Compare commits
42
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
1267d20fb8 | ||
|
|
35f697ed08 | ||
|
|
d0fb6befb1 | ||
|
|
81c96019b5 | ||
|
|
baed8ff8f2 | ||
|
|
f80fb4dd1b | ||
|
|
b71b793681 | ||
|
|
61c87ad1d9 | ||
|
|
4dee508e12 | ||
|
|
76da5134ef | ||
|
|
4c630c9a13 | ||
|
|
d8dbefcd07 | ||
|
|
974208fc05 | ||
|
|
2eb71e8244 | ||
|
|
668cccbf24 | ||
|
|
4ea4585402 | ||
|
|
dc6d102473 | ||
|
|
683afc0a47 | ||
|
|
2c3e20512e | ||
|
|
51b2f04672 | ||
|
|
f190c8251e | ||
|
|
1945404eaa | ||
|
|
6476722372 | ||
|
|
4078334d85 | ||
|
|
0a6689b29e | ||
|
|
ef22167b1a | ||
|
|
cbda0fee0a | ||
|
|
7234817adb | ||
|
|
f1c36d6eea | ||
|
|
bb21ceb7f5 | ||
|
|
a8970b8ace | ||
|
|
4996bc0984 | ||
|
|
a0de597d69 | ||
|
|
544da29863 | ||
|
|
daaa0fda41 | ||
|
|
56a276bcc1 | ||
|
|
754bbcf1fa | ||
|
|
63e6b709f8 | ||
|
|
8612b0a9e9 | ||
|
|
96d9cfaa63 | ||
|
|
3c7536dba0 | ||
|
|
3c92e1cbfd |
@@ -118,6 +118,79 @@ concurrency:
|
||||
cancel-in-progress: true
|
||||
|
||||
jobs:
|
||||
# ---------------------------------------------------------------------------
|
||||
# THE TEST GATE (2026-07-26). Everything below `needs:` this job, so a red test
|
||||
# stops the release instead of shipping with it.
|
||||
#
|
||||
# WHY IT IS A JOB HERE AND NOT JUST .gitea/workflows/test.yml: a separate
|
||||
# workflow cannot block another one — they run side by side and a red `test`
|
||||
# workflow would have published anyway. Only a `needs:` edge inside THIS
|
||||
# workflow is a gate. test.yml exists too, for fast feedback on `main`; both
|
||||
# call the same scripts/run-tests.sh so they cannot drift.
|
||||
#
|
||||
# WHAT WAS BROKEN: the release tract ran two `go test` invocations in total —
|
||||
# build-shaterd.sh's one-package buildtags check and check-router-tags.sh's
|
||||
# three named tests. 115 of the 116 test files under shater/** had never run in
|
||||
# CI (upstream's .github/workflows/test.yml triggers on branches this fork does
|
||||
# not have, and Gitea ignores .github/workflows entirely once .gitea/workflows
|
||||
# exists). TestDNSFilterRemoteBlocklistHTTPClient shipped red twice.
|
||||
#
|
||||
# WHAT IT COVERS: the whole suite under the SHIPPED build tags
|
||||
# (scripts/router-tags.sh) on linux — the two dimensions that were missing.
|
||||
# transport/wireguard compiles 1 test file without the tag set and 7 with it
|
||||
# (the AmneziaWG ones); shater/generate has 44 test files on linux against 32
|
||||
# elsewhere. Plus a -race pass and the panel's TypeScript tests. Details and
|
||||
# the named, reasoned exclusions are in scripts/run-tests.sh.
|
||||
test:
|
||||
name: test gate
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- name: Checkout
|
||||
uses: actions/checkout@v4
|
||||
|
||||
# go.mod `replace`s wireguard-go to ./submodules/wireguard-go, so without
|
||||
# this even `go list` fails. Same step/reason as in build-apk below.
|
||||
- name: Init wireguard-go submodule (awg)
|
||||
run: git submodule update --init --depth 1 submodules/wireguard-go
|
||||
|
||||
- name: Set up Go
|
||||
uses: actions/setup-go@v5
|
||||
with:
|
||||
go-version-file: go.mod
|
||||
cache: false # explicit actions/cache@v3.3.2 below
|
||||
|
||||
# Same cache key as build-apk: this job runs first, so it warms the module
|
||||
# + build cache the SDK-lane build then restores. (v3.3.2 pin: see header.)
|
||||
- name: Cache Go modules + build cache
|
||||
uses: actions/cache@v3.3.2
|
||||
with:
|
||||
path: |
|
||||
~/go/pkg/mod
|
||||
~/.cache/go-build
|
||||
key: go-${{ hashFiles('go.sum') }}
|
||||
restore-keys: |
|
||||
go-
|
||||
|
||||
# Node 24, NOT the 20 build-apk uses for the SPA: panel's tests are
|
||||
# TypeScript run directly by `node --test`, and type stripping only exists
|
||||
# from 22.6 — on node 20 `npm test` dies before running a single case.
|
||||
- name: Set up Node
|
||||
uses: actions/setup-node@v4
|
||||
with:
|
||||
node-version: '24'
|
||||
|
||||
- name: Cache panel node_modules
|
||||
uses: actions/cache@v3.3.2
|
||||
with:
|
||||
path: panel/node_modules
|
||||
key: npm-${{ hashFiles('panel/package-lock.json') }}
|
||||
|
||||
- name: Panel tests
|
||||
run: bash scripts/run-panel-tests.sh
|
||||
|
||||
- name: Go tests (shipped tags, linux, + race)
|
||||
run: bash scripts/run-tests.sh
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Build the 4 packages through the ImmortalWrt 25.12 apk-SDK for the 25.12/apk
|
||||
# fleet (BPI-R3 mini on BananaWRT 25.12-mtk-vendor, BPI-R4 on OpenWrt 25.12,
|
||||
@@ -125,6 +198,9 @@ jobs:
|
||||
# packages.adb + shater-apk.pem, uploaded as the artifact `apkfeed-<arch>`.
|
||||
build-apk:
|
||||
name: apk ${{ matrix.arch }}
|
||||
# THE GATE EDGE. A red test skips this job, which leaves no artifact, which
|
||||
# (with the guards in release-apk) leaves nothing published.
|
||||
needs: test
|
||||
runs-on: ubuntu-latest
|
||||
strategy:
|
||||
fail-fast: false
|
||||
@@ -302,11 +378,17 @@ jobs:
|
||||
# of that URL, so it is now written unconditionally and asserted afterwards.
|
||||
release-apk:
|
||||
name: release apk
|
||||
needs: build-apk
|
||||
needs: [test, build-apk]
|
||||
# Publish whatever arch feeds succeeded — do NOT block the aarch64 release
|
||||
# when an unrelated arch (e.g. x86_64) fails. download-artifact only fetches
|
||||
# artifacts that exist, and the publish loop skips missing apkfeed-* dirs.
|
||||
if: ${{ !cancelled() }}
|
||||
#
|
||||
# `needs.test.result == 'success'` is the second half of the gate. Without
|
||||
# it, `!cancelled()` is true when the test job FAILS (build-apk is then
|
||||
# skipped), this job runs with no artifacts at all, and — see the guard at
|
||||
# the end of the publish step — used to exit 0 having published nothing. Red
|
||||
# tests must SKIP this job, not "succeed" through it.
|
||||
if: ${{ !cancelled() && needs.test.result == 'success' }}
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- name: Checkout
|
||||
@@ -342,6 +424,14 @@ jobs:
|
||||
ROLLING: ${{ steps.rel.outputs.rolling }}
|
||||
run: |
|
||||
set -euo pipefail
|
||||
# Counted, and asserted non-zero at the end. Until 2026-07-26 this loop
|
||||
# was the step's whole body: with no artifacts the glob stayed
|
||||
# unexpanded, `[ -d ... ]` was false, `continue` ran once, the loop
|
||||
# ended and the step exited 0 — "release apk" went GREEN having
|
||||
# published absolutely nothing. Any upstream failure (all arches
|
||||
# failing to build, an artifact-name change, a download-artifact
|
||||
# hiccup) therefore looked like a successful release.
|
||||
published=0
|
||||
for d in artifacts/apkfeed-*; do
|
||||
[ -d "$d" ] || continue
|
||||
arch="${d#artifacts/apkfeed-}"
|
||||
@@ -432,4 +522,15 @@ jobs:
|
||||
echo " own rules, not by our intent."
|
||||
exit 14; }
|
||||
echo "[release-apk] OK — $ROLL serves $want"
|
||||
published=$((published + 1))
|
||||
done
|
||||
|
||||
# The assert the loop above never had. Zero feeds published is a failed
|
||||
# release, not a quiet success — say so with a non-zero exit.
|
||||
if [ "$published" -eq 0 ]; then
|
||||
echo "[release-apk] ERROR: no apkfeed-* artifact reached this job, so"
|
||||
echo " NOTHING was published. Downloaded tree:"
|
||||
ls -la artifacts 2>&1 | sed 's/^/ /' || echo " (no artifacts/ dir at all)"
|
||||
exit 10
|
||||
fi
|
||||
echo "[release-apk] published $published arch feed(s)"
|
||||
|
||||
@@ -0,0 +1,85 @@
|
||||
# Shater — the test gate, on every push to `main`.
|
||||
#
|
||||
# WHY THIS FILE EXISTS (2026-07-26)
|
||||
# The fork had a full suite and no CI that ran it. Upstream's
|
||||
# .github/workflows/test.yml triggers on `stable`/`testing`/`unstable`; this
|
||||
# repo only has `main`. And Gitea does not read .github/workflows AT ALL once
|
||||
# .gitea/workflows exists — so those files are decoration here. Result: 115 of
|
||||
# the 116 test files under shater/** had never once executed in CI, and
|
||||
# TestDNSFilterRemoteBlocklistHTTPClient stayed red across two published
|
||||
# releases.
|
||||
#
|
||||
# RELATIONSHIP TO release.yml
|
||||
# This workflow is the FAST FEEDBACK loop on `main`. It is NOT the release
|
||||
# gate: a separate workflow cannot block another one. The gate is the `test`
|
||||
# JOB inside .gitea/workflows/release.yml, which build-apk `needs:` — see the
|
||||
# comment there. Both run the very same scripts/run-tests.sh, so they cannot
|
||||
# drift apart.
|
||||
name: test
|
||||
|
||||
on:
|
||||
push:
|
||||
branches: [main]
|
||||
paths-ignore:
|
||||
- '**.md'
|
||||
- 'docs-shater/**'
|
||||
pull_request:
|
||||
branches: [main]
|
||||
workflow_dispatch:
|
||||
|
||||
concurrency:
|
||||
group: test-${{ github.ref }}
|
||||
cancel-in-progress: true
|
||||
|
||||
jobs:
|
||||
test:
|
||||
name: go + panel tests
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- name: Checkout
|
||||
uses: actions/checkout@v4
|
||||
|
||||
# go.mod has `replace github.com/sagernet/wireguard-go => ./submodules/
|
||||
# wireguard-go`, so WITHOUT this every `go list`/`go test` fails before it
|
||||
# starts. Same step, same reason, as in release.yml's build job.
|
||||
- name: Init wireguard-go submodule (awg)
|
||||
run: git submodule update --init --depth 1 submodules/wireguard-go
|
||||
|
||||
- name: Set up Go
|
||||
uses: actions/setup-go@v5
|
||||
with:
|
||||
go-version-file: go.mod
|
||||
cache: false # explicit actions/cache@v3.3.2 below
|
||||
|
||||
# v3.3.2 is the last release speaking the cache API act_runner implements
|
||||
# (see the header of release.yml). Same key as the release build job, so
|
||||
# whichever runs first warms the other.
|
||||
- name: Cache Go modules + build cache
|
||||
uses: actions/cache@v3.3.2
|
||||
with:
|
||||
path: |
|
||||
~/go/pkg/mod
|
||||
~/.cache/go-build
|
||||
key: go-${{ hashFiles('go.sum') }}
|
||||
restore-keys: |
|
||||
go-
|
||||
|
||||
# Node 24, NOT the 20 the SPA build uses: panel's tests are TypeScript run
|
||||
# through `node --test`, and type stripping only exists from 22.6. On
|
||||
# node 20 `npm test` dies with a syntax error before running anything.
|
||||
- name: Set up Node
|
||||
uses: actions/setup-node@v4
|
||||
with:
|
||||
node-version: '24'
|
||||
|
||||
- name: Cache panel node_modules
|
||||
uses: actions/cache@v3.3.2
|
||||
with:
|
||||
path: panel/node_modules
|
||||
key: npm-${{ hashFiles('panel/package-lock.json') }}
|
||||
|
||||
- name: Panel tests
|
||||
run: bash scripts/run-panel-tests.sh
|
||||
|
||||
- name: Go tests (shipped tags, linux, + race)
|
||||
run: bash scripts/run-tests.sh
|
||||
+7
-1
@@ -7,4 +7,10 @@
|
||||
[submodule "submodules/wireguard-go"]
|
||||
path = submodules/wireguard-go
|
||||
url = https://github.com/Leadaxe/wireguard-go-awg2-lx
|
||||
branch = lx
|
||||
# The pin lives on lx-awg2-v005, NOT on lx: the two are separate lines (42
|
||||
# commits apart one way, 131 the other). `lx` has no hasReserved() gate in
|
||||
# conn/bind_std.go at all, so a `git submodule update --remote` against it
|
||||
# would silently restore the bug where ClientBind/StdNetBind shred the
|
||||
# AmneziaWG magic header and no chain carries traffic. Keep this pointing at
|
||||
# the line the pin is actually on.
|
||||
branch = lx-awg2-v005
|
||||
|
||||
+3
-1
@@ -32,7 +32,9 @@ single-use token into the standalone SPA the daemon serves on its own port
|
||||
|
||||
## Highlights
|
||||
|
||||
- Transparent **TPROXY** data plane (TCP + UDP), SNI/Host/QUIC sniffing, no DNS leaks.
|
||||
- Transparent **TPROXY** data plane (TCP + UDP), SNI/Host/QUIC sniffing, no DNS leaks
|
||||
— `:53` interception is on by default and covers the queries a client sends to the
|
||||
router itself, not just the ones aimed around it (`globals.dns_intercept`, D24).
|
||||
- First-match routing by source / destination / list / geo / client → outbound /
|
||||
selector / chain / direct / block; node groups with balancer/observatory;
|
||||
multi-hop chains; per-rule egress.
|
||||
|
||||
@@ -3,7 +3,33 @@
|
||||
| Поле | Значение |
|
||||
|------|----------|
|
||||
| Тип | B (bug) |
|
||||
| Статус | C (complete) |
|
||||
| Статус | C (complete) — guard **снят** (см. баннер ниже) |
|
||||
|
||||
> ## ⛔️ Guard снят (2026-07-26) — первопричина к shater не относится
|
||||
> **Оба guard'а (Start-guard в `protocol/wireguard/endpoint.go` и
|
||||
> selector-guard в `protocol/group/awg_selector_guard.go`) удалены**, вместе с
|
||||
> их adapter-хуками (`OutboundManager.ConsumersOf`, `AmneziaWGSuspendable`).
|
||||
> Апстрим снял их коммитом `5fa3a0a17`; сюда снятие приехало отдельно.
|
||||
>
|
||||
> **Почему.** Зависание было **Android-специфичным** (`Libbox.newService` не
|
||||
> возвращал управление). Android для shater не платформа и ей не станет —
|
||||
> мы собираем роутерный бинарь под OpenWrt/aarch64. При этом лекарство для
|
||||
> самой AWG-за-detour связки у нас уже есть: reserved-clear gate в
|
||||
> `ClientBind` (`d971eb85e` + пин сабмодуля `7d15f33`), без которого AWG не
|
||||
> поднимался вообще ни за каким detour'ом. Мы носили и лекарство, и запрет
|
||||
> на его применение.
|
||||
>
|
||||
> **Чем это было плохо на практике.** Guard отказывал **молча**: не ошибкой,
|
||||
> а `started=false`, после чего каждый дозвон падал с «WireGuard is not ready
|
||||
> yet». Конфигурация «AmneziaWG за WireGuard-хопом» выглядела не как
|
||||
> отклонённая, а как «нода почему-то не работает».
|
||||
>
|
||||
> **Регрессия:** `protocol/wireguard/awg_over_wireguard_start_lx_test.go`
|
||||
> (`with_gvisor && with_awg`) — AWG-эндпоинт с `detour` на outbound типа
|
||||
> `wireguard` доходит до PostStart и поднимает `started`. До снятия guard'а
|
||||
> тест краснел.
|
||||
>
|
||||
> **Осталось:** сквозной прогон на железе (AWG поверх реального WG-хопа).
|
||||
|
||||
Отклонять (по образцу ядрового запрета «empty direct detour») конфигурацию, где
|
||||
AmneziaWG-endpoint (источник с AWG-полями) имеет `detour` на **любой
|
||||
|
||||
@@ -0,0 +1,145 @@
|
||||
// lx:begin l3-honest-drop
|
||||
package adapter
|
||||
|
||||
import (
|
||||
"net/netip"
|
||||
"testing"
|
||||
|
||||
"github.com/sagernet/sing-tun"
|
||||
"github.com/sagernet/sing-tun/gtcpip/header"
|
||||
|
||||
"github.com/stretchr/testify/require"
|
||||
)
|
||||
|
||||
// judgeFlowRouter answers PreMatch with a canned verdict; JudgeFlow reads
|
||||
// nothing else off the Router.
|
||||
type judgeFlowRouter struct {
|
||||
Router
|
||||
result PreMatchResult
|
||||
}
|
||||
|
||||
func (r *judgeFlowRouter) PreMatch(InboundContext, []byte) PreMatchResult { return r.result }
|
||||
|
||||
// judgeFlowPort is the tun.Port half of a FlowOutbound. inet4 is what
|
||||
// PortAddresses reports for IPv4 — the one field the two ICMP consumers in
|
||||
// sing-tun disagree about (see the comment on
|
||||
// TestJudgeFlowICMPToBoundPortStaysAFlow).
|
||||
type judgeFlowPort struct {
|
||||
Outbound
|
||||
inet4 netip.Addr
|
||||
}
|
||||
|
||||
func (o *judgeFlowPort) Tag() string { return "wg-out" }
|
||||
func (o *judgeFlowPort) Type() string { return "wireguard" }
|
||||
func (o *judgeFlowPort) PortAddresses() (netip.Addr, netip.Addr) {
|
||||
return o.inet4, netip.Addr{}
|
||||
}
|
||||
func (o *judgeFlowPort) PortMTU() uint32 { return 1420 }
|
||||
func (o *judgeFlowPort) AttachReturn(tun.Return) error { return nil }
|
||||
func (o *judgeFlowPort) DetachReturn(tun.Return) error { return nil }
|
||||
func (o *judgeFlowPort) WritePackets(packets [][]byte) error { return nil }
|
||||
|
||||
// judgeFlowNonPort is a FlowOutbound-shaped result that is NOT a tun.Port — the
|
||||
// interface drift the second line of defense in JudgeFlow exists for.
|
||||
type judgeFlowNonPort struct {
|
||||
Outbound
|
||||
}
|
||||
|
||||
func (o *judgeFlowNonPort) Tag() string { return "drifted" }
|
||||
func (o *judgeFlowNonPort) Type() string { return "drifted" }
|
||||
|
||||
func judgeFlow(t *testing.T, protocol uint8, result PreMatchResult) tun.FlowVerdict {
|
||||
t.Helper()
|
||||
return JudgeFlow(
|
||||
&judgeFlowRouter{result: result},
|
||||
"l3-in", "tun", protocol,
|
||||
netip.MustParseAddrPort("192.168.1.2:1234"),
|
||||
netip.MustParseAddrPort("1.1.1.1:1234"),
|
||||
nil,
|
||||
)
|
||||
}
|
||||
|
||||
const (
|
||||
judgeFlowICMP = uint8(header.ICMPv4ProtocolNumber)
|
||||
judgeFlowTCP = uint8(header.TCPProtocolNumber)
|
||||
)
|
||||
|
||||
// TestJudgeFlowICMPToBoundPortStaysAFlow is the guard on the ONE fix that must
|
||||
// not be made here.
|
||||
//
|
||||
// sing-tun has two ICMP consumers with different requirements on the port:
|
||||
//
|
||||
// - ForwardDispatcher.createFlow (flow_dispatch.go) needs only a VALID port
|
||||
// address — it NATs the echo identifier and rewrites the source to that
|
||||
// address. This is the path every unfragmented LAN ping takes, and it is
|
||||
// what makes ping-through-WireGuard/AWG work at all.
|
||||
// - ICMPForwarder.installFlow (stack_gvisor_icmp.go) additionally requires the
|
||||
// address to be UNSPECIFIED, because it writes the packet to the port
|
||||
// unmodified. A WireGuard endpoint reports its concrete interface address
|
||||
// (transport/wireguard/port.go), so installFlow declines and HandlePacket
|
||||
// falls through to forging the echo reply.
|
||||
//
|
||||
// The tempting fix — "for ICMP, refuse ActionFlow when PortAddresses() is not
|
||||
// unspecified, so the verdict becomes a drop and the forgery is unreachable" —
|
||||
// is applied HERE, in the one function both consumers share, with byte-identical
|
||||
// arguments from either. It would therefore kill the working path too: every
|
||||
// ping through WireGuard/AWG, fragmented or not, would drop, and l3_tunnel would
|
||||
// carry nothing but `direct`. Keep this test failing loudly if anyone tries.
|
||||
func TestJudgeFlowICMPToBoundPortStaysAFlow(t *testing.T) {
|
||||
t.Parallel()
|
||||
port := &judgeFlowPort{inet4: netip.MustParseAddr("10.2.0.2")}
|
||||
verdict := judgeFlow(t, judgeFlowICMP, PreMatchResult{Action: PreMatchFlow, Outbound: port})
|
||||
require.Equal(t, tun.ActionFlow, verdict.Action,
|
||||
"ICMP to a WireGuard/AWG endpoint must stay a flow: the forward dispatcher NATs it by echo identifier and this is the whole point of l3_tunnel")
|
||||
require.Same(t, tun.Port(port), verdict.Port)
|
||||
}
|
||||
|
||||
// The `direct` shape: an unspecified port address. Both consumers accept it.
|
||||
func TestJudgeFlowICMPToUnspecifiedPortStaysAFlow(t *testing.T) {
|
||||
t.Parallel()
|
||||
port := &judgeFlowPort{inet4: netip.IPv4Unspecified()}
|
||||
verdict := judgeFlow(t, judgeFlowICMP, PreMatchResult{Action: PreMatchFlow, Outbound: port})
|
||||
require.Equal(t, tun.ActionFlow, verdict.Action)
|
||||
require.Same(t, tun.Port(port), verdict.Port)
|
||||
}
|
||||
|
||||
// PreMatchDrop is the honest verdict and must arrive as ActionDrop: it is the
|
||||
// only value (besides Reject) that stops ICMPForwarder.HandlePacket before the
|
||||
// Echo -> EchoReply rewrite.
|
||||
func TestJudgeFlowICMPDropReachesTheStackAsDrop(t *testing.T) {
|
||||
t.Parallel()
|
||||
verdict := judgeFlow(t, judgeFlowICMP, PreMatchResult{Action: PreMatchDrop})
|
||||
require.Equal(t, tun.ActionDrop, verdict.Action)
|
||||
}
|
||||
|
||||
// The second line of defense: a PreMatchFlow whose outbound is not a tun.Port
|
||||
// must not degrade ICMP to ActionAccept, because Accept is the forged reply.
|
||||
func TestJudgeFlowICMPNonPortOutboundDrops(t *testing.T) {
|
||||
t.Parallel()
|
||||
verdict := judgeFlow(t, judgeFlowICMP, PreMatchResult{Action: PreMatchFlow, Outbound: &judgeFlowNonPort{}})
|
||||
require.Equal(t, tun.ActionDrop, verdict.Action,
|
||||
"FlowOutbound and tun.Port are distinct interfaces; a drift between them must not silently re-enable the echo forger")
|
||||
}
|
||||
|
||||
func TestJudgeFlowTCPNonPortOutboundAccepts(t *testing.T) {
|
||||
t.Parallel()
|
||||
verdict := judgeFlow(t, judgeFlowTCP, PreMatchResult{Action: PreMatchFlow, Outbound: &judgeFlowNonPort{}})
|
||||
require.Equal(t, tun.ActionAccept, verdict.Action,
|
||||
"for TCP, falling back to Accept is upstream behaviour and must stay untouched")
|
||||
}
|
||||
|
||||
// TCP keeps every mapping it had, including the Continue -> Accept default that
|
||||
// is a forgery only for ICMP.
|
||||
func TestJudgeFlowTCPContinueStaysAccept(t *testing.T) {
|
||||
t.Parallel()
|
||||
verdict := judgeFlow(t, judgeFlowTCP, PreMatchResult{Action: PreMatchContinue})
|
||||
require.Equal(t, tun.ActionAccept, verdict.Action)
|
||||
}
|
||||
|
||||
func TestJudgeFlowTCPBypassStaysBypass(t *testing.T) {
|
||||
t.Parallel()
|
||||
verdict := judgeFlow(t, judgeFlowTCP, PreMatchResult{Action: PreMatchBypass})
|
||||
require.Equal(t, tun.ActionBypass, verdict.Action)
|
||||
}
|
||||
|
||||
// lx:end l3-honest-drop
|
||||
@@ -45,30 +45,8 @@ type OutboundManager interface {
|
||||
Default() Outbound
|
||||
Remove(tag string) error
|
||||
Create(ctx context.Context, router Router, logger log.ContextLogger, tag string, outboundType string, options any) error
|
||||
// lx:begin awg
|
||||
// ConsumersOf returns the tags of outbounds that depend on (detour through)
|
||||
// the given tag — the reverse of Dependencies(). Used by the selector guard to
|
||||
// walk up to AmneziaWG consumers when a group switches to a WireGuard member.
|
||||
ConsumersOf(tag string) []string
|
||||
// lx:end awg
|
||||
}
|
||||
|
||||
// lx:begin awg
|
||||
// AmneziaWGSuspendable is implemented by an AmneziaWG endpoint so the selector
|
||||
// guard can suspend it (bring its device down) when a group it detours through
|
||||
// switches to a WireGuard member — AmneziaWG inside a WireGuard tunnel hangs the
|
||||
// kernel on Android. The marker lives in adapter so protocol/group can act on it
|
||||
// without importing protocol/wireguard.
|
||||
type AmneziaWGSuspendable interface {
|
||||
// IsAmneziaWG reports whether this endpoint runs AmneziaWG (has AWG params).
|
||||
IsAmneziaWG() bool
|
||||
// SuspendAmneziaWG brings the device down so no junk handshake is sent. It is
|
||||
// idempotent and safe to call on a not-yet-started or already-suspended endpoint.
|
||||
SuspendAmneziaWG()
|
||||
}
|
||||
|
||||
// lx:end awg
|
||||
|
||||
// lx:begin idle-suspend
|
||||
// IdleSuspendable is implemented by a WG/AWG endpoint so the router's idle tick
|
||||
// (SPEC 020) can suspend it when it is idle and unreachable, without importing
|
||||
|
||||
@@ -208,21 +208,6 @@ func (m *Manager) Outbound(tag string) (adapter.Outbound, bool) {
|
||||
return m.endpoint.Get(tag)
|
||||
}
|
||||
|
||||
// lx:begin awg
|
||||
// ConsumersOf returns a copy of the tags that detour through tag (reverse of
|
||||
// Dependencies()), built from the dependByTag ledger populated at Create time.
|
||||
func (m *Manager) ConsumersOf(tag string) []string {
|
||||
m.access.RLock()
|
||||
defer m.access.RUnlock()
|
||||
consumers := m.dependByTag[tag]
|
||||
if len(consumers) == 0 {
|
||||
return nil
|
||||
}
|
||||
return append([]string(nil), consumers...)
|
||||
}
|
||||
|
||||
// lx:end awg
|
||||
|
||||
func (m *Manager) Default() adapter.Outbound {
|
||||
m.access.RLock()
|
||||
defer m.access.RUnlock()
|
||||
|
||||
@@ -75,7 +75,18 @@ func JudgeFlow(router Router, inbound string, inboundType string, network uint8,
|
||||
case PreMatchFlow:
|
||||
port, isPort := result.Outbound.(tun.Port)
|
||||
if !isPort {
|
||||
// lx:begin l3-honest-drop
|
||||
// Second line of defense behind route.(*Router).preMatchFlow: a
|
||||
// PreMatchFlow result already implies the outbound is an
|
||||
// adapter.FlowOutbound, but FlowOutbound and tun.Port are distinct
|
||||
// interfaces, and a drift between them must not degrade ICMP to
|
||||
// ActionAccept — the TUN stack would then forge the echo reply
|
||||
// itself instead of admitting the tunnel cannot carry the packet.
|
||||
if networkName == N.NetworkICMP {
|
||||
return tun.FlowVerdict{Action: tun.ActionDrop}
|
||||
}
|
||||
return tun.FlowVerdict{Action: tun.ActionAccept}
|
||||
// lx:end l3-honest-drop
|
||||
}
|
||||
verdict := tun.FlowVerdict{Action: tun.ActionFlow, Port: port, UDPTimeout: result.UDPTimeout, NewTracker: result.NewTracker}
|
||||
if result.Destination.IsValid() {
|
||||
|
||||
@@ -0,0 +1,35 @@
|
||||
//go:build darwin
|
||||
|
||||
package dialer
|
||||
|
||||
import (
|
||||
"syscall"
|
||||
"testing"
|
||||
|
||||
"golang.org/x/sys/unix"
|
||||
)
|
||||
|
||||
// udpSocketDFSet reports whether the socket has "don't fragment" forced on
|
||||
// (control.DisableUDPFragment sets IP_DONTFRAG=1 on darwin).
|
||||
func udpSocketDFSet(t *testing.T, sysConn syscall.Conn) bool {
|
||||
t.Helper()
|
||||
rawConn, err := sysConn.SyscallConn()
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
var (
|
||||
value int
|
||||
sockErr error
|
||||
ctrlErr error
|
||||
)
|
||||
ctrlErr = rawConn.Control(func(fd uintptr) {
|
||||
value, sockErr = unix.GetsockoptInt(int(fd), unix.IPPROTO_IP, unix.IP_DONTFRAG)
|
||||
})
|
||||
if ctrlErr != nil {
|
||||
t.Fatal(ctrlErr)
|
||||
}
|
||||
if sockErr != nil {
|
||||
t.Fatal(sockErr)
|
||||
}
|
||||
return value != 0
|
||||
}
|
||||
@@ -0,0 +1,36 @@
|
||||
//go:build linux
|
||||
|
||||
package dialer
|
||||
|
||||
import (
|
||||
"syscall"
|
||||
"testing"
|
||||
|
||||
"golang.org/x/sys/unix"
|
||||
)
|
||||
|
||||
// udpSocketDFSet reports whether the socket has "don't fragment" forced on
|
||||
// (control.DisableUDPFragment sets IP_MTU_DISCOVER=IP_PMTUDISC_DO on linux,
|
||||
// the same flag the user-visible failure was traced to on android).
|
||||
func udpSocketDFSet(t *testing.T, sysConn syscall.Conn) bool {
|
||||
t.Helper()
|
||||
rawConn, err := sysConn.SyscallConn()
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
var (
|
||||
value int
|
||||
sockErr error
|
||||
ctrlErr error
|
||||
)
|
||||
ctrlErr = rawConn.Control(func(fd uintptr) {
|
||||
value, sockErr = unix.GetsockoptInt(int(fd), unix.IPPROTO_IP, unix.IP_MTU_DISCOVER)
|
||||
})
|
||||
if ctrlErr != nil {
|
||||
t.Fatal(ctrlErr)
|
||||
}
|
||||
if sockErr != nil {
|
||||
t.Fatal(sockErr)
|
||||
}
|
||||
return value == unix.IP_PMTUDISC_DO
|
||||
}
|
||||
@@ -0,0 +1,14 @@
|
||||
//go:build !darwin && !linux && !windows
|
||||
|
||||
package dialer
|
||||
|
||||
import (
|
||||
"syscall"
|
||||
"testing"
|
||||
)
|
||||
|
||||
func udpSocketDFSet(t *testing.T, _ syscall.Conn) bool {
|
||||
t.Helper()
|
||||
t.Skip("DF socket-flag introspection implemented for darwin, linux and windows only")
|
||||
return false
|
||||
}
|
||||
@@ -0,0 +1,43 @@
|
||||
//go:build windows
|
||||
|
||||
package dialer
|
||||
|
||||
import (
|
||||
"syscall"
|
||||
"testing"
|
||||
|
||||
"golang.org/x/sys/windows"
|
||||
)
|
||||
|
||||
// IP_MTU_DISCOVER on windows (ws2ipdef.h); control.DisableUDPFragment sets it to
|
||||
// IP_PMTUDISC_DO, the same "don't fragment" state the linux helper checks.
|
||||
const (
|
||||
windowsIPMTUDiscover = 71
|
||||
windowsPMTUDiscDo = 1
|
||||
)
|
||||
|
||||
// udpSocketDFSet reports whether the socket has "don't fragment" forced on.
|
||||
// shater addition: upstream ships linux + darwin only, so the whole suite
|
||||
// skipped on the dev host — where it is the one platform we can actually run it
|
||||
// on before the router build.
|
||||
func udpSocketDFSet(t *testing.T, sysConn syscall.Conn) bool {
|
||||
t.Helper()
|
||||
rawConn, err := sysConn.SyscallConn()
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
var (
|
||||
value int
|
||||
sockErr error
|
||||
)
|
||||
ctrlErr := rawConn.Control(func(fd uintptr) {
|
||||
value, sockErr = windows.GetsockoptInt(windows.Handle(fd), windows.IPPROTO_IP, windowsIPMTUDiscover)
|
||||
})
|
||||
if ctrlErr != nil {
|
||||
t.Fatal(ctrlErr)
|
||||
}
|
||||
if sockErr != nil {
|
||||
t.Skip("IP_MTU_DISCOVER is not readable on this host: ", sockErr)
|
||||
}
|
||||
return value == windowsPMTUDiscDo
|
||||
}
|
||||
@@ -0,0 +1,99 @@
|
||||
// lx: regression tests for the udp_fragment / UDPFragmentDefault
|
||||
// plumbing. The WireGuard endpoint (and MASQUE outbound) rely on
|
||||
// UDPFragmentDefault=true reaching the real UDP socket as "DF clear": with DF
|
||||
// set, an outer datagram larger than the path MTU is silently dropped instead
|
||||
// of fragmented, which blackholes nested tunnels (AWG-over-AWG, MASQUE-over-AWG)
|
||||
// and AWG s4 transport junk. These tests assert the socket flag itself, on both
|
||||
// paths a WireGuard bind can take: the dialer (ClientBind, detour case) and the
|
||||
// listener control (StdNetBind via WireGuardControl, no-detour case).
|
||||
package dialer
|
||||
|
||||
import (
|
||||
"context"
|
||||
"net"
|
||||
"syscall"
|
||||
"testing"
|
||||
|
||||
"github.com/sagernet/sing-box/option"
|
||||
M "github.com/sagernet/sing/common/metadata"
|
||||
N "github.com/sagernet/sing/common/network"
|
||||
)
|
||||
|
||||
func dialUDPForDF(t *testing.T, options option.DialerOptions) syscall.Conn {
|
||||
t.Helper()
|
||||
d, err := NewDefault(context.Background(), options)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
conn, err := d.DialContext(context.Background(), N.NetworkUDP, M.ParseSocksaddr("127.0.0.1:9"))
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
t.Cleanup(func() { _ = conn.Close() })
|
||||
sysConn, isSysConn := conn.(syscall.Conn)
|
||||
if !isSysConn {
|
||||
t.Fatalf("dialed UDP conn %T does not expose SyscallConn", conn)
|
||||
}
|
||||
return sysConn
|
||||
}
|
||||
|
||||
func listenUDPForDF(t *testing.T, options option.DialerOptions) syscall.Conn {
|
||||
t.Helper()
|
||||
d, err := NewDefault(context.Background(), options)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
// WireGuardControl() is the listener control conn.StdNetBind installs on the
|
||||
// socket a no-detour WireGuard endpoint sends its outer datagrams from — the
|
||||
// exact socket the DF default decides the fate of.
|
||||
listenConfig := net.ListenConfig{Control: d.WireGuardControl()}
|
||||
packetConn, err := listenConfig.ListenPacket(context.Background(), "udp4", "127.0.0.1:0")
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
t.Cleanup(func() { _ = packetConn.Close() })
|
||||
sysConn, isSysConn := packetConn.(syscall.Conn)
|
||||
if !isSysConn {
|
||||
t.Fatalf("listened UDP conn %T does not expose SyscallConn", packetConn)
|
||||
}
|
||||
return sysConn
|
||||
}
|
||||
|
||||
// Upstream default: no UDPFragmentDefault, no udp_fragment → DF is set on both
|
||||
// the dial and listener paths. Pins the baseline the endpoint fix opts out of.
|
||||
func TestUDPFragmentDFByDefault_LX(t *testing.T) {
|
||||
if !udpSocketDFSet(t, dialUDPForDF(t, option.DialerOptions{})) {
|
||||
t.Fatal("default dialer must set DF on dialed UDP sockets")
|
||||
}
|
||||
if !udpSocketDFSet(t, listenUDPForDF(t, option.DialerOptions{})) {
|
||||
t.Fatal("default dialer must set DF on listener-control UDP sockets")
|
||||
}
|
||||
}
|
||||
|
||||
// UDPFragmentDefault=true (what the WireGuard endpoint and MASQUE outbound now
|
||||
// set) → DF clear on both paths, so oversize outer datagrams fragment instead
|
||||
// of vanishing.
|
||||
func TestUDPFragmentDefaultClearsDF_LX(t *testing.T) {
|
||||
options := option.DialerOptions{UDPFragmentDefault: true}
|
||||
if udpSocketDFSet(t, dialUDPForDF(t, options)) {
|
||||
t.Fatal("UDPFragmentDefault=true must leave DF clear on dialed UDP sockets")
|
||||
}
|
||||
if udpSocketDFSet(t, listenUDPForDF(t, options)) {
|
||||
t.Fatal("UDPFragmentDefault=true must leave DF clear on listener-control UDP sockets")
|
||||
}
|
||||
}
|
||||
|
||||
// Explicit user config always wins over the protocol default, in both
|
||||
// directions.
|
||||
func TestUDPFragmentExplicitOverride_LX(t *testing.T) {
|
||||
fragmentOff := false
|
||||
options := option.DialerOptions{UDPFragment: &fragmentOff, UDPFragmentDefault: true}
|
||||
if !udpSocketDFSet(t, dialUDPForDF(t, options)) {
|
||||
t.Fatal("udp_fragment=false must set DF even when the protocol default allows fragmentation")
|
||||
}
|
||||
fragmentOn := true
|
||||
options = option.DialerOptions{UDPFragment: &fragmentOn}
|
||||
if udpSocketDFSet(t, dialUDPForDF(t, options)) {
|
||||
t.Fatal("udp_fragment=true must leave DF clear even without a protocol default")
|
||||
}
|
||||
}
|
||||
@@ -25,6 +25,21 @@ func requireRoot(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
// requireTCPDump skips when tcpdump is not installed.
|
||||
//
|
||||
// The same honesty this package's callers demand of a health reading: a missing
|
||||
// INSTRUMENT is "not checked", never "broken". Without it every test in this
|
||||
// file fails on `cmd.Start()` — sixteen red results that say nothing about the
|
||||
// code and hide any real failure among them — on a machine where the only thing
|
||||
// wrong is that a capture tool is absent. requireRoot has always drawn that line
|
||||
// for privileges; this draws it for the tool.
|
||||
func requireTCPDump(t *testing.T) {
|
||||
t.Helper()
|
||||
if _, err := exec.LookPath("tcpdump"); err != nil {
|
||||
t.Skip("integration test requires tcpdump on PATH; install it to run this suite")
|
||||
}
|
||||
}
|
||||
|
||||
func tcpdumpObserver(t *testing.T, iface string, port uint16, needle string, do func(), wait time.Duration) bool {
|
||||
t.Helper()
|
||||
return tcpdumpObserverMulti(t, iface, port, []string{needle}, do, wait)[needle]
|
||||
@@ -36,6 +51,9 @@ func tcpdumpObserver(t *testing.T, iface string, port uint16, needle string, do
|
||||
// the wire.
|
||||
func tcpdumpObserverMulti(t *testing.T, iface string, port uint16, needles []string, do func(), wait time.Duration) map[string]bool {
|
||||
t.Helper()
|
||||
// Every capture in this file funnels through here, so one guard covers the
|
||||
// whole suite and no future test can forget it.
|
||||
requireTCPDump(t)
|
||||
ctx, cancel := context.WithTimeout(context.Background(), wait)
|
||||
defer cancel()
|
||||
cmd := exec.CommandContext(ctx, "tcpdump", "-i", iface, "-n", "-A", "-l",
|
||||
|
||||
@@ -0,0 +1,141 @@
|
||||
// lx:begin health-board
|
||||
|
||||
package urltest
|
||||
|
||||
import (
|
||||
"strconv"
|
||||
"strings"
|
||||
"sync"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"github.com/sagernet/sing-box/adapter"
|
||||
)
|
||||
|
||||
// captureEvictions swaps the eviction notice sink for the duration of a test and
|
||||
// returns a func that reads back everything reported.
|
||||
func captureEvictions(t *testing.T) func() []string {
|
||||
t.Helper()
|
||||
var (
|
||||
mu sync.Mutex
|
||||
msgs []string
|
||||
)
|
||||
orig := boardEvictionLog
|
||||
boardEvictionLog = func(m string) {
|
||||
mu.Lock()
|
||||
msgs = append(msgs, m)
|
||||
mu.Unlock()
|
||||
}
|
||||
t.Cleanup(func() { boardEvictionLog = orig })
|
||||
return func() []string {
|
||||
mu.Lock()
|
||||
defer mu.Unlock()
|
||||
return append([]string(nil), msgs...)
|
||||
}
|
||||
}
|
||||
|
||||
// TestBoardHoldsAGenerationWithoutEvicting is the "what it holds" half of the
|
||||
// bound. A live generation on this box is ~1200 tags (≈380 nodes plus their
|
||||
// per-group egress copies and chain hops); the board must carry that — and a
|
||||
// second generation's worth of overlap during a subscription rename — with no
|
||||
// eviction at all, or the ceiling would be silently degrading real health data.
|
||||
func TestBoardHoldsAGenerationWithoutEvicting(t *testing.T) {
|
||||
read := captureEvictions(t)
|
||||
s := NewHistoryStorage()
|
||||
|
||||
const generation = 1200
|
||||
for gen := 0; gen < 2; gen++ {
|
||||
for i := 0; i < generation; i++ {
|
||||
s.StoreURLTestHistory("gen"+strconv.Itoa(gen)+"-node-"+strconv.Itoa(i),
|
||||
&adapter.URLTestHistory{LastOK: time.Now(), Delay: 20})
|
||||
}
|
||||
}
|
||||
if got := s.Evicted(); got != 0 {
|
||||
t.Fatalf("two full generations (%d tags) evicted %d entries; the board must hold them",
|
||||
2*generation, got)
|
||||
}
|
||||
if msgs := read(); len(msgs) != 0 {
|
||||
t.Fatalf("unexpected eviction notices: %v", msgs)
|
||||
}
|
||||
// Everything is still readable.
|
||||
if s.LoadURLTestHistory("gen0-node-0") == nil {
|
||||
t.Fatalf("the first tag of the first generation was lost without an eviction")
|
||||
}
|
||||
}
|
||||
|
||||
// TestBoardEvictsOldestAndSaysSo is the "what happens when it overflows" half.
|
||||
// Overflow must (a) actually bound the map, (b) drop the LEAST RECENTLY MEASURED
|
||||
// tags — on this box, exactly the ones no config names any more — and (c) be
|
||||
// audible: a silent eviction is a health board quietly forgetting nodes it is
|
||||
// still being asked about.
|
||||
func TestBoardEvictsOldestAndSaysSo(t *testing.T) {
|
||||
read := captureEvictions(t)
|
||||
s := NewHistoryStorage()
|
||||
|
||||
base := time.Now().Add(-24 * time.Hour)
|
||||
// Stale generation first: measured a day ago, nothing since.
|
||||
const stale = 1500
|
||||
for i := 0; i < stale; i++ {
|
||||
s.StoreURLTestHistory("stale-"+strconv.Itoa(i),
|
||||
&adapter.URLTestHistory{LastOK: base.Add(time.Duration(i) * time.Millisecond), Delay: 30})
|
||||
}
|
||||
if s.Evicted() != 0 {
|
||||
t.Fatalf("evicted before the ceiling was reached")
|
||||
}
|
||||
// Now push past the ceiling with fresh measurements.
|
||||
for i := 0; i <= maxBoardEntries; i++ {
|
||||
s.StoreURLTestHistory("fresh-"+strconv.Itoa(i),
|
||||
&adapter.URLTestHistory{LastOK: time.Now(), Delay: 15})
|
||||
}
|
||||
|
||||
if got := s.Evicted(); got == 0 {
|
||||
t.Fatalf("board grew past %d entries without evicting anything — it is still unbounded", maxBoardEntries)
|
||||
}
|
||||
s.access.RLock()
|
||||
size := len(s.delayHistory)
|
||||
s.access.RUnlock()
|
||||
if size > maxBoardEntries {
|
||||
t.Fatalf("board holds %d entries, above the %d ceiling", size, maxBoardEntries)
|
||||
}
|
||||
|
||||
// The day-old generation is what went, not the fresh one.
|
||||
if s.LoadURLTestHistory("stale-0") != nil {
|
||||
t.Fatalf("the oldest observation survived while newer ones were dropped")
|
||||
}
|
||||
if s.LoadURLTestHistory("fresh-"+strconv.Itoa(maxBoardEntries)) == nil {
|
||||
t.Fatalf("the newest measurement was evicted")
|
||||
}
|
||||
|
||||
msgs := read()
|
||||
if len(msgs) == 0 {
|
||||
t.Fatalf("entries were evicted with no notice — eviction must never be silent")
|
||||
}
|
||||
m := msgs[0]
|
||||
for _, want := range []string{"health board full", "evicted", "re-probed"} {
|
||||
if !strings.Contains(m, want) {
|
||||
t.Fatalf("eviction notice %q does not say %q", m, want)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestBoardEvictionThroughMarkFailed pins the OTHER write path. MarkFailed is how
|
||||
// a dead node is recorded, and a flood of dead renamed nodes is exactly the shape
|
||||
// of the leak — so it has to prune too, not just the success path.
|
||||
func TestBoardEvictionThroughMarkFailed(t *testing.T) {
|
||||
captureEvictions(t)
|
||||
s := NewHistoryStorage()
|
||||
for i := 0; i <= maxBoardEntries; i++ {
|
||||
s.MarkFailed("dead-" + strconv.Itoa(i))
|
||||
}
|
||||
s.access.RLock()
|
||||
size := len(s.delayHistory)
|
||||
s.access.RUnlock()
|
||||
if size > maxBoardEntries {
|
||||
t.Fatalf("MarkFailed grew the board to %d, above the %d ceiling", size, maxBoardEntries)
|
||||
}
|
||||
if s.Evicted() == 0 {
|
||||
t.Fatalf("MarkFailed never prunes — the failure path is still unbounded")
|
||||
}
|
||||
}
|
||||
|
||||
// lx:end health-board
|
||||
@@ -10,11 +10,128 @@
|
||||
package urltest
|
||||
|
||||
import (
|
||||
"sort"
|
||||
"strconv"
|
||||
"sync"
|
||||
"time"
|
||||
|
||||
"github.com/sagernet/sing-box/adapter"
|
||||
"github.com/sagernet/sing-box/log"
|
||||
)
|
||||
|
||||
// --- board capacity ---------------------------------------------------------
|
||||
//
|
||||
// The board is the one structure in the daemon whose key space is chosen by
|
||||
// somebody else. Its keys are outbound TAGS, and on this box a tag is a node
|
||||
// NAME straight out of the subscription — plus the derived per-group egress
|
||||
// copies ("group-<g>-m<i>-<node>") and per-chain hop copies the probe planner
|
||||
// creates for the same nodes. Providers rename their nodes freely, so a daily
|
||||
// subscription refresh introduces a whole new generation of keys, while the
|
||||
// store itself is pinned to the ENGINE's context (shater/engine.New) and so
|
||||
// outlives every generation and every Apply — by design, so health survives a
|
||||
// config change.
|
||||
//
|
||||
// Nothing ever removed a key. DeleteURLTestHistory exists but no shater path
|
||||
// calls it (only daemon/ and clashapi/, which this fork does not run), so the
|
||||
// map was strictly append-only for the life of the process — and the process is
|
||||
// expected to live for months.
|
||||
//
|
||||
// The arithmetic: ~380 nodes, and a config with a couple of egress-bound groups
|
||||
// plus a handful of chains puts a LIVE generation at roughly 380 base tags +
|
||||
// 2x380 group copies + ~100 chain copies ≈ 1200 keys. One new generation per day
|
||||
// is ~440k keys a year, at ~200 B per entry (map bucket + a tag string that is
|
||||
// routinely 30-50 B with flag emoji, + a 56 B URLTestHistory) ≈ 88 MB of a
|
||||
// 512 MB box — spent entirely on nodes that no longer exist.
|
||||
const (
|
||||
// maxBoardEntries is the hard ceiling. 4096 is ~3.4 live generations, so the
|
||||
// board comfortably holds the current config plus the overlap while a
|
||||
// subscription refresh swaps names, and still costs under a megabyte. A tighter
|
||||
// bound would start evicting tags the running config actually uses; a looser one
|
||||
// would stop being a bound in any useful sense.
|
||||
maxBoardEntries = 4096
|
||||
// keepBoardEntries is the prune target: drop a quarter at a time so the
|
||||
// O(n log n) selection is amortised over ~1024 inserts instead of running on
|
||||
// every probe once the board is full.
|
||||
keepBoardEntries = 3072
|
||||
)
|
||||
|
||||
// boardEvictionLog reports an eviction. A package var so tests can capture it;
|
||||
// production leaves it writing to the process log, which under procd is the same
|
||||
// syslog/logsink stream every other daemon line lands in.
|
||||
//
|
||||
// Eviction is NEVER silent. It is not free either: an evicted tag reverts to
|
||||
// "untested" and its next probe re-measures it, so a board that evicts entries
|
||||
// belonging to the LIVE config is a board whose ceiling is too low — and the only
|
||||
// way anyone finds that out is this line.
|
||||
var boardEvictionLog = func(msg string) { boardLogger().Warn(msg) }
|
||||
|
||||
// pruneLocked drops the least-recently-OBSERVED entries when the board exceeds
|
||||
// maxBoardEntries. "Least recently observed" is max(LastOK, LastFail): the entry
|
||||
// nothing has measured for the longest is, on this box, precisely a tag that no
|
||||
// longer exists in any config — a renamed node, a removed group copy, a retired
|
||||
// chain hop. Caller holds access.
|
||||
func (s *HistoryStorage) pruneLocked() {
|
||||
if len(s.delayHistory) <= maxBoardEntries {
|
||||
return
|
||||
}
|
||||
type kv struct {
|
||||
tag string
|
||||
seen time.Time
|
||||
}
|
||||
all := make([]kv, 0, len(s.delayHistory))
|
||||
for tag, h := range s.delayHistory {
|
||||
seen := h.LastOK
|
||||
if h.LastFail.After(seen) {
|
||||
seen = h.LastFail
|
||||
}
|
||||
all = append(all, kv{tag, seen})
|
||||
}
|
||||
sort.Slice(all, func(i, j int) bool { return all[i].seen.Before(all[j].seen) })
|
||||
drop := len(all) - keepBoardEntries
|
||||
var oldest time.Time
|
||||
for i := 0; i < drop; i++ {
|
||||
if i == 0 {
|
||||
oldest = all[i].seen
|
||||
}
|
||||
delete(s.delayHistory, all[i].tag)
|
||||
}
|
||||
s.evicted += uint64(drop)
|
||||
|
||||
msg := "urltest: health board full (" + strconv.Itoa(maxBoardEntries) + " tags) — evicted " +
|
||||
strconv.Itoa(drop) + " least-recently-measured entries (" + strconv.FormatUint(s.evicted, 10) +
|
||||
" total since start); they revert to untested and will be re-probed"
|
||||
if !oldest.IsZero() {
|
||||
msg += "; oldest observation was " + time.Since(oldest).Truncate(time.Second).String() + " ago"
|
||||
}
|
||||
boardEvictionLog(msg)
|
||||
}
|
||||
|
||||
// Evicted reports how many entries the capacity bound has dropped since the store
|
||||
// was created. Nonzero means the board reached maxBoardEntries at least once.
|
||||
func (s *HistoryStorage) Evicted() uint64 {
|
||||
if s == nil {
|
||||
return 0
|
||||
}
|
||||
s.access.RLock()
|
||||
defer s.access.RUnlock()
|
||||
return s.evicted
|
||||
}
|
||||
|
||||
// boardLogger is the process-wide fallback logger for eviction notices. The store
|
||||
// is built from a plain constructor with no logger in sight (box.New, the daemon,
|
||||
// shater/engine all call NewHistoryStorage()), so rather than change that
|
||||
// signature everywhere the notice goes to the standard logger — which on the
|
||||
// router is the daemon's own stderr, i.e. the same sink logsink owns.
|
||||
var (
|
||||
boardLogOnce sync.Once
|
||||
boardLog log.ContextLogger
|
||||
)
|
||||
|
||||
func boardLogger() log.ContextLogger {
|
||||
boardLogOnce.Do(func() { boardLog = log.StdLogger() })
|
||||
return boardLog
|
||||
}
|
||||
|
||||
// HealthVerdict classifies a stored history entry at read time.
|
||||
type HealthVerdict int
|
||||
|
||||
@@ -54,6 +171,7 @@ func (s *HistoryStorage) MarkFailed(tag string) {
|
||||
updated.Delay = previous.Delay
|
||||
}
|
||||
s.delayHistory[tag] = updated
|
||||
s.pruneLocked()
|
||||
s.notifyUpdated()
|
||||
s.access.Unlock()
|
||||
}
|
||||
|
||||
@@ -0,0 +1,61 @@
|
||||
package urltest
|
||||
|
||||
// lx: health board §5.C — the reachability half of "should this be probed".
|
||||
//
|
||||
// # Two different reasons not to probe, and why they cannot be one flag
|
||||
//
|
||||
// A group's OWN probing schedule is stood down for two unrelated reasons, and
|
||||
// conflating them breaks one of the two:
|
||||
//
|
||||
// - NOT USED — no enabled routing rule reaches this group, so probing it
|
||||
// measures a path nothing travels. That is a property of the CONFIG, it is
|
||||
// decided once when the config is generated, and it travels in the config
|
||||
// itself (option.URLTestOutboundOptions.SelfCheck). It cannot change while
|
||||
// the box runs, because the rules cannot change while the box runs.
|
||||
//
|
||||
// - NOT REACHABLE RIGHT NOW — the group is a hop of a chain and a hop in
|
||||
// FRONT of it is currently dead. Every member of this group dials through
|
||||
// that hop, so every probe would fail inside it: the measurement would be
|
||||
// about the broken hop, and would be recorded against this one. That is a
|
||||
// property of the WORLD, it changes minute by minute, and it must be
|
||||
// re-asked every time rather than baked into the config — a hop that comes
|
||||
// back must resume probing on its own, with no reapply and nobody pressing
|
||||
// anything.
|
||||
//
|
||||
// ProbeGate is the second one. It is deliberately a QUESTION asked at the
|
||||
// moment of probing and never a stored answer: there is no flag to set, so
|
||||
// there is no flag to forget to clear.
|
||||
//
|
||||
// The gate governs the group's own SCHEDULE only — the warm-up sweep and the
|
||||
// ticker. An explicit check (a human, an API call) is a deliberate request and
|
||||
// is never refused, exactly as with SelfCheck.
|
||||
type ProbeGate interface {
|
||||
// ProbeAllowed reports whether the outbound tagged tag may run its own
|
||||
// scheduled probe right now.
|
||||
//
|
||||
// Implementations MUST answer true when they do not know: a gate that
|
||||
// refuses on missing information would silence probing precisely when the
|
||||
// system has the least idea what is going on, and nothing would ever
|
||||
// measure its way out of that. A nil ProbeGate means "no gate" and every
|
||||
// probe proceeds.
|
||||
ProbeAllowed(tag string) bool
|
||||
|
||||
// ProbeWhenIdle reports whether the outbound tagged tag must keep measuring
|
||||
// even when no traffic is passing through it.
|
||||
//
|
||||
// A urltest group normally probes only while it is in use: Touch arms the
|
||||
// ticker on a dial, and the idle timeout stops it again. That is right for a
|
||||
// group whose readings matter only while somebody is dialling it, and wrong
|
||||
// for one the routing config REACHES: a rule that matches rarely — a narrow
|
||||
// domain list, say — is in force the whole time, so the health of its target
|
||||
// is a live question the whole time. Letting it go quiet means the panel
|
||||
// reports "untested" about a rule that is armed, and the first real request
|
||||
// pays a cold probe instead of picking an already-known-good member.
|
||||
//
|
||||
// Unlike ProbeAllowed, the safe answer here is FALSE when nothing is known.
|
||||
// This one ADDS work, and a gate that claimed it on missing information would
|
||||
// keep every group in the process probing forever — not a default anybody
|
||||
// asked for. Absent gate, unknown tag, nothing configured yet: false, and the
|
||||
// idle timeout behaves exactly as it always has.
|
||||
ProbeWhenIdle(tag string) bool
|
||||
}
|
||||
@@ -21,6 +21,10 @@ type HistoryStorage struct {
|
||||
access sync.RWMutex
|
||||
delayHistory map[string]*adapter.URLTestHistory
|
||||
updateHooks []*observable.Subscriber[struct{}]
|
||||
// evicted counts entries dropped by the capacity bound (board_lx.go). The map
|
||||
// is keyed by outbound tags chosen by a subscription provider, so it needs a
|
||||
// ceiling; see the comment on maxBoardEntries.
|
||||
evicted uint64
|
||||
}
|
||||
|
||||
func NewHistoryStorage() *HistoryStorage {
|
||||
@@ -71,6 +75,11 @@ func (s *HistoryStorage) StoreURLTestHistory(tag string, history *adapter.URLTes
|
||||
}
|
||||
// lx:end health-board
|
||||
s.delayHistory[tag] = history
|
||||
// lx:begin health-board — the map is keyed by provider-chosen tags and the
|
||||
// store outlives every engine generation, so it must bound itself here: no
|
||||
// shater path ever calls DeleteURLTestHistory. See maxBoardEntries.
|
||||
s.pruneLocked()
|
||||
// lx:end health-board
|
||||
s.notifyUpdated()
|
||||
s.access.Unlock()
|
||||
}
|
||||
|
||||
@@ -126,6 +126,12 @@ func (t *HTTP3Transport) newTransport() *http3.Transport {
|
||||
conn.Close()
|
||||
return nil, dialErr
|
||||
}
|
||||
// quic-go does not take ownership of the packet conn passed to
|
||||
// DialEarly: when the connection ends it only stops reading.
|
||||
go func() {
|
||||
<-quicConn.Context().Done()
|
||||
conn.Close()
|
||||
}()
|
||||
return quicConn, nil
|
||||
},
|
||||
TLSClientConfig: t.tlsConfig,
|
||||
|
||||
@@ -0,0 +1,351 @@
|
||||
package quic
|
||||
|
||||
import (
|
||||
"context"
|
||||
"crypto/tls"
|
||||
"net"
|
||||
"net/http"
|
||||
"net/url"
|
||||
"sync"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"github.com/sagernet/quic-go"
|
||||
"github.com/sagernet/quic-go/http3"
|
||||
sbTLS "github.com/sagernet/sing-box/common/tls"
|
||||
C "github.com/sagernet/sing-box/constant"
|
||||
"github.com/sagernet/sing-box/dns"
|
||||
"github.com/sagernet/sing-box/dns/transport"
|
||||
"github.com/sagernet/sing-box/option"
|
||||
"github.com/sagernet/sing/common"
|
||||
"github.com/sagernet/sing/common/logger"
|
||||
M "github.com/sagernet/sing/common/metadata"
|
||||
N "github.com/sagernet/sing/common/network"
|
||||
|
||||
mDNS "github.com/miekg/dns"
|
||||
)
|
||||
|
||||
var _ N.Dialer = (*trackingDialer)(nil)
|
||||
|
||||
// These tests pin down who owns the UDP socket handed to quic-go.
|
||||
//
|
||||
// quic-go's Dial/DialEarly take a net.PacketConn but do NOT take ownership of
|
||||
// it: quic.setupTransport() builds a Transport with createdConn=false, and
|
||||
// Transport.Close() then only calls conn.SetReadDeadline(time.Now()) instead of
|
||||
// conn.Close(). So every QUIC connection torn down here — idle timeout, a
|
||||
// retryable error, an engine reload calling Reset() — used to strand the UDP
|
||||
// socket that carried it for the rest of the process's life. On a router that
|
||||
// resolves through DoQ/DoH3 for months that is an unbounded fd leak.
|
||||
//
|
||||
// Both tests reconnect once and assert the socket from the FIRST connection is
|
||||
// actually closed. Without the `<-conn.Context().Done() -> rawConn.Close()`
|
||||
// watchdogs in quic.go / http3.go they fail on that assertion.
|
||||
|
||||
type trackedConn struct {
|
||||
net.Conn
|
||||
closeOnce sync.Once
|
||||
closed chan struct{}
|
||||
}
|
||||
|
||||
func (c *trackedConn) Close() error {
|
||||
c.closeOnce.Do(func() { close(c.closed) })
|
||||
return c.Conn.Close()
|
||||
}
|
||||
|
||||
// trackingDialer hands out real UDP sockets and remembers every one of them.
|
||||
type trackingDialer struct {
|
||||
access sync.Mutex
|
||||
conns []*trackedConn
|
||||
}
|
||||
|
||||
func (d *trackingDialer) DialContext(ctx context.Context, network string, destination M.Socksaddr) (net.Conn, error) {
|
||||
conn, err := (&net.Dialer{}).DialContext(ctx, network, destination.String())
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
tracked := &trackedConn{Conn: conn, closed: make(chan struct{})}
|
||||
d.access.Lock()
|
||||
d.conns = append(d.conns, tracked)
|
||||
d.access.Unlock()
|
||||
return tracked, nil
|
||||
}
|
||||
|
||||
func (d *trackingDialer) ListenPacket(ctx context.Context, destination M.Socksaddr) (net.PacketConn, error) {
|
||||
return net.ListenUDP("udp", nil)
|
||||
}
|
||||
|
||||
func (d *trackingDialer) count() int {
|
||||
d.access.Lock()
|
||||
defer d.access.Unlock()
|
||||
return len(d.conns)
|
||||
}
|
||||
|
||||
func (d *trackingDialer) at(index int) *trackedConn {
|
||||
d.access.Lock()
|
||||
defer d.access.Unlock()
|
||||
return d.conns[index]
|
||||
}
|
||||
|
||||
func (d *trackingDialer) closeAll() {
|
||||
d.access.Lock()
|
||||
defer d.access.Unlock()
|
||||
for _, conn := range d.conns {
|
||||
conn.Close()
|
||||
}
|
||||
}
|
||||
|
||||
func requireClosed(t *testing.T, conn *trackedConn, what string) {
|
||||
t.Helper()
|
||||
select {
|
||||
case <-conn.closed:
|
||||
case <-time.After(5 * time.Second):
|
||||
t.Fatalf("%s: the UDP socket of the retired QUIC connection was never closed — quic-go does not own it, we must", what)
|
||||
}
|
||||
}
|
||||
|
||||
func requireDialed(t *testing.T, dialer *trackingDialer, want int) {
|
||||
t.Helper()
|
||||
deadline := time.Now().Add(5 * time.Second)
|
||||
for time.Now().Before(deadline) {
|
||||
if dialer.count() >= want {
|
||||
return
|
||||
}
|
||||
time.Sleep(10 * time.Millisecond)
|
||||
}
|
||||
t.Fatalf("expected at least %d dial(s), got %d", want, dialer.count())
|
||||
}
|
||||
|
||||
func testServerTLSConfig(t *testing.T, nextProtos []string) *tls.Config {
|
||||
t.Helper()
|
||||
certificate, err := sbTLS.GenerateKeyPair(nil, nil, nil, "localhost")
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
return &tls.Config{
|
||||
Certificates: []tls.Certificate{*certificate},
|
||||
NextProtos: nextProtos,
|
||||
MinVersion: tls.VersionTLS13,
|
||||
}
|
||||
}
|
||||
|
||||
func testClientTLSConfig(t *testing.T, nextProtos []string) sbTLS.Config {
|
||||
t.Helper()
|
||||
config, err := sbTLS.NewClient(context.Background(), logger.NOP(), "localhost", option.OutboundTLSOptions{
|
||||
Enabled: true,
|
||||
Insecure: true,
|
||||
ServerName: "localhost",
|
||||
})
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
config.SetNextProtos(nextProtos)
|
||||
return config
|
||||
}
|
||||
|
||||
// startDoQServer serves a minimal DoQ responder and returns its address.
|
||||
func startDoQServer(t *testing.T) M.Socksaddr {
|
||||
t.Helper()
|
||||
listener, err := quic.ListenAddr("127.0.0.1:0", testServerTLSConfig(t, []string{"doq"}), nil)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
ctx, cancel := context.WithCancel(context.Background())
|
||||
t.Cleanup(func() {
|
||||
cancel()
|
||||
listener.Close()
|
||||
})
|
||||
go func() {
|
||||
for {
|
||||
conn, acceptErr := listener.Accept(ctx)
|
||||
if acceptErr != nil {
|
||||
return
|
||||
}
|
||||
go func(conn *quic.Conn) {
|
||||
for {
|
||||
stream, streamErr := conn.AcceptStream(ctx)
|
||||
if streamErr != nil {
|
||||
return
|
||||
}
|
||||
go func(stream *quic.Stream) {
|
||||
defer stream.Close()
|
||||
request, readErr := transport.ReadMessage(stream)
|
||||
if readErr != nil {
|
||||
return
|
||||
}
|
||||
response := new(mDNS.Msg)
|
||||
response.SetReply(request)
|
||||
transport.WriteMessage(stream, 0, response)
|
||||
}(stream)
|
||||
}
|
||||
}(conn)
|
||||
}
|
||||
}()
|
||||
return M.ParseSocksaddr(listener.Addr().String())
|
||||
}
|
||||
|
||||
func testQuery() *mDNS.Msg {
|
||||
message := new(mDNS.Msg)
|
||||
message.SetQuestion("example.com.", mDNS.TypeA)
|
||||
return message
|
||||
}
|
||||
|
||||
func TestQUICTransportClosesPacketConnOnReconnect(t *testing.T) {
|
||||
t.Parallel()
|
||||
serverAddr := startDoQServer(t)
|
||||
dialer := &trackingDialer{}
|
||||
t.Cleanup(dialer.closeAll)
|
||||
|
||||
dnsTransport := &Transport{
|
||||
TransportAdapter: dns.NewTransportAdapter(C.DNSTypeQUIC, "test-doq", nil),
|
||||
dialer: dialer,
|
||||
serverAddr: serverAddr,
|
||||
tlsConfig: testClientTLSConfig(t, []string{"doq"}),
|
||||
connection: transport.NewConnPool(transport.ConnPoolOptions[*quic.Conn]{
|
||||
Mode: transport.ConnPoolSingle,
|
||||
IsAlive: func(conn *quic.Conn) bool {
|
||||
return conn != nil && !common.Done(conn.Context())
|
||||
},
|
||||
Close: func(conn *quic.Conn, _ error) {
|
||||
conn.CloseWithError(0, "")
|
||||
},
|
||||
}),
|
||||
}
|
||||
t.Cleanup(func() { dnsTransport.Close() })
|
||||
|
||||
ctx, cancel := context.WithTimeout(context.Background(), 15*time.Second)
|
||||
defer cancel()
|
||||
|
||||
if _, err := dnsTransport.Exchange(ctx, testQuery()); err != nil {
|
||||
t.Fatal("first exchange: ", err)
|
||||
}
|
||||
requireDialed(t, dialer, 1)
|
||||
first := dialer.at(0)
|
||||
|
||||
// Retire the connection the way a retryable error or an engine reload does.
|
||||
dnsTransport.Reset()
|
||||
requireClosed(t, first, "Reset()")
|
||||
|
||||
// The reconnect must still work, on a fresh socket.
|
||||
if _, err := dnsTransport.Exchange(ctx, testQuery()); err != nil {
|
||||
t.Fatal("second exchange: ", err)
|
||||
}
|
||||
requireDialed(t, dialer, 2)
|
||||
second := dialer.at(1)
|
||||
if second == first {
|
||||
t.Fatal("expected a new UDP socket for the reconnect")
|
||||
}
|
||||
|
||||
if err := dnsTransport.Close(); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
requireClosed(t, second, "Close()")
|
||||
}
|
||||
|
||||
func TestHTTP3TransportClosesPacketConnOnReconnect(t *testing.T) {
|
||||
t.Parallel()
|
||||
mux := http.NewServeMux()
|
||||
mux.HandleFunc("/dns-query", func(writer http.ResponseWriter, request *http.Request) {
|
||||
message, err := readRequestMessage(request)
|
||||
if err != nil {
|
||||
writer.WriteHeader(http.StatusBadRequest)
|
||||
return
|
||||
}
|
||||
response := new(mDNS.Msg)
|
||||
response.SetReply(message)
|
||||
rawResponse, err := response.Pack()
|
||||
if err != nil {
|
||||
writer.WriteHeader(http.StatusInternalServerError)
|
||||
return
|
||||
}
|
||||
writer.Header().Set("Content-Type", transport.MimeType)
|
||||
writer.Write(rawResponse)
|
||||
})
|
||||
listener, err := quic.ListenAddrEarly("127.0.0.1:0", testServerTLSConfig(t, []string{http3.NextProtoH3}), nil)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
server := &http3.Server{Handler: mux}
|
||||
go server.ServeListener(listener)
|
||||
t.Cleanup(func() {
|
||||
server.Close()
|
||||
listener.Close()
|
||||
})
|
||||
serverAddr := M.ParseSocksaddr(listener.Addr().String())
|
||||
|
||||
dialer := &trackingDialer{}
|
||||
t.Cleanup(dialer.closeAll)
|
||||
|
||||
stdConfig := &tls.Config{
|
||||
InsecureSkipVerify: true,
|
||||
ServerName: "localhost",
|
||||
NextProtos: []string{http3.NextProtoH3},
|
||||
MinVersion: tls.VersionTLS13,
|
||||
}
|
||||
dnsTransport := &HTTP3Transport{
|
||||
TransportAdapter: dns.NewTransportAdapter(C.DNSTypeHTTP3, "test-doh3", nil),
|
||||
logger: logger.NOP(),
|
||||
dialer: dialer,
|
||||
destination: &url.URL{Scheme: "https", Host: "localhost", Path: "/dns-query"},
|
||||
headers: http.Header{},
|
||||
serverAddr: serverAddr,
|
||||
tlsConfig: stdConfig,
|
||||
}
|
||||
dnsTransport.transport = dnsTransport.newTransport()
|
||||
t.Cleanup(func() { dnsTransport.Close() })
|
||||
|
||||
ctx, cancel := context.WithTimeout(context.Background(), 15*time.Second)
|
||||
defer cancel()
|
||||
|
||||
if _, err = dnsTransport.Exchange(ctx, testQuery()); err != nil {
|
||||
t.Fatal("first exchange: ", err)
|
||||
}
|
||||
requireDialed(t, dialer, 1)
|
||||
first := dialer.at(0)
|
||||
|
||||
dnsTransport.Reset()
|
||||
requireClosed(t, first, "Reset()")
|
||||
|
||||
if _, err = dnsTransport.Exchange(ctx, testQuery()); err != nil {
|
||||
t.Fatal("second exchange: ", err)
|
||||
}
|
||||
requireDialed(t, dialer, 2)
|
||||
second := dialer.at(1)
|
||||
if second == first {
|
||||
t.Fatal("expected a new UDP socket for the reconnect")
|
||||
}
|
||||
|
||||
if err = dnsTransport.Close(); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
requireClosed(t, second, "Close()")
|
||||
}
|
||||
|
||||
func readRequestMessage(request *http.Request) (*mDNS.Msg, error) {
|
||||
defer request.Body.Close()
|
||||
rawMessage := make([]byte, 4096)
|
||||
n, err := readFull(request.Body, rawMessage)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
var message mDNS.Msg
|
||||
err = message.Unpack(rawMessage[:n])
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
return &message, nil
|
||||
}
|
||||
|
||||
func readFull(reader interface{ Read([]byte) (int, error) }, buffer []byte) (int, error) {
|
||||
var total int
|
||||
for total < len(buffer) {
|
||||
n, err := reader.Read(buffer[total:])
|
||||
total += n
|
||||
if err != nil {
|
||||
if total > 0 {
|
||||
return total, nil
|
||||
}
|
||||
return total, err
|
||||
}
|
||||
}
|
||||
return total, nil
|
||||
}
|
||||
@@ -4,6 +4,7 @@ import (
|
||||
"context"
|
||||
"errors"
|
||||
"os"
|
||||
"time"
|
||||
|
||||
"github.com/sagernet/quic-go"
|
||||
"github.com/sagernet/sing-box/adapter"
|
||||
@@ -117,6 +118,12 @@ func (t *Transport) Exchange(ctx context.Context, message *mDNS.Msg) (*mDNS.Msg,
|
||||
rawConn.Close()
|
||||
return nil, E.Cause(err, "establish QUIC connection")
|
||||
}
|
||||
// quic-go does not take ownership of the packet conn passed to
|
||||
// DialEarly: when the connection ends it only stops reading.
|
||||
go func() {
|
||||
<-earlyConnection.Context().Done()
|
||||
rawConn.Close()
|
||||
}()
|
||||
return earlyConnection, nil
|
||||
})
|
||||
if err != nil {
|
||||
@@ -144,6 +151,11 @@ func (t *Transport) exchange(ctx context.Context, message *mDNS.Msg, conn *quic.
|
||||
return nil, E.Cause(err, "open stream")
|
||||
}
|
||||
defer stream.CancelRead(0)
|
||||
stopWatch := context.AfterFunc(ctx, func() {
|
||||
stream.CancelRead(0)
|
||||
_ = stream.SetWriteDeadline(time.Now())
|
||||
})
|
||||
defer stopWatch()
|
||||
err = transport.WriteMessage(stream, 0, message)
|
||||
if err != nil {
|
||||
stream.Close()
|
||||
|
||||
@@ -12,6 +12,49 @@ as GitHub **pre-releases** and never become "Latest".
|
||||
|
||||
#### Unreleased (shater)
|
||||
|
||||
**`l3-honest-drop` — ICMP routed to an L4-only outbound is dropped, not
|
||||
forged** — ships with `shaterd` (part of the shater L3 ingress,
|
||||
`docs-shater/DECISIONS.md` D25), not as an lx release tag; recorded here because
|
||||
it edits two upstream files. Without it the TUN stack answers an unroutable echo
|
||||
ITSELF — sing-tun's `ICMPForwarder.HandlePacket` rewrites Echo→EchoReply
|
||||
whenever the flow judgment comes back Accept (`stack_gvisor_icmp.go`) — so a
|
||||
ping routed to vless/vmess/… would read as a working tunnel while the packet
|
||||
never left the router.
|
||||
|
||||
* **`route/route.go` (`PreMatch`)** — the pre-match walk was renamed to
|
||||
`preMatch` and the exported `PreMatch` became a thin FUNNEL that rewrites
|
||||
`PreMatchContinue` and `PreMatchBypass` to `PreMatchDrop` for
|
||||
`N.NetworkICMP`. An earlier version overrode `continueResult` inside
|
||||
`preMatchFlow` instead; that covered only the exits reaching that function and
|
||||
left three of the walk's own exits forging — the `prepareMatchMetadata` error
|
||||
return, the sniff bail-outs, and the `default:` arm of the rule-action switch
|
||||
(every action pre-match has no arm for: `hijack-dns`, `direct`, …). A guard on
|
||||
the single return value cannot be outgrown by a new exit. `PreMatchBypass` is
|
||||
folded in because sing-tun implements `ActionBypass` on the nfqueue plane only
|
||||
— on the TUN path it lands in the same `default:` arm as Accept, i.e. forges.
|
||||
* **`adapter/router.go` (`JudgeFlow`, the `!isPort` branch)** — ICMP returns
|
||||
`ActionDrop` where it fell through to `ActionAccept`. Second line of defense:
|
||||
`adapter.FlowOutbound` and `tun.Port` are distinct interfaces, and a drift
|
||||
between them must not quietly re-enable the forged reply.
|
||||
* **TCP/UDP behaviour is unchanged** — `PreMatchContinue` still means "take the
|
||||
ordinary connection route" for both, `PreMatchBypass` still means bypass, and
|
||||
the `!isPort` fallthrough still returns `ActionAccept` for them; pinned by
|
||||
`route/prematch_icmp_lx_test.go` and `adapter/judgeflow_icmp_lx_test.go`
|
||||
(both inside the marker), each ICMP case having an explicit TCP/UDP twin.
|
||||
* **NOT covered: a FRAGMENTED echo to a WireGuard/AWG outbound is still
|
||||
forged** — sing-tun's `ForwardDispatcher.Dispatch` returns before asking for a
|
||||
verdict at all when `parsed.fragment`, and the reassembled packet reaches
|
||||
`ICMPForwarder.HandlePacket`, whose `installFlow` demands an UNSPECIFIED port
|
||||
address that a WireGuard endpoint never has. Fixing it inside `JudgeFlow`
|
||||
is NOT possible — both consumers call it with identical arguments and the
|
||||
working path needs the concrete address. Full chain, the two viable fixes and
|
||||
the trap are in `docs-shater/DECISIONS.md` D25, "KNOWN HOLE".
|
||||
* **Rebase cost: two small marked blocks** (`lx:begin/end l3-honest-drop`, a
|
||||
wrapper function in `route/route.go` and one branch body in
|
||||
`adapter/router.go`) plus the two self-contained test files — carried across
|
||||
an upstream rebase by eye. Note that `PreMatch`'s own body now lives in
|
||||
`preMatch`, so an upstream change to the walk applies to that function.
|
||||
|
||||
**Fork-layer + control-plane rework of proxy health** — ships with `shaterd`
|
||||
(the shater router daemon), not as an lx release tag; recorded here because the
|
||||
load-bearing half lives in fork zones (`common/urltest`, `protocol/group`).
|
||||
|
||||
@@ -319,6 +319,10 @@ Three values, not two, because the leaks differ in *kind*: an ICMP echo is ephem
|
||||
user-initiated and reveals the address only to a host the user deliberately contacted,
|
||||
whereas ESP/GRE is a standing second tunnel carrying arbitrary traffic beside ours. A
|
||||
single toggle would make "I want ping to work" mean "I allow a parallel VPN bypass".
|
||||
*(Refined 2026-07-26 by D25: still true of TPROXY — but ICMP echo now has an
|
||||
opt-in data plane of its own, the dedicated L3 TUN, so the policy no longer
|
||||
speaks alone for ping; it keeps sole charge of ESP/GRE/IGMP and of the degraded
|
||||
paths.)*
|
||||
|
||||
**Fail-open degradations must be visible in the panel, not only in `logread`.** The
|
||||
audit deliberately converted many aborts into warn-and-continue (an unfetchable list,
|
||||
@@ -666,3 +670,535 @@ Neither is a substitute for the other. A new protocol in `shater/parse` +
|
||||
stays only so the router set remains a subset of the lx desktop set. Noted
|
||||
because "a tag that buys nothing" is the mirror image of this bug and should be
|
||||
removed deliberately, not silently.
|
||||
|
||||
## D24 — DNS interception is the DEFAULT (`dns_intercept=1`), not an opt-in
|
||||
Decided 2026-07-26. `Globals.DNSIntercept` shipped as opt-in (`default false`, and
|
||||
absent from both `DefaultGlobals` and the shipped `/etc/config/shater`). The result
|
||||
was an **inverted** posture, which is the reason this is a decision and not a
|
||||
preference:
|
||||
|
||||
- a client with **standard** settings — DNS = the router's address, exactly what
|
||||
DHCP hands out — sent its queries to the router. The nft `:53` divert was behind
|
||||
the flag (`netplane/nft.go`), and the rule right after it is an unconditional
|
||||
`fib daddr type local accept`, so the query was delivered locally to dnsmasq and
|
||||
forwarded to the ISP **in the clear**: no blocklists, no per-device DNS rules,
|
||||
no Block-DoH, no resolver detour, nothing;
|
||||
- a client that hard-coded `8.8.8.8` "to bypass the router" was addressing a
|
||||
non-local IP and **was** caught by the ordinary tproxy catch-all.
|
||||
|
||||
The obedient client leaked; the evader did not. Meanwhile `FEATURES.md`, `README.md`
|
||||
and D14 all promised "no DNS leaks" and "dnsmasq never sees LAN queries" — true only
|
||||
for the traffic pattern the default did not cover. `dns_intercept` appeared nowhere
|
||||
in `docs-shater/` at all.
|
||||
|
||||
**Decision: `DNSIntercept` is seeded ON in `model.DefaultGlobals`, and the shipped
|
||||
`/etc/config/shater` carries an explicit `option dns_intercept '1'`.** Nothing about
|
||||
the interception MECHANISM changed — only which side of the switch is the default.
|
||||
|
||||
**`.lan` and the private PTR zones keep working, and that is a pre-existing part of
|
||||
the mechanism, not something bolted on for this flip.** `generate/dns.go` adds a
|
||||
synthetic DNS server (`shater-local-dns`, plain UDP to `127.0.0.1:53`, detour
|
||||
`direct`, so the daemon's own loop-mark keeps it out of the divert) and PREPENDS a
|
||||
`domain_suffix` rule for `lan` + the RFC6303 private reverse zones, ahead of every
|
||||
device/filter rule. Two honest limitations: it hardcodes `lan` (a router whose
|
||||
dnsmasq domain was changed needs a `config dns_rule` for the new suffix), and it
|
||||
only exists when the model has at least one `config resolver` — with none, buildDNS
|
||||
emits no DNS plane at all and the engine falls back to its built-in `local`
|
||||
transport, which reads `/etc/resolv.conf` (127.0.0.1 → dnsmasq), so local names
|
||||
still resolve but nothing is filtered.
|
||||
|
||||
**A dead engine does NOT black out the LAN's DNS.** This was the first thing checked,
|
||||
because "intercept everything" invites the reading "engine down = no DNS anywhere",
|
||||
and that is not what happens:
|
||||
- the fail-closed **holding plane** (D17, `RenderHoldNft`) hooks `forward` ONLY.
|
||||
A query addressed to the router is INPUT-hook traffic, so dnsmasq answers it as
|
||||
it always did — unfiltered and plaintext to the ISP. Deliberate: blocking it
|
||||
would also cut the daemon's own name resolution and with it any chance of
|
||||
self-recovery;
|
||||
- with the FULL plane loaded and the engine's tproxy socket gone, the `tproxy`
|
||||
statement returns `NFT_BREAK`, which aborts its own rule; the packet continues
|
||||
down the chain into the same `fib daddr type local accept` and reaches dnsmasq.
|
||||
|
||||
So the failure mode is a DNS **fail-open** (working, unfiltered) while client
|
||||
TRAFFIC stays fail-closed — and a query aimed at an EXTERNAL resolver is dropped
|
||||
with the rest of the forwarded traffic. Operators must know this: "the tunnel is
|
||||
down" does not mean "DNS is private".
|
||||
|
||||
**Existing installs.** `/etc/config/shater` is a conffile
|
||||
(`openwrt/shater-core/Makefile`), so an upgrade never replaces it:
|
||||
- a config that never mentioned the option (all of them, before this change) now
|
||||
parses over the ON seed and **starts intercepting on the next apply**. That is the
|
||||
intended behaviour change, and the only one this decision makes;
|
||||
- an explicit `option dns_intercept '0'` keeps winning. It survives the
|
||||
`WriteUCI→ReadUCI` round-trip because `render.go` emits booleans ALWAYS —
|
||||
the trap a default-true bool has and a default-false one does not: a value
|
||||
omitted at false would come back as the seed and silently re-enable itself.
|
||||
`shater/model/dnsintercept_test.go` pins both directions, plus the shipped file.
|
||||
|
||||
**Not done: silencing the "no resolvers configured" warning by shipping a resolver.**
|
||||
With interception on and no `config resolver`, generate warns — and it is right to:
|
||||
every client query now lands in an engine that has no resolver plane, so it is
|
||||
answered by the system resolver (dnsmasq → the ISP, in the clear) with filtering and
|
||||
anti-leak inert. Shipping a `type local` resolver would make the warning disappear
|
||||
while changing nothing about where the queries go: the panel would show a configured
|
||||
resolver and the operator would believe DNS was handled. That is the inverted lie
|
||||
this project keeps deleting. The warning stays; what it needs is the accurate
|
||||
wording (it currently claims `.lan` breaks, which the fallback above disproves), not
|
||||
a workaround. Note also that a fresh install ships INERT (`enabled '0'`) and
|
||||
`Reconcile` tears down instead of generating, so the warning cannot appear before the
|
||||
operator has enabled the stack — at which point it describes their live config.
|
||||
|
||||
**OPEN, and it gates shipping this default: the synthetic local server changes how
|
||||
proxy-endpoint DOMAINS are resolved.** Found while landing D24, reproduced on Linux
|
||||
with one resolver and a node addressed by a hostname:
|
||||
|
||||
- `common/dialer/dialer.go` resolves a domain server address through
|
||||
`route.default_domain_resolver`; when that is unset it uses
|
||||
`dnsTransport.Default()` — the engine's built-in `local` transport, i.e. a
|
||||
bootstrap-DIRECT lookup — but **only while fewer than two DNS transports exist**.
|
||||
With two or more and no default, it reports the `missing-domain-resolver`
|
||||
deprecation and leaves the query transport nil, so `dns.Router.Lookup` falls back
|
||||
to `lookupWithRules`: the CLIENT DNS plane.
|
||||
- `dns_intercept` adds `shater-local-dns`, which takes a single-resolver config from
|
||||
one transport to two. So a config whose only resolver is DoH-through-the-tunnel —
|
||||
the recommended anti-leak setup — would start resolving its own node's hostname
|
||||
through that same tunnel: a bootstrap loop where there was none.
|
||||
- Evidence: the same model emits no deprecation notice with `dns_intercept=0` and
|
||||
two `missing-domain-resolver` notices with `dns_intercept=1`;
|
||||
`generate.TestDNSFilterRemoteBlocklistHTTPClient` (Linux-only) fails on exactly
|
||||
that notice and is deliberately left failing rather than relaxed.
|
||||
|
||||
The fix belongs in `generate` (`route.go:160` already sets
|
||||
`route.default_domain_resolver` from `endpointResolver()`, which is opt-in and unset
|
||||
by default): when buildDNS emits the synthetic local server and no endpoint resolver
|
||||
is configured, `default_domain_resolver` must be pointed at a bootstrap-direct
|
||||
server, which restores exactly the pre-D24 behaviour and clears the notice. Until
|
||||
that lands, an operator can get the same result by setting `endpoint_resolver` to a
|
||||
direct resolver. Note the hazard is **not** created by D24 — any config with two
|
||||
resolvers has it today; the default merely makes it universal.
|
||||
|
||||
## D25 — L3 ingress: LAN ICMP rides a dedicated TUN through the tunnel, not a policy verdict
|
||||
Decided 2026-07-26. D17 made everything TPROXY cannot divert an explicit policy
|
||||
(`Globals.Untunnelable` = block | icmp | direct) — and its premise still holds:
|
||||
kernel TPROXY delivers a packet by handing it to a listening SOCKET, and sockets
|
||||
exist for TCP and UDP only, so an ICMP echo has nothing to be handed to. But a
|
||||
policy can only choose between losing the packet and leaking it with the
|
||||
client's real source address; neither ever puts a ping THROUGH the tunnel. This
|
||||
decision adds the data plane D17 could not have: **`globals.l3_tunnel` (opt-in,
|
||||
default off; `model.Globals.L3Tunnel`) opens a second, dedicated ingress — a TUN
|
||||
device — and LAN ICMP enters the engine as raw IP packets**, where the ordinary
|
||||
route rules pick an outbound exactly as for any flow. The policy is refined, not
|
||||
repealed: it keeps sole charge of the protocols the engine cannot ingest at all,
|
||||
and of the degraded paths (both below).
|
||||
|
||||
**The whole mechanism is one mark, one rule, one device — the TPROXY plane is
|
||||
untouched.** The nft prerouting chain stamps `L3Mark` (= fwmark_base + 0x80,
|
||||
`netplane/nft.go` `l3MarkOffset`) on LAN `ip protocol icmp` / `meta l4proto
|
||||
ipv6-icmp` ONLY, and only after every local plane was already accepted
|
||||
(fib-local, RFC1918/link-local/multicast daddr sets) and — for v6 — after a
|
||||
unicast ND/NA carve-out, because one tunnelled neighbour probe is enough to take
|
||||
the LAN's v6 plane down (`renderNft`, the L3 block). `addL3Routing`
|
||||
(`netplane/apply.go`) binds that mark to a table (= table_base + 0x08) whose
|
||||
only content is `default dev shater-l3`; del-then-add idempotent, and a failed
|
||||
rule or route is a NAMED operator warning, never an apply abort. `generate`
|
||||
emits the synthetic `l3-in` TUN inbound bound to exactly `netplane.L3Device`,
|
||||
MTU 65535 (the largest IP datagram there can be, so the KERNEL can never
|
||||
fragment on the way in — see "the device MTU is not a tunnel budget" below),
|
||||
point-to-point /30 + /126 addresses from private space,
|
||||
the v6 one only when `globals.ipv6` is on — and only next to a tproxy inbound:
|
||||
the ingress rides the same LAN divert plane, and without one the TUN would sit
|
||||
dark while the config claims ICMP is tunnelled, so it is skipped with a warning
|
||||
(`generate/inbound.go`, `appendL3TunInbound`). `shater/registry` registers the
|
||||
`tun` inbound type; that costs no new build tag and no meaningful size because
|
||||
`with_wireguard` already requires `with_gvisor` (D23, `scripts/router-tags.sh`).
|
||||
|
||||
**`auto_route: false` is load-bearing, not a default we happened to keep.**
|
||||
sing-box's auto_route rewrites the router's MAIN routing table — it would drag
|
||||
everything the router itself sends (WAN traffic, DNS, the tunnel's own underlay)
|
||||
into this TUN. The fwmark rule + dedicated table above is deliberately the ONLY
|
||||
entrance, and disabling the feature can never strand a stale default route in
|
||||
main (`generate/inbound.go`; pinned by `TestL3TunnelEmitsTunInbound`).
|
||||
|
||||
**`stack: "gvisor"` is a deliberate choice, and the tempting reason for it is
|
||||
wrong.** It is TRUE that sing-tun's system stack answers an ICMP echo LOCALLY —
|
||||
`processIPv4ICMP` rewrites Echo→EchoReply in place and swaps the addresses
|
||||
(sing-tun `stack_system.go:648`; the v6 twin sits right under it). It is FALSE
|
||||
that this makes the system stack unusable here: `dispatchIPv4`
|
||||
(`stack_system.go:355-372`) hands the packet to the SAME `ForwardDispatcher`
|
||||
first and only falls through to that forger for packets addressed to the TUN
|
||||
itself, exactly as the gVisor filter does (`stack_gvisor_filter.go:52-113`).
|
||||
Both stacks would forward. gvisor is chosen because it is already linked —
|
||||
`with_wireguard` requires `with_gvisor` (D23), so it costs no tag and no new
|
||||
code path — and because it is the combination the integration test actually
|
||||
exercises. Do not re-derive this as "the system stack fakes ping": it fakes ping
|
||||
only where the dispatcher declined the packet.
|
||||
|
||||
**The ceiling is ICMP echo, and it is upstream's dispatcher — NOT the netstack.**
|
||||
This distinction matters because the netstack answer is the intuitive one and it
|
||||
is wrong. On the forward path a WireGuard/AWG endpoint never consults gVisor at
|
||||
all: `Endpoint.WritePackets` (`transport/wireguard/port.go:21-58`) reads the IP
|
||||
version and the destination address and hands the raw bytes to
|
||||
`wgDevice.InputPackets` — the protocol byte is never examined — and
|
||||
`returnDeviceWrapper.Write` (`:127-157`) offers every decrypted packet to
|
||||
`returnPath.ReturnPackets` before the stack sees it. WireGuard would carry ESP
|
||||
today if anything handed it one. What refuses is `ForwardDispatcher`: its parser
|
||||
sets `hasFlow` for TCP, UDP and ICMP echo alone (`flow_parse.go`,
|
||||
`parseTransport`, the echo identifier serving as the pseudo-port), and
|
||||
`createFlow` NATs through a port-shaped selector (`flow_dispatch.go:325`,
|
||||
`allocateSelector`) that ESP, AH and GRE do not have. So ESP/AH/GRE/IGMP/SCTP
|
||||
cannot enter the engine in ANY configuration and REMAIN on the D17 policy —
|
||||
or on the kernel egress of D26, which sidesteps the dispatcher entirely. The nft
|
||||
plane encodes the same boundary on purpose: it marks `icmp`/`ipv6-icmp` only,
|
||||
never `l4proto != { tcp, udp }`, because a marked ESP packet would enter the
|
||||
device and vanish — a black hole wearing a tunnel's name — instead of receiving
|
||||
the policy's honest verdict (`netplane/nft.go`, the prerouting L3 comment).
|
||||
|
||||
**What works and what does not, read off the upstream source.** ping v4/v6 —
|
||||
yes. Windows `tracert` — yes: the gVisor return path recognises
|
||||
`ICMPv4TimeExceeded` and `ICMPv4DstUnreachable` alongside EchoReply and NATs
|
||||
them back to the LAN client (`stack_gvisor_icmp.go:341+`, `returnPacket`). IPv6
|
||||
traceroute — intermediate hops stay invisible: the v6 branch of the same
|
||||
function accepts EchoReply only, so just the final destination answers. Several
|
||||
LAN clients behind the one tunnel address are already solved upstream:
|
||||
`ForwardDispatcher` NATs by echo identifier and rewrites the source to the
|
||||
outbound's port address (`flow_dispatch.go:325+`, `createFlow`; `icmpFlowKey`) —
|
||||
we wrote no NAT of our own.
|
||||
|
||||
**Which outbounds can carry it.** The contract is `adapter.FlowOutbound`
|
||||
(= `Outbound` + `tun.Port` + `PreMatchFlow`, `adapter/outbound.go`). In-tree
|
||||
implementors: the WireGuard/AWG endpoint (`protocol/wireguard`), `direct`
|
||||
(`protocol/direct`), `bridge` (`protocol/bridge`), `tailscale`
|
||||
(`protocol/tailscale`). Of those, the shaterd registry can construct only
|
||||
WireGuard/AWG and direct (`shater/registry/registry.go` — bridge and tailscale
|
||||
are not registered). Every proxy protocol — vless/vmess/trojan/shadowsocks/
|
||||
hysteria2/tuic/socks/http/shadowtls — is L4-only and cannot. Recorded as a known
|
||||
gap: `masque` is L3 by nature (CONNECT-IP; it builds a userspace gVisor stack
|
||||
per tunnel, `protocol/masque/outbound.go`) but implements no `tun.Port` and is
|
||||
not in the shater registry, so today it cannot carry the ingress. Wiring it up
|
||||
is possible future work, not a promise.
|
||||
|
||||
**ICMP to an L4-only outbound is DROPPED, and that took patching upstream files
|
||||
(the `lx:l3-honest-drop` delta — see `docs-lx/lx-changelog.md`).** In the gVisor
|
||||
stack the fallthrough verdict is a forgery: `ICMPForwarder.HandlePacket` answers
|
||||
the echo ITSELF (Echo→EchoReply + address swap) whenever the flow judgment comes
|
||||
back Accept (`stack_gvisor_icmp.go:120`), and upstream maps "no flow route" to
|
||||
exactly that Accept — so a ping routed to vless would read as tunnelled while
|
||||
the packet died on the router. Two small marked hunks make the truth observable:
|
||||
`route/route.go` wraps the whole pre-match walk — the walk itself became
|
||||
`preMatch`, and the exported `PreMatch` is now a FUNNEL that rewrites
|
||||
`PreMatchContinue` and `PreMatchBypass` to `PreMatchDrop` for `N.NetworkICMP` —
|
||||
and `adapter/router.go` (`JudgeFlow`, the `!isPort` branch) returns `ActionDrop`
|
||||
for ICMP where it fell through to `ActionAccept` — the second line of defense,
|
||||
because `FlowOutbound` and `tun.Port` are distinct interfaces and a drift
|
||||
between them must not quietly re-enable the forger. TCP/UDP verdicts are
|
||||
byte-identical; `route/prematch_icmp_lx_test.go` and
|
||||
`adapter/judgeflow_icmp_lx_test.go` pin both directions. The operator-facing
|
||||
text says the same out loud (`shater/apply/warnings.go`): proxy-routed addresses
|
||||
"cannot be pinged at all — deliberately".
|
||||
|
||||
> **Why a funnel and not an override inside the walk.** The first version of
|
||||
> this delta overrode the pre-declared `continueResult` inside `preMatchFlow`
|
||||
> and claimed to cover "every exit point of the function at once". It covered
|
||||
> every exit of THAT function; the walk above it has exits of its own that never
|
||||
> reach it — the `prepareMatchMetadata` error return (which arrived later, with
|
||||
> the shared-metadata refactor, upstream `b911fb078`), the sniff bail-outs, and
|
||||
> the `default:` arm of the rule-action switch, which catches every action
|
||||
> pre-match has no arm for (`hijack-dns`, `direct`, and whatever upstream adds
|
||||
> next). Each of those returned `PreMatchContinue`, i.e. `tun.ActionAccept`,
|
||||
> i.e. the forged reply. A guard on the single return value cannot be outgrown
|
||||
> by a new exit. `PreMatchBypass` joined the drop for the same reason: sing-tun
|
||||
> implements `ActionBypass` on the nfqueue plane only — the name appears nowhere
|
||||
> in `flow_dispatch.go` or `stack_gvisor_icmp.go` — so on the TUN path it lands
|
||||
> in the same `default:` arm as Accept and forges too. There is no honest bypass
|
||||
> for a packet that is already inside the engine's TUN.
|
||||
|
||||
**The device MTU is NOT a tunnel budget, and pretending it was manufactured
|
||||
forged replies.** `l3-in` is created with MTU **65535**, not the tunnel's 1420,
|
||||
and the maximum is the whole argument. This MTU governs exactly one thing:
|
||||
whether the KERNEL splits a packet on its way INTO the device. What the engine
|
||||
then puts into the tunnel is sized separately and correctly, against the
|
||||
OUTBOUND's MTU — `ForwardDispatcher.forwardToPort` (`flow_dispatch.go:445-481`)
|
||||
measures every forwarded packet against `Port.PortMTU()` and either fragments to
|
||||
it (no DF, `fragmentIPv4Packet`) or answers a well-formed `fragmentation needed`
|
||||
quoting it (DF, `buildFragmentationNeeded`, source = the far host, so PMTU
|
||||
discovery works end to end). That machinery was always there; it was simply
|
||||
never handed a whole packet.
|
||||
|
||||
At 1420 it wasn't. Anything above 1392 bytes of payload was fragmented by the
|
||||
kernel at this device, and a fragment is the one thing sing-tun will not judge:
|
||||
`Dispatch` (`flow_dispatch.go:176-177`) returns on `parsed.fragment` BEFORE
|
||||
calling `JudgeFlow` at all. The fragments fell through to the gVisor stack —
|
||||
promiscuous and spoofing (`stack_gvisor.go:219-223`) — which reassembled them
|
||||
and handed the echo to `ICMPForwarder.HandlePacket` (`stack_gvisor_icmp.go:105+`),
|
||||
whose `installFlow` (`:233-244`) writes to the port UNMODIFIED and therefore
|
||||
demands a port address that is valid **and UNSPECIFIED**. `direct` qualifies
|
||||
(`IPv4Unspecified()`); a WireGuard/AWG endpoint reports its concrete interface
|
||||
address (`transport/wireguard/port.go:13`) and does not. So it declined, and
|
||||
`HandlePacket` fell past the switch and FORGED the reply: `SetType(EchoReply)` +
|
||||
address swap. Net effect on the operator's bench: `ping -s 1392` honest,
|
||||
`ping -s 1393` a lie told by the router — and the lie was, of course, only for
|
||||
the outbounds this feature exists for. (Upstream applies the very same
|
||||
unspecified test and answers it honestly in the cloudflared ICMP handler,
|
||||
`protocol/cloudflare/inbound.go:163-167`: it drops. Only the TUN path forges.)
|
||||
|
||||
65535 rather than "big enough": no IP datagram can exceed it, so the kernel
|
||||
CANNOT fragment at this device, for any packet, ever. Any smaller value leaves
|
||||
a band of sizes open and re-opens the class. It is also sing-box's own default
|
||||
TUN MTU on Linux. Pinned by `TestL3TunnelMTULeavesNothingForTheKernelToFragment`
|
||||
and `TestL3TunnelMTUIsNotATunnelBudget` (`generate/l3mtu_test.go`), and — the
|
||||
assertion that matters — by the integration test reading the MTU back off the
|
||||
real kernel device, since a kernel that clamped it would restore the forgery
|
||||
without changing a generated byte.
|
||||
|
||||
Memory was MEASURED, not reasoned about: three paired runs of
|
||||
`TestIntegrationL3TunInboundStarts` under `-test.memprofilerate=1` (exact
|
||||
accounting, not sampled) allocate 5.41 / 5.48 / 5.47 MB at 65535 against
|
||||
5.76 / 5.46 / 5.70 MB at 1420, and a `-diff_base` profile attributes every
|
||||
difference to netlink interface enumeration. Nothing in the read path scales
|
||||
with the MTU: gVisor reads through `fdbased.BufConfig`, which sing-tun's `init`
|
||||
pins to a single 65535-byte view regardless of MTU, and `fdbased` keeps `mtu`
|
||||
only to return it from `MTU()`. Two adjacent facts, recorded because both are
|
||||
easy to derive wrongly: (a) `protocol/tun` computes
|
||||
`enableGSO = stack == gvisor && mtu < 49152`, so this MTU turns GSO off there —
|
||||
and then `StartStateStart` turns it back ON unconditionally because an
|
||||
`adapter.FlowOutbound` exists in the config, so the ~1.98 MB of TCP/UDP GRO
|
||||
scaffolding is present at BOTH MTUs and is priced by the flow-capable outbound,
|
||||
not by this number; (b) the `mtu_fix` on the `shater_l3` fw4 zone is now inert —
|
||||
only ICMP is ever marked into the device — and its uci-defaults comment still
|
||||
says "the tunnel MTU is 1420".
|
||||
|
||||
**What is still NOT covered, said plainly.**
|
||||
|
||||
1. **A big ping does not start WORKING — it starts FAILING HONESTLY.** Upstream's
|
||||
ICMP NAT is unfragmented-only in BOTH directions: `classifyReturn`
|
||||
(`flow_dispatch.go:703-710`) returns `returnPass` on `parsed.fragment` exactly
|
||||
as the forward path does. So a non-DF `ping -s 2000` now genuinely leaves the
|
||||
router (fragmented to the tunnel MTU by `forwardToPort`), the far host really
|
||||
answers, and the reply — fragmented by the peer to fit the tunnel — is not
|
||||
NAT'd back to the LAN client. The operator sees a timeout. That is the
|
||||
feature's promise ("travels or fails honestly"), not a capability claim.
|
||||
Carrying oversized ICMP end to end would need reassembly upstream does not
|
||||
have; it is not planned.
|
||||
2. **A client that puts fragments on the wire ITSELF.** The device MTU cannot
|
||||
un-fragment what already arrived fragmented, so such packets still reach the
|
||||
gVisor stack, still get reassembled there, and still receive a forged reply
|
||||
when the outbound is WireGuard/AWG. This is the residue the planned
|
||||
`ip frag-off & 0x3fff != 0` prerouting carve-out (`netplane/nft.go`) is for.
|
||||
**Whoever writes that rule must first check whether it can ever match:** fw4's
|
||||
ruleset uses conntrack, conntrack pulls in `nf_defrag_ipv4`/`nf_defrag_ipv6`,
|
||||
and defrag REASSEMBLES in PREROUTING before our marking rules run. Where
|
||||
defrag is active the case does not arise (the MTU covers it) and the rule is
|
||||
dead; where it is not, the rule is the only cover. Verify on the bench with
|
||||
`nft list ruleset | grep -c ct` and a fragment counter, do not assume.
|
||||
3. **The DF path changed hands and is untested on hardware.** It used to be the
|
||||
kernel that answered `fragmentation needed` (from the router's LAN address,
|
||||
MTU 1420); it is now the engine (from the far host's address, quoting
|
||||
`Port.PortMTU()`). Both are correct PMTUD; only the first has ever run on a
|
||||
real router.
|
||||
**fw4 has to be told about the device, and `list device` is the only spelling
|
||||
that works.** nftables runs EVERY table on every packet and a drop in any one of
|
||||
them wins — an accept in `inet shater` cannot override fw4, and fw4 WILL reject
|
||||
this forward: netifd never learns about a device the daemon creates at runtime,
|
||||
so `shater-l3` belongs to no zone and falls into fw4's zone-less defaults. Hence
|
||||
a real fw4 zone `shater_l3` + a lan→shater_l3 forwarding, seeded idempotently
|
||||
(NAMED sections) and unconditionally in uci-defaults
|
||||
(`openwrt/shater-core/files/etc/uci-defaults/30_shater-core`, `seed_l3_zone`),
|
||||
with `mtu_fix` set. That `mtu_fix` is now inert and should be read as such: it
|
||||
clamps forwarded TCP MSS to the route MTU, the device MTU is 65535, and nothing
|
||||
but ICMP is ever marked into this device — the uci-defaults comment still says
|
||||
"the tunnel MTU is 1420" and is stale. The device is attached
|
||||
via `list device`, deliberately NOT `list network`: fw4 resolves a zone's
|
||||
networks through netifd, which yields an EMPTY device set for a runtime-created
|
||||
TUN (a proto-none stub would have to be brought UP to contribute an l3_device,
|
||||
and nothing ever brings it up), while `list device` compiles to a plain
|
||||
iifname/oifname string match — valid before the TUN exists, matching from the
|
||||
moment shaterd creates it. `kmod-tun` joined DEPENDS so a slimmed image cannot
|
||||
lose `/dev/net/tun` (`openwrt/shater-core/Makefile`). Our own forward chain
|
||||
accepts both TUN legs ahead of the fail-closed drops — accepts that speak for
|
||||
OUR table only (`netplane/nft.go`, forward chain step 4).
|
||||
|
||||
**What the policy still owns, and the one combination that now warns.** With the
|
||||
ingress on, the mark is stamped in prerouting and the ROUTING decision carries
|
||||
echo into the TUN before the forward chain — where the policy's verdicts live —
|
||||
is ever consulted; that holds under every `untunnelable` value. The policy
|
||||
therefore governs exactly two things: the never-markable protocols above, and
|
||||
the fallback when the L3 rule/route did not come up (engine down, partial apply)
|
||||
— `block` turns that failure into an honest loss, `direct` into a silent leak
|
||||
with the real address. That is why `l3_tunnel` + `untunnelable=direct` draws a
|
||||
validation warning naming the safe choice (`model/validate.go`), why every
|
||||
rule/route failure surfaces as a named panel warning rather than an abort
|
||||
(`addL3Routing`), and why the D17 HOLDING plane never marks: the TUN is created
|
||||
BY the engine, and the holding plane exists precisely because the engine is not
|
||||
running — marking would dead-end ping in a device that does not exist
|
||||
(`netplane/nft.go`, hold comment).
|
||||
|
||||
- **Rejected: `auto_route` / letting the engine own the routing.** It rewrites
|
||||
the main table and intercepts the router's own WAN/DNS/underlay traffic; the
|
||||
blast radius of a toggle meant for LAN ping would be the whole router.
|
||||
- **Rejected: marking all `l4proto != { tcp, udp }` into the TUN.** ESP/AH/GRE/
|
||||
IGMP/SCTP cannot be parsed into flows upstream; they would vanish inside the
|
||||
device. A drop with a name (the policy's) beats a silent black hole.
|
||||
- **Rejected: keeping upstream's accept-and-forge for unroutable ICMP.** A ping
|
||||
that "works" without leaving the router is the inverted lie this project keeps
|
||||
deleting (D17's fiction purge, D23's dead WireGuard, D24's obedient-client
|
||||
leak).
|
||||
|
||||
**Proven, and not proven, said plainly.** The cold start is PROVEN, not assumed:
|
||||
`TestIntegrationL3TunInboundStarts`
|
||||
(`shater/generate/l3_integration_linux_test.go`, run as root with NET_ADMIN and
|
||||
`/dev/net/tun`, PASS) drives an `l3_tunnel=1` config through the SLIM registry
|
||||
(`registry.Context`, not upstream's `include.Context`) under the shipped router
|
||||
tag set: `box.New` + `Start` accept it, the kernel really ends up with the
|
||||
`shater-l3` device at the contract MTU 65535 — the assertion the value exists
|
||||
for, since a kernel that clamped it would silently restore the forged-reply
|
||||
band — and Close removes it; precisely
|
||||
the "built with X, verified with Y" gap class D23 exists for (a lost
|
||||
`tun.RegisterInbound` or a trimmed `with_gvisor` changes no generated byte and
|
||||
would otherwise surface only on the operator's router). Each layer contract is
|
||||
pinned besides (`generate` `TestL3Tunnel*`, `netplane` `TestL3Ingress*`,
|
||||
`route/prematch_icmp_lx_test.go`). Exactly two things remain UNVERIFIED:
|
||||
(a) the end-to-end path on live hardware — LAN client → prerouting mark →
|
||||
ip rule → TUN → WireGuard peer → reply back to the client — has not been
|
||||
exercised on a real router; (b) the steady-state memory cost of the second
|
||||
gVisor netstack (the `l3-in` TUN beside the WireGuard endpoint's own) is
|
||||
unmeasured on the target hardware. An indicative figure exists and is only
|
||||
that: on x86_64 in a container, idle and carrying no flows, peak RSS of a
|
||||
process that brought the same engine up went from ~26.0-26.8 MB without
|
||||
`l3_tunnel` to ~28.3-28.7 MB with it over three paired runs — about +2.2 MB.
|
||||
That was measured on a throwaway harness, not on aarch64, not under load, and
|
||||
with an empty ICMP NAT table, so it bounds nothing on the router. Neither
|
||||
item is folded into any claim above.
|
||||
|
||||
Consequence: a ping from the LAN either genuinely travels through the tunnel
|
||||
(WireGuard/AWG, direct) or fails honestly, at every size the router itself can
|
||||
put into the device — and a router that never opts in renders the pre-feature
|
||||
plane byte-for-byte (`TestL3IngressOptIn` pins the off-state render). Read
|
||||
"fails honestly" strictly: above the tunnel MTU a non-DF ping now leaves the
|
||||
router for real and then times out, because upstream's ICMP NAT does not carry
|
||||
fragments back either. The one qualifier left is item 2 above — a client that
|
||||
puts fragments on the wire ITSELF, on a router where conntrack defrag is not
|
||||
reassembling them first. This paragraph has been overclaimed twice already;
|
||||
extend it only against a bench result, never against a reading.
|
||||
|
||||
## D26 — What the engine cannot carry, the kernel carries: `untunnelable_egress`
|
||||
Decided 2026-07-26. D25 ended with ESP/AH/GRE/IGMP/SCTP still owned by the D17
|
||||
policy — that is, with a choice between dropping them and leaking them out the
|
||||
WAN, never a data plane. This decision gives them one, and deliberately NOT
|
||||
ours: **`globals.untunnelable_egress` (default empty;
|
||||
`model.Globals.UntunnelableEgress`) names an existing egress of type
|
||||
interface/tunnel, and LAN traffic that is neither TCP nor UDP is stamped in
|
||||
prerouting with that egress's own mark, so the KERNEL routes it out that
|
||||
egress's device with the kernel's own NAT.** No proxy, no engine, no userspace
|
||||
stack ever touches the packet — which is exactly why every protocol works.
|
||||
|
||||
**Where the engine's boundary actually is — recorded so nobody digs for it
|
||||
twice.** It is NOT the gVisor stack, and it is not WireGuard: on the forward
|
||||
path the WG/AWG endpoint never consults gVisor at all. `Endpoint.WritePackets`
|
||||
(`transport/wireguard/port.go:21-58`) takes the raw IP packet bytes, reads
|
||||
exactly the IP version and the destination address, and hands
|
||||
`device.InputPacketRef`s to `wgDevice.InputPackets` — the protocol byte is
|
||||
never read; on the way back (`port.go:127-157`) `returnDeviceWrapper.Write`
|
||||
offers every decrypted packet to `returnPath.ReturnPackets` first and only the
|
||||
unconsumed remainder falls through to the gVisor device. gVisor serves
|
||||
`DialContext`/`ListenPacket` — traffic the ENGINE originates — while forwarded
|
||||
traffic bypasses the stack in both directions, indifferent to protocol. The
|
||||
real ceiling sits one step earlier, in sing-tun's `ForwardDispatcher`:
|
||||
`parseTransport` (`flow_parse.go:106-153`) sets `hasFlow` for exactly TCP, UDP,
|
||||
ICMPv4 Echo/EchoReply and ICMPv6 EchoRequest/EchoReply — a packet of any other
|
||||
protocol is never dispatched as a flow — and `createFlow`
|
||||
(`flow_dispatch.go:325`) builds its NAT through
|
||||
`allocateSelector(packet.protocol, …, packet.source.Port())` (line 355), which
|
||||
needs a port-like selector that ESP/AH/GRE simply do not have (SCTP has ports,
|
||||
but the parser above never grants it a flow either). Tailscale documents the
|
||||
same frontier for its own userspace mode — "Any IP protocol other than TCP or
|
||||
UDP (such as SCTP) is not supported in userspace mode… All IP protocols are
|
||||
supported" in kernel mode
|
||||
(https://tailscale.com/docs/reference/kernel-vs-userspace-routers) — useful as
|
||||
external corroboration of where userspace data planes generally end, though OUR
|
||||
boundary is the dispatcher, not the stack. The kernel egress was therefore
|
||||
chosen not because userspace "cannot" in principle, but because the kernel
|
||||
delivers all protocols with zero new code on the hot path.
|
||||
|
||||
**The mechanism already existed; the feature is one binding and one marking
|
||||
step.** `addEgressRouting` (`netplane/apply.go`) has always installed, for
|
||||
every interface/tunnel egress, an `ip rule fwmark <EgressMark> lookup
|
||||
<EgressTable>` plus a `default dev <device>` route in that table — per-rule
|
||||
egress selection rides on it. The only missing piece was that nothing ever
|
||||
marked non-TCP/UDP traffic: `untunnelable=direct` merely ACCEPTED it in the
|
||||
forward chain, so it left over the main table, i.e. the WAN.
|
||||
`UntunnelableEgressBinding` (`netplane/nft.go`) resolves the option to the
|
||||
egress's index, its OWN mark and its OWN device — deliberately no third
|
||||
mark/table pair to keep coherent — and the prerouting chain stamps that mark on
|
||||
the untunnelable protocols. A name that does not resolve to an interface/tunnel
|
||||
egress with a device renders nothing and is reported: the D17 policy stays in
|
||||
sole charge, which is the fail-closed reading of a typo.
|
||||
|
||||
**Why `l4proto != { tcp, udp }` is safe here when D25 banned it.** D25 rejected
|
||||
the broad filter because the receiving side was the `ForwardDispatcher`, which
|
||||
classifies nothing beyond TCP/UDP/ICMP echo — a marked ESP packet would enter
|
||||
the TUN and vanish, a black hole wearing a tunnel's name. Here the receiving
|
||||
side is the kernel, which forwards ANY IP protocol and NATs what it has
|
||||
machinery for: SCTP carries ports and NATs like TCP/UDP; GRE is NATed only
|
||||
through the PPTP helper keyed on the call-id — the kernel's own comment calls
|
||||
GRE "generally not very suited for NAT, as it has no protocol-specific part as
|
||||
port numbers" (`net/netfilter/nf_conntrack_proto_gre.c`); ESP/AH pass as plain
|
||||
routed IP. Nothing on this path can silently swallow a protocol it does not
|
||||
understand, which was the entire objection.
|
||||
|
||||
**Order against D25: the L3 ingress claims ICMP first.** With `l3_tunnel` on,
|
||||
LAN ICMP is marked into the engine's TUN before the egress carrier is consulted
|
||||
— the engine path routes ping by the operator's rules, which a kernel egress
|
||||
cannot do — and only the remaining protocols go to the egress. With `l3_tunnel`
|
||||
off, ICMP goes to the egress with everything else. In both shapes marked
|
||||
traffic is settled by ROUTING before the forward chain speaks, so the D17
|
||||
policy now governs exactly the failure case — the rule or route that did not
|
||||
come up — the same division D25 already established for the L3 mark.
|
||||
|
||||
**What the feature refuses to promise — and the operator text refuses with it
|
||||
(`shater/apply/warnings.go`, the egress-carrier note).** (a) It is not a tunnel
|
||||
per se: the option accepts any interface/tunnel egress, and on the target
|
||||
routers a WireGuard device is the exception (`kmod-wireguard` is usually
|
||||
absent) while a second WAN is routine. Through a WireGuard egress this
|
||||
genuinely is a tunnel; through a second WAN it is simply another uplink, and
|
||||
the destination sees that uplink's real address. No text, comment or doc line
|
||||
may call it a tunnel unconditionally. (b) It does not revive IPTV: IGMP is
|
||||
LAN-side multicast group management, WireGuard is L3 point-to-point and carries
|
||||
no multicast, and multicast never crossed this router under any setting —
|
||||
routing IGMP out an egress restores nothing, and no wording may hint otherwise.
|
||||
(c) IPsec through NAT-T never needed it: RFC 3948 encapsulates ESP in UDP/4500,
|
||||
so a modern IPsec client behind NAT is ordinary UDP that already follows the
|
||||
routing rules; the raw-ESP case this feature carries is the no-NAT-T remainder.
|
||||
|
||||
- **Rejected: teaching the engine these protocols.** Extending `parseTransport`
|
||||
and the selector NAT upstream would be new hot-path code in an
|
||||
actively-maintained adversarial area, for protocols the kernel already
|
||||
forwards for free — and for ESP/AH/GRE there is no port-like selector to NAT
|
||||
by in the first place.
|
||||
- **Rejected 2026-07-26: carrying them through the userspace AWG endpoint
|
||||
site-to-site, with no NAT at all.** This is the alternative the "no port-like
|
||||
selector" line above does NOT dispose of, and it is written down because the
|
||||
obvious reading of that line — "impossible" — is wrong and would be
|
||||
re-derived. The endpoint is already protocol-blind in both directions
|
||||
(`transport/wireguard/port.go:21-58`, `:127-157`), so an ESP packet could be
|
||||
forwarded UNTOUCHED, keeping the LAN client's own source address, and the
|
||||
reply would come back addressed to that client and need only be written to the
|
||||
TUN. No selector, no NAT, every protocol. It needs two things we declined to
|
||||
take on: lx-owned code in the forward hot path, bypassing `ForwardDispatcher`
|
||||
on both legs — precisely the surface CONSTITUTION §2 exists to keep small on
|
||||
an actively-maintained upstream — and a SERVER-side prerequisite (our LAN
|
||||
prefix in the peer's `AllowedIPs`, plus a route back), which turns a router
|
||||
option into a deployment contract. The kernel egress above buys the same
|
||||
protocols with zero hot-path code, so this stays a design on file, not a gap.
|
||||
- **Rejected: a dedicated mark/table pair for the carrier.** `addEgressRouting`
|
||||
already binds `EgressMark`/`EgressTable` to the device; a third pair would be
|
||||
a second copy of the same route that could drift from the first.
|
||||
|
||||
**Not verified, said plainly.** The end-to-end path — LAN client → prerouting
|
||||
mark → ip rule → egress device → far end and back — has not been exercised with
|
||||
real ESP or GRE on live hardware. Nothing above claims it has.
|
||||
|
||||
Consequence: raw IPsec, PPTP/GRE, SCTP — and ICMP when the L3 ingress is off —
|
||||
leave through an egress the operator explicitly named, under kernel routing and
|
||||
kernel NAT, instead of being dropped or silently leaking out the WAN; and with
|
||||
the option empty (the default) the plane renders byte-for-byte as before, with
|
||||
the D17 policy in sole charge.
|
||||
|
||||
@@ -12,6 +12,28 @@ usable release, **[T1]** next, **[T2]** later. Phases refer to `ROADMAP.md`.
|
||||
## Transparent proxying & routing
|
||||
- **[MVP]** TPROXY transparent proxy for multiple LAN interfaces (TCP + UDP), SNI/
|
||||
Host/QUIC sniffing.
|
||||
- **[MVP]** **L3 ingress for ICMP** (`globals.l3_tunnel`, opt-in, default off):
|
||||
LAN ping travels THROUGH the tunnel instead of being dropped or answered by a
|
||||
forged local reply. The engine opens a dedicated TUN (`shater-l3`, gVisor
|
||||
stack, `auto_route` off); nft marks LAN icmp/icmpv6 only and a scoped
|
||||
`ip rule` routes it in — the TPROXY plane and the main routing table stay
|
||||
untouched (D25). Carried only by L3-capable egresses (WireGuard/AmneziaWG,
|
||||
direct); ICMP routed to vless/vmess/… is honestly dropped, never faked.
|
||||
Ceiling is upstream sing-tun's: ICMP echo only — Windows tracert works, IPv6
|
||||
traceroute shows just the destination; ESP/AH/GRE/IGMP stay with the
|
||||
`untunnelable` policy (D17) unless `untunnelable_egress` carries them (D26).
|
||||
- **[MVP]** **Kernel egress for untunnelable protocols**
|
||||
(`globals.untunnelable_egress`, opt-in, default empty): names an existing
|
||||
interface/tunnel egress, and IPsec (ESP/AH), PPTP/GRE, SCTP — everything that
|
||||
is neither TCP nor UDP, plus ICMP when the L3 ingress is off — is routed out
|
||||
that egress's device by the KERNEL with kernel NAT, reusing the egress's own
|
||||
fwmark/table from `addEgressRouting`; the proxy never sees a byte, which is
|
||||
why every protocol works (D26). What that buys depends on the device: a
|
||||
WireGuard interface really is a tunnel, a second WAN is just another uplink
|
||||
whose real address the destination sees. It does not revive multicast IPTV,
|
||||
and UDP-based VPNs (WireGuard, OpenVPN-UDP, IPsec NAT-T) never needed it —
|
||||
they follow the routing rules as before. The `untunnelable` policy (D17)
|
||||
keeps only the failure case: a route that did not come up.
|
||||
- **[MVP]** First-match routing rules by source (IP/CIDR/MAC/interface/zone),
|
||||
destination, port, proto → target (outbound/selector/chain/direct/block) + egress.
|
||||
A rule names its **destination through a rule-set only** — a reusable named list
|
||||
@@ -40,6 +62,16 @@ usable release, **[T1]** next, **[T2]** later. Phases refer to `ROADMAP.md`.
|
||||
type=fakeip + pool — there is no global "FakeIP mode"); no DNS leaks. Routing
|
||||
is decided by in-engine rule-sets — the v0.1 dnsmasq→nftset population
|
||||
mechanism does not exist in v0.2 (see generate/dns.go).
|
||||
The hijack covers the queries a client sends **to the router itself** — the
|
||||
address DHCP hands out — because `globals.dns_intercept` is **ON by default**
|
||||
(D24). With it off, those queries go to dnsmasq and out to the ISP in the clear,
|
||||
so the well-behaved client leaks while the one that hard-codes 8.8.8.8 does not.
|
||||
`.lan` and the private PTR zones are preserved through dnsmasq either way. Two
|
||||
things the promise does NOT cover, both by design: while the engine is DOWN the
|
||||
holding plane hooks `forward` only, so dnsmasq still answers router-addressed
|
||||
:53 unfiltered (client traffic and DNS to external resolvers stay blocked); and
|
||||
with no `config resolver` at all there is no DNS plane to filter with — queries
|
||||
fall through to the system resolver and generate says so.
|
||||
- **[MVP]** Client DoT/DoH blocking (stop devices bypassing the filter).
|
||||
- **[MVP]** **Blocklists** with **flexible sources**: `inline` (type your own) /
|
||||
`file` / `url` (auto-update) / `geosite` category (only when geodata present).
|
||||
|
||||
@@ -186,6 +186,38 @@ daemon (`shaterd run`), which owns the engine, the `inet shater` data plane, pol
|
||||
routing, in-process DNS, and the admin panel (default `:8088`). The LuCI app's
|
||||
"Open panel" button mints a single-use token and hands the browser off to the panel.
|
||||
|
||||
### What enabling does to DNS
|
||||
|
||||
From the first apply, **every** LAN plaintext `:53` goes into the engine — including
|
||||
the queries a client sends to the router's own address, which is what DHCP hands out.
|
||||
That is `globals.dns_intercept`, and it is **on by default** (D24); without it those
|
||||
queries reach dnsmasq and the ISP unfiltered, i.e. the client with default settings
|
||||
leaks while the one that hard-coded `8.8.8.8` does not. What follows from it:
|
||||
|
||||
- `.lan` and private reverse (PTR) lookups still go to dnsmasq — the engine gets a
|
||||
rule for those suffixes. If you renamed dnsmasq's domain away from `lan`, add a
|
||||
`config dns_rule` for the new suffix.
|
||||
- Configure at least one `config resolver`. With none, the engine has no resolver
|
||||
plane: intercepted queries fall through to the system resolver (dnsmasq → your
|
||||
ISP, in the clear), blocklists and per-device DNS rules are inert, and the apply
|
||||
says so in its warnings.
|
||||
- While the engine is DOWN, DNS is **not** blacked out: the fail-closed holding
|
||||
plane hooks `forward` only, so dnsmasq keeps answering router-addressed `:53`
|
||||
(unfiltered, plaintext) while client traffic and DNS to external resolvers stay
|
||||
blocked. "The tunnel is down" is not "DNS is private".
|
||||
|
||||
To opt out, on the router:
|
||||
|
||||
```sh
|
||||
uci set shater.globals.dns_intercept=0
|
||||
uci commit shater
|
||||
shaterd apply
|
||||
```
|
||||
|
||||
Your `0` is kept: `/etc/config/shater` is a conffile (upgrades never replace it) and
|
||||
the daemon always writes the option back explicitly, so it is never re-enabled by a
|
||||
default.
|
||||
|
||||
## 5. The signed apk repo (the normal install path)
|
||||
|
||||
OpenWrt/ImmortalWrt **25.12** packages with Alpine's **apk**: `.apk` files, a
|
||||
|
||||
+34
-1
@@ -251,7 +251,40 @@ Apply/rollback: `apSnapshot` (run→last-good, nft→last-good.nft, route marks)
|
||||
> v0.2: "restart engine only on change" → config-hash gate + Close+New box (no reload).
|
||||
|
||||
### uci.go — `/etc/config/shater` schema
|
||||
- `config globals`: enabled, loglevel, kill_switch, dns_mode, ipv6, fwmark_base, table_base, confirm_timeout, resolver_default, resolver_fallback, probe_url, probe_interval, schema_version, active_profile.
|
||||
- `config globals` — the full option set, with the value used when the option is
|
||||
ABSENT (the `model.DefaultGlobals` seed). Booleans are always written back as
|
||||
`'1'`/`'0'` by `render.go`, so an explicit value never decays into the seed:
|
||||
|
||||
| option | default | meaning |
|
||||
|---|---|---|
|
||||
| `enabled` | `0` as shipped | master switch; `0` ⇒ `Reconcile` tears the stack down instead of applying |
|
||||
| `loglevel` (alias `log_level`) | `warning` | engine + daemon level; `none/off/silent/disabled` ⇒ log disabled, unknown ⇒ `warn` + a validation warning |
|
||||
| `log_syslog` / `log_file` / `log_persist` | `1` / `1` / `0` | operational log (`shater/logsink`): syslog, rotated file, and whether that file lives on flash instead of tmpfs |
|
||||
| `log_max_kb` | `2048` | size cap of the log file, clamped to 128…8192; `0` = "use the default", not "off" |
|
||||
| `kill_switch` | `closed` | `closed` = fail-closed (block on engine loss, incl. a holding plane when the engine never started); `open` = plain routing |
|
||||
| `ipv6` | `1` | `0` drops LAN IPv6 in the forward chain instead of leaving it unproxied |
|
||||
| `fwmark_base` / `table_base` | `0x2000` | reserved fwmark / routing-table bases (must not collide with fw4 or other apps) |
|
||||
| `confirm_timeout` | `0` | seconds before an unconfirmed apply auto-rolls back; `0` = commit-confirm off |
|
||||
| `resolver_default` / `resolver_fallback` / `endpoint_resolver` | unset | `config resolver` names: the DNS catch-all, its failover chain, and the bootstrap-direct server that resolves proxy endpoint DOMAINS |
|
||||
| `probe_url` / `probe_interval` | engine defaults | the ONE instrument all health probing uses (D20 — there are no per-group overrides) |
|
||||
| `panel_port` | `0` ⇒ `8088` | admin-panel HTTP port |
|
||||
| `dns_filter` | `0` | master enable of the blocklist/allowlist filter (D15); needs at least one `config resolver` |
|
||||
| `dns_intercept` | **`1`** | force ALL LAN plaintext `:53` into the engine, INCLUDING queries addressed to the router itself. See D24 for why this is the default, what preserves `.lan`, and what happens while the engine is down |
|
||||
| `block_doh` | `0` | NXDOMAIN the known public DoH hostnames + the Firefox canary and reject `:443` to their IPs, so clients fall back to `:53` (which the engine catches) |
|
||||
| `group_health` | `1` | OUR background group probing (the observatory). Does not touch sing-box's own urltest inside a group |
|
||||
| `untunnelable` | `block` | policy for what TPROXY cannot carry (ICMP/IGMP/ESP/AH/GRE/SCTP): `block` \| `icmp` (echo out, rest dropped) \| `direct` (all out, bypassing the tunnel) |
|
||||
| `geo_provider` | unset = auto | `sagernet` \| `loyalsoldier` \| `metacubex` \| `custom`; auto = country codes from SagerNet, everything else from Loyalsoldier |
|
||||
| `geosite_url` / `geoip_url` | unset | `{category}` templates, honoured only when `geo_provider=custom` |
|
||||
| `geosite_index_url` / `geoip_index_url` | unset | git-trees URLs used to SUGGEST categories in the panel; empty = no suggestions |
|
||||
| `stats_backend` | `memory` | `off` (no aggregation at all) \| `memory` (RAM, lost on restart) \| `sqlite` (aggregates in RAM + query/connection log on disk) |
|
||||
| `stats_ring_size` / `stats_timeline_minutes` / `stats_max_domains` | `200` / `60` / `5000` | live-log length, sparkline minutes, domain-map cap. **`0` = UNLIMITED** (grows with traffic), which is why these three are always emitted |
|
||||
| `stats_disk_limit_mb` | `64` | on-disk cap of `stats.db`; only meaningful for `stats_backend=sqlite`; `0` = unlimited |
|
||||
| `stats_retention_disabled` | `0` | master switch that turns OFF all trimming/pruning — every aggregate then grows unbounded |
|
||||
| `schema_version` | `0` = pre-versioned | UCI schema revision; `shaterd migrate` writes `2` |
|
||||
| `active_profile` | unset | display bookkeeping: the last profile switched to |
|
||||
|
||||
Deleted options still parse (unknown keys are ignored) and drain out on the next
|
||||
render: `dns_mode` (D17 — fake-IP is a resolver TYPE), `sweep_interval` (D19).
|
||||
- `config inbound`: name, enabled, type, network, tproxy_port(12345), listen, port, auth, user, pass, target_addr, target_port, target_network, tcp, udp, sniff.
|
||||
- `config subscription`: name, enabled, url, update_interval, fetch_via(direct|proxy), ua, hwid, device_os, ver_os, device_model, list header, format, list include/exclude/filter_proto/filter_country, dedup, expire_alert_days.
|
||||
- `config node`: name, enabled, uri, mux, mux_concurrency, xudp_concurrency, xudp_udp443, sockopt_mark, tcp_fast_open, tcp_keepalive_idle.
|
||||
|
||||
@@ -40,6 +40,10 @@ define Package/shater-core
|
||||
# shaterd : the daemon our init supervises (`shaterd run`)
|
||||
# kmod-nft-tproxy : kernel TPROXY (shaterd emits the `inet shater` rules)
|
||||
# kmod-nft-socket : socket match used by the tproxy divert chain
|
||||
# kmod-tun : /dev/net/tun — the daemon opens the `shater-l3` TUN
|
||||
# for L3 ingress (globals.l3_tunnel); usually built-in
|
||||
# on stock images, but a slimmed image without it would
|
||||
# make the option fail with a cryptic open() error.
|
||||
# ip-full : `ip rule`/`ip route`/rt_tables for policy routing
|
||||
# nftables-json : shaterd shells out to `nft`, and netplane/stats.go
|
||||
# parses `nft -j list ...` — the JSON output only exists
|
||||
@@ -49,7 +53,7 @@ define Package/shater-core
|
||||
# ca-bundle : the daemon is CGO_ENABLED=0, so crypto/x509 has no
|
||||
# host cert fallback — without /etc/ssl/certs every
|
||||
# HTTPS subscription / .srs ruleset fetch fails.
|
||||
DEPENDS:=+shaterd +kmod-nft-tproxy +kmod-nft-socket +ip-full +nftables-json +ca-bundle
|
||||
DEPENDS:=+shaterd +kmod-nft-tproxy +kmod-nft-socket +kmod-tun +ip-full +nftables-json +ca-bundle
|
||||
PKGARCH:=all
|
||||
endef
|
||||
|
||||
@@ -80,6 +84,10 @@ define Package/shater-core/install
|
||||
$(INSTALL_DIR) $(1)/etc/init.d
|
||||
$(INSTALL_BIN) ./files/etc/init.d/shater $(1)/etc/init.d/shater
|
||||
$(INSTALL_BIN) ./files/etc/init.d/shater-cron $(1)/etc/init.d/shater-cron
|
||||
# START=21 one-shot that loads the persisted fail-closed plane before fw4's
|
||||
# `lan -> wan ACCEPT` can be the only thing on the box (the main init is
|
||||
# START=99, i.e. seconds of plaintext forwarding on every boot).
|
||||
$(INSTALL_BIN) ./files/etc/init.d/shater-armor $(1)/etc/init.d/shater-armor
|
||||
|
||||
$(INSTALL_DIR) $(1)/etc/hotplug.d/iface
|
||||
$(INSTALL_BIN) ./files/etc/hotplug.d/iface/99-shater $(1)/etc/hotplug.d/iface/99-shater
|
||||
|
||||
@@ -23,6 +23,32 @@ config globals 'globals'
|
||||
option kill_switch 'closed'
|
||||
# There is no dns_mode option: routing is decided by in-engine rule-sets and
|
||||
# fake-IP is a resolver type (`config resolver` with type=fakeip + pool).
|
||||
#
|
||||
# Force ALL LAN plaintext DNS (:53) into the engine, INCLUDING queries the
|
||||
# client sends to the router itself (the address DHCP hands out). ON by
|
||||
# default: with it off, a client using the router as its resolver is answered
|
||||
# by dnsmasq and forwarded to the ISP in the clear — no blocklists, no
|
||||
# per-device DNS rules, no resolver detour — while a client that hard-codes
|
||||
# 8.8.8.8 IS intercepted. The obedient client leaked; the evader did not.
|
||||
#
|
||||
# Set to '0' to opt out (dnsmasq answers router-addressed :53 again). Your
|
||||
# explicit value is never overwritten: this file is a conffile, and the daemon
|
||||
# always writes the option back as '1'/'0'.
|
||||
#
|
||||
# .lan and the private reverse (PTR) zones keep working: with at least one
|
||||
# `config resolver` present the engine gets a synthetic server pointed at
|
||||
# dnsmasq on 127.0.0.1:53 plus a rule that sends those suffixes to it; with no
|
||||
# resolver at all the engine falls back to the system resolver, which is
|
||||
# dnsmasq too. If you changed dnsmasq's domain away from `lan`, add a
|
||||
# `config dns_rule` for it (only `lan` + RFC6303 reverse zones are built in).
|
||||
#
|
||||
# While the engine is DOWN the LAN is NOT left without DNS: the fail-closed
|
||||
# holding plane hooks `forward` only, so dnsmasq still answers router-addressed
|
||||
# :53 — unfiltered and in the clear, the documented trade-off (blocking it
|
||||
# would also cut the daemon's own name resolution and its chance to recover).
|
||||
# Queries aimed at an EXTERNAL resolver are dropped with the rest of the LAN's
|
||||
# forwarded traffic.
|
||||
option dns_intercept '1'
|
||||
option ipv6 '1'
|
||||
# Reserved fwmark base and routing-table base (do not overlap fw4/other apps).
|
||||
option fwmark_base '0x2000'
|
||||
|
||||
@@ -33,6 +33,30 @@
|
||||
# be running. `start` raises ACTIVE_FLAG, `stop` clears it; hotplug/cron
|
||||
# reconcile ONLY while the flag is up, so an admin `stop` STICKS — no
|
||||
# background actor may resurrect interception behind a stopped daemon.
|
||||
# * BEING REPLACED IS NOT BEING SWITCHED OFF. `restart` and `reload` (which is
|
||||
# stop+start, i.e. every LuCI Save & Apply) both run through `stop`, and the
|
||||
# daemon's SIGTERM teardown removes the fail-closed table unconditionally — it
|
||||
# does not consult kill_switch at all. Between that teardown and the
|
||||
# successor's first apply the init GUARANTEES a gap: it waits for the old
|
||||
# process to exit (shater_wait_stopped), then runs `shaterd migrate`, then
|
||||
# starts a daemon that still has to build an engine. So a restart is announced
|
||||
# with RESTART_FLAG, which tells the outgoing daemon to leave the fail-closed
|
||||
# holding plane STANDING — apply.TeardownExiting swaps it in with one nft
|
||||
# transaction and then skips the delete, so the table is never absent, not even
|
||||
# for the 80-90 ms the old arm-after-teardown order measured. A real `stop`
|
||||
# raises no flag and therefore still means what it says.
|
||||
# (A package UPGRADE does not come through here at all on apk v3: shater-core's
|
||||
# script table is post-install / pre-deinstall / post-upgrade, with no
|
||||
# pre-upgrade, so default_prerm — and its `stop` — runs only on REMOVAL.)
|
||||
# * The FAIL-CLOSED PLANE MUST ALSO EXIST BEFORE THIS SCRIPT DOES. START=99 is
|
||||
# after fw4 (19) and netifd (20), so at every boot the LAN forwards to the WAN
|
||||
# in the clear for as long as it takes procd to decompress the daemon off
|
||||
# flash and get an engine up. /etc/init.d/shater-armor (START=21) loads
|
||||
# BOOT_ARMOR — a copy of the holding plane the daemon persists on every apply
|
||||
# — to close that window. This script owns the DISARM half, and it owns it
|
||||
# with a CLOSED LIST: an operator's `stop`, or a removal, and nothing else.
|
||||
# Powering the box down must not — `shutdown` reaches stop_service too, and it
|
||||
# is not a person switching the product off (see shater_stop_disarms).
|
||||
# * The engine must never be permanently abandoned while interception stands:
|
||||
# respawn retries are infinite (procd never gives up); a sustained-dead
|
||||
# daemon is additionally escalated by the shater-cron watchdog.
|
||||
@@ -50,11 +74,166 @@ ACTIVE_FLAG=/var/run/shater.active
|
||||
# Written by `shaterd run`; the single-owner token this init waits on so a
|
||||
# restart never overlaps a new data plane with the previous one's teardown.
|
||||
PIDFILE=/var/run/shaterd.pid
|
||||
# Raised around a restart/reload, read by the OUTGOING `shaterd run` at SIGTERM:
|
||||
# present => "you are being replaced, leave the fail-closed plane standing";
|
||||
# absent => "you are being switched off, take everything down". tmpfs, so a
|
||||
# power cut can never make the next boot look like a restart.
|
||||
RESTART_FLAG=/var/run/shater.restarting
|
||||
# The persisted fail-closed holding plane. Written by the daemon on every apply,
|
||||
# loaded by /etc/init.d/shater-armor at boot. Its PRESENCE is the arm token, so
|
||||
# removing it here is how a deliberate stop stops the next boot from blocking.
|
||||
BOOT_ARMOR=/etc/shater/boot.nft
|
||||
# Seconds `start` will wait for a predecessor to finish its teardown. Must be
|
||||
# >= term_timeout below (procd's hard cap on a predecessor's life after SIGTERM)
|
||||
# so we never give up while procd is still letting it shut down cleanly.
|
||||
STOP_WAIT_SECS=40
|
||||
|
||||
# WHICH ACTION rc.common was invoked with, frozen at source time.
|
||||
#
|
||||
# rc.common does, in this order:
|
||||
# initscript=$1; action=${2:-help}; shift 2; ...; . "$initscript"; $action "$@"
|
||||
# so `action` is ALREADY assigned when this file is sourced, and every action then
|
||||
# runs as a function in THAT SAME shell. MEASURED on the target (ImmortalWrt
|
||||
# 25.12.1 r37978) with a throwaway probe init script, not read off documentation:
|
||||
#
|
||||
# /etc/init.d/X restart -> stop_service action=[restart], start_service [restart]
|
||||
# /etc/init.d/X stop -> stop_service action=[stop]
|
||||
# /etc/init.d/X reload -> reload_service action=[reload]
|
||||
# `reboot` -> stop_service action=[SHUTDOWN] <-- see below
|
||||
# the boot after it -> start_service action=[boot]
|
||||
#
|
||||
# A previous probe reported this variable EMPTY and the emptiness was written up as
|
||||
# the defect. It was the probe: `sh -x /etc/init.d/shater restart` bypasses the
|
||||
# `#!/bin/sh /etc/rc.common` shebang, so rc.common never runs, never assigns
|
||||
# `action`, and the variable reads empty no matter what this file does.
|
||||
#
|
||||
# Frozen into our own variable because `action` is a short, generic name that other
|
||||
# framework helpers also use as a local; a snapshot taken before any function runs
|
||||
# cannot be shadowed later.
|
||||
SHATER_RC_ACTION="$action"
|
||||
|
||||
# --- what an action MEANS --------------------------------------------------
|
||||
#
|
||||
# THE BUG THESE TWO PREDICATES REPLACE (v0.2.17, measured on the live router).
|
||||
# The old stop_service was `case $action in restart|reload) keep;; *) DISARM;; esac`
|
||||
# — an open default that swept up every action nobody had enumerated. `reboot` is
|
||||
# one of them: procd runs the K-links with the action `shutdown`, so the shutdown
|
||||
# path deleted the arm token on the way down and the next boot had nothing to load.
|
||||
# The mechanism destroyed itself at exactly the moment it exists for. Instrument
|
||||
# reading from the router, one minute apart across a reboot:
|
||||
#
|
||||
# 13:28 /etc/shater/boot.nft present
|
||||
# ---- reboot (stop_service action=[shutdown] -> old `*` branch -> rm)
|
||||
# 18s at_S22: NO_TABLE armor_file=NO_FILE
|
||||
#
|
||||
# So both lists below are POSITIVE and CLOSED. An action nobody thought about —
|
||||
# `shutdown` above all, but also whatever a future procd invents — falls through
|
||||
# both and changes nothing. The default now fails in the recoverable direction: at
|
||||
# worst a boot arms when it need not have, which costs the second before the daemon
|
||||
# applies and is still gated by shater-armor's own four state refusals. The old
|
||||
# default failed in the direction of the plaintext window the feature was built to
|
||||
# close.
|
||||
#
|
||||
# They are predicates rather than an inline `case` so the test gate can execute the
|
||||
# real thing: it sources THIS FILE in /bin/sh and calls them with every action procd
|
||||
# actually uses (shater/cmd/shaterd/initscript_test.go). A comment claiming
|
||||
# `shutdown` is handled is what shipped last time.
|
||||
|
||||
# True only for the ONE action that means "the operator switched the product off".
|
||||
# Deliberately not `shutdown`: powering a router down is not turning a feature off.
|
||||
#
|
||||
# NOT sufficient on its own — see shater_stop_disarms. `stop` is also how the
|
||||
# package manager's plumbing reaches us, and a package manager is not a person.
|
||||
shater_action_disarms() {
|
||||
case "$1" in
|
||||
stop) return 0 ;;
|
||||
*) return 1 ;;
|
||||
esac
|
||||
}
|
||||
|
||||
# Is a package manager in the middle of a transaction RIGHT NOW?
|
||||
#
|
||||
# This is a state, read at the moment the decision is made, exactly like
|
||||
# shater-armor's four refusals — not a record of an event. The same question is
|
||||
# already asked (for the same reason: prerm/postinst plumbing is not a user
|
||||
# action) by the detached bring-up in /etc/uci-defaults/30_shater-core.
|
||||
shater_pkg_transaction() {
|
||||
pidof apk >/dev/null 2>&1 && return 0
|
||||
pidof opkg >/dev/null 2>&1 && return 0
|
||||
return 1
|
||||
}
|
||||
|
||||
# Is the main service still enabled at boot? Same glob, and for the same reason,
|
||||
# as shater-armor's own check: `/etc/init.d/shater enabled` would source procd.sh
|
||||
# and take a blocking flock, which is not something to do from inside a package
|
||||
# manager's transaction.
|
||||
shater_rc_enabled() {
|
||||
local f
|
||||
for f in /etc/rc.d/S[0-9][0-9]shater; do
|
||||
[ -e "$f" ] && return 0
|
||||
done
|
||||
return 1
|
||||
}
|
||||
|
||||
# THE ACTUAL DISARM DECISION.
|
||||
# $1 = action
|
||||
# $2 = 1 when a package transaction is in flight
|
||||
# $3 = 1 when the service is still enabled in rc.d
|
||||
# All three are passed in rather than read inside, so the gate can drive every
|
||||
# combination without a package manager or an /etc/rc.d.
|
||||
#
|
||||
# WHY IT IS NOT JUST THE ACTION. base-files' default_prerm runs, in this order:
|
||||
#
|
||||
# if [ "$PKG_UPGRADE" != "1" ]; then "$i" disable; fi
|
||||
# "$i" stop
|
||||
#
|
||||
# so a package manager reaches stop_service wearing the operator's clothes. Two
|
||||
# different intentions arrive as the same action, and the difference between them
|
||||
# is readable at the moment of the decision:
|
||||
#
|
||||
# REMOVAL — prerm has ALREADY run `disable`, so S99shater is gone. The product
|
||||
# is going away; the armor goes with it. (It is belt-and-braces even
|
||||
# so: shater-armor refuses to arm without that symlink, and the whole
|
||||
# init script is about to be deleted anyway.)
|
||||
# REPLACED — the service is still enabled, so something intends to bring it
|
||||
# back. That is not an operator switching anything off, and deleting
|
||||
# the armor here would leave the next boot unprotected. "The next
|
||||
# apply will rewrite it" is not an answer: the armor exists precisely
|
||||
# to cover a reboot, and a reboot between an update and the first
|
||||
# apply is how this product is deployed.
|
||||
#
|
||||
# MEASURED, because the paragraph above is about a path I got wrong once already.
|
||||
# On THIS target (apk-tools 3.0.5, ImmortalWrt 25.12.1) shater-core's script table
|
||||
# is post-install / pre-deinstall / post-upgrade, with NO pre-upgrade — so an apk
|
||||
# UPGRADE never executes default_prerm and never calls `stop` at all. Verified with
|
||||
# a real `apk fix --reinstall shater-core` while sampling the armor file: 245 625
|
||||
# samples, zero disappearances, even with this guard mutated off. The upgrade half
|
||||
# of this predicate is therefore defence-in-depth for a shape that is one
|
||||
# `pre-upgrade` script (or a returning opkg lane) away, NOT a fix for an observed
|
||||
# failure. The removal half is live today.
|
||||
shater_stop_disarms() {
|
||||
shater_action_disarms "$1" || return 1
|
||||
# No package manager involved => a person typed it. The escape hatch must work.
|
||||
[ "$2" = "1" ] || return 0
|
||||
# A package transaction that has NOT disabled the service is replacing it.
|
||||
[ "$3" = "1" ] && return 1
|
||||
return 0
|
||||
}
|
||||
|
||||
# True when a successor is coming, so the outgoing daemon should leave the
|
||||
# fail-closed holding plane standing instead of removing it.
|
||||
#
|
||||
# `shutdown` is deliberately NOT a handoff either: nothing is coming, and the
|
||||
# kernel that would hold the plane is going away with it. Leaving the flag down
|
||||
# there also keeps the marker's meaning exact — it says "you are being replaced",
|
||||
# and at shutdown nothing is.
|
||||
shater_action_handoff() {
|
||||
case "$1" in
|
||||
restart|reload) return 0 ;;
|
||||
*) return 1 ;;
|
||||
esac
|
||||
}
|
||||
|
||||
# --- helpers ---------------------------------------------------------------
|
||||
|
||||
# True only when the stack is explicitly enabled in UCI.
|
||||
@@ -73,6 +252,25 @@ _slog() {
|
||||
[ "$(uci -q get shater.globals.log_syslog)" = "0" ] || logger -t shater "$@"
|
||||
}
|
||||
|
||||
# Announce/withdraw "this daemon is being replaced, not switched off". Read by
|
||||
# `shaterd run` when it receives SIGTERM.
|
||||
shater_mark_restart() {
|
||||
mkdir -p "$(dirname "$RESTART_FLAG")" 2>/dev/null
|
||||
: > "$RESTART_FLAG"
|
||||
}
|
||||
shater_clear_restart() { rm -f "$RESTART_FLAG"; }
|
||||
|
||||
# Remove the persisted boot armor, so the LAN is NOT blocked at the next boot
|
||||
# before the daemon starts. Called from exactly two places, both of which are a
|
||||
# statement about the PRODUCT rather than about this process: an operator typing
|
||||
# `stop`, and a daemon binary that is no longer on the box. In neither case is
|
||||
# anything going to come along and replace the armor with a real data plane, and a
|
||||
# kill switch with nothing behind it is just a brick.
|
||||
#
|
||||
# NOT called on the shutdown path. That is the whole fix — see
|
||||
# shater_action_disarms.
|
||||
shater_disarm_boot() { rm -f "$BOOT_ARMOR"; }
|
||||
|
||||
# Echo the pid of a LIVE `shaterd run`, or fail. The pidfile is written by the
|
||||
# daemon itself and removed only by the daemon that owns it, AFTER its teardown
|
||||
# has completed — so "pidfile names a live process" is precisely "the previous
|
||||
@@ -136,9 +334,20 @@ start_service() {
|
||||
# Guard: never claim to run without the daemon binary. A half-removed/failed
|
||||
# shaterd upgrade must degrade to "plugin off", not to a box that thinks
|
||||
# interception is live with nothing behind it.
|
||||
#
|
||||
# "Plugin off" now has to include DISARMING. With the boot armor in play, a
|
||||
# missing binary is the one case where the fail-closed plane could stand
|
||||
# forever with nothing able to replace it: the armor loads at START=21, the
|
||||
# daemon never starts, and every later boot repeats it. The product being gone
|
||||
# is not a security event — it is an uninstall — so the plane comes down and
|
||||
# the LAN returns to plain routing, loudly.
|
||||
if [ ! -x "$PROG" ]; then
|
||||
shater_clear_restart
|
||||
shater_disarm_boot
|
||||
rm -f "$ACTIVE_FLAG"
|
||||
nft delete table inet shater 2>/dev/null
|
||||
_slog -p daemon.err \
|
||||
"shaterd binary missing/not executable at $PROG — refusing to start (LAN stays on plain routing)"
|
||||
"shaterd binary missing/not executable at $PROG — refusing to start; the fail-closed plane and its boot armor have been REMOVED (LAN back to plain routing, unprotected). Reinstall shaterd."
|
||||
return 0
|
||||
fi
|
||||
|
||||
@@ -150,6 +359,11 @@ start_service() {
|
||||
# running, which is the boot case.
|
||||
shater_wait_stopped
|
||||
|
||||
# The predecessor is gone and has already consumed the flag (it reads it in its
|
||||
# SIGTERM handler). Withdraw it now, so a LATER `stop` is unambiguous even if
|
||||
# this start fails further down.
|
||||
shater_clear_restart
|
||||
|
||||
# Bring the UCI schema forward before the daemon reads it (idempotent;
|
||||
# refuses a newer schema) so an upgraded package never applies a stale config.
|
||||
#
|
||||
@@ -214,6 +428,45 @@ start_service() {
|
||||
}
|
||||
|
||||
stop_service() {
|
||||
# Say WHY we are stopping before procd sends the signal, because the daemon
|
||||
# cannot tell from the signal alone and the answer changes what it leaves in
|
||||
# the kernel. Two INDEPENDENT questions, and the old code conflated them into
|
||||
# one two-armed `case` whose else-branch answered both wrongly for `shutdown`:
|
||||
#
|
||||
# 1. IS A SUCCESSOR COMING (this process only)? restart / reload.
|
||||
# Raise RESTART_FLAG so the outgoing daemon replaces its data plane with
|
||||
# the fail-closed HOLDING plane instead of removing it. The gap until the
|
||||
# successor applies is not a moment: this script waits out the old
|
||||
# process, runs `shaterd migrate`, then starts a daemon that must build an
|
||||
# engine — all of it, before this flag existed, with `lan -> wan ACCEPT`
|
||||
# and nothing else.
|
||||
#
|
||||
# 2. IS THE PRODUCT BEING SWITCHED OFF (across boots)? `stop` — and only
|
||||
# `stop`, and only when a PERSON is behind it (shater_stop_disarms; the
|
||||
# package manager reaches us through `stop` too). Then the boot armor goes
|
||||
# with it, so the next boot does not quietly reinstate what the operator
|
||||
# just switched off — the same rule ACTIVE_FLAG has always enforced for
|
||||
# hotplug/cron.
|
||||
#
|
||||
# `shutdown` answers NO to both, which is the defect this replaced: a reboot is
|
||||
# not a successor and it is certainly not an operator switching the product off.
|
||||
# It is the boot the armor exists for. An upgrade answers NO to the second for
|
||||
# the same kind of reason.
|
||||
if shater_action_handoff "$SHATER_RC_ACTION"; then
|
||||
shater_mark_restart
|
||||
else
|
||||
shater_clear_restart
|
||||
fi
|
||||
local in_pkg=0 rc_en=0
|
||||
shater_pkg_transaction && in_pkg=1
|
||||
shater_rc_enabled && rc_en=1
|
||||
if shater_stop_disarms "$SHATER_RC_ACTION" "$in_pkg" "$rc_en"; then
|
||||
shater_disarm_boot
|
||||
elif [ "$in_pkg" = "1" ] && shater_action_disarms "$SHATER_RC_ACTION"; then
|
||||
_slog -p daemon.info \
|
||||
"stop came from a package transaction that left the service enabled — keeping the boot armor, so being replaced cannot leave the next boot unprotected"
|
||||
fi
|
||||
|
||||
# Drop the live-flag FIRST so a concurrent hotplug/cron tick cannot rebuild
|
||||
# what we are about to tear down. procd then sends SIGTERM to `shaterd run`,
|
||||
# which runs its OWN honest teardown (engine.Close + netplane restore) — we
|
||||
@@ -233,6 +486,11 @@ reload_service() {
|
||||
# disabled, `start` is a no-op, so a disable+apply cleanly tears everything
|
||||
# down. Because the wait lives in start_service, this path gets the same
|
||||
# stop-then-start ordering guarantee as `restart`.
|
||||
#
|
||||
# Marked EXPLICITLY as well as via SHATER_RC_ACTION: this is the path a routine
|
||||
# Save & Apply takes, so it is the one that must not depend on reading an
|
||||
# rc.common variable correctly. Belt and braces, one line.
|
||||
shater_mark_restart
|
||||
stop
|
||||
start
|
||||
}
|
||||
|
||||
@@ -0,0 +1,162 @@
|
||||
#!/bin/sh /etc/rc.common
|
||||
# /etc/init.d/shater-armor — the fail-closed plane, before the daemon exists.
|
||||
#
|
||||
# WHAT THIS CLOSES
|
||||
#
|
||||
# /etc/init.d/shater is START=99. By then fw4 (START=19) has long since loaded
|
||||
# `lan -> wan ACCEPT` and netifd (START=20) has brought the LAN bridge up, so the
|
||||
# router forwards LAN traffic to the WAN in the clear from the moment the link
|
||||
# comes up until `shaterd run` has been decompressed off flash, has waited out any
|
||||
# predecessor, has migrated UCI, has read the config and has installed its first
|
||||
# table. On router-class hardware with a UPX-packed binary that is seconds — and
|
||||
# they are exactly the seconds in which Wi-Fi finishes associating and every
|
||||
# client on the network reconnects and starts talking. `kill_switch=closed` was
|
||||
# configured the whole time and covered none of it.
|
||||
#
|
||||
# There was nothing in the package that could cover it either: no /etc/nftables.d
|
||||
# include, no `nft -f` in uci-defaults. Protection existed only inside a Go
|
||||
# process that had not started yet.
|
||||
#
|
||||
# HOW
|
||||
#
|
||||
# The daemon persists a copy of its fail-closed HOLDING plane (the same ruleset it
|
||||
# installs when the engine is down: one forward chain, LAN-to-LAN and router
|
||||
# traffic accepted, everything else from the diverted devices dropped) to
|
||||
# $ARMOR on every apply. This script loads it early. When the daemon comes up it
|
||||
# replaces the table atomically — the ruleset begins with `delete table` and adds
|
||||
# its own in one netlink transaction — so there is never a moment with no table.
|
||||
#
|
||||
# `iifname` matches by NAME at packet time, not by ifindex at load time, so
|
||||
# loading this before netifd has created br-lan is fine: the rules simply start
|
||||
# matching when the device appears. That is why START can sit here rather than
|
||||
# racing netifd.
|
||||
#
|
||||
# START=21: after fw4 (19) and netifd (20), because fw4's own start tears its
|
||||
# table down and rebuilds it and we do not want to be in the middle of that, and
|
||||
# because there is nothing to protect before the LAN device is being created. The
|
||||
# residual exposure is the fraction of a second between netifd's `ifup` and this
|
||||
# script, against seconds-to-a-minute before.
|
||||
#
|
||||
# THE ESCAPE HATCHES (a kill switch that cannot be switched off is a brick)
|
||||
#
|
||||
# These are STATE checks, evaluated here, at the moment of arming — not a record
|
||||
# of something that happened on the way down. That distinction is the whole
|
||||
# lesson of v0.2.17: the arm token was deleted by an EVENT on the shutdown path
|
||||
# ("this looks like a stop"), and since `reboot` also runs the K-links, the
|
||||
# mechanism reliably erased itself on the one transition it was built for. An
|
||||
# event on the way down cannot be trusted to describe the world on the way up; a
|
||||
# question asked on the way up can be.
|
||||
#
|
||||
# * $ARMOR only exists while the daemon's last applied config was BOTH enabled
|
||||
# and fail-closed. `globals.enabled=0` and `kill_switch=open` each remove it
|
||||
# at the next apply, and an operator typing `/etc/init.d/shater stop` removes
|
||||
# it there and then. Powering the box off does NOT.
|
||||
# * We refuse to arm when the main service is disabled in rc.d, or when the
|
||||
# daemon binary is gone — in either case nothing would ever come along to
|
||||
# replace the armor with a real data plane. These two are what makes a
|
||||
# genuinely uninstalled/disabled product safe REGARDLESS of what the file
|
||||
# says, which is why they are checked here rather than trusted to have been
|
||||
# acted on earlier.
|
||||
# * We refuse to arm when UCI can be read AND says the stack is disabled. A
|
||||
# config that cannot be read is NOT a refusal: that case is precisely why the
|
||||
# armor is a file rather than a query.
|
||||
# * The chain hooks `forward` only, so SSH, LuCI and the admin panel (all input
|
||||
# hook, to the router's own addresses) stay reachable. The operator can always
|
||||
# get in and undo this.
|
||||
#
|
||||
# Note what a bare `/etc/init.d/shater stop` does NOT mean: it does not survive a
|
||||
# reboot, because S99shater is still linked and procd starts the daemon again. So
|
||||
# "stopped" is not a durable off-state and this script must not be designed as if
|
||||
# it were — the durable ones are `disable` (no S??shater) and `globals.enabled=0`,
|
||||
# and those are the two refusals above.
|
||||
#
|
||||
# busybox ash only — no bashisms.
|
||||
|
||||
START=21 # after firewall (19) and network (20), long before shater (99)
|
||||
STOP=89
|
||||
|
||||
ARMOR=/etc/shater/boot.nft
|
||||
PROG=/usr/bin/shaterd
|
||||
|
||||
# Syslog line that honors globals.log_syslog, like the other two inits. An
|
||||
# unreadable UCI leaves the option empty => ON, which is what we want here: the
|
||||
# one boot where the config cannot be read is the boot worth logging.
|
||||
_slog() {
|
||||
[ "$(uci -q get shater.globals.log_syslog)" = "0" ] || logger -t shater-armor "$@"
|
||||
}
|
||||
|
||||
# Is the MAIN service enabled at boot? Answered by looking for its rc.d symlink
|
||||
# rather than by running `/etc/init.d/shater enabled`: that is a USE_PROCD script,
|
||||
# so every action of it sources procd.sh, which takes a blocking flock — and this
|
||||
# runs at START=21, in the middle of boot, for a question a glob answers exactly
|
||||
# as well. The START number is not hardcoded; any S<NN>shater counts.
|
||||
shater_service_enabled() {
|
||||
local f
|
||||
for f in /etc/rc.d/S[0-9][0-9]shater; do
|
||||
[ -e "$f" ] && return 0
|
||||
done
|
||||
return 1
|
||||
}
|
||||
|
||||
start() {
|
||||
# No saved plane => the stack has never applied an enabled, fail-closed config
|
||||
# (or it was explicitly switched off). Nothing to do, and nothing to say.
|
||||
[ -f "$ARMOR" ] || return 0
|
||||
[ -s "$ARMOR" ] || {
|
||||
_slog -p daemon.err "$ARMOR is empty — NOT arming; the LAN is unprotected until shaterd starts"
|
||||
return 0
|
||||
}
|
||||
|
||||
# Never arm something nothing can disarm.
|
||||
[ -x "$PROG" ] || {
|
||||
_slog -p daemon.err \
|
||||
"$PROG is missing — NOT arming (nothing would replace the block with a working data plane); the LAN stays on plain routing"
|
||||
return 0
|
||||
}
|
||||
shater_service_enabled || {
|
||||
_slog -p daemon.warn \
|
||||
"the shater service is disabled in rc.d — NOT arming (nothing would replace the block with a working data plane); the LAN stays on plain routing"
|
||||
return 0
|
||||
}
|
||||
|
||||
# A READABLE config that says "off" wins over the saved plane (it means the
|
||||
# daemon was stopped before it could disarm). An UNREADABLE config does not:
|
||||
# that is the case this whole mechanism exists for.
|
||||
en=$(uci -q get shater.globals.enabled 2>/dev/null)
|
||||
if [ -n "$en" ] && [ "$en" != "1" ]; then
|
||||
rm -f "$ARMOR"
|
||||
_slog -p daemon.info "globals.enabled=$en — boot armor removed, not arming"
|
||||
return 0
|
||||
fi
|
||||
|
||||
command -v nft >/dev/null 2>&1 || {
|
||||
_slog -p daemon.err "nft is not installed — cannot arm; the LAN is unprotected until shaterd starts"
|
||||
return 0
|
||||
}
|
||||
|
||||
# Validate before loading: a truncated/incompatible snapshot must not leave a
|
||||
# half-built table behind on the one boot it is needed.
|
||||
if ! nft -c -f "$ARMOR" >/dev/null 2>&1; then
|
||||
_slog -p daemon.err \
|
||||
"$ARMOR did not validate (nft -c) — NOT arming; the LAN is unprotected until shaterd starts"
|
||||
return 0
|
||||
fi
|
||||
if nft -f "$ARMOR" >/dev/null 2>&1; then
|
||||
_slog -p daemon.warn \
|
||||
"fail-closed plane armed from $ARMOR: LAN->WAN forwarding is BLOCKED until shaterd applies. SSH, LuCI and the admin panel stay reachable."
|
||||
else
|
||||
_slog -p daemon.err \
|
||||
"could not load $ARMOR — the LAN is unprotected until shaterd starts"
|
||||
fi
|
||||
return 0
|
||||
}
|
||||
|
||||
stop() {
|
||||
# Deliberately a NO-OP. By the time anything stops this service the daemon owns
|
||||
# `inet shater`, and deleting the table here would dismantle a LIVE data plane
|
||||
# on the strength of a service that only ever ran for one second at boot. The
|
||||
# disarm paths that matter live where the decision is actually made:
|
||||
# /etc/init.d/shater stop (operator switched it off) and the daemon itself
|
||||
# (globals.enabled=0 / kill_switch=open).
|
||||
return 0
|
||||
}
|
||||
@@ -59,6 +59,68 @@ if uci -q get shater.globals >/dev/null 2>&1 || [ -f /etc/config/shater ]; then
|
||||
uci -q commit shater
|
||||
fi
|
||||
|
||||
# Introduce the daemon-created `shater-l3` TUN to fw4 (L3 ingress, D-L3). The
|
||||
# daemon policy-routes LAN ICMP into that device from OUR nft table
|
||||
# `inet shater`, but nftables runs EVERY table on every packet and a drop in
|
||||
# any one of them wins — an accept in `inet shater` cannot override fw4. And
|
||||
# fw4 WILL drop this forward: netifd knows nothing about a device the daemon
|
||||
# creates at runtime, so it belongs to no zone and falls into fw4's zone-less
|
||||
# defaults (REJECT). The device has to be declared to fw4 itself; it cannot be
|
||||
# fixed from our own table.
|
||||
#
|
||||
# Seeded UNCONDITIONALLY (not gated on globals.l3_tunnel): uci-defaults run
|
||||
# once, so gating on the option would require re-running this script when the
|
||||
# option is flipped later — which never happens. An idle zone is harmless: its
|
||||
# device match is a plain iifname/oifname STRING compare that simply never hits
|
||||
# while the TUN does not exist.
|
||||
#
|
||||
# Idempotency: `config zone`/`config forwarding` are normally ANONYMOUS
|
||||
# sections, and a naive `uci add firewall zone` would append a duplicate on
|
||||
# every re-run (uci-defaults re-run on package upgrade/reinstall). All sections
|
||||
# here are NAMED instead, guarded by an existence check — a re-run re-finds the
|
||||
# section and touches nothing.
|
||||
seed_l3_zone() {
|
||||
# No fw4 on this image (bare nftables build) => nothing drops the forward
|
||||
# on fw4's behalf and there is nothing to punch through.
|
||||
[ -f /etc/config/firewall ] || return 0
|
||||
if ! uci -q get firewall.shater_l3 >/dev/null; then
|
||||
uci set firewall.shater_l3=zone
|
||||
uci set firewall.shater_l3.name='shater_l3'
|
||||
uci set firewall.shater_l3.input='REJECT'
|
||||
uci set firewall.shater_l3.output='ACCEPT'
|
||||
uci set firewall.shater_l3.forward='REJECT'
|
||||
uci set firewall.shater_l3.masq='0'
|
||||
# INERT TODAY, kept for the day it is not. mtu_fix clamps forwarded TCP
|
||||
# MSS to the route MTU — but shater-l3 is 65535 (deliberately: at any
|
||||
# smaller value the kernel fragments into the device, and the flow
|
||||
# dispatcher refuses to judge a fragment and lets the stack forge the
|
||||
# echo reply — see l3MTU in shater/generate/inbound.go), so the clamp has
|
||||
# nothing to clamp to. And only ICMP is ever marked into this device, so
|
||||
# no TCP rides here to be clamped in the first place. It earns its keep
|
||||
# the moment either of those changes; removing it would make that day
|
||||
# silent.
|
||||
uci set firewall.shater_l3.mtu_fix='1'
|
||||
# `list device`, deliberately NOT the usual `list network`: fw4
|
||||
# resolves a zone's networks through netifd, and netifd never learns
|
||||
# about a device the daemon creates at runtime — a stub interface
|
||||
# (proto none) would need to be brought UP to contribute an l3_device,
|
||||
# and nothing ever brings it up, so `list network` resolves to an
|
||||
# EMPTY device set and fw4 keeps dropping the forward. `list device`
|
||||
# instead compiles to an iifname/oifname STRING match, valid before
|
||||
# the TUN exists and matching from the moment shaterd creates it —
|
||||
# no netifd involvement and no firewall reload at enable time. Do not
|
||||
# "normalize" this to `list network` in a refactor; it breaks silently.
|
||||
uci add_list firewall.shater_l3.device='shater-l3'
|
||||
fi
|
||||
if ! uci -q get firewall.shater_l3_fwd >/dev/null; then
|
||||
uci set firewall.shater_l3_fwd=forwarding
|
||||
uci set firewall.shater_l3_fwd.src='lan'
|
||||
uci set firewall.shater_l3_fwd.dest='shater_l3'
|
||||
fi
|
||||
uci -q commit firewall
|
||||
}
|
||||
seed_l3_zone
|
||||
|
||||
# Bring the UCI schema forward on upgrade (idempotent; refuses a newer schema).
|
||||
[ -x /usr/bin/shaterd ] && /usr/bin/shaterd migrate >/dev/null 2>&1
|
||||
|
||||
@@ -110,8 +172,27 @@ SHATER_BRINGUP='
|
||||
done
|
||||
[ -x /etc/init.d/shater ] && /etc/init.d/shater enable
|
||||
[ -x /etc/init.d/shater-cron ] && /etc/init.d/shater-cron enable
|
||||
# The boot-time fail-closed armor. `enable` only — it is a one-shot that loads
|
||||
# the persisted holding plane at START=21, and running it NOW would install a
|
||||
# block on a live box moments before the daemon replaces it anyway. It has to
|
||||
# be enabled here regardless of whether the stack is on: the file it loads only
|
||||
# exists while the daemon wants it to, so an enabled-but-unarmed service is a
|
||||
# no-op, and enabling it later would mean the first boot after an upgrade is
|
||||
# the one boot still exposed.
|
||||
[ -x /etc/init.d/shater-armor ] && /etc/init.d/shater-armor enable
|
||||
[ -x /etc/init.d/shater ] && /etc/init.d/shater restart
|
||||
[ -x /etc/init.d/shater-cron ] && /etc/init.d/shater-cron restart
|
||||
# Fold the seeded shater_l3 zone into the LIVE ruleset — matters on a live
|
||||
# opkg/apk install only, where firewall started long before our commit and
|
||||
# nothing else would re-read it until the next reboot. Gated on the fw4
|
||||
# table actually being loaded: at FIRST boot this job can run before the
|
||||
# S19 firewall start, and an early reload would install a ruleset built
|
||||
# from a half-initialized netifd AND make the later start a no-op (fw4
|
||||
# start skips when its table already exists). No table => the pending S19
|
||||
# start reads the committed config by itself, no reload needed.
|
||||
if nft list tables 2>/dev/null | grep -q "inet fw4"; then
|
||||
[ -x /etc/init.d/firewall ] && /etc/init.d/firewall reload
|
||||
fi
|
||||
exit 0
|
||||
'
|
||||
SHATER_TMO=""
|
||||
|
||||
@@ -18,6 +18,32 @@ type URLTestOutboundOptions struct {
|
||||
// lx: SPEC 019 v2 — load-balancing.
|
||||
Mode string `json:"mode,omitempty"` // least_test (default) | round_robin
|
||||
Balancer *URLTestBalancerOptions `json:"balancer,omitempty"`
|
||||
// lx: health board §5.C — SelfCheck stands the group's OWN background
|
||||
// health-check up or down. nil/absent == true, so every existing config keeps
|
||||
// today's behaviour.
|
||||
//
|
||||
// Why this exists at all: a urltest group probes its members BY ITSELF — a
|
||||
// warm-up sweep at PostStart and a ticker for as long as traffic keeps
|
||||
// touching it — and it dials the members' outbounds DIRECTLY, from the
|
||||
// router, over whatever the default WAN route is. For a group that traffic
|
||||
// actually flows through, that is exactly right: the probe travels the same
|
||||
// path the connections do. But for a group NO routing rule reaches, that
|
||||
// same probe measures a path nothing uses — and it stores the result under
|
||||
// the members' BASE tags, which every health consumer then reads as "the
|
||||
// node's health". A node that is blocked on the direct WAN and perfectly
|
||||
// alive behind a tunnel therefore reads "dead" the moment such a group
|
||||
// probes it; the reading is not merely stale, it is FALSE, and it poisons
|
||||
// the shared board for everyone (selection, the panel, the observatory's
|
||||
// freshness gate). SelfCheck=false is how the control plane stands such a
|
||||
// group's own schedule down: the shater engine computes which groups the
|
||||
// applied rules actually reach (the observatory's used-set) and disables
|
||||
// the self-check on the rest, so the ONLY prober left is the observatory —
|
||||
// which probes along the real dial paths and nothing else.
|
||||
//
|
||||
// The flag suppresses only the group's own SCHEDULE (the PostStart warm-up
|
||||
// and the Touch ticker). An EXPLICIT CheckOutbounds/URLTest call — the
|
||||
// adapter interface a human or an API invokes on purpose — still works.
|
||||
SelfCheck *bool `json:"self_check,omitempty"`
|
||||
}
|
||||
|
||||
// URLTestBalancerOptions configures round_robin: a fixed-size pool of live nodes, lazily
|
||||
|
||||
@@ -262,6 +262,150 @@
|
||||
}
|
||||
}
|
||||
|
||||
/* ---- fixture band (dev builds only; see App.tsx MockBanner) ----
|
||||
Deliberately outside the crit/amber vocabulary: nothing is wrong with the
|
||||
router, there is no router. The hazard hatch is the service-sticker language a
|
||||
piece of network hardware already uses for "this unit is not in service". */
|
||||
.mock-band {
|
||||
display: flex;
|
||||
align-items: center;
|
||||
gap: calc(var(--u, 8px) * 1.5);
|
||||
margin-top: calc(var(--u, 8px) * 2);
|
||||
padding: 10px 14px;
|
||||
border: 1px dashed var(--faint);
|
||||
border-radius: 9px;
|
||||
background: repeating-linear-gradient(
|
||||
-45deg,
|
||||
var(--sink),
|
||||
var(--sink) 9px,
|
||||
var(--panel) 9px,
|
||||
var(--panel) 18px
|
||||
);
|
||||
}
|
||||
.mock-band-tag {
|
||||
flex-shrink: 0;
|
||||
align-self: flex-start;
|
||||
padding: 3px 7px;
|
||||
border: 1px solid var(--faint);
|
||||
border-radius: 4px;
|
||||
background: var(--raised);
|
||||
font-family: var(--font-mono);
|
||||
font-size: 10px;
|
||||
font-weight: 700;
|
||||
letter-spacing: 0.14em;
|
||||
color: var(--dim);
|
||||
}
|
||||
.mock-band-copy {
|
||||
flex: 1;
|
||||
min-width: 0;
|
||||
display: flex;
|
||||
flex-direction: column;
|
||||
gap: 2px;
|
||||
}
|
||||
.mock-band-headline {
|
||||
font-family: var(--font-mono);
|
||||
font-size: 12.5px;
|
||||
font-weight: 700;
|
||||
letter-spacing: 0.02em;
|
||||
color: var(--ink);
|
||||
}
|
||||
.mock-band-detail {
|
||||
font-size: 12.5px;
|
||||
line-height: 1.5;
|
||||
color: var(--dim);
|
||||
max-width: 76ch;
|
||||
}
|
||||
.mock-band-detail code {
|
||||
font-family: var(--font-mono);
|
||||
font-size: 11.5px;
|
||||
color: var(--ink);
|
||||
}
|
||||
|
||||
/* ---- commit-confirm band (every page except Apply, which has the full panel) ----
|
||||
Same plate as the protection banner so the two read as one family; the seconds
|
||||
are the loud element because they are the only thing that is running out. */
|
||||
.cfm-band {
|
||||
display: flex;
|
||||
align-items: center;
|
||||
gap: calc(var(--u, 8px) * 1.5);
|
||||
margin-top: calc(var(--u, 8px) * 2);
|
||||
padding: 10px 14px;
|
||||
border: 1px solid color-mix(in srgb, var(--amber) 50%, var(--groove));
|
||||
border-radius: 9px;
|
||||
background: linear-gradient(180deg, color-mix(in srgb, var(--amber) 10%, var(--raised)), var(--raised));
|
||||
box-shadow: 0 1px 0 var(--edge) inset;
|
||||
}
|
||||
.cfm-band-count {
|
||||
display: flex;
|
||||
align-items: baseline;
|
||||
gap: 2px;
|
||||
flex-shrink: 0;
|
||||
font-family: var(--font-mono);
|
||||
color: var(--amber);
|
||||
}
|
||||
.cfm-band-num {
|
||||
font-size: 22px;
|
||||
font-weight: 700;
|
||||
font-variant-numeric: tabular-nums;
|
||||
line-height: 1;
|
||||
}
|
||||
.cfm-band-unit {
|
||||
font-size: 11px;
|
||||
letter-spacing: 0.06em;
|
||||
}
|
||||
.cfm-band-copy {
|
||||
flex: 1;
|
||||
min-width: 0;
|
||||
display: flex;
|
||||
flex-direction: column;
|
||||
gap: 3px;
|
||||
}
|
||||
.cfm-band-headline {
|
||||
font-family: var(--font-mono);
|
||||
font-size: 12.5px;
|
||||
font-weight: 700;
|
||||
letter-spacing: 0.02em;
|
||||
color: var(--ink);
|
||||
}
|
||||
.cfm-band-detail {
|
||||
font-size: 12.5px;
|
||||
line-height: 1.5;
|
||||
color: var(--dim);
|
||||
max-width: 76ch;
|
||||
}
|
||||
.cfm-band-actions {
|
||||
display: flex;
|
||||
align-items: center;
|
||||
gap: calc(var(--u, 8px) * 1);
|
||||
flex-shrink: 0;
|
||||
}
|
||||
.cfm-band-link {
|
||||
padding: 6px 11px;
|
||||
border: 1px solid var(--groove);
|
||||
border-radius: 6px;
|
||||
font-family: var(--font-mono);
|
||||
font-size: 11px;
|
||||
letter-spacing: 0.06em;
|
||||
text-transform: uppercase;
|
||||
text-decoration: none;
|
||||
color: var(--ink);
|
||||
background: var(--raised);
|
||||
}
|
||||
.cfm-band-link:hover {
|
||||
border-color: var(--accent);
|
||||
color: var(--accent);
|
||||
}
|
||||
|
||||
@media (max-width: 720px) {
|
||||
.cfm-band {
|
||||
flex-wrap: wrap;
|
||||
}
|
||||
.cfm-band-actions {
|
||||
width: 100%;
|
||||
justify-content: flex-end;
|
||||
}
|
||||
}
|
||||
|
||||
/* ---- last-apply findings (Overview) ----
|
||||
Severity carries the colour; the accent is reserved for interactive controls. */
|
||||
.findings {
|
||||
@@ -312,6 +456,13 @@
|
||||
.finding--warning {
|
||||
border-color: color-mix(in srgb, var(--amber) 40%, var(--groove));
|
||||
}
|
||||
/* The daemon's "the list is capped" disclosure. Dashed, because the row is about
|
||||
what ISN'T here — it must not read as one more finding to work through. */
|
||||
.finding--truncated {
|
||||
border-style: dashed;
|
||||
border-color: color-mix(in srgb, var(--amber) 40%, var(--groove));
|
||||
background: var(--panel);
|
||||
}
|
||||
.finding-copy {
|
||||
flex: 1;
|
||||
min-width: 0;
|
||||
|
||||
+118
-7
@@ -1,12 +1,14 @@
|
||||
import './App.css'
|
||||
import { useCallback, useEffect, useState } from 'react'
|
||||
import { Faceplate, FaceplateHeader, Led, Module } from './components'
|
||||
import { Button, Faceplate, FaceplateHeader, Led, Module } from './components'
|
||||
import type { LedVariant } from './components'
|
||||
import { ApiError, MOCK, getStatus } from './api'
|
||||
import { ApiError, MOCK, confirm as apiConfirm, getStatus } from './api'
|
||||
import type { Status } from './api'
|
||||
import { usePendingConfirm } from './pendingConfirm'
|
||||
import { bootstrapSession } from './session'
|
||||
import { ROUTES, navigate, useRoute } from './router'
|
||||
import { protectionState } from './planeState'
|
||||
import { engineState, protectionState } from './planeState'
|
||||
import { truncationNote } from './findings'
|
||||
import type { Route } from './router'
|
||||
import { Overview, Placeholder, Nodes, Routing, Apply, DNS, Devices, Targets, Settings, Profiles, Insights, Networks } from './pages'
|
||||
|
||||
@@ -99,12 +101,102 @@ export function App() {
|
||||
footer={<StatusBar status={status} />}
|
||||
>
|
||||
<Nav route={route} />
|
||||
<MockBanner />
|
||||
<PlaneBanner status={status} route={route} />
|
||||
<ConfirmBand route={route} onChanged={() => void refreshStatus()} />
|
||||
<Page route={route} status={status} onStatusChange={() => void refreshStatus()} />
|
||||
</Faceplate>
|
||||
)
|
||||
}
|
||||
|
||||
/**
|
||||
* The commit-confirm countdown, on every page.
|
||||
*
|
||||
* The daemon arms an auto-rollback on EVERY apply, but only the Apply page ever
|
||||
* said so: press Apply on Routing, read "Applied", walk away, and the router
|
||||
* reverts a minute later with nothing on screen having mentioned it. This band
|
||||
* carries that deadline — and the button that stops it — to wherever the operator
|
||||
* actually is.
|
||||
*
|
||||
* Suppressed on Apply, which renders the full control room for the same window
|
||||
* (and reads the same record, so a reload no longer loses the countdown there
|
||||
* either).
|
||||
*/
|
||||
function ConfirmBand({ route, onChanged }: { route: Route; onChanged: () => void }) {
|
||||
const armed = usePendingConfirm()
|
||||
const [busy, setBusy] = useState(false)
|
||||
const [error, setError] = useState<string | null>(null)
|
||||
|
||||
// Keeping the config is the only action offered here; rolling back early is a
|
||||
// deliberate act with its own before/after readout, and that lives on Apply.
|
||||
const keep = useCallback(async () => {
|
||||
setBusy(true)
|
||||
setError(null)
|
||||
try {
|
||||
const r = await apiConfirm()
|
||||
if (r.error) setError(r.error)
|
||||
} catch (e) {
|
||||
setError(e instanceof Error ? e.message : 'request failed')
|
||||
} finally {
|
||||
setBusy(false)
|
||||
onChanged()
|
||||
}
|
||||
}, [onChanged])
|
||||
|
||||
if (!armed || route === 'apply') return null
|
||||
|
||||
return (
|
||||
<div className="cfm-band" role="alert">
|
||||
<Led variant="amber" pulse />
|
||||
<div className="cfm-band-count" role="timer" aria-label={`${armed.remaining} seconds until auto-rollback`}>
|
||||
<span className="cfm-band-num">{armed.remaining}</span>
|
||||
<span className="cfm-band-unit">s</span>
|
||||
</div>
|
||||
<div className="cfm-band-copy">
|
||||
<span className="cfm-band-headline">This config is live but not kept</span>
|
||||
<span className="cfm-band-detail">
|
||||
{error
|
||||
? `Couldn’t keep it — ${error}. Try again, or open Apply.`
|
||||
: 'Every apply arms an auto-rollback. Keep this config before the timer runs out, or the router reverts to the last-good one.'}
|
||||
</span>
|
||||
</div>
|
||||
<div className="cfm-band-actions">
|
||||
<Button variant="primary" onClick={() => void keep()} disabled={busy}>
|
||||
{busy ? 'Keeping…' : 'Keep this config'}
|
||||
</Button>
|
||||
<a className="cfm-band-link" href="#/apply" onClick={() => navigate('apply')}>
|
||||
Apply page
|
||||
</a>
|
||||
</div>
|
||||
</div>
|
||||
)
|
||||
}
|
||||
|
||||
/**
|
||||
* Says, on every page, that nothing on screen came from a router.
|
||||
*
|
||||
* Only a DEV build can ever render this — the fixtures are not in a production
|
||||
* bundle (api.ts initMockBackend), so an operator cannot reach this state at all.
|
||||
* It is here for the person who CAN: a footer line reading "DEMO DATA" is easy to
|
||||
* work past for an afternoon and then screenshot into a bug report, and every
|
||||
* number above it is invented.
|
||||
*/
|
||||
function MockBanner() {
|
||||
if (!MOCK) return null
|
||||
return (
|
||||
<div className="mock-band" role="status">
|
||||
<span className="mock-band-tag">FIXTURES</span>
|
||||
<div className="mock-band-copy">
|
||||
<span className="mock-band-headline">No router is being read</span>
|
||||
<span className="mock-band-detail">
|
||||
Every reading on this page is invented by <code>src/mock.ts</code> for offline
|
||||
development. Drop <code>?mock</code> from the address to talk to a daemon.
|
||||
</span>
|
||||
</div>
|
||||
</div>
|
||||
)
|
||||
}
|
||||
|
||||
/**
|
||||
* The protection state, pinned under the nav on every page EXCEPT Overview
|
||||
* (which shows the same state as its own headline readout — see planeState.ts).
|
||||
@@ -130,12 +222,15 @@ function PlaneBanner({ status, route }: { status: Status | null; route: Route })
|
||||
|
||||
const criticals = (status.warnings ?? []).filter((w) => w.severity === 'critical').length
|
||||
const state = protectionState(status)
|
||||
// The published list is capped at 50, so with a note attached the count is a
|
||||
// floor. Say "at least" rather than quoting a total the daemon didn't send.
|
||||
const atLeast = truncationNote(status.warnings) ? 'At least ' : ''
|
||||
|
||||
// Wording comes from the shared source of truth so the banner and Overview can
|
||||
// never describe the same router differently.
|
||||
const headline = state.alarm
|
||||
? state.headline
|
||||
: `${criticals} protection ${criticals === 1 ? 'gap' : 'gaps'} from the last apply`
|
||||
: `${atLeast}${criticals} protection ${criticals === 1 ? 'gap' : 'gaps'} from the last apply`
|
||||
const detail = state.alarm
|
||||
? state.detail
|
||||
: 'Something you configured isn’t in effect. Review the findings before relying on it.'
|
||||
@@ -225,14 +320,30 @@ function StatusBar({ status }: { status: Status | null }) {
|
||||
)
|
||||
}
|
||||
|
||||
/**
|
||||
* The one lamp that is on screen no matter which page you are on.
|
||||
*
|
||||
* It used to read `status.running`, which the daemon hardcoded to `true` — so the
|
||||
* "Offline" branch could never be reached and the plate said "Online" through an
|
||||
* engine that had failed to start. It now asks {@link engineState}, whose whole
|
||||
* job is to be able to answer "down", and refuses to guess when nothing has been
|
||||
* reported: an unlit socket, not a green light.
|
||||
*/
|
||||
function masterIndicator(
|
||||
phase: Phase,
|
||||
status: Status | null,
|
||||
): { label: string; variant: LedVariant; pulse?: boolean } {
|
||||
if (phase === 'loading' || !status) return { label: 'Linking', variant: 'off' }
|
||||
if (status.running && status.active) return { label: 'Online', variant: 'on', pulse: true }
|
||||
if (status.running) return { label: 'Standby', variant: 'amber' }
|
||||
return { label: 'Offline', variant: 'crit' }
|
||||
switch (engineState(status)) {
|
||||
case 'down':
|
||||
return { label: 'Engine down', variant: 'crit' }
|
||||
case 'up':
|
||||
return status.active
|
||||
? { label: 'Online', variant: 'on', pulse: true }
|
||||
: { label: 'Standby', variant: 'amber' }
|
||||
default:
|
||||
return { label: 'Unknown', variant: 'off' }
|
||||
}
|
||||
}
|
||||
|
||||
function UnauthPlate() {
|
||||
|
||||
+279
-47
@@ -13,8 +13,13 @@
|
||||
// serves in-memory fixtures instead of hitting the network, so `npm run dev`
|
||||
// and screenshot runs render without a live backend. A real backend in dev is
|
||||
// reachable instead via the Vite proxy in vite.config.ts (no flag ⇒ real fetch).
|
||||
//
|
||||
// THE FIXTURES ARE A DEV-BUILD-ONLY ARTEFACT — see initMockBackend below. They
|
||||
// used to be a plain static import, decided at RUNTIME off `location.search`, so
|
||||
// the invented router shipped inside the binary that goes on real hardware and a
|
||||
// link ending in `?dev` painted a healthy appliance without making one request.
|
||||
|
||||
import * as mock from './mock'
|
||||
import { armPendingConfirm, clearPendingConfirm, noteConfirmTimeout } from './pendingConfirm'
|
||||
|
||||
// --- error type -------------------------------------------------------------
|
||||
|
||||
@@ -390,19 +395,142 @@ export interface GroupHealth {
|
||||
* any more and nothing to report here beyond the groups themselves.
|
||||
*/
|
||||
|
||||
/**
|
||||
* One hop of one chain, measured where that hop actually sits in the path.
|
||||
*
|
||||
* This is the reading the daemon always took and never showed. A chain is not a
|
||||
* target with a single health — it is an ordered series of them, and the only
|
||||
* question an operator ever asks about a broken chain is WHICH hop broke. The
|
||||
* end-to-end exit reading cannot answer that: it says "the path is dead" for a
|
||||
* four-hop chain and leaves the person to guess between four suspects.
|
||||
*
|
||||
* WIRE ORDER. `index` is 1-based and counts hops in the order the router dials
|
||||
* them: hop 1 is the first physical hop, and each later hop is dialled THROUGH
|
||||
* the ones before it. The hop carrying `exit: true` — always the largest index —
|
||||
* is where traffic leaves for the internet. A leading `egress:` in the chain's
|
||||
* configured Hops is NOT a numbered hop: the daemon lifts it into the entry
|
||||
* detour of hop 1, so a chain written `egress:ewan → node:awgout → group:sub0`
|
||||
* reports two hops, not three. Anything zipping this against the model's Hops
|
||||
* must drop that leading egress first and give up on labelling entirely if the
|
||||
* counts still disagree — a chain that splices sub-chains gets flattened here,
|
||||
* and a confidently WRONG hop name is worse than no name.
|
||||
*
|
||||
* `tag` is the engine-side outbound (`chain-<name>-h2`). Debugging and tooltips
|
||||
* only; it is never a label to put in front of a person.
|
||||
*
|
||||
* ORDERED WALK — THE READING STOPS AT THE FIRST DEAD HOP. Hops are NOT measured
|
||||
* independently, and never were measurable that way: hop 3 is dialled THROUGH
|
||||
* hop 2, so probing hop 3 while hop 2 is down measures hop 2 a second time and
|
||||
* learns nothing about hop 3. The daemon therefore walks the path in wire order
|
||||
* and stops at the first hop that does not answer. Every hop below that one is
|
||||
* left undialled and reported `state: "untested"` — no measurement exists —
|
||||
* carrying {@link ChainHopBlock} in `blocked_by` to name the hop that stopped the
|
||||
* walk. So a chain never reports a dead hop with a live hop below it; that shape
|
||||
* is not a rare case, it is unreachable.
|
||||
*
|
||||
* NODE HOP vs GROUP HOP. For `kind: "node"` the hop IS the measurement: `total`
|
||||
* is 1, the counters follow its own state, and `selected` is ''. For
|
||||
* `kind: "group"` the counters roll up that hop's per-hop member COPIES — the
|
||||
* copies dialled through the hops in front of it, which is exactly why they can
|
||||
* read alive here while the same group's standalone card reads dead. Both
|
||||
* readings are true; they measure different dial paths. `selected` is the node
|
||||
* NAME the hop routes through right now, and `delay_ms` / `age_seconds` belong
|
||||
* to that selected member (or the freshest alive one).
|
||||
*
|
||||
* Invariants the daemon guarantees — never re-derive them, just read them:
|
||||
* `tested === alive + dead` and `alive + dead + untested === total`.
|
||||
*
|
||||
* `state` is a closed set of THREE. `untested` is NEVER "dead" and never
|
||||
* "healthy": it means nothing fresh enough is known. Without `blocked_by` that is
|
||||
* a matter of timing — for a used chain it resolves on its own within seconds.
|
||||
* With `blocked_by` it will not resolve until the named hop is fixed. There is no
|
||||
* fourth state for that; the state stays `untested` because that is what it is.
|
||||
* `age_seconds: -1` means the age is unknown.
|
||||
*/
|
||||
export interface ChainHopHealth {
|
||||
/** 1-based WIRE order. Hop 1 is dialled first; see the note above. */
|
||||
index: number
|
||||
/** Engine outbound tag (`chain-<name>-h2`) — tooltips/debugging, never a label. */
|
||||
tag: string
|
||||
/** `node` ⇒ the hop is the measurement. `group` ⇒ the counters roll up members. */
|
||||
kind: 'node' | 'group'
|
||||
/** This hop is where traffic leaves for the internet. Always the largest index. */
|
||||
exit: boolean
|
||||
/** Closed set — switch on it exhaustively. `untested` is never "dead". */
|
||||
state: 'alive' | 'dead' | 'untested'
|
||||
/** RTT of the selected/freshest alive member; 0 (meaningless) when not alive. */
|
||||
delay_ms: number
|
||||
/** Age of that measurement in seconds; -1 when unknown. */
|
||||
age_seconds: number
|
||||
/** Node name this GROUP hop routes through right now; '' for a node hop. */
|
||||
selected: string
|
||||
total: number
|
||||
tested: number
|
||||
alive: number
|
||||
dead: number
|
||||
untested: number
|
||||
/**
|
||||
* PRESENT ONLY on a hop the ordered walk never reached — i.e. a hop sitting
|
||||
* below one the prober found `dead`. The key is omitted otherwise; absent is
|
||||
* the normal case and means "this hop was actually dialled".
|
||||
*
|
||||
* Its presence is the daemon's own statement that this hop has NO measurement,
|
||||
* and it comes with the rest of that statement already filled in: `state` is
|
||||
* `untested`, `delay_ms` is 0, `age_seconds` is -1, and the counters are
|
||||
* `alive: 0, dead: 0, tested: 0, untested: total`. Read those; do not re-derive
|
||||
* a verdict from them, and do not infer a block from zeroed counters either —
|
||||
* an unprobed-yet hop has the same numbers and a very different meaning.
|
||||
* `selected` MAY still be non-empty: the wrapper does have a pick, it simply
|
||||
* was not measured, so it says which node the hop would use, not which node is
|
||||
* carrying traffic.
|
||||
*/
|
||||
blocked_by?: ChainHopBlock
|
||||
}
|
||||
|
||||
/**
|
||||
* The hop that stopped the ordered walk, as reported on every hop below it.
|
||||
*
|
||||
* This exists because "no reading" and "no reading, and here is whose fault that
|
||||
* is" are different answers to the operator's actual question. Without it a
|
||||
* blocked hop is indistinguishable from one the observatory has not come round to
|
||||
* yet, and the interface can only shrug.
|
||||
*
|
||||
* `index` is the 1-based WIRE index of the blocking hop and is ALWAYS smaller
|
||||
* than the index of the hop carrying it, so it points at a hop already on screen.
|
||||
* `tag` is that hop's engine outbound (`chain-<name>-h3`) — debugging and
|
||||
* tooltips only, never a label to put in front of a person, exactly as on
|
||||
* {@link ChainHopHealth}.tag.
|
||||
*/
|
||||
export interface ChainHopBlock {
|
||||
/** 1-based wire index of the hop that did not answer. Always < this hop's index. */
|
||||
index: number
|
||||
/** That hop's engine outbound tag — tooltips/debugging, never a label. */
|
||||
tag: string
|
||||
}
|
||||
|
||||
/** Per-chain reachability, the chain analogue of {@link GroupHealth}.used (plan
|
||||
* §5.E): a chain no enabled routing rule routes through is outside the
|
||||
* observatory's plan, so its exit is never probed and the Targets card renders it
|
||||
* "unused" instead of an exit-test readout. A chain has no membership counters —
|
||||
* it is a fixed path, and its end-to-end health is the exit test's job. */
|
||||
* observatory's plan, so nothing probes it and the Targets card says so instead
|
||||
* of rendering a health reading. A chain has no membership counters of its own —
|
||||
* it is a fixed path, and its health lives on its {@link ChainHopHealth} hops. */
|
||||
export interface ChainHealth {
|
||||
name: string
|
||||
/** An enabled routing rule (the Final target, a DNS-resolver detour, a device
|
||||
* target, …) reaches this chain, so the observatory probes its exit in the
|
||||
* target, …) reaches this chain, so the observatory probes its hops in the
|
||||
* background. false ⇒ nothing routes through the chain: it is skipped by the
|
||||
* background probing and its end-to-end health stays untested. That is an
|
||||
* "unused" note about the ROUTING CONFIG, never a health problem. */
|
||||
* background probing and its health stays untested. That is an "unused" note
|
||||
* about the ROUTING CONFIG, never a health problem. */
|
||||
used: boolean
|
||||
/**
|
||||
* Per-hop health in wire order (see {@link ChainHopHealth}).
|
||||
*
|
||||
* MAY BE ABSENT, and absent does not mean "this chain has no hops". It means
|
||||
* the engine never materialised per-hop outbounds for it: the chain is unused,
|
||||
* or it collapses to a single hop and the daemon points traffic straight at
|
||||
* that target instead of building a copy of it. Read a missing key as "nothing
|
||||
* measured per hop", never as an empty path or as a fault.
|
||||
*/
|
||||
hops?: ChainHopHealth[]
|
||||
}
|
||||
|
||||
export interface GroupsHealth {
|
||||
@@ -996,12 +1124,59 @@ export interface Model {
|
||||
|
||||
// --- transport --------------------------------------------------------------
|
||||
|
||||
/** True when the URL asks for the offline fixture backend (?mock or ?dev). */
|
||||
export const MOCK: boolean = (() => {
|
||||
// --- the offline fixture backend (dev builds only) ---------------------------
|
||||
|
||||
/**
|
||||
* True when the in-memory fixtures are serving this session instead of the
|
||||
* daemon. ALWAYS false in a production build — see {@link initMockBackend}.
|
||||
*
|
||||
* A live binding, not a constant: it is decided once during boot, before the
|
||||
* first render, and every importer sees the same value for the whole session.
|
||||
*/
|
||||
export let MOCK = false
|
||||
|
||||
/** The loaded fixture module. `null` unless a dev build was asked for `?mock`. */
|
||||
let fixtures: typeof import('./mock') | null = null
|
||||
|
||||
/**
|
||||
* Load the fixture backend, if this build has one and the URL asks for it.
|
||||
* Call ONCE from the entry point and await it before the first render — the
|
||||
* pages read {@link MOCK} while they render, so flipping it afterwards would
|
||||
* leave a half-mocked screen.
|
||||
*
|
||||
* Two gates, and the order matters. `import.meta.env.DEV` is folded to a literal
|
||||
* `false` by Vite at build time, so in a production build the whole body is
|
||||
* unreachable, `import('./mock')` is tree-shaken out of the module graph, and the
|
||||
* fixtures are not in the emitted bundle AT ALL — not lazily, not behind a flag.
|
||||
* `vite.config.ts` fails the build if that ever stops being true.
|
||||
*
|
||||
* This is deliberately stronger than "hide the mock behind a query flag". The
|
||||
* flag was the bug: `?dev` on a production URL rendered an invented healthy
|
||||
* router — 119 of 122 nodes alive, "Protected" — with no request made and one
|
||||
* line of small print in the footer to say so. A person cannot audit a bundle;
|
||||
* the only honest guarantee is that the invented data is not in it.
|
||||
*/
|
||||
export async function initMockBackend(): Promise<boolean> {
|
||||
if (import.meta.env.DEV && mockRequested()) {
|
||||
fixtures = await import('./mock')
|
||||
MOCK = true
|
||||
}
|
||||
return MOCK
|
||||
}
|
||||
|
||||
/** Does the URL ask for the offline fixture backend (`?mock` or `?dev`)? */
|
||||
function mockRequested(): boolean {
|
||||
if (typeof location === 'undefined') return false
|
||||
const q = new URLSearchParams(location.search)
|
||||
return q.has('mock') || q.has('dev')
|
||||
})()
|
||||
}
|
||||
|
||||
/** The fixture backend, for the `MOCK ? … : …` branches below. Throws rather
|
||||
* than inventing data if it is ever reached without having been loaded. */
|
||||
function mock(): NonNullable<typeof fixtures> {
|
||||
if (!fixtures) throw new Error('mock backend not loaded — call initMockBackend() first')
|
||||
return fixtures
|
||||
}
|
||||
|
||||
/** A decoded response plus the raw Headers, for endpoints whose contract puts
|
||||
* pagination metadata outside the JSON body (see the stats log endpoints). */
|
||||
@@ -1050,33 +1225,54 @@ async function req<T>(path: string, init?: RequestInit): Promise<T> {
|
||||
// --- endpoints --------------------------------------------------------------
|
||||
|
||||
export function getStatus(): Promise<Status> {
|
||||
return MOCK ? mock.getStatus() : req<Status>('api/status')
|
||||
return MOCK ? mock().getStatus() : req<Status>('api/status')
|
||||
}
|
||||
|
||||
export function getConfig(): Promise<Model> {
|
||||
return MOCK ? mock.getConfig() : req<Model>('api/config')
|
||||
export async function getConfig(): Promise<Model> {
|
||||
const m = await (MOCK ? mock().getConfig() : req<Model>('api/config'))
|
||||
// Every page reads the config, and the commit-confirm window's length is the
|
||||
// only thing needed to arm a countdown — so it is captured here once instead of
|
||||
// being threaded through eight pages. See pendingConfirm.ts.
|
||||
noteConfirmTimeout(m.Globals?.ConfirmTimeout)
|
||||
return m
|
||||
}
|
||||
|
||||
export function putConfig(m: Model): Promise<{ ok: boolean; applied: boolean }> {
|
||||
return MOCK
|
||||
? mock.putConfig(m)
|
||||
? mock().putConfig(m)
|
||||
: req('api/config', { method: 'PUT', body: JSON.stringify(m) })
|
||||
}
|
||||
|
||||
export function apply(): Promise<ApplyResult> {
|
||||
return MOCK ? mock.apply() : req<ApplyResult>('api/apply', { method: 'POST' })
|
||||
/**
|
||||
* POST /api/apply.
|
||||
*
|
||||
* The daemon arms an auto-rollback on EVERY successful apply that changed
|
||||
* something (panel/api.go handleApply → ArmRollback), whichever page's button was
|
||||
* pressed. Recording it here — the one place every one of those buttons goes
|
||||
* through — is what lets the countdown and the "Keep this config" control follow
|
||||
* the operator around the panel instead of living in the Apply page's local
|
||||
* state. See pendingConfirm.ts.
|
||||
*/
|
||||
export async function apply(): Promise<ApplyResult> {
|
||||
const r = await (MOCK ? mock().apply() : req<ApplyResult>('api/apply', { method: 'POST' }))
|
||||
if (!r.error && r.changed) armPendingConfirm()
|
||||
return r
|
||||
}
|
||||
|
||||
export function confirm(): Promise<ApplyResult> {
|
||||
return MOCK ? mock.confirm() : req<ApplyResult>('api/confirm', { method: 'POST' })
|
||||
export async function confirm(): Promise<ApplyResult> {
|
||||
const r = await (MOCK ? mock().confirm() : req<ApplyResult>('api/confirm', { method: 'POST' }))
|
||||
if (!r.error) clearPendingConfirm()
|
||||
return r
|
||||
}
|
||||
|
||||
export function rollback(): Promise<ApplyResult> {
|
||||
return MOCK ? mock.rollback() : req<ApplyResult>('api/rollback', { method: 'POST' })
|
||||
export async function rollback(): Promise<ApplyResult> {
|
||||
const r = await (MOCK ? mock().rollback() : req<ApplyResult>('api/rollback', { method: 'POST' }))
|
||||
if (!r.error) clearPendingConfirm()
|
||||
return r
|
||||
}
|
||||
|
||||
export function getStats(): Promise<Stats> {
|
||||
return MOCK ? mock.getStats() : req<Stats>('api/stats')
|
||||
return MOCK ? mock().getStats() : req<Stats>('api/stats')
|
||||
}
|
||||
|
||||
// --- daemon log download ------------------------------------------------------
|
||||
@@ -1128,7 +1324,7 @@ function saveBlob(blob: Blob, filename: string): void {
|
||||
*/
|
||||
export async function downloadLog(range: LogRange): Promise<void> {
|
||||
if (MOCK) {
|
||||
saveBlob(new Blob([mock.getLogText(range)], { type: 'text/plain' }), `shater-log-${range}.txt`)
|
||||
saveBlob(new Blob([mock().getLogText(range)], { type: 'text/plain' }), `shater-log-${range}.txt`)
|
||||
return
|
||||
}
|
||||
let res: Response
|
||||
@@ -1207,14 +1403,14 @@ function logPage<T>(env: { body: T[] | null; headers: Headers }): StatsLogPage<T
|
||||
/** GET /api/stats/log — one page of the DNS query log with its cursor metadata. */
|
||||
export function getStatsLogPage(q: StatsLogQuery = {}): Promise<StatsLogPage<QueryLogEntry>> {
|
||||
return MOCK
|
||||
? mock.getStatsLogPage(q)
|
||||
? mock().getStatsLogPage(q)
|
||||
: reqFull<QueryLogEntry[] | null>(`api/stats/log${statsLogQS(q)}`).then(logPage)
|
||||
}
|
||||
|
||||
/** GET /api/stats/conns — one page of the connection log with its cursor metadata. */
|
||||
export function getStatsConnsPage(q: StatsLogQuery = {}): Promise<StatsLogPage<ConnLogEntry>> {
|
||||
return MOCK
|
||||
? mock.getStatsConnsPage(q)
|
||||
? mock().getStatsConnsPage(q)
|
||||
: reqFull<ConnLogEntry[] | null>(`api/stats/conns${statsLogQS(q)}`).then(logPage)
|
||||
}
|
||||
|
||||
@@ -1222,13 +1418,13 @@ export function getStatsConnsPage(q: StatsLogQuery = {}): Promise<StatsLogPage<C
|
||||
* Rows only; callers that tail the stream want {@link getStatsLogPage} instead. */
|
||||
export function getStatsLog(q: number | StatsLogQuery = {}): Promise<QueryLogEntry[]> {
|
||||
const o: StatsLogQuery = typeof q === 'number' ? { limit: q } : q
|
||||
return MOCK ? mock.getStatsLog(o) : req<QueryLogEntry[]>(`api/stats/log${statsLogQS(o)}`)
|
||||
return MOCK ? mock().getStatsLog(o) : req<QueryLogEntry[]>(`api/stats/log${statsLogQS(o)}`)
|
||||
}
|
||||
|
||||
/** GET /api/stats/conns — the live connection-event log (device→dest), newest first. */
|
||||
export function getStatsConns(q: number | StatsLogQuery = {}): Promise<ConnLogEntry[]> {
|
||||
const o: StatsLogQuery = typeof q === 'number' ? { limit: q } : q
|
||||
return MOCK ? mock.getStatsConns(o) : req<ConnLogEntry[]>(`api/stats/conns${statsLogQS(o)}`)
|
||||
return MOCK ? mock().getStatsConns(o) : req<ConnLogEntry[]>(`api/stats/conns${statsLogQS(o)}`)
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -1283,12 +1479,12 @@ export interface RulesReachability {
|
||||
|
||||
/** GET /api/rules/reachability — which routing rules can never fire, and why. */
|
||||
export function getRulesReachability(): Promise<RulesReachability> {
|
||||
return MOCK ? mock.getRulesReachability() : req<RulesReachability>('api/rules/reachability')
|
||||
return MOCK ? mock().getRulesReachability() : req<RulesReachability>('api/rules/reachability')
|
||||
}
|
||||
|
||||
/** GET /api/ruleset/status — remote rule-set / blocklist freshness + rule counts. */
|
||||
export function getRulesetStatus(): Promise<RulesetStatus[]> {
|
||||
return MOCK ? mock.getRulesetStatus() : req<RulesetStatus[]>('api/ruleset/status')
|
||||
return MOCK ? mock().getRulesetStatus() : req<RulesetStatus[]>('api/ruleset/status')
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -1299,7 +1495,7 @@ export function getRulesetStatus(): Promise<RulesetStatus[]> {
|
||||
*/
|
||||
export function updateRuleset(tag: string): Promise<RulesetStatus | { ok: boolean }> {
|
||||
return MOCK
|
||||
? mock.updateRuleset(tag)
|
||||
? mock().updateRuleset(tag)
|
||||
: req('api/ruleset/update', { method: 'POST', body: JSON.stringify({ tag }) })
|
||||
}
|
||||
|
||||
@@ -1327,7 +1523,7 @@ export interface RulesetCheck {
|
||||
*/
|
||||
export function checkRulesetCategory(source: string, category: string): Promise<RulesetCheck> {
|
||||
return MOCK
|
||||
? mock.checkRulesetCategory(source, category)
|
||||
? mock().checkRulesetCategory(source, category)
|
||||
: req<RulesetCheck>('api/ruleset/check', {
|
||||
method: 'POST',
|
||||
body: JSON.stringify({ source, category }),
|
||||
@@ -1356,18 +1552,18 @@ export interface RulesetCategories {
|
||||
*/
|
||||
export function getRulesetCategories(source: string): Promise<RulesetCategories> {
|
||||
return MOCK
|
||||
? mock.getRulesetCategories(source)
|
||||
? mock().getRulesetCategories(source)
|
||||
: req<RulesetCategories>(`api/ruleset/categories?source=${encodeURIComponent(source)}`)
|
||||
}
|
||||
|
||||
/** GET /api/devices — discovered LAN clients merged with per-device config. */
|
||||
export function getDevices(): Promise<DiscoveredDevice[]> {
|
||||
return MOCK ? mock.getDevices() : req<DiscoveredDevice[]>('api/devices')
|
||||
return MOCK ? mock().getDevices() : req<DiscoveredDevice[]>('api/devices')
|
||||
}
|
||||
|
||||
/** GET /api/interfaces — the router's UCI network interfaces for the egress picker. */
|
||||
export function getInterfaces(): Promise<Interface[]> {
|
||||
return MOCK ? mock.getInterfaces() : req<Interface[]>('api/interfaces')
|
||||
return MOCK ? mock().getInterfaces() : req<Interface[]>('api/interfaces')
|
||||
}
|
||||
|
||||
/** POST /api/session — exchange a single-use handoff token for a session cookie. */
|
||||
@@ -1403,7 +1599,7 @@ export function importWg(conf: string): Promise<{ uri: string; name: string }> {
|
||||
*/
|
||||
export function updateSubscription(name: string): Promise<{ added: number }> {
|
||||
return MOCK
|
||||
? mock.updateSubscription(name)
|
||||
? mock().updateSubscription(name)
|
||||
: req('api/subscription/update', { method: 'POST', body: JSON.stringify({ name }) })
|
||||
}
|
||||
|
||||
@@ -1423,7 +1619,7 @@ export function updateSubscription(name: string): Promise<{ added: number }> {
|
||||
export function getGroupsHealth(
|
||||
opts: { group?: string; members?: boolean } = {},
|
||||
): Promise<GroupsHealth> {
|
||||
if (MOCK) return mock.getGroupsHealth(opts)
|
||||
if (MOCK) return mock().getGroupsHealth(opts)
|
||||
const p = new URLSearchParams()
|
||||
if (opts.group) p.set('group', opts.group)
|
||||
if (opts.members) p.set('members', '1')
|
||||
@@ -1432,8 +1628,21 @@ export function getGroupsHealth(
|
||||
}
|
||||
|
||||
/**
|
||||
* One group's (or chain's) last test: which member the balancer picked, how fast
|
||||
* it answered, and what the internet saw as the source address.
|
||||
* What the OBSERVATORY measured for one group or chain — not a dial the panel
|
||||
* made.
|
||||
*
|
||||
* This shape used to come from a fresh connection opened on demand, straight at
|
||||
* the target. That was a lie on any router whose proxies are blocked when dialled
|
||||
* directly and work only as a hop behind a tunnel: the card reported dead for a
|
||||
* path that carries traffic all day. The daemon now has exactly one thing that
|
||||
* measures — the background observatory, which probes along the REAL dial path,
|
||||
* per-hop copies and all — and this endpoint reports what it found. There is no
|
||||
* second measurement anywhere, and the panel never opens a connection of its own.
|
||||
*
|
||||
* So read the fields as a READ, not as a test run: `ok` and `delay_ms` are the
|
||||
* observatory's verdict for the path traffic actually takes, and `tested_unix`
|
||||
* (router clock, seconds) is when the OBSERVATORY took that measurement — which
|
||||
* can be a few seconds before the refresh was asked for.
|
||||
*
|
||||
* `ok:true` with an EMPTY `exit_ip`/`exit_country` is a valid, successful result,
|
||||
* not a partial failure: the delay was measured but the exit address could not be
|
||||
@@ -1442,10 +1651,23 @@ export function getGroupsHealth(
|
||||
*
|
||||
* Chains ride the same endpoint. For a chain row, `group` carries the CHAIN's
|
||||
* name and `selected` the node its last group hop picked ('' when the exit hop
|
||||
* isn't a group). Everything else reads the same way.
|
||||
* isn't a group). Per-hop detail is a different read: {@link ChainHopHealth}.
|
||||
*
|
||||
* `ok:false` ⇒ the test failed and `error` carries the human reason; every other
|
||||
* field is meaningless. `tested_unix` is the router's clock, in seconds.
|
||||
* `ok:false` ⇒ there is no usable measurement and `error` carries the human
|
||||
* reason; every other field is meaningless. Four of those reasons are about the
|
||||
* observatory rather than the path, and must not be rendered as "your target is
|
||||
* broken":
|
||||
*
|
||||
* "not routed by any enabled rule, so nothing measures it — the observatory
|
||||
* only probes paths the rules use"
|
||||
* "the observatory has not reached this target yet — it refreshes on the
|
||||
* global probe interval"
|
||||
* "background probing is disabled, so there is nothing to measure this target
|
||||
* with"
|
||||
* "the observatory's probe through this path failed"
|
||||
*
|
||||
* Only the last one is a health finding. The first three say the measurement
|
||||
* does not exist, which is a different thing and a different fix.
|
||||
*/
|
||||
export interface GroupTestResult {
|
||||
group: string // group name — or a chain name for a chain row
|
||||
@@ -1462,7 +1684,11 @@ export interface GroupTestResult {
|
||||
* GET /api/groups/test — progress plus every result so far. `results` is ALWAYS
|
||||
* an array (never null); `done`/`total` count finished vs targeted groups and
|
||||
* chains while `running` is true. Idle reads `{running:false}` with the last
|
||||
* run's results still attached, so a reload after a test still shows what it found.
|
||||
* run's results still attached, so a reload still shows what was last read.
|
||||
*
|
||||
* "Running" means the observatory is working through an out-of-turn refresh pass
|
||||
* over the named targets and this endpoint is collecting what it measures. It is
|
||||
* not the panel dialling anything.
|
||||
*/
|
||||
export interface GroupTestStatus {
|
||||
running: boolean
|
||||
@@ -1495,18 +1721,24 @@ export interface GroupTestStart {
|
||||
}
|
||||
|
||||
/**
|
||||
* POST /api/groups/test — measure a target's delay and exit address. Pass a
|
||||
* group or chain name to test one; pass nothing (or '') to test every group
|
||||
* and every chain. Singleton: a second call while a run is in flight resolves
|
||||
* to `{started:false, reason:'already running'}` rather than failing.
|
||||
* POST /api/groups/test — ask the observatory for an out-of-turn refresh pass,
|
||||
* then report what it measured. Pass a group or chain name to refresh one; pass
|
||||
* nothing (or '') for every group and every chain.
|
||||
*
|
||||
* It does NOT dial. The observatory is the only thing in the daemon that
|
||||
* measures anything, and it measures along the real dial path — so this is the
|
||||
* "don't wait for the next probe interval" button, not a second opinion. The
|
||||
* numbers it returns are the same numbers the cards are already showing, just
|
||||
* fresher. Singleton: a second call while a pass is in flight resolves to
|
||||
* `{started:false, reason:'already running'}` rather than failing.
|
||||
*/
|
||||
export function postGroupsTest(name = ''): Promise<GroupTestStart> {
|
||||
return MOCK
|
||||
? mock.postGroupsTest(name)
|
||||
? mock().postGroupsTest(name)
|
||||
: req<GroupTestStart>('api/groups/test', { method: 'POST', body: JSON.stringify({ name }) })
|
||||
}
|
||||
|
||||
/** GET /api/groups/test — progress + results of the current/last group test. */
|
||||
export function getGroupsTest(): Promise<GroupTestStatus> {
|
||||
return MOCK ? mock.getGroupsTest() : req<GroupTestStatus>('api/groups/test')
|
||||
return MOCK ? mock().getGroupsTest() : req<GroupTestStatus>('api/groups/test')
|
||||
}
|
||||
|
||||
@@ -1,11 +1,49 @@
|
||||
import { useEffect, useState } from 'react'
|
||||
|
||||
/**
|
||||
* The panel's wall clock, in the SAME timezone as every timestamp under it.
|
||||
*
|
||||
* It used to read `getUTCHours()` and print "UTC", while `format.ts` renders every
|
||||
* log line, connection event and date through `toLocaleTimeString` — i.e. the
|
||||
* browser's zone. In Moscow that put two clocks three hours apart on one plate,
|
||||
* and the header was the one nobody could reconcile: the router's "started" time
|
||||
* read later than the current time while the uptime said it had been up for hours.
|
||||
*
|
||||
* So the clock follows the rest of the panel — local, and it SAYS which offset
|
||||
* that is, because a bare "12:41:07" beside a router in another zone is the
|
||||
* ambiguity that started this. The zone label is the browser's UTC offset, not an
|
||||
* abbreviation: "MSK"/"CEST" are not derivable everywhere, an offset always is.
|
||||
*
|
||||
* This is the BROWSER's clock, not the router's — the appliance has no RTC. Every
|
||||
* router-sourced instant in the panel is converted to this clock before it is
|
||||
* shown, which is what makes one label at the top honest for the whole page.
|
||||
*/
|
||||
function zoneLabel(d: Date): string {
|
||||
// getTimezoneOffset() is minutes WEST of UTC, so the sign is inverted.
|
||||
const min = -d.getTimezoneOffset()
|
||||
if (min === 0) return 'UTC'
|
||||
const sign = min < 0 ? '−' : '+'
|
||||
const a = Math.abs(min)
|
||||
const h = Math.floor(a / 60)
|
||||
const m = a % 60
|
||||
return `UTC${sign}${h}${m ? `:${String(m).padStart(2, '0')}` : ''}`
|
||||
}
|
||||
|
||||
function format(d: Date): string {
|
||||
const p = (n: number) => String(n).padStart(2, '0')
|
||||
return `${p(d.getUTCHours())}:${p(d.getUTCMinutes())}:${p(d.getUTCSeconds())} UTC`
|
||||
return `${p(d.getHours())}:${p(d.getMinutes())}:${p(d.getSeconds())} ${zoneLabel(d)}`
|
||||
}
|
||||
|
||||
/** Live UTC readout, tabular digits, ticking once a second. */
|
||||
/** The full zone name, for the title — "Europe/Moscow" says more than "+3" does. */
|
||||
function zoneName(): string {
|
||||
try {
|
||||
return Intl.DateTimeFormat().resolvedOptions().timeZone || ''
|
||||
} catch {
|
||||
return ''
|
||||
}
|
||||
}
|
||||
|
||||
/** Live local readout, tabular digits, ticking once a second. */
|
||||
export function Clock({ className }: { className?: string }) {
|
||||
const [now, setNow] = useState(() => format(new Date()))
|
||||
|
||||
@@ -14,5 +52,13 @@ export function Clock({ className }: { className?: string }) {
|
||||
return () => window.clearInterval(id)
|
||||
}, [])
|
||||
|
||||
return <span className={['clock', className].filter(Boolean).join(' ')}>{now}</span>
|
||||
const zone = zoneName()
|
||||
return (
|
||||
<span
|
||||
className={['clock', className].filter(Boolean).join(' ')}
|
||||
title={zone ? `Your device's clock — ${zone}. Every time in the panel is shown in this zone.` : undefined}
|
||||
>
|
||||
{now}
|
||||
</span>
|
||||
)
|
||||
}
|
||||
|
||||
@@ -0,0 +1,127 @@
|
||||
// findings.ts — which apply-time finding is shown where.
|
||||
//
|
||||
// Run with `npm test` (node's built-in test runner + native TypeScript
|
||||
// stripping; no test dependency is added to the SPA, which ships inside the
|
||||
// daemon binary).
|
||||
//
|
||||
// Two defects are pinned here.
|
||||
//
|
||||
// 1. THE TRUNCATION NOTE WAS UNREACHABLE. The daemon caps Status.warnings at 50
|
||||
// and overwrites the last slot with an `info` note counting what it dropped.
|
||||
// Overview filtered `info` away wholesale, and the settings-page route keys on
|
||||
// a section (`generate`) that no page owns — so the single line telling the
|
||||
// operator "you are not seeing all of it" reached no screen at all.
|
||||
//
|
||||
// 2. FINDINGS ABOUT AN ENTITY NEVER REACHED THAT ENTITY'S PAGE. The generator
|
||||
// drops a node it cannot build and names it; the Nodes page rendered that node
|
||||
// as an ordinary row with a green toggle, because it never read the findings.
|
||||
|
||||
import { test } from 'node:test'
|
||||
import assert from 'node:assert/strict'
|
||||
|
||||
import {
|
||||
attentionFindings,
|
||||
entityFindings,
|
||||
findingsByName,
|
||||
sectionNotes,
|
||||
truncationNote,
|
||||
worstSeverity,
|
||||
} from './findings.ts'
|
||||
import type { StatusWarning } from './api.ts'
|
||||
|
||||
const crit = (section: string, name: string, message = 'broken'): StatusWarning => ({
|
||||
severity: 'critical',
|
||||
section,
|
||||
name,
|
||||
message,
|
||||
})
|
||||
const warn = (section: string, name: string, message = 'degraded'): StatusWarning => ({
|
||||
severity: 'warning',
|
||||
section,
|
||||
name,
|
||||
message,
|
||||
})
|
||||
const info = (section: string, name: string, message: string): StatusWarning => ({
|
||||
severity: 'info',
|
||||
section,
|
||||
name,
|
||||
message,
|
||||
})
|
||||
|
||||
/** Verbatim from apply/warnings.go finalizeWarnings. */
|
||||
const SUPPRESSED = info(
|
||||
'generate',
|
||||
'',
|
||||
'7 further warning(s) suppressed; run `logread -e shater` for the full list',
|
||||
)
|
||||
|
||||
// --- the truncation note ----------------------------------------------------
|
||||
|
||||
test('the truncation note is found, whatever else is in the list', () => {
|
||||
const note = truncationNote([crit('rule', 'a'), warn('node', 'b'), SUPPRESSED])
|
||||
assert.notEqual(note, null)
|
||||
assert.match(note!.message, /7 further warning/)
|
||||
})
|
||||
|
||||
test('a whole list has no truncation note', () => {
|
||||
assert.equal(truncationNote([crit('rule', 'a'), warn('node', 'b')]), null)
|
||||
assert.equal(truncationNote([]), null)
|
||||
assert.equal(truncationNote(undefined), null)
|
||||
})
|
||||
|
||||
test('an ordinary info note is not mistaken for the truncation note', () => {
|
||||
const notes = [info('untunnelable', 'block', 'Ping and traceroute do not work…')]
|
||||
assert.equal(truncationNote(notes), null)
|
||||
})
|
||||
|
||||
test('the truncation note is kept out of the settings-page notes it would pollute', () => {
|
||||
const all = [info('generate', '', 'cache: moved to /overlay'), SUPPRESSED]
|
||||
const notes = sectionNotes(all, 'generate')
|
||||
assert.equal(notes.length, 1)
|
||||
assert.match(notes[0].message, /cache:/)
|
||||
})
|
||||
|
||||
test('the attention list still carries only critical and warning', () => {
|
||||
const all = [crit('rule', 'a'), warn('node', 'b'), info('untunnelable', 'block', 'x'), SUPPRESSED]
|
||||
const attention = attentionFindings(all)
|
||||
assert.equal(attention.length, 2)
|
||||
assert.ok(attention.every((w) => w.severity !== 'info'))
|
||||
})
|
||||
|
||||
// --- per-entity findings ----------------------------------------------------
|
||||
|
||||
test('a page takes only the sections it owns', () => {
|
||||
const all = [
|
||||
crit('node', 'tokyo-01', 'parse share-link: bad scheme (skipped)'),
|
||||
warn('subscription', 'qomar', 'fetch failed'),
|
||||
crit('rule', 'default', 'never applies'),
|
||||
info('generate', '', 'cache: x'),
|
||||
]
|
||||
const mine = entityFindings(all, ['node', 'subscription'])
|
||||
assert.deepEqual(
|
||||
mine.map((w) => w.name),
|
||||
['tokyo-01', 'qomar'],
|
||||
)
|
||||
})
|
||||
|
||||
test('entity findings never include info notes', () => {
|
||||
const all = [info('node', 'tokyo-01', 'just a note'), SUPPRESSED]
|
||||
assert.equal(entityFindings(all, ['node', 'generate']).length, 0)
|
||||
})
|
||||
|
||||
test('findings index by name, and global (unnamed) ones are left out', () => {
|
||||
const all = [
|
||||
crit('node', 'tokyo-01', 'first'),
|
||||
warn('node', 'tokyo-01', 'second'),
|
||||
crit('node', '', 'global to the section'),
|
||||
]
|
||||
const byName = findingsByName(entityFindings(all, ['node']))
|
||||
assert.equal(byName.size, 1)
|
||||
assert.equal(byName.get('tokyo-01')!.length, 2)
|
||||
})
|
||||
|
||||
test('one lamp per row takes the loudest severity', () => {
|
||||
assert.equal(worstSeverity([warn('node', 'a'), crit('node', 'a')]), 'critical')
|
||||
assert.equal(worstSeverity([warn('node', 'a')]), 'warning')
|
||||
assert.equal(worstSeverity([]), null)
|
||||
})
|
||||
+91
-2
@@ -6,7 +6,8 @@
|
||||
//
|
||||
// critical / warning — something needs attention: a protection promise is
|
||||
// broken, or something you configured isn't in effect. These belong on
|
||||
// Overview, where the operator looks first.
|
||||
// Overview, where the operator looks first — and, when they name an entity,
|
||||
// ALSO on the page that owns that entity (see `entityFindings`).
|
||||
//
|
||||
// info — a statement ABOUT the configuration, not a problem. It never clears,
|
||||
// because nothing is wrong: it is simply describing a choice that was made.
|
||||
@@ -16,9 +17,49 @@
|
||||
// page that never goes away and never asks for anything trains people to skim
|
||||
// the list — which is exactly how a real critical finding gets missed. Anything
|
||||
// standing in the findings list should be something you could act on.
|
||||
//
|
||||
// The one exception is carved out below: the daemon's own note that it dropped
|
||||
// findings to fit the cap. It is `info` by severity and unactionable by nature,
|
||||
// and it is the single most important line in the list, because it is the list
|
||||
// telling you it is not the whole list.
|
||||
|
||||
import type { StatusWarning } from './api'
|
||||
|
||||
/**
|
||||
* The daemon's truncation disclosure, verbatim from apply/warnings.go
|
||||
* finalizeWarnings:
|
||||
*
|
||||
* "%d further warning(s) suppressed; run `logread -e shater` for the full list"
|
||||
*
|
||||
* Matched on the stable clause rather than the whole sentence so a reworded tail
|
||||
* still registers. If this ever stops matching, the failure mode is a list that
|
||||
* silently claims to be complete — which is why `truncationNote` is tested.
|
||||
*/
|
||||
const SUPPRESSED_RE = /further warning\(s\) suppressed/
|
||||
|
||||
/**
|
||||
* The daemon's "this list is incomplete" note, or null when the list is whole.
|
||||
*
|
||||
* Status.warnings is capped at 50, sorted critical-first, and the last slot is
|
||||
* REPLACED by an `info` note counting what was dropped. That note therefore
|
||||
* arrives on the one channel the panel filtered away wholesale: `info` never
|
||||
* reached Overview, and the settings-page route (`sectionNotes`) keys on
|
||||
* section `generate`, which no page owns. So the single line saying "there are
|
||||
* findings you are not being shown" was the only one guaranteed to be invisible.
|
||||
*
|
||||
* Callers must render this WITH the attention list, not instead of it.
|
||||
*/
|
||||
export function truncationNote(warnings: StatusWarning[] | undefined): StatusWarning | null {
|
||||
return (
|
||||
(warnings ?? []).find((w) => w.severity === 'info' && SUPPRESSED_RE.test(w.message)) ?? null
|
||||
)
|
||||
}
|
||||
|
||||
/** Is this the truncation disclosure rather than an ordinary note? */
|
||||
function isTruncationNote(w: StatusWarning): boolean {
|
||||
return w.severity === 'info' && SUPPRESSED_RE.test(w.message)
|
||||
}
|
||||
|
||||
/** Findings that need attention — the Overview list. Info notes are excluded. */
|
||||
export function attentionFindings(warnings: StatusWarning[] | undefined): StatusWarning[] {
|
||||
return (warnings ?? []).filter((w) => w.severity === 'critical' || w.severity === 'warning')
|
||||
@@ -29,10 +70,58 @@ export function attentionFindings(warnings: StatusWarning[] | undefined): Status
|
||||
* (e.g. `untunnelable` → the Networks page's "Other traffic" section). Only info:
|
||||
* a critical/warning is an attention item and stays on Overview, so it can't be
|
||||
* quietly buried on a settings page instead.
|
||||
*
|
||||
* The truncation note is excluded: it is about the LIST, not about any section,
|
||||
* and it has its own home beside the list ({@link truncationNote}).
|
||||
*/
|
||||
export function sectionNotes(
|
||||
warnings: StatusWarning[] | undefined,
|
||||
section: string,
|
||||
): StatusWarning[] {
|
||||
return (warnings ?? []).filter((w) => w.severity === 'info' && w.section === section)
|
||||
return (warnings ?? []).filter(
|
||||
(w) => w.severity === 'info' && w.section === section && !isTruncationNote(w),
|
||||
)
|
||||
}
|
||||
|
||||
/**
|
||||
* The attention findings about entities ONE page owns — for that page to show
|
||||
* beside the entities themselves.
|
||||
*
|
||||
* Overview is where you look when you already suspect something; a page like
|
||||
* Nodes is where you look when you don't. The generator drops a node it cannot
|
||||
* build — an unparseable share link, a WireGuard key materialised twice — and
|
||||
* says so by name ("node \"x\": parse share-link: … (skipped)"), yet that node
|
||||
* kept rendering as an ordinary row with a green toggle, because the page never
|
||||
* read the findings at all. The switch says on; the engine has no such outbound.
|
||||
*
|
||||
* This does NOT move anything off Overview: the same finding appears in both
|
||||
* places, which is correct — one list is "what is wrong with this router", the
|
||||
* other is "what is wrong with this node".
|
||||
*/
|
||||
export function entityFindings(
|
||||
warnings: StatusWarning[] | undefined,
|
||||
sections: readonly string[],
|
||||
): StatusWarning[] {
|
||||
const want = new Set(sections)
|
||||
return attentionFindings(warnings).filter((w) => want.has(w.section))
|
||||
}
|
||||
|
||||
/** Index attention findings by entity name, for badging a row directly. Entries
|
||||
* with an empty `name` are global to their section and are left out. */
|
||||
export function findingsByName(findings: StatusWarning[]): Map<string, StatusWarning[]> {
|
||||
const out = new Map<string, StatusWarning[]>()
|
||||
for (const f of findings) {
|
||||
if (!f.name) continue
|
||||
const list = out.get(f.name)
|
||||
if (list) list.push(f)
|
||||
else out.set(f.name, [f])
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
/** The loudest severity in a set — for a row badge that has room for one lamp. */
|
||||
export function worstSeverity(findings: StatusWarning[]): 'critical' | 'warning' | null {
|
||||
if (findings.some((f) => f.severity === 'critical')) return 'critical'
|
||||
if (findings.length > 0) return 'warning'
|
||||
return null
|
||||
}
|
||||
|
||||
@@ -77,6 +77,39 @@ export function fmtDateTime(unix: number): string {
|
||||
return d && t ? `${d}, ${t}` : d || t
|
||||
}
|
||||
|
||||
// --- remote-list freshness ---------------------------------------------------
|
||||
// A url/geo-sourced list re-fetches on a cadence and the engine reports when it
|
||||
// last pulled (GET /api/ruleset/status). Routing shows this for rule-sets and DNS
|
||||
// shows it for blocklists, so the two readings live here and cannot drift apart.
|
||||
|
||||
/** "updated 3h ago" / "never updated" for a remote list's last fetch (RFC3339). */
|
||||
export function relFetch(iso: string): string {
|
||||
if (!iso) return 'never updated'
|
||||
const t = Date.parse(iso)
|
||||
if (Number.isNaN(t)) return 'never updated'
|
||||
const s = Math.max(0, Math.floor((Date.now() - t) / 1000))
|
||||
if (s < 45) return 'updated just now'
|
||||
const m = Math.floor(s / 60)
|
||||
if (m < 60) return `updated ${m}m ago`
|
||||
const h = Math.floor(m / 60)
|
||||
if (h < 24) return `updated ${h}h ago`
|
||||
const d = Math.floor(h / 24)
|
||||
return `updated ${d}d ago`
|
||||
}
|
||||
|
||||
/** "every 24h" for an auto-update cadence in seconds ("" when there is none). */
|
||||
export function everyLabel(sec: number): string {
|
||||
if (!sec || sec <= 0) return ''
|
||||
if (sec % 3600 === 0) {
|
||||
const h = sec / 3600
|
||||
if (h < 48) return `every ${h}h`
|
||||
if (sec % 86400 === 0) return `every ${sec / 86400}d`
|
||||
return `every ${h}h`
|
||||
}
|
||||
if (sec % 60 === 0) return `every ${sec / 60}m`
|
||||
return `every ${sec}s`
|
||||
}
|
||||
|
||||
/**
|
||||
* A coarse "how long until / since" reading for a unix deadline, relative to now.
|
||||
*
|
||||
|
||||
+25
-9
@@ -2,17 +2,33 @@ import { StrictMode } from 'react'
|
||||
import { createRoot } from 'react-dom/client'
|
||||
import './tokens.css'
|
||||
import { App } from './App'
|
||||
import { initMockBackend } from './api'
|
||||
import { ConfirmProvider } from './components'
|
||||
|
||||
const rootEl = document.getElementById('root')
|
||||
if (!rootEl) throw new Error('#root not found')
|
||||
|
||||
// ConfirmProvider sits ABOVE <App> so it survives App's early returns (the
|
||||
// unauth / no-link plates) — useConfirm() can never find itself without a host.
|
||||
createRoot(rootEl).render(
|
||||
<StrictMode>
|
||||
<ConfirmProvider>
|
||||
<App />
|
||||
</ConfirmProvider>
|
||||
</StrictMode>,
|
||||
)
|
||||
// Settle the fixture question BEFORE the first render: pages read `MOCK` while
|
||||
// they render, so a backend that arrives afterwards would paint half a screen
|
||||
// from the daemon and half from fixtures. In a production build this resolves
|
||||
// immediately and to `false` — the fixtures are not in the bundle to load (see
|
||||
// api.ts initMockBackend and the assertNoMockFixtures plugin in vite.config.ts).
|
||||
function mount() {
|
||||
// ConfirmProvider sits ABOVE <App> so it survives App's early returns (the
|
||||
// unauth / no-link plates) — useConfirm() can never find itself without a host.
|
||||
createRoot(rootEl!).render(
|
||||
<StrictMode>
|
||||
<ConfirmProvider>
|
||||
<App />
|
||||
</ConfirmProvider>
|
||||
</StrictMode>,
|
||||
)
|
||||
}
|
||||
|
||||
// A fixture module that fails to load is a broken dev checkout, not a reason to
|
||||
// hand the operator a blank plate — mount anyway and let the shell report that it
|
||||
// cannot reach a daemon, which by then is the truth.
|
||||
void initMockBackend().then(mount, (e) => {
|
||||
console.error('mock backend failed to load; continuing against the real API', e)
|
||||
mount()
|
||||
})
|
||||
|
||||
+202
-37
@@ -6,7 +6,7 @@
|
||||
// state mutates in-memory so the Apply / Confirm / Rollback flow is exercisable.
|
||||
//
|
||||
// Type-only imports from api.ts (erased at build) keep this free of a runtime cycle.
|
||||
import type { ApplyResult, ChainHealth, ConnLogEntry, DiscoveredDevice, GroupHealth, GroupMemberHealth, GroupsHealth, GroupTestResult, GroupTestStart, GroupTestStatus, Interface, Model, Profile, QueryLogEntry, RuleReach, RulesReachability, RulesetCategories, RulesetCheck, RulesetStatus, Stats, StatsLogPage, StatsLogQuery, Status, StatusWarning, Traffic } from './api'
|
||||
import type { ApplyResult, ChainHealth, ChainHopHealth, ConnLogEntry, DiscoveredDevice, GroupHealth, GroupMemberHealth, GroupsHealth, GroupTestResult, GroupTestStart, GroupTestStatus, Interface, Model, Profile, QueryLogEntry, RuleReach, RulesReachability, RulesetCategories, RulesetCheck, RulesetStatus, Stats, StatsLogPage, StatsLogQuery, Status, StatusWarning, Traffic } from './api'
|
||||
|
||||
let armed = false // a pending commit-confirm auto-rollback
|
||||
let hasLastGood = false // a predecessor config exists to roll back to (post-apply)
|
||||
@@ -129,9 +129,25 @@ const CONFIG: Model = {
|
||||
{ Name: 'via-tunnel', Source: 'subscription', Subscription: 'primary', Strategy: 'leastping', Egress: 'awg' },
|
||||
{ Name: 'fallback', Source: 'subscription', Subscription: 'backup', Strategy: 'roundrobin', Egress: '' },
|
||||
],
|
||||
// One multi-hop chain so `?mock` exercises the chain card's Test button and
|
||||
// its result readout: enters through the awg tunnel, exits via the auto group.
|
||||
Chains: [{ Name: 'relay', Hops: ['egress:awg', 'group:auto'] }],
|
||||
// Three chains, one per state the hop readout has to render.
|
||||
Chains: [
|
||||
// The owner's real production shape: leave through a WAN interface, cross an
|
||||
// AmneziaWG node, then three subscription groups in series. The leading
|
||||
// `egress:` is NOT a numbered hop — the daemon lifts it into hop 1's entry
|
||||
// detour — so this reports FOUR hops, and hop 3 is dead while its neighbours
|
||||
// answer. That single red notch in the middle of a live path is the entire
|
||||
// reason per-hop health exists, so `?mock` must show it at a glance.
|
||||
{
|
||||
Name: 'ewan-wg-subs',
|
||||
Hops: ['egress:wan', 'node:home-wg', 'group:auto', 'group:stealth', 'group:via-tunnel'],
|
||||
},
|
||||
// Used, but the observatory hasn't come round yet — every hop untested. Not
|
||||
// dead and not healthy: the state the panel most easily renders as a fault.
|
||||
{ Name: 'sub-fresh', Hops: ['node:home-wg', 'group:fallback'] },
|
||||
// No enabled rule targets it, so the observatory skips it entirely and the
|
||||
// daemon never materialises its hops: `used:false` and NO `hops` key.
|
||||
{ Name: 'relay', Hops: ['egress:awg', 'group:auto'] },
|
||||
],
|
||||
Egresses: [
|
||||
{ Name: 'wan', Type: 'interface', Interface: 'wan' },
|
||||
// An AmneziaWG tunnel — the whole point of a group-level egress binding.
|
||||
@@ -145,6 +161,13 @@ const CONFIG: Model = {
|
||||
Rules: [
|
||||
{ Name: 'block-ads', Enabled: true, Order: 10, DstRuleset: ['ad-hosts'], Target: 'block' },
|
||||
{ Name: 'ru-bypass', Enabled: true, Order: 20, DstRuleset: ['ru-inside'], Target: 'direct' },
|
||||
// These two are what make the chains USED — the observatory probes only the
|
||||
// paths an enabled rule can reach, so without them every chain card would
|
||||
// read "not routed" and the hop rail would never appear in `?mock`. Kept
|
||||
// ABOVE the condition-less rule at Order 40, which would otherwise swallow
|
||||
// everything below it and mark them "never applies".
|
||||
{ Name: 'media-via-chain', Enabled: true, Order: 22, DstRuleset: ['yt-geosite'], Target: 'chain:ewan-wg-subs' },
|
||||
{ Name: 'spare-via-chain', Enabled: true, Order: 24, DstRuleset: ['ad-hosts'], Target: 'chain:sub-fresh' },
|
||||
{ Name: 'private-direct', Enabled: true, Order: 30, DstRuleset: ['private-nets'], Target: 'direct' },
|
||||
// A SECOND condition-less rule, above the real default. It reads like a working
|
||||
// rule and does nothing: a rule with no conditions becomes the router's default,
|
||||
@@ -171,6 +194,7 @@ const CONFIG: Model = {
|
||||
// to an official remote list; the others are the usual url / inline lists.
|
||||
Blocklists: [
|
||||
{ Name: 'StevenBlack', Enabled: true, Source: 'url', URL: 'https://raw.githubusercontent.com/StevenBlack/hosts/master/hosts', Response: 'nxdomain', UpdateInterval: '24h' },
|
||||
{ Name: 'oisd-basic', Enabled: true, Source: 'url', URL: 'https://big.oisd.nl/domainswild', Response: 'nxdomain', UpdateInterval: '24h' },
|
||||
{ Name: 'telegram-block', Enabled: false, Source: 'geosite', Categories: ['telegram'], Response: 'nxdomain', UpdateInterval: '24h' },
|
||||
],
|
||||
Resolvers: [
|
||||
@@ -300,6 +324,24 @@ const RULESET_STATUS: RulesetStatus[] = [
|
||||
rule_count: 903,
|
||||
},
|
||||
{ tag: 'rs-ru-geoip-ru', name: 'ru-geoip', category: 'ru', kind: 'ruleset', remote: true, last_updated: '', interval_seconds: 86_400, rule_count: 0 },
|
||||
// Blocklists report through the same endpoint under `bl-<name>`, which the DNS
|
||||
// page never asked for — so a list that has NEVER been fetched still read
|
||||
// "filtering". StevenBlack is that case here; oisd-basic is the healthy one, so
|
||||
// both readings are exercisable offline.
|
||||
{ tag: 'bl-StevenBlack', name: 'StevenBlack', category: '', kind: 'blocklist', remote: true, last_updated: '', interval_seconds: 86_400, rule_count: 0 },
|
||||
{
|
||||
tag: 'bl-oisd-basic',
|
||||
name: 'oisd-basic',
|
||||
category: '',
|
||||
kind: 'blocklist',
|
||||
remote: true,
|
||||
last_updated: new Date(Date.now() - 6 * 3600_000).toISOString(),
|
||||
interval_seconds: 86_400,
|
||||
rule_count: 218_431,
|
||||
},
|
||||
// Disabled in CONFIG, so the row reads "off" whatever this says — it exists to
|
||||
// prove the row does not start claiming things the moment a status appears.
|
||||
{ tag: 'bl-telegram-block-telegram', name: 'telegram-block', category: 'telegram', kind: 'blocklist', remote: true, last_updated: '', interval_seconds: 86_400, rule_count: 0 },
|
||||
]
|
||||
|
||||
/** GET /api/rules/reachability. Mirrors the daemon's analysis over CONFIG.Rules:
|
||||
@@ -458,7 +500,15 @@ export async function getRulesetCategories(source: string): Promise<RulesetCateg
|
||||
// the field case the readout used to call "Protected" (one
|
||||
// rule, `default → direct`); `unknown` is a daemon too old to
|
||||
// report. Default: tunnel.
|
||||
function mockPlane(): { plane: 'full' | 'hold' | 'none'; engine: boolean; killSwitch: string } {
|
||||
// ?mock&plane=unreported → a daemon that sends NO `plane` field. The panel then
|
||||
// knows nothing about what is installed, which is the state
|
||||
// the Kill-switch module used to render as a green "ARMED"
|
||||
// (`undefined !== 'none'` is true).
|
||||
function mockPlane(): {
|
||||
plane: 'full' | 'hold' | 'none' | undefined
|
||||
engine: boolean
|
||||
killSwitch: string
|
||||
} {
|
||||
const q = typeof location === 'undefined' ? '' : location.search
|
||||
const params = new URLSearchParams(q)
|
||||
const killSwitch = params.get('ks') === 'open' ? 'open' : 'closed'
|
||||
@@ -466,6 +516,7 @@ function mockPlane(): { plane: 'full' | 'hold' | 'none'; engine: boolean; killSw
|
||||
if (p === 'hold') return { plane: 'hold', engine: false, killSwitch: 'closed' }
|
||||
if (p === 'none') return { plane: 'none', engine: false, killSwitch }
|
||||
if (p === 'open') return { plane: 'none', engine: false, killSwitch: 'open' }
|
||||
if (p === 'unreported') return { plane: undefined, engine: true, killSwitch }
|
||||
return { plane: 'full', engine: true, killSwitch }
|
||||
}
|
||||
|
||||
@@ -473,7 +524,7 @@ function mockPlane(): { plane: 'full' | 'hold' | 'none'; engine: boolean; killSw
|
||||
// meaningful with the plane installed: with the engine down there is no running
|
||||
// config to judge, and the daemon reports the unknown/zero value — so do the same
|
||||
// here rather than leaving a stale "tunnel" behind a dead engine.
|
||||
function mockTraffic(plane: 'full' | 'hold' | 'none'): Traffic | undefined {
|
||||
function mockTraffic(plane: 'full' | 'hold' | 'none' | undefined): Traffic | undefined {
|
||||
if (plane !== 'full') return { verdict: '', default: '', tunnel_rules: 0 }
|
||||
const params = new URLSearchParams(typeof location === 'undefined' ? '' : location.search)
|
||||
switch (params.get('traffic')) {
|
||||
@@ -518,6 +569,29 @@ const MOCK_WARNINGS: StatusWarning[] = [
|
||||
name: 'fakeip-pool',
|
||||
message: 'fake-IP resolver cannot be used as a fallback; the failover chain was not built',
|
||||
},
|
||||
// Two findings the generator attributes to a NODE by name — the class that the
|
||||
// Nodes page never showed, leaving a node the engine threw away rendered as an
|
||||
// ordinary row with a green toggle. Both name real fixture nodes so the row
|
||||
// badge, the collapsed-bucket "N flagged" count and the per-row strip all fire.
|
||||
{
|
||||
severity: 'warning',
|
||||
section: 'node',
|
||||
name: 'fi-trojan',
|
||||
message: 'parse share-link: unsupported scheme "trojan+ws" (skipped)',
|
||||
},
|
||||
{
|
||||
severity: 'warning',
|
||||
section: 'node',
|
||||
name: 'home-wg',
|
||||
message:
|
||||
'this WireGuard node is materialised twice in the engine config — as "home-wg" and as "group-stealth-m1-home-wg" — and traffic can reach both. A WireGuard peer keeps ONE session per public key, so two devices built from one private key evict each other continuously and NEITHER tunnel passes traffic. Only "home-wg" is kept; everything that routed through "group-stealth-m1-home-wg" is fail-closed (blocked) instead of leaving over the plain WAN',
|
||||
},
|
||||
{
|
||||
severity: 'warning',
|
||||
section: 'subscription',
|
||||
name: 'backup',
|
||||
message: 'fetch failed: dial tcp 203.0.113.9:443: i/o timeout — serving the nodes cached earlier',
|
||||
},
|
||||
{
|
||||
severity: 'info',
|
||||
section: 'generate',
|
||||
@@ -526,6 +600,19 @@ const MOCK_WARNINGS: StatusWarning[] = [
|
||||
},
|
||||
]
|
||||
|
||||
/**
|
||||
* The daemon's truncation disclosure, exactly as apply/warnings.go writes it when
|
||||
* the published set overflows the 50-entry cap. Served under `?mock&trunc` so the
|
||||
* "this list is incomplete" rendering is exercisable — it used to be dropped
|
||||
* wholesale by the panel's `info` filter and reached no screen at all.
|
||||
*/
|
||||
const MOCK_TRUNCATION: StatusWarning = {
|
||||
severity: 'info',
|
||||
section: 'generate',
|
||||
name: '',
|
||||
message: '7 further warning(s) suppressed; run `logread -e shater` for the full list',
|
||||
}
|
||||
|
||||
/**
|
||||
* The standing `untunnelable` note the daemon reports. It is INFO, never a
|
||||
* problem: it states a correct, chosen configuration. Two shapes, mirroring the
|
||||
@@ -572,9 +659,12 @@ function mockWarnings(killSwitch: string): StatusWarning[] {
|
||||
const params = new URLSearchParams(q)
|
||||
const mode = (CONFIG.Globals as { Untunnelable?: string }).Untunnelable ?? 'block'
|
||||
const notes = untunnelableNote(mode, killSwitch)
|
||||
// `?trunc` adds the daemon's "the published list is capped" disclosure, which
|
||||
// it appends IN PLACE OF the last entry it had room for.
|
||||
const trunc = params.has('trunc') ? [{ ...MOCK_TRUNCATION }] : []
|
||||
// A degraded plane always comes with the findings that explain it.
|
||||
if (params.has('warn') || params.get('plane')) {
|
||||
return [...MOCK_WARNINGS.map((w) => ({ ...w })), ...notes]
|
||||
if (params.has('warn') || params.get('plane') || trunc.length > 0) {
|
||||
return [...MOCK_WARNINGS.map((w) => ({ ...w })), ...notes, ...trunc]
|
||||
}
|
||||
return notes
|
||||
}
|
||||
@@ -1172,20 +1262,59 @@ function healthList(): GroupHealth[] {
|
||||
return (CONFIG.Groups ?? []).map((g) => summarise(g.Name, GROUP_MEMBERS.get(g.Name) ?? []))
|
||||
}
|
||||
|
||||
/** Per-chain reachability for the Targets page's "unused" badge (plan §5.E) — the
|
||||
* chain analogue of healthList's `used` field. The mock's single chain `relay` is
|
||||
* NOT referenced by any rule in CONFIG.Rules (they target group:auto / block /
|
||||
* direct), so it reads used=false and its card renders "unused" — exactly the case
|
||||
* the badge exists to surface. A stopped engine reports no chains. */
|
||||
/**
|
||||
* Per-hop health, keyed by chain name — what the observatory measured at each
|
||||
* position of the path, in WIRE order.
|
||||
*
|
||||
* `ewan-wg-subs` is the fixture that matters, and it encodes the ORDERED WALK.
|
||||
* Hop 1 is the WireGuard node and answers; hop 2 is a subscription group whose
|
||||
* copies answer THROUGH it — 119 of 122 tested alive, which is the reading only a
|
||||
* per-hop probe can produce, since the same members are dialled differently on
|
||||
* their own card. Hop 3 is a group whose members all time out at that position,
|
||||
* and the walk STOPS there: hop 4 is dialled through hop 3, so it was never
|
||||
* dialled at all. It comes back `untested` with `blocked_by` naming hop 3, its
|
||||
* counters zeroed, and `selected` still set — the wrapper has a pick, nothing
|
||||
* crossed it to measure. A dead hop with a live hop under it is not in this
|
||||
* fixture because the daemon can no longer produce one.
|
||||
*
|
||||
* `sub-fresh` is used but never yet reached: every hop untested, nothing dead,
|
||||
* no block — the other reason a lamp is unlit, and the one that fixes itself.
|
||||
* `relay` is absent from this map on purpose — an unused chain is never
|
||||
* materialised, so the daemon sends no `hops` key at all, which is "nothing
|
||||
* measured", not "no hops".
|
||||
*/
|
||||
const CHAIN_HOPS: Record<string, ChainHopHealth[]> = {
|
||||
'ewan-wg-subs': [
|
||||
{ index: 1, tag: 'chain-ewan-wg-subs-h1', kind: 'node', exit: false, state: 'alive', delay_ms: 41, age_seconds: 22, selected: '', total: 1, tested: 1, alive: 1, dead: 0, untested: 0 },
|
||||
{ index: 2, tag: 'chain-ewan-wg-subs-h2', kind: 'group', exit: false, state: 'alive', delay_ms: 96, age_seconds: 18, selected: '🇳🇱 Amsterdam-01', total: 298, tested: 122, alive: 119, dead: 3, untested: 176 },
|
||||
{ index: 3, tag: 'chain-ewan-wg-subs-h3', kind: 'group', exit: false, state: 'dead', delay_ms: 0, age_seconds: 15, selected: '', total: 2, tested: 2, alive: 0, dead: 2, untested: 0 },
|
||||
{ index: 4, tag: 'chain-ewan-wg-subs-h4', kind: 'group', exit: true, state: 'untested', delay_ms: 0, age_seconds: -1, selected: '🇸🇬 Singapore-09', total: 6, tested: 0, alive: 0, dead: 0, untested: 6, blocked_by: { index: 3, tag: 'chain-ewan-wg-subs-h3' } },
|
||||
],
|
||||
'sub-fresh': [
|
||||
{ index: 1, tag: 'chain-sub-fresh-h1', kind: 'node', exit: false, state: 'untested', delay_ms: 0, age_seconds: -1, selected: '', total: 1, tested: 0, alive: 0, dead: 0, untested: 1 },
|
||||
{ index: 2, tag: 'chain-sub-fresh-h2', kind: 'group', exit: true, state: 'untested', delay_ms: 0, age_seconds: -1, selected: '', total: 24, tested: 0, alive: 0, dead: 0, untested: 24 },
|
||||
],
|
||||
}
|
||||
|
||||
/** Per-chain reachability plus per-hop health for the Targets page. `used` is the
|
||||
* chain analogue of healthList's field; `hops` is OMITTED (never null, never []),
|
||||
* exactly like the daemon, for a chain the engine never materialised. A stopped
|
||||
* engine reports no chains at all. */
|
||||
function chainHealthList(): ChainHealth[] {
|
||||
if (!mockPlane().engine) return []
|
||||
return (CONFIG.Chains ?? []).map((c) => ({ name: c.Name, used: chainUsed(c.Name) }))
|
||||
return (CONFIG.Chains ?? []).map((c) => {
|
||||
const hops = CHAIN_HOPS[c.Name]
|
||||
const h: ChainHealth = { name: c.Name, used: chainUsed(c.Name) }
|
||||
if (hops) h.hops = hops.map((x) => ({ ...x }))
|
||||
return h
|
||||
})
|
||||
}
|
||||
|
||||
/** A chain is "used" when some enabled routing rule (or Final, or a DNS detour)
|
||||
* targets `chain:<name>` — the same reachability the daemon's observatory derives.
|
||||
* The mock's rules never target a chain, so every chain reads used=false; a real
|
||||
* config would mark the ones rules point at used=true. */
|
||||
* Two of the mock's rules do (`media-via-chain` → ewan-wg-subs, `spare-via-chain`
|
||||
* → sub-fresh), so those two chains read used=true and get a hop rail; `relay`
|
||||
* is targeted by nothing and reads used=false, which is the unused note. */
|
||||
function chainUsed(name: string): boolean {
|
||||
const target = `chain:${name}`
|
||||
return (CONFIG.Rules ?? []).some(
|
||||
@@ -1237,15 +1366,23 @@ class ApiErrorLike extends Error {
|
||||
}
|
||||
}
|
||||
|
||||
// Mock group/chain test. Deliberately covers every state the UI has to render,
|
||||
// one per target, so a single offline run exercises all of them:
|
||||
// auto → ok WITH an exit address
|
||||
// stealth → ok WITHOUT one (delay measured, address undeterminable) — a
|
||||
// SUCCESS, and the case the UI most easily gets wrong
|
||||
// relay → the chain: same wire shape, `group` carries the CHAIN's name and
|
||||
// `selected` the node its exit group picked
|
||||
// fallback → a failure carrying a human reason
|
||||
// Results land one per GET poll, so the running/progress state is visible too.
|
||||
// Mock refresh results. The endpoint no longer dials anything: it asks the
|
||||
// observatory to measure out of turn and reports what the observatory found, so
|
||||
// every row here is a READ of a background measurement. Deliberately covers every
|
||||
// state the UI has to render, one per target, so a single offline run exercises
|
||||
// all of them:
|
||||
// auto → ok WITH an exit address
|
||||
// stealth → ok WITHOUT one (delay measured, address undeterminable) — a
|
||||
// SUCCESS, and the case the UI most easily gets wrong
|
||||
// ewan-wg-subs → the chain: same wire shape, `group` carries the CHAIN's name
|
||||
// and `selected` the node its exit hop picked
|
||||
// via-tunnel → the one honest health FAILURE: a probe that ran and failed
|
||||
// fallback,
|
||||
// relay → not routed at all, so no measurement exists to report
|
||||
// sub-fresh → routed, but the observatory hasn't come round yet
|
||||
// The last three are absence of measurement, not a broken target, and the copy
|
||||
// has to keep them apart. Results land one per GET poll, so the running/progress
|
||||
// state is visible too.
|
||||
const GROUP_TEST_SHAPE: Record<string, Omit<GroupTestResult, 'group' | 'tested_unix'>> = {
|
||||
auto: {
|
||||
selected: 'nl-reality-2',
|
||||
@@ -1263,32 +1400,60 @@ const GROUP_TEST_SHAPE: Record<string, Omit<GroupTestResult, 'group' | 'tested_u
|
||||
ok: true,
|
||||
error: '',
|
||||
},
|
||||
// The chain — Selected is the node the chain's exit group (auto) picked.
|
||||
relay: {
|
||||
selected: 'nl-reality-2',
|
||||
delay_ms: 61,
|
||||
exit_ip: '185.12.34.56',
|
||||
exit_country: 'NL',
|
||||
ok: true,
|
||||
error: '',
|
||||
// The chain, and the pairing that makes the whole feature worth building. A
|
||||
// chain is one series path, so with hop 3 dead the end-to-end probe is never
|
||||
// even attempted — the daemon stops walking there. This row and the hop rail
|
||||
// therefore have to tell one story, not two: both name hop 3, and neither
|
||||
// offers hop 4 as a second suspect. Note the row does NOT say "the probe
|
||||
// failed" — no probe of this chain's exit ran at all — which is why the daemon
|
||||
// has a separate message for it.
|
||||
'ewan-wg-subs': {
|
||||
selected: '',
|
||||
delay_ms: 0,
|
||||
exit_ip: '',
|
||||
exit_country: '',
|
||||
ok: false,
|
||||
error:
|
||||
'hop 3 of this chain was probed and did not answer, so nothing reaches the exit through it — fix that hop first',
|
||||
},
|
||||
// Dead through its tunnel, exactly as its membership health says — the exit
|
||||
// test and the member health tell the same story about the same group.
|
||||
// The one real health failure in the fixture: the observatory's probe ran along
|
||||
// this path and did not come back.
|
||||
'via-tunnel': {
|
||||
selected: '',
|
||||
delay_ms: 0,
|
||||
exit_ip: '',
|
||||
exit_country: '',
|
||||
ok: false,
|
||||
error: 'no member answered through egress awg (6 of 6 timed out)',
|
||||
error: 'the observatory’s probe through this path failed',
|
||||
},
|
||||
// Not a health verdict — nothing routes here, so no measurement of it exists.
|
||||
fallback: {
|
||||
selected: '',
|
||||
delay_ms: 0,
|
||||
exit_ip: '',
|
||||
exit_country: '',
|
||||
ok: false,
|
||||
error: 'no reachable node in the group (all 3 members timed out)',
|
||||
error:
|
||||
'not routed by any enabled rule, so nothing measures it — the observatory only probes paths the rules use',
|
||||
},
|
||||
relay: {
|
||||
selected: '',
|
||||
delay_ms: 0,
|
||||
exit_ip: '',
|
||||
exit_country: '',
|
||||
ok: false,
|
||||
error:
|
||||
'not routed by any enabled rule, so nothing measures it — the observatory only probes paths the rules use',
|
||||
},
|
||||
// Routed, materialised, simply not reached yet. Untested is not dead.
|
||||
'sub-fresh': {
|
||||
selected: '',
|
||||
delay_ms: 0,
|
||||
exit_ip: '',
|
||||
exit_country: '',
|
||||
ok: false,
|
||||
error:
|
||||
'the observatory has not reached this target yet — it refreshes on the global probe interval',
|
||||
},
|
||||
}
|
||||
|
||||
|
||||
@@ -0,0 +1,443 @@
|
||||
/* Alerts section (rendered on Settings) — inherits the Faceplate tokens and the
|
||||
* shared page chrome from App.css (.toast, .mono). Every rule below is a
|
||||
* one-to-one copy of the DNS.css rule the markup used before the section moved
|
||||
* here, renamed `dns-*` → `alr-*` so nothing collides. Orange stays an accent. */
|
||||
|
||||
/* ---- section shell (matches the Settings group plates one-to-one) ---- */
|
||||
.alr-section {
|
||||
margin-top: calc(var(--u, 8px) * 3.5);
|
||||
}
|
||||
.alr-sec-hd {
|
||||
display: flex;
|
||||
align-items: baseline;
|
||||
gap: 12px;
|
||||
padding-bottom: 10px;
|
||||
border-bottom: 1px solid var(--groove);
|
||||
}
|
||||
.alr-sec-title {
|
||||
margin: 0;
|
||||
font-family: var(--font-mono);
|
||||
font-size: 13px;
|
||||
font-weight: 700;
|
||||
letter-spacing: var(--track-label, 0.18em);
|
||||
text-transform: uppercase;
|
||||
color: var(--dim);
|
||||
}
|
||||
.alr-sec-count {
|
||||
font-size: 11px;
|
||||
letter-spacing: 0.06em;
|
||||
color: var(--faint);
|
||||
}
|
||||
.alr-sec-note {
|
||||
margin: 10px 2px 0;
|
||||
font-family: var(--font-sans);
|
||||
font-size: 12.5px;
|
||||
line-height: 1.55;
|
||||
color: var(--dim);
|
||||
max-width: 56ch;
|
||||
}
|
||||
|
||||
/* ---- add form ---- */
|
||||
.alr-add {
|
||||
display: flex;
|
||||
flex-direction: column;
|
||||
gap: 10px;
|
||||
margin-top: calc(var(--u, 8px) * 2);
|
||||
}
|
||||
.alr-add-top {
|
||||
display: flex;
|
||||
flex-wrap: wrap;
|
||||
align-items: center;
|
||||
gap: 10px;
|
||||
}
|
||||
.alr-input {
|
||||
min-width: 0;
|
||||
padding: 9px 12px;
|
||||
border: 1px solid var(--groove);
|
||||
border-radius: 7px;
|
||||
background: var(--sink);
|
||||
color: var(--ink);
|
||||
font-family: var(--font-mono);
|
||||
font-size: 12.5px;
|
||||
letter-spacing: 0.02em;
|
||||
box-shadow: 0 1px 2px var(--shadow) inset;
|
||||
transition: border-color 0.15s, box-shadow 0.15s;
|
||||
}
|
||||
.alr-input::placeholder {
|
||||
color: var(--faint);
|
||||
}
|
||||
.alr-input:focus-visible {
|
||||
border-color: var(--accent);
|
||||
outline: 2px solid var(--accent);
|
||||
outline-offset: 1px;
|
||||
}
|
||||
.alr-input:disabled {
|
||||
opacity: 0.55;
|
||||
}
|
||||
.alr-input--name {
|
||||
flex: 0 1 14rem;
|
||||
}
|
||||
|
||||
/* segmented type picker */
|
||||
.alr-seg {
|
||||
display: inline-flex;
|
||||
border: 1px solid var(--groove);
|
||||
border-radius: 7px;
|
||||
overflow: hidden;
|
||||
background: var(--sink);
|
||||
}
|
||||
.alr-seg-btn {
|
||||
padding: 8px 14px;
|
||||
border: 0;
|
||||
background: transparent;
|
||||
color: var(--dim);
|
||||
font-family: var(--font-mono);
|
||||
font-size: 11px;
|
||||
letter-spacing: 0.08em;
|
||||
text-transform: uppercase;
|
||||
cursor: pointer;
|
||||
transition: background 0.15s, color 0.15s;
|
||||
}
|
||||
.alr-seg-btn + .alr-seg-btn {
|
||||
border-left: 1px solid var(--groove);
|
||||
}
|
||||
.alr-seg-btn.on {
|
||||
background: var(--accent);
|
||||
color: #fff;
|
||||
}
|
||||
.alr-seg-btn:focus-visible {
|
||||
outline: 2px solid var(--accent);
|
||||
outline-offset: -2px;
|
||||
}
|
||||
|
||||
.alr-resp {
|
||||
display: inline-flex;
|
||||
align-items: center;
|
||||
gap: 8px;
|
||||
}
|
||||
.alr-resp-label {
|
||||
font-size: 10px;
|
||||
letter-spacing: var(--track-label, 0.18em);
|
||||
text-transform: uppercase;
|
||||
color: var(--faint);
|
||||
}
|
||||
.alr-select {
|
||||
padding: 8px 10px;
|
||||
border: 1px solid var(--groove);
|
||||
border-radius: 7px;
|
||||
background: var(--sink);
|
||||
color: var(--ink);
|
||||
font-family: var(--font-mono);
|
||||
font-size: 11.5px;
|
||||
letter-spacing: 0.04em;
|
||||
cursor: pointer;
|
||||
}
|
||||
.alr-select:focus-visible {
|
||||
border-color: var(--accent);
|
||||
outline: 2px solid var(--accent);
|
||||
outline-offset: 1px;
|
||||
}
|
||||
|
||||
.alr-add-actions {
|
||||
display: flex;
|
||||
align-items: center;
|
||||
justify-content: flex-end;
|
||||
gap: 14px;
|
||||
flex-wrap: wrap;
|
||||
}
|
||||
.alr-field-err {
|
||||
flex: 1;
|
||||
min-width: 0;
|
||||
margin: 0;
|
||||
font-family: var(--font-mono);
|
||||
font-size: 11.5px;
|
||||
line-height: 1.5;
|
||||
color: var(--crit);
|
||||
}
|
||||
|
||||
/* ---- rows ---- */
|
||||
.alr-rows {
|
||||
list-style: none;
|
||||
margin: calc(var(--u, 8px) * 2) 0 0;
|
||||
padding: 0;
|
||||
display: flex;
|
||||
flex-direction: column;
|
||||
gap: 8px;
|
||||
}
|
||||
.alr-row {
|
||||
display: flex;
|
||||
align-items: center;
|
||||
gap: calc(var(--u, 8px) * 1.5);
|
||||
padding: 12px 14px;
|
||||
border: 1px solid var(--groove);
|
||||
border-radius: 8px;
|
||||
background: linear-gradient(
|
||||
180deg,
|
||||
var(--raised),
|
||||
color-mix(in srgb, var(--raised) 82%, var(--panel))
|
||||
);
|
||||
box-shadow: 0 1px 0 var(--edge) inset;
|
||||
}
|
||||
.alr-row-main {
|
||||
flex: 1;
|
||||
min-width: 0;
|
||||
display: flex;
|
||||
flex-direction: column;
|
||||
gap: 4px;
|
||||
}
|
||||
.alr-row-l1 {
|
||||
display: flex;
|
||||
align-items: center;
|
||||
gap: 8px;
|
||||
flex-wrap: wrap;
|
||||
}
|
||||
.alr-row-name {
|
||||
font-family: var(--font-mono);
|
||||
font-size: 13px;
|
||||
font-weight: 600;
|
||||
letter-spacing: 0.01em;
|
||||
color: var(--ink);
|
||||
overflow: hidden;
|
||||
text-overflow: ellipsis;
|
||||
white-space: nowrap;
|
||||
max-width: 24ch;
|
||||
}
|
||||
.alr-row-l2 {
|
||||
display: flex;
|
||||
align-items: center;
|
||||
gap: 10px;
|
||||
flex-wrap: wrap;
|
||||
font-size: 11.5px;
|
||||
letter-spacing: 0.02em;
|
||||
}
|
||||
.alr-row-detail {
|
||||
color: var(--dim);
|
||||
overflow: hidden;
|
||||
text-overflow: ellipsis;
|
||||
white-space: nowrap;
|
||||
max-width: 40ch;
|
||||
}
|
||||
|
||||
/* badge — groove-bordered, not orange (accent stays reserved) */
|
||||
.alr-badge {
|
||||
display: inline-block;
|
||||
padding: 2px 7px;
|
||||
border: 1px solid var(--groove);
|
||||
border-radius: 5px;
|
||||
background: color-mix(in srgb, var(--sink) 60%, transparent);
|
||||
font-family: var(--font-mono);
|
||||
font-size: 10px;
|
||||
font-weight: 600;
|
||||
letter-spacing: 0.1em;
|
||||
text-transform: uppercase;
|
||||
color: var(--dim);
|
||||
white-space: nowrap;
|
||||
}
|
||||
.alr-badge--accent {
|
||||
border-color: color-mix(in srgb, var(--accent) 55%, var(--groove));
|
||||
color: var(--accent);
|
||||
}
|
||||
|
||||
.alr-masked {
|
||||
font-family: var(--font-mono);
|
||||
font-size: 10px;
|
||||
letter-spacing: 0.08em;
|
||||
color: var(--faint);
|
||||
text-transform: uppercase;
|
||||
cursor: help;
|
||||
}
|
||||
|
||||
.alr-del {
|
||||
flex: none;
|
||||
padding: 6px 12px;
|
||||
font-size: 10.5px;
|
||||
}
|
||||
|
||||
/* ---- empty plate ---- */
|
||||
.alr-empty {
|
||||
margin-top: calc(var(--u, 8px) * 2);
|
||||
padding: calc(var(--u, 8px) * 3);
|
||||
border: 1px dashed var(--groove);
|
||||
border-radius: 9px;
|
||||
background: color-mix(in srgb, var(--raised) 55%, transparent);
|
||||
text-align: center;
|
||||
}
|
||||
.alr-empty-title {
|
||||
display: block;
|
||||
font-size: 13px;
|
||||
font-weight: 700;
|
||||
letter-spacing: 0.06em;
|
||||
color: var(--dim);
|
||||
}
|
||||
.alr-empty-body {
|
||||
margin: 8px auto 0;
|
||||
max-width: 48ch;
|
||||
font-family: var(--font-sans);
|
||||
font-size: 13px;
|
||||
line-height: 1.55;
|
||||
color: var(--dim);
|
||||
}
|
||||
|
||||
/* ---- loading skeleton ---- */
|
||||
.alr-skel {
|
||||
height: 62px;
|
||||
border: 1px solid var(--groove);
|
||||
border-radius: 8px;
|
||||
background: linear-gradient(90deg, var(--raised), var(--sink), var(--raised));
|
||||
background-size: 200% 100%;
|
||||
animation: alr-skel-shift 1.4s ease-in-out infinite;
|
||||
}
|
||||
@keyframes alr-skel-shift {
|
||||
from {
|
||||
background-position: 200% 0;
|
||||
}
|
||||
to {
|
||||
background-position: -200% 0;
|
||||
}
|
||||
}
|
||||
|
||||
/* the per-row delivery picker sits inline in the row */
|
||||
.alr-detour {
|
||||
flex: none;
|
||||
display: flex;
|
||||
flex-direction: column;
|
||||
gap: 5px;
|
||||
min-width: 0;
|
||||
}
|
||||
.alr-detour-label {
|
||||
font-size: 10px;
|
||||
letter-spacing: var(--track-label, 0.18em);
|
||||
text-transform: uppercase;
|
||||
color: var(--faint);
|
||||
}
|
||||
.alr-detour-select {
|
||||
max-width: 22rem;
|
||||
}
|
||||
|
||||
/* current delivery-path readout on the row */
|
||||
.alr-path {
|
||||
color: var(--faint);
|
||||
white-space: nowrap;
|
||||
}
|
||||
.alr-path[data-active='on'] {
|
||||
color: var(--dim);
|
||||
}
|
||||
.alr-path-name {
|
||||
color: var(--led-on);
|
||||
font-weight: 600;
|
||||
}
|
||||
.alr-path[data-missing='y'] .alr-path-name {
|
||||
color: var(--amber);
|
||||
}
|
||||
.alr-path-flag {
|
||||
color: var(--amber);
|
||||
}
|
||||
|
||||
/* alert delivery: deliver-via picker + fallback toggle + caution note */
|
||||
.alr-delivery {
|
||||
display: flex;
|
||||
flex-wrap: wrap;
|
||||
align-items: center;
|
||||
gap: 10px 20px;
|
||||
}
|
||||
.alr-fallback {
|
||||
display: inline-flex;
|
||||
align-items: center;
|
||||
gap: 8px;
|
||||
cursor: pointer;
|
||||
}
|
||||
.alr-fallback-label {
|
||||
font-size: 10px;
|
||||
letter-spacing: var(--track-label, 0.18em);
|
||||
text-transform: uppercase;
|
||||
color: var(--faint);
|
||||
}
|
||||
/* the per-row delivery controls sit inline in the row (like .alr-detour) */
|
||||
.alr-ctl {
|
||||
flex: none;
|
||||
display: flex;
|
||||
flex-direction: column;
|
||||
gap: 8px;
|
||||
min-width: 0;
|
||||
}
|
||||
.alr-note {
|
||||
margin: 0;
|
||||
font-family: var(--font-sans);
|
||||
font-size: 11.5px;
|
||||
line-height: 1.5;
|
||||
color: var(--amber);
|
||||
max-width: 56ch;
|
||||
}
|
||||
.alr-note--row {
|
||||
margin-top: 2px;
|
||||
}
|
||||
|
||||
/* alert event checkboxes */
|
||||
.alr-events {
|
||||
display: flex;
|
||||
flex-wrap: wrap;
|
||||
gap: 8px 16px;
|
||||
margin: 0;
|
||||
padding: 0;
|
||||
border: 0;
|
||||
}
|
||||
.alr-event {
|
||||
display: inline-flex;
|
||||
align-items: center;
|
||||
gap: 6px;
|
||||
font-size: 13px;
|
||||
color: var(--fp-text, inherit);
|
||||
cursor: pointer;
|
||||
}
|
||||
.alr-event input {
|
||||
accent-color: var(--fp-accent, currentColor);
|
||||
}
|
||||
|
||||
/* ---- responsive ---- */
|
||||
@media (max-width: 640px) {
|
||||
.alr-row {
|
||||
flex-wrap: wrap;
|
||||
}
|
||||
.alr-row-main {
|
||||
flex-basis: calc(100% - 90px);
|
||||
}
|
||||
.alr-del {
|
||||
margin-left: auto;
|
||||
}
|
||||
.alr-input--name {
|
||||
flex-basis: 100%;
|
||||
}
|
||||
.alr-detour {
|
||||
flex-basis: 100%;
|
||||
order: 3;
|
||||
flex-wrap: wrap;
|
||||
}
|
||||
/* A <select> won't shrink below its widest option unless it's allowed to:
|
||||
without min-width:0 the long detour labels push the page into a horizontal
|
||||
scroll at 390px. Let them fill the row and clip instead. */
|
||||
.alr-detour-select,
|
||||
.alr-resp .alr-select {
|
||||
max-width: 100%;
|
||||
width: 100%;
|
||||
min-width: 0;
|
||||
}
|
||||
.alr-resp {
|
||||
display: flex;
|
||||
flex-wrap: wrap;
|
||||
max-width: 100%;
|
||||
}
|
||||
.alr-ctl {
|
||||
flex-basis: 100%;
|
||||
order: 3;
|
||||
}
|
||||
}
|
||||
|
||||
@media (prefers-reduced-motion: reduce) {
|
||||
.alr-skel {
|
||||
animation: none;
|
||||
}
|
||||
.alr-input,
|
||||
.alr-seg-btn {
|
||||
transition: none;
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,690 @@
|
||||
import './Alerts.css'
|
||||
import { useCallback, useMemo, useState } from 'react'
|
||||
import { Button, Toggle, useConfirm } from '../components'
|
||||
import type { Alert, Model } from '../api'
|
||||
|
||||
// The Alerts section — out-of-band notifications (Telegram bot / webhook) for
|
||||
// kill-switch trips, apply failures, new devices and subscription expiry. It
|
||||
// lived at the bottom of the DNS page, which is the last place an operator
|
||||
// looking for "tell me when the tunnel dies" would think to look; it now renders
|
||||
// as a group on Settings. The component owns no I/O: every mutation goes through
|
||||
// the `onSave` prop so Settings keeps a single dirty banner and a single toast.
|
||||
//
|
||||
// NOTE on duplication: the detour helpers below (DetourCatalog, canonDetour,
|
||||
// detourValues, describeDetour, DetourSelect) plus asArray / uniqueName /
|
||||
// maskUrl / EmptyPlate are deliberate copies of the ones in DNS.tsx. DNS keeps
|
||||
// its own for resolvers and DNS rules; extracting a shared module would couple
|
||||
// two pages that otherwise share nothing, and that refactor is out of scope
|
||||
// here. If a third consumer ever appears, promote them then.
|
||||
|
||||
// ---- local Model extension --------------------------------------------------
|
||||
|
||||
/** The Model with the Alerts slice surfaced (index-signature passthrough). */
|
||||
type AlertsModel = Model & { Alerts?: Alert[] | null }
|
||||
|
||||
/**
|
||||
* Every event the daemon actually sends. A retired health-probe event was left
|
||||
* out on purpose: nothing ever fired it, so a channel that subscribed to it would
|
||||
* just stay quiet forever — the one failure mode an alert must not have. Only
|
||||
* events with a live firing path are offered here.
|
||||
*/
|
||||
const ALERT_EVENTS: ReadonlyArray<{ id: string; label: string }> = [
|
||||
{ id: 'killswitch', label: 'Kill-switch' },
|
||||
{ id: 'apply_fail', label: 'Apply failure' },
|
||||
{ id: 'new_device', label: 'New device' },
|
||||
{ id: 'sub_expiry', label: 'Subscription expiring' },
|
||||
]
|
||||
|
||||
// Shown when an alert routes through a detour with no direct fallback — the exact
|
||||
// case where a tunnel-down alert could fail to send. The user asked for this.
|
||||
const VIA_NO_FALLBACK_NOTE =
|
||||
'A kill-switch/tunnel-down alert may not send if it routes through the affected tunnel — enable fallback.'
|
||||
|
||||
// ---- helpers (copies of DNS.tsx — see the header note) ----------------------
|
||||
|
||||
const asArray = <T,>(a: T[] | null | undefined): T[] => (a ? a : [])
|
||||
|
||||
const HTTP_RE = /^https?:\/\//i
|
||||
|
||||
/** A remote URL often carries a token in its query/path — show host only. */
|
||||
function maskUrl(url: string): { host: string; masked: boolean } {
|
||||
try {
|
||||
const u = new URL(url)
|
||||
return { host: u.host, masked: u.search !== '' || u.pathname.replace(/\/+$/, '') !== '' }
|
||||
} catch {
|
||||
return { host: url || '—', masked: false }
|
||||
}
|
||||
}
|
||||
|
||||
function uniqueName(base: string, taken: Set<string>): string {
|
||||
const seed = base.trim() || 'alert'
|
||||
if (!taken.has(seed)) return seed
|
||||
let i = 2
|
||||
while (taken.has(`${seed}-${i}`)) i++
|
||||
return `${seed}-${i}`
|
||||
}
|
||||
|
||||
/** The live targets an alert's delivery can be pinned to (the picker). */
|
||||
interface DetourCatalog {
|
||||
groups: string[]
|
||||
chains: string[]
|
||||
egresses: { name: string; type: string }[]
|
||||
nodes: string[]
|
||||
}
|
||||
|
||||
/**
|
||||
* Normalise a stored `Via` to a picker option value. Empty/`direct` ⇒
|
||||
* `direct`; already-prefixed values (`group:`/`chain:`/`egress:`/`node:`) pass
|
||||
* through; a bare legacy name is resolved against the catalog so a still-valid
|
||||
* setup isn't mislabelled; anything unresolved is kept verbatim (shown stale).
|
||||
*/
|
||||
function canonDetour(raw: string | undefined, cat: DetourCatalog): string {
|
||||
const d = (raw ?? '').trim()
|
||||
if (!d || d.toLowerCase() === 'direct') return 'direct'
|
||||
if (/^(node|group|chain|egress):/i.test(d)) return d
|
||||
if (cat.egresses.some((e) => e.name === d)) return `egress:${d}`
|
||||
if (cat.groups.includes(d)) return `group:${d}`
|
||||
if (cat.chains.includes(d)) return `chain:${d}`
|
||||
if (cat.nodes.includes(d)) return `node:${d}`
|
||||
return d
|
||||
}
|
||||
|
||||
/** Every valid option value for a catalog, including `direct`. */
|
||||
function detourValues(cat: DetourCatalog): Set<string> {
|
||||
const s = new Set<string>(['direct'])
|
||||
for (const g of cat.groups) s.add(`group:${g}`)
|
||||
for (const c of cat.chains) s.add(`chain:${c}`)
|
||||
for (const e of cat.egresses) s.add(`egress:${e.name}`)
|
||||
for (const n of cat.nodes) s.add(`node:${n}`)
|
||||
return s
|
||||
}
|
||||
|
||||
/** Describe a canonical detour value for the row readout. */
|
||||
function describeDetour(
|
||||
canon: string,
|
||||
cat: DetourCatalog,
|
||||
valid: Set<string>,
|
||||
): { direct: boolean; prefix: string; name: string; missing: boolean } {
|
||||
if (canon === 'direct') return { direct: true, prefix: '', name: '', missing: false }
|
||||
const i = canon.indexOf(':')
|
||||
const kind = i === -1 ? '' : canon.slice(0, i)
|
||||
const name = i === -1 ? canon : canon.slice(i + 1)
|
||||
const missing = !valid.has(canon)
|
||||
let prefix = 'via'
|
||||
if (kind === 'group') prefix = 'via group'
|
||||
else if (kind === 'chain') prefix = 'via chain'
|
||||
else if (kind === 'node') prefix = 'via node'
|
||||
else if (kind === 'egress') {
|
||||
const eg = cat.egresses.find((e) => e.name === name)
|
||||
prefix = eg?.type === 'interface' ? 'via interface' : 'via egress'
|
||||
}
|
||||
return { direct: false, prefix, name, missing }
|
||||
}
|
||||
|
||||
// ---- section ----------------------------------------------------------------
|
||||
|
||||
export function AlertsSection({
|
||||
config,
|
||||
busy,
|
||||
loading,
|
||||
onSave,
|
||||
}: {
|
||||
/** Full desired-state model; null until it has loaded. */
|
||||
config: Model | null
|
||||
/** A save/apply is in flight — controls lock. */
|
||||
busy: boolean
|
||||
/** The config is still loading — show a skeleton row. */
|
||||
loading: boolean
|
||||
/** Persist the whole next model; resolves true on success (Settings' `save`). */
|
||||
onSave: (next: Model, okMsg: string) => Promise<boolean>
|
||||
}): JSX.Element {
|
||||
const confirm = useConfirm()
|
||||
const model = config as AlertsModel | null
|
||||
const alerts = useMemo<Alert[]>(() => asArray(model?.Alerts), [model])
|
||||
|
||||
// Alerts route through Direct/group/node/egress only (no chains) — the contract
|
||||
// vocabulary for Alert.Via. Built straight from the Model with chains dropped.
|
||||
const alertCatalog = useMemo<DetourCatalog>(
|
||||
() => ({
|
||||
groups: asArray(config?.Groups).map((g) => g.Name),
|
||||
chains: [],
|
||||
egresses: asArray(config?.Egresses).map((e) => ({ name: e.Name, type: e.Type })),
|
||||
nodes: asArray(config?.Nodes).map((n) => n.Name),
|
||||
}),
|
||||
[config],
|
||||
)
|
||||
const alertValid = useMemo(() => detourValues(alertCatalog), [alertCatalog])
|
||||
|
||||
const alertNames = useMemo(() => new Set(alerts.map((a) => a.Name)), [alerts])
|
||||
const alertsOn = alerts.filter((a) => a.Enabled).length
|
||||
|
||||
// ---- mutations — all writes go through onSave -----------------------------
|
||||
const addAlert = useCallback(
|
||||
(draft: Alert): Promise<boolean> => {
|
||||
if (!model) return Promise.resolve(false)
|
||||
const taken = new Set(alerts.map((a) => a.Name))
|
||||
const a: Alert = { ...draft, Name: uniqueName(draft.Name, taken) }
|
||||
return onSave({ ...model, Alerts: [...alerts, a] }, `Added ${a.Name}`)
|
||||
},
|
||||
[model, alerts, onSave],
|
||||
)
|
||||
|
||||
const toggleAlert = useCallback(
|
||||
(idx: number, on: boolean) => {
|
||||
if (!model) return
|
||||
const next = alerts.map((a, i) => (i === idx ? { ...a, Enabled: on } : a))
|
||||
void onSave({ ...model, Alerts: next }, `${next[idx].Name} ${on ? 'enabled' : 'disabled'}`)
|
||||
},
|
||||
[model, alerts, onSave],
|
||||
)
|
||||
|
||||
const removeAlert = useCallback(
|
||||
async (idx: number) => {
|
||||
if (!model) return
|
||||
const target = alerts[idx]
|
||||
const ok = await confirm({
|
||||
label: 'Delete alert',
|
||||
title: `Delete alert “${target.Name}”?`,
|
||||
body: 'This removes it from the config.',
|
||||
})
|
||||
if (!ok) return
|
||||
const next = alerts.filter((_, i) => i !== idx)
|
||||
void onSave({ ...model, Alerts: next }, `Deleted ${target.Name}`)
|
||||
},
|
||||
[model, alerts, onSave, confirm],
|
||||
)
|
||||
|
||||
const setAlertVia = useCallback(
|
||||
(idx: number, v: string) => {
|
||||
if (!model) return
|
||||
const via = v === 'direct' ? '' : v
|
||||
const next = alerts.map((a, i) => (i === idx ? { ...a, Via: via || undefined } : a))
|
||||
void onSave(
|
||||
{ ...model, Alerts: next },
|
||||
via ? `${next[idx].Name} delivers via ${via}` : `${next[idx].Name} delivers direct`,
|
||||
)
|
||||
},
|
||||
[model, alerts, onSave],
|
||||
)
|
||||
|
||||
const setAlertFallback = useCallback(
|
||||
(idx: number, on: boolean) => {
|
||||
if (!model) return
|
||||
const next = alerts.map((a, i) => (i === idx ? { ...a, Fallback: on || undefined } : a))
|
||||
void onSave(
|
||||
{ ...model, Alerts: next },
|
||||
`${next[idx].Name} direct fallback ${on ? 'on' : 'off'}`,
|
||||
)
|
||||
},
|
||||
[model, alerts, onSave],
|
||||
)
|
||||
|
||||
return (
|
||||
<div className="alr-section" aria-label="Alerts">
|
||||
<header className="alr-sec-hd">
|
||||
<h2 className="alr-sec-title">Alerts</h2>
|
||||
<span className="alr-sec-count mono">
|
||||
{alertsOn} / {alerts.length} on
|
||||
</span>
|
||||
</header>
|
||||
<p className="alr-sec-note">
|
||||
Out-of-band notifications. Delivered <strong>direct to the internet</strong> by default — so a
|
||||
kill-switch or engine-down alert still reaches you when the proxy is down. You can route one
|
||||
through a group, node or egress instead, with a direct fallback if that detour fails.
|
||||
</p>
|
||||
|
||||
<AddAlertForm
|
||||
busy={busy}
|
||||
disabled={!config}
|
||||
taken={alertNames}
|
||||
catalog={alertCatalog}
|
||||
valid={alertValid}
|
||||
onAdd={addAlert}
|
||||
/>
|
||||
|
||||
{loading ? (
|
||||
<ul className="alr-rows" aria-hidden="true">
|
||||
<li className="alr-skel" />
|
||||
</ul>
|
||||
) : alerts.length === 0 ? (
|
||||
<EmptyPlate
|
||||
title="No alerts"
|
||||
body="Add a Telegram bot or a webhook above to get notified when the kill-switch trips, a new device joins, or an apply fails."
|
||||
/>
|
||||
) : (
|
||||
<ul className="alr-rows">
|
||||
{alerts.map((a, i) => (
|
||||
<AlertRow
|
||||
key={`${a.Name}-${i}`}
|
||||
alert={a}
|
||||
busy={busy}
|
||||
catalog={alertCatalog}
|
||||
valid={alertValid}
|
||||
onToggle={(on) => toggleAlert(i, on)}
|
||||
onVia={(v) => setAlertVia(i, v)}
|
||||
onFallback={(on) => setAlertFallback(i, on)}
|
||||
onDelete={() => removeAlert(i)}
|
||||
/>
|
||||
))}
|
||||
</ul>
|
||||
)}
|
||||
</div>
|
||||
)
|
||||
}
|
||||
|
||||
// ---- alert add form + row ----------------------------------------------------
|
||||
|
||||
function AddAlertForm({
|
||||
busy,
|
||||
disabled,
|
||||
taken,
|
||||
catalog,
|
||||
valid,
|
||||
onAdd,
|
||||
}: {
|
||||
busy: boolean
|
||||
disabled: boolean
|
||||
taken: Set<string>
|
||||
catalog: DetourCatalog
|
||||
valid: Set<string>
|
||||
onAdd: (a: Alert) => Promise<boolean>
|
||||
}) {
|
||||
const [name, setName] = useState('')
|
||||
const [type, setType] = useState<'telegram' | 'webhook'>('telegram')
|
||||
const [token, setToken] = useState('')
|
||||
const [chatId, setChatId] = useState('')
|
||||
const [url, setUrl] = useState('')
|
||||
const [events, setEvents] = useState<string[]>(['killswitch'])
|
||||
const [via, setVia] = useState('direct')
|
||||
const [fallback, setFallback] = useState(false)
|
||||
const [err, setErr] = useState<string | null>(null)
|
||||
|
||||
const reset = () => {
|
||||
setName('')
|
||||
setType('telegram')
|
||||
setToken('')
|
||||
setChatId('')
|
||||
setUrl('')
|
||||
setEvents(['killswitch'])
|
||||
setVia('direct')
|
||||
setFallback(false)
|
||||
}
|
||||
|
||||
const toggleEvent = (id: string) =>
|
||||
setEvents((prev) => (prev.includes(id) ? prev.filter((e) => e !== id) : [...prev, id]))
|
||||
|
||||
const submit = async () => {
|
||||
const nm = name.trim()
|
||||
if (!nm) {
|
||||
setErr('Give the alert a name.')
|
||||
return
|
||||
}
|
||||
if (taken.has(nm)) {
|
||||
setErr(`An alert named “${nm}” already exists.`)
|
||||
return
|
||||
}
|
||||
if (type === 'telegram') {
|
||||
if (!token.trim() || !chatId.trim()) {
|
||||
setErr('Telegram needs a bot token and a chat ID.')
|
||||
return
|
||||
}
|
||||
} else if (!HTTP_RE.test(url.trim())) {
|
||||
setErr('Enter an http(s):// webhook URL.')
|
||||
return
|
||||
}
|
||||
if (events.length === 0) {
|
||||
setErr('Pick at least one event to notify on.')
|
||||
return
|
||||
}
|
||||
setErr(null)
|
||||
const routed = via !== 'direct'
|
||||
const routing = { Via: routed ? via : undefined, Fallback: routed && fallback ? true : undefined }
|
||||
const draft: Alert =
|
||||
type === 'telegram'
|
||||
? { Name: nm, Enabled: true, Type: 'telegram', Token: token.trim(), ChatID: chatId.trim(), Events: events, ...routing }
|
||||
: { Name: nm, Enabled: true, Type: 'webhook', URL: url.trim(), Events: events, ...routing }
|
||||
const ok = await onAdd(draft)
|
||||
if (ok) reset()
|
||||
}
|
||||
|
||||
const routed = via !== 'direct'
|
||||
|
||||
return (
|
||||
<form
|
||||
className="alr-add"
|
||||
onSubmit={(e) => {
|
||||
e.preventDefault()
|
||||
void submit()
|
||||
}}
|
||||
>
|
||||
<div className="alr-add-top">
|
||||
<input
|
||||
className="alr-input alr-input--name"
|
||||
type="text"
|
||||
spellCheck={false}
|
||||
autoComplete="off"
|
||||
placeholder="Alert name"
|
||||
aria-label="Alert name"
|
||||
value={name}
|
||||
onChange={(e) => {
|
||||
setName(e.target.value)
|
||||
if (err) setErr(null)
|
||||
}}
|
||||
disabled={busy || disabled}
|
||||
/>
|
||||
<div className="alr-seg" role="group" aria-label="Alert type">
|
||||
<button
|
||||
type="button"
|
||||
className={type === 'telegram' ? 'alr-seg-btn on' : 'alr-seg-btn'}
|
||||
aria-pressed={type === 'telegram'}
|
||||
onClick={() => setType('telegram')}
|
||||
disabled={busy || disabled}
|
||||
>
|
||||
Telegram
|
||||
</button>
|
||||
<button
|
||||
type="button"
|
||||
className={type === 'webhook' ? 'alr-seg-btn on' : 'alr-seg-btn'}
|
||||
aria-pressed={type === 'webhook'}
|
||||
onClick={() => setType('webhook')}
|
||||
disabled={busy || disabled}
|
||||
>
|
||||
Webhook
|
||||
</button>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
{type === 'telegram' ? (
|
||||
<>
|
||||
<input
|
||||
className="alr-input"
|
||||
type="password"
|
||||
spellCheck={false}
|
||||
autoComplete="off"
|
||||
placeholder="Bot token (kept secret)"
|
||||
aria-label="Telegram bot token"
|
||||
value={token}
|
||||
onChange={(e) => {
|
||||
setToken(e.target.value)
|
||||
if (err) setErr(null)
|
||||
}}
|
||||
disabled={busy || disabled}
|
||||
/>
|
||||
<input
|
||||
className="alr-input"
|
||||
type="text"
|
||||
spellCheck={false}
|
||||
autoComplete="off"
|
||||
placeholder="Chat ID (e.g. -1001234567890)"
|
||||
aria-label="Telegram chat ID"
|
||||
value={chatId}
|
||||
onChange={(e) => {
|
||||
setChatId(e.target.value)
|
||||
if (err) setErr(null)
|
||||
}}
|
||||
disabled={busy || disabled}
|
||||
/>
|
||||
</>
|
||||
) : (
|
||||
<input
|
||||
className="alr-input"
|
||||
type="text"
|
||||
inputMode="url"
|
||||
spellCheck={false}
|
||||
autoComplete="off"
|
||||
placeholder="https://hooks.example.com/…"
|
||||
aria-label="Webhook URL"
|
||||
value={url}
|
||||
onChange={(e) => {
|
||||
setUrl(e.target.value)
|
||||
if (err) setErr(null)
|
||||
}}
|
||||
disabled={busy || disabled}
|
||||
/>
|
||||
)}
|
||||
|
||||
<fieldset className="alr-events" aria-label="Events to notify on">
|
||||
{ALERT_EVENTS.map((ev) => (
|
||||
<label key={ev.id} className="alr-event">
|
||||
<input
|
||||
type="checkbox"
|
||||
checked={events.includes(ev.id)}
|
||||
onChange={() => toggleEvent(ev.id)}
|
||||
disabled={busy || disabled}
|
||||
/>
|
||||
<span>{ev.label}</span>
|
||||
</label>
|
||||
))}
|
||||
</fieldset>
|
||||
|
||||
<div className="alr-delivery">
|
||||
<label className="alr-resp">
|
||||
<span className="alr-resp-label mono">Deliver via</span>
|
||||
<DetourSelect
|
||||
value={via}
|
||||
catalog={catalog}
|
||||
valid={valid}
|
||||
busy={busy}
|
||||
disabled={disabled}
|
||||
ariaLabel="Deliver alert via"
|
||||
onChange={setVia}
|
||||
directLabel="Direct (default)"
|
||||
/>
|
||||
</label>
|
||||
<label className="alr-fallback">
|
||||
<Toggle
|
||||
pressed={fallback}
|
||||
onChange={setFallback}
|
||||
label={fallback ? 'Disable direct fallback' : 'Enable direct fallback'}
|
||||
disabled={busy || disabled || !routed}
|
||||
/>
|
||||
<span className="alr-fallback-label mono">Fallback to direct</span>
|
||||
</label>
|
||||
</div>
|
||||
{routed && !fallback && (
|
||||
<p className="alr-note" role="note">
|
||||
{VIA_NO_FALLBACK_NOTE}
|
||||
</p>
|
||||
)}
|
||||
|
||||
<div className="alr-add-actions">
|
||||
{err && (
|
||||
<p className="alr-field-err" role="alert">
|
||||
{err}
|
||||
</p>
|
||||
)}
|
||||
<Button type="submit" variant="primary" disabled={busy || disabled}>
|
||||
{busy ? 'Saving…' : 'Add alert'}
|
||||
</Button>
|
||||
</div>
|
||||
</form>
|
||||
)
|
||||
}
|
||||
|
||||
function AlertRow({
|
||||
alert,
|
||||
busy,
|
||||
catalog,
|
||||
valid,
|
||||
onToggle,
|
||||
onVia,
|
||||
onFallback,
|
||||
onDelete,
|
||||
}: {
|
||||
alert: Alert
|
||||
busy: boolean
|
||||
catalog: DetourCatalog
|
||||
valid: Set<string>
|
||||
onToggle: (on: boolean) => void
|
||||
onVia: (v: string) => void
|
||||
onFallback: (on: boolean) => void
|
||||
onDelete: () => void
|
||||
}) {
|
||||
// Never render the token/URL in clear — show a masked descriptor only.
|
||||
const detail = useMemo(() => {
|
||||
if (alert.Type === 'telegram') {
|
||||
return { text: `chat ${alert.ChatID || '—'}`, masked: !!alert.Token }
|
||||
}
|
||||
const { host, masked } = maskUrl(alert.URL ?? '')
|
||||
return { text: host, masked: masked || !!alert.URL }
|
||||
}, [alert.Type, alert.ChatID, alert.Token, alert.URL])
|
||||
|
||||
const events = asArray(alert.Events)
|
||||
const canon = useMemo(() => canonDetour(alert.Via, catalog), [alert.Via, catalog])
|
||||
const route = useMemo(() => describeDetour(canon, catalog, valid), [canon, catalog, valid])
|
||||
const routed = canon !== 'direct'
|
||||
const fallback = alert.Fallback ?? false
|
||||
|
||||
return (
|
||||
<li className="alr-row">
|
||||
<Toggle
|
||||
pressed={alert.Enabled}
|
||||
onChange={onToggle}
|
||||
label={`${alert.Enabled ? 'Disable' : 'Enable'} alert ${alert.Name}`}
|
||||
disabled={busy}
|
||||
/>
|
||||
<div className="alr-row-main">
|
||||
<div className="alr-row-l1">
|
||||
<span className="alr-row-name">{alert.Name}</span>
|
||||
<span className="alr-badge">{alert.Type}</span>
|
||||
{events.map((e) => (
|
||||
<span key={e} className="alr-badge alr-badge--accent">
|
||||
{e}
|
||||
</span>
|
||||
))}
|
||||
</div>
|
||||
<div className="alr-row-l2 mono">
|
||||
<span className="alr-row-detail">{detail.text}</span>
|
||||
{detail.masked && (
|
||||
<span className="alr-masked" title="Secret is stored but hidden here">
|
||||
secret hidden
|
||||
</span>
|
||||
)}
|
||||
{route.direct ? (
|
||||
<span className="alr-path">direct</span>
|
||||
) : (
|
||||
<span className="alr-path" data-active="on" data-missing={route.missing ? 'y' : undefined}>
|
||||
{route.prefix} <strong className="alr-path-name">{route.name}</strong>
|
||||
{route.missing && <span className="alr-path-flag"> (missing)</span>}
|
||||
{fallback ? ' · +direct fallback' : ' · no fallback'}
|
||||
</span>
|
||||
)}
|
||||
</div>
|
||||
{routed && !fallback && <p className="alr-note alr-note--row">{VIA_NO_FALLBACK_NOTE}</p>}
|
||||
</div>
|
||||
<div className="alr-ctl">
|
||||
<label className="alr-detour">
|
||||
<span className="alr-detour-label mono">Deliver via</span>
|
||||
<DetourSelect
|
||||
value={canon}
|
||||
catalog={catalog}
|
||||
valid={valid}
|
||||
busy={busy}
|
||||
disabled={false}
|
||||
ariaLabel={`Deliver alert ${alert.Name} via`}
|
||||
onChange={onVia}
|
||||
directLabel="Direct (default)"
|
||||
/>
|
||||
</label>
|
||||
<label className="alr-fallback">
|
||||
<Toggle
|
||||
pressed={fallback}
|
||||
onChange={onFallback}
|
||||
label={`${fallback ? 'Disable' : 'Enable'} direct fallback for ${alert.Name}`}
|
||||
disabled={busy || !routed}
|
||||
/>
|
||||
<span className="alr-fallback-label mono">Fallback to direct</span>
|
||||
</label>
|
||||
</div>
|
||||
<Button
|
||||
className="alr-del"
|
||||
onClick={onDelete}
|
||||
disabled={busy}
|
||||
aria-label={`Delete alert ${alert.Name}`}
|
||||
>
|
||||
Delete
|
||||
</Button>
|
||||
</li>
|
||||
)
|
||||
}
|
||||
|
||||
/** The live delivery picker: option list built from the Model's targets. */
|
||||
function DetourSelect({
|
||||
value,
|
||||
catalog,
|
||||
valid,
|
||||
busy,
|
||||
disabled,
|
||||
ariaLabel,
|
||||
onChange,
|
||||
directLabel = 'Direct (no proxy)',
|
||||
}: {
|
||||
value: string // canonical value
|
||||
catalog: DetourCatalog
|
||||
valid: Set<string>
|
||||
busy: boolean
|
||||
disabled: boolean
|
||||
ariaLabel: string
|
||||
onChange: (v: string) => void
|
||||
directLabel?: string
|
||||
}) {
|
||||
const missing = value !== 'direct' && !valid.has(value)
|
||||
return (
|
||||
<select
|
||||
className="alr-select alr-detour-select"
|
||||
value={value}
|
||||
onChange={(e) => onChange(e.target.value)}
|
||||
disabled={busy || disabled}
|
||||
aria-label={ariaLabel}
|
||||
>
|
||||
<option value="direct">{directLabel}</option>
|
||||
{catalog.groups.length > 0 && (
|
||||
<optgroup label="Groups">
|
||||
{catalog.groups.map((g) => (
|
||||
<option key={g} value={`group:${g}`}>
|
||||
Group {g} (balancer)
|
||||
</option>
|
||||
))}
|
||||
</optgroup>
|
||||
)}
|
||||
{catalog.chains.length > 0 && (
|
||||
<optgroup label="Chains">
|
||||
{catalog.chains.map((c) => (
|
||||
<option key={c} value={`chain:${c}`}>
|
||||
Chain {c}
|
||||
</option>
|
||||
))}
|
||||
</optgroup>
|
||||
)}
|
||||
{catalog.egresses.length > 0 && (
|
||||
<optgroup label="Interfaces / egresses">
|
||||
{catalog.egresses.map((e) => (
|
||||
<option key={e.name} value={`egress:${e.name}`}>
|
||||
Interface/egress {e.name}
|
||||
{e.type ? ` (${e.type})` : ''}
|
||||
</option>
|
||||
))}
|
||||
</optgroup>
|
||||
)}
|
||||
{catalog.nodes.length > 0 && (
|
||||
<optgroup label="Nodes">
|
||||
{catalog.nodes.map((n) => (
|
||||
<option key={n} value={`node:${n}`}>
|
||||
Node {n}
|
||||
</option>
|
||||
))}
|
||||
</optgroup>
|
||||
)}
|
||||
{missing && <option value={value}>{value} (missing)</option>}
|
||||
</select>
|
||||
)
|
||||
}
|
||||
|
||||
function EmptyPlate({ title, body }: { title: string; body: string }) {
|
||||
return (
|
||||
<div className="alr-empty">
|
||||
<span className="alr-empty-title mono">{title}</span>
|
||||
<p className="alr-empty-body">{body}</p>
|
||||
</div>
|
||||
)
|
||||
}
|
||||
+108
-53
@@ -11,6 +11,8 @@ import {
|
||||
ApiError,
|
||||
} from '../api'
|
||||
import type { Globals, Status } from '../api'
|
||||
import { engineReadout, killSwitchReadout } from '../planeState'
|
||||
import { onPendingConfirmExpire, usePendingConfirm } from '../pendingConfirm'
|
||||
|
||||
// Short, readable config hash — drops the "sha256:" prefix like the footer does.
|
||||
function short(hash: string): string {
|
||||
@@ -25,13 +27,6 @@ function msg(e: unknown): string {
|
||||
|
||||
type Busy = 'apply' | 'confirm' | 'rollback' | null
|
||||
|
||||
/** A pending commit-confirm window: the daemon has armed an auto-rollback. */
|
||||
interface Armed {
|
||||
total: number // the ConfirmTimeout the window started with
|
||||
remaining: number // seconds left before the daemon reverts
|
||||
appliedHash: string // the hash that went live on apply (the "after" of apply)
|
||||
}
|
||||
|
||||
type ActionKind = 'apply' | 'confirm' | 'rollback' | 'expire'
|
||||
interface ActionResult {
|
||||
kind: ActionKind
|
||||
@@ -57,7 +52,11 @@ export default function Apply() {
|
||||
const [configError, setConfigError] = useState<string | null>(null)
|
||||
|
||||
const [busy, setBusy] = useState<Busy>(null)
|
||||
const [armed, setArmed] = useState<Armed | null>(null)
|
||||
// The armed window is app-wide state, not this page's: it is recorded by the
|
||||
// api layer on every apply and survives a reload. Keeping it local is what made
|
||||
// refreshing this tab lose both the countdown and the only button that could
|
||||
// stop it. See pendingConfirm.ts.
|
||||
const armed = usePendingConfirm()
|
||||
const [result, setResult] = useState<ActionResult | null>(null)
|
||||
const [confirmingRollback, setConfirmingRollback] = useState(false)
|
||||
|
||||
@@ -104,32 +103,64 @@ export default function Apply() {
|
||||
void loadConfig()
|
||||
}, [loadConfig])
|
||||
|
||||
// ---- commit-confirm countdown: a calm 1s numeric tick, effect-scoped so the
|
||||
// timer is always cleared on unmount / confirm / rollback (no leaked intervals) ----
|
||||
// The window running out does NOT mean the daemon rolled back.
|
||||
//
|
||||
// apply.ArmRollback captures the data-plane generation when it arms, and on
|
||||
// expiry it compares. If anything re-applied the plane in between — another
|
||||
// panel apply, SIGHUP, a hotplug or the once-a-minute cron reconcile, the WAN
|
||||
// profile auto-switch — it disarms and KEEPS the running config, logging "NOT
|
||||
// rolling back" and nothing else. That is the common case on a production
|
||||
// router, and this page used to print "daemon auto-rolled back to last-good
|
||||
// config" for it: a confident report of an event that did not happen, with a
|
||||
// hash pair underneath that quietly said "unchanged".
|
||||
//
|
||||
// The panel cannot see which branch ran — the daemon says so only in its log.
|
||||
// So it reports the one thing it CAN observe, the live config hash, and waits
|
||||
// for the revert to land before reading it (a rollback is a full re-apply and
|
||||
// does not complete the instant the timer fires).
|
||||
const liveHashRef = useRef('')
|
||||
liveHashRef.current = status?.hash ?? ''
|
||||
useEffect(() => {
|
||||
if (!armed) return
|
||||
if (armed.remaining <= 0) {
|
||||
// Window elapsed — the daemon reverts to last-good on its own. Observe it.
|
||||
const before = armed.appliedHash
|
||||
setArmed(null)
|
||||
flash('Auto-rolled back')
|
||||
let cancelled = false
|
||||
const off = onPendingConfirmExpire(() => {
|
||||
const before = liveHashRef.current
|
||||
flash('Confirm window elapsed')
|
||||
setResult({
|
||||
kind: 'expire',
|
||||
tone: 'warn',
|
||||
text: 'Confirm window elapsed. Reading what the daemon did…',
|
||||
before,
|
||||
after: before,
|
||||
})
|
||||
void (async () => {
|
||||
const after = (await refreshStatus())?.hash ?? ''
|
||||
let after = before
|
||||
for (let i = 0; i < 4 && !cancelled; i++) {
|
||||
await new Promise((r) => window.setTimeout(r, 1500))
|
||||
if (cancelled) return
|
||||
after = (await refreshStatus())?.hash ?? after
|
||||
if (after !== before) break
|
||||
}
|
||||
if (cancelled) return
|
||||
setResult({
|
||||
kind: 'expire',
|
||||
tone: 'warn',
|
||||
text: 'Confirm window elapsed — daemon auto-rolled back to last-good config.',
|
||||
text:
|
||||
after !== before
|
||||
? 'Confirm window elapsed and the live config changed — the daemon reverted to its last-good config.'
|
||||
: 'Confirm window elapsed and the live config has not changed, so this config is still running. ' +
|
||||
'The daemon only reverts if nothing else re-applied the data plane while the window was open; ' +
|
||||
'otherwise it stands down and keeps what is live. Which one happened is in the daemon log — ' +
|
||||
'download it from Settings, or run `logread -e shater`.',
|
||||
before,
|
||||
after,
|
||||
})
|
||||
})()
|
||||
return
|
||||
})
|
||||
return () => {
|
||||
cancelled = true
|
||||
off()
|
||||
}
|
||||
const id = window.setTimeout(() => {
|
||||
setArmed((a) => (a ? { ...a, remaining: a.remaining - 1 } : a))
|
||||
}, 1000)
|
||||
return () => window.clearTimeout(id)
|
||||
}, [armed, flash, refreshStatus])
|
||||
}, [flash, refreshStatus])
|
||||
|
||||
const confirmWindow = globals?.ConfirmTimeout ?? 0
|
||||
|
||||
@@ -146,8 +177,9 @@ export default function Apply() {
|
||||
return
|
||||
}
|
||||
const after = (await refreshStatus())?.hash ?? before
|
||||
// The window itself was recorded by api.apply(); this branch only writes the
|
||||
// readout for it.
|
||||
if (r.changed && confirmWindow > 0) {
|
||||
setArmed({ total: confirmWindow, remaining: confirmWindow, appliedHash: after })
|
||||
setResult({
|
||||
kind: 'apply',
|
||||
tone: 'good',
|
||||
@@ -179,7 +211,8 @@ export default function Apply() {
|
||||
const doConfirm = useCallback(async () => {
|
||||
const before = status?.hash ?? ''
|
||||
setBusy('confirm')
|
||||
setArmed(null) // stop the countdown immediately; confirm cancels the auto-rollback
|
||||
// api.confirm() clears the shared window on success — the countdown stops the
|
||||
// moment the daemon agrees, not the moment we asked.
|
||||
try {
|
||||
const r = await apiConfirm()
|
||||
if (r.error) {
|
||||
@@ -208,7 +241,7 @@ export default function Apply() {
|
||||
const before = status?.hash ?? ''
|
||||
setConfirmingRollback(false)
|
||||
setBusy('rollback')
|
||||
setArmed(null) // rolling back also cancels any pending confirm window
|
||||
// api.rollback() clears the shared window on success (rolling back ends it).
|
||||
try {
|
||||
const r = await apiRollback()
|
||||
if (r.error) {
|
||||
@@ -237,19 +270,37 @@ export default function Apply() {
|
||||
}, [status, flash, refreshStatus])
|
||||
|
||||
// ---- derived display state (mirrors Overview's LED semantics) ----
|
||||
const killArmed = globals ? globals.KillSwitch === 'closed' : false
|
||||
const engineVariant: LedVariant = !status
|
||||
? 'off'
|
||||
: status.running && status.active
|
||||
? 'on'
|
||||
: status.running
|
||||
? 'amber'
|
||||
: 'crit'
|
||||
const dataVariant: LedVariant = status?.table ? 'on' : status?.running ? 'amber' : 'off'
|
||||
//
|
||||
// The LIVE kill-switch wins over the saved one, exactly as on Overview: this row
|
||||
// is a status readout, and the config on disk can already differ from what is
|
||||
// installed. Falls back to the config only while /api/status is unread.
|
||||
const killArmed = (status?.kill_switch ?? globals?.KillSwitch ?? 'closed') === 'closed'
|
||||
// Whether that setting is actually installed — same three-plus-unknown reading
|
||||
// as Overview, so the two pages cannot disagree about the same router.
|
||||
const kill = killSwitchReadout(status, globals?.KillSwitch)
|
||||
const killWord =
|
||||
kill.state === 'open'
|
||||
? 'open'
|
||||
: kill.state === 'armed'
|
||||
? 'fail-closed'
|
||||
: kill.state === 'inert'
|
||||
? 'closed · not in effect'
|
||||
: 'closed · not reported'
|
||||
// Every engine mark on this page comes from ONE reading, and that reading is
|
||||
// able to say "stopped" — see planeState.engineState for why `status.running`
|
||||
// could not. This page is where someone lands when the network is down; three
|
||||
// green lamps here were the difference between "I broke it" and "nothing broke".
|
||||
const engine = engineReadout(status)
|
||||
const engineVariant: LedVariant = engine.variant
|
||||
// No nft table means there is no data plane at all. Under a fail-closed switch
|
||||
// that is a leak (crit); under an open one it is the documented choice (amber).
|
||||
// It used to go amber whenever `running` was true — i.e. always — and unlit
|
||||
// otherwise, so the one state worth shouting about had no colour of its own.
|
||||
const dataVariant: LedVariant = status?.table ? 'on' : !status ? 'off' : killArmed ? 'crit' : 'amber'
|
||||
const configVariant: LedVariant = status?.enabled ? 'on' : 'amber'
|
||||
|
||||
const liveHash = short(status?.hash ?? '')
|
||||
const pct = armed ? Math.max(0, Math.round((armed.remaining / armed.total) * 100)) : 0
|
||||
const pct = armed ? Math.max(0, Math.round((armed.remaining / armed.pending.total) * 100)) : 0
|
||||
|
||||
// Only offer rollback when the daemon says one would revert something: an armed
|
||||
// commit-confirm snapshot, or an engine last-good predecessor. When false there
|
||||
@@ -264,9 +315,7 @@ export default function Apply() {
|
||||
label="Engine"
|
||||
variant={engineVariant}
|
||||
pulse={engineVariant === 'on'}
|
||||
value={
|
||||
!status ? 'checking…' : status.running ? (status.active ? 'active' : 'idle') : 'stopped'
|
||||
}
|
||||
value={engine.word}
|
||||
/>
|
||||
<StatusPip
|
||||
label="Config"
|
||||
@@ -278,11 +327,7 @@ export default function Apply() {
|
||||
variant={dataVariant}
|
||||
value={status?.table ? 'nft installed' : 'no table'}
|
||||
/>
|
||||
<StatusPip
|
||||
label="Kill-switch"
|
||||
variant={killArmed ? 'on' : 'amber'}
|
||||
value={killArmed ? 'fail-closed' : 'open'}
|
||||
/>
|
||||
<StatusPip label="Kill-switch" variant={kill.variant} value={killWord} />
|
||||
</div>
|
||||
|
||||
{statusError && (
|
||||
@@ -310,9 +355,13 @@ export default function Apply() {
|
||||
unit="· sha256"
|
||||
led={{ variant: configVariant }}
|
||||
rows={[
|
||||
{ k: 'engine', v: status?.running ? 'running' : 'stopped', hot: !status?.running },
|
||||
{ k: 'data plane', v: status?.table ? 'nft installed' : 'no table' },
|
||||
{ k: 'kill-switch', v: killArmed ? 'fail-closed' : 'open', hot: !killArmed },
|
||||
{ k: 'engine', v: engine.word, hot: engineVariant === 'crit' },
|
||||
{
|
||||
k: 'data plane',
|
||||
v: status?.table ? 'nft installed' : 'no table',
|
||||
hot: dataVariant === 'crit',
|
||||
},
|
||||
{ k: 'kill-switch', v: killWord, hot: kill.variant === 'crit' || !killArmed },
|
||||
]}
|
||||
/>
|
||||
<Module
|
||||
@@ -325,7 +374,7 @@ export default function Apply() {
|
||||
}
|
||||
led={{ variant: engineVariant }}
|
||||
rows={[
|
||||
{ k: 'state', v: !status ? 'checking…' : status.active ? 'active' : 'idle' },
|
||||
{ k: 'state', v: engine.word, hot: engineVariant === 'crit' },
|
||||
{ k: 'config', v: status?.enabled ? 'enabled' : 'disabled' },
|
||||
{ k: 'schema', v: globals ? `v${globals.SchemaVersion}` : '—' },
|
||||
]}
|
||||
@@ -362,9 +411,13 @@ export default function Apply() {
|
||||
</div>
|
||||
<div className="cc-info">
|
||||
<p className="cc-copy">
|
||||
Applied config <span className="mono">{short(armed.appliedHash)}</span> is live but
|
||||
not yet kept. Confirm to keep it — otherwise the daemon rolls back to the last-good
|
||||
config when the timer hits zero.
|
||||
{/* The live hash IS the applied one while a window is open — that
|
||||
is what "live but not kept" means — so the readout survives a
|
||||
reload instead of depending on what this tab remembers. */}
|
||||
Applied config <span className="mono">{liveHash}</span> is live but not yet kept.
|
||||
Confirm to keep it. At zero the daemon rolls back to the last-good config — unless
|
||||
something else re-applies the data plane first, in which case it stands down and
|
||||
keeps whatever is live.
|
||||
</p>
|
||||
<div className="cc-bar" aria-hidden="true">
|
||||
<span className="cc-bar-fill" style={{ width: `${pct}%` }} />
|
||||
@@ -481,7 +534,9 @@ function labelFor(kind: ActionKind): string {
|
||||
case 'rollback':
|
||||
return 'Rollback'
|
||||
case 'expire':
|
||||
return 'Auto-rollback'
|
||||
// NOT "Auto-rollback": on expiry the daemon either reverts or stands down,
|
||||
// and this page cannot tell which. Name the event it did observe.
|
||||
return 'Window elapsed'
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
+80
-68
@@ -387,6 +387,79 @@
|
||||
.dns-row-state[data-active='on'] {
|
||||
color: var(--led-on);
|
||||
}
|
||||
/* A list that is switched on but has nothing loaded is not "off" and is certainly
|
||||
not "filtering" — warn semantics, the same amber the badges use. */
|
||||
.dns-row-state[data-active='warn'] {
|
||||
color: var(--amber);
|
||||
}
|
||||
|
||||
/* ---- remote-list freshness (mirrors the rule-set rows on Routing) ---- */
|
||||
.dns-row-sync {
|
||||
display: flex;
|
||||
align-items: center;
|
||||
flex-wrap: wrap;
|
||||
gap: 10px;
|
||||
margin-top: 3px;
|
||||
}
|
||||
.dns-sync-fresh {
|
||||
font-family: var(--font-mono);
|
||||
font-size: 11px;
|
||||
color: var(--dim);
|
||||
}
|
||||
.dns-sync-fresh[data-never='y'] {
|
||||
color: var(--amber);
|
||||
}
|
||||
.dns-sync-every,
|
||||
.dns-sync-rules {
|
||||
font-family: var(--font-mono);
|
||||
font-size: 10.5px;
|
||||
letter-spacing: 0.02em;
|
||||
color: var(--faint);
|
||||
}
|
||||
.dns-sync-every::before {
|
||||
content: '↻ ';
|
||||
}
|
||||
.dns-sync-update {
|
||||
display: inline-flex;
|
||||
align-items: center;
|
||||
gap: 6px;
|
||||
padding: 3px 10px;
|
||||
border: 1px solid var(--accent-soft);
|
||||
border-radius: 5px;
|
||||
background: var(--raised);
|
||||
color: var(--accent);
|
||||
font-family: var(--font-mono);
|
||||
font-size: 10px;
|
||||
letter-spacing: var(--track-label);
|
||||
text-transform: uppercase;
|
||||
cursor: pointer;
|
||||
transition: color 0.12s, border-color 0.12s, background 0.12s;
|
||||
}
|
||||
.dns-sync-update:hover:not(:disabled) {
|
||||
border-color: var(--accent);
|
||||
background: color-mix(in srgb, var(--accent) 12%, transparent);
|
||||
}
|
||||
.dns-sync-update:focus-visible {
|
||||
outline: 2px solid var(--accent);
|
||||
outline-offset: 2px;
|
||||
}
|
||||
.dns-sync-update:disabled {
|
||||
opacity: 0.6;
|
||||
cursor: not-allowed;
|
||||
}
|
||||
.dns-sync-spin {
|
||||
width: 10px;
|
||||
height: 10px;
|
||||
border: 2px solid color-mix(in srgb, var(--accent) 35%, transparent);
|
||||
border-top-color: var(--accent);
|
||||
border-radius: 50%;
|
||||
animation: dns-sync-spin 0.7s linear infinite;
|
||||
}
|
||||
@keyframes dns-sync-spin {
|
||||
to {
|
||||
transform: rotate(360deg);
|
||||
}
|
||||
}
|
||||
|
||||
/* badge — groove-bordered, not orange (accent stays reserved) */
|
||||
.dns-badge {
|
||||
@@ -575,80 +648,19 @@
|
||||
}
|
||||
}
|
||||
|
||||
/* alert delivery: deliver-via picker + fallback toggle + caution note */
|
||||
.dns-alert-delivery {
|
||||
display: flex;
|
||||
flex-wrap: wrap;
|
||||
align-items: center;
|
||||
gap: 10px 20px;
|
||||
}
|
||||
.dns-fallback {
|
||||
display: inline-flex;
|
||||
align-items: center;
|
||||
gap: 8px;
|
||||
cursor: pointer;
|
||||
}
|
||||
.dns-fallback-label {
|
||||
font-size: 10px;
|
||||
letter-spacing: var(--track-label, 0.18em);
|
||||
text-transform: uppercase;
|
||||
color: var(--faint);
|
||||
}
|
||||
/* the per-alert-row delivery controls sit inline in the row (like .dns-detour) */
|
||||
.dns-alert-ctl {
|
||||
flex: none;
|
||||
display: flex;
|
||||
flex-direction: column;
|
||||
gap: 8px;
|
||||
min-width: 0;
|
||||
}
|
||||
.dns-alert-note {
|
||||
margin: 0;
|
||||
font-family: var(--font-sans);
|
||||
font-size: 11.5px;
|
||||
line-height: 1.5;
|
||||
color: var(--amber);
|
||||
max-width: 56ch;
|
||||
}
|
||||
.dns-alert-note--row {
|
||||
margin-top: 2px;
|
||||
}
|
||||
|
||||
@media (max-width: 640px) {
|
||||
.dns-alert-ctl {
|
||||
flex-basis: 100%;
|
||||
order: 3;
|
||||
}
|
||||
}
|
||||
|
||||
/* alert event checkboxes */
|
||||
.dns-events {
|
||||
display: flex;
|
||||
flex-wrap: wrap;
|
||||
gap: 8px 16px;
|
||||
margin: 0;
|
||||
padding: 0;
|
||||
border: 0;
|
||||
}
|
||||
.dns-event {
|
||||
display: inline-flex;
|
||||
align-items: center;
|
||||
gap: 6px;
|
||||
font-size: 13px;
|
||||
color: var(--fp-text, inherit);
|
||||
cursor: pointer;
|
||||
}
|
||||
.dns-event input {
|
||||
accent-color: var(--fp-accent, currentColor);
|
||||
}
|
||||
|
||||
@media (prefers-reduced-motion: reduce) {
|
||||
.dns-skel {
|
||||
animation: none;
|
||||
}
|
||||
/* No spin under reduced motion — the static ring + "Updating…" label carry it. */
|
||||
.dns-sync-spin {
|
||||
animation: none;
|
||||
border-top-color: color-mix(in srgb, var(--accent) 35%, transparent);
|
||||
}
|
||||
.dns-chip,
|
||||
.dns-input,
|
||||
.dns-seg-btn {
|
||||
.dns-seg-btn,
|
||||
.dns-sync-update {
|
||||
transition: none;
|
||||
}
|
||||
}
|
||||
|
||||
+205
-489
@@ -1,8 +1,16 @@
|
||||
import './DNS.css'
|
||||
import { useCallback, useEffect, useMemo, useRef, useState } from 'react'
|
||||
import { Button, CatSuggest, Led, SrcPicker, Toggle, useConfirm } from '../components'
|
||||
import { apply as apiApply, getConfig, putConfig, ApiError } from '../api'
|
||||
import type { Alert, DNSRule, Model, Resolver } from '../api'
|
||||
import {
|
||||
apply as apiApply,
|
||||
getConfig,
|
||||
getRulesetStatus,
|
||||
putConfig,
|
||||
updateRuleset as apiUpdateRuleset,
|
||||
ApiError,
|
||||
} from '../api'
|
||||
import type { DNSRule, Model, Resolver, RulesetStatus } from '../api'
|
||||
import { everyLabel, relFetch } from '../format'
|
||||
|
||||
// The DNS / Blocklists page is a thin editor over the desired-state Model —
|
||||
// exactly like Nodes.tsx. Every edit rewrites the relevant slice in-place, PUTs
|
||||
@@ -52,28 +60,9 @@ type GlobalsX = Model['Globals'] & { DNSFilter?: boolean }
|
||||
type DNSModel = Model & {
|
||||
Blocklists?: Blocklist[] | null
|
||||
Allowlists?: Allowlist[] | null
|
||||
Alerts?: Alert[] | null
|
||||
DNSRules?: DNSRule[] | null
|
||||
}
|
||||
|
||||
/**
|
||||
* Every event the daemon actually sends. A retired health-probe event was left
|
||||
* out on purpose: nothing ever fired it, so a channel that subscribed to it would
|
||||
* just stay quiet forever — the one failure mode an alert must not have. Only
|
||||
* events with a live firing path are offered here.
|
||||
*/
|
||||
const ALERT_EVENTS: ReadonlyArray<{ id: string; label: string }> = [
|
||||
{ id: 'killswitch', label: 'Kill-switch' },
|
||||
{ id: 'apply_fail', label: 'Apply failure' },
|
||||
{ id: 'new_device', label: 'New device' },
|
||||
{ id: 'sub_expiry', label: 'Subscription expiring' },
|
||||
]
|
||||
|
||||
// Shown when an alert routes through a detour with no direct fallback — the exact
|
||||
// case where a tunnel-down alert could fail to send. The user asked for this.
|
||||
const VIA_NO_FALLBACK_NOTE =
|
||||
'A kill-switch/tunnel-down alert may not send if it routes through the affected tunnel — enable fallback.'
|
||||
|
||||
// ---- helpers ---------------------------------------------------------------
|
||||
|
||||
const asArray = <T,>(a: T[] | null | undefined): T[] => (a ? a : [])
|
||||
@@ -302,7 +291,6 @@ export default function DNS() {
|
||||
const blocklists = useMemo(() => asArray(config?.Blocklists), [config])
|
||||
const allowlists = useMemo(() => asArray(config?.Allowlists), [config])
|
||||
const resolvers = useMemo<Resolver[]>(() => asArray(config?.Resolvers), [config])
|
||||
const alerts = useMemo<Alert[]>(() => asArray(config?.Alerts), [config])
|
||||
// Ascending Order — the engine evaluates DNS rules first-match, so the list is
|
||||
// shown and edited in the order it actually runs.
|
||||
const dnsRules = useMemo<DNSRule[]>(
|
||||
@@ -310,6 +298,68 @@ export default function DNS() {
|
||||
[config],
|
||||
)
|
||||
|
||||
// ---- did the lists actually LOAD? -----------------------------------------
|
||||
//
|
||||
// A blocklist row said "filtering" whenever the list and the master switch were
|
||||
// both on. Neither of those is evidence that anything is being blocked: a
|
||||
// url/geosite list is fetched by the engine, the daemon treats a failed fetch as
|
||||
// a CRITICAL apply finding, and the row went on saying "filtering" through it.
|
||||
// The Routing page had already been given this reading for rule-sets — the same
|
||||
// endpoint, the same tags (`bl-<name>` / `al-<name>`) — and the DNS page never
|
||||
// asked. Slow poll: lists refresh on a ~24h cadence, so 15s only has to catch a
|
||||
// manual Update-now. Grouped by NAME because a geosite list with N categories
|
||||
// reports N records.
|
||||
const [listStatus, setListStatus] = useState<Map<string, RulesetStatus[]>>(new Map())
|
||||
const [updatingLists, setUpdatingLists] = useState<Set<string>>(new Set())
|
||||
const loadListStatus = useCallback(async () => {
|
||||
try {
|
||||
const all = await getRulesetStatus()
|
||||
const m = new Map<string, RulesetStatus[]>()
|
||||
for (const s of all) {
|
||||
if (s.kind !== 'blocklist' && s.kind !== 'allowlist') continue
|
||||
const key = `${s.kind}:${s.name}`
|
||||
const arr = m.get(key)
|
||||
if (arr) arr.push(s)
|
||||
else m.set(key, [s])
|
||||
}
|
||||
setListStatus(m)
|
||||
} catch {
|
||||
// Engine stopped or an older daemon — keep the last reading. The row falls
|
||||
// back to "load not reported", which claims nothing either way.
|
||||
}
|
||||
}, [])
|
||||
useEffect(() => {
|
||||
void loadListStatus()
|
||||
const id = window.setInterval(() => void loadListStatus(), 15000)
|
||||
return () => window.clearInterval(id)
|
||||
}, [loadListStatus])
|
||||
|
||||
const updateList = useCallback(
|
||||
async (kind: 'blocklist' | 'allowlist', name: string) => {
|
||||
const key = `${kind}:${name}`
|
||||
setUpdatingLists((prev) => new Set(prev).add(key))
|
||||
try {
|
||||
// One geo list can hold several categories, each its own engine tag.
|
||||
const recs = listStatus.get(key) ?? []
|
||||
const tags = recs.length
|
||||
? recs.map((r) => r.tag)
|
||||
: [`${kind === 'blocklist' ? 'bl' : 'al'}-${name}`]
|
||||
for (const tag of tags) await apiUpdateRuleset(tag)
|
||||
await loadListStatus()
|
||||
flash(`${name} refreshed`)
|
||||
} catch (e) {
|
||||
flash(`Refresh failed — ${errText(e)}`)
|
||||
} finally {
|
||||
setUpdatingLists((prev) => {
|
||||
const next = new Set(prev)
|
||||
next.delete(key)
|
||||
return next
|
||||
})
|
||||
}
|
||||
},
|
||||
[listStatus, loadListStatus, flash],
|
||||
)
|
||||
|
||||
const blOn = blocklists.filter((b) => b.Enabled).length
|
||||
const alOn = allowlists.filter((a) => a.Enabled).length
|
||||
const blNames = useMemo(() => new Set(blocklists.map((b) => b.Name)), [blocklists])
|
||||
@@ -500,19 +550,58 @@ export default function DNS() {
|
||||
[config, resolvers, save],
|
||||
)
|
||||
|
||||
/**
|
||||
* Delete a resolver, saying what it was still wired into.
|
||||
*
|
||||
* The three GLOBAL slots (default, fallback, endpoint) are cleared here, because
|
||||
* a global pointing at nothing is never what anyone meant. The DNS RULES are a
|
||||
* different matter: each one is a decision about which queries go where, and
|
||||
* silently deleting or repointing them would change where a device's DNS goes
|
||||
* without saying so. So they are named instead and left alone — the dialog is
|
||||
* where the operator finds out they exist, which is precisely what this page
|
||||
* used to skip: it cleared the two globals without a word and never mentioned
|
||||
* the rules at all.
|
||||
*/
|
||||
const removeResolver = useCallback(
|
||||
async (idx: number) => {
|
||||
if (!config) return
|
||||
const target = resolvers[idx]
|
||||
const g = { ...config.Globals }
|
||||
const slots: string[] = []
|
||||
if (g.ResolverDefault === target.Name) slots.push('the default resolver')
|
||||
if (g.ResolverFallback === target.Name) slots.push('the fallback resolver')
|
||||
if (g.EndpointResolver === target.Name) slots.push('the endpoint resolver')
|
||||
const usedBy = dnsRules.filter((r) => r.Resolver === target.Name)
|
||||
|
||||
const parts: string[] = []
|
||||
if (slots.length > 0) {
|
||||
parts.push(
|
||||
`It is ${slots.join(' and ')} — ${
|
||||
slots.length === 1 ? 'that slot is' : 'those slots are'
|
||||
} cleared, so DNS falls back to the engine's built-in resolution.`,
|
||||
)
|
||||
}
|
||||
if (usedBy.length === 1) {
|
||||
parts.push(
|
||||
`One DNS rule still sends queries to it (order ${usedBy[0].Order}). It is left as it is and will have nowhere to resolve — repoint it before you apply.`,
|
||||
)
|
||||
} else if (usedBy.length > 1) {
|
||||
parts.push(
|
||||
`${usedBy.length} DNS rules still send queries to it (orders ${usedBy
|
||||
.map((r) => r.Order)
|
||||
.join(', ')}). They are left as they are and will have nowhere to resolve — repoint them before you apply.`,
|
||||
)
|
||||
}
|
||||
if (parts.length === 0) parts.push('Nothing else in the config points at it.')
|
||||
|
||||
const ok = await confirm({
|
||||
label: 'Delete resolver',
|
||||
title: `Delete resolver “${target.Name}”?`,
|
||||
body: 'This removes it from the config.',
|
||||
body: parts.join(' '),
|
||||
})
|
||||
if (!ok) return
|
||||
const next = resolvers.filter((_, i) => i !== idx)
|
||||
// Don't leave default/fallback pointing at a resolver that no longer exists.
|
||||
const g = { ...config.Globals }
|
||||
// Don't leave default/fallback/endpoint pointing at a resolver that's gone.
|
||||
const cleared: string[] = []
|
||||
if (g.ResolverDefault === target.Name) {
|
||||
g.ResolverDefault = ''
|
||||
@@ -522,12 +611,16 @@ export default function DNS() {
|
||||
g.ResolverFallback = ''
|
||||
cleared.push('fallback')
|
||||
}
|
||||
if (g.EndpointResolver === target.Name) {
|
||||
g.EndpointResolver = ''
|
||||
cleared.push('endpoint')
|
||||
}
|
||||
const msg = cleared.length
|
||||
? `Deleted ${target.Name} — cleared ${cleared.join(' & ')}`
|
||||
: `Deleted ${target.Name}`
|
||||
void save({ ...config, Globals: g, Resolvers: next }, msg)
|
||||
},
|
||||
[config, resolvers, save],
|
||||
[config, resolvers, dnsRules, save, confirm],
|
||||
)
|
||||
|
||||
const setResolverDefault = useCallback(
|
||||
@@ -606,78 +699,6 @@ export default function DNS() {
|
||||
[config, dnsRules, save, confirm],
|
||||
)
|
||||
|
||||
// ---- alert mutations ------------------------------------------------------
|
||||
const addAlert = useCallback(
|
||||
(draft: Alert): Promise<boolean> => {
|
||||
if (!config) return Promise.resolve(false)
|
||||
const taken = new Set(alerts.map((a) => a.Name))
|
||||
const a: Alert = { ...draft, Name: uniqueName(draft.Name, taken) }
|
||||
return save({ ...config, Alerts: [...alerts, a] }, `Added ${a.Name}`)
|
||||
},
|
||||
[config, alerts, save],
|
||||
)
|
||||
|
||||
const toggleAlert = useCallback(
|
||||
(idx: number, on: boolean) => {
|
||||
if (!config) return
|
||||
const next = alerts.map((a, i) => (i === idx ? { ...a, Enabled: on } : a))
|
||||
void save({ ...config, Alerts: next }, `${next[idx].Name} ${on ? 'enabled' : 'disabled'}`)
|
||||
},
|
||||
[config, alerts, save],
|
||||
)
|
||||
|
||||
const removeAlert = useCallback(
|
||||
async (idx: number) => {
|
||||
if (!config) return
|
||||
const target = alerts[idx]
|
||||
const ok = await confirm({
|
||||
label: 'Delete alert',
|
||||
title: `Delete alert “${target.Name}”?`,
|
||||
body: 'This removes it from the config.',
|
||||
})
|
||||
if (!ok) return
|
||||
const next = alerts.filter((_, i) => i !== idx)
|
||||
void save({ ...config, Alerts: next }, `Deleted ${target.Name}`)
|
||||
},
|
||||
[config, alerts, save, confirm],
|
||||
)
|
||||
|
||||
const setAlertVia = useCallback(
|
||||
(idx: number, v: string) => {
|
||||
if (!config) return
|
||||
const via = v === 'direct' ? '' : v
|
||||
const next = alerts.map((a, i) => (i === idx ? { ...a, Via: via || undefined } : a))
|
||||
void save(
|
||||
{ ...config, Alerts: next },
|
||||
via ? `${next[idx].Name} delivers via ${via}` : `${next[idx].Name} delivers direct`,
|
||||
)
|
||||
},
|
||||
[config, alerts, save],
|
||||
)
|
||||
|
||||
const setAlertFallback = useCallback(
|
||||
(idx: number, on: boolean) => {
|
||||
if (!config) return
|
||||
const next = alerts.map((a, i) => (i === idx ? { ...a, Fallback: on || undefined } : a))
|
||||
void save(
|
||||
{ ...config, Alerts: next },
|
||||
`${next[idx].Name} direct fallback ${on ? 'on' : 'off'}`,
|
||||
)
|
||||
},
|
||||
[config, alerts, save],
|
||||
)
|
||||
|
||||
// Alerts route through Direct/group/node/egress only (no chains) — the contract
|
||||
// vocabulary for Alert.Via. Reuse the resolver detour catalog with chains dropped.
|
||||
const alertCatalog = useMemo<DetourCatalog>(
|
||||
() => ({ ...detourCatalog, chains: [] }),
|
||||
[detourCatalog],
|
||||
)
|
||||
const alertValid = useMemo(() => detourValues(alertCatalog), [alertCatalog])
|
||||
|
||||
const alertNames = useMemo(() => new Set(alerts.map((a) => a.Name)), [alerts])
|
||||
const alertsOn = alerts.filter((a) => a.Enabled).length
|
||||
|
||||
const loading = config === null && loadError === null
|
||||
|
||||
return (
|
||||
@@ -905,8 +926,11 @@ export default function DNS() {
|
||||
categories={b.Categories}
|
||||
response={b.Response}
|
||||
filterOn={dnsFilterOn}
|
||||
statuses={listStatus.get(`blocklist:${b.Name}`) ?? null}
|
||||
updating={updatingLists.has(`blocklist:${b.Name}`)}
|
||||
busy={busy}
|
||||
onToggle={(on) => toggleBlocklist(i, on)}
|
||||
onUpdateNow={() => void updateList('blocklist', b.Name)}
|
||||
onDelete={() => removeBlocklist(i)}
|
||||
/>
|
||||
))}
|
||||
@@ -964,9 +988,13 @@ export default function DNS() {
|
||||
url={a.URL}
|
||||
path={a.Path}
|
||||
entries={a.Entries}
|
||||
categories={a.Categories}
|
||||
filterOn={dnsFilterOn}
|
||||
statuses={listStatus.get(`allowlist:${a.Name}`) ?? null}
|
||||
updating={updatingLists.has(`allowlist:${a.Name}`)}
|
||||
busy={busy}
|
||||
onToggle={(on) => toggleAllowlist(i, on)}
|
||||
onUpdateNow={() => void updateList('allowlist', a.Name)}
|
||||
onDelete={() => removeAllowlist(i)}
|
||||
/>
|
||||
))}
|
||||
@@ -1091,57 +1119,6 @@ export default function DNS() {
|
||||
)}
|
||||
</div>
|
||||
|
||||
{/* ---- 5. ALERTS ---- */}
|
||||
<div className="dns-section" aria-label="Alerts">
|
||||
<header className="dns-sec-hd">
|
||||
<h2 className="dns-sec-title">Alerts</h2>
|
||||
<span className="dns-sec-count mono">
|
||||
{alertsOn} / {alerts.length} on
|
||||
</span>
|
||||
</header>
|
||||
<p className="dns-sec-note">
|
||||
Out-of-band notifications. Delivered <strong>direct to the internet</strong> by default — so a
|
||||
kill-switch or engine-down alert still reaches you when the proxy is down. You can route one
|
||||
through a group, node or egress instead, with a direct fallback if that detour fails.
|
||||
</p>
|
||||
|
||||
<AddAlertForm
|
||||
busy={busy}
|
||||
disabled={!config}
|
||||
taken={alertNames}
|
||||
catalog={alertCatalog}
|
||||
valid={alertValid}
|
||||
onAdd={addAlert}
|
||||
/>
|
||||
|
||||
{loading ? (
|
||||
<ul className="dns-rows" aria-hidden="true">
|
||||
<li className="dns-skel" />
|
||||
</ul>
|
||||
) : alerts.length === 0 ? (
|
||||
<EmptyPlate
|
||||
title="No alerts"
|
||||
body="Add a Telegram bot or a webhook above to get notified when the kill-switch trips, a new device joins, or an apply fails."
|
||||
/>
|
||||
) : (
|
||||
<ul className="dns-rows">
|
||||
{alerts.map((a, i) => (
|
||||
<AlertRow
|
||||
key={`${a.Name}-${i}`}
|
||||
alert={a}
|
||||
busy={busy}
|
||||
catalog={alertCatalog}
|
||||
valid={alertValid}
|
||||
onToggle={(on) => toggleAlert(i, on)}
|
||||
onVia={(v) => setAlertVia(i, v)}
|
||||
onFallback={(on) => setAlertFallback(i, on)}
|
||||
onDelete={() => removeAlert(i)}
|
||||
/>
|
||||
))}
|
||||
</ul>
|
||||
)}
|
||||
</div>
|
||||
|
||||
{toast && (
|
||||
<div className="toast" role="status">
|
||||
{toast}
|
||||
@@ -1151,342 +1128,6 @@ export default function DNS() {
|
||||
)
|
||||
}
|
||||
|
||||
// ---- alert add form + row --------------------------------------------------
|
||||
|
||||
function AddAlertForm({
|
||||
busy,
|
||||
disabled,
|
||||
taken,
|
||||
catalog,
|
||||
valid,
|
||||
onAdd,
|
||||
}: {
|
||||
busy: boolean
|
||||
disabled: boolean
|
||||
taken: Set<string>
|
||||
catalog: DetourCatalog
|
||||
valid: Set<string>
|
||||
onAdd: (a: Alert) => Promise<boolean>
|
||||
}) {
|
||||
const [name, setName] = useState('')
|
||||
const [type, setType] = useState<'telegram' | 'webhook'>('telegram')
|
||||
const [token, setToken] = useState('')
|
||||
const [chatId, setChatId] = useState('')
|
||||
const [url, setUrl] = useState('')
|
||||
const [events, setEvents] = useState<string[]>(['killswitch'])
|
||||
const [via, setVia] = useState('direct')
|
||||
const [fallback, setFallback] = useState(false)
|
||||
const [err, setErr] = useState<string | null>(null)
|
||||
|
||||
const reset = () => {
|
||||
setName('')
|
||||
setType('telegram')
|
||||
setToken('')
|
||||
setChatId('')
|
||||
setUrl('')
|
||||
setEvents(['killswitch'])
|
||||
setVia('direct')
|
||||
setFallback(false)
|
||||
}
|
||||
|
||||
const toggleEvent = (id: string) =>
|
||||
setEvents((prev) => (prev.includes(id) ? prev.filter((e) => e !== id) : [...prev, id]))
|
||||
|
||||
const submit = async () => {
|
||||
const nm = name.trim()
|
||||
if (!nm) {
|
||||
setErr('Give the alert a name.')
|
||||
return
|
||||
}
|
||||
if (taken.has(nm)) {
|
||||
setErr(`An alert named “${nm}” already exists.`)
|
||||
return
|
||||
}
|
||||
if (type === 'telegram') {
|
||||
if (!token.trim() || !chatId.trim()) {
|
||||
setErr('Telegram needs a bot token and a chat ID.')
|
||||
return
|
||||
}
|
||||
} else if (!HTTP_RE.test(url.trim())) {
|
||||
setErr('Enter an http(s):// webhook URL.')
|
||||
return
|
||||
}
|
||||
if (events.length === 0) {
|
||||
setErr('Pick at least one event to notify on.')
|
||||
return
|
||||
}
|
||||
setErr(null)
|
||||
const routed = via !== 'direct'
|
||||
const routing = { Via: routed ? via : undefined, Fallback: routed && fallback ? true : undefined }
|
||||
const draft: Alert =
|
||||
type === 'telegram'
|
||||
? { Name: nm, Enabled: true, Type: 'telegram', Token: token.trim(), ChatID: chatId.trim(), Events: events, ...routing }
|
||||
: { Name: nm, Enabled: true, Type: 'webhook', URL: url.trim(), Events: events, ...routing }
|
||||
const ok = await onAdd(draft)
|
||||
if (ok) reset()
|
||||
}
|
||||
|
||||
const routed = via !== 'direct'
|
||||
|
||||
return (
|
||||
<form
|
||||
className="dns-add"
|
||||
onSubmit={(e) => {
|
||||
e.preventDefault()
|
||||
void submit()
|
||||
}}
|
||||
>
|
||||
<div className="dns-add-top">
|
||||
<input
|
||||
className="dns-input dns-input--name"
|
||||
type="text"
|
||||
spellCheck={false}
|
||||
autoComplete="off"
|
||||
placeholder="Alert name"
|
||||
aria-label="Alert name"
|
||||
value={name}
|
||||
onChange={(e) => {
|
||||
setName(e.target.value)
|
||||
if (err) setErr(null)
|
||||
}}
|
||||
disabled={busy || disabled}
|
||||
/>
|
||||
<div className="dns-seg" role="group" aria-label="Alert type">
|
||||
<button
|
||||
type="button"
|
||||
className={type === 'telegram' ? 'dns-seg-btn on' : 'dns-seg-btn'}
|
||||
aria-pressed={type === 'telegram'}
|
||||
onClick={() => setType('telegram')}
|
||||
disabled={busy || disabled}
|
||||
>
|
||||
Telegram
|
||||
</button>
|
||||
<button
|
||||
type="button"
|
||||
className={type === 'webhook' ? 'dns-seg-btn on' : 'dns-seg-btn'}
|
||||
aria-pressed={type === 'webhook'}
|
||||
onClick={() => setType('webhook')}
|
||||
disabled={busy || disabled}
|
||||
>
|
||||
Webhook
|
||||
</button>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
{type === 'telegram' ? (
|
||||
<>
|
||||
<input
|
||||
className="dns-input"
|
||||
type="password"
|
||||
spellCheck={false}
|
||||
autoComplete="off"
|
||||
placeholder="Bot token (kept secret)"
|
||||
aria-label="Telegram bot token"
|
||||
value={token}
|
||||
onChange={(e) => {
|
||||
setToken(e.target.value)
|
||||
if (err) setErr(null)
|
||||
}}
|
||||
disabled={busy || disabled}
|
||||
/>
|
||||
<input
|
||||
className="dns-input"
|
||||
type="text"
|
||||
spellCheck={false}
|
||||
autoComplete="off"
|
||||
placeholder="Chat ID (e.g. -1001234567890)"
|
||||
aria-label="Telegram chat ID"
|
||||
value={chatId}
|
||||
onChange={(e) => {
|
||||
setChatId(e.target.value)
|
||||
if (err) setErr(null)
|
||||
}}
|
||||
disabled={busy || disabled}
|
||||
/>
|
||||
</>
|
||||
) : (
|
||||
<input
|
||||
className="dns-input"
|
||||
type="text"
|
||||
inputMode="url"
|
||||
spellCheck={false}
|
||||
autoComplete="off"
|
||||
placeholder="https://hooks.example.com/…"
|
||||
aria-label="Webhook URL"
|
||||
value={url}
|
||||
onChange={(e) => {
|
||||
setUrl(e.target.value)
|
||||
if (err) setErr(null)
|
||||
}}
|
||||
disabled={busy || disabled}
|
||||
/>
|
||||
)}
|
||||
|
||||
<fieldset className="dns-events" aria-label="Events to notify on">
|
||||
{ALERT_EVENTS.map((ev) => (
|
||||
<label key={ev.id} className="dns-event">
|
||||
<input
|
||||
type="checkbox"
|
||||
checked={events.includes(ev.id)}
|
||||
onChange={() => toggleEvent(ev.id)}
|
||||
disabled={busy || disabled}
|
||||
/>
|
||||
<span>{ev.label}</span>
|
||||
</label>
|
||||
))}
|
||||
</fieldset>
|
||||
|
||||
<div className="dns-alert-delivery">
|
||||
<label className="dns-resp">
|
||||
<span className="dns-resp-label mono">Deliver via</span>
|
||||
<DetourSelect
|
||||
value={via}
|
||||
catalog={catalog}
|
||||
valid={valid}
|
||||
busy={busy}
|
||||
disabled={disabled}
|
||||
ariaLabel="Deliver alert via"
|
||||
onChange={setVia}
|
||||
directLabel="Direct (default)"
|
||||
/>
|
||||
</label>
|
||||
<label className="dns-fallback">
|
||||
<Toggle
|
||||
pressed={fallback}
|
||||
onChange={setFallback}
|
||||
label={fallback ? 'Disable direct fallback' : 'Enable direct fallback'}
|
||||
disabled={busy || disabled || !routed}
|
||||
/>
|
||||
<span className="dns-fallback-label mono">Fallback to direct</span>
|
||||
</label>
|
||||
</div>
|
||||
{routed && !fallback && (
|
||||
<p className="dns-alert-note" role="note">
|
||||
{VIA_NO_FALLBACK_NOTE}
|
||||
</p>
|
||||
)}
|
||||
|
||||
<div className="dns-add-actions">
|
||||
{err && (
|
||||
<p className="dns-field-err" role="alert">
|
||||
{err}
|
||||
</p>
|
||||
)}
|
||||
<Button type="submit" variant="primary" disabled={busy || disabled}>
|
||||
{busy ? 'Saving…' : 'Add alert'}
|
||||
</Button>
|
||||
</div>
|
||||
</form>
|
||||
)
|
||||
}
|
||||
|
||||
function AlertRow({
|
||||
alert,
|
||||
busy,
|
||||
catalog,
|
||||
valid,
|
||||
onToggle,
|
||||
onVia,
|
||||
onFallback,
|
||||
onDelete,
|
||||
}: {
|
||||
alert: Alert
|
||||
busy: boolean
|
||||
catalog: DetourCatalog
|
||||
valid: Set<string>
|
||||
onToggle: (on: boolean) => void
|
||||
onVia: (v: string) => void
|
||||
onFallback: (on: boolean) => void
|
||||
onDelete: () => void
|
||||
}) {
|
||||
// Never render the token/URL in clear — show a masked descriptor only.
|
||||
const detail = useMemo(() => {
|
||||
if (alert.Type === 'telegram') {
|
||||
return { text: `chat ${alert.ChatID || '—'}`, masked: !!alert.Token }
|
||||
}
|
||||
const { host, masked } = maskUrl(alert.URL ?? '')
|
||||
return { text: host, masked: masked || !!alert.URL }
|
||||
}, [alert.Type, alert.ChatID, alert.Token, alert.URL])
|
||||
|
||||
const events = asArray(alert.Events)
|
||||
const canon = useMemo(() => canonDetour(alert.Via, catalog), [alert.Via, catalog])
|
||||
const route = useMemo(() => describeDetour(canon, catalog, valid), [canon, catalog, valid])
|
||||
const routed = canon !== 'direct'
|
||||
const fallback = alert.Fallback ?? false
|
||||
|
||||
return (
|
||||
<li className="dns-row">
|
||||
<Toggle
|
||||
pressed={alert.Enabled}
|
||||
onChange={onToggle}
|
||||
label={`${alert.Enabled ? 'Disable' : 'Enable'} alert ${alert.Name}`}
|
||||
disabled={busy}
|
||||
/>
|
||||
<div className="dns-row-main">
|
||||
<div className="dns-row-l1">
|
||||
<span className="dns-row-name">{alert.Name}</span>
|
||||
<span className="dns-badge">{alert.Type}</span>
|
||||
{events.map((e) => (
|
||||
<span key={e} className="dns-badge dns-badge--accent">
|
||||
{e}
|
||||
</span>
|
||||
))}
|
||||
</div>
|
||||
<div className="dns-row-l2 mono">
|
||||
<span className="dns-row-detail">{detail.text}</span>
|
||||
{detail.masked && (
|
||||
<span className="dns-masked" title="Secret is stored but hidden here">
|
||||
secret hidden
|
||||
</span>
|
||||
)}
|
||||
{route.direct ? (
|
||||
<span className="dns-path">direct</span>
|
||||
) : (
|
||||
<span className="dns-path" data-active="on" data-missing={route.missing ? 'y' : undefined}>
|
||||
{route.prefix} <strong className="dns-path-name">{route.name}</strong>
|
||||
{route.missing && <span className="dns-path-flag"> (missing)</span>}
|
||||
{fallback ? ' · +direct fallback' : ' · no fallback'}
|
||||
</span>
|
||||
)}
|
||||
</div>
|
||||
{routed && !fallback && <p className="dns-alert-note dns-alert-note--row">{VIA_NO_FALLBACK_NOTE}</p>}
|
||||
</div>
|
||||
<div className="dns-alert-ctl">
|
||||
<label className="dns-detour">
|
||||
<span className="dns-detour-label mono">Deliver via</span>
|
||||
<DetourSelect
|
||||
value={canon}
|
||||
catalog={catalog}
|
||||
valid={valid}
|
||||
busy={busy}
|
||||
disabled={false}
|
||||
ariaLabel={`Deliver alert ${alert.Name} via`}
|
||||
onChange={onVia}
|
||||
directLabel="Direct (default)"
|
||||
/>
|
||||
</label>
|
||||
<label className="dns-fallback">
|
||||
<Toggle
|
||||
pressed={fallback}
|
||||
onChange={onFallback}
|
||||
label={`${fallback ? 'Disable' : 'Enable'} direct fallback for ${alert.Name}`}
|
||||
disabled={busy || !routed}
|
||||
/>
|
||||
<span className="dns-fallback-label mono">Fallback to direct</span>
|
||||
</label>
|
||||
</div>
|
||||
<Button
|
||||
className="dns-del"
|
||||
onClick={onDelete}
|
||||
disabled={busy}
|
||||
aria-label={`Delete alert ${alert.Name}`}
|
||||
>
|
||||
Delete
|
||||
</Button>
|
||||
</li>
|
||||
)
|
||||
}
|
||||
|
||||
// ---- add form --------------------------------------------------------------
|
||||
|
||||
interface AddDraft {
|
||||
@@ -1717,8 +1358,11 @@ function ListRow({
|
||||
categories,
|
||||
response,
|
||||
filterOn,
|
||||
statuses,
|
||||
updating,
|
||||
busy,
|
||||
onToggle,
|
||||
onUpdateNow,
|
||||
onDelete,
|
||||
}: {
|
||||
name: string
|
||||
@@ -1730,8 +1374,14 @@ function ListRow({
|
||||
categories?: string[] | null
|
||||
response?: BlockResponse
|
||||
filterOn: boolean
|
||||
/** What the running engine reports about this list, one record per geo category.
|
||||
* null/[] ⇒ nothing reported: an older daemon, a stopped engine, or a list that
|
||||
* has not been applied yet. The row then says so instead of guessing. */
|
||||
statuses: RulesetStatus[] | null
|
||||
updating: boolean
|
||||
busy: boolean
|
||||
onToggle: (on: boolean) => void
|
||||
onUpdateNow: () => void
|
||||
onDelete: () => void
|
||||
}) {
|
||||
const detail = useMemo<{ text: string; masked: boolean; title?: string }>(() => {
|
||||
@@ -1755,8 +1405,45 @@ function ListRow({
|
||||
}
|
||||
}, [source, url, path, entries, categories])
|
||||
|
||||
// A list only actually filters when both it and the master switch are on.
|
||||
const active = enabled && filterOn
|
||||
// url and geosite lists are FETCHED by the engine; inline and file ones are read
|
||||
// straight from the config and are loaded the moment they are applied.
|
||||
const remote = source === 'url' || source === 'geosite'
|
||||
const recs = statuses ?? []
|
||||
const hasStatus = recs.length > 0
|
||||
// A geo list with several categories: the OLDEST fetch (so a category that never
|
||||
// arrived is never hidden behind a fresh sibling) and the SUM of the counts.
|
||||
let ruleCount = 0
|
||||
let neverAny = false
|
||||
let oldestIso = ''
|
||||
for (const s of recs) {
|
||||
ruleCount += s.rule_count
|
||||
if (!s.last_updated) neverAny = true
|
||||
else if (!oldestIso || Date.parse(s.last_updated) < Date.parse(oldestIso)) oldestIso = s.last_updated
|
||||
}
|
||||
const interval = everyLabel(recs[0]?.interval_seconds ?? 0)
|
||||
|
||||
/**
|
||||
* Whether this list is BLOCKING ANYTHING, which is a different question from
|
||||
* whether it is switched on — and the one the row used to answer wrongly.
|
||||
*
|
||||
* "filtering" is now only said when the engine reports rules loaded for it. A
|
||||
* remote list that has never been fetched (the daemon raises this as a critical
|
||||
* apply finding) reads "not loaded", and one that fetched an empty list reads
|
||||
* "empty". Nothing reported at all is "load not reported": unknown, not green.
|
||||
*/
|
||||
const state: { text: string; tone: 'on' | 'off' | 'warn' } = !enabled
|
||||
? { text: 'off', tone: 'off' }
|
||||
: !filterOn
|
||||
? { text: 'inactive', tone: 'off' }
|
||||
: !remote
|
||||
? { text: 'filtering', tone: 'on' }
|
||||
: !hasStatus
|
||||
? { text: 'load not reported', tone: 'off' }
|
||||
: neverAny
|
||||
? { text: 'not loaded — nothing blocked', tone: 'warn' }
|
||||
: ruleCount === 0
|
||||
? { text: 'loaded empty — nothing blocked', tone: 'warn' }
|
||||
: { text: 'filtering', tone: 'on' }
|
||||
|
||||
return (
|
||||
<li className="dns-row">
|
||||
@@ -1786,10 +1473,39 @@ function ListRow({
|
||||
token hidden
|
||||
</span>
|
||||
)}
|
||||
<span className="dns-row-state" data-active={active ? 'on' : 'off'}>
|
||||
{active ? 'filtering' : 'inactive'}
|
||||
<span className="dns-row-state" data-active={state.tone}>
|
||||
{state.text}
|
||||
</span>
|
||||
</div>
|
||||
{remote && (
|
||||
<div className="dns-row-sync">
|
||||
<span className="dns-sync-fresh" data-never={hasStatus && neverAny ? 'y' : undefined}>
|
||||
{hasStatus ? relFetch(oldestIso) : 'status pending'}
|
||||
</span>
|
||||
{interval && <span className="dns-sync-every">{interval}</span>}
|
||||
{ruleCount > 0 && (
|
||||
<span className="dns-sync-rules">
|
||||
{ruleCount.toLocaleString('en-US')} rule{ruleCount === 1 ? '' : 's'}
|
||||
</span>
|
||||
)}
|
||||
<button
|
||||
type="button"
|
||||
className="dns-sync-update"
|
||||
onClick={onUpdateNow}
|
||||
disabled={busy || updating}
|
||||
aria-label={`Update ${name} now`}
|
||||
>
|
||||
{updating ? (
|
||||
<>
|
||||
<span className="dns-sync-spin" aria-hidden="true" />
|
||||
<span>Updating…</span>
|
||||
</>
|
||||
) : (
|
||||
'Update now'
|
||||
)}
|
||||
</button>
|
||||
</div>
|
||||
)}
|
||||
</div>
|
||||
<Button
|
||||
className="dns-del"
|
||||
|
||||
@@ -85,6 +85,19 @@ const DEFAULT_TPROXY_PORT = 12345
|
||||
* Collapsing those into one switch would make "I want ping to work" silently mean
|
||||
* "I permit a parallel VPN bypass", so the middle option exists to remove that
|
||||
* false choice — and the labels push anyone who wants diagnostics to `icmp`.
|
||||
*
|
||||
* WHY THIS COPY WAS REWRITTEN. `block` used to say "Nothing leaves except through
|
||||
* the tunnel", and it was not true. The daemon let untunnelable traffic out toward
|
||||
* every destination the ROUTING RULES send direct, on the argument that such a host
|
||||
* already has your address from ordinary TCP. Under the commonest setup here —
|
||||
* "tunnel what's blocked, send the rest direct" — the routing default IS direct, so
|
||||
* that covered everything: `block` behaved exactly like `direct`, including ESP/GRE,
|
||||
* i.e. the parallel-VPN case the middle rung exists to exclude. The daemon now drops
|
||||
* unconditionally under `block`, and this copy states the price instead of hiding it
|
||||
* (the owner's call: this router does not do ping and does not do IPTV).
|
||||
*
|
||||
* `icmp` still carries that destination-dependence for its NON-ping half, so its
|
||||
* cost line says so rather than claiming "nothing else gets out".
|
||||
*/
|
||||
type Untunnelable = 'block' | 'icmp' | 'direct'
|
||||
|
||||
@@ -98,7 +111,10 @@ function normUntunnelable(raw: string | undefined): Untunnelable {
|
||||
|
||||
const UNTUNNELABLE_OPTIONS: ReadonlyArray<{ value: string; label: string }> = [
|
||||
{ value: 'block', label: 'Block everything — most private' },
|
||||
{ value: 'icmp', label: 'Allow ping only — for diagnostics' },
|
||||
// Not "Allow ping only": the rung also lets the other untunnelable protocols
|
||||
// out toward directly-routed addresses, and the cost line below says so. A
|
||||
// label that promised "only" would be contradicted two lines under itself.
|
||||
{ value: 'icmp', label: 'Allow ping — for diagnostics' },
|
||||
{ value: 'direct', label: 'Allow everything — most compatible' },
|
||||
]
|
||||
|
||||
@@ -110,13 +126,18 @@ interface PolicyCopy {
|
||||
|
||||
const UNTUNNELABLE_COPY: Record<Untunnelable, PolicyCopy> = {
|
||||
block: {
|
||||
works: 'Nothing leaves except through the tunnel.',
|
||||
cost: 'Ping and traceroute won’t work from your devices, and neither will multicast IPTV or connecting to a VPN from a device on your network.',
|
||||
// Scoped to "this traffic" on purpose. The old line — "Nothing leaves except
|
||||
// through the tunnel" — was doubly loose: it was false (see the note above),
|
||||
// and even read charitably it collides with directly-routed TCP, which does
|
||||
// leave outside the tunnel by design.
|
||||
works:
|
||||
'None of this traffic leaves the router — it’s dropped, whatever your routing rules say. It’s the only setting whose promise doesn’t depend on how the rules are written.',
|
||||
cost: 'Ping and traceroute stop working from your devices. So do IPsec and PPTP VPN connections made from a device on your network, multicast IPTV, and SCTP. VPNs that run over UDP — WireGuard, OpenVPN-UDP, and IPsec through NAT (IKEv2/NAT-T) — are unaffected: they go through the tunnel like everything else.',
|
||||
tone: 'good',
|
||||
},
|
||||
icmp: {
|
||||
works: 'Ping and traceroute work, so you can check whether something is reachable.',
|
||||
cost: 'Whatever you ping sees your real IP address instead of the tunnel’s. Only for hosts you deliberately ping, and nothing else gets out — IPTV and VPN connections stay blocked.',
|
||||
works: 'Ping and traceroute work everywhere, so you can check whether something is reachable.',
|
||||
cost: 'Whatever you ping sees your real IP address instead of the tunnel’s. IPsec, PPTP and IPTV also get out — but only toward addresses your routing rules already send direct, so a VPN app on a device can still open its own connection beside this one if its server is one of those.',
|
||||
tone: 'warn',
|
||||
},
|
||||
direct: {
|
||||
@@ -551,7 +572,7 @@ export default function Networks({ status }: { status?: Status | null }) {
|
||||
|
||||
{untunnelable === 'block' && (
|
||||
<p className="nw-sec-note nw-policy-hint">
|
||||
If you just want to check whether a site is reachable, choose <strong>Allow ping only</strong>{' '}
|
||||
If you just want to check whether a site is reachable, choose <strong>Allow ping</strong>{' '}
|
||||
rather than allowing everything — it’s the narrower of the two.
|
||||
</p>
|
||||
)}
|
||||
|
||||
@@ -196,6 +196,52 @@
|
||||
border-color: color-mix(in srgb, var(--amber) 55%, var(--groove));
|
||||
color: var(--amber);
|
||||
}
|
||||
/* "not built" — the saved switch says on and the engine has no such outbound. */
|
||||
.badge--crit {
|
||||
border-color: color-mix(in srgb, var(--crit) 55%, var(--groove));
|
||||
color: var(--crit);
|
||||
}
|
||||
|
||||
/* ---- last-apply findings, attached to the row they are about ----
|
||||
Sits under the row's own two lines, inside the row plate, so a node the
|
||||
generator threw away cannot read as an ordinary enabled node. Severity carries
|
||||
the colour; the accent stays reserved for controls. */
|
||||
.row-findings {
|
||||
margin: 6px 0 0;
|
||||
padding: 0;
|
||||
list-style: none;
|
||||
display: flex;
|
||||
flex-direction: column;
|
||||
gap: 5px;
|
||||
}
|
||||
.row-finding {
|
||||
display: flex;
|
||||
align-items: flex-start;
|
||||
gap: 8px;
|
||||
padding: 7px 9px;
|
||||
border: 1px solid color-mix(in srgb, var(--amber) 40%, var(--groove));
|
||||
border-radius: 6px;
|
||||
background: color-mix(in srgb, var(--sink) 35%, transparent);
|
||||
}
|
||||
.row-finding--critical {
|
||||
border-color: color-mix(in srgb, var(--crit) 45%, var(--groove));
|
||||
}
|
||||
.row-finding-msg {
|
||||
flex: 1;
|
||||
min-width: 0;
|
||||
font-size: 12px;
|
||||
line-height: 1.5;
|
||||
color: var(--ink);
|
||||
max-width: 82ch;
|
||||
overflow-wrap: anywhere;
|
||||
}
|
||||
/* Findings that belong to no single row (see Nodes.tsx globalFindings). */
|
||||
.node-findings {
|
||||
margin-bottom: calc(var(--u, 8px) * 2);
|
||||
}
|
||||
.node-findings .row-findings {
|
||||
margin-top: 0;
|
||||
}
|
||||
|
||||
/* masked-credential marker */
|
||||
.masked {
|
||||
@@ -641,6 +687,20 @@ select.fp-input {
|
||||
letter-spacing: 0.06em;
|
||||
color: var(--faint);
|
||||
}
|
||||
/* A collapsed bucket has to carry its own bad news: a 300-node subscription is
|
||||
closed by default, and the per-row findings inside it are otherwise unreachable
|
||||
without knowing to look. */
|
||||
.group-flagged {
|
||||
flex: none;
|
||||
display: inline-flex;
|
||||
align-items: center;
|
||||
gap: 6px;
|
||||
font-family: var(--font-mono);
|
||||
font-size: 10.5px;
|
||||
letter-spacing: 0.06em;
|
||||
text-transform: uppercase;
|
||||
color: var(--amber);
|
||||
}
|
||||
.group-rows {
|
||||
margin-top: 8px;
|
||||
}
|
||||
|
||||
+153
-3
@@ -6,12 +6,14 @@ import { Button, Led, Toggle, useConfirm } from '../components'
|
||||
import {
|
||||
apply as apiApply,
|
||||
getConfig,
|
||||
getStatus,
|
||||
putConfig,
|
||||
importWg,
|
||||
updateSubscription,
|
||||
ApiError,
|
||||
} from '../api'
|
||||
import type { Model, Node as NodeCfg, Subscription } from '../api'
|
||||
import type { Model, Node as NodeCfg, StatusWarning, Subscription } from '../api'
|
||||
import { entityFindings, findingsByName } from '../findings'
|
||||
import { fmtBytes, fmtDate, fmtUntil } from '../format'
|
||||
|
||||
// The whole page is a thin editor over the desired-state Model: every mutation
|
||||
@@ -519,6 +521,43 @@ export default function Nodes() {
|
||||
void loadConfig()
|
||||
}, [loadConfig])
|
||||
|
||||
// ---- what the last apply said about these nodes ---------------------------
|
||||
//
|
||||
// The generator drops a node it cannot build and names it: an unparseable
|
||||
// share link (generate/outbound.go), a name colliding with a reserved tag, a
|
||||
// WireGuard private key materialised twice (generate/wgdedup.go). Until now
|
||||
// this page never read /api/status, so a node the engine had thrown away
|
||||
// rendered as an ordinary row with a green toggle — the switch said on and
|
||||
// there was no such outbound anywhere in the running config.
|
||||
//
|
||||
// Findings are attached to the ROWS, not summarised at the top: a 300-node
|
||||
// subscription makes a list of names useless, and the row is where the false
|
||||
// reassurance was.
|
||||
const [findings, setFindings] = useState<StatusWarning[]>([])
|
||||
const loadFindings = useCallback(async () => {
|
||||
try {
|
||||
const s = await getStatus()
|
||||
setFindings(entityFindings(s.warnings, ['node', 'subscription']))
|
||||
} catch {
|
||||
// Status is a supplement here, not the page. Keep the last set rather than
|
||||
// clearing it — a dropped poll is not the same as "the problem is fixed".
|
||||
}
|
||||
}, [])
|
||||
useEffect(() => {
|
||||
void loadFindings()
|
||||
}, [loadFindings])
|
||||
|
||||
const nodeFindings = useMemo(
|
||||
() => findingsByName(findings.filter((w) => w.section === 'node')),
|
||||
[findings],
|
||||
)
|
||||
const subFindings = useMemo(
|
||||
() => findingsByName(findings.filter((w) => w.section === 'subscription')),
|
||||
[findings],
|
||||
)
|
||||
// Findings about nodes/subscriptions in general, which belong to no single row.
|
||||
const globalFindings = useMemo(() => findings.filter((w) => !w.name), [findings])
|
||||
|
||||
// ---- toast + persistent apply banner --------------------------------------
|
||||
const [toast, setToast] = useState<string | null>(null)
|
||||
const toastTimer = useRef<number | undefined>(undefined)
|
||||
@@ -573,8 +612,11 @@ export default function Nodes() {
|
||||
flash(`Apply failed — ${errText(e)}`)
|
||||
} finally {
|
||||
setApplying(false)
|
||||
// An apply is exactly what rewrites the findings — including clearing the
|
||||
// ones the operator just fixed.
|
||||
void loadFindings()
|
||||
}
|
||||
}, [flash, loadConfig])
|
||||
}, [flash, loadConfig, loadFindings])
|
||||
|
||||
// ---- node mutations -------------------------------------------------------
|
||||
const nodes = useMemo(() => asArray(config?.Nodes), [config])
|
||||
@@ -741,14 +783,35 @@ export default function Nodes() {
|
||||
[config, nodes, save],
|
||||
)
|
||||
|
||||
/**
|
||||
* Delete a node, naming everything that still points at it.
|
||||
*
|
||||
* `findNodeReferences` was already here and already right — it just wasn't asked
|
||||
* on the one path where the answer matters. A RENAME carried its references and
|
||||
* said so; a DELETE said "This removes it from the config", which is true of the
|
||||
* node and silent about the rule, group member, chain hop or resolver detour
|
||||
* left spelling a name nothing answers to. That is not a cosmetic dangle: an
|
||||
* unresolved target does not fall through to the default route, so the traffic
|
||||
* aimed at it is blocked.
|
||||
*/
|
||||
const removeNode = useCallback(
|
||||
async (idx: number) => {
|
||||
if (!config) return
|
||||
const target = nodes[idx]
|
||||
const refs = findNodeReferences(config, target.Name)
|
||||
const shown = refs.slice(0, 4).map((r) => r.label)
|
||||
const more = refs.length - shown.length
|
||||
const ok = await confirm({
|
||||
label: 'Delete node',
|
||||
title: `Delete node “${target.Name}”?`,
|
||||
body: 'This removes it from the config.',
|
||||
body:
|
||||
refs.length === 0
|
||||
? 'Nothing else in the config points at it.'
|
||||
: `${refSummary(refs)} still ${refs.length === 1 ? 'points' : 'point'} at it — ${shown.join(
|
||||
', ',
|
||||
)}${
|
||||
more > 0 ? `, and ${more} more` : ''
|
||||
}. Nothing rewrites them, and a target that no longer resolves does not fall through to the default route: the traffic aimed at it is blocked.`,
|
||||
})
|
||||
if (!ok) return
|
||||
const next = nodes.filter((_, i) => i !== idx)
|
||||
@@ -973,6 +1036,14 @@ export default function Nodes() {
|
||||
</div>
|
||||
)}
|
||||
|
||||
{/* Findings about nodes in general — no single row owns them, so they sit
|
||||
above the lists rather than being dropped for having no name. */}
|
||||
{globalFindings.length > 0 && (
|
||||
<div className="node-findings">
|
||||
<RowFindings findings={globalFindings} />
|
||||
</div>
|
||||
)}
|
||||
|
||||
{/* ---- NODES ---- */}
|
||||
<div className="node-section" aria-label="Nodes">
|
||||
<header className="sec-hd">
|
||||
@@ -1128,6 +1199,7 @@ export default function Nodes() {
|
||||
<NodeGroup
|
||||
key={g.key || '__manual__'}
|
||||
group={g}
|
||||
findings={nodeFindings}
|
||||
open={isGroupOpen(g)}
|
||||
busy={busy}
|
||||
egressNames={egressNames}
|
||||
@@ -1217,6 +1289,7 @@ export default function Nodes() {
|
||||
busy={busy}
|
||||
catalog={detourCatalog}
|
||||
valid={detourValid}
|
||||
findings={subFindings.get(s.Name) ?? EMPTY_FINDINGS}
|
||||
onToggle={(on) => toggleSub(i, on)}
|
||||
onDelete={() => removeSub(i)}
|
||||
onEdit={(patch) => editSub(i, patch)}
|
||||
@@ -1243,6 +1316,7 @@ function NodeGroup({
|
||||
open,
|
||||
busy,
|
||||
egressNames,
|
||||
findings,
|
||||
onToggle,
|
||||
onToggleNode,
|
||||
onRemoveNode,
|
||||
@@ -1253,6 +1327,8 @@ function NodeGroup({
|
||||
open: boolean
|
||||
busy: boolean
|
||||
egressNames: string[]
|
||||
/** Last-apply findings per node name (findings.ts findingsByName). */
|
||||
findings: Map<string, StatusWarning[]>
|
||||
onToggle: () => void
|
||||
onToggleNode: (idx: number, on: boolean) => void
|
||||
onRemoveNode: (idx: number) => void
|
||||
@@ -1262,6 +1338,13 @@ function NodeGroup({
|
||||
const panelId = `node-group-${group.key || 'manual'}`
|
||||
// The same inventory count as the section header, scoped to this bucket.
|
||||
const count = useMemo(() => fmtEnabled(group.items.map((i) => i.node)), [group.items])
|
||||
// How many nodes in this bucket the last apply had something to say about —
|
||||
// shown on the COLLAPSED header, because a subscription of 300 nodes is
|
||||
// collapsed by default and the row badge below would never be seen otherwise.
|
||||
const flagged = useMemo(
|
||||
() => group.items.filter(({ node }) => findings.has(node.Name)).length,
|
||||
[group.items, findings],
|
||||
)
|
||||
return (
|
||||
<section className={`node-group${open ? ' node-group--open' : ''}`}>
|
||||
<h3 className="group-hd-wrap">
|
||||
@@ -1275,6 +1358,12 @@ function NodeGroup({
|
||||
<span className="group-caret" aria-hidden="true" />
|
||||
<span className="group-name">{group.label}</span>
|
||||
<span className="group-count mono">{count}</span>
|
||||
{flagged > 0 && (
|
||||
<span className="group-flagged" title="Findings from the last apply">
|
||||
<Led variant="amber" />
|
||||
{flagged} flagged
|
||||
</span>
|
||||
)}
|
||||
</button>
|
||||
</h3>
|
||||
{open && group.key !== '' && (
|
||||
@@ -1291,6 +1380,7 @@ function NodeGroup({
|
||||
node={node}
|
||||
busy={busy}
|
||||
egressNames={egressNames}
|
||||
findings={findings.get(node.Name) ?? EMPTY_FINDINGS}
|
||||
onToggle={(on) => onToggleNode(idx, on)}
|
||||
onDelete={() => onRemoveNode(idx)}
|
||||
onRename={(name, onError) => onRenameNode(idx, name, onError)}
|
||||
@@ -1303,10 +1393,29 @@ function NodeGroup({
|
||||
)
|
||||
}
|
||||
|
||||
/** One shared empty array, so a clean row doesn't get a fresh identity per render. */
|
||||
const EMPTY_FINDINGS: StatusWarning[] = []
|
||||
|
||||
/**
|
||||
* Did the generator say it left this entity OUT of the engine config?
|
||||
*
|
||||
* The producers all end the sentence with the same word — "(skipped)" for an
|
||||
* unparseable share link or a bad WireGuard endpoint (generate/outbound.go),
|
||||
* "skipped" for a name colliding with a reserved tag — and wgdedup says only one
|
||||
* of the duplicates "is kept". Read the daemon's word rather than inventing a
|
||||
* verdict: a finding that does NOT say this may well be about a node that is
|
||||
* running perfectly, and badging it "not built" would be a new lie in place of
|
||||
* the old one.
|
||||
*/
|
||||
function skipped(findings: StatusWarning[]): boolean {
|
||||
return findings.some((f) => /\bskipped\b|\bis kept\b/i.test(f.message))
|
||||
}
|
||||
|
||||
function NodeRow({
|
||||
node,
|
||||
busy,
|
||||
egressNames,
|
||||
findings,
|
||||
onToggle,
|
||||
onDelete,
|
||||
onRename,
|
||||
@@ -1315,6 +1424,8 @@ function NodeRow({
|
||||
node: NodeCfg
|
||||
busy: boolean
|
||||
egressNames: string[]
|
||||
/** What the last apply said about THIS node; empty when it said nothing. */
|
||||
findings: StatusWarning[]
|
||||
onToggle: (on: boolean) => void
|
||||
onDelete: () => void
|
||||
onRename: (name: string, onError: (msg: string) => void) => Promise<boolean>
|
||||
@@ -1462,6 +1573,16 @@ function NodeRow({
|
||||
)}
|
||||
<span className="badge">{proto}</span>
|
||||
{node.Stale && <span className="badge badge--warn">stale</span>}
|
||||
{/* The toggle above is the SAVED state. When the last apply couldn't
|
||||
build this node the engine has no such outbound, and the two
|
||||
disagree — so the row says which, rather than leaving a green
|
||||
switch to imply the node is carrying traffic. The word is the
|
||||
daemon's own where it used one. */}
|
||||
{findings.length > 0 && (
|
||||
<span className={`badge badge--${skipped(findings) ? 'crit' : 'warn'}`}>
|
||||
{skipped(findings) ? 'not built' : 'flagged'}
|
||||
</span>
|
||||
)}
|
||||
</div>
|
||||
{renameErr && (
|
||||
<p className="row-err" role="alert">
|
||||
@@ -1485,6 +1606,7 @@ function NodeRow({
|
||||
</span>
|
||||
)}
|
||||
</div>
|
||||
<RowFindings findings={findings} />
|
||||
</div>
|
||||
<div className="row-actions">
|
||||
{canPin && (
|
||||
@@ -1604,6 +1726,7 @@ function SubRow({
|
||||
busy,
|
||||
catalog,
|
||||
valid,
|
||||
findings,
|
||||
onToggle,
|
||||
onDelete,
|
||||
onEdit,
|
||||
@@ -1614,6 +1737,8 @@ function SubRow({
|
||||
busy: boolean
|
||||
catalog: DetourCatalog
|
||||
valid: Set<string>
|
||||
/** What the last apply said about THIS subscription; empty when it said nothing. */
|
||||
findings: StatusWarning[]
|
||||
onToggle: (on: boolean) => void
|
||||
onDelete: () => void
|
||||
onEdit: (patch: Subscription) => Promise<boolean>
|
||||
@@ -1638,6 +1763,7 @@ function SubRow({
|
||||
<span className="row-name">{sub.Name}</span>
|
||||
{sub.Format && sub.Format !== 'auto' && <span className="badge">{sub.Format}</span>}
|
||||
{sub.FetchVia === 'proxy' && <span className="badge">via proxy</span>}
|
||||
{findings.length > 0 && <span className="badge badge--warn">flagged</span>}
|
||||
</div>
|
||||
<div className="row-line2 mono">
|
||||
<span className="row-host">{host}</span>
|
||||
@@ -1650,6 +1776,7 @@ function SubRow({
|
||||
every {interval} · {count} node{count === 1 ? '' : 's'}
|
||||
</span>
|
||||
</div>
|
||||
<RowFindings findings={findings} />
|
||||
</div>
|
||||
<div className="row-actions">
|
||||
<Button
|
||||
@@ -2121,6 +2248,29 @@ function HeaderRows({
|
||||
)
|
||||
}
|
||||
|
||||
/**
|
||||
* What the last apply said about THIS row, under the row it is about.
|
||||
*
|
||||
* Deliberately inside the row rather than in a list at the top of the page: the
|
||||
* failure being fixed is a node that looks fine, and a name in a summary three
|
||||
* screens up does not fix that. The wording is the daemon's own — these messages
|
||||
* already name the entity and say what was done about it ("(skipped)", "only X
|
||||
* is kept"), so paraphrasing them here would only invent a second vocabulary.
|
||||
*/
|
||||
function RowFindings({ findings }: { findings: StatusWarning[] }) {
|
||||
if (findings.length === 0) return null
|
||||
return (
|
||||
<ul className="row-findings" aria-label="Findings from the last apply">
|
||||
{findings.map((f, i) => (
|
||||
<li key={i} className={`row-finding row-finding--${f.severity}`}>
|
||||
<Led variant={f.severity === 'critical' ? 'crit' : 'amber'} />
|
||||
<span className="row-finding-msg">{f.message}</span>
|
||||
</li>
|
||||
))}
|
||||
</ul>
|
||||
)
|
||||
}
|
||||
|
||||
function EmptyPlate({ title, body }: { title: string; body: string }) {
|
||||
return (
|
||||
<div className="empty-plate">
|
||||
|
||||
+154
-45
@@ -4,17 +4,18 @@ import type { LedVariant } from '../components'
|
||||
import { fmtDateTime, fmtDuration } from '../format'
|
||||
import {
|
||||
apply as apiApply,
|
||||
confirm as apiConfirm,
|
||||
rollback as apiRollback,
|
||||
getConfig,
|
||||
getRulesReachability,
|
||||
getStats,
|
||||
ApiError,
|
||||
} from '../api'
|
||||
import type { Model, Stats, Status, StatusWarning } from '../api'
|
||||
import { confirmTimeout } from '../pendingConfirm'
|
||||
import { navigate } from '../router'
|
||||
import type { Route } from '../router'
|
||||
import { attentionFindings } from '../findings'
|
||||
import { protectionState } from '../planeState'
|
||||
import { attentionFindings, truncationNote } from '../findings'
|
||||
import { engineReadout, killSwitchReadout, protectionState } from '../planeState'
|
||||
|
||||
// null-safe length for a Go slice that may arrive as null.
|
||||
const len = (a: unknown[] | null | undefined): number => (a ? a.length : 0)
|
||||
@@ -27,7 +28,8 @@ function short(hash: string): string {
|
||||
return h.length > 12 ? h.slice(0, 12) : h
|
||||
}
|
||||
|
||||
type ControlKind = 'apply' | 'confirm' | 'rollback'
|
||||
// Confirm is no longer one of them — see the note beside the controls row.
|
||||
type ControlKind = 'apply' | 'rollback'
|
||||
|
||||
/**
|
||||
* Live service uptime in seconds, ticking between status polls.
|
||||
@@ -42,17 +44,32 @@ type ControlKind = 'apply' | 'confirm' | 'rollback'
|
||||
* Returns null when the daemon doesn't report uptime (older builds) — the caller
|
||||
* then renders nothing rather than inventing a number.
|
||||
*/
|
||||
function useUptime(status: Status | null): number | null {
|
||||
function useUptime(status: Status | null): { seconds: number; startedUnix: number } | null {
|
||||
const base = useRef<{ uptime: number; at: number } | null>(null)
|
||||
// The instant the daemon came up, ON THE BROWSER'S CLOCK.
|
||||
//
|
||||
// `status.started_unix` is the router's own clock, and the router has no RTC —
|
||||
// it runs on UTC with no tzdata. Rendering it through the browser's timezone
|
||||
// printed a start time three hours in the FUTURE for a Moscow operator, beside
|
||||
// an uptime of "2 h 41 min". Deriving it instead as now-minus-uptime is a
|
||||
// difference of two client timestamps, so it is skew-proof and can never land
|
||||
// ahead of the clock in the header.
|
||||
const started = useRef<number | null>(null)
|
||||
const [, forceTick] = useState(0)
|
||||
|
||||
const reported = status?.uptime_seconds
|
||||
useEffect(() => {
|
||||
if (typeof reported !== 'number' || !Number.isFinite(reported)) {
|
||||
base.current = null
|
||||
started.current = null
|
||||
return
|
||||
}
|
||||
base.current = { uptime: reported, at: Date.now() }
|
||||
const now = Date.now()
|
||||
base.current = { uptime: reported, at: now }
|
||||
// Re-baselining every poll would jitter the displayed second back and forth;
|
||||
// only move it when the estimate has genuinely drifted (a daemon restart).
|
||||
const est = Math.round(now / 1000 - reported)
|
||||
if (started.current === null || Math.abs(started.current - est) > 5) started.current = est
|
||||
forceTick((n) => n + 1)
|
||||
}, [reported])
|
||||
|
||||
@@ -64,8 +81,11 @@ function useUptime(status: Status | null): number | null {
|
||||
return () => window.clearInterval(id)
|
||||
}, [])
|
||||
|
||||
if (!base.current) return null
|
||||
return base.current.uptime + Math.max(0, (Date.now() - base.current.at) / 1000)
|
||||
if (!base.current || started.current === null) return null
|
||||
return {
|
||||
seconds: base.current.uptime + Math.max(0, (Date.now() - base.current.at) / 1000),
|
||||
startedUnix: started.current,
|
||||
}
|
||||
}
|
||||
|
||||
export function Overview({
|
||||
@@ -93,6 +113,29 @@ export function Overview({
|
||||
void loadConfig()
|
||||
}, [loadConfig])
|
||||
|
||||
// ---- how many rules are actually IN FORCE ----------------------------------
|
||||
//
|
||||
// `Rule.Enabled` from /api/config is the DESIRED state; the active WAN profile
|
||||
// overrides it in either direction, and the daemon reports the result as
|
||||
// `effective_enabled`. Counting the saved switches told a router running one
|
||||
// chain that it had "2 / 2" — the Routing page had already been fixed to read
|
||||
// the verdicts, and the home page kept summing the config beside it.
|
||||
//
|
||||
// null ⇒ no verdicts (older daemon, engine stopped, endpoint unreachable). The
|
||||
// module then says so rather than passing the saved count off as the live one.
|
||||
const [inForce, setInForce] = useState<number | null>(null)
|
||||
const loadReach = useCallback(async () => {
|
||||
try {
|
||||
const { rules } = await getRulesReachability()
|
||||
setInForce(rules.filter((r) => r.effective_enabled).length)
|
||||
} catch {
|
||||
setInForce(null)
|
||||
}
|
||||
}, [])
|
||||
useEffect(() => {
|
||||
void loadReach()
|
||||
}, [loadReach])
|
||||
|
||||
// ---- live filter stats: poll the aggregate snapshot, degrade to honest empty states ----
|
||||
const [stats, setStats] = useState<Stats | null>(null)
|
||||
useEffect(() => {
|
||||
@@ -121,7 +164,7 @@ export function Overview({
|
||||
}
|
||||
}, [])
|
||||
|
||||
// ---- apply / confirm / rollback ----
|
||||
// ---- apply / rollback ----
|
||||
const [busy, setBusy] = useState<ControlKind | null>(null)
|
||||
const [result, setResult] = useState<{ ok: boolean; msg: string } | null>(null)
|
||||
const [toast, setToast] = useState<string | null>(null)
|
||||
@@ -139,22 +182,25 @@ export function Overview({
|
||||
setBusy(kind)
|
||||
setResult(null)
|
||||
try {
|
||||
const fn = kind === 'apply' ? apiApply : kind === 'confirm' ? apiConfirm : apiRollback
|
||||
const r = await fn()
|
||||
const r = kind === 'apply' ? await apiApply() : await apiRollback()
|
||||
if (r.error) {
|
||||
setResult({ ok: false, msg: r.error })
|
||||
flash(`${kind} failed`)
|
||||
} else {
|
||||
// An apply that changed something armed an auto-rollback, and saying
|
||||
// "data plane reconciled" while a timer runs is how someone walks away
|
||||
// from a config that then reverts. Name the window when there is one.
|
||||
const window = confirmTimeout()
|
||||
const msg =
|
||||
kind === 'apply'
|
||||
? r.changed
|
||||
? 'Applied — data plane reconciled'
|
||||
? window > 0
|
||||
? `Applied — keep this config within ${window}s or it rolls back`
|
||||
: 'Applied — data plane reconciled'
|
||||
: 'Applied — already up to date'
|
||||
: kind === 'confirm'
|
||||
? 'Confirmed — auto-rollback cancelled'
|
||||
: 'Rolled back to last-good config'
|
||||
: 'Rolled back to last-good config'
|
||||
setResult({ ok: true, msg })
|
||||
flash(kind === 'apply' ? 'Applied' : kind === 'confirm' ? 'Confirmed' : 'Rolled back')
|
||||
flash(kind === 'apply' ? 'Applied' : 'Rolled back')
|
||||
}
|
||||
} catch (e) {
|
||||
const msg = e instanceof Error ? e.message : 'request failed'
|
||||
@@ -163,16 +209,20 @@ export function Overview({
|
||||
} finally {
|
||||
setBusy(null)
|
||||
onStatusChange()
|
||||
if (kind !== 'confirm') void loadConfig()
|
||||
void loadConfig()
|
||||
// An apply or a rollback is exactly what changes which rules are in force.
|
||||
void loadReach()
|
||||
}
|
||||
},
|
||||
[flash, loadConfig, onStatusChange],
|
||||
[flash, loadConfig, loadReach, onStatusChange],
|
||||
)
|
||||
|
||||
// ---- service uptime (PROCESS uptime, not "time since the last apply") ----
|
||||
const uptime = useUptime(status)
|
||||
const uptimeText = uptime === null ? '' : fmtDuration(uptime)
|
||||
const startedAt = status?.started_unix ? fmtDateTime(status.started_unix) : ''
|
||||
const uptimeText = uptime === null ? '' : fmtDuration(uptime.seconds)
|
||||
// On YOUR clock, derived from the uptime — never `status.started_unix`, which is
|
||||
// the router's clock and has no timezone to convert from. See useUptime.
|
||||
const startedAt = uptime === null ? '' : fmtDateTime(uptime.startedUnix)
|
||||
|
||||
// ---- derived display state ----
|
||||
const g = config?.Globals
|
||||
@@ -251,24 +301,27 @@ export function Overview({
|
||||
? `${worstGroup.group} — no answer`
|
||||
: `${worstGroup.group} — ${worstGroup.dead} down`
|
||||
|
||||
const engineVariant: LedVariant = !status
|
||||
? 'off'
|
||||
: status.running && status.active
|
||||
? 'on'
|
||||
: status.running
|
||||
? 'amber'
|
||||
: 'crit'
|
||||
// One reading for the engine, and it is able to say "stopped": `status.running`
|
||||
// was a constant `true` on the daemon, so this LED could never go crit and the
|
||||
// Engine module was green through a process that had failed to start. See
|
||||
// planeState.engineState.
|
||||
const engine = engineReadout(status)
|
||||
const engineVariant: LedVariant = engine.variant
|
||||
|
||||
const protection = protectionState(status)
|
||||
// Configured fail-closed AND actually enforcing it. `none` means nothing is
|
||||
// installed, so the setting is inert no matter what it says.
|
||||
const killInEffect = killArmed && status?.plane !== 'none'
|
||||
// Configured fail-closed, actually enforcing it, or not known — three answers,
|
||||
// and the third is not folded into the first. See planeState.killSwitchReadout.
|
||||
const kill = killSwitchReadout(status, g?.KillSwitch)
|
||||
|
||||
// Findings that need attention. `info` notes are statements about the config,
|
||||
// not problems, so they live beside the setting they describe (see findings.ts)
|
||||
// — keeping this list to things someone could actually act on.
|
||||
const warnings = attentionFindings(status?.warnings)
|
||||
const criticalCount = warnings.filter((w) => w.severity === 'critical').length
|
||||
// The daemon caps the published list at 50 and says so in an `info` note — the
|
||||
// one channel this page filters away. Carried separately so the list can admit
|
||||
// it is not the whole list. See findings.ts truncationNote.
|
||||
const truncated = truncationNote(status?.warnings)
|
||||
|
||||
return (
|
||||
<section className="page" aria-label="Overview">
|
||||
@@ -291,7 +344,7 @@ export function Overview({
|
||||
</p>
|
||||
)}
|
||||
|
||||
<Findings warnings={warnings} criticalCount={criticalCount} />
|
||||
<Findings warnings={warnings} criticalCount={criticalCount} truncated={truncated} />
|
||||
|
||||
<div className="grid">
|
||||
{/* Groups, not nodes: a group is where a dial path is defined, so it is the
|
||||
@@ -334,11 +387,28 @@ export function Overview({
|
||||
/>
|
||||
)}
|
||||
|
||||
{/* "N / M in force", the same reading the Routing page shows — never the
|
||||
count of saved switches. The lamp follows the same rule: a table of
|
||||
rules none of which are in force routes exactly nothing, and it used
|
||||
to sit under a green light saying "0 / 7". */}
|
||||
<Module
|
||||
name="Routing"
|
||||
value={String(enabledCount(config?.Rules))}
|
||||
unit={`/ ${len(config?.Rules)} rules`}
|
||||
led={{ variant: len(config?.Rules) ? 'on' : 'amber' }}
|
||||
value={inForce === null ? String(enabledCount(config?.Rules)) : String(inForce)}
|
||||
unit={
|
||||
inForce === null
|
||||
? `/ ${len(config?.Rules)} rules saved`
|
||||
: `/ ${len(config?.Rules)} in force`
|
||||
}
|
||||
led={{
|
||||
variant:
|
||||
len(config?.Rules) === 0
|
||||
? 'amber'
|
||||
: inForce === null
|
||||
? 'off'
|
||||
: inForce === 0
|
||||
? 'amber'
|
||||
: 'on',
|
||||
}}
|
||||
rows={[
|
||||
{ k: 'egresses', v: String(len(config?.Egresses)) },
|
||||
{ k: 'default', v: defaultTarget(status, config), hot: true },
|
||||
@@ -381,17 +451,17 @@ export function Overview({
|
||||
/>
|
||||
|
||||
{/* A kill-switch set to fail-closed is only ARMED if something is actually
|
||||
installed to enforce it. With no plane it is configured but inert, and
|
||||
saying "ARMED" there would be a false reassurance next to a readout
|
||||
that says nothing is protected. */}
|
||||
installed to enforce it, and "we haven't been told" is neither. With no
|
||||
plane it is configured but inert; with no reading the lamp stays unlit
|
||||
rather than joining the healthy branch by default. */}
|
||||
<Module
|
||||
name="Kill-switch"
|
||||
value={killInEffect ? 'ARMED' : killArmed ? 'NOT IN EFFECT' : 'OPEN'}
|
||||
led={{ variant: killInEffect ? 'on' : killArmed ? 'crit' : 'amber' }}
|
||||
value={kill.value}
|
||||
led={{ variant: kill.variant }}
|
||||
rows={[
|
||||
{ k: 'setting', v: killArmed ? 'fail-closed' : 'fail-open', hot: !killArmed },
|
||||
...(killArmed && !killInEffect
|
||||
? [{ k: 'blocking now', v: 'no — nothing installed', hot: true }]
|
||||
...(kill.blockingNow
|
||||
? [{ k: 'blocking now', v: kill.blockingNow, hot: kill.hot }]
|
||||
: [{ k: 'ipv6', v: g?.IPv6 ? 'covered' : 'off' }]),
|
||||
{ k: 'confirm', v: g?.ConfirmTimeout ? `${g.ConfirmTimeout}s window` : 'no auto-rollback' },
|
||||
]}
|
||||
@@ -414,6 +484,7 @@ export function Overview({
|
||||
unit={status?.version?.includes('-') ? '· ' + status.version.split('-').slice(1).join('-') : ''}
|
||||
led={{ variant: engineVariant }}
|
||||
rows={[
|
||||
{ k: 'process', v: engine.word, hot: engineVariant === 'crit' },
|
||||
{ k: 'config hash', v: <span className="mono">{short(status?.hash ?? '')}</span> },
|
||||
// Uptime of the daemon PROCESS. "started" is the moment it came up,
|
||||
// by the router's clock — not the moment a config was applied.
|
||||
@@ -431,9 +502,14 @@ export function Overview({
|
||||
<Button variant="primary" onClick={() => void run('apply')} disabled={busy !== null}>
|
||||
{busy === 'apply' ? 'Applying…' : 'Apply config'}
|
||||
</Button>
|
||||
<Button onClick={() => void run('confirm')} disabled={busy !== null}>
|
||||
{busy === 'confirm' ? 'Confirming…' : 'Confirm'}
|
||||
</Button>
|
||||
{/* A "Confirm" button used to sit here permanently, and pressing it
|
||||
always printed "Confirmed — auto-rollback cancelled": `apply.Confirm()`
|
||||
returns nil whether or not a window was ever armed, so the message was
|
||||
a success report for an event that usually had not happened.
|
||||
Keeping a config is now offered only while a window is actually open,
|
||||
and that is announced by the app-wide band directly above this page —
|
||||
which is where the button lives, beside the countdown it belongs to,
|
||||
rather than duplicated here. */}
|
||||
{canRollback && (
|
||||
<Button onClick={() => void run('rollback')} disabled={busy !== null}>
|
||||
{busy === 'rollback' ? 'Rolling back…' : 'Rollback'}
|
||||
@@ -464,11 +540,23 @@ const SECTION_ROUTE: Record<string, Route> = {
|
||||
rule: 'routing',
|
||||
ruleset: 'routing',
|
||||
blocklist: 'dns',
|
||||
allowlist: 'dns',
|
||||
resolver: 'dns',
|
||||
dns_rule: 'dns',
|
||||
device: 'devices',
|
||||
chain: 'targets',
|
||||
group: 'targets',
|
||||
// A node the generator dropped (unparseable share link, duplicate WireGuard
|
||||
// key, name colliding with a reserved tag) is reported under `node` — and had
|
||||
// nowhere to jump to, so the one page that could show it a green toggle was
|
||||
// also the one page the finding could not reach.
|
||||
node: 'nodes',
|
||||
subscription: 'nodes',
|
||||
egress: 'targets',
|
||||
inbound: 'networks',
|
||||
interface: 'networks',
|
||||
profile: 'profiles',
|
||||
alert: 'settings',
|
||||
// The standing note about non-TCP/UDP traffic — its control lives on Networks.
|
||||
untunnelable: 'networks',
|
||||
}
|
||||
@@ -486,11 +574,14 @@ const SECTION_ROUTE: Record<string, Route> = {
|
||||
function Findings({
|
||||
warnings,
|
||||
criticalCount,
|
||||
truncated,
|
||||
}: {
|
||||
warnings: StatusWarning[]
|
||||
criticalCount: number
|
||||
/** The daemon's "N further suppressed" note, when the list was capped. */
|
||||
truncated: StatusWarning | null
|
||||
}) {
|
||||
if (warnings.length === 0) return null
|
||||
if (warnings.length === 0 && !truncated) return null
|
||||
|
||||
const rank = { critical: 0, warning: 1, info: 2 } as const
|
||||
const sorted = [...warnings].sort((a, b) => rank[a.severity] - rank[b.severity])
|
||||
@@ -500,6 +591,10 @@ function Findings({
|
||||
<header className="findings-hd">
|
||||
<h2 className="findings-title">Last apply</h2>
|
||||
<span className="findings-count mono">
|
||||
{/* "at least" whenever the list was capped: the counts below it are a
|
||||
floor, not a total, and the cap drops the least severe FIRST — so
|
||||
on a config with fifty criticals the thing it drops is a critical. */}
|
||||
{truncated ? 'at least ' : ''}
|
||||
{criticalCount > 0
|
||||
? `${criticalCount} critical · ${warnings.length} total`
|
||||
: `${warnings.length} note${warnings.length === 1 ? '' : 's'}`}
|
||||
@@ -538,6 +633,20 @@ function Findings({
|
||||
</li>
|
||||
)
|
||||
})}
|
||||
{/* The list saying it is not the whole list. Last, because it is about
|
||||
everything above it — and never filtered out with the other `info`
|
||||
notes, which is where it used to disappear. */}
|
||||
{truncated && (
|
||||
<li className="finding finding--truncated">
|
||||
<Led variant="amber" />
|
||||
<div className="finding-copy">
|
||||
<span className="finding-where mono">list truncated</span>
|
||||
<span className="finding-msg">
|
||||
Some findings are missing from this list. {truncated.message}
|
||||
</span>
|
||||
</div>
|
||||
</li>
|
||||
)}
|
||||
</ul>
|
||||
</section>
|
||||
)
|
||||
|
||||
+162
-46
@@ -12,6 +12,7 @@ import {
|
||||
ApiError,
|
||||
} from '../api'
|
||||
import type { Model, Rule, RuleReach, Ruleset, RulesetStatus } from '../api'
|
||||
import { everyLabel, relFetch } from '../format'
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// The api.ts `Rule` is a deliberately thin subset (Name/Enabled/Order/Target/
|
||||
@@ -59,16 +60,31 @@ type RuleForce = {
|
||||
}
|
||||
|
||||
/**
|
||||
* Everything `Proto` can match, and nothing else. The engine understands two
|
||||
* transports and exactly ten application protocols its sniffers can name
|
||||
* (generate/route.go sniffedProtocols); a value outside this set builds a rule
|
||||
* that is perfectly valid and can never fire — so its traffic quietly falls
|
||||
* Everything `Proto` can match, and nothing else. A value outside this set builds
|
||||
* a rule that is perfectly valid and can never fire — so its traffic quietly falls
|
||||
* through to whatever rule sits below it. That is why this is a closed list and
|
||||
* not a text box.
|
||||
*
|
||||
* Split into two groups because they answer different questions: the transport is
|
||||
* known the moment a packet arrives, while an app protocol is only known once the
|
||||
* first bytes have been read and labelled.
|
||||
* Three groups, because the engine reads them through three different matchers
|
||||
* (generate/route.go, ruleMatchers) and they answer different questions:
|
||||
*
|
||||
* - Transport — the L4 network. Known the moment a packet arrives.
|
||||
* - Detected protocol — the L7 label a sniffer puts on a connection once its
|
||||
* first bytes have been read. This group, and ONLY this group, is the engine's
|
||||
* `sniffedProtocols` set; anything else routed into that matcher is inert.
|
||||
* - Layer 3 — ICMP. Not a sniffed label: it lands in the emitted rule's
|
||||
* `network`, never in `protocol` (the sniffers are skipped outright for an
|
||||
* ICMP flow, so they never report "icmp"). All three spellings are the SAME
|
||||
* one network; `icmpv4`/`icmpv6` additionally pin `ip_version`, which the
|
||||
* engine derives from the destination address.
|
||||
*
|
||||
* ICMP carries caveats the picker deliberately does not try to enforce, because
|
||||
* the daemon reports each one against the whole config on apply: it reaches the
|
||||
* engine only while globals l3_tunnel is on, it has no ports (a port matcher
|
||||
* beside it can never be satisfied), `icmpv6` also needs globals ipv6 on, and it
|
||||
* is DROPPED rather than falling through when routed at a target that cannot
|
||||
* carry layer 3 — i.e. every proxy protocol. Only wireguard/AmneziaWG nodes and
|
||||
* direct/interface egresses can carry a ping.
|
||||
*/
|
||||
const PROTO_TRANSPORT: ReadonlyArray<{ id: string; label: string }> = [
|
||||
{ id: 'tcp', label: 'TCP' },
|
||||
@@ -86,10 +102,27 @@ const PROTO_APP: ReadonlyArray<{ id: string; label: string }> = [
|
||||
{ id: 'rdp', label: 'RDP' },
|
||||
{ id: 'ntp', label: 'NTP' },
|
||||
]
|
||||
const PROTO_VALUES = new Set([...PROTO_TRANSPORT, ...PROTO_APP].map((p) => p.id))
|
||||
/**
|
||||
* The family-qualified spellings are offered next to plain `icmp` rather than
|
||||
* hidden behind it: the engine treats them as first-class and the difference is
|
||||
* observable (an `ip_version` item on the same rule), so hiding them would leave a
|
||||
* capability reachable only by hand-editing /etc/config/shater — and would mean
|
||||
* that anyone who edited such a rule here lost the narrowing on the next save.
|
||||
*/
|
||||
const PROTO_L3: ReadonlyArray<{ id: string; label: string }> = [
|
||||
{ id: 'icmp', label: 'ICMP (ping)' },
|
||||
{ id: 'icmpv4', label: 'ICMP — IPv4 only' },
|
||||
{ id: 'icmpv6', label: 'ICMP — IPv6 only' },
|
||||
]
|
||||
const PROTO_VALUES = new Set(
|
||||
[...PROTO_TRANSPORT, ...PROTO_APP, ...PROTO_L3].map((p) => p.id),
|
||||
)
|
||||
|
||||
/** The Proto picker's option list — shared by the inline add row and the editor. */
|
||||
function ProtoOptions({ value }: { value: string }) {
|
||||
// The engine lower-cases `Proto` before matching it, so a hand-written `ICMP`
|
||||
// is a working rule; judge it the same way and flag only what really is inert.
|
||||
const matches = PROTO_VALUES.has(value.trim().toLowerCase())
|
||||
return (
|
||||
<>
|
||||
<option value="">any</option>
|
||||
@@ -107,10 +140,18 @@ function ProtoOptions({ value }: { value: string }) {
|
||||
</option>
|
||||
))}
|
||||
</optgroup>
|
||||
{/* A stored value the engine can't detect is kept and flagged, never
|
||||
silently rewritten — the rule it belongs to is live right now. */}
|
||||
<optgroup label="Layer 3">
|
||||
{PROTO_L3.map((p) => (
|
||||
<option key={p.id} value={p.id}>
|
||||
{p.label}
|
||||
</option>
|
||||
))}
|
||||
</optgroup>
|
||||
{/* A stored value none of the groups spells verbatim is kept and offered as
|
||||
written, never silently rewritten — the rule it belongs to is live right
|
||||
now. It is flagged only when the engine cannot match it either. */}
|
||||
{value !== '' && !PROTO_VALUES.has(value) && (
|
||||
<option value={value}>{value} — never matches</option>
|
||||
<option value={value}>{matches ? value : `${value} — never matches`}</option>
|
||||
)}
|
||||
</>
|
||||
)
|
||||
@@ -203,35 +244,8 @@ function errMsg(e: unknown): string {
|
||||
// --- remote-list freshness (feedback #9) ------------------------------------
|
||||
// A url-source ruleset re-fetches on a cadence; the engine reports when it last
|
||||
// pulled and how many rules the list holds. Match a ruleset to its status by the
|
||||
// engine tag `rs-<name>`.
|
||||
|
||||
/** "updated 3h ago" / "never updated" for a remote list's last fetch. */
|
||||
function relFetch(iso: string): string {
|
||||
if (!iso) return 'never updated'
|
||||
const t = Date.parse(iso)
|
||||
if (Number.isNaN(t)) return 'never updated'
|
||||
const s = Math.max(0, Math.floor((Date.now() - t) / 1000))
|
||||
if (s < 45) return 'updated just now'
|
||||
const m = Math.floor(s / 60)
|
||||
if (m < 60) return `updated ${m}m ago`
|
||||
const h = Math.floor(m / 60)
|
||||
if (h < 24) return `updated ${h}h ago`
|
||||
const d = Math.floor(h / 24)
|
||||
return `updated ${d}d ago`
|
||||
}
|
||||
|
||||
/** "every 24h" for an auto-update cadence in seconds ("" when there is none). */
|
||||
function everyLabel(sec: number): string {
|
||||
if (!sec || sec <= 0) return ''
|
||||
if (sec % 3600 === 0) {
|
||||
const h = sec / 3600
|
||||
if (h < 48) return `every ${h}h`
|
||||
if (sec % 86400 === 0) return `every ${sec / 86400}d`
|
||||
return `every ${h}h`
|
||||
}
|
||||
if (sec % 60 === 0) return `every ${sec / 60}m`
|
||||
return `every ${sec}s`
|
||||
}
|
||||
// engine tag `rs-<name>`. The two readings (relFetch / everyLabel) live in
|
||||
// format.ts because the DNS page shows the same ones for blocklists.
|
||||
|
||||
/** A rule with no matcher of any kind is the effective catch-all (route Final).
|
||||
* Mirrors model.IsCatchAll on the daemon side — the two must agree or the
|
||||
@@ -286,6 +300,57 @@ function formHasNoMatchers(f: {
|
||||
)
|
||||
}
|
||||
|
||||
/**
|
||||
* What happens to the network when the default route stops being emitted —
|
||||
* whether it is deleted or merely switched off (the engine emits neither).
|
||||
*
|
||||
* The old text was one sentence for every rule: "Traffic it matched will fall
|
||||
* through to the next rule." For an ordinary rule that is true. For the catch-all
|
||||
* there IS no next rule, and what happens instead is decided by the kill-switch:
|
||||
* generate/route.go sets `final := tagBlock` and only `kill_switch=open` swaps
|
||||
* that for `direct`. So removing the default either takes the whole network
|
||||
* offline or puts the whole network on the naked WAN — and the page said "falls
|
||||
* through to the next rule" for both, on a row it had already badged
|
||||
* "DEFAULT ROUTE · FINAL".
|
||||
*
|
||||
* `successor` is the rule that would inherit route.Final instead (a config can
|
||||
* carry more than one conditionless rule; the last one wins). When there is one,
|
||||
* nothing is lost — the honest warning is that the destination changes.
|
||||
*/
|
||||
function defaultRouteConsequence(
|
||||
killSwitch: string,
|
||||
successor: { name: string; order: number; target: string } | null,
|
||||
): ReactNode {
|
||||
if (successor) {
|
||||
return (
|
||||
<>
|
||||
This is the router’s <strong>default route</strong> — everything no other rule matches
|
||||
follows it. Remove it and “{successor.name}” (order {successor.order}) has no conditions
|
||||
either, so it takes over: unmatched traffic goes to{' '}
|
||||
<strong className="mono">{successor.target}</strong> instead.
|
||||
</>
|
||||
)
|
||||
}
|
||||
if (killSwitch === 'open') {
|
||||
return (
|
||||
<>
|
||||
This is the router’s <strong>default route</strong> — everything no other rule matches
|
||||
follows it, and no other rule matches everything. With the kill-switch set to{' '}
|
||||
<strong>fail-open</strong>, unmatched traffic then leaves through your normal internet
|
||||
connection with your real address — unproxied and unfiltered.
|
||||
</>
|
||||
)
|
||||
}
|
||||
return (
|
||||
<>
|
||||
This is the router’s <strong>default route</strong> — everything no other rule matches
|
||||
follows it, and no other rule matches everything. With the kill-switch set to{' '}
|
||||
<strong>fail-closed</strong>, unmatched traffic is then <strong>blocked</strong>: devices on
|
||||
your network lose the internet until you add a default back.
|
||||
</>
|
||||
)
|
||||
}
|
||||
|
||||
/** Effective routing target for a rule (Target wins; a bare Egress is a target too). */
|
||||
function effectiveTarget(r: RRule): string {
|
||||
if (r.Target && r.Target.trim()) return r.Target.trim()
|
||||
@@ -678,17 +743,61 @@ export default function Routing() {
|
||||
[config, rules, rulesets, rulesetUsage, persist, confirm],
|
||||
)
|
||||
|
||||
/**
|
||||
* Is this rule the one the ENGINE uses as route.Final right now, and in force?
|
||||
*
|
||||
* Both halves matter. A conditionless rule the daemon reports as shadowed owns
|
||||
* nothing (the row already says "never applies"), and one that is switched off
|
||||
* is not being emitted either — removing either changes no traffic, so neither
|
||||
* earns a warning.
|
||||
*/
|
||||
const isLiveDefault = useCallback(
|
||||
(r: RRule): boolean => isCatchAll(r) && shadowOf(r) === null && forceOf(r).on,
|
||||
[shadowOf, forceOf],
|
||||
)
|
||||
|
||||
/** Which rule would inherit route.Final if `name` stopped being emitted: the
|
||||
* LAST remaining conditionless, switched-on rule. null when there is none. */
|
||||
const successorDefault = useCallback(
|
||||
(name: string): { name: string; order: number; target: string } | null => {
|
||||
for (let i = rules.length - 1; i >= 0; i--) {
|
||||
const r = rules[i]
|
||||
if (r.Name === name) continue
|
||||
if (r.Enabled && isCatchAll(r)) {
|
||||
return { name: r.Name, order: r.Order, target: effectiveTarget(r) }
|
||||
}
|
||||
}
|
||||
return null
|
||||
},
|
||||
[rules],
|
||||
)
|
||||
|
||||
const killSwitch = (config?.Globals?.KillSwitch ?? 'closed') === 'open' ? 'open' : 'closed'
|
||||
|
||||
const onToggle = useCallback(
|
||||
(name: string) => {
|
||||
async (name: string) => {
|
||||
const target = rules.find((r) => r.Name === name)
|
||||
if (!target) return
|
||||
const nextState = !target.Enabled
|
||||
// Switching the default route OFF is the same event as deleting it — the
|
||||
// generator emits only enabled rules — so it asks the same question. It used
|
||||
// to ask nothing at all, which made the least reversible control on the page
|
||||
// the only one with no confirmation.
|
||||
if (!nextState && isLiveDefault(target)) {
|
||||
const ok = await confirm({
|
||||
label: 'Turn off default route',
|
||||
title: `Turn off the default route “${name}”?`,
|
||||
body: defaultRouteConsequence(killSwitch, successorDefault(name)),
|
||||
confirmLabel: 'Turn it off',
|
||||
})
|
||||
if (!ok) return
|
||||
}
|
||||
commitRules(
|
||||
rules.map((r) => (r.Name === name ? { ...r, Enabled: nextState } : r)),
|
||||
`${name} ${nextState ? 'enabled' : 'disabled'}`,
|
||||
)
|
||||
},
|
||||
[rules, commitRules],
|
||||
[rules, commitRules, confirm, isLiveDefault, killSwitch, successorDefault],
|
||||
)
|
||||
|
||||
const onMove = useCallback(
|
||||
@@ -716,10 +825,17 @@ export default function Routing() {
|
||||
|
||||
const onDelete = useCallback(
|
||||
async (name: string) => {
|
||||
const target = rules.find((r) => r.Name === name)
|
||||
if (!target) return
|
||||
// The catch-all has no "next rule" to fall through to — see
|
||||
// defaultRouteConsequence. Every other rule keeps the plain sentence.
|
||||
const isDefault = isLiveDefault(target)
|
||||
const ok = await confirm({
|
||||
label: 'Delete rule',
|
||||
title: `Delete rule "${name}"?`,
|
||||
body: 'Traffic it matched will fall through to the next rule.',
|
||||
label: isDefault ? 'Delete default route' : 'Delete rule',
|
||||
title: isDefault ? `Delete the default route “${name}”?` : `Delete rule “${name}”?`,
|
||||
body: isDefault
|
||||
? defaultRouteConsequence(killSwitch, successorDefault(name))
|
||||
: 'Traffic it matched will fall through to the next rule.',
|
||||
})
|
||||
if (!ok) return
|
||||
commitRules(
|
||||
@@ -727,7 +843,7 @@ export default function Routing() {
|
||||
`${name} deleted`,
|
||||
)
|
||||
},
|
||||
[rules, commitRules, confirm],
|
||||
[rules, commitRules, confirm, isLiveDefault, killSwitch, successorDefault],
|
||||
)
|
||||
|
||||
// Insert a new rule just above the catch-all (so a specific rule can actually match).
|
||||
|
||||
@@ -1,7 +1,8 @@
|
||||
import './Settings.css'
|
||||
import { useCallback, useEffect, useRef, useState } from 'react'
|
||||
import type { ReactNode } from 'react'
|
||||
import { Button, Led, Select, Toggle } from '../components'
|
||||
import { Button, Led, Select, Toggle, useConfirm } from '../components'
|
||||
import { AlertsSection } from './Alerts'
|
||||
import { apply as apiApply, downloadLog, getConfig, putConfig, ApiError } from '../api'
|
||||
import type { Globals, LogRange, Model } from '../api'
|
||||
|
||||
@@ -125,6 +126,7 @@ const STATS_BACKENDS: ReadonlyArray<{ value: string; label: string }> = [
|
||||
// ---- page ------------------------------------------------------------------
|
||||
|
||||
export default function Settings() {
|
||||
const confirm = useConfirm()
|
||||
const [config, setConfig] = useState<Model | null>(null)
|
||||
const [loadError, setLoadError] = useState<string | null>(null)
|
||||
|
||||
@@ -257,6 +259,48 @@ export default function Settings() {
|
||||
const groupHealthOn = globals?.GroupHealth !== false
|
||||
|
||||
const killSwitch = globals?.KillSwitch === 'open' ? 'open' : 'closed'
|
||||
|
||||
/**
|
||||
* The master switch, which is the most destructive control in the panel and was
|
||||
* the only one that asked nothing.
|
||||
*
|
||||
* Turning it off is not "pausing the proxy": apply.go runs Teardown() — the nft
|
||||
* table goes, the policy routing goes, `plane` becomes `none`. The kill-switch
|
||||
* does not save you, because a kill-switch is a rule in a table that no longer
|
||||
* exists. Everything on the LAN then leaves through the plain WAN, unproxied and
|
||||
* unfiltered. Deleting a rule-set asked for confirmation; this did not.
|
||||
*
|
||||
* Turning it back ON is not destructive and is not gated.
|
||||
*/
|
||||
const toggleService = useCallback(
|
||||
async (on: boolean) => {
|
||||
if (!on) {
|
||||
const ok = await confirm({
|
||||
label: 'Turn off the service',
|
||||
title: 'Turn the proxy engine off?',
|
||||
body: (
|
||||
<>
|
||||
This tears the whole data plane down — the firewall table, the policy routing and the
|
||||
DNS interception are removed, not paused. Nothing is proxied, filtered or blocked, and
|
||||
every device leaves through your normal internet connection with its real address.{' '}
|
||||
{killSwitch === 'closed' ? (
|
||||
<>
|
||||
The kill-switch does not hold here: with nothing installed there is nothing left
|
||||
to block with.
|
||||
</>
|
||||
) : (
|
||||
<>The kill-switch is already open, so nothing changes about that.</>
|
||||
)}
|
||||
</>
|
||||
),
|
||||
confirmLabel: 'Turn it off',
|
||||
})
|
||||
if (!ok) return
|
||||
}
|
||||
setGlobal('Enabled', on, on ? 'Engine enabled' : 'Engine disabled')
|
||||
},
|
||||
[confirm, killSwitch, setGlobal],
|
||||
)
|
||||
const killNote =
|
||||
killSwitch === 'open'
|
||||
? 'Fail-open — if the engine stops, traffic falls back to the direct WAN. Stays online, but unprotected.'
|
||||
@@ -296,10 +340,13 @@ export default function Settings() {
|
||||
<div className="set-groups">
|
||||
{/* ---- SERVICE ---- */}
|
||||
<Group title="Service" count={globals?.Enabled ? 'enabled' : 'disabled'}>
|
||||
<Field label="Proxy engine" note="Master on/off for the whole appliance.">
|
||||
<Field
|
||||
label="Proxy engine"
|
||||
note="Master on/off for the whole appliance. Off removes the firewall table and the policy routing — every device goes out directly, with no kill-switch to catch it."
|
||||
>
|
||||
<Toggle
|
||||
pressed={globals?.Enabled ?? false}
|
||||
onChange={(on) => setGlobal('Enabled', on, on ? 'Engine enabled' : 'Engine disabled')}
|
||||
onChange={(on) => void toggleService(on)}
|
||||
label={globals?.Enabled ? 'Disable proxy engine' : 'Enable proxy engine'}
|
||||
size="md"
|
||||
disabled={busy || !ready}
|
||||
@@ -360,7 +407,7 @@ export default function Settings() {
|
||||
|
||||
<Field
|
||||
label="Log level"
|
||||
note="Verbosity of the daemon log. “none” silences the engine and drops the control-plane to panic-only — a turn-down, not a true off: even warnings and errors are hidden. The toggles below decide where whatever is emitted gets written; turning both off is the only full silence. Failures still raise alerts regardless of this level."
|
||||
note="Verbosity of the daemon log. “none” silences the engine and drops the control-plane to panic-only — a turn-down, not a true off: even warnings and errors are hidden. The toggles below decide where whatever is emitted gets written; turning both off is the only full silence. Failures still raise alerts regardless of this level — set up where they go in the Alerts section below."
|
||||
>
|
||||
<Select
|
||||
value={globals?.LogLevel || 'warning'}
|
||||
@@ -574,6 +621,13 @@ export default function Settings() {
|
||||
</Field>
|
||||
</Group>
|
||||
|
||||
{/* ---- ALERTS ---- */}
|
||||
{/* Extracted from the DNS page — out-of-band notifications belong with
|
||||
the appliance-wide knobs, next to the log level whose note points
|
||||
here. Renders its own section header (same plate as a Group); all
|
||||
writes go through `save`, so the dirty banner and toast stay one. */}
|
||||
<AlertsSection config={config} busy={busy} loading={loading} onSave={save} />
|
||||
|
||||
{/* ---- STATISTICS & LOGGING ---- */}
|
||||
<Group
|
||||
title="Statistics & logging"
|
||||
|
||||
@@ -704,6 +704,13 @@
|
||||
.tg-test--bad .tg-test-msg {
|
||||
color: var(--crit);
|
||||
}
|
||||
/* "Nothing measured this" is not a failure and must never be dressed as one: an
|
||||
unlit lamp and the faintest text on the card, the same register the group
|
||||
readout uses for its unmeasured state. */
|
||||
.tg-test--none .tg-test-msg {
|
||||
font-family: var(--font-sans);
|
||||
color: var(--faint);
|
||||
}
|
||||
.tg-test--wait .tg-test-msg {
|
||||
color: var(--amber);
|
||||
}
|
||||
@@ -1038,6 +1045,235 @@
|
||||
}
|
||||
|
||||
/* ---- responsive ---- */
|
||||
/* ---- chain hop rail ----
|
||||
* The chain section's signature, and the one place this card spends any
|
||||
* boldness: the path is drawn as a CONDUCTOR with a numbered lamp at each hop,
|
||||
* and the conductor is SEVERED below the first hop that was probed and did not
|
||||
* answer. A chain is a single series path, so the question is never "how many
|
||||
* hops are green", it is "where does my traffic stop" — and a broken line answers
|
||||
* that before a word has been read.
|
||||
*
|
||||
* The two marks carry two different facts and must not be conflated:
|
||||
* - the LAMP is that hop's own measurement (good / warn / crit / unlit). Below
|
||||
* the break there is no measurement to draw: the daemon stops walking at the
|
||||
* first dead hop, so those lamps are UNLIT and the row says which hop stopped
|
||||
* the walk. Unlit is never a shade of red — it claims nothing, which is the
|
||||
* truth about a hop nobody dialled;
|
||||
* - the CONDUCTOR is reachability through the path, which really does stop.
|
||||
*
|
||||
* Orange is untouched here. Semantics carry every colour, and everything that is
|
||||
* not a lamp is groove-grey. No transitions and no animation anywhere in the
|
||||
* rail, so there is nothing for reduced-motion to switch off. */
|
||||
.ch-rail {
|
||||
gap: 8px;
|
||||
}
|
||||
.ch-eyebrow {
|
||||
font-family: var(--font-mono);
|
||||
font-size: 9px;
|
||||
letter-spacing: var(--track-label);
|
||||
text-transform: uppercase;
|
||||
color: var(--faint);
|
||||
}
|
||||
.ch-hops {
|
||||
--ch-num: 1.8ch; /* the engraved hop number's gutter */
|
||||
--ch-gap: 8px;
|
||||
--ch-led: 10px; /* must match .led's width */
|
||||
--ch-lampy: 14px; /* row top → lamp centre; the conductor's anchor */
|
||||
/* x of the conductor: the number gutter, one gap, then the lamp's centre */
|
||||
--ch-spine: calc(var(--ch-num) + var(--ch-gap) + var(--ch-led) / 2);
|
||||
list-style: none;
|
||||
margin: 0;
|
||||
padding: 0;
|
||||
}
|
||||
.ch-hop {
|
||||
position: relative;
|
||||
display: grid;
|
||||
grid-template-columns: var(--ch-num) var(--ch-led) minmax(0, 1fr);
|
||||
column-gap: var(--ch-gap);
|
||||
align-items: start;
|
||||
}
|
||||
|
||||
/* the conductor — two halves per row, so a break lands on one link only */
|
||||
.ch-hop::before,
|
||||
.ch-hop::after {
|
||||
content: '';
|
||||
position: absolute;
|
||||
left: var(--ch-spine);
|
||||
width: 2px;
|
||||
margin-left: -1px;
|
||||
/* Brighter than a plain groove: this line IS the readout, and at groove
|
||||
strength it disappeared into the panel and took the whole idea with it. */
|
||||
background: color-mix(in srgb, var(--dim) 55%, var(--groove));
|
||||
}
|
||||
.ch-hop::before {
|
||||
top: 0;
|
||||
height: calc(var(--ch-lampy) - var(--ch-led) / 2 - 3px);
|
||||
}
|
||||
.ch-hop::after {
|
||||
top: calc(var(--ch-lampy) + var(--ch-led) / 2 + 3px);
|
||||
bottom: 0;
|
||||
}
|
||||
/* Nothing feeds hop 1 from above, and nothing leaves the exit downward — the
|
||||
path starts and ends inside this rail. */
|
||||
.ch-hop--first::before {
|
||||
display: none;
|
||||
}
|
||||
.ch-hop--exit::after {
|
||||
bottom: auto;
|
||||
height: 9px;
|
||||
}
|
||||
/* …the exit ends on a crossbar instead of trailing off: end of line. */
|
||||
.ch-hop--exit .ch-socket {
|
||||
position: relative;
|
||||
}
|
||||
.ch-hop--exit .ch-socket::after {
|
||||
content: '';
|
||||
position: absolute;
|
||||
left: 50%;
|
||||
transform: translateX(-50%);
|
||||
top: calc(var(--ch-lampy) + var(--ch-led) / 2 + 12px);
|
||||
width: 11px;
|
||||
height: 2px;
|
||||
background: color-mix(in srgb, var(--dim) 55%, var(--groove));
|
||||
}
|
||||
|
||||
/* THE SEVER. Everything from the dead hop's outgoing link downward is drawn as a
|
||||
broken conductor: unmistakably not-a-line at a glance, and unmistakably not a
|
||||
colour, because a colour here would compete with the lamps that carry health. */
|
||||
.ch-hop--dead::after,
|
||||
.ch-hop--severed::before,
|
||||
.ch-hop--severed::after {
|
||||
background: repeating-linear-gradient(
|
||||
to bottom,
|
||||
color-mix(in srgb, var(--dim) 45%, var(--groove)) 0 3px,
|
||||
transparent 3px 7px
|
||||
);
|
||||
}
|
||||
|
||||
.ch-num {
|
||||
font-size: 10px;
|
||||
line-height: calc(var(--ch-lampy) * 2);
|
||||
text-align: right;
|
||||
color: var(--faint);
|
||||
}
|
||||
.ch-socket {
|
||||
display: flex;
|
||||
align-items: center;
|
||||
height: calc(var(--ch-lampy) * 2);
|
||||
}
|
||||
.ch-body {
|
||||
min-width: 0;
|
||||
/* Separates one hop from the next. The conductor runs through this space, so
|
||||
too little of it and two hops read as one wrapped row. */
|
||||
padding-bottom: 8px;
|
||||
}
|
||||
.ch-l1 {
|
||||
display: flex;
|
||||
align-items: center;
|
||||
flex-wrap: wrap;
|
||||
gap: 8px;
|
||||
min-height: calc(var(--ch-lampy) * 2);
|
||||
}
|
||||
.ch-name {
|
||||
font-size: 12px;
|
||||
color: var(--ink);
|
||||
overflow-wrap: anywhere;
|
||||
}
|
||||
.ch-hop--untested .ch-name {
|
||||
color: var(--dim);
|
||||
}
|
||||
/* The exit marker is NEUTRAL on purpose. The config path above this rail tags its
|
||||
exit green, which is free there — but in here green means "answering", and a
|
||||
green badge on the last hop would read as a health claim about it. */
|
||||
.ch-tag {
|
||||
font-family: var(--font-mono);
|
||||
font-size: 8.5px;
|
||||
letter-spacing: 0.14em;
|
||||
text-transform: uppercase;
|
||||
color: var(--faint);
|
||||
}
|
||||
.ch-delay {
|
||||
font-size: 11.5px;
|
||||
font-weight: 700;
|
||||
color: var(--ink);
|
||||
}
|
||||
.ch-quiet {
|
||||
font-family: var(--font-sans);
|
||||
font-size: 12px;
|
||||
color: var(--faint);
|
||||
}
|
||||
/* A blocked hop's phrase carries a tooltip with the blocking hop's engine
|
||||
outbound, so it takes the same help cursor as .ch-dead. No colour of its own:
|
||||
the finding is red once, on the hop that actually failed. */
|
||||
.ch-blocked {
|
||||
cursor: help;
|
||||
}
|
||||
.ch-age {
|
||||
margin-left: auto;
|
||||
font-size: 10.5px;
|
||||
color: var(--faint);
|
||||
white-space: nowrap;
|
||||
}
|
||||
|
||||
/* The counters read exactly as they do on a group card — alive out of TESTED,
|
||||
with the untested remainder as a quiet aside only when there is one. Same
|
||||
register, same weights, deliberately not a second dialect. */
|
||||
.ch-l2 {
|
||||
display: flex;
|
||||
align-items: center;
|
||||
flex-wrap: wrap;
|
||||
gap: 8px;
|
||||
margin-top: 1px;
|
||||
font-size: 11.5px;
|
||||
letter-spacing: 0.02em;
|
||||
color: var(--dim);
|
||||
}
|
||||
.ch-count {
|
||||
font-size: 12px;
|
||||
color: var(--dim);
|
||||
white-space: nowrap;
|
||||
}
|
||||
.ch-count b {
|
||||
font-size: 14px;
|
||||
font-weight: 700;
|
||||
color: var(--ink);
|
||||
}
|
||||
.ch-hop--dead .ch-count b {
|
||||
color: var(--crit);
|
||||
}
|
||||
.ch-word {
|
||||
font-size: 10.5px;
|
||||
letter-spacing: 0.12em;
|
||||
text-transform: uppercase;
|
||||
color: var(--faint);
|
||||
}
|
||||
.ch-dead {
|
||||
padding: 1px 6px;
|
||||
border-radius: 4px;
|
||||
background: color-mix(in srgb, var(--crit) 12%, transparent);
|
||||
font-size: 10.5px;
|
||||
color: var(--crit);
|
||||
white-space: nowrap;
|
||||
cursor: help;
|
||||
}
|
||||
.ch-rest {
|
||||
font-size: 10.5px;
|
||||
color: var(--faint);
|
||||
}
|
||||
/* On a blocked hop this chip says "set to", not "now": a pick nothing crossed.
|
||||
It steps back to faint so it can't be mistaken for a live reading. */
|
||||
.ch-hop--blocked .ch-now {
|
||||
color: var(--faint);
|
||||
}
|
||||
.ch-now {
|
||||
max-width: 28ch;
|
||||
overflow: hidden;
|
||||
text-overflow: ellipsis;
|
||||
white-space: nowrap;
|
||||
font-size: 10.5px;
|
||||
color: var(--dim);
|
||||
}
|
||||
|
||||
@media (max-width: 640px) {
|
||||
.tg-sec-hd {
|
||||
flex-wrap: wrap;
|
||||
@@ -1060,6 +1296,20 @@
|
||||
.gh-now {
|
||||
max-width: 100%;
|
||||
}
|
||||
/* On a phone the age stamp stops being pushed to a lonely right edge and just
|
||||
joins the end of the hop's line; the selected node gets the full width
|
||||
instead of an ellipsis it doesn't need. */
|
||||
.ch-age {
|
||||
margin-left: 0;
|
||||
}
|
||||
.ch-now {
|
||||
max-width: 100%;
|
||||
}
|
||||
/* Every field of a hop wraps onto its own line at this width, so the gap
|
||||
between hops has to grow with them or the rail reads as one block of text. */
|
||||
.ch-body {
|
||||
padding-bottom: 12px;
|
||||
}
|
||||
.gh-mems {
|
||||
max-height: 260px;
|
||||
}
|
||||
|
||||
+426
-84
@@ -22,6 +22,8 @@ import type {
|
||||
GroupTestResult,
|
||||
GroupTestStatus,
|
||||
Chain,
|
||||
ChainHealth,
|
||||
ChainHopHealth,
|
||||
Egress,
|
||||
Interface,
|
||||
Node,
|
||||
@@ -436,14 +438,14 @@ const normalizeTest = (st: GroupTestStatus): GroupTestStatus => ({
|
||||
})
|
||||
|
||||
/**
|
||||
* How the header names the reach of a running exit test. The name matters more
|
||||
* How the header names the reach of a running refresh pass. The name matters more
|
||||
* than the number when there is only one: "auto" tells the operator which button
|
||||
* they pressed; "1 target" tells them nothing they didn't already know.
|
||||
*/
|
||||
function scopeLabel(scope: string[], targetCount: number): string {
|
||||
if (scope.length === 1) return scope[0]
|
||||
if (scope.length === 0) return 'exits' // pre-scope daemon — say nothing false
|
||||
return scope.length >= targetCount ? 'every exit' : `${scope.length} exits`
|
||||
if (scope.length === 0) return 'targets' // pre-scope daemon — say nothing false
|
||||
return scope.length >= targetCount ? 'every target' : `${scope.length} targets`
|
||||
}
|
||||
|
||||
/** Which editor (add or edit-by-name) is open within a section. */
|
||||
@@ -590,10 +592,12 @@ export default function Targets() {
|
||||
[health],
|
||||
)
|
||||
|
||||
// ---- group/chain exit test: how fast, through which node, out which address --
|
||||
// The POST only kicks a run off, and a 2 s poll of the GET carries progress
|
||||
// plus every result so far. One endpoint covers groups and chains alike:
|
||||
// POST with a group or chain name tests that one; an empty name tests them all.
|
||||
// ---- out-of-turn refresh: how fast, through which node, out which address ----
|
||||
// The POST does NOT dial. It asks the observatory — the only thing in the daemon
|
||||
// that measures anything, and it measures along the real dial path — to come
|
||||
// round out of turn; a 2 s poll of the GET carries progress plus every reading
|
||||
// so far. One endpoint covers groups and chains alike: POST with a name refreshes
|
||||
// that one, an empty name refreshes them all.
|
||||
const [gtest, setGtest] = useState<GroupTestStatus>(IDLE_TEST)
|
||||
const [gtestErr, setGtestErr] = useState<string | null>(null)
|
||||
const [polling, setPolling] = useState(false)
|
||||
@@ -636,7 +640,7 @@ export default function Targets() {
|
||||
void readTest().then((st) => {
|
||||
if (!alive || !st || st.running) return
|
||||
setPolling(false)
|
||||
flash('Group test complete')
|
||||
flash('Readings refreshed')
|
||||
})
|
||||
}, 2000)
|
||||
return () => {
|
||||
@@ -652,16 +656,16 @@ export default function Targets() {
|
||||
if (r.started) {
|
||||
setGtestErr(null)
|
||||
setPolling(true)
|
||||
flash(name ? `Testing ${name}…` : 'Testing every exit…')
|
||||
flash(name ? `Refreshing ${name}…` : 'Refreshing every reading…')
|
||||
void readTest()
|
||||
} else if (r.reason === 'already running') {
|
||||
setPolling(true) // pick up the run someone else started
|
||||
flash('A group test is already running')
|
||||
setPolling(true) // pick up the pass someone else started
|
||||
flash('The prober is already refreshing')
|
||||
} else {
|
||||
flash(`Couldn’t start the test — ${r.reason || 'the daemon refused it'}`)
|
||||
flash(`Couldn’t ask for a refresh — ${r.reason || 'the daemon refused it'}`)
|
||||
}
|
||||
} catch (e) {
|
||||
flash(`Couldn’t start the test — ${errText(e)}`)
|
||||
flash(`Couldn’t ask for a refresh — ${errText(e)}`)
|
||||
}
|
||||
},
|
||||
[flash, readTest],
|
||||
@@ -910,10 +914,11 @@ export default function Targets() {
|
||||
<h2 className="tg-sec-title">Groups</h2>
|
||||
<span className="tg-sec-count mono">{groups.length} configured</span>
|
||||
{/* The observatory's background probing is invisible by design — it
|
||||
keeps every used group's and chain's numbers fresh on its own. The
|
||||
one manual run left is the exit test: it is scoped to the groups
|
||||
and chains it names, so its progress says WHICH, and its badge
|
||||
lands only on those cards. */}
|
||||
keeps every used group's and chain's numbers fresh on its own, along
|
||||
the path traffic actually takes. The one manual control left does
|
||||
not measure anything itself: it asks that prober to come round out
|
||||
of turn. It is scoped to the groups and chains it names, so its
|
||||
progress says WHICH, and its badge lands only on those cards. */}
|
||||
<div className="tg-sec-ctl">
|
||||
{groupHealthOn && (
|
||||
<>
|
||||
@@ -921,11 +926,11 @@ export default function Targets() {
|
||||
<span
|
||||
className="tg-run tg-run--exit"
|
||||
role="status"
|
||||
title="An exit test sends one connection through each group or chain it covers and reports the delay and the address the internet sees."
|
||||
title="The background prober is measuring the targets this refresh covers, along the path each one's traffic really takes."
|
||||
>
|
||||
<Led variant="amber" pulse />
|
||||
<span className="tg-run-what">
|
||||
exit test · {scopeLabel(asArray(gtest.scope), groups.length + chains.length)}
|
||||
refreshing · {scopeLabel(asArray(gtest.scope), groups.length + chains.length)}
|
||||
</span>
|
||||
<span className="tg-run-n mono">
|
||||
{gtest.done}/{gtest.total}
|
||||
@@ -935,9 +940,9 @@ export default function Targets() {
|
||||
<Button
|
||||
onClick={() => void runTest()}
|
||||
disabled={busy || !config || (groups.length === 0 && chains.length === 0) || gtest.running}
|
||||
title="Send one connection through each group and chain and report the delay and the exit address the internet sees"
|
||||
title="Ask the background prober to measure every group and chain out of turn, then show what it measured. The panel opens no connection of its own."
|
||||
>
|
||||
{gtest.running ? 'Testing…' : 'Test every exit'}
|
||||
{gtest.running ? 'Refreshing…' : 'Refresh every reading'}
|
||||
</Button>
|
||||
</>
|
||||
)}
|
||||
@@ -957,6 +962,13 @@ export default function Targets() {
|
||||
dials out through a tunnel measures them through that tunnel, so the same node can be alive
|
||||
in one group and dead in another.
|
||||
</p>
|
||||
<p className="tg-sec-note">
|
||||
One thing measures, and the panel is not it. A background prober walks every path your
|
||||
rules use — hop by hop, exactly as traffic goes — and every number on this page is a read
|
||||
of what it found. <strong>Refresh every reading</strong> asks it to come round out of turn
|
||||
instead of waiting for the next pass; it opens no connection of its own, so a target no
|
||||
rule routes through has nothing to report and says so.
|
||||
</p>
|
||||
|
||||
{groupHealthOn && healthErr && (
|
||||
<p className="tg-test-err" role="alert">
|
||||
@@ -967,7 +979,7 @@ export default function Targets() {
|
||||
|
||||
{groupHealthOn && gtestErr && (
|
||||
<p className="tg-test-err" role="alert">
|
||||
Couldn’t read the test results — {gtestErr}.{' '}
|
||||
Couldn’t read the refreshed numbers — {gtestErr}.{' '}
|
||||
<button className="linkish" onClick={() => void readTest()}>
|
||||
Retry
|
||||
</button>
|
||||
@@ -1110,7 +1122,9 @@ export default function Targets() {
|
||||
chain={c}
|
||||
busy={busy}
|
||||
showHealth={groupHealthOn}
|
||||
used={healthByChain.get(c.Name)?.used}
|
||||
// The whole chain health record, not just `.used` — the card
|
||||
// renders the observatory's per-hop measurements from it.
|
||||
health={healthByChain.get(c.Name)}
|
||||
test={testByGroup.get(c.Name)}
|
||||
// The badge is this card's business only when the run names it.
|
||||
testing={gtest.running && testScope.has(c.Name)}
|
||||
@@ -1232,7 +1246,7 @@ function GroupRow({
|
||||
group: Group
|
||||
busy: boolean
|
||||
/** Group health checks are on (Settings). When false, the card drops its health
|
||||
* readout, its exit-test readout and its Test button — it is config only. */
|
||||
* readout, its end-to-end reading and its Refresh button — it is config only. */
|
||||
showHealth: boolean
|
||||
/** This group's membership health, or undefined when the engine hasn't built
|
||||
* it (not applied yet, or dropped for having no usable members). */
|
||||
@@ -1242,14 +1256,14 @@ function GroupRow({
|
||||
healthKnown: boolean
|
||||
test?: GroupTestResult
|
||||
/**
|
||||
* A group exit test covering THIS group is in flight.
|
||||
* A refresh pass covering THIS group is in flight.
|
||||
*
|
||||
* Deliberately not "a test is running": the caller resolves it against the run's
|
||||
* scope. There is no per-card equivalent for the health run — that one measures
|
||||
* every group at once and is reported once, in the section header.
|
||||
*/
|
||||
testing: boolean
|
||||
/** Any exit test is in flight; the daemon runs one at a time. */
|
||||
/** Any refresh pass is in flight; the daemon runs one at a time. */
|
||||
testBusy: boolean
|
||||
onTest: () => void
|
||||
onEdit: () => void
|
||||
@@ -1309,7 +1323,11 @@ function GroupRow({
|
||||
health={health}
|
||||
healthKnown={healthKnown}
|
||||
/>
|
||||
<GroupTestReadout test={test} pending={testing && !test} />
|
||||
<GroupTestReadout
|
||||
test={test}
|
||||
pending={testing && !test}
|
||||
hideAbsence={health?.used === false}
|
||||
/>
|
||||
</>
|
||||
)}
|
||||
</div>
|
||||
@@ -1320,7 +1338,7 @@ function GroupRow({
|
||||
editLabel={`Edit group ${group.Name}`}
|
||||
deleteLabel={`Delete group ${group.Name}`}
|
||||
onTest={showHealth ? onTest : undefined}
|
||||
testLabel={showHealth ? `Test the exit of group ${group.Name}` : undefined}
|
||||
testLabel={showHealth ? `Refresh the reading for group ${group.Name}` : undefined}
|
||||
testDisabled={testBusy}
|
||||
/>
|
||||
</li>
|
||||
@@ -1379,21 +1397,7 @@ function GroupHealthReadout({
|
||||
// its members would stay "untested" forever. That is a fact about the ROUTING
|
||||
// CONFIG, not about the members — so instead of counters that could only ever
|
||||
// read as a permanent unknown, the card says so, quietly: unused, not unwell.
|
||||
if (!health.used) {
|
||||
return (
|
||||
<div className="gh gh--unused">
|
||||
<div className="gh-line">
|
||||
<span
|
||||
className="gh-unused"
|
||||
title="No enabled rule routes through this group, so its members are not probed. Add it to a rule to see health."
|
||||
>
|
||||
unused
|
||||
</span>
|
||||
<span className="gh-quiet">not probed — no enabled rule routes through this group</span>
|
||||
</div>
|
||||
</div>
|
||||
)
|
||||
}
|
||||
if (!health.used) return <NotRoutedNote kind="group" />
|
||||
|
||||
const v = verdictOf(health)
|
||||
const { total, tested, alive, dead, untested } = health
|
||||
@@ -1556,6 +1560,56 @@ function GroupHealthReadout({
|
||||
)
|
||||
}
|
||||
|
||||
/**
|
||||
* The card's answer when NOTHING ROUTES THROUGH THIS TARGET. Shared by the group
|
||||
* card and the chain card, because it is the same misunderstanding on both.
|
||||
*
|
||||
* It has to carry two statements, and the old one-liner ("not probed — no enabled
|
||||
* rule routes through this group") only carried the first. Read fast it still
|
||||
* landed as a verdict: a card that normally shows health and today shows a grey
|
||||
* pill reads as "the health is bad". So the two meanings are now separated, on
|
||||
* purpose and in this order:
|
||||
*
|
||||
* 1. the ROUTING FACT — nothing routes here, so nothing measures it;
|
||||
* 2. the NON-FACT — this is not a health reading at all. Absent numbers are
|
||||
* absence of measurement, never failure.
|
||||
*
|
||||
* For a GROUP there is a third line, and it is the confusion this whole change
|
||||
* exists to end: a group used only as a hop inside a chain is never routed to
|
||||
* DIRECTLY, so it correctly reads unused here while carrying real traffic as a
|
||||
* hop. Its health is measured at that hop, on the chain's card.
|
||||
*
|
||||
* Unused is neutral — groove-grey, never amber, never crit. It is a state of the
|
||||
* config, and the config is not sick.
|
||||
*/
|
||||
function NotRoutedNote({ kind }: { kind: 'group' | 'chain' }) {
|
||||
return (
|
||||
<div className="gh gh--unused">
|
||||
<div className="gh-line">
|
||||
<span className="gh-unused">unused</span>
|
||||
<span className="gh-quiet">
|
||||
No enabled rule routes through this {kind}, so the observatory never probes it.
|
||||
</span>
|
||||
</div>
|
||||
<p className="gh-say">
|
||||
That is a routing fact, not a health reading. There are no numbers here because nothing
|
||||
measured this {kind} — not because it failed.
|
||||
</p>
|
||||
{kind === 'group' ? (
|
||||
<p className="gh-say">
|
||||
A group used only as a hop inside a chain reads unused here on purpose: the rules point at
|
||||
the chain, not at the group. Its members are measured at that hop, so its real health is on
|
||||
that chain’s card, hop by hop.
|
||||
</p>
|
||||
) : (
|
||||
<p className="gh-say">
|
||||
Point a rule at this chain and the observatory starts measuring every hop within seconds.
|
||||
</p>
|
||||
)}
|
||||
</div>
|
||||
)
|
||||
}
|
||||
|
||||
/**
|
||||
* One group's member rows, fetched on demand.
|
||||
*
|
||||
@@ -1661,7 +1715,9 @@ function MemberRow({ member }: { member: GroupMemberHealth }) {
|
||||
}
|
||||
|
||||
/**
|
||||
* What a group test found, in the four states it actually has.
|
||||
* What the OBSERVATORY measured for this target end to end, in the four states it
|
||||
* actually has. Nothing here was dialled by the panel — it is a read of the
|
||||
* background prober's own measurement along the real path.
|
||||
*
|
||||
* The one worth spelling out: `ok` with an EMPTY `exit_ip` is a SUCCESS. The
|
||||
* delay was measured; only the address lookup came back empty. Rendering that as
|
||||
@@ -1669,15 +1725,45 @@ function MemberRow({ member }: { member: GroupMemberHealth }) {
|
||||
* traffic, so it reads as a result with the address slot marked unknown — dim,
|
||||
* not red, and the LED stays green.
|
||||
*/
|
||||
function GroupTestReadout({ test, pending }: { test?: GroupTestResult; pending: boolean }) {
|
||||
/**
|
||||
* Errors that mean NO MEASUREMENT EXISTS, as opposed to "this target is broken".
|
||||
*
|
||||
* Three of the observatory's four failure reasons are about the observatory, not
|
||||
* about the path: nothing routes here, nothing has reached it yet, or background
|
||||
* probing is switched off. Painting those crit-red — which is what `ok:false`
|
||||
* used to buy you — reports a fault that nobody has found, on a target that may
|
||||
* be carrying traffic perfectly. Only "the observatory's probe through this path
|
||||
* failed" is a health finding, and it is deliberately NOT in this list.
|
||||
*
|
||||
* Matched on a stable fragment rather than the whole sentence, so a daemon that
|
||||
* rewords the tail still classifies. An error we don't recognise stays red: an
|
||||
* unknown failure is likelier to be real than not, and that is the safe default.
|
||||
*/
|
||||
const NO_MEASUREMENT = [
|
||||
'not routed by any enabled rule',
|
||||
'has not reached this target yet',
|
||||
'background probing is disabled',
|
||||
]
|
||||
const isAbsence = (err: string): boolean => NO_MEASUREMENT.some((frag) => err.includes(frag))
|
||||
|
||||
function GroupTestReadout({
|
||||
test,
|
||||
pending,
|
||||
hideAbsence,
|
||||
}: {
|
||||
test?: GroupTestResult
|
||||
pending: boolean
|
||||
/** The card already explains why nothing measures this target (the unused
|
||||
* note), so an absence error here would just say it a second time. */
|
||||
hideAbsence?: boolean
|
||||
}) {
|
||||
if (pending) {
|
||||
// "testing", never "measuring": the health run owns that word and covers every
|
||||
// group at once. Two runs that read the same on a card is how one group's test
|
||||
// came to look like all four were busy.
|
||||
// Names who is working and on what: the prober, on this target. The badge is
|
||||
// scoped to the cards the run covers, so it can say "this one" honestly.
|
||||
return (
|
||||
<div className="tg-test tg-test--wait" role="status">
|
||||
<Led variant="amber" pulse />
|
||||
<span className="tg-test-msg">testing this exit…</span>
|
||||
<span className="tg-test-msg">waiting for the prober to measure this…</span>
|
||||
</div>
|
||||
)
|
||||
}
|
||||
@@ -1686,10 +1772,22 @@ function GroupTestReadout({ test, pending }: { test?: GroupTestResult; pending:
|
||||
const at = test.tested_unix ? fmtClock(test.tested_unix) : ''
|
||||
|
||||
if (!test.ok) {
|
||||
// No measurement exists. Unlit lamp, quiet text: this panel's way of saying
|
||||
// "no verdict", which is precisely the state — never a red one.
|
||||
if (isAbsence(test.error)) {
|
||||
if (hideAbsence) return null
|
||||
return (
|
||||
<div className="tg-test tg-test--none" role="status">
|
||||
<Led variant="off" />
|
||||
<span className="tg-test-msg">{test.error}</span>
|
||||
{at && <span className="tg-test-at mono">{at}</span>}
|
||||
</div>
|
||||
)
|
||||
}
|
||||
return (
|
||||
<div className="tg-test tg-test--bad" role="status">
|
||||
<Led variant="crit" />
|
||||
<span className="tg-test-msg">{test.error || 'the test failed'}</span>
|
||||
<span className="tg-test-msg">{test.error || 'the probe failed'}</span>
|
||||
{at && <span className="tg-test-at mono">{at}</span>}
|
||||
</div>
|
||||
)
|
||||
@@ -2114,7 +2212,7 @@ function ChainRow({
|
||||
chain,
|
||||
busy,
|
||||
showHealth,
|
||||
used,
|
||||
health,
|
||||
test,
|
||||
testing,
|
||||
testBusy,
|
||||
@@ -2125,31 +2223,41 @@ function ChainRow({
|
||||
chain: Chain
|
||||
busy: boolean
|
||||
/** Group health checks are on (Settings). When false, the card drops its
|
||||
* exit-test readout and Test button — it is config only. */
|
||||
* health readout and Refresh button — it is config only. */
|
||||
showHealth: boolean
|
||||
/** This chain's reachability (GroupHealth.Used's chain analogue, plan §5.E).
|
||||
* undefined ⇒ the health endpoint hasn't reported this chain (not applied yet, or
|
||||
* a daemon version without chains): no badge. false ⇒ no enabled rule routes
|
||||
* through the chain, so the observatory never probes it and the card renders
|
||||
* "unused" instead of an exit-test readout. */
|
||||
used?: boolean
|
||||
/** Everything the observatory knows about this chain: whether any enabled rule
|
||||
* routes through it, and the per-hop measurements along it.
|
||||
* undefined ⇒ the health endpoint hasn't reported this chain at all (not
|
||||
* applied yet, or a daemon version without chains): the card says nothing
|
||||
* rather than guessing. */
|
||||
health?: ChainHealth
|
||||
test?: GroupTestResult
|
||||
/** An exit test covering THIS chain is in flight (the caller resolves it
|
||||
/** A refresh pass covering THIS chain is in flight (the caller resolves it
|
||||
* against the run's scope, exactly as for a group card). */
|
||||
testing: boolean
|
||||
/** Any exit test is in flight; the daemon runs one at a time. */
|
||||
/** Any refresh pass is in flight; the daemon runs one at a time. */
|
||||
testBusy: boolean
|
||||
onTest: () => void
|
||||
onEdit: () => void
|
||||
onDelete: () => void
|
||||
}) {
|
||||
const hops = asArray(chain.Hops)
|
||||
// A LEADING `egress:` is not a hop and the rail below already knows it: the
|
||||
// daemon lifts it into hop 1's entry detour (see hopLabels), so it is tagged
|
||||
// "entry" and never numbered. The badge counted it anyway, which is how a chain
|
||||
// drawn with four hops came to be labelled "5 hops" directly above them.
|
||||
const entryEgress = hops.length > 0 && hops[0].startsWith('egress:')
|
||||
const numbered = entryEgress ? hops.length - 1 : hops.length
|
||||
return (
|
||||
<li className="tg-row">
|
||||
<div className="tg-row-main">
|
||||
<div className="tg-row-l1">
|
||||
<span className="tg-row-name">{chain.Name}</span>
|
||||
<span className="tg-badge">{hops.length} hop{hops.length === 1 ? '' : 's'}</span>
|
||||
<span className="tg-badge">
|
||||
{numbered === 0 && entryEgress
|
||||
? 'entry only · no exit'
|
||||
: `${numbered} hop${numbered === 1 ? '' : 's'}`}
|
||||
</span>
|
||||
</div>
|
||||
<div className="tg-row-l2">
|
||||
{hops.length === 0 ? (
|
||||
@@ -2181,25 +2289,21 @@ function ChainRow({
|
||||
{showHealth && (
|
||||
<>
|
||||
{/* A chain no enabled rule routes through is never probed (the
|
||||
observatory walks only reachable paths), so instead of an exit-test
|
||||
readout the card says so, quietly — the same "unused" pattern the
|
||||
group card uses (GroupHealthReadout), not a new design. `used` is
|
||||
undefined until the health endpoint reports this chain (or from a
|
||||
daemon version without chains): no badge then. */}
|
||||
{used === false && (
|
||||
<div className="gh gh--unused">
|
||||
<div className="gh-line">
|
||||
<span
|
||||
className="gh-unused"
|
||||
title="No enabled rule routes through this chain, so its exit is not probed. Add it to a rule to see health."
|
||||
>
|
||||
unused
|
||||
</span>
|
||||
<span className="gh-quiet">not probed — no enabled rule routes through this chain</span>
|
||||
</div>
|
||||
</div>
|
||||
)}
|
||||
<GroupTestReadout test={test} pending={testing && !test} />
|
||||
observatory walks only reachable paths), so instead of a health
|
||||
readout the card says so — the same "unused" note the group card
|
||||
uses, not a new design. `health` is undefined until the endpoint
|
||||
reports this chain (or on a daemon without chains): say nothing
|
||||
then rather than guess. */}
|
||||
{health?.used === false ? (
|
||||
<NotRoutedNote kind="chain" />
|
||||
) : health?.used ? (
|
||||
<ChainHopRail chain={chain.Name} defs={hops} hops={health.hops} />
|
||||
) : null}
|
||||
<GroupTestReadout
|
||||
test={test}
|
||||
pending={testing && !test}
|
||||
hideAbsence={health?.used === false}
|
||||
/>
|
||||
</>
|
||||
)}
|
||||
</div>
|
||||
@@ -2210,13 +2314,250 @@ function ChainRow({
|
||||
editLabel={`Edit chain ${chain.Name}`}
|
||||
deleteLabel={`Delete chain ${chain.Name}`}
|
||||
onTest={showHealth ? onTest : undefined}
|
||||
testLabel={showHealth ? `Test the exit of chain ${chain.Name}` : undefined}
|
||||
testLabel={showHealth ? `Refresh the reading for chain ${chain.Name}` : undefined}
|
||||
testDisabled={testBusy}
|
||||
/>
|
||||
</li>
|
||||
)
|
||||
}
|
||||
|
||||
// ---- chain hop rail --------------------------------------------------------
|
||||
|
||||
/**
|
||||
* The API gives hops an index and no name. The page already knows the names — the
|
||||
* model's own `Hops` strings ("egress:ewan", "node:awgout", "group:sub0") — so
|
||||
* zip the two by POSITION.
|
||||
*
|
||||
* Two things make that safe rather than clever. A LEADING `egress:` is not a
|
||||
* numbered hop: the daemon lifts it into hop 1's entry detour, so it is dropped
|
||||
* before counting. And if the counts still disagree — a chain that splices
|
||||
* sub-chains gets FLATTENED by the daemon, producing more wire hops than the
|
||||
* config lists — every label is dropped. A hop labelled with its neighbour's name
|
||||
* is worse than a hop with no name at all: it would send someone to fix the wrong
|
||||
* target.
|
||||
*/
|
||||
function hopLabels(defs: string[], hops: ChainHopHealth[]): (string | undefined)[] {
|
||||
const numbered = defs.length > 0 && defs[0].startsWith('egress:') ? defs.slice(1) : defs
|
||||
if (numbered.length !== hops.length) return hops.map(() => undefined)
|
||||
return hops.map((h) => (h.index >= 1 && h.index <= numbered.length ? numbered[h.index - 1] : undefined))
|
||||
}
|
||||
|
||||
/**
|
||||
* One hop's lamp.
|
||||
*
|
||||
* `dead` is crit and `untested` is an UNLIT socket — never red, because nothing
|
||||
* has been measured and an unlit lamp is this panel's way of saying "no verdict".
|
||||
* That covers a hop the walk never reached (`blocked_by`) too: it is neither
|
||||
* healthy nor broken, and unlit is the only mark that claims neither.
|
||||
* The fourth case is the page's existing house reading, applied here for
|
||||
* consistency rather than invented: a group hop that is carrying traffic but has
|
||||
* confirmed failures on its board is amber. `state` stays the daemon's word for
|
||||
* "can this hop carry traffic"; the amber only qualifies HOW WELL.
|
||||
*/
|
||||
function hopLed(h: ChainHopHealth): LedVariant {
|
||||
if (h.state === 'dead') return 'crit'
|
||||
if (h.state === 'untested') return 'off'
|
||||
return h.dead > 0 ? 'amber' : 'on'
|
||||
}
|
||||
|
||||
/**
|
||||
* What the observatory measured at each position of a chain — the reading the
|
||||
* daemon always took and the panel never showed.
|
||||
*
|
||||
* THE DESIGN RISK, and the one place this card spends any boldness: the rail
|
||||
* draws the CONDUCTOR as well as the lamps, and severs it below the first dead
|
||||
* hop. A chain is a single series path, so the operator's real question is never
|
||||
* "how many hops are green" — it is "where does my traffic stop". Four lamps in a
|
||||
* column answer the first question and leave the second to arithmetic. A broken
|
||||
* conductor answers the second one before you have read a single word, which is
|
||||
* the whole reason this feature exists.
|
||||
*
|
||||
* It stays honest by keeping two different facts on two different marks. The
|
||||
* CONDUCTOR is reachability through the path, and that genuinely does stop at the
|
||||
* break. The LAMPS are measurements — and there are none below the break to show:
|
||||
* the daemon walks the path in order and stops at the first hop that does not
|
||||
* answer, because every later hop is dialled THROUGH that one. Those hops arrive
|
||||
* `untested` with `blocked_by` naming the hop that stopped the walk, so their
|
||||
* lamps stay UNLIT: not a soft red, not a pale green, just this panel's way of
|
||||
* saying no verdict exists about a hop nobody reached. The row says so in words
|
||||
* too, naming that hop, because "why is this row empty" is the question the shape
|
||||
* alone cannot answer.
|
||||
*
|
||||
* Nothing here is re-derived from the daemon's counters — the block, the zeroed
|
||||
* numbers and the state all come off the wire. The only thing the panel adds is
|
||||
* what a chain structurally is.
|
||||
*
|
||||
* Everything around the rail is deliberately quiet: no colour but the semantic
|
||||
* lamps, no motion at all, the orange accent untouched.
|
||||
*/
|
||||
function ChainHopRail({
|
||||
chain,
|
||||
defs,
|
||||
hops,
|
||||
}: {
|
||||
chain: string
|
||||
/** The chain's configured hops, straight off the model — the only source of names. */
|
||||
defs: string[]
|
||||
/** Absent ⇒ the engine never materialised per-hop outbounds. NOT "no hops". */
|
||||
hops?: ChainHopHealth[]
|
||||
}) {
|
||||
const ordered = useMemo(() => [...asArray(hops)].sort((a, b) => a.index - b.index), [hops])
|
||||
const labels = useMemo(() => hopLabels(defs, ordered), [defs, ordered])
|
||||
|
||||
// The first hop that was probed and did not answer. Everything after it is
|
||||
// unreachable THROUGH THIS CHAIN, whatever its own lamp says. `untested` is
|
||||
// never a break: nothing was measured, so nothing is known to be severed.
|
||||
const breakAt = ordered.findIndex((h) => h.state === 'dead')
|
||||
|
||||
if (ordered.length === 0) {
|
||||
// Say why, in one line, instead of an empty rail. The daemon collapses a
|
||||
// single-target chain into a plain alias and never builds copies to measure,
|
||||
// so we can tell the two absences apart from the config alone.
|
||||
const numbered = defs.filter((d, i) => !(i === 0 && d.startsWith('egress:')))
|
||||
return (
|
||||
<div className="gh gh--absent">
|
||||
<span className="gh-absent-msg">
|
||||
{numbered.length <= 1
|
||||
? 'This chain has a single hop, so the engine points traffic straight at that target instead of building a path to measure. Its health is on that target’s own card.'
|
||||
: 'The engine hasn’t built this chain’s hops yet, so there is nothing measured per hop. They appear once it is running with this config applied.'}
|
||||
</span>
|
||||
</div>
|
||||
)
|
||||
}
|
||||
|
||||
return (
|
||||
<div className="gh ch-rail">
|
||||
<span className="ch-eyebrow">measured, hop by hop</span>
|
||||
<ol className="ch-hops">
|
||||
{ordered.map((h, i) => {
|
||||
const label = labels[i]
|
||||
const severed = breakAt >= 0 && i > breakAt
|
||||
// Straight off the wire: present ⇒ the walk never reached this hop, so
|
||||
// there is nothing measured here and the daemon has already named the
|
||||
// hop that stopped it. Never inferred from the counters.
|
||||
const blocked = h.blocked_by
|
||||
const cls = [
|
||||
'ch-hop',
|
||||
`ch-hop--${h.state}`,
|
||||
blocked ? 'ch-hop--blocked' : '',
|
||||
severed ? 'ch-hop--severed' : '',
|
||||
h.exit ? 'ch-hop--exit' : '',
|
||||
i === 0 ? 'ch-hop--first' : '',
|
||||
]
|
||||
.filter(Boolean)
|
||||
.join(' ')
|
||||
const age = fmtAge(h.age_seconds)
|
||||
return (
|
||||
<li key={h.tag || h.index} className={cls}>
|
||||
<span className="ch-num mono" aria-hidden="true">
|
||||
{h.index}
|
||||
</span>
|
||||
<span className="ch-socket">
|
||||
<Led variant={hopLed(h)} />
|
||||
</span>
|
||||
<div className="ch-body">
|
||||
<div className="ch-l1">
|
||||
<span className="ch-name mono" title={`engine outbound ${h.tag}`}>
|
||||
{label ?? (h.kind === 'group' ? 'a group hop' : 'a node hop')}
|
||||
</span>
|
||||
{h.exit && <span className="ch-tag">exit</span>}
|
||||
{h.state === 'alive' && h.delay_ms > 0 && (
|
||||
<span className="ch-delay mono">{h.delay_ms} ms</span>
|
||||
)}
|
||||
{/* Two different silences. A plain untested hop is a timing
|
||||
gap that fills in by itself; a BLOCKED one never will,
|
||||
because the walk stopped above it — so it says which hop
|
||||
stopped it instead of implying someone should wait. */}
|
||||
{blocked ? (
|
||||
<span
|
||||
className="ch-quiet ch-blocked"
|
||||
title={`hop ${blocked.index} did not answer, so nothing was dialled through it (engine outbound ${blocked.tag})`}
|
||||
>
|
||||
no reading — the probe stopped at hop {blocked.index}
|
||||
</span>
|
||||
) : (
|
||||
h.state === 'untested' && <span className="ch-quiet">not measured yet</span>
|
||||
)}
|
||||
{age && <span className="ch-age mono">{age}</span>}
|
||||
</div>
|
||||
|
||||
{/* A node hop IS its own measurement (total 1), so counters would
|
||||
only restate the lamp. A group hop rolls up its per-hop member
|
||||
copies, and those read exactly as they do everywhere else in
|
||||
this app: alive out of TESTED, with the untested remainder as a
|
||||
quiet aside only when there is one. A blocked hop has those
|
||||
counters zeroed by the daemon, so it lands in the tested === 0
|
||||
branch — and there it must say the members were never REACHED,
|
||||
not that they are still waiting their turn. */}
|
||||
{h.kind === 'group' && h.total > 0 && (
|
||||
<div className="ch-l2">
|
||||
{h.tested === 0 ? (
|
||||
<span className="ch-rest mono">
|
||||
{h.total} member{h.total === 1 ? '' : 's'},{' '}
|
||||
{blocked ? 'none of them reached' : 'none measured'}
|
||||
</span>
|
||||
) : (
|
||||
<>
|
||||
<span className="ch-count mono">
|
||||
<b>{h.alive}</b> / {h.tested}
|
||||
</span>
|
||||
<span className="ch-word">alive</span>
|
||||
{h.dead > 0 && (
|
||||
<span
|
||||
className="ch-dead mono"
|
||||
title={`${h.dead} member${h.dead === 1 ? '' : 's'} were probed at this hop and did not answer`}
|
||||
>
|
||||
{h.dead} not answering
|
||||
</span>
|
||||
)}
|
||||
{h.untested > 0 && (
|
||||
<span className="ch-rest mono">
|
||||
tested {h.tested} of {h.total}
|
||||
</span>
|
||||
)}
|
||||
</>
|
||||
)}
|
||||
{/* The wrapper keeps its pick even when nothing crossed it,
|
||||
so on a blocked hop this is the node it WOULD use — say
|
||||
that, rather than "now", which claims live traffic. */}
|
||||
{h.selected && (
|
||||
<span
|
||||
className="ch-now mono"
|
||||
title={
|
||||
blocked
|
||||
? `Hop ${h.index} of “${chain}” is set to ${h.selected}; nothing crossed it to measure`
|
||||
: `Traffic crossing hop ${h.index} of “${chain}” is on ${h.selected}`
|
||||
}
|
||||
>
|
||||
{blocked ? 'set to' : 'now'} → {h.selected}
|
||||
</span>
|
||||
)}
|
||||
</div>
|
||||
)}
|
||||
</div>
|
||||
</li>
|
||||
)
|
||||
})}
|
||||
</ol>
|
||||
|
||||
{/* The sentence the rail's shape implies, written out — because the break is
|
||||
the answer someone came here for, and a graphic alone should never be the
|
||||
only place a finding exists. It names the dead hop, since that is the one
|
||||
thing here anybody can act on. */}
|
||||
{breakAt >= 0 && (
|
||||
<p className="gh-say gh-say--bad">
|
||||
Hop {ordered[breakAt].index}
|
||||
{labels[breakAt] ? ` (${labels[breakAt]})` : ''} was probed and did not answer, so traffic
|
||||
stops there
|
||||
{breakAt < ordered.length - 1
|
||||
? ' — and the hops below it are dialled through it, so nothing reached them and nothing is known about them.'
|
||||
: '.'}
|
||||
</p>
|
||||
)}
|
||||
</div>
|
||||
)
|
||||
}
|
||||
|
||||
function ChainEditor({
|
||||
initial,
|
||||
hopOptions,
|
||||
@@ -2691,8 +3032,8 @@ function RowActions({
|
||||
busy: boolean
|
||||
editLabel: string
|
||||
deleteLabel: string
|
||||
// Only groups and chains can be tested, so the control is optional and absent
|
||||
// everywhere else rather than a disabled stub on every row.
|
||||
// Only groups and chains are probed, so the refresh control is optional and
|
||||
// absent everywhere else rather than a disabled stub on every row.
|
||||
onTest?: () => void
|
||||
testLabel?: string
|
||||
testDisabled?: boolean
|
||||
@@ -2705,8 +3046,9 @@ function RowActions({
|
||||
onClick={onTest}
|
||||
disabled={busy || testDisabled}
|
||||
aria-label={testLabel}
|
||||
title="Ask the background prober to measure this target out of turn. It does not open a connection from the panel."
|
||||
>
|
||||
Test
|
||||
Refresh
|
||||
</Button>
|
||||
)}
|
||||
<Button className="tg-act" onClick={onEdit} disabled={busy} aria-label={editLabel}>
|
||||
|
||||
@@ -0,0 +1,91 @@
|
||||
// pendingConfirm — the record of an armed auto-rollback, shared by the whole panel.
|
||||
//
|
||||
// Run with `npm test`. The module imports React only for its hook; the plain
|
||||
// functions exercised here touch neither React nor the DOM, and `localStorage` is
|
||||
// absent under node, which is itself one of the cases worth pinning (the panel
|
||||
// must still work, it just forgets on reload).
|
||||
//
|
||||
// What these protect:
|
||||
// - arming when commit-confirm is OFF must record nothing. The daemon does not
|
||||
// arm a window then, and a countdown for a rollback that will never happen is
|
||||
// the same class of lie as the "Confirmed" message this module replaced.
|
||||
// - a window that has elapsed reads as gone, so nothing renders "0 s left".
|
||||
// - expiry notifies exactly once even though several components watch it.
|
||||
|
||||
import { test } from 'node:test'
|
||||
import assert from 'node:assert/strict'
|
||||
|
||||
import {
|
||||
armPendingConfirm,
|
||||
clearPendingConfirm,
|
||||
confirmTimeout,
|
||||
noteConfirmTimeout,
|
||||
onPendingConfirmExpire,
|
||||
readPendingConfirm,
|
||||
} from './pendingConfirm.ts'
|
||||
|
||||
test('commit-confirm off ⇒ arming records nothing', () => {
|
||||
noteConfirmTimeout(0)
|
||||
assert.equal(confirmTimeout(), 0)
|
||||
armPendingConfirm()
|
||||
assert.equal(readPendingConfirm(), null)
|
||||
})
|
||||
|
||||
test('a window is recorded with the timeout the config reported', () => {
|
||||
noteConfirmTimeout(90)
|
||||
armPendingConfirm()
|
||||
const p = readPendingConfirm()
|
||||
assert.notEqual(p, null)
|
||||
assert.equal(p!.total, 90)
|
||||
// Deadline is in the future and within a second of now + the window.
|
||||
const left = (p!.until - Date.now()) / 1000
|
||||
assert.ok(left > 89 && left <= 90, `expected ~90s left, got ${left}`)
|
||||
clearPendingConfirm()
|
||||
assert.equal(readPendingConfirm(), null)
|
||||
})
|
||||
|
||||
test('switching commit-confirm off drops a window that was already armed', () => {
|
||||
noteConfirmTimeout(60)
|
||||
armPendingConfirm()
|
||||
assert.notEqual(readPendingConfirm(), null)
|
||||
noteConfirmTimeout(0)
|
||||
assert.equal(readPendingConfirm(), null)
|
||||
})
|
||||
|
||||
test('an elapsed window reads as gone, never as a countdown at zero', () => {
|
||||
noteConfirmTimeout(1)
|
||||
armPendingConfirm()
|
||||
const p = readPendingConfirm()
|
||||
assert.notEqual(p, null)
|
||||
// Wind the clock forward rather than sleeping through the window.
|
||||
const realNow = Date.now
|
||||
Date.now = () => realNow() + 5000
|
||||
try {
|
||||
assert.equal(readPendingConfirm(), null)
|
||||
} finally {
|
||||
Date.now = realNow
|
||||
}
|
||||
clearPendingConfirm()
|
||||
})
|
||||
|
||||
test('confirming does NOT fire the expiry listeners', () => {
|
||||
noteConfirmTimeout(30)
|
||||
let fired = 0
|
||||
const off = onPendingConfirmExpire(() => {
|
||||
fired++
|
||||
})
|
||||
armPendingConfirm()
|
||||
clearPendingConfirm()
|
||||
off()
|
||||
assert.equal(fired, 0)
|
||||
})
|
||||
|
||||
test('a nonsense timeout is ignored rather than taken as "off"', () => {
|
||||
noteConfirmTimeout(45)
|
||||
noteConfirmTimeout(Number.NaN)
|
||||
noteConfirmTimeout(-1)
|
||||
noteConfirmTimeout(undefined)
|
||||
assert.equal(confirmTimeout(), 45)
|
||||
clearPendingConfirm()
|
||||
noteConfirmTimeout(0)
|
||||
})
|
||||
@@ -0,0 +1,186 @@
|
||||
// The commit-confirm window, as ONE fact the whole panel can see.
|
||||
//
|
||||
// WHY THIS EXISTS. `POST /api/apply` always arms an auto-rollback for
|
||||
// `Globals.ConfirmTimeout` seconds (shater/panel/api.go handleApply →
|
||||
// ArmRollback) — EVERY Apply button does that, not just the one on the Apply
|
||||
// page. But the countdown, and the button that stops it, lived in one component's
|
||||
// local state. So:
|
||||
//
|
||||
// - pressing Apply on Routing/DNS/Nodes/Settings/Devices/Profiles/Targets said
|
||||
// "Applied" and nothing else; the operator walked away and the router quietly
|
||||
// reverted a minute later;
|
||||
// - reloading the tab wiped the countdown AND the "Keep this config" button, so
|
||||
// there was no way left to confirm from the panel at all.
|
||||
//
|
||||
// The daemon does not report a deadline, so this module records the one the panel
|
||||
// itself armed, in `localStorage`. That is a deliberately modest claim — it knows
|
||||
// about windows THIS BROWSER opened and says nothing about one opened elsewhere —
|
||||
// but it survives a reload, a new tab and a navigation, which is what the two
|
||||
// failures above needed.
|
||||
//
|
||||
// It is also the answer to "is there anything to confirm?". `apply.Confirm()`
|
||||
// returns nil unconditionally, so a Confirm button that is always live can only
|
||||
// ever report success. Gating it on a record here means the panel offers the
|
||||
// action when it knows a window is open, and then reports an outcome it knows.
|
||||
|
||||
import { useEffect, useState } from 'react'
|
||||
|
||||
/** A live commit-confirm window the panel armed. */
|
||||
export interface PendingConfirm {
|
||||
/** Epoch ms at which the daemon auto-rolls back if nobody confirms. */
|
||||
until: number
|
||||
/** The window it started with, in seconds — the progress bar's denominator. */
|
||||
total: number
|
||||
}
|
||||
|
||||
const KEY = 'shater.pendingConfirm'
|
||||
|
||||
type Listener = () => void
|
||||
const listeners = new Set<Listener>()
|
||||
const expiryListeners = new Set<Listener>()
|
||||
|
||||
/** localStorage is absent under SSR/tests and throws in some privacy modes. A
|
||||
* panel that cannot remember a window must still work — it just forgets on
|
||||
* reload, which is exactly the old behaviour and no worse. */
|
||||
function store(): Storage | null {
|
||||
try {
|
||||
return typeof localStorage === 'undefined' ? null : localStorage
|
||||
} catch {
|
||||
return null
|
||||
}
|
||||
}
|
||||
|
||||
function load(): PendingConfirm | null {
|
||||
const s = store()
|
||||
if (!s) return null
|
||||
try {
|
||||
const raw = s.getItem(KEY)
|
||||
if (!raw) return null
|
||||
const v = JSON.parse(raw) as Partial<PendingConfirm>
|
||||
if (typeof v.until !== 'number' || typeof v.total !== 'number') return null
|
||||
if (!Number.isFinite(v.until)) return null
|
||||
return { until: v.until, total: v.total }
|
||||
} catch {
|
||||
return null
|
||||
}
|
||||
}
|
||||
|
||||
// The single in-process copy. Storage is the durable mirror, not the source of
|
||||
// truth for a running tab: a `storage` event re-hydrates it when another tab
|
||||
// writes.
|
||||
let armed: PendingConfirm | null = load()
|
||||
|
||||
function emit() {
|
||||
for (const l of [...listeners]) l()
|
||||
}
|
||||
|
||||
function write(v: PendingConfirm | null) {
|
||||
armed = v
|
||||
const s = store()
|
||||
if (s) {
|
||||
try {
|
||||
if (v) s.setItem(KEY, JSON.stringify(v))
|
||||
else s.removeItem(KEY)
|
||||
} catch {
|
||||
// Storage full or blocked — the in-process copy still drives this tab.
|
||||
}
|
||||
}
|
||||
emit()
|
||||
}
|
||||
|
||||
/** The armed window, or null when there is none or it has already elapsed. */
|
||||
export function readPendingConfirm(): PendingConfirm | null {
|
||||
if (!armed) return null
|
||||
return armed.until > Date.now() ? armed : null
|
||||
}
|
||||
|
||||
// ---- the window's length ----------------------------------------------------
|
||||
|
||||
// `Globals.ConfirmTimeout` is all that is needed to arm a window, and every page
|
||||
// reads the config anyway — so api.getConfig() feeds it here rather than each
|
||||
// caller threading it through. 0 (or never seen) means commit-confirm is off, and
|
||||
// arming then does nothing: an apply on such a router really is immediate.
|
||||
let timeout = 0
|
||||
|
||||
export function noteConfirmTimeout(seconds: number | undefined) {
|
||||
if (typeof seconds !== 'number' || !Number.isFinite(seconds) || seconds < 0) return
|
||||
timeout = Math.floor(seconds)
|
||||
// A window armed before commit-confirm was switched off is no longer real —
|
||||
// drop it rather than count down to an event that will not happen.
|
||||
if (timeout === 0 && armed) write(null)
|
||||
}
|
||||
|
||||
export function confirmTimeout(): number {
|
||||
return timeout
|
||||
}
|
||||
|
||||
/** Record the window an apply just opened. Call only when the apply CHANGED
|
||||
* something: an unchanged apply reconciles nothing and arms nothing. */
|
||||
export function armPendingConfirm() {
|
||||
if (timeout <= 0) return
|
||||
write({ until: Date.now() + timeout * 1000, total: timeout })
|
||||
}
|
||||
|
||||
/** Confirm and rollback both end the window. */
|
||||
export function clearPendingConfirm() {
|
||||
if (armed) write(null)
|
||||
}
|
||||
|
||||
/** Fires when a window ran out on its own — i.e. the daemon has reverted — and
|
||||
* NOT when it was confirmed or rolled back. Returns an unsubscribe. */
|
||||
export function onPendingConfirmExpire(fn: Listener): () => void {
|
||||
expiryListeners.add(fn)
|
||||
return () => {
|
||||
expiryListeners.delete(fn)
|
||||
}
|
||||
}
|
||||
|
||||
/** Idempotent: several mounted countdowns race to notice the same deadline, and
|
||||
* only the first one gets to announce it. */
|
||||
function expire() {
|
||||
if (!armed) return
|
||||
write(null)
|
||||
for (const l of [...expiryListeners]) l()
|
||||
}
|
||||
|
||||
// Another tab confirming, rolling back or applying is the same event as this one
|
||||
// doing it.
|
||||
if (typeof window !== 'undefined') {
|
||||
window.addEventListener('storage', (e) => {
|
||||
if (e.key !== KEY && e.key !== null) return
|
||||
armed = load()
|
||||
emit()
|
||||
})
|
||||
}
|
||||
|
||||
/**
|
||||
* The armed window and its remaining seconds, ticking once a second.
|
||||
*
|
||||
* Returns null when nothing is armed. While non-null `remaining` is at least 1:
|
||||
* reaching zero clears the record and notifies {@link onPendingConfirmExpire}, so
|
||||
* no component ever renders "0 s left" for a window that is already over.
|
||||
*/
|
||||
export function usePendingConfirm(): { pending: PendingConfirm; remaining: number } | null {
|
||||
const [, tick] = useState(0)
|
||||
|
||||
useEffect(() => {
|
||||
const sync = () => tick((n) => n + 1)
|
||||
listeners.add(sync)
|
||||
sync()
|
||||
return () => {
|
||||
listeners.delete(sync)
|
||||
}
|
||||
}, [])
|
||||
|
||||
useEffect(() => {
|
||||
const id = window.setInterval(() => {
|
||||
if (armed && armed.until <= Date.now()) expire()
|
||||
else if (armed) tick((n) => n + 1)
|
||||
}, 1000)
|
||||
return () => window.clearInterval(id)
|
||||
}, [])
|
||||
|
||||
const pending = readPendingConfirm()
|
||||
if (!pending) return null
|
||||
return { pending, remaining: Math.max(1, Math.ceil((pending.until - Date.now()) / 1000)) }
|
||||
}
|
||||
@@ -15,7 +15,7 @@
|
||||
import { test } from 'node:test'
|
||||
import assert from 'node:assert/strict'
|
||||
|
||||
import { protectionState } from './planeState.ts'
|
||||
import { engineReadout, engineState, killSwitchReadout, protectionState } from './planeState.ts'
|
||||
import type { Status, Traffic } from './api.ts'
|
||||
|
||||
/** A healthy, fully-installed router; `traffic` is what each case varies. */
|
||||
@@ -148,3 +148,106 @@ test('daemon too old to send `plane` keeps its own fallback', () => {
|
||||
'Starting up',
|
||||
)
|
||||
})
|
||||
|
||||
// --- engineState: the reading that could not say "down" ----------------------
|
||||
//
|
||||
// `apply.Status.running` was a hardcoded `true` on the daemon, so every panel LED
|
||||
// derived from it was lit before it was read: App's master indicator could not
|
||||
// reach its "Offline" branch, and Apply's "engine: running / stopped" row had one
|
||||
// reachable value. These pin the three answers, and that "up" needs agreement.
|
||||
|
||||
test('engine_running:false is down even while the daemon claims it is running', () => {
|
||||
assert.equal(engineState(status({ running: true, engine_running: false })), 'down')
|
||||
assert.equal(engineReadout(status({ running: true, engine_running: false })).variant, 'crit')
|
||||
assert.equal(engineReadout(status({ running: true, engine_running: false })).word, 'stopped')
|
||||
})
|
||||
|
||||
test('a daemon that reports itself stopped is down whatever engine_running says', () => {
|
||||
assert.equal(engineState(status({ running: false, engine_running: true })), 'down')
|
||||
})
|
||||
|
||||
test('up needs both, and then active/idle splits the lamp', () => {
|
||||
assert.equal(engineState(status({ running: true, engine_running: true })), 'up')
|
||||
assert.equal(engineReadout(status({ active: true })).variant, 'on')
|
||||
assert.equal(engineReadout(status({ active: true })).word, 'active')
|
||||
assert.equal(engineReadout(status({ active: false })).variant, 'amber')
|
||||
assert.equal(engineReadout(status({ active: false })).word, 'idle')
|
||||
})
|
||||
|
||||
test('an older daemon with no engine_running is unknown — an unlit lamp, never green', () => {
|
||||
const { engine_running, ...old } = status()
|
||||
void engine_running
|
||||
assert.equal(engineState(old as Status), 'unknown')
|
||||
const r = engineReadout(old as Status)
|
||||
assert.equal(r.variant, 'off')
|
||||
assert.notEqual(r.variant, 'on')
|
||||
assert.equal(r.word, 'not reported')
|
||||
})
|
||||
|
||||
test('no status at all is unknown, not down', () => {
|
||||
assert.equal(engineState(null), 'unknown')
|
||||
assert.equal(engineReadout(null).variant, 'off')
|
||||
assert.equal(engineReadout(null).word, 'checking…')
|
||||
})
|
||||
|
||||
// --- killSwitchReadout: "I don't know" is not "it's armed" -------------------
|
||||
//
|
||||
// The Overview module read `killArmed && status?.plane !== 'none'`, and
|
||||
// `undefined !== 'none'` is true — so a daemon that never reported `plane`, and
|
||||
// the seconds before the first status arrives, both lit a green lamp over the
|
||||
// word ARMED. These pin the fourth answer that expression could not express.
|
||||
|
||||
test('a daemon that does not report `plane` reads as not reported, never ARMED', () => {
|
||||
const { plane, ...noPlane } = status()
|
||||
void plane
|
||||
const k = killSwitchReadout(noPlane as Status)
|
||||
assert.equal(k.state, 'unknown')
|
||||
assert.notEqual(k.value, 'ARMED')
|
||||
assert.equal(k.variant, 'off')
|
||||
assert.notEqual(k.variant, 'on')
|
||||
assert.equal(k.blockingNow, 'not known')
|
||||
})
|
||||
|
||||
test('no status at all is unknown too, and says there is no reading', () => {
|
||||
const k = killSwitchReadout(null)
|
||||
assert.equal(k.state, 'unknown')
|
||||
assert.equal(k.variant, 'off')
|
||||
assert.equal(k.blockingNow, 'no reading yet')
|
||||
})
|
||||
|
||||
test('fail-closed with a plane installed is armed', () => {
|
||||
for (const plane of ['full', 'hold'] as const) {
|
||||
const k = killSwitchReadout(status({ plane }))
|
||||
assert.equal(k.state, 'armed')
|
||||
assert.equal(k.value, 'ARMED')
|
||||
assert.equal(k.variant, 'on')
|
||||
assert.equal(k.blockingNow, null)
|
||||
}
|
||||
})
|
||||
|
||||
test('fail-closed with no plane is configured but blocking nothing', () => {
|
||||
const k = killSwitchReadout(status({ plane: 'none', table: false }))
|
||||
assert.equal(k.state, 'inert')
|
||||
assert.equal(k.value, 'NOT IN EFFECT')
|
||||
assert.equal(k.variant, 'crit')
|
||||
assert.equal(k.hot, true)
|
||||
})
|
||||
|
||||
test('fail-open is the operator’s choice — amber, and never a plane question', () => {
|
||||
for (const plane of ['full', 'none', undefined] as const) {
|
||||
const k = killSwitchReadout(status({ kill_switch: 'open', plane }))
|
||||
assert.equal(k.state, 'open')
|
||||
assert.equal(k.value, 'OPEN')
|
||||
assert.equal(k.variant, 'amber')
|
||||
}
|
||||
})
|
||||
|
||||
test('the live kill_switch wins over the saved one; the saved one only fills a gap', () => {
|
||||
const { kill_switch, ...noKill } = status()
|
||||
void kill_switch
|
||||
// Live says open, config says closed → live wins.
|
||||
assert.equal(killSwitchReadout(status({ kill_switch: 'open' }), 'closed').state, 'open')
|
||||
// Nothing live → fall back to the saved policy.
|
||||
assert.equal(killSwitchReadout(noKill as Status, 'open').state, 'open')
|
||||
assert.equal(killSwitchReadout(noKill as Status, 'closed').state, 'armed')
|
||||
})
|
||||
|
||||
+141
-10
@@ -21,6 +21,135 @@ export interface ProtectionState {
|
||||
alarm: boolean
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Is the engine actually up?
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
/**
|
||||
* Three answers, and "unknown" is one of them.
|
||||
*
|
||||
* up — the sing-box process is running.
|
||||
* down — it is not. Nothing is being proxied or filtered.
|
||||
* unknown — nobody has told us. Never paint this green.
|
||||
*
|
||||
* THIS EXISTS BECAUSE `status.running` COULD NOT SAY "down". It is the DAEMON's
|
||||
* own liveness, and on the daemons this panel shipped against it was a hardcoded
|
||||
* `true` (apply.go) — so `running ? 'running' : 'stopped'` had exactly one
|
||||
* reachable branch, and every LED derived from it was lit before it was read. A
|
||||
* router whose engine failed to start, with a fail-closed holding plan installed
|
||||
* and no internet on the LAN, showed three green lamps on the page people go to
|
||||
* when they are trying to fix it.
|
||||
*
|
||||
* `engine_running` is the field that answers the question honestly, so it decides
|
||||
* `up`. Either field may still prove a NEGATIVE — a daemon that reports itself
|
||||
* stopped cannot be running an engine — and a negative always wins, so "up" needs
|
||||
* both to agree. Neither field asserting anything leaves `unknown`.
|
||||
*/
|
||||
export type EngineState = 'up' | 'down' | 'unknown'
|
||||
|
||||
export function engineState(status: Status | null): EngineState {
|
||||
if (!status) return 'unknown'
|
||||
if (!status.running) return 'down'
|
||||
if (typeof status.engine_running === 'boolean') return status.engine_running ? 'up' : 'down'
|
||||
return 'unknown'
|
||||
}
|
||||
|
||||
/** How the engine's lamp is painted and what the readout beside it says.
|
||||
*
|
||||
* `unknown` is an UNLIT socket, never amber and never green: amber is this
|
||||
* panel's "degraded", and there is nothing to be degraded about when no reading
|
||||
* has arrived. `down` is crit even when the kill-switch caught it — the engine
|
||||
* being dead is the fault; whether traffic leaks is a separate lamp. */
|
||||
export function engineReadout(status: Status | null): { variant: LedVariant; word: string } {
|
||||
switch (engineState(status)) {
|
||||
case 'down':
|
||||
return { variant: 'crit', word: 'stopped' }
|
||||
case 'up':
|
||||
return status?.active ? { variant: 'on', word: 'active' } : { variant: 'amber', word: 'idle' }
|
||||
default:
|
||||
return { variant: 'off', word: status ? 'not reported' : 'checking…' }
|
||||
}
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Is the kill-switch actually blocking anything?
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
/**
|
||||
* Four answers, and "unknown" is one of them.
|
||||
*
|
||||
* armed — configured fail-closed AND a data plane is installed to enforce it.
|
||||
* inert — configured fail-closed, but there is no plane. Nothing is blocking.
|
||||
* unknown — the daemon has not said how much plane is installed, so whether the
|
||||
* setting is in force is not known. NEVER paint this green.
|
||||
* open — configured fail-open. Nothing is meant to be blocked.
|
||||
*/
|
||||
export type KillSwitchState = 'armed' | 'inert' | 'unknown' | 'open'
|
||||
|
||||
export interface KillSwitchReadout {
|
||||
state: KillSwitchState
|
||||
/** The word the module puts in its readout. */
|
||||
value: string
|
||||
variant: LedVariant
|
||||
/** Is it blocking right now — the row under the readout. `null` ⇒ nothing to add. */
|
||||
blockingNow: string | null
|
||||
/** True when `blockingNow` is bad news and should be drawn hot. */
|
||||
hot: boolean
|
||||
}
|
||||
|
||||
/**
|
||||
* THE UNKNOWN BRANCH IS THE WHOLE POINT. This used to be
|
||||
*
|
||||
* killArmed && status?.plane !== 'none'
|
||||
*
|
||||
* and `undefined !== 'none'` is true — so a daemon that had not reported `plane`
|
||||
* at all, and a panel that had not yet received its first status, both landed in
|
||||
* the "ARMED" branch under a green lamp. Every other unknown in this file is an
|
||||
* unlit socket for exactly this reason (see engineReadout): the kill-switch is
|
||||
* the last thing standing between the LAN and the plain WAN, and "I don't know
|
||||
* whether it is installed" must never be dressed as "it is".
|
||||
*
|
||||
* `configured` is the SAVED policy from /api/config, used only while
|
||||
* /api/status has not reported one. The live value wins wherever it exists, as
|
||||
* everywhere else in the panel: this is a status readout, and the config on disk
|
||||
* can already differ from what is installed.
|
||||
*/
|
||||
export function killSwitchReadout(
|
||||
status: Status | null,
|
||||
configured?: string,
|
||||
): KillSwitchReadout {
|
||||
const closed = (status?.kill_switch ?? configured ?? 'closed') === 'closed'
|
||||
|
||||
if (!closed) {
|
||||
return { state: 'open', value: 'OPEN', variant: 'amber', blockingNow: null, hot: false }
|
||||
}
|
||||
|
||||
switch (status?.plane) {
|
||||
case 'full':
|
||||
case 'hold':
|
||||
// Something is installed, so the fail-closed guard is really in the path.
|
||||
return { state: 'armed', value: 'ARMED', variant: 'on', blockingNow: null, hot: false }
|
||||
case 'none':
|
||||
return {
|
||||
state: 'inert',
|
||||
value: 'NOT IN EFFECT',
|
||||
variant: 'crit',
|
||||
blockingNow: 'no — nothing installed',
|
||||
hot: true,
|
||||
}
|
||||
default:
|
||||
return {
|
||||
state: 'unknown',
|
||||
value: 'NOT REPORTED',
|
||||
variant: 'off',
|
||||
// Terse on purpose: this is a two-column readout row, and the long form
|
||||
// wrapped onto three lines beside a one-word key.
|
||||
blockingNow: status ? 'not known' : 'no reading yet',
|
||||
hot: false,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* `plane` + `engine_running` express the state more precisely than the three
|
||||
* booleans the old status strip exposed (engine active / config enabled / nft
|
||||
@@ -101,8 +230,18 @@ export function protectionState(status: Status | null): ProtectionState {
|
||||
}
|
||||
}
|
||||
|
||||
// Older daemon with no `plane` field: fall back to what we can observe.
|
||||
if (status.running && status.active && status.table) {
|
||||
// Older daemon with no `plane` field: fall back to what we can observe. The
|
||||
// engine's own state is asked FIRST — "stopped" is the answer that matters, and
|
||||
// reading it off `running` is what used to make it unreachable (see engineState).
|
||||
if (engineState(status) === 'down') {
|
||||
return {
|
||||
variant: 'crit',
|
||||
headline: 'Service stopped',
|
||||
detail: 'The engine isn’t running, so traffic isn’t being proxied or filtered.',
|
||||
alarm: true,
|
||||
}
|
||||
}
|
||||
if (status.active && status.table) {
|
||||
return {
|
||||
variant: 'on',
|
||||
headline: 'Protected',
|
||||
@@ -110,14 +249,6 @@ export function protectionState(status: Status | null): ProtectionState {
|
||||
alarm: false,
|
||||
}
|
||||
}
|
||||
if (!status.running) {
|
||||
return {
|
||||
variant: 'crit',
|
||||
headline: 'Service stopped',
|
||||
detail: 'The service isn’t running, so traffic isn’t being handled.',
|
||||
alarm: true,
|
||||
}
|
||||
}
|
||||
return {
|
||||
variant: 'amber',
|
||||
headline: 'Starting up',
|
||||
|
||||
+48
-1
@@ -1,15 +1,62 @@
|
||||
import { defineConfig } from 'vite'
|
||||
import type { Plugin } from 'vite'
|
||||
import react from '@vitejs/plugin-react'
|
||||
|
||||
// Minimal ambient for the dev-proxy target override — avoids pulling in @types/node
|
||||
// just for one env read. Vite runs this file under Node where `process` exists.
|
||||
declare const process: { env: Record<string, string | undefined> }
|
||||
|
||||
/** `src/mock.ts`, as the module graph spells it (POSIX-normalised for Windows). */
|
||||
const MOCK_MODULE = 'src/mock.ts'
|
||||
|
||||
/**
|
||||
* Refuse to emit a production bundle that contains the offline fixture backend.
|
||||
*
|
||||
* `src/mock.ts` describes an invented, healthy router: a full config, 122 nodes
|
||||
* with 119 of them alive, "Protected". It exists so `npm run dev` renders without
|
||||
* a daemon. It shipped inside the binary that goes on real hardware, switched on
|
||||
* by nothing more than a `?dev` on the end of the URL — so a link someone was
|
||||
* sent, or a bookmark they saved, showed an appliance in perfect health while
|
||||
* making no request to the appliance at all.
|
||||
*
|
||||
* api.ts now loads it behind `import.meta.env.DEV`, which Vite folds to a literal
|
||||
* `false` for a build, so Rollup drops the dynamic import and the module never
|
||||
* enters the graph. That is a property of a build tool's optimiser, and an
|
||||
* optimiser is not a promise: one refactor that makes the condition non-static
|
||||
* silently puts the fixtures back. So the property is CHECKED rather than
|
||||
* trusted — if `src/mock.ts` reaches any emitted chunk, the build fails here
|
||||
* instead of shipping.
|
||||
*/
|
||||
function assertNoMockFixtures(): Plugin {
|
||||
return {
|
||||
name: 'shater:assert-no-mock-fixtures',
|
||||
apply: 'build',
|
||||
generateBundle(_options, bundle) {
|
||||
const guilty: string[] = []
|
||||
for (const [file, output] of Object.entries(bundle)) {
|
||||
if (output.type !== 'chunk') continue
|
||||
for (const id of output.moduleIds) {
|
||||
if (id.replace(/\\/g, '/').endsWith(MOCK_MODULE)) guilty.push(`${file} ← ${id}`)
|
||||
}
|
||||
}
|
||||
if (guilty.length > 0) {
|
||||
this.error(
|
||||
`the offline fixture backend (${MOCK_MODULE}) reached the production bundle:\n ` +
|
||||
guilty.join('\n ') +
|
||||
`\nFixtures describe a router that does not exist. Keep every path to them behind ` +
|
||||
`\`import.meta.env.DEV\` so Rollup can drop them, and never gate them on a runtime ` +
|
||||
`flag such as a query parameter.`,
|
||||
)
|
||||
}
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
// The SPA is embedded in the forked sing-box binary and served by the daemon on
|
||||
// its own port. Relative base so it works under any mount path; single small
|
||||
// bundle (no code-splitting) keeps the embed simple and the flash budget low.
|
||||
export default defineConfig({
|
||||
plugins: [react()],
|
||||
plugins: [react(), assertNoMockFixtures()],
|
||||
base: './',
|
||||
build: {
|
||||
outDir: 'dist',
|
||||
|
||||
@@ -1,93 +0,0 @@
|
||||
// lx:begin awg
|
||||
|
||||
package group
|
||||
|
||||
import (
|
||||
"github.com/sagernet/sing-box/adapter"
|
||||
C "github.com/sagernet/sing-box/constant"
|
||||
)
|
||||
|
||||
// suspendAmneziaWGConsumersOnWireGuardSwitch is called from Selector.SelectOutbound
|
||||
// BEFORE the switch is committed. If the member about to be selected is — or chains
|
||||
// down via detour to — a WireGuard-based endpoint (type "wireguard", covering plain
|
||||
// WG and AmneziaWG), it walks UP from this group to every AmneziaWG endpoint that
|
||||
// detours through it and suspends each one (brings its device down). Rationale:
|
||||
// AmneziaWG traffic encapsulated inside a WireGuard tunnel hangs the kernel on
|
||||
// Android; the static Start-guard cannot cover this because a selector's chosen
|
||||
// member is only known at runtime.
|
||||
//
|
||||
// Called before s.selected is updated, so the race is closed: once the group
|
||||
// points at the WireGuard member, the AmneziaWG consumers are already suspended
|
||||
// (started=false) and a concurrent reconnect fails with "not ready" instead of
|
||||
// sending a junk handshake into WireGuard.
|
||||
func suspendAmneziaWGConsumersOnWireGuardSwitch(outboundManager adapter.OutboundManager, groupTag string, selected adapter.Outbound) {
|
||||
if outboundManager == nil || groupTag == "" {
|
||||
return
|
||||
}
|
||||
if !chainReachesWireGuard(outboundManager, selected, make(map[string]bool)) {
|
||||
return
|
||||
}
|
||||
suspendAmneziaWGConsumers(outboundManager, groupTag, make(map[string]bool))
|
||||
}
|
||||
|
||||
// chainReachesWireGuard reports whether outbound is — or transitively detours
|
||||
// down to, or (being a group) contains a member that is — a WireGuard-based
|
||||
// endpoint. visited guards against cycles.
|
||||
func chainReachesWireGuard(outboundManager adapter.OutboundManager, outbound adapter.Outbound, visited map[string]bool) bool {
|
||||
if outbound == nil {
|
||||
return false
|
||||
}
|
||||
tag := outbound.Tag()
|
||||
if tag != "" {
|
||||
if visited[tag] {
|
||||
return false
|
||||
}
|
||||
visited[tag] = true
|
||||
}
|
||||
if outbound.Type() == C.TypeWireGuard {
|
||||
return true
|
||||
}
|
||||
// Down the detour chain (vless -> ... -> wireguard).
|
||||
for _, dependency := range outbound.Dependencies() {
|
||||
if member, loaded := outboundManager.Outbound(dependency); loaded {
|
||||
if chainReachesWireGuard(outboundManager, member, visited) {
|
||||
return true
|
||||
}
|
||||
}
|
||||
}
|
||||
// A nested group: any member reaching WireGuard counts.
|
||||
if group, isGroup := outbound.(adapter.OutboundGroup); isGroup {
|
||||
for _, memberTag := range group.All() {
|
||||
if member, loaded := outboundManager.Outbound(memberTag); loaded {
|
||||
if chainReachesWireGuard(outboundManager, member, visited) {
|
||||
return true
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
// suspendAmneziaWGConsumers walks UP from tag via the reverse-dependency ledger
|
||||
// (ConsumersOf) and suspends every AmneziaWG endpoint that detours through it,
|
||||
// directly or transitively (e.g. AWG -> vless -> group). visited guards cycles.
|
||||
func suspendAmneziaWGConsumers(outboundManager adapter.OutboundManager, tag string, visited map[string]bool) {
|
||||
for _, consumerTag := range outboundManager.ConsumersOf(tag) {
|
||||
if visited[consumerTag] {
|
||||
continue
|
||||
}
|
||||
visited[consumerTag] = true
|
||||
consumer, loaded := outboundManager.Outbound(consumerTag)
|
||||
if !loaded {
|
||||
continue
|
||||
}
|
||||
if awg, isAWG := consumer.(adapter.AmneziaWGSuspendable); isAWG && awg.IsAmneziaWG() {
|
||||
awg.SuspendAmneziaWG()
|
||||
}
|
||||
// Keep walking up: a non-AWG hop (vless) or a parent group may itself have
|
||||
// an AmneziaWG consumer above it.
|
||||
suspendAmneziaWGConsumers(outboundManager, consumerTag, visited)
|
||||
}
|
||||
}
|
||||
|
||||
// lx:end awg
|
||||
@@ -1,120 +0,0 @@
|
||||
// lx:begin awg
|
||||
|
||||
package group
|
||||
|
||||
import (
|
||||
"testing"
|
||||
|
||||
"github.com/sagernet/sing-box/adapter"
|
||||
C "github.com/sagernet/sing-box/constant"
|
||||
)
|
||||
|
||||
// fakeOutbound is a minimal adapter.Outbound; only Type/Tag/Dependencies are read.
|
||||
type fakeOutbound struct {
|
||||
adapter.Outbound
|
||||
tag string
|
||||
outboundTyp string
|
||||
detour string
|
||||
}
|
||||
|
||||
func (o *fakeOutbound) Type() string { return o.outboundTyp }
|
||||
func (o *fakeOutbound) Tag() string { return o.tag }
|
||||
func (o *fakeOutbound) Dependencies() []string {
|
||||
if o.detour == "" {
|
||||
return nil
|
||||
}
|
||||
return []string{o.detour}
|
||||
}
|
||||
|
||||
// fakeAWG implements adapter.AmneziaWGSuspendable and records suspension.
|
||||
type fakeAWG struct {
|
||||
fakeOutbound
|
||||
awg bool
|
||||
suspended bool
|
||||
}
|
||||
|
||||
func (a *fakeAWG) IsAmneziaWG() bool { return a.awg }
|
||||
func (a *fakeAWG) SuspendAmneziaWG() { a.suspended = true }
|
||||
|
||||
// fakeManager resolves tags and reverse-deps (ConsumersOf) from fixed maps.
|
||||
type fakeManager struct {
|
||||
adapter.OutboundManager
|
||||
byTag map[string]adapter.Outbound
|
||||
consumers map[string][]string
|
||||
}
|
||||
|
||||
func (m *fakeManager) Outbound(tag string) (adapter.Outbound, bool) {
|
||||
ob, ok := m.byTag[tag]
|
||||
return ob, ok
|
||||
}
|
||||
func (m *fakeManager) ConsumersOf(tag string) []string { return m.consumers[tag] }
|
||||
|
||||
func TestChainReachesWireGuard(t *testing.T) {
|
||||
wg := &fakeOutbound{tag: "wg", outboundTyp: C.TypeWireGuard}
|
||||
vlessToWG := &fakeOutbound{tag: "v2wg", outboundTyp: C.TypeVLESS, detour: "wg"}
|
||||
vlessLeaf := &fakeOutbound{tag: "vleaf", outboundTyp: C.TypeVLESS}
|
||||
mgr := &fakeManager{byTag: map[string]adapter.Outbound{
|
||||
"wg": wg, "v2wg": vlessToWG, "vleaf": vlessLeaf,
|
||||
}}
|
||||
|
||||
if !chainReachesWireGuard(mgr, wg, map[string]bool{}) {
|
||||
t.Fatal("direct wireguard member must reach wireguard")
|
||||
}
|
||||
if !chainReachesWireGuard(mgr, vlessToWG, map[string]bool{}) {
|
||||
t.Fatal("vless detouring to wireguard must reach wireguard")
|
||||
}
|
||||
if chainReachesWireGuard(mgr, vlessLeaf, map[string]bool{}) {
|
||||
t.Fatal("plain vless must not reach wireguard")
|
||||
}
|
||||
}
|
||||
|
||||
func TestSuspendAmneziaWGConsumers(t *testing.T) {
|
||||
// awg-direct detours through the group "sel"
|
||||
awgDirect := &fakeAWG{fakeOutbound: fakeOutbound{tag: "awg-direct", outboundTyp: C.TypeWireGuard, detour: "sel"}, awg: true}
|
||||
// awg-via-hop -> vless-hop -> sel
|
||||
awgViaHop := &fakeAWG{fakeOutbound: fakeOutbound{tag: "awg-hop", outboundTyp: C.TypeWireGuard, detour: "vless-hop"}, awg: true}
|
||||
vlessHop := &fakeOutbound{tag: "vless-hop", outboundTyp: C.TypeVLESS, detour: "sel"}
|
||||
// plain-wg detours through sel but is NOT amneziawg — must stay untouched
|
||||
plainWG := &fakeAWG{fakeOutbound: fakeOutbound{tag: "plain-wg", outboundTyp: C.TypeWireGuard, detour: "sel"}, awg: false}
|
||||
|
||||
mgr := &fakeManager{
|
||||
byTag: map[string]adapter.Outbound{
|
||||
"awg-direct": awgDirect, "awg-hop": awgViaHop,
|
||||
"vless-hop": vlessHop, "plain-wg": plainWG,
|
||||
},
|
||||
consumers: map[string][]string{
|
||||
"sel": {"awg-direct", "vless-hop", "plain-wg"},
|
||||
"vless-hop": {"awg-hop"},
|
||||
},
|
||||
}
|
||||
|
||||
suspendAmneziaWGConsumers(mgr, "sel", map[string]bool{})
|
||||
|
||||
if !awgDirect.suspended {
|
||||
t.Error("direct AmneziaWG consumer must be suspended")
|
||||
}
|
||||
if !awgViaHop.suspended {
|
||||
t.Error("transitive AmneziaWG consumer (via vless hop) must be suspended")
|
||||
}
|
||||
if plainWG.suspended {
|
||||
t.Error("plain (non-AmneziaWG) wireguard consumer must NOT be suspended")
|
||||
}
|
||||
}
|
||||
|
||||
// A switch to a non-wireguard member must suspend nothing.
|
||||
func TestSuspendSkippedForNonWireGuardSwitch(t *testing.T) {
|
||||
awg := &fakeAWG{fakeOutbound: fakeOutbound{tag: "awg", outboundTyp: C.TypeWireGuard, detour: "sel"}, awg: true}
|
||||
vlessLeaf := &fakeOutbound{tag: "vleaf", outboundTyp: C.TypeVLESS}
|
||||
mgr := &fakeManager{
|
||||
byTag: map[string]adapter.Outbound{"awg": awg, "vleaf": vlessLeaf},
|
||||
consumers: map[string][]string{"sel": {"awg"}},
|
||||
}
|
||||
|
||||
// selected member is plain vless (does not reach wireguard) → no suspension
|
||||
suspendAmneziaWGConsumersOnWireGuardSwitch(mgr, "sel", vlessLeaf)
|
||||
if awg.suspended {
|
||||
t.Error("must not suspend when the selected member does not reach wireguard")
|
||||
}
|
||||
}
|
||||
|
||||
// lx:end awg
|
||||
@@ -128,16 +128,6 @@ func (s *Selector) SelectOutbound(tag string) bool {
|
||||
if s.selected.Load() == detour {
|
||||
return true
|
||||
}
|
||||
// lx:begin awg
|
||||
// Suspend AmneziaWG consumers BEFORE switching: if the new member is (or chains
|
||||
// to) a WireGuard endpoint, any AmneziaWG endpoint that detours through this
|
||||
// group would tunnel AWG inside WireGuard and hang the kernel on Android. Doing
|
||||
// this before s.selected.Swap closes the race — by the time the group points at
|
||||
// the WireGuard member, those consumers are already down (started=false), so a
|
||||
// concurrent reconnect fails with "not ready" instead of sending a junk
|
||||
// handshake into WireGuard.
|
||||
suspendAmneziaWGConsumersOnWireGuardSwitch(s.outbound, s.Tag(), detour)
|
||||
// lx:end awg
|
||||
s.selected.Store(detour)
|
||||
invalidateReachability(s.ctx) // lx: SPEC 020 — active selection changed
|
||||
if s.Tag() != "" {
|
||||
|
||||
+292
-45
@@ -44,6 +44,20 @@ type URLTest struct {
|
||||
group *URLTestGroup
|
||||
interruptExternalConnections bool
|
||||
balancer *balancer // lx: SPEC 019 — nil for least_test (default)
|
||||
// lx: health board §5.C — true when options.SelfCheck == false: the group's
|
||||
// OWN probing schedule (PostStart warm-up + Touch ticker) is stood down and
|
||||
// the observatory is the only thing that measures its members. Stored
|
||||
// INVERTED so the zero value keeps today's behaviour for every construction
|
||||
// path that does not go through NewURLTest (hand-built groups in tests).
|
||||
// See option.URLTestOutboundOptions.SelfCheck for the full reasoning.
|
||||
selfCheckDisabled bool
|
||||
// lx: health board §5.C — the RUNTIME half of the same question, read from
|
||||
// the context registry at construction. selfCheckDisabled above says "this
|
||||
// group is not used by the config"; this says "this group cannot be reached
|
||||
// right now", which changes while the box runs and is therefore asked
|
||||
// afresh at every scheduled check rather than stored. nil = no gate.
|
||||
// See urltest.ProbeGate for why the two must stay separate.
|
||||
probeGate urltest.ProbeGate
|
||||
}
|
||||
|
||||
func NewURLTest(ctx context.Context, router adapter.Router, logger log.ContextLogger, tag string, options option.URLTestOutboundOptions) (adapter.Outbound, error) {
|
||||
@@ -71,6 +85,12 @@ func NewURLTest(ctx context.Context, router adapter.Router, logger log.ContextLo
|
||||
idleTimeout: time.Duration(options.IdleTimeout),
|
||||
interruptExternalConnections: options.InterruptExistConnections,
|
||||
balancer: balancer,
|
||||
// nil/absent means true (self-check on) — the documented default, so a
|
||||
// config written before the flag existed behaves exactly as it always has.
|
||||
selfCheckDisabled: options.SelfCheck != nil && !*options.SelfCheck,
|
||||
// Absent from the registry (plain sing-box, tests) yields nil, which
|
||||
// means "no gate" — every scheduled probe proceeds, as before.
|
||||
probeGate: service.FromContext[urltest.ProbeGate](ctx),
|
||||
}
|
||||
if len(outbound.tags) == 0 {
|
||||
return nil, E.New("missing tags")
|
||||
@@ -92,6 +112,13 @@ func (s *URLTest) Start() error {
|
||||
return err
|
||||
}
|
||||
group.balancer = s.balancer // lx: SPEC 019 v2 — health-check drives the pool through it
|
||||
// lx: health board §5.C — carry the stand-down flag onto the group the same
|
||||
// way the balancer travels: set after construction, immutable from then on.
|
||||
group.selfCheckDisabled = s.selfCheckDisabled
|
||||
// The gate and the tag to ask it about travel together: the gate answers
|
||||
// per-outbound, and the group is the thing whose schedule is being gated.
|
||||
group.probeGate = s.probeGate
|
||||
group.tag = s.Tag()
|
||||
if s.balancer != nil {
|
||||
// lx: health board §5.B — slot liveness reads through the board verdict, so a
|
||||
// death recorded by any prober or a failed dial takes effect on the next pick,
|
||||
@@ -122,12 +149,14 @@ func (s *URLTest) Now() string {
|
||||
if s.balancer != nil {
|
||||
return s.group.lastSelected.Load()
|
||||
}
|
||||
if s.group.selectedOutboundTCP != nil {
|
||||
return s.group.selectedOutboundTCP.Tag()
|
||||
} else if s.group.selectedOutboundUDP != nil {
|
||||
return s.group.selectedOutboundUDP.Tag()
|
||||
// One load, so the two halves reported here are the SAME decision.
|
||||
selected := s.group.selected.Load()
|
||||
if selected.tcp != nil {
|
||||
return selected.tcp.Tag()
|
||||
} else if selected.udp != nil {
|
||||
return selected.udp.Tag()
|
||||
}
|
||||
// lx: SPEC 019 — cold start: before the first URL-test, selectedOutbound* is nil but
|
||||
// lx: SPEC 019 — cold start: before the first URL-test the pair is empty but
|
||||
// traffic already flows via the Select() fallback (outbounds[0] when no history yet).
|
||||
// Mirror exactly what the next DialContext would pick, so the UI shows the real node
|
||||
// instead of blank. Select() is the same source of truth DialContext uses.
|
||||
@@ -296,29 +325,155 @@ func (s *URLTest) NewPacketConnection(ctx context.Context, conn N.PacketConn, me
|
||||
}
|
||||
|
||||
type URLTestGroup struct {
|
||||
ctx context.Context
|
||||
outbound adapter.OutboundManager
|
||||
pause pause.Manager
|
||||
pauseCallback *list.Element[pause.Callback]
|
||||
logger log.Logger
|
||||
outbounds []adapter.Outbound
|
||||
link string
|
||||
interval time.Duration
|
||||
tolerance uint16
|
||||
idleTimeout time.Duration
|
||||
history *urltest.HistoryStorage
|
||||
checking atomic.Bool
|
||||
selectedOutboundTCP adapter.Outbound
|
||||
selectedOutboundUDP adapter.Outbound
|
||||
ctx context.Context
|
||||
outbound adapter.OutboundManager
|
||||
pause pause.Manager
|
||||
pauseCallback *list.Element[pause.Callback]
|
||||
logger log.Logger
|
||||
outbounds []adapter.Outbound
|
||||
link string
|
||||
interval time.Duration
|
||||
tolerance uint16
|
||||
idleTimeout time.Duration
|
||||
history *urltest.HistoryStorage
|
||||
checking atomic.Bool
|
||||
// selected is the least_test cache: the member this group currently prefers, per
|
||||
// network. It is written by the probing goroutine and read on EVERY dial through
|
||||
// the group (selectExcluding / dialSelect) and by the panel (Now), so it is an
|
||||
// atomic value rather than two plain fields — the same thing Selector does one file
|
||||
// over (selector.go, common.TypedValue[adapter.Outbound]). An interface field is two
|
||||
// words; a torn read of one hands a dial a type descriptor with the wrong data
|
||||
// pointer, which is not a wrong node but a corrupt one.
|
||||
//
|
||||
// The TCP and UDP halves live in ONE value on purpose. They are decided together, by
|
||||
// one pass over one board reading, and publishing them separately let a reader pick
|
||||
// up the new TCP choice against the previous UDP choice — the group's hysteresis
|
||||
// silently applied to a decision that was never made.
|
||||
selected common.TypedValue[selectedPair]
|
||||
interruptGroup *interrupt.Group
|
||||
interruptExternalConnections bool
|
||||
access sync.Mutex
|
||||
ticker *time.Ticker
|
||||
close chan struct{}
|
||||
started bool
|
||||
lastActive common.TypedValue[time.Time]
|
||||
lastSelected common.TypedValue[string] // lx: SPEC 019 — Now() in balanced modes
|
||||
balancer *balancer // lx: SPEC 019 v2 — round_robin pool; nil for least_test
|
||||
// started is read by Touch on every dial, outside g.access, and written by
|
||||
// PostStart under it — an atomic because that is what it always was in effect.
|
||||
started atomic.Bool
|
||||
// closed latches in Close and is what makes Close FINAL. Guarded by access.
|
||||
//
|
||||
// It exists because "has a ticker" is not the same question as "is shut down", and
|
||||
// Close used to ask the first one: with no ticker armed it returned before closing
|
||||
// g.close, leaving the group indistinguishable from a running one. A Touch arriving
|
||||
// afterwards — an outbound snapshot taken before an Apply is still dialable for up
|
||||
// to two minutes, see shater/engine/grouptest.go — then armed a fresh ticker whose
|
||||
// loopCheck waits on a channel nobody will ever close, in a box whose context is
|
||||
// already cancelled. Every tick of it fails instantly and files a "dead" verdict on
|
||||
// the SHARED health board that the live generation selects nodes from. One retired
|
||||
// group can go on declaring the whole node set dead for the uptime of the daemon.
|
||||
closed bool
|
||||
lastActive common.TypedValue[time.Time]
|
||||
lastSelected common.TypedValue[string] // lx: SPEC 019 — Now() in balanced modes
|
||||
balancer *balancer // lx: SPEC 019 v2 — round_robin pool; nil for least_test
|
||||
// lx: health board §5.C — mirrors URLTest.selfCheckDisabled (set by Start,
|
||||
// immutable afterwards, zero value = probing on). Guards ONLY the group's
|
||||
// own schedule: the PostStart warm-up sweep and the Touch ticker. An
|
||||
// explicit CheckOutbounds/URLTest call is untouched — the flag stands down
|
||||
// the schedule, not the capability.
|
||||
selfCheckDisabled bool
|
||||
// lx: health board §5.C — the runtime gate and the tag it is asked about.
|
||||
// Both mirror URLTest's fields (set by Start, immutable afterwards); nil
|
||||
// gate or empty tag means every scheduled check proceeds. Consulted only
|
||||
// through selfCheckAllowed, and only on the SCHEDULE.
|
||||
probeGate urltest.ProbeGate
|
||||
tag string
|
||||
}
|
||||
|
||||
// selectedPair is one published least_test decision: the member chosen for TCP and the
|
||||
// member chosen for UDP, as of the same probing round. Either half may be nil (nothing
|
||||
// picked yet for that network).
|
||||
type selectedPair struct {
|
||||
tcp adapter.Outbound
|
||||
udp adapter.Outbound
|
||||
}
|
||||
|
||||
// selectedFor returns the cached choice for one network (nil when there is none, or when
|
||||
// network is neither TCP nor UDP — the caller then falls through to a fresh selection).
|
||||
func (g *URLTestGroup) selectedFor(network string) adapter.Outbound {
|
||||
pair := g.selected.Load()
|
||||
switch network {
|
||||
case N.NetworkTCP:
|
||||
return pair.tcp
|
||||
case N.NetworkUDP:
|
||||
return pair.udp
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// setSelected publishes a decision. It is the ONLY writer of g.selected, and it writes
|
||||
// the pair whole — see the field comment for why the two halves may not be split.
|
||||
func (g *URLTestGroup) setSelected(tcp, udp adapter.Outbound) {
|
||||
g.selected.Store(selectedPair{tcp: tcp, udp: udp})
|
||||
}
|
||||
|
||||
// selfCheckAllowed reports whether the group's OWN probing schedule may dial
|
||||
// right now. lx: health board §5.C.
|
||||
//
|
||||
// Two independent refusals, in the order they can be answered cheapest first:
|
||||
//
|
||||
// selfCheckDisabled — the config says no rule reaches this group. Fixed for
|
||||
// the life of the box; see standDownUnusedSelfCheck.
|
||||
// probeGate — the world says this group cannot be reached right now,
|
||||
// typically a chain hop sitting behind a dead hop. Asked
|
||||
// fresh EVERY time, which is the entire mechanism by which
|
||||
// a recovered hop resumes probing: there is no state here
|
||||
// to reset, so there is none to get stuck.
|
||||
//
|
||||
// Neither refusal touches an explicit CheckOutbounds/URLTest — a deliberate
|
||||
// request is never a scheduled one.
|
||||
func (g *URLTestGroup) selfCheckAllowed() bool {
|
||||
if g.selfCheckDisabled {
|
||||
return false
|
||||
}
|
||||
if g.probeGate == nil || g.tag == "" {
|
||||
return true
|
||||
}
|
||||
return g.probeGate.ProbeAllowed(g.tag)
|
||||
}
|
||||
|
||||
// scheduledCheck is one firing of the group's own schedule — the warm-up sweep
|
||||
// and every ticker tick go through here, and nothing else does. Having exactly
|
||||
// one gated entry point is what keeps the two callers from drifting apart, and
|
||||
// it is the seam the tests drive to assert that a gated group makes no dial
|
||||
// attempt at all.
|
||||
func (g *URLTestGroup) scheduledCheck() {
|
||||
if !g.selfCheckAllowed() {
|
||||
return
|
||||
}
|
||||
g.CheckOutbounds(false)
|
||||
}
|
||||
|
||||
// keepWarm reports whether this group must keep measuring with no traffic
|
||||
// flowing through it. lx: health board §5.C — see urltest.ProbeGate.ProbeWhenIdle.
|
||||
//
|
||||
// The default is NO, in every direction: no gate, no tag, or a group whose
|
||||
// self-check is stood down anyway. Only a gate that positively says "the routing
|
||||
// config reaches this group" turns the idle timeout off, so plain sing-box and
|
||||
// every hand-built group keep the lifecycle they have always had.
|
||||
func (g *URLTestGroup) keepWarm() bool {
|
||||
if g.selfCheckDisabled || g.probeGate == nil || g.tag == "" {
|
||||
return false
|
||||
}
|
||||
return g.probeGate.ProbeWhenIdle(g.tag)
|
||||
}
|
||||
|
||||
// startTickerLocked arms the group's own probing ticker. g.access MUST be held
|
||||
// and g.ticker MUST be nil. Extracted so PostStart and Touch arm it identically
|
||||
// — two ways in, one construction, no chance of one of them forgetting the pause
|
||||
// registration.
|
||||
func (g *URLTestGroup) startTickerLocked() {
|
||||
ticker := time.NewTicker(g.interval)
|
||||
g.ticker = ticker
|
||||
g.pauseCallback = pause.RegisterTicker(g.pause, ticker, g.interval, nil)
|
||||
go g.loopCheck(ticker, g.close)
|
||||
}
|
||||
|
||||
func NewURLTestGroup(ctx context.Context, outboundManager adapter.OutboundManager, logger log.Logger, outbounds []adapter.Outbound, link string, interval time.Duration, tolerance uint16, idleTimeout time.Duration, interruptExternalConnections bool) (*URLTestGroup, error) {
|
||||
@@ -358,41 +513,112 @@ func NewURLTestGroup(ctx context.Context, outboundManager adapter.OutboundManage
|
||||
func (g *URLTestGroup) PostStart() {
|
||||
g.access.Lock()
|
||||
defer g.access.Unlock()
|
||||
g.started = true
|
||||
if g.closed {
|
||||
return
|
||||
}
|
||||
g.started.Store(true)
|
||||
g.lastActive.Store(time.Now())
|
||||
// lx: SPEC 019 v2 — seed the pool so round_robin can route from the first connection,
|
||||
// before the first health-check completes (history-warm nodes first, else config order).
|
||||
// The seed only READS the board, so it runs even with the self-check stood down.
|
||||
g.seedPool()
|
||||
go g.CheckOutbounds(false)
|
||||
// lx: health board §5.C — the warm-up sweep is the first half of the group's
|
||||
// own probing schedule, and it fires for EVERY group at box start, including
|
||||
// groups no routing rule reaches. For those, the sweep dials every member
|
||||
// directly from the router — a path nothing uses — and records the outcome
|
||||
// under the members' base tags, forging the board reading the observatory
|
||||
// exists to keep honest. A stood-down group therefore skips it entirely; the
|
||||
// observatory (or nothing, for a truly unused group) is what measures its
|
||||
// members.
|
||||
//
|
||||
// The same call is now also where a chain hop behind a DEAD hop declines to
|
||||
// sweep: every member of such a group dials through the broken hop, so the
|
||||
// sweep would measure that hop once per member and file the result against
|
||||
// this one. selfCheckAllowed keeps both refusals in one place.
|
||||
go g.scheduledCheck()
|
||||
// A group the routing config REACHES keeps measuring whether or not anybody
|
||||
// dials it, so its ticker is armed here instead of waiting for a Touch that
|
||||
// may never come. Without this, a used group with no traffic gets this one
|
||||
// warm-up sweep and then nothing: its members age past the verdict TTL and
|
||||
// the panel reports "untested" about a rule that is in force, while the first
|
||||
// real request pays a cold probe. Nothing else would fill the gap — the
|
||||
// observatory stands off a urltest group's members entirely (probeplan.go
|
||||
// SelfChecked), which is the whole point of one dialler per target.
|
||||
//
|
||||
// lastActive was stored a moment ago, so loopCheck's opening "idle longer
|
||||
// than the interval" check does not fire and this cannot double up with the
|
||||
// sweep above.
|
||||
if g.keepWarm() && g.ticker == nil {
|
||||
g.startTickerLocked()
|
||||
}
|
||||
}
|
||||
|
||||
func (g *URLTestGroup) Touch() {
|
||||
if !g.started {
|
||||
if !g.started.Load() {
|
||||
return
|
||||
}
|
||||
// lx: health board §5.C — Touch's only job is to keep the group's OWN
|
||||
// probing ticker alive while traffic flows. With the self-check stood down
|
||||
// there is deliberately no ticker to start or feed: the observatory owns the
|
||||
// schedule, and a stray dial through an unused group (a stale rule cache, a
|
||||
// manual pin) must not arm 30 minutes of direct probing under the members'
|
||||
// base tags. Checked before the lock because the flag is immutable after
|
||||
// Start, exactly like the started fast-path above.
|
||||
//
|
||||
// The runtime gate is deliberately NOT consulted here. Touch only arms the
|
||||
// ticker; refusing to arm it would mean a hop that recovers has no ticker
|
||||
// left to notice — the block would outlive the failure, which is the one
|
||||
// outcome this must never have. The ticker runs and each tick re-asks the
|
||||
// gate (loopCheck -> scheduledCheck), so a blocked hop costs a predicate
|
||||
// call per interval and resumes the moment the hop in front answers.
|
||||
if g.selfCheckDisabled {
|
||||
return
|
||||
}
|
||||
g.access.Lock()
|
||||
defer g.access.Unlock()
|
||||
// A closed group arms nothing. Touch is reachable long after Close — a caller
|
||||
// holding an outbound from a snapshot taken before an Apply keeps dialling it (up
|
||||
// to the 120s budget of shater/engine/grouptest.go) — and the ticker it would arm
|
||||
// has no way left to stop: see the `closed` field for what that costs.
|
||||
if g.closed {
|
||||
return
|
||||
}
|
||||
if g.ticker != nil {
|
||||
g.lastActive.Store(time.Now())
|
||||
return
|
||||
}
|
||||
ticker := time.NewTicker(g.interval)
|
||||
g.ticker = ticker
|
||||
g.pauseCallback = pause.RegisterTicker(g.pause, ticker, g.interval, nil)
|
||||
go g.loopCheck(ticker, g.close)
|
||||
g.startTickerLocked()
|
||||
}
|
||||
|
||||
// Close shuts the group down for good. It is idempotent, and it is FINAL: no later Touch
|
||||
// can bring the probing schedule back.
|
||||
//
|
||||
// It used to return early when no ticker happened to be armed, without ever closing
|
||||
// g.close — so a group that was closed while idle stayed, from the point of view of every
|
||||
// other method, a perfectly live group. That is the whole defect: the close channel is the
|
||||
// only way a loopCheck goroutine ever exits (its idle-timeout escape does not fire for a
|
||||
// group the routing config reaches, keepWarm), so a ticker armed after such a Close is
|
||||
// immortal, and every one of its ticks writes a failure to the shared health board on
|
||||
// behalf of a box that no longer exists.
|
||||
func (g *URLTestGroup) Close() error {
|
||||
g.access.Lock()
|
||||
defer g.access.Unlock()
|
||||
if g.ticker == nil {
|
||||
if g.closed {
|
||||
return nil
|
||||
}
|
||||
g.ticker.Stop()
|
||||
g.ticker = nil
|
||||
g.pause.UnregisterCallback(g.pauseCallback)
|
||||
g.pauseCallback = nil
|
||||
close(g.close)
|
||||
g.closed = true
|
||||
// Unconditionally, BEFORE looking at the ticker: this is the signal every loopCheck
|
||||
// waits on, including any that a Touch armed after the last one was retired by the
|
||||
// idle timeout.
|
||||
if g.close != nil {
|
||||
close(g.close)
|
||||
}
|
||||
if g.ticker != nil {
|
||||
g.ticker.Stop()
|
||||
g.ticker = nil
|
||||
g.pause.UnregisterCallback(g.pauseCallback)
|
||||
g.pauseCallback = nil
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
@@ -405,10 +631,15 @@ func (g *URLTestGroup) Select(network string) (adapter.Outbound, bool) {
|
||||
return g.selectExcluding(network, nil)
|
||||
}
|
||||
|
||||
// loopCheck is the group's own schedule. lx: health board §5.C — every probe it
|
||||
// fires goes through scheduledCheck, so a stood-down or currently-unreachable
|
||||
// group ticks without dialling. The ticker's LIFECYCLE (the idle timeout below)
|
||||
// is deliberately left alone: a gated group keeps its ticker exactly as long as
|
||||
// an ungated one would, because the ticker is what will notice the recovery.
|
||||
func (g *URLTestGroup) loopCheck(ticker *time.Ticker, closeChan <-chan struct{}) {
|
||||
if time.Since(g.lastActive.Load()) > g.interval {
|
||||
g.lastActive.Store(time.Now())
|
||||
g.CheckOutbounds(false)
|
||||
g.scheduledCheck()
|
||||
}
|
||||
for {
|
||||
select {
|
||||
@@ -416,7 +647,13 @@ func (g *URLTestGroup) loopCheck(ticker *time.Ticker, closeChan <-chan struct{})
|
||||
return
|
||||
case <-ticker.C:
|
||||
}
|
||||
if time.Since(g.lastActive.Load()) > g.idleTimeout {
|
||||
// The idle timeout retires the ticker of a group nobody is dialling —
|
||||
// unless the routing config reaches it, in which case its health is a
|
||||
// live question whether or not traffic is flowing and the ticker must
|
||||
// outlive the silence. Asked here rather than remembered from PostStart
|
||||
// so it tracks the running config, and asked OUTSIDE g.access because the
|
||||
// answer comes from the engine, which has locks of its own.
|
||||
if !g.keepWarm() && time.Since(g.lastActive.Load()) > g.idleTimeout {
|
||||
g.access.Lock()
|
||||
if g.ticker == ticker {
|
||||
g.ticker.Stop()
|
||||
@@ -427,7 +664,7 @@ func (g *URLTestGroup) loopCheck(ticker *time.Ticker, closeChan <-chan struct{})
|
||||
g.access.Unlock()
|
||||
return
|
||||
}
|
||||
g.CheckOutbounds(false)
|
||||
g.scheduledCheck()
|
||||
}
|
||||
}
|
||||
|
||||
@@ -525,19 +762,29 @@ func (g *URLTestGroup) testNodes(ctx context.Context, outbounds []adapter.Outbou
|
||||
return result
|
||||
}
|
||||
|
||||
// performUpdateCheck re-ranks the members after a probing round and publishes the result.
|
||||
// It is the only writer of g.selected: it reads the current pair ONCE, decides both
|
||||
// networks against that one snapshot, and stores the outcome as a single value, so no
|
||||
// reader can ever observe a half-applied decision. Callers are serialised by g.checking
|
||||
// (urlTest), which is what makes the read-decide-store sequence safe without a lock.
|
||||
func (g *URLTestGroup) performUpdateCheck() {
|
||||
current := g.selected.Load()
|
||||
next := current
|
||||
var updated bool
|
||||
if outbound, exists := g.Select(N.NetworkTCP); outbound != nil && (g.selectedOutboundTCP == nil || (exists && outbound != g.selectedOutboundTCP)) {
|
||||
if g.selectedOutboundTCP != nil {
|
||||
if outbound, exists := g.Select(N.NetworkTCP); outbound != nil && (current.tcp == nil || (exists && outbound != current.tcp)) {
|
||||
if current.tcp != nil {
|
||||
updated = true
|
||||
}
|
||||
g.selectedOutboundTCP = outbound
|
||||
next.tcp = outbound
|
||||
}
|
||||
if outbound, exists := g.Select(N.NetworkUDP); outbound != nil && (g.selectedOutboundUDP == nil || (exists && outbound != g.selectedOutboundUDP)) {
|
||||
if g.selectedOutboundUDP != nil {
|
||||
if outbound, exists := g.Select(N.NetworkUDP); outbound != nil && (current.udp == nil || (exists && outbound != current.udp)) {
|
||||
if current.udp != nil {
|
||||
updated = true
|
||||
}
|
||||
g.selectedOutboundUDP = outbound
|
||||
next.udp = outbound
|
||||
}
|
||||
if next != current {
|
||||
g.setSelected(next.tcp, next.udp)
|
||||
}
|
||||
if updated {
|
||||
g.interruptGroup.Interrupt(g.interruptExternalConnections)
|
||||
|
||||
@@ -74,13 +74,7 @@ func (g *URLTestGroup) selectExcluding(network string, exclude map[string]bool)
|
||||
var minOutbound adapter.Outbound
|
||||
// Keep the upstream hysteresis: the currently selected outbound only yields to a
|
||||
// member faster by more than tolerance — but only while it is still alive itself.
|
||||
var current adapter.Outbound
|
||||
switch network {
|
||||
case N.NetworkTCP:
|
||||
current = g.selectedOutboundTCP
|
||||
case N.NetworkUDP:
|
||||
current = g.selectedOutboundUDP
|
||||
}
|
||||
current := g.selectedFor(network)
|
||||
if current != nil {
|
||||
currentTag := RealTag(current)
|
||||
if !exclude[currentTag] && g.history.Verdict(currentTag, ttl) == urltest.VerdictAlive {
|
||||
@@ -144,13 +138,7 @@ func (s *URLTest) dialSelect(ctx context.Context, network string, destination M.
|
||||
if s.balancer != nil {
|
||||
return s.selectBalanced(ctx, network, destination, tried)
|
||||
}
|
||||
var outbound adapter.Outbound
|
||||
switch N.NetworkName(network) {
|
||||
case N.NetworkTCP:
|
||||
outbound = s.group.selectedOutboundTCP
|
||||
case N.NetworkUDP:
|
||||
outbound = s.group.selectedOutboundUDP
|
||||
}
|
||||
outbound := s.group.selectedFor(N.NetworkName(network))
|
||||
if outbound != nil {
|
||||
realTag := RealTag(outbound)
|
||||
if !tried[realTag] && s.group.history.Verdict(realTag, s.group.healthTTL()) != urltest.VerdictDead {
|
||||
|
||||
@@ -7,6 +7,7 @@ import (
|
||||
"context"
|
||||
"errors"
|
||||
"net"
|
||||
"sync"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
@@ -32,10 +33,31 @@ type healthNode struct {
|
||||
func (n *healthNode) Tag() string { return n.tag }
|
||||
func (n *healthNode) Network() []string { return []string{N.NetworkTCP, N.NetworkUDP} }
|
||||
|
||||
func (n *healthNode) DialContext(ctx context.Context, network string, destination M.Socksaddr) (net.Conn, error) {
|
||||
if n.dialed != nil {
|
||||
*n.dialed = append(*n.dialed, n.tag)
|
||||
// healthDialMu guards the shared dialed slice. testNodes probes a group's
|
||||
// members CONCURRENTLY (a batch of 10), so two nodes pointing at one slice
|
||||
// append from two goroutines; the append is what the -race build trips on, not
|
||||
// the code under test.
|
||||
var healthDialMu sync.Mutex
|
||||
|
||||
func (n *healthNode) record() {
|
||||
if n.dialed == nil {
|
||||
return
|
||||
}
|
||||
healthDialMu.Lock()
|
||||
*n.dialed = append(*n.dialed, n.tag)
|
||||
healthDialMu.Unlock()
|
||||
}
|
||||
|
||||
// dialsOf reads a dial log under the same lock. Every assertion on a log a
|
||||
// concurrent sweep may still be writing must go through it.
|
||||
func dialsOf(dialed *[]string) []string {
|
||||
healthDialMu.Lock()
|
||||
defer healthDialMu.Unlock()
|
||||
return append([]string(nil), *dialed...)
|
||||
}
|
||||
|
||||
func (n *healthNode) DialContext(ctx context.Context, network string, destination M.Socksaddr) (net.Conn, error) {
|
||||
n.record()
|
||||
if n.fail {
|
||||
return nil, errors.New("dial refused")
|
||||
}
|
||||
@@ -45,9 +67,7 @@ func (n *healthNode) DialContext(ctx context.Context, network string, destinatio
|
||||
}
|
||||
|
||||
func (n *healthNode) ListenPacket(ctx context.Context, destination M.Socksaddr) (net.PacketConn, error) {
|
||||
if n.dialed != nil {
|
||||
*n.dialed = append(*n.dialed, n.tag)
|
||||
}
|
||||
n.record()
|
||||
if n.fail {
|
||||
return nil, errors.New("listen refused")
|
||||
}
|
||||
@@ -85,6 +105,9 @@ func healthTestGroup(hist *urltest.HistoryStorage, manager adapter.OutboundManag
|
||||
tolerance: 50,
|
||||
logger: log.NewNOPFactory().Logger(),
|
||||
interruptGroup: interrupt.NewGroup(),
|
||||
// Real groups always have this channel (NewURLTestGroup); it is what Close
|
||||
// signals every loopCheck through, so a hand-built group needs it too.
|
||||
close: make(chan struct{}),
|
||||
}
|
||||
}
|
||||
|
||||
@@ -161,7 +184,7 @@ func TestSelectHysteresisKeepsAliveCurrent(t *testing.T) {
|
||||
hist := urltest.NewHistoryStorage()
|
||||
a, b := &balNode{tag: "a"}, &balNode{tag: "b"}
|
||||
g := healthTestGroup(hist, nil, a, b)
|
||||
g.selectedOutboundTCP = a
|
||||
g.setSelected(a, nil)
|
||||
storeAlive(hist, "a", 100)
|
||||
storeAlive(hist, "b", 60) // within tolerance (100 ≤ 60+50) → keep a
|
||||
if selected, _ := g.Select(N.NetworkTCP); selected != adapter.Outbound(a) {
|
||||
@@ -178,7 +201,7 @@ func TestSelectDeadCurrentLosesToAlive(t *testing.T) {
|
||||
hist := urltest.NewHistoryStorage()
|
||||
a, b := &balNode{tag: "a"}, &balNode{tag: "b"}
|
||||
g := healthTestGroup(hist, nil, a, b)
|
||||
g.selectedOutboundTCP = a
|
||||
g.setSelected(a, nil)
|
||||
storeAlive(hist, "a", 10)
|
||||
hist.MarkFailed("a")
|
||||
storeAlive(hist, "b", 500)
|
||||
@@ -331,7 +354,7 @@ func TestDialContextRetriesThroughNextAlive(t *testing.T) {
|
||||
s := healthURLTest(g, nil, nil)
|
||||
storeAlive(hist, "a", 10)
|
||||
storeAlive(hist, "b", 100)
|
||||
g.selectedOutboundTCP = a // the checker had picked a; it dies between ticks
|
||||
g.setSelected(a, nil) // the checker had picked a; it dies between ticks
|
||||
conn, err := s.DialContext(context.Background(), N.NetworkTCP, destDomain("example.com"))
|
||||
if err != nil {
|
||||
t.Fatalf("DialContext failed despite live member b: %v", err)
|
||||
@@ -427,7 +450,7 @@ func TestListenPacketRetriesBeforeFirstSend(t *testing.T) {
|
||||
s := healthURLTest(g, nil, nil)
|
||||
storeAlive(hist, "a", 10)
|
||||
storeAlive(hist, "b", 100)
|
||||
g.selectedOutboundUDP = a
|
||||
g.setSelected(nil, a)
|
||||
conn, err := s.ListenPacket(context.Background(), destDomain("example.com"))
|
||||
if err != nil {
|
||||
t.Fatalf("ListenPacket failed despite live member b: %v", err)
|
||||
|
||||
@@ -0,0 +1,330 @@
|
||||
package group
|
||||
|
||||
// Concurrency tests for the urltest group: the group's own probing schedule running at
|
||||
// the same time as traffic going through it.
|
||||
//
|
||||
// This combination had no coverage at all. Every existing test either probes OR dials,
|
||||
// never both at once, so the race detector had nothing to detect: the cached least_test
|
||||
// choice was written by the prober goroutine and read on every single dial, with no
|
||||
// synchronisation whatsoever, and the suite stayed green for as long as those two things
|
||||
// never happened in the same test.
|
||||
//
|
||||
// An unsynchronised interface field is not a "usually fine" race. It is two words — type
|
||||
// descriptor and data pointer — and a reader that catches the store half way holds a
|
||||
// descriptor addressing the wrong value. What comes out is not a suboptimal node, it is a
|
||||
// corrupt one, on the path of every connection the group carries.
|
||||
|
||||
import (
|
||||
"context"
|
||||
"sync"
|
||||
"sync/atomic"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"github.com/sagernet/sing-box/adapter"
|
||||
"github.com/sagernet/sing-box/common/urltest"
|
||||
M "github.com/sagernet/sing/common/metadata"
|
||||
N "github.com/sagernet/sing/common/network"
|
||||
"github.com/sagernet/sing/service/pause"
|
||||
)
|
||||
|
||||
// TestURLTestDialRacesProbeTicker drives real dials through a least_test group while the
|
||||
// group's own probing schedule keeps re-deciding which member to use — the situation on
|
||||
// every router where a urltest group carries traffic, since the ticker fires on its own
|
||||
// interval regardless of what the connections are doing.
|
||||
//
|
||||
// It is a -race test first and an assertion test second: the failure it was written for
|
||||
// is reported by the detector, not by a wrong value. Run it under -race or it proves
|
||||
// almost nothing (the gate's [4/4] pass does).
|
||||
func TestURLTestDialRacesProbeTicker(t *testing.T) {
|
||||
hist := urltest.NewHistoryStorage()
|
||||
a, b := &healthNode{tag: "a"}, &healthNode{tag: "b"}
|
||||
manager := managerOf(a, b)
|
||||
g := healthTestGroup(hist, manager, a, b)
|
||||
s := healthURLTest(g, nil, manager)
|
||||
storeAlive(hist, "a", 20)
|
||||
storeAlive(hist, "b", 500)
|
||||
|
||||
var proberWG, dialWG sync.WaitGroup
|
||||
stop := make(chan struct{})
|
||||
|
||||
// The prober: one full turn of the group's own schedule per iteration. CheckOutbounds
|
||||
// is the real tick (probe every member, then publish); the probes cannot reach
|
||||
// anything from a unit test, so the board is then re-armed with a winner that MOVES
|
||||
// and the publish step is run again — otherwise the cached choice is written once and
|
||||
// the window in which a reader can catch a torn write is a few nanoseconds wide.
|
||||
proberWG.Add(1)
|
||||
go func() {
|
||||
defer proberWG.Done()
|
||||
for i := 0; ; i++ {
|
||||
select {
|
||||
case <-stop:
|
||||
return
|
||||
default:
|
||||
}
|
||||
g.CheckOutbounds(true)
|
||||
fast, slow := "a", "b"
|
||||
if i%2 == 1 {
|
||||
fast, slow = "b", "a"
|
||||
}
|
||||
storeAlive(hist, fast, 20)
|
||||
storeAlive(hist, slow, 500)
|
||||
g.performUpdateCheck()
|
||||
}
|
||||
}()
|
||||
|
||||
// The traffic: every dial reads the cached choice (dialSelect), and so does the panel
|
||||
// (Now). Both are the read side of the race.
|
||||
var dials, nows atomic.Int64
|
||||
for range 4 {
|
||||
dialWG.Add(1)
|
||||
go func() {
|
||||
defer dialWG.Done()
|
||||
for range 300 {
|
||||
if conn, err := s.DialContext(context.Background(), N.NetworkTCP, M.Socksaddr{}); err == nil {
|
||||
_ = conn.Close()
|
||||
dials.Add(1)
|
||||
}
|
||||
if pc, err := s.ListenPacket(context.Background(), M.Socksaddr{}); err == nil {
|
||||
_ = pc.Close()
|
||||
}
|
||||
// The panel polls this while everything above is happening.
|
||||
if tag := s.Now(); tag != "" && tag != "a" && tag != "b" {
|
||||
t.Errorf("Now() = %q, which is not a member of the group — a torn read of the cached choice", tag)
|
||||
}
|
||||
nows.Add(1)
|
||||
}
|
||||
}()
|
||||
}
|
||||
|
||||
// The dialers are the bounded side; the prober runs until they are done.
|
||||
dialersDone := make(chan struct{})
|
||||
go func() { dialWG.Wait(); close(dialersDone) }()
|
||||
select {
|
||||
case <-dialersDone:
|
||||
case <-time.After(60 * time.Second):
|
||||
close(stop)
|
||||
proberWG.Wait()
|
||||
t.Fatal("dialers did not finish — the group deadlocked against its own prober")
|
||||
}
|
||||
close(stop)
|
||||
proberWG.Wait()
|
||||
|
||||
if dials.Load() == 0 {
|
||||
t.Fatal("no dial succeeded — the test never exercised the read side it exists to race")
|
||||
}
|
||||
if nows.Load() == 0 {
|
||||
t.Fatal("Now() was never polled")
|
||||
}
|
||||
}
|
||||
|
||||
// TestSelectedPairPublishedTogether pins the pairing half of the same defect: the TCP and
|
||||
// UDP choices are one decision, taken from one board reading, and they become visible
|
||||
// together. They used to be two separate field writes, so a reader could take the new TCP
|
||||
// choice against the previous UDP one — a combination no probing round ever decided, and
|
||||
// the group's hysteresis silently applied to it.
|
||||
func TestSelectedPairPublishedTogether(t *testing.T) {
|
||||
hist := urltest.NewHistoryStorage()
|
||||
a, b := &healthNode{tag: "a"}, &healthNode{tag: "b"}
|
||||
g := healthTestGroup(hist, managerOf(a, b), a, b)
|
||||
|
||||
// Round 1: a wins both networks.
|
||||
storeAlive(hist, "a", 20)
|
||||
storeAlive(hist, "b", 500)
|
||||
g.performUpdateCheck()
|
||||
if got := g.selected.Load(); got.tcp != adapter.Outbound(a) || got.udp != adapter.Outbound(a) {
|
||||
t.Fatalf("after round 1 the pair is (%v, %v), want (a, a)", tagOrNil(got.tcp), tagOrNil(got.udp))
|
||||
}
|
||||
|
||||
// Round 2: b wins both, by more than the tolerance.
|
||||
storeAlive(hist, "a", 500)
|
||||
storeAlive(hist, "b", 20)
|
||||
g.performUpdateCheck()
|
||||
got := g.selected.Load()
|
||||
if got.tcp != adapter.Outbound(b) || got.udp != adapter.Outbound(b) {
|
||||
t.Fatalf("after round 2 the pair is (%v, %v), want (b, b) — both halves move together",
|
||||
tagOrNil(got.tcp), tagOrNil(got.udp))
|
||||
}
|
||||
|
||||
// And what the dial path reads per network agrees with the published pair.
|
||||
if g.selectedFor(N.NetworkTCP) != got.tcp || g.selectedFor(N.NetworkUDP) != got.udp {
|
||||
t.Fatal("selectedFor disagrees with the published pair")
|
||||
}
|
||||
if g.selectedFor("icmp") != nil {
|
||||
t.Fatal("selectedFor on an unknown network must yield nothing, not a TCP choice")
|
||||
}
|
||||
}
|
||||
|
||||
// TestURLTestGroupProbeRacesPanelRead is the narrower of the pair: the panel's Now() poll
|
||||
// against the prober, with no dialling at all. shater/engine/grouphealth.go,
|
||||
// shater/stats/stats.go and shater/engine/grouptest.go all call Now() from their own
|
||||
// goroutines while the group's ticker runs.
|
||||
func TestURLTestGroupProbeRacesPanelRead(t *testing.T) {
|
||||
hist := urltest.NewHistoryStorage()
|
||||
a, b := &healthNode{tag: "a"}, &healthNode{tag: "b"}
|
||||
manager := managerOf(a, b)
|
||||
g := healthTestGroup(hist, manager, a, b)
|
||||
s := healthURLTest(g, nil, manager)
|
||||
|
||||
var wg sync.WaitGroup
|
||||
stop := make(chan struct{})
|
||||
wg.Add(1)
|
||||
go func() {
|
||||
defer wg.Done()
|
||||
for i := 0; ; i++ {
|
||||
select {
|
||||
case <-stop:
|
||||
return
|
||||
default:
|
||||
}
|
||||
fast, slow := "a", "b"
|
||||
if i%2 == 1 {
|
||||
fast, slow = "b", "a"
|
||||
}
|
||||
storeAlive(hist, fast, 20)
|
||||
storeAlive(hist, slow, 500)
|
||||
g.performUpdateCheck()
|
||||
}
|
||||
}()
|
||||
for range 3 {
|
||||
wg.Add(1)
|
||||
go func() {
|
||||
defer wg.Done()
|
||||
for range 2000 {
|
||||
_ = s.Now()
|
||||
}
|
||||
}()
|
||||
}
|
||||
time.Sleep(50 * time.Millisecond)
|
||||
close(stop)
|
||||
wg.Wait()
|
||||
}
|
||||
|
||||
// tagOrNil renders a possibly-nil outbound for a failure message.
|
||||
func tagOrNil(o adapter.Outbound) string {
|
||||
if o == nil {
|
||||
return "<nil>"
|
||||
}
|
||||
return o.Tag()
|
||||
}
|
||||
|
||||
// --- Close is final ---------------------------------------------------------
|
||||
|
||||
// TestGroupCloseIsFinalForALaterTouch is the immortal-ticker regression.
|
||||
//
|
||||
// Close used to return early whenever no ticker happened to be armed — which is the
|
||||
// normal state of a group nobody is dialling — WITHOUT closing g.close. Nothing else
|
||||
// records that a group was shut down (started is never cleared), so a Touch arriving
|
||||
// afterwards armed a fresh ticker and a fresh loopCheck goroutine waiting on a channel
|
||||
// that would never be closed. Its only other exit, the idle timeout, does not fire for a
|
||||
// group the routing config reaches.
|
||||
//
|
||||
// A later Touch is not hypothetical: shater/engine/grouptest.go dials through outbounds
|
||||
// taken from a snapshot at the start of a run and keeps doing so for up to 120s, so
|
||||
// "press Test in the panel, then apply a config within two minutes" is enough. The
|
||||
// retired group then probes forever through a cancelled context — every probe fails
|
||||
// instantly — and files "dead" for its members on the SHARED health board that the LIVE
|
||||
// generation picks nodes from.
|
||||
func TestGroupCloseIsFinalForALaterTouch(t *testing.T) {
|
||||
hist := urltest.NewHistoryStorage()
|
||||
defer hist.Close()
|
||||
var dialed []string
|
||||
a := &healthNode{tag: "a", fail: true, dialed: &dialed}
|
||||
g := healthTestGroup(hist, managerOf(a), a)
|
||||
g.pause = pause.ManagerFromContext(pause.WithDefaultManager(context.Background()))
|
||||
// Fast enough that a surviving ticker proves itself within the test's patience.
|
||||
g.interval = 10 * time.Millisecond
|
||||
g.idleTimeout = time.Hour
|
||||
// Started, but idle: no ticker armed. This is the state Close mishandled.
|
||||
g.started.Store(true)
|
||||
|
||||
if err := g.Close(); err != nil {
|
||||
t.Fatalf("Close: %v", err)
|
||||
}
|
||||
g.Touch()
|
||||
|
||||
g.access.Lock()
|
||||
ticker := g.ticker
|
||||
g.access.Unlock()
|
||||
if ticker != nil {
|
||||
t.Fatal("Touch armed a probing ticker on a CLOSED group — nothing can stop it: " +
|
||||
"its loopCheck waits on a channel that will never be closed")
|
||||
}
|
||||
|
||||
// The consequence, stated in the terms that actually hurt: no probe, so no forged
|
||||
// verdict on the shared board.
|
||||
time.Sleep(150 * time.Millisecond)
|
||||
if got := dialsOf(&dialed); len(got) != 0 {
|
||||
t.Fatalf("a closed group dialled %v — a retired generation is writing to the live health board", got)
|
||||
}
|
||||
if v := hist.Verdict("a", 10*time.Minute); v != urltest.VerdictUntested {
|
||||
t.Fatalf("verdict(a) = %v after closing the group, want untested — the dead marks are forged", v)
|
||||
}
|
||||
}
|
||||
|
||||
// TestGroupCloseStopsAnArmedTicker keeps the original behaviour honest: when a ticker IS
|
||||
// armed, Close still stops it, unregisters the pause callback and signals loopCheck.
|
||||
func TestGroupCloseStopsAnArmedTicker(t *testing.T) {
|
||||
hist := urltest.NewHistoryStorage()
|
||||
defer hist.Close()
|
||||
a := &healthNode{tag: "a", fail: true}
|
||||
g := healthTestGroup(hist, managerOf(a), a)
|
||||
g.pause = pause.ManagerFromContext(pause.WithDefaultManager(context.Background()))
|
||||
g.interval = 10 * time.Millisecond
|
||||
g.idleTimeout = time.Hour
|
||||
g.started.Store(true)
|
||||
g.lastActive.Store(time.Now())
|
||||
|
||||
g.Touch()
|
||||
g.access.Lock()
|
||||
armed := g.ticker != nil
|
||||
g.access.Unlock()
|
||||
if !armed {
|
||||
t.Fatal("Touch did not arm the ticker on a live group")
|
||||
}
|
||||
|
||||
if err := g.Close(); err != nil {
|
||||
t.Fatalf("Close: %v", err)
|
||||
}
|
||||
g.access.Lock()
|
||||
stillArmed := g.ticker != nil
|
||||
g.access.Unlock()
|
||||
if stillArmed {
|
||||
t.Fatal("Close left the ticker armed")
|
||||
}
|
||||
select {
|
||||
case <-g.close:
|
||||
default:
|
||||
t.Fatal("Close did not signal loopCheck")
|
||||
}
|
||||
// Idempotent: a second Close must not close an already-closed channel (panic) or
|
||||
// undo anything.
|
||||
if err := g.Close(); err != nil {
|
||||
t.Fatalf("second Close: %v", err)
|
||||
}
|
||||
}
|
||||
|
||||
// TestGroupPostStartAfterCloseDoesNothing: the other way a retired group can be woken.
|
||||
// PostStart is called on every member of a box at start-up; a group closed by a racing
|
||||
// shutdown must not be brought back by it.
|
||||
func TestGroupPostStartAfterCloseDoesNothing(t *testing.T) {
|
||||
hist := urltest.NewHistoryStorage()
|
||||
defer hist.Close()
|
||||
var dialed []string
|
||||
a := &healthNode{tag: "a", fail: true, dialed: &dialed}
|
||||
g := healthTestGroup(hist, managerOf(a), a)
|
||||
g.pause = pause.ManagerFromContext(pause.WithDefaultManager(context.Background()))
|
||||
g.interval = 10 * time.Millisecond
|
||||
g.idleTimeout = time.Hour
|
||||
|
||||
_ = g.Close()
|
||||
g.PostStart()
|
||||
time.Sleep(150 * time.Millisecond)
|
||||
|
||||
if g.started.Load() {
|
||||
t.Fatal("PostStart marked a closed group as started")
|
||||
}
|
||||
if got := dialsOf(&dialed); len(got) != 0 {
|
||||
t.Fatalf("PostStart on a closed group ran the warm-up sweep: dialled %v", got)
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,272 @@
|
||||
package group
|
||||
|
||||
// lx: health board §5.C tests — SelfCheck stands the group's OWN probing
|
||||
// schedule down: no PostStart warm-up sweep, no Touch ticker. The explicit
|
||||
// CheckOutbounds path stays available, and the nil default keeps probing.
|
||||
|
||||
import (
|
||||
"context"
|
||||
"sync"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"github.com/sagernet/sing-box/common/urltest"
|
||||
"github.com/sagernet/sing-box/log"
|
||||
"github.com/sagernet/sing-box/option"
|
||||
"github.com/sagernet/sing/service/pause"
|
||||
)
|
||||
|
||||
// waitForHistory polls until the store holds an entry for tag or the deadline
|
||||
// passes; reports whether it appeared. PostStart's sweep runs on its own
|
||||
// goroutine, so both directions of the assertion need a bounded wait.
|
||||
func waitForHistory(hist *urltest.HistoryStorage, tag string, deadline time.Duration) bool {
|
||||
stop := time.Now().Add(deadline)
|
||||
for time.Now().Before(stop) {
|
||||
if hist.LoadURLTestHistory(tag) != nil {
|
||||
return true
|
||||
}
|
||||
time.Sleep(5 * time.Millisecond)
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
// A group with the self-check stood down writes NOTHING to the history storage
|
||||
// on PostStart: the warm-up sweep — which would dial the member directly from
|
||||
// the router and mark the failure under its base tag — must not fire. And
|
||||
// Touch, the other half of the schedule, must not start a ticker either.
|
||||
func TestSelfCheckDisabledPostStartWritesNothing(t *testing.T) {
|
||||
hist := urltest.NewHistoryStorage()
|
||||
a := &healthNode{tag: "a", fail: true}
|
||||
manager := managerOf(a)
|
||||
g := healthTestGroup(hist, manager, a)
|
||||
g.selfCheckDisabled = true
|
||||
|
||||
g.PostStart()
|
||||
// The absence of a write is the assertion, so give the (non-existent) sweep
|
||||
// real time to have happened before declaring victory.
|
||||
if waitForHistory(hist, "a", 150*time.Millisecond) {
|
||||
t.Fatal("a stood-down group's PostStart wrote to the board; the warm-up sweep must not fire")
|
||||
}
|
||||
|
||||
g.Touch()
|
||||
g.access.Lock()
|
||||
ticker := g.ticker
|
||||
g.access.Unlock()
|
||||
if ticker != nil {
|
||||
t.Fatal("Touch armed the probing ticker on a stood-down group")
|
||||
}
|
||||
}
|
||||
|
||||
// The default (SelfCheck nil, i.e. the zero-value field on a hand-built group)
|
||||
// keeps today's behaviour: PostStart's warm-up sweep runs and records the
|
||||
// failing member on the board.
|
||||
func TestSelfCheckDefaultStillProbesOnPostStart(t *testing.T) {
|
||||
hist := urltest.NewHistoryStorage()
|
||||
a := &healthNode{tag: "a", fail: true}
|
||||
manager := managerOf(a)
|
||||
g := healthTestGroup(hist, manager, a)
|
||||
|
||||
g.PostStart()
|
||||
if !waitForHistory(hist, "a", 5*time.Second) {
|
||||
t.Fatal("default group's PostStart never probed; the self-check must stay on unless stood down")
|
||||
}
|
||||
if v := hist.Verdict("a", 10*time.Minute); v != urltest.VerdictDead {
|
||||
t.Fatalf("verdict(a) = %v, want dead from the warm-up sweep", v)
|
||||
}
|
||||
}
|
||||
|
||||
// An EXPLICIT CheckOutbounds still probes a stood-down group: the flag
|
||||
// suppresses the group's own schedule, never a deliberate request (the adapter
|
||||
// interface a human or an API invokes on purpose).
|
||||
func TestSelfCheckDisabledExplicitCheckStillProbes(t *testing.T) {
|
||||
hist := urltest.NewHistoryStorage()
|
||||
a := &healthNode{tag: "a", fail: true}
|
||||
manager := managerOf(a)
|
||||
g := healthTestGroup(hist, manager, a)
|
||||
g.selfCheckDisabled = true
|
||||
|
||||
g.CheckOutbounds(true)
|
||||
if hist.LoadURLTestHistory("a") == nil {
|
||||
t.Fatal("an explicit CheckOutbounds(true) did not probe; the flag must only stand down the schedule")
|
||||
}
|
||||
}
|
||||
|
||||
// The option → outbound plumbing: nil/absent means on, an explicit false means
|
||||
// stood down, an explicit true means on. NewURLTest is the only place the
|
||||
// option is read, so this is where a plumbing regression would hide.
|
||||
func TestSelfCheckOptionPlumbing(t *testing.T) {
|
||||
build := func(selfCheck *bool) *URLTest {
|
||||
t.Helper()
|
||||
opts := option.URLTestOutboundOptions{Outbounds: []string{"a"}}
|
||||
opts.SelfCheck = selfCheck
|
||||
ob, err := NewURLTest(context.Background(), nil, log.NewNOPFactory().Logger(), "t", opts)
|
||||
if err != nil {
|
||||
t.Fatalf("NewURLTest: %v", err)
|
||||
}
|
||||
return ob.(*URLTest)
|
||||
}
|
||||
if build(nil).selfCheckDisabled {
|
||||
t.Fatal("nil SelfCheck must keep the self-check ON (the compatibility default)")
|
||||
}
|
||||
on, off := true, false
|
||||
if build(&on).selfCheckDisabled {
|
||||
t.Fatal("SelfCheck=true must keep the self-check on")
|
||||
}
|
||||
if !build(&off).selfCheckDisabled {
|
||||
t.Fatal("SelfCheck=false must stand the self-check down")
|
||||
}
|
||||
}
|
||||
|
||||
// lx: health board §5.C — the RUNTIME gate (urltest.ProbeGate). The self-check
|
||||
// flag above says "the config reaches nothing here"; the gate says "the path in
|
||||
// front of this group is down right now". Both stand the SCHEDULE down; neither
|
||||
// touches an explicit check; and only the gate is allowed to change its mind
|
||||
// while the box runs.
|
||||
|
||||
// fakeGate answers from a mutable set of blocked tags, so one test can watch a
|
||||
// group stop dialling and start again without rebuilding anything.
|
||||
type fakeGate struct {
|
||||
mu sync.Mutex
|
||||
blocked map[string]bool
|
||||
warm map[string]bool
|
||||
asked int
|
||||
}
|
||||
|
||||
func (g *fakeGate) ProbeAllowed(tag string) bool {
|
||||
g.mu.Lock()
|
||||
defer g.mu.Unlock()
|
||||
g.asked++
|
||||
return !g.blocked[tag]
|
||||
}
|
||||
|
||||
func (g *fakeGate) ProbeWhenIdle(tag string) bool {
|
||||
g.mu.Lock()
|
||||
defer g.mu.Unlock()
|
||||
return g.warm[tag]
|
||||
}
|
||||
|
||||
func (g *fakeGate) set(tag string, blocked bool) {
|
||||
g.mu.Lock()
|
||||
defer g.mu.Unlock()
|
||||
g.blocked[tag] = blocked
|
||||
}
|
||||
|
||||
func (g *fakeGate) asks() int {
|
||||
g.mu.Lock()
|
||||
defer g.mu.Unlock()
|
||||
return g.asked
|
||||
}
|
||||
|
||||
// A gated group makes NO DIAL ATTEMPT on its own schedule — the assertion is on
|
||||
// the attempt log, not on the board, because a probe that ran and failed leaves
|
||||
// the same "nothing useful known" as one that never ran, and only the attempt
|
||||
// log tells them apart. This is the waste half of the chain-hop fix: a hop
|
||||
// sitting behind a dead hop would otherwise spend one probe timeout per member
|
||||
// rediscovering the same broken hop.
|
||||
func TestProbeGateBlocksScheduledDials(t *testing.T) {
|
||||
hist := urltest.NewHistoryStorage()
|
||||
defer hist.Close()
|
||||
var dialed []string
|
||||
a := &healthNode{tag: "chain-c-h3-a", fail: true, dialed: &dialed}
|
||||
b := &healthNode{tag: "chain-c-h3-b", fail: true, dialed: &dialed}
|
||||
gate := &fakeGate{blocked: map[string]bool{"chain-c-h3": true}}
|
||||
|
||||
g := healthTestGroup(hist, managerOf(a, b), a, b)
|
||||
g.tag = "chain-c-h3"
|
||||
g.probeGate = gate
|
||||
// Touch arms a real ticker, so this group needs the two things
|
||||
// healthTestGroup leaves out because nothing else in that suite starts one:
|
||||
// a pause manager to register the ticker with, and the close channel Close
|
||||
// shuts the loop down through.
|
||||
g.pause = pause.ManagerFromContext(pause.WithDefaultManager(context.Background()))
|
||||
g.close = make(chan struct{})
|
||||
|
||||
// The warm-up sweep: gated, so nothing is dialled. Give the (non-existent)
|
||||
// sweep real time to have happened — the absence is the assertion.
|
||||
g.PostStart()
|
||||
if waitForHistory(hist, "chain-c-h3-a", 150*time.Millisecond) {
|
||||
t.Fatal("a gated group's PostStart wrote to the board")
|
||||
}
|
||||
// A ticker tick, driven directly: this is the exact call loopCheck makes.
|
||||
g.scheduledCheck()
|
||||
if got := dialsOf(&dialed); len(got) != 0 {
|
||||
t.Fatalf("gated group dialled %v; a hop behind a dead hop must not dial at all", got)
|
||||
}
|
||||
|
||||
// Touch still arms the ticker. Refusing to arm it would leave a recovered
|
||||
// hop with nothing to notice — the block would outlive the failure.
|
||||
g.Touch()
|
||||
g.access.Lock()
|
||||
ticker := g.ticker
|
||||
g.access.Unlock()
|
||||
if ticker == nil {
|
||||
t.Fatal("Touch did not arm the ticker on a gated group; nothing would be left to spot the recovery")
|
||||
}
|
||||
_ = g.Close()
|
||||
|
||||
// The hop in front comes back. Nothing is reset, nothing is reapplied — the
|
||||
// next scheduled check simply asks again and gets a different answer.
|
||||
gate.set("chain-c-h3", false)
|
||||
before := gate.asks()
|
||||
g.scheduledCheck()
|
||||
if gate.asks() <= before {
|
||||
t.Error("scheduledCheck did not re-ask the gate; a cached answer is a block that outlives its cause")
|
||||
}
|
||||
if len(dialsOf(&dialed)) == 0 {
|
||||
t.Fatal("the group did not resume dialling after the hop in front recovered")
|
||||
}
|
||||
}
|
||||
|
||||
// An EXPLICIT check is a deliberate request and is never gated — the same rule
|
||||
// SelfCheck already follows. The gate stands down the schedule, not the
|
||||
// capability.
|
||||
func TestProbeGateDoesNotBlockExplicitCheck(t *testing.T) {
|
||||
hist := urltest.NewHistoryStorage()
|
||||
defer hist.Close()
|
||||
var dialed []string
|
||||
a := &healthNode{tag: "chain-c-h3-a", fail: true, dialed: &dialed}
|
||||
g := healthTestGroup(hist, managerOf(a), a)
|
||||
g.tag = "chain-c-h3"
|
||||
g.probeGate = &fakeGate{blocked: map[string]bool{"chain-c-h3": true}}
|
||||
|
||||
g.CheckOutbounds(true)
|
||||
if len(dialsOf(&dialed)) == 0 {
|
||||
t.Fatal("an explicit CheckOutbounds was refused by the gate")
|
||||
}
|
||||
}
|
||||
|
||||
// The two refusals are independent and compose the obvious way; and the absent
|
||||
// cases (no gate at all, an ungated tag) leave today's behaviour untouched,
|
||||
// which is what every plain sing-box config and every hand-built group relies
|
||||
// on.
|
||||
func TestSelfCheckAllowedCombinations(t *testing.T) {
|
||||
hist := urltest.NewHistoryStorage()
|
||||
defer hist.Close()
|
||||
a := &healthNode{tag: "a"}
|
||||
base := func() *URLTestGroup { return healthTestGroup(hist, managerOf(a), a) }
|
||||
|
||||
if g := base(); !g.selfCheckAllowed() {
|
||||
t.Error("a plain group with no gate must probe (the zero value is the compatibility default)")
|
||||
}
|
||||
g := base()
|
||||
g.selfCheckDisabled = true
|
||||
g.tag, g.probeGate = "chain-c-h3", &fakeGate{blocked: map[string]bool{}}
|
||||
if g.selfCheckAllowed() {
|
||||
t.Error("an UNUSED group must stay down even when the path in front is fine")
|
||||
}
|
||||
g = base()
|
||||
g.tag, g.probeGate = "chain-c-h3", &fakeGate{blocked: map[string]bool{"chain-c-h3": true}}
|
||||
if g.selfCheckAllowed() {
|
||||
t.Error("a group behind a dead hop must not run its schedule")
|
||||
}
|
||||
g = base()
|
||||
g.tag, g.probeGate = "auto", &fakeGate{blocked: map[string]bool{"chain-c-h3": true}}
|
||||
if !g.selfCheckAllowed() {
|
||||
t.Error("an unrelated group was gated by another tag's block")
|
||||
}
|
||||
g = base()
|
||||
g.probeGate = &fakeGate{blocked: map[string]bool{"": true}}
|
||||
if !g.selfCheckAllowed() {
|
||||
t.Error("a group with no tag must not be gated; there is nothing to ask about")
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,129 @@
|
||||
//go:build with_gvisor && with_awg
|
||||
|
||||
// lx: regression for the removal of the AmneziaWG-over-WireGuard start guard.
|
||||
//
|
||||
// The guard refused to bring up an AmneziaWG endpoint whose detour chain reached
|
||||
// a WireGuard-based endpoint — and refused *silently*: Start returned nil with
|
||||
// started=false, so the endpoint looked configured but every dial through it
|
||||
// failed with "WireGuard is not ready yet". The root cause it protected against
|
||||
// (a kernel hang on Android) is gone on this graft (ClientBind reserved-gate),
|
||||
// and Android is not a supported platform here at all.
|
||||
//
|
||||
// This test builds a real AmneziaWG endpoint (junk + ranged magic headers) whose
|
||||
// detour points at an outbound of type "wireguard", drives both start stages,
|
||||
// and asserts the endpoint reports itself started. With the guard in place the
|
||||
// first stage short-circuits and started stays false — this test fails.
|
||||
package wireguard
|
||||
|
||||
import (
|
||||
"context"
|
||||
"crypto/rand"
|
||||
"encoding/base64"
|
||||
"net"
|
||||
"net/netip"
|
||||
"os"
|
||||
"testing"
|
||||
|
||||
"github.com/sagernet/sing-box/adapter"
|
||||
C "github.com/sagernet/sing-box/constant"
|
||||
"github.com/sagernet/sing-box/log"
|
||||
"github.com/sagernet/sing-box/option"
|
||||
"github.com/sagernet/sing/common/json/badoption"
|
||||
M "github.com/sagernet/sing/common/metadata"
|
||||
"github.com/sagernet/sing/service"
|
||||
"github.com/sagernet/sing/service/pause"
|
||||
)
|
||||
|
||||
// wgTypedOutbound is an adapter.Outbound that reports type "wireguard" — the hop
|
||||
// the guard used to refuse to start behind. Dialling through it always fails:
|
||||
// the point of the test is that the upper endpoint comes UP, not that it carries
|
||||
// traffic (that is the job of the transport-level e2e stand).
|
||||
type wgTypedOutbound struct {
|
||||
adapter.Outbound
|
||||
tag string
|
||||
}
|
||||
|
||||
func (o *wgTypedOutbound) Type() string { return C.TypeWireGuard }
|
||||
func (o *wgTypedOutbound) Tag() string { return o.tag }
|
||||
func (o *wgTypedOutbound) Dependencies() []string { return nil }
|
||||
|
||||
func (o *wgTypedOutbound) DialContext(ctx context.Context, network string, destination M.Socksaddr) (net.Conn, error) {
|
||||
return nil, os.ErrClosed
|
||||
}
|
||||
|
||||
func (o *wgTypedOutbound) ListenPacket(ctx context.Context, destination M.Socksaddr) (net.PacketConn, error) {
|
||||
return nil, os.ErrClosed
|
||||
}
|
||||
|
||||
// startChainManager resolves tags from a fixed map. adapter.OutboundManager is
|
||||
// embedded so this compiles against either shape of the interface.
|
||||
type startChainManager struct {
|
||||
adapter.OutboundManager
|
||||
byTag map[string]adapter.Outbound
|
||||
}
|
||||
|
||||
func (m *startChainManager) Outbound(tag string) (adapter.Outbound, bool) {
|
||||
ob, loaded := m.byTag[tag]
|
||||
return ob, loaded
|
||||
}
|
||||
|
||||
func randomKey(t *testing.T) string {
|
||||
t.Helper()
|
||||
var key [32]byte
|
||||
if _, err := rand.Read(key[:]); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
// Clamp so wireguard-go accepts it as a curve25519 private key.
|
||||
key[0] &= 248
|
||||
key[31] = (key[31] & 127) | 64
|
||||
return base64.StdEncoding.EncodeToString(key[:])
|
||||
}
|
||||
|
||||
// TestAmneziaWGOverWireGuardDetourStarts pins the invariant: an AmneziaWG
|
||||
// endpoint detouring through a WireGuard hop must come up like any other.
|
||||
func TestAmneziaWGOverWireGuardDetourStarts(t *testing.T) {
|
||||
ctx := pause.WithDefaultManager(context.Background())
|
||||
ctx = service.ContextWith[adapter.OutboundManager](ctx, &startChainManager{
|
||||
byTag: map[string]adapter.Outbound{
|
||||
"wg-hop": &wgTypedOutbound{tag: "wg-hop"},
|
||||
},
|
||||
})
|
||||
options := option.WireGuardEndpointOptions{
|
||||
MTU: 1280,
|
||||
Address: badoption.Listable[netip.Prefix]{netip.MustParsePrefix("10.7.0.2/32")},
|
||||
PrivateKey: randomKey(t),
|
||||
Peers: []option.WireGuardPeer{{
|
||||
Address: "10.9.9.9",
|
||||
Port: 51820,
|
||||
PublicKey: randomKey(t),
|
||||
AllowedIPs: badoption.Listable[netip.Prefix]{netip.MustParsePrefix("0.0.0.0/0")},
|
||||
}},
|
||||
AmneziaWGOptions: option.AmneziaWGOptions{
|
||||
Jc: 3,
|
||||
Jmin: 8,
|
||||
Jmax: 80,
|
||||
S4: 16,
|
||||
H1: "10-20",
|
||||
H2: "30-40",
|
||||
H3: "50-60",
|
||||
H4: "70-80",
|
||||
},
|
||||
}
|
||||
options.Detour = "wg-hop"
|
||||
|
||||
ep, err := NewEndpoint(ctx, nil, log.NewNOPFactory().NewLogger("wg-awg"), "wg-awg", options)
|
||||
if err != nil {
|
||||
t.Fatal("create amneziawg endpoint over a wireguard detour: ", err)
|
||||
}
|
||||
defer ep.Close()
|
||||
|
||||
if err = ep.Start(adapter.StartStateStart); err != nil {
|
||||
t.Fatal("start stage: ", err)
|
||||
}
|
||||
if err = ep.Start(adapter.StartStatePostStart); err != nil {
|
||||
t.Fatal("post-start stage: ", err)
|
||||
}
|
||||
if !ep.(*Endpoint).started.Load() {
|
||||
t.Fatal("an amneziawg endpoint behind a wireguard hop must start; it is silently held down")
|
||||
}
|
||||
}
|
||||
@@ -1,100 +0,0 @@
|
||||
// lx:begin awg
|
||||
|
||||
package wireguard
|
||||
|
||||
import (
|
||||
"testing"
|
||||
|
||||
"github.com/sagernet/sing-box/adapter"
|
||||
C "github.com/sagernet/sing-box/constant"
|
||||
)
|
||||
|
||||
// fakeOutbound is a minimal adapter.Outbound for the start-guard chain walk:
|
||||
// only Type() and Dependencies() (the detour) are consulted. The embedded
|
||||
// interface is nil — any other method would panic, which never happens here.
|
||||
type fakeOutbound struct {
|
||||
adapter.Outbound
|
||||
tag string
|
||||
outboundTyp string
|
||||
detour string
|
||||
}
|
||||
|
||||
func (o *fakeOutbound) Type() string { return o.outboundTyp }
|
||||
func (o *fakeOutbound) Tag() string { return o.tag }
|
||||
func (o *fakeOutbound) Dependencies() []string {
|
||||
if o.detour == "" {
|
||||
return nil
|
||||
}
|
||||
return []string{o.detour}
|
||||
}
|
||||
|
||||
// fakeGroup is an adapter.OutboundGroup (selector/urltest stand-in); the chain
|
||||
// walk must stop at it without expanding All().
|
||||
type fakeGroup struct {
|
||||
fakeOutbound
|
||||
members []string
|
||||
}
|
||||
|
||||
func (g *fakeGroup) Now() string { return "" }
|
||||
func (g *fakeGroup) All() []string { return g.members }
|
||||
|
||||
type fakeOutboundManager struct {
|
||||
adapter.OutboundManager
|
||||
byTag map[string]adapter.Outbound
|
||||
}
|
||||
|
||||
func (m *fakeOutboundManager) Outbound(tag string) (adapter.Outbound, bool) {
|
||||
ob, loaded := m.byTag[tag]
|
||||
return ob, loaded
|
||||
}
|
||||
|
||||
func TestAwgDetourChainReachesWireGuard(t *testing.T) {
|
||||
mgr := &fakeOutboundManager{byTag: map[string]adapter.Outbound{
|
||||
// AWG -> wg-out (direct)
|
||||
"wg-out": &fakeOutbound{tag: "wg-out", outboundTyp: C.TypeWireGuard},
|
||||
// AWG -> vless-hop -> wg-deep (transitive)
|
||||
"vless-hop": &fakeOutbound{tag: "vless-hop", outboundTyp: C.TypeVLESS, detour: "wg-deep"},
|
||||
"wg-deep": &fakeOutbound{tag: "wg-deep", outboundTyp: C.TypeWireGuard},
|
||||
// AWG -> vless-leaf -> direct-leaf (no wireguard anywhere)
|
||||
"vless-leaf": &fakeOutbound{tag: "vless-leaf", outboundTyp: C.TypeVLESS, detour: "direct-leaf"},
|
||||
"direct-leaf": &fakeOutbound{tag: "direct-leaf", outboundTyp: C.TypeDirect},
|
||||
// AWG -> sel (selector hiding a wireguard member) — walk must stop, return ""
|
||||
"sel": &fakeGroup{
|
||||
fakeOutbound: fakeOutbound{tag: "sel", outboundTyp: C.TypeSelector},
|
||||
members: []string{"wg-out"},
|
||||
},
|
||||
// cyclic detour: a -> b -> a, no wireguard
|
||||
"cyc-a": &fakeOutbound{tag: "cyc-a", outboundTyp: C.TypeVLESS, detour: "cyc-b"},
|
||||
"cyc-b": &fakeOutbound{tag: "cyc-b", outboundTyp: C.TypeVLESS, detour: "cyc-a"},
|
||||
}}
|
||||
|
||||
cases := []struct {
|
||||
name string
|
||||
start string
|
||||
wantEmpty bool
|
||||
wantTag string
|
||||
}{
|
||||
{"direct wireguard", "wg-out", false, "wg-out"},
|
||||
{"transitive via vless", "vless-hop", false, "wg-deep"},
|
||||
{"no wireguard in chain", "vless-leaf", true, ""},
|
||||
{"selector in the middle is skipped", "sel", true, ""},
|
||||
{"cyclic chain terminates", "cyc-a", true, ""},
|
||||
{"unknown tag", "nope", true, ""},
|
||||
}
|
||||
for _, tc := range cases {
|
||||
t.Run(tc.name, func(t *testing.T) {
|
||||
got := awgDetourChainReachesWireGuard(mgr, tc.start, make(map[string]bool))
|
||||
if tc.wantEmpty {
|
||||
if got != "" {
|
||||
t.Fatalf("expected no wireguard in chain, got %q", got)
|
||||
}
|
||||
return
|
||||
}
|
||||
if got != tc.wantTag {
|
||||
t.Fatalf("expected blocked-by %q, got %q", tc.wantTag, got)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
// lx:end awg
|
||||
+20
-118
@@ -4,7 +4,6 @@ import (
|
||||
"context"
|
||||
"net"
|
||||
"net/netip"
|
||||
"strconv"
|
||||
"sync"
|
||||
"sync/atomic"
|
||||
"time"
|
||||
@@ -45,28 +44,14 @@ type Endpoint struct {
|
||||
localAddresses []netip.Prefix
|
||||
endpoint *wireguard.Endpoint
|
||||
started atomic.Bool
|
||||
// lx:begin awg
|
||||
// awgActive marks this endpoint as running AmneziaWG (AmneziaWGOptions.IsSet());
|
||||
// detour is its configured upstream tag. Start uses them to refuse to bring up
|
||||
// an AmneziaWG-over-WireGuard chain, which hangs the kernel on Android — see
|
||||
// awgDetourChainReachesWireGuard. The ledger lives here (not just in the dialer
|
||||
// guard) because the hang happens synchronously in Start, before any dial.
|
||||
awgActive bool
|
||||
detour string
|
||||
// awgChainBlocked is set by Start when the AmneziaWG-over-WireGuard guard
|
||||
// fires: the device is left unstarted (started stays false) so no junk
|
||||
// handshake runs and the kernel cannot hang, while the rest of the instance
|
||||
// comes up. PostStart then skips this endpoint too.
|
||||
awgChainBlocked bool
|
||||
// lx:end awg
|
||||
// lx:begin idle-suspend
|
||||
// SPEC 020 idle-suspend state. lastActivity is the unix-nano timestamp of the
|
||||
// last dial through this endpoint, stamped at PostStart and on every dial entry.
|
||||
// idleAsleep is true while the endpoint is Down due to idle-suspend (distinct
|
||||
// from a guard-suspend, which sets started=false and clears idleAsleep, so a
|
||||
// guard-suspended endpoint fast-paths out of resumeOnDial and is never
|
||||
// idle-woken). resumeMu serialises the idle tick's suspend decision, a dial's
|
||||
// wake, and the AmneziaWG guard-suspend against one another.
|
||||
// from a deliberately-stopped endpoint, which has started=false and
|
||||
// idleAsleep=false, so it fast-paths out of resumeOnDial and is never
|
||||
// idle-woken). resumeMu serialises the idle tick's suspend decision against a
|
||||
// dial's wake.
|
||||
lastActivity atomic.Int64
|
||||
idleAsleep atomic.Bool
|
||||
resumeMu sync.Mutex
|
||||
@@ -74,6 +59,16 @@ type Endpoint struct {
|
||||
}
|
||||
|
||||
func NewEndpoint(ctx context.Context, router adapter.Router, logger log.ContextLogger, tag string, options option.WireGuardEndpointOptions) (adapter.Endpoint, error) {
|
||||
// lx: allow OS-level fragmentation of the OUTER UDP socket by default, the
|
||||
// same opt-out direct/hysteria/hysteria2/tuic already take. Without it the
|
||||
// dialer sets DF (IP_MTU_DISCOVER=IP_PMTUDISC_DO on linux), and an outer
|
||||
// datagram over the path MTU — routine once anything is encapsulated: WG's
|
||||
// own ~32 B header, AmneziaWG s4 transport junk, or this endpoint carrying a
|
||||
// nested tunnel — is dropped by the kernel ("message too long") instead of
|
||||
// fragmented, so the tunnel comes up and then carries nothing. An explicit
|
||||
// `udp_fragment: false` on the node still restores DF (UDPFragment wins over
|
||||
// UDPFragmentDefault in common/dialer).
|
||||
options.UDPFragmentDefault = true
|
||||
ep := &Endpoint{
|
||||
Adapter: endpoint.NewAdapterWithDialerOptions(C.TypeWireGuard, tag, []string{N.NetworkTCP, N.NetworkUDP, N.NetworkICMP}, options.DialerOptions),
|
||||
ctx: ctx,
|
||||
@@ -81,10 +76,6 @@ func NewEndpoint(ctx context.Context, router adapter.Router, logger log.ContextL
|
||||
dnsRouter: service.FromContext[adapter.DNSRouter](ctx),
|
||||
logger: logger,
|
||||
localAddresses: options.Address,
|
||||
// lx:begin awg
|
||||
awgActive: options.AmneziaWGOptions.IsSet(),
|
||||
detour: options.Detour,
|
||||
// lx:end awg
|
||||
}
|
||||
if options.Detour != "" && options.ListenPort != 0 {
|
||||
return nil, E.New("`listen_port` is conflict with `detour`")
|
||||
@@ -116,7 +107,8 @@ func NewEndpoint(ctx context.Context, router adapter.Router, logger log.ContextL
|
||||
Dialer: outboundDialer,
|
||||
CreateDialer: func(interfaceName string) N.Dialer {
|
||||
return common.Must1(dialer.NewDefault(ctx, option.DialerOptions{
|
||||
BindInterface: interfaceName,
|
||||
BindInterface: interfaceName,
|
||||
UDPFragmentDefault: true, // lx: same reason as above — this is the bind-to-interface twin of the outer socket
|
||||
}))
|
||||
},
|
||||
Name: options.Name,
|
||||
@@ -157,33 +149,6 @@ func NewEndpoint(ctx context.Context, router adapter.Router, logger log.ContextL
|
||||
}
|
||||
|
||||
func (w *Endpoint) Start(stage adapter.StartStage) error {
|
||||
// lx:begin awg
|
||||
// Refuse to bring up an AmneziaWG endpoint whose detour chain reaches a
|
||||
// WireGuard-based endpoint: encapsulating AWG (junk handshake) inside a
|
||||
// WireGuard tunnel hangs the kernel on Android. The hang happens here, in the
|
||||
// synchronous Start path (peer-domain resolution over the detour, then the
|
||||
// device's junk handshake) — before any dial — so the lazy DetourDialer guard
|
||||
// never gets a chance to fire. We must catch it at Start instead.
|
||||
//
|
||||
// Behaviour is "variant B": do NOT return an error (that would abort the whole
|
||||
// instance start). Instead log, skip device startup, and leave started=false
|
||||
// so the rest of the config comes up and every dial through this endpoint
|
||||
// fails cleanly with "WireGuard is not ready yet". A selector/urltest in the
|
||||
// middle hides the real target at start time, so the chain walk stops at a
|
||||
// group and that case is left to the lazy DetourDialer guard at dial time.
|
||||
if stage == adapter.StartStateStart && w.awgActive && w.detour != "" {
|
||||
if outboundManager := service.FromContext[adapter.OutboundManager](w.ctx); outboundManager != nil {
|
||||
if blockedBy := awgDetourChainReachesWireGuard(outboundManager, w.detour, make(map[string]bool)); blockedBy != "" {
|
||||
w.awgChainBlocked = true
|
||||
w.logger.Error("amneziawg endpoint will not start: its detour chain reaches wireguard-based endpoint ", strconv.Quote(blockedBy), " — amneziawg over wireguard is not supported. Use a non-wireguard detour (e.g. vless).")
|
||||
return nil
|
||||
}
|
||||
}
|
||||
}
|
||||
if w.awgChainBlocked {
|
||||
return nil
|
||||
}
|
||||
// lx:end awg
|
||||
switch stage {
|
||||
case adapter.StartStateStart:
|
||||
return w.endpoint.Start(false)
|
||||
@@ -200,69 +165,6 @@ func (w *Endpoint) Start(stage adapter.StartStage) error {
|
||||
return nil
|
||||
}
|
||||
|
||||
// lx:begin awg
|
||||
// awgDetourChainReachesWireGuard walks the transitive detour chain starting at
|
||||
// tag and returns the tag of the first WireGuard-based outbound it reaches
|
||||
// (type "wireguard", covering plain WireGuard and AmneziaWG), or "" if none. It
|
||||
// follows each outbound's detour dependency; it deliberately does NOT expand
|
||||
// selector/urltest groups, whose chosen member is only known at runtime — that
|
||||
// case is handled lazily by the DetourDialer guard. visited guards against cyclic
|
||||
// detour configs. All outbounds are registered before any Start, so every tag in
|
||||
// the chain is resolvable here even though some may not have started yet.
|
||||
func awgDetourChainReachesWireGuard(outboundManager adapter.OutboundManager, tag string, visited map[string]bool) string {
|
||||
if tag == "" || visited[tag] {
|
||||
return ""
|
||||
}
|
||||
visited[tag] = true
|
||||
outbound, loaded := outboundManager.Outbound(tag)
|
||||
if !loaded {
|
||||
return ""
|
||||
}
|
||||
if outbound.Type() == C.TypeWireGuard {
|
||||
return tag
|
||||
}
|
||||
if _, isGroup := outbound.(adapter.OutboundGroup); isGroup {
|
||||
// Runtime-resolved target — leave it to the lazy DetourDialer guard.
|
||||
return ""
|
||||
}
|
||||
for _, dependency := range outbound.Dependencies() {
|
||||
if blockedBy := awgDetourChainReachesWireGuard(outboundManager, dependency, visited); blockedBy != "" {
|
||||
return blockedBy
|
||||
}
|
||||
}
|
||||
return ""
|
||||
}
|
||||
|
||||
// IsAmneziaWG reports whether this endpoint runs AmneziaWG. Implements
|
||||
// adapter.AmneziaWGSuspendable.
|
||||
func (w *Endpoint) IsAmneziaWG() bool {
|
||||
return w.awgActive
|
||||
}
|
||||
|
||||
// SuspendAmneziaWG brings the device down and marks the endpoint not-ready, so a
|
||||
// junk handshake is never sent and every dial fails with "WireGuard is not ready
|
||||
// yet". Called by the selector guard when a group this endpoint detours through
|
||||
// switches to a WireGuard member (AmneziaWG over WireGuard hangs the kernel on
|
||||
// Android). Idempotent. Implements adapter.AmneziaWGSuspendable.
|
||||
func (w *Endpoint) SuspendAmneziaWG() {
|
||||
// Take resumeMu so this is ordered against resumeOnDial/SuspendIfIdle: without
|
||||
// it, a dial that already passed resumeOnDial's idleAsleep checks could wake
|
||||
// the endpoint back up right after we clear the flag, defeating the guard.
|
||||
w.resumeMu.Lock()
|
||||
defer w.resumeMu.Unlock()
|
||||
if w.started.CompareAndSwap(true, false) {
|
||||
w.logger.Error("amneziawg endpoint suspended: a selector in its detour chain switched to a wireguard-based member — amneziawg over wireguard is not supported")
|
||||
}
|
||||
// Clear any idle-suspend state so resumeOnDial does not resurrect a
|
||||
// guard-suspended endpoint: if it was idle-asleep first, idleAsleep would still
|
||||
// be true and the next dial would wake it (SPEC 022 #2). With idleAsleep=false
|
||||
// resumeOnDial's fast path returns started (now false) and the endpoint stays down.
|
||||
w.idleAsleep.Store(false)
|
||||
w.endpoint.Suspend()
|
||||
}
|
||||
|
||||
// lx:end awg
|
||||
|
||||
// lx:begin idle-suspend
|
||||
|
||||
// stampActivity records the current time as the last dial through this endpoint.
|
||||
@@ -286,8 +188,8 @@ func (w *Endpoint) IdleSince() time.Duration {
|
||||
// holder — when it is unreachable from the active routing tree AND has been idle
|
||||
// past the threshold. Silent on every non-transition (edge-triggered logging).
|
||||
//
|
||||
// It never touches a guard-suspended endpoint: that one already has
|
||||
// started==false but idleAsleep==false, and the `!started` guard below short-
|
||||
// It never touches a deliberately-stopped endpoint: that one already has
|
||||
// started==false but idleAsleep==false, and the `!started` check below short-
|
||||
// circuits before the CAS. resumeMu mutually excludes this against resumeOnDial.
|
||||
func (w *Endpoint) SuspendIfIdle(reachable bool, threshold time.Duration) {
|
||||
w.resumeMu.Lock()
|
||||
@@ -296,7 +198,7 @@ func (w *Endpoint) SuspendIfIdle(reachable bool, threshold time.Duration) {
|
||||
return
|
||||
}
|
||||
if !w.started.Load() {
|
||||
// Already down some other way (guard-suspend, awg-chain-blocked, closed).
|
||||
// Already down some other way (deliberately stopped, closed).
|
||||
return
|
||||
}
|
||||
if w.idleAsleep.CompareAndSwap(false, true) {
|
||||
@@ -313,7 +215,7 @@ func (w *Endpoint) SuspendIfIdle(reachable bool, threshold time.Duration) {
|
||||
// session); that cost is on the first packet, as for any cold WG dial.
|
||||
//
|
||||
// Returns true if the endpoint is dialable (awake), false if it must stay down
|
||||
// (guard-suspend / chain-blocked — not an idle-suspend, so we do not resurrect it).
|
||||
// (deliberately stopped / closed — not an idle-suspend, so we do not resurrect it).
|
||||
func (w *Endpoint) resumeOnDial() bool {
|
||||
w.stampActivity()
|
||||
if !w.idleAsleep.Load() {
|
||||
|
||||
@@ -103,21 +103,21 @@ func TestSuspendIfIdle_idempotentCAS(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
// TestSuspendIfIdle_guardSuspendedNotTouched is the §8 invariant verified live on
|
||||
// an AWG-over-WG endpoint (wg-3 in the prod run): a guard-suspended endpoint has
|
||||
// started=false WITHOUT idleAsleep. The idle tick must early-return on !started and
|
||||
// NOT flip idleAsleep — otherwise a later resumeOnDial would idle-wake it and
|
||||
// re-trigger the AWG-over-WG kernel hang the guard exists to prevent.
|
||||
func TestSuspendIfIdle_guardSuspendedNotTouched(t *testing.T) {
|
||||
// TestSuspendIfIdle_stoppedNotTouched is the §8 invariant: a deliberately-stopped
|
||||
// endpoint (Close, or a start that never completed) has started=false WITHOUT
|
||||
// idleAsleep. The idle tick must early-return on !started and NOT flip idleAsleep
|
||||
// — otherwise a later resumeOnDial would idle-wake a device that was
|
||||
// intentionally down.
|
||||
func TestSuspendIfIdle_stoppedNotTouched(t *testing.T) {
|
||||
w := newIdleTestEndpoint()
|
||||
w.started.Store(false) // guard-suspend (device.Down at Start), idleAsleep stays false
|
||||
w.started.Store(false) // stopped, idleAsleep stays false
|
||||
w.lastActivity.Store(time.Now().Add(-time.Hour).UnixNano())
|
||||
w.SuspendIfIdle(false, 30*time.Second)
|
||||
if w.idleAsleep.Load() {
|
||||
t.Fatal("a guard-suspended endpoint must NOT be flagged idleAsleep by the tick")
|
||||
t.Fatal("a stopped endpoint must NOT be flagged idleAsleep by the tick")
|
||||
}
|
||||
if w.started.Load() {
|
||||
t.Fatal("the tick must not change started for a guard-suspended endpoint")
|
||||
t.Fatal("the tick must not change started for a stopped endpoint")
|
||||
}
|
||||
}
|
||||
|
||||
@@ -152,17 +152,17 @@ func TestResumeOnDial_dialBeforeTickRace(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
func TestResumeOnDial_guardSuspendedNotWoken(t *testing.T) {
|
||||
// A guard-suspended endpoint has started=false but idleAsleep=false.
|
||||
// resumeOnDial must NOT wake it (returns started, i.e. false).
|
||||
func TestResumeOnDial_stoppedNotWoken(t *testing.T) {
|
||||
// A deliberately-stopped endpoint (Close / failed start) has started=false but
|
||||
// idleAsleep=false. resumeOnDial must NOT wake it (returns started, i.e. false).
|
||||
w := newIdleTestEndpoint()
|
||||
w.started.Store(false) // simulate guard/awg-chain suspend (not idle)
|
||||
w.started.Store(false) // stopped, not idle-suspended
|
||||
ok := w.resumeOnDial()
|
||||
if ok {
|
||||
t.Fatal("resumeOnDial must not resurrect a guard-suspended (non-idle) endpoint")
|
||||
t.Fatal("resumeOnDial must not resurrect a stopped (non-idle) endpoint")
|
||||
}
|
||||
if w.idleAsleep.Load() {
|
||||
t.Fatal("guard-suspended endpoint must not be flagged idleAsleep")
|
||||
t.Fatal("stopped endpoint must not be flagged idleAsleep")
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -0,0 +1,268 @@
|
||||
// lx:begin l3-honest-drop
|
||||
package route
|
||||
|
||||
import (
|
||||
"context"
|
||||
"net/netip"
|
||||
"testing"
|
||||
|
||||
"github.com/sagernet/sing-box/adapter"
|
||||
C "github.com/sagernet/sing-box/constant"
|
||||
"github.com/sagernet/sing-box/log"
|
||||
"github.com/sagernet/sing-box/option"
|
||||
R "github.com/sagernet/sing-box/route/rule"
|
||||
"github.com/sagernet/sing/common/json/badoption"
|
||||
M "github.com/sagernet/sing/common/metadata"
|
||||
N "github.com/sagernet/sing/common/network"
|
||||
|
||||
"github.com/stretchr/testify/require"
|
||||
)
|
||||
|
||||
// The contract under test: PreMatch never answers "continue" (nor "bypass") for
|
||||
// an ICMP flow. adapter.JudgeFlow maps both to tun.ActionAccept, and the TUN
|
||||
// stack answers Accept by FORGING the echo reply itself
|
||||
// (sing-tun stack_gvisor_icmp.go ICMPForwarder.HandlePacket, the fallthrough
|
||||
// under the Flow/Reject/Drop switch). A verdict of "continue" therefore reads to
|
||||
// the operator as a working ping off a tunnel that never carried the packet.
|
||||
//
|
||||
// Every test below has a TCP/UDP twin: the honest drop must not leak into the
|
||||
// protocols where "continue" really does mean "take the ordinary connection
|
||||
// route".
|
||||
|
||||
// icmpL4Outbound is a minimal L4-only outbound (the vless/vmess/... shape): it
|
||||
// does NOT implement adapter.FlowOutbound, and Network() lists only TCP/UDP.
|
||||
// Unused Outbound methods come from the embedded nil interface and are never
|
||||
// called on the pre-match paths under test.
|
||||
type icmpL4Outbound struct {
|
||||
adapter.Outbound
|
||||
tag string
|
||||
}
|
||||
|
||||
func (o *icmpL4Outbound) Tag() string { return o.tag }
|
||||
func (o *icmpL4Outbound) Type() string { return "vless" }
|
||||
func (o *icmpL4Outbound) Network() []string { return []string{N.NetworkTCP, N.NetworkUDP} }
|
||||
|
||||
// icmpOutboundManager resolves tags from a fixed map and hands the same L4-only
|
||||
// outbound out as the default; the rest of the OutboundManager surface is never
|
||||
// touched by the pre-match walk.
|
||||
type icmpOutboundManager struct {
|
||||
adapter.OutboundManager
|
||||
defaultOutbound adapter.Outbound
|
||||
outbounds map[string]adapter.Outbound
|
||||
}
|
||||
|
||||
func (m *icmpOutboundManager) Default() adapter.Outbound { return m.defaultOutbound }
|
||||
|
||||
func (m *icmpOutboundManager) Outbound(tag string) (adapter.Outbound, bool) {
|
||||
outbound, loaded := m.outbounds[tag]
|
||||
return outbound, loaded
|
||||
}
|
||||
|
||||
// icmpDNSRouter / icmpDNSTransportManager implement only what
|
||||
// prepareMatchMetadata reaches. FakeIP returns nil unless a transport is
|
||||
// installed, which is how the "fakeip lookup failed" exit is driven below.
|
||||
type icmpDNSRouter struct {
|
||||
adapter.DNSRouter
|
||||
}
|
||||
|
||||
func (s *icmpDNSRouter) LookupReverseMapping(netip.Addr) (string, bool) { return "", false }
|
||||
|
||||
type icmpDNSTransportManager struct {
|
||||
adapter.DNSTransportManager
|
||||
fakeIP adapter.FakeIPTransport
|
||||
}
|
||||
|
||||
func (s *icmpDNSTransportManager) FakeIP() adapter.FakeIPTransport {
|
||||
if s.fakeIP == nil {
|
||||
return nil
|
||||
}
|
||||
return s.fakeIP
|
||||
}
|
||||
|
||||
// icmpMissingFakeIPTransport claims every address and then fails to look any of
|
||||
// them up — exactly the "missing fakeip record, try enable
|
||||
// `experimental.cache_file`" error prepareMatchMetadata returns.
|
||||
type icmpMissingFakeIPTransport struct {
|
||||
adapter.FakeIPTransport
|
||||
}
|
||||
|
||||
func (t *icmpMissingFakeIPTransport) Store() adapter.FakeIPStore {
|
||||
return &icmpMissingFakeIPStore{}
|
||||
}
|
||||
|
||||
type icmpMissingFakeIPStore struct {
|
||||
adapter.FakeIPStore
|
||||
}
|
||||
|
||||
func (s *icmpMissingFakeIPStore) Contains(netip.Addr) bool { return true }
|
||||
func (s *icmpMissingFakeIPStore) Lookup(netip.Addr) (string, bool) { return "", false }
|
||||
|
||||
type icmpRouterOptions struct {
|
||||
fakeIP adapter.FakeIPTransport
|
||||
rules []option.Rule
|
||||
}
|
||||
|
||||
func icmpTestRouter(t *testing.T, options icmpRouterOptions) *Router {
|
||||
t.Helper()
|
||||
logger := log.NewNOPFactory().NewLogger("test")
|
||||
defaultOutbound := &icmpL4Outbound{tag: "proxy-out"}
|
||||
router := &Router{
|
||||
ctx: context.Background(),
|
||||
logger: logger,
|
||||
dns: &icmpDNSRouter{},
|
||||
dnsTransport: &icmpDNSTransportManager{fakeIP: options.fakeIP},
|
||||
outbound: &icmpOutboundManager{
|
||||
defaultOutbound: defaultOutbound,
|
||||
outbounds: map[string]adapter.Outbound{defaultOutbound.Tag(): defaultOutbound},
|
||||
},
|
||||
}
|
||||
for i, ruleOptions := range options.rules {
|
||||
rule, err := R.NewRule(router.ctx, logger, ruleOptions, false)
|
||||
require.NoError(t, err, "build rule[%d]", i)
|
||||
router.rules = append(router.rules, rule)
|
||||
}
|
||||
return router
|
||||
}
|
||||
|
||||
func icmpTestMetadata(network string) adapter.InboundContext {
|
||||
return adapter.InboundContext{
|
||||
Inbound: "l3-in",
|
||||
InboundType: C.TypeTun,
|
||||
Network: network,
|
||||
Source: M.SocksaddrFrom(netip.MustParseAddr("192.168.1.2"), 0),
|
||||
Destination: M.SocksaddrFrom(netip.MustParseAddr("1.1.1.1"), 0),
|
||||
}
|
||||
}
|
||||
|
||||
// lanRuleWithAction matches every packet from the test source, so the action is
|
||||
// what the test is actually about.
|
||||
func lanRuleWithAction(action option.RuleAction) option.Rule {
|
||||
return option.Rule{
|
||||
Type: C.RuleTypeDefault,
|
||||
DefaultOptions: option.DefaultRule{
|
||||
RawDefaultRule: option.RawDefaultRule{
|
||||
SourceIPCIDR: badoption.Listable[string]{"192.168.1.0/24"},
|
||||
},
|
||||
RuleAction: action,
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
// --- exit 1: an outbound that cannot carry layer 3 --------------------------
|
||||
|
||||
func TestPreMatchICMPToL4OutboundDrops(t *testing.T) {
|
||||
t.Parallel()
|
||||
router := icmpTestRouter(t, icmpRouterOptions{})
|
||||
result := router.PreMatch(icmpTestMetadata(N.NetworkICMP), nil)
|
||||
require.Equal(t, adapter.PreMatchDrop, result.Action,
|
||||
"ICMP to an L4-only outbound fell through to the ordinary pre-match path: the TUN stack will forge the echo reply and ping will lie about a tunnel that never saw the packet")
|
||||
}
|
||||
|
||||
func TestPreMatchTCPToL4OutboundContinues(t *testing.T) {
|
||||
t.Parallel()
|
||||
router := icmpTestRouter(t, icmpRouterOptions{})
|
||||
result := router.PreMatch(icmpTestMetadata(N.NetworkTCP), nil)
|
||||
require.Equal(t, adapter.PreMatchContinue, result.Action,
|
||||
"TCP to an L4-only outbound must keep taking the ordinary connection route; the ICMP honest-drop must not leak into TCP/UDP pre-match")
|
||||
}
|
||||
|
||||
func TestPreMatchUDPToL4OutboundContinues(t *testing.T) {
|
||||
t.Parallel()
|
||||
router := icmpTestRouter(t, icmpRouterOptions{})
|
||||
result := router.PreMatch(icmpTestMetadata(N.NetworkUDP), nil)
|
||||
require.Equal(t, adapter.PreMatchContinue, result.Action,
|
||||
"UDP to an L4-only outbound must keep taking the ordinary connection route")
|
||||
}
|
||||
|
||||
// --- exit 2: prepareMatchMetadata failed before any rule was walked ---------
|
||||
|
||||
// This exit arrived with the shared prepareMatchMetadata refactor (upstream
|
||||
// b911fb078): it returns before the rule walk, so it never reaches preMatchFlow
|
||||
// where the ICMP override used to live.
|
||||
func TestPreMatchICMPMetadataErrorDrops(t *testing.T) {
|
||||
t.Parallel()
|
||||
router := icmpTestRouter(t, icmpRouterOptions{fakeIP: &icmpMissingFakeIPTransport{}})
|
||||
result := router.PreMatch(icmpTestMetadata(N.NetworkICMP), nil)
|
||||
require.Equal(t, adapter.PreMatchDrop, result.Action,
|
||||
"a fakeip record that cannot be resolved must not degrade ICMP to continue: continue is tun.ActionAccept, and Accept is a forged echo reply")
|
||||
}
|
||||
|
||||
func TestPreMatchTCPMetadataErrorContinues(t *testing.T) {
|
||||
t.Parallel()
|
||||
router := icmpTestRouter(t, icmpRouterOptions{fakeIP: &icmpMissingFakeIPTransport{}})
|
||||
result := router.PreMatch(icmpTestMetadata(N.NetworkTCP), nil)
|
||||
require.Equal(t, adapter.PreMatchContinue, result.Action,
|
||||
"for TCP the metadata-error exit must keep meaning `take the ordinary connection route`")
|
||||
}
|
||||
|
||||
// --- exit 3: a rule action the pre-match walk does not handle ---------------
|
||||
|
||||
// hijack-dns is one of the actions PreMatch's switch has no arm for, so it lands
|
||||
// in the default arm. Any future unhandled action lands there too — that is why
|
||||
// the guard is a funnel on the return value and not a per-arm override.
|
||||
func TestPreMatchICMPUnhandledRuleActionDrops(t *testing.T) {
|
||||
t.Parallel()
|
||||
router := icmpTestRouter(t, icmpRouterOptions{
|
||||
rules: []option.Rule{lanRuleWithAction(option.RuleAction{Action: C.RuleActionTypeHijackDNS})},
|
||||
})
|
||||
result := router.PreMatch(icmpTestMetadata(N.NetworkICMP), nil)
|
||||
require.Equal(t, adapter.PreMatchDrop, result.Action,
|
||||
"an unhandled rule action must not degrade ICMP to continue: continue is tun.ActionAccept, and Accept is a forged echo reply")
|
||||
}
|
||||
|
||||
func TestPreMatchTCPUnhandledRuleActionContinues(t *testing.T) {
|
||||
t.Parallel()
|
||||
router := icmpTestRouter(t, icmpRouterOptions{
|
||||
rules: []option.Rule{lanRuleWithAction(option.RuleAction{Action: C.RuleActionTypeHijackDNS})},
|
||||
})
|
||||
result := router.PreMatch(icmpTestMetadata(N.NetworkTCP), nil)
|
||||
require.Equal(t, adapter.PreMatchContinue, result.Action,
|
||||
"the unhandled-action exit must stay a continue for TCP")
|
||||
}
|
||||
|
||||
// --- exit 4: an explicit bypass ---------------------------------------------
|
||||
|
||||
// sing-tun implements ActionBypass on the nfqueue plane only; on the TUN path it
|
||||
// falls into the same default arm as Accept (flow_dispatch.go judgeAndInstall,
|
||||
// and the ICMP forwarder's switch has no Bypass case either), i.e. into the same
|
||||
// forgery. There is no honest bypass for a packet already inside the engine's
|
||||
// TUN.
|
||||
func TestPreMatchICMPBypassDrops(t *testing.T) {
|
||||
t.Parallel()
|
||||
router := icmpTestRouter(t, icmpRouterOptions{
|
||||
rules: []option.Rule{lanRuleWithAction(option.RuleAction{Action: C.RuleActionTypeBypass})},
|
||||
})
|
||||
result := router.PreMatch(icmpTestMetadata(N.NetworkICMP), nil)
|
||||
require.Equal(t, adapter.PreMatchDrop, result.Action,
|
||||
"bypass degrades to tun.ActionAccept on the TUN path, which is the forged echo reply again")
|
||||
}
|
||||
|
||||
func TestPreMatchTCPBypassIsStillBypass(t *testing.T) {
|
||||
t.Parallel()
|
||||
router := icmpTestRouter(t, icmpRouterOptions{
|
||||
rules: []option.Rule{lanRuleWithAction(option.RuleAction{Action: C.RuleActionTypeBypass})},
|
||||
})
|
||||
result := router.PreMatch(icmpTestMetadata(N.NetworkTCP), nil)
|
||||
require.Equal(t, adapter.PreMatchBypass, result.Action,
|
||||
"the ICMP honest-drop must not turn a TCP bypass rule into a drop")
|
||||
}
|
||||
|
||||
// --- the verdicts that must pass through untouched ---------------------------
|
||||
|
||||
// A reject rule already carries its own honest verdict; the funnel must not
|
||||
// rewrite it (a Reject sends an ICMP unreachable, which is information, not a
|
||||
// forged liveness signal).
|
||||
func TestPreMatchICMPRejectIsNotRewritten(t *testing.T) {
|
||||
t.Parallel()
|
||||
router := icmpTestRouter(t, icmpRouterOptions{
|
||||
rules: []option.Rule{lanRuleWithAction(option.RuleAction{
|
||||
Action: C.RuleActionTypeReject,
|
||||
RejectOptions: option.RejectActionOptions{Method: C.RuleActionRejectMethodDefault},
|
||||
})},
|
||||
})
|
||||
result := router.PreMatch(icmpTestMetadata(N.NetworkICMP), nil)
|
||||
require.Equal(t, adapter.PreMatchReject, result.Action,
|
||||
"the ICMP funnel must only rewrite continue/bypass, never an explicit reject")
|
||||
}
|
||||
|
||||
// lx:end l3-honest-drop
|
||||
@@ -0,0 +1,194 @@
|
||||
package route
|
||||
|
||||
import (
|
||||
"context"
|
||||
"net"
|
||||
"net/netip"
|
||||
"testing"
|
||||
|
||||
"github.com/sagernet/sing-box/adapter"
|
||||
C "github.com/sagernet/sing-box/constant"
|
||||
"github.com/sagernet/sing-box/log"
|
||||
"github.com/sagernet/sing-box/option"
|
||||
R "github.com/sagernet/sing-box/route/rule"
|
||||
"github.com/sagernet/sing/common/json/badoption"
|
||||
M "github.com/sagernet/sing/common/metadata"
|
||||
N "github.com/sagernet/sing/common/network"
|
||||
|
||||
"github.com/stretchr/testify/require"
|
||||
)
|
||||
|
||||
// The pre-match path (adapter.JudgeFlow -> Router.PreMatch, used by the TUN/
|
||||
// WireGuard-endpoint flow dispatcher) used to prepare only fakeip and the IP
|
||||
// version. Everything a rule matches on that the inbound cannot know — the
|
||||
// connection owner and the neighbor behind the source address — was resolved in
|
||||
// matchRule only, so a `source_mac_address` / `source_hostname` rule silently
|
||||
// failed to match in pre-match and the flow fell through to the default
|
||||
// outbound. Upstream b911fb078 shares one prepareMatchMetadata between both
|
||||
// paths; these tests pin that.
|
||||
|
||||
// stubNeighborResolver answers for exactly one address.
|
||||
type stubNeighborResolver struct {
|
||||
address netip.Addr
|
||||
mac net.HardwareAddr
|
||||
hostname string
|
||||
}
|
||||
|
||||
func (r *stubNeighborResolver) LookupMAC(address netip.Addr) (net.HardwareAddr, bool) {
|
||||
if address != r.address || r.mac == nil {
|
||||
return nil, false
|
||||
}
|
||||
return r.mac, true
|
||||
}
|
||||
|
||||
func (r *stubNeighborResolver) LookupHostname(address netip.Addr) (string, bool) {
|
||||
if address != r.address || r.hostname == "" {
|
||||
return "", false
|
||||
}
|
||||
return r.hostname, true
|
||||
}
|
||||
|
||||
func (r *stubNeighborResolver) LookupAddresses(hostname string) []netip.Addr {
|
||||
if hostname != r.hostname {
|
||||
return nil
|
||||
}
|
||||
return []netip.Addr{r.address}
|
||||
}
|
||||
|
||||
func (r *stubNeighborResolver) Start() error { return nil }
|
||||
func (r *stubNeighborResolver) Close() error { return nil }
|
||||
|
||||
// stubDNSRouter / stubDNSTransportManager implement only what
|
||||
// prepareMatchMetadata reaches; every other method is left to the embedded nil
|
||||
// interface and would panic if it were ever called.
|
||||
type stubDNSRouter struct {
|
||||
adapter.DNSRouter
|
||||
}
|
||||
|
||||
func (s *stubDNSRouter) LookupReverseMapping(netip.Addr) (string, bool) { return "", false }
|
||||
|
||||
type stubDNSTransportManager struct {
|
||||
adapter.DNSTransportManager
|
||||
}
|
||||
|
||||
func (s *stubDNSTransportManager) FakeIP() adapter.FakeIPTransport { return nil }
|
||||
|
||||
// stubOutboundManager's default outbound supports no network at all, so a flow
|
||||
// that reaches preMatchFlow bails out with PreMatchContinue instead of nil-
|
||||
// dereferencing. That is exactly the pre-fix verdict we assert against.
|
||||
type stubOutboundManager struct {
|
||||
adapter.OutboundManager
|
||||
defaultOutbound adapter.Outbound
|
||||
}
|
||||
|
||||
func (s *stubOutboundManager) Default() adapter.Outbound { return s.defaultOutbound }
|
||||
|
||||
type stubNoNetworkOutbound struct {
|
||||
adapter.Outbound
|
||||
}
|
||||
|
||||
func (o *stubNoNetworkOutbound) Tag() string { return "stub" }
|
||||
func (o *stubNoNetworkOutbound) Type() string { return "direct" }
|
||||
func (o *stubNoNetworkOutbound) Network() []string { return nil }
|
||||
|
||||
func newPreMatchTestRouter(t *testing.T, resolver adapter.NeighborResolver, rules ...option.Rule) *Router {
|
||||
t.Helper()
|
||||
logger := log.NewNOPFactory().NewLogger("test")
|
||||
router := &Router{
|
||||
ctx: context.Background(),
|
||||
logger: logger,
|
||||
dns: &stubDNSRouter{},
|
||||
dnsTransport: &stubDNSTransportManager{},
|
||||
outbound: &stubOutboundManager{defaultOutbound: &stubNoNetworkOutbound{}},
|
||||
neighborResolver: resolver,
|
||||
needFindNeighbor: true,
|
||||
}
|
||||
for i, ruleOptions := range rules {
|
||||
rule, err := R.NewRule(router.ctx, logger, ruleOptions, false)
|
||||
require.NoError(t, err, "build rule[%d]", i)
|
||||
router.rules = append(router.rules, rule)
|
||||
}
|
||||
return router
|
||||
}
|
||||
|
||||
func rejectOnSourceMAC(macAddress string) option.Rule {
|
||||
return option.Rule{
|
||||
Type: C.RuleTypeDefault,
|
||||
DefaultOptions: option.DefaultRule{
|
||||
RawDefaultRule: option.RawDefaultRule{
|
||||
SourceMACAddress: badoption.Listable[string]{macAddress},
|
||||
},
|
||||
RuleAction: option.RuleAction{
|
||||
Action: C.RuleActionTypeReject,
|
||||
RejectOptions: option.RejectActionOptions{Method: C.RuleActionRejectMethodDefault},
|
||||
},
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
func rejectOnSourceHostname(hostname string) option.Rule {
|
||||
return option.Rule{
|
||||
Type: C.RuleTypeDefault,
|
||||
DefaultOptions: option.DefaultRule{
|
||||
RawDefaultRule: option.RawDefaultRule{
|
||||
SourceHostname: badoption.Listable[string]{hostname},
|
||||
},
|
||||
RuleAction: option.RuleAction{
|
||||
Action: C.RuleActionTypeReject,
|
||||
RejectOptions: option.RejectActionOptions{Method: C.RuleActionRejectMethodDefault},
|
||||
},
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
func preMatchMetadata() adapter.InboundContext {
|
||||
return adapter.InboundContext{
|
||||
Inbound: "tun-in",
|
||||
InboundType: C.TypeTun,
|
||||
Network: N.NetworkUDP,
|
||||
Source: M.ParseSocksaddr("192.168.1.5:41234"),
|
||||
Destination: M.ParseSocksaddr("1.1.1.1:443"),
|
||||
}
|
||||
}
|
||||
|
||||
func TestPreMatchResolvesNeighborMAC(t *testing.T) {
|
||||
t.Parallel()
|
||||
mac, err := net.ParseMAC("de:ad:be:ef:00:01")
|
||||
require.NoError(t, err)
|
||||
resolver := &stubNeighborResolver{
|
||||
address: netip.MustParseAddr("192.168.1.5"),
|
||||
mac: mac,
|
||||
hostname: "kitchen-tv",
|
||||
}
|
||||
router := newPreMatchTestRouter(t, resolver, rejectOnSourceMAC("de:ad:be:ef:00:01"))
|
||||
result := router.PreMatch(preMatchMetadata(), nil)
|
||||
require.Equal(t, adapter.PreMatchReject, result.Action,
|
||||
"source_mac_address rule must match in pre-match; the MAC has to be resolved there too")
|
||||
}
|
||||
|
||||
func TestPreMatchResolvesNeighborHostname(t *testing.T) {
|
||||
t.Parallel()
|
||||
resolver := &stubNeighborResolver{
|
||||
address: netip.MustParseAddr("192.168.1.5"),
|
||||
hostname: "kitchen-tv",
|
||||
}
|
||||
router := newPreMatchTestRouter(t, resolver, rejectOnSourceHostname("kitchen-tv"))
|
||||
result := router.PreMatch(preMatchMetadata(), nil)
|
||||
require.Equal(t, adapter.PreMatchReject, result.Action,
|
||||
"source_hostname rule must match in pre-match; the hostname has to be resolved there too")
|
||||
}
|
||||
|
||||
// A source the neighbor resolver does not know must still fall through, not
|
||||
// match on a half-filled metadata.
|
||||
func TestPreMatchNeighborMissDoesNotMatch(t *testing.T) {
|
||||
t.Parallel()
|
||||
mac, err := net.ParseMAC("de:ad:be:ef:00:01")
|
||||
require.NoError(t, err)
|
||||
resolver := &stubNeighborResolver{
|
||||
address: netip.MustParseAddr("192.168.1.9"),
|
||||
mac: mac,
|
||||
}
|
||||
router := newPreMatchTestRouter(t, resolver, rejectOnSourceMAC("de:ad:be:ef:00:01"))
|
||||
result := router.PreMatch(preMatchMetadata(), nil)
|
||||
require.Equal(t, adapter.PreMatchContinue, result.Action)
|
||||
}
|
||||
+74
-25
@@ -314,27 +314,60 @@ func (r *Router) routePacketConnection(ctx context.Context, conn N.PacketConn, m
|
||||
return nil
|
||||
}
|
||||
|
||||
// lx:begin l3-honest-drop
|
||||
// PreMatch funnels every verdict of the pre-match walk through one ICMP check.
|
||||
//
|
||||
// An ICMP flow has no fallback path, so PreMatchContinue is not "try the
|
||||
// ordinary connection route" the way it is for TCP and UDP: the TUN stack takes
|
||||
// the packet back and answers the echo ITSELF (sing-tun stack_gvisor_icmp.go —
|
||||
// adapter.JudgeFlow maps Continue to tun.ActionAccept, and the ICMP forwarder
|
||||
// answers Accept by rewriting Echo into EchoReply and swapping the addresses).
|
||||
// A ping routed to an outbound that cannot carry layer 3 — every proxy
|
||||
// protocol; only adapter.FlowOutbound can — would therefore return a FORGED
|
||||
// reply, and the operator would read a working ping off a tunnel that never saw
|
||||
// the packet. Dropping instead reports the truth.
|
||||
//
|
||||
// PreMatchBypass is folded into the same drop because sing-tun implements
|
||||
// bypass for the nfqueue plane only (`ActionBypass` appears nowhere in
|
||||
// flow_dispatch.go / stack_gvisor_icmp.go): on the TUN path it degrades to the
|
||||
// same Accept, i.e. to the same forgery. There is no honest bypass for an ICMP
|
||||
// packet that is already inside the engine's TUN.
|
||||
//
|
||||
// This is a funnel and not an override inside the walk on purpose: the walk has
|
||||
// several independent exits that say "continue" (the prepareMatchMetadata error
|
||||
// return, the sniff bail-outs, the un-routable `bypass`, and the default arm of
|
||||
// the rule-action switch), and an earlier version of this delta guarded only
|
||||
// the ones that pass through preMatchFlow — leaving the others as narrow paths
|
||||
// to the forged reply. Guarding the single return value cannot be outgrown by a
|
||||
// new exit.
|
||||
func (r *Router) PreMatch(metadata adapter.InboundContext, firstPacket []byte) adapter.PreMatchResult {
|
||||
result := r.preMatch(metadata, firstPacket)
|
||||
if metadata.Network == N.NetworkICMP {
|
||||
switch result.Action {
|
||||
case adapter.PreMatchContinue, adapter.PreMatchBypass:
|
||||
return adapter.PreMatchResult{Action: adapter.PreMatchDrop}
|
||||
}
|
||||
}
|
||||
return result
|
||||
}
|
||||
|
||||
// preMatch is upstream's PreMatch body, unchanged; only the name moved, so that
|
||||
// the funnel above owns the exported entry point. An upstream change to the
|
||||
// pre-match walk applies to THIS function.
|
||||
func (r *Router) preMatch(metadata adapter.InboundContext, firstPacket []byte) adapter.PreMatchResult {
|
||||
// lx:end l3-honest-drop
|
||||
ctx := log.ContextWithNewID(r.ctx)
|
||||
metadata.PreMatch = true
|
||||
continueResult := adapter.PreMatchResult{Action: adapter.PreMatchContinue}
|
||||
packetDestination := metadata.Destination
|
||||
if metadata.Destination.Addr.IsValid() && r.dnsTransport.FakeIP() != nil && r.dnsTransport.FakeIP().Store().Contains(metadata.Destination.Addr) {
|
||||
domain, loaded := r.dnsTransport.FakeIP().Store().Lookup(metadata.Destination.Addr)
|
||||
if !loaded || domain == "" {
|
||||
return continueResult
|
||||
}
|
||||
metadata.OriginDestination = metadata.Destination
|
||||
metadata.Destination = M.Socksaddr{
|
||||
Fqdn: domain,
|
||||
Port: metadata.Destination.Port,
|
||||
}
|
||||
metadata.FakeIP = true
|
||||
}
|
||||
if metadata.Destination.IsIPv4() {
|
||||
metadata.IPVersion = 4
|
||||
} else if metadata.Destination.IsIPv6() {
|
||||
metadata.IPVersion = 6
|
||||
// lx: pre-match used to prepare only fakeip + IP version, so process/neighbor
|
||||
// rule items (process_name, source_mac_address, source_hostname, …) never had
|
||||
// their metadata filled here and silently failed to match — they were resolved
|
||||
// in matchRule only. Both paths now share prepareMatchMetadata (upstream
|
||||
// b911fb078).
|
||||
err := r.prepareMatchMetadata(ctx, &metadata)
|
||||
if err != nil {
|
||||
return continueResult
|
||||
}
|
||||
for currentRuleIndex, currentRule := range r.rules {
|
||||
metadata.ResetRuleCache()
|
||||
@@ -448,6 +481,11 @@ func applyRouteOptionsOverride(metadata *adapter.InboundContext, routeOptions *R
|
||||
|
||||
func (r *Router) preMatchFlow(ctx context.Context, metadata *adapter.InboundContext, packetDestination M.Socksaddr, matchedRule adapter.Rule, outboundTag string) adapter.PreMatchResult {
|
||||
continueResult := adapter.PreMatchResult{Action: adapter.PreMatchContinue}
|
||||
// lx: ICMP does NOT get a local override here any more — the honest drop is
|
||||
// applied once, to the single return value of PreMatch (see the funnel
|
||||
// there, marker l3-honest-drop). Overriding continueResult in this function
|
||||
// covered only the exits that reach it and left the walk's own exits
|
||||
// forging.
|
||||
var outbound adapter.Outbound
|
||||
if outboundTag == "" {
|
||||
outbound = r.outbound.Default()
|
||||
@@ -540,13 +578,11 @@ func (r *Router) preMatchFlow(ctx context.Context, metadata *adapter.InboundCont
|
||||
return result
|
||||
}
|
||||
|
||||
func (r *Router) matchRule(
|
||||
ctx context.Context, metadata *adapter.InboundContext,
|
||||
inputConn net.Conn, inputPacketConn N.PacketConn,
|
||||
) (
|
||||
selectedRule adapter.Rule, selectedRuleIndex int,
|
||||
buffers []*buf.Buffer, packetBuffers []*N.PacketBuffer, fatalErr error,
|
||||
) {
|
||||
// prepareMatchMetadata fills in everything a rule may match on but the inbound
|
||||
// cannot know: the connection owner, the neighbor (MAC/hostname) behind the
|
||||
// source address, the fakeip / reverse-mapped domain and the IP version. Shared
|
||||
// by matchRule and PreMatch — see the note at the PreMatch call site.
|
||||
func (r *Router) prepareMatchMetadata(ctx context.Context, metadata *adapter.InboundContext) error {
|
||||
r.searchProcessInfo(ctx, metadata)
|
||||
if r.neighborResolver != nil && metadata.SourceMACAddress == nil && metadata.Source.Addr.IsValid() {
|
||||
mac, macFound := r.neighborResolver.LookupMAC(metadata.Source.Addr)
|
||||
@@ -568,8 +604,7 @@ func (r *Router) matchRule(
|
||||
if metadata.Destination.Addr.IsValid() && r.dnsTransport.FakeIP() != nil && r.dnsTransport.FakeIP().Store().Contains(metadata.Destination.Addr) {
|
||||
domain, loaded := r.dnsTransport.FakeIP().Store().Lookup(metadata.Destination.Addr)
|
||||
if !loaded {
|
||||
fatalErr = E.New("missing fakeip record, try enable `experimental.cache_file`")
|
||||
return
|
||||
return E.New("missing fakeip record, try enable `experimental.cache_file`")
|
||||
}
|
||||
if domain != "" {
|
||||
metadata.OriginDestination = metadata.Destination
|
||||
@@ -592,6 +627,20 @@ func (r *Router) matchRule(
|
||||
} else if metadata.Destination.IsIPv6() {
|
||||
metadata.IPVersion = 6
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
func (r *Router) matchRule(
|
||||
ctx context.Context, metadata *adapter.InboundContext,
|
||||
inputConn net.Conn, inputPacketConn N.PacketConn,
|
||||
) (
|
||||
selectedRule adapter.Rule, selectedRuleIndex int,
|
||||
buffers []*buf.Buffer, packetBuffers []*N.PacketBuffer, fatalErr error,
|
||||
) {
|
||||
fatalErr = r.prepareMatchMetadata(ctx, metadata)
|
||||
if fatalErr != nil {
|
||||
return
|
||||
}
|
||||
|
||||
match:
|
||||
for currentRuleIndex, currentRule := range r.rules {
|
||||
|
||||
@@ -0,0 +1,88 @@
|
||||
#!/usr/bin/env bash
|
||||
#
|
||||
# run-panel-tests.sh — the admin-panel half of the release test gate.
|
||||
#
|
||||
# panel/package.json has declared a `test` script since the SPA was scaffolded
|
||||
# and nothing had ever called it: not release.yml, not build-shaterd.sh (which
|
||||
# only runs `npm ci` + `npm run build`). This script is what CI calls, and it
|
||||
# does the one thing `npm test` alone cannot: prove that tests actually RAN.
|
||||
#
|
||||
# `node --test src/*.test.ts` with no matching file leaves the glob unexpanded;
|
||||
# node then reports `pass 0` and exits 0 — a green CI step that ran nothing,
|
||||
# which is the exact failure class this whole change is about. So: the test
|
||||
# files are counted BEFORE the run (a rename to *.spec.ts is named as such
|
||||
# rather than showing up as a mystery), and the pass count is asserted > 0 and
|
||||
# the fail count 0 after it.
|
||||
#
|
||||
# KNOWN LIMIT: node --test counts a *.test.ts file that declares no cases at
|
||||
# all as one passing "test" (the module loaded). So an emptied-out file still
|
||||
# reads as pass 1 here. Deleting, renaming or breaking the file is caught;
|
||||
# gutting its contents while keeping the name is not.
|
||||
#
|
||||
# NODE VERSION: >= 22.6. The tests are TypeScript executed directly by
|
||||
# `node --test`; type stripping does not exist before then, so on node 20 the
|
||||
# run dies with a syntax error. CI pins node 24 for this step (the SPA *build*
|
||||
# still uses node 20 — that one goes through vite/tsc and does not care).
|
||||
#
|
||||
# Usage: scripts/run-panel-tests.sh
|
||||
set -euo pipefail
|
||||
|
||||
SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
|
||||
REPO="$(cd "$SCRIPT_DIR/.." && pwd)"
|
||||
cd "$REPO/panel"
|
||||
|
||||
node_major="$(node -p 'process.versions.node.split(".")[0]')"
|
||||
node_minor="$(node -p 'process.versions.node.split(".")[1]')"
|
||||
if [ "$node_major" -lt 22 ] || { [ "$node_major" -eq 22 ] && [ "$node_minor" -lt 6 ]; }; then
|
||||
echo " ERROR: panel tests are TypeScript under \`node --test\` and need node >= 22.6" >&2
|
||||
echo " (got $(node --version)). Type stripping does not exist before that." >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
# The glob `npm test` itself uses. Counted here so "somebody renamed the tests"
|
||||
# is reported as that, instead of as a suspiciously fast green step.
|
||||
shopt -s nullglob
|
||||
files=(src/*.test.ts)
|
||||
shopt -u nullglob
|
||||
if [ "${#files[@]}" -eq 0 ]; then
|
||||
echo " ERROR: no panel/src/*.test.ts — panel/package.json's \`test\` script" >&2
|
||||
echo " would match nothing and still exit 0. Fix the glob or the files." >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
echo "== panel tests (node $(node --version)), ${#files[@]} file(s) =="
|
||||
if [ ! -d node_modules ]; then
|
||||
npm ci
|
||||
fi
|
||||
|
||||
out=""
|
||||
rc=0
|
||||
set +e
|
||||
out="$(npm test --silent 2>&1)"
|
||||
rc=$?
|
||||
set -e
|
||||
sed 's/^/ /' <<<"$out"
|
||||
|
||||
if [ "$rc" -ne 0 ]; then
|
||||
echo " FAILED: npm test exited $rc" >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
# node --test's summary is `ℹ pass N` (spec reporter) or `# pass N` (tap).
|
||||
passed="$(sed -n 's/.*[[:space:]]pass[[:space:]]\{1,\}\([0-9]\{1,\}\).*/\1/p' <<<"$out" | tail -1)"
|
||||
failed="$(sed -n 's/.*[[:space:]]fail[[:space:]]\{1,\}\([0-9]\{1,\}\).*/\1/p' <<<"$out" | tail -1)"
|
||||
if [ -z "$passed" ]; then
|
||||
echo " FAILED: could not find a pass count in node --test output — the gate" >&2
|
||||
echo " cannot tell a green run from an empty one." >&2
|
||||
exit 1
|
||||
fi
|
||||
if [ "$passed" -lt 1 ]; then
|
||||
echo " FAILED: 0 panel tests ran. \`npm test\` returned success having done nothing." >&2
|
||||
exit 1
|
||||
fi
|
||||
if [ -n "$failed" ] && [ "$failed" -gt 0 ]; then
|
||||
echo " FAILED: $failed panel test(s) failed." >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
echo " OK: $passed panel test(s) passed."
|
||||
@@ -0,0 +1,417 @@
|
||||
#!/usr/bin/env bash
|
||||
#
|
||||
# run-tests.sh — THE test gate of the release tract.
|
||||
#
|
||||
# WHY THIS EXISTS (2026-07-26)
|
||||
# Until now the release tract ran almost no tests. The only `go test` calls in
|
||||
# the whole publishing path were scripts/build-shaterd.sh's one-package
|
||||
# buildtags check and the three named tests scripts/check-router-tags.sh runs.
|
||||
# The upstream .github/workflows/test.yml triggers on `stable`/`testing`/
|
||||
# `unstable` — branches this fork does not have — and Gitea does not read
|
||||
# .github/workflows at all once .gitea/workflows exists. Net effect: 115 of the
|
||||
# 116 test files under shater/** had never executed in CI, and
|
||||
# TestDNSFilterRemoteBlocklistHTTPClient shipped red through two releases
|
||||
# before anyone ran it by hand.
|
||||
#
|
||||
# WHAT IT GUARANTEES
|
||||
# 1. The suite runs under the SHIPPED build tags (scripts/router-tags.sh), not
|
||||
# under some CI-local tag set. This is not cosmetic: the AmneziaWG tests in
|
||||
# transport/wireguard are `//go:build with_awg` — 1 test file compiles
|
||||
# without the tag set, 7 with it. The 2026-07-25 WireGuard outage was
|
||||
# exactly a "built with X, verified with Y" gap.
|
||||
# 2. It runs on linux. shater/generate has 44 test files on linux against 32 on
|
||||
# windows/darwin; the linux-only half is where the routing, ruleset, DNS and
|
||||
# health tests live.
|
||||
# 3. Nothing is skipped SILENTLY. Three machine checks:
|
||||
# - the tag set may only ADD test files, never hide them (a test behind
|
||||
# `//go:build !with_awg` would vanish from the gate — this fails first);
|
||||
# - every package that has tests must report `ok` by name; a suite that
|
||||
# compiles down to "no test files" fails the gate instead of passing it;
|
||||
# - every ^TestIntegration under the fork's trees must produce a verdict
|
||||
# BY NAME ([5/5]). `ok <pkg>` is printed whether the privileged tests in
|
||||
# that package ran or called t.Skip, so the second check cannot see them
|
||||
# — and the gate would keep saying "passes every test we own" while the
|
||||
# tests that need a real kernel never executed.
|
||||
# A guard that silently runs nothing is worse than no guard (same rule as
|
||||
# scripts/check-router-tags.sh).
|
||||
#
|
||||
# Usage:
|
||||
# scripts/run-tests.sh # full gate (~3 min warm on the runner)
|
||||
# scripts/run-tests.sh --no-race # skip the -race pass (faster; local loop)
|
||||
#
|
||||
# Env:
|
||||
# SHATER_GO_IMAGE docker image used to reach linux from a non-linux host
|
||||
# (default golang:1.26 — keep it >= go.mod's toolchain).
|
||||
# SHATER_NO_DOCKER=1 fail instead of falling back to docker.
|
||||
# SHATER_REQUIRE_PRIVILEGED=1
|
||||
# turn [5/5]'s "did not run here" report into a hard
|
||||
# failure. Use it on the OpenWrt VM or in any pre-release
|
||||
# run that must actually have exercised the kernel paths.
|
||||
set -euo pipefail
|
||||
|
||||
SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
|
||||
REPO="$(cd "$SCRIPT_DIR/.." && pwd)"
|
||||
cd "$REPO"
|
||||
|
||||
RACE=1
|
||||
for a in "$@"; do
|
||||
case "$a" in
|
||||
--no-race) RACE=0 ;;
|
||||
-h|--help) sed -n '2,49p' "$0" | sed 's/^# \{0,1\}//'; exit 0 ;;
|
||||
*) echo "run-tests: unknown flag: $a" >&2; exit 2 ;;
|
||||
esac
|
||||
done
|
||||
|
||||
# shellcheck source=router-tags.sh
|
||||
. "$SCRIPT_DIR/router-tags.sh"
|
||||
|
||||
# The fork's own trees plus the upstream trees the fork edits (adapter/, route/,
|
||||
# option/, dns/ all carry shater changes). ROOTS, not a hand-kept package list: a
|
||||
# new package with tests joins the gate the moment it is created, which is the
|
||||
# whole point.
|
||||
ROOTS=(./shater/... ./protocol/... ./transport/... ./adapter/... ./route/... ./option/... ./dns/...)
|
||||
|
||||
# common/ is mostly upstream, but common/tls, common/tlsfragment, common/sniff
|
||||
# and common/urltest carry fork behaviour (D13 DPI-bypass, urltest health), so it
|
||||
# is in — minus the privileged integration tests below.
|
||||
ROOTS_COMMON=(./common/...)
|
||||
# SKIP, WITH REASON: common/tlsspoof's TestIntegration* enter TCP_REPAIR and
|
||||
# need CAP_NET_ADMIN. The act_runner job container runs as root but WITHOUT
|
||||
# that capability, so they do not skip — they FAIL. Excluded by name so the
|
||||
# rest of common/ can be a real gate instead of a permanently red one. (On
|
||||
# linux every tlsspoof test is a TestIntegration*, so that package is
|
||||
# effectively uncovered here; it is covered by the VM runs.)
|
||||
# The ^TestIntegration prefix is the fork-wide marker for "needs capabilities
|
||||
# the ordinary gate lacks", and [5/5] below leans on the same convention to
|
||||
# catch privileged tests inside ROOTS, which are NOT name-filtered and would
|
||||
# otherwise skip behind a green `ok <pkg>`. ROOTS_COMMON stays out of [5/5]:
|
||||
# these fail rather than skip without the capability, and that is a decision
|
||||
# about upstream code, not about the fork's own coverage.
|
||||
SKIP_COMMON='^TestIntegration'
|
||||
|
||||
# SKIP, WITH REASON: the first -race run over this tree (2026-07-26 — nobody
|
||||
# had ever run one) turned up two failures. One was a REAL product race:
|
||||
# ClientBind.connect() touched its fields from both the Send() path and
|
||||
# RoutineReceiveIncoming() with no lock, caught by
|
||||
# transport/wireguard.TestAwgDetourClientBindDelivers; that one has since been
|
||||
# fixed in client_bind.go and is NOT skipped — it is exactly what this pass is
|
||||
# for. What was left:
|
||||
# - shater/alert.TestExpiryDedupWithinDay — the test's own closure read a
|
||||
# variable the test body wrote while Notifier.dispatch's goroutine was
|
||||
# still delivering. FIXED 2026-07-26 (the simulated clock now has a mutex),
|
||||
# so the entry is gone and the -race pass covers the whole tree again.
|
||||
# Nothing is skipped under -race any more. Keep it that way: an entry here is a
|
||||
# hole in the gate, so add one only with a named reason and delete it the moment
|
||||
# the race is fixed.
|
||||
RACE_SKIP='^$'
|
||||
|
||||
echo "== shater test gate =="
|
||||
echo " tags : $SHATER_ROUTER_TAGS"
|
||||
echo " ldflags: $SHATER_ROUTER_LDFLAGS"
|
||||
echo " race : $([ "$RACE" -eq 1 ] && echo yes || echo no)"
|
||||
echo
|
||||
|
||||
# --- linux, or re-exec on linux ---------------------------------------------
|
||||
# The linux-only half of the suite is the half worth running (see header). From a
|
||||
# non-linux host, re-exec inside a golang container rather than quietly testing
|
||||
# 32 of shater/generate's 44 files — a partial gate reads exactly like a passing
|
||||
# one.
|
||||
if [ "$(go env GOOS)" != "linux" ] && [ "${SHATER_TESTS_IN_DOCKER:-0}" != "1" ]; then
|
||||
if [ "${SHATER_NO_DOCKER:-0}" = "1" ] || ! command -v docker >/dev/null 2>&1; then
|
||||
echo " ERROR: the gate needs linux (GOOS=$(go env GOOS)) and docker is unavailable/disabled." >&2
|
||||
echo " Run it on the linux CI runner or the OpenWrt VM." >&2
|
||||
exit 1
|
||||
fi
|
||||
image="${SHATER_GO_IMAGE:-golang:1.26}"
|
||||
echo "== re-exec on linux via docker ($image) =="
|
||||
host_repo="$REPO"
|
||||
command -v cygpath >/dev/null 2>&1 && host_repo="$(cygpath -w "$REPO")"
|
||||
# Hand the container CAP_NET_ADMIN and /dev/net/tun when this host's docker
|
||||
# can. shater/generate's ^TestIntegration tests open a real TUN and stand a
|
||||
# real engine on it; without the device they skip, and a dev running the gate
|
||||
# by hand would get a green result that never touched the kernel path the
|
||||
# branch is about. The dev host CAN give them (Docker Desktop's VM has the tun
|
||||
# module) — the CI runner cannot, which is what [5/5] exists to say out loud.
|
||||
# PROBED, never assumed: a docker whose kernel lacks tun refuses --device and
|
||||
# would take the whole gate down with it.
|
||||
priv_flags=()
|
||||
if MSYS2_ARG_CONV_EXCL='*' MSYS_NO_PATHCONV=1 docker run --rm \
|
||||
--cap-add NET_ADMIN --device /dev/net/tun "$image" true >/dev/null 2>&1; then
|
||||
priv_flags=(--cap-add NET_ADMIN --device /dev/net/tun)
|
||||
echo " CAP_NET_ADMIN + /dev/net/tun: available — the privileged tests will really run"
|
||||
else
|
||||
echo " CAP_NET_ADMIN + /dev/net/tun: NOT available from this docker — [5/5] will report the gap"
|
||||
fi
|
||||
MSYS2_ARG_CONV_EXCL='*' MSYS_NO_PATHCONV=1 docker run --rm \
|
||||
"${priv_flags[@]+"${priv_flags[@]}"}" \
|
||||
-v "$host_repo":/src \
|
||||
-v shater-tagcheck-gomod:/go/pkg/mod \
|
||||
-v shater-tagcheck-gocache:/root/.cache/go-build \
|
||||
-w /src \
|
||||
-e SHATER_TESTS_IN_DOCKER=1 \
|
||||
-e SHATER_REQUIRE_PRIVILEGED="${SHATER_REQUIRE_PRIVILEGED:-0}" \
|
||||
"$image" bash scripts/run-tests.sh "$@"
|
||||
exit $?
|
||||
fi
|
||||
|
||||
ALL_ROOTS=("${ROOTS[@]}" "${ROOTS_COMMON[@]}")
|
||||
|
||||
# --- [1/4] the tag set may only ADD test files, never hide them --------------
|
||||
# `go list` counts the test files the compiler would actually take. If adding the
|
||||
# shipped tags REMOVES a test file from any package, that test exists but the
|
||||
# gate would never see it — which is the failure mode this whole script is about,
|
||||
# just pointed the other way.
|
||||
echo "== [1/5] no test file is hidden by the shipped tag set =="
|
||||
LISTFMT='{{.ImportPath}} {{len .TestGoFiles}} {{len .XTestGoFiles}}'
|
||||
plain="$(go list -f "$LISTFMT" "${ALL_ROOTS[@]}")"
|
||||
tagged="$(go list -tags "$SHATER_ROUTER_TAGS" -f "$LISTFMT" "${ALL_ROOTS[@]}")"
|
||||
hidden=0
|
||||
while read -r pkg t x; do
|
||||
[ -n "${pkg:-}" ] || continue
|
||||
n_plain=$((t + x))
|
||||
[ "$n_plain" -gt 0 ] || continue
|
||||
line="$(awk -v p="$pkg" '$1 == p { print; exit }' <<<"$tagged")"
|
||||
if [ -z "$line" ]; then
|
||||
echo " HIDDEN: $pkg has $n_plain test file(s) untagged but no package at all under the shipped tags" >&2
|
||||
hidden=1
|
||||
continue
|
||||
fi
|
||||
read -r _ tt tx <<<"$line"
|
||||
n_tagged=$((tt + tx))
|
||||
if [ "$n_tagged" -lt "$n_plain" ]; then
|
||||
echo " HIDDEN: $pkg — $n_plain test file(s) untagged, only $n_tagged under the shipped tags" >&2
|
||||
hidden=1
|
||||
elif [ "$n_tagged" -gt "$n_plain" ]; then
|
||||
echo " +$((n_tagged - n_plain)) tag-gated test file(s): $pkg ($n_plain -> $n_tagged)"
|
||||
fi
|
||||
done <<<"$plain"
|
||||
if [ "$hidden" -ne 0 ]; then
|
||||
echo >&2
|
||||
echo " FAILED: a test file is invisible to the tag set we ship. Either the" >&2
|
||||
echo " constraint is wrong or the tag set is — do not paper over it" >&2
|
||||
echo " by testing with different tags than we build with." >&2
|
||||
exit 1
|
||||
fi
|
||||
echo
|
||||
|
||||
# --- the runner --------------------------------------------------------------
|
||||
# Runs one suite and then PROVES it ran: every package `go list` says has tests
|
||||
# must appear as `ok <pkg>` in the output. `go test` over a package whose tests
|
||||
# all vanished behind a build constraint prints "[no test files]" and exits 0 —
|
||||
# a green run that verified nothing.
|
||||
LOG="$(mktemp)"
|
||||
trap 'rm -f "$LOG"' EXIT
|
||||
FAILED=0
|
||||
|
||||
run_suite() { # $1=label $2=extra go-test flags (may be empty) $3..=packages
|
||||
local label="$1" extra="$2"
|
||||
shift 2
|
||||
local pkgs=("$@") rc=0 expect missing=0 pkg
|
||||
|
||||
expect="$(go list -tags "$SHATER_ROUTER_TAGS" \
|
||||
-f '{{if or .TestGoFiles .XTestGoFiles}}{{.ImportPath}}{{end}}' \
|
||||
"${pkgs[@]}" | grep -v '^$' || true)"
|
||||
if [ -z "$expect" ]; then
|
||||
echo " FAILED [$label]: go list reports no package with tests here — the gate" >&2
|
||||
echo " would have run nothing and passed." >&2
|
||||
FAILED=1
|
||||
return
|
||||
fi
|
||||
echo " packages with tests: $(wc -l <<<"$expect" | tr -d ' ')"
|
||||
|
||||
set +e
|
||||
# shellcheck disable=SC2086 # $extra is a deliberate word-split flag list
|
||||
go test -count=1 $extra \
|
||||
-tags "$SHATER_ROUTER_TAGS" -ldflags "$SHATER_ROUTER_LDFLAGS" \
|
||||
"${pkgs[@]}" >"$LOG" 2>&1
|
||||
rc=$?
|
||||
set -e
|
||||
sed 's/^/ /' "$LOG"
|
||||
|
||||
if [ "$rc" -ne 0 ]; then
|
||||
echo " FAILED [$label]: go test exited $rc" >&2
|
||||
FAILED=1
|
||||
return
|
||||
fi
|
||||
|
||||
while read -r pkg; do
|
||||
[ -n "$pkg" ] || continue
|
||||
grep -qE "^ok[[:space:]]+$pkg([[:space:]]|\$)" "$LOG" || {
|
||||
echo " DID NOT RUN [$label]: $pkg" >&2
|
||||
missing=1
|
||||
}
|
||||
done <<<"$expect"
|
||||
if [ "$missing" -ne 0 ]; then
|
||||
echo " FAILED [$label]: package(s) above have test files but produced no 'ok'" >&2
|
||||
echo " line. Build-constraint or file-name drift emptied them." >&2
|
||||
FAILED=1
|
||||
return
|
||||
fi
|
||||
echo " OK [$label]"
|
||||
}
|
||||
|
||||
# --- [2/5] the fork's trees, shipped tags, linux -----------------------------
|
||||
echo "== [2/5] go test — the fork's trees (shipped tags, linux) =="
|
||||
run_suite main "" "${ROOTS[@]}"
|
||||
echo
|
||||
|
||||
# --- [3/5] common/, minus the tests that need CAP_NET_ADMIN ------------------
|
||||
echo "== [3/5] go test — common/ (minus the CAP_NET_ADMIN integration tests) =="
|
||||
run_suite common "-skip $SKIP_COMMON" "${ROOTS_COMMON[@]}"
|
||||
echo
|
||||
|
||||
# --- [4/5] -race over the same trees -----------------------------------------
|
||||
# Everything, not a subset: shater/netplane alone is ~110 s under -race and it is
|
||||
# the single most concurrency-critical package we own (the nft data plane), so
|
||||
# once it is in, adding the rest costs ~40 s more. common/ is left out — it is
|
||||
# upstream code exercised by upstream CI.
|
||||
if [ "$RACE" -eq 1 ]; then
|
||||
echo "== [4/5] go test -race — the fork's trees =="
|
||||
echo " nothing is skipped under -race"
|
||||
|
||||
run_suite race "-race -skip $RACE_SKIP" "${ROOTS[@]}"
|
||||
else
|
||||
echo "== [4/5] -race pass skipped (--no-race) =="
|
||||
fi
|
||||
echo
|
||||
|
||||
# --- [5/5] the privileged tests may not skip in silence ----------------------
|
||||
# THE HOLE THIS CLOSES. Some tests can only prove what they claim against a real
|
||||
# kernel: shater/generate's TestIntegrationL3TunInboundStarts opens /dev/net/tun
|
||||
# and stands a real engine on it, TestIntegrationL3EgressICMPIsAFlow binds a real
|
||||
# socket to a real device. Both guard themselves with t.Skip when root or the
|
||||
# device is missing — the honest thing for a test to do, and completely INVISIBLE
|
||||
# above: `go test` prints `ok <pkg>` whether they ran or skipped, so [2/5]'s
|
||||
# per-package `ok` check is satisfied either way and the gate closes by claiming
|
||||
# it "passes every test we own". That is precisely the failure this whole script
|
||||
# was written for (115 of 116 test files never running while CI stayed green),
|
||||
# one level down and harder to see.
|
||||
#
|
||||
# The list is DISCOVERED, not hand-kept — `go test -list` over the same ROOTS —
|
||||
# so a privileged test written next month joins this check on the day it is
|
||||
# named, with no edit here. It keys on the ^TestIntegration prefix, already this
|
||||
# fork's marker for "needs capabilities the ordinary gate lacks" (SKIP_COMMON
|
||||
# above excludes common/tlsspoof's TestIntegration* for exactly that reason).
|
||||
# Name a privileged test anything else and it is invisible again — so don't.
|
||||
#
|
||||
# Verdicts, per test, by name:
|
||||
# RAN — it executed here; printed so that is visible rather than assumed.
|
||||
# FAILED — fatal, like any other failure.
|
||||
# MISSING — `go test -list` named it and the run produced no verdict for it:
|
||||
# fatal. A test that vanished between listing and running is the
|
||||
# same class of hole as one hidden by a build tag.
|
||||
# SKIPPED while this environment HAS root and /dev/net/tun — fatal. The
|
||||
# capability guard cannot be what skipped it, so something else did
|
||||
# and only the test knows what.
|
||||
# SKIPPED because the environment genuinely cannot run it — reported loudly,
|
||||
# by name, and it REPLACES the closing banner, so the last line of
|
||||
# the gate can never claim coverage it does not have. Deliberately
|
||||
# not fatal by default: the act_runner is an LXC guest whose kernel
|
||||
# has no tun module at all (checked 2026-07-26 on 10.10.10.211 —
|
||||
# `modprobe tun` answers "Module tun not found", /dev/net does not
|
||||
# exist, and act_runner runs job containers with privileged:false
|
||||
# and no container.options), so the device cannot be handed down
|
||||
# without reconfiguring the Proxmox host. Making it fatal would
|
||||
# paint CI permanently red and teach everyone to ignore the gate.
|
||||
# SHATER_REQUIRE_PRIVILEGED=1 makes it fatal for the runs that can.
|
||||
echo "== [5/5] the privileged tests (^TestIntegration) produced a verdict by name =="
|
||||
PRIV_RE='^TestIntegration'
|
||||
PRIV_UNVERIFIED=""
|
||||
# -ldflags is NOT optional on the discovery call either: `go test -list` LINKS
|
||||
# each test binary before it can enumerate its tests, and without
|
||||
# -checklinkname=0 every package that pulls common/badtls fails to link. The
|
||||
# first cut of this step omitted it, swallowed the error with `2>/dev/null ||
|
||||
# true`, and reported "none declared" — a check against silent skipping that was
|
||||
# itself silently skipping. Hence also: the exit status is inspected, and an
|
||||
# empty list is only ever reported after a SUCCESSFUL enumeration.
|
||||
set +e
|
||||
priv_expect_raw="$(go test -list "$PRIV_RE" \
|
||||
-tags "$SHATER_ROUTER_TAGS" -ldflags "$SHATER_ROUTER_LDFLAGS" "${ROOTS[@]}" 2>&1)"
|
||||
priv_list_rc=$?
|
||||
set -e
|
||||
priv_expect="$(grep -E "$PRIV_RE" <<<"$priv_expect_raw" | sort -u || true)"
|
||||
if [ "$priv_list_rc" -ne 0 ]; then
|
||||
echo " FAILED [privileged]: could not enumerate the privileged tests (go test -list exited $priv_list_rc)." >&2
|
||||
echo " An unreadable list is NOT an empty list — this check refuses to" >&2
|
||||
echo " report 'nothing to verify' on the strength of a failed command." >&2
|
||||
sed 's/^/ /' <<<"$priv_expect_raw" | grep -vE '^\s+(ok|\?)\s' >&2 || true
|
||||
FAILED=1
|
||||
elif [ -z "$priv_expect" ]; then
|
||||
echo " none declared under the fork's trees — nothing to verify"
|
||||
else
|
||||
priv_capable=0
|
||||
if [ "$(id -u)" = "0" ] && [ -e /dev/net/tun ]; then
|
||||
priv_capable=1
|
||||
fi
|
||||
echo " declared: $(wc -l <<<"$priv_expect" | tr -d ' ')"
|
||||
echo " this environment: uid=$(id -u), /dev/net/tun $([ -e /dev/net/tun ] && echo present || echo MISSING) => can run them: $([ "$priv_capable" -eq 1 ] && echo yes || echo NO)"
|
||||
set +e
|
||||
go test -count=1 -v -run "$PRIV_RE" \
|
||||
-tags "$SHATER_ROUTER_TAGS" -ldflags "$SHATER_ROUTER_LDFLAGS" \
|
||||
"${ROOTS[@]}" >"$LOG" 2>&1
|
||||
priv_rc=$?
|
||||
set -e
|
||||
# The verdict lines plus whatever reason the test printed just before them,
|
||||
# so a skip is readable here and not just counted.
|
||||
grep -E '^(--- (PASS|SKIP|FAIL): |[[:space:]]+[^[:space:]]+\.go:[0-9]+: )' "$LOG" \
|
||||
| sed 's/^/ | /' || true
|
||||
priv_bad=0
|
||||
while read -r name; do
|
||||
[ -n "$name" ] || continue
|
||||
if grep -qE "^--- PASS: ${name}([[:space:]]|\$)" "$LOG"; then
|
||||
echo " RAN $name"
|
||||
elif grep -qE "^--- FAIL: ${name}([[:space:]]|\$)" "$LOG"; then
|
||||
echo " FAILED $name" >&2
|
||||
priv_bad=1
|
||||
elif grep -qE "^--- SKIP: ${name}([[:space:]]|\$)" "$LOG"; then
|
||||
if [ "$priv_capable" -eq 1 ]; then
|
||||
echo " SKIPPED $name — but this environment HAS root and /dev/net/tun, so the capability guard is NOT what skipped it" >&2
|
||||
priv_bad=1
|
||||
else
|
||||
echo " DID NOT RUN $name — skipped: no root and/or no /dev/net/tun here"
|
||||
PRIV_UNVERIFIED="$PRIV_UNVERIFIED $name"
|
||||
fi
|
||||
else
|
||||
echo " MISSING $name — go test -list named it, the run produced no verdict for it" >&2
|
||||
priv_bad=1
|
||||
fi
|
||||
done <<<"$priv_expect"
|
||||
if [ "$priv_rc" -ne 0 ] && [ "$priv_bad" -eq 0 ]; then
|
||||
echo " FAILED [privileged]: go test exited $priv_rc with every named test accounted for —" >&2
|
||||
echo " a build or package-level failure, see the log above." >&2
|
||||
priv_bad=1
|
||||
fi
|
||||
if [ "$priv_bad" -ne 0 ]; then
|
||||
FAILED=1
|
||||
fi
|
||||
fi
|
||||
echo
|
||||
|
||||
if [ "$FAILED" -ne 0 ]; then
|
||||
echo "== TEST GATE FAILED — nothing may be published from this run. ==" >&2
|
||||
exit 1
|
||||
fi
|
||||
if [ -n "$PRIV_UNVERIFIED" ]; then
|
||||
echo "== !! PASSED, BUT NOT FULLY VERIFIED !! =================================="
|
||||
echo " Every test that COULD run here passed. These did not run at all:"
|
||||
for t in $PRIV_UNVERIFIED; do
|
||||
echo " - $t"
|
||||
done
|
||||
echo
|
||||
echo " They need root + CAP_NET_ADMIN + /dev/net/tun, which this environment"
|
||||
echo " does not have. Nothing about the kernel paths they cover was verified"
|
||||
echo " by this run. To actually run them, from a host whose docker can:"
|
||||
echo " scripts/run-tests.sh # the re-exec hands the container both"
|
||||
echo " or directly:"
|
||||
echo " docker run --rm --cap-add NET_ADMIN --device /dev/net/tun \\"
|
||||
echo " -v \"\$PWD\":/src -w /src golang:1.26 bash scripts/run-tests.sh"
|
||||
echo " or on the OpenWrt VM. SHATER_REQUIRE_PRIVILEGED=1 makes this a hard"
|
||||
echo " failure instead of this notice."
|
||||
echo "=========================================================================="
|
||||
if [ "${SHATER_REQUIRE_PRIVILEGED:-0}" = "1" ]; then
|
||||
echo "== TEST GATE FAILED: SHATER_REQUIRE_PRIVILEGED=1 and the tests above did not run. ==" >&2
|
||||
exit 1
|
||||
fi
|
||||
exit 0
|
||||
fi
|
||||
echo "== OK: the shipped tag set, on linux, passes every test we own. =="
|
||||
@@ -0,0 +1,141 @@
|
||||
package alert
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"strings"
|
||||
"sync"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"github.com/sagernet/sing-box/log"
|
||||
"github.com/sagernet/sing-box/shater/model"
|
||||
)
|
||||
|
||||
// warnLogger records Warn lines so a test can assert that an eviction was
|
||||
// actually announced. Everything else falls through to the standard logger.
|
||||
type warnLogger struct {
|
||||
log.ContextLogger
|
||||
mu sync.Mutex
|
||||
warns []string
|
||||
}
|
||||
|
||||
func newWarnLogger() *warnLogger { return &warnLogger{ContextLogger: log.StdLogger()} }
|
||||
|
||||
func (l *warnLogger) Warn(args ...any) {
|
||||
l.mu.Lock()
|
||||
l.warns = append(l.warns, fmt.Sprint(args...))
|
||||
l.mu.Unlock()
|
||||
}
|
||||
|
||||
func (l *warnLogger) lines() []string {
|
||||
l.mu.Lock()
|
||||
defer l.mu.Unlock()
|
||||
return append([]string(nil), l.warns...)
|
||||
}
|
||||
|
||||
// notifierAt builds a Notifier with no channels (so nothing is ever delivered —
|
||||
// only the dedup bookkeeping runs) and a clock the test drives.
|
||||
func notifierAt(t *testing.T, clock *time.Time) (*Notifier, *warnLogger) {
|
||||
t.Helper()
|
||||
lg := newWarnLogger()
|
||||
n := New([]model.Alert{}, lg)
|
||||
n.now = func() time.Time { return *clock }
|
||||
return n, lg
|
||||
}
|
||||
|
||||
func (n *Notifier) dedupSize() int {
|
||||
n.mu.Lock()
|
||||
defer n.mu.Unlock()
|
||||
return len(n.dedup)
|
||||
}
|
||||
|
||||
// TestDedupTableIsBoundedOverTime is the leak itself: a guest network whose
|
||||
// clients randomise their MAC produces an endless stream of distinct
|
||||
// "new_device:<MAC>" keys, and nothing ever deleted one. Spread over time — which
|
||||
// is how it actually happens — the table must stay small, and nothing may be
|
||||
// reported as evicted, because an entry past the dedup window could no longer
|
||||
// suppress anything anyway.
|
||||
func TestDedupTableIsBoundedOverTime(t *testing.T) {
|
||||
clock := time.Now()
|
||||
n, lg := notifierAt(t, &clock)
|
||||
|
||||
// Ten times the cap, at one incident per second: every key is long past the
|
||||
// 60s window by the time the next batch arrives.
|
||||
const fires = maxDedupKeys * 10
|
||||
for i := 0; i < fires; i++ {
|
||||
clock = clock.Add(time.Second)
|
||||
n.FireIncident(Incident{
|
||||
Events: []string{"new_device"},
|
||||
Title: "New device on the LAN",
|
||||
Key: fmt.Sprintf("new_device:02:00:00:%02x:%02x:%02x", i>>16&0xff, i>>8&0xff, i&0xff),
|
||||
})
|
||||
}
|
||||
|
||||
if got := n.dedupSize(); got > maxDedupKeys {
|
||||
t.Fatalf("dedup table holds %d entries after %d distinct incidents; cap is %d",
|
||||
got, fires, maxDedupKeys)
|
||||
}
|
||||
if got := n.DedupEvicted(); got != 0 {
|
||||
t.Fatalf("reported %d LIVE evictions; entries aged out of the window and losing them costs nothing", got)
|
||||
}
|
||||
if lines := lg.lines(); len(lines) != 0 {
|
||||
t.Fatalf("expiry sweep must be silent (it loses nothing), got: %v", lines)
|
||||
}
|
||||
}
|
||||
|
||||
// TestDedupTableEvictionIsAnnounced is the other half: when the cap genuinely
|
||||
// bites — more distinct incidents inside ONE dedup window than the table holds —
|
||||
// live suppression state is lost and repeats may notify twice. That must be said
|
||||
// out loud, not absorbed.
|
||||
func TestDedupTableEvictionIsAnnounced(t *testing.T) {
|
||||
clock := time.Now()
|
||||
n, lg := notifierAt(t, &clock)
|
||||
|
||||
// The clock does not move: every key stays inside its window.
|
||||
for i := 0; i < maxDedupKeys+10; i++ {
|
||||
clock = clock.Add(time.Millisecond) // still far inside dedupWindow
|
||||
n.FireIncident(Incident{
|
||||
Events: []string{"new_device"},
|
||||
Title: "New device on the LAN",
|
||||
Key: fmt.Sprintf("new_device:flood-%d", i),
|
||||
})
|
||||
}
|
||||
|
||||
if got := n.dedupSize(); got > maxDedupKeys {
|
||||
t.Fatalf("dedup table grew to %d, above the %d cap", got, maxDedupKeys)
|
||||
}
|
||||
if n.DedupEvicted() == 0 {
|
||||
t.Fatalf("a flood of %d in-window incidents evicted nothing — the cap is not enforced", maxDedupKeys+10)
|
||||
}
|
||||
lines := lg.lines()
|
||||
if len(lines) == 0 {
|
||||
t.Fatalf("live suppression entries were dropped with no notice")
|
||||
}
|
||||
if !strings.Contains(lines[0], "dedup table full") || !strings.Contains(lines[0], "notify twice") {
|
||||
t.Fatalf("eviction notice does not explain the consequence: %q", lines[0])
|
||||
}
|
||||
}
|
||||
|
||||
// TestDedupStillSuppressesWithinTheWindow guards the behaviour the bound must not
|
||||
// break: a repeat inside the window is still collapsed, and one outside it is not.
|
||||
func TestDedupStillSuppressesWithinTheWindow(t *testing.T) {
|
||||
clock := time.Now()
|
||||
n, _ := notifierAt(t, &clock)
|
||||
|
||||
fire := func() bool {
|
||||
n.mu.Lock()
|
||||
defer n.mu.Unlock()
|
||||
return n.suppressedLocked("killswitch\x00Kill-switch engaged")
|
||||
}
|
||||
if fire() {
|
||||
t.Fatalf("first fire was suppressed")
|
||||
}
|
||||
clock = clock.Add(dedupWindow / 2)
|
||||
if !fire() {
|
||||
t.Fatalf("a repeat inside the window was NOT suppressed")
|
||||
}
|
||||
clock = clock.Add(dedupWindow)
|
||||
if fire() {
|
||||
t.Fatalf("a repeat past the window was suppressed")
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,87 @@
|
||||
package alert
|
||||
|
||||
import (
|
||||
"net/http"
|
||||
"net/http/httptest"
|
||||
"sync/atomic"
|
||||
"testing"
|
||||
|
||||
"github.com/sagernet/sing-box/shater/model"
|
||||
)
|
||||
|
||||
// closingRT is an http.RoundTripper that also implements the CloseIdleConnections
|
||||
// hook http.Client forwards to, so a test can observe whether the client was ever
|
||||
// released. http.Transport implements the same hook — this stands in for it.
|
||||
type closingRT struct {
|
||||
rt http.RoundTripper
|
||||
closed atomic.Int32
|
||||
}
|
||||
|
||||
func (c *closingRT) RoundTrip(r *http.Request) (*http.Response, error) { return c.rt.RoundTrip(r) }
|
||||
func (c *closingRT) CloseIdleConnections() { c.closed.Add(1) }
|
||||
|
||||
// TestDetourClientIsClosedAfterDelivery: the detour factory (engine.HTTPClient)
|
||||
// mints a NEW http.Transport for every call, and that transport's idle connections
|
||||
// are live proxying sessions through an engine outbound whose object the dial
|
||||
// closure pins. The notifier used one per delivery and dropped it, so every alert
|
||||
// left a keep-alive session — and a reference to a possibly-retired engine
|
||||
// generation — behind for the whole idle timeout.
|
||||
func TestDetourClientIsClosedAfterDelivery(t *testing.T) {
|
||||
c, srv := newSink(t)
|
||||
n := New([]model.Alert{{
|
||||
Name: "hook", Enabled: true, Type: "webhook", URL: srv.URL,
|
||||
Events: []string{"killswitch"}, Via: "node:tunnel",
|
||||
}}, nil)
|
||||
|
||||
var made []*closingRT
|
||||
n.SetClientFactory(func(via string) (*http.Client, error) {
|
||||
rt := &closingRT{rt: http.DefaultTransport}
|
||||
made = append(made, rt)
|
||||
return &http.Client{Transport: rt}, nil
|
||||
})
|
||||
|
||||
n.Fire("killswitch", "Kill-switch engaged", "the tunnel is down")
|
||||
n.Wait()
|
||||
|
||||
if c.n() != 1 {
|
||||
t.Fatalf("delivery count = %d, want 1", c.n())
|
||||
}
|
||||
if len(made) != 1 {
|
||||
t.Fatalf("factory called %d times, want 1", len(made))
|
||||
}
|
||||
if got := made[0].closed.Load(); got == 0 {
|
||||
t.Fatalf("the per-delivery detour client was never closed — its idle connections " +
|
||||
"(and the engine outbound its dialer pins) outlive the alert")
|
||||
}
|
||||
}
|
||||
|
||||
// TestDetourClientIsClosedWhenTheSendFails covers the fallback path: a detour that
|
||||
// errors and falls back to direct still built a transport, and that one leaked too.
|
||||
func TestDetourClientIsClosedWhenTheSendFails(t *testing.T) {
|
||||
// A sink that rejects, so the detour send fails and Fallback kicks in.
|
||||
reject := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
||||
w.WriteHeader(http.StatusInternalServerError)
|
||||
}))
|
||||
defer reject.Close()
|
||||
|
||||
n := New([]model.Alert{{
|
||||
Name: "hook", Enabled: true, Type: "webhook", URL: reject.URL,
|
||||
Events: []string{"killswitch"}, Via: "node:tunnel", Fallback: true,
|
||||
}}, nil)
|
||||
|
||||
var rt *closingRT
|
||||
n.SetClientFactory(func(via string) (*http.Client, error) {
|
||||
rt = &closingRT{rt: http.DefaultTransport}
|
||||
return &http.Client{Transport: rt}, nil
|
||||
})
|
||||
|
||||
n.Fire("killswitch", "Kill-switch engaged", "the tunnel is down")
|
||||
n.Wait()
|
||||
|
||||
if rt == nil {
|
||||
t.Fatalf("factory was never called")
|
||||
}
|
||||
if got := rt.closed.Load(); got == 0 {
|
||||
t.Fatalf("a failed detour delivery still leaked its transport")
|
||||
}
|
||||
}
|
||||
@@ -5,6 +5,7 @@ import (
|
||||
"os"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
"sync"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
@@ -74,22 +75,37 @@ func TestExpiryDedupWithinDay(t *testing.T) {
|
||||
// test compresses 24 simulated hours into a few milliseconds of wall clock — so
|
||||
// without this the second delivery is (correctly) suppressed by that window and
|
||||
// the test would be measuring the fixture, not the behaviour.
|
||||
// The clock is shared with the notifier's DELIVERY goroutines (buildPayload
|
||||
// stamps the payload with n.now()), which are still in flight while this body
|
||||
// advances the simulated time — so it needs a lock, not a bare variable.
|
||||
var clockMu sync.Mutex
|
||||
simNow := now
|
||||
e.n.now = func() time.Time { return simNow }
|
||||
setNow := func(t time.Time) {
|
||||
clockMu.Lock()
|
||||
simNow = t
|
||||
clockMu.Unlock()
|
||||
}
|
||||
e.n.now = func() time.Time {
|
||||
clockMu.Lock()
|
||||
defer clockMu.Unlock()
|
||||
return simNow
|
||||
}
|
||||
|
||||
if got := e.Run(m, now); got != 1 {
|
||||
t.Fatalf("first Run fired %d, want 1", got)
|
||||
}
|
||||
// Simulate a minute-by-minute reconcile for the next 23 hours.
|
||||
for i := 1; i <= 23; i++ {
|
||||
simNow = now.Add(time.Duration(i) * time.Hour)
|
||||
if got := e.Run(m, simNow); got != 0 {
|
||||
at := now.Add(time.Duration(i) * time.Hour)
|
||||
setNow(at)
|
||||
if got := e.Run(m, at); got != 0 {
|
||||
t.Fatalf("Run at +%dh fired %d alerts, want 0 (dedup window is 24h)", i, got)
|
||||
}
|
||||
}
|
||||
// Just past the window it may speak again.
|
||||
simNow = now.Add(24*time.Hour + time.Minute)
|
||||
if got := e.Run(m, simNow); got != 1 {
|
||||
past := now.Add(24*time.Hour + time.Minute)
|
||||
setNow(past)
|
||||
if got := e.Run(m, past); got != 1 {
|
||||
t.Errorf("Run just past 24h fired %d, want 1", got)
|
||||
}
|
||||
e.n.Wait()
|
||||
|
||||
@@ -20,6 +20,7 @@ import (
|
||||
"fmt"
|
||||
"io"
|
||||
"net/http"
|
||||
"sort"
|
||||
"strings"
|
||||
"sync"
|
||||
"time"
|
||||
@@ -31,6 +32,29 @@ import (
|
||||
const (
|
||||
httpTimeout = 8 * time.Second
|
||||
dedupWindow = 60 * time.Second
|
||||
|
||||
// maxDedupKeys / keepDedupKeys bound the dedup table.
|
||||
//
|
||||
// Nothing ever deleted from it. Every key that had EVER fired stayed forever,
|
||||
// and its highest-cardinality producer is "new_device:<MAC>" — on a guest
|
||||
// network where clients randomise their MAC per association, that is a fresh
|
||||
// key per device per join, for the life of a daemon that runs for months.
|
||||
//
|
||||
// The real bound is not this cap, it is the window: an entry older than
|
||||
// dedupWindow (60s) can never suppress anything again, so it is pure garbage
|
||||
// and sweeping it costs nothing and changes no behaviour. compactDedupLocked
|
||||
// does that first. The cap only bites when 1024 DISTINCT incidents fired inside
|
||||
// one 60-second window — and there the eviction is a real loss of suppression
|
||||
// state, so it is reported rather than done quietly.
|
||||
//
|
||||
// Why 1024: the new_device watcher polls every ~45s, so a poll would have to
|
||||
// discover a thousand previously-unseen MACs at once to reach it — roughly 20x
|
||||
// the worst guest-network churn this box has seen. At ~100 B per entry (a
|
||||
// 28-byte "new_device:<MAC>" key plus a time.Time plus map overhead) the full
|
||||
// table is ~100 KB of a 512 MB router: cheap enough that a generous headroom
|
||||
// costs nothing.
|
||||
maxDedupKeys = 1024
|
||||
keepDedupKeys = 512
|
||||
)
|
||||
|
||||
// telegramAPIBase is the Telegram Bot API root. A package var so tests can point
|
||||
@@ -43,6 +67,10 @@ type Notifier struct {
|
||||
mu sync.Mutex
|
||||
alerts []model.Alert
|
||||
dedup map[string]time.Time // (event\x00title) -> last fire time
|
||||
// dedupEvicted counts LIVE dedup entries dropped by the capacity bound (see
|
||||
// compactDedupLocked). Expired entries swept out are NOT counted: they could no
|
||||
// longer suppress anything, so dropping them loses nothing.
|
||||
dedupEvicted uint64
|
||||
|
||||
client *http.Client
|
||||
log log.ContextLogger
|
||||
@@ -227,10 +255,69 @@ func (n *Notifier) suppressedLocked(key string) bool {
|
||||
if last, ok := n.dedup[key]; ok && now.Sub(last) < dedupWindow {
|
||||
return true
|
||||
}
|
||||
if len(n.dedup) >= maxDedupKeys {
|
||||
n.compactDedupLocked(now)
|
||||
}
|
||||
n.dedup[key] = now
|
||||
return false
|
||||
}
|
||||
|
||||
// compactDedupLocked bounds the dedup table. Caller holds n.mu.
|
||||
//
|
||||
// Two stages, deliberately distinct because only one of them loses anything:
|
||||
//
|
||||
// 1. Drop every entry older than dedupWindow. Such an entry cannot suppress a
|
||||
// future fire (suppressedLocked already ignores it), so this is garbage
|
||||
// collection, not eviction: no notification changes, and nothing is reported.
|
||||
// On any realistic traffic this stage alone keeps the table at "keys seen in
|
||||
// the last minute".
|
||||
// 2. If the table is STILL full, a thousand distinct incidents fired inside one
|
||||
// window. Now eviction is real — the oldest live entries go, and a repeat of
|
||||
// one of them within its window will notify a second time instead of being
|
||||
// collapsed. That is a visible change in behaviour, so it is logged, with the
|
||||
// running total, rather than silently absorbed.
|
||||
func (n *Notifier) compactDedupLocked(now time.Time) {
|
||||
for k, last := range n.dedup {
|
||||
if now.Sub(last) >= dedupWindow {
|
||||
delete(n.dedup, k)
|
||||
}
|
||||
}
|
||||
if len(n.dedup) < maxDedupKeys {
|
||||
return
|
||||
}
|
||||
|
||||
type kv struct {
|
||||
key string
|
||||
at time.Time
|
||||
}
|
||||
all := make([]kv, 0, len(n.dedup))
|
||||
for k, at := range n.dedup {
|
||||
all = append(all, kv{k, at})
|
||||
}
|
||||
sort.Slice(all, func(i, j int) bool { return all[i].at.Before(all[j].at) })
|
||||
drop := len(all) - keepDedupKeys
|
||||
for i := 0; i < drop; i++ {
|
||||
delete(n.dedup, all[i].key)
|
||||
}
|
||||
n.dedupEvicted += uint64(drop)
|
||||
n.log.Warn("alert: dedup table full (", maxDedupKeys,
|
||||
" incidents inside one ", dedupWindow, " window) — dropped ", drop,
|
||||
" live suppression entries (", n.dedupEvicted,
|
||||
" total); repeats of those incidents may notify twice")
|
||||
}
|
||||
|
||||
// DedupEvicted reports how many LIVE suppression entries the cap has dropped since
|
||||
// start (stage 2 of compactDedupLocked only — the expiry sweep is not counted,
|
||||
// because it loses nothing). Nonzero means alerts may have been delivered twice.
|
||||
func (n *Notifier) DedupEvicted() uint64 {
|
||||
if n == nil {
|
||||
return 0
|
||||
}
|
||||
n.mu.Lock()
|
||||
defer n.mu.Unlock()
|
||||
return n.dedupEvicted
|
||||
}
|
||||
|
||||
// dispatch delivers to a single alert in its own goroutine. Panics are recovered
|
||||
// and logged; a delivery error is logged. It never crashes the daemon.
|
||||
func (n *Notifier) dispatch(a model.Alert, event, title, body string) {
|
||||
@@ -276,6 +363,18 @@ func (n *Notifier) deliver(a model.Alert, event, title, body string) error {
|
||||
|
||||
if detour && factory != nil {
|
||||
client, cerr := factory(via)
|
||||
if cerr == nil && client != nil {
|
||||
// The factory (engine.HTTPClient) builds a BRAND NEW http.Transport per
|
||||
// call, with keep-alive and a 90s idle timeout, and we use it for exactly
|
||||
// one POST. Dropping it without this leaves the idle connection — a real
|
||||
// proxying session through an engine outbound, plus its read and write
|
||||
// loops — alive for the whole idle timeout. Worse, the transport's
|
||||
// DialContext closure captures that outbound object, so the idle
|
||||
// connection PINS a retired engine generation whose close budget is 5
|
||||
// seconds. One alert delivery per minute keeps a permanent rolling set of
|
||||
// them.
|
||||
defer client.CloseIdleConnections()
|
||||
}
|
||||
if cerr != nil {
|
||||
if a.Fallback {
|
||||
n.log.Warn("alert: ", a.Name, " detour ", via, " unavailable (", cerr, ") — falling back to direct")
|
||||
|
||||
+533
-134
@@ -61,10 +61,16 @@ type Applier struct {
|
||||
// silently skipped. "" until the first successful nft load.
|
||||
lastNft string
|
||||
|
||||
// holding is true while the fail-closed HOLDING PLANE is installed: the engine
|
||||
// holding is the LATCH half of the hold state: true while a fail-closed HOLDING
|
||||
// PLANE that THIS Applier installed (holdLocked) is in the kernel — the engine
|
||||
// is down and LAN->WAN forwarding is blocked. Surfaced in Status so the panel
|
||||
// can say "protected but not proxying" instead of showing a healthy-looking UI
|
||||
// over an unprotected router.
|
||||
//
|
||||
// It is deliberately NOT the whole answer: a plane installed by the boot armor
|
||||
// or left by a predecessor blocks the LAN just as hard and never touches this
|
||||
// field. Read it through Holding()/holdingWith(), which add the derived half;
|
||||
// writing an inference into this latch would create a claim nothing clears.
|
||||
// holding + lastWarnings live behind their OWN lock, NOT the apply mutex.
|
||||
// Status() reads them, and Status is polled by the LuCI dashboard, the cron
|
||||
// watchdog and hotplug; making those reads queue behind a running apply
|
||||
@@ -205,11 +211,15 @@ func (a *Applier) configureObservatory(m *model.Model, opts option.Options) {
|
||||
})
|
||||
}
|
||||
|
||||
// TestGroups launches the engine's one-shot exit test (delay + exit address, F2)
|
||||
// of the named groups/chains, returning started=false when a run is already in
|
||||
// flight or the engine is absent. names empty/nil = every group and every chain.
|
||||
// It takes NEITHER the apply mutex nor the flock, so kicking off a test never
|
||||
// blocks behind an Apply.
|
||||
// TestGroups launches the engine's one-shot group/chain test (F2): the engine
|
||||
// asks its observatory for an out-of-turn pass and reports what it measured,
|
||||
// plus the exit address for alive rule-routed targets — it no longer dials
|
||||
// health probes of its own. Returns started=false when a run is already in
|
||||
// flight or the engine is absent. names empty/nil = every group and every
|
||||
// chain. probeURL is passed through for signature stability and IGNORED by the
|
||||
// engine (the probe URL is a global observatory setting now). It takes NEITHER
|
||||
// the apply mutex nor the flock, so kicking off a test never blocks behind an
|
||||
// Apply.
|
||||
func (a *Applier) TestGroups(names []string, probeURL string) (started bool) {
|
||||
if a.eng == nil {
|
||||
return false
|
||||
@@ -338,6 +348,19 @@ func (a *Applier) UpdateSubscription(name string) (added int, err error) {
|
||||
if err != nil {
|
||||
return 0, fmt.Errorf("subscription %q detour %q: %w", name, sub.FetchDetour, err)
|
||||
}
|
||||
// engine.HTTPClient builds a FRESH http.Transport per call, and its dialer
|
||||
// closes over the resolved outbound. Dropping the client on the floor leaves
|
||||
// that transport's idle keep-alives open — those are real proxied sessions
|
||||
// through the engine, a goroutine pair each — and the closure PINS the engine
|
||||
// generation they were dialled on. A generation's shutdown budget is five
|
||||
// seconds; an idle connection outlives it, which is how a retired box stays
|
||||
// alive with its WireGuard devices still held.
|
||||
//
|
||||
// Deferred rather than closed at the end on purpose: every exit from here is
|
||||
// covered, including the four error returns below, and those are the ones that
|
||||
// leak in the field — a subscription whose feed is broken is retried by cron
|
||||
// once a refresh interval, forever.
|
||||
defer client.CloseIdleConnections()
|
||||
}
|
||||
|
||||
// FetchWithInfo additionally parses the provider's `subscription-userinfo`
|
||||
@@ -414,7 +437,7 @@ func (a *Applier) applyLocked(m *model.Model) (bool, error) {
|
||||
|
||||
// (2) engine swap. On error the old engine keeps running and we do NOT touch
|
||||
// netplane — abort and surface the error.
|
||||
changed, err := a.eng.Apply(opts)
|
||||
changed, err := engineApply(a, opts)
|
||||
if err != nil {
|
||||
// ...unless the engine is not running AT ALL. A failed apply that left the
|
||||
// PREVIOUS engine running still has a valid, loaded data plane — replacing
|
||||
@@ -430,99 +453,19 @@ func (a *Applier) applyLocked(m *model.Model) (bool, error) {
|
||||
|
||||
// (3) netplane, fail-closed: on any failure return the error WITHOUT tearing
|
||||
// the engine/table down (kill-switch/table stay up; the watchdog decides).
|
||||
//
|
||||
// Render FIRST, then decide whether anything needs re-asserting. Gating the
|
||||
// whole data plane on the engine's `changed` flag alone was wrong: the engine
|
||||
// hash covers option.Options only, so a purely netplane-visible change (the
|
||||
// kill-switch flipping open->closed, DNS force-intercept, a rule gaining an
|
||||
// iface:/zone: source, an inbound device rename) hashed identical, took the
|
||||
// fast-path, and left the OLD ruleset loaded while the panel reported success.
|
||||
// A stale-but-loaded fail-open table is exactly the leak this audit is about.
|
||||
// Resolve WHERE the untunnelable-protocol drop applies before rendering. The
|
||||
// engine is already running by this point (step 2), so its loaded rule-sets can
|
||||
// supply the addresses a geoip list contributes; a stopped engine or a list that
|
||||
// has not downloaded yet yields no plan and the conservative blanket drop.
|
||||
//
|
||||
// The resolved addresses are rendered INTO the ruleset text, so the idempotence
|
||||
// check below sees them: a refreshed geoip list changes the text and triggers a
|
||||
// real reload, rather than leaving a stale plan loaded (the D3 trap).
|
||||
untunPlan := a.untunnelablePlanFor(m, opts)
|
||||
ruleset, nftWarnings, err := netplane.RenderNftPlanAt(m, untunPlan, now)
|
||||
if err != nil {
|
||||
// Includes the refusal on an unusable interface name with a closed
|
||||
// kill-switch: the previous ruleset stays loaded and keeps protecting the
|
||||
// LAN while the operator fixes the name.
|
||||
return changed, err
|
||||
}
|
||||
nftCurrent := ruleset == a.lastNft && netplane.TableExists()
|
||||
if !nftCurrent {
|
||||
if err := netplane.ApplyNft(ruleset); err != nil {
|
||||
// The engine may be up, but with no table loaded nothing is diverted into
|
||||
// it — LAN traffic goes straight out the WAN. That is the same silent
|
||||
// fail-open as a dead engine, so it gets the same answer: if there is no
|
||||
// table at all, hold the line rather than leave the LAN exposed.
|
||||
if !netplane.TableExists() {
|
||||
a.holdLocked(m, err)
|
||||
}
|
||||
return changed, err
|
||||
}
|
||||
a.lastNft = ruleset
|
||||
}
|
||||
// ApplyRouting is idempotent by del-then-add, which means it opens a brief
|
||||
// window with NO fwmark rule installed — during it, diverted packets miss the
|
||||
// `local default dev lo` table. Harmless on a real change (the plane is being
|
||||
// rebuilt anyway), but pointless churn on a no-op reconcile, so skip it when
|
||||
// the ruleset is unchanged AND the rule is verifiably still installed.
|
||||
// routeWarnings carries an egress that was BUILT but cannot route (no nexthop on a
|
||||
// non-point-to-point device). It is deliberately part of the status warning set:
|
||||
// such an egress looks applied everywhere in the UI while being unable to reach
|
||||
// anything off its own subnet. On the fast path nothing was rebuilt, so there is
|
||||
// nothing new to report and the previous set stands.
|
||||
var routeWarnings []string
|
||||
if !nftCurrent || !netplane.RoutingPresent(m.Globals) {
|
||||
var rerr error
|
||||
routeWarnings, rerr = netplane.ApplyRoutingWithWarnings(m)
|
||||
if rerr != nil {
|
||||
return changed, rerr
|
||||
}
|
||||
}
|
||||
if err := netplane.ApplySysctl(); err != nil {
|
||||
return changed, err
|
||||
}
|
||||
// Per-diverted-ingress-iface knobs (accept_local/rp_filter): the static
|
||||
// ApplySysctl above cannot know the LAN device names, and without
|
||||
// accept_local=1 on the ingress iface the tproxied packet never reaches the
|
||||
// engine socket — it escapes to the fail-closed forward drop and the LAN goes
|
||||
// dark. Re-asserted on EVERY apply, never fast-pathed: netifd recreating a
|
||||
// bridge (`ifup lan`, a VLAN change) hands back a device with the kernel
|
||||
// defaults, silently un-setting accept_local behind our back. Fail-closed like
|
||||
// the other netplane steps: return without teardown.
|
||||
if err := netplane.ApplyIfaceSysctlsAt(m, now); err != nil {
|
||||
return changed, err
|
||||
}
|
||||
|
||||
// The plane is COMPLETE only here: table + policy routing + sysctls. ApplyNft
|
||||
// already flushed the DNS conntrack when it loaded the ruleset, but that is
|
||||
// one step too early — ApplyRouting is idempotent BY del-then-add, so it opens
|
||||
// a window in which the fwmark rule is momentarily absent, and any DNS flow
|
||||
// that crosses that window is tracked against a plane that is still being
|
||||
// assembled. Flushing once more now that every piece is in place is what makes
|
||||
// "no entry survives the transition" actually true. Only on a real change (the
|
||||
// fast path assembled nothing), best-effort, and cheap: the :53 entry count is
|
||||
// bounded by the number of clients.
|
||||
if !nftCurrent {
|
||||
if n, ferr := netplane.FlushDNSConntrack(); ferr != nil {
|
||||
a.log.Debug("flush DNS conntrack after plane change: ", ferr)
|
||||
} else if n > 0 {
|
||||
a.log.Debug("plane changed: dropped ", n, " stale DNS conntrack entries")
|
||||
}
|
||||
plane, perr := applyDataPlane(a, m, opts, now)
|
||||
if perr != nil {
|
||||
// The engine is ALREADY running the new config at this point, and the data
|
||||
// plane is not — so nothing the previous apply published is true any more.
|
||||
// Publishing is what this branch used to skip entirely; see abortAfterSwap.
|
||||
return changed, a.abortAfterSwap(m, plane, warnings, configWarnings, perr)
|
||||
}
|
||||
|
||||
// (4) success. Bump the effective-state generation ONLY when something really
|
||||
// moved: a no-op reconcile must not invalidate an armed commit-confirm window
|
||||
// (cron reconciles every minute — counting those would cancel every rollback
|
||||
// that commit-confirm exists to guarantee).
|
||||
if changed || !nftCurrent {
|
||||
if changed || plane.changed {
|
||||
a.stateGen.Add(1)
|
||||
}
|
||||
a.setHolding(false)
|
||||
@@ -540,7 +483,8 @@ func (a *Applier) applyLocked(m *model.Model) (bool, error) {
|
||||
// correctly so here: an egress that cannot reach off its own subnet is a configured
|
||||
// path that silently carries nothing, exactly the class of fault that channel exists
|
||||
// for.
|
||||
ws := collectWarnings(m.Globals, warnings, append(nftWarnings, routeWarnings...), configWarnings, untunPlan.Notes()...)
|
||||
ws := collectWarnings(m.Globals, warnings,
|
||||
append(plane.nftWarnings, plane.routeWarnings...), configWarnings, plane.planNotes...)
|
||||
a.setWarnings(ws)
|
||||
// The log only hears about a CHANGE. Status above always carries the full set;
|
||||
// reprinting it on every no-op reconcile (cron, once a minute, plus every
|
||||
@@ -555,6 +499,187 @@ func (a *Applier) applyLocked(m *model.Model) (bool, error) {
|
||||
return changed, nil
|
||||
}
|
||||
|
||||
// planeOutcome is what the netplane half of an apply did, carried back to
|
||||
// applyLocked so the SAME facts can be published whether it succeeded or failed.
|
||||
// Before it existed, everything the netplane stage learned — the warnings it
|
||||
// rendered, the notes the untunnelable plan produced — was thrown away on any
|
||||
// error, which is why a failed apply left the previous config's verdict standing.
|
||||
type planeOutcome struct {
|
||||
// changed is true when the nft ruleset was actually (re)loaded, i.e. the data
|
||||
// plane moved. Distinct from the engine's own `changed`: the engine hash covers
|
||||
// option.Options only, so a purely netplane-visible change hashes identical.
|
||||
changed bool
|
||||
// stage names the netplane step that failed, in operator words, or "" on
|
||||
// success. It is what the failure warning is addressed to.
|
||||
stage string
|
||||
planNotes []string
|
||||
nftWarnings []string
|
||||
routeWarnings []string
|
||||
}
|
||||
|
||||
// engineApply and applyDataPlane are the two heavy halves of applyLocked, behind
|
||||
// package-level seams for exactly one reason: the PUBLISHING behaviour around
|
||||
// them (what Status says after a stage fails) is the thing this file gets wrong
|
||||
// most easily and can otherwise only be tested on a router, with root, a real
|
||||
// sing-box instance and a real nft binary — i.e. never, in the gate. Production
|
||||
// always runs the real methods; a test substitutes a stage that fails and asserts
|
||||
// what the operator is then told. Same seam pattern as applyHoldNft.
|
||||
var (
|
||||
engineApply = func(a *Applier, opts option.Options) (bool, error) { return a.eng.Apply(opts) }
|
||||
applyDataPlane = (*Applier).applyDataPlaneLocked
|
||||
)
|
||||
|
||||
// applyDataPlaneLocked is step (3) of the pipeline: nft ruleset, policy routing
|
||||
// and sysctls, in that order, fail-closed. Caller holds a.mu and has ALREADY
|
||||
// swapped the engine, so every failure here leaves the router in a mixed state —
|
||||
// which is why the outcome is returned even on error.
|
||||
func (a *Applier) applyDataPlaneLocked(m *model.Model, opts option.Options, now time.Time) (planeOutcome, error) {
|
||||
// Render FIRST, then decide whether anything needs re-asserting. Gating the
|
||||
// whole data plane on the engine's `changed` flag alone was wrong: the engine
|
||||
// hash covers option.Options only, so a purely netplane-visible change (the
|
||||
// kill-switch flipping open->closed, DNS force-intercept, a rule gaining an
|
||||
// iface:/zone: source, an inbound device rename) hashed identical, took the
|
||||
// fast-path, and left the OLD ruleset loaded while the panel reported success.
|
||||
// A stale-but-loaded fail-open table is exactly the leak this audit is about.
|
||||
// Resolve WHERE the untunnelable-protocol drop applies before rendering. The
|
||||
// engine is already running by this point (step 2), so its loaded rule-sets can
|
||||
// supply the addresses a geoip list contributes; a stopped engine or a list that
|
||||
// has not downloaded yet yields no plan and the conservative blanket drop.
|
||||
//
|
||||
// The resolved addresses are rendered INTO the ruleset text, so the idempotence
|
||||
// check below sees them: a refreshed geoip list changes the text and triggers a
|
||||
// real reload, rather than leaving a stale plan loaded (the D3 trap).
|
||||
untunPlan := a.untunnelablePlanFor(m, opts)
|
||||
out := planeOutcome{planNotes: untunPlan.Notes()}
|
||||
|
||||
ruleset, nftWarnings, err := netplane.RenderNftPlanAt(m, untunPlan, now)
|
||||
out.nftWarnings = nftWarnings
|
||||
if err != nil {
|
||||
// Includes the refusal on an unusable interface name with a closed
|
||||
// kill-switch: the previous ruleset stays loaded and keeps protecting the
|
||||
// LAN while the operator fixes the name.
|
||||
out.stage = "rendering the nft ruleset"
|
||||
return out, err
|
||||
}
|
||||
nftCurrent := ruleset == a.lastNft && tableExists()
|
||||
if !nftCurrent {
|
||||
if err := netplane.ApplyNft(ruleset); err != nil {
|
||||
// The engine may be up, but with no table loaded nothing is diverted into
|
||||
// it — LAN traffic goes straight out the WAN. That is the same silent
|
||||
// fail-open as a dead engine, so it gets the same answer: if there is no
|
||||
// table at all, hold the line rather than leave the LAN exposed.
|
||||
if !tableExists() {
|
||||
a.holdLocked(m, err)
|
||||
}
|
||||
out.stage = "loading the nft ruleset"
|
||||
return out, err
|
||||
}
|
||||
a.lastNft = ruleset
|
||||
out.changed = true
|
||||
}
|
||||
// ApplyRouting is idempotent by del-then-add, which means it opens a brief
|
||||
// window with NO fwmark rule installed — during it, diverted packets miss the
|
||||
// `local default dev lo` table. Harmless on a real change (the plane is being
|
||||
// rebuilt anyway), but pointless churn on a no-op reconcile, so skip it when
|
||||
// the ruleset is unchanged AND the rule is verifiably still installed.
|
||||
// routeWarnings carries an egress that was BUILT but cannot route (no nexthop on a
|
||||
// non-point-to-point device). It is deliberately part of the status warning set:
|
||||
// such an egress looks applied everywhere in the UI while being unable to reach
|
||||
// anything off its own subnet. On the fast path nothing was rebuilt, so there is
|
||||
// nothing new to report and the previous set stands.
|
||||
if !nftCurrent || !netplane.RoutingPresent(m.Globals) {
|
||||
routeWarnings, rerr := netplane.ApplyRoutingWithWarnings(m)
|
||||
out.routeWarnings = routeWarnings
|
||||
if rerr != nil {
|
||||
out.stage = "installing the policy routing"
|
||||
return out, rerr
|
||||
}
|
||||
}
|
||||
if err := netplane.ApplySysctl(); err != nil {
|
||||
out.stage = "setting the kernel sysctls"
|
||||
return out, err
|
||||
}
|
||||
// Per-diverted-ingress-iface knobs (accept_local/rp_filter): the static
|
||||
// ApplySysctl above cannot know the LAN device names, and without
|
||||
// accept_local=1 on the ingress iface the tproxied packet never reaches the
|
||||
// engine socket — it escapes to the fail-closed forward drop and the LAN goes
|
||||
// dark. Re-asserted on EVERY apply, never fast-pathed: netifd recreating a
|
||||
// bridge (`ifup lan`, a VLAN change) hands back a device with the kernel
|
||||
// defaults, silently un-setting accept_local behind our back. Fail-closed like
|
||||
// the other netplane steps: return without teardown.
|
||||
if err := netplane.ApplyIfaceSysctlsAt(m, now); err != nil {
|
||||
out.stage = "setting the per-interface sysctls"
|
||||
return out, err
|
||||
}
|
||||
|
||||
// The plane is COMPLETE only here: table + policy routing + sysctls. ApplyNft
|
||||
// already flushed the DNS conntrack when it loaded the ruleset, but that is
|
||||
// one step too early — ApplyRouting is idempotent BY del-then-add, so it opens
|
||||
// a window in which the fwmark rule is momentarily absent, and any DNS flow
|
||||
// that crosses that window is tracked against a plane that is still being
|
||||
// assembled. Flushing once more now that every piece is in place is what makes
|
||||
// "no entry survives the transition" actually true. Only on a real change (the
|
||||
// fast path assembled nothing), best-effort, and cheap: the :53 entry count is
|
||||
// bounded by the number of clients.
|
||||
if out.changed {
|
||||
if n, ferr := netplane.FlushDNSConntrack(); ferr != nil {
|
||||
a.log.Debug("flush DNS conntrack after plane change: ", ferr)
|
||||
} else if n > 0 {
|
||||
a.log.Debug("plane changed: dropped ", n, " stale DNS conntrack entries")
|
||||
}
|
||||
}
|
||||
return out, nil
|
||||
}
|
||||
|
||||
// abortAfterSwap publishes an honest status for an apply that got PAST the engine
|
||||
// swap and then failed in the data plane, and returns the cause unchanged.
|
||||
//
|
||||
// The defect it exists for: every netplane failure used to `return changed, err`
|
||||
// before setTraffic/setWarnings, so Status kept serving the verdict and the
|
||||
// findings of the PREVIOUS config while the engine was already running the new
|
||||
// one. Worse, it did not self-heal — the next cron reconcile hashes identical,
|
||||
// fails at the same stage, and returns at the same place, so the stale verdict
|
||||
// stood for as long as the fault did. A green "Protected" over a half-installed
|
||||
// plane is the exact inversion this audit is about: not an error shown when
|
||||
// things are fine, but calm shown when they are not.
|
||||
//
|
||||
// What it publishes:
|
||||
//
|
||||
// - Traffic goes back to UNKNOWN (the zero value). It is tempting to publish
|
||||
// TrafficOf(opts) here, since the ENGINE really is running those options — but
|
||||
// the verdict describes where the LAN's traffic ends up, and that is decided by
|
||||
// the engine and the data plane together. With one of them from this config and
|
||||
// the other from the last one, the honest answer is that we do not know; the
|
||||
// panel renders unknown, and is required never to render it as protected.
|
||||
// - The warning set is replaced by THIS config's warnings, with a critical entry
|
||||
// naming the stage that failed at the front. The operator gets the news about
|
||||
// the config that is actually loaded, plus the fact that it is only half loaded.
|
||||
//
|
||||
// Caller holds a.mu.
|
||||
func (a *Applier) abortAfterSwap(m *model.Model, plane planeOutcome, generateWarnings []string, configWarnings []model.Warning, cause error) error {
|
||||
a.setTraffic(generate.Traffic{})
|
||||
|
||||
stage := plane.stage
|
||||
if stage == "" {
|
||||
stage = "installing the data plane"
|
||||
}
|
||||
ws := gatherWarnings(m.Globals, generateWarnings,
|
||||
append(plane.nftWarnings, plane.routeWarnings...), configWarnings, plane.planNotes...)
|
||||
ws = finalizeWarnings(append([]Warning{{
|
||||
Severity: SeverityCritical,
|
||||
Section: "netplane",
|
||||
Name: stage,
|
||||
Message: fmt.Sprintf("the engine was switched to this configuration but the data plane could NOT be "+
|
||||
"completed — %s failed: %v. What the kernel holds is part of this configuration and part of the "+
|
||||
"previous one, so where your traffic goes is UNKNOWN: treat this router as unprotected until a "+
|
||||
"reconcile succeeds. It is retried every minute; if it keeps failing, fix the cause or roll back.",
|
||||
stage, cause),
|
||||
}}, ws...))
|
||||
a.setWarnings(ws)
|
||||
a.logWarningsIfChanged(ws)
|
||||
return cause
|
||||
}
|
||||
|
||||
// holdLocked installs the fail-closed HOLDING PLANE when the engine is not
|
||||
// running and the kill-switch is closed. Caller holds a.mu.
|
||||
//
|
||||
@@ -620,10 +745,35 @@ func killSwitchClosed(g model.Globals) bool {
|
||||
// attempted, so the window between daemon start and a successful engine start is
|
||||
// protected rather than open. The daemon calls it at startup.
|
||||
//
|
||||
// It is a no-op when the plane is disabled, the kill-switch is open, or a table
|
||||
// is already loaded. A successful apply replaces the holding plane atomically
|
||||
// (the full ruleset is a single `delete table` + `table` transaction), so the
|
||||
// cost on a healthy boot is a few seconds of blocked forwarding — which is
|
||||
// It is a no-op when the plane is disabled or the kill-switch is open: fail-open
|
||||
// is the operator's documented choice and must not be quietly overridden.
|
||||
//
|
||||
// It is NO LONGER a no-op when a table is already loaded, and that is the change.
|
||||
// The early return on netplane.TableExists() cost two things:
|
||||
//
|
||||
// - It deferred to a table this process cannot inspect. Since the boot armor
|
||||
// landed, the table found here is usually /etc/init.d/shater-armor's snapshot
|
||||
// — which may have been rendered before an interface rename and therefore
|
||||
// protects a device that no longer exists. cmd/shaterd's own armorRender
|
||||
// preference says a fresh render beats a snapshot for exactly this reason.
|
||||
// - It skipped setHolding entirely, so with the boot armor loaded the daemon
|
||||
// reported holding=false over a LAN that really was cut off. Status then
|
||||
// published plane="full" and the apply-failure alert (cmd/shaterd
|
||||
// fireApplyFail) told the operator traffic was NOT being blocked at the very
|
||||
// moment it was — sending them to dismantle a protection that was working.
|
||||
//
|
||||
// Rather than adopt a table it cannot vouch for, ArmHold installs its own,
|
||||
// rendered from the model it was handed: what it then publishes is a fact about
|
||||
// something this process did, not a guess about something it found. That is also
|
||||
// why the fix is not simply "call setHolding(true) when a table is present" — see
|
||||
// foreignHold for the case where installing is impossible and a claim has to be
|
||||
// derived instead.
|
||||
//
|
||||
// Replacing a loaded table costs nothing in exposure: `nft -f` commits as ONE
|
||||
// netlink transaction (netplane.runNftStdin), so there is no window between the
|
||||
// delete and the new table, and a ruleset that fails to load leaves the previous
|
||||
// one in place. A successful apply then replaces the holding plane the same way,
|
||||
// so the cost on a healthy boot is a few seconds of blocked forwarding — which is
|
||||
// precisely what fail-closed is supposed to mean.
|
||||
func (a *Applier) ArmHold(m *model.Model) {
|
||||
if m == nil || !m.Globals.Enabled || !killSwitchClosed(m.Globals) {
|
||||
@@ -636,9 +786,6 @@ func (a *Applier) ArmHold(m *model.Model) {
|
||||
defer release()
|
||||
a.mu.Lock()
|
||||
defer a.mu.Unlock()
|
||||
if netplane.TableExists() {
|
||||
return
|
||||
}
|
||||
a.holdLocked(m, errEngineNotStartedYet)
|
||||
}
|
||||
|
||||
@@ -649,13 +796,107 @@ var errEngineNotStartedYet = errors.New("engine has not started yet")
|
||||
// running nft.
|
||||
var applyHoldNft = netplane.ApplyNft
|
||||
|
||||
// Holding reports whether the fail-closed holding plane is currently installed
|
||||
// (engine down, LAN->WAN blocked). Takes only the leaf lock, so it never waits
|
||||
// on a running apply.
|
||||
// tableExists and bootArmorPresent are the two OUTSIDE-WORLD facts this file's
|
||||
// honesty now rests on: is our table in the kernel, and does the persisted
|
||||
// fail-closed plane exist on flash. Seams for the same reason as engineApply and
|
||||
// applyDataPlane — what the daemon SAYS about a plane it did not install can
|
||||
// otherwise only be exercised on a router, with root, a real nft and a real
|
||||
// /etc/shater, i.e. never, in the gate. Production reads netplane.
|
||||
var (
|
||||
tableExists = netplane.TableExists
|
||||
bootArmorPresent = netplane.BootArmorPresent
|
||||
// teardownNft is a seam for the same reason: whether the table is REMOVED or
|
||||
// REPLACED on the way out is the whole of the restart-gap fix, and it can
|
||||
// otherwise only be observed on a router with a real nft.
|
||||
teardownNft = netplane.TeardownNft
|
||||
)
|
||||
|
||||
// Holding reports whether forwarded LAN traffic is currently being BLOCKED by a
|
||||
// fail-closed plane while the engine is down. It never waits on a running apply.
|
||||
//
|
||||
// It has two sources, and the second one is the whole point.
|
||||
//
|
||||
// The LATCH (a.holding) is written by holdLocked, i.e. when THIS process
|
||||
// installed the plane. It used to be the only source, which quietly made this a
|
||||
// record of what the process had DONE rather than a description of the router.
|
||||
// The boot armor (netplane/armor.go) opened the gap: /etc/init.d/shater-armor
|
||||
// loads the persisted holding plane at START=21, long before the daemon exists,
|
||||
// and the daemon's own unreadable-config path (cmd/shaterd armOnUnreadableConfig)
|
||||
// reinstates it without going anywhere near the apply pipeline. In both cases the
|
||||
// LAN really is cut off and the latch reads false. With an unreadable config
|
||||
// NOTHING ever corrects it, because the correction only happens on an apply and
|
||||
// no apply can run: Status publishes plane="full" over a blocked LAN, and
|
||||
// fireApplyFail words its incident as "traffic is NOT being blocked" at the exact
|
||||
// moment it is. That is the inverted lie — it does not hide a fault, it invents
|
||||
// one, and the obvious response to it is to tear down the protection that works.
|
||||
//
|
||||
// The DERIVED half (foreignHold) closes that, and it self-clears: it is computed
|
||||
// at read time from the kernel and the engine, so it goes false the instant
|
||||
// either fact changes, whereas a latch set from an inference would have to be
|
||||
// remembered to be cleared. Same reasoning as Warnings' live half.
|
||||
func (a *Applier) Holding() bool {
|
||||
return a.holdingWith(tableExists(), a.eng != nil && a.eng.Running())
|
||||
}
|
||||
|
||||
// holdingWith is Holding against facts the caller has already established, so
|
||||
// Status does not shell out to `nft list table` twice for one poll.
|
||||
func (a *Applier) holdingWith(tableLoaded, engineUp bool) bool {
|
||||
a.stateMu.RLock()
|
||||
defer a.stateMu.RUnlock()
|
||||
return a.holding
|
||||
latched := a.holding
|
||||
a.stateMu.RUnlock()
|
||||
return latched || foreignHold(tableLoaded, engineUp)
|
||||
}
|
||||
|
||||
// foreignHold answers: is a plane THIS PROCESS DID NOT INSTALL holding the LAN?
|
||||
//
|
||||
// It deliberately does not try to identify the loaded table, because it cannot.
|
||||
// netplane exposes no read-back of the loaded ruleset, `#` comments do not
|
||||
// survive `nft -f`, and the two candidates — our holding plane, or a full ruleset
|
||||
// left behind by a generation that died without tearing down — are not
|
||||
// distinguishable from here. Guessing which one it is would be a second lie
|
||||
// inside the fix for the first.
|
||||
//
|
||||
// It does not need to. What `holding` asserts is not "the holding plane is the
|
||||
// object in the kernel", it is "the engine is down and forwarded traffic is being
|
||||
// dropped" — and BOTH candidates do that, provided they were rendered from an
|
||||
// enabled, fail-closed config:
|
||||
//
|
||||
// - the holding plane drops by construction: netplane.RenderHoldNftAt ends in
|
||||
// `<iif> meta nfproto ipv4 drop` and its v6 twin;
|
||||
// - a leftover FULL ruleset drops too, by design. With no engine socket the
|
||||
// `tproxy` statement returns NFT_BREAK, which aborts its own rule before the
|
||||
// trailing `meta mark set ... accept`, so the packet reaches the forward chain
|
||||
// unmarked and meets the PRIMARY FAIL-CLOSED DROP there (netplane/nft.go).
|
||||
// That is the kill-switch leak this project fixed in the forward chain, and
|
||||
// netplane/armor.go restates the property in prose.
|
||||
//
|
||||
// So the question that actually decides the claim is whether the last config this
|
||||
// router applied was enabled AND fail-closed — and the boot armor's PRESENCE is
|
||||
// exactly that fact, by its own contract ("the file IS the arm token",
|
||||
// netplane/armor.go): written after every successful config read while enabled
|
||||
// with the kill switch closed, removed the moment either stops being true. A
|
||||
// leftover full plane from a kill_switch=open config — the one that really does
|
||||
// drop nothing — therefore reads FALSE here, which is the answer the operator
|
||||
// needs and the case condition (1) of this fix exists to protect.
|
||||
//
|
||||
// engineUp must be false. With the engine up, a loaded table is the working full
|
||||
// plane doing its job and nothing is being held; this is also what keeps a
|
||||
// successful apply's setHolding(false) from being undone one line later.
|
||||
//
|
||||
// Two residual gaps, named rather than papered over:
|
||||
//
|
||||
// - a full plane rendered under kill_switch=open, still loaded after the
|
||||
// operator closed the switch and the armor was rewritten, reads as holding
|
||||
// while it is not. Any reconcile closes it, and cron runs one a minute.
|
||||
// - a boot armor whose device set predates an interface rename protects the old
|
||||
// name. That is a property of the snapshot, not of this predicate, and it is
|
||||
// why ArmHold now replaces it with a fresh render the moment this daemon has a
|
||||
// readable model.
|
||||
func foreignHold(tableLoaded, engineUp bool) bool {
|
||||
if !tableLoaded || engineUp {
|
||||
return false
|
||||
}
|
||||
return bootArmorPresent()
|
||||
}
|
||||
|
||||
func (a *Applier) setHolding(v bool) {
|
||||
@@ -730,7 +971,42 @@ func (a *Applier) Reconcile() (changed bool, err error) {
|
||||
|
||||
// Teardown is the honest teardown: engine.Close + netplane routing/nft teardown +
|
||||
// clear ACTIVE_FLAG, under the flock. Safe to call when nothing is up (idempotent).
|
||||
func (a *Applier) Teardown() error {
|
||||
//
|
||||
// This is the "everything goes" form — the operator disabled the stack or stopped
|
||||
// the service. A process that is being REPLACED wants TeardownExiting instead.
|
||||
func (a *Applier) Teardown() error { return a.teardown(nil) }
|
||||
|
||||
// TeardownExiting is Teardown for a daemon that is going away, with the one
|
||||
// ordering that never leaves the LAN uncovered.
|
||||
//
|
||||
// THE GAP THIS CLOSES (MEASURED on the stand: 80-90 ms, twice). The exit path used
|
||||
// to be `Teardown(); armOnExit()` — TeardownNft DELETED the table, and only then
|
||||
// was the fail-closed holding plane installed. Two nft transactions, and between
|
||||
// them the `inet shater` table does not exist at all, so fw4's `lan -> wan ACCEPT`
|
||||
// is the only policy on the box and the whole LAN forwards in the clear. That is
|
||||
// not a boot-time window: it is every `restart`, every `reload_service` (i.e.
|
||||
// every LuCI Save & Apply) and every package upgrade. The width is two `nft`
|
||||
// invocations, so it does NOT grow with the engine — eng.Close runs before the
|
||||
// table is touched — but it is the whole LAN, in the clear, every time.
|
||||
//
|
||||
// The comment that used to sit on the call site — "AFTER the teardown, never
|
||||
// before: Teardown deletes the table, so a plane installed first would simply be
|
||||
// removed again" — described the mechanism correctly and drew the wrong conclusion
|
||||
// from it: the answer is not to arm later, it is to stop deleting.
|
||||
//
|
||||
// So arm FIRST and then skip the delete. RenderHoldNft's output is a single
|
||||
// `nft -f` script that opens with `table inet shater` / `delete table inet shater`
|
||||
// / `table inet shater { ... }` — one netlink transaction, in which the table is
|
||||
// REPLACED rather than removed and re-added. The kernel never observes its
|
||||
// absence, so a sampler cannot either.
|
||||
//
|
||||
// arm reports whether it actually installed a plane. When it did NOT — a real
|
||||
// `stop`, or a handoff with kill_switch=open, where fail-open is the operator's
|
||||
// documented choice — the table is removed exactly as before. "Keep the table"
|
||||
// therefore follows from "a plane is standing", never from the caller's intent.
|
||||
func (a *Applier) TeardownExiting(arm func() bool) error { return a.teardown(arm) }
|
||||
|
||||
func (a *Applier) teardown(arm func() bool) error {
|
||||
release, err := lockExclusive()
|
||||
if err != nil {
|
||||
return err
|
||||
@@ -739,6 +1015,15 @@ func (a *Applier) Teardown() error {
|
||||
a.mu.Lock()
|
||||
defer a.mu.Unlock()
|
||||
|
||||
// Install the successor plane BEFORE anything is dismantled, and do it while
|
||||
// still holding the apply lock, so a concurrent apply cannot slip between the
|
||||
// swap and the teardown. arm must not call back into the Applier (cmd/shaterd's
|
||||
// armOnExit goes straight to model + netplane) or this deadlocks.
|
||||
kept := false
|
||||
if arm != nil {
|
||||
kept = arm()
|
||||
}
|
||||
|
||||
// Stop the observatory BEFORE closing the engine, and wait for an in-flight
|
||||
// tick: a teardown must not leave probe dials racing a box that is going away.
|
||||
a.eng.StopObservatory()
|
||||
@@ -761,8 +1046,14 @@ func (a *Applier) Teardown() error {
|
||||
if err := netplane.TeardownRouting(m); err != nil && firstErr == nil {
|
||||
firstErr = err
|
||||
}
|
||||
if err := netplane.TeardownNft(); err != nil && firstErr == nil {
|
||||
firstErr = err
|
||||
// The policy routing above is safe to remove either way: the holding plane is a
|
||||
// single `forward` chain of accepts and drops and consults no routing table, so
|
||||
// it keeps working with the ip rules gone. The TABLE is the one thing that must
|
||||
// not be removed out from under it.
|
||||
if !kept {
|
||||
if err := teardownNft(); err != nil && firstErr == nil {
|
||||
firstErr = err
|
||||
}
|
||||
}
|
||||
// Put the per-ingress-iface knobs back the way we found them. With the table
|
||||
// and the policy routing gone, a lingering accept_local=1 / rp_filter=0 on a
|
||||
@@ -774,7 +1065,11 @@ func (a *Applier) Teardown() error {
|
||||
clearActiveFlag(a.log)
|
||||
a.lastGood = nil
|
||||
a.lastNft = ""
|
||||
a.setHolding(false)
|
||||
// Not a blanket false any more: with a holding plane standing, forwarded LAN
|
||||
// traffic really IS being blocked, and saying otherwise here is the inverted lie
|
||||
// Holding()'s doc comment is about — the process is exiting, but Status can
|
||||
// still be read over the control socket before it does.
|
||||
a.setHolding(kept)
|
||||
a.setTraffic(generate.Traffic{})
|
||||
a.setWarnings(nil)
|
||||
// The plane is gone, so the logged set no longer describes anything. Forget it,
|
||||
@@ -936,12 +1231,28 @@ func (a *Applier) Rollback() error {
|
||||
defer release()
|
||||
a.mu.Lock()
|
||||
defer a.mu.Unlock()
|
||||
if err := a.eng.Rollback(); err != nil {
|
||||
m, err := rollbackEngineAndPlane(a)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
a.publishEngineRollback(m)
|
||||
return nil
|
||||
}
|
||||
|
||||
// rollbackEngineAndPlane is the ACTION half of the no-snapshot rollback: drive the
|
||||
// engine back to its predecessor config and re-assert the data plane from current
|
||||
// UCI. It returns the model the plane was rebuilt from. Caller holds a.mu.
|
||||
//
|
||||
// It is a variable for the same reason as engineApply/applyDataPlane: what this
|
||||
// rollback PUBLISHES afterwards is the part that was wrong, and it cannot be
|
||||
// exercised at all without two real engine generations and a real nft binary.
|
||||
var rollbackEngineAndPlane = func(a *Applier) (*model.Model, error) {
|
||||
if err := a.eng.Rollback(); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
m, err := model.ReadUCI()
|
||||
if err != nil {
|
||||
return err
|
||||
return nil, err
|
||||
}
|
||||
// One clock for the whole re-assert, same as applyLocked: the ruleset's
|
||||
// divert set and the iface sysctls below must agree on the profile-effective
|
||||
@@ -953,14 +1264,14 @@ func (a *Applier) Rollback() error {
|
||||
a.log.Warn("netplane: ", w)
|
||||
}
|
||||
if err != nil {
|
||||
return err
|
||||
return nil, err
|
||||
}
|
||||
if err := netplane.ApplyNft(ruleset); err != nil {
|
||||
return err
|
||||
return nil, err
|
||||
}
|
||||
a.lastNft = ruleset
|
||||
if err := netplane.ApplyRouting(m); err != nil {
|
||||
return err
|
||||
return nil, err
|
||||
}
|
||||
// The sysctl half must be re-asserted here too. It used to be missing: a
|
||||
// rollback that changes the set of diverted ingress devices (a different
|
||||
@@ -969,9 +1280,60 @@ func (a *Applier) Rollback() error {
|
||||
// then never reaches the engine socket and that network goes dark after a
|
||||
// rollback, which is precisely when the operator can least afford it.
|
||||
if err := netplane.ApplySysctl(); err != nil {
|
||||
return err
|
||||
return nil, err
|
||||
}
|
||||
return netplane.ApplyIfaceSysctlsAt(m, now)
|
||||
if err := netplane.ApplyIfaceSysctlsAt(m, now); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
return m, nil
|
||||
}
|
||||
|
||||
// publishEngineRollback makes Status describe the router the no-snapshot rollback
|
||||
// just produced, instead of the one it rolled away FROM. Caller holds a.mu.
|
||||
//
|
||||
// The defect: this path touched none of the publishers. Apply a tunnel config,
|
||||
// Confirm it (which consumes the snapshot), then roll back later — the engine goes
|
||||
// to its predecessor, which may well be the `default -> direct` config, and the
|
||||
// panel keeps showing the tunnel verdict and the tunnel config's warnings, in
|
||||
// green, indefinitely. The whole LAN is on the plain WAN with its real address and
|
||||
// the UI says Protected. Nothing else corrects it: the verdict is only ever
|
||||
// rewritten by a successful apply, and a rollback is not one.
|
||||
//
|
||||
// The verdict published is UNKNOWN, not a computed one, and that is the honest
|
||||
// answer rather than a lazy one: engine.Rollback re-applies option.Options that
|
||||
// this process no longer holds (the engine keeps them, apply does not), so there
|
||||
// is nothing here to run generate.TrafficOf over. Guessing from current UCI would
|
||||
// be worse than saying nothing — UCI is the config we rolled AWAY from. Unknown is
|
||||
// rendered as unknown by the panel and never as protected, and the next reconcile
|
||||
// (cron, within a minute) replaces it with the truth.
|
||||
func (a *Applier) publishEngineRollback(m *model.Model) {
|
||||
a.setTraffic(generate.Traffic{})
|
||||
a.setHolding(false) // a full plane was just loaded; whatever hold there was is over
|
||||
// lastGood is the teardown/rollback target, and the data plane was just built
|
||||
// from m — so m is what a later Teardown must know about to remove the right
|
||||
// marks and routing tables.
|
||||
a.lastGood = m
|
||||
// The observatory's reachability plan was built from the config we rolled away
|
||||
// from: left running it probes outbounds that may no longer exist and files the
|
||||
// results against tags the running box does not have. There is no plan to
|
||||
// replace it with (see above), so stop probing until the next apply installs one.
|
||||
if a.eng != nil {
|
||||
a.eng.StopObservatory()
|
||||
}
|
||||
ws := finalizeWarnings([]Warning{{
|
||||
Severity: SeverityCritical,
|
||||
Section: "engine",
|
||||
Name: "rollback",
|
||||
Message: "the engine was rolled back to the configuration that ran before the current one. " +
|
||||
"That configuration is not the one on disk, so where your traffic goes and what was left " +
|
||||
"un-applied are both UNKNOWN until the next reconcile (within a minute) re-applies the " +
|
||||
"saved config and reports on it. Do not read this router as protected in the meantime.",
|
||||
}})
|
||||
a.setWarnings(ws)
|
||||
a.logWarningsIfChanged(ws)
|
||||
// The plane moved, so an armed commit-confirm watcher must see a changed
|
||||
// generation and stand down rather than clobber what we just restored.
|
||||
a.stateGen.Add(1)
|
||||
}
|
||||
|
||||
// canRollback reports whether Rollback would actually revert something: an armed
|
||||
@@ -1015,11 +1377,30 @@ func ActiveFlagPresent() bool {
|
||||
|
||||
// Status is the read-side snapshot printed by `shaterd status` as JSON.
|
||||
//
|
||||
// running the daemon process is up (a live socket reply => true; the offline
|
||||
// stub reports false). Whether the ENGINE is intercepting is carried by
|
||||
// active/table/hash, not by running — a daemon can be up but inert.
|
||||
// running SHATER IS RUNNING: the daemon answered AND its engine has a started
|
||||
// sing-box instance carrying a config. false therefore covers every way
|
||||
// of not proxying — daemon down, daemon up with a dead engine, disabled,
|
||||
// torn down — and the fields below say which.
|
||||
//
|
||||
// It used to be the literal `true`, on the reasoning that Status() is
|
||||
// only ever called from inside the live daemon. That was true and it was
|
||||
// useless: a constant cannot report anything, and the panel built its
|
||||
// headline on `running && active`, so the "not running" branch was
|
||||
// physically unreachable and an engine that never started showed green.
|
||||
// A field whose only possible value is the reassuring one is worse than
|
||||
// no field: it is a promise the code cannot break.
|
||||
//
|
||||
// "Is the daemon process alive?" is a different question and is answered
|
||||
// by whether the status call returned at all (plus uptime_seconds, which
|
||||
// only a live daemon can produce).
|
||||
// enabled globals.enabled in UCI.
|
||||
// active ACTIVE_FLAG present (a successful enabled apply raised it).
|
||||
// active ACTIVE_FLAG present. This is the "the service is meant to be running"
|
||||
// latch that gates hotplug and cron, NOT a health signal: it is raised by
|
||||
// a successful enabled apply and cleared only by teardown, so it stays up
|
||||
// while the engine is down and the fail-closed holding plane is blocking
|
||||
// the LAN — deliberately, because clearing it would switch off the very
|
||||
// cron reconcile that brings the engine back. Never render it as "we are
|
||||
// proxying"; that is what running/plane/traffic are for.
|
||||
// table the `inet shater` nft table is loaded.
|
||||
// hash the running engine's config hash ("" when the engine is not started).
|
||||
// kill_switch globals.kill_switch in UCI (closed = fail-closed, open = leaky).
|
||||
@@ -1040,9 +1421,14 @@ type Status struct {
|
||||
PanelPort int `json:"panel_port"`
|
||||
CanRollback bool `json:"can_rollback"`
|
||||
|
||||
// EngineRunning is whether a sing-box instance is actually started. It is the
|
||||
// honest answer to "are we proxying?", which running/active/table each only
|
||||
// approximate.
|
||||
// EngineRunning is whether a sing-box instance is actually started.
|
||||
//
|
||||
// It was added as the honest field to stand beside a `running` that was hard-wired
|
||||
// true, and no consumer ever read it. Now that running carries the same fact it is
|
||||
// kept as its explicit, unambiguous name — the two are equal by construction from
|
||||
// the daemon — because it is already in the published API and reading
|
||||
// `engine_running` in a client is self-documenting where `running` needs this
|
||||
// comment.
|
||||
EngineRunning bool `json:"engine_running"`
|
||||
|
||||
// Plane describes what is loaded in the kernel RIGHT NOW:
|
||||
@@ -1137,26 +1523,39 @@ func processUptime(now time.Time) (startedUnix, uptimeSeconds int64) {
|
||||
return now.Unix() - uptimeSeconds, uptimeSeconds
|
||||
}
|
||||
|
||||
// Status returns the live status from this daemon's engine + kernel state. It is
|
||||
// only ever called from within the running daemon, so running=true; engine state
|
||||
// is reflected by Active/Table/Hash (an inert daemon reports running=true but
|
||||
// active=false/table=false/hash="").
|
||||
// Status returns the live status from this daemon's engine + kernel state.
|
||||
//
|
||||
// It is only ever called from within the running daemon, which is exactly why
|
||||
// `running` is read off the engine rather than set to true: from in here the
|
||||
// daemon's own liveness is a tautology, and the only thing left worth reporting
|
||||
// under that name is whether shater is carrying any traffic. A daemon that is up
|
||||
// with a dead engine reports running=false, plane="hold"/"none" and an unknown
|
||||
// traffic verdict — which is the state this field exists to make expressible.
|
||||
func (a *Applier) Status() Status {
|
||||
engineUp := a.eng != nil && a.eng.Running()
|
||||
// Read the kernel ONCE and hand the same fact to both Table and the hold
|
||||
// verdict. Asking twice is not just an extra `nft` fork per poll: the two reads
|
||||
// could disagree across a teardown and publish table=false with plane="hold".
|
||||
tableLoaded := tableExists()
|
||||
s := Status{
|
||||
Running: true,
|
||||
Running: engineUp,
|
||||
Active: ActiveFlagPresent(),
|
||||
Table: netplane.TableExists(),
|
||||
Table: tableLoaded,
|
||||
Hash: a.eng.Hash(),
|
||||
CanRollback: a.canRollback(),
|
||||
EngineRunning: a.eng.Running(),
|
||||
EngineRunning: engineUp,
|
||||
Traffic: a.Traffic(),
|
||||
Warnings: a.Warnings(),
|
||||
}
|
||||
s.StartedUnix, s.UptimeSeconds = processUptime(time.Now())
|
||||
switch {
|
||||
case !s.Table:
|
||||
case !tableLoaded:
|
||||
s.Plane = "none"
|
||||
case a.Holding():
|
||||
case a.holdingWith(tableLoaded, engineUp):
|
||||
// Includes the plane THIS PROCESS DID NOT INSTALL — the boot armor loaded by
|
||||
// /etc/init.d/shater-armor, or what a predecessor left behind. Without that
|
||||
// the branch below claimed "full" (documented as "traffic is diverted into a
|
||||
// RUNNING engine") over a dead engine and a blocked LAN.
|
||||
s.Plane = "hold"
|
||||
default:
|
||||
s.Plane = "full"
|
||||
|
||||
@@ -0,0 +1,270 @@
|
||||
package apply
|
||||
|
||||
// The hold state must describe THE ROUTER, not this process's memory of what it
|
||||
// did.
|
||||
//
|
||||
// The boot armor (netplane/armor.go) put a fail-closed plane in the kernel that
|
||||
// the Applier never installs: /etc/init.d/shater-armor loads it at START=21,
|
||||
// before the daemon exists, and cmd/shaterd reinstates it when the config cannot
|
||||
// be read. The latch behind Holding() knew nothing about either, so the daemon
|
||||
// reported `holding=false` and `plane="full"` over a LAN that was blocked — and
|
||||
// with an unreadable config that state is PERMANENT, because the only thing that
|
||||
// ever wrote the latch was an apply and no apply can run.
|
||||
//
|
||||
// What comes out the other end is an alert (cmd/shaterd fireApplyFail) that
|
||||
// chooses its wording from exactly this bool and tells the operator "Traffic is
|
||||
// NOT being blocked" while it is. That is the inverted failure: not a fault
|
||||
// hidden, but a fault invented — and the obvious response to it is to go and
|
||||
// dismantle the protection that is doing its job.
|
||||
|
||||
import (
|
||||
"errors"
|
||||
"os"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"github.com/sagernet/sing-box/option"
|
||||
"github.com/sagernet/sing-box/shater/engine"
|
||||
"github.com/sagernet/sing-box/shater/model"
|
||||
)
|
||||
|
||||
// stubPlaneFacts pins the two outside-world seams — is our table in the kernel,
|
||||
// is the persisted fail-closed plane on flash — for the duration of a test.
|
||||
func stubPlaneFacts(t *testing.T, table, armor bool) func() {
|
||||
t.Helper()
|
||||
origTable, origArmor := tableExists, bootArmorPresent
|
||||
tableExists = func() bool { return table }
|
||||
bootArmorPresent = func() bool { return armor }
|
||||
return func() { tableExists, bootArmorPresent = origTable, origArmor }
|
||||
}
|
||||
|
||||
// TestHoldingSeesAPlaneThisProcessDidNotInstall is the core regression.
|
||||
//
|
||||
// All three cases share the same latch value (false — this Applier has installed
|
||||
// nothing) and must still produce three different, correct answers.
|
||||
func TestHoldingSeesAPlaneThisProcessDidNotInstall(t *testing.T) {
|
||||
a := New(engine.New(), nil)
|
||||
|
||||
// (1) No table at all. Nothing is protecting the LAN and nothing may claim to.
|
||||
restore := stubPlaneFacts(t, false, true)
|
||||
if a.Holding() {
|
||||
t.Errorf("Holding() = true with no table loaded")
|
||||
}
|
||||
if got := a.Status().Plane; got != "none" {
|
||||
t.Errorf("Plane = %q with no table loaded, want \"none\"", got)
|
||||
}
|
||||
restore()
|
||||
|
||||
// (2) The boot armor's plane IS loaded and this engine never started. This is
|
||||
// what /etc/init.d/shater-armor leaves behind at every boot, and what the
|
||||
// daemon reinstates when /etc/config/shater cannot be read — the case where
|
||||
// nothing else will ever correct the answer.
|
||||
restore = stubPlaneFacts(t, true, true)
|
||||
if !a.Holding() {
|
||||
t.Errorf("Holding() = false while the boot armor's fail-closed plane is loaded and the " +
|
||||
"engine is down — the LAN is blocked and the daemon says it is not; the apply-failure " +
|
||||
"alert would send the operator to fix a protection that is working")
|
||||
}
|
||||
s := a.Status()
|
||||
if s.Plane != "hold" {
|
||||
t.Errorf("Plane = %q over a blocked LAN with a dead engine, want \"hold\" "+
|
||||
"(\"full\" means traffic is diverted into a RUNNING engine)", s.Plane)
|
||||
}
|
||||
if !s.Table {
|
||||
t.Errorf("Table = false although a table is loaded")
|
||||
}
|
||||
if s.Running || s.EngineRunning {
|
||||
t.Errorf("running/engine_running must stay false while holding: %+v", s)
|
||||
}
|
||||
restore()
|
||||
|
||||
// (3) A table is loaded but there is NO armor on flash. refreshBootArmor
|
||||
// removes that file exactly when the operator disables the stack or opens the
|
||||
// kill switch, so what is loaded here is a leftover that drops nothing.
|
||||
// Claiming a hold would be the new lie: it would tell someone who deliberately
|
||||
// chose fail-open that their LAN is cut off.
|
||||
restore = stubPlaneFacts(t, true, false)
|
||||
if a.Holding() {
|
||||
t.Errorf("Holding() = true with no fail-closed armor on flash — a leftover plane from a " +
|
||||
"kill_switch=open config blocks nothing, and saying otherwise is the same lie inverted")
|
||||
}
|
||||
restore()
|
||||
}
|
||||
|
||||
// TestSuccessfulApplyStillClearsTheHold: the self-healing path must survive the
|
||||
// derived half. A read-time inference that ignored the engine would re-assert the
|
||||
// hold one line after setHolding(false) and pin the router in "protected, not
|
||||
// proxying" forever — with the table loaded and the armor on flash, which is the
|
||||
// steady state of every healthy router.
|
||||
func TestSuccessfulApplyStillClearsTheHold(t *testing.T) {
|
||||
a := New(engine.New(), nil)
|
||||
t.Cleanup(func() { _ = a.eng.Close() })
|
||||
t.Cleanup(func() { _ = os.Remove(ActiveFlag) })
|
||||
|
||||
// A REAL started instance: the derived half asks the engine, so a stub would
|
||||
// not exercise the thing under test.
|
||||
if _, err := a.eng.Apply(mixedOn(18841)); err != nil {
|
||||
t.Fatalf("bring a real engine up: %v", err)
|
||||
}
|
||||
if !a.eng.Running() {
|
||||
t.Fatalf("precondition: the engine must be running")
|
||||
}
|
||||
|
||||
// The steady state of a healthy router: our table loaded, armor on flash.
|
||||
defer stubPlaneFacts(t, true, true)()
|
||||
|
||||
a.setHolding(true) // whatever put us on hold before this apply
|
||||
if !a.Holding() {
|
||||
t.Fatalf("precondition: the latch must read through")
|
||||
}
|
||||
|
||||
m := holdModel("closed")
|
||||
m.Globals.GroupHealth = false // no background probing from a unit test
|
||||
defer stubApplyStages(t,
|
||||
func(*Applier, option.Options) (bool, error) { return true, nil },
|
||||
func(*Applier, *model.Model, option.Options, time.Time) (planeOutcome, error) {
|
||||
return planeOutcome{changed: true}, nil
|
||||
})()
|
||||
|
||||
a.mu.Lock()
|
||||
_, err := a.applyLocked(m)
|
||||
a.mu.Unlock()
|
||||
if err != nil {
|
||||
t.Fatalf("applyLocked: %v", err)
|
||||
}
|
||||
|
||||
if a.Holding() {
|
||||
t.Fatalf("a successful apply did not clear the hold — the router is proxying and the " +
|
||||
"panel would still show \"protected, not proxying\"")
|
||||
}
|
||||
if got := a.Status().Plane; got != "full" {
|
||||
t.Errorf("Plane = %q after a successful apply with the engine up, want \"full\"", got)
|
||||
}
|
||||
}
|
||||
|
||||
// TestApplyFailureOverAForeignPlaneReportsBlocked pins the input cmd/shaterd's
|
||||
// fireApplyFail words its incident from.
|
||||
//
|
||||
// The shape is the real one: the engine will not start, so applyLocked calls
|
||||
// holdLocked — and holdLocked cannot install anything either (nft refuses, the
|
||||
// overlay is full). The latch therefore stays false. But the boot armor's plane
|
||||
// is still standing in the kernel, so forwarded traffic IS being dropped, and the
|
||||
// incident must say so. With only the latch, this is precisely where the daemon
|
||||
// said "Traffic is NOT being blocked (kill switch is open)" on a router whose
|
||||
// kill switch was closed and whose LAN was cut off.
|
||||
func TestApplyFailureOverAForeignPlaneReportsBlocked(t *testing.T) {
|
||||
a := New(engine.New(), nil)
|
||||
defer stubPlaneFacts(t, true, true)()
|
||||
|
||||
origHold := applyHoldNft
|
||||
applyHoldNft = func(string) error { return errors.New("nft -f (load) failed: no space left on device") }
|
||||
defer func() { applyHoldNft = origHold }()
|
||||
|
||||
defer stubApplyStages(t,
|
||||
func(*Applier, option.Options) (bool, error) {
|
||||
return false, errors.New("start rule-set[geosite]: connection refused")
|
||||
},
|
||||
func(*Applier, *model.Model, option.Options, time.Time) (planeOutcome, error) {
|
||||
t.Errorf("the data-plane stage must not run after the engine stage failed")
|
||||
return planeOutcome{}, nil
|
||||
})()
|
||||
|
||||
a.mu.Lock()
|
||||
_, err := a.applyLocked(holdModel("closed"))
|
||||
a.mu.Unlock()
|
||||
if err == nil {
|
||||
t.Fatalf("applyLocked must surface the engine failure")
|
||||
}
|
||||
|
||||
a.stateMu.RLock()
|
||||
latched := a.holding
|
||||
a.stateMu.RUnlock()
|
||||
if latched {
|
||||
t.Fatalf("precondition: holdLocked could not install a plane, so the latch must be false")
|
||||
}
|
||||
|
||||
// fireApplyFail(notifier, err, applier.Holding()) — this bool picks between
|
||||
// "forwarded LAN traffic is being dropped" and "Traffic is NOT being blocked".
|
||||
if !a.Holding() {
|
||||
t.Fatalf("Holding() = false while a fail-closed plane blocks the LAN: the incident would " +
|
||||
"read \"Traffic is NOT being blocked\" at the exact moment it is being blocked")
|
||||
}
|
||||
}
|
||||
|
||||
// TestArmHoldReplacesAPlaneItDidNotInstall: ArmHold used to return early on
|
||||
// TableExists() and publish nothing, which is how the boot armor's plane came to
|
||||
// be loaded with holding=false. It now installs its own — rendered from the model
|
||||
// it was handed, so the state it publishes is a fact about something this process
|
||||
// did rather than a guess about something it found (and the fresh render also
|
||||
// replaces a snapshot that may predate an interface rename).
|
||||
func TestArmHoldReplacesAPlaneItDidNotInstall(t *testing.T) {
|
||||
loaded := withHoldProbe(t)
|
||||
defer stubPlaneFacts(t, true, true)() // a table is ALREADY loaded
|
||||
a := New(engine.New(), nil)
|
||||
|
||||
a.ArmHold(holdModel("closed"))
|
||||
|
||||
if len(*loaded) != 1 {
|
||||
t.Fatalf("ArmHold loaded %d rulesets over an existing table, want 1 — it deferred to a "+
|
||||
"table it cannot inspect instead of installing one it can vouch for", len(*loaded))
|
||||
}
|
||||
if !a.Holding() {
|
||||
t.Errorf("ArmHold installed the holding plane but did not publish it")
|
||||
}
|
||||
if got := a.Status().Plane; got != "hold" {
|
||||
t.Errorf("Plane = %q after ArmHold, want \"hold\"", got)
|
||||
}
|
||||
}
|
||||
|
||||
// TestArmHoldStillRespectsFailOpen: replacing a foreign table must not become a
|
||||
// licence to install a plane the operator did not ask for. kill_switch=open and
|
||||
// globals.enabled=0 are explicit choices and ArmHold must keep obeying both.
|
||||
func TestArmHoldStillRespectsFailOpen(t *testing.T) {
|
||||
loaded := withHoldProbe(t)
|
||||
defer stubPlaneFacts(t, true, false)()
|
||||
a := New(engine.New(), nil)
|
||||
|
||||
a.ArmHold(holdModel("open"))
|
||||
if len(*loaded) != 0 {
|
||||
t.Errorf("kill_switch=open loaded %d rulesets, want 0", len(*loaded))
|
||||
}
|
||||
|
||||
disabled := holdModel("closed")
|
||||
disabled.Globals.Enabled = false
|
||||
a.ArmHold(disabled)
|
||||
if len(*loaded) != 0 {
|
||||
t.Errorf("globals.enabled=0 loaded %d rulesets, want 0", len(*loaded))
|
||||
}
|
||||
|
||||
a.ArmHold(nil)
|
||||
if len(*loaded) != 0 {
|
||||
t.Errorf("a nil model loaded %d rulesets, want 0", len(*loaded))
|
||||
}
|
||||
if a.Holding() {
|
||||
t.Errorf("nothing was installed, so nothing may be reported as holding")
|
||||
}
|
||||
}
|
||||
|
||||
// TestForeignHoldMatrix pins the predicate itself, including the two facts it
|
||||
// refuses to guess about.
|
||||
func TestForeignHoldMatrix(t *testing.T) {
|
||||
for _, tc := range []struct {
|
||||
name string
|
||||
table, engine, armor bool
|
||||
want bool
|
||||
}{
|
||||
{"no table", false, false, true, false},
|
||||
{"engine up: the table is the working full plane", true, true, true, false},
|
||||
{"armor on flash, engine down: blocked", true, false, true, true},
|
||||
{"no armor: the last config was disabled or fail-open", true, false, false, false},
|
||||
{"nothing at all", false, false, false, false},
|
||||
} {
|
||||
t.Run(tc.name, func(t *testing.T) {
|
||||
defer stubPlaneFacts(t, tc.table, tc.armor)()
|
||||
if got := foreignHold(tc.table, tc.engine); got != tc.want {
|
||||
t.Errorf("foreignHold(table=%v, engineUp=%v) = %v, want %v",
|
||||
tc.table, tc.engine, got, tc.want)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,178 @@
|
||||
package apply
|
||||
|
||||
// Regression tests for the three ways this package used to report calm over a
|
||||
// router that was not doing what its config said. Each of them is the INVERTED
|
||||
// failure — not an error shown when things are fine, but green shown when they
|
||||
// are not — which is the only kind that gets someone hurt.
|
||||
|
||||
import (
|
||||
"errors"
|
||||
"strings"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"github.com/sagernet/sing-box/option"
|
||||
"github.com/sagernet/sing-box/shater/engine"
|
||||
"github.com/sagernet/sing-box/shater/generate"
|
||||
"github.com/sagernet/sing-box/shater/model"
|
||||
)
|
||||
|
||||
// TestStatusRunningReportsTheEngine pins the contract the panel headline is built
|
||||
// on.
|
||||
//
|
||||
// `running` used to be the literal `true` in the only code path that produces it,
|
||||
// so `running && active` — what the panel reads — could not go false however dead
|
||||
// the engine was, and the "not running" branch was unreachable code. A status
|
||||
// field that can only ever hold the reassuring value is not a weak signal, it is
|
||||
// an unfalsifiable claim.
|
||||
func TestStatusRunningReportsTheEngine(t *testing.T) {
|
||||
a := New(engine.New(), nil)
|
||||
|
||||
// A fresh applier's engine has never started: nothing is being proxied, and
|
||||
// the status must be able to say so.
|
||||
s := a.Status()
|
||||
if s.Running {
|
||||
t.Errorf("Status().Running = true with a stopped engine — the field is a constant again")
|
||||
}
|
||||
if s.Running != s.EngineRunning {
|
||||
t.Errorf("running (%v) and engine_running (%v) must agree: they are the same fact",
|
||||
s.Running, s.EngineRunning)
|
||||
}
|
||||
|
||||
// And the hold state — engine down, LAN blocked — must not read as running
|
||||
// either. This is the three-green-lamps case: holding does not clear the
|
||||
// ACTIVE flag (cron needs it to keep retrying), so `active` alone cannot say it.
|
||||
g := model.DefaultGlobals()
|
||||
g.KillSwitch = "open" // the early-return branch of holdLocked
|
||||
a.mu.Lock()
|
||||
a.holdLocked(&model.Model{Globals: g}, errors.New("engine start failed"))
|
||||
a.mu.Unlock()
|
||||
if a.Status().Running {
|
||||
t.Errorf("Status().Running = true while the engine is down and the plane is held")
|
||||
}
|
||||
}
|
||||
|
||||
// TestApplyFailingAfterEngineSwapDropsTheOldVerdict is the defect-3 regression.
|
||||
//
|
||||
// Everything after the engine swap used to `return changed, err` before
|
||||
// setTraffic/setWarnings, so a netplane failure left Status serving the VERDICT
|
||||
// and the FINDINGS of the configuration that no longer runs. It did not self-heal
|
||||
// either: the next cron reconcile hashes identical, fails at the same stage and
|
||||
// returns at the same place, so the stale green stood for as long as the fault.
|
||||
func TestApplyFailingAfterEngineSwapDropsTheOldVerdict(t *testing.T) {
|
||||
a := New(engine.New(), nil)
|
||||
|
||||
// What the previous, fully successful apply published.
|
||||
a.setTraffic(generate.Traffic{Verdict: generate.VerdictTunnel, Default: "auto", TunnelRules: 3})
|
||||
a.setWarnings([]Warning{{
|
||||
Severity: SeverityCritical, Section: "ruleset", Name: "stale",
|
||||
Message: "a finding of the configuration that is no longer running",
|
||||
}})
|
||||
|
||||
// The engine swap succeeds; the data plane does not.
|
||||
restore := stubApplyStages(t,
|
||||
func(a *Applier, opts option.Options) (bool, error) { return true, nil },
|
||||
func(a *Applier, m *model.Model, opts option.Options, now time.Time) (planeOutcome, error) {
|
||||
return planeOutcome{stage: "loading the nft ruleset"}, errors.New("nft: permission denied")
|
||||
})
|
||||
defer restore()
|
||||
|
||||
a.mu.Lock()
|
||||
_, err := a.applyLocked(holdModel("closed"))
|
||||
a.mu.Unlock()
|
||||
if err == nil {
|
||||
t.Fatalf("applyLocked must surface the netplane failure")
|
||||
}
|
||||
|
||||
s := a.Status()
|
||||
if s.Traffic.Verdict != "" {
|
||||
t.Errorf("Traffic.Verdict = %q after a half-installed plane, want \"\" (unknown): "+
|
||||
"the engine runs the new config and the kernel does not, so nobody knows where traffic goes",
|
||||
s.Traffic.Verdict)
|
||||
}
|
||||
var sawAbort bool
|
||||
for _, w := range s.Warnings {
|
||||
if strings.Contains(w.Message, "a finding of the configuration that is no longer running") {
|
||||
t.Errorf("the previous config's findings are still published: %+v", w)
|
||||
}
|
||||
if w.Section == "netplane" && strings.Contains(w.Message, "could NOT be completed") {
|
||||
sawAbort = true
|
||||
if w.Severity != SeverityCritical {
|
||||
t.Errorf("an incomplete data plane is critical, got %q", w.Severity)
|
||||
}
|
||||
if !strings.Contains(w.Message, "loading the nft ruleset") {
|
||||
t.Errorf("the warning must name the stage that failed: %q", w.Message)
|
||||
}
|
||||
}
|
||||
}
|
||||
if !sawAbort {
|
||||
t.Errorf("no warning says the data plane is incomplete; warnings = %+v", s.Warnings)
|
||||
}
|
||||
}
|
||||
|
||||
// TestRollbackWithoutSnapshotRepublishes is the defect-2 regression.
|
||||
//
|
||||
// Apply a tunnel config, Confirm it (which consumes the commit-confirm snapshot),
|
||||
// then roll back later. The no-snapshot branch drives engine.Rollback and rebuilds
|
||||
// the plane — and used to touch none of the publishers, so the panel kept showing
|
||||
// the tunnel verdict and the tunnel config's warnings in green while the engine
|
||||
// had gone back to a predecessor that may route `default -> direct`. The entire
|
||||
// LAN on the plain WAN, under a green "Protected", indefinitely.
|
||||
func TestRollbackWithoutSnapshotRepublishes(t *testing.T) {
|
||||
a := New(engine.New(), nil)
|
||||
a.setTraffic(generate.Traffic{Verdict: generate.VerdictTunnel, Default: "auto", TunnelRules: 2})
|
||||
a.setWarnings([]Warning{{
|
||||
Severity: SeverityWarning, Section: "chain", Name: "hop",
|
||||
Message: "a finding of the configuration we are rolling away from",
|
||||
}})
|
||||
|
||||
m := holdModel("closed")
|
||||
orig := rollbackEngineAndPlane
|
||||
rollbackEngineAndPlane = func(*Applier) (*model.Model, error) { return m, nil }
|
||||
defer func() { rollbackEngineAndPlane = orig }()
|
||||
|
||||
before := a.stateGen.Load()
|
||||
if err := a.Rollback(); err != nil {
|
||||
t.Fatalf("Rollback: %v", err)
|
||||
}
|
||||
|
||||
s := a.Status()
|
||||
if s.Traffic.Verdict != "" {
|
||||
t.Errorf("Traffic.Verdict = %q after an engine rollback, want \"\" (unknown): the running "+
|
||||
"config is one this process cannot describe", s.Traffic.Verdict)
|
||||
}
|
||||
var sawRollback bool
|
||||
for _, w := range s.Warnings {
|
||||
if strings.Contains(w.Message, "rolling away from") {
|
||||
t.Errorf("the pre-rollback findings are still published: %+v", w)
|
||||
}
|
||||
if w.Section == "engine" && w.Name == "rollback" {
|
||||
sawRollback = true
|
||||
if w.Severity != SeverityCritical {
|
||||
t.Errorf("an undescribable running config is critical, got %q", w.Severity)
|
||||
}
|
||||
}
|
||||
}
|
||||
if !sawRollback {
|
||||
t.Errorf("nothing says the router is running a rolled-back config; warnings = %+v", s.Warnings)
|
||||
}
|
||||
if a.LastGood() != m {
|
||||
t.Errorf("last-good must become the model the data plane was rebuilt from")
|
||||
}
|
||||
if a.stateGen.Load() == before {
|
||||
t.Errorf("the plane moved but stateGen did not: an armed commit-confirm watcher " +
|
||||
"would clobber the config we just restored")
|
||||
}
|
||||
}
|
||||
|
||||
// stubApplyStages replaces the two heavy halves of applyLocked for the duration of
|
||||
// a test and returns the restore func.
|
||||
func stubApplyStages(t *testing.T,
|
||||
eng func(*Applier, option.Options) (bool, error),
|
||||
plane func(*Applier, *model.Model, option.Options, time.Time) (planeOutcome, error),
|
||||
) func() {
|
||||
t.Helper()
|
||||
origEngine, origPlane := engineApply, applyDataPlane
|
||||
engineApply, applyDataPlane = eng, plane
|
||||
return func() { engineApply, applyDataPlane = origEngine, origPlane }
|
||||
}
|
||||
@@ -0,0 +1,149 @@
|
||||
package apply
|
||||
|
||||
// The exit path must never leave the LAN uncovered, and "never" is an ORDER, not
|
||||
// an intention.
|
||||
//
|
||||
// WHAT THIS PINS. The daemon's SIGTERM path used to be:
|
||||
//
|
||||
// applier.Teardown() // netplane.TeardownNft() -> `nft delete table inet shater`
|
||||
// armOnExit(handoff) // then, separately, install the fail-closed holding plane
|
||||
//
|
||||
// Two nft transactions. Between them the `inet shater` table does not exist, so
|
||||
// fw4's `lan -> wan ACCEPT` is the only policy on the box and every forwarded LAN
|
||||
// packet leaves in the clear. MEASURED on the stand at 80-90 ms, reproduced twice
|
||||
// with a 35 000-sample run at ~1.3 ms resolution — and this is not a boot-time
|
||||
// window that heals itself: it is every `restart`, every `reload_service` (i.e.
|
||||
// every LuCI Save & Apply), and every package upgrade.
|
||||
//
|
||||
// The old call site even carried a comment explaining the mechanism — "AFTER the
|
||||
// teardown, never before: Teardown deletes the table, so a plane installed first
|
||||
// would simply be removed again" — and drew the wrong conclusion from a correct
|
||||
// observation. The fix is not to arm later, it is to stop deleting: arm first (a
|
||||
// single `nft -f` that opens with `delete table` and closes with the new one, so
|
||||
// the kernel replaces rather than removes), then skip the delete.
|
||||
//
|
||||
// So the property under test is a SEQUENCE, and the test records the order the
|
||||
// two seams are called in. A test that only asserted "the table still exists at
|
||||
// the end" would pass against the broken code.
|
||||
|
||||
import (
|
||||
"errors"
|
||||
"testing"
|
||||
|
||||
"github.com/sagernet/sing-box/shater/engine"
|
||||
)
|
||||
|
||||
// recordTeardownSeams captures the order in which the exit path touches the
|
||||
// kernel: "arm" when a holding plane is installed, "delete" when the table is
|
||||
// removed.
|
||||
func recordTeardownSeams(t *testing.T) (*[]string, func()) {
|
||||
t.Helper()
|
||||
var calls []string
|
||||
orig := teardownNft
|
||||
teardownNft = func() error {
|
||||
calls = append(calls, "delete")
|
||||
return nil
|
||||
}
|
||||
return &calls, func() { teardownNft = orig }
|
||||
}
|
||||
|
||||
// TestTeardownExitingReplacesThePlaneInsteadOfRemovingIt is the regression: with a
|
||||
// successor coming, the table must be swapped and NEVER deleted.
|
||||
func TestTeardownExitingReplacesThePlaneInsteadOfRemovingIt(t *testing.T) {
|
||||
calls, restore := recordTeardownSeams(t)
|
||||
defer restore()
|
||||
|
||||
a := New(engine.New(), nil)
|
||||
if err := a.TeardownExiting(func() bool {
|
||||
*calls = append(*calls, "arm")
|
||||
return true
|
||||
}); err != nil {
|
||||
t.Fatalf("TeardownExiting: %v", err)
|
||||
}
|
||||
|
||||
if len(*calls) != 1 || (*calls)[0] != "arm" {
|
||||
t.Fatalf("exit path did %v, want exactly [arm]: the holding plane must be installed "+
|
||||
"and the table must NOT be deleted — a `delete` here is the 80-90 ms window in which "+
|
||||
"fw4's lan->wan ACCEPT is the only policy on the box", *calls)
|
||||
}
|
||||
if !a.Holding() && tableExists == nil {
|
||||
t.Errorf("unreachable; keeps the linter honest about the seam")
|
||||
}
|
||||
}
|
||||
|
||||
// TestTeardownExitingArmsBeforeItTearsDown pins the ORDER even in the case where
|
||||
// the table does still get removed. Arming has to be the first thing that touches
|
||||
// the kernel; if it ran after the delete we would be back to the two-transaction
|
||||
// gap with extra steps.
|
||||
func TestTeardownExitingArmsBeforeItTearsDown(t *testing.T) {
|
||||
calls, restore := recordTeardownSeams(t)
|
||||
defer restore()
|
||||
|
||||
a := New(engine.New(), nil)
|
||||
// arm reports FALSE: nothing was installed (a render failure, or kill_switch=open
|
||||
// where fail-open is the operator's documented choice). The table must then come
|
||||
// down exactly as it always did.
|
||||
if err := a.TeardownExiting(func() bool {
|
||||
*calls = append(*calls, "arm")
|
||||
return false
|
||||
}); err != nil {
|
||||
t.Fatalf("TeardownExiting: %v", err)
|
||||
}
|
||||
|
||||
if len(*calls) != 2 || (*calls)[0] != "arm" || (*calls)[1] != "delete" {
|
||||
t.Fatalf("exit path did %v, want [arm delete]: arming must precede the delete, and a "+
|
||||
"plane that was NOT installed must not keep the table alive", *calls)
|
||||
}
|
||||
}
|
||||
|
||||
// TestTeardownStillRemovesEverything is the escape hatch. A deliberate `stop`, and
|
||||
// a Reconcile that finds the stack disabled, both come through the plain Teardown
|
||||
// and must dismantle the plane completely — a kill switch that cannot be switched
|
||||
// off is a brick.
|
||||
func TestTeardownStillRemovesEverything(t *testing.T) {
|
||||
calls, restore := recordTeardownSeams(t)
|
||||
defer restore()
|
||||
|
||||
a := New(engine.New(), nil)
|
||||
if err := a.Teardown(); err != nil {
|
||||
t.Fatalf("Teardown: %v", err)
|
||||
}
|
||||
if len(*calls) != 1 || (*calls)[0] != "delete" {
|
||||
t.Fatalf("Teardown did %v, want [delete]: an operator's stop must take the table with it", *calls)
|
||||
}
|
||||
}
|
||||
|
||||
// TestTeardownExitingReportsHolding pins the honesty half: with a plane left
|
||||
// standing the Applier must not go on saying it installed nothing. Status can
|
||||
// still be read over the control socket between the swap and the exit, and
|
||||
// "holding=false over a blocked LAN" is the inverted lie holdstate_test.go is
|
||||
// about, just reached down a different path.
|
||||
func TestTeardownExitingReportsHolding(t *testing.T) {
|
||||
_, restore := recordTeardownSeams(t)
|
||||
defer restore()
|
||||
restoreFacts := stubPlaneFacts(t, true, true)
|
||||
defer restoreFacts()
|
||||
|
||||
a := New(engine.New(), nil)
|
||||
if err := a.TeardownExiting(func() bool { return true }); err != nil {
|
||||
t.Fatalf("TeardownExiting: %v", err)
|
||||
}
|
||||
if !a.Holding() {
|
||||
t.Errorf("Holding() = false right after the exit path left a fail-closed plane standing")
|
||||
}
|
||||
}
|
||||
|
||||
// TestTeardownExitingSurvivesANilArm keeps the plain-Teardown contract explicit:
|
||||
// a nil arm is "nothing to install", not a panic.
|
||||
func TestTeardownExitingSurvivesANilArm(t *testing.T) {
|
||||
calls, restore := recordTeardownSeams(t)
|
||||
defer restore()
|
||||
|
||||
a := New(engine.New(), nil)
|
||||
if err := a.TeardownExiting(nil); err != nil && !errors.Is(err, nil) {
|
||||
t.Fatalf("TeardownExiting(nil): %v", err)
|
||||
}
|
||||
if len(*calls) != 1 || (*calls)[0] != "delete" {
|
||||
t.Fatalf("TeardownExiting(nil) did %v, want [delete]", *calls)
|
||||
}
|
||||
}
|
||||
@@ -1,13 +1,19 @@
|
||||
package apply
|
||||
|
||||
// Deciding WHERE the untunnelable-protocol drop applies.
|
||||
// Deciding WHERE the untunnelable-protocol drop applies, for the ONE policy that
|
||||
// asks: `icmp`.
|
||||
//
|
||||
// The data plane cannot tunnel anything that is not TCP or UDP (kernel TPROXY
|
||||
// needs a socket; the proxy protocols carry TCP streams and UDP datagrams). The
|
||||
// drop that follows from that is only justified for destinations the routing
|
||||
// rules actually send THROUGH the tunnel: where a rule routes direct, the
|
||||
// client's real address already reaches that destination over TCP, so dropping
|
||||
// its ICMP hides nothing and merely breaks diagnostics.
|
||||
// needs a socket; the proxy protocols carry TCP streams and UDP datagrams). Where
|
||||
// a rule routes direct, the client's real address already reaches that
|
||||
// destination over TCP, so dropping its ICMP hides nothing and merely breaks
|
||||
// diagnostics.
|
||||
//
|
||||
// That reasoning used to govern `block` as well, which made `block` identical to
|
||||
// `direct` under the ordinary "tunnel the blocked list, send the rest direct"
|
||||
// configuration — see the essay in netplane/untunnelable.go. `block` now drops
|
||||
// unconditionally and `direct` allows unconditionally; neither reads this file,
|
||||
// and untunnelablePlanFor no longer builds a plan for them at all.
|
||||
//
|
||||
// This file computes the difference, by walking the FULLY RESOLVED routing rules
|
||||
// that generate produced. Using generate's output rather than the raw model is
|
||||
@@ -239,9 +245,31 @@ func buildUntunnelablePlan(opts option.Options, lookup ruleSetCIDRs) *netplane.U
|
||||
}
|
||||
mt.AnyDst = !hasDstMatcher
|
||||
mt.Dst4, mt.Dst6 = netplane.PrefixStrings(dst)
|
||||
if !mt.AnyDst && len(mt.Dst4) == 0 && len(mt.Dst6) == 0 {
|
||||
// The rule names addresses and not one of them survived into a form the
|
||||
// data plane can express — unparseable, or IPv4-mapped IPv6, which
|
||||
// PrefixStrings drops because nftables has no set type for it. The step
|
||||
// would render no line at all, so the walk would silently step OVER a
|
||||
// rule that CAN claim this traffic and let a later rule decide in its
|
||||
// place. That is the over-permissive mistake this file exists to avoid,
|
||||
// so it is undecidable rather than skippable.
|
||||
plan.Warnings = append(plan.Warnings,
|
||||
"a routing rule's addresses cannot be expressed by the firewall, so ping/IPTV/"+
|
||||
"VPN-passthrough traffic is blocked from that rule onwards")
|
||||
return plan
|
||||
}
|
||||
|
||||
// Source predicate, when the rule is scoped to particular clients.
|
||||
// Source predicate, when the rule is scoped to particular clients. Same
|
||||
// reasoning as above, and here the consequence is worse than a skipped step:
|
||||
// an empty source pair reads as "any source", so an ALLOW scoped to three lab
|
||||
// machines would render as an allow for the whole LAN.
|
||||
mt.Src4, mt.Src6 = netplane.PrefixStrings(parsePrefixes(d.SourceIPCIDR))
|
||||
if len(d.SourceIPCIDR) > 0 && len(mt.Src4) == 0 && len(mt.Src6) == 0 {
|
||||
plan.Warnings = append(plan.Warnings,
|
||||
"a routing rule's source addresses cannot be expressed by the firewall, so ping/IPTV/"+
|
||||
"VPN-passthrough traffic is blocked from that rule onwards")
|
||||
return plan
|
||||
}
|
||||
|
||||
plan.Matches = append(plan.Matches, mt)
|
||||
|
||||
@@ -419,10 +447,17 @@ func parsePrefixes(in []string) []netip.Prefix {
|
||||
// rule-set addresses against the RUNNING engine. A nil engine (or a stopped one)
|
||||
// yields lookups that always report "not loaded", so the plan degrades to the
|
||||
// conservative blanket drop on its own.
|
||||
//
|
||||
// ONLY `icmp` has a use for it. The other two rungs are unconditional — `direct`
|
||||
// allows every untunnelable protocol wherever it was going, `block` allows none —
|
||||
// and netplane.untunnelableRules refuses to consult the plan for either. Building
|
||||
// one anyway would push the routing rules' whole address space into kernel memory
|
||||
// (geoip-us alone is ~159 000 prefixes, ~20 MB) to answer a question nothing asks;
|
||||
// worse, it would render the sets into the ruleset text, so a geoip refresh would
|
||||
// churn the data plane for a policy that cannot use it. Since `block` is the
|
||||
// DEFAULT, this is the stock install's path.
|
||||
func (a *Applier) untunnelablePlanFor(m *model.Model, opts option.Options) *netplane.UntunnelablePlan {
|
||||
if netplane.EffectiveUntunnelable(m.Globals) == netplane.UntunnelableDirect {
|
||||
// Everything untunnelable is allowed regardless of destination; computing
|
||||
// (and loading into the kernel) thousands of prefixes would change nothing.
|
||||
if netplane.EffectiveUntunnelable(m.Globals) != netplane.UntunnelableICMP {
|
||||
return nil
|
||||
}
|
||||
lookup := func(tag string) ([]netip.Prefix, bool) { return nil, false }
|
||||
|
||||
@@ -63,11 +63,31 @@ func renderPlan(t *testing.T, m *model.Model, plan *netplane.UntunnelablePlan) s
|
||||
return rs
|
||||
}
|
||||
|
||||
// icmpPolicy puts the model on the ONE policy that consults the destination plan.
|
||||
//
|
||||
// Every assertion about the WALK's rendered form has to be made under it, and
|
||||
// that is a change of contract rather than test bookkeeping: `block` now drops
|
||||
// every untunnelable protocol unconditionally and `direct` accepts every one of
|
||||
// them unconditionally, so netplane.untunnelableRules refuses to read the plan for
|
||||
// either. A render-level test left on the default policy would be asserting
|
||||
// against a section the renderer no longer writes — which is exactly how the
|
||||
// defect survived: it was `block` rendering `direct`, under a test that read the
|
||||
// resulting accept as the feature working.
|
||||
func icmpPolicy(m *model.Model) *model.Model {
|
||||
m.Globals.Untunnelable = netplane.UntunnelableICMP
|
||||
return m
|
||||
}
|
||||
|
||||
// TestOnlyPinnedAddressIsTunnelled is the first scenario from the brief: a rule
|
||||
// sends ONLY 8.8.8.8/32 through the tunnel and everything else goes direct, so
|
||||
// only 8.8.8.8 may be un-pingable and the rest of the internet must answer.
|
||||
//
|
||||
// The PLAN half is unchanged — the walk still resolves the pinned address to a
|
||||
// deny and everything else to the routing default. Only the rendering moved to
|
||||
// `icmp` (see icmpPolicy). What `block` renders for this same configuration is
|
||||
// TestBlockDropsEvenWhenEverythingRoutesDirect, and it is nothing at all.
|
||||
func TestOnlyPinnedAddressIsTunnelled(t *testing.T) {
|
||||
m := tunnelModel()
|
||||
m := icmpPolicy(tunnelModel())
|
||||
m.Rulesets = []model.Ruleset{pinnedIPSet("pin", "8.8.8.8/32")}
|
||||
m.Rules = []model.Rule{
|
||||
{Name: "pin", Enabled: true, Order: 10, DstRuleset: []string{"pin"}, Target: "group:auto"},
|
||||
@@ -106,10 +126,128 @@ func TestOnlyPinnedAddressIsTunnelled(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
// TestBlockDropsEvenWhenEverythingRoutesDirect is the defect, in the exact
|
||||
// configuration that makes it bite.
|
||||
//
|
||||
// "Tunnel the pinned address, send the rest direct" leaves the ROUTING DEFAULT
|
||||
// direct, so the plan's DefaultAllow is true — and `block` used to walk the plan
|
||||
// like `icmp` and inherit that default as a blanket
|
||||
// `meta l4proto != { tcp, udp } accept`. ICMP, ESP, AH, GRE, IGMP and SCTP all
|
||||
// left with the client's real address. That includes a client-run IPsec or PPTP
|
||||
// tunnel: a standing second tunnel beside ours, carrying arbitrary traffic under a
|
||||
// peer we neither route nor filter, for as long as it stays up — which is the very
|
||||
// thing the middle rung exists to keep out of "I just want ping". Meanwhile the
|
||||
// panel promised "Nothing leaves except through the tunnel."
|
||||
//
|
||||
// `block` is the DEFAULT policy, so this was the stock install.
|
||||
//
|
||||
// RED BEFORE THE FIX: with the old untunnelableRules the rendered forward chain
|
||||
// carries `... meta l4proto != { tcp, udp } accept`, emitted from plan.DefaultAllow.
|
||||
func TestBlockDropsEvenWhenEverythingRoutesDirect(t *testing.T) {
|
||||
m := tunnelModel()
|
||||
m.Globals.IPv6 = true
|
||||
m.Globals.Untunnelable = netplane.UntunnelableBlock // the default, spelled out: it is the subject
|
||||
m.Rulesets = []model.Ruleset{pinnedIPSet("pin", "8.8.8.8/32")}
|
||||
m.Rules = []model.Rule{
|
||||
{Name: "pin", Enabled: true, Order: 10, DstRuleset: []string{"pin"}, Target: "group:auto"},
|
||||
{Name: "rest", Enabled: true, Order: 99, Target: "direct"},
|
||||
}
|
||||
plan := planFor(t, m, map[string][]string{"rs-pin": {"8.8.8.8/32"}})
|
||||
if !plan.DefaultAllow {
|
||||
t.Fatalf("precondition: this config must yield a plan whose default is ALLOW, or the "+
|
||||
"test is not exercising the defect at all; plan=%+v", plan)
|
||||
}
|
||||
|
||||
fwd := renderPlan(t, m, plan)
|
||||
for _, line := range strings.Split(fwd, "\n") {
|
||||
if !strings.Contains(line, netplane.UntunnelableFilterExpr()) &&
|
||||
!strings.Contains(line, "echo-request") {
|
||||
continue
|
||||
}
|
||||
t.Errorf("block emitted an untunnelable exception although its whole promise is that "+
|
||||
"there are none: %q", strings.TrimSpace(line))
|
||||
}
|
||||
// The drops now carry the entire policy, so losing one would be silent.
|
||||
if !strings.Contains(fwd, "meta nfproto ipv4 drop") ||
|
||||
!strings.Contains(fwd, "meta nfproto ipv6 drop") {
|
||||
t.Fatalf("block lost a fail-closed drop, so nothing enforces it:\n%s", fwd)
|
||||
}
|
||||
|
||||
// And the applier must not even BUILD a plan for block: nothing reads it, and
|
||||
// building one pushes the routing rules' whole address space into kernel memory
|
||||
// and into the ruleset text (a geoip refresh would then churn the data plane
|
||||
// for a policy that cannot use it).
|
||||
opts, _, err := generate.GenerateWithWarnings(m)
|
||||
if err != nil {
|
||||
t.Fatalf("generate: %v", err)
|
||||
}
|
||||
var a Applier
|
||||
if got := a.untunnelablePlanFor(m, opts); got != nil {
|
||||
t.Errorf("block must build no destination plan at all, got %d step(s) / defaultAllow=%v",
|
||||
len(got.Matches), got.DefaultAllow)
|
||||
}
|
||||
m.Globals.Untunnelable = netplane.UntunnelableDirect
|
||||
if got := a.untunnelablePlanFor(m, opts); got != nil {
|
||||
t.Errorf("direct must build no destination plan either, got %+v", got)
|
||||
}
|
||||
if got := a.untunnelablePlanFor(icmpPolicy(m), opts); got == nil {
|
||||
t.Errorf("icmp is the policy that needs the plan; it must still get one")
|
||||
}
|
||||
}
|
||||
|
||||
// TestSourceScopedRuleDoesNotWidenTheOtherFamily: a step scoped to IPv6 clients
|
||||
// must emit no IPv4 line at all.
|
||||
//
|
||||
// emit() only wrote `ip saddr @set` when THAT family had prefixes, so a rule whose
|
||||
// source_ip_cidr held only IPv6 prefixes rendered an IPv4 line with no source
|
||||
// clause whatsoever — an accept for every IPv4 host on the LAN, out of a rule the
|
||||
// operator scoped to a handful of v6 addresses. The destination half had always
|
||||
// skipped in that situation (`!AnyDst && len(dst) == 0`); the source half was the
|
||||
// asymmetry, and the catch-all collapse in buildUntunnelablePlan reads "scoped"
|
||||
// as the union of both families, so the two disagreed.
|
||||
//
|
||||
// RED BEFORE THE FIX: `... ip daddr @unt_d4_0 accept`, with no `ip saddr`.
|
||||
func TestSourceScopedRuleDoesNotWidenTheOtherFamily(t *testing.T) {
|
||||
m := icmpPolicy(tunnelModel())
|
||||
m.Globals.IPv6 = true
|
||||
m.Rulesets = []model.Ruleset{pinnedIPSet("lab", "198.51.100.0/24", "2001:db8:70::/48")}
|
||||
m.Rules = []model.Rule{
|
||||
{Name: "lab", Enabled: true, Order: 10, Src: []string{"2001:db8:9::/48"},
|
||||
DstRuleset: []string{"lab"}, Target: "direct"},
|
||||
{Name: "dflt", Enabled: true, Order: 99, Target: "group:auto"},
|
||||
}
|
||||
plan := planFor(t, m, map[string][]string{"rs-lab": {"198.51.100.0/24", "2001:db8:70::/48"}})
|
||||
if len(plan.Matches) != 1 {
|
||||
t.Fatalf("expected one step, got %+v", plan.Matches)
|
||||
}
|
||||
step := plan.Matches[0]
|
||||
if len(step.Src4) != 0 || len(step.Src6) != 1 {
|
||||
t.Fatalf("the step must carry v6 sources only: src4=%v src6=%v", step.Src4, step.Src6)
|
||||
}
|
||||
if len(step.Dst4) != 1 || len(step.Dst6) != 1 {
|
||||
t.Fatalf("the destination list is dual-family: dst4=%v dst6=%v", step.Dst4, step.Dst6)
|
||||
}
|
||||
|
||||
fwd := renderPlan(t, m, plan)
|
||||
for _, line := range strings.Split(fwd, "\n") {
|
||||
if !strings.Contains(line, "@unt_d4_0") {
|
||||
continue
|
||||
}
|
||||
t.Errorf("a rule scoped to IPv6 sources emitted an IPv4 line; with no source clause on "+
|
||||
"it that is an accept for the whole LAN: %q", strings.TrimSpace(line))
|
||||
}
|
||||
// ...and the family the rule really does scope must survive, or the guard
|
||||
// over-corrected into dropping the step entirely.
|
||||
if !strings.Contains(fwd, "ip6 saddr @unt_s6_0") ||
|
||||
!strings.Contains(fwd, "ip6 daddr @unt_d6_0 accept") {
|
||||
t.Errorf("the v6 half of the step was lost:\n%s", fwd)
|
||||
}
|
||||
}
|
||||
|
||||
// TestCatchAllTunnelDropsEverything is the mirror case: a catch-all rule into the
|
||||
// tunnel means nothing is provably direct, so everything untunnelable is dropped.
|
||||
func TestCatchAllTunnelDropsEverything(t *testing.T) {
|
||||
m := tunnelModel()
|
||||
m := icmpPolicy(tunnelModel())
|
||||
m.Rules = []model.Rule{{Name: "all", Enabled: true, Order: 99, Target: "group:auto"}}
|
||||
plan := planFor(t, m, nil)
|
||||
|
||||
@@ -132,7 +270,7 @@ func TestCatchAllTunnelDropsEverything(t *testing.T) {
|
||||
// `ru-direct` routes a geoip list direct while everything else is tunnelled, so
|
||||
// exactly those addresses become pingable.
|
||||
func TestGeoIPRulesetResolvesToPingableAddresses(t *testing.T) {
|
||||
m := tunnelModel()
|
||||
m := icmpPolicy(tunnelModel())
|
||||
m.Globals.IPv6 = true
|
||||
m.Rulesets = []model.Ruleset{{
|
||||
Name: "ru", Type: "ipcidr", Source: "geoip", Categories: []string{"ru"},
|
||||
@@ -173,7 +311,7 @@ func TestGeoIPRulesetResolvesToPingableAddresses(t *testing.T) {
|
||||
// NOT read as "contains no addresses". That would let later rules decide and
|
||||
// could allow traffic the plan cannot actually account for.
|
||||
func TestUnloadedRuleSetStaysConservative(t *testing.T) {
|
||||
m := tunnelModel()
|
||||
m := icmpPolicy(tunnelModel())
|
||||
m.Rulesets = []model.Ruleset{{
|
||||
Name: "ru", Type: "ipcidr", Source: "geoip", Categories: []string{"ru"},
|
||||
}}
|
||||
@@ -268,7 +406,7 @@ func TestMigratedDomainAndIPRuleStaysOutOfTheUntunnelablePlan(t *testing.T) {
|
||||
for _, target := range []string{"direct", "block", "group:auto"} {
|
||||
for _, engine := range []string{"up", "down"} {
|
||||
t.Run(target+"/engine-"+engine, func(t *testing.T) {
|
||||
m := migratedRule(target)
|
||||
m := icmpPolicy(migratedRule(target))
|
||||
var loaded map[string][]string
|
||||
if engine == "up" {
|
||||
loaded = map[string][]string{"rs-rule-x": {}, "rs-rule-x-ip": {"203.0.113.0/24"}}
|
||||
@@ -395,7 +533,7 @@ func TestHugePlanLoadsInFullAndReportsItsSize(t *testing.T) {
|
||||
addr := netip.AddrFrom4([4]byte{10, byte(i >> 16), byte(i >> 8), byte(i)})
|
||||
huge = append(huge, netip.PrefixFrom(addr, 32).String())
|
||||
}
|
||||
m := tunnelModel()
|
||||
m := icmpPolicy(tunnelModel())
|
||||
m.Rulesets = []model.Ruleset{{
|
||||
Name: "big", Type: "ipcidr", Source: "geoip", Categories: []string{"us"},
|
||||
}}
|
||||
@@ -532,7 +670,14 @@ func TestPlanNeverAcceptsTCPOrUDP(t *testing.T) {
|
||||
strings.TrimSpace(line))
|
||||
}
|
||||
}
|
||||
if policyLines == 0 {
|
||||
// `block` is the exception, and now it is the point: it emits NO
|
||||
// exception line whatsoever. That silence IS the policy — everything
|
||||
// untunnelable falls through to the fail-closed drops below.
|
||||
if policy == netplane.UntunnelableBlock {
|
||||
if policyLines != 0 {
|
||||
t.Errorf("block must emit no untunnelable exception at all:\n%s", fwd)
|
||||
}
|
||||
} else if policyLines == 0 {
|
||||
t.Errorf("policy %q emitted no lines at all:\n%s", policy, fwd)
|
||||
}
|
||||
if !strings.Contains(fwd, "meta nfproto ipv4 drop") ||
|
||||
@@ -569,7 +714,7 @@ func TestLocalPlaneSurvivesEveryPlan(t *testing.T) {
|
||||
// TestSourceScopedRuleNarrowsTheAllow: a rule scoped to particular clients must
|
||||
// only grant those clients, not everyone.
|
||||
func TestSourceScopedRuleNarrowsTheAllow(t *testing.T) {
|
||||
m := tunnelModel()
|
||||
m := icmpPolicy(tunnelModel())
|
||||
m.Rulesets = []model.Ruleset{pinnedIPSet("lab", "198.51.100.0/24")}
|
||||
m.Rules = []model.Rule{
|
||||
{Name: "lab", Enabled: true, Order: 10, Src: []string{"192.168.9.0/24"},
|
||||
|
||||
+297
-25
@@ -74,25 +74,76 @@ type Warning struct {
|
||||
Message string `json:"message"`
|
||||
}
|
||||
|
||||
// notAppliedTags are the SCREAMING-KEBAB prefixes generate stamps on the one
|
||||
// class of warning that means "you configured this protection and it is NOT in
|
||||
// force right now". They exist precisely so the condition is greppable and
|
||||
// machine-recognisable (generate/ruleset.go, generate/dnsfilter.go say so where
|
||||
// they emit them), which makes them a STRUCTURAL signal rather than a guess at
|
||||
// wording — so they decide severity outright, before anything else is consulted.
|
||||
//
|
||||
// This is the fix for the defect that made this whole classifier untrustworthy:
|
||||
// the tagged texts say "NOT ACTIVE"/"unreachable right now" in words that matched
|
||||
// none of the old markers ("UNREACHABLE" upper-case against "unreachable"
|
||||
// lower-case, "is NOT applied" against "is configured but NOT ACTIVE"), and the
|
||||
// tag also breaks entityRe below, so a blocklist that failed to download — the
|
||||
// single most common real-world fault on this router, and the one the panel has
|
||||
// no other way to show — was published as a plain `warning` under section
|
||||
// "generate" with no name. Meanwhile `ruleset "x": url source with empty url`
|
||||
// parsed cleanly and was graded critical. Severity was, in effect, inverted:
|
||||
// a typo shouted, a network outage whispered.
|
||||
var notAppliedTags = []string{
|
||||
"RULESET-NOT-APPLIED", // generate/ruleset.go:821,997,1242
|
||||
"DNS-FILTER-NOT-APPLIED", // generate/dnsfilter.go:132
|
||||
}
|
||||
|
||||
// criticalMarkers are substrings that identify a warning as "protection you
|
||||
// configured is not in effect".
|
||||
// configured is not in effect", for the texts that carry neither a tag above nor
|
||||
// a protection section below.
|
||||
//
|
||||
// This is a heuristic over free text, and it is one on purpose: generate emits
|
||||
// plain strings today, and inventing a parallel structured warning API across a
|
||||
// package boundary owned by another agent would be a far larger change than the
|
||||
// problem warrants. The markers below are taken verbatim from the actual warning
|
||||
// texts, so they are exact rather than speculative. If generate ever emits its
|
||||
// own severity, this list becomes dead code and the conversion simplifies.
|
||||
// problem warrants. Every entry below is quoted from a warning that a producer
|
||||
// ACTUALLY emits, with the file it comes from — because the previous list had
|
||||
// drifted into fiction: five of its nine entries matched no living text at all.
|
||||
// Three of those five ("not covered", "fail-closed", "REJECTED") described ONE
|
||||
// netplane message (netplane/nft.go:606), which reaches us on the netplane
|
||||
// channel and is graded critical wholesale before classify() ever runs; one
|
||||
// ("left un-blocked") named a message generate/doh.go:178 records as deleted;
|
||||
// one ("UNREACHABLE") was upper-case against a lower-case text. A marker with no
|
||||
// producer is not harmless: it reads as coverage, and it is what let the real
|
||||
// texts go ungraded for as long as they did.
|
||||
//
|
||||
// If generate ever emits its own severity, this list becomes dead code and the
|
||||
// conversion simplifies.
|
||||
var criticalMarkers = []string{
|
||||
"UNREACHABLE", // remote rule-set/blocklist not applied
|
||||
"is NOT applied", // ''
|
||||
"NOT emitted", // block_doh NXDOMAIN rules missing
|
||||
"left un-blocked", // block_doh: upstream resolver excluded
|
||||
"inert", // dns_filter / per-device DNS configured but not working
|
||||
"has NO effect", // dns_mode=fakeip with no fakeip resolver
|
||||
"not covered", // an interface outside the fail-closed guard
|
||||
"fail-closed", // ''
|
||||
"REJECTED", // an unusable interface name
|
||||
// The DoH NXDOMAIN rules were not built (generate/dns.go:41,67), and a routing
|
||||
// rule whose sources the engine cannot see is not built either
|
||||
// (generate/route.go:113) — in both cases the operator's block simply is not there.
|
||||
"NOT emitted",
|
||||
// dns_filter / per-device DNS / dns_intercept configured but not working
|
||||
// (generate/dns.go:32,35,38,58,61,64) — the filter is on in the UI and filtering nothing.
|
||||
"inert",
|
||||
// A dns_rule that survived parsing but matches nothing (generate/dns.go:987).
|
||||
"has NO effect",
|
||||
// A routing rule that was emitted but whose target is never reached
|
||||
// (generate/route.go:113,115): the traffic the operator sent through a tunnel
|
||||
// follows the rules below it and the default instead. Present tense on purpose —
|
||||
// "never applied" (past) is warnUnreachableRules' wording, which is graded by
|
||||
// consequence a few lines below, not swept in here.
|
||||
"never applies",
|
||||
// A rule scoped to one source that now matches the WHOLE network
|
||||
// (generate/route.go:407, generate/dns.go:992). Whatever the rule does — send a
|
||||
// device direct, point it at another resolver — it now does it to every client,
|
||||
// and nothing else in the UI shows that the scope collapsed.
|
||||
"applies to EVERY client on the router",
|
||||
"apply to ALL clients",
|
||||
// DNS that leaves the router in plaintext to the provider while the UI shows a
|
||||
// configured resolver (generate/dns.go:38,64,410) and node hostnames resolved
|
||||
// direct from the real address (generate/dns.go:469). These are leaks of exactly
|
||||
// the kind the tunnel exists to prevent.
|
||||
"in the clear",
|
||||
"your provider sees",
|
||||
// A condition-less rule retired by a later condition-less rule whose target is
|
||||
// `direct` (generate/route.go warnUnreachableRules): the operator's default
|
||||
// policy — a tunnel, or a block — is not the one the router uses, so everything
|
||||
@@ -117,6 +168,22 @@ var protectionSections = map[string]bool{
|
||||
"allowlist": true,
|
||||
}
|
||||
|
||||
// degradedProtectionMarkers are the exceptions to the section rule above: texts
|
||||
// about a protection list that is STILL IN EFFECT.
|
||||
//
|
||||
// Grading these critical is the same defect pointed the other way. The panel's
|
||||
// alarm banner lights on critical and on nothing else, so every critical that
|
||||
// turns out to be cosmetic teaches the operator that the banner means nothing —
|
||||
// and the next one, the one about the blocklist that really did not load, is the
|
||||
// one they will not read. In particular the refresh failure is the NORMAL state
|
||||
// of a Russian router for minutes at a time: the list is served from the copy
|
||||
// compiled earlier and keeps blocking, which is a degradation, not a gap.
|
||||
var degradedProtectionMarkers = []string{
|
||||
"continuing with the copy compiled earlier", // generate/ruleset.go:819 — stale but blocking
|
||||
"is IGNORED", // generate/ruleset.go:445,472 — a redundant field, the list loads
|
||||
"bad update_interval", // generate/ruleset.go:862,1010 — falls back to the default interval
|
||||
}
|
||||
|
||||
// infoMarkers identify operational notes that are not protection gaps.
|
||||
var infoMarkers = []string{"cache:"}
|
||||
|
||||
@@ -124,26 +191,67 @@ var infoMarkers = []string{"cache:"}
|
||||
// consistently, so Section/Name can be recovered from a plain string.
|
||||
var entityRe = regexp.MustCompile(`^([a-z_]+) "([^"]*)": (.*)$`)
|
||||
|
||||
// tagRe matches the SCREAMING-KEBAB prefix of a tagged warning (see notAppliedTags).
|
||||
var tagRe = regexp.MustCompile(`^([A-Z][A-Z0-9-]*): `)
|
||||
|
||||
// taggedEntityRe recovers the entity from a TAGGED warning, whose shape is
|
||||
//
|
||||
// RULESET-NOT-APPLIED: ruleset "ads" is configured but NOT ACTIVE: ...
|
||||
//
|
||||
// i.e. the tag sits where entityRe expects the kind, and the entity is followed by
|
||||
// prose rather than by ": ". Without this the single most important warning on the
|
||||
// router arrived with Section "generate" and no Name, so the panel could neither
|
||||
// group it nor link to the list it is about.
|
||||
var taggedEntityRe = regexp.MustCompile(`^[A-Z][A-Z0-9-]*: ([a-z_]+) "([^"]*)"`)
|
||||
|
||||
// warningFromText normalises one free-text warning. defaultSection is used when
|
||||
// the text carries no `kind "name":` prefix.
|
||||
func warningFromText(text, defaultSection, severity string) Warning {
|
||||
w := Warning{Severity: severity, Section: defaultSection, Message: strings.TrimSpace(text)}
|
||||
if m := entityRe.FindStringSubmatch(w.Message); m != nil {
|
||||
w.Section, w.Name, w.Message = m[1], m[2], m[3]
|
||||
return w
|
||||
}
|
||||
if m := taggedEntityRe.FindStringSubmatch(w.Message); m != nil {
|
||||
// Attribution only — the message is deliberately left WHOLE. The tag is the
|
||||
// operator's grep handle into `logread` (it is documented as such where it is
|
||||
// emitted), so stripping it to save one repetition of the list's name would
|
||||
// cost the one thing the tag exists for.
|
||||
w.Section, w.Name = m[1], m[2]
|
||||
}
|
||||
return w
|
||||
}
|
||||
|
||||
// classify picks a severity from the parsed section plus the message text.
|
||||
// Section wins where it is decisive (see protectionSections); the markers then
|
||||
// catch the global warnings that carry no entity prefix at all.
|
||||
//
|
||||
// Order is the whole design:
|
||||
//
|
||||
// 1. a not-applied TAG is structural and decides outright — it is the producer
|
||||
// saying "this protection is off", not us guessing from prose;
|
||||
// 2. info markers, so a cache relocation never reads as a fault;
|
||||
// 3. the section, for the entity kinds whose entire purpose is to block
|
||||
// something — minus the handful of texts that say the list still works;
|
||||
// 4. the free-text markers, which catch the global warnings that carry no entity
|
||||
// prefix at all.
|
||||
func classify(section, text string) string {
|
||||
if m := tagRe.FindStringSubmatch(text); m != nil {
|
||||
for _, tag := range notAppliedTags {
|
||||
if m[1] == tag {
|
||||
return SeverityCritical
|
||||
}
|
||||
}
|
||||
}
|
||||
for _, m := range infoMarkers {
|
||||
if strings.Contains(text, m) {
|
||||
return SeverityInfo
|
||||
}
|
||||
}
|
||||
if protectionSections[section] {
|
||||
for _, m := range degradedProtectionMarkers {
|
||||
if strings.Contains(text, m) {
|
||||
return SeverityWarning
|
||||
}
|
||||
}
|
||||
return SeverityCritical
|
||||
}
|
||||
for _, m := range criticalMarkers {
|
||||
@@ -163,6 +271,15 @@ func classify(section, text string) string {
|
||||
// fail-closed guard does not cover that interface — always critical.
|
||||
// - configWarnings come from model.Validate (already structured).
|
||||
func collectWarnings(g model.Globals, generateWarnings, netplaneWarnings []string, configWarnings []model.Warning, planWarnings ...string) []Warning {
|
||||
return finalizeWarnings(gatherWarnings(g, generateWarnings, netplaneWarnings, configWarnings, planWarnings...))
|
||||
}
|
||||
|
||||
// gatherWarnings is collectWarnings without the sort and the cap, so a caller
|
||||
// that must FOLD IN a warning of its own (applyLocked's post-swap failure, which
|
||||
// has to say that the data plane is incomplete) can do so and then finalize once.
|
||||
// Sorting and capping a list twice is not equivalent: the second pass would drop
|
||||
// the "N further warning(s) suppressed" disclosure the first pass appended.
|
||||
func gatherWarnings(g model.Globals, generateWarnings, netplaneWarnings []string, configWarnings []model.Warning, planWarnings ...string) []Warning {
|
||||
out := make([]Warning, 0, len(generateWarnings)+len(netplaneWarnings)+len(configWarnings)+1)
|
||||
// The untunnelable-protocol policy always reports what it costs the user; it is
|
||||
// the only one of these that describes correct behaviour rather than a fault.
|
||||
@@ -186,7 +303,12 @@ func collectWarnings(g model.Globals, generateWarnings, netplaneWarnings []strin
|
||||
Message: cw.Message,
|
||||
})
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
// finalizeWarnings sorts critical-first and applies the cap. Call it exactly once
|
||||
// per published set.
|
||||
func finalizeWarnings(out []Warning) []Warning {
|
||||
// Stable sort by descending severity so the cap can only drop the least
|
||||
// important entries, and the panel gets the worst news first.
|
||||
sort.SliceStable(out, func(i, j int) bool {
|
||||
@@ -238,36 +360,186 @@ func untunnelablePolicyWarnings(g model.Globals, planNotes []string) []Warning {
|
||||
return append(out, Warning{Severity: SeverityInfo, Section: section, Name: name, Message: msg})
|
||||
}
|
||||
|
||||
// The egress carrier owns the whole story the moment untunnelable_egress
|
||||
// names an egress: the carried protocols get that egress's own mark in
|
||||
// prerouting and the ROUTING decision sends them out its device with the
|
||||
// kernel's NAT — the forward chain, where the policy's verdicts live, no
|
||||
// longer decides their fate. So this branch sits above every other and
|
||||
// returns its own text; with the option empty, the notes below are
|
||||
// byte-for-byte what they were. Two phrasing rules here are load-bearing.
|
||||
// First, the egress is NEVER called a tunnel unconditionally: the option
|
||||
// accepts any interface/tunnel egress, and on the routers this ships to
|
||||
// that is at least as often a second WAN — another uplink, whose real
|
||||
// address the far end sees — as a WireGuard device. Second, the note must
|
||||
// say out loud that UDP-based VPNs are none of this option's business: an
|
||||
// operator who reads "VPN passthrough" and enables it for a WireGuard
|
||||
// client that already worked through the ordinary tunnel has been misled,
|
||||
// not helped. What the policy still owns is exactly the failure path — a
|
||||
// name that resolves to no interface/tunnel egress, a rule or route that
|
||||
// did not come up — and each tail below says what that failure looks like,
|
||||
// because under `direct` (or an open kill switch) it is a silent leak
|
||||
// through the normal uplink with the real address, and nothing anywhere
|
||||
// else would say so.
|
||||
if egressName := strings.TrimSpace(g.UntunnelableEgress); egressName != "" {
|
||||
var msg, failure string
|
||||
if netplane.L3Enabled(g) {
|
||||
msg = "Ping and Windows tracert keep travelling THROUGH the tunnel, toward every " +
|
||||
"address your rules send to an outbound that can carry plain IP " +
|
||||
"(WireGuard/AmneziaWG) — the L3 ingress claims ICMP before this option is " +
|
||||
"consulted, and addresses your rules send anywhere else " +
|
||||
"(vless/vmess/trojan/shadowsocks and the like) still cannot be pinged at " +
|
||||
"all, deliberately. Everything else the proxy cannot carry — IPsec " +
|
||||
"(ESP/AH), PPTP/GRE, SCTP and every other protocol that is neither TCP " +
|
||||
"nor UDP — now leaves through egress \"" + egressName + "\": the kernel " +
|
||||
"routes it out that interface with that interface's own NAT, and none of " +
|
||||
"it goes through the proxy or follows your routing rules. "
|
||||
failure = "when one of the routes is not in place — the L3 route for ping, the " +
|
||||
"egress route for the rest (a name that matches no interface/tunnel " +
|
||||
"egress, or a rule or route that failed to come up): "
|
||||
} else {
|
||||
msg = "Ping, Windows tracert, IPsec (ESP/AH), PPTP/GRE, SCTP and every other " +
|
||||
"protocol that is neither TCP nor UDP now leave through egress \"" +
|
||||
egressName + "\": the kernel routes them out that interface with that " +
|
||||
"interface's own NAT, and none of it goes through the proxy or follows " +
|
||||
"your routing rules — the hops tracert prints are that interface's path " +
|
||||
"(on Linux and macOS traceroute sends UDP probes instead, which still " +
|
||||
"follow your rules). "
|
||||
failure = "when the egress route is not in place (a name that matches no " +
|
||||
"interface/tunnel egress, or a rule or route that failed to come up): "
|
||||
}
|
||||
msg += "What that buys depends entirely on what the interface IS: a WireGuard " +
|
||||
"interface really is a tunnel, but a second WAN is not — it is just another " +
|
||||
"uplink, and the host on the far end sees that uplink's real address. Two " +
|
||||
"things this option does NOT do: multicast IPTV does not pass this router " +
|
||||
"under any setting, and carrying IGMP out an egress cannot change that; and " +
|
||||
"VPNs that run over UDP (WireGuard, OpenVPN-UDP, IPsec through NAT — IKE on " +
|
||||
"UDP 500, NAT-T on UDP 4500) never needed it: they are ordinary tunnelled " +
|
||||
"traffic, keep following your routing rules exactly as before, and gain " +
|
||||
"nothing from this option. The `untunnelable` policy no longer decides this " +
|
||||
"traffic's fate — routing settles it before the forward chain gets a say — " +
|
||||
"and answers only for failure, " + failure
|
||||
switch {
|
||||
case !killSwitchClosed(g):
|
||||
msg += "with the kill switch open nothing is dropped, so whatever loses its " +
|
||||
"route quietly leaves through your normal uplink with your real IP address."
|
||||
case policy == netplane.UntunnelableDirect:
|
||||
msg += "\"direct\" quietly lets it leave through your normal uplink with " +
|
||||
"your real IP address."
|
||||
case policy == netplane.UntunnelableICMP:
|
||||
msg += "\"icmp\" drops it, excepting only ping — which then quietly leaves " +
|
||||
"with your real IP address instead of failing."
|
||||
default:
|
||||
msg += "\"block\" drops it — an honest loss rather than a silent leak."
|
||||
}
|
||||
return note(egressName, msg)
|
||||
}
|
||||
|
||||
// The L3 ingress rewrites the ICMP half of every note below, so it gets one
|
||||
// text of its own rather than four patched variants: echo is marked in
|
||||
// prerouting and the ROUTING decision carries it into the engine's TUN before
|
||||
// the forward chain — where the policy accepts and the kill-switch drops
|
||||
// live — is ever consulted. That holds under all three policy values and
|
||||
// with the kill switch open alike, which is why this branch sits above the
|
||||
// kill-switch note: "reaches the internet with your real IP address" stops
|
||||
// being true for ping the moment the divert exists. What the policy still
|
||||
// owns is exactly two things, and both are said: the protocols the engine
|
||||
// cannot ingest at all (raw IPsec, PPTP/GRE), and the fallback path a marked
|
||||
// packet takes when the L3 route failed to install — under `direct` (or an
|
||||
// open kill switch) that failure is a SILENT leak with the real address,
|
||||
// under `block` an honest packet loss. The unpingable-through-proxy sentence
|
||||
// is deliberate too: those pings used to be answered by the router itself,
|
||||
// and a fake "alive" is worse than a truthful timeout.
|
||||
if netplane.L3Enabled(g) {
|
||||
msg := "Ping and Windows tracert work and travel THROUGH the tunnel, toward every " +
|
||||
"address your rules send to an outbound that can carry plain IP " +
|
||||
"(WireGuard/AmneziaWG). Addresses your rules send anywhere else " +
|
||||
"(vless/vmess/trojan/shadowsocks and the like) cannot be pinged at all — " +
|
||||
"deliberately: those pings used to be answered by the router itself, reporting " +
|
||||
"hosts alive it had never reached. The hops tracert prints are the tunnel's " +
|
||||
"path, not your own, and IPv6 traceroute shows only the destination, none of " +
|
||||
"the hops on the way. Raw VPN passthrough (IPsec ESP/AH, PPTP/GRE) cannot " +
|
||||
"enter the tunnel at all and stays with the untunnelable policy: "
|
||||
switch {
|
||||
case !killSwitchClosed(g):
|
||||
msg += "with the kill switch open none of it is dropped, so it leaves with your " +
|
||||
"real IP address — and if the L3 route ever fails to come up, ping quietly " +
|
||||
"does the same instead of failing."
|
||||
case policy == netplane.UntunnelableDirect:
|
||||
msg += "\"direct\" lets it out with your real IP address — and if the L3 route " +
|
||||
"ever fails to come up, ping quietly does the same instead of failing."
|
||||
case policy == netplane.UntunnelableICMP:
|
||||
msg += "\"icmp\" drops it, excepting only echo — which now rides the tunnel " +
|
||||
"anyway, so the exception matters just once: if the L3 route ever fails to " +
|
||||
"come up, it lets ping quietly leave with your real IP address instead of " +
|
||||
"failing."
|
||||
default:
|
||||
msg += "\"block\" drops it — and if the L3 route ever fails to come up, ping " +
|
||||
"fails outright rather than leaking."
|
||||
}
|
||||
msg += " VPNs that run over UDP (WireGuard, OpenVPN-UDP, IPsec through NAT) are " +
|
||||
"ordinary tunnelled traffic and are unaffected either way. Multicast IPTV does " +
|
||||
"not pass this router on any setting; the L3 ingress does not change that."
|
||||
return note(policy, msg)
|
||||
}
|
||||
|
||||
// With the kill switch open the forward chain has no drops at all, so nothing
|
||||
// is restricted whatever the policy says. Saying that is more useful than
|
||||
// repeating a promise which is not being kept.
|
||||
//
|
||||
// Neither this note nor the `direct` one below may claim IPTV, for the same
|
||||
// reason the `block` note disclaims it: multicast does not cross this router
|
||||
// under ANY of the three settings. The stream itself is WAN-side inbound and
|
||||
// these rules never match it, and a client's outbound multicast UDP is dropped
|
||||
// by the fail-closed guard regardless of the policy. Promising it here would be
|
||||
// the identical lie to the one just removed from `block`, only in the branch
|
||||
// where the operator is least likely to go looking for the cause.
|
||||
if !killSwitchClosed(g) {
|
||||
if policy == netplane.UntunnelableDirect {
|
||||
return out
|
||||
}
|
||||
return note(policy,
|
||||
"This setting has no effect while the kill switch is open: with the kill switch open "+
|
||||
"nothing is blocked, so ping, IPTV and VPN passthrough all work — and all of them "+
|
||||
"reach the internet with your real IP address.")
|
||||
"This setting has no effect while the kill switch is open: with the kill switch open the "+
|
||||
"forward chain has no drops at all, so ping, traceroute and raw VPN passthrough "+
|
||||
"(IPsec ESP/AH, PPTP/GRE) all work — and every one of them reaches the internet with "+
|
||||
"your real IP address. IPTV is not part of that: multicast does not pass this router "+
|
||||
"on any setting, which is a separate matter from this one.")
|
||||
}
|
||||
|
||||
switch policy {
|
||||
case netplane.UntunnelableDirect:
|
||||
return note(policy,
|
||||
"Ping, IPTV and VPN passthrough (IPsec/PPTP) work everywhere, but they go straight out "+
|
||||
"with your real IP address instead of through the tunnel — they are the kinds of "+
|
||||
"traffic a tunnel cannot carry.")
|
||||
"Ping and traceroute work everywhere, and so does raw VPN passthrough (IPsec ESP/AH, "+
|
||||
"PPTP/GRE) — but all of it goes straight out with your real IP address instead of "+
|
||||
"through the tunnel, because a tunnel cannot carry this kind of traffic. VPNs that "+
|
||||
"run over UDP (WireGuard, OpenVPN-UDP, IPsec through NAT) are ordinary tunnelled "+
|
||||
"traffic and are unaffected either way. IPTV is not covered by this setting at all: "+
|
||||
"multicast does not pass this router on any of the three, so switching to `direct` "+
|
||||
"will not bring it back.")
|
||||
case netplane.UntunnelableICMP:
|
||||
return note(policy,
|
||||
"Ping and traceroute work everywhere, including addresses you send through the tunnel; "+
|
||||
"the host you ping sees your real IP address. IPTV and VPN passthrough (IPsec/PPTP) "+
|
||||
"work only toward addresses your rules route directly.")
|
||||
default:
|
||||
// This text used to say these things "work only toward addresses your rules
|
||||
// route directly". That was written when a `direct` route final made the
|
||||
// untunnelable drop degenerate into a blanket accept — i.e. when the note was
|
||||
// describing the bug rather than the policy. netplane now blocks what it says
|
||||
// it blocks, so the honest sentence is that none of it works at all, and the
|
||||
// note has to name what is and is NOT affected: "ping does not work" sends an
|
||||
// operator hunting a fault, and the difference between raw ESP and IPsec
|
||||
// through NAT is the difference between "my VPN broke" and "my VPN is fine".
|
||||
return note(netplane.UntunnelableBlock,
|
||||
"Ping, traceroute, IPTV and VPN passthrough work only toward addresses your rules route "+
|
||||
"directly — those already see your real IP address anyway. Toward addresses you send "+
|
||||
"through the tunnel they will not work, because a tunnel cannot carry them and they "+
|
||||
"would otherwise leak your real IP address.")
|
||||
"Ping, traceroute, IPsec/PPTP VPN passthrough and IPTV do not work from your devices at "+
|
||||
"all — not even toward addresses your rules route directly. None of this traffic can "+
|
||||
"travel through a tunnel, so rather than let it out with your real IP address it is "+
|
||||
"dropped. Concretely: ping and Windows tracert fail (on Linux and macOS traceroute "+
|
||||
"sends UDP probes instead, which ARE tunnelled — the hops it prints are the tunnel's "+
|
||||
"path, not your own), and so do raw IPsec (ESP/AH) and PPTP/GRE — a PPTP session will "+
|
||||
"even look connected, because its control channel is TCP and only the payload is "+
|
||||
"dropped. VPNs that run over UDP are NOT affected: WireGuard, OpenVPN-UDP and IPsec "+
|
||||
"through NAT (IKE on UDP 500, NAT-T on UDP 4500) keep working normally. Multicast "+
|
||||
"IPTV does not cross this router under any setting; that one is not this policy.")
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -17,7 +17,7 @@ import (
|
||||
func TestCollectWarningsAttributesEntities(t *testing.T) {
|
||||
got := collectWarnings(blockGlobals(),
|
||||
[]string{
|
||||
`ruleset "ads": remote list "https://x/y.srs" is UNREACHABLE right now, so it is NOT applied`,
|
||||
ruleSetNotApplied,
|
||||
`device "kids-tablet": no current IP (ip unset and MAC "aa:bb" not leased), skipped`,
|
||||
`chain "hop": has no hops, target skipped`,
|
||||
`dns_filter enabled but no resolvers configured; filter inert`,
|
||||
@@ -38,14 +38,23 @@ func TestCollectWarningsAttributesEntities(t *testing.T) {
|
||||
if w.Severity != SeverityCritical {
|
||||
t.Errorf("an unapplied blocklist is a protection gap; severity = %q, want critical", w.Severity)
|
||||
}
|
||||
if strings.Contains(w.Message, `ruleset "ads":`) {
|
||||
t.Errorf("the entity prefix must move into Section/Name, not stay in Message: %q", w.Message)
|
||||
// The RULESET-NOT-APPLIED tag stays in the message on purpose: generate
|
||||
// documents it as the operator's grep handle into logread.
|
||||
if !strings.HasPrefix(w.Message, "RULESET-NOT-APPLIED:") {
|
||||
t.Errorf("a tagged warning must keep its greppable tag in the message: %q", w.Message)
|
||||
}
|
||||
}
|
||||
if w, ok := byName["device/kids-tablet"]; !ok {
|
||||
t.Errorf("device warning not attributed; got %+v", got)
|
||||
} else if w.Severity != SeverityWarning {
|
||||
t.Errorf("a skipped device is not a protection gap; severity = %q, want warning", w.Severity)
|
||||
} else {
|
||||
if w.Severity != SeverityWarning {
|
||||
t.Errorf("a skipped device is not a protection gap; severity = %q, want warning", w.Severity)
|
||||
}
|
||||
// The plain `kind "name": message` prefix, by contrast, MOVES into
|
||||
// Section/Name — it carries no information the fields do not.
|
||||
if strings.Contains(w.Message, `device "kids-tablet":`) {
|
||||
t.Errorf("the entity prefix must move into Section/Name, not stay in Message: %q", w.Message)
|
||||
}
|
||||
}
|
||||
if w, ok := byName["chain/hop"]; !ok || w.Severity != SeverityWarning {
|
||||
t.Errorf("chain warning: got %+v", w)
|
||||
@@ -115,7 +124,7 @@ func TestCollectWarningsCapKeepsCriticals(t *testing.T) {
|
||||
for i := 0; i < 200; i++ {
|
||||
noisy = append(noisy, `chain "c": has no hops, target skipped`)
|
||||
}
|
||||
noisy = append(noisy, `ruleset "ads": remote list is UNREACHABLE right now, so it is NOT applied`)
|
||||
noisy = append(noisy, ruleSetNotApplied)
|
||||
|
||||
got := collectWarnings(blockGlobals(), noisy, nil, nil)
|
||||
if len(got) != maxStatusWarnings {
|
||||
@@ -232,6 +241,170 @@ func TestWarningsAgainstRealGenerateOutput(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
// ruleSetNotApplied is the message generate/ruleset.go:997 actually produces when
|
||||
// a remote blocklist cannot be fetched — the single most common real fault on a
|
||||
// router in Russia, and the one the panel has no other way to show (an omitted
|
||||
// rule-set produces no row in GET /api/ruleset/status, so the UI is
|
||||
// indistinguishable from "not configured").
|
||||
//
|
||||
// Quoted verbatim, tag and all, because the classifier is a heuristic over free
|
||||
// text and a test written against invented text proves nothing about it. The
|
||||
// version this replaced asserted on `... is UNREACHABLE ... is NOT applied`, a
|
||||
// sentence no producer has ever emitted; it passed for as long as the real
|
||||
// sentence was being graded a plain `warning` under section "generate" with no
|
||||
// name at all.
|
||||
const ruleSetNotApplied = `RULESET-NOT-APPLIED: ruleset "ads" is configured but NOT ACTIVE: ` +
|
||||
`its source "https://big.oisd.nl/domainswild" is unreachable right now, so rule-set "ads" was ` +
|
||||
`omitted and matches NOTHING until it loads (a blocklist blocks nothing; a routing rule is skipped). ` +
|
||||
`Handing an unusable list to the engine would abort engine start and take the LAN down instead. ` +
|
||||
`Retried automatically on the next reconcile (~1 min) — no action needed unless this persists.`
|
||||
|
||||
// TestClassifyRealGenerateTexts grades the sentences generate REALLY emits, by
|
||||
// consequence.
|
||||
//
|
||||
// The defect this pins: severity was inverted. A blocklist that could not be
|
||||
// downloaded — protection the operator configured, not in force — came out
|
||||
// `warning`, because its tag broke the entity regexp and its wording matched no
|
||||
// marker. A typo in the same list's URL came out `critical`, because that one
|
||||
// parsed cleanly into section "ruleset". The panel's alarm banner lights on
|
||||
// critical and on nothing else, so the router shouted about the typo and stayed
|
||||
// quiet about the outage.
|
||||
func TestClassifyRealGenerateTexts(t *testing.T) {
|
||||
cases := []struct {
|
||||
name string
|
||||
text string
|
||||
want string
|
||||
// section/entity attribution, when the panel must be able to deep-link.
|
||||
section, entity string
|
||||
}{{
|
||||
name: "remote blocklist could not be fetched",
|
||||
text: ruleSetNotApplied,
|
||||
want: SeverityCritical,
|
||||
section: "ruleset", entity: "ads",
|
||||
}, {
|
||||
name: "not one DNS filter list could be built",
|
||||
text: "DNS-FILTER-NOT-APPLIED: dns_filter is ON and lists are enabled, but NOT ONE of them " +
|
||||
"could be built right now — nothing is being filtered or allowed. See the per-list " +
|
||||
"warnings above for why. Rebuilt on the next reconcile (~1 min).",
|
||||
want: SeverityCritical,
|
||||
section: "generate",
|
||||
}, {
|
||||
// generate/ruleset.go:1214 — the counterpart the outage used to be graded
|
||||
// BELOW. It stays critical: the consequence is identical (the list is not
|
||||
// loaded and matches nothing), which is the whole point — the inversion is
|
||||
// fixed by lifting the outage, not by lowering the typo.
|
||||
name: "url blocklist with an empty url",
|
||||
text: `ruleset "ads": url source with empty url, skipped`,
|
||||
want: SeverityCritical,
|
||||
section: "ruleset", entity: "ads",
|
||||
}, {
|
||||
// generate/ruleset.go:819 — the list IS still in force, from the copy
|
||||
// compiled earlier. Grading this critical would light the alarm banner on
|
||||
// most reconciles of a healthy router behind a flaky link, and a banner that
|
||||
// is always on is a banner nobody reads when it finally matters.
|
||||
name: "blocklist refresh failed but the compiled copy still blocks",
|
||||
text: `ruleset "ads": could not refresh the list from "https://big.oisd.nl/domainswild" ` +
|
||||
`(dial tcp: i/o timeout); continuing with the copy compiled earlier. Retried on the next reconcile.`,
|
||||
want: SeverityWarning,
|
||||
section: "ruleset", entity: "ads",
|
||||
}, {
|
||||
// generate/route.go:407 — a rule the operator scoped to one device now
|
||||
// applies to the entire network. Whatever it does, it now does to everyone.
|
||||
name: "a rule's source scope collapsed to the whole LAN",
|
||||
text: `rule "kids": none of its source entries can be matched by the engine (interface/zone/MAC ` +
|
||||
`selectors and invalid addresses are dropped), so the rule now applies to EVERY client on ` +
|
||||
`the router instead of that source — check it is still what you want, and use IP ` +
|
||||
`addresses/subnets as the source`,
|
||||
want: SeverityCritical,
|
||||
section: "rule", entity: "kids",
|
||||
}, {
|
||||
// generate/route.go:115 — the rule exists in the UI and routes nothing.
|
||||
name: "a rule whose target is never reached",
|
||||
text: `rule "work": no matcher the engine can evaluate, skipped — its target "group:auto" ` +
|
||||
`never applies and the traffic follows the rules below it and the default`,
|
||||
want: SeverityCritical,
|
||||
section: "rule", entity: "work",
|
||||
}, {
|
||||
// generate/ruleset.go:445 — a redundant field on a list that loads fine.
|
||||
name: "a redundant field on a working blocklist",
|
||||
text: `ruleset "ads": format "binary" is IGNORED for source=geosite — the category decides. Remove it to avoid confusion.`,
|
||||
want: SeverityWarning,
|
||||
section: "ruleset", entity: "ads",
|
||||
}, {
|
||||
// generate/chain.go:108 — a configured path that does not resolve. Its
|
||||
// traffic is blocked fail-closed, so no protection claim is broken.
|
||||
name: "a chain with no hops",
|
||||
text: `chain "hop": has no hops, target skipped`,
|
||||
want: SeverityWarning,
|
||||
section: "chain", entity: "hop",
|
||||
}, {
|
||||
// generate/cache.go:92 — operational, the lists still compile.
|
||||
name: "the compiled lists moved to tmpfs",
|
||||
text: "cache: only 3 MiB free on /overlay (need 8 MiB), using tmpfs /tmp/shater instead — " +
|
||||
"remote rule-sets will be re-downloaded after every reboot, so free some space",
|
||||
want: SeverityInfo,
|
||||
section: "generate",
|
||||
}}
|
||||
|
||||
for _, tc := range cases {
|
||||
t.Run(tc.name, func(t *testing.T) {
|
||||
var got *Warning
|
||||
for _, w := range collectWarnings(blockGlobals(), []string{tc.text}, nil, nil) {
|
||||
if w.Section == "untunnelable" {
|
||||
continue // the always-present policy notice
|
||||
}
|
||||
w := w
|
||||
got = &w
|
||||
}
|
||||
if got == nil {
|
||||
t.Fatalf("the warning was dropped entirely")
|
||||
}
|
||||
if got.Severity != tc.want {
|
||||
t.Errorf("severity = %q, want %q\n text: %s", got.Severity, tc.want, tc.text)
|
||||
}
|
||||
if got.Section != tc.section || got.Name != tc.entity {
|
||||
t.Errorf("attribution = %q/%q, want %q/%q — the panel deep-links on these",
|
||||
got.Section, got.Name, tc.section, tc.entity)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
// TestUnreachableRuleSetFromRealGenerate drives the ACTUAL producer, so this stays
|
||||
// correct if generate rewords or re-tags the message. A url rule-set pointing at a
|
||||
// closed local port fails the reachability probe exactly the way an unreachable
|
||||
// public blocklist does, with no network needed.
|
||||
func TestUnreachableRuleSetFromRealGenerate(t *testing.T) {
|
||||
m := holdModel("closed")
|
||||
m.Rulesets = []model.Ruleset{
|
||||
{Name: "ads", Type: "domain", Source: "url", URL: "http://127.0.0.1:1/blocklist.srs", Format: "binary"},
|
||||
}
|
||||
m.Rules = []model.Rule{
|
||||
{Name: "blockads", Enabled: true, Order: 10, DstRuleset: []string{"ads"}, Target: "block"},
|
||||
}
|
||||
|
||||
_, genWarnings, err := generate.GenerateWithWarnings(m)
|
||||
if err != nil {
|
||||
t.Fatalf("GenerateWithWarnings: %v", err)
|
||||
}
|
||||
t.Logf("real generate warnings: %q", genWarnings)
|
||||
|
||||
var found *Warning
|
||||
for _, w := range collectWarnings(blockGlobals(), genWarnings, nil, nil) {
|
||||
if w.Section == "ruleset" && w.Name == "ads" {
|
||||
w := w
|
||||
found = &w
|
||||
}
|
||||
}
|
||||
if found == nil {
|
||||
t.Fatalf("an unfetchable blocklist produced no warning attributed to it: %q", genWarnings)
|
||||
}
|
||||
if found.Severity != SeverityCritical {
|
||||
t.Errorf("a blocklist that did not load is a protection gap the operator cannot otherwise "+
|
||||
"see; severity = %q, want critical (message: %q)", found.Severity, found.Message)
|
||||
}
|
||||
}
|
||||
|
||||
// blockGlobals is the default policy fixture: kill-switch closed, untunnelable
|
||||
// traffic blocked — i.e. what a stock install runs.
|
||||
func blockGlobals() model.Globals {
|
||||
|
||||
@@ -0,0 +1,281 @@
|
||||
// Keeping the fail-closed plane alive across the moments the daemon is not.
|
||||
//
|
||||
// The daemon owns the `inet shater` table, which means the table exists exactly
|
||||
// while the daemon does. Three of those moments are not covered by anything else,
|
||||
// and all three are the same defect wearing different clothes: the protection is
|
||||
// an in-process thing, and the process is not always there.
|
||||
//
|
||||
// BOOT /etc/init.d/shater is START=99. fw4 loaded `lan -> wan ACCEPT` at 19
|
||||
// and netifd brought the LAN up at 20; the clients that reconnect in
|
||||
// between are unprotected until the daemon has been decompressed off
|
||||
// flash, waited out any predecessor, migrated UCI and applied.
|
||||
// RESTART SIGTERM ran an unconditional Teardown — kill_switch was not so much
|
||||
// as consulted — and the successor cannot apply until the init's
|
||||
// shater_wait_stopped loop, `shaterd migrate` and engine start have all
|
||||
// finished. `reload_service` is stop+start, and so is every package
|
||||
// upgrade, so this ran on a routine `Save & Apply`.
|
||||
// NO CONFIG model.ReadUCI failing left the arming call unreached: it sat in the
|
||||
// else-branch of the successful read. Nothing recovered from it either
|
||||
// — Reconcile returns before any plane work, and the cron watchdog sees
|
||||
// a live pidof and its own `uci -q get` fails the same way.
|
||||
//
|
||||
// The answer to all three is one artifact: netplane's BOOT ARMOR, a persisted copy
|
||||
// of the fail-closed holding plane (netplane/armor.go). This file is the daemon's
|
||||
// half — it keeps that copy honest, and it reinstates it in the two cases the
|
||||
// daemon is the only one who can.
|
||||
//
|
||||
// Everything here is deliberately conservative in ONE direction: it never installs
|
||||
// a plane the operator did not ask for. `globals.enabled=0` or `kill_switch=open`
|
||||
// removes the armor and installs nothing, and a deliberate `/etc/init.d/shater
|
||||
// stop` is a handoff-free exit that leaves nothing behind. Fail-closed is a
|
||||
// policy, and a kill switch that outlives its own off switch is not one.
|
||||
package main
|
||||
|
||||
import (
|
||||
"os"
|
||||
|
||||
"github.com/sagernet/sing-box/log"
|
||||
"github.com/sagernet/sing-box/shater/model"
|
||||
"github.com/sagernet/sing-box/shater/netplane"
|
||||
)
|
||||
|
||||
// restartHandoffPath is raised by /etc/init.d/shater around a restart/reload and
|
||||
// cleared by its start (and by a real stop). Its presence at SIGTERM means "this
|
||||
// daemon is being REPLACED", as opposed to "this daemon is being switched off".
|
||||
//
|
||||
// tmpfs on purpose: a marker that survived a power cut would make the first boot
|
||||
// after it look like a restart.
|
||||
//
|
||||
// A var, not a const, only so tests can point it at a temp dir.
|
||||
var restartHandoffPath = "/var/run/shater.restarting"
|
||||
|
||||
// restartHandoffPending reports whether the init script announced a restart.
|
||||
//
|
||||
// Absent is read as "a real stop", which is the SAFE direction to be wrong in: it
|
||||
// degrades to exactly the behaviour that shipped before this file existed (full
|
||||
// teardown), whereas the other default would leave a stopped router blocked.
|
||||
func restartHandoffPending() bool {
|
||||
_, err := os.Stat(restartHandoffPath)
|
||||
return err == nil
|
||||
}
|
||||
|
||||
// armorPlan is what to do about the fail-closed plane at a decision point.
|
||||
type armorPlan int
|
||||
|
||||
const (
|
||||
// armorNothing: install nothing. Either the operator does not want a plane
|
||||
// (disabled / kill_switch=open), or there is nothing to install from.
|
||||
armorNothing armorPlan = iota
|
||||
// armorRender: build the holding plane from the model we just read. Preferred
|
||||
// whenever a model is readable — it reflects the CURRENT interface set, where a
|
||||
// snapshot may predate an interface rename.
|
||||
armorRender
|
||||
// armorSnapshot: reinstate the persisted boot armor. The only option when the
|
||||
// config cannot be read, which is precisely when it is needed.
|
||||
armorSnapshot
|
||||
)
|
||||
|
||||
// armorWanted reports whether m asks for a fail-closed plane at all: the stack is
|
||||
// enabled AND the kill switch is closed. Both halves are the operator's explicit
|
||||
// choice and neither may be second-guessed — `kill_switch=open` is a documented
|
||||
// decision to let traffic through when the engine is down, not an oversight.
|
||||
func armorWanted(m *model.Model) bool {
|
||||
return m != nil && m.Globals.Enabled && netplane.KillSwitchClosed(m.Globals)
|
||||
}
|
||||
|
||||
// planStartupArmor decides what to install when the daemon starts and could NOT
|
||||
// read its config.
|
||||
//
|
||||
// It is only ever consulted on the read-failure path: with a readable model the
|
||||
// applier's own ArmHold does this job (and does it better — it holds the apply
|
||||
// lock while it works). With no model there is nothing to render from, so the
|
||||
// persisted snapshot is the entire answer; with no snapshot either, nothing is
|
||||
// installed, because "this router has never applied an enabled, fail-closed
|
||||
// config" is then the most likely truth and blacking out a LAN on a guess is not
|
||||
// a recovery.
|
||||
func planStartupArmor(readErr error, snapshot bool) armorPlan {
|
||||
if readErr == nil {
|
||||
return armorNothing
|
||||
}
|
||||
if snapshot {
|
||||
return armorSnapshot
|
||||
}
|
||||
return armorNothing
|
||||
}
|
||||
|
||||
// planExitArmor decides what the daemon leaves behind when it is asked to exit.
|
||||
//
|
||||
// The handoff flag is the whole distinction the old code was missing. A restart,
|
||||
// a reload and a package upgrade all reach this point, and in all three the
|
||||
// operator has not asked for protection to end — only for this process to be
|
||||
// replaced. A `stop` has asked for exactly that, and must be obeyed: it is the
|
||||
// operator's escape hatch, and a kill switch that cannot be switched off is a
|
||||
// brick.
|
||||
func planExitArmor(handoff bool, m *model.Model, readErr error, snapshot bool) armorPlan {
|
||||
if !handoff {
|
||||
return armorNothing
|
||||
}
|
||||
if readErr != nil {
|
||||
// Being replaced with an unreadable config: the snapshot is the last thing
|
||||
// this router is known to have wanted, and it is still the honest answer.
|
||||
if snapshot {
|
||||
return armorSnapshot
|
||||
}
|
||||
return armorNothing
|
||||
}
|
||||
if !armorWanted(m) {
|
||||
return armorNothing
|
||||
}
|
||||
return armorRender
|
||||
}
|
||||
|
||||
// refreshBootArmor keeps the persisted holding plane in step with the desired
|
||||
// state. Called after every successful UCI read, so the snapshot on flash always
|
||||
// describes the config the router is actually running.
|
||||
//
|
||||
// Writing is content-gated inside netplane.SaveBootArmor (this runs once a minute
|
||||
// under cron; rewriting an identical file that often is how flash dies), and the
|
||||
// REMOVE half matters just as much as the write: turning the stack off, or opening
|
||||
// the kill switch, has to disarm the next boot too, or the operator's change would
|
||||
// silently come back after a power cut.
|
||||
func refreshBootArmor(m *model.Model, logger log.ContextLogger) {
|
||||
if !armorWanted(m) {
|
||||
if netplane.BootArmorPresent() {
|
||||
if err := netplane.RemoveBootArmor(); err != nil {
|
||||
logger.Warn("boot armor: could not remove ", netplane.BootArmorPath, ": ", err)
|
||||
} else {
|
||||
logger.Info("boot armor removed (the stack is disabled or the kill switch is open): ",
|
||||
"the LAN is no longer blocked at boot before the daemon starts")
|
||||
}
|
||||
}
|
||||
return
|
||||
}
|
||||
ruleset, err := netplane.RenderHoldNft(m)
|
||||
if err != nil {
|
||||
// A transient render failure must not disarm: a stale fail-closed plane is
|
||||
// recoverable (the daemon replaces it seconds into the next boot), an absent
|
||||
// one is a leak.
|
||||
logger.Warn("boot armor: could not render the fail-closed plane: ", err)
|
||||
return
|
||||
}
|
||||
if ruleset == "" {
|
||||
// No divert devices at all — this config intercepts nothing, so there is
|
||||
// nothing for a boot-time plane to protect. Blocking the LAN at boot on
|
||||
// behalf of a config that does not touch it would be a pure outage.
|
||||
if netplane.BootArmorPresent() {
|
||||
if rerr := netplane.RemoveBootArmor(); rerr != nil {
|
||||
logger.Warn("boot armor: could not remove ", netplane.BootArmorPath, ": ", rerr)
|
||||
}
|
||||
}
|
||||
return
|
||||
}
|
||||
changed, serr := netplane.SaveBootArmor(ruleset)
|
||||
switch {
|
||||
case serr != nil:
|
||||
logger.Warn("boot armor: could not write ", netplane.BootArmorPath, ": ", serr,
|
||||
" — the LAN will be unprotected between boot and this daemon's first apply")
|
||||
case changed:
|
||||
logger.Info("boot armor updated (", netplane.BootArmorPath,
|
||||
"): the LAN is fail-closed from early boot until the engine is up")
|
||||
}
|
||||
}
|
||||
|
||||
// armFromSnapshot reinstates the persisted holding plane and reports whether a
|
||||
// plane is now standing. why is a short phrase for the log ("config is
|
||||
// unreadable", "restart handoff").
|
||||
func armFromSnapshot(why string, logger log.ContextLogger) bool {
|
||||
loaded, err := netplane.LoadBootArmor()
|
||||
switch {
|
||||
case err != nil:
|
||||
logger.Error("FAIL-CLOSED PLANE NOT INSTALLED (", why, "): the saved plane ",
|
||||
netplane.BootArmorPath, " could not be loaded: ", err,
|
||||
" — LAN traffic may be reaching the WAN unprotected")
|
||||
return false
|
||||
case loaded:
|
||||
logger.Error("fail-closed plane reinstated from ", netplane.BootArmorPath,
|
||||
" (", why, "): LAN->WAN forwarding is BLOCKED. ",
|
||||
"SSH, LuCI and the admin panel remain reachable.")
|
||||
return true
|
||||
default:
|
||||
logger.Warn("no saved fail-closed plane at ", netplane.BootArmorPath, " (", why,
|
||||
"): nothing was installed")
|
||||
return false
|
||||
}
|
||||
}
|
||||
|
||||
// armFromModel renders the holding plane for m, installs it, and reports whether
|
||||
// a plane is now standing.
|
||||
//
|
||||
// The return value is load-bearing on the exit path: Applier.TeardownExiting keeps
|
||||
// the nft table only when a plane really was installed, so a render failure or a
|
||||
// fail-open config falls back to the old remove-everything behaviour instead of
|
||||
// leaving whatever the engine happened to have in the kernel.
|
||||
func armFromModel(m *model.Model, why string, logger log.ContextLogger) bool {
|
||||
ruleset, err := netplane.RenderHoldNft(m)
|
||||
if err != nil {
|
||||
logger.Error("FAIL-CLOSED PLANE NOT INSTALLED (", why, "): render failed: ", err)
|
||||
return false
|
||||
}
|
||||
if ruleset == "" {
|
||||
// No divert devices: there is nothing this plane would protect.
|
||||
return false
|
||||
}
|
||||
// One `nft -f` that opens with `delete table` and closes with the new table:
|
||||
// the swap is a single netlink transaction, so this REPLACES whatever plane is
|
||||
// loaded without the table ever being absent. That property is why the exit
|
||||
// path can arm before it tears down.
|
||||
if err := netplane.ApplyNft(ruleset); err != nil {
|
||||
logger.Error("FAIL-CLOSED PLANE NOT INSTALLED (", why, "): ", err,
|
||||
" — LAN traffic may be reaching the WAN unprotected")
|
||||
return false
|
||||
}
|
||||
logger.Info("fail-closed plane left in place (", why,
|
||||
"): LAN->WAN forwarding stays BLOCKED until the next daemon applies. ",
|
||||
"SSH, LuCI and the admin panel remain reachable.")
|
||||
return true
|
||||
}
|
||||
|
||||
// armOnUnreadableConfig is the window-3 answer: the daemon is up but cannot read
|
||||
// its own desired state, so it falls back to the last state it persisted.
|
||||
//
|
||||
// A table that is ALREADY loaded is left alone. This runs on every reconcile —
|
||||
// cron fires one a minute — and a `nft -f` is a delete-and-recreate of the whole
|
||||
// table plus a DNS conntrack flush, so re-installing an identical plane sixty
|
||||
// times an hour would be pure churn, and each replacement is itself a brief hole.
|
||||
// The question this path answers is "is there anything at all standing", and once
|
||||
// the answer is yes it stays yes until an apply succeeds and replaces it properly.
|
||||
func armOnUnreadableConfig(logger log.ContextLogger) {
|
||||
if netplane.TableExists() {
|
||||
return
|
||||
}
|
||||
if planStartupArmor(errUnreadableConfig, netplane.BootArmorPresent()) != armorSnapshot {
|
||||
logger.Warn("the config could not be read and there is no saved fail-closed plane at ",
|
||||
netplane.BootArmorPath, " — nothing is protecting the LAN; fix /etc/config/shater ",
|
||||
"(a full /overlay is the usual cause) and reconcile")
|
||||
return
|
||||
}
|
||||
armFromSnapshot("the config could not be read", logger)
|
||||
}
|
||||
|
||||
// errUnreadableConfig is a stand-in for "the read failed" in the call above,
|
||||
// where the concrete error has already been logged by the caller.
|
||||
var errUnreadableConfig = os.ErrInvalid
|
||||
|
||||
// armOnExit is the window-2 answer: what this daemon leaves in the kernel when it
|
||||
// is asked to go away, and whether anything is now standing there. See
|
||||
// planExitArmor for the policy.
|
||||
//
|
||||
// It is called BY Applier.TeardownExiting, before the teardown and under the apply
|
||||
// lock, so that the plane is swapped rather than removed-then-rebuilt. It must
|
||||
// therefore never call back into the Applier — everything here goes straight to
|
||||
// model.ReadUCI and netplane.
|
||||
func armOnExit(handoff bool, logger log.ContextLogger) bool {
|
||||
m, err := model.ReadUCI()
|
||||
switch planExitArmor(handoff, m, err, netplane.BootArmorPresent()) {
|
||||
case armorRender:
|
||||
return armFromModel(m, "restart handoff", logger)
|
||||
case armorSnapshot:
|
||||
return armFromSnapshot("restart handoff, config unreadable", logger)
|
||||
}
|
||||
return false
|
||||
}
|
||||
@@ -0,0 +1,182 @@
|
||||
package main
|
||||
|
||||
// The daemon's half of the fail-closed armor: the two decisions that decide
|
||||
// whether the LAN is protected in the moments this process is not running.
|
||||
//
|
||||
// Both are pure functions on purpose. The behaviour they encode can otherwise
|
||||
// only be observed on a router, with root, by killing a daemon at the right
|
||||
// moment and reading `nft list ruleset` — i.e. never, in a gate.
|
||||
|
||||
import (
|
||||
"errors"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
"testing"
|
||||
|
||||
"github.com/sagernet/sing-box/log"
|
||||
"github.com/sagernet/sing-box/shater/model"
|
||||
"github.com/sagernet/sing-box/shater/netplane"
|
||||
)
|
||||
|
||||
func enabledClosed() *model.Model {
|
||||
return &model.Model{Globals: model.Globals{
|
||||
Enabled: true, KillSwitch: "closed", FwmarkBase: 0x2000, TableBase: 0x2000,
|
||||
}, Inbounds: []model.Inbound{{
|
||||
Name: "lan", Enabled: true, Type: "tproxy", Network: "lan",
|
||||
TproxyPort: 12345, TCP: true, UDP: true,
|
||||
}}}
|
||||
}
|
||||
|
||||
// TestPlanExitArmor is W2. SIGTERM used to run an unconditional Teardown — the
|
||||
// only place in this codebase that removes the fail-closed plane without so much
|
||||
// as reading kill_switch — and every `restart`, every `reload_service` (which is
|
||||
// what a LuCI Save & Apply runs) and every package upgrade went through it. The
|
||||
// gap that follows is guaranteed non-empty by the init script itself.
|
||||
//
|
||||
// So the exit has to know WHY it is exiting. What it must never do is confuse the
|
||||
// two directions: a restart that leaves nothing behind is a plaintext window, and
|
||||
// a stop that leaves a block behind is a router the operator cannot un-brick.
|
||||
//
|
||||
// RED BEFORE: there was no such decision — the daemon always tore everything down.
|
||||
func TestPlanExitArmor(t *testing.T) {
|
||||
open := enabledClosed()
|
||||
open.Globals.KillSwitch = "open"
|
||||
disabled := enabledClosed()
|
||||
disabled.Globals.Enabled = false
|
||||
|
||||
cases := []struct {
|
||||
name string
|
||||
handoff bool
|
||||
m *model.Model
|
||||
readErr error
|
||||
snapshot bool
|
||||
want armorPlan
|
||||
}{
|
||||
{"restart, enabled + fail-closed => leave the plane standing",
|
||||
true, enabledClosed(), nil, true, armorRender},
|
||||
{"restart, no snapshot on disk => still render from the live model",
|
||||
true, enabledClosed(), nil, false, armorRender},
|
||||
{"restart, kill_switch=open => the operator chose fail-open; install nothing",
|
||||
true, open, nil, true, armorNothing},
|
||||
{"restart, stack disabled => nothing to protect",
|
||||
true, disabled, nil, true, armorNothing},
|
||||
{"restart, config unreadable => the persisted plane is the last known truth",
|
||||
true, nil, errors.New("uci: no such file"), true, armorSnapshot},
|
||||
{"restart, config unreadable and nothing persisted => nothing to install",
|
||||
true, nil, errors.New("uci: no such file"), false, armorNothing},
|
||||
|
||||
// The escape hatch. A deliberate `/etc/init.d/shater stop` must mean what it
|
||||
// says in every one of these, or the kill switch has no off switch.
|
||||
{"stop, enabled + fail-closed => the plane goes away",
|
||||
false, enabledClosed(), nil, true, armorNothing},
|
||||
{"stop, config unreadable => still goes away",
|
||||
false, nil, errors.New("uci: no such file"), true, armorNothing},
|
||||
}
|
||||
for _, c := range cases {
|
||||
t.Run(c.name, func(t *testing.T) {
|
||||
if got := planExitArmor(c.handoff, c.m, c.readErr, c.snapshot); got != c.want {
|
||||
t.Errorf("planExitArmor(handoff=%v, readErr=%v, snapshot=%v) = %v, want %v",
|
||||
c.handoff, c.readErr, c.snapshot, got, c.want)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
// TestPlanStartupArmor is W3. An unreadable /etc/config/shater — a full /overlay
|
||||
// caught mid `uci commit` is the cause the init script itself documents — left the
|
||||
// arming call unreached, because it sat in the else-branch of the successful read.
|
||||
// Nothing recovered from that: Reconcile returns before any plane work and the
|
||||
// cron watchdog's own `uci -q get` fails identically, so the box sat there with a
|
||||
// live daemon, an answering panel and no table at all.
|
||||
//
|
||||
// RED BEFORE: no decision existed; the read-failure branch only logged.
|
||||
func TestPlanStartupArmor(t *testing.T) {
|
||||
readErr := errors.New("uci: cannot read /etc/config/shater")
|
||||
if got := planStartupArmor(readErr, true); got != armorSnapshot {
|
||||
t.Errorf("unreadable config with a persisted plane = %v, want armorSnapshot", got)
|
||||
}
|
||||
// Nothing persisted means this router has never applied an enabled,
|
||||
// fail-closed config. Blacking out a LAN on that guess is not a recovery.
|
||||
if got := planStartupArmor(readErr, false); got != armorNothing {
|
||||
t.Errorf("unreadable config with no persisted plane = %v, want armorNothing", got)
|
||||
}
|
||||
// A readable config is the applier's business (ArmHold), not this path's.
|
||||
if got := planStartupArmor(nil, true); got != armorNothing {
|
||||
t.Errorf("readable config = %v, want armorNothing", got)
|
||||
}
|
||||
}
|
||||
|
||||
// TestRefreshBootArmorTracksDesiredState is W1's durable half: the file
|
||||
// /etc/init.d/shater-armor loads at START=21 only exists while the operator wants
|
||||
// it to. Writing it is half the contract; REMOVING it when the stack is switched
|
||||
// off or the kill switch is opened is the other half, and the more dangerous one
|
||||
// to get wrong — a stale armor would reinstate, at the next power cut, a block the
|
||||
// operator had already turned off.
|
||||
//
|
||||
// RED BEFORE: neither the file nor this function existed.
|
||||
func TestRefreshBootArmorTracksDesiredState(t *testing.T) {
|
||||
orig := netplane.BootArmorPath
|
||||
netplane.BootArmorPath = filepath.Join(t.TempDir(), "shater", "boot.nft")
|
||||
defer func() { netplane.BootArmorPath = orig }()
|
||||
logger := log.StdLogger()
|
||||
|
||||
refreshBootArmor(enabledClosed(), logger)
|
||||
if !netplane.BootArmorPresent() {
|
||||
t.Fatalf("an enabled, fail-closed config must persist a boot armor")
|
||||
}
|
||||
b, err := os.ReadFile(netplane.BootArmorPath)
|
||||
if err != nil {
|
||||
t.Fatalf("read: %v", err)
|
||||
}
|
||||
// It must be the HOLDING plane — a forward chain that drops — and not the full
|
||||
// tproxy ruleset, which would reference an engine that is not running at boot.
|
||||
for _, must := range []string{"table inet shater", "hook forward", "drop"} {
|
||||
if !strings.Contains(string(b), must) {
|
||||
t.Errorf("the persisted armor must contain %q; got:\n%s", must, string(b))
|
||||
}
|
||||
}
|
||||
if strings.Contains(string(b), "tproxy") {
|
||||
t.Errorf("the persisted armor must NOT divert to an engine that is not running:\n%s", string(b))
|
||||
}
|
||||
|
||||
openKS := enabledClosed()
|
||||
openKS.Globals.KillSwitch = "open"
|
||||
refreshBootArmor(openKS, logger)
|
||||
if netplane.BootArmorPresent() {
|
||||
t.Errorf("kill_switch=open is a documented choice to let traffic through; " +
|
||||
"the boot armor must be removed, not left to block the next boot")
|
||||
}
|
||||
|
||||
refreshBootArmor(enabledClosed(), logger)
|
||||
if !netplane.BootArmorPresent() {
|
||||
t.Fatalf("re-arming after a disarm must work")
|
||||
}
|
||||
off := enabledClosed()
|
||||
off.Globals.Enabled = false
|
||||
refreshBootArmor(off, logger)
|
||||
if netplane.BootArmorPresent() {
|
||||
t.Errorf("globals.enabled=0 must remove the boot armor")
|
||||
}
|
||||
}
|
||||
|
||||
// TestRestartHandoffPending pins the marker's read side, including the default
|
||||
// that matters: an ABSENT marker means "a real stop". Defaulting the other way
|
||||
// would leave a stopped router blocked whenever the init script failed to write
|
||||
// the flag.
|
||||
func TestRestartHandoffPending(t *testing.T) {
|
||||
orig := restartHandoffPath
|
||||
defer func() { restartHandoffPath = orig }()
|
||||
dir := t.TempDir()
|
||||
restartHandoffPath = filepath.Join(dir, "shater.restarting")
|
||||
|
||||
if restartHandoffPending() {
|
||||
t.Errorf("an absent marker must read as a real stop")
|
||||
}
|
||||
if err := os.WriteFile(restartHandoffPath, nil, 0o644); err != nil {
|
||||
t.Fatalf("write marker: %v", err)
|
||||
}
|
||||
if !restartHandoffPending() {
|
||||
t.Errorf("a present marker must read as a restart handoff")
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,169 @@
|
||||
// Executing the SHIPPED init script's action classification.
|
||||
//
|
||||
// WHY THIS TEST IS SHAPED LIKE THIS
|
||||
//
|
||||
// The boot-armor defect that shipped in v0.2.17 was not in any Go file. The Go
|
||||
// half was correct and fully covered: refreshBootArmor tracked desired state,
|
||||
// planExitArmor made the right call, LoadBootArmor validated before loading.
|
||||
// Every one of those tests was green while the feature did not work at all on
|
||||
// hardware, because the thing that broke it was one shell `case` in
|
||||
// /etc/init.d/shater whose default arm swept up procd's `shutdown` action — so
|
||||
// the arm token was deleted on the way down, every reboot, and the boot it
|
||||
// existed to protect always found no file.
|
||||
//
|
||||
// A unit test that cannot see the shell file cannot catch that, and a comment in
|
||||
// the shell file claiming `shutdown` is handled is precisely what shipped. So
|
||||
// this runs the real thing: it sources the actual packaged
|
||||
// openwrt/shater-core/files/etc/init.d/shater in /bin/sh and calls its two
|
||||
// classification predicates with every action procd actually uses.
|
||||
//
|
||||
// Sourcing the whole file is safe and deliberate — at top level it contains only
|
||||
// variable assignments and function definitions, nothing that touches the system —
|
||||
// and sourcing the WHOLE file is the point: a test that copy-pasted the `case`
|
||||
// would pass while the shipped script said something else.
|
||||
//
|
||||
// The action names are not invented. They were measured on the target
|
||||
// (ImmortalWrt 25.12.1 r37978) with a throwaway probe init script:
|
||||
//
|
||||
// /etc/init.d/X restart -> stop_service action=[restart]
|
||||
// /etc/init.d/X stop -> stop_service action=[stop]
|
||||
// /etc/init.d/X reload -> reload_service action=[reload]
|
||||
// `reboot` -> stop_service action=[shutdown]
|
||||
// the boot after it -> start_service action=[boot]
|
||||
package main
|
||||
|
||||
import (
|
||||
"os/exec"
|
||||
"path/filepath"
|
||||
"runtime"
|
||||
"strings"
|
||||
"testing"
|
||||
)
|
||||
|
||||
// initScriptPath is the packaged init script, relative to this package dir.
|
||||
const initScriptPath = "../../../openwrt/shater-core/files/etc/init.d/shater"
|
||||
|
||||
// askInitScript sources the init script in /bin/sh and reports whether fn
|
||||
// returns true for the given arguments.
|
||||
func askInitScript(t *testing.T, fn string, args ...string) bool {
|
||||
t.Helper()
|
||||
abs, err := filepath.Abs(initScriptPath)
|
||||
if err != nil {
|
||||
t.Fatalf("resolve %s: %v", initScriptPath, err)
|
||||
}
|
||||
// `. script` then call the predicate. `set -e` is deliberately NOT used: the
|
||||
// predicates report by exit status, and a false answer is not an error.
|
||||
script := `. "$1" || exit 3; shift; if ` + fn + ` "$@"; then echo yes; else echo no; fi`
|
||||
argv := append([]string{"-c", script, "sh", abs}, args...)
|
||||
out, err := exec.Command("/bin/sh", argv...).CombinedOutput()
|
||||
if err != nil {
|
||||
t.Fatalf("%s(%q): %v\n%s", fn, args, err, out)
|
||||
}
|
||||
switch strings.TrimSpace(string(out)) {
|
||||
case "yes":
|
||||
return true
|
||||
case "no":
|
||||
return false
|
||||
default:
|
||||
t.Fatalf("%s(%q): unreadable answer %q", fn, args, out)
|
||||
return false
|
||||
}
|
||||
}
|
||||
|
||||
// TestInitScriptActionClassification pins the two closed lists. The `shutdown`
|
||||
// rows are the regression: both must be false, because a reboot is neither a
|
||||
// handoff (nothing is coming) nor an operator switching the product off.
|
||||
func TestInitScriptActionClassification(t *testing.T) {
|
||||
if runtime.GOOS == "windows" {
|
||||
t.Skip("needs a POSIX /bin/sh; the gate runs on linux")
|
||||
}
|
||||
for _, tc := range []struct {
|
||||
action string
|
||||
disarms bool
|
||||
handoff bool
|
||||
why string
|
||||
}{
|
||||
{"stop", true, false, "the operator switched the product off"},
|
||||
{"shutdown", false, false, "REBOOT/POWEROFF — must not disarm; this is the boot the armor exists for"},
|
||||
{"restart", false, true, "a successor is coming"},
|
||||
{"reload", false, true, "Save & Apply is stop+start"},
|
||||
{"boot", false, false, "start side, never reaches stop_service"},
|
||||
{"start", false, false, "start side"},
|
||||
{"", false, false, "unknown/empty degrades to changing nothing"},
|
||||
{"enable", false, false, "not a lifecycle transition"},
|
||||
{"disable", false, false, "durable off, but handled by shater-armor's rc.d refusal, not here"},
|
||||
} {
|
||||
if got := askInitScript(t, "shater_action_disarms", tc.action); got != tc.disarms {
|
||||
t.Errorf("shater_action_disarms(%q) = %v, want %v (%s)", tc.action, got, tc.disarms, tc.why)
|
||||
}
|
||||
if got := askInitScript(t, "shater_action_handoff", tc.action); got != tc.handoff {
|
||||
t.Errorf("shater_action_handoff(%q) = %v, want %v (%s)", tc.action, got, tc.handoff, tc.why)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestInitScriptStopDisarmsOnlyForAPerson pins the rest of the decision: `stop`
|
||||
// disarms when a PERSON is behind it, or when the product is being removed — and
|
||||
// not when something is merely replacing it.
|
||||
//
|
||||
// base-files' default_prerm reaches stop_service as a plain `stop`:
|
||||
//
|
||||
// if [ "$PKG_UPGRADE" != "1" ]; then "$i" disable; fi
|
||||
// "$i" stop
|
||||
//
|
||||
// so two different intentions arrive as one action. The rc.d state separates them:
|
||||
// a removal has already run `disable`, a replacement has not.
|
||||
//
|
||||
// On this target an apk UPGRADE turns out never to run default_prerm at all
|
||||
// (no pre-upgrade script — verified with a real `apk fix --reinstall` while
|
||||
// sampling the armor file), so the upgrade rows below are defence-in-depth rather
|
||||
// than a reproduction. The removal rows are live behaviour.
|
||||
func TestInitScriptStopDisarmsOnlyForAPerson(t *testing.T) {
|
||||
if runtime.GOOS == "windows" {
|
||||
t.Skip("needs a POSIX /bin/sh; the gate runs on linux")
|
||||
}
|
||||
for _, tc := range []struct {
|
||||
action string
|
||||
inPkg string
|
||||
rcEnable string
|
||||
want bool
|
||||
why string
|
||||
}{
|
||||
{"stop", "0", "1", true, "an operator typed it — the escape hatch must keep working"},
|
||||
{"stop", "0", "0", true, "an operator typed it on an already-disabled service"},
|
||||
{"stop", "1", "1", false, "BEING REPLACED — prerm left the service enabled, so something is coming back"},
|
||||
{"stop", "1", "0", true, "REMOVAL — prerm already ran `disable`; the product is going away"},
|
||||
{"shutdown", "0", "1", false, "reboot never disarms"},
|
||||
{"shutdown", "1", "1", false, "reboot never disarms, package manager or not"},
|
||||
{"shutdown", "1", "0", false, "still a reboot; the action decides first"},
|
||||
{"restart", "0", "1", false, "a successor is coming"},
|
||||
{"restart", "1", "0", false, "a successor is coming; action decides before any state"},
|
||||
{"reload", "1", "1", false, "Save & Apply"},
|
||||
{"", "1", "0", false, "unknown action changes nothing"},
|
||||
} {
|
||||
got := askInitScript(t, "shater_stop_disarms", tc.action, tc.inPkg, tc.rcEnable)
|
||||
if got != tc.want {
|
||||
t.Errorf("shater_stop_disarms(%q, in_pkg=%s, rc_enabled=%s) = %v, want %v (%s)",
|
||||
tc.action, tc.inPkg, tc.rcEnable, got, tc.want, tc.why)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestInitScriptsParse is the cheapest possible guard against the class of bug
|
||||
// that no Go test can otherwise see: a shell file that ships syntactically
|
||||
// broken. An init script that fails to parse takes the whole service down and
|
||||
// `go build` is perfectly happy about it.
|
||||
func TestInitScriptsParse(t *testing.T) {
|
||||
if runtime.GOOS == "windows" {
|
||||
t.Skip("needs a POSIX /bin/sh; the gate runs on linux")
|
||||
}
|
||||
for _, name := range []string{"shater", "shater-armor", "shater-cron"} {
|
||||
p, err := filepath.Abs(filepath.Join(filepath.Dir(initScriptPath), name))
|
||||
if err != nil {
|
||||
t.Fatalf("resolve %s: %v", name, err)
|
||||
}
|
||||
if out, err := exec.Command("/bin/sh", "-n", p).CombinedOutput(); err != nil {
|
||||
t.Errorf("/etc/init.d/%s does not parse: %v\n%s", name, err, out)
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -289,10 +289,27 @@ func cmdRun() int {
|
||||
// reconcile. No-op when no iface-driven profiles are configured.
|
||||
go watchActiveProfile(applier, logger)
|
||||
|
||||
// Keep the persisted fail-closed plane in step with the config BEFORE anything
|
||||
// is attempted: it is what protects the LAN at the NEXT boot (and across the
|
||||
// next restart), and an engine start that hangs for a minute must not be what
|
||||
// stands between a config change and its armor being written.
|
||||
if readErr == nil {
|
||||
refreshBootArmor(m, logger)
|
||||
}
|
||||
|
||||
// Initial apply. A failed initial apply must NOT crash-loop the box into a
|
||||
// blackout: log it and stay up so a later SIGHUP/apply can fix the config.
|
||||
if readErr != nil {
|
||||
logger.Error("initial ReadUCI failed (staying up): ", readErr)
|
||||
// ...but STAYING UP IS NOT THE SAME AS BEING SAFE. This branch used to end
|
||||
// here, which meant an unreadable /etc/config/shater — a full /overlay caught
|
||||
// mid `uci commit` is the documented cause — left the router with no table at
|
||||
// all, permanently: nothing else installs one (Reconcile returns before any
|
||||
// plane work, and the cron watchdog's own `uci -q get` fails identically), and
|
||||
// the panel reported a live daemon the whole time. The persisted holding plane
|
||||
// is the last thing this router is KNOWN to have wanted, and it needs nothing
|
||||
// readable to be true.
|
||||
armOnUnreadableConfig(logger)
|
||||
} else if m.Globals.Enabled {
|
||||
// ARM FIRST, THEN TRY. Install the fail-closed holding plane BEFORE the
|
||||
// engine is attempted, so the gap between daemon start and a working engine
|
||||
@@ -329,6 +346,13 @@ func cmdRun() int {
|
||||
if addr, ok := panelAddr(globals); ok {
|
||||
panelSrv = panel.NewServer(applier, panel.Options{Addr: addr, Logger: logger})
|
||||
panelSrv.SetStats(statsAgg)
|
||||
// The log download reads the sink's FILE, and the sink writes to it from
|
||||
// its own goroutine — so without a barrier a download can miss the last
|
||||
// milliseconds of lines, which are the ones the operator came for. Hand
|
||||
// the panel the barrier only (not the sink): it is a consumer of the log,
|
||||
// not its owner. Bounded on the panel side, so a stuck writer cannot turn
|
||||
// /api/log into the hang the async sink was built to prevent.
|
||||
panelSrv.SetLogSync(sink.Sync)
|
||||
go func() {
|
||||
if err := panelSrv.Start(); err != nil && err != http.ErrServerClosed {
|
||||
logger.Warn("panel server unavailable (daemon continues): ", err)
|
||||
@@ -362,9 +386,20 @@ func cmdRun() int {
|
||||
// learn whether the plane is meant to be up (globals.enabled) so a
|
||||
// reconcile error can raise the kill-switch alert.
|
||||
enabled := false
|
||||
if mm, e := model.ReadUCI(); e == nil {
|
||||
if mm, e := model.ReadUCI(); e != nil {
|
||||
// The live config just became unreadable. Same answer as at startup:
|
||||
// reinstate what this router last persisted, rather than run on with
|
||||
// whatever the kernel happens to hold.
|
||||
logger.Error("reconcile: could not read the config: ", e)
|
||||
armOnUnreadableConfig(logger)
|
||||
} else {
|
||||
notifier.Update(mm.Alerts)
|
||||
enabled = mm.Globals.Enabled
|
||||
// Keep the persisted fail-closed plane in step with the config the
|
||||
// operator just changed — including the disarm half, so turning the
|
||||
// stack off (or opening the kill switch) also stops the next boot from
|
||||
// blocking the LAN.
|
||||
refreshBootArmor(mm, logger)
|
||||
// Pick up a changed stats backend / sizing without a daemon restart.
|
||||
// A no-op when nothing changed, so a routine reconcile never churns
|
||||
// the store (and never restarts its log cursors).
|
||||
@@ -386,8 +421,31 @@ func cmdRun() int {
|
||||
// DNS-query manager). Re-point the aggregator at the current manager.
|
||||
statsAgg.Resubscribe()
|
||||
case syscall.SIGTERM, syscall.SIGINT:
|
||||
logger.Info("signal ", sig, ": honest teardown + exit")
|
||||
if err := applier.Teardown(); err != nil {
|
||||
// Being REPLACED is not the same as being switched off, and until now
|
||||
// this path could not tell the difference: Teardown does not consult
|
||||
// kill_switch at all (compare holdLocked, which does), so `restart`,
|
||||
// `reload_service` — which is stop+start, i.e. every LuCI Save & Apply —
|
||||
// and every package upgrade dismantled the fail-closed plane and left the
|
||||
// LAN forwarding in the clear for as long as the successor needed to come
|
||||
// up. That interval is guaranteed non-empty by the init itself: it waits
|
||||
// for this process to exit, then runs `shaterd migrate`, then starts the
|
||||
// daemon, which then has to build an engine.
|
||||
//
|
||||
// The init script announces a restart with a tmpfs marker; absent it, this
|
||||
// is a deliberate stop and the plane goes away for good, which is the
|
||||
// operator's escape hatch and must keep working.
|
||||
handoff := restartHandoffPending()
|
||||
logger.Info("signal ", sig, ": honest teardown + exit (restart handoff: ", handoff, ")")
|
||||
// BEFORE the teardown, not after. This used to read `Teardown(); armOnExit()`
|
||||
// on the reasoning that Teardown deletes the table so arming first would be
|
||||
// undone — true, and the wrong conclusion: it left a measured 80-90 ms window per
|
||||
// restart (80-90 ms measured) in which no `inet shater` table existed at all and fw4's
|
||||
// `lan -> wan ACCEPT` was the only policy on the box. TeardownExiting arms
|
||||
// first (one nft transaction that REPLACES the table) and then skips the
|
||||
// delete iff a plane really went in.
|
||||
if err := applier.TeardownExiting(func() bool {
|
||||
return armOnExit(handoff, logger)
|
||||
}); err != nil {
|
||||
logger.Error("teardown: ", err)
|
||||
}
|
||||
return 0
|
||||
@@ -984,14 +1042,23 @@ func handleCtl(conn net.Conn, a *apply.Applier, ps *panel.Server, sa stats.Stats
|
||||
writeLine(conn, string(b))
|
||||
case "reconcile":
|
||||
ch, err := a.Reconcile()
|
||||
// The panel and the CLI reconcile over this socket, not by SIGHUP, so the
|
||||
// persisted fail-closed plane has to be refreshed here too — otherwise
|
||||
// enabling the stack from the panel left the next boot unarmed (and
|
||||
// DISABLING it left the next boot armed) until some unrelated SIGHUP
|
||||
// happened along.
|
||||
if mm, e := model.ReadUCI(); e == nil {
|
||||
refreshBootArmor(mm, l)
|
||||
}
|
||||
writeResult(conn, ch, err)
|
||||
case "apply":
|
||||
a.Snapshot()
|
||||
changed, err := a.Reconcile()
|
||||
if err == nil {
|
||||
if m, e := model.ReadUCI(); e == nil {
|
||||
if m, e := model.ReadUCI(); e == nil {
|
||||
if err == nil {
|
||||
a.ArmRollback(m.Globals.ConfirmTimeout)
|
||||
}
|
||||
refreshBootArmor(m, l)
|
||||
}
|
||||
writeResult(conn, changed, err)
|
||||
case "confirm":
|
||||
|
||||
@@ -33,8 +33,47 @@ var newStatsStore = stats.NewStore
|
||||
var _ stats.StatsStore = (*statsHolder)(nil)
|
||||
|
||||
// statsHolder delegates every StatsStore call to the store currently installed.
|
||||
// Reads take a read lock so a swap never blocks concurrent panel queries for
|
||||
// longer than the pointer exchange.
|
||||
//
|
||||
// The read lock is held for the WHOLE delegated call, not just long enough to
|
||||
// pick the pointer up. That distinction is the entire safety property of this
|
||||
// type, and getting it wrong is invisible in every test that does not race a
|
||||
// reader against a swap:
|
||||
//
|
||||
// // WRONG — this is what it used to do.
|
||||
// func (h *statsHolder) get() stats.StatsStore {
|
||||
// h.mu.RLock(); defer h.mu.RUnlock(); return h.inner
|
||||
// }
|
||||
// func (h *statsHolder) Snapshot() stats.Snapshot { return h.get().Snapshot() }
|
||||
//
|
||||
// The lock is gone by the time Snapshot runs. A reader that has just taken the
|
||||
// pointer is holding nothing: the swap acquires the write lock unopposed, closes
|
||||
// that very store, and the reader then queries a closed one. For the persistent
|
||||
// backend that is not an error the panel can see — a closed bolt ring answers
|
||||
// every read with errRingUnavailable, i.e. an EMPTY PAGE, which the panel renders
|
||||
// as "you have no traffic". A lie, presented as data, from the one component
|
||||
// whose whole job is to report what is happening.
|
||||
//
|
||||
// Ordering the close correctly (see Reconfigure) protects the FILE. It cannot
|
||||
// protect a caller that already holds the pointer; only the lock can.
|
||||
//
|
||||
// The cost is bounded and lands where it should. Readers do not block each other
|
||||
// — RWMutex admits them all concurrently — so the panel's own load is unchanged.
|
||||
// What now waits is the SWAP: it must let the in-flight reads finish, which is
|
||||
// one page query (at most stats.MaxLogLimit rows) plus the old store's close. A
|
||||
// swap happens when a human changes a setting.
|
||||
//
|
||||
// The two alternatives were weighed and rejected:
|
||||
//
|
||||
// - Reference-count the store and close it when the last reader leaves. It
|
||||
// lets the swap return sooner, but it moves the close — flush, final bbolt
|
||||
// commit, and any error it reports — onto whichever panel goroutine happens
|
||||
// to drop the last reference. That is real lifecycle machinery, and error
|
||||
// reporting from an arbitrary place, bought to avoid a wait measured in
|
||||
// milliseconds on a once-per-Apply operation.
|
||||
// - Make a closed store safe to read (serve its last snapshot). It cannot be
|
||||
// done for the part that matters: the query and connection logs live in the
|
||||
// bolt file, and a closed ring has nothing to serve but emptiness. That is
|
||||
// the lie itself, not a defence against it.
|
||||
type statsHolder struct {
|
||||
mu sync.RWMutex
|
||||
inner stats.StatsStore
|
||||
@@ -56,23 +95,60 @@ func newStatsHolder(eng *engine.Engine, logger log.ContextLogger, backend string
|
||||
}
|
||||
}
|
||||
|
||||
func (h *statsHolder) get() stats.StatsStore {
|
||||
// --- stats.StatsStore -------------------------------------------------------
|
||||
//
|
||||
// Every one of these runs the inner call INSIDE the read lock. There is
|
||||
// deliberately no `get()` accessor any more: a helper that hands the pointer out
|
||||
// is a helper that hands out an unguarded reference, and the whole defect above
|
||||
// was one such call site.
|
||||
|
||||
func (h *statsHolder) Start() {
|
||||
h.mu.RLock()
|
||||
defer h.mu.RUnlock()
|
||||
return h.inner
|
||||
h.inner.Start()
|
||||
}
|
||||
|
||||
// --- stats.StatsStore -------------------------------------------------------
|
||||
func (h *statsHolder) Close() error {
|
||||
h.mu.RLock()
|
||||
defer h.mu.RUnlock()
|
||||
return h.inner.Close()
|
||||
}
|
||||
|
||||
func (h *statsHolder) Start() { h.get().Start() }
|
||||
func (h *statsHolder) Close() error { return h.get().Close() }
|
||||
func (h *statsHolder) Resubscribe() { h.get().Resubscribe() }
|
||||
func (h *statsHolder) Resubscribe() {
|
||||
h.mu.RLock()
|
||||
defer h.mu.RUnlock()
|
||||
h.inner.Resubscribe()
|
||||
}
|
||||
|
||||
func (h *statsHolder) Queries(q stats.LogQuery) []stats.LogEntry { return h.get().Queries(q) }
|
||||
func (h *statsHolder) Conns(q stats.LogQuery) []stats.ConnLogEntry { return h.get().Conns(q) }
|
||||
func (h *statsHolder) QueriesPage(q stats.LogQuery) stats.LogPage { return h.get().QueriesPage(q) }
|
||||
func (h *statsHolder) ConnsPage(q stats.LogQuery) stats.ConnPage { return h.get().ConnsPage(q) }
|
||||
func (h *statsHolder) Snapshot() stats.Snapshot { return h.get().Snapshot() }
|
||||
func (h *statsHolder) Queries(q stats.LogQuery) []stats.LogEntry {
|
||||
h.mu.RLock()
|
||||
defer h.mu.RUnlock()
|
||||
return h.inner.Queries(q)
|
||||
}
|
||||
|
||||
func (h *statsHolder) Conns(q stats.LogQuery) []stats.ConnLogEntry {
|
||||
h.mu.RLock()
|
||||
defer h.mu.RUnlock()
|
||||
return h.inner.Conns(q)
|
||||
}
|
||||
|
||||
func (h *statsHolder) QueriesPage(q stats.LogQuery) stats.LogPage {
|
||||
h.mu.RLock()
|
||||
defer h.mu.RUnlock()
|
||||
return h.inner.QueriesPage(q)
|
||||
}
|
||||
|
||||
func (h *statsHolder) ConnsPage(q stats.LogQuery) stats.ConnPage {
|
||||
h.mu.RLock()
|
||||
defer h.mu.RUnlock()
|
||||
return h.inner.ConnsPage(q)
|
||||
}
|
||||
|
||||
func (h *statsHolder) Snapshot() stats.Snapshot {
|
||||
h.mu.RLock()
|
||||
defer h.mu.RUnlock()
|
||||
return h.inner.Snapshot()
|
||||
}
|
||||
|
||||
// --- swapping ---------------------------------------------------------------
|
||||
|
||||
@@ -90,17 +166,35 @@ func statsSpecFrom(g model.Globals) (string, stats.Config) {
|
||||
// Reconfigure swaps in a new store when backend/cfg differ from what is running.
|
||||
// It reports whether a swap happened.
|
||||
//
|
||||
// Ordering is deliberate: the NEW store is built and started BEFORE the old one
|
||||
// is closed, so a failure to construct it leaves the running store untouched and
|
||||
// stats keep working. The old store is then closed, which for the persistent
|
||||
// backend flushes the pending write buffer into its final bbolt commit — dropping
|
||||
// it without Close would lose the not-yet-committed tail of the log.
|
||||
// Ordering is deliberate, and it is the OPPOSITE of what it was: the old store is
|
||||
// CLOSED FIRST, and only then is the new one built.
|
||||
//
|
||||
// Building first looked safer — a construction failure would leave the running
|
||||
// store untouched — but the two stores of a persistent-backend swap resolve the
|
||||
// SAME on-disk path, and the second bbolt.Open there cannot win a file the first
|
||||
// one still holds. It sat on the flock for a second, timed out, and the old ring's
|
||||
// "an unopenable file must be a corrupt file" recovery deleted the live database
|
||||
// out from under its own writer (unlink of an open file succeeds on Linux, so
|
||||
// nothing complained): months of query log gone, replaced by an empty DB whose
|
||||
// Snapshot still reported the same backend. Changing ANY of the five stats knobs
|
||||
// was enough. The overlap was the whole defect, so the overlap is what had to go —
|
||||
// the classification bug is fixed alongside it in stats.newBoltRing, but a swap
|
||||
// that hands the file over cleanly does not depend on that fix being right.
|
||||
//
|
||||
// Closing first also flushes the old store's pending write buffer into its final
|
||||
// bbolt commit; dropping it without Close would lose the not-yet-committed tail.
|
||||
//
|
||||
// The whole exchange runs under the write lock, so a concurrent /api/stats poll
|
||||
// sees either the OLD OPEN store or the new one — never the closed one in between,
|
||||
// which would answer an honest-looking empty page. Readers pay one flush+commit of
|
||||
// blocking on a knob change; a torn read costs the panel its trust.
|
||||
//
|
||||
// stats.NewStore itself never returns nil and never panics: an unopenable stats
|
||||
// DB degrades to the in-memory ring internally and reports Backend "memory" from
|
||||
// Snapshot, so the panel shows what is actually running rather than what was
|
||||
// asked for. That is the "degrade, don't die" requirement, and it means
|
||||
// Reconfigure has no error path of its own.
|
||||
// Reconfigure has no error path of its own — and it is why closing first cannot
|
||||
// strand the daemon without a store.
|
||||
//
|
||||
// Log cursor continuity is DELIBERATELY not preserved across a swap. Each store
|
||||
// owns its own seq counter (memRing counts from zero; boltRing seeds from its own
|
||||
@@ -112,12 +206,19 @@ func statsSpecFrom(g model.Globals) (string, stats.Config) {
|
||||
// honest Backend in the snapshot — which it gets for free by delegating.
|
||||
func (h *statsHolder) Reconfigure(backend string, cfg stats.Config) bool {
|
||||
h.mu.Lock()
|
||||
defer h.mu.Unlock()
|
||||
if backend == h.backend && cfg == h.cfg {
|
||||
h.mu.Unlock()
|
||||
return false
|
||||
}
|
||||
old, oldBackend := h.inner, h.backend
|
||||
|
||||
// Hand the resources over BEFORE asking for them again: the old store releases
|
||||
// its DB file (and its subscriptions) here, so the new one opens a path nobody
|
||||
// holds. See the ordering note above — this line is the fix.
|
||||
if err := old.Close(); err != nil {
|
||||
h.log.Warn("stats: closing the previous ", oldBackend, " store: ", err)
|
||||
}
|
||||
|
||||
next := newStatsStore(backend, h.eng, h.log, cfg)
|
||||
next.Start()
|
||||
// Point the fresh store at the running box's event managers. Start() subscribes,
|
||||
@@ -130,13 +231,7 @@ func (h *statsHolder) Reconfigure(backend string, cfg stats.Config) bool {
|
||||
h.inner = next
|
||||
h.backend = backend
|
||||
h.cfg = cfg
|
||||
h.mu.Unlock()
|
||||
|
||||
// Close the old store OUTSIDE the lock: a persistent-store Close flushes and
|
||||
// commits, which can take a moment, and no panel read should block behind it.
|
||||
if err := old.Close(); err != nil {
|
||||
h.log.Warn("stats: closing the previous ", oldBackend, " store: ", err)
|
||||
}
|
||||
h.log.Info("stats: backend ", oldBackend, " -> ", backend,
|
||||
" (log cursors restart; reload the panel to resume live tailing)")
|
||||
return true
|
||||
|
||||
@@ -1,8 +1,10 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"sync"
|
||||
"sync/atomic"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"github.com/sagernet/sing-box/log"
|
||||
"github.com/sagernet/sing-box/shater/engine"
|
||||
@@ -150,3 +152,234 @@ func TestStatsHolderDelegatesAfterSwap(t *testing.T) {
|
||||
t.Errorf("Close: %v", err)
|
||||
}
|
||||
}
|
||||
|
||||
// orderedStore is a fakeStore that timestamps its construction and its Close against a
|
||||
// shared step counter, so a test can assert the ORDER of the swap choreography rather
|
||||
// than only its outcome.
|
||||
type orderedStore struct {
|
||||
fakeStore
|
||||
step *atomic.Int32
|
||||
builtAt int32
|
||||
closedAt int32
|
||||
}
|
||||
|
||||
func (o *orderedStore) Close() error {
|
||||
o.closedAt = o.step.Add(1)
|
||||
return nil
|
||||
}
|
||||
|
||||
// TestStatsHolderClosesOldBeforeOpeningNew is the data-loss regression: changing ANY of
|
||||
// the five stats knobs used to destroy the persistent log.
|
||||
//
|
||||
// The swap built the replacement store BEFORE closing the one it replaced, and both
|
||||
// halves resolve the SAME on-disk DB path (stats.statsFilePath). The second bbolt.Open
|
||||
// therefore lost the flock race against a database that was still open, timed out after a
|
||||
// second, and the ring's "an unopenable file must be a corrupt file" recovery deleted it —
|
||||
// on Linux, unlinking an open file succeeds, so the old store went on writing into a
|
||||
// nameless inode while the panel was told nothing had changed. Months of query log, gone,
|
||||
// for a ring-size edit.
|
||||
//
|
||||
// The store the holder builds in production owns an exclusive resource, so the ONLY safe
|
||||
// order is release-then-acquire. That is what is pinned here; the constructor seam makes
|
||||
// it assertable without standing up a real DB.
|
||||
func TestStatsHolderClosesOldBeforeOpeningNew(t *testing.T) {
|
||||
var step atomic.Int32
|
||||
var built []*orderedStore
|
||||
orig := newStatsStore
|
||||
newStatsStore = func(backend string, _ *engine.Engine, _ log.ContextLogger, cfg ...stats.Config) stats.StatsStore {
|
||||
s := &orderedStore{step: &step}
|
||||
s.backend = backend
|
||||
if len(cfg) > 0 {
|
||||
s.cfg = cfg[0]
|
||||
}
|
||||
s.builtAt = step.Add(1)
|
||||
built = append(built, s)
|
||||
return s
|
||||
}
|
||||
t.Cleanup(func() { newStatsStore = orig })
|
||||
|
||||
h := newStatsHolderFrom(nil, log.StdLogger(), globalsWith("sqlite", 200))
|
||||
h.Start()
|
||||
if !h.ReconfigureFrom(globalsWith("sqlite", 5000)) {
|
||||
t.Fatalf("changing the ring size must swap the store")
|
||||
}
|
||||
if len(built) != 2 {
|
||||
t.Fatalf("built %d stores, want 2", len(built))
|
||||
}
|
||||
old, next := built[0], built[1]
|
||||
|
||||
if old.closedAt == 0 {
|
||||
t.Fatal("the replaced store was never closed — its buffered rows are lost and its DB file stays locked")
|
||||
}
|
||||
if old.closedAt > next.builtAt {
|
||||
t.Fatalf("the new store was constructed at step %d while the old one was still open (closed at step %d): "+
|
||||
"both resolve the same stats DB path, so the second open loses the flock race and the loser's "+
|
||||
"recovery path deletes the live database", next.builtAt, old.closedAt)
|
||||
}
|
||||
}
|
||||
|
||||
// TestStatsHolderReadsNeverSeeTheClosedStore is the second half of the swap defect,
|
||||
// and it is about the READER rather than the file.
|
||||
//
|
||||
// Closing the outgoing store before opening its replacement is what stops the DB file
|
||||
// from being deleted, but it does nothing for a caller that is already inside a call on
|
||||
// that store. The holder used to take the read lock only long enough to copy the pointer
|
||||
// out (`get()`), so the sequence below was ordinary, not exotic:
|
||||
//
|
||||
// 1. a panel poll takes the pointer to the running store and drops the lock;
|
||||
// 2. Reconfigure takes the write lock — unopposed, nobody is holding the read side —
|
||||
// and closes that store;
|
||||
// 3. the poll now runs its query against a store that is shut.
|
||||
//
|
||||
// For the persistent backend step 3 is not a visible error: a closed bolt ring answers
|
||||
// every read with errRingUnavailable, which reaches the panel as an empty page and is
|
||||
// drawn as "no traffic". That is the same class of lie as everything else in this pass,
|
||||
// arriving from the other direction — the component whose entire job is to report what
|
||||
// happened, reporting that nothing did.
|
||||
//
|
||||
// The test drives the three steps EXPLICITLY rather than hoping a hammering loop lands in
|
||||
// the window (it did, about one run in five, which is exactly the kind of failure that
|
||||
// gets dismissed as a flake). The reader is parked inside the outgoing store's Snapshot;
|
||||
// the swap is then given every chance to close it underneath.
|
||||
func TestStatsHolderReadsNeverSeeTheClosedStore(t *testing.T) {
|
||||
var stores []*gatedStore
|
||||
orig := newStatsStore
|
||||
newStatsStore = func(backend string, _ *engine.Engine, _ log.ContextLogger, cfg ...stats.Config) stats.StatsStore {
|
||||
g := &gatedStore{entered: make(chan struct{}), release: make(chan struct{})}
|
||||
g.backend = backend
|
||||
if len(stores) > 0 {
|
||||
close(g.release) // only the OUTGOING store parks its reader
|
||||
}
|
||||
stores = append(stores, g)
|
||||
return g
|
||||
}
|
||||
t.Cleanup(func() { newStatsStore = orig })
|
||||
|
||||
h := newStatsHolderFrom(nil, log.StdLogger(), globalsWith("sqlite", 200))
|
||||
h.Start()
|
||||
outgoing := stores[0]
|
||||
|
||||
// 1. A panel poll is inside the outgoing store and cannot be hurried.
|
||||
read := make(chan string, 1)
|
||||
go func() { read <- h.Snapshot().Backend }()
|
||||
select {
|
||||
case <-outgoing.entered:
|
||||
case <-time.After(10 * time.Second):
|
||||
close(outgoing.release)
|
||||
t.Fatal("the reader never reached the store")
|
||||
}
|
||||
|
||||
// 2. The swap runs while that reader is still inside.
|
||||
swapped := make(chan struct{})
|
||||
go func() {
|
||||
defer close(swapped)
|
||||
h.ReconfigureFrom(globalsWith("sqlite", 5000))
|
||||
}()
|
||||
|
||||
// Give it every opportunity to close the store out from under the parked reader.
|
||||
// A correct holder cannot: the reader holds the read lock, so the swap is still
|
||||
// waiting for the write lock when this deadline expires. A holder that only guards
|
||||
// the pointer gets there in microseconds, and this loop ends the moment it does.
|
||||
deadline := time.Now().Add(300 * time.Millisecond)
|
||||
for !outgoing.shut.Load() && time.Now().Before(deadline) {
|
||||
time.Sleep(time.Millisecond)
|
||||
}
|
||||
|
||||
// 3. Let the poll finish and see what it was served.
|
||||
close(outgoing.release)
|
||||
got := <-read
|
||||
select {
|
||||
case <-swapped:
|
||||
case <-time.After(10 * time.Second):
|
||||
t.Fatal("the swap never completed after the reader left")
|
||||
}
|
||||
|
||||
if got == closedStoreBackend {
|
||||
t.Fatal("a panel read was served by the store that was being closed — an empty page presented as data")
|
||||
}
|
||||
if got != "sqlite" {
|
||||
t.Fatalf("the read returned %q, want the running backend", got)
|
||||
}
|
||||
if len(stores) != 2 {
|
||||
t.Fatalf("built %d stores, want 2", len(stores))
|
||||
}
|
||||
if !outgoing.shut.Load() {
|
||||
t.Fatal("the outgoing store was never closed")
|
||||
}
|
||||
if got := h.Snapshot().Backend; got != "sqlite" {
|
||||
t.Fatalf("after the swap Snapshot reports %q, want sqlite", got)
|
||||
}
|
||||
}
|
||||
|
||||
// closedStoreBackend is what gatedStore reports once it has been closed. No real store
|
||||
// says this; it exists so that a read served by a shut store is unmistakable instead of
|
||||
// merely empty — which is precisely what makes the real defect hard to see.
|
||||
const closedStoreBackend = "closed"
|
||||
|
||||
// gatedStore parks the first caller of Snapshot until the test releases it, and reports
|
||||
// closedStoreBackend from the moment Close has run.
|
||||
type gatedStore struct {
|
||||
fakeStore
|
||||
entered chan struct{}
|
||||
release chan struct{}
|
||||
enterOnce sync.Once
|
||||
shut atomic.Bool
|
||||
}
|
||||
|
||||
func (s *gatedStore) Close() error {
|
||||
s.shut.Store(true)
|
||||
return nil
|
||||
}
|
||||
|
||||
func (s *gatedStore) Snapshot() stats.Snapshot {
|
||||
s.enterOnce.Do(func() { close(s.entered) })
|
||||
<-s.release
|
||||
if s.shut.Load() {
|
||||
return stats.Snapshot{Backend: closedStoreBackend}
|
||||
}
|
||||
return stats.Snapshot{Backend: s.backend}
|
||||
}
|
||||
|
||||
// TestStatsHolderSwapWaitsForInFlightReads states the cost of the above out loud, so a
|
||||
// later reader of this file knows the wait is intended and not an oversight: a swap does
|
||||
// not begin while a read is in progress.
|
||||
func TestStatsHolderSwapWaitsForInFlightReads(t *testing.T) {
|
||||
var stores []*gatedStore
|
||||
orig := newStatsStore
|
||||
newStatsStore = func(backend string, _ *engine.Engine, _ log.ContextLogger, cfg ...stats.Config) stats.StatsStore {
|
||||
g := &gatedStore{entered: make(chan struct{}), release: make(chan struct{})}
|
||||
g.backend = backend
|
||||
if len(stores) > 0 {
|
||||
close(g.release)
|
||||
}
|
||||
stores = append(stores, g)
|
||||
return g
|
||||
}
|
||||
t.Cleanup(func() { newStatsStore = orig })
|
||||
|
||||
h := newStatsHolderFrom(nil, log.StdLogger(), globalsWith("sqlite", 200))
|
||||
h.Start()
|
||||
outgoing := stores[0]
|
||||
|
||||
go func() { _ = h.Snapshot() }()
|
||||
<-outgoing.entered
|
||||
|
||||
swapped := make(chan struct{})
|
||||
go func() {
|
||||
defer close(swapped)
|
||||
h.ReconfigureFrom(globalsWith("sqlite", 5000))
|
||||
}()
|
||||
|
||||
select {
|
||||
case <-swapped:
|
||||
close(outgoing.release)
|
||||
t.Fatal("the swap completed while a read was still inside the store it retired")
|
||||
case <-time.After(200 * time.Millisecond):
|
||||
}
|
||||
close(outgoing.release)
|
||||
select {
|
||||
case <-swapped:
|
||||
case <-time.After(10 * time.Second):
|
||||
t.Fatal("the swap never completed after the read finished")
|
||||
}
|
||||
}
|
||||
|
||||
+33
-1
@@ -159,7 +159,21 @@ func New(logger ...log.ContextLogger) *Engine {
|
||||
// URLTestHistory() below. The pointer is stable across Apply swaps, so health
|
||||
// history survives config changes instead of being reset on every apply.
|
||||
ctx = service.ContextWithPtr(ctx, urltest.NewHistoryStorage())
|
||||
return &Engine{ctx: ctx, log: l}
|
||||
// Pre-register the engine itself as the probe GATE, into the same shared
|
||||
// registry and for the same reason as the two above: every urltest group
|
||||
// built by every box.New reads it out of ctx (protocol/group/urltest.go), and
|
||||
// the registration must survive Apply swaps because the question it answers
|
||||
// outlives any single box.
|
||||
//
|
||||
// The gate answers "may this outbound run its own scheduled probe right
|
||||
// now" — see Engine.ProbeAllowed. It is a live call, not a stored flag, so
|
||||
// a chain hop that comes back resumes probing with no reapply. Registering
|
||||
// the Engine here (rather than a snapshot) is what makes that possible;
|
||||
// the struct is built first purely so ctx can point at it.
|
||||
e := &Engine{log: l}
|
||||
ctx = service.ContextWith[urltest.ProbeGate](ctx, e)
|
||||
e.ctx = ctx
|
||||
return e
|
||||
}
|
||||
|
||||
// Apply installs opts as the running configuration.
|
||||
@@ -193,6 +207,24 @@ func (e *Engine) Apply(opts option.Options) (changed bool, err error) {
|
||||
}
|
||||
|
||||
func (e *Engine) applyLocked(opts option.Options) (bool, error) {
|
||||
// Stand down the self-check of urltest groups no rule reaches (see
|
||||
// selfcheck.go for the whole argument). This MUTATES opts in place — the
|
||||
// option structs are pointers behind an `any` — and it must run BEFORE the
|
||||
// hash below, so the hash describes the config that is really built: a rule
|
||||
// change that flips a group used<->unused is then a real change that
|
||||
// triggers a swap, and an unchanged config hashes identically on every
|
||||
// reconcile because the stand-down is deterministic.
|
||||
stood, used := standDownUnusedSelfCheck(opts)
|
||||
if stood > 0 && e.log != nil {
|
||||
e.log.Info("apply: stood down self-check on ", stood, " unused urltest group(s); the observatory is their only prober")
|
||||
}
|
||||
// Publish the same walk's used-set for ProbeWhenIdle BEFORE the new box is
|
||||
// built, because the question is asked during that box's PostStart: a used
|
||||
// group arms its probing ticker there rather than waiting for traffic. The
|
||||
// observatory's own copy is published later, after a SUCCESSFUL swap, which
|
||||
// is too late to be the answer.
|
||||
e.publishKeepWarm(used)
|
||||
|
||||
newHash, err := e.hashOptions(opts)
|
||||
if err != nil {
|
||||
return false, E.Cause(err, "hash options")
|
||||
|
||||
+269
-14
@@ -1,6 +1,7 @@
|
||||
package engine
|
||||
|
||||
import (
|
||||
"sort"
|
||||
"strings"
|
||||
|
||||
"github.com/sagernet/sing-box/adapter"
|
||||
@@ -320,11 +321,15 @@ func parseGroupCopyTag(group, tag string) (member string, ok bool) {
|
||||
return member, true
|
||||
}
|
||||
|
||||
// ChainHealth is one configured chain's reachability, mirroring GroupHealth.Used
|
||||
// for the chain card (plan §5.E): a chain no enabled rule routes through is never
|
||||
// probed — the observatory walks only reachable paths — and the panel renders it
|
||||
// "unused" rather than as a health problem. A chain has no membership counters: it
|
||||
// is a fixed path, and its end-to-end health is the exit test's job, not a roll-up.
|
||||
// ChainHealth is one configured chain's reachability plus its PER-HOP health,
|
||||
// mirroring GroupHealth.Used for the chain card (plan §5.E): a chain no enabled
|
||||
// rule routes through is never probed — the observatory walks only reachable
|
||||
// paths — and the panel renders it "unused" rather than as a health problem.
|
||||
// A chain has no membership counters of its own: it is a fixed path, and its
|
||||
// end-to-end health is the exit probe's job. What it DOES have is hops, and
|
||||
// since the observatory now probes every hop wrapper (probeplan.go walkDetour),
|
||||
// each hop's health is on the board and is projected here so an operator can
|
||||
// see WHICH hop died instead of only that the chain did.
|
||||
type ChainHealth struct {
|
||||
// Name is the chain's model name (config chain "chain:<name>"), what the Targets
|
||||
// page lists and what a rule targets.
|
||||
@@ -336,34 +341,284 @@ type ChainHealth struct {
|
||||
// when the observatory is disabled or not yet configured: no badge is better
|
||||
// than a wrong one (the same rule as GroupHealth.Used).
|
||||
Used bool `json:"used"`
|
||||
// Hops is the per-hop health readout, L1..Ln in wire order (see ChainHopHealth).
|
||||
// It is EMPTY for a chain the running box never materialised: a chain no rule
|
||||
// references is resolved lazily and never built, and a 1-hop chain without an
|
||||
// egress entry resolves straight to its target with no wrapper — in both cases
|
||||
// there are no "chain-<name>-h…" outbounds to project. An absent "hops" key
|
||||
// therefore means "nothing materialised to report on", NEVER "this chain has
|
||||
// no hops" — the model, not this projection, knows how many hops were
|
||||
// configured.
|
||||
Hops []ChainHopHealth `json:"hops,omitempty"`
|
||||
}
|
||||
|
||||
// ChainHealth reports the reachability (used/unused) of every named chain, one row
|
||||
// per name, in the order given. It is the chain analogue of GroupHealth.Used: a
|
||||
// chain the observatory's used-set does not cover is reported Used=false so the
|
||||
// panel can mark it "unused" instead of running an exit test against a path nothing
|
||||
// routes through.
|
||||
// ChainHopHealth is one hop of one materialised chain, as the health board saw
|
||||
// it — a projection, like everything in this file: nothing here dials.
|
||||
//
|
||||
// A NODE hop is a single measurement: the observatory dials the hop wrapper
|
||||
// "chain-<name>-h<i>", which pulls exactly the path prefix up to and including
|
||||
// this hop, so Total=1, the counters follow the hop's own state, DelayMs and
|
||||
// AgeSeconds are its own observation, and Selected is "" (a fixed hop selects
|
||||
// nothing).
|
||||
//
|
||||
// A GROUP hop rolls up its member copies "chain-<name>-h<i>-<member>", each of
|
||||
// which the observatory probes through its own prefix of the chain. The
|
||||
// counters obey the same invariants as GroupHealth — Tested == Alive+Dead and
|
||||
// Alive+Dead+Untested == Total — so the panel needs no arithmetic of its own.
|
||||
// State summarises them: "alive" when at least one member is alive (the hop can
|
||||
// carry traffic), "dead" when at least one was tested and none is alive (a
|
||||
// positive finding of a dead hop), "untested" when nothing was tested. Selected
|
||||
// is the node NAME the wrapper currently picks; DelayMs/AgeSeconds are the
|
||||
// SELECTED member's observation, or the freshest ALIVE member's when the
|
||||
// selection has no measurement of its own — the number shown must always be a
|
||||
// measurement somebody took, never an average nobody did.
|
||||
//
|
||||
// # The hop nobody reached: BlockedBy
|
||||
//
|
||||
// The prober walks a chain in wire order and stops at the first dead hop
|
||||
// (observatory.go probeChainOrdered), because a probe of hop 3 dials THROUGH
|
||||
// hop 2 and a hop 2 with no live member makes that probe a measurement of hop 2.
|
||||
// Everything behind the break is therefore not dialled at all, and this struct
|
||||
// says so: State is untested — no fresh knowledge, which is the truth — the
|
||||
// counters collapse into Untested, and BlockedBy names the hop that stopped the
|
||||
// walk. A row like that must never be read as a fault of its own; the fault is
|
||||
// at BlockedBy.Index, and that is the hop to go and fix.
|
||||
type ChainHopHealth struct {
|
||||
Index int `json:"index"` // 1-based position on the wire, L1..Ln
|
||||
Tag string `json:"tag"` // "chain-<name>-h<i>" — the wrapper actually dialled
|
||||
Kind string `json:"kind"` // "node" | "group"
|
||||
Exit bool `json:"exit"` // the LAST hop: where traffic leaves to the internet
|
||||
State string `json:"state"` // "alive" | "dead" | "untested"
|
||||
DelayMs int `json:"delay_ms"`
|
||||
AgeSeconds int64 `json:"age_seconds"` // -1 when unknown
|
||||
Selected string `json:"selected"` // group hop: the node NAME it currently selects; "" otherwise
|
||||
Total int `json:"total"`
|
||||
Tested int `json:"tested"`
|
||||
Alive int `json:"alive"`
|
||||
Dead int `json:"dead"`
|
||||
Untested int `json:"untested"`
|
||||
// BlockedBy is set ONLY on a hop the ordered walk never reached, and names the
|
||||
// dead hop in front of it. Absent (omitted, never null) on every hop that was
|
||||
// itself measured. When it is present State is always "untested" and the
|
||||
// counters are always 0/0/0/Total — see the type comment.
|
||||
BlockedBy *ChainHopBlock `json:"blocked_by,omitempty"`
|
||||
}
|
||||
|
||||
// ChainHopBlock identifies the hop whose failure stopped a chain's ordered probe
|
||||
// walk: its 1-based wire index and the engine outbound tag that was dialled.
|
||||
//
|
||||
// The index is the field that matters — it is what the operator acts on, and it
|
||||
// lines up with ChainHopHealth.Index on the same chain, so a reader can point
|
||||
// straight at the offending row. The tag is diagnostic detail (logs, tooltips)
|
||||
// and is not a label to put in front of a person: "chain-ewan-wg-subs-h2" is not
|
||||
// a name anybody chose.
|
||||
type ChainHopBlock struct {
|
||||
Index int `json:"index"`
|
||||
Tag string `json:"tag"`
|
||||
}
|
||||
|
||||
// ChainHealth reports the reachability (used/unused) of every named chain plus
|
||||
// the per-hop health of each one the running box materialised, one row per
|
||||
// name, in the order given. The Used half is the chain analogue of
|
||||
// GroupHealth.Used: a chain the observatory's used-set does not cover is
|
||||
// reported Used=false so the panel can mark it "unused" instead of a health
|
||||
// readout against a path nothing routes through.
|
||||
//
|
||||
// names come from the desired-state model, NOT the running box: a chain no rule
|
||||
// references is never materialised (generate/chain.go resolveChain is lazy), so it
|
||||
// is invisible to a box-only enumeration — yet the panel lists it from the config
|
||||
// and must be able to badge it. The engine supplies the only fact a box read can
|
||||
// add here, the observatory's published used-set. Pure apart from that read; nil
|
||||
// names or a stopped engine (nil used-set) yield an empty/used-everything result.
|
||||
// and must be able to badge it. The engine supplies what only it can: the
|
||||
// observatory's published used-set, and the running box's outbound/endpoint
|
||||
// pool the hop projection reads. Still no dialling anywhere; nil names or a
|
||||
// stopped engine (nil used-set, empty pool) yield an empty/used-everything
|
||||
// result with no hops.
|
||||
func (e *Engine) ChainHealth(names []string) []ChainHealth {
|
||||
out := make([]ChainHealth, 0, len(names))
|
||||
used := e.observatoryUsed()
|
||||
usedChains := usedChainNames(used)
|
||||
pool := e.runningPool()
|
||||
view := e.HealthView()
|
||||
for _, name := range names {
|
||||
name = strings.TrimSpace(name)
|
||||
if name == "" {
|
||||
continue
|
||||
}
|
||||
out = append(out, ChainHealth{Name: name, Used: used == nil || usedChains[name]})
|
||||
out = append(out, ChainHealth{
|
||||
Name: name,
|
||||
Used: used == nil || usedChains[name],
|
||||
Hops: chainHopHealthOf(pool, name, view),
|
||||
})
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
// runningPool snapshots the running box's outbounds AND endpoints into one
|
||||
// list. The endpoints matter: a chain hop rebuilt from a wireguard/AmneziaWG
|
||||
// node is an ENDPOINT copy, invisible in Outbounds() — the same trap
|
||||
// groupTargets documents — and a hop projection that missed it would silently
|
||||
// drop the very hop this feature exists to localise (the production chain's
|
||||
// first hop IS an AWG endpoint). Empty (never nil-unsafe) on a stopped engine.
|
||||
func (e *Engine) runningPool() []adapter.Outbound {
|
||||
var pool []adapter.Outbound
|
||||
inst := e.Instance()
|
||||
if inst == nil {
|
||||
return pool
|
||||
}
|
||||
if om := inst.Outbound(); om != nil {
|
||||
pool = append(pool, om.Outbounds()...)
|
||||
}
|
||||
if em := inst.Endpoint(); em != nil {
|
||||
for _, ep := range em.Endpoints() {
|
||||
pool = append(pool, ep)
|
||||
}
|
||||
}
|
||||
return pool
|
||||
}
|
||||
|
||||
// chainHopHealthOf projects one chain's hop wrappers out of an outbound pool
|
||||
// against one health view. Pure apart from the view reads — the same
|
||||
// unit-testing contract as groupHealthOf: hand it a fake pool and a hand-built
|
||||
// store and every branch is reachable without a box.
|
||||
//
|
||||
// A pool entry belongs to chain <name> when its tag is exactly
|
||||
// "chain-<name>-h<digits>" — the hop WRAPPER the observatory dials. Member
|
||||
// copies ("chain-<name>-h<i>-<member>") are not hops themselves; they are
|
||||
// reached through the wrapper's own member list (adapter.OutboundGroup.All), so
|
||||
// the roll-up sees exactly what the wrapper can select, in its order.
|
||||
func chainHopHealthOf(pool []adapter.Outbound, name string, view HealthView) []ChainHopHealth {
|
||||
prefix := "chain-" + name + "-h"
|
||||
var hops []ChainHopHealth
|
||||
for _, ob := range pool {
|
||||
tag := ob.Tag()
|
||||
rest, ok := strings.CutPrefix(tag, prefix)
|
||||
if !ok {
|
||||
continue
|
||||
}
|
||||
idx, ok := parseAllDigits(rest)
|
||||
if !ok {
|
||||
continue // a member copy, or another chain sharing the prefix
|
||||
}
|
||||
if g, isGroup := ob.(adapter.OutboundGroup); isGroup {
|
||||
hops = append(hops, chainGroupHop(g, name, idx, view))
|
||||
} else {
|
||||
hops = append(hops, chainNodeHop(tag, idx, view))
|
||||
}
|
||||
}
|
||||
sort.Slice(hops, func(i, j int) bool { return hops[i].Index < hops[j].Index })
|
||||
if len(hops) > 0 {
|
||||
// The largest index is the exit — the wrapper whose probe leaves to the
|
||||
// internet. Marked after sorting so the flag cannot depend on pool order.
|
||||
hops[len(hops)-1].Exit = true
|
||||
}
|
||||
markBlockedHops(hops)
|
||||
return hops
|
||||
}
|
||||
|
||||
// markBlockedHops applies the ordered walk's outcome to the projection: from the
|
||||
// first hop that reads DEAD, every later hop is rewritten as not-reached.
|
||||
//
|
||||
// This is a rewrite and not an annotation on purpose. The prober stops at the
|
||||
// first dead hop, so a hop behind the break has not been dialled since the break
|
||||
// appeared — but its board records do not vanish, they age out on the TTL. For
|
||||
// as long as that takes, the raw projection would keep publishing a verdict about
|
||||
// a hop nothing has attempted, and the most damaging version of that is a stale
|
||||
// "dead" pointing the operator at a hop that may be perfectly fine. Untested is
|
||||
// the honest reading of "nothing current is known", and BlockedBy is why.
|
||||
//
|
||||
// The counters collapse the same way, into Untested, so the invariants the whole
|
||||
// health surface promises still hold verbatim: Tested == Alive+Dead and
|
||||
// Alive+Dead+Untested == Total.
|
||||
//
|
||||
// Selected SURVIVES. It is not a measurement — it is the member the wrapper would
|
||||
// route through — and it stays true (and useful) whether or not anything reached
|
||||
// this hop.
|
||||
func markBlockedHops(hops []ChainHopHealth) {
|
||||
for i := range hops {
|
||||
if hops[i].State != HealthDead {
|
||||
continue
|
||||
}
|
||||
blocker := &ChainHopBlock{Index: hops[i].Index, Tag: hops[i].Tag}
|
||||
for j := i + 1; j < len(hops); j++ {
|
||||
h := &hops[j]
|
||||
h.State = HealthUntested
|
||||
h.DelayMs, h.AgeSeconds = 0, -1
|
||||
h.Alive, h.Dead, h.Tested = 0, 0, 0
|
||||
h.Untested = h.Total
|
||||
h.BlockedBy = blocker
|
||||
}
|
||||
return // the first break is the only one that can be observed
|
||||
}
|
||||
}
|
||||
|
||||
// chainNodeHop is the one-measurement hop: the wrapper itself was dialled by
|
||||
// the observatory, so its own board state IS the hop's health and the counters
|
||||
// degenerate to whichever bucket that state fills.
|
||||
func chainNodeHop(tag string, idx int, view HealthView) ChainHopHealth {
|
||||
hop := ChainHopHealth{Index: idx, Tag: tag, Kind: "node", Total: 1}
|
||||
hop.State, hop.DelayMs, hop.AgeSeconds = view.State(tag)
|
||||
switch hop.State {
|
||||
case HealthAlive:
|
||||
hop.Alive = 1
|
||||
case HealthDead:
|
||||
hop.Dead = 1
|
||||
default:
|
||||
hop.Untested = 1
|
||||
}
|
||||
hop.Tested = hop.Alive + hop.Dead
|
||||
return hop
|
||||
}
|
||||
|
||||
// chainGroupHop rolls a group hop up over its member copies. The counters carry
|
||||
// the GroupHealth invariants; the summary State answers the only question a hop
|
||||
// row asks — "can this hop carry the chain": alive while anything answers,
|
||||
// dead only on a positive all-tested-dead finding, untested when nothing is
|
||||
// known (never dead-by-absence, the same honesty rule as everywhere else).
|
||||
func chainGroupHop(g adapter.OutboundGroup, chain string, idx int, view HealthView) ChainHopHealth {
|
||||
hop := ChainHopHealth{Index: idx, Tag: g.Tag(), Kind: "group", AgeSeconds: -1}
|
||||
selected := g.Now()
|
||||
if selected != "" {
|
||||
hop.Selected = chainMemberName(selected, chain)
|
||||
}
|
||||
|
||||
// The number a hop row shows must be a real observation: the selected
|
||||
// member's when it has one, else the freshest alive member's.
|
||||
freshDelay, freshAge := 0, int64(-1)
|
||||
selDelay, selAge := 0, int64(-1)
|
||||
for _, tag := range g.All() {
|
||||
state, delayMs, age := view.State(tag)
|
||||
switch state {
|
||||
case HealthAlive:
|
||||
hop.Alive++
|
||||
if age >= 0 && (freshAge < 0 || age < freshAge) {
|
||||
freshDelay, freshAge = delayMs, age
|
||||
}
|
||||
case HealthDead:
|
||||
hop.Dead++
|
||||
default:
|
||||
hop.Untested++
|
||||
}
|
||||
hop.Total++
|
||||
if tag == selected && age >= 0 {
|
||||
selDelay, selAge = delayMs, age
|
||||
}
|
||||
}
|
||||
hop.Tested = hop.Alive + hop.Dead
|
||||
switch {
|
||||
case hop.Alive > 0:
|
||||
hop.State = HealthAlive
|
||||
case hop.Tested > 0:
|
||||
hop.State = HealthDead
|
||||
default:
|
||||
hop.State = HealthUntested
|
||||
}
|
||||
if selAge >= 0 {
|
||||
hop.DelayMs, hop.AgeSeconds = selDelay, selAge
|
||||
} else {
|
||||
hop.DelayMs, hop.AgeSeconds = freshDelay, freshAge
|
||||
}
|
||||
return hop
|
||||
}
|
||||
|
||||
// usedChainNames recovers the set of chain NAMES the observatory's used-set covers.
|
||||
//
|
||||
// The used-set is keyed by outbound TAG (the generator's schema, materialised in
|
||||
|
||||
@@ -309,8 +309,167 @@ func TestChainHealthNilUsedSet(t *testing.T) {
|
||||
t.Fatalf("ChainHealth = %+v, want %+v (nil used-set ⇒ all used, blanks dropped)", got, want)
|
||||
}
|
||||
for i, w := range want {
|
||||
if got[i] != w {
|
||||
if got[i].Name != w.Name || got[i].Used != w.Used {
|
||||
t.Errorf("ChainHealth[%d] = %+v, want %+v", i, got[i], w)
|
||||
}
|
||||
// A stopped engine materialised nothing: hops must be empty, and the
|
||||
// contract says empty means "nothing materialised", not "no hops".
|
||||
if len(got[i].Hops) != 0 {
|
||||
t.Errorf("ChainHealth[%d].Hops = %+v, want empty on a stopped engine", i, got[i].Hops)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestChainHopHealthProjection drives the per-hop readout over a 3-hop chain —
|
||||
// node hop L1, group hop L2, group hop L3 (the exit) — with exactly ONE hop's
|
||||
// members dead. The dead state must land on THAT hop's index and nowhere else,
|
||||
// the counters must obey the GroupHealth invariants on every hop, and the exit
|
||||
// flag must sit on the largest index. This is the "which hop died" question the
|
||||
// whole hop surface exists to answer.
|
||||
//
|
||||
// L3 additionally pins the ordered walk's half of the contract: it sits BEHIND
|
||||
// the dead L2, so whatever its stale board records say, it is reported untested
|
||||
// with L2 named — never as a hop of its own with a verdict nobody took.
|
||||
func TestChainHopHealthProjection(t *testing.T) {
|
||||
hist := urltest.NewHistoryStorage()
|
||||
now := time.Now()
|
||||
|
||||
// L1 (node hop wrapper): alive — the prefix up to hop 1 works.
|
||||
hist.StoreURLTestHistory("chain-c-h1", &adapter.URLTestHistory{LastOK: now.Add(-5 * time.Second), Delay: 40})
|
||||
// L2 (group hop): BOTH member copies dead — this is the hop that died.
|
||||
hist.StoreURLTestHistory("chain-c-h2-x", &adapter.URLTestHistory{LastFail: now.Add(-3 * time.Second)})
|
||||
hist.StoreURLTestHistory("chain-c-h2-y", &adapter.URLTestHistory{LastFail: now.Add(-2 * time.Second)})
|
||||
// L3 (group hop, exit): one member alive, one never measured.
|
||||
hist.StoreURLTestHistory("chain-c-h3-a", &adapter.URLTestHistory{LastOK: now.Add(-7 * time.Second), Delay: 200})
|
||||
// chain-c-h3-b: nothing at all.
|
||||
|
||||
pool := []adapter.Outbound{
|
||||
&depOutbound{failingOutbound{tag: "chain-c-h1"}, nil},
|
||||
&depGroup{fakeGroup{
|
||||
tag: "chain-c-h2", kind: C.TypeSelector,
|
||||
all: []string{"chain-c-h2-x", "chain-c-h2-y"},
|
||||
now: "chain-c-h2-x",
|
||||
}, []string{"chain-c-h2-x", "chain-c-h2-y"}},
|
||||
&depGroup{fakeGroup{
|
||||
tag: "chain-c-h3", kind: C.TypeSelector,
|
||||
all: []string{"chain-c-h3-a", "chain-c-h3-b"},
|
||||
now: "chain-c-h3-a",
|
||||
}, []string{"chain-c-h3-a", "chain-c-h3-b"}},
|
||||
// Noise the projection must ignore: a member copy is not a hop, another
|
||||
// chain's wrapper is not this chain's.
|
||||
&depOutbound{failingOutbound{tag: "chain-c-h2-x"}, nil},
|
||||
&depOutbound{failingOutbound{tag: "chain-other-h1"}, nil},
|
||||
}
|
||||
|
||||
hops := chainHopHealthOf(pool, "c", newHealthView(hist, healthTTLFloor, now))
|
||||
if len(hops) != 3 {
|
||||
t.Fatalf("got %d hops, want 3: %+v", len(hops), hops)
|
||||
}
|
||||
|
||||
// Ordered by index, exit on the largest.
|
||||
for i, wantIdx := range []int{1, 2, 3} {
|
||||
if hops[i].Index != wantIdx {
|
||||
t.Fatalf("hops out of order: %+v", hops)
|
||||
}
|
||||
if got, want := hops[i].Exit, wantIdx == 3; got != want {
|
||||
t.Errorf("hop %d Exit = %v, want %v", wantIdx, got, want)
|
||||
}
|
||||
}
|
||||
|
||||
h1, h2, h3 := hops[0], hops[1], hops[2]
|
||||
|
||||
// L1: one measurement, its own numbers, no selection.
|
||||
if h1.Kind != "node" || h1.State != HealthAlive || h1.Total != 1 || h1.Alive != 1 ||
|
||||
h1.DelayMs != 40 || h1.AgeSeconds != 5 || h1.Selected != "" {
|
||||
t.Errorf("h1 = %+v, want an alive node hop with its own 40ms/5s and no selection", h1)
|
||||
}
|
||||
|
||||
// L2: the dead hop. Both members tested, none alive => a POSITIVE dead
|
||||
// finding on exactly this index. The selected member (x) is dead, and a
|
||||
// dead observation carries delay 0 with the failure's age.
|
||||
if h2.Kind != "group" || h2.State != HealthDead {
|
||||
t.Fatalf("h2 = %+v, want the DEAD group hop — this is the answer to 'which hop died'", h2)
|
||||
}
|
||||
if h2.Total != 2 || h2.Tested != 2 || h2.Alive != 0 || h2.Dead != 2 || h2.Untested != 0 {
|
||||
t.Errorf("h2 counters = %+v, want 2 tested / 2 dead", h2)
|
||||
}
|
||||
if h2.Selected != "x" {
|
||||
t.Errorf("h2 Selected = %q, want the node NAME x", h2.Selected)
|
||||
}
|
||||
if h2.DelayMs != 0 || h2.AgeSeconds != 3 {
|
||||
t.Errorf("h2 delay/age = %d/%d, want 0/3 (the selected member's failure observation)", h2.DelayMs, h2.AgeSeconds)
|
||||
}
|
||||
|
||||
// L3: BEHIND the break. Its board still holds an alive record for member "a"
|
||||
// (probed before L2 died), but the prober has not dialled this hop since, so
|
||||
// publishing that record would be a verdict nobody currently holds — and the
|
||||
// symmetric case, a stale FAILURE, would send the operator to fix a hop that
|
||||
// is fine. It reads untested, counters collapsed into Untested, with L2 named.
|
||||
if h3.Kind != "group" || h3.State != HealthUntested {
|
||||
t.Fatalf("h3 = %+v, want an UNTESTED hop: the walk stopped at L2 and never dialled it", h3)
|
||||
}
|
||||
if h3.BlockedBy == nil {
|
||||
t.Fatalf("h3 = %+v, want BlockedBy naming L2 — 'why is this empty' must be answered on the row", h3)
|
||||
}
|
||||
if h3.BlockedBy.Index != 2 || h3.BlockedBy.Tag != "chain-c-h2" {
|
||||
t.Errorf("h3.BlockedBy = %+v, want the dead hop 2 / chain-c-h2", *h3.BlockedBy)
|
||||
}
|
||||
if h3.Total != 2 || h3.Tested != 0 || h3.Alive != 0 || h3.Dead != 0 || h3.Untested != 2 {
|
||||
t.Errorf("h3 counters = %+v, want every member untested (nothing was measured at this hop)", h3)
|
||||
}
|
||||
if h3.DelayMs != 0 || h3.AgeSeconds != -1 {
|
||||
t.Errorf("h3 delay/age = %d/%d, want 0/-1 — there is no observation to age", h3.DelayMs, h3.AgeSeconds)
|
||||
}
|
||||
// The selection is not a measurement: the wrapper really would route through
|
||||
// "a", and that stays true whether or not anything reached this hop.
|
||||
if h3.Selected != "a" {
|
||||
t.Errorf("h3 Selected = %q, want a — a block hides measurements, not configuration", h3.Selected)
|
||||
}
|
||||
// Nothing in FRONT of the break may be touched by it.
|
||||
if h1.BlockedBy != nil || h2.BlockedBy != nil {
|
||||
t.Errorf("BlockedBy leaked onto a measured hop: h1=%+v h2=%+v", h1.BlockedBy, h2.BlockedBy)
|
||||
}
|
||||
|
||||
// The invariants, on every hop, exactly as GroupHealth promises.
|
||||
for _, h := range hops {
|
||||
if h.Tested != h.Alive+h.Dead {
|
||||
t.Errorf("hop %d: Tested = %d, want alive+dead = %d", h.Index, h.Tested, h.Alive+h.Dead)
|
||||
}
|
||||
if h.Alive+h.Dead+h.Untested != h.Total {
|
||||
t.Errorf("hop %d: alive+dead+untested = %d, want total = %d",
|
||||
h.Index, h.Alive+h.Dead+h.Untested, h.Total)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// A group hop whose SELECTION has no measurement shows the freshest ALIVE
|
||||
// member's numbers instead — the shown number must always be an observation
|
||||
// somebody took.
|
||||
func TestChainHopSelectionWithoutMeasurementFallsBack(t *testing.T) {
|
||||
hist := urltest.NewHistoryStorage()
|
||||
now := time.Now()
|
||||
hist.StoreURLTestHistory("chain-c-h1-a", &adapter.URLTestHistory{LastOK: now.Add(-30 * time.Second), Delay: 90})
|
||||
hist.StoreURLTestHistory("chain-c-h1-b", &adapter.URLTestHistory{LastOK: now.Add(-4 * time.Second), Delay: 150})
|
||||
// The selection points at c, which has no observation at all.
|
||||
pool := []adapter.Outbound{
|
||||
&depGroup{fakeGroup{
|
||||
tag: "chain-c-h1", kind: C.TypeSelector,
|
||||
all: []string{"chain-c-h1-a", "chain-c-h1-b", "chain-c-h1-c"},
|
||||
now: "chain-c-h1-c",
|
||||
}, nil},
|
||||
}
|
||||
hops := chainHopHealthOf(pool, "c", newHealthView(hist, healthTTLFloor, now))
|
||||
if len(hops) != 1 {
|
||||
t.Fatalf("got %d hops, want 1", len(hops))
|
||||
}
|
||||
h := hops[0]
|
||||
if h.Selected != "c" {
|
||||
t.Errorf("Selected = %q, want c (the selection is reported even unmeasured)", h.Selected)
|
||||
}
|
||||
if h.DelayMs != 150 || h.AgeSeconds != 4 {
|
||||
t.Errorf("delay/age = %d/%d, want the freshest ALIVE member's 150/4", h.DelayMs, h.AgeSeconds)
|
||||
}
|
||||
if h.State != HealthAlive || h.Alive != 2 || h.Untested != 1 {
|
||||
t.Errorf("hop = %+v, want alive with 2 alive / 1 untested", h)
|
||||
}
|
||||
}
|
||||
+332
-65
@@ -12,33 +12,46 @@ import (
|
||||
"time"
|
||||
|
||||
"github.com/sagernet/sing-box/adapter"
|
||||
"github.com/sagernet/sing-box/common/urltest"
|
||||
C "github.com/sagernet/sing-box/constant"
|
||||
)
|
||||
|
||||
// Exit test — "what am I actually going out through, and how fast" (F2, plan §5.E).
|
||||
// Group/chain test — "ask the observatory to refresh, then report what it
|
||||
// measured, plus the exit address" (F2, plan §5.E; reworked under the one-probe
|
||||
// rule).
|
||||
//
|
||||
// # Why this is not just another node probe
|
||||
// # This file no longer measures latency. On purpose.
|
||||
//
|
||||
// The observatory (observatory.go) answers "which nodes are alive". It cannot
|
||||
// answer the question an operator actually asks after switching a group: *through
|
||||
// which address am I leaving the country right now*. A group is an indirection —
|
||||
// a selector or urltest over many members — so the delay of the group and the
|
||||
// public address it exits from are properties of the CURRENT selection, not of
|
||||
// any node the panel can point at. A CHAIN is the same question over a longer
|
||||
// path: its exit wrapper "chain-<name>-hN" tunnels through every hop, so dialling
|
||||
// it measures the whole L1..Ln path end to end.
|
||||
// It used to: one urltest.URLTest per target, dialled right here. That was the
|
||||
// defect, not a feature. The observatory (observatory.go) probes every path the
|
||||
// routing rules actually use — per-chain member copies, egress-bound group
|
||||
// copies, chain exits — and it is the ONLY thing allowed to dial for health.
|
||||
// A second dial path from this file measured the WRONG thing: pressing "Test"
|
||||
// dialled the base groups directly from the router over the default WAN, a path
|
||||
// no rule routes through, and (worse) URLTest's DialContext touched the group,
|
||||
// arming its own 30-minute probing ticker, which kept writing those false
|
||||
// direct measurements under the members' base tags. For a node that is blocked
|
||||
// on the direct WAN and alive only behind a tunnel, that reading is not stale —
|
||||
// it is FALSE, and it poisoned selection and the panel alike.
|
||||
//
|
||||
// So one test per target (group or chain) measures two things:
|
||||
// So a manual test is now a READ with a refresh request in front of it:
|
||||
//
|
||||
// delay_ms — reusing urltest.URLTest, the SAME primitive the observatory uses
|
||||
// for a node. There is deliberately no second latency mechanism
|
||||
// here: the target is just an outbound, and probing it exercises
|
||||
// exactly the path traffic will take.
|
||||
// exit_ip — a real HTTP request THROUGH the target to a service that echoes
|
||||
// the client address back. Nothing else can produce this number: the
|
||||
// router cannot know its own public address, and the proxy protocol
|
||||
// does not report it.
|
||||
// delay_ms / ok — the observatory's OWN measurement of the target's dial
|
||||
// path. TestGroups rewinds the observatory (RefreshObservatory)
|
||||
// so the numbers are fresh — taken AFTER the button press —
|
||||
// then each target waits for an observation newer than the
|
||||
// run's start instant and reports it verbatim. tested_unix is
|
||||
// the instant that observation was taken, not "now".
|
||||
// exit_ip — a real HTTP request THROUGH the target to a service that
|
||||
// echoes the client address back. This is the ONE connection
|
||||
// this file still opens, and it stays because no probe can
|
||||
// answer it: the router cannot know its own public address,
|
||||
// and the proxy protocol does not report it. It is NOT a
|
||||
// second health probe — it travels the target's own routed
|
||||
// path (the same outbound object the rules dial), and it is
|
||||
// only ever issued for a target that (a) the observatory's
|
||||
// used-set covers and (b) just resolved ALIVE. A target
|
||||
// outside the rules, or one whose path is down, gets an empty
|
||||
// exit_ip — never a connection.
|
||||
//
|
||||
// # The direct-egress trap
|
||||
//
|
||||
@@ -51,16 +64,15 @@ import (
|
||||
// exit_ip instead of somebody else's number.
|
||||
|
||||
const (
|
||||
// groupTestDelayTimeout bounds the latency probe of one group.
|
||||
groupTestDelayTimeout = 5 * time.Second
|
||||
// groupTestExitTimeout bounds the whole exit-address lookup for one group
|
||||
// (connect through the tunnel + TLS + response). Short on purpose: this runs
|
||||
// while a human waits on a panel button, and a slow answer is worth less than a
|
||||
// prompt "could not determine".
|
||||
groupTestExitTimeout = 6 * time.Second
|
||||
// groupTestConcurrency bounds how many groups are tested in parallel. Small:
|
||||
// every one of these opens a real tunnelled connection, and a router with a
|
||||
// handful of groups is the normal case.
|
||||
// groupTestConcurrency bounds how many EXIT-ADDRESS lookups run in parallel.
|
||||
// Small: each one opens a real tunnelled connection, and a router with a
|
||||
// handful of groups is the normal case. The board polling itself is not
|
||||
// bounded by this — it is sleep-and-read, no I/O.
|
||||
groupTestConcurrency = 4
|
||||
// exitBodyLimit caps what is read from the exit-address service. These responses
|
||||
// are a few hundred bytes; anything larger is a hijacked/captive-portal answer
|
||||
@@ -68,19 +80,66 @@ const (
|
||||
exitBodyLimit = 4 << 10
|
||||
)
|
||||
|
||||
// groupTestWaitDeadline / groupTestPollEvery pace the wait for the observatory:
|
||||
// each covered target polls the health board once a second until its measured
|
||||
// tag carries an observation newer than the run's start, giving up after the
|
||||
// deadline. 120s covers a forced pass of a large plan (batches chain
|
||||
// back-to-back during a forced pass, see observatoryTickOnce) with room for the
|
||||
// probe timeouts of a mostly-dead population. Package variables, not constants,
|
||||
// for exactly one reason: the timeout tests must not take two minutes.
|
||||
var (
|
||||
groupTestWaitDeadline = 120 * time.Second
|
||||
groupTestPollEvery = time.Second
|
||||
)
|
||||
|
||||
// The fixed result texts for the targets that are never (or not yet) measured.
|
||||
// They are contract, not decoration — the panel shows them verbatim.
|
||||
const (
|
||||
// groupTestErrNotRouted: the target is outside the observatory's used-set,
|
||||
// so no rule routes through it and nothing measures it. Dialling it anyway
|
||||
// would recreate the false direct measurement this rework removed.
|
||||
groupTestErrNotRouted = "not routed by any enabled rule, so nothing measures it — the observatory only probes paths the rules use"
|
||||
// groupTestErrNotReached: the deadline passed without a fresh observation.
|
||||
groupTestErrNotReached = "the observatory has not reached this target yet — it refreshes on the global probe interval"
|
||||
// groupTestErrProbingOff: the observatory is disabled (the GroupHealth
|
||||
// master switch), so a refresh request has nothing to wake and waiting for
|
||||
// the deadline would just delay the same answer by two minutes.
|
||||
groupTestErrProbingOff = "background probing is disabled, so there is nothing to measure this target with"
|
||||
// groupTestErrPathDead: the fresh observation exists and it is a FAILURE —
|
||||
// the observatory probed the target's path after the button press and the
|
||||
// path did not answer. An honest negative, not a missing measurement.
|
||||
groupTestErrPathDead = "the observatory's probe through this path failed"
|
||||
// groupTestErrChainBlockedFmt: a CHAIN whose exit was never dialled because
|
||||
// an earlier hop was probed and did not answer. The prober walks a chain in
|
||||
// order and stops at the first dead hop, so there is no end-to-end
|
||||
// measurement to wait for and there never will be while that hop is down.
|
||||
// Reported at once, naming the hop, instead of spending the full deadline to
|
||||
// answer "not reached yet" about a path that is known to be broken — and
|
||||
// naming it is the whole value: the operator's next action is at that hop.
|
||||
groupTestErrChainBlockedFmt = "hop %d of this chain was probed and did not answer, so nothing reaches the exit through it — fix that hop first"
|
||||
)
|
||||
|
||||
// GroupTestResult is one target's test outcome — a group's or a chain's (Group
|
||||
// then carries the chain's model name). The JSON tags are the panel contract —
|
||||
// see the shater API docs for /api/groups/test.
|
||||
// see the shater API docs for /api/groups/test. The field names and types are
|
||||
// FROZEN; what changed in the rework is where the numbers come from.
|
||||
//
|
||||
// DelayMs and OK are a READ of the observatory's measurement of the target's
|
||||
// dial path — not a fresh dial performed by this file. OK is true exactly when
|
||||
// the health board's state for the measured tag is alive; DelayMs is that
|
||||
// observation's RTT; TestedUnix is the instant the OBSERVATION was taken (now
|
||||
// minus its age), so a result honestly says how old its number is instead of
|
||||
// stamping the poll time over it.
|
||||
//
|
||||
// Selected is the group's current pick (OutboundGroup.Now()); for a chain it is
|
||||
// the node NAME the chain's last group hop currently selects, "" when the chain
|
||||
// has no group hop (a fixed path selects nothing).
|
||||
//
|
||||
// OK reports whether the LATENCY measurement succeeded, which is the test's primary
|
||||
// question. A failed exit-address lookup deliberately does NOT clear it: knowing the
|
||||
// target is up and fast is useful on its own, and a probe service being unreachable
|
||||
// says nothing about the tunnel. In that case OK stays true and ExitIP is empty —
|
||||
// "not determined", never a guess and never somebody else's address.
|
||||
// A failed exit-address lookup deliberately does NOT clear OK: knowing the
|
||||
// target is up and fast is useful on its own, and a probe service being
|
||||
// unreachable says nothing about the tunnel. In that case OK stays true and
|
||||
// ExitIP is empty — "not determined", never a guess and never somebody else's
|
||||
// address.
|
||||
type GroupTestResult struct {
|
||||
Group string `json:"group"`
|
||||
Selected string `json:"selected"`
|
||||
@@ -117,6 +176,13 @@ type groupTestTarget struct {
|
||||
name string
|
||||
ob adapter.Outbound
|
||||
sel func() string
|
||||
// isChain marks a CHAIN target. A chain is the one target whose measurement
|
||||
// can be legitimately absent while the target is known to be broken: the
|
||||
// prober walks a chain in order and stops at the first dead hop, so a chain
|
||||
// whose hop 2 died never has its exit dialled at all. Without this flag a run
|
||||
// would sit out its whole deadline and then report "not reached yet" about a
|
||||
// path it knows is down — see testOneTarget.
|
||||
isChain bool
|
||||
}
|
||||
|
||||
// chainTargetsFrom discovers each materialised chain in the running
|
||||
@@ -161,7 +227,7 @@ func chainTargetsFrom(pool []adapter.Outbound) []groupTestTarget {
|
||||
if !ok {
|
||||
continue
|
||||
}
|
||||
t := groupTestTarget{name: name, ob: ob}
|
||||
t := groupTestTarget{name: name, ob: ob, isChain: true}
|
||||
if hop := chainLastGroupHop(byTag, name); hop != nil {
|
||||
t.sel = func() string { return chainMemberName(hop.Now(), name) }
|
||||
}
|
||||
@@ -175,18 +241,38 @@ func chainTargetsFrom(pool []adapter.Outbound) []groupTestTarget {
|
||||
// "chain-<name>-h<digits>". Member copies ("…-h<i>-<member>") and non-chain tags
|
||||
// parse false.
|
||||
func parseChainExitTag(tag string) (string, bool) {
|
||||
name, _, ok := parseChainHopTag(tag)
|
||||
return name, ok
|
||||
}
|
||||
|
||||
// parseChainHopTag is parseChainExitTag plus the number: it recovers the chain
|
||||
// NAME and the 1-based HOP INDEX from a hop wrapper tag "chain-<name>-h<digits>"
|
||||
// (generate/chain.go buildHopWrapper). Member copies ("…-h<i>-<member>") and
|
||||
// non-chain tags parse false.
|
||||
//
|
||||
// The index is what lets the probe plan ORDER a chain's measurements. A chain is
|
||||
// a series path — hop i is dialled through hops 1..i-1 — so the difference
|
||||
// between "this hop failed" and "the hop in front of it failed" is the whole
|
||||
// diagnosis, and it cannot be recovered from an unordered set of tags.
|
||||
//
|
||||
// The split takes the LAST "-h<digits>", so a chain literally named "a-h2" is
|
||||
// read as chain "a-h2" hop 1 rather than chain "a" hop 2 — the same documented
|
||||
// reserved-namespace edge case parseGroupCopyTag accepts, pinned by
|
||||
// TestParseChainExitTag.
|
||||
func parseChainHopTag(tag string) (name string, hop int, ok bool) {
|
||||
rest, ok := strings.CutPrefix(tag, "chain-")
|
||||
if !ok {
|
||||
return "", false
|
||||
return "", 0, false
|
||||
}
|
||||
i := strings.LastIndex(rest, "-h")
|
||||
if i <= 0 {
|
||||
return "", false
|
||||
return "", 0, false
|
||||
}
|
||||
if _, ok := parseAllDigits(rest[i+2:]); !ok {
|
||||
return "", false
|
||||
hop, ok = parseAllDigits(rest[i+2:])
|
||||
if !ok {
|
||||
return "", 0, false
|
||||
}
|
||||
return rest[:i], true
|
||||
return rest[:i], hop, true
|
||||
}
|
||||
|
||||
// chainLastGroupHop finds the chain's LAST group hop — the wrapper selector
|
||||
@@ -259,19 +345,30 @@ func parseAllDigits(s string) (int, bool) {
|
||||
// and chains (empty/nil = every group and every chain in the running box),
|
||||
// returning started=false when a run is already in flight.
|
||||
//
|
||||
// It is a SINGLETON: a second request while a run is in flight is refused rather
|
||||
// than queued or run in parallel, because these runs open real tunnelled
|
||||
// connections and a panel that double-fires a button must not multiply the load
|
||||
// on the uplink. The observatory checks the same guard and skips its tick while
|
||||
// a run is in flight (observatory.go), so a manual test never competes with
|
||||
// background probing for the uplink.
|
||||
// It does NOT probe. It records the run's start instant, asks the observatory
|
||||
// for an out-of-turn full pass (RefreshObservatory), and then each target waits
|
||||
// for the health board to carry an observation NEWER than that instant on the
|
||||
// tag the observatory actually measures for it — see testOneTarget. The only
|
||||
// connection a run may still open is the exit-address lookup, and only for a
|
||||
// target that resolved alive on a rule-routed path.
|
||||
//
|
||||
// probeURL is the latency-probe URL; "" falls back to urltest's gstatic default.
|
||||
// It is a SINGLETON: a second request while a run is in flight is refused
|
||||
// rather than queued, because two overlapping runs would each rewind the
|
||||
// observatory's cursor and neither pass would ever complete — and the panel's
|
||||
// progress contract assumes one run's counters at a time anyway.
|
||||
//
|
||||
// Apply-swap safety: the target outbounds are snapshotted up front, so a config swap
|
||||
// mid-run cannot change what is being tested. A target torn down mid-run simply fails
|
||||
// its probe and is reported not-ok.
|
||||
// probeURL is accepted and IGNORED. The probe URL is a global observatory
|
||||
// setting now (ObservatoryConfig.ProbeURL, set at apply time); a per-run URL
|
||||
// would mean this run measures something different from what the board holds,
|
||||
// which is exactly the two-instruments split the rework removed. The parameter
|
||||
// stays so the callers (shater/apply, shater/panel) keep compiling and the
|
||||
// control-plane API shape does not churn.
|
||||
//
|
||||
// Apply-swap safety: the target outbounds are snapshotted up front, so a config
|
||||
// swap mid-run cannot change what is being tested. A target torn down mid-run
|
||||
// simply never receives a fresh observation and resolves on the deadline.
|
||||
func (e *Engine) TestGroups(names []string, probeURL string) (started bool) {
|
||||
_ = probeURL // ignored — see the doc comment above
|
||||
if !e.groupTestRunning.CompareAndSwap(false, true) {
|
||||
return false
|
||||
}
|
||||
@@ -285,6 +382,13 @@ func (e *Engine) TestGroups(names []string, probeURL string) (started bool) {
|
||||
// GroupTestStatus for why the scope is a set of names.
|
||||
e.setGroupTestScope(scopeOf(targets, missing))
|
||||
|
||||
// The freshness watermark: only an observation taken AFTER this instant may
|
||||
// answer this run. Recorded BEFORE the refresh request so a probe that lands
|
||||
// between the two can never be missed, only double-counted as fresh — the
|
||||
// harmless direction.
|
||||
t0 := time.Now()
|
||||
e.RefreshObservatory()
|
||||
|
||||
go func() {
|
||||
defer e.groupTestRunning.Store(false)
|
||||
|
||||
@@ -303,15 +407,22 @@ func (e *Engine) TestGroups(names []string, probeURL string) (started bool) {
|
||||
e.setGroupTestResults(results)
|
||||
e.groupTestDone.Add(int64(len(missing)))
|
||||
|
||||
// The used-set and the enabled bit are snapshotted once for the whole
|
||||
// run: they only change on an apply, and a run that straddles an apply
|
||||
// is already best-effort (see the swap-safety note above).
|
||||
used := e.observatoryUsed()
|
||||
obsEnabled, _, _ := e.ObservatoryStatus()
|
||||
|
||||
// One goroutine per target: they spend their life sleeping on the board
|
||||
// poll, so there is nothing to bound — the semaphore below bounds the
|
||||
// exit-address lookups, the only real connections left in a run.
|
||||
sem := make(chan struct{}, groupTestConcurrency)
|
||||
var wg sync.WaitGroup
|
||||
for i, tgt := range targets {
|
||||
wg.Add(1)
|
||||
sem <- struct{}{}
|
||||
go func(i int, tgt groupTestTarget) {
|
||||
defer wg.Done()
|
||||
defer func() { <-sem }()
|
||||
res := e.testOneTarget(tgt, probeURL)
|
||||
res := e.testOneTarget(tgt, used, obsEnabled, t0, sem)
|
||||
e.storeGroupTestResult(i, res)
|
||||
e.groupTestDone.Add(1)
|
||||
}(i, tgt)
|
||||
@@ -393,29 +504,185 @@ func (e *Engine) groupTargets(names []string) (targets []groupTestTarget, missin
|
||||
return targets, missing
|
||||
}
|
||||
|
||||
// testOneTarget measures one target: what it currently selects, the latency
|
||||
// through it, and the public address it exits from. For a chain the dialled
|
||||
// outbound is the exit wrapper, so the delay and the exit address are end-to-end
|
||||
// properties of the whole L1..Ln path.
|
||||
func (e *Engine) testOneTarget(t groupTestTarget, probeURL string) GroupTestResult {
|
||||
// testOneTarget resolves one target WITHOUT probing it: it decides whether the
|
||||
// observatory measures this target at all, and if so waits for a fresh
|
||||
// observation and reports it. The three ways out, in order:
|
||||
//
|
||||
// 1. the used-set does not cover the target — no enabled rule routes through
|
||||
// it, so nothing measures it and nothing SHOULD: resolved immediately with
|
||||
// groupTestErrNotRouted, never dialled, empty exit address. A nil used-set
|
||||
// means "unknown" (observatory not yet configured) and is treated as
|
||||
// covered — no refusal is better than a wrong one;
|
||||
// 2. the observatory is disabled — the refresh request went nowhere, so the
|
||||
// wait below could only ever end on its deadline: resolved immediately
|
||||
// with groupTestErrProbingOff instead of stalling the panel for two
|
||||
// minutes to say the same thing;
|
||||
// 3. covered and enabled — poll the health board once a second until the
|
||||
// MEASURED TAG carries an observation newer than since, then report that
|
||||
// observation verbatim (readFreshObservation). On the deadline:
|
||||
// groupTestErrNotReached.
|
||||
//
|
||||
// The exit-address lookup runs ONLY on the alive path of (3) — a rule-routed
|
||||
// target whose path just answered a probe — and through sem, so a run never
|
||||
// opens more than groupTestConcurrency tunnelled connections at once.
|
||||
func (e *Engine) testOneTarget(t groupTestTarget, used map[string]bool, obsEnabled bool, since time.Time, sem chan struct{}) GroupTestResult {
|
||||
res := GroupTestResult{Group: t.name, TestedUnix: time.Now().Unix()}
|
||||
if t.sel != nil {
|
||||
res.Selected = t.sel()
|
||||
}
|
||||
|
||||
ctx, cancel := context.WithTimeout(context.Background(), groupTestDelayTimeout)
|
||||
delay, err := urltest.URLTest(ctx, probeURL, t.ob)
|
||||
cancel()
|
||||
if err != nil {
|
||||
res.Error = err.Error()
|
||||
if used != nil && !used[t.ob.Tag()] {
|
||||
res.Error = groupTestErrNotRouted
|
||||
return res
|
||||
}
|
||||
if !obsEnabled {
|
||||
res.Error = groupTestErrProbingOff
|
||||
return res
|
||||
}
|
||||
res.OK = true
|
||||
res.DelayMs = int(delay)
|
||||
|
||||
// The exit address is best-effort by design: see GroupTestResult.OK.
|
||||
res.ExitIP, res.ExitCountry = e.exitAddress(t.ob)
|
||||
return res
|
||||
deadline := time.Now().Add(groupTestWaitDeadline)
|
||||
for {
|
||||
// Asked BEFORE the board read, and only for a chain. The exit of a blocked
|
||||
// chain does carry a fresh dead verdict now — the observatory derives it
|
||||
// from the broken hop (markChainExitDead) so selection cannot be tempted
|
||||
// down a path that provably does not work — but "the probe through this
|
||||
// path failed" would be the wrong sentence to show a person: no probe of
|
||||
// the exit ran, and the actionable fact is WHICH hop stopped it. Same
|
||||
// verdict, better answer.
|
||||
if idx, ok := e.blockedChainHop(t, since); ok {
|
||||
res.OK = false
|
||||
res.Error = fmt.Sprintf(groupTestErrChainBlockedFmt, idx)
|
||||
return res
|
||||
}
|
||||
if e.readFreshObservation(&res, t, since) {
|
||||
if res.OK {
|
||||
// The exit address is best-effort by design: see GroupTestResult.
|
||||
sem <- struct{}{}
|
||||
res.ExitIP, res.ExitCountry = e.exitAddress(t.ob)
|
||||
<-sem
|
||||
}
|
||||
return res
|
||||
}
|
||||
if !time.Now().Before(deadline) {
|
||||
res.Error = groupTestErrNotReached
|
||||
return res
|
||||
}
|
||||
time.Sleep(groupTestPollEvery)
|
||||
}
|
||||
}
|
||||
|
||||
// blockedChainHop reports the hop that stopped this CHAIN target's ordered walk
|
||||
// before its exit, and only when the finding is FRESH enough to answer this run.
|
||||
//
|
||||
// Both halves matter. The projection (chainHopHealthOf) already computes the
|
||||
// break exactly once, and reading it here rather than re-deriving it keeps the
|
||||
// run and the Targets card from ever naming different hops. And the freshness
|
||||
// check is the same watermark readFreshObservation uses: a run rewinds the
|
||||
// observatory and then reports what it measures AFTERWARDS, so a break left over
|
||||
// from before the button press must not short-circuit the wait — the forced pass
|
||||
// may be about to find that hop alive again.
|
||||
//
|
||||
// Non-chain targets, unmaterialised chains and chains with no break all report
|
||||
// false, and the caller goes on waiting exactly as before.
|
||||
func (e *Engine) blockedChainHop(t groupTestTarget, since time.Time) (int, bool) {
|
||||
if !t.isChain {
|
||||
return 0, false
|
||||
}
|
||||
return blockingHopOf(chainHopHealthOf(e.runningPool(), t.name, e.HealthView()), time.Now(), since)
|
||||
}
|
||||
|
||||
// blockingHopOf is the decision itself, split out so it is testable without a
|
||||
// box: given one chain's hop readout, the hop whose failure kept the EXIT from
|
||||
// being dialled, and only when that failure was observed at or after since.
|
||||
//
|
||||
// The exit's own BlockedBy is what is consulted, not "any dead hop": a chain
|
||||
// whose exit was itself probed and failed has a real end-to-end measurement, and
|
||||
// that measurement is the honest answer (groupTestErrPathDead). Only when the
|
||||
// exit was never reached is there nothing to wait for.
|
||||
func blockingHopOf(hops []ChainHopHealth, now, since time.Time) (int, bool) {
|
||||
if len(hops) == 0 {
|
||||
return 0, false
|
||||
}
|
||||
blocker := hops[len(hops)-1].BlockedBy
|
||||
if blocker == nil {
|
||||
return 0, false
|
||||
}
|
||||
for _, h := range hops {
|
||||
if h.Index != blocker.Index {
|
||||
continue
|
||||
}
|
||||
if h.AgeSeconds < 0 {
|
||||
return 0, false
|
||||
}
|
||||
if now.Add(-time.Duration(h.AgeSeconds) * time.Second).Before(since) {
|
||||
return 0, false // a break observed before this run started; keep waiting
|
||||
}
|
||||
return h.Index, true
|
||||
}
|
||||
return 0, false
|
||||
}
|
||||
|
||||
// measuredTagOf is the tag the observatory actually probes for this target's
|
||||
// dial path — the tag whose board entry answers "how is this target doing":
|
||||
//
|
||||
// - a GROUP target dials whatever it currently selects, so the group's health
|
||||
// IS its selection's health: OutboundGroup.Now(). For an egress-bound group
|
||||
// that is a per-group copy tag, which is exactly what the plan probes;
|
||||
// - a CHAIN whose exit wrapper is a PLAIN outbound is probed end-to-end under
|
||||
// that wrapper tag (probeplan.go: a chain exit is its own measurement);
|
||||
// - a CHAIN whose exit wrapper is a GROUP (the last hop is a group) has its
|
||||
// member copies probed instead of the wrapper, so the wrapper's Now() — the
|
||||
// member copy the chain currently dials through — is the measured tag. The
|
||||
// wrapper here IS the last group hop, so this is chainLastGroupHop's Now()
|
||||
// without a second lookup;
|
||||
// - fallback: the target's own tag, for a group that has not selected yet
|
||||
// (cold selector mid-swap). Its board entry is almost certainly empty, and
|
||||
// the caller then honestly reports "not reached" rather than inventing one.
|
||||
func measuredTagOf(t groupTestTarget) string {
|
||||
if g, ok := t.ob.(adapter.OutboundGroup); ok {
|
||||
if now := g.Now(); now != "" {
|
||||
return now
|
||||
}
|
||||
}
|
||||
return t.ob.Tag()
|
||||
}
|
||||
|
||||
// readFreshObservation reads the board once: if the target's measured tag holds
|
||||
// an observation taken at or after since, it is written into res (state, delay,
|
||||
// the observation's own timestamp, and the current selection so Selected and
|
||||
// the measurement describe the same pick) and true is returned. Otherwise res
|
||||
// is left for the next poll.
|
||||
//
|
||||
// Precision note: the board reports ages in whole seconds, so "at or after
|
||||
// since" is accurate to one second — an observation taken up to a second
|
||||
// BEFORE the refresh can slip through as fresh. That is the acceptable
|
||||
// direction: it is still a real measurement of the same path, at most a second
|
||||
// older than requested; the strict direction (discarding genuinely fresh
|
||||
// observations) would make every run one probe interval slower for nothing.
|
||||
func (e *Engine) readFreshObservation(res *GroupTestResult, t groupTestTarget, since time.Time) bool {
|
||||
// Re-read the selection at every poll: a forced observatory pass is exactly
|
||||
// the kind of event that makes a urltest group switch members, and the
|
||||
// measurement below is taken against the CURRENT pick.
|
||||
if t.sel != nil {
|
||||
res.Selected = t.sel()
|
||||
}
|
||||
tag := measuredTagOf(t)
|
||||
now := time.Now()
|
||||
state, delayMs, age := e.HealthView().State(tag)
|
||||
if age < 0 {
|
||||
return false // no observation at all (untested)
|
||||
}
|
||||
observedAt := now.Add(-time.Duration(age) * time.Second)
|
||||
if observedAt.Before(since) {
|
||||
return false // an old reading; the refresh has not reached this tag yet
|
||||
}
|
||||
res.OK = state == HealthAlive
|
||||
res.DelayMs = delayMs
|
||||
res.TestedUnix = observedAt.Unix()
|
||||
if !res.OK {
|
||||
res.Error = groupTestErrPathDead
|
||||
}
|
||||
return true
|
||||
}
|
||||
|
||||
// exitProbe is one exit-address service: a URL and the parser for its body.
|
||||
|
||||
+196
-15
@@ -3,6 +3,7 @@ package engine
|
||||
import (
|
||||
"context"
|
||||
"crypto/tls"
|
||||
"fmt"
|
||||
"net"
|
||||
"net/http"
|
||||
"net/http/httptest"
|
||||
@@ -143,10 +144,10 @@ func TestChainMemberName(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
// dialableOutbound routes every dial to a fixed local address — a stand-in for a
|
||||
// chain exit whose whole path is up. urltest.URLTest dials the probe URL's host
|
||||
// through the outbound, so pointing every dial at a local HTTP server makes the
|
||||
// latency probe succeed without any network.
|
||||
// dialableOutbound routes every dial to a fixed local address — the sink the
|
||||
// exit-address lookup lands in during tests. Since the rework nothing else in
|
||||
// this file dials at all; the local server refuses to be an exit-address
|
||||
// service (wrong status, no TLS), so the lookup honestly comes back empty.
|
||||
type dialableOutbound struct {
|
||||
failingOutbound
|
||||
addr string
|
||||
@@ -158,20 +159,42 @@ func (d *dialableOutbound) DialContext(ctx context.Context, network string, _ M.
|
||||
return (&net.Dialer{}).DialContext(ctx, network, d.addr)
|
||||
}
|
||||
|
||||
// TestChainExitTestMeasuresEndToEnd is the §6-S4 acceptance path for chains: the
|
||||
// exit test dials the chain's EXIT TAG, returns a measured delay, carries the
|
||||
// chain's model name (not the wrapper tag) as the result's Group, and reports the
|
||||
// last group hop's pick as Selected. The exit address is measured through the
|
||||
// same outbound (unreachable from a test => empty, never a guess).
|
||||
func TestChainExitTestMeasuresEndToEnd(t *testing.T) {
|
||||
probe := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
||||
// countingOutbound counts every DialContext/ListenPacket. It is the tripwire of
|
||||
// the rework: a target the observatory does not cover must NEVER be dialled,
|
||||
// and the counter is the proof.
|
||||
type countingOutbound struct {
|
||||
failingOutbound
|
||||
dials int
|
||||
}
|
||||
|
||||
func (c *countingOutbound) DialContext(ctx context.Context, network string, dst M.Socksaddr) (net.Conn, error) {
|
||||
c.dials++
|
||||
return nil, fmt.Errorf("dial refused (test)")
|
||||
}
|
||||
|
||||
func (c *countingOutbound) ListenPacket(ctx context.Context, dst M.Socksaddr) (net.PacketConn, error) {
|
||||
c.dials++
|
||||
return nil, fmt.Errorf("dial refused (test)")
|
||||
}
|
||||
|
||||
// groupTestSem is a fresh exit-lookup semaphore for direct testOneTarget calls.
|
||||
func groupTestSem() chan struct{} { return make(chan struct{}, groupTestConcurrency) }
|
||||
|
||||
// TestChainTestReportsObservatoryMeasurement is the §6-S4 acceptance path for
|
||||
// chains under the one-probe rule: the manual test does NOT dial the exit — it
|
||||
// reads the observatory's board entry for the exit tag, reports its delay and
|
||||
// ITS timestamp, carries the chain's model name as Group and the last group
|
||||
// hop's pick as Selected. The exit-address lookup still travels the exit
|
||||
// outbound (unreachable from a test => empty, never a guess).
|
||||
func TestChainTestReportsObservatoryMeasurement(t *testing.T) {
|
||||
sink := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
||||
w.WriteHeader(http.StatusNoContent)
|
||||
}))
|
||||
defer probe.Close()
|
||||
defer sink.Close()
|
||||
|
||||
exit := &dialableOutbound{
|
||||
failingOutbound: failingOutbound{tag: "chain-x-h2"},
|
||||
addr: probe.Listener.Addr().String(),
|
||||
addr: sink.Listener.Addr().String(),
|
||||
deps: []string{"chain-x-h1"},
|
||||
}
|
||||
pool := []adapter.Outbound{
|
||||
@@ -187,14 +210,31 @@ func TestChainExitTestMeasuresEndToEnd(t *testing.T) {
|
||||
if len(targets) != 1 || targets[0].ob.Tag() != "chain-x-h2" {
|
||||
t.Fatalf("targets = %+v, want the chain dialled via its exit tag", targets)
|
||||
}
|
||||
if got := measuredTagOf(targets[0]); got != "chain-x-h2" {
|
||||
t.Fatalf("measuredTagOf = %q, want the plain exit wrapper itself", got)
|
||||
}
|
||||
|
||||
e := New()
|
||||
res := e.testOneTarget(targets[0], probe.URL)
|
||||
// The observatory measured the exit 10 seconds ago, AFTER the run's start
|
||||
// instant below: that observation — delay and timestamp both — is the answer.
|
||||
observedAt := time.Now().Add(-10 * time.Second)
|
||||
e.URLTestHistory().StoreURLTestHistory("chain-x-h2", &adapter.URLTestHistory{LastOK: observedAt, Delay: 77})
|
||||
since := time.Now().Add(-time.Minute)
|
||||
|
||||
res := e.testOneTarget(targets[0], map[string]bool{"chain-x-h2": true}, true, since, groupTestSem())
|
||||
if res.Group != "x" {
|
||||
t.Errorf("Group = %q, want the chain's model name x", res.Group)
|
||||
}
|
||||
if !res.OK || res.Error != "" {
|
||||
t.Fatalf("result = %+v, want a successful measurement through the exit tag (a local roundtrip may legitimately read 0ms)", res)
|
||||
t.Fatalf("result = %+v, want the board's alive observation reported", res)
|
||||
}
|
||||
if res.DelayMs != 77 {
|
||||
t.Errorf("DelayMs = %d, want the observatory's 77 — this file measures nothing itself", res.DelayMs)
|
||||
}
|
||||
// tested_unix is the OBSERVATION's instant (whole-second precision), never
|
||||
// the poll's.
|
||||
if got, want := res.TestedUnix, observedAt.Unix(); got < want-1 || got > want+1 {
|
||||
t.Errorf("TestedUnix = %d, want the observation's instant ~%d", got, want)
|
||||
}
|
||||
if res.Selected != "relay" {
|
||||
t.Errorf("Selected = %q, want the last group hop's pick relay", res.Selected)
|
||||
@@ -204,6 +244,93 @@ func TestChainExitTestMeasuresEndToEnd(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
// TestTestOneTargetUnroutedNeverDialled is the rework's core promise: a target
|
||||
// outside the observatory's used-set is resolved immediately with the
|
||||
// documented explanation and ZERO dials — no latency probe, no exit-address
|
||||
// lookup, nothing. Dialling it would manufacture exactly the false direct
|
||||
// measurement the rework removed.
|
||||
func TestTestOneTargetUnroutedNeverDialled(t *testing.T) {
|
||||
e := New()
|
||||
ob := &countingOutbound{failingOutbound: failingOutbound{tag: "idle"}}
|
||||
tgt := groupTestTarget{name: "idle", ob: ob}
|
||||
|
||||
res := e.testOneTarget(tgt, map[string]bool{"something-else": true}, true, time.Now(), groupTestSem())
|
||||
if ob.dials != 0 {
|
||||
t.Fatalf("an unrouted target was dialled %d time(s); it must never be", ob.dials)
|
||||
}
|
||||
if res.OK {
|
||||
t.Fatal("an unrouted target reported ok=true")
|
||||
}
|
||||
if res.Error != groupTestErrNotRouted {
|
||||
t.Fatalf("Error = %q, want the not-routed explanation", res.Error)
|
||||
}
|
||||
if res.ExitIP != "" || res.ExitCountry != "" {
|
||||
t.Fatalf("an unrouted target must carry no exit address: %+v", res)
|
||||
}
|
||||
}
|
||||
|
||||
// A disabled observatory resolves covered targets immediately too: the refresh
|
||||
// request went nowhere, so waiting out the deadline would only delay the same
|
||||
// honest answer — and still, nothing is dialled.
|
||||
func TestTestOneTargetProbingOffNeverDialled(t *testing.T) {
|
||||
e := New()
|
||||
ob := &countingOutbound{failingOutbound: failingOutbound{tag: "auto"}}
|
||||
tgt := groupTestTarget{name: "auto", ob: ob}
|
||||
|
||||
res := e.testOneTarget(tgt, nil, false, time.Now(), groupTestSem())
|
||||
if ob.dials != 0 {
|
||||
t.Fatalf("target dialled %d time(s) with probing off; want 0", ob.dials)
|
||||
}
|
||||
if res.OK || res.Error != groupTestErrProbingOff {
|
||||
t.Fatalf("result = %+v, want ok=false with the probing-off explanation", res)
|
||||
}
|
||||
}
|
||||
|
||||
// The deadline path: a covered target whose measured tag never receives a fresh
|
||||
// observation resolves with the not-reached explanation — and, again, without a
|
||||
// single dial of its own.
|
||||
func TestTestOneTargetDeadlineWithoutObservation(t *testing.T) {
|
||||
// Shrink the wait so the test answers in milliseconds; restore afterwards.
|
||||
oldDeadline, oldPoll := groupTestWaitDeadline, groupTestPollEvery
|
||||
groupTestWaitDeadline, groupTestPollEvery = 30*time.Millisecond, 5*time.Millisecond
|
||||
defer func() { groupTestWaitDeadline, groupTestPollEvery = oldDeadline, oldPoll }()
|
||||
|
||||
e := New()
|
||||
ob := &countingOutbound{failingOutbound: failingOutbound{tag: "auto"}}
|
||||
tgt := groupTestTarget{name: "auto", ob: ob}
|
||||
|
||||
res := e.testOneTarget(tgt, map[string]bool{"auto": true}, true, time.Now(), groupTestSem())
|
||||
if ob.dials != 0 {
|
||||
t.Fatalf("target dialled %d time(s) while waiting on the board; want 0", ob.dials)
|
||||
}
|
||||
if res.OK || res.Error != groupTestErrNotReached {
|
||||
t.Fatalf("result = %+v, want ok=false with the not-reached explanation", res)
|
||||
}
|
||||
}
|
||||
|
||||
// A fresh DEAD observation is an answer, not a timeout: ok=false with the
|
||||
// path-dead explanation, delay 0, and no exit-address connection for a path
|
||||
// that just failed its probe.
|
||||
func TestTestOneTargetReportsFreshDeath(t *testing.T) {
|
||||
e := New()
|
||||
ob := &countingOutbound{failingOutbound: failingOutbound{tag: "auto"}}
|
||||
tgt := groupTestTarget{name: "auto", ob: ob}
|
||||
|
||||
since := time.Now().Add(-time.Minute)
|
||||
e.URLTestHistory().MarkFailed("auto")
|
||||
|
||||
res := e.testOneTarget(tgt, map[string]bool{"auto": true}, true, since, groupTestSem())
|
||||
if ob.dials != 0 {
|
||||
t.Fatalf("a dead target was dialled %d time(s); the exit lookup is for alive paths only", ob.dials)
|
||||
}
|
||||
if res.OK || res.Error != groupTestErrPathDead {
|
||||
t.Fatalf("result = %+v, want ok=false with the path-dead explanation", res)
|
||||
}
|
||||
if res.DelayMs != 0 || res.ExitIP != "" {
|
||||
t.Fatalf("a dead path must carry no delay and no exit address: %+v", res)
|
||||
}
|
||||
}
|
||||
|
||||
// TestGroupTestSingleton is the "no parallel runs" invariant: while a run holds the
|
||||
// guard, a second TestGroups must be refused rather than starting a concurrent run
|
||||
// (each of these opens real tunnelled connections).
|
||||
@@ -460,3 +587,57 @@ func TestGroupTestScopeDedupedAndSorted(t *testing.T) {
|
||||
t.Fatalf("scope = %v, want [alpha zeta]", scope)
|
||||
}
|
||||
}
|
||||
|
||||
// A manual run against a chain whose walk is blocked must say so, at once and by
|
||||
// hop number. The prober stops at the first dead hop, so such a chain's exit is
|
||||
// never dialled and no end-to-end observation is ever coming: waiting for one
|
||||
// would burn the full 120s deadline and then answer "the observatory has not
|
||||
// reached this target yet" — which reads like a timing artefact about a chain
|
||||
// that is, in fact, down and whose broken hop is already known.
|
||||
//
|
||||
// The freshness rule is the same one readFreshObservation uses: a run rewinds
|
||||
// the observatory and reports what it measures AFTERWARDS, so a break left over
|
||||
// from before the button press must not short-circuit the wait — the forced pass
|
||||
// may be about to find that hop alive again.
|
||||
func TestBlockingHopOfChainWalk(t *testing.T) {
|
||||
since := time.Now()
|
||||
// A 3-hop readout as the projection builds it: hop 2 dead, hop 3 not reached.
|
||||
blocked := func(ageOfBreak int64) []ChainHopHealth {
|
||||
blocker := &ChainHopBlock{Index: 2, Tag: "chain-c-h2"}
|
||||
return []ChainHopHealth{
|
||||
{Index: 1, Tag: "chain-c-h1", State: HealthAlive, AgeSeconds: 1},
|
||||
{Index: 2, Tag: "chain-c-h2", State: HealthDead, AgeSeconds: ageOfBreak},
|
||||
{Index: 3, Tag: "chain-c-h3", State: HealthUntested, AgeSeconds: -1, Exit: true, BlockedBy: blocker},
|
||||
}
|
||||
}
|
||||
|
||||
// Measured after the run started: answer now, naming hop 2.
|
||||
if idx, ok := blockingHopOf(blocked(0), since.Add(time.Second), since); !ok || idx != 2 {
|
||||
t.Errorf("fresh break = (%d,%v), want (2,true) — the run must name the hop instead of timing out", idx, ok)
|
||||
}
|
||||
|
||||
// Measured BEFORE the run started: keep waiting. The forced pass has not
|
||||
// re-probed that hop yet, and it may be about to answer.
|
||||
if idx, ok := blockingHopOf(blocked(30), since.Add(time.Second), since); ok {
|
||||
t.Errorf("stale break reported as this run's finding (hop %d); a run answers with what IT measured", idx)
|
||||
}
|
||||
|
||||
// A chain whose exit WAS reached has a real end-to-end reading; the ordinary
|
||||
// path reports it (alive, or groupTestErrPathDead). Nothing to short-circuit.
|
||||
reached := []ChainHopHealth{
|
||||
{Index: 1, Tag: "chain-c-h1", State: HealthAlive, AgeSeconds: 1},
|
||||
{Index: 2, Tag: "chain-c-h2", State: HealthDead, AgeSeconds: 0, Exit: true},
|
||||
}
|
||||
if _, ok := blockingHopOf(reached, since.Add(time.Second), since); ok {
|
||||
t.Error("a dead EXIT was reported as a block; its own probe is the answer")
|
||||
}
|
||||
if _, ok := blockingHopOf(nil, since, since); ok {
|
||||
t.Error("an unmaterialised chain reported a block")
|
||||
}
|
||||
|
||||
// A non-chain target never consults any of this.
|
||||
e := New()
|
||||
if _, ok := e.blockedChainHop(groupTestTarget{name: "auto"}, since); ok {
|
||||
t.Error("a GROUP target was treated as a blocked chain")
|
||||
}
|
||||
}
|
||||
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user