Compare commits
13
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
7234817adb | ||
|
|
f1c36d6eea | ||
|
|
bb21ceb7f5 | ||
|
|
a8970b8ace | ||
|
|
4996bc0984 | ||
|
|
a0de597d69 | ||
|
|
544da29863 | ||
|
|
daaa0fda41 | ||
|
|
56a276bcc1 | ||
|
|
754bbcf1fa | ||
|
|
63e6b709f8 | ||
|
|
8612b0a9e9 | ||
|
|
96d9cfaa63 |
@@ -118,6 +118,79 @@ concurrency:
|
|||||||
cancel-in-progress: true
|
cancel-in-progress: true
|
||||||
|
|
||||||
jobs:
|
jobs:
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
# THE TEST GATE (2026-07-26). Everything below `needs:` this job, so a red test
|
||||||
|
# stops the release instead of shipping with it.
|
||||||
|
#
|
||||||
|
# WHY IT IS A JOB HERE AND NOT JUST .gitea/workflows/test.yml: a separate
|
||||||
|
# workflow cannot block another one — they run side by side and a red `test`
|
||||||
|
# workflow would have published anyway. Only a `needs:` edge inside THIS
|
||||||
|
# workflow is a gate. test.yml exists too, for fast feedback on `main`; both
|
||||||
|
# call the same scripts/run-tests.sh so they cannot drift.
|
||||||
|
#
|
||||||
|
# WHAT WAS BROKEN: the release tract ran two `go test` invocations in total —
|
||||||
|
# build-shaterd.sh's one-package buildtags check and check-router-tags.sh's
|
||||||
|
# three named tests. 115 of the 116 test files under shater/** had never run in
|
||||||
|
# CI (upstream's .github/workflows/test.yml triggers on branches this fork does
|
||||||
|
# not have, and Gitea ignores .github/workflows entirely once .gitea/workflows
|
||||||
|
# exists). TestDNSFilterRemoteBlocklistHTTPClient shipped red twice.
|
||||||
|
#
|
||||||
|
# WHAT IT COVERS: the whole suite under the SHIPPED build tags
|
||||||
|
# (scripts/router-tags.sh) on linux — the two dimensions that were missing.
|
||||||
|
# transport/wireguard compiles 1 test file without the tag set and 7 with it
|
||||||
|
# (the AmneziaWG ones); shater/generate has 44 test files on linux against 32
|
||||||
|
# elsewhere. Plus a -race pass and the panel's TypeScript tests. Details and
|
||||||
|
# the named, reasoned exclusions are in scripts/run-tests.sh.
|
||||||
|
test:
|
||||||
|
name: test gate
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
steps:
|
||||||
|
- name: Checkout
|
||||||
|
uses: actions/checkout@v4
|
||||||
|
|
||||||
|
# go.mod `replace`s wireguard-go to ./submodules/wireguard-go, so without
|
||||||
|
# this even `go list` fails. Same step/reason as in build-apk below.
|
||||||
|
- name: Init wireguard-go submodule (awg)
|
||||||
|
run: git submodule update --init --depth 1 submodules/wireguard-go
|
||||||
|
|
||||||
|
- name: Set up Go
|
||||||
|
uses: actions/setup-go@v5
|
||||||
|
with:
|
||||||
|
go-version-file: go.mod
|
||||||
|
cache: false # explicit actions/cache@v3.3.2 below
|
||||||
|
|
||||||
|
# Same cache key as build-apk: this job runs first, so it warms the module
|
||||||
|
# + build cache the SDK-lane build then restores. (v3.3.2 pin: see header.)
|
||||||
|
- name: Cache Go modules + build cache
|
||||||
|
uses: actions/cache@v3.3.2
|
||||||
|
with:
|
||||||
|
path: |
|
||||||
|
~/go/pkg/mod
|
||||||
|
~/.cache/go-build
|
||||||
|
key: go-${{ hashFiles('go.sum') }}
|
||||||
|
restore-keys: |
|
||||||
|
go-
|
||||||
|
|
||||||
|
# Node 24, NOT the 20 build-apk uses for the SPA: panel's tests are
|
||||||
|
# TypeScript run directly by `node --test`, and type stripping only exists
|
||||||
|
# from 22.6 — on node 20 `npm test` dies before running a single case.
|
||||||
|
- name: Set up Node
|
||||||
|
uses: actions/setup-node@v4
|
||||||
|
with:
|
||||||
|
node-version: '24'
|
||||||
|
|
||||||
|
- name: Cache panel node_modules
|
||||||
|
uses: actions/cache@v3.3.2
|
||||||
|
with:
|
||||||
|
path: panel/node_modules
|
||||||
|
key: npm-${{ hashFiles('panel/package-lock.json') }}
|
||||||
|
|
||||||
|
- name: Panel tests
|
||||||
|
run: bash scripts/run-panel-tests.sh
|
||||||
|
|
||||||
|
- name: Go tests (shipped tags, linux, + race)
|
||||||
|
run: bash scripts/run-tests.sh
|
||||||
|
|
||||||
# ---------------------------------------------------------------------------
|
# ---------------------------------------------------------------------------
|
||||||
# Build the 4 packages through the ImmortalWrt 25.12 apk-SDK for the 25.12/apk
|
# Build the 4 packages through the ImmortalWrt 25.12 apk-SDK for the 25.12/apk
|
||||||
# fleet (BPI-R3 mini on BananaWRT 25.12-mtk-vendor, BPI-R4 on OpenWrt 25.12,
|
# fleet (BPI-R3 mini on BananaWRT 25.12-mtk-vendor, BPI-R4 on OpenWrt 25.12,
|
||||||
@@ -125,6 +198,9 @@ jobs:
|
|||||||
# packages.adb + shater-apk.pem, uploaded as the artifact `apkfeed-<arch>`.
|
# packages.adb + shater-apk.pem, uploaded as the artifact `apkfeed-<arch>`.
|
||||||
build-apk:
|
build-apk:
|
||||||
name: apk ${{ matrix.arch }}
|
name: apk ${{ matrix.arch }}
|
||||||
|
# THE GATE EDGE. A red test skips this job, which leaves no artifact, which
|
||||||
|
# (with the guards in release-apk) leaves nothing published.
|
||||||
|
needs: test
|
||||||
runs-on: ubuntu-latest
|
runs-on: ubuntu-latest
|
||||||
strategy:
|
strategy:
|
||||||
fail-fast: false
|
fail-fast: false
|
||||||
@@ -302,11 +378,17 @@ jobs:
|
|||||||
# of that URL, so it is now written unconditionally and asserted afterwards.
|
# of that URL, so it is now written unconditionally and asserted afterwards.
|
||||||
release-apk:
|
release-apk:
|
||||||
name: release apk
|
name: release apk
|
||||||
needs: build-apk
|
needs: [test, build-apk]
|
||||||
# Publish whatever arch feeds succeeded — do NOT block the aarch64 release
|
# Publish whatever arch feeds succeeded — do NOT block the aarch64 release
|
||||||
# when an unrelated arch (e.g. x86_64) fails. download-artifact only fetches
|
# when an unrelated arch (e.g. x86_64) fails. download-artifact only fetches
|
||||||
# artifacts that exist, and the publish loop skips missing apkfeed-* dirs.
|
# artifacts that exist, and the publish loop skips missing apkfeed-* dirs.
|
||||||
if: ${{ !cancelled() }}
|
#
|
||||||
|
# `needs.test.result == 'success'` is the second half of the gate. Without
|
||||||
|
# it, `!cancelled()` is true when the test job FAILS (build-apk is then
|
||||||
|
# skipped), this job runs with no artifacts at all, and — see the guard at
|
||||||
|
# the end of the publish step — used to exit 0 having published nothing. Red
|
||||||
|
# tests must SKIP this job, not "succeed" through it.
|
||||||
|
if: ${{ !cancelled() && needs.test.result == 'success' }}
|
||||||
runs-on: ubuntu-latest
|
runs-on: ubuntu-latest
|
||||||
steps:
|
steps:
|
||||||
- name: Checkout
|
- name: Checkout
|
||||||
@@ -342,6 +424,14 @@ jobs:
|
|||||||
ROLLING: ${{ steps.rel.outputs.rolling }}
|
ROLLING: ${{ steps.rel.outputs.rolling }}
|
||||||
run: |
|
run: |
|
||||||
set -euo pipefail
|
set -euo pipefail
|
||||||
|
# Counted, and asserted non-zero at the end. Until 2026-07-26 this loop
|
||||||
|
# was the step's whole body: with no artifacts the glob stayed
|
||||||
|
# unexpanded, `[ -d ... ]` was false, `continue` ran once, the loop
|
||||||
|
# ended and the step exited 0 — "release apk" went GREEN having
|
||||||
|
# published absolutely nothing. Any upstream failure (all arches
|
||||||
|
# failing to build, an artifact-name change, a download-artifact
|
||||||
|
# hiccup) therefore looked like a successful release.
|
||||||
|
published=0
|
||||||
for d in artifacts/apkfeed-*; do
|
for d in artifacts/apkfeed-*; do
|
||||||
[ -d "$d" ] || continue
|
[ -d "$d" ] || continue
|
||||||
arch="${d#artifacts/apkfeed-}"
|
arch="${d#artifacts/apkfeed-}"
|
||||||
@@ -432,4 +522,15 @@ jobs:
|
|||||||
echo " own rules, not by our intent."
|
echo " own rules, not by our intent."
|
||||||
exit 14; }
|
exit 14; }
|
||||||
echo "[release-apk] OK — $ROLL serves $want"
|
echo "[release-apk] OK — $ROLL serves $want"
|
||||||
|
published=$((published + 1))
|
||||||
done
|
done
|
||||||
|
|
||||||
|
# The assert the loop above never had. Zero feeds published is a failed
|
||||||
|
# release, not a quiet success — say so with a non-zero exit.
|
||||||
|
if [ "$published" -eq 0 ]; then
|
||||||
|
echo "[release-apk] ERROR: no apkfeed-* artifact reached this job, so"
|
||||||
|
echo " NOTHING was published. Downloaded tree:"
|
||||||
|
ls -la artifacts 2>&1 | sed 's/^/ /' || echo " (no artifacts/ dir at all)"
|
||||||
|
exit 10
|
||||||
|
fi
|
||||||
|
echo "[release-apk] published $published arch feed(s)"
|
||||||
|
|||||||
@@ -0,0 +1,85 @@
|
|||||||
|
# Shater — the test gate, on every push to `main`.
|
||||||
|
#
|
||||||
|
# WHY THIS FILE EXISTS (2026-07-26)
|
||||||
|
# The fork had a full suite and no CI that ran it. Upstream's
|
||||||
|
# .github/workflows/test.yml triggers on `stable`/`testing`/`unstable`; this
|
||||||
|
# repo only has `main`. And Gitea does not read .github/workflows AT ALL once
|
||||||
|
# .gitea/workflows exists — so those files are decoration here. Result: 115 of
|
||||||
|
# the 116 test files under shater/** had never once executed in CI, and
|
||||||
|
# TestDNSFilterRemoteBlocklistHTTPClient stayed red across two published
|
||||||
|
# releases.
|
||||||
|
#
|
||||||
|
# RELATIONSHIP TO release.yml
|
||||||
|
# This workflow is the FAST FEEDBACK loop on `main`. It is NOT the release
|
||||||
|
# gate: a separate workflow cannot block another one. The gate is the `test`
|
||||||
|
# JOB inside .gitea/workflows/release.yml, which build-apk `needs:` — see the
|
||||||
|
# comment there. Both run the very same scripts/run-tests.sh, so they cannot
|
||||||
|
# drift apart.
|
||||||
|
name: test
|
||||||
|
|
||||||
|
on:
|
||||||
|
push:
|
||||||
|
branches: [main]
|
||||||
|
paths-ignore:
|
||||||
|
- '**.md'
|
||||||
|
- 'docs-shater/**'
|
||||||
|
pull_request:
|
||||||
|
branches: [main]
|
||||||
|
workflow_dispatch:
|
||||||
|
|
||||||
|
concurrency:
|
||||||
|
group: test-${{ github.ref }}
|
||||||
|
cancel-in-progress: true
|
||||||
|
|
||||||
|
jobs:
|
||||||
|
test:
|
||||||
|
name: go + panel tests
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
steps:
|
||||||
|
- name: Checkout
|
||||||
|
uses: actions/checkout@v4
|
||||||
|
|
||||||
|
# go.mod has `replace github.com/sagernet/wireguard-go => ./submodules/
|
||||||
|
# wireguard-go`, so WITHOUT this every `go list`/`go test` fails before it
|
||||||
|
# starts. Same step, same reason, as in release.yml's build job.
|
||||||
|
- name: Init wireguard-go submodule (awg)
|
||||||
|
run: git submodule update --init --depth 1 submodules/wireguard-go
|
||||||
|
|
||||||
|
- name: Set up Go
|
||||||
|
uses: actions/setup-go@v5
|
||||||
|
with:
|
||||||
|
go-version-file: go.mod
|
||||||
|
cache: false # explicit actions/cache@v3.3.2 below
|
||||||
|
|
||||||
|
# v3.3.2 is the last release speaking the cache API act_runner implements
|
||||||
|
# (see the header of release.yml). Same key as the release build job, so
|
||||||
|
# whichever runs first warms the other.
|
||||||
|
- name: Cache Go modules + build cache
|
||||||
|
uses: actions/cache@v3.3.2
|
||||||
|
with:
|
||||||
|
path: |
|
||||||
|
~/go/pkg/mod
|
||||||
|
~/.cache/go-build
|
||||||
|
key: go-${{ hashFiles('go.sum') }}
|
||||||
|
restore-keys: |
|
||||||
|
go-
|
||||||
|
|
||||||
|
# Node 24, NOT the 20 the SPA build uses: panel's tests are TypeScript run
|
||||||
|
# through `node --test`, and type stripping only exists from 22.6. On
|
||||||
|
# node 20 `npm test` dies with a syntax error before running anything.
|
||||||
|
- name: Set up Node
|
||||||
|
uses: actions/setup-node@v4
|
||||||
|
with:
|
||||||
|
node-version: '24'
|
||||||
|
|
||||||
|
- name: Cache panel node_modules
|
||||||
|
uses: actions/cache@v3.3.2
|
||||||
|
with:
|
||||||
|
path: panel/node_modules
|
||||||
|
key: npm-${{ hashFiles('panel/package-lock.json') }}
|
||||||
|
|
||||||
|
- name: Panel tests
|
||||||
|
run: bash scripts/run-panel-tests.sh
|
||||||
|
|
||||||
|
- name: Go tests (shipped tags, linux, + race)
|
||||||
|
run: bash scripts/run-tests.sh
|
||||||
+7
-1
@@ -7,4 +7,10 @@
|
|||||||
[submodule "submodules/wireguard-go"]
|
[submodule "submodules/wireguard-go"]
|
||||||
path = submodules/wireguard-go
|
path = submodules/wireguard-go
|
||||||
url = https://github.com/Leadaxe/wireguard-go-awg2-lx
|
url = https://github.com/Leadaxe/wireguard-go-awg2-lx
|
||||||
branch = lx
|
# The pin lives on lx-awg2-v005, NOT on lx: the two are separate lines (42
|
||||||
|
# commits apart one way, 131 the other). `lx` has no hasReserved() gate in
|
||||||
|
# conn/bind_std.go at all, so a `git submodule update --remote` against it
|
||||||
|
# would silently restore the bug where ClientBind/StdNetBind shred the
|
||||||
|
# AmneziaWG magic header and no chain carries traffic. Keep this pointing at
|
||||||
|
# the line the pin is actually on.
|
||||||
|
branch = lx-awg2-v005
|
||||||
|
|||||||
+3
-1
@@ -32,7 +32,9 @@ single-use token into the standalone SPA the daemon serves on its own port
|
|||||||
|
|
||||||
## Highlights
|
## Highlights
|
||||||
|
|
||||||
- Transparent **TPROXY** data plane (TCP + UDP), SNI/Host/QUIC sniffing, no DNS leaks.
|
- Transparent **TPROXY** data plane (TCP + UDP), SNI/Host/QUIC sniffing, no DNS leaks
|
||||||
|
— `:53` interception is on by default and covers the queries a client sends to the
|
||||||
|
router itself, not just the ones aimed around it (`globals.dns_intercept`, D24).
|
||||||
- First-match routing by source / destination / list / geo / client → outbound /
|
- First-match routing by source / destination / list / geo / client → outbound /
|
||||||
selector / chain / direct / block; node groups with balancer/observatory;
|
selector / chain / direct / block; node groups with balancer/observatory;
|
||||||
multi-hop chains; per-rule egress.
|
multi-hop chains; per-rule egress.
|
||||||
|
|||||||
@@ -3,7 +3,33 @@
|
|||||||
| Поле | Значение |
|
| Поле | Значение |
|
||||||
|------|----------|
|
|------|----------|
|
||||||
| Тип | B (bug) |
|
| Тип | B (bug) |
|
||||||
| Статус | C (complete) |
|
| Статус | C (complete) — guard **снят** (см. баннер ниже) |
|
||||||
|
|
||||||
|
> ## ⛔️ Guard снят (2026-07-26) — первопричина к shater не относится
|
||||||
|
> **Оба guard'а (Start-guard в `protocol/wireguard/endpoint.go` и
|
||||||
|
> selector-guard в `protocol/group/awg_selector_guard.go`) удалены**, вместе с
|
||||||
|
> их adapter-хуками (`OutboundManager.ConsumersOf`, `AmneziaWGSuspendable`).
|
||||||
|
> Апстрим снял их коммитом `5fa3a0a17`; сюда снятие приехало отдельно.
|
||||||
|
>
|
||||||
|
> **Почему.** Зависание было **Android-специфичным** (`Libbox.newService` не
|
||||||
|
> возвращал управление). Android для shater не платформа и ей не станет —
|
||||||
|
> мы собираем роутерный бинарь под OpenWrt/aarch64. При этом лекарство для
|
||||||
|
> самой AWG-за-detour связки у нас уже есть: reserved-clear gate в
|
||||||
|
> `ClientBind` (`d971eb85e` + пин сабмодуля `7d15f33`), без которого AWG не
|
||||||
|
> поднимался вообще ни за каким detour'ом. Мы носили и лекарство, и запрет
|
||||||
|
> на его применение.
|
||||||
|
>
|
||||||
|
> **Чем это было плохо на практике.** Guard отказывал **молча**: не ошибкой,
|
||||||
|
> а `started=false`, после чего каждый дозвон падал с «WireGuard is not ready
|
||||||
|
> yet». Конфигурация «AmneziaWG за WireGuard-хопом» выглядела не как
|
||||||
|
> отклонённая, а как «нода почему-то не работает».
|
||||||
|
>
|
||||||
|
> **Регрессия:** `protocol/wireguard/awg_over_wireguard_start_lx_test.go`
|
||||||
|
> (`with_gvisor && with_awg`) — AWG-эндпоинт с `detour` на outbound типа
|
||||||
|
> `wireguard` доходит до PostStart и поднимает `started`. До снятия guard'а
|
||||||
|
> тест краснел.
|
||||||
|
>
|
||||||
|
> **Осталось:** сквозной прогон на железе (AWG поверх реального WG-хопа).
|
||||||
|
|
||||||
Отклонять (по образцу ядрового запрета «empty direct detour») конфигурацию, где
|
Отклонять (по образцу ядрового запрета «empty direct detour») конфигурацию, где
|
||||||
AmneziaWG-endpoint (источник с AWG-полями) имеет `detour` на **любой
|
AmneziaWG-endpoint (источник с AWG-полями) имеет `detour` на **любой
|
||||||
|
|||||||
@@ -45,30 +45,8 @@ type OutboundManager interface {
|
|||||||
Default() Outbound
|
Default() Outbound
|
||||||
Remove(tag string) error
|
Remove(tag string) error
|
||||||
Create(ctx context.Context, router Router, logger log.ContextLogger, tag string, outboundType string, options any) error
|
Create(ctx context.Context, router Router, logger log.ContextLogger, tag string, outboundType string, options any) error
|
||||||
// lx:begin awg
|
|
||||||
// ConsumersOf returns the tags of outbounds that depend on (detour through)
|
|
||||||
// the given tag — the reverse of Dependencies(). Used by the selector guard to
|
|
||||||
// walk up to AmneziaWG consumers when a group switches to a WireGuard member.
|
|
||||||
ConsumersOf(tag string) []string
|
|
||||||
// lx:end awg
|
|
||||||
}
|
}
|
||||||
|
|
||||||
// lx:begin awg
|
|
||||||
// AmneziaWGSuspendable is implemented by an AmneziaWG endpoint so the selector
|
|
||||||
// guard can suspend it (bring its device down) when a group it detours through
|
|
||||||
// switches to a WireGuard member — AmneziaWG inside a WireGuard tunnel hangs the
|
|
||||||
// kernel on Android. The marker lives in adapter so protocol/group can act on it
|
|
||||||
// without importing protocol/wireguard.
|
|
||||||
type AmneziaWGSuspendable interface {
|
|
||||||
// IsAmneziaWG reports whether this endpoint runs AmneziaWG (has AWG params).
|
|
||||||
IsAmneziaWG() bool
|
|
||||||
// SuspendAmneziaWG brings the device down so no junk handshake is sent. It is
|
|
||||||
// idempotent and safe to call on a not-yet-started or already-suspended endpoint.
|
|
||||||
SuspendAmneziaWG()
|
|
||||||
}
|
|
||||||
|
|
||||||
// lx:end awg
|
|
||||||
|
|
||||||
// lx:begin idle-suspend
|
// lx:begin idle-suspend
|
||||||
// IdleSuspendable is implemented by a WG/AWG endpoint so the router's idle tick
|
// IdleSuspendable is implemented by a WG/AWG endpoint so the router's idle tick
|
||||||
// (SPEC 020) can suspend it when it is idle and unreachable, without importing
|
// (SPEC 020) can suspend it when it is idle and unreachable, without importing
|
||||||
|
|||||||
@@ -208,21 +208,6 @@ func (m *Manager) Outbound(tag string) (adapter.Outbound, bool) {
|
|||||||
return m.endpoint.Get(tag)
|
return m.endpoint.Get(tag)
|
||||||
}
|
}
|
||||||
|
|
||||||
// lx:begin awg
|
|
||||||
// ConsumersOf returns a copy of the tags that detour through tag (reverse of
|
|
||||||
// Dependencies()), built from the dependByTag ledger populated at Create time.
|
|
||||||
func (m *Manager) ConsumersOf(tag string) []string {
|
|
||||||
m.access.RLock()
|
|
||||||
defer m.access.RUnlock()
|
|
||||||
consumers := m.dependByTag[tag]
|
|
||||||
if len(consumers) == 0 {
|
|
||||||
return nil
|
|
||||||
}
|
|
||||||
return append([]string(nil), consumers...)
|
|
||||||
}
|
|
||||||
|
|
||||||
// lx:end awg
|
|
||||||
|
|
||||||
func (m *Manager) Default() adapter.Outbound {
|
func (m *Manager) Default() adapter.Outbound {
|
||||||
m.access.RLock()
|
m.access.RLock()
|
||||||
defer m.access.RUnlock()
|
defer m.access.RUnlock()
|
||||||
|
|||||||
@@ -0,0 +1,35 @@
|
|||||||
|
//go:build darwin
|
||||||
|
|
||||||
|
package dialer
|
||||||
|
|
||||||
|
import (
|
||||||
|
"syscall"
|
||||||
|
"testing"
|
||||||
|
|
||||||
|
"golang.org/x/sys/unix"
|
||||||
|
)
|
||||||
|
|
||||||
|
// udpSocketDFSet reports whether the socket has "don't fragment" forced on
|
||||||
|
// (control.DisableUDPFragment sets IP_DONTFRAG=1 on darwin).
|
||||||
|
func udpSocketDFSet(t *testing.T, sysConn syscall.Conn) bool {
|
||||||
|
t.Helper()
|
||||||
|
rawConn, err := sysConn.SyscallConn()
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
var (
|
||||||
|
value int
|
||||||
|
sockErr error
|
||||||
|
ctrlErr error
|
||||||
|
)
|
||||||
|
ctrlErr = rawConn.Control(func(fd uintptr) {
|
||||||
|
value, sockErr = unix.GetsockoptInt(int(fd), unix.IPPROTO_IP, unix.IP_DONTFRAG)
|
||||||
|
})
|
||||||
|
if ctrlErr != nil {
|
||||||
|
t.Fatal(ctrlErr)
|
||||||
|
}
|
||||||
|
if sockErr != nil {
|
||||||
|
t.Fatal(sockErr)
|
||||||
|
}
|
||||||
|
return value != 0
|
||||||
|
}
|
||||||
@@ -0,0 +1,36 @@
|
|||||||
|
//go:build linux
|
||||||
|
|
||||||
|
package dialer
|
||||||
|
|
||||||
|
import (
|
||||||
|
"syscall"
|
||||||
|
"testing"
|
||||||
|
|
||||||
|
"golang.org/x/sys/unix"
|
||||||
|
)
|
||||||
|
|
||||||
|
// udpSocketDFSet reports whether the socket has "don't fragment" forced on
|
||||||
|
// (control.DisableUDPFragment sets IP_MTU_DISCOVER=IP_PMTUDISC_DO on linux,
|
||||||
|
// the same flag the user-visible failure was traced to on android).
|
||||||
|
func udpSocketDFSet(t *testing.T, sysConn syscall.Conn) bool {
|
||||||
|
t.Helper()
|
||||||
|
rawConn, err := sysConn.SyscallConn()
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
var (
|
||||||
|
value int
|
||||||
|
sockErr error
|
||||||
|
ctrlErr error
|
||||||
|
)
|
||||||
|
ctrlErr = rawConn.Control(func(fd uintptr) {
|
||||||
|
value, sockErr = unix.GetsockoptInt(int(fd), unix.IPPROTO_IP, unix.IP_MTU_DISCOVER)
|
||||||
|
})
|
||||||
|
if ctrlErr != nil {
|
||||||
|
t.Fatal(ctrlErr)
|
||||||
|
}
|
||||||
|
if sockErr != nil {
|
||||||
|
t.Fatal(sockErr)
|
||||||
|
}
|
||||||
|
return value == unix.IP_PMTUDISC_DO
|
||||||
|
}
|
||||||
@@ -0,0 +1,14 @@
|
|||||||
|
//go:build !darwin && !linux && !windows
|
||||||
|
|
||||||
|
package dialer
|
||||||
|
|
||||||
|
import (
|
||||||
|
"syscall"
|
||||||
|
"testing"
|
||||||
|
)
|
||||||
|
|
||||||
|
func udpSocketDFSet(t *testing.T, _ syscall.Conn) bool {
|
||||||
|
t.Helper()
|
||||||
|
t.Skip("DF socket-flag introspection implemented for darwin, linux and windows only")
|
||||||
|
return false
|
||||||
|
}
|
||||||
@@ -0,0 +1,43 @@
|
|||||||
|
//go:build windows
|
||||||
|
|
||||||
|
package dialer
|
||||||
|
|
||||||
|
import (
|
||||||
|
"syscall"
|
||||||
|
"testing"
|
||||||
|
|
||||||
|
"golang.org/x/sys/windows"
|
||||||
|
)
|
||||||
|
|
||||||
|
// IP_MTU_DISCOVER on windows (ws2ipdef.h); control.DisableUDPFragment sets it to
|
||||||
|
// IP_PMTUDISC_DO, the same "don't fragment" state the linux helper checks.
|
||||||
|
const (
|
||||||
|
windowsIPMTUDiscover = 71
|
||||||
|
windowsPMTUDiscDo = 1
|
||||||
|
)
|
||||||
|
|
||||||
|
// udpSocketDFSet reports whether the socket has "don't fragment" forced on.
|
||||||
|
// shater addition: upstream ships linux + darwin only, so the whole suite
|
||||||
|
// skipped on the dev host — where it is the one platform we can actually run it
|
||||||
|
// on before the router build.
|
||||||
|
func udpSocketDFSet(t *testing.T, sysConn syscall.Conn) bool {
|
||||||
|
t.Helper()
|
||||||
|
rawConn, err := sysConn.SyscallConn()
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
var (
|
||||||
|
value int
|
||||||
|
sockErr error
|
||||||
|
)
|
||||||
|
ctrlErr := rawConn.Control(func(fd uintptr) {
|
||||||
|
value, sockErr = windows.GetsockoptInt(windows.Handle(fd), windows.IPPROTO_IP, windowsIPMTUDiscover)
|
||||||
|
})
|
||||||
|
if ctrlErr != nil {
|
||||||
|
t.Fatal(ctrlErr)
|
||||||
|
}
|
||||||
|
if sockErr != nil {
|
||||||
|
t.Skip("IP_MTU_DISCOVER is not readable on this host: ", sockErr)
|
||||||
|
}
|
||||||
|
return value == windowsPMTUDiscDo
|
||||||
|
}
|
||||||
@@ -0,0 +1,99 @@
|
|||||||
|
// lx: regression tests for the udp_fragment / UDPFragmentDefault
|
||||||
|
// plumbing. The WireGuard endpoint (and MASQUE outbound) rely on
|
||||||
|
// UDPFragmentDefault=true reaching the real UDP socket as "DF clear": with DF
|
||||||
|
// set, an outer datagram larger than the path MTU is silently dropped instead
|
||||||
|
// of fragmented, which blackholes nested tunnels (AWG-over-AWG, MASQUE-over-AWG)
|
||||||
|
// and AWG s4 transport junk. These tests assert the socket flag itself, on both
|
||||||
|
// paths a WireGuard bind can take: the dialer (ClientBind, detour case) and the
|
||||||
|
// listener control (StdNetBind via WireGuardControl, no-detour case).
|
||||||
|
package dialer
|
||||||
|
|
||||||
|
import (
|
||||||
|
"context"
|
||||||
|
"net"
|
||||||
|
"syscall"
|
||||||
|
"testing"
|
||||||
|
|
||||||
|
"github.com/sagernet/sing-box/option"
|
||||||
|
M "github.com/sagernet/sing/common/metadata"
|
||||||
|
N "github.com/sagernet/sing/common/network"
|
||||||
|
)
|
||||||
|
|
||||||
|
func dialUDPForDF(t *testing.T, options option.DialerOptions) syscall.Conn {
|
||||||
|
t.Helper()
|
||||||
|
d, err := NewDefault(context.Background(), options)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
conn, err := d.DialContext(context.Background(), N.NetworkUDP, M.ParseSocksaddr("127.0.0.1:9"))
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
t.Cleanup(func() { _ = conn.Close() })
|
||||||
|
sysConn, isSysConn := conn.(syscall.Conn)
|
||||||
|
if !isSysConn {
|
||||||
|
t.Fatalf("dialed UDP conn %T does not expose SyscallConn", conn)
|
||||||
|
}
|
||||||
|
return sysConn
|
||||||
|
}
|
||||||
|
|
||||||
|
func listenUDPForDF(t *testing.T, options option.DialerOptions) syscall.Conn {
|
||||||
|
t.Helper()
|
||||||
|
d, err := NewDefault(context.Background(), options)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
// WireGuardControl() is the listener control conn.StdNetBind installs on the
|
||||||
|
// socket a no-detour WireGuard endpoint sends its outer datagrams from — the
|
||||||
|
// exact socket the DF default decides the fate of.
|
||||||
|
listenConfig := net.ListenConfig{Control: d.WireGuardControl()}
|
||||||
|
packetConn, err := listenConfig.ListenPacket(context.Background(), "udp4", "127.0.0.1:0")
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
t.Cleanup(func() { _ = packetConn.Close() })
|
||||||
|
sysConn, isSysConn := packetConn.(syscall.Conn)
|
||||||
|
if !isSysConn {
|
||||||
|
t.Fatalf("listened UDP conn %T does not expose SyscallConn", packetConn)
|
||||||
|
}
|
||||||
|
return sysConn
|
||||||
|
}
|
||||||
|
|
||||||
|
// Upstream default: no UDPFragmentDefault, no udp_fragment → DF is set on both
|
||||||
|
// the dial and listener paths. Pins the baseline the endpoint fix opts out of.
|
||||||
|
func TestUDPFragmentDFByDefault_LX(t *testing.T) {
|
||||||
|
if !udpSocketDFSet(t, dialUDPForDF(t, option.DialerOptions{})) {
|
||||||
|
t.Fatal("default dialer must set DF on dialed UDP sockets")
|
||||||
|
}
|
||||||
|
if !udpSocketDFSet(t, listenUDPForDF(t, option.DialerOptions{})) {
|
||||||
|
t.Fatal("default dialer must set DF on listener-control UDP sockets")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// UDPFragmentDefault=true (what the WireGuard endpoint and MASQUE outbound now
|
||||||
|
// set) → DF clear on both paths, so oversize outer datagrams fragment instead
|
||||||
|
// of vanishing.
|
||||||
|
func TestUDPFragmentDefaultClearsDF_LX(t *testing.T) {
|
||||||
|
options := option.DialerOptions{UDPFragmentDefault: true}
|
||||||
|
if udpSocketDFSet(t, dialUDPForDF(t, options)) {
|
||||||
|
t.Fatal("UDPFragmentDefault=true must leave DF clear on dialed UDP sockets")
|
||||||
|
}
|
||||||
|
if udpSocketDFSet(t, listenUDPForDF(t, options)) {
|
||||||
|
t.Fatal("UDPFragmentDefault=true must leave DF clear on listener-control UDP sockets")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// Explicit user config always wins over the protocol default, in both
|
||||||
|
// directions.
|
||||||
|
func TestUDPFragmentExplicitOverride_LX(t *testing.T) {
|
||||||
|
fragmentOff := false
|
||||||
|
options := option.DialerOptions{UDPFragment: &fragmentOff, UDPFragmentDefault: true}
|
||||||
|
if !udpSocketDFSet(t, dialUDPForDF(t, options)) {
|
||||||
|
t.Fatal("udp_fragment=false must set DF even when the protocol default allows fragmentation")
|
||||||
|
}
|
||||||
|
fragmentOn := true
|
||||||
|
options = option.DialerOptions{UDPFragment: &fragmentOn}
|
||||||
|
if udpSocketDFSet(t, dialUDPForDF(t, options)) {
|
||||||
|
t.Fatal("udp_fragment=true must leave DF clear even without a protocol default")
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -25,6 +25,21 @@ func requireRoot(t *testing.T) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// requireTCPDump skips when tcpdump is not installed.
|
||||||
|
//
|
||||||
|
// The same honesty this package's callers demand of a health reading: a missing
|
||||||
|
// INSTRUMENT is "not checked", never "broken". Without it every test in this
|
||||||
|
// file fails on `cmd.Start()` — sixteen red results that say nothing about the
|
||||||
|
// code and hide any real failure among them — on a machine where the only thing
|
||||||
|
// wrong is that a capture tool is absent. requireRoot has always drawn that line
|
||||||
|
// for privileges; this draws it for the tool.
|
||||||
|
func requireTCPDump(t *testing.T) {
|
||||||
|
t.Helper()
|
||||||
|
if _, err := exec.LookPath("tcpdump"); err != nil {
|
||||||
|
t.Skip("integration test requires tcpdump on PATH; install it to run this suite")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
func tcpdumpObserver(t *testing.T, iface string, port uint16, needle string, do func(), wait time.Duration) bool {
|
func tcpdumpObserver(t *testing.T, iface string, port uint16, needle string, do func(), wait time.Duration) bool {
|
||||||
t.Helper()
|
t.Helper()
|
||||||
return tcpdumpObserverMulti(t, iface, port, []string{needle}, do, wait)[needle]
|
return tcpdumpObserverMulti(t, iface, port, []string{needle}, do, wait)[needle]
|
||||||
@@ -36,6 +51,9 @@ func tcpdumpObserver(t *testing.T, iface string, port uint16, needle string, do
|
|||||||
// the wire.
|
// the wire.
|
||||||
func tcpdumpObserverMulti(t *testing.T, iface string, port uint16, needles []string, do func(), wait time.Duration) map[string]bool {
|
func tcpdumpObserverMulti(t *testing.T, iface string, port uint16, needles []string, do func(), wait time.Duration) map[string]bool {
|
||||||
t.Helper()
|
t.Helper()
|
||||||
|
// Every capture in this file funnels through here, so one guard covers the
|
||||||
|
// whole suite and no future test can forget it.
|
||||||
|
requireTCPDump(t)
|
||||||
ctx, cancel := context.WithTimeout(context.Background(), wait)
|
ctx, cancel := context.WithTimeout(context.Background(), wait)
|
||||||
defer cancel()
|
defer cancel()
|
||||||
cmd := exec.CommandContext(ctx, "tcpdump", "-i", iface, "-n", "-A", "-l",
|
cmd := exec.CommandContext(ctx, "tcpdump", "-i", iface, "-n", "-A", "-l",
|
||||||
|
|||||||
@@ -0,0 +1,61 @@
|
|||||||
|
package urltest
|
||||||
|
|
||||||
|
// lx: health board §5.C — the reachability half of "should this be probed".
|
||||||
|
//
|
||||||
|
// # Two different reasons not to probe, and why they cannot be one flag
|
||||||
|
//
|
||||||
|
// A group's OWN probing schedule is stood down for two unrelated reasons, and
|
||||||
|
// conflating them breaks one of the two:
|
||||||
|
//
|
||||||
|
// - NOT USED — no enabled routing rule reaches this group, so probing it
|
||||||
|
// measures a path nothing travels. That is a property of the CONFIG, it is
|
||||||
|
// decided once when the config is generated, and it travels in the config
|
||||||
|
// itself (option.URLTestOutboundOptions.SelfCheck). It cannot change while
|
||||||
|
// the box runs, because the rules cannot change while the box runs.
|
||||||
|
//
|
||||||
|
// - NOT REACHABLE RIGHT NOW — the group is a hop of a chain and a hop in
|
||||||
|
// FRONT of it is currently dead. Every member of this group dials through
|
||||||
|
// that hop, so every probe would fail inside it: the measurement would be
|
||||||
|
// about the broken hop, and would be recorded against this one. That is a
|
||||||
|
// property of the WORLD, it changes minute by minute, and it must be
|
||||||
|
// re-asked every time rather than baked into the config — a hop that comes
|
||||||
|
// back must resume probing on its own, with no reapply and nobody pressing
|
||||||
|
// anything.
|
||||||
|
//
|
||||||
|
// ProbeGate is the second one. It is deliberately a QUESTION asked at the
|
||||||
|
// moment of probing and never a stored answer: there is no flag to set, so
|
||||||
|
// there is no flag to forget to clear.
|
||||||
|
//
|
||||||
|
// The gate governs the group's own SCHEDULE only — the warm-up sweep and the
|
||||||
|
// ticker. An explicit check (a human, an API call) is a deliberate request and
|
||||||
|
// is never refused, exactly as with SelfCheck.
|
||||||
|
type ProbeGate interface {
|
||||||
|
// ProbeAllowed reports whether the outbound tagged tag may run its own
|
||||||
|
// scheduled probe right now.
|
||||||
|
//
|
||||||
|
// Implementations MUST answer true when they do not know: a gate that
|
||||||
|
// refuses on missing information would silence probing precisely when the
|
||||||
|
// system has the least idea what is going on, and nothing would ever
|
||||||
|
// measure its way out of that. A nil ProbeGate means "no gate" and every
|
||||||
|
// probe proceeds.
|
||||||
|
ProbeAllowed(tag string) bool
|
||||||
|
|
||||||
|
// ProbeWhenIdle reports whether the outbound tagged tag must keep measuring
|
||||||
|
// even when no traffic is passing through it.
|
||||||
|
//
|
||||||
|
// A urltest group normally probes only while it is in use: Touch arms the
|
||||||
|
// ticker on a dial, and the idle timeout stops it again. That is right for a
|
||||||
|
// group whose readings matter only while somebody is dialling it, and wrong
|
||||||
|
// for one the routing config REACHES: a rule that matches rarely — a narrow
|
||||||
|
// domain list, say — is in force the whole time, so the health of its target
|
||||||
|
// is a live question the whole time. Letting it go quiet means the panel
|
||||||
|
// reports "untested" about a rule that is armed, and the first real request
|
||||||
|
// pays a cold probe instead of picking an already-known-good member.
|
||||||
|
//
|
||||||
|
// Unlike ProbeAllowed, the safe answer here is FALSE when nothing is known.
|
||||||
|
// This one ADDS work, and a gate that claimed it on missing information would
|
||||||
|
// keep every group in the process probing forever — not a default anybody
|
||||||
|
// asked for. Absent gate, unknown tag, nothing configured yet: false, and the
|
||||||
|
// idle timeout behaves exactly as it always has.
|
||||||
|
ProbeWhenIdle(tag string) bool
|
||||||
|
}
|
||||||
@@ -666,3 +666,109 @@ Neither is a substitute for the other. A new protocol in `shater/parse` +
|
|||||||
stays only so the router set remains a subset of the lx desktop set. Noted
|
stays only so the router set remains a subset of the lx desktop set. Noted
|
||||||
because "a tag that buys nothing" is the mirror image of this bug and should be
|
because "a tag that buys nothing" is the mirror image of this bug and should be
|
||||||
removed deliberately, not silently.
|
removed deliberately, not silently.
|
||||||
|
|
||||||
|
## D24 — DNS interception is the DEFAULT (`dns_intercept=1`), not an opt-in
|
||||||
|
Decided 2026-07-26. `Globals.DNSIntercept` shipped as opt-in (`default false`, and
|
||||||
|
absent from both `DefaultGlobals` and the shipped `/etc/config/shater`). The result
|
||||||
|
was an **inverted** posture, which is the reason this is a decision and not a
|
||||||
|
preference:
|
||||||
|
|
||||||
|
- a client with **standard** settings — DNS = the router's address, exactly what
|
||||||
|
DHCP hands out — sent its queries to the router. The nft `:53` divert was behind
|
||||||
|
the flag (`netplane/nft.go`), and the rule right after it is an unconditional
|
||||||
|
`fib daddr type local accept`, so the query was delivered locally to dnsmasq and
|
||||||
|
forwarded to the ISP **in the clear**: no blocklists, no per-device DNS rules,
|
||||||
|
no Block-DoH, no resolver detour, nothing;
|
||||||
|
- a client that hard-coded `8.8.8.8` "to bypass the router" was addressing a
|
||||||
|
non-local IP and **was** caught by the ordinary tproxy catch-all.
|
||||||
|
|
||||||
|
The obedient client leaked; the evader did not. Meanwhile `FEATURES.md`, `README.md`
|
||||||
|
and D14 all promised "no DNS leaks" and "dnsmasq never sees LAN queries" — true only
|
||||||
|
for the traffic pattern the default did not cover. `dns_intercept` appeared nowhere
|
||||||
|
in `docs-shater/` at all.
|
||||||
|
|
||||||
|
**Decision: `DNSIntercept` is seeded ON in `model.DefaultGlobals`, and the shipped
|
||||||
|
`/etc/config/shater` carries an explicit `option dns_intercept '1'`.** Nothing about
|
||||||
|
the interception MECHANISM changed — only which side of the switch is the default.
|
||||||
|
|
||||||
|
**`.lan` and the private PTR zones keep working, and that is a pre-existing part of
|
||||||
|
the mechanism, not something bolted on for this flip.** `generate/dns.go` adds a
|
||||||
|
synthetic DNS server (`shater-local-dns`, plain UDP to `127.0.0.1:53`, detour
|
||||||
|
`direct`, so the daemon's own loop-mark keeps it out of the divert) and PREPENDS a
|
||||||
|
`domain_suffix` rule for `lan` + the RFC6303 private reverse zones, ahead of every
|
||||||
|
device/filter rule. Two honest limitations: it hardcodes `lan` (a router whose
|
||||||
|
dnsmasq domain was changed needs a `config dns_rule` for the new suffix), and it
|
||||||
|
only exists when the model has at least one `config resolver` — with none, buildDNS
|
||||||
|
emits no DNS plane at all and the engine falls back to its built-in `local`
|
||||||
|
transport, which reads `/etc/resolv.conf` (127.0.0.1 → dnsmasq), so local names
|
||||||
|
still resolve but nothing is filtered.
|
||||||
|
|
||||||
|
**A dead engine does NOT black out the LAN's DNS.** This was the first thing checked,
|
||||||
|
because "intercept everything" invites the reading "engine down = no DNS anywhere",
|
||||||
|
and that is not what happens:
|
||||||
|
- the fail-closed **holding plane** (D17, `RenderHoldNft`) hooks `forward` ONLY.
|
||||||
|
A query addressed to the router is INPUT-hook traffic, so dnsmasq answers it as
|
||||||
|
it always did — unfiltered and plaintext to the ISP. Deliberate: blocking it
|
||||||
|
would also cut the daemon's own name resolution and with it any chance of
|
||||||
|
self-recovery;
|
||||||
|
- with the FULL plane loaded and the engine's tproxy socket gone, the `tproxy`
|
||||||
|
statement returns `NFT_BREAK`, which aborts its own rule; the packet continues
|
||||||
|
down the chain into the same `fib daddr type local accept` and reaches dnsmasq.
|
||||||
|
|
||||||
|
So the failure mode is a DNS **fail-open** (working, unfiltered) while client
|
||||||
|
TRAFFIC stays fail-closed — and a query aimed at an EXTERNAL resolver is dropped
|
||||||
|
with the rest of the forwarded traffic. Operators must know this: "the tunnel is
|
||||||
|
down" does not mean "DNS is private".
|
||||||
|
|
||||||
|
**Existing installs.** `/etc/config/shater` is a conffile
|
||||||
|
(`openwrt/shater-core/Makefile`), so an upgrade never replaces it:
|
||||||
|
- a config that never mentioned the option (all of them, before this change) now
|
||||||
|
parses over the ON seed and **starts intercepting on the next apply**. That is the
|
||||||
|
intended behaviour change, and the only one this decision makes;
|
||||||
|
- an explicit `option dns_intercept '0'` keeps winning. It survives the
|
||||||
|
`WriteUCI→ReadUCI` round-trip because `render.go` emits booleans ALWAYS —
|
||||||
|
the trap a default-true bool has and a default-false one does not: a value
|
||||||
|
omitted at false would come back as the seed and silently re-enable itself.
|
||||||
|
`shater/model/dnsintercept_test.go` pins both directions, plus the shipped file.
|
||||||
|
|
||||||
|
**Not done: silencing the "no resolvers configured" warning by shipping a resolver.**
|
||||||
|
With interception on and no `config resolver`, generate warns — and it is right to:
|
||||||
|
every client query now lands in an engine that has no resolver plane, so it is
|
||||||
|
answered by the system resolver (dnsmasq → the ISP, in the clear) with filtering and
|
||||||
|
anti-leak inert. Shipping a `type local` resolver would make the warning disappear
|
||||||
|
while changing nothing about where the queries go: the panel would show a configured
|
||||||
|
resolver and the operator would believe DNS was handled. That is the inverted lie
|
||||||
|
this project keeps deleting. The warning stays; what it needs is the accurate
|
||||||
|
wording (it currently claims `.lan` breaks, which the fallback above disproves), not
|
||||||
|
a workaround. Note also that a fresh install ships INERT (`enabled '0'`) and
|
||||||
|
`Reconcile` tears down instead of generating, so the warning cannot appear before the
|
||||||
|
operator has enabled the stack — at which point it describes their live config.
|
||||||
|
|
||||||
|
**OPEN, and it gates shipping this default: the synthetic local server changes how
|
||||||
|
proxy-endpoint DOMAINS are resolved.** Found while landing D24, reproduced on Linux
|
||||||
|
with one resolver and a node addressed by a hostname:
|
||||||
|
|
||||||
|
- `common/dialer/dialer.go` resolves a domain server address through
|
||||||
|
`route.default_domain_resolver`; when that is unset it uses
|
||||||
|
`dnsTransport.Default()` — the engine's built-in `local` transport, i.e. a
|
||||||
|
bootstrap-DIRECT lookup — but **only while fewer than two DNS transports exist**.
|
||||||
|
With two or more and no default, it reports the `missing-domain-resolver`
|
||||||
|
deprecation and leaves the query transport nil, so `dns.Router.Lookup` falls back
|
||||||
|
to `lookupWithRules`: the CLIENT DNS plane.
|
||||||
|
- `dns_intercept` adds `shater-local-dns`, which takes a single-resolver config from
|
||||||
|
one transport to two. So a config whose only resolver is DoH-through-the-tunnel —
|
||||||
|
the recommended anti-leak setup — would start resolving its own node's hostname
|
||||||
|
through that same tunnel: a bootstrap loop where there was none.
|
||||||
|
- Evidence: the same model emits no deprecation notice with `dns_intercept=0` and
|
||||||
|
two `missing-domain-resolver` notices with `dns_intercept=1`;
|
||||||
|
`generate.TestDNSFilterRemoteBlocklistHTTPClient` (Linux-only) fails on exactly
|
||||||
|
that notice and is deliberately left failing rather than relaxed.
|
||||||
|
|
||||||
|
The fix belongs in `generate` (`route.go:160` already sets
|
||||||
|
`route.default_domain_resolver` from `endpointResolver()`, which is opt-in and unset
|
||||||
|
by default): when buildDNS emits the synthetic local server and no endpoint resolver
|
||||||
|
is configured, `default_domain_resolver` must be pointed at a bootstrap-direct
|
||||||
|
server, which restores exactly the pre-D24 behaviour and clears the notice. Until
|
||||||
|
that lands, an operator can get the same result by setting `endpoint_resolver` to a
|
||||||
|
direct resolver. Note the hazard is **not** created by D24 — any config with two
|
||||||
|
resolvers has it today; the default merely makes it universal.
|
||||||
|
|||||||
@@ -40,6 +40,16 @@ usable release, **[T1]** next, **[T2]** later. Phases refer to `ROADMAP.md`.
|
|||||||
type=fakeip + pool — there is no global "FakeIP mode"); no DNS leaks. Routing
|
type=fakeip + pool — there is no global "FakeIP mode"); no DNS leaks. Routing
|
||||||
is decided by in-engine rule-sets — the v0.1 dnsmasq→nftset population
|
is decided by in-engine rule-sets — the v0.1 dnsmasq→nftset population
|
||||||
mechanism does not exist in v0.2 (see generate/dns.go).
|
mechanism does not exist in v0.2 (see generate/dns.go).
|
||||||
|
The hijack covers the queries a client sends **to the router itself** — the
|
||||||
|
address DHCP hands out — because `globals.dns_intercept` is **ON by default**
|
||||||
|
(D24). With it off, those queries go to dnsmasq and out to the ISP in the clear,
|
||||||
|
so the well-behaved client leaks while the one that hard-codes 8.8.8.8 does not.
|
||||||
|
`.lan` and the private PTR zones are preserved through dnsmasq either way. Two
|
||||||
|
things the promise does NOT cover, both by design: while the engine is DOWN the
|
||||||
|
holding plane hooks `forward` only, so dnsmasq still answers router-addressed
|
||||||
|
:53 unfiltered (client traffic and DNS to external resolvers stay blocked); and
|
||||||
|
with no `config resolver` at all there is no DNS plane to filter with — queries
|
||||||
|
fall through to the system resolver and generate says so.
|
||||||
- **[MVP]** Client DoT/DoH blocking (stop devices bypassing the filter).
|
- **[MVP]** Client DoT/DoH blocking (stop devices bypassing the filter).
|
||||||
- **[MVP]** **Blocklists** with **flexible sources**: `inline` (type your own) /
|
- **[MVP]** **Blocklists** with **flexible sources**: `inline` (type your own) /
|
||||||
`file` / `url` (auto-update) / `geosite` category (only when geodata present).
|
`file` / `url` (auto-update) / `geosite` category (only when geodata present).
|
||||||
|
|||||||
@@ -186,6 +186,38 @@ daemon (`shaterd run`), which owns the engine, the `inet shater` data plane, pol
|
|||||||
routing, in-process DNS, and the admin panel (default `:8088`). The LuCI app's
|
routing, in-process DNS, and the admin panel (default `:8088`). The LuCI app's
|
||||||
"Open panel" button mints a single-use token and hands the browser off to the panel.
|
"Open panel" button mints a single-use token and hands the browser off to the panel.
|
||||||
|
|
||||||
|
### What enabling does to DNS
|
||||||
|
|
||||||
|
From the first apply, **every** LAN plaintext `:53` goes into the engine — including
|
||||||
|
the queries a client sends to the router's own address, which is what DHCP hands out.
|
||||||
|
That is `globals.dns_intercept`, and it is **on by default** (D24); without it those
|
||||||
|
queries reach dnsmasq and the ISP unfiltered, i.e. the client with default settings
|
||||||
|
leaks while the one that hard-coded `8.8.8.8` does not. What follows from it:
|
||||||
|
|
||||||
|
- `.lan` and private reverse (PTR) lookups still go to dnsmasq — the engine gets a
|
||||||
|
rule for those suffixes. If you renamed dnsmasq's domain away from `lan`, add a
|
||||||
|
`config dns_rule` for the new suffix.
|
||||||
|
- Configure at least one `config resolver`. With none, the engine has no resolver
|
||||||
|
plane: intercepted queries fall through to the system resolver (dnsmasq → your
|
||||||
|
ISP, in the clear), blocklists and per-device DNS rules are inert, and the apply
|
||||||
|
says so in its warnings.
|
||||||
|
- While the engine is DOWN, DNS is **not** blacked out: the fail-closed holding
|
||||||
|
plane hooks `forward` only, so dnsmasq keeps answering router-addressed `:53`
|
||||||
|
(unfiltered, plaintext) while client traffic and DNS to external resolvers stay
|
||||||
|
blocked. "The tunnel is down" is not "DNS is private".
|
||||||
|
|
||||||
|
To opt out, on the router:
|
||||||
|
|
||||||
|
```sh
|
||||||
|
uci set shater.globals.dns_intercept=0
|
||||||
|
uci commit shater
|
||||||
|
shaterd apply
|
||||||
|
```
|
||||||
|
|
||||||
|
Your `0` is kept: `/etc/config/shater` is a conffile (upgrades never replace it) and
|
||||||
|
the daemon always writes the option back explicitly, so it is never re-enabled by a
|
||||||
|
default.
|
||||||
|
|
||||||
## 5. The signed apk repo (the normal install path)
|
## 5. The signed apk repo (the normal install path)
|
||||||
|
|
||||||
OpenWrt/ImmortalWrt **25.12** packages with Alpine's **apk**: `.apk` files, a
|
OpenWrt/ImmortalWrt **25.12** packages with Alpine's **apk**: `.apk` files, a
|
||||||
|
|||||||
+34
-1
@@ -251,7 +251,40 @@ Apply/rollback: `apSnapshot` (run→last-good, nft→last-good.nft, route marks)
|
|||||||
> v0.2: "restart engine only on change" → config-hash gate + Close+New box (no reload).
|
> v0.2: "restart engine only on change" → config-hash gate + Close+New box (no reload).
|
||||||
|
|
||||||
### uci.go — `/etc/config/shater` schema
|
### uci.go — `/etc/config/shater` schema
|
||||||
- `config globals`: enabled, loglevel, kill_switch, dns_mode, ipv6, fwmark_base, table_base, confirm_timeout, resolver_default, resolver_fallback, probe_url, probe_interval, schema_version, active_profile.
|
- `config globals` — the full option set, with the value used when the option is
|
||||||
|
ABSENT (the `model.DefaultGlobals` seed). Booleans are always written back as
|
||||||
|
`'1'`/`'0'` by `render.go`, so an explicit value never decays into the seed:
|
||||||
|
|
||||||
|
| option | default | meaning |
|
||||||
|
|---|---|---|
|
||||||
|
| `enabled` | `0` as shipped | master switch; `0` ⇒ `Reconcile` tears the stack down instead of applying |
|
||||||
|
| `loglevel` (alias `log_level`) | `warning` | engine + daemon level; `none/off/silent/disabled` ⇒ log disabled, unknown ⇒ `warn` + a validation warning |
|
||||||
|
| `log_syslog` / `log_file` / `log_persist` | `1` / `1` / `0` | operational log (`shater/logsink`): syslog, rotated file, and whether that file lives on flash instead of tmpfs |
|
||||||
|
| `log_max_kb` | `2048` | size cap of the log file, clamped to 128…8192; `0` = "use the default", not "off" |
|
||||||
|
| `kill_switch` | `closed` | `closed` = fail-closed (block on engine loss, incl. a holding plane when the engine never started); `open` = plain routing |
|
||||||
|
| `ipv6` | `1` | `0` drops LAN IPv6 in the forward chain instead of leaving it unproxied |
|
||||||
|
| `fwmark_base` / `table_base` | `0x2000` | reserved fwmark / routing-table bases (must not collide with fw4 or other apps) |
|
||||||
|
| `confirm_timeout` | `0` | seconds before an unconfirmed apply auto-rolls back; `0` = commit-confirm off |
|
||||||
|
| `resolver_default` / `resolver_fallback` / `endpoint_resolver` | unset | `config resolver` names: the DNS catch-all, its failover chain, and the bootstrap-direct server that resolves proxy endpoint DOMAINS |
|
||||||
|
| `probe_url` / `probe_interval` | engine defaults | the ONE instrument all health probing uses (D20 — there are no per-group overrides) |
|
||||||
|
| `panel_port` | `0` ⇒ `8088` | admin-panel HTTP port |
|
||||||
|
| `dns_filter` | `0` | master enable of the blocklist/allowlist filter (D15); needs at least one `config resolver` |
|
||||||
|
| `dns_intercept` | **`1`** | force ALL LAN plaintext `:53` into the engine, INCLUDING queries addressed to the router itself. See D24 for why this is the default, what preserves `.lan`, and what happens while the engine is down |
|
||||||
|
| `block_doh` | `0` | NXDOMAIN the known public DoH hostnames + the Firefox canary and reject `:443` to their IPs, so clients fall back to `:53` (which the engine catches) |
|
||||||
|
| `group_health` | `1` | OUR background group probing (the observatory). Does not touch sing-box's own urltest inside a group |
|
||||||
|
| `untunnelable` | `block` | policy for what TPROXY cannot carry (ICMP/IGMP/ESP/AH/GRE/SCTP): `block` \| `icmp` (echo out, rest dropped) \| `direct` (all out, bypassing the tunnel) |
|
||||||
|
| `geo_provider` | unset = auto | `sagernet` \| `loyalsoldier` \| `metacubex` \| `custom`; auto = country codes from SagerNet, everything else from Loyalsoldier |
|
||||||
|
| `geosite_url` / `geoip_url` | unset | `{category}` templates, honoured only when `geo_provider=custom` |
|
||||||
|
| `geosite_index_url` / `geoip_index_url` | unset | git-trees URLs used to SUGGEST categories in the panel; empty = no suggestions |
|
||||||
|
| `stats_backend` | `memory` | `off` (no aggregation at all) \| `memory` (RAM, lost on restart) \| `sqlite` (aggregates in RAM + query/connection log on disk) |
|
||||||
|
| `stats_ring_size` / `stats_timeline_minutes` / `stats_max_domains` | `200` / `60` / `5000` | live-log length, sparkline minutes, domain-map cap. **`0` = UNLIMITED** (grows with traffic), which is why these three are always emitted |
|
||||||
|
| `stats_disk_limit_mb` | `64` | on-disk cap of `stats.db`; only meaningful for `stats_backend=sqlite`; `0` = unlimited |
|
||||||
|
| `stats_retention_disabled` | `0` | master switch that turns OFF all trimming/pruning — every aggregate then grows unbounded |
|
||||||
|
| `schema_version` | `0` = pre-versioned | UCI schema revision; `shaterd migrate` writes `2` |
|
||||||
|
| `active_profile` | unset | display bookkeeping: the last profile switched to |
|
||||||
|
|
||||||
|
Deleted options still parse (unknown keys are ignored) and drain out on the next
|
||||||
|
render: `dns_mode` (D17 — fake-IP is a resolver TYPE), `sweep_interval` (D19).
|
||||||
- `config inbound`: name, enabled, type, network, tproxy_port(12345), listen, port, auth, user, pass, target_addr, target_port, target_network, tcp, udp, sniff.
|
- `config inbound`: name, enabled, type, network, tproxy_port(12345), listen, port, auth, user, pass, target_addr, target_port, target_network, tcp, udp, sniff.
|
||||||
- `config subscription`: name, enabled, url, update_interval, fetch_via(direct|proxy), ua, hwid, device_os, ver_os, device_model, list header, format, list include/exclude/filter_proto/filter_country, dedup, expire_alert_days.
|
- `config subscription`: name, enabled, url, update_interval, fetch_via(direct|proxy), ua, hwid, device_os, ver_os, device_model, list header, format, list include/exclude/filter_proto/filter_country, dedup, expire_alert_days.
|
||||||
- `config node`: name, enabled, uri, mux, mux_concurrency, xudp_concurrency, xudp_udp443, sockopt_mark, tcp_fast_open, tcp_keepalive_idle.
|
- `config node`: name, enabled, uri, mux, mux_concurrency, xudp_concurrency, xudp_udp443, sockopt_mark, tcp_fast_open, tcp_keepalive_idle.
|
||||||
|
|||||||
@@ -23,6 +23,32 @@ config globals 'globals'
|
|||||||
option kill_switch 'closed'
|
option kill_switch 'closed'
|
||||||
# There is no dns_mode option: routing is decided by in-engine rule-sets and
|
# There is no dns_mode option: routing is decided by in-engine rule-sets and
|
||||||
# fake-IP is a resolver type (`config resolver` with type=fakeip + pool).
|
# fake-IP is a resolver type (`config resolver` with type=fakeip + pool).
|
||||||
|
#
|
||||||
|
# Force ALL LAN plaintext DNS (:53) into the engine, INCLUDING queries the
|
||||||
|
# client sends to the router itself (the address DHCP hands out). ON by
|
||||||
|
# default: with it off, a client using the router as its resolver is answered
|
||||||
|
# by dnsmasq and forwarded to the ISP in the clear — no blocklists, no
|
||||||
|
# per-device DNS rules, no resolver detour — while a client that hard-codes
|
||||||
|
# 8.8.8.8 IS intercepted. The obedient client leaked; the evader did not.
|
||||||
|
#
|
||||||
|
# Set to '0' to opt out (dnsmasq answers router-addressed :53 again). Your
|
||||||
|
# explicit value is never overwritten: this file is a conffile, and the daemon
|
||||||
|
# always writes the option back as '1'/'0'.
|
||||||
|
#
|
||||||
|
# .lan and the private reverse (PTR) zones keep working: with at least one
|
||||||
|
# `config resolver` present the engine gets a synthetic server pointed at
|
||||||
|
# dnsmasq on 127.0.0.1:53 plus a rule that sends those suffixes to it; with no
|
||||||
|
# resolver at all the engine falls back to the system resolver, which is
|
||||||
|
# dnsmasq too. If you changed dnsmasq's domain away from `lan`, add a
|
||||||
|
# `config dns_rule` for it (only `lan` + RFC6303 reverse zones are built in).
|
||||||
|
#
|
||||||
|
# While the engine is DOWN the LAN is NOT left without DNS: the fail-closed
|
||||||
|
# holding plane hooks `forward` only, so dnsmasq still answers router-addressed
|
||||||
|
# :53 — unfiltered and in the clear, the documented trade-off (blocking it
|
||||||
|
# would also cut the daemon's own name resolution and its chance to recover).
|
||||||
|
# Queries aimed at an EXTERNAL resolver are dropped with the rest of the LAN's
|
||||||
|
# forwarded traffic.
|
||||||
|
option dns_intercept '1'
|
||||||
option ipv6 '1'
|
option ipv6 '1'
|
||||||
# Reserved fwmark base and routing-table base (do not overlap fw4/other apps).
|
# Reserved fwmark base and routing-table base (do not overlap fw4/other apps).
|
||||||
option fwmark_base '0x2000'
|
option fwmark_base '0x2000'
|
||||||
|
|||||||
@@ -262,6 +262,91 @@
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/* ---- commit-confirm band (every page except Apply, which has the full panel) ----
|
||||||
|
Same plate as the protection banner so the two read as one family; the seconds
|
||||||
|
are the loud element because they are the only thing that is running out. */
|
||||||
|
.cfm-band {
|
||||||
|
display: flex;
|
||||||
|
align-items: center;
|
||||||
|
gap: calc(var(--u, 8px) * 1.5);
|
||||||
|
margin-top: calc(var(--u, 8px) * 2);
|
||||||
|
padding: 10px 14px;
|
||||||
|
border: 1px solid color-mix(in srgb, var(--amber) 50%, var(--groove));
|
||||||
|
border-radius: 9px;
|
||||||
|
background: linear-gradient(180deg, color-mix(in srgb, var(--amber) 10%, var(--raised)), var(--raised));
|
||||||
|
box-shadow: 0 1px 0 var(--edge) inset;
|
||||||
|
}
|
||||||
|
.cfm-band-count {
|
||||||
|
display: flex;
|
||||||
|
align-items: baseline;
|
||||||
|
gap: 2px;
|
||||||
|
flex-shrink: 0;
|
||||||
|
font-family: var(--font-mono);
|
||||||
|
color: var(--amber);
|
||||||
|
}
|
||||||
|
.cfm-band-num {
|
||||||
|
font-size: 22px;
|
||||||
|
font-weight: 700;
|
||||||
|
font-variant-numeric: tabular-nums;
|
||||||
|
line-height: 1;
|
||||||
|
}
|
||||||
|
.cfm-band-unit {
|
||||||
|
font-size: 11px;
|
||||||
|
letter-spacing: 0.06em;
|
||||||
|
}
|
||||||
|
.cfm-band-copy {
|
||||||
|
flex: 1;
|
||||||
|
min-width: 0;
|
||||||
|
display: flex;
|
||||||
|
flex-direction: column;
|
||||||
|
gap: 3px;
|
||||||
|
}
|
||||||
|
.cfm-band-headline {
|
||||||
|
font-family: var(--font-mono);
|
||||||
|
font-size: 12.5px;
|
||||||
|
font-weight: 700;
|
||||||
|
letter-spacing: 0.02em;
|
||||||
|
color: var(--ink);
|
||||||
|
}
|
||||||
|
.cfm-band-detail {
|
||||||
|
font-size: 12.5px;
|
||||||
|
line-height: 1.5;
|
||||||
|
color: var(--dim);
|
||||||
|
max-width: 76ch;
|
||||||
|
}
|
||||||
|
.cfm-band-actions {
|
||||||
|
display: flex;
|
||||||
|
align-items: center;
|
||||||
|
gap: calc(var(--u, 8px) * 1);
|
||||||
|
flex-shrink: 0;
|
||||||
|
}
|
||||||
|
.cfm-band-link {
|
||||||
|
padding: 6px 11px;
|
||||||
|
border: 1px solid var(--groove);
|
||||||
|
border-radius: 6px;
|
||||||
|
font-family: var(--font-mono);
|
||||||
|
font-size: 11px;
|
||||||
|
letter-spacing: 0.06em;
|
||||||
|
text-transform: uppercase;
|
||||||
|
text-decoration: none;
|
||||||
|
color: var(--ink);
|
||||||
|
background: var(--raised);
|
||||||
|
}
|
||||||
|
.cfm-band-link:hover {
|
||||||
|
border-color: var(--accent);
|
||||||
|
color: var(--accent);
|
||||||
|
}
|
||||||
|
|
||||||
|
@media (max-width: 720px) {
|
||||||
|
.cfm-band {
|
||||||
|
flex-wrap: wrap;
|
||||||
|
}
|
||||||
|
.cfm-band-actions {
|
||||||
|
width: 100%;
|
||||||
|
justify-content: flex-end;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
/* ---- last-apply findings (Overview) ----
|
/* ---- last-apply findings (Overview) ----
|
||||||
Severity carries the colour; the accent is reserved for interactive controls. */
|
Severity carries the colour; the accent is reserved for interactive controls. */
|
||||||
.findings {
|
.findings {
|
||||||
|
|||||||
+87
-6
@@ -1,12 +1,13 @@
|
|||||||
import './App.css'
|
import './App.css'
|
||||||
import { useCallback, useEffect, useState } from 'react'
|
import { useCallback, useEffect, useState } from 'react'
|
||||||
import { Faceplate, FaceplateHeader, Led, Module } from './components'
|
import { Button, Faceplate, FaceplateHeader, Led, Module } from './components'
|
||||||
import type { LedVariant } from './components'
|
import type { LedVariant } from './components'
|
||||||
import { ApiError, MOCK, getStatus } from './api'
|
import { ApiError, MOCK, confirm as apiConfirm, getStatus } from './api'
|
||||||
import type { Status } from './api'
|
import type { Status } from './api'
|
||||||
|
import { usePendingConfirm } from './pendingConfirm'
|
||||||
import { bootstrapSession } from './session'
|
import { bootstrapSession } from './session'
|
||||||
import { ROUTES, navigate, useRoute } from './router'
|
import { ROUTES, navigate, useRoute } from './router'
|
||||||
import { protectionState } from './planeState'
|
import { engineState, protectionState } from './planeState'
|
||||||
import type { Route } from './router'
|
import type { Route } from './router'
|
||||||
import { Overview, Placeholder, Nodes, Routing, Apply, DNS, Devices, Targets, Settings, Profiles, Insights, Networks } from './pages'
|
import { Overview, Placeholder, Nodes, Routing, Apply, DNS, Devices, Targets, Settings, Profiles, Insights, Networks } from './pages'
|
||||||
|
|
||||||
@@ -100,11 +101,75 @@ export function App() {
|
|||||||
>
|
>
|
||||||
<Nav route={route} />
|
<Nav route={route} />
|
||||||
<PlaneBanner status={status} route={route} />
|
<PlaneBanner status={status} route={route} />
|
||||||
|
<ConfirmBand route={route} onChanged={() => void refreshStatus()} />
|
||||||
<Page route={route} status={status} onStatusChange={() => void refreshStatus()} />
|
<Page route={route} status={status} onStatusChange={() => void refreshStatus()} />
|
||||||
</Faceplate>
|
</Faceplate>
|
||||||
)
|
)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* The commit-confirm countdown, on every page.
|
||||||
|
*
|
||||||
|
* The daemon arms an auto-rollback on EVERY apply, but only the Apply page ever
|
||||||
|
* said so: press Apply on Routing, read "Applied", walk away, and the router
|
||||||
|
* reverts a minute later with nothing on screen having mentioned it. This band
|
||||||
|
* carries that deadline — and the button that stops it — to wherever the operator
|
||||||
|
* actually is.
|
||||||
|
*
|
||||||
|
* Suppressed on Apply, which renders the full control room for the same window
|
||||||
|
* (and reads the same record, so a reload no longer loses the countdown there
|
||||||
|
* either).
|
||||||
|
*/
|
||||||
|
function ConfirmBand({ route, onChanged }: { route: Route; onChanged: () => void }) {
|
||||||
|
const armed = usePendingConfirm()
|
||||||
|
const [busy, setBusy] = useState(false)
|
||||||
|
const [error, setError] = useState<string | null>(null)
|
||||||
|
|
||||||
|
// Keeping the config is the only action offered here; rolling back early is a
|
||||||
|
// deliberate act with its own before/after readout, and that lives on Apply.
|
||||||
|
const keep = useCallback(async () => {
|
||||||
|
setBusy(true)
|
||||||
|
setError(null)
|
||||||
|
try {
|
||||||
|
const r = await apiConfirm()
|
||||||
|
if (r.error) setError(r.error)
|
||||||
|
} catch (e) {
|
||||||
|
setError(e instanceof Error ? e.message : 'request failed')
|
||||||
|
} finally {
|
||||||
|
setBusy(false)
|
||||||
|
onChanged()
|
||||||
|
}
|
||||||
|
}, [onChanged])
|
||||||
|
|
||||||
|
if (!armed || route === 'apply') return null
|
||||||
|
|
||||||
|
return (
|
||||||
|
<div className="cfm-band" role="alert">
|
||||||
|
<Led variant="amber" pulse />
|
||||||
|
<div className="cfm-band-count" role="timer" aria-label={`${armed.remaining} seconds until auto-rollback`}>
|
||||||
|
<span className="cfm-band-num">{armed.remaining}</span>
|
||||||
|
<span className="cfm-band-unit">s</span>
|
||||||
|
</div>
|
||||||
|
<div className="cfm-band-copy">
|
||||||
|
<span className="cfm-band-headline">This config is live but not kept</span>
|
||||||
|
<span className="cfm-band-detail">
|
||||||
|
{error
|
||||||
|
? `Couldn’t keep it — ${error}. Try again, or open Apply.`
|
||||||
|
: 'Every apply arms an auto-rollback. Keep this config before the timer runs out, or the router reverts to the last-good one.'}
|
||||||
|
</span>
|
||||||
|
</div>
|
||||||
|
<div className="cfm-band-actions">
|
||||||
|
<Button variant="primary" onClick={() => void keep()} disabled={busy}>
|
||||||
|
{busy ? 'Keeping…' : 'Keep this config'}
|
||||||
|
</Button>
|
||||||
|
<a className="cfm-band-link" href="#/apply" onClick={() => navigate('apply')}>
|
||||||
|
Apply page
|
||||||
|
</a>
|
||||||
|
</div>
|
||||||
|
</div>
|
||||||
|
)
|
||||||
|
}
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* The protection state, pinned under the nav on every page EXCEPT Overview
|
* The protection state, pinned under the nav on every page EXCEPT Overview
|
||||||
* (which shows the same state as its own headline readout — see planeState.ts).
|
* (which shows the same state as its own headline readout — see planeState.ts).
|
||||||
@@ -225,14 +290,30 @@ function StatusBar({ status }: { status: Status | null }) {
|
|||||||
)
|
)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* The one lamp that is on screen no matter which page you are on.
|
||||||
|
*
|
||||||
|
* It used to read `status.running`, which the daemon hardcoded to `true` — so the
|
||||||
|
* "Offline" branch could never be reached and the plate said "Online" through an
|
||||||
|
* engine that had failed to start. It now asks {@link engineState}, whose whole
|
||||||
|
* job is to be able to answer "down", and refuses to guess when nothing has been
|
||||||
|
* reported: an unlit socket, not a green light.
|
||||||
|
*/
|
||||||
function masterIndicator(
|
function masterIndicator(
|
||||||
phase: Phase,
|
phase: Phase,
|
||||||
status: Status | null,
|
status: Status | null,
|
||||||
): { label: string; variant: LedVariant; pulse?: boolean } {
|
): { label: string; variant: LedVariant; pulse?: boolean } {
|
||||||
if (phase === 'loading' || !status) return { label: 'Linking', variant: 'off' }
|
if (phase === 'loading' || !status) return { label: 'Linking', variant: 'off' }
|
||||||
if (status.running && status.active) return { label: 'Online', variant: 'on', pulse: true }
|
switch (engineState(status)) {
|
||||||
if (status.running) return { label: 'Standby', variant: 'amber' }
|
case 'down':
|
||||||
return { label: 'Offline', variant: 'crit' }
|
return { label: 'Engine down', variant: 'crit' }
|
||||||
|
case 'up':
|
||||||
|
return status.active
|
||||||
|
? { label: 'Online', variant: 'on', pulse: true }
|
||||||
|
: { label: 'Standby', variant: 'amber' }
|
||||||
|
default:
|
||||||
|
return { label: 'Unknown', variant: 'off' }
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
function UnauthPlate() {
|
function UnauthPlate() {
|
||||||
|
|||||||
+83
-11
@@ -15,6 +15,7 @@
|
|||||||
// reachable instead via the Vite proxy in vite.config.ts (no flag ⇒ real fetch).
|
// reachable instead via the Vite proxy in vite.config.ts (no flag ⇒ real fetch).
|
||||||
|
|
||||||
import * as mock from './mock'
|
import * as mock from './mock'
|
||||||
|
import { armPendingConfirm, clearPendingConfirm, noteConfirmTimeout } from './pendingConfirm'
|
||||||
|
|
||||||
// --- error type -------------------------------------------------------------
|
// --- error type -------------------------------------------------------------
|
||||||
|
|
||||||
@@ -413,6 +414,16 @@ export interface GroupHealth {
|
|||||||
* `tag` is the engine-side outbound (`chain-<name>-h2`). Debugging and tooltips
|
* `tag` is the engine-side outbound (`chain-<name>-h2`). Debugging and tooltips
|
||||||
* only; it is never a label to put in front of a person.
|
* only; it is never a label to put in front of a person.
|
||||||
*
|
*
|
||||||
|
* ORDERED WALK — THE READING STOPS AT THE FIRST DEAD HOP. Hops are NOT measured
|
||||||
|
* independently, and never were measurable that way: hop 3 is dialled THROUGH
|
||||||
|
* hop 2, so probing hop 3 while hop 2 is down measures hop 2 a second time and
|
||||||
|
* learns nothing about hop 3. The daemon therefore walks the path in wire order
|
||||||
|
* and stops at the first hop that does not answer. Every hop below that one is
|
||||||
|
* left undialled and reported `state: "untested"` — no measurement exists —
|
||||||
|
* carrying {@link ChainHopBlock} in `blocked_by` to name the hop that stopped the
|
||||||
|
* walk. So a chain never reports a dead hop with a live hop below it; that shape
|
||||||
|
* is not a rare case, it is unreachable.
|
||||||
|
*
|
||||||
* NODE HOP vs GROUP HOP. For `kind: "node"` the hop IS the measurement: `total`
|
* NODE HOP vs GROUP HOP. For `kind: "node"` the hop IS the measurement: `total`
|
||||||
* is 1, the counters follow its own state, and `selected` is ''. For
|
* is 1, the counters follow its own state, and `selected` is ''. For
|
||||||
* `kind: "group"` the counters roll up that hop's per-hop member COPIES — the
|
* `kind: "group"` the counters roll up that hop's per-hop member COPIES — the
|
||||||
@@ -425,9 +436,12 @@ export interface GroupHealth {
|
|||||||
* Invariants the daemon guarantees — never re-derive them, just read them:
|
* Invariants the daemon guarantees — never re-derive them, just read them:
|
||||||
* `tested === alive + dead` and `alive + dead + untested === total`.
|
* `tested === alive + dead` and `alive + dead + untested === total`.
|
||||||
*
|
*
|
||||||
* `state` is a closed set. `untested` is NEVER "dead" and never "healthy": it
|
* `state` is a closed set of THREE. `untested` is NEVER "dead" and never
|
||||||
* means nothing fresh enough is known, which for a used chain is seconds away
|
* "healthy": it means nothing fresh enough is known. Without `blocked_by` that is
|
||||||
* from resolving on its own. `age_seconds: -1` means the age is unknown.
|
* a matter of timing — for a used chain it resolves on its own within seconds.
|
||||||
|
* With `blocked_by` it will not resolve until the named hop is fixed. There is no
|
||||||
|
* fourth state for that; the state stays `untested` because that is what it is.
|
||||||
|
* `age_seconds: -1` means the age is unknown.
|
||||||
*/
|
*/
|
||||||
export interface ChainHopHealth {
|
export interface ChainHopHealth {
|
||||||
/** 1-based WIRE order. Hop 1 is dialled first; see the note above. */
|
/** 1-based WIRE order. Hop 1 is dialled first; see the note above. */
|
||||||
@@ -451,6 +465,43 @@ export interface ChainHopHealth {
|
|||||||
alive: number
|
alive: number
|
||||||
dead: number
|
dead: number
|
||||||
untested: number
|
untested: number
|
||||||
|
/**
|
||||||
|
* PRESENT ONLY on a hop the ordered walk never reached — i.e. a hop sitting
|
||||||
|
* below one the prober found `dead`. The key is omitted otherwise; absent is
|
||||||
|
* the normal case and means "this hop was actually dialled".
|
||||||
|
*
|
||||||
|
* Its presence is the daemon's own statement that this hop has NO measurement,
|
||||||
|
* and it comes with the rest of that statement already filled in: `state` is
|
||||||
|
* `untested`, `delay_ms` is 0, `age_seconds` is -1, and the counters are
|
||||||
|
* `alive: 0, dead: 0, tested: 0, untested: total`. Read those; do not re-derive
|
||||||
|
* a verdict from them, and do not infer a block from zeroed counters either —
|
||||||
|
* an unprobed-yet hop has the same numbers and a very different meaning.
|
||||||
|
* `selected` MAY still be non-empty: the wrapper does have a pick, it simply
|
||||||
|
* was not measured, so it says which node the hop would use, not which node is
|
||||||
|
* carrying traffic.
|
||||||
|
*/
|
||||||
|
blocked_by?: ChainHopBlock
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* The hop that stopped the ordered walk, as reported on every hop below it.
|
||||||
|
*
|
||||||
|
* This exists because "no reading" and "no reading, and here is whose fault that
|
||||||
|
* is" are different answers to the operator's actual question. Without it a
|
||||||
|
* blocked hop is indistinguishable from one the observatory has not come round to
|
||||||
|
* yet, and the interface can only shrug.
|
||||||
|
*
|
||||||
|
* `index` is the 1-based WIRE index of the blocking hop and is ALWAYS smaller
|
||||||
|
* than the index of the hop carrying it, so it points at a hop already on screen.
|
||||||
|
* `tag` is that hop's engine outbound (`chain-<name>-h3`) — debugging and
|
||||||
|
* tooltips only, never a label to put in front of a person, exactly as on
|
||||||
|
* {@link ChainHopHealth}.tag.
|
||||||
|
*/
|
||||||
|
export interface ChainHopBlock {
|
||||||
|
/** 1-based wire index of the hop that did not answer. Always < this hop's index. */
|
||||||
|
index: number
|
||||||
|
/** That hop's engine outbound tag — tooltips/debugging, never a label. */
|
||||||
|
tag: string
|
||||||
}
|
}
|
||||||
|
|
||||||
/** Per-chain reachability, the chain analogue of {@link GroupHealth}.used (plan
|
/** Per-chain reachability, the chain analogue of {@link GroupHealth}.used (plan
|
||||||
@@ -1126,8 +1177,13 @@ export function getStatus(): Promise<Status> {
|
|||||||
return MOCK ? mock.getStatus() : req<Status>('api/status')
|
return MOCK ? mock.getStatus() : req<Status>('api/status')
|
||||||
}
|
}
|
||||||
|
|
||||||
export function getConfig(): Promise<Model> {
|
export async function getConfig(): Promise<Model> {
|
||||||
return MOCK ? mock.getConfig() : req<Model>('api/config')
|
const m = await (MOCK ? mock.getConfig() : req<Model>('api/config'))
|
||||||
|
// Every page reads the config, and the commit-confirm window's length is the
|
||||||
|
// only thing needed to arm a countdown — so it is captured here once instead of
|
||||||
|
// being threaded through eight pages. See pendingConfirm.ts.
|
||||||
|
noteConfirmTimeout(m.Globals?.ConfirmTimeout)
|
||||||
|
return m
|
||||||
}
|
}
|
||||||
|
|
||||||
export function putConfig(m: Model): Promise<{ ok: boolean; applied: boolean }> {
|
export function putConfig(m: Model): Promise<{ ok: boolean; applied: boolean }> {
|
||||||
@@ -1136,16 +1192,32 @@ export function putConfig(m: Model): Promise<{ ok: boolean; applied: boolean }>
|
|||||||
: req('api/config', { method: 'PUT', body: JSON.stringify(m) })
|
: req('api/config', { method: 'PUT', body: JSON.stringify(m) })
|
||||||
}
|
}
|
||||||
|
|
||||||
export function apply(): Promise<ApplyResult> {
|
/**
|
||||||
return MOCK ? mock.apply() : req<ApplyResult>('api/apply', { method: 'POST' })
|
* POST /api/apply.
|
||||||
|
*
|
||||||
|
* The daemon arms an auto-rollback on EVERY successful apply that changed
|
||||||
|
* something (panel/api.go handleApply → ArmRollback), whichever page's button was
|
||||||
|
* pressed. Recording it here — the one place every one of those buttons goes
|
||||||
|
* through — is what lets the countdown and the "Keep this config" control follow
|
||||||
|
* the operator around the panel instead of living in the Apply page's local
|
||||||
|
* state. See pendingConfirm.ts.
|
||||||
|
*/
|
||||||
|
export async function apply(): Promise<ApplyResult> {
|
||||||
|
const r = await (MOCK ? mock.apply() : req<ApplyResult>('api/apply', { method: 'POST' }))
|
||||||
|
if (!r.error && r.changed) armPendingConfirm()
|
||||||
|
return r
|
||||||
}
|
}
|
||||||
|
|
||||||
export function confirm(): Promise<ApplyResult> {
|
export async function confirm(): Promise<ApplyResult> {
|
||||||
return MOCK ? mock.confirm() : req<ApplyResult>('api/confirm', { method: 'POST' })
|
const r = await (MOCK ? mock.confirm() : req<ApplyResult>('api/confirm', { method: 'POST' }))
|
||||||
|
if (!r.error) clearPendingConfirm()
|
||||||
|
return r
|
||||||
}
|
}
|
||||||
|
|
||||||
export function rollback(): Promise<ApplyResult> {
|
export async function rollback(): Promise<ApplyResult> {
|
||||||
return MOCK ? mock.rollback() : req<ApplyResult>('api/rollback', { method: 'POST' })
|
const r = await (MOCK ? mock.rollback() : req<ApplyResult>('api/rollback', { method: 'POST' }))
|
||||||
|
if (!r.error) clearPendingConfirm()
|
||||||
|
return r
|
||||||
}
|
}
|
||||||
|
|
||||||
export function getStats(): Promise<Stats> {
|
export function getStats(): Promise<Stats> {
|
||||||
|
|||||||
@@ -1,11 +1,49 @@
|
|||||||
import { useEffect, useState } from 'react'
|
import { useEffect, useState } from 'react'
|
||||||
|
|
||||||
|
/**
|
||||||
|
* The panel's wall clock, in the SAME timezone as every timestamp under it.
|
||||||
|
*
|
||||||
|
* It used to read `getUTCHours()` and print "UTC", while `format.ts` renders every
|
||||||
|
* log line, connection event and date through `toLocaleTimeString` — i.e. the
|
||||||
|
* browser's zone. In Moscow that put two clocks three hours apart on one plate,
|
||||||
|
* and the header was the one nobody could reconcile: the router's "started" time
|
||||||
|
* read later than the current time while the uptime said it had been up for hours.
|
||||||
|
*
|
||||||
|
* So the clock follows the rest of the panel — local, and it SAYS which offset
|
||||||
|
* that is, because a bare "12:41:07" beside a router in another zone is the
|
||||||
|
* ambiguity that started this. The zone label is the browser's UTC offset, not an
|
||||||
|
* abbreviation: "MSK"/"CEST" are not derivable everywhere, an offset always is.
|
||||||
|
*
|
||||||
|
* This is the BROWSER's clock, not the router's — the appliance has no RTC. Every
|
||||||
|
* router-sourced instant in the panel is converted to this clock before it is
|
||||||
|
* shown, which is what makes one label at the top honest for the whole page.
|
||||||
|
*/
|
||||||
|
function zoneLabel(d: Date): string {
|
||||||
|
// getTimezoneOffset() is minutes WEST of UTC, so the sign is inverted.
|
||||||
|
const min = -d.getTimezoneOffset()
|
||||||
|
if (min === 0) return 'UTC'
|
||||||
|
const sign = min < 0 ? '−' : '+'
|
||||||
|
const a = Math.abs(min)
|
||||||
|
const h = Math.floor(a / 60)
|
||||||
|
const m = a % 60
|
||||||
|
return `UTC${sign}${h}${m ? `:${String(m).padStart(2, '0')}` : ''}`
|
||||||
|
}
|
||||||
|
|
||||||
function format(d: Date): string {
|
function format(d: Date): string {
|
||||||
const p = (n: number) => String(n).padStart(2, '0')
|
const p = (n: number) => String(n).padStart(2, '0')
|
||||||
return `${p(d.getUTCHours())}:${p(d.getUTCMinutes())}:${p(d.getUTCSeconds())} UTC`
|
return `${p(d.getHours())}:${p(d.getMinutes())}:${p(d.getSeconds())} ${zoneLabel(d)}`
|
||||||
}
|
}
|
||||||
|
|
||||||
/** Live UTC readout, tabular digits, ticking once a second. */
|
/** The full zone name, for the title — "Europe/Moscow" says more than "+3" does. */
|
||||||
|
function zoneName(): string {
|
||||||
|
try {
|
||||||
|
return Intl.DateTimeFormat().resolvedOptions().timeZone || ''
|
||||||
|
} catch {
|
||||||
|
return ''
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/** Live local readout, tabular digits, ticking once a second. */
|
||||||
export function Clock({ className }: { className?: string }) {
|
export function Clock({ className }: { className?: string }) {
|
||||||
const [now, setNow] = useState(() => format(new Date()))
|
const [now, setNow] = useState(() => format(new Date()))
|
||||||
|
|
||||||
@@ -14,5 +52,13 @@ export function Clock({ className }: { className?: string }) {
|
|||||||
return () => window.clearInterval(id)
|
return () => window.clearInterval(id)
|
||||||
}, [])
|
}, [])
|
||||||
|
|
||||||
return <span className={['clock', className].filter(Boolean).join(' ')}>{now}</span>
|
const zone = zoneName()
|
||||||
|
return (
|
||||||
|
<span
|
||||||
|
className={['clock', className].filter(Boolean).join(' ')}
|
||||||
|
title={zone ? `Your device's clock — ${zone}. Every time in the panel is shown in this zone.` : undefined}
|
||||||
|
>
|
||||||
|
{now}
|
||||||
|
</span>
|
||||||
|
)
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -77,6 +77,39 @@ export function fmtDateTime(unix: number): string {
|
|||||||
return d && t ? `${d}, ${t}` : d || t
|
return d && t ? `${d}, ${t}` : d || t
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// --- remote-list freshness ---------------------------------------------------
|
||||||
|
// A url/geo-sourced list re-fetches on a cadence and the engine reports when it
|
||||||
|
// last pulled (GET /api/ruleset/status). Routing shows this for rule-sets and DNS
|
||||||
|
// shows it for blocklists, so the two readings live here and cannot drift apart.
|
||||||
|
|
||||||
|
/** "updated 3h ago" / "never updated" for a remote list's last fetch (RFC3339). */
|
||||||
|
export function relFetch(iso: string): string {
|
||||||
|
if (!iso) return 'never updated'
|
||||||
|
const t = Date.parse(iso)
|
||||||
|
if (Number.isNaN(t)) return 'never updated'
|
||||||
|
const s = Math.max(0, Math.floor((Date.now() - t) / 1000))
|
||||||
|
if (s < 45) return 'updated just now'
|
||||||
|
const m = Math.floor(s / 60)
|
||||||
|
if (m < 60) return `updated ${m}m ago`
|
||||||
|
const h = Math.floor(m / 60)
|
||||||
|
if (h < 24) return `updated ${h}h ago`
|
||||||
|
const d = Math.floor(h / 24)
|
||||||
|
return `updated ${d}d ago`
|
||||||
|
}
|
||||||
|
|
||||||
|
/** "every 24h" for an auto-update cadence in seconds ("" when there is none). */
|
||||||
|
export function everyLabel(sec: number): string {
|
||||||
|
if (!sec || sec <= 0) return ''
|
||||||
|
if (sec % 3600 === 0) {
|
||||||
|
const h = sec / 3600
|
||||||
|
if (h < 48) return `every ${h}h`
|
||||||
|
if (sec % 86400 === 0) return `every ${sec / 86400}d`
|
||||||
|
return `every ${h}h`
|
||||||
|
}
|
||||||
|
if (sec % 60 === 0) return `every ${sec / 60}m`
|
||||||
|
return `every ${sec}s`
|
||||||
|
}
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* A coarse "how long until / since" reading for a unix deadline, relative to now.
|
* A coarse "how long until / since" reading for a unix deadline, relative to now.
|
||||||
*
|
*
|
||||||
|
|||||||
+44
-16
@@ -194,6 +194,7 @@ const CONFIG: Model = {
|
|||||||
// to an official remote list; the others are the usual url / inline lists.
|
// to an official remote list; the others are the usual url / inline lists.
|
||||||
Blocklists: [
|
Blocklists: [
|
||||||
{ Name: 'StevenBlack', Enabled: true, Source: 'url', URL: 'https://raw.githubusercontent.com/StevenBlack/hosts/master/hosts', Response: 'nxdomain', UpdateInterval: '24h' },
|
{ Name: 'StevenBlack', Enabled: true, Source: 'url', URL: 'https://raw.githubusercontent.com/StevenBlack/hosts/master/hosts', Response: 'nxdomain', UpdateInterval: '24h' },
|
||||||
|
{ Name: 'oisd-basic', Enabled: true, Source: 'url', URL: 'https://big.oisd.nl/domainswild', Response: 'nxdomain', UpdateInterval: '24h' },
|
||||||
{ Name: 'telegram-block', Enabled: false, Source: 'geosite', Categories: ['telegram'], Response: 'nxdomain', UpdateInterval: '24h' },
|
{ Name: 'telegram-block', Enabled: false, Source: 'geosite', Categories: ['telegram'], Response: 'nxdomain', UpdateInterval: '24h' },
|
||||||
],
|
],
|
||||||
Resolvers: [
|
Resolvers: [
|
||||||
@@ -323,6 +324,24 @@ const RULESET_STATUS: RulesetStatus[] = [
|
|||||||
rule_count: 903,
|
rule_count: 903,
|
||||||
},
|
},
|
||||||
{ tag: 'rs-ru-geoip-ru', name: 'ru-geoip', category: 'ru', kind: 'ruleset', remote: true, last_updated: '', interval_seconds: 86_400, rule_count: 0 },
|
{ tag: 'rs-ru-geoip-ru', name: 'ru-geoip', category: 'ru', kind: 'ruleset', remote: true, last_updated: '', interval_seconds: 86_400, rule_count: 0 },
|
||||||
|
// Blocklists report through the same endpoint under `bl-<name>`, which the DNS
|
||||||
|
// page never asked for — so a list that has NEVER been fetched still read
|
||||||
|
// "filtering". StevenBlack is that case here; oisd-basic is the healthy one, so
|
||||||
|
// both readings are exercisable offline.
|
||||||
|
{ tag: 'bl-StevenBlack', name: 'StevenBlack', category: '', kind: 'blocklist', remote: true, last_updated: '', interval_seconds: 86_400, rule_count: 0 },
|
||||||
|
{
|
||||||
|
tag: 'bl-oisd-basic',
|
||||||
|
name: 'oisd-basic',
|
||||||
|
category: '',
|
||||||
|
kind: 'blocklist',
|
||||||
|
remote: true,
|
||||||
|
last_updated: new Date(Date.now() - 6 * 3600_000).toISOString(),
|
||||||
|
interval_seconds: 86_400,
|
||||||
|
rule_count: 218_431,
|
||||||
|
},
|
||||||
|
// Disabled in CONFIG, so the row reads "off" whatever this says — it exists to
|
||||||
|
// prove the row does not start claiming things the moment a status appears.
|
||||||
|
{ tag: 'bl-telegram-block-telegram', name: 'telegram-block', category: 'telegram', kind: 'blocklist', remote: true, last_updated: '', interval_seconds: 86_400, rule_count: 0 },
|
||||||
]
|
]
|
||||||
|
|
||||||
/** GET /api/rules/reachability. Mirrors the daemon's analysis over CONFIG.Rules:
|
/** GET /api/rules/reachability. Mirrors the daemon's analysis over CONFIG.Rules:
|
||||||
@@ -1199,14 +1218,19 @@ function healthList(): GroupHealth[] {
|
|||||||
* Per-hop health, keyed by chain name — what the observatory measured at each
|
* Per-hop health, keyed by chain name — what the observatory measured at each
|
||||||
* position of the path, in WIRE order.
|
* position of the path, in WIRE order.
|
||||||
*
|
*
|
||||||
* `ewan-wg-subs` is the fixture that matters. Its hop 1 is the AmneziaWG node and
|
* `ewan-wg-subs` is the fixture that matters, and it encodes the ORDERED WALK.
|
||||||
* answers; hops 2 and 4 are subscription groups that answer THROUGH it; hop 3 is
|
* Hop 1 is the WireGuard node and answers; hop 2 is a subscription group whose
|
||||||
* a group whose members all time out at that position. Note that hop 4 rolls up
|
* copies answer THROUGH it — 119 of 122 tested alive, which is the reading only a
|
||||||
* `via-tunnel`, the same group whose standalone card reads 0 of 6 alive — alive
|
* per-hop probe can produce, since the same members are dialled differently on
|
||||||
* as a hop, dead on its own, both true, because they measure different dial
|
* their own card. Hop 3 is a group whose members all time out at that position,
|
||||||
* paths. That contradiction is the whole point of measuring per hop.
|
* and the walk STOPS there: hop 4 is dialled through hop 3, so it was never
|
||||||
|
* dialled at all. It comes back `untested` with `blocked_by` naming hop 3, its
|
||||||
|
* counters zeroed, and `selected` still set — the wrapper has a pick, nothing
|
||||||
|
* crossed it to measure. A dead hop with a live hop under it is not in this
|
||||||
|
* fixture because the daemon can no longer produce one.
|
||||||
*
|
*
|
||||||
* `sub-fresh` is used but never yet reached: every hop untested, nothing dead.
|
* `sub-fresh` is used but never yet reached: every hop untested, nothing dead,
|
||||||
|
* no block — the other reason a lamp is unlit, and the one that fixes itself.
|
||||||
* `relay` is absent from this map on purpose — an unused chain is never
|
* `relay` is absent from this map on purpose — an unused chain is never
|
||||||
* materialised, so the daemon sends no `hops` key at all, which is "nothing
|
* materialised, so the daemon sends no `hops` key at all, which is "nothing
|
||||||
* measured", not "no hops".
|
* measured", not "no hops".
|
||||||
@@ -1215,8 +1239,8 @@ const CHAIN_HOPS: Record<string, ChainHopHealth[]> = {
|
|||||||
'ewan-wg-subs': [
|
'ewan-wg-subs': [
|
||||||
{ index: 1, tag: 'chain-ewan-wg-subs-h1', kind: 'node', exit: false, state: 'alive', delay_ms: 41, age_seconds: 22, selected: '', total: 1, tested: 1, alive: 1, dead: 0, untested: 0 },
|
{ index: 1, tag: 'chain-ewan-wg-subs-h1', kind: 'node', exit: false, state: 'alive', delay_ms: 41, age_seconds: 22, selected: '', total: 1, tested: 1, alive: 1, dead: 0, untested: 0 },
|
||||||
{ index: 2, tag: 'chain-ewan-wg-subs-h2', kind: 'group', exit: false, state: 'alive', delay_ms: 96, age_seconds: 18, selected: '🇳🇱 Amsterdam-01', total: 298, tested: 122, alive: 119, dead: 3, untested: 176 },
|
{ index: 2, tag: 'chain-ewan-wg-subs-h2', kind: 'group', exit: false, state: 'alive', delay_ms: 96, age_seconds: 18, selected: '🇳🇱 Amsterdam-01', total: 298, tested: 122, alive: 119, dead: 3, untested: 176 },
|
||||||
{ index: 3, tag: 'chain-ewan-wg-subs-h3', kind: 'group', exit: false, state: 'dead', delay_ms: 0, age_seconds: 15, selected: '', total: 24, tested: 24, alive: 0, dead: 24, untested: 0 },
|
{ index: 3, tag: 'chain-ewan-wg-subs-h3', kind: 'group', exit: false, state: 'dead', delay_ms: 0, age_seconds: 15, selected: '', total: 2, tested: 2, alive: 0, dead: 2, untested: 0 },
|
||||||
{ index: 4, tag: 'chain-ewan-wg-subs-h4', kind: 'group', exit: true, state: 'alive', delay_ms: 148, age_seconds: 19, selected: '🇸🇬 Singapore-09', total: 6, tested: 6, alive: 6, dead: 0, untested: 0 },
|
{ index: 4, tag: 'chain-ewan-wg-subs-h4', kind: 'group', exit: true, state: 'untested', delay_ms: 0, age_seconds: -1, selected: '🇸🇬 Singapore-09', total: 6, tested: 0, alive: 0, dead: 0, untested: 6, blocked_by: { index: 3, tag: 'chain-ewan-wg-subs-h3' } },
|
||||||
],
|
],
|
||||||
'sub-fresh': [
|
'sub-fresh': [
|
||||||
{ index: 1, tag: 'chain-sub-fresh-h1', kind: 'node', exit: false, state: 'untested', delay_ms: 0, age_seconds: -1, selected: '', total: 1, tested: 0, alive: 0, dead: 0, untested: 1 },
|
{ index: 1, tag: 'chain-sub-fresh-h1', kind: 'node', exit: false, state: 'untested', delay_ms: 0, age_seconds: -1, selected: '', total: 1, tested: 0, alive: 0, dead: 0, untested: 1 },
|
||||||
@@ -1240,8 +1264,9 @@ function chainHealthList(): ChainHealth[] {
|
|||||||
|
|
||||||
/** A chain is "used" when some enabled routing rule (or Final, or a DNS detour)
|
/** A chain is "used" when some enabled routing rule (or Final, or a DNS detour)
|
||||||
* targets `chain:<name>` — the same reachability the daemon's observatory derives.
|
* targets `chain:<name>` — the same reachability the daemon's observatory derives.
|
||||||
* The mock's rules never target a chain, so every chain reads used=false; a real
|
* Two of the mock's rules do (`media-via-chain` → ewan-wg-subs, `spare-via-chain`
|
||||||
* config would mark the ones rules point at used=true. */
|
* → sub-fresh), so those two chains read used=true and get a hop rail; `relay`
|
||||||
|
* is targeted by nothing and reads used=false, which is the unused note. */
|
||||||
function chainUsed(name: string): boolean {
|
function chainUsed(name: string): boolean {
|
||||||
const target = `chain:${name}`
|
const target = `chain:${name}`
|
||||||
return (CONFIG.Rules ?? []).some(
|
return (CONFIG.Rules ?? []).some(
|
||||||
@@ -1328,17 +1353,20 @@ const GROUP_TEST_SHAPE: Record<string, Omit<GroupTestResult, 'group' | 'tested_u
|
|||||||
error: '',
|
error: '',
|
||||||
},
|
},
|
||||||
// The chain, and the pairing that makes the whole feature worth building. A
|
// The chain, and the pairing that makes the whole feature worth building. A
|
||||||
// chain is one series path, so with hop 3 dead the end-to-end probe CANNOT
|
// chain is one series path, so with hop 3 dead the end-to-end probe is never
|
||||||
// succeed — this row and the hop rail have to tell one story, not two. The
|
// even attempted — the daemon stops walking there. This row and the hop rail
|
||||||
// row says the path is down; the rail says which of the four hops did it,
|
// therefore have to tell one story, not two: both name hop 3, and neither
|
||||||
// which is the part nobody could see before.
|
// offers hop 4 as a second suspect. Note the row does NOT say "the probe
|
||||||
|
// failed" — no probe of this chain's exit ran at all — which is why the daemon
|
||||||
|
// has a separate message for it.
|
||||||
'ewan-wg-subs': {
|
'ewan-wg-subs': {
|
||||||
selected: '',
|
selected: '',
|
||||||
delay_ms: 0,
|
delay_ms: 0,
|
||||||
exit_ip: '',
|
exit_ip: '',
|
||||||
exit_country: '',
|
exit_country: '',
|
||||||
ok: false,
|
ok: false,
|
||||||
error: 'the observatory’s probe through this path failed',
|
error:
|
||||||
|
'hop 3 of this chain was probed and did not answer, so nothing reaches the exit through it — fix that hop first',
|
||||||
},
|
},
|
||||||
// The one real health failure in the fixture: the observatory's probe ran along
|
// The one real health failure in the fixture: the observatory's probe ran along
|
||||||
// this path and did not come back.
|
// this path and did not come back.
|
||||||
|
|||||||
+67
-56
@@ -11,6 +11,8 @@ import {
|
|||||||
ApiError,
|
ApiError,
|
||||||
} from '../api'
|
} from '../api'
|
||||||
import type { Globals, Status } from '../api'
|
import type { Globals, Status } from '../api'
|
||||||
|
import { engineReadout } from '../planeState'
|
||||||
|
import { onPendingConfirmExpire, usePendingConfirm } from '../pendingConfirm'
|
||||||
|
|
||||||
// Short, readable config hash — drops the "sha256:" prefix like the footer does.
|
// Short, readable config hash — drops the "sha256:" prefix like the footer does.
|
||||||
function short(hash: string): string {
|
function short(hash: string): string {
|
||||||
@@ -25,13 +27,6 @@ function msg(e: unknown): string {
|
|||||||
|
|
||||||
type Busy = 'apply' | 'confirm' | 'rollback' | null
|
type Busy = 'apply' | 'confirm' | 'rollback' | null
|
||||||
|
|
||||||
/** A pending commit-confirm window: the daemon has armed an auto-rollback. */
|
|
||||||
interface Armed {
|
|
||||||
total: number // the ConfirmTimeout the window started with
|
|
||||||
remaining: number // seconds left before the daemon reverts
|
|
||||||
appliedHash: string // the hash that went live on apply (the "after" of apply)
|
|
||||||
}
|
|
||||||
|
|
||||||
type ActionKind = 'apply' | 'confirm' | 'rollback' | 'expire'
|
type ActionKind = 'apply' | 'confirm' | 'rollback' | 'expire'
|
||||||
interface ActionResult {
|
interface ActionResult {
|
||||||
kind: ActionKind
|
kind: ActionKind
|
||||||
@@ -57,7 +52,11 @@ export default function Apply() {
|
|||||||
const [configError, setConfigError] = useState<string | null>(null)
|
const [configError, setConfigError] = useState<string | null>(null)
|
||||||
|
|
||||||
const [busy, setBusy] = useState<Busy>(null)
|
const [busy, setBusy] = useState<Busy>(null)
|
||||||
const [armed, setArmed] = useState<Armed | null>(null)
|
// The armed window is app-wide state, not this page's: it is recorded by the
|
||||||
|
// api layer on every apply and survives a reload. Keeping it local is what made
|
||||||
|
// refreshing this tab lose both the countdown and the only button that could
|
||||||
|
// stop it. See pendingConfirm.ts.
|
||||||
|
const armed = usePendingConfirm()
|
||||||
const [result, setResult] = useState<ActionResult | null>(null)
|
const [result, setResult] = useState<ActionResult | null>(null)
|
||||||
const [confirmingRollback, setConfirmingRollback] = useState(false)
|
const [confirmingRollback, setConfirmingRollback] = useState(false)
|
||||||
|
|
||||||
@@ -104,32 +103,30 @@ export default function Apply() {
|
|||||||
void loadConfig()
|
void loadConfig()
|
||||||
}, [loadConfig])
|
}, [loadConfig])
|
||||||
|
|
||||||
// ---- commit-confirm countdown: a calm 1s numeric tick, effect-scoped so the
|
// The window running out is the daemon reverting on its own — observe it and
|
||||||
// timer is always cleared on unmount / confirm / rollback (no leaked intervals) ----
|
// say so. The countdown itself ticks inside usePendingConfirm; this only reacts
|
||||||
useEffect(() => {
|
// to the end of it, and the store makes sure that fires exactly once even with
|
||||||
if (!armed) return
|
// the app-wide band mounted alongside.
|
||||||
if (armed.remaining <= 0) {
|
const liveHashRef = useRef('')
|
||||||
// Window elapsed — the daemon reverts to last-good on its own. Observe it.
|
liveHashRef.current = status?.hash ?? ''
|
||||||
const before = armed.appliedHash
|
useEffect(
|
||||||
setArmed(null)
|
() =>
|
||||||
flash('Auto-rolled back')
|
onPendingConfirmExpire(() => {
|
||||||
void (async () => {
|
const before = liveHashRef.current
|
||||||
const after = (await refreshStatus())?.hash ?? ''
|
flash('Auto-rolled back')
|
||||||
setResult({
|
void (async () => {
|
||||||
kind: 'expire',
|
const after = (await refreshStatus())?.hash ?? ''
|
||||||
tone: 'warn',
|
setResult({
|
||||||
text: 'Confirm window elapsed — daemon auto-rolled back to last-good config.',
|
kind: 'expire',
|
||||||
before,
|
tone: 'warn',
|
||||||
after,
|
text: 'Confirm window elapsed — daemon auto-rolled back to last-good config.',
|
||||||
})
|
before,
|
||||||
})()
|
after,
|
||||||
return
|
})
|
||||||
}
|
})()
|
||||||
const id = window.setTimeout(() => {
|
}),
|
||||||
setArmed((a) => (a ? { ...a, remaining: a.remaining - 1 } : a))
|
[flash, refreshStatus],
|
||||||
}, 1000)
|
)
|
||||||
return () => window.clearTimeout(id)
|
|
||||||
}, [armed, flash, refreshStatus])
|
|
||||||
|
|
||||||
const confirmWindow = globals?.ConfirmTimeout ?? 0
|
const confirmWindow = globals?.ConfirmTimeout ?? 0
|
||||||
|
|
||||||
@@ -146,8 +143,9 @@ export default function Apply() {
|
|||||||
return
|
return
|
||||||
}
|
}
|
||||||
const after = (await refreshStatus())?.hash ?? before
|
const after = (await refreshStatus())?.hash ?? before
|
||||||
|
// The window itself was recorded by api.apply(); this branch only writes the
|
||||||
|
// readout for it.
|
||||||
if (r.changed && confirmWindow > 0) {
|
if (r.changed && confirmWindow > 0) {
|
||||||
setArmed({ total: confirmWindow, remaining: confirmWindow, appliedHash: after })
|
|
||||||
setResult({
|
setResult({
|
||||||
kind: 'apply',
|
kind: 'apply',
|
||||||
tone: 'good',
|
tone: 'good',
|
||||||
@@ -179,7 +177,8 @@ export default function Apply() {
|
|||||||
const doConfirm = useCallback(async () => {
|
const doConfirm = useCallback(async () => {
|
||||||
const before = status?.hash ?? ''
|
const before = status?.hash ?? ''
|
||||||
setBusy('confirm')
|
setBusy('confirm')
|
||||||
setArmed(null) // stop the countdown immediately; confirm cancels the auto-rollback
|
// api.confirm() clears the shared window on success — the countdown stops the
|
||||||
|
// moment the daemon agrees, not the moment we asked.
|
||||||
try {
|
try {
|
||||||
const r = await apiConfirm()
|
const r = await apiConfirm()
|
||||||
if (r.error) {
|
if (r.error) {
|
||||||
@@ -208,7 +207,7 @@ export default function Apply() {
|
|||||||
const before = status?.hash ?? ''
|
const before = status?.hash ?? ''
|
||||||
setConfirmingRollback(false)
|
setConfirmingRollback(false)
|
||||||
setBusy('rollback')
|
setBusy('rollback')
|
||||||
setArmed(null) // rolling back also cancels any pending confirm window
|
// api.rollback() clears the shared window on success (rolling back ends it).
|
||||||
try {
|
try {
|
||||||
const r = await apiRollback()
|
const r = await apiRollback()
|
||||||
if (r.error) {
|
if (r.error) {
|
||||||
@@ -237,19 +236,26 @@ export default function Apply() {
|
|||||||
}, [status, flash, refreshStatus])
|
}, [status, flash, refreshStatus])
|
||||||
|
|
||||||
// ---- derived display state (mirrors Overview's LED semantics) ----
|
// ---- derived display state (mirrors Overview's LED semantics) ----
|
||||||
const killArmed = globals ? globals.KillSwitch === 'closed' : false
|
//
|
||||||
const engineVariant: LedVariant = !status
|
// The LIVE kill-switch wins over the saved one, exactly as on Overview: this row
|
||||||
? 'off'
|
// is a status readout, and the config on disk can already differ from what is
|
||||||
: status.running && status.active
|
// installed. Falls back to the config only while /api/status is unread.
|
||||||
? 'on'
|
const killArmed = (status?.kill_switch ?? globals?.KillSwitch ?? 'closed') === 'closed'
|
||||||
: status.running
|
// Every engine mark on this page comes from ONE reading, and that reading is
|
||||||
? 'amber'
|
// able to say "stopped" — see planeState.engineState for why `status.running`
|
||||||
: 'crit'
|
// could not. This page is where someone lands when the network is down; three
|
||||||
const dataVariant: LedVariant = status?.table ? 'on' : status?.running ? 'amber' : 'off'
|
// green lamps here were the difference between "I broke it" and "nothing broke".
|
||||||
|
const engine = engineReadout(status)
|
||||||
|
const engineVariant: LedVariant = engine.variant
|
||||||
|
// No nft table means there is no data plane at all. Under a fail-closed switch
|
||||||
|
// that is a leak (crit); under an open one it is the documented choice (amber).
|
||||||
|
// It used to go amber whenever `running` was true — i.e. always — and unlit
|
||||||
|
// otherwise, so the one state worth shouting about had no colour of its own.
|
||||||
|
const dataVariant: LedVariant = status?.table ? 'on' : !status ? 'off' : killArmed ? 'crit' : 'amber'
|
||||||
const configVariant: LedVariant = status?.enabled ? 'on' : 'amber'
|
const configVariant: LedVariant = status?.enabled ? 'on' : 'amber'
|
||||||
|
|
||||||
const liveHash = short(status?.hash ?? '')
|
const liveHash = short(status?.hash ?? '')
|
||||||
const pct = armed ? Math.max(0, Math.round((armed.remaining / armed.total) * 100)) : 0
|
const pct = armed ? Math.max(0, Math.round((armed.remaining / armed.pending.total) * 100)) : 0
|
||||||
|
|
||||||
// Only offer rollback when the daemon says one would revert something: an armed
|
// Only offer rollback when the daemon says one would revert something: an armed
|
||||||
// commit-confirm snapshot, or an engine last-good predecessor. When false there
|
// commit-confirm snapshot, or an engine last-good predecessor. When false there
|
||||||
@@ -264,9 +270,7 @@ export default function Apply() {
|
|||||||
label="Engine"
|
label="Engine"
|
||||||
variant={engineVariant}
|
variant={engineVariant}
|
||||||
pulse={engineVariant === 'on'}
|
pulse={engineVariant === 'on'}
|
||||||
value={
|
value={engine.word}
|
||||||
!status ? 'checking…' : status.running ? (status.active ? 'active' : 'idle') : 'stopped'
|
|
||||||
}
|
|
||||||
/>
|
/>
|
||||||
<StatusPip
|
<StatusPip
|
||||||
label="Config"
|
label="Config"
|
||||||
@@ -310,8 +314,12 @@ export default function Apply() {
|
|||||||
unit="· sha256"
|
unit="· sha256"
|
||||||
led={{ variant: configVariant }}
|
led={{ variant: configVariant }}
|
||||||
rows={[
|
rows={[
|
||||||
{ k: 'engine', v: status?.running ? 'running' : 'stopped', hot: !status?.running },
|
{ k: 'engine', v: engine.word, hot: engineVariant === 'crit' },
|
||||||
{ k: 'data plane', v: status?.table ? 'nft installed' : 'no table' },
|
{
|
||||||
|
k: 'data plane',
|
||||||
|
v: status?.table ? 'nft installed' : 'no table',
|
||||||
|
hot: dataVariant === 'crit',
|
||||||
|
},
|
||||||
{ k: 'kill-switch', v: killArmed ? 'fail-closed' : 'open', hot: !killArmed },
|
{ k: 'kill-switch', v: killArmed ? 'fail-closed' : 'open', hot: !killArmed },
|
||||||
]}
|
]}
|
||||||
/>
|
/>
|
||||||
@@ -325,7 +333,7 @@ export default function Apply() {
|
|||||||
}
|
}
|
||||||
led={{ variant: engineVariant }}
|
led={{ variant: engineVariant }}
|
||||||
rows={[
|
rows={[
|
||||||
{ k: 'state', v: !status ? 'checking…' : status.active ? 'active' : 'idle' },
|
{ k: 'state', v: engine.word, hot: engineVariant === 'crit' },
|
||||||
{ k: 'config', v: status?.enabled ? 'enabled' : 'disabled' },
|
{ k: 'config', v: status?.enabled ? 'enabled' : 'disabled' },
|
||||||
{ k: 'schema', v: globals ? `v${globals.SchemaVersion}` : '—' },
|
{ k: 'schema', v: globals ? `v${globals.SchemaVersion}` : '—' },
|
||||||
]}
|
]}
|
||||||
@@ -362,9 +370,12 @@ export default function Apply() {
|
|||||||
</div>
|
</div>
|
||||||
<div className="cc-info">
|
<div className="cc-info">
|
||||||
<p className="cc-copy">
|
<p className="cc-copy">
|
||||||
Applied config <span className="mono">{short(armed.appliedHash)}</span> is live but
|
{/* The live hash IS the applied one while a window is open — that
|
||||||
not yet kept. Confirm to keep it — otherwise the daemon rolls back to the last-good
|
is what "live but not kept" means — so the readout survives a
|
||||||
config when the timer hits zero.
|
reload instead of depending on what this tab remembers. */}
|
||||||
|
Applied config <span className="mono">{liveHash}</span> is live but not yet kept.
|
||||||
|
Confirm to keep it — otherwise the daemon rolls back to the last-good config when
|
||||||
|
the timer hits zero.
|
||||||
</p>
|
</p>
|
||||||
<div className="cc-bar" aria-hidden="true">
|
<div className="cc-bar" aria-hidden="true">
|
||||||
<span className="cc-bar-fill" style={{ width: `${pct}%` }} />
|
<span className="cc-bar-fill" style={{ width: `${pct}%` }} />
|
||||||
|
|||||||
+80
-1
@@ -387,6 +387,79 @@
|
|||||||
.dns-row-state[data-active='on'] {
|
.dns-row-state[data-active='on'] {
|
||||||
color: var(--led-on);
|
color: var(--led-on);
|
||||||
}
|
}
|
||||||
|
/* A list that is switched on but has nothing loaded is not "off" and is certainly
|
||||||
|
not "filtering" — warn semantics, the same amber the badges use. */
|
||||||
|
.dns-row-state[data-active='warn'] {
|
||||||
|
color: var(--amber);
|
||||||
|
}
|
||||||
|
|
||||||
|
/* ---- remote-list freshness (mirrors the rule-set rows on Routing) ---- */
|
||||||
|
.dns-row-sync {
|
||||||
|
display: flex;
|
||||||
|
align-items: center;
|
||||||
|
flex-wrap: wrap;
|
||||||
|
gap: 10px;
|
||||||
|
margin-top: 3px;
|
||||||
|
}
|
||||||
|
.dns-sync-fresh {
|
||||||
|
font-family: var(--font-mono);
|
||||||
|
font-size: 11px;
|
||||||
|
color: var(--dim);
|
||||||
|
}
|
||||||
|
.dns-sync-fresh[data-never='y'] {
|
||||||
|
color: var(--amber);
|
||||||
|
}
|
||||||
|
.dns-sync-every,
|
||||||
|
.dns-sync-rules {
|
||||||
|
font-family: var(--font-mono);
|
||||||
|
font-size: 10.5px;
|
||||||
|
letter-spacing: 0.02em;
|
||||||
|
color: var(--faint);
|
||||||
|
}
|
||||||
|
.dns-sync-every::before {
|
||||||
|
content: '↻ ';
|
||||||
|
}
|
||||||
|
.dns-sync-update {
|
||||||
|
display: inline-flex;
|
||||||
|
align-items: center;
|
||||||
|
gap: 6px;
|
||||||
|
padding: 3px 10px;
|
||||||
|
border: 1px solid var(--accent-soft);
|
||||||
|
border-radius: 5px;
|
||||||
|
background: var(--raised);
|
||||||
|
color: var(--accent);
|
||||||
|
font-family: var(--font-mono);
|
||||||
|
font-size: 10px;
|
||||||
|
letter-spacing: var(--track-label);
|
||||||
|
text-transform: uppercase;
|
||||||
|
cursor: pointer;
|
||||||
|
transition: color 0.12s, border-color 0.12s, background 0.12s;
|
||||||
|
}
|
||||||
|
.dns-sync-update:hover:not(:disabled) {
|
||||||
|
border-color: var(--accent);
|
||||||
|
background: color-mix(in srgb, var(--accent) 12%, transparent);
|
||||||
|
}
|
||||||
|
.dns-sync-update:focus-visible {
|
||||||
|
outline: 2px solid var(--accent);
|
||||||
|
outline-offset: 2px;
|
||||||
|
}
|
||||||
|
.dns-sync-update:disabled {
|
||||||
|
opacity: 0.6;
|
||||||
|
cursor: not-allowed;
|
||||||
|
}
|
||||||
|
.dns-sync-spin {
|
||||||
|
width: 10px;
|
||||||
|
height: 10px;
|
||||||
|
border: 2px solid color-mix(in srgb, var(--accent) 35%, transparent);
|
||||||
|
border-top-color: var(--accent);
|
||||||
|
border-radius: 50%;
|
||||||
|
animation: dns-sync-spin 0.7s linear infinite;
|
||||||
|
}
|
||||||
|
@keyframes dns-sync-spin {
|
||||||
|
to {
|
||||||
|
transform: rotate(360deg);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
/* badge — groove-bordered, not orange (accent stays reserved) */
|
/* badge — groove-bordered, not orange (accent stays reserved) */
|
||||||
.dns-badge {
|
.dns-badge {
|
||||||
@@ -646,9 +719,15 @@
|
|||||||
.dns-skel {
|
.dns-skel {
|
||||||
animation: none;
|
animation: none;
|
||||||
}
|
}
|
||||||
|
/* No spin under reduced motion — the static ring + "Updating…" label carry it. */
|
||||||
|
.dns-sync-spin {
|
||||||
|
animation: none;
|
||||||
|
border-top-color: color-mix(in srgb, var(--accent) 35%, transparent);
|
||||||
|
}
|
||||||
.dns-chip,
|
.dns-chip,
|
||||||
.dns-input,
|
.dns-input,
|
||||||
.dns-seg-btn {
|
.dns-seg-btn,
|
||||||
|
.dns-sync-update {
|
||||||
transition: none;
|
transition: none;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
+205
-10
@@ -1,8 +1,16 @@
|
|||||||
import './DNS.css'
|
import './DNS.css'
|
||||||
import { useCallback, useEffect, useMemo, useRef, useState } from 'react'
|
import { useCallback, useEffect, useMemo, useRef, useState } from 'react'
|
||||||
import { Button, CatSuggest, Led, SrcPicker, Toggle, useConfirm } from '../components'
|
import { Button, CatSuggest, Led, SrcPicker, Toggle, useConfirm } from '../components'
|
||||||
import { apply as apiApply, getConfig, putConfig, ApiError } from '../api'
|
import {
|
||||||
import type { Alert, DNSRule, Model, Resolver } from '../api'
|
apply as apiApply,
|
||||||
|
getConfig,
|
||||||
|
getRulesetStatus,
|
||||||
|
putConfig,
|
||||||
|
updateRuleset as apiUpdateRuleset,
|
||||||
|
ApiError,
|
||||||
|
} from '../api'
|
||||||
|
import type { Alert, DNSRule, Model, Resolver, RulesetStatus } from '../api'
|
||||||
|
import { everyLabel, relFetch } from '../format'
|
||||||
|
|
||||||
// The DNS / Blocklists page is a thin editor over the desired-state Model —
|
// The DNS / Blocklists page is a thin editor over the desired-state Model —
|
||||||
// exactly like Nodes.tsx. Every edit rewrites the relevant slice in-place, PUTs
|
// exactly like Nodes.tsx. Every edit rewrites the relevant slice in-place, PUTs
|
||||||
@@ -310,6 +318,68 @@ export default function DNS() {
|
|||||||
[config],
|
[config],
|
||||||
)
|
)
|
||||||
|
|
||||||
|
// ---- did the lists actually LOAD? -----------------------------------------
|
||||||
|
//
|
||||||
|
// A blocklist row said "filtering" whenever the list and the master switch were
|
||||||
|
// both on. Neither of those is evidence that anything is being blocked: a
|
||||||
|
// url/geosite list is fetched by the engine, the daemon treats a failed fetch as
|
||||||
|
// a CRITICAL apply finding, and the row went on saying "filtering" through it.
|
||||||
|
// The Routing page had already been given this reading for rule-sets — the same
|
||||||
|
// endpoint, the same tags (`bl-<name>` / `al-<name>`) — and the DNS page never
|
||||||
|
// asked. Slow poll: lists refresh on a ~24h cadence, so 15s only has to catch a
|
||||||
|
// manual Update-now. Grouped by NAME because a geosite list with N categories
|
||||||
|
// reports N records.
|
||||||
|
const [listStatus, setListStatus] = useState<Map<string, RulesetStatus[]>>(new Map())
|
||||||
|
const [updatingLists, setUpdatingLists] = useState<Set<string>>(new Set())
|
||||||
|
const loadListStatus = useCallback(async () => {
|
||||||
|
try {
|
||||||
|
const all = await getRulesetStatus()
|
||||||
|
const m = new Map<string, RulesetStatus[]>()
|
||||||
|
for (const s of all) {
|
||||||
|
if (s.kind !== 'blocklist' && s.kind !== 'allowlist') continue
|
||||||
|
const key = `${s.kind}:${s.name}`
|
||||||
|
const arr = m.get(key)
|
||||||
|
if (arr) arr.push(s)
|
||||||
|
else m.set(key, [s])
|
||||||
|
}
|
||||||
|
setListStatus(m)
|
||||||
|
} catch {
|
||||||
|
// Engine stopped or an older daemon — keep the last reading. The row falls
|
||||||
|
// back to "load not reported", which claims nothing either way.
|
||||||
|
}
|
||||||
|
}, [])
|
||||||
|
useEffect(() => {
|
||||||
|
void loadListStatus()
|
||||||
|
const id = window.setInterval(() => void loadListStatus(), 15000)
|
||||||
|
return () => window.clearInterval(id)
|
||||||
|
}, [loadListStatus])
|
||||||
|
|
||||||
|
const updateList = useCallback(
|
||||||
|
async (kind: 'blocklist' | 'allowlist', name: string) => {
|
||||||
|
const key = `${kind}:${name}`
|
||||||
|
setUpdatingLists((prev) => new Set(prev).add(key))
|
||||||
|
try {
|
||||||
|
// One geo list can hold several categories, each its own engine tag.
|
||||||
|
const recs = listStatus.get(key) ?? []
|
||||||
|
const tags = recs.length
|
||||||
|
? recs.map((r) => r.tag)
|
||||||
|
: [`${kind === 'blocklist' ? 'bl' : 'al'}-${name}`]
|
||||||
|
for (const tag of tags) await apiUpdateRuleset(tag)
|
||||||
|
await loadListStatus()
|
||||||
|
flash(`${name} refreshed`)
|
||||||
|
} catch (e) {
|
||||||
|
flash(`Refresh failed — ${errText(e)}`)
|
||||||
|
} finally {
|
||||||
|
setUpdatingLists((prev) => {
|
||||||
|
const next = new Set(prev)
|
||||||
|
next.delete(key)
|
||||||
|
return next
|
||||||
|
})
|
||||||
|
}
|
||||||
|
},
|
||||||
|
[listStatus, loadListStatus, flash],
|
||||||
|
)
|
||||||
|
|
||||||
const blOn = blocklists.filter((b) => b.Enabled).length
|
const blOn = blocklists.filter((b) => b.Enabled).length
|
||||||
const alOn = allowlists.filter((a) => a.Enabled).length
|
const alOn = allowlists.filter((a) => a.Enabled).length
|
||||||
const blNames = useMemo(() => new Set(blocklists.map((b) => b.Name)), [blocklists])
|
const blNames = useMemo(() => new Set(blocklists.map((b) => b.Name)), [blocklists])
|
||||||
@@ -500,19 +570,58 @@ export default function DNS() {
|
|||||||
[config, resolvers, save],
|
[config, resolvers, save],
|
||||||
)
|
)
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Delete a resolver, saying what it was still wired into.
|
||||||
|
*
|
||||||
|
* The three GLOBAL slots (default, fallback, endpoint) are cleared here, because
|
||||||
|
* a global pointing at nothing is never what anyone meant. The DNS RULES are a
|
||||||
|
* different matter: each one is a decision about which queries go where, and
|
||||||
|
* silently deleting or repointing them would change where a device's DNS goes
|
||||||
|
* without saying so. So they are named instead and left alone — the dialog is
|
||||||
|
* where the operator finds out they exist, which is precisely what this page
|
||||||
|
* used to skip: it cleared the two globals without a word and never mentioned
|
||||||
|
* the rules at all.
|
||||||
|
*/
|
||||||
const removeResolver = useCallback(
|
const removeResolver = useCallback(
|
||||||
async (idx: number) => {
|
async (idx: number) => {
|
||||||
if (!config) return
|
if (!config) return
|
||||||
const target = resolvers[idx]
|
const target = resolvers[idx]
|
||||||
|
const g = { ...config.Globals }
|
||||||
|
const slots: string[] = []
|
||||||
|
if (g.ResolverDefault === target.Name) slots.push('the default resolver')
|
||||||
|
if (g.ResolverFallback === target.Name) slots.push('the fallback resolver')
|
||||||
|
if (g.EndpointResolver === target.Name) slots.push('the endpoint resolver')
|
||||||
|
const usedBy = dnsRules.filter((r) => r.Resolver === target.Name)
|
||||||
|
|
||||||
|
const parts: string[] = []
|
||||||
|
if (slots.length > 0) {
|
||||||
|
parts.push(
|
||||||
|
`It is ${slots.join(' and ')} — ${
|
||||||
|
slots.length === 1 ? 'that slot is' : 'those slots are'
|
||||||
|
} cleared, so DNS falls back to the engine's built-in resolution.`,
|
||||||
|
)
|
||||||
|
}
|
||||||
|
if (usedBy.length === 1) {
|
||||||
|
parts.push(
|
||||||
|
`One DNS rule still sends queries to it (order ${usedBy[0].Order}). It is left as it is and will have nowhere to resolve — repoint it before you apply.`,
|
||||||
|
)
|
||||||
|
} else if (usedBy.length > 1) {
|
||||||
|
parts.push(
|
||||||
|
`${usedBy.length} DNS rules still send queries to it (orders ${usedBy
|
||||||
|
.map((r) => r.Order)
|
||||||
|
.join(', ')}). They are left as they are and will have nowhere to resolve — repoint them before you apply.`,
|
||||||
|
)
|
||||||
|
}
|
||||||
|
if (parts.length === 0) parts.push('Nothing else in the config points at it.')
|
||||||
|
|
||||||
const ok = await confirm({
|
const ok = await confirm({
|
||||||
label: 'Delete resolver',
|
label: 'Delete resolver',
|
||||||
title: `Delete resolver “${target.Name}”?`,
|
title: `Delete resolver “${target.Name}”?`,
|
||||||
body: 'This removes it from the config.',
|
body: parts.join(' '),
|
||||||
})
|
})
|
||||||
if (!ok) return
|
if (!ok) return
|
||||||
const next = resolvers.filter((_, i) => i !== idx)
|
const next = resolvers.filter((_, i) => i !== idx)
|
||||||
// Don't leave default/fallback pointing at a resolver that no longer exists.
|
// Don't leave default/fallback/endpoint pointing at a resolver that's gone.
|
||||||
const g = { ...config.Globals }
|
|
||||||
const cleared: string[] = []
|
const cleared: string[] = []
|
||||||
if (g.ResolverDefault === target.Name) {
|
if (g.ResolverDefault === target.Name) {
|
||||||
g.ResolverDefault = ''
|
g.ResolverDefault = ''
|
||||||
@@ -522,12 +631,16 @@ export default function DNS() {
|
|||||||
g.ResolverFallback = ''
|
g.ResolverFallback = ''
|
||||||
cleared.push('fallback')
|
cleared.push('fallback')
|
||||||
}
|
}
|
||||||
|
if (g.EndpointResolver === target.Name) {
|
||||||
|
g.EndpointResolver = ''
|
||||||
|
cleared.push('endpoint')
|
||||||
|
}
|
||||||
const msg = cleared.length
|
const msg = cleared.length
|
||||||
? `Deleted ${target.Name} — cleared ${cleared.join(' & ')}`
|
? `Deleted ${target.Name} — cleared ${cleared.join(' & ')}`
|
||||||
: `Deleted ${target.Name}`
|
: `Deleted ${target.Name}`
|
||||||
void save({ ...config, Globals: g, Resolvers: next }, msg)
|
void save({ ...config, Globals: g, Resolvers: next }, msg)
|
||||||
},
|
},
|
||||||
[config, resolvers, save],
|
[config, resolvers, dnsRules, save, confirm],
|
||||||
)
|
)
|
||||||
|
|
||||||
const setResolverDefault = useCallback(
|
const setResolverDefault = useCallback(
|
||||||
@@ -905,8 +1018,11 @@ export default function DNS() {
|
|||||||
categories={b.Categories}
|
categories={b.Categories}
|
||||||
response={b.Response}
|
response={b.Response}
|
||||||
filterOn={dnsFilterOn}
|
filterOn={dnsFilterOn}
|
||||||
|
statuses={listStatus.get(`blocklist:${b.Name}`) ?? null}
|
||||||
|
updating={updatingLists.has(`blocklist:${b.Name}`)}
|
||||||
busy={busy}
|
busy={busy}
|
||||||
onToggle={(on) => toggleBlocklist(i, on)}
|
onToggle={(on) => toggleBlocklist(i, on)}
|
||||||
|
onUpdateNow={() => void updateList('blocklist', b.Name)}
|
||||||
onDelete={() => removeBlocklist(i)}
|
onDelete={() => removeBlocklist(i)}
|
||||||
/>
|
/>
|
||||||
))}
|
))}
|
||||||
@@ -964,9 +1080,13 @@ export default function DNS() {
|
|||||||
url={a.URL}
|
url={a.URL}
|
||||||
path={a.Path}
|
path={a.Path}
|
||||||
entries={a.Entries}
|
entries={a.Entries}
|
||||||
|
categories={a.Categories}
|
||||||
filterOn={dnsFilterOn}
|
filterOn={dnsFilterOn}
|
||||||
|
statuses={listStatus.get(`allowlist:${a.Name}`) ?? null}
|
||||||
|
updating={updatingLists.has(`allowlist:${a.Name}`)}
|
||||||
busy={busy}
|
busy={busy}
|
||||||
onToggle={(on) => toggleAllowlist(i, on)}
|
onToggle={(on) => toggleAllowlist(i, on)}
|
||||||
|
onUpdateNow={() => void updateList('allowlist', a.Name)}
|
||||||
onDelete={() => removeAllowlist(i)}
|
onDelete={() => removeAllowlist(i)}
|
||||||
/>
|
/>
|
||||||
))}
|
))}
|
||||||
@@ -1717,8 +1837,11 @@ function ListRow({
|
|||||||
categories,
|
categories,
|
||||||
response,
|
response,
|
||||||
filterOn,
|
filterOn,
|
||||||
|
statuses,
|
||||||
|
updating,
|
||||||
busy,
|
busy,
|
||||||
onToggle,
|
onToggle,
|
||||||
|
onUpdateNow,
|
||||||
onDelete,
|
onDelete,
|
||||||
}: {
|
}: {
|
||||||
name: string
|
name: string
|
||||||
@@ -1730,8 +1853,14 @@ function ListRow({
|
|||||||
categories?: string[] | null
|
categories?: string[] | null
|
||||||
response?: BlockResponse
|
response?: BlockResponse
|
||||||
filterOn: boolean
|
filterOn: boolean
|
||||||
|
/** What the running engine reports about this list, one record per geo category.
|
||||||
|
* null/[] ⇒ nothing reported: an older daemon, a stopped engine, or a list that
|
||||||
|
* has not been applied yet. The row then says so instead of guessing. */
|
||||||
|
statuses: RulesetStatus[] | null
|
||||||
|
updating: boolean
|
||||||
busy: boolean
|
busy: boolean
|
||||||
onToggle: (on: boolean) => void
|
onToggle: (on: boolean) => void
|
||||||
|
onUpdateNow: () => void
|
||||||
onDelete: () => void
|
onDelete: () => void
|
||||||
}) {
|
}) {
|
||||||
const detail = useMemo<{ text: string; masked: boolean; title?: string }>(() => {
|
const detail = useMemo<{ text: string; masked: boolean; title?: string }>(() => {
|
||||||
@@ -1755,8 +1884,45 @@ function ListRow({
|
|||||||
}
|
}
|
||||||
}, [source, url, path, entries, categories])
|
}, [source, url, path, entries, categories])
|
||||||
|
|
||||||
// A list only actually filters when both it and the master switch are on.
|
// url and geosite lists are FETCHED by the engine; inline and file ones are read
|
||||||
const active = enabled && filterOn
|
// straight from the config and are loaded the moment they are applied.
|
||||||
|
const remote = source === 'url' || source === 'geosite'
|
||||||
|
const recs = statuses ?? []
|
||||||
|
const hasStatus = recs.length > 0
|
||||||
|
// A geo list with several categories: the OLDEST fetch (so a category that never
|
||||||
|
// arrived is never hidden behind a fresh sibling) and the SUM of the counts.
|
||||||
|
let ruleCount = 0
|
||||||
|
let neverAny = false
|
||||||
|
let oldestIso = ''
|
||||||
|
for (const s of recs) {
|
||||||
|
ruleCount += s.rule_count
|
||||||
|
if (!s.last_updated) neverAny = true
|
||||||
|
else if (!oldestIso || Date.parse(s.last_updated) < Date.parse(oldestIso)) oldestIso = s.last_updated
|
||||||
|
}
|
||||||
|
const interval = everyLabel(recs[0]?.interval_seconds ?? 0)
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Whether this list is BLOCKING ANYTHING, which is a different question from
|
||||||
|
* whether it is switched on — and the one the row used to answer wrongly.
|
||||||
|
*
|
||||||
|
* "filtering" is now only said when the engine reports rules loaded for it. A
|
||||||
|
* remote list that has never been fetched (the daemon raises this as a critical
|
||||||
|
* apply finding) reads "not loaded", and one that fetched an empty list reads
|
||||||
|
* "empty". Nothing reported at all is "load not reported": unknown, not green.
|
||||||
|
*/
|
||||||
|
const state: { text: string; tone: 'on' | 'off' | 'warn' } = !enabled
|
||||||
|
? { text: 'off', tone: 'off' }
|
||||||
|
: !filterOn
|
||||||
|
? { text: 'inactive', tone: 'off' }
|
||||||
|
: !remote
|
||||||
|
? { text: 'filtering', tone: 'on' }
|
||||||
|
: !hasStatus
|
||||||
|
? { text: 'load not reported', tone: 'off' }
|
||||||
|
: neverAny
|
||||||
|
? { text: 'not loaded — nothing blocked', tone: 'warn' }
|
||||||
|
: ruleCount === 0
|
||||||
|
? { text: 'loaded empty — nothing blocked', tone: 'warn' }
|
||||||
|
: { text: 'filtering', tone: 'on' }
|
||||||
|
|
||||||
return (
|
return (
|
||||||
<li className="dns-row">
|
<li className="dns-row">
|
||||||
@@ -1786,10 +1952,39 @@ function ListRow({
|
|||||||
token hidden
|
token hidden
|
||||||
</span>
|
</span>
|
||||||
)}
|
)}
|
||||||
<span className="dns-row-state" data-active={active ? 'on' : 'off'}>
|
<span className="dns-row-state" data-active={state.tone}>
|
||||||
{active ? 'filtering' : 'inactive'}
|
{state.text}
|
||||||
</span>
|
</span>
|
||||||
</div>
|
</div>
|
||||||
|
{remote && (
|
||||||
|
<div className="dns-row-sync">
|
||||||
|
<span className="dns-sync-fresh" data-never={hasStatus && neverAny ? 'y' : undefined}>
|
||||||
|
{hasStatus ? relFetch(oldestIso) : 'status pending'}
|
||||||
|
</span>
|
||||||
|
{interval && <span className="dns-sync-every">{interval}</span>}
|
||||||
|
{ruleCount > 0 && (
|
||||||
|
<span className="dns-sync-rules">
|
||||||
|
{ruleCount.toLocaleString('en-US')} rule{ruleCount === 1 ? '' : 's'}
|
||||||
|
</span>
|
||||||
|
)}
|
||||||
|
<button
|
||||||
|
type="button"
|
||||||
|
className="dns-sync-update"
|
||||||
|
onClick={onUpdateNow}
|
||||||
|
disabled={busy || updating}
|
||||||
|
aria-label={`Update ${name} now`}
|
||||||
|
>
|
||||||
|
{updating ? (
|
||||||
|
<>
|
||||||
|
<span className="dns-sync-spin" aria-hidden="true" />
|
||||||
|
<span>Updating…</span>
|
||||||
|
</>
|
||||||
|
) : (
|
||||||
|
'Update now'
|
||||||
|
)}
|
||||||
|
</button>
|
||||||
|
</div>
|
||||||
|
)}
|
||||||
</div>
|
</div>
|
||||||
<Button
|
<Button
|
||||||
className="dns-del"
|
className="dns-del"
|
||||||
|
|||||||
@@ -85,6 +85,19 @@ const DEFAULT_TPROXY_PORT = 12345
|
|||||||
* Collapsing those into one switch would make "I want ping to work" silently mean
|
* Collapsing those into one switch would make "I want ping to work" silently mean
|
||||||
* "I permit a parallel VPN bypass", so the middle option exists to remove that
|
* "I permit a parallel VPN bypass", so the middle option exists to remove that
|
||||||
* false choice — and the labels push anyone who wants diagnostics to `icmp`.
|
* false choice — and the labels push anyone who wants diagnostics to `icmp`.
|
||||||
|
*
|
||||||
|
* WHY THIS COPY WAS REWRITTEN. `block` used to say "Nothing leaves except through
|
||||||
|
* the tunnel", and it was not true. The daemon let untunnelable traffic out toward
|
||||||
|
* every destination the ROUTING RULES send direct, on the argument that such a host
|
||||||
|
* already has your address from ordinary TCP. Under the commonest setup here —
|
||||||
|
* "tunnel what's blocked, send the rest direct" — the routing default IS direct, so
|
||||||
|
* that covered everything: `block` behaved exactly like `direct`, including ESP/GRE,
|
||||||
|
* i.e. the parallel-VPN case the middle rung exists to exclude. The daemon now drops
|
||||||
|
* unconditionally under `block`, and this copy states the price instead of hiding it
|
||||||
|
* (the owner's call: this router does not do ping and does not do IPTV).
|
||||||
|
*
|
||||||
|
* `icmp` still carries that destination-dependence for its NON-ping half, so its
|
||||||
|
* cost line says so rather than claiming "nothing else gets out".
|
||||||
*/
|
*/
|
||||||
type Untunnelable = 'block' | 'icmp' | 'direct'
|
type Untunnelable = 'block' | 'icmp' | 'direct'
|
||||||
|
|
||||||
@@ -98,7 +111,10 @@ function normUntunnelable(raw: string | undefined): Untunnelable {
|
|||||||
|
|
||||||
const UNTUNNELABLE_OPTIONS: ReadonlyArray<{ value: string; label: string }> = [
|
const UNTUNNELABLE_OPTIONS: ReadonlyArray<{ value: string; label: string }> = [
|
||||||
{ value: 'block', label: 'Block everything — most private' },
|
{ value: 'block', label: 'Block everything — most private' },
|
||||||
{ value: 'icmp', label: 'Allow ping only — for diagnostics' },
|
// Not "Allow ping only": the rung also lets the other untunnelable protocols
|
||||||
|
// out toward directly-routed addresses, and the cost line below says so. A
|
||||||
|
// label that promised "only" would be contradicted two lines under itself.
|
||||||
|
{ value: 'icmp', label: 'Allow ping — for diagnostics' },
|
||||||
{ value: 'direct', label: 'Allow everything — most compatible' },
|
{ value: 'direct', label: 'Allow everything — most compatible' },
|
||||||
]
|
]
|
||||||
|
|
||||||
@@ -110,13 +126,18 @@ interface PolicyCopy {
|
|||||||
|
|
||||||
const UNTUNNELABLE_COPY: Record<Untunnelable, PolicyCopy> = {
|
const UNTUNNELABLE_COPY: Record<Untunnelable, PolicyCopy> = {
|
||||||
block: {
|
block: {
|
||||||
works: 'Nothing leaves except through the tunnel.',
|
// Scoped to "this traffic" on purpose. The old line — "Nothing leaves except
|
||||||
cost: 'Ping and traceroute won’t work from your devices, and neither will multicast IPTV or connecting to a VPN from a device on your network.',
|
// through the tunnel" — was doubly loose: it was false (see the note above),
|
||||||
|
// and even read charitably it collides with directly-routed TCP, which does
|
||||||
|
// leave outside the tunnel by design.
|
||||||
|
works:
|
||||||
|
'None of this traffic leaves the router — it’s dropped, whatever your routing rules say. It’s the only setting whose promise doesn’t depend on how the rules are written.',
|
||||||
|
cost: 'Ping and traceroute stop working from your devices. So do IPsec and PPTP VPN connections made from a device on your network, multicast IPTV, and SCTP. VPNs that run over UDP — WireGuard, OpenVPN-UDP, and IPsec through NAT (IKEv2/NAT-T) — are unaffected: they go through the tunnel like everything else.',
|
||||||
tone: 'good',
|
tone: 'good',
|
||||||
},
|
},
|
||||||
icmp: {
|
icmp: {
|
||||||
works: 'Ping and traceroute work, so you can check whether something is reachable.',
|
works: 'Ping and traceroute work everywhere, so you can check whether something is reachable.',
|
||||||
cost: 'Whatever you ping sees your real IP address instead of the tunnel’s. Only for hosts you deliberately ping, and nothing else gets out — IPTV and VPN connections stay blocked.',
|
cost: 'Whatever you ping sees your real IP address instead of the tunnel’s. IPsec, PPTP and IPTV also get out — but only toward addresses your routing rules already send direct, so a VPN app on a device can still open its own connection beside this one if its server is one of those.',
|
||||||
tone: 'warn',
|
tone: 'warn',
|
||||||
},
|
},
|
||||||
direct: {
|
direct: {
|
||||||
@@ -551,7 +572,7 @@ export default function Networks({ status }: { status?: Status | null }) {
|
|||||||
|
|
||||||
{untunnelable === 'block' && (
|
{untunnelable === 'block' && (
|
||||||
<p className="nw-sec-note nw-policy-hint">
|
<p className="nw-sec-note nw-policy-hint">
|
||||||
If you just want to check whether a site is reachable, choose <strong>Allow ping only</strong>{' '}
|
If you just want to check whether a site is reachable, choose <strong>Allow ping</strong>{' '}
|
||||||
rather than allowing everything — it’s the narrower of the two.
|
rather than allowing everything — it’s the narrower of the two.
|
||||||
</p>
|
</p>
|
||||||
)}
|
)}
|
||||||
|
|||||||
@@ -741,14 +741,35 @@ export default function Nodes() {
|
|||||||
[config, nodes, save],
|
[config, nodes, save],
|
||||||
)
|
)
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Delete a node, naming everything that still points at it.
|
||||||
|
*
|
||||||
|
* `findNodeReferences` was already here and already right — it just wasn't asked
|
||||||
|
* on the one path where the answer matters. A RENAME carried its references and
|
||||||
|
* said so; a DELETE said "This removes it from the config", which is true of the
|
||||||
|
* node and silent about the rule, group member, chain hop or resolver detour
|
||||||
|
* left spelling a name nothing answers to. That is not a cosmetic dangle: an
|
||||||
|
* unresolved target does not fall through to the default route, so the traffic
|
||||||
|
* aimed at it is blocked.
|
||||||
|
*/
|
||||||
const removeNode = useCallback(
|
const removeNode = useCallback(
|
||||||
async (idx: number) => {
|
async (idx: number) => {
|
||||||
if (!config) return
|
if (!config) return
|
||||||
const target = nodes[idx]
|
const target = nodes[idx]
|
||||||
|
const refs = findNodeReferences(config, target.Name)
|
||||||
|
const shown = refs.slice(0, 4).map((r) => r.label)
|
||||||
|
const more = refs.length - shown.length
|
||||||
const ok = await confirm({
|
const ok = await confirm({
|
||||||
label: 'Delete node',
|
label: 'Delete node',
|
||||||
title: `Delete node “${target.Name}”?`,
|
title: `Delete node “${target.Name}”?`,
|
||||||
body: 'This removes it from the config.',
|
body:
|
||||||
|
refs.length === 0
|
||||||
|
? 'Nothing else in the config points at it.'
|
||||||
|
: `${refSummary(refs)} still ${refs.length === 1 ? 'points' : 'point'} at it — ${shown.join(
|
||||||
|
', ',
|
||||||
|
)}${
|
||||||
|
more > 0 ? `, and ${more} more` : ''
|
||||||
|
}. Nothing rewrites them, and a target that no longer resolves does not fall through to the default route: the traffic aimed at it is blocked.`,
|
||||||
})
|
})
|
||||||
if (!ok) return
|
if (!ok) return
|
||||||
const next = nodes.filter((_, i) => i !== idx)
|
const next = nodes.filter((_, i) => i !== idx)
|
||||||
|
|||||||
+104
-32
@@ -4,17 +4,18 @@ import type { LedVariant } from '../components'
|
|||||||
import { fmtDateTime, fmtDuration } from '../format'
|
import { fmtDateTime, fmtDuration } from '../format'
|
||||||
import {
|
import {
|
||||||
apply as apiApply,
|
apply as apiApply,
|
||||||
confirm as apiConfirm,
|
|
||||||
rollback as apiRollback,
|
rollback as apiRollback,
|
||||||
getConfig,
|
getConfig,
|
||||||
|
getRulesReachability,
|
||||||
getStats,
|
getStats,
|
||||||
ApiError,
|
ApiError,
|
||||||
} from '../api'
|
} from '../api'
|
||||||
import type { Model, Stats, Status, StatusWarning } from '../api'
|
import type { Model, Stats, Status, StatusWarning } from '../api'
|
||||||
|
import { confirmTimeout } from '../pendingConfirm'
|
||||||
import { navigate } from '../router'
|
import { navigate } from '../router'
|
||||||
import type { Route } from '../router'
|
import type { Route } from '../router'
|
||||||
import { attentionFindings } from '../findings'
|
import { attentionFindings } from '../findings'
|
||||||
import { protectionState } from '../planeState'
|
import { engineReadout, protectionState } from '../planeState'
|
||||||
|
|
||||||
// null-safe length for a Go slice that may arrive as null.
|
// null-safe length for a Go slice that may arrive as null.
|
||||||
const len = (a: unknown[] | null | undefined): number => (a ? a.length : 0)
|
const len = (a: unknown[] | null | undefined): number => (a ? a.length : 0)
|
||||||
@@ -27,7 +28,8 @@ function short(hash: string): string {
|
|||||||
return h.length > 12 ? h.slice(0, 12) : h
|
return h.length > 12 ? h.slice(0, 12) : h
|
||||||
}
|
}
|
||||||
|
|
||||||
type ControlKind = 'apply' | 'confirm' | 'rollback'
|
// Confirm is no longer one of them — see the note beside the controls row.
|
||||||
|
type ControlKind = 'apply' | 'rollback'
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* Live service uptime in seconds, ticking between status polls.
|
* Live service uptime in seconds, ticking between status polls.
|
||||||
@@ -42,17 +44,32 @@ type ControlKind = 'apply' | 'confirm' | 'rollback'
|
|||||||
* Returns null when the daemon doesn't report uptime (older builds) — the caller
|
* Returns null when the daemon doesn't report uptime (older builds) — the caller
|
||||||
* then renders nothing rather than inventing a number.
|
* then renders nothing rather than inventing a number.
|
||||||
*/
|
*/
|
||||||
function useUptime(status: Status | null): number | null {
|
function useUptime(status: Status | null): { seconds: number; startedUnix: number } | null {
|
||||||
const base = useRef<{ uptime: number; at: number } | null>(null)
|
const base = useRef<{ uptime: number; at: number } | null>(null)
|
||||||
|
// The instant the daemon came up, ON THE BROWSER'S CLOCK.
|
||||||
|
//
|
||||||
|
// `status.started_unix` is the router's own clock, and the router has no RTC —
|
||||||
|
// it runs on UTC with no tzdata. Rendering it through the browser's timezone
|
||||||
|
// printed a start time three hours in the FUTURE for a Moscow operator, beside
|
||||||
|
// an uptime of "2 h 41 min". Deriving it instead as now-minus-uptime is a
|
||||||
|
// difference of two client timestamps, so it is skew-proof and can never land
|
||||||
|
// ahead of the clock in the header.
|
||||||
|
const started = useRef<number | null>(null)
|
||||||
const [, forceTick] = useState(0)
|
const [, forceTick] = useState(0)
|
||||||
|
|
||||||
const reported = status?.uptime_seconds
|
const reported = status?.uptime_seconds
|
||||||
useEffect(() => {
|
useEffect(() => {
|
||||||
if (typeof reported !== 'number' || !Number.isFinite(reported)) {
|
if (typeof reported !== 'number' || !Number.isFinite(reported)) {
|
||||||
base.current = null
|
base.current = null
|
||||||
|
started.current = null
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
base.current = { uptime: reported, at: Date.now() }
|
const now = Date.now()
|
||||||
|
base.current = { uptime: reported, at: now }
|
||||||
|
// Re-baselining every poll would jitter the displayed second back and forth;
|
||||||
|
// only move it when the estimate has genuinely drifted (a daemon restart).
|
||||||
|
const est = Math.round(now / 1000 - reported)
|
||||||
|
if (started.current === null || Math.abs(started.current - est) > 5) started.current = est
|
||||||
forceTick((n) => n + 1)
|
forceTick((n) => n + 1)
|
||||||
}, [reported])
|
}, [reported])
|
||||||
|
|
||||||
@@ -64,8 +81,11 @@ function useUptime(status: Status | null): number | null {
|
|||||||
return () => window.clearInterval(id)
|
return () => window.clearInterval(id)
|
||||||
}, [])
|
}, [])
|
||||||
|
|
||||||
if (!base.current) return null
|
if (!base.current || started.current === null) return null
|
||||||
return base.current.uptime + Math.max(0, (Date.now() - base.current.at) / 1000)
|
return {
|
||||||
|
seconds: base.current.uptime + Math.max(0, (Date.now() - base.current.at) / 1000),
|
||||||
|
startedUnix: started.current,
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
export function Overview({
|
export function Overview({
|
||||||
@@ -93,6 +113,29 @@ export function Overview({
|
|||||||
void loadConfig()
|
void loadConfig()
|
||||||
}, [loadConfig])
|
}, [loadConfig])
|
||||||
|
|
||||||
|
// ---- how many rules are actually IN FORCE ----------------------------------
|
||||||
|
//
|
||||||
|
// `Rule.Enabled` from /api/config is the DESIRED state; the active WAN profile
|
||||||
|
// overrides it in either direction, and the daemon reports the result as
|
||||||
|
// `effective_enabled`. Counting the saved switches told a router running one
|
||||||
|
// chain that it had "2 / 2" — the Routing page had already been fixed to read
|
||||||
|
// the verdicts, and the home page kept summing the config beside it.
|
||||||
|
//
|
||||||
|
// null ⇒ no verdicts (older daemon, engine stopped, endpoint unreachable). The
|
||||||
|
// module then says so rather than passing the saved count off as the live one.
|
||||||
|
const [inForce, setInForce] = useState<number | null>(null)
|
||||||
|
const loadReach = useCallback(async () => {
|
||||||
|
try {
|
||||||
|
const { rules } = await getRulesReachability()
|
||||||
|
setInForce(rules.filter((r) => r.effective_enabled).length)
|
||||||
|
} catch {
|
||||||
|
setInForce(null)
|
||||||
|
}
|
||||||
|
}, [])
|
||||||
|
useEffect(() => {
|
||||||
|
void loadReach()
|
||||||
|
}, [loadReach])
|
||||||
|
|
||||||
// ---- live filter stats: poll the aggregate snapshot, degrade to honest empty states ----
|
// ---- live filter stats: poll the aggregate snapshot, degrade to honest empty states ----
|
||||||
const [stats, setStats] = useState<Stats | null>(null)
|
const [stats, setStats] = useState<Stats | null>(null)
|
||||||
useEffect(() => {
|
useEffect(() => {
|
||||||
@@ -121,7 +164,7 @@ export function Overview({
|
|||||||
}
|
}
|
||||||
}, [])
|
}, [])
|
||||||
|
|
||||||
// ---- apply / confirm / rollback ----
|
// ---- apply / rollback ----
|
||||||
const [busy, setBusy] = useState<ControlKind | null>(null)
|
const [busy, setBusy] = useState<ControlKind | null>(null)
|
||||||
const [result, setResult] = useState<{ ok: boolean; msg: string } | null>(null)
|
const [result, setResult] = useState<{ ok: boolean; msg: string } | null>(null)
|
||||||
const [toast, setToast] = useState<string | null>(null)
|
const [toast, setToast] = useState<string | null>(null)
|
||||||
@@ -139,22 +182,25 @@ export function Overview({
|
|||||||
setBusy(kind)
|
setBusy(kind)
|
||||||
setResult(null)
|
setResult(null)
|
||||||
try {
|
try {
|
||||||
const fn = kind === 'apply' ? apiApply : kind === 'confirm' ? apiConfirm : apiRollback
|
const r = kind === 'apply' ? await apiApply() : await apiRollback()
|
||||||
const r = await fn()
|
|
||||||
if (r.error) {
|
if (r.error) {
|
||||||
setResult({ ok: false, msg: r.error })
|
setResult({ ok: false, msg: r.error })
|
||||||
flash(`${kind} failed`)
|
flash(`${kind} failed`)
|
||||||
} else {
|
} else {
|
||||||
|
// An apply that changed something armed an auto-rollback, and saying
|
||||||
|
// "data plane reconciled" while a timer runs is how someone walks away
|
||||||
|
// from a config that then reverts. Name the window when there is one.
|
||||||
|
const window = confirmTimeout()
|
||||||
const msg =
|
const msg =
|
||||||
kind === 'apply'
|
kind === 'apply'
|
||||||
? r.changed
|
? r.changed
|
||||||
? 'Applied — data plane reconciled'
|
? window > 0
|
||||||
|
? `Applied — keep this config within ${window}s or it rolls back`
|
||||||
|
: 'Applied — data plane reconciled'
|
||||||
: 'Applied — already up to date'
|
: 'Applied — already up to date'
|
||||||
: kind === 'confirm'
|
: 'Rolled back to last-good config'
|
||||||
? 'Confirmed — auto-rollback cancelled'
|
|
||||||
: 'Rolled back to last-good config'
|
|
||||||
setResult({ ok: true, msg })
|
setResult({ ok: true, msg })
|
||||||
flash(kind === 'apply' ? 'Applied' : kind === 'confirm' ? 'Confirmed' : 'Rolled back')
|
flash(kind === 'apply' ? 'Applied' : 'Rolled back')
|
||||||
}
|
}
|
||||||
} catch (e) {
|
} catch (e) {
|
||||||
const msg = e instanceof Error ? e.message : 'request failed'
|
const msg = e instanceof Error ? e.message : 'request failed'
|
||||||
@@ -163,16 +209,20 @@ export function Overview({
|
|||||||
} finally {
|
} finally {
|
||||||
setBusy(null)
|
setBusy(null)
|
||||||
onStatusChange()
|
onStatusChange()
|
||||||
if (kind !== 'confirm') void loadConfig()
|
void loadConfig()
|
||||||
|
// An apply or a rollback is exactly what changes which rules are in force.
|
||||||
|
void loadReach()
|
||||||
}
|
}
|
||||||
},
|
},
|
||||||
[flash, loadConfig, onStatusChange],
|
[flash, loadConfig, loadReach, onStatusChange],
|
||||||
)
|
)
|
||||||
|
|
||||||
// ---- service uptime (PROCESS uptime, not "time since the last apply") ----
|
// ---- service uptime (PROCESS uptime, not "time since the last apply") ----
|
||||||
const uptime = useUptime(status)
|
const uptime = useUptime(status)
|
||||||
const uptimeText = uptime === null ? '' : fmtDuration(uptime)
|
const uptimeText = uptime === null ? '' : fmtDuration(uptime.seconds)
|
||||||
const startedAt = status?.started_unix ? fmtDateTime(status.started_unix) : ''
|
// On YOUR clock, derived from the uptime — never `status.started_unix`, which is
|
||||||
|
// the router's clock and has no timezone to convert from. See useUptime.
|
||||||
|
const startedAt = uptime === null ? '' : fmtDateTime(uptime.startedUnix)
|
||||||
|
|
||||||
// ---- derived display state ----
|
// ---- derived display state ----
|
||||||
const g = config?.Globals
|
const g = config?.Globals
|
||||||
@@ -251,13 +301,12 @@ export function Overview({
|
|||||||
? `${worstGroup.group} — no answer`
|
? `${worstGroup.group} — no answer`
|
||||||
: `${worstGroup.group} — ${worstGroup.dead} down`
|
: `${worstGroup.group} — ${worstGroup.dead} down`
|
||||||
|
|
||||||
const engineVariant: LedVariant = !status
|
// One reading for the engine, and it is able to say "stopped": `status.running`
|
||||||
? 'off'
|
// was a constant `true` on the daemon, so this LED could never go crit and the
|
||||||
: status.running && status.active
|
// Engine module was green through a process that had failed to start. See
|
||||||
? 'on'
|
// planeState.engineState.
|
||||||
: status.running
|
const engine = engineReadout(status)
|
||||||
? 'amber'
|
const engineVariant: LedVariant = engine.variant
|
||||||
: 'crit'
|
|
||||||
|
|
||||||
const protection = protectionState(status)
|
const protection = protectionState(status)
|
||||||
// Configured fail-closed AND actually enforcing it. `none` means nothing is
|
// Configured fail-closed AND actually enforcing it. `none` means nothing is
|
||||||
@@ -334,11 +383,28 @@ export function Overview({
|
|||||||
/>
|
/>
|
||||||
)}
|
)}
|
||||||
|
|
||||||
|
{/* "N / M in force", the same reading the Routing page shows — never the
|
||||||
|
count of saved switches. The lamp follows the same rule: a table of
|
||||||
|
rules none of which are in force routes exactly nothing, and it used
|
||||||
|
to sit under a green light saying "0 / 7". */}
|
||||||
<Module
|
<Module
|
||||||
name="Routing"
|
name="Routing"
|
||||||
value={String(enabledCount(config?.Rules))}
|
value={inForce === null ? String(enabledCount(config?.Rules)) : String(inForce)}
|
||||||
unit={`/ ${len(config?.Rules)} rules`}
|
unit={
|
||||||
led={{ variant: len(config?.Rules) ? 'on' : 'amber' }}
|
inForce === null
|
||||||
|
? `/ ${len(config?.Rules)} rules saved`
|
||||||
|
: `/ ${len(config?.Rules)} in force`
|
||||||
|
}
|
||||||
|
led={{
|
||||||
|
variant:
|
||||||
|
len(config?.Rules) === 0
|
||||||
|
? 'amber'
|
||||||
|
: inForce === null
|
||||||
|
? 'off'
|
||||||
|
: inForce === 0
|
||||||
|
? 'amber'
|
||||||
|
: 'on',
|
||||||
|
}}
|
||||||
rows={[
|
rows={[
|
||||||
{ k: 'egresses', v: String(len(config?.Egresses)) },
|
{ k: 'egresses', v: String(len(config?.Egresses)) },
|
||||||
{ k: 'default', v: defaultTarget(status, config), hot: true },
|
{ k: 'default', v: defaultTarget(status, config), hot: true },
|
||||||
@@ -414,6 +480,7 @@ export function Overview({
|
|||||||
unit={status?.version?.includes('-') ? '· ' + status.version.split('-').slice(1).join('-') : ''}
|
unit={status?.version?.includes('-') ? '· ' + status.version.split('-').slice(1).join('-') : ''}
|
||||||
led={{ variant: engineVariant }}
|
led={{ variant: engineVariant }}
|
||||||
rows={[
|
rows={[
|
||||||
|
{ k: 'process', v: engine.word, hot: engineVariant === 'crit' },
|
||||||
{ k: 'config hash', v: <span className="mono">{short(status?.hash ?? '')}</span> },
|
{ k: 'config hash', v: <span className="mono">{short(status?.hash ?? '')}</span> },
|
||||||
// Uptime of the daemon PROCESS. "started" is the moment it came up,
|
// Uptime of the daemon PROCESS. "started" is the moment it came up,
|
||||||
// by the router's clock — not the moment a config was applied.
|
// by the router's clock — not the moment a config was applied.
|
||||||
@@ -431,9 +498,14 @@ export function Overview({
|
|||||||
<Button variant="primary" onClick={() => void run('apply')} disabled={busy !== null}>
|
<Button variant="primary" onClick={() => void run('apply')} disabled={busy !== null}>
|
||||||
{busy === 'apply' ? 'Applying…' : 'Apply config'}
|
{busy === 'apply' ? 'Applying…' : 'Apply config'}
|
||||||
</Button>
|
</Button>
|
||||||
<Button onClick={() => void run('confirm')} disabled={busy !== null}>
|
{/* A "Confirm" button used to sit here permanently, and pressing it
|
||||||
{busy === 'confirm' ? 'Confirming…' : 'Confirm'}
|
always printed "Confirmed — auto-rollback cancelled": `apply.Confirm()`
|
||||||
</Button>
|
returns nil whether or not a window was ever armed, so the message was
|
||||||
|
a success report for an event that usually had not happened.
|
||||||
|
Keeping a config is now offered only while a window is actually open,
|
||||||
|
and that is announced by the app-wide band directly above this page —
|
||||||
|
which is where the button lives, beside the countdown it belongs to,
|
||||||
|
rather than duplicated here. */}
|
||||||
{canRollback && (
|
{canRollback && (
|
||||||
<Button onClick={() => void run('rollback')} disabled={busy !== null}>
|
<Button onClick={() => void run('rollback')} disabled={busy !== null}>
|
||||||
{busy === 'rollback' ? 'Rolling back…' : 'Rollback'}
|
{busy === 'rollback' ? 'Rolling back…' : 'Rollback'}
|
||||||
|
|||||||
+111
-35
@@ -12,6 +12,7 @@ import {
|
|||||||
ApiError,
|
ApiError,
|
||||||
} from '../api'
|
} from '../api'
|
||||||
import type { Model, Rule, RuleReach, Ruleset, RulesetStatus } from '../api'
|
import type { Model, Rule, RuleReach, Ruleset, RulesetStatus } from '../api'
|
||||||
|
import { everyLabel, relFetch } from '../format'
|
||||||
|
|
||||||
// ---------------------------------------------------------------------------
|
// ---------------------------------------------------------------------------
|
||||||
// The api.ts `Rule` is a deliberately thin subset (Name/Enabled/Order/Target/
|
// The api.ts `Rule` is a deliberately thin subset (Name/Enabled/Order/Target/
|
||||||
@@ -203,35 +204,8 @@ function errMsg(e: unknown): string {
|
|||||||
// --- remote-list freshness (feedback #9) ------------------------------------
|
// --- remote-list freshness (feedback #9) ------------------------------------
|
||||||
// A url-source ruleset re-fetches on a cadence; the engine reports when it last
|
// A url-source ruleset re-fetches on a cadence; the engine reports when it last
|
||||||
// pulled and how many rules the list holds. Match a ruleset to its status by the
|
// pulled and how many rules the list holds. Match a ruleset to its status by the
|
||||||
// engine tag `rs-<name>`.
|
// engine tag `rs-<name>`. The two readings (relFetch / everyLabel) live in
|
||||||
|
// format.ts because the DNS page shows the same ones for blocklists.
|
||||||
/** "updated 3h ago" / "never updated" for a remote list's last fetch. */
|
|
||||||
function relFetch(iso: string): string {
|
|
||||||
if (!iso) return 'never updated'
|
|
||||||
const t = Date.parse(iso)
|
|
||||||
if (Number.isNaN(t)) return 'never updated'
|
|
||||||
const s = Math.max(0, Math.floor((Date.now() - t) / 1000))
|
|
||||||
if (s < 45) return 'updated just now'
|
|
||||||
const m = Math.floor(s / 60)
|
|
||||||
if (m < 60) return `updated ${m}m ago`
|
|
||||||
const h = Math.floor(m / 60)
|
|
||||||
if (h < 24) return `updated ${h}h ago`
|
|
||||||
const d = Math.floor(h / 24)
|
|
||||||
return `updated ${d}d ago`
|
|
||||||
}
|
|
||||||
|
|
||||||
/** "every 24h" for an auto-update cadence in seconds ("" when there is none). */
|
|
||||||
function everyLabel(sec: number): string {
|
|
||||||
if (!sec || sec <= 0) return ''
|
|
||||||
if (sec % 3600 === 0) {
|
|
||||||
const h = sec / 3600
|
|
||||||
if (h < 48) return `every ${h}h`
|
|
||||||
if (sec % 86400 === 0) return `every ${sec / 86400}d`
|
|
||||||
return `every ${h}h`
|
|
||||||
}
|
|
||||||
if (sec % 60 === 0) return `every ${sec / 60}m`
|
|
||||||
return `every ${sec}s`
|
|
||||||
}
|
|
||||||
|
|
||||||
/** A rule with no matcher of any kind is the effective catch-all (route Final).
|
/** A rule with no matcher of any kind is the effective catch-all (route Final).
|
||||||
* Mirrors model.IsCatchAll on the daemon side — the two must agree or the
|
* Mirrors model.IsCatchAll on the daemon side — the two must agree or the
|
||||||
@@ -286,6 +260,57 @@ function formHasNoMatchers(f: {
|
|||||||
)
|
)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* What happens to the network when the default route stops being emitted —
|
||||||
|
* whether it is deleted or merely switched off (the engine emits neither).
|
||||||
|
*
|
||||||
|
* The old text was one sentence for every rule: "Traffic it matched will fall
|
||||||
|
* through to the next rule." For an ordinary rule that is true. For the catch-all
|
||||||
|
* there IS no next rule, and what happens instead is decided by the kill-switch:
|
||||||
|
* generate/route.go sets `final := tagBlock` and only `kill_switch=open` swaps
|
||||||
|
* that for `direct`. So removing the default either takes the whole network
|
||||||
|
* offline or puts the whole network on the naked WAN — and the page said "falls
|
||||||
|
* through to the next rule" for both, on a row it had already badged
|
||||||
|
* "DEFAULT ROUTE · FINAL".
|
||||||
|
*
|
||||||
|
* `successor` is the rule that would inherit route.Final instead (a config can
|
||||||
|
* carry more than one conditionless rule; the last one wins). When there is one,
|
||||||
|
* nothing is lost — the honest warning is that the destination changes.
|
||||||
|
*/
|
||||||
|
function defaultRouteConsequence(
|
||||||
|
killSwitch: string,
|
||||||
|
successor: { name: string; order: number; target: string } | null,
|
||||||
|
): ReactNode {
|
||||||
|
if (successor) {
|
||||||
|
return (
|
||||||
|
<>
|
||||||
|
This is the router’s <strong>default route</strong> — everything no other rule matches
|
||||||
|
follows it. Remove it and “{successor.name}” (order {successor.order}) has no conditions
|
||||||
|
either, so it takes over: unmatched traffic goes to{' '}
|
||||||
|
<strong className="mono">{successor.target}</strong> instead.
|
||||||
|
</>
|
||||||
|
)
|
||||||
|
}
|
||||||
|
if (killSwitch === 'open') {
|
||||||
|
return (
|
||||||
|
<>
|
||||||
|
This is the router’s <strong>default route</strong> — everything no other rule matches
|
||||||
|
follows it, and no other rule matches everything. With the kill-switch set to{' '}
|
||||||
|
<strong>fail-open</strong>, unmatched traffic then leaves through your normal internet
|
||||||
|
connection with your real address — unproxied and unfiltered.
|
||||||
|
</>
|
||||||
|
)
|
||||||
|
}
|
||||||
|
return (
|
||||||
|
<>
|
||||||
|
This is the router’s <strong>default route</strong> — everything no other rule matches
|
||||||
|
follows it, and no other rule matches everything. With the kill-switch set to{' '}
|
||||||
|
<strong>fail-closed</strong>, unmatched traffic is then <strong>blocked</strong>: devices on
|
||||||
|
your network lose the internet until you add a default back.
|
||||||
|
</>
|
||||||
|
)
|
||||||
|
}
|
||||||
|
|
||||||
/** Effective routing target for a rule (Target wins; a bare Egress is a target too). */
|
/** Effective routing target for a rule (Target wins; a bare Egress is a target too). */
|
||||||
function effectiveTarget(r: RRule): string {
|
function effectiveTarget(r: RRule): string {
|
||||||
if (r.Target && r.Target.trim()) return r.Target.trim()
|
if (r.Target && r.Target.trim()) return r.Target.trim()
|
||||||
@@ -678,17 +703,61 @@ export default function Routing() {
|
|||||||
[config, rules, rulesets, rulesetUsage, persist, confirm],
|
[config, rules, rulesets, rulesetUsage, persist, confirm],
|
||||||
)
|
)
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Is this rule the one the ENGINE uses as route.Final right now, and in force?
|
||||||
|
*
|
||||||
|
* Both halves matter. A conditionless rule the daemon reports as shadowed owns
|
||||||
|
* nothing (the row already says "never applies"), and one that is switched off
|
||||||
|
* is not being emitted either — removing either changes no traffic, so neither
|
||||||
|
* earns a warning.
|
||||||
|
*/
|
||||||
|
const isLiveDefault = useCallback(
|
||||||
|
(r: RRule): boolean => isCatchAll(r) && shadowOf(r) === null && forceOf(r).on,
|
||||||
|
[shadowOf, forceOf],
|
||||||
|
)
|
||||||
|
|
||||||
|
/** Which rule would inherit route.Final if `name` stopped being emitted: the
|
||||||
|
* LAST remaining conditionless, switched-on rule. null when there is none. */
|
||||||
|
const successorDefault = useCallback(
|
||||||
|
(name: string): { name: string; order: number; target: string } | null => {
|
||||||
|
for (let i = rules.length - 1; i >= 0; i--) {
|
||||||
|
const r = rules[i]
|
||||||
|
if (r.Name === name) continue
|
||||||
|
if (r.Enabled && isCatchAll(r)) {
|
||||||
|
return { name: r.Name, order: r.Order, target: effectiveTarget(r) }
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return null
|
||||||
|
},
|
||||||
|
[rules],
|
||||||
|
)
|
||||||
|
|
||||||
|
const killSwitch = (config?.Globals?.KillSwitch ?? 'closed') === 'open' ? 'open' : 'closed'
|
||||||
|
|
||||||
const onToggle = useCallback(
|
const onToggle = useCallback(
|
||||||
(name: string) => {
|
async (name: string) => {
|
||||||
const target = rules.find((r) => r.Name === name)
|
const target = rules.find((r) => r.Name === name)
|
||||||
if (!target) return
|
if (!target) return
|
||||||
const nextState = !target.Enabled
|
const nextState = !target.Enabled
|
||||||
|
// Switching the default route OFF is the same event as deleting it — the
|
||||||
|
// generator emits only enabled rules — so it asks the same question. It used
|
||||||
|
// to ask nothing at all, which made the least reversible control on the page
|
||||||
|
// the only one with no confirmation.
|
||||||
|
if (!nextState && isLiveDefault(target)) {
|
||||||
|
const ok = await confirm({
|
||||||
|
label: 'Turn off default route',
|
||||||
|
title: `Turn off the default route “${name}”?`,
|
||||||
|
body: defaultRouteConsequence(killSwitch, successorDefault(name)),
|
||||||
|
confirmLabel: 'Turn it off',
|
||||||
|
})
|
||||||
|
if (!ok) return
|
||||||
|
}
|
||||||
commitRules(
|
commitRules(
|
||||||
rules.map((r) => (r.Name === name ? { ...r, Enabled: nextState } : r)),
|
rules.map((r) => (r.Name === name ? { ...r, Enabled: nextState } : r)),
|
||||||
`${name} ${nextState ? 'enabled' : 'disabled'}`,
|
`${name} ${nextState ? 'enabled' : 'disabled'}`,
|
||||||
)
|
)
|
||||||
},
|
},
|
||||||
[rules, commitRules],
|
[rules, commitRules, confirm, isLiveDefault, killSwitch, successorDefault],
|
||||||
)
|
)
|
||||||
|
|
||||||
const onMove = useCallback(
|
const onMove = useCallback(
|
||||||
@@ -716,10 +785,17 @@ export default function Routing() {
|
|||||||
|
|
||||||
const onDelete = useCallback(
|
const onDelete = useCallback(
|
||||||
async (name: string) => {
|
async (name: string) => {
|
||||||
|
const target = rules.find((r) => r.Name === name)
|
||||||
|
if (!target) return
|
||||||
|
// The catch-all has no "next rule" to fall through to — see
|
||||||
|
// defaultRouteConsequence. Every other rule keeps the plain sentence.
|
||||||
|
const isDefault = isLiveDefault(target)
|
||||||
const ok = await confirm({
|
const ok = await confirm({
|
||||||
label: 'Delete rule',
|
label: isDefault ? 'Delete default route' : 'Delete rule',
|
||||||
title: `Delete rule "${name}"?`,
|
title: isDefault ? `Delete the default route “${name}”?` : `Delete rule “${name}”?`,
|
||||||
body: 'Traffic it matched will fall through to the next rule.',
|
body: isDefault
|
||||||
|
? defaultRouteConsequence(killSwitch, successorDefault(name))
|
||||||
|
: 'Traffic it matched will fall through to the next rule.',
|
||||||
})
|
})
|
||||||
if (!ok) return
|
if (!ok) return
|
||||||
commitRules(
|
commitRules(
|
||||||
@@ -727,7 +803,7 @@ export default function Routing() {
|
|||||||
`${name} deleted`,
|
`${name} deleted`,
|
||||||
)
|
)
|
||||||
},
|
},
|
||||||
[rules, commitRules, confirm],
|
[rules, commitRules, confirm, isLiveDefault, killSwitch, successorDefault],
|
||||||
)
|
)
|
||||||
|
|
||||||
// Insert a new rule just above the catch-all (so a specific rule can actually match).
|
// Insert a new rule just above the catch-all (so a specific rule can actually match).
|
||||||
|
|||||||
@@ -1,7 +1,7 @@
|
|||||||
import './Settings.css'
|
import './Settings.css'
|
||||||
import { useCallback, useEffect, useRef, useState } from 'react'
|
import { useCallback, useEffect, useRef, useState } from 'react'
|
||||||
import type { ReactNode } from 'react'
|
import type { ReactNode } from 'react'
|
||||||
import { Button, Led, Select, Toggle } from '../components'
|
import { Button, Led, Select, Toggle, useConfirm } from '../components'
|
||||||
import { apply as apiApply, downloadLog, getConfig, putConfig, ApiError } from '../api'
|
import { apply as apiApply, downloadLog, getConfig, putConfig, ApiError } from '../api'
|
||||||
import type { Globals, LogRange, Model } from '../api'
|
import type { Globals, LogRange, Model } from '../api'
|
||||||
|
|
||||||
@@ -125,6 +125,7 @@ const STATS_BACKENDS: ReadonlyArray<{ value: string; label: string }> = [
|
|||||||
// ---- page ------------------------------------------------------------------
|
// ---- page ------------------------------------------------------------------
|
||||||
|
|
||||||
export default function Settings() {
|
export default function Settings() {
|
||||||
|
const confirm = useConfirm()
|
||||||
const [config, setConfig] = useState<Model | null>(null)
|
const [config, setConfig] = useState<Model | null>(null)
|
||||||
const [loadError, setLoadError] = useState<string | null>(null)
|
const [loadError, setLoadError] = useState<string | null>(null)
|
||||||
|
|
||||||
@@ -257,6 +258,48 @@ export default function Settings() {
|
|||||||
const groupHealthOn = globals?.GroupHealth !== false
|
const groupHealthOn = globals?.GroupHealth !== false
|
||||||
|
|
||||||
const killSwitch = globals?.KillSwitch === 'open' ? 'open' : 'closed'
|
const killSwitch = globals?.KillSwitch === 'open' ? 'open' : 'closed'
|
||||||
|
|
||||||
|
/**
|
||||||
|
* The master switch, which is the most destructive control in the panel and was
|
||||||
|
* the only one that asked nothing.
|
||||||
|
*
|
||||||
|
* Turning it off is not "pausing the proxy": apply.go runs Teardown() — the nft
|
||||||
|
* table goes, the policy routing goes, `plane` becomes `none`. The kill-switch
|
||||||
|
* does not save you, because a kill-switch is a rule in a table that no longer
|
||||||
|
* exists. Everything on the LAN then leaves through the plain WAN, unproxied and
|
||||||
|
* unfiltered. Deleting a rule-set asked for confirmation; this did not.
|
||||||
|
*
|
||||||
|
* Turning it back ON is not destructive and is not gated.
|
||||||
|
*/
|
||||||
|
const toggleService = useCallback(
|
||||||
|
async (on: boolean) => {
|
||||||
|
if (!on) {
|
||||||
|
const ok = await confirm({
|
||||||
|
label: 'Turn off the service',
|
||||||
|
title: 'Turn the proxy engine off?',
|
||||||
|
body: (
|
||||||
|
<>
|
||||||
|
This tears the whole data plane down — the firewall table, the policy routing and the
|
||||||
|
DNS interception are removed, not paused. Nothing is proxied, filtered or blocked, and
|
||||||
|
every device leaves through your normal internet connection with its real address.{' '}
|
||||||
|
{killSwitch === 'closed' ? (
|
||||||
|
<>
|
||||||
|
The kill-switch does not hold here: with nothing installed there is nothing left
|
||||||
|
to block with.
|
||||||
|
</>
|
||||||
|
) : (
|
||||||
|
<>The kill-switch is already open, so nothing changes about that.</>
|
||||||
|
)}
|
||||||
|
</>
|
||||||
|
),
|
||||||
|
confirmLabel: 'Turn it off',
|
||||||
|
})
|
||||||
|
if (!ok) return
|
||||||
|
}
|
||||||
|
setGlobal('Enabled', on, on ? 'Engine enabled' : 'Engine disabled')
|
||||||
|
},
|
||||||
|
[confirm, killSwitch, setGlobal],
|
||||||
|
)
|
||||||
const killNote =
|
const killNote =
|
||||||
killSwitch === 'open'
|
killSwitch === 'open'
|
||||||
? 'Fail-open — if the engine stops, traffic falls back to the direct WAN. Stays online, but unprotected.'
|
? 'Fail-open — if the engine stops, traffic falls back to the direct WAN. Stays online, but unprotected.'
|
||||||
@@ -296,10 +339,13 @@ export default function Settings() {
|
|||||||
<div className="set-groups">
|
<div className="set-groups">
|
||||||
{/* ---- SERVICE ---- */}
|
{/* ---- SERVICE ---- */}
|
||||||
<Group title="Service" count={globals?.Enabled ? 'enabled' : 'disabled'}>
|
<Group title="Service" count={globals?.Enabled ? 'enabled' : 'disabled'}>
|
||||||
<Field label="Proxy engine" note="Master on/off for the whole appliance.">
|
<Field
|
||||||
|
label="Proxy engine"
|
||||||
|
note="Master on/off for the whole appliance. Off removes the firewall table and the policy routing — every device goes out directly, with no kill-switch to catch it."
|
||||||
|
>
|
||||||
<Toggle
|
<Toggle
|
||||||
pressed={globals?.Enabled ?? false}
|
pressed={globals?.Enabled ?? false}
|
||||||
onChange={(on) => setGlobal('Enabled', on, on ? 'Engine enabled' : 'Engine disabled')}
|
onChange={(on) => void toggleService(on)}
|
||||||
label={globals?.Enabled ? 'Disable proxy engine' : 'Enable proxy engine'}
|
label={globals?.Enabled ? 'Disable proxy engine' : 'Enable proxy engine'}
|
||||||
size="md"
|
size="md"
|
||||||
disabled={busy || !ready}
|
disabled={busy || !ready}
|
||||||
|
|||||||
@@ -1054,8 +1054,11 @@
|
|||||||
* that before a word has been read.
|
* that before a word has been read.
|
||||||
*
|
*
|
||||||
* The two marks carry two different facts and must not be conflated:
|
* The two marks carry two different facts and must not be conflated:
|
||||||
* - the LAMP is that hop's own measurement (good / warn / crit / unlit), and a
|
* - the LAMP is that hop's own measurement (good / warn / crit / unlit). Below
|
||||||
* hop below the break that answered keeps its green, because it did answer;
|
* the break there is no measurement to draw: the daemon stops walking at the
|
||||||
|
* first dead hop, so those lamps are UNLIT and the row says which hop stopped
|
||||||
|
* the walk. Unlit is never a shade of red — it claims nothing, which is the
|
||||||
|
* truth about a hop nobody dialled;
|
||||||
* - the CONDUCTOR is reachability through the path, which really does stop.
|
* - the CONDUCTOR is reachability through the path, which really does stop.
|
||||||
*
|
*
|
||||||
* Orange is untouched here. Semantics carry every colour, and everything that is
|
* Orange is untouched here. Semantics carry every colour, and everything that is
|
||||||
@@ -1199,6 +1202,12 @@
|
|||||||
font-size: 12px;
|
font-size: 12px;
|
||||||
color: var(--faint);
|
color: var(--faint);
|
||||||
}
|
}
|
||||||
|
/* A blocked hop's phrase carries a tooltip with the blocking hop's engine
|
||||||
|
outbound, so it takes the same help cursor as .ch-dead. No colour of its own:
|
||||||
|
the finding is red once, on the hop that actually failed. */
|
||||||
|
.ch-blocked {
|
||||||
|
cursor: help;
|
||||||
|
}
|
||||||
.ch-age {
|
.ch-age {
|
||||||
margin-left: auto;
|
margin-left: auto;
|
||||||
font-size: 10.5px;
|
font-size: 10.5px;
|
||||||
@@ -1251,6 +1260,11 @@
|
|||||||
font-size: 10.5px;
|
font-size: 10.5px;
|
||||||
color: var(--faint);
|
color: var(--faint);
|
||||||
}
|
}
|
||||||
|
/* On a blocked hop this chip says "set to", not "now": a pick nothing crossed.
|
||||||
|
It steps back to faint so it can't be mistaken for a live reading. */
|
||||||
|
.ch-hop--blocked .ch-now {
|
||||||
|
color: var(--faint);
|
||||||
|
}
|
||||||
.ch-now {
|
.ch-now {
|
||||||
max-width: 28ch;
|
max-width: 28ch;
|
||||||
overflow: hidden;
|
overflow: hidden;
|
||||||
|
|||||||
+66
-16
@@ -2242,12 +2242,22 @@ function ChainRow({
|
|||||||
onDelete: () => void
|
onDelete: () => void
|
||||||
}) {
|
}) {
|
||||||
const hops = asArray(chain.Hops)
|
const hops = asArray(chain.Hops)
|
||||||
|
// A LEADING `egress:` is not a hop and the rail below already knows it: the
|
||||||
|
// daemon lifts it into hop 1's entry detour (see hopLabels), so it is tagged
|
||||||
|
// "entry" and never numbered. The badge counted it anyway, which is how a chain
|
||||||
|
// drawn with four hops came to be labelled "5 hops" directly above them.
|
||||||
|
const entryEgress = hops.length > 0 && hops[0].startsWith('egress:')
|
||||||
|
const numbered = entryEgress ? hops.length - 1 : hops.length
|
||||||
return (
|
return (
|
||||||
<li className="tg-row">
|
<li className="tg-row">
|
||||||
<div className="tg-row-main">
|
<div className="tg-row-main">
|
||||||
<div className="tg-row-l1">
|
<div className="tg-row-l1">
|
||||||
<span className="tg-row-name">{chain.Name}</span>
|
<span className="tg-row-name">{chain.Name}</span>
|
||||||
<span className="tg-badge">{hops.length} hop{hops.length === 1 ? '' : 's'}</span>
|
<span className="tg-badge">
|
||||||
|
{numbered === 0 && entryEgress
|
||||||
|
? 'entry only · no exit'
|
||||||
|
: `${numbered} hop${numbered === 1 ? '' : 's'}`}
|
||||||
|
</span>
|
||||||
</div>
|
</div>
|
||||||
<div className="tg-row-l2">
|
<div className="tg-row-l2">
|
||||||
{hops.length === 0 ? (
|
{hops.length === 0 ? (
|
||||||
@@ -2337,6 +2347,8 @@ function hopLabels(defs: string[], hops: ChainHopHealth[]): (string | undefined)
|
|||||||
*
|
*
|
||||||
* `dead` is crit and `untested` is an UNLIT socket — never red, because nothing
|
* `dead` is crit and `untested` is an UNLIT socket — never red, because nothing
|
||||||
* has been measured and an unlit lamp is this panel's way of saying "no verdict".
|
* has been measured and an unlit lamp is this panel's way of saying "no verdict".
|
||||||
|
* That covers a hop the walk never reached (`blocked_by`) too: it is neither
|
||||||
|
* healthy nor broken, and unlit is the only mark that claims neither.
|
||||||
* The fourth case is the page's existing house reading, applied here for
|
* The fourth case is the page's existing house reading, applied here for
|
||||||
* consistency rather than invented: a group hop that is carrying traffic but has
|
* consistency rather than invented: a group hop that is carrying traffic but has
|
||||||
* confirmed failures on its board is amber. `state` stays the daemon's word for
|
* confirmed failures on its board is amber. `state` stays the daemon's word for
|
||||||
@@ -2360,12 +2372,20 @@ function hopLed(h: ChainHopHealth): LedVariant {
|
|||||||
* conductor answers the second one before you have read a single word, which is
|
* conductor answers the second one before you have read a single word, which is
|
||||||
* the whole reason this feature exists.
|
* the whole reason this feature exists.
|
||||||
*
|
*
|
||||||
* It stays honest by keeping the two facts on two different marks. Each LAMP is
|
* It stays honest by keeping two different facts on two different marks. The
|
||||||
* that hop's own measurement and never changes because of a hop in front of it —
|
* CONDUCTOR is reachability through the path, and that genuinely does stop at the
|
||||||
* a hop after the break that answers still shows green, because it really did
|
* break. The LAMPS are measurements — and there are none below the break to show:
|
||||||
* answer. The CONDUCTOR is reachability through the path, and that genuinely does
|
* the daemon walks the path in order and stops at the first hop that does not
|
||||||
* stop at the break. Nothing here is re-derived from the daemon's counters; the
|
* answer, because every later hop is dialled THROUGH that one. Those hops arrive
|
||||||
* only thing the panel adds is what a chain structurally is.
|
* `untested` with `blocked_by` naming the hop that stopped the walk, so their
|
||||||
|
* lamps stay UNLIT: not a soft red, not a pale green, just this panel's way of
|
||||||
|
* saying no verdict exists about a hop nobody reached. The row says so in words
|
||||||
|
* too, naming that hop, because "why is this row empty" is the question the shape
|
||||||
|
* alone cannot answer.
|
||||||
|
*
|
||||||
|
* Nothing here is re-derived from the daemon's counters — the block, the zeroed
|
||||||
|
* numbers and the state all come off the wire. The only thing the panel adds is
|
||||||
|
* what a chain structurally is.
|
||||||
*
|
*
|
||||||
* Everything around the rail is deliberately quiet: no colour but the semantic
|
* Everything around the rail is deliberately quiet: no colour but the semantic
|
||||||
* lamps, no motion at all, the orange accent untouched.
|
* lamps, no motion at all, the orange accent untouched.
|
||||||
@@ -2412,9 +2432,14 @@ function ChainHopRail({
|
|||||||
{ordered.map((h, i) => {
|
{ordered.map((h, i) => {
|
||||||
const label = labels[i]
|
const label = labels[i]
|
||||||
const severed = breakAt >= 0 && i > breakAt
|
const severed = breakAt >= 0 && i > breakAt
|
||||||
|
// Straight off the wire: present ⇒ the walk never reached this hop, so
|
||||||
|
// there is nothing measured here and the daemon has already named the
|
||||||
|
// hop that stopped it. Never inferred from the counters.
|
||||||
|
const blocked = h.blocked_by
|
||||||
const cls = [
|
const cls = [
|
||||||
'ch-hop',
|
'ch-hop',
|
||||||
`ch-hop--${h.state}`,
|
`ch-hop--${h.state}`,
|
||||||
|
blocked ? 'ch-hop--blocked' : '',
|
||||||
severed ? 'ch-hop--severed' : '',
|
severed ? 'ch-hop--severed' : '',
|
||||||
h.exit ? 'ch-hop--exit' : '',
|
h.exit ? 'ch-hop--exit' : '',
|
||||||
i === 0 ? 'ch-hop--first' : '',
|
i === 0 ? 'ch-hop--first' : '',
|
||||||
@@ -2439,7 +2464,20 @@ function ChainHopRail({
|
|||||||
{h.state === 'alive' && h.delay_ms > 0 && (
|
{h.state === 'alive' && h.delay_ms > 0 && (
|
||||||
<span className="ch-delay mono">{h.delay_ms} ms</span>
|
<span className="ch-delay mono">{h.delay_ms} ms</span>
|
||||||
)}
|
)}
|
||||||
{h.state === 'untested' && <span className="ch-quiet">not measured yet</span>}
|
{/* Two different silences. A plain untested hop is a timing
|
||||||
|
gap that fills in by itself; a BLOCKED one never will,
|
||||||
|
because the walk stopped above it — so it says which hop
|
||||||
|
stopped it instead of implying someone should wait. */}
|
||||||
|
{blocked ? (
|
||||||
|
<span
|
||||||
|
className="ch-quiet ch-blocked"
|
||||||
|
title={`hop ${blocked.index} did not answer, so nothing was dialled through it (engine outbound ${blocked.tag})`}
|
||||||
|
>
|
||||||
|
no reading — the probe stopped at hop {blocked.index}
|
||||||
|
</span>
|
||||||
|
) : (
|
||||||
|
h.state === 'untested' && <span className="ch-quiet">not measured yet</span>
|
||||||
|
)}
|
||||||
{age && <span className="ch-age mono">{age}</span>}
|
{age && <span className="ch-age mono">{age}</span>}
|
||||||
</div>
|
</div>
|
||||||
|
|
||||||
@@ -2447,12 +2485,16 @@ function ChainHopRail({
|
|||||||
only restate the lamp. A group hop rolls up its per-hop member
|
only restate the lamp. A group hop rolls up its per-hop member
|
||||||
copies, and those read exactly as they do everywhere else in
|
copies, and those read exactly as they do everywhere else in
|
||||||
this app: alive out of TESTED, with the untested remainder as a
|
this app: alive out of TESTED, with the untested remainder as a
|
||||||
quiet aside only when there is one. */}
|
quiet aside only when there is one. A blocked hop has those
|
||||||
|
counters zeroed by the daemon, so it lands in the tested === 0
|
||||||
|
branch — and there it must say the members were never REACHED,
|
||||||
|
not that they are still waiting their turn. */}
|
||||||
{h.kind === 'group' && h.total > 0 && (
|
{h.kind === 'group' && h.total > 0 && (
|
||||||
<div className="ch-l2">
|
<div className="ch-l2">
|
||||||
{h.tested === 0 ? (
|
{h.tested === 0 ? (
|
||||||
<span className="ch-rest mono">
|
<span className="ch-rest mono">
|
||||||
{h.total} member{h.total === 1 ? '' : 's'}, none measured
|
{h.total} member{h.total === 1 ? '' : 's'},{' '}
|
||||||
|
{blocked ? 'none of them reached' : 'none measured'}
|
||||||
</span>
|
</span>
|
||||||
) : (
|
) : (
|
||||||
<>
|
<>
|
||||||
@@ -2475,12 +2517,19 @@ function ChainHopRail({
|
|||||||
)}
|
)}
|
||||||
</>
|
</>
|
||||||
)}
|
)}
|
||||||
|
{/* The wrapper keeps its pick even when nothing crossed it,
|
||||||
|
so on a blocked hop this is the node it WOULD use — say
|
||||||
|
that, rather than "now", which claims live traffic. */}
|
||||||
{h.selected && (
|
{h.selected && (
|
||||||
<span
|
<span
|
||||||
className="ch-now mono"
|
className="ch-now mono"
|
||||||
title={`Traffic crossing hop ${h.index} of “${chain}” is on ${h.selected}`}
|
title={
|
||||||
|
blocked
|
||||||
|
? `Hop ${h.index} of “${chain}” is set to ${h.selected}; nothing crossed it to measure`
|
||||||
|
: `Traffic crossing hop ${h.index} of “${chain}” is on ${h.selected}`
|
||||||
|
}
|
||||||
>
|
>
|
||||||
now → {h.selected}
|
{blocked ? 'set to' : 'now'} → {h.selected}
|
||||||
</span>
|
</span>
|
||||||
)}
|
)}
|
||||||
</div>
|
</div>
|
||||||
@@ -2493,14 +2542,15 @@ function ChainHopRail({
|
|||||||
|
|
||||||
{/* The sentence the rail's shape implies, written out — because the break is
|
{/* The sentence the rail's shape implies, written out — because the break is
|
||||||
the answer someone came here for, and a graphic alone should never be the
|
the answer someone came here for, and a graphic alone should never be the
|
||||||
only place a finding exists. */}
|
only place a finding exists. It names the dead hop, since that is the one
|
||||||
|
thing here anybody can act on. */}
|
||||||
{breakAt >= 0 && (
|
{breakAt >= 0 && (
|
||||||
<p className="gh-say gh-say--bad">
|
<p className="gh-say gh-say--bad">
|
||||||
Hop {ordered[breakAt].index}
|
Hop {ordered[breakAt].index}
|
||||||
{labels[breakAt] ? ` (${labels[breakAt]})` : ''} was probed and did not answer. A chain is
|
{labels[breakAt] ? ` (${labels[breakAt]})` : ''} was probed and did not answer, so traffic
|
||||||
one path, so traffic stops there
|
stops there
|
||||||
{breakAt < ordered.length - 1
|
{breakAt < ordered.length - 1
|
||||||
? ' — the hops after it answer on their own, but nothing reaches them through this chain.'
|
? ' — and the hops below it are dialled through it, so nothing reached them and nothing is known about them.'
|
||||||
: '.'}
|
: '.'}
|
||||||
</p>
|
</p>
|
||||||
)}
|
)}
|
||||||
|
|||||||
@@ -0,0 +1,91 @@
|
|||||||
|
// pendingConfirm — the record of an armed auto-rollback, shared by the whole panel.
|
||||||
|
//
|
||||||
|
// Run with `npm test`. The module imports React only for its hook; the plain
|
||||||
|
// functions exercised here touch neither React nor the DOM, and `localStorage` is
|
||||||
|
// absent under node, which is itself one of the cases worth pinning (the panel
|
||||||
|
// must still work, it just forgets on reload).
|
||||||
|
//
|
||||||
|
// What these protect:
|
||||||
|
// - arming when commit-confirm is OFF must record nothing. The daemon does not
|
||||||
|
// arm a window then, and a countdown for a rollback that will never happen is
|
||||||
|
// the same class of lie as the "Confirmed" message this module replaced.
|
||||||
|
// - a window that has elapsed reads as gone, so nothing renders "0 s left".
|
||||||
|
// - expiry notifies exactly once even though several components watch it.
|
||||||
|
|
||||||
|
import { test } from 'node:test'
|
||||||
|
import assert from 'node:assert/strict'
|
||||||
|
|
||||||
|
import {
|
||||||
|
armPendingConfirm,
|
||||||
|
clearPendingConfirm,
|
||||||
|
confirmTimeout,
|
||||||
|
noteConfirmTimeout,
|
||||||
|
onPendingConfirmExpire,
|
||||||
|
readPendingConfirm,
|
||||||
|
} from './pendingConfirm.ts'
|
||||||
|
|
||||||
|
test('commit-confirm off ⇒ arming records nothing', () => {
|
||||||
|
noteConfirmTimeout(0)
|
||||||
|
assert.equal(confirmTimeout(), 0)
|
||||||
|
armPendingConfirm()
|
||||||
|
assert.equal(readPendingConfirm(), null)
|
||||||
|
})
|
||||||
|
|
||||||
|
test('a window is recorded with the timeout the config reported', () => {
|
||||||
|
noteConfirmTimeout(90)
|
||||||
|
armPendingConfirm()
|
||||||
|
const p = readPendingConfirm()
|
||||||
|
assert.notEqual(p, null)
|
||||||
|
assert.equal(p!.total, 90)
|
||||||
|
// Deadline is in the future and within a second of now + the window.
|
||||||
|
const left = (p!.until - Date.now()) / 1000
|
||||||
|
assert.ok(left > 89 && left <= 90, `expected ~90s left, got ${left}`)
|
||||||
|
clearPendingConfirm()
|
||||||
|
assert.equal(readPendingConfirm(), null)
|
||||||
|
})
|
||||||
|
|
||||||
|
test('switching commit-confirm off drops a window that was already armed', () => {
|
||||||
|
noteConfirmTimeout(60)
|
||||||
|
armPendingConfirm()
|
||||||
|
assert.notEqual(readPendingConfirm(), null)
|
||||||
|
noteConfirmTimeout(0)
|
||||||
|
assert.equal(readPendingConfirm(), null)
|
||||||
|
})
|
||||||
|
|
||||||
|
test('an elapsed window reads as gone, never as a countdown at zero', () => {
|
||||||
|
noteConfirmTimeout(1)
|
||||||
|
armPendingConfirm()
|
||||||
|
const p = readPendingConfirm()
|
||||||
|
assert.notEqual(p, null)
|
||||||
|
// Wind the clock forward rather than sleeping through the window.
|
||||||
|
const realNow = Date.now
|
||||||
|
Date.now = () => realNow() + 5000
|
||||||
|
try {
|
||||||
|
assert.equal(readPendingConfirm(), null)
|
||||||
|
} finally {
|
||||||
|
Date.now = realNow
|
||||||
|
}
|
||||||
|
clearPendingConfirm()
|
||||||
|
})
|
||||||
|
|
||||||
|
test('confirming does NOT fire the expiry listeners', () => {
|
||||||
|
noteConfirmTimeout(30)
|
||||||
|
let fired = 0
|
||||||
|
const off = onPendingConfirmExpire(() => {
|
||||||
|
fired++
|
||||||
|
})
|
||||||
|
armPendingConfirm()
|
||||||
|
clearPendingConfirm()
|
||||||
|
off()
|
||||||
|
assert.equal(fired, 0)
|
||||||
|
})
|
||||||
|
|
||||||
|
test('a nonsense timeout is ignored rather than taken as "off"', () => {
|
||||||
|
noteConfirmTimeout(45)
|
||||||
|
noteConfirmTimeout(Number.NaN)
|
||||||
|
noteConfirmTimeout(-1)
|
||||||
|
noteConfirmTimeout(undefined)
|
||||||
|
assert.equal(confirmTimeout(), 45)
|
||||||
|
clearPendingConfirm()
|
||||||
|
noteConfirmTimeout(0)
|
||||||
|
})
|
||||||
@@ -0,0 +1,186 @@
|
|||||||
|
// The commit-confirm window, as ONE fact the whole panel can see.
|
||||||
|
//
|
||||||
|
// WHY THIS EXISTS. `POST /api/apply` always arms an auto-rollback for
|
||||||
|
// `Globals.ConfirmTimeout` seconds (shater/panel/api.go handleApply →
|
||||||
|
// ArmRollback) — EVERY Apply button does that, not just the one on the Apply
|
||||||
|
// page. But the countdown, and the button that stops it, lived in one component's
|
||||||
|
// local state. So:
|
||||||
|
//
|
||||||
|
// - pressing Apply on Routing/DNS/Nodes/Settings/Devices/Profiles/Targets said
|
||||||
|
// "Applied" and nothing else; the operator walked away and the router quietly
|
||||||
|
// reverted a minute later;
|
||||||
|
// - reloading the tab wiped the countdown AND the "Keep this config" button, so
|
||||||
|
// there was no way left to confirm from the panel at all.
|
||||||
|
//
|
||||||
|
// The daemon does not report a deadline, so this module records the one the panel
|
||||||
|
// itself armed, in `localStorage`. That is a deliberately modest claim — it knows
|
||||||
|
// about windows THIS BROWSER opened and says nothing about one opened elsewhere —
|
||||||
|
// but it survives a reload, a new tab and a navigation, which is what the two
|
||||||
|
// failures above needed.
|
||||||
|
//
|
||||||
|
// It is also the answer to "is there anything to confirm?". `apply.Confirm()`
|
||||||
|
// returns nil unconditionally, so a Confirm button that is always live can only
|
||||||
|
// ever report success. Gating it on a record here means the panel offers the
|
||||||
|
// action when it knows a window is open, and then reports an outcome it knows.
|
||||||
|
|
||||||
|
import { useEffect, useState } from 'react'
|
||||||
|
|
||||||
|
/** A live commit-confirm window the panel armed. */
|
||||||
|
export interface PendingConfirm {
|
||||||
|
/** Epoch ms at which the daemon auto-rolls back if nobody confirms. */
|
||||||
|
until: number
|
||||||
|
/** The window it started with, in seconds — the progress bar's denominator. */
|
||||||
|
total: number
|
||||||
|
}
|
||||||
|
|
||||||
|
const KEY = 'shater.pendingConfirm'
|
||||||
|
|
||||||
|
type Listener = () => void
|
||||||
|
const listeners = new Set<Listener>()
|
||||||
|
const expiryListeners = new Set<Listener>()
|
||||||
|
|
||||||
|
/** localStorage is absent under SSR/tests and throws in some privacy modes. A
|
||||||
|
* panel that cannot remember a window must still work — it just forgets on
|
||||||
|
* reload, which is exactly the old behaviour and no worse. */
|
||||||
|
function store(): Storage | null {
|
||||||
|
try {
|
||||||
|
return typeof localStorage === 'undefined' ? null : localStorage
|
||||||
|
} catch {
|
||||||
|
return null
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
function load(): PendingConfirm | null {
|
||||||
|
const s = store()
|
||||||
|
if (!s) return null
|
||||||
|
try {
|
||||||
|
const raw = s.getItem(KEY)
|
||||||
|
if (!raw) return null
|
||||||
|
const v = JSON.parse(raw) as Partial<PendingConfirm>
|
||||||
|
if (typeof v.until !== 'number' || typeof v.total !== 'number') return null
|
||||||
|
if (!Number.isFinite(v.until)) return null
|
||||||
|
return { until: v.until, total: v.total }
|
||||||
|
} catch {
|
||||||
|
return null
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// The single in-process copy. Storage is the durable mirror, not the source of
|
||||||
|
// truth for a running tab: a `storage` event re-hydrates it when another tab
|
||||||
|
// writes.
|
||||||
|
let armed: PendingConfirm | null = load()
|
||||||
|
|
||||||
|
function emit() {
|
||||||
|
for (const l of [...listeners]) l()
|
||||||
|
}
|
||||||
|
|
||||||
|
function write(v: PendingConfirm | null) {
|
||||||
|
armed = v
|
||||||
|
const s = store()
|
||||||
|
if (s) {
|
||||||
|
try {
|
||||||
|
if (v) s.setItem(KEY, JSON.stringify(v))
|
||||||
|
else s.removeItem(KEY)
|
||||||
|
} catch {
|
||||||
|
// Storage full or blocked — the in-process copy still drives this tab.
|
||||||
|
}
|
||||||
|
}
|
||||||
|
emit()
|
||||||
|
}
|
||||||
|
|
||||||
|
/** The armed window, or null when there is none or it has already elapsed. */
|
||||||
|
export function readPendingConfirm(): PendingConfirm | null {
|
||||||
|
if (!armed) return null
|
||||||
|
return armed.until > Date.now() ? armed : null
|
||||||
|
}
|
||||||
|
|
||||||
|
// ---- the window's length ----------------------------------------------------
|
||||||
|
|
||||||
|
// `Globals.ConfirmTimeout` is all that is needed to arm a window, and every page
|
||||||
|
// reads the config anyway — so api.getConfig() feeds it here rather than each
|
||||||
|
// caller threading it through. 0 (or never seen) means commit-confirm is off, and
|
||||||
|
// arming then does nothing: an apply on such a router really is immediate.
|
||||||
|
let timeout = 0
|
||||||
|
|
||||||
|
export function noteConfirmTimeout(seconds: number | undefined) {
|
||||||
|
if (typeof seconds !== 'number' || !Number.isFinite(seconds) || seconds < 0) return
|
||||||
|
timeout = Math.floor(seconds)
|
||||||
|
// A window armed before commit-confirm was switched off is no longer real —
|
||||||
|
// drop it rather than count down to an event that will not happen.
|
||||||
|
if (timeout === 0 && armed) write(null)
|
||||||
|
}
|
||||||
|
|
||||||
|
export function confirmTimeout(): number {
|
||||||
|
return timeout
|
||||||
|
}
|
||||||
|
|
||||||
|
/** Record the window an apply just opened. Call only when the apply CHANGED
|
||||||
|
* something: an unchanged apply reconciles nothing and arms nothing. */
|
||||||
|
export function armPendingConfirm() {
|
||||||
|
if (timeout <= 0) return
|
||||||
|
write({ until: Date.now() + timeout * 1000, total: timeout })
|
||||||
|
}
|
||||||
|
|
||||||
|
/** Confirm and rollback both end the window. */
|
||||||
|
export function clearPendingConfirm() {
|
||||||
|
if (armed) write(null)
|
||||||
|
}
|
||||||
|
|
||||||
|
/** Fires when a window ran out on its own — i.e. the daemon has reverted — and
|
||||||
|
* NOT when it was confirmed or rolled back. Returns an unsubscribe. */
|
||||||
|
export function onPendingConfirmExpire(fn: Listener): () => void {
|
||||||
|
expiryListeners.add(fn)
|
||||||
|
return () => {
|
||||||
|
expiryListeners.delete(fn)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/** Idempotent: several mounted countdowns race to notice the same deadline, and
|
||||||
|
* only the first one gets to announce it. */
|
||||||
|
function expire() {
|
||||||
|
if (!armed) return
|
||||||
|
write(null)
|
||||||
|
for (const l of [...expiryListeners]) l()
|
||||||
|
}
|
||||||
|
|
||||||
|
// Another tab confirming, rolling back or applying is the same event as this one
|
||||||
|
// doing it.
|
||||||
|
if (typeof window !== 'undefined') {
|
||||||
|
window.addEventListener('storage', (e) => {
|
||||||
|
if (e.key !== KEY && e.key !== null) return
|
||||||
|
armed = load()
|
||||||
|
emit()
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* The armed window and its remaining seconds, ticking once a second.
|
||||||
|
*
|
||||||
|
* Returns null when nothing is armed. While non-null `remaining` is at least 1:
|
||||||
|
* reaching zero clears the record and notifies {@link onPendingConfirmExpire}, so
|
||||||
|
* no component ever renders "0 s left" for a window that is already over.
|
||||||
|
*/
|
||||||
|
export function usePendingConfirm(): { pending: PendingConfirm; remaining: number } | null {
|
||||||
|
const [, tick] = useState(0)
|
||||||
|
|
||||||
|
useEffect(() => {
|
||||||
|
const sync = () => tick((n) => n + 1)
|
||||||
|
listeners.add(sync)
|
||||||
|
sync()
|
||||||
|
return () => {
|
||||||
|
listeners.delete(sync)
|
||||||
|
}
|
||||||
|
}, [])
|
||||||
|
|
||||||
|
useEffect(() => {
|
||||||
|
const id = window.setInterval(() => {
|
||||||
|
if (armed && armed.until <= Date.now()) expire()
|
||||||
|
else if (armed) tick((n) => n + 1)
|
||||||
|
}, 1000)
|
||||||
|
return () => window.clearInterval(id)
|
||||||
|
}, [])
|
||||||
|
|
||||||
|
const pending = readPendingConfirm()
|
||||||
|
if (!pending) return null
|
||||||
|
return { pending, remaining: Math.max(1, Math.ceil((pending.until - Date.now()) / 1000)) }
|
||||||
|
}
|
||||||
@@ -15,7 +15,7 @@
|
|||||||
import { test } from 'node:test'
|
import { test } from 'node:test'
|
||||||
import assert from 'node:assert/strict'
|
import assert from 'node:assert/strict'
|
||||||
|
|
||||||
import { protectionState } from './planeState.ts'
|
import { engineReadout, engineState, protectionState } from './planeState.ts'
|
||||||
import type { Status, Traffic } from './api.ts'
|
import type { Status, Traffic } from './api.ts'
|
||||||
|
|
||||||
/** A healthy, fully-installed router; `traffic` is what each case varies. */
|
/** A healthy, fully-installed router; `traffic` is what each case varies. */
|
||||||
@@ -148,3 +148,44 @@ test('daemon too old to send `plane` keeps its own fallback', () => {
|
|||||||
'Starting up',
|
'Starting up',
|
||||||
)
|
)
|
||||||
})
|
})
|
||||||
|
|
||||||
|
// --- engineState: the reading that could not say "down" ----------------------
|
||||||
|
//
|
||||||
|
// `apply.Status.running` was a hardcoded `true` on the daemon, so every panel LED
|
||||||
|
// derived from it was lit before it was read: App's master indicator could not
|
||||||
|
// reach its "Offline" branch, and Apply's "engine: running / stopped" row had one
|
||||||
|
// reachable value. These pin the three answers, and that "up" needs agreement.
|
||||||
|
|
||||||
|
test('engine_running:false is down even while the daemon claims it is running', () => {
|
||||||
|
assert.equal(engineState(status({ running: true, engine_running: false })), 'down')
|
||||||
|
assert.equal(engineReadout(status({ running: true, engine_running: false })).variant, 'crit')
|
||||||
|
assert.equal(engineReadout(status({ running: true, engine_running: false })).word, 'stopped')
|
||||||
|
})
|
||||||
|
|
||||||
|
test('a daemon that reports itself stopped is down whatever engine_running says', () => {
|
||||||
|
assert.equal(engineState(status({ running: false, engine_running: true })), 'down')
|
||||||
|
})
|
||||||
|
|
||||||
|
test('up needs both, and then active/idle splits the lamp', () => {
|
||||||
|
assert.equal(engineState(status({ running: true, engine_running: true })), 'up')
|
||||||
|
assert.equal(engineReadout(status({ active: true })).variant, 'on')
|
||||||
|
assert.equal(engineReadout(status({ active: true })).word, 'active')
|
||||||
|
assert.equal(engineReadout(status({ active: false })).variant, 'amber')
|
||||||
|
assert.equal(engineReadout(status({ active: false })).word, 'idle')
|
||||||
|
})
|
||||||
|
|
||||||
|
test('an older daemon with no engine_running is unknown — an unlit lamp, never green', () => {
|
||||||
|
const { engine_running, ...old } = status()
|
||||||
|
void engine_running
|
||||||
|
assert.equal(engineState(old as Status), 'unknown')
|
||||||
|
const r = engineReadout(old as Status)
|
||||||
|
assert.equal(r.variant, 'off')
|
||||||
|
assert.notEqual(r.variant, 'on')
|
||||||
|
assert.equal(r.word, 'not reported')
|
||||||
|
})
|
||||||
|
|
||||||
|
test('no status at all is unknown, not down', () => {
|
||||||
|
assert.equal(engineState(null), 'unknown')
|
||||||
|
assert.equal(engineReadout(null).variant, 'off')
|
||||||
|
assert.equal(engineReadout(null).word, 'checking…')
|
||||||
|
})
|
||||||
|
|||||||
+62
-10
@@ -21,6 +21,56 @@ export interface ProtectionState {
|
|||||||
alarm: boolean
|
alarm: boolean
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// ---------------------------------------------------------------------------
|
||||||
|
// Is the engine actually up?
|
||||||
|
// ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Three answers, and "unknown" is one of them.
|
||||||
|
*
|
||||||
|
* up — the sing-box process is running.
|
||||||
|
* down — it is not. Nothing is being proxied or filtered.
|
||||||
|
* unknown — nobody has told us. Never paint this green.
|
||||||
|
*
|
||||||
|
* THIS EXISTS BECAUSE `status.running` COULD NOT SAY "down". It is the DAEMON's
|
||||||
|
* own liveness, and on the daemons this panel shipped against it was a hardcoded
|
||||||
|
* `true` (apply.go) — so `running ? 'running' : 'stopped'` had exactly one
|
||||||
|
* reachable branch, and every LED derived from it was lit before it was read. A
|
||||||
|
* router whose engine failed to start, with a fail-closed holding plan installed
|
||||||
|
* and no internet on the LAN, showed three green lamps on the page people go to
|
||||||
|
* when they are trying to fix it.
|
||||||
|
*
|
||||||
|
* `engine_running` is the field that answers the question honestly, so it decides
|
||||||
|
* `up`. Either field may still prove a NEGATIVE — a daemon that reports itself
|
||||||
|
* stopped cannot be running an engine — and a negative always wins, so "up" needs
|
||||||
|
* both to agree. Neither field asserting anything leaves `unknown`.
|
||||||
|
*/
|
||||||
|
export type EngineState = 'up' | 'down' | 'unknown'
|
||||||
|
|
||||||
|
export function engineState(status: Status | null): EngineState {
|
||||||
|
if (!status) return 'unknown'
|
||||||
|
if (!status.running) return 'down'
|
||||||
|
if (typeof status.engine_running === 'boolean') return status.engine_running ? 'up' : 'down'
|
||||||
|
return 'unknown'
|
||||||
|
}
|
||||||
|
|
||||||
|
/** How the engine's lamp is painted and what the readout beside it says.
|
||||||
|
*
|
||||||
|
* `unknown` is an UNLIT socket, never amber and never green: amber is this
|
||||||
|
* panel's "degraded", and there is nothing to be degraded about when no reading
|
||||||
|
* has arrived. `down` is crit even when the kill-switch caught it — the engine
|
||||||
|
* being dead is the fault; whether traffic leaks is a separate lamp. */
|
||||||
|
export function engineReadout(status: Status | null): { variant: LedVariant; word: string } {
|
||||||
|
switch (engineState(status)) {
|
||||||
|
case 'down':
|
||||||
|
return { variant: 'crit', word: 'stopped' }
|
||||||
|
case 'up':
|
||||||
|
return status?.active ? { variant: 'on', word: 'active' } : { variant: 'amber', word: 'idle' }
|
||||||
|
default:
|
||||||
|
return { variant: 'off', word: status ? 'not reported' : 'checking…' }
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* `plane` + `engine_running` express the state more precisely than the three
|
* `plane` + `engine_running` express the state more precisely than the three
|
||||||
* booleans the old status strip exposed (engine active / config enabled / nft
|
* booleans the old status strip exposed (engine active / config enabled / nft
|
||||||
@@ -101,8 +151,18 @@ export function protectionState(status: Status | null): ProtectionState {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
// Older daemon with no `plane` field: fall back to what we can observe.
|
// Older daemon with no `plane` field: fall back to what we can observe. The
|
||||||
if (status.running && status.active && status.table) {
|
// engine's own state is asked FIRST — "stopped" is the answer that matters, and
|
||||||
|
// reading it off `running` is what used to make it unreachable (see engineState).
|
||||||
|
if (engineState(status) === 'down') {
|
||||||
|
return {
|
||||||
|
variant: 'crit',
|
||||||
|
headline: 'Service stopped',
|
||||||
|
detail: 'The engine isn’t running, so traffic isn’t being proxied or filtered.',
|
||||||
|
alarm: true,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if (status.active && status.table) {
|
||||||
return {
|
return {
|
||||||
variant: 'on',
|
variant: 'on',
|
||||||
headline: 'Protected',
|
headline: 'Protected',
|
||||||
@@ -110,14 +170,6 @@ export function protectionState(status: Status | null): ProtectionState {
|
|||||||
alarm: false,
|
alarm: false,
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
if (!status.running) {
|
|
||||||
return {
|
|
||||||
variant: 'crit',
|
|
||||||
headline: 'Service stopped',
|
|
||||||
detail: 'The service isn’t running, so traffic isn’t being handled.',
|
|
||||||
alarm: true,
|
|
||||||
}
|
|
||||||
}
|
|
||||||
return {
|
return {
|
||||||
variant: 'amber',
|
variant: 'amber',
|
||||||
headline: 'Starting up',
|
headline: 'Starting up',
|
||||||
|
|||||||
@@ -1,93 +0,0 @@
|
|||||||
// lx:begin awg
|
|
||||||
|
|
||||||
package group
|
|
||||||
|
|
||||||
import (
|
|
||||||
"github.com/sagernet/sing-box/adapter"
|
|
||||||
C "github.com/sagernet/sing-box/constant"
|
|
||||||
)
|
|
||||||
|
|
||||||
// suspendAmneziaWGConsumersOnWireGuardSwitch is called from Selector.SelectOutbound
|
|
||||||
// BEFORE the switch is committed. If the member about to be selected is — or chains
|
|
||||||
// down via detour to — a WireGuard-based endpoint (type "wireguard", covering plain
|
|
||||||
// WG and AmneziaWG), it walks UP from this group to every AmneziaWG endpoint that
|
|
||||||
// detours through it and suspends each one (brings its device down). Rationale:
|
|
||||||
// AmneziaWG traffic encapsulated inside a WireGuard tunnel hangs the kernel on
|
|
||||||
// Android; the static Start-guard cannot cover this because a selector's chosen
|
|
||||||
// member is only known at runtime.
|
|
||||||
//
|
|
||||||
// Called before s.selected is updated, so the race is closed: once the group
|
|
||||||
// points at the WireGuard member, the AmneziaWG consumers are already suspended
|
|
||||||
// (started=false) and a concurrent reconnect fails with "not ready" instead of
|
|
||||||
// sending a junk handshake into WireGuard.
|
|
||||||
func suspendAmneziaWGConsumersOnWireGuardSwitch(outboundManager adapter.OutboundManager, groupTag string, selected adapter.Outbound) {
|
|
||||||
if outboundManager == nil || groupTag == "" {
|
|
||||||
return
|
|
||||||
}
|
|
||||||
if !chainReachesWireGuard(outboundManager, selected, make(map[string]bool)) {
|
|
||||||
return
|
|
||||||
}
|
|
||||||
suspendAmneziaWGConsumers(outboundManager, groupTag, make(map[string]bool))
|
|
||||||
}
|
|
||||||
|
|
||||||
// chainReachesWireGuard reports whether outbound is — or transitively detours
|
|
||||||
// down to, or (being a group) contains a member that is — a WireGuard-based
|
|
||||||
// endpoint. visited guards against cycles.
|
|
||||||
func chainReachesWireGuard(outboundManager adapter.OutboundManager, outbound adapter.Outbound, visited map[string]bool) bool {
|
|
||||||
if outbound == nil {
|
|
||||||
return false
|
|
||||||
}
|
|
||||||
tag := outbound.Tag()
|
|
||||||
if tag != "" {
|
|
||||||
if visited[tag] {
|
|
||||||
return false
|
|
||||||
}
|
|
||||||
visited[tag] = true
|
|
||||||
}
|
|
||||||
if outbound.Type() == C.TypeWireGuard {
|
|
||||||
return true
|
|
||||||
}
|
|
||||||
// Down the detour chain (vless -> ... -> wireguard).
|
|
||||||
for _, dependency := range outbound.Dependencies() {
|
|
||||||
if member, loaded := outboundManager.Outbound(dependency); loaded {
|
|
||||||
if chainReachesWireGuard(outboundManager, member, visited) {
|
|
||||||
return true
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
// A nested group: any member reaching WireGuard counts.
|
|
||||||
if group, isGroup := outbound.(adapter.OutboundGroup); isGroup {
|
|
||||||
for _, memberTag := range group.All() {
|
|
||||||
if member, loaded := outboundManager.Outbound(memberTag); loaded {
|
|
||||||
if chainReachesWireGuard(outboundManager, member, visited) {
|
|
||||||
return true
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
return false
|
|
||||||
}
|
|
||||||
|
|
||||||
// suspendAmneziaWGConsumers walks UP from tag via the reverse-dependency ledger
|
|
||||||
// (ConsumersOf) and suspends every AmneziaWG endpoint that detours through it,
|
|
||||||
// directly or transitively (e.g. AWG -> vless -> group). visited guards cycles.
|
|
||||||
func suspendAmneziaWGConsumers(outboundManager adapter.OutboundManager, tag string, visited map[string]bool) {
|
|
||||||
for _, consumerTag := range outboundManager.ConsumersOf(tag) {
|
|
||||||
if visited[consumerTag] {
|
|
||||||
continue
|
|
||||||
}
|
|
||||||
visited[consumerTag] = true
|
|
||||||
consumer, loaded := outboundManager.Outbound(consumerTag)
|
|
||||||
if !loaded {
|
|
||||||
continue
|
|
||||||
}
|
|
||||||
if awg, isAWG := consumer.(adapter.AmneziaWGSuspendable); isAWG && awg.IsAmneziaWG() {
|
|
||||||
awg.SuspendAmneziaWG()
|
|
||||||
}
|
|
||||||
// Keep walking up: a non-AWG hop (vless) or a parent group may itself have
|
|
||||||
// an AmneziaWG consumer above it.
|
|
||||||
suspendAmneziaWGConsumers(outboundManager, consumerTag, visited)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
// lx:end awg
|
|
||||||
@@ -1,120 +0,0 @@
|
|||||||
// lx:begin awg
|
|
||||||
|
|
||||||
package group
|
|
||||||
|
|
||||||
import (
|
|
||||||
"testing"
|
|
||||||
|
|
||||||
"github.com/sagernet/sing-box/adapter"
|
|
||||||
C "github.com/sagernet/sing-box/constant"
|
|
||||||
)
|
|
||||||
|
|
||||||
// fakeOutbound is a minimal adapter.Outbound; only Type/Tag/Dependencies are read.
|
|
||||||
type fakeOutbound struct {
|
|
||||||
adapter.Outbound
|
|
||||||
tag string
|
|
||||||
outboundTyp string
|
|
||||||
detour string
|
|
||||||
}
|
|
||||||
|
|
||||||
func (o *fakeOutbound) Type() string { return o.outboundTyp }
|
|
||||||
func (o *fakeOutbound) Tag() string { return o.tag }
|
|
||||||
func (o *fakeOutbound) Dependencies() []string {
|
|
||||||
if o.detour == "" {
|
|
||||||
return nil
|
|
||||||
}
|
|
||||||
return []string{o.detour}
|
|
||||||
}
|
|
||||||
|
|
||||||
// fakeAWG implements adapter.AmneziaWGSuspendable and records suspension.
|
|
||||||
type fakeAWG struct {
|
|
||||||
fakeOutbound
|
|
||||||
awg bool
|
|
||||||
suspended bool
|
|
||||||
}
|
|
||||||
|
|
||||||
func (a *fakeAWG) IsAmneziaWG() bool { return a.awg }
|
|
||||||
func (a *fakeAWG) SuspendAmneziaWG() { a.suspended = true }
|
|
||||||
|
|
||||||
// fakeManager resolves tags and reverse-deps (ConsumersOf) from fixed maps.
|
|
||||||
type fakeManager struct {
|
|
||||||
adapter.OutboundManager
|
|
||||||
byTag map[string]adapter.Outbound
|
|
||||||
consumers map[string][]string
|
|
||||||
}
|
|
||||||
|
|
||||||
func (m *fakeManager) Outbound(tag string) (adapter.Outbound, bool) {
|
|
||||||
ob, ok := m.byTag[tag]
|
|
||||||
return ob, ok
|
|
||||||
}
|
|
||||||
func (m *fakeManager) ConsumersOf(tag string) []string { return m.consumers[tag] }
|
|
||||||
|
|
||||||
func TestChainReachesWireGuard(t *testing.T) {
|
|
||||||
wg := &fakeOutbound{tag: "wg", outboundTyp: C.TypeWireGuard}
|
|
||||||
vlessToWG := &fakeOutbound{tag: "v2wg", outboundTyp: C.TypeVLESS, detour: "wg"}
|
|
||||||
vlessLeaf := &fakeOutbound{tag: "vleaf", outboundTyp: C.TypeVLESS}
|
|
||||||
mgr := &fakeManager{byTag: map[string]adapter.Outbound{
|
|
||||||
"wg": wg, "v2wg": vlessToWG, "vleaf": vlessLeaf,
|
|
||||||
}}
|
|
||||||
|
|
||||||
if !chainReachesWireGuard(mgr, wg, map[string]bool{}) {
|
|
||||||
t.Fatal("direct wireguard member must reach wireguard")
|
|
||||||
}
|
|
||||||
if !chainReachesWireGuard(mgr, vlessToWG, map[string]bool{}) {
|
|
||||||
t.Fatal("vless detouring to wireguard must reach wireguard")
|
|
||||||
}
|
|
||||||
if chainReachesWireGuard(mgr, vlessLeaf, map[string]bool{}) {
|
|
||||||
t.Fatal("plain vless must not reach wireguard")
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
func TestSuspendAmneziaWGConsumers(t *testing.T) {
|
|
||||||
// awg-direct detours through the group "sel"
|
|
||||||
awgDirect := &fakeAWG{fakeOutbound: fakeOutbound{tag: "awg-direct", outboundTyp: C.TypeWireGuard, detour: "sel"}, awg: true}
|
|
||||||
// awg-via-hop -> vless-hop -> sel
|
|
||||||
awgViaHop := &fakeAWG{fakeOutbound: fakeOutbound{tag: "awg-hop", outboundTyp: C.TypeWireGuard, detour: "vless-hop"}, awg: true}
|
|
||||||
vlessHop := &fakeOutbound{tag: "vless-hop", outboundTyp: C.TypeVLESS, detour: "sel"}
|
|
||||||
// plain-wg detours through sel but is NOT amneziawg — must stay untouched
|
|
||||||
plainWG := &fakeAWG{fakeOutbound: fakeOutbound{tag: "plain-wg", outboundTyp: C.TypeWireGuard, detour: "sel"}, awg: false}
|
|
||||||
|
|
||||||
mgr := &fakeManager{
|
|
||||||
byTag: map[string]adapter.Outbound{
|
|
||||||
"awg-direct": awgDirect, "awg-hop": awgViaHop,
|
|
||||||
"vless-hop": vlessHop, "plain-wg": plainWG,
|
|
||||||
},
|
|
||||||
consumers: map[string][]string{
|
|
||||||
"sel": {"awg-direct", "vless-hop", "plain-wg"},
|
|
||||||
"vless-hop": {"awg-hop"},
|
|
||||||
},
|
|
||||||
}
|
|
||||||
|
|
||||||
suspendAmneziaWGConsumers(mgr, "sel", map[string]bool{})
|
|
||||||
|
|
||||||
if !awgDirect.suspended {
|
|
||||||
t.Error("direct AmneziaWG consumer must be suspended")
|
|
||||||
}
|
|
||||||
if !awgViaHop.suspended {
|
|
||||||
t.Error("transitive AmneziaWG consumer (via vless hop) must be suspended")
|
|
||||||
}
|
|
||||||
if plainWG.suspended {
|
|
||||||
t.Error("plain (non-AmneziaWG) wireguard consumer must NOT be suspended")
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
// A switch to a non-wireguard member must suspend nothing.
|
|
||||||
func TestSuspendSkippedForNonWireGuardSwitch(t *testing.T) {
|
|
||||||
awg := &fakeAWG{fakeOutbound: fakeOutbound{tag: "awg", outboundTyp: C.TypeWireGuard, detour: "sel"}, awg: true}
|
|
||||||
vlessLeaf := &fakeOutbound{tag: "vleaf", outboundTyp: C.TypeVLESS}
|
|
||||||
mgr := &fakeManager{
|
|
||||||
byTag: map[string]adapter.Outbound{"awg": awg, "vleaf": vlessLeaf},
|
|
||||||
consumers: map[string][]string{"sel": {"awg"}},
|
|
||||||
}
|
|
||||||
|
|
||||||
// selected member is plain vless (does not reach wireguard) → no suspension
|
|
||||||
suspendAmneziaWGConsumersOnWireGuardSwitch(mgr, "sel", vlessLeaf)
|
|
||||||
if awg.suspended {
|
|
||||||
t.Error("must not suspend when the selected member does not reach wireguard")
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
// lx:end awg
|
|
||||||
@@ -128,16 +128,6 @@ func (s *Selector) SelectOutbound(tag string) bool {
|
|||||||
if s.selected.Load() == detour {
|
if s.selected.Load() == detour {
|
||||||
return true
|
return true
|
||||||
}
|
}
|
||||||
// lx:begin awg
|
|
||||||
// Suspend AmneziaWG consumers BEFORE switching: if the new member is (or chains
|
|
||||||
// to) a WireGuard endpoint, any AmneziaWG endpoint that detours through this
|
|
||||||
// group would tunnel AWG inside WireGuard and hang the kernel on Android. Doing
|
|
||||||
// this before s.selected.Swap closes the race — by the time the group points at
|
|
||||||
// the WireGuard member, those consumers are already down (started=false), so a
|
|
||||||
// concurrent reconnect fails with "not ready" instead of sending a junk
|
|
||||||
// handshake into WireGuard.
|
|
||||||
suspendAmneziaWGConsumersOnWireGuardSwitch(s.outbound, s.Tag(), detour)
|
|
||||||
// lx:end awg
|
|
||||||
s.selected.Store(detour)
|
s.selected.Store(detour)
|
||||||
invalidateReachability(s.ctx) // lx: SPEC 020 — active selection changed
|
invalidateReachability(s.ctx) // lx: SPEC 020 — active selection changed
|
||||||
if s.Tag() != "" {
|
if s.Tag() != "" {
|
||||||
|
|||||||
+253
-46
@@ -51,6 +51,13 @@ type URLTest struct {
|
|||||||
// path that does not go through NewURLTest (hand-built groups in tests).
|
// path that does not go through NewURLTest (hand-built groups in tests).
|
||||||
// See option.URLTestOutboundOptions.SelfCheck for the full reasoning.
|
// See option.URLTestOutboundOptions.SelfCheck for the full reasoning.
|
||||||
selfCheckDisabled bool
|
selfCheckDisabled bool
|
||||||
|
// lx: health board §5.C — the RUNTIME half of the same question, read from
|
||||||
|
// the context registry at construction. selfCheckDisabled above says "this
|
||||||
|
// group is not used by the config"; this says "this group cannot be reached
|
||||||
|
// right now", which changes while the box runs and is therefore asked
|
||||||
|
// afresh at every scheduled check rather than stored. nil = no gate.
|
||||||
|
// See urltest.ProbeGate for why the two must stay separate.
|
||||||
|
probeGate urltest.ProbeGate
|
||||||
}
|
}
|
||||||
|
|
||||||
func NewURLTest(ctx context.Context, router adapter.Router, logger log.ContextLogger, tag string, options option.URLTestOutboundOptions) (adapter.Outbound, error) {
|
func NewURLTest(ctx context.Context, router adapter.Router, logger log.ContextLogger, tag string, options option.URLTestOutboundOptions) (adapter.Outbound, error) {
|
||||||
@@ -81,6 +88,9 @@ func NewURLTest(ctx context.Context, router adapter.Router, logger log.ContextLo
|
|||||||
// nil/absent means true (self-check on) — the documented default, so a
|
// nil/absent means true (self-check on) — the documented default, so a
|
||||||
// config written before the flag existed behaves exactly as it always has.
|
// config written before the flag existed behaves exactly as it always has.
|
||||||
selfCheckDisabled: options.SelfCheck != nil && !*options.SelfCheck,
|
selfCheckDisabled: options.SelfCheck != nil && !*options.SelfCheck,
|
||||||
|
// Absent from the registry (plain sing-box, tests) yields nil, which
|
||||||
|
// means "no gate" — every scheduled probe proceeds, as before.
|
||||||
|
probeGate: service.FromContext[urltest.ProbeGate](ctx),
|
||||||
}
|
}
|
||||||
if len(outbound.tags) == 0 {
|
if len(outbound.tags) == 0 {
|
||||||
return nil, E.New("missing tags")
|
return nil, E.New("missing tags")
|
||||||
@@ -105,6 +115,10 @@ func (s *URLTest) Start() error {
|
|||||||
// lx: health board §5.C — carry the stand-down flag onto the group the same
|
// lx: health board §5.C — carry the stand-down flag onto the group the same
|
||||||
// way the balancer travels: set after construction, immutable from then on.
|
// way the balancer travels: set after construction, immutable from then on.
|
||||||
group.selfCheckDisabled = s.selfCheckDisabled
|
group.selfCheckDisabled = s.selfCheckDisabled
|
||||||
|
// The gate and the tag to ask it about travel together: the gate answers
|
||||||
|
// per-outbound, and the group is the thing whose schedule is being gated.
|
||||||
|
group.probeGate = s.probeGate
|
||||||
|
group.tag = s.Tag()
|
||||||
if s.balancer != nil {
|
if s.balancer != nil {
|
||||||
// lx: health board §5.B — slot liveness reads through the board verdict, so a
|
// lx: health board §5.B — slot liveness reads through the board verdict, so a
|
||||||
// death recorded by any prober or a failed dial takes effect on the next pick,
|
// death recorded by any prober or a failed dial takes effect on the next pick,
|
||||||
@@ -135,12 +149,14 @@ func (s *URLTest) Now() string {
|
|||||||
if s.balancer != nil {
|
if s.balancer != nil {
|
||||||
return s.group.lastSelected.Load()
|
return s.group.lastSelected.Load()
|
||||||
}
|
}
|
||||||
if s.group.selectedOutboundTCP != nil {
|
// One load, so the two halves reported here are the SAME decision.
|
||||||
return s.group.selectedOutboundTCP.Tag()
|
selected := s.group.selected.Load()
|
||||||
} else if s.group.selectedOutboundUDP != nil {
|
if selected.tcp != nil {
|
||||||
return s.group.selectedOutboundUDP.Tag()
|
return selected.tcp.Tag()
|
||||||
|
} else if selected.udp != nil {
|
||||||
|
return selected.udp.Tag()
|
||||||
}
|
}
|
||||||
// lx: SPEC 019 — cold start: before the first URL-test, selectedOutbound* is nil but
|
// lx: SPEC 019 — cold start: before the first URL-test the pair is empty but
|
||||||
// traffic already flows via the Select() fallback (outbounds[0] when no history yet).
|
// traffic already flows via the Select() fallback (outbounds[0] when no history yet).
|
||||||
// Mirror exactly what the next DialContext would pick, so the UI shows the real node
|
// Mirror exactly what the next DialContext would pick, so the UI shows the real node
|
||||||
// instead of blank. Select() is the same source of truth DialContext uses.
|
// instead of blank. Select() is the same source of truth DialContext uses.
|
||||||
@@ -309,35 +325,155 @@ func (s *URLTest) NewPacketConnection(ctx context.Context, conn N.PacketConn, me
|
|||||||
}
|
}
|
||||||
|
|
||||||
type URLTestGroup struct {
|
type URLTestGroup struct {
|
||||||
ctx context.Context
|
ctx context.Context
|
||||||
outbound adapter.OutboundManager
|
outbound adapter.OutboundManager
|
||||||
pause pause.Manager
|
pause pause.Manager
|
||||||
pauseCallback *list.Element[pause.Callback]
|
pauseCallback *list.Element[pause.Callback]
|
||||||
logger log.Logger
|
logger log.Logger
|
||||||
outbounds []adapter.Outbound
|
outbounds []adapter.Outbound
|
||||||
link string
|
link string
|
||||||
interval time.Duration
|
interval time.Duration
|
||||||
tolerance uint16
|
tolerance uint16
|
||||||
idleTimeout time.Duration
|
idleTimeout time.Duration
|
||||||
history *urltest.HistoryStorage
|
history *urltest.HistoryStorage
|
||||||
checking atomic.Bool
|
checking atomic.Bool
|
||||||
selectedOutboundTCP adapter.Outbound
|
// selected is the least_test cache: the member this group currently prefers, per
|
||||||
selectedOutboundUDP adapter.Outbound
|
// network. It is written by the probing goroutine and read on EVERY dial through
|
||||||
|
// the group (selectExcluding / dialSelect) and by the panel (Now), so it is an
|
||||||
|
// atomic value rather than two plain fields — the same thing Selector does one file
|
||||||
|
// over (selector.go, common.TypedValue[adapter.Outbound]). An interface field is two
|
||||||
|
// words; a torn read of one hands a dial a type descriptor with the wrong data
|
||||||
|
// pointer, which is not a wrong node but a corrupt one.
|
||||||
|
//
|
||||||
|
// The TCP and UDP halves live in ONE value on purpose. They are decided together, by
|
||||||
|
// one pass over one board reading, and publishing them separately let a reader pick
|
||||||
|
// up the new TCP choice against the previous UDP choice — the group's hysteresis
|
||||||
|
// silently applied to a decision that was never made.
|
||||||
|
selected common.TypedValue[selectedPair]
|
||||||
interruptGroup *interrupt.Group
|
interruptGroup *interrupt.Group
|
||||||
interruptExternalConnections bool
|
interruptExternalConnections bool
|
||||||
access sync.Mutex
|
access sync.Mutex
|
||||||
ticker *time.Ticker
|
ticker *time.Ticker
|
||||||
close chan struct{}
|
close chan struct{}
|
||||||
started bool
|
// started is read by Touch on every dial, outside g.access, and written by
|
||||||
lastActive common.TypedValue[time.Time]
|
// PostStart under it — an atomic because that is what it always was in effect.
|
||||||
lastSelected common.TypedValue[string] // lx: SPEC 019 — Now() in balanced modes
|
started atomic.Bool
|
||||||
balancer *balancer // lx: SPEC 019 v2 — round_robin pool; nil for least_test
|
// closed latches in Close and is what makes Close FINAL. Guarded by access.
|
||||||
|
//
|
||||||
|
// It exists because "has a ticker" is not the same question as "is shut down", and
|
||||||
|
// Close used to ask the first one: with no ticker armed it returned before closing
|
||||||
|
// g.close, leaving the group indistinguishable from a running one. A Touch arriving
|
||||||
|
// afterwards — an outbound snapshot taken before an Apply is still dialable for up
|
||||||
|
// to two minutes, see shater/engine/grouptest.go — then armed a fresh ticker whose
|
||||||
|
// loopCheck waits on a channel nobody will ever close, in a box whose context is
|
||||||
|
// already cancelled. Every tick of it fails instantly and files a "dead" verdict on
|
||||||
|
// the SHARED health board that the live generation selects nodes from. One retired
|
||||||
|
// group can go on declaring the whole node set dead for the uptime of the daemon.
|
||||||
|
closed bool
|
||||||
|
lastActive common.TypedValue[time.Time]
|
||||||
|
lastSelected common.TypedValue[string] // lx: SPEC 019 — Now() in balanced modes
|
||||||
|
balancer *balancer // lx: SPEC 019 v2 — round_robin pool; nil for least_test
|
||||||
// lx: health board §5.C — mirrors URLTest.selfCheckDisabled (set by Start,
|
// lx: health board §5.C — mirrors URLTest.selfCheckDisabled (set by Start,
|
||||||
// immutable afterwards, zero value = probing on). Guards ONLY the group's
|
// immutable afterwards, zero value = probing on). Guards ONLY the group's
|
||||||
// own schedule: the PostStart warm-up sweep and the Touch ticker. An
|
// own schedule: the PostStart warm-up sweep and the Touch ticker. An
|
||||||
// explicit CheckOutbounds/URLTest call is untouched — the flag stands down
|
// explicit CheckOutbounds/URLTest call is untouched — the flag stands down
|
||||||
// the schedule, not the capability.
|
// the schedule, not the capability.
|
||||||
selfCheckDisabled bool
|
selfCheckDisabled bool
|
||||||
|
// lx: health board §5.C — the runtime gate and the tag it is asked about.
|
||||||
|
// Both mirror URLTest's fields (set by Start, immutable afterwards); nil
|
||||||
|
// gate or empty tag means every scheduled check proceeds. Consulted only
|
||||||
|
// through selfCheckAllowed, and only on the SCHEDULE.
|
||||||
|
probeGate urltest.ProbeGate
|
||||||
|
tag string
|
||||||
|
}
|
||||||
|
|
||||||
|
// selectedPair is one published least_test decision: the member chosen for TCP and the
|
||||||
|
// member chosen for UDP, as of the same probing round. Either half may be nil (nothing
|
||||||
|
// picked yet for that network).
|
||||||
|
type selectedPair struct {
|
||||||
|
tcp adapter.Outbound
|
||||||
|
udp adapter.Outbound
|
||||||
|
}
|
||||||
|
|
||||||
|
// selectedFor returns the cached choice for one network (nil when there is none, or when
|
||||||
|
// network is neither TCP nor UDP — the caller then falls through to a fresh selection).
|
||||||
|
func (g *URLTestGroup) selectedFor(network string) adapter.Outbound {
|
||||||
|
pair := g.selected.Load()
|
||||||
|
switch network {
|
||||||
|
case N.NetworkTCP:
|
||||||
|
return pair.tcp
|
||||||
|
case N.NetworkUDP:
|
||||||
|
return pair.udp
|
||||||
|
}
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// setSelected publishes a decision. It is the ONLY writer of g.selected, and it writes
|
||||||
|
// the pair whole — see the field comment for why the two halves may not be split.
|
||||||
|
func (g *URLTestGroup) setSelected(tcp, udp adapter.Outbound) {
|
||||||
|
g.selected.Store(selectedPair{tcp: tcp, udp: udp})
|
||||||
|
}
|
||||||
|
|
||||||
|
// selfCheckAllowed reports whether the group's OWN probing schedule may dial
|
||||||
|
// right now. lx: health board §5.C.
|
||||||
|
//
|
||||||
|
// Two independent refusals, in the order they can be answered cheapest first:
|
||||||
|
//
|
||||||
|
// selfCheckDisabled — the config says no rule reaches this group. Fixed for
|
||||||
|
// the life of the box; see standDownUnusedSelfCheck.
|
||||||
|
// probeGate — the world says this group cannot be reached right now,
|
||||||
|
// typically a chain hop sitting behind a dead hop. Asked
|
||||||
|
// fresh EVERY time, which is the entire mechanism by which
|
||||||
|
// a recovered hop resumes probing: there is no state here
|
||||||
|
// to reset, so there is none to get stuck.
|
||||||
|
//
|
||||||
|
// Neither refusal touches an explicit CheckOutbounds/URLTest — a deliberate
|
||||||
|
// request is never a scheduled one.
|
||||||
|
func (g *URLTestGroup) selfCheckAllowed() bool {
|
||||||
|
if g.selfCheckDisabled {
|
||||||
|
return false
|
||||||
|
}
|
||||||
|
if g.probeGate == nil || g.tag == "" {
|
||||||
|
return true
|
||||||
|
}
|
||||||
|
return g.probeGate.ProbeAllowed(g.tag)
|
||||||
|
}
|
||||||
|
|
||||||
|
// scheduledCheck is one firing of the group's own schedule — the warm-up sweep
|
||||||
|
// and every ticker tick go through here, and nothing else does. Having exactly
|
||||||
|
// one gated entry point is what keeps the two callers from drifting apart, and
|
||||||
|
// it is the seam the tests drive to assert that a gated group makes no dial
|
||||||
|
// attempt at all.
|
||||||
|
func (g *URLTestGroup) scheduledCheck() {
|
||||||
|
if !g.selfCheckAllowed() {
|
||||||
|
return
|
||||||
|
}
|
||||||
|
g.CheckOutbounds(false)
|
||||||
|
}
|
||||||
|
|
||||||
|
// keepWarm reports whether this group must keep measuring with no traffic
|
||||||
|
// flowing through it. lx: health board §5.C — see urltest.ProbeGate.ProbeWhenIdle.
|
||||||
|
//
|
||||||
|
// The default is NO, in every direction: no gate, no tag, or a group whose
|
||||||
|
// self-check is stood down anyway. Only a gate that positively says "the routing
|
||||||
|
// config reaches this group" turns the idle timeout off, so plain sing-box and
|
||||||
|
// every hand-built group keep the lifecycle they have always had.
|
||||||
|
func (g *URLTestGroup) keepWarm() bool {
|
||||||
|
if g.selfCheckDisabled || g.probeGate == nil || g.tag == "" {
|
||||||
|
return false
|
||||||
|
}
|
||||||
|
return g.probeGate.ProbeWhenIdle(g.tag)
|
||||||
|
}
|
||||||
|
|
||||||
|
// startTickerLocked arms the group's own probing ticker. g.access MUST be held
|
||||||
|
// and g.ticker MUST be nil. Extracted so PostStart and Touch arm it identically
|
||||||
|
// — two ways in, one construction, no chance of one of them forgetting the pause
|
||||||
|
// registration.
|
||||||
|
func (g *URLTestGroup) startTickerLocked() {
|
||||||
|
ticker := time.NewTicker(g.interval)
|
||||||
|
g.ticker = ticker
|
||||||
|
g.pauseCallback = pause.RegisterTicker(g.pause, ticker, g.interval, nil)
|
||||||
|
go g.loopCheck(ticker, g.close)
|
||||||
}
|
}
|
||||||
|
|
||||||
func NewURLTestGroup(ctx context.Context, outboundManager adapter.OutboundManager, logger log.Logger, outbounds []adapter.Outbound, link string, interval time.Duration, tolerance uint16, idleTimeout time.Duration, interruptExternalConnections bool) (*URLTestGroup, error) {
|
func NewURLTestGroup(ctx context.Context, outboundManager adapter.OutboundManager, logger log.Logger, outbounds []adapter.Outbound, link string, interval time.Duration, tolerance uint16, idleTimeout time.Duration, interruptExternalConnections bool) (*URLTestGroup, error) {
|
||||||
@@ -377,7 +513,10 @@ func NewURLTestGroup(ctx context.Context, outboundManager adapter.OutboundManage
|
|||||||
func (g *URLTestGroup) PostStart() {
|
func (g *URLTestGroup) PostStart() {
|
||||||
g.access.Lock()
|
g.access.Lock()
|
||||||
defer g.access.Unlock()
|
defer g.access.Unlock()
|
||||||
g.started = true
|
if g.closed {
|
||||||
|
return
|
||||||
|
}
|
||||||
|
g.started.Store(true)
|
||||||
g.lastActive.Store(time.Now())
|
g.lastActive.Store(time.Now())
|
||||||
// lx: SPEC 019 v2 — seed the pool so round_robin can route from the first connection,
|
// lx: SPEC 019 v2 — seed the pool so round_robin can route from the first connection,
|
||||||
// before the first health-check completes (history-warm nodes first, else config order).
|
// before the first health-check completes (history-warm nodes first, else config order).
|
||||||
@@ -391,13 +530,31 @@ func (g *URLTestGroup) PostStart() {
|
|||||||
// exists to keep honest. A stood-down group therefore skips it entirely; the
|
// exists to keep honest. A stood-down group therefore skips it entirely; the
|
||||||
// observatory (or nothing, for a truly unused group) is what measures its
|
// observatory (or nothing, for a truly unused group) is what measures its
|
||||||
// members.
|
// members.
|
||||||
if !g.selfCheckDisabled {
|
//
|
||||||
go g.CheckOutbounds(false)
|
// The same call is now also where a chain hop behind a DEAD hop declines to
|
||||||
|
// sweep: every member of such a group dials through the broken hop, so the
|
||||||
|
// sweep would measure that hop once per member and file the result against
|
||||||
|
// this one. selfCheckAllowed keeps both refusals in one place.
|
||||||
|
go g.scheduledCheck()
|
||||||
|
// A group the routing config REACHES keeps measuring whether or not anybody
|
||||||
|
// dials it, so its ticker is armed here instead of waiting for a Touch that
|
||||||
|
// may never come. Without this, a used group with no traffic gets this one
|
||||||
|
// warm-up sweep and then nothing: its members age past the verdict TTL and
|
||||||
|
// the panel reports "untested" about a rule that is in force, while the first
|
||||||
|
// real request pays a cold probe. Nothing else would fill the gap — the
|
||||||
|
// observatory stands off a urltest group's members entirely (probeplan.go
|
||||||
|
// SelfChecked), which is the whole point of one dialler per target.
|
||||||
|
//
|
||||||
|
// lastActive was stored a moment ago, so loopCheck's opening "idle longer
|
||||||
|
// than the interval" check does not fire and this cannot double up with the
|
||||||
|
// sweep above.
|
||||||
|
if g.keepWarm() && g.ticker == nil {
|
||||||
|
g.startTickerLocked()
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
func (g *URLTestGroup) Touch() {
|
func (g *URLTestGroup) Touch() {
|
||||||
if !g.started {
|
if !g.started.Load() {
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
// lx: health board §5.C — Touch's only job is to keep the group's OWN
|
// lx: health board §5.C — Touch's only job is to keep the group's OWN
|
||||||
@@ -407,32 +564,61 @@ func (g *URLTestGroup) Touch() {
|
|||||||
// manual pin) must not arm 30 minutes of direct probing under the members'
|
// manual pin) must not arm 30 minutes of direct probing under the members'
|
||||||
// base tags. Checked before the lock because the flag is immutable after
|
// base tags. Checked before the lock because the flag is immutable after
|
||||||
// Start, exactly like the started fast-path above.
|
// Start, exactly like the started fast-path above.
|
||||||
|
//
|
||||||
|
// The runtime gate is deliberately NOT consulted here. Touch only arms the
|
||||||
|
// ticker; refusing to arm it would mean a hop that recovers has no ticker
|
||||||
|
// left to notice — the block would outlive the failure, which is the one
|
||||||
|
// outcome this must never have. The ticker runs and each tick re-asks the
|
||||||
|
// gate (loopCheck -> scheduledCheck), so a blocked hop costs a predicate
|
||||||
|
// call per interval and resumes the moment the hop in front answers.
|
||||||
if g.selfCheckDisabled {
|
if g.selfCheckDisabled {
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
g.access.Lock()
|
g.access.Lock()
|
||||||
defer g.access.Unlock()
|
defer g.access.Unlock()
|
||||||
|
// A closed group arms nothing. Touch is reachable long after Close — a caller
|
||||||
|
// holding an outbound from a snapshot taken before an Apply keeps dialling it (up
|
||||||
|
// to the 120s budget of shater/engine/grouptest.go) — and the ticker it would arm
|
||||||
|
// has no way left to stop: see the `closed` field for what that costs.
|
||||||
|
if g.closed {
|
||||||
|
return
|
||||||
|
}
|
||||||
if g.ticker != nil {
|
if g.ticker != nil {
|
||||||
g.lastActive.Store(time.Now())
|
g.lastActive.Store(time.Now())
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
ticker := time.NewTicker(g.interval)
|
g.startTickerLocked()
|
||||||
g.ticker = ticker
|
|
||||||
g.pauseCallback = pause.RegisterTicker(g.pause, ticker, g.interval, nil)
|
|
||||||
go g.loopCheck(ticker, g.close)
|
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// Close shuts the group down for good. It is idempotent, and it is FINAL: no later Touch
|
||||||
|
// can bring the probing schedule back.
|
||||||
|
//
|
||||||
|
// It used to return early when no ticker happened to be armed, without ever closing
|
||||||
|
// g.close — so a group that was closed while idle stayed, from the point of view of every
|
||||||
|
// other method, a perfectly live group. That is the whole defect: the close channel is the
|
||||||
|
// only way a loopCheck goroutine ever exits (its idle-timeout escape does not fire for a
|
||||||
|
// group the routing config reaches, keepWarm), so a ticker armed after such a Close is
|
||||||
|
// immortal, and every one of its ticks writes a failure to the shared health board on
|
||||||
|
// behalf of a box that no longer exists.
|
||||||
func (g *URLTestGroup) Close() error {
|
func (g *URLTestGroup) Close() error {
|
||||||
g.access.Lock()
|
g.access.Lock()
|
||||||
defer g.access.Unlock()
|
defer g.access.Unlock()
|
||||||
if g.ticker == nil {
|
if g.closed {
|
||||||
return nil
|
return nil
|
||||||
}
|
}
|
||||||
g.ticker.Stop()
|
g.closed = true
|
||||||
g.ticker = nil
|
// Unconditionally, BEFORE looking at the ticker: this is the signal every loopCheck
|
||||||
g.pause.UnregisterCallback(g.pauseCallback)
|
// waits on, including any that a Touch armed after the last one was retired by the
|
||||||
g.pauseCallback = nil
|
// idle timeout.
|
||||||
close(g.close)
|
if g.close != nil {
|
||||||
|
close(g.close)
|
||||||
|
}
|
||||||
|
if g.ticker != nil {
|
||||||
|
g.ticker.Stop()
|
||||||
|
g.ticker = nil
|
||||||
|
g.pause.UnregisterCallback(g.pauseCallback)
|
||||||
|
g.pauseCallback = nil
|
||||||
|
}
|
||||||
return nil
|
return nil
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -445,10 +631,15 @@ func (g *URLTestGroup) Select(network string) (adapter.Outbound, bool) {
|
|||||||
return g.selectExcluding(network, nil)
|
return g.selectExcluding(network, nil)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// loopCheck is the group's own schedule. lx: health board §5.C — every probe it
|
||||||
|
// fires goes through scheduledCheck, so a stood-down or currently-unreachable
|
||||||
|
// group ticks without dialling. The ticker's LIFECYCLE (the idle timeout below)
|
||||||
|
// is deliberately left alone: a gated group keeps its ticker exactly as long as
|
||||||
|
// an ungated one would, because the ticker is what will notice the recovery.
|
||||||
func (g *URLTestGroup) loopCheck(ticker *time.Ticker, closeChan <-chan struct{}) {
|
func (g *URLTestGroup) loopCheck(ticker *time.Ticker, closeChan <-chan struct{}) {
|
||||||
if time.Since(g.lastActive.Load()) > g.interval {
|
if time.Since(g.lastActive.Load()) > g.interval {
|
||||||
g.lastActive.Store(time.Now())
|
g.lastActive.Store(time.Now())
|
||||||
g.CheckOutbounds(false)
|
g.scheduledCheck()
|
||||||
}
|
}
|
||||||
for {
|
for {
|
||||||
select {
|
select {
|
||||||
@@ -456,7 +647,13 @@ func (g *URLTestGroup) loopCheck(ticker *time.Ticker, closeChan <-chan struct{})
|
|||||||
return
|
return
|
||||||
case <-ticker.C:
|
case <-ticker.C:
|
||||||
}
|
}
|
||||||
if time.Since(g.lastActive.Load()) > g.idleTimeout {
|
// The idle timeout retires the ticker of a group nobody is dialling —
|
||||||
|
// unless the routing config reaches it, in which case its health is a
|
||||||
|
// live question whether or not traffic is flowing and the ticker must
|
||||||
|
// outlive the silence. Asked here rather than remembered from PostStart
|
||||||
|
// so it tracks the running config, and asked OUTSIDE g.access because the
|
||||||
|
// answer comes from the engine, which has locks of its own.
|
||||||
|
if !g.keepWarm() && time.Since(g.lastActive.Load()) > g.idleTimeout {
|
||||||
g.access.Lock()
|
g.access.Lock()
|
||||||
if g.ticker == ticker {
|
if g.ticker == ticker {
|
||||||
g.ticker.Stop()
|
g.ticker.Stop()
|
||||||
@@ -467,7 +664,7 @@ func (g *URLTestGroup) loopCheck(ticker *time.Ticker, closeChan <-chan struct{})
|
|||||||
g.access.Unlock()
|
g.access.Unlock()
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
g.CheckOutbounds(false)
|
g.scheduledCheck()
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -565,19 +762,29 @@ func (g *URLTestGroup) testNodes(ctx context.Context, outbounds []adapter.Outbou
|
|||||||
return result
|
return result
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// performUpdateCheck re-ranks the members after a probing round and publishes the result.
|
||||||
|
// It is the only writer of g.selected: it reads the current pair ONCE, decides both
|
||||||
|
// networks against that one snapshot, and stores the outcome as a single value, so no
|
||||||
|
// reader can ever observe a half-applied decision. Callers are serialised by g.checking
|
||||||
|
// (urlTest), which is what makes the read-decide-store sequence safe without a lock.
|
||||||
func (g *URLTestGroup) performUpdateCheck() {
|
func (g *URLTestGroup) performUpdateCheck() {
|
||||||
|
current := g.selected.Load()
|
||||||
|
next := current
|
||||||
var updated bool
|
var updated bool
|
||||||
if outbound, exists := g.Select(N.NetworkTCP); outbound != nil && (g.selectedOutboundTCP == nil || (exists && outbound != g.selectedOutboundTCP)) {
|
if outbound, exists := g.Select(N.NetworkTCP); outbound != nil && (current.tcp == nil || (exists && outbound != current.tcp)) {
|
||||||
if g.selectedOutboundTCP != nil {
|
if current.tcp != nil {
|
||||||
updated = true
|
updated = true
|
||||||
}
|
}
|
||||||
g.selectedOutboundTCP = outbound
|
next.tcp = outbound
|
||||||
}
|
}
|
||||||
if outbound, exists := g.Select(N.NetworkUDP); outbound != nil && (g.selectedOutboundUDP == nil || (exists && outbound != g.selectedOutboundUDP)) {
|
if outbound, exists := g.Select(N.NetworkUDP); outbound != nil && (current.udp == nil || (exists && outbound != current.udp)) {
|
||||||
if g.selectedOutboundUDP != nil {
|
if current.udp != nil {
|
||||||
updated = true
|
updated = true
|
||||||
}
|
}
|
||||||
g.selectedOutboundUDP = outbound
|
next.udp = outbound
|
||||||
|
}
|
||||||
|
if next != current {
|
||||||
|
g.setSelected(next.tcp, next.udp)
|
||||||
}
|
}
|
||||||
if updated {
|
if updated {
|
||||||
g.interruptGroup.Interrupt(g.interruptExternalConnections)
|
g.interruptGroup.Interrupt(g.interruptExternalConnections)
|
||||||
|
|||||||
@@ -74,13 +74,7 @@ func (g *URLTestGroup) selectExcluding(network string, exclude map[string]bool)
|
|||||||
var minOutbound adapter.Outbound
|
var minOutbound adapter.Outbound
|
||||||
// Keep the upstream hysteresis: the currently selected outbound only yields to a
|
// Keep the upstream hysteresis: the currently selected outbound only yields to a
|
||||||
// member faster by more than tolerance — but only while it is still alive itself.
|
// member faster by more than tolerance — but only while it is still alive itself.
|
||||||
var current adapter.Outbound
|
current := g.selectedFor(network)
|
||||||
switch network {
|
|
||||||
case N.NetworkTCP:
|
|
||||||
current = g.selectedOutboundTCP
|
|
||||||
case N.NetworkUDP:
|
|
||||||
current = g.selectedOutboundUDP
|
|
||||||
}
|
|
||||||
if current != nil {
|
if current != nil {
|
||||||
currentTag := RealTag(current)
|
currentTag := RealTag(current)
|
||||||
if !exclude[currentTag] && g.history.Verdict(currentTag, ttl) == urltest.VerdictAlive {
|
if !exclude[currentTag] && g.history.Verdict(currentTag, ttl) == urltest.VerdictAlive {
|
||||||
@@ -144,13 +138,7 @@ func (s *URLTest) dialSelect(ctx context.Context, network string, destination M.
|
|||||||
if s.balancer != nil {
|
if s.balancer != nil {
|
||||||
return s.selectBalanced(ctx, network, destination, tried)
|
return s.selectBalanced(ctx, network, destination, tried)
|
||||||
}
|
}
|
||||||
var outbound adapter.Outbound
|
outbound := s.group.selectedFor(N.NetworkName(network))
|
||||||
switch N.NetworkName(network) {
|
|
||||||
case N.NetworkTCP:
|
|
||||||
outbound = s.group.selectedOutboundTCP
|
|
||||||
case N.NetworkUDP:
|
|
||||||
outbound = s.group.selectedOutboundUDP
|
|
||||||
}
|
|
||||||
if outbound != nil {
|
if outbound != nil {
|
||||||
realTag := RealTag(outbound)
|
realTag := RealTag(outbound)
|
||||||
if !tried[realTag] && s.group.history.Verdict(realTag, s.group.healthTTL()) != urltest.VerdictDead {
|
if !tried[realTag] && s.group.history.Verdict(realTag, s.group.healthTTL()) != urltest.VerdictDead {
|
||||||
|
|||||||
@@ -7,6 +7,7 @@ import (
|
|||||||
"context"
|
"context"
|
||||||
"errors"
|
"errors"
|
||||||
"net"
|
"net"
|
||||||
|
"sync"
|
||||||
"testing"
|
"testing"
|
||||||
"time"
|
"time"
|
||||||
|
|
||||||
@@ -32,10 +33,31 @@ type healthNode struct {
|
|||||||
func (n *healthNode) Tag() string { return n.tag }
|
func (n *healthNode) Tag() string { return n.tag }
|
||||||
func (n *healthNode) Network() []string { return []string{N.NetworkTCP, N.NetworkUDP} }
|
func (n *healthNode) Network() []string { return []string{N.NetworkTCP, N.NetworkUDP} }
|
||||||
|
|
||||||
func (n *healthNode) DialContext(ctx context.Context, network string, destination M.Socksaddr) (net.Conn, error) {
|
// healthDialMu guards the shared dialed slice. testNodes probes a group's
|
||||||
if n.dialed != nil {
|
// members CONCURRENTLY (a batch of 10), so two nodes pointing at one slice
|
||||||
*n.dialed = append(*n.dialed, n.tag)
|
// append from two goroutines; the append is what the -race build trips on, not
|
||||||
|
// the code under test.
|
||||||
|
var healthDialMu sync.Mutex
|
||||||
|
|
||||||
|
func (n *healthNode) record() {
|
||||||
|
if n.dialed == nil {
|
||||||
|
return
|
||||||
}
|
}
|
||||||
|
healthDialMu.Lock()
|
||||||
|
*n.dialed = append(*n.dialed, n.tag)
|
||||||
|
healthDialMu.Unlock()
|
||||||
|
}
|
||||||
|
|
||||||
|
// dialsOf reads a dial log under the same lock. Every assertion on a log a
|
||||||
|
// concurrent sweep may still be writing must go through it.
|
||||||
|
func dialsOf(dialed *[]string) []string {
|
||||||
|
healthDialMu.Lock()
|
||||||
|
defer healthDialMu.Unlock()
|
||||||
|
return append([]string(nil), *dialed...)
|
||||||
|
}
|
||||||
|
|
||||||
|
func (n *healthNode) DialContext(ctx context.Context, network string, destination M.Socksaddr) (net.Conn, error) {
|
||||||
|
n.record()
|
||||||
if n.fail {
|
if n.fail {
|
||||||
return nil, errors.New("dial refused")
|
return nil, errors.New("dial refused")
|
||||||
}
|
}
|
||||||
@@ -45,9 +67,7 @@ func (n *healthNode) DialContext(ctx context.Context, network string, destinatio
|
|||||||
}
|
}
|
||||||
|
|
||||||
func (n *healthNode) ListenPacket(ctx context.Context, destination M.Socksaddr) (net.PacketConn, error) {
|
func (n *healthNode) ListenPacket(ctx context.Context, destination M.Socksaddr) (net.PacketConn, error) {
|
||||||
if n.dialed != nil {
|
n.record()
|
||||||
*n.dialed = append(*n.dialed, n.tag)
|
|
||||||
}
|
|
||||||
if n.fail {
|
if n.fail {
|
||||||
return nil, errors.New("listen refused")
|
return nil, errors.New("listen refused")
|
||||||
}
|
}
|
||||||
@@ -85,6 +105,9 @@ func healthTestGroup(hist *urltest.HistoryStorage, manager adapter.OutboundManag
|
|||||||
tolerance: 50,
|
tolerance: 50,
|
||||||
logger: log.NewNOPFactory().Logger(),
|
logger: log.NewNOPFactory().Logger(),
|
||||||
interruptGroup: interrupt.NewGroup(),
|
interruptGroup: interrupt.NewGroup(),
|
||||||
|
// Real groups always have this channel (NewURLTestGroup); it is what Close
|
||||||
|
// signals every loopCheck through, so a hand-built group needs it too.
|
||||||
|
close: make(chan struct{}),
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -161,7 +184,7 @@ func TestSelectHysteresisKeepsAliveCurrent(t *testing.T) {
|
|||||||
hist := urltest.NewHistoryStorage()
|
hist := urltest.NewHistoryStorage()
|
||||||
a, b := &balNode{tag: "a"}, &balNode{tag: "b"}
|
a, b := &balNode{tag: "a"}, &balNode{tag: "b"}
|
||||||
g := healthTestGroup(hist, nil, a, b)
|
g := healthTestGroup(hist, nil, a, b)
|
||||||
g.selectedOutboundTCP = a
|
g.setSelected(a, nil)
|
||||||
storeAlive(hist, "a", 100)
|
storeAlive(hist, "a", 100)
|
||||||
storeAlive(hist, "b", 60) // within tolerance (100 ≤ 60+50) → keep a
|
storeAlive(hist, "b", 60) // within tolerance (100 ≤ 60+50) → keep a
|
||||||
if selected, _ := g.Select(N.NetworkTCP); selected != adapter.Outbound(a) {
|
if selected, _ := g.Select(N.NetworkTCP); selected != adapter.Outbound(a) {
|
||||||
@@ -178,7 +201,7 @@ func TestSelectDeadCurrentLosesToAlive(t *testing.T) {
|
|||||||
hist := urltest.NewHistoryStorage()
|
hist := urltest.NewHistoryStorage()
|
||||||
a, b := &balNode{tag: "a"}, &balNode{tag: "b"}
|
a, b := &balNode{tag: "a"}, &balNode{tag: "b"}
|
||||||
g := healthTestGroup(hist, nil, a, b)
|
g := healthTestGroup(hist, nil, a, b)
|
||||||
g.selectedOutboundTCP = a
|
g.setSelected(a, nil)
|
||||||
storeAlive(hist, "a", 10)
|
storeAlive(hist, "a", 10)
|
||||||
hist.MarkFailed("a")
|
hist.MarkFailed("a")
|
||||||
storeAlive(hist, "b", 500)
|
storeAlive(hist, "b", 500)
|
||||||
@@ -331,7 +354,7 @@ func TestDialContextRetriesThroughNextAlive(t *testing.T) {
|
|||||||
s := healthURLTest(g, nil, nil)
|
s := healthURLTest(g, nil, nil)
|
||||||
storeAlive(hist, "a", 10)
|
storeAlive(hist, "a", 10)
|
||||||
storeAlive(hist, "b", 100)
|
storeAlive(hist, "b", 100)
|
||||||
g.selectedOutboundTCP = a // the checker had picked a; it dies between ticks
|
g.setSelected(a, nil) // the checker had picked a; it dies between ticks
|
||||||
conn, err := s.DialContext(context.Background(), N.NetworkTCP, destDomain("example.com"))
|
conn, err := s.DialContext(context.Background(), N.NetworkTCP, destDomain("example.com"))
|
||||||
if err != nil {
|
if err != nil {
|
||||||
t.Fatalf("DialContext failed despite live member b: %v", err)
|
t.Fatalf("DialContext failed despite live member b: %v", err)
|
||||||
@@ -427,7 +450,7 @@ func TestListenPacketRetriesBeforeFirstSend(t *testing.T) {
|
|||||||
s := healthURLTest(g, nil, nil)
|
s := healthURLTest(g, nil, nil)
|
||||||
storeAlive(hist, "a", 10)
|
storeAlive(hist, "a", 10)
|
||||||
storeAlive(hist, "b", 100)
|
storeAlive(hist, "b", 100)
|
||||||
g.selectedOutboundUDP = a
|
g.setSelected(nil, a)
|
||||||
conn, err := s.ListenPacket(context.Background(), destDomain("example.com"))
|
conn, err := s.ListenPacket(context.Background(), destDomain("example.com"))
|
||||||
if err != nil {
|
if err != nil {
|
||||||
t.Fatalf("ListenPacket failed despite live member b: %v", err)
|
t.Fatalf("ListenPacket failed despite live member b: %v", err)
|
||||||
|
|||||||
@@ -0,0 +1,330 @@
|
|||||||
|
package group
|
||||||
|
|
||||||
|
// Concurrency tests for the urltest group: the group's own probing schedule running at
|
||||||
|
// the same time as traffic going through it.
|
||||||
|
//
|
||||||
|
// This combination had no coverage at all. Every existing test either probes OR dials,
|
||||||
|
// never both at once, so the race detector had nothing to detect: the cached least_test
|
||||||
|
// choice was written by the prober goroutine and read on every single dial, with no
|
||||||
|
// synchronisation whatsoever, and the suite stayed green for as long as those two things
|
||||||
|
// never happened in the same test.
|
||||||
|
//
|
||||||
|
// An unsynchronised interface field is not a "usually fine" race. It is two words — type
|
||||||
|
// descriptor and data pointer — and a reader that catches the store half way holds a
|
||||||
|
// descriptor addressing the wrong value. What comes out is not a suboptimal node, it is a
|
||||||
|
// corrupt one, on the path of every connection the group carries.
|
||||||
|
|
||||||
|
import (
|
||||||
|
"context"
|
||||||
|
"sync"
|
||||||
|
"sync/atomic"
|
||||||
|
"testing"
|
||||||
|
"time"
|
||||||
|
|
||||||
|
"github.com/sagernet/sing-box/adapter"
|
||||||
|
"github.com/sagernet/sing-box/common/urltest"
|
||||||
|
M "github.com/sagernet/sing/common/metadata"
|
||||||
|
N "github.com/sagernet/sing/common/network"
|
||||||
|
"github.com/sagernet/sing/service/pause"
|
||||||
|
)
|
||||||
|
|
||||||
|
// TestURLTestDialRacesProbeTicker drives real dials through a least_test group while the
|
||||||
|
// group's own probing schedule keeps re-deciding which member to use — the situation on
|
||||||
|
// every router where a urltest group carries traffic, since the ticker fires on its own
|
||||||
|
// interval regardless of what the connections are doing.
|
||||||
|
//
|
||||||
|
// It is a -race test first and an assertion test second: the failure it was written for
|
||||||
|
// is reported by the detector, not by a wrong value. Run it under -race or it proves
|
||||||
|
// almost nothing (the gate's [4/4] pass does).
|
||||||
|
func TestURLTestDialRacesProbeTicker(t *testing.T) {
|
||||||
|
hist := urltest.NewHistoryStorage()
|
||||||
|
a, b := &healthNode{tag: "a"}, &healthNode{tag: "b"}
|
||||||
|
manager := managerOf(a, b)
|
||||||
|
g := healthTestGroup(hist, manager, a, b)
|
||||||
|
s := healthURLTest(g, nil, manager)
|
||||||
|
storeAlive(hist, "a", 20)
|
||||||
|
storeAlive(hist, "b", 500)
|
||||||
|
|
||||||
|
var proberWG, dialWG sync.WaitGroup
|
||||||
|
stop := make(chan struct{})
|
||||||
|
|
||||||
|
// The prober: one full turn of the group's own schedule per iteration. CheckOutbounds
|
||||||
|
// is the real tick (probe every member, then publish); the probes cannot reach
|
||||||
|
// anything from a unit test, so the board is then re-armed with a winner that MOVES
|
||||||
|
// and the publish step is run again — otherwise the cached choice is written once and
|
||||||
|
// the window in which a reader can catch a torn write is a few nanoseconds wide.
|
||||||
|
proberWG.Add(1)
|
||||||
|
go func() {
|
||||||
|
defer proberWG.Done()
|
||||||
|
for i := 0; ; i++ {
|
||||||
|
select {
|
||||||
|
case <-stop:
|
||||||
|
return
|
||||||
|
default:
|
||||||
|
}
|
||||||
|
g.CheckOutbounds(true)
|
||||||
|
fast, slow := "a", "b"
|
||||||
|
if i%2 == 1 {
|
||||||
|
fast, slow = "b", "a"
|
||||||
|
}
|
||||||
|
storeAlive(hist, fast, 20)
|
||||||
|
storeAlive(hist, slow, 500)
|
||||||
|
g.performUpdateCheck()
|
||||||
|
}
|
||||||
|
}()
|
||||||
|
|
||||||
|
// The traffic: every dial reads the cached choice (dialSelect), and so does the panel
|
||||||
|
// (Now). Both are the read side of the race.
|
||||||
|
var dials, nows atomic.Int64
|
||||||
|
for range 4 {
|
||||||
|
dialWG.Add(1)
|
||||||
|
go func() {
|
||||||
|
defer dialWG.Done()
|
||||||
|
for range 300 {
|
||||||
|
if conn, err := s.DialContext(context.Background(), N.NetworkTCP, M.Socksaddr{}); err == nil {
|
||||||
|
_ = conn.Close()
|
||||||
|
dials.Add(1)
|
||||||
|
}
|
||||||
|
if pc, err := s.ListenPacket(context.Background(), M.Socksaddr{}); err == nil {
|
||||||
|
_ = pc.Close()
|
||||||
|
}
|
||||||
|
// The panel polls this while everything above is happening.
|
||||||
|
if tag := s.Now(); tag != "" && tag != "a" && tag != "b" {
|
||||||
|
t.Errorf("Now() = %q, which is not a member of the group — a torn read of the cached choice", tag)
|
||||||
|
}
|
||||||
|
nows.Add(1)
|
||||||
|
}
|
||||||
|
}()
|
||||||
|
}
|
||||||
|
|
||||||
|
// The dialers are the bounded side; the prober runs until they are done.
|
||||||
|
dialersDone := make(chan struct{})
|
||||||
|
go func() { dialWG.Wait(); close(dialersDone) }()
|
||||||
|
select {
|
||||||
|
case <-dialersDone:
|
||||||
|
case <-time.After(60 * time.Second):
|
||||||
|
close(stop)
|
||||||
|
proberWG.Wait()
|
||||||
|
t.Fatal("dialers did not finish — the group deadlocked against its own prober")
|
||||||
|
}
|
||||||
|
close(stop)
|
||||||
|
proberWG.Wait()
|
||||||
|
|
||||||
|
if dials.Load() == 0 {
|
||||||
|
t.Fatal("no dial succeeded — the test never exercised the read side it exists to race")
|
||||||
|
}
|
||||||
|
if nows.Load() == 0 {
|
||||||
|
t.Fatal("Now() was never polled")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestSelectedPairPublishedTogether pins the pairing half of the same defect: the TCP and
|
||||||
|
// UDP choices are one decision, taken from one board reading, and they become visible
|
||||||
|
// together. They used to be two separate field writes, so a reader could take the new TCP
|
||||||
|
// choice against the previous UDP one — a combination no probing round ever decided, and
|
||||||
|
// the group's hysteresis silently applied to it.
|
||||||
|
func TestSelectedPairPublishedTogether(t *testing.T) {
|
||||||
|
hist := urltest.NewHistoryStorage()
|
||||||
|
a, b := &healthNode{tag: "a"}, &healthNode{tag: "b"}
|
||||||
|
g := healthTestGroup(hist, managerOf(a, b), a, b)
|
||||||
|
|
||||||
|
// Round 1: a wins both networks.
|
||||||
|
storeAlive(hist, "a", 20)
|
||||||
|
storeAlive(hist, "b", 500)
|
||||||
|
g.performUpdateCheck()
|
||||||
|
if got := g.selected.Load(); got.tcp != adapter.Outbound(a) || got.udp != adapter.Outbound(a) {
|
||||||
|
t.Fatalf("after round 1 the pair is (%v, %v), want (a, a)", tagOrNil(got.tcp), tagOrNil(got.udp))
|
||||||
|
}
|
||||||
|
|
||||||
|
// Round 2: b wins both, by more than the tolerance.
|
||||||
|
storeAlive(hist, "a", 500)
|
||||||
|
storeAlive(hist, "b", 20)
|
||||||
|
g.performUpdateCheck()
|
||||||
|
got := g.selected.Load()
|
||||||
|
if got.tcp != adapter.Outbound(b) || got.udp != adapter.Outbound(b) {
|
||||||
|
t.Fatalf("after round 2 the pair is (%v, %v), want (b, b) — both halves move together",
|
||||||
|
tagOrNil(got.tcp), tagOrNil(got.udp))
|
||||||
|
}
|
||||||
|
|
||||||
|
// And what the dial path reads per network agrees with the published pair.
|
||||||
|
if g.selectedFor(N.NetworkTCP) != got.tcp || g.selectedFor(N.NetworkUDP) != got.udp {
|
||||||
|
t.Fatal("selectedFor disagrees with the published pair")
|
||||||
|
}
|
||||||
|
if g.selectedFor("icmp") != nil {
|
||||||
|
t.Fatal("selectedFor on an unknown network must yield nothing, not a TCP choice")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestURLTestGroupProbeRacesPanelRead is the narrower of the pair: the panel's Now() poll
|
||||||
|
// against the prober, with no dialling at all. shater/engine/grouphealth.go,
|
||||||
|
// shater/stats/stats.go and shater/engine/grouptest.go all call Now() from their own
|
||||||
|
// goroutines while the group's ticker runs.
|
||||||
|
func TestURLTestGroupProbeRacesPanelRead(t *testing.T) {
|
||||||
|
hist := urltest.NewHistoryStorage()
|
||||||
|
a, b := &healthNode{tag: "a"}, &healthNode{tag: "b"}
|
||||||
|
manager := managerOf(a, b)
|
||||||
|
g := healthTestGroup(hist, manager, a, b)
|
||||||
|
s := healthURLTest(g, nil, manager)
|
||||||
|
|
||||||
|
var wg sync.WaitGroup
|
||||||
|
stop := make(chan struct{})
|
||||||
|
wg.Add(1)
|
||||||
|
go func() {
|
||||||
|
defer wg.Done()
|
||||||
|
for i := 0; ; i++ {
|
||||||
|
select {
|
||||||
|
case <-stop:
|
||||||
|
return
|
||||||
|
default:
|
||||||
|
}
|
||||||
|
fast, slow := "a", "b"
|
||||||
|
if i%2 == 1 {
|
||||||
|
fast, slow = "b", "a"
|
||||||
|
}
|
||||||
|
storeAlive(hist, fast, 20)
|
||||||
|
storeAlive(hist, slow, 500)
|
||||||
|
g.performUpdateCheck()
|
||||||
|
}
|
||||||
|
}()
|
||||||
|
for range 3 {
|
||||||
|
wg.Add(1)
|
||||||
|
go func() {
|
||||||
|
defer wg.Done()
|
||||||
|
for range 2000 {
|
||||||
|
_ = s.Now()
|
||||||
|
}
|
||||||
|
}()
|
||||||
|
}
|
||||||
|
time.Sleep(50 * time.Millisecond)
|
||||||
|
close(stop)
|
||||||
|
wg.Wait()
|
||||||
|
}
|
||||||
|
|
||||||
|
// tagOrNil renders a possibly-nil outbound for a failure message.
|
||||||
|
func tagOrNil(o adapter.Outbound) string {
|
||||||
|
if o == nil {
|
||||||
|
return "<nil>"
|
||||||
|
}
|
||||||
|
return o.Tag()
|
||||||
|
}
|
||||||
|
|
||||||
|
// --- Close is final ---------------------------------------------------------
|
||||||
|
|
||||||
|
// TestGroupCloseIsFinalForALaterTouch is the immortal-ticker regression.
|
||||||
|
//
|
||||||
|
// Close used to return early whenever no ticker happened to be armed — which is the
|
||||||
|
// normal state of a group nobody is dialling — WITHOUT closing g.close. Nothing else
|
||||||
|
// records that a group was shut down (started is never cleared), so a Touch arriving
|
||||||
|
// afterwards armed a fresh ticker and a fresh loopCheck goroutine waiting on a channel
|
||||||
|
// that would never be closed. Its only other exit, the idle timeout, does not fire for a
|
||||||
|
// group the routing config reaches.
|
||||||
|
//
|
||||||
|
// A later Touch is not hypothetical: shater/engine/grouptest.go dials through outbounds
|
||||||
|
// taken from a snapshot at the start of a run and keeps doing so for up to 120s, so
|
||||||
|
// "press Test in the panel, then apply a config within two minutes" is enough. The
|
||||||
|
// retired group then probes forever through a cancelled context — every probe fails
|
||||||
|
// instantly — and files "dead" for its members on the SHARED health board that the LIVE
|
||||||
|
// generation picks nodes from.
|
||||||
|
func TestGroupCloseIsFinalForALaterTouch(t *testing.T) {
|
||||||
|
hist := urltest.NewHistoryStorage()
|
||||||
|
defer hist.Close()
|
||||||
|
var dialed []string
|
||||||
|
a := &healthNode{tag: "a", fail: true, dialed: &dialed}
|
||||||
|
g := healthTestGroup(hist, managerOf(a), a)
|
||||||
|
g.pause = pause.ManagerFromContext(pause.WithDefaultManager(context.Background()))
|
||||||
|
// Fast enough that a surviving ticker proves itself within the test's patience.
|
||||||
|
g.interval = 10 * time.Millisecond
|
||||||
|
g.idleTimeout = time.Hour
|
||||||
|
// Started, but idle: no ticker armed. This is the state Close mishandled.
|
||||||
|
g.started.Store(true)
|
||||||
|
|
||||||
|
if err := g.Close(); err != nil {
|
||||||
|
t.Fatalf("Close: %v", err)
|
||||||
|
}
|
||||||
|
g.Touch()
|
||||||
|
|
||||||
|
g.access.Lock()
|
||||||
|
ticker := g.ticker
|
||||||
|
g.access.Unlock()
|
||||||
|
if ticker != nil {
|
||||||
|
t.Fatal("Touch armed a probing ticker on a CLOSED group — nothing can stop it: " +
|
||||||
|
"its loopCheck waits on a channel that will never be closed")
|
||||||
|
}
|
||||||
|
|
||||||
|
// The consequence, stated in the terms that actually hurt: no probe, so no forged
|
||||||
|
// verdict on the shared board.
|
||||||
|
time.Sleep(150 * time.Millisecond)
|
||||||
|
if got := dialsOf(&dialed); len(got) != 0 {
|
||||||
|
t.Fatalf("a closed group dialled %v — a retired generation is writing to the live health board", got)
|
||||||
|
}
|
||||||
|
if v := hist.Verdict("a", 10*time.Minute); v != urltest.VerdictUntested {
|
||||||
|
t.Fatalf("verdict(a) = %v after closing the group, want untested — the dead marks are forged", v)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestGroupCloseStopsAnArmedTicker keeps the original behaviour honest: when a ticker IS
|
||||||
|
// armed, Close still stops it, unregisters the pause callback and signals loopCheck.
|
||||||
|
func TestGroupCloseStopsAnArmedTicker(t *testing.T) {
|
||||||
|
hist := urltest.NewHistoryStorage()
|
||||||
|
defer hist.Close()
|
||||||
|
a := &healthNode{tag: "a", fail: true}
|
||||||
|
g := healthTestGroup(hist, managerOf(a), a)
|
||||||
|
g.pause = pause.ManagerFromContext(pause.WithDefaultManager(context.Background()))
|
||||||
|
g.interval = 10 * time.Millisecond
|
||||||
|
g.idleTimeout = time.Hour
|
||||||
|
g.started.Store(true)
|
||||||
|
g.lastActive.Store(time.Now())
|
||||||
|
|
||||||
|
g.Touch()
|
||||||
|
g.access.Lock()
|
||||||
|
armed := g.ticker != nil
|
||||||
|
g.access.Unlock()
|
||||||
|
if !armed {
|
||||||
|
t.Fatal("Touch did not arm the ticker on a live group")
|
||||||
|
}
|
||||||
|
|
||||||
|
if err := g.Close(); err != nil {
|
||||||
|
t.Fatalf("Close: %v", err)
|
||||||
|
}
|
||||||
|
g.access.Lock()
|
||||||
|
stillArmed := g.ticker != nil
|
||||||
|
g.access.Unlock()
|
||||||
|
if stillArmed {
|
||||||
|
t.Fatal("Close left the ticker armed")
|
||||||
|
}
|
||||||
|
select {
|
||||||
|
case <-g.close:
|
||||||
|
default:
|
||||||
|
t.Fatal("Close did not signal loopCheck")
|
||||||
|
}
|
||||||
|
// Idempotent: a second Close must not close an already-closed channel (panic) or
|
||||||
|
// undo anything.
|
||||||
|
if err := g.Close(); err != nil {
|
||||||
|
t.Fatalf("second Close: %v", err)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestGroupPostStartAfterCloseDoesNothing: the other way a retired group can be woken.
|
||||||
|
// PostStart is called on every member of a box at start-up; a group closed by a racing
|
||||||
|
// shutdown must not be brought back by it.
|
||||||
|
func TestGroupPostStartAfterCloseDoesNothing(t *testing.T) {
|
||||||
|
hist := urltest.NewHistoryStorage()
|
||||||
|
defer hist.Close()
|
||||||
|
var dialed []string
|
||||||
|
a := &healthNode{tag: "a", fail: true, dialed: &dialed}
|
||||||
|
g := healthTestGroup(hist, managerOf(a), a)
|
||||||
|
g.pause = pause.ManagerFromContext(pause.WithDefaultManager(context.Background()))
|
||||||
|
g.interval = 10 * time.Millisecond
|
||||||
|
g.idleTimeout = time.Hour
|
||||||
|
|
||||||
|
_ = g.Close()
|
||||||
|
g.PostStart()
|
||||||
|
time.Sleep(150 * time.Millisecond)
|
||||||
|
|
||||||
|
if g.started.Load() {
|
||||||
|
t.Fatal("PostStart marked a closed group as started")
|
||||||
|
}
|
||||||
|
if got := dialsOf(&dialed); len(got) != 0 {
|
||||||
|
t.Fatalf("PostStart on a closed group ran the warm-up sweep: dialled %v", got)
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -6,12 +6,14 @@ package group
|
|||||||
|
|
||||||
import (
|
import (
|
||||||
"context"
|
"context"
|
||||||
|
"sync"
|
||||||
"testing"
|
"testing"
|
||||||
"time"
|
"time"
|
||||||
|
|
||||||
"github.com/sagernet/sing-box/common/urltest"
|
"github.com/sagernet/sing-box/common/urltest"
|
||||||
"github.com/sagernet/sing-box/log"
|
"github.com/sagernet/sing-box/log"
|
||||||
"github.com/sagernet/sing-box/option"
|
"github.com/sagernet/sing-box/option"
|
||||||
|
"github.com/sagernet/sing/service/pause"
|
||||||
)
|
)
|
||||||
|
|
||||||
// waitForHistory polls until the store holds an entry for tag or the deadline
|
// waitForHistory polls until the store holds an entry for tag or the deadline
|
||||||
@@ -114,3 +116,157 @@ func TestSelfCheckOptionPlumbing(t *testing.T) {
|
|||||||
t.Fatal("SelfCheck=false must stand the self-check down")
|
t.Fatal("SelfCheck=false must stand the self-check down")
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// lx: health board §5.C — the RUNTIME gate (urltest.ProbeGate). The self-check
|
||||||
|
// flag above says "the config reaches nothing here"; the gate says "the path in
|
||||||
|
// front of this group is down right now". Both stand the SCHEDULE down; neither
|
||||||
|
// touches an explicit check; and only the gate is allowed to change its mind
|
||||||
|
// while the box runs.
|
||||||
|
|
||||||
|
// fakeGate answers from a mutable set of blocked tags, so one test can watch a
|
||||||
|
// group stop dialling and start again without rebuilding anything.
|
||||||
|
type fakeGate struct {
|
||||||
|
mu sync.Mutex
|
||||||
|
blocked map[string]bool
|
||||||
|
warm map[string]bool
|
||||||
|
asked int
|
||||||
|
}
|
||||||
|
|
||||||
|
func (g *fakeGate) ProbeAllowed(tag string) bool {
|
||||||
|
g.mu.Lock()
|
||||||
|
defer g.mu.Unlock()
|
||||||
|
g.asked++
|
||||||
|
return !g.blocked[tag]
|
||||||
|
}
|
||||||
|
|
||||||
|
func (g *fakeGate) ProbeWhenIdle(tag string) bool {
|
||||||
|
g.mu.Lock()
|
||||||
|
defer g.mu.Unlock()
|
||||||
|
return g.warm[tag]
|
||||||
|
}
|
||||||
|
|
||||||
|
func (g *fakeGate) set(tag string, blocked bool) {
|
||||||
|
g.mu.Lock()
|
||||||
|
defer g.mu.Unlock()
|
||||||
|
g.blocked[tag] = blocked
|
||||||
|
}
|
||||||
|
|
||||||
|
func (g *fakeGate) asks() int {
|
||||||
|
g.mu.Lock()
|
||||||
|
defer g.mu.Unlock()
|
||||||
|
return g.asked
|
||||||
|
}
|
||||||
|
|
||||||
|
// A gated group makes NO DIAL ATTEMPT on its own schedule — the assertion is on
|
||||||
|
// the attempt log, not on the board, because a probe that ran and failed leaves
|
||||||
|
// the same "nothing useful known" as one that never ran, and only the attempt
|
||||||
|
// log tells them apart. This is the waste half of the chain-hop fix: a hop
|
||||||
|
// sitting behind a dead hop would otherwise spend one probe timeout per member
|
||||||
|
// rediscovering the same broken hop.
|
||||||
|
func TestProbeGateBlocksScheduledDials(t *testing.T) {
|
||||||
|
hist := urltest.NewHistoryStorage()
|
||||||
|
defer hist.Close()
|
||||||
|
var dialed []string
|
||||||
|
a := &healthNode{tag: "chain-c-h3-a", fail: true, dialed: &dialed}
|
||||||
|
b := &healthNode{tag: "chain-c-h3-b", fail: true, dialed: &dialed}
|
||||||
|
gate := &fakeGate{blocked: map[string]bool{"chain-c-h3": true}}
|
||||||
|
|
||||||
|
g := healthTestGroup(hist, managerOf(a, b), a, b)
|
||||||
|
g.tag = "chain-c-h3"
|
||||||
|
g.probeGate = gate
|
||||||
|
// Touch arms a real ticker, so this group needs the two things
|
||||||
|
// healthTestGroup leaves out because nothing else in that suite starts one:
|
||||||
|
// a pause manager to register the ticker with, and the close channel Close
|
||||||
|
// shuts the loop down through.
|
||||||
|
g.pause = pause.ManagerFromContext(pause.WithDefaultManager(context.Background()))
|
||||||
|
g.close = make(chan struct{})
|
||||||
|
|
||||||
|
// The warm-up sweep: gated, so nothing is dialled. Give the (non-existent)
|
||||||
|
// sweep real time to have happened — the absence is the assertion.
|
||||||
|
g.PostStart()
|
||||||
|
if waitForHistory(hist, "chain-c-h3-a", 150*time.Millisecond) {
|
||||||
|
t.Fatal("a gated group's PostStart wrote to the board")
|
||||||
|
}
|
||||||
|
// A ticker tick, driven directly: this is the exact call loopCheck makes.
|
||||||
|
g.scheduledCheck()
|
||||||
|
if got := dialsOf(&dialed); len(got) != 0 {
|
||||||
|
t.Fatalf("gated group dialled %v; a hop behind a dead hop must not dial at all", got)
|
||||||
|
}
|
||||||
|
|
||||||
|
// Touch still arms the ticker. Refusing to arm it would leave a recovered
|
||||||
|
// hop with nothing to notice — the block would outlive the failure.
|
||||||
|
g.Touch()
|
||||||
|
g.access.Lock()
|
||||||
|
ticker := g.ticker
|
||||||
|
g.access.Unlock()
|
||||||
|
if ticker == nil {
|
||||||
|
t.Fatal("Touch did not arm the ticker on a gated group; nothing would be left to spot the recovery")
|
||||||
|
}
|
||||||
|
_ = g.Close()
|
||||||
|
|
||||||
|
// The hop in front comes back. Nothing is reset, nothing is reapplied — the
|
||||||
|
// next scheduled check simply asks again and gets a different answer.
|
||||||
|
gate.set("chain-c-h3", false)
|
||||||
|
before := gate.asks()
|
||||||
|
g.scheduledCheck()
|
||||||
|
if gate.asks() <= before {
|
||||||
|
t.Error("scheduledCheck did not re-ask the gate; a cached answer is a block that outlives its cause")
|
||||||
|
}
|
||||||
|
if len(dialsOf(&dialed)) == 0 {
|
||||||
|
t.Fatal("the group did not resume dialling after the hop in front recovered")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// An EXPLICIT check is a deliberate request and is never gated — the same rule
|
||||||
|
// SelfCheck already follows. The gate stands down the schedule, not the
|
||||||
|
// capability.
|
||||||
|
func TestProbeGateDoesNotBlockExplicitCheck(t *testing.T) {
|
||||||
|
hist := urltest.NewHistoryStorage()
|
||||||
|
defer hist.Close()
|
||||||
|
var dialed []string
|
||||||
|
a := &healthNode{tag: "chain-c-h3-a", fail: true, dialed: &dialed}
|
||||||
|
g := healthTestGroup(hist, managerOf(a), a)
|
||||||
|
g.tag = "chain-c-h3"
|
||||||
|
g.probeGate = &fakeGate{blocked: map[string]bool{"chain-c-h3": true}}
|
||||||
|
|
||||||
|
g.CheckOutbounds(true)
|
||||||
|
if len(dialsOf(&dialed)) == 0 {
|
||||||
|
t.Fatal("an explicit CheckOutbounds was refused by the gate")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// The two refusals are independent and compose the obvious way; and the absent
|
||||||
|
// cases (no gate at all, an ungated tag) leave today's behaviour untouched,
|
||||||
|
// which is what every plain sing-box config and every hand-built group relies
|
||||||
|
// on.
|
||||||
|
func TestSelfCheckAllowedCombinations(t *testing.T) {
|
||||||
|
hist := urltest.NewHistoryStorage()
|
||||||
|
defer hist.Close()
|
||||||
|
a := &healthNode{tag: "a"}
|
||||||
|
base := func() *URLTestGroup { return healthTestGroup(hist, managerOf(a), a) }
|
||||||
|
|
||||||
|
if g := base(); !g.selfCheckAllowed() {
|
||||||
|
t.Error("a plain group with no gate must probe (the zero value is the compatibility default)")
|
||||||
|
}
|
||||||
|
g := base()
|
||||||
|
g.selfCheckDisabled = true
|
||||||
|
g.tag, g.probeGate = "chain-c-h3", &fakeGate{blocked: map[string]bool{}}
|
||||||
|
if g.selfCheckAllowed() {
|
||||||
|
t.Error("an UNUSED group must stay down even when the path in front is fine")
|
||||||
|
}
|
||||||
|
g = base()
|
||||||
|
g.tag, g.probeGate = "chain-c-h3", &fakeGate{blocked: map[string]bool{"chain-c-h3": true}}
|
||||||
|
if g.selfCheckAllowed() {
|
||||||
|
t.Error("a group behind a dead hop must not run its schedule")
|
||||||
|
}
|
||||||
|
g = base()
|
||||||
|
g.tag, g.probeGate = "auto", &fakeGate{blocked: map[string]bool{"chain-c-h3": true}}
|
||||||
|
if !g.selfCheckAllowed() {
|
||||||
|
t.Error("an unrelated group was gated by another tag's block")
|
||||||
|
}
|
||||||
|
g = base()
|
||||||
|
g.probeGate = &fakeGate{blocked: map[string]bool{"": true}}
|
||||||
|
if !g.selfCheckAllowed() {
|
||||||
|
t.Error("a group with no tag must not be gated; there is nothing to ask about")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|||||||
@@ -0,0 +1,129 @@
|
|||||||
|
//go:build with_gvisor && with_awg
|
||||||
|
|
||||||
|
// lx: regression for the removal of the AmneziaWG-over-WireGuard start guard.
|
||||||
|
//
|
||||||
|
// The guard refused to bring up an AmneziaWG endpoint whose detour chain reached
|
||||||
|
// a WireGuard-based endpoint — and refused *silently*: Start returned nil with
|
||||||
|
// started=false, so the endpoint looked configured but every dial through it
|
||||||
|
// failed with "WireGuard is not ready yet". The root cause it protected against
|
||||||
|
// (a kernel hang on Android) is gone on this graft (ClientBind reserved-gate),
|
||||||
|
// and Android is not a supported platform here at all.
|
||||||
|
//
|
||||||
|
// This test builds a real AmneziaWG endpoint (junk + ranged magic headers) whose
|
||||||
|
// detour points at an outbound of type "wireguard", drives both start stages,
|
||||||
|
// and asserts the endpoint reports itself started. With the guard in place the
|
||||||
|
// first stage short-circuits and started stays false — this test fails.
|
||||||
|
package wireguard
|
||||||
|
|
||||||
|
import (
|
||||||
|
"context"
|
||||||
|
"crypto/rand"
|
||||||
|
"encoding/base64"
|
||||||
|
"net"
|
||||||
|
"net/netip"
|
||||||
|
"os"
|
||||||
|
"testing"
|
||||||
|
|
||||||
|
"github.com/sagernet/sing-box/adapter"
|
||||||
|
C "github.com/sagernet/sing-box/constant"
|
||||||
|
"github.com/sagernet/sing-box/log"
|
||||||
|
"github.com/sagernet/sing-box/option"
|
||||||
|
"github.com/sagernet/sing/common/json/badoption"
|
||||||
|
M "github.com/sagernet/sing/common/metadata"
|
||||||
|
"github.com/sagernet/sing/service"
|
||||||
|
"github.com/sagernet/sing/service/pause"
|
||||||
|
)
|
||||||
|
|
||||||
|
// wgTypedOutbound is an adapter.Outbound that reports type "wireguard" — the hop
|
||||||
|
// the guard used to refuse to start behind. Dialling through it always fails:
|
||||||
|
// the point of the test is that the upper endpoint comes UP, not that it carries
|
||||||
|
// traffic (that is the job of the transport-level e2e stand).
|
||||||
|
type wgTypedOutbound struct {
|
||||||
|
adapter.Outbound
|
||||||
|
tag string
|
||||||
|
}
|
||||||
|
|
||||||
|
func (o *wgTypedOutbound) Type() string { return C.TypeWireGuard }
|
||||||
|
func (o *wgTypedOutbound) Tag() string { return o.tag }
|
||||||
|
func (o *wgTypedOutbound) Dependencies() []string { return nil }
|
||||||
|
|
||||||
|
func (o *wgTypedOutbound) DialContext(ctx context.Context, network string, destination M.Socksaddr) (net.Conn, error) {
|
||||||
|
return nil, os.ErrClosed
|
||||||
|
}
|
||||||
|
|
||||||
|
func (o *wgTypedOutbound) ListenPacket(ctx context.Context, destination M.Socksaddr) (net.PacketConn, error) {
|
||||||
|
return nil, os.ErrClosed
|
||||||
|
}
|
||||||
|
|
||||||
|
// startChainManager resolves tags from a fixed map. adapter.OutboundManager is
|
||||||
|
// embedded so this compiles against either shape of the interface.
|
||||||
|
type startChainManager struct {
|
||||||
|
adapter.OutboundManager
|
||||||
|
byTag map[string]adapter.Outbound
|
||||||
|
}
|
||||||
|
|
||||||
|
func (m *startChainManager) Outbound(tag string) (adapter.Outbound, bool) {
|
||||||
|
ob, loaded := m.byTag[tag]
|
||||||
|
return ob, loaded
|
||||||
|
}
|
||||||
|
|
||||||
|
func randomKey(t *testing.T) string {
|
||||||
|
t.Helper()
|
||||||
|
var key [32]byte
|
||||||
|
if _, err := rand.Read(key[:]); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
// Clamp so wireguard-go accepts it as a curve25519 private key.
|
||||||
|
key[0] &= 248
|
||||||
|
key[31] = (key[31] & 127) | 64
|
||||||
|
return base64.StdEncoding.EncodeToString(key[:])
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestAmneziaWGOverWireGuardDetourStarts pins the invariant: an AmneziaWG
|
||||||
|
// endpoint detouring through a WireGuard hop must come up like any other.
|
||||||
|
func TestAmneziaWGOverWireGuardDetourStarts(t *testing.T) {
|
||||||
|
ctx := pause.WithDefaultManager(context.Background())
|
||||||
|
ctx = service.ContextWith[adapter.OutboundManager](ctx, &startChainManager{
|
||||||
|
byTag: map[string]adapter.Outbound{
|
||||||
|
"wg-hop": &wgTypedOutbound{tag: "wg-hop"},
|
||||||
|
},
|
||||||
|
})
|
||||||
|
options := option.WireGuardEndpointOptions{
|
||||||
|
MTU: 1280,
|
||||||
|
Address: badoption.Listable[netip.Prefix]{netip.MustParsePrefix("10.7.0.2/32")},
|
||||||
|
PrivateKey: randomKey(t),
|
||||||
|
Peers: []option.WireGuardPeer{{
|
||||||
|
Address: "10.9.9.9",
|
||||||
|
Port: 51820,
|
||||||
|
PublicKey: randomKey(t),
|
||||||
|
AllowedIPs: badoption.Listable[netip.Prefix]{netip.MustParsePrefix("0.0.0.0/0")},
|
||||||
|
}},
|
||||||
|
AmneziaWGOptions: option.AmneziaWGOptions{
|
||||||
|
Jc: 3,
|
||||||
|
Jmin: 8,
|
||||||
|
Jmax: 80,
|
||||||
|
S4: 16,
|
||||||
|
H1: "10-20",
|
||||||
|
H2: "30-40",
|
||||||
|
H3: "50-60",
|
||||||
|
H4: "70-80",
|
||||||
|
},
|
||||||
|
}
|
||||||
|
options.Detour = "wg-hop"
|
||||||
|
|
||||||
|
ep, err := NewEndpoint(ctx, nil, log.NewNOPFactory().NewLogger("wg-awg"), "wg-awg", options)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal("create amneziawg endpoint over a wireguard detour: ", err)
|
||||||
|
}
|
||||||
|
defer ep.Close()
|
||||||
|
|
||||||
|
if err = ep.Start(adapter.StartStateStart); err != nil {
|
||||||
|
t.Fatal("start stage: ", err)
|
||||||
|
}
|
||||||
|
if err = ep.Start(adapter.StartStatePostStart); err != nil {
|
||||||
|
t.Fatal("post-start stage: ", err)
|
||||||
|
}
|
||||||
|
if !ep.(*Endpoint).started.Load() {
|
||||||
|
t.Fatal("an amneziawg endpoint behind a wireguard hop must start; it is silently held down")
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -1,100 +0,0 @@
|
|||||||
// lx:begin awg
|
|
||||||
|
|
||||||
package wireguard
|
|
||||||
|
|
||||||
import (
|
|
||||||
"testing"
|
|
||||||
|
|
||||||
"github.com/sagernet/sing-box/adapter"
|
|
||||||
C "github.com/sagernet/sing-box/constant"
|
|
||||||
)
|
|
||||||
|
|
||||||
// fakeOutbound is a minimal adapter.Outbound for the start-guard chain walk:
|
|
||||||
// only Type() and Dependencies() (the detour) are consulted. The embedded
|
|
||||||
// interface is nil — any other method would panic, which never happens here.
|
|
||||||
type fakeOutbound struct {
|
|
||||||
adapter.Outbound
|
|
||||||
tag string
|
|
||||||
outboundTyp string
|
|
||||||
detour string
|
|
||||||
}
|
|
||||||
|
|
||||||
func (o *fakeOutbound) Type() string { return o.outboundTyp }
|
|
||||||
func (o *fakeOutbound) Tag() string { return o.tag }
|
|
||||||
func (o *fakeOutbound) Dependencies() []string {
|
|
||||||
if o.detour == "" {
|
|
||||||
return nil
|
|
||||||
}
|
|
||||||
return []string{o.detour}
|
|
||||||
}
|
|
||||||
|
|
||||||
// fakeGroup is an adapter.OutboundGroup (selector/urltest stand-in); the chain
|
|
||||||
// walk must stop at it without expanding All().
|
|
||||||
type fakeGroup struct {
|
|
||||||
fakeOutbound
|
|
||||||
members []string
|
|
||||||
}
|
|
||||||
|
|
||||||
func (g *fakeGroup) Now() string { return "" }
|
|
||||||
func (g *fakeGroup) All() []string { return g.members }
|
|
||||||
|
|
||||||
type fakeOutboundManager struct {
|
|
||||||
adapter.OutboundManager
|
|
||||||
byTag map[string]adapter.Outbound
|
|
||||||
}
|
|
||||||
|
|
||||||
func (m *fakeOutboundManager) Outbound(tag string) (adapter.Outbound, bool) {
|
|
||||||
ob, loaded := m.byTag[tag]
|
|
||||||
return ob, loaded
|
|
||||||
}
|
|
||||||
|
|
||||||
func TestAwgDetourChainReachesWireGuard(t *testing.T) {
|
|
||||||
mgr := &fakeOutboundManager{byTag: map[string]adapter.Outbound{
|
|
||||||
// AWG -> wg-out (direct)
|
|
||||||
"wg-out": &fakeOutbound{tag: "wg-out", outboundTyp: C.TypeWireGuard},
|
|
||||||
// AWG -> vless-hop -> wg-deep (transitive)
|
|
||||||
"vless-hop": &fakeOutbound{tag: "vless-hop", outboundTyp: C.TypeVLESS, detour: "wg-deep"},
|
|
||||||
"wg-deep": &fakeOutbound{tag: "wg-deep", outboundTyp: C.TypeWireGuard},
|
|
||||||
// AWG -> vless-leaf -> direct-leaf (no wireguard anywhere)
|
|
||||||
"vless-leaf": &fakeOutbound{tag: "vless-leaf", outboundTyp: C.TypeVLESS, detour: "direct-leaf"},
|
|
||||||
"direct-leaf": &fakeOutbound{tag: "direct-leaf", outboundTyp: C.TypeDirect},
|
|
||||||
// AWG -> sel (selector hiding a wireguard member) — walk must stop, return ""
|
|
||||||
"sel": &fakeGroup{
|
|
||||||
fakeOutbound: fakeOutbound{tag: "sel", outboundTyp: C.TypeSelector},
|
|
||||||
members: []string{"wg-out"},
|
|
||||||
},
|
|
||||||
// cyclic detour: a -> b -> a, no wireguard
|
|
||||||
"cyc-a": &fakeOutbound{tag: "cyc-a", outboundTyp: C.TypeVLESS, detour: "cyc-b"},
|
|
||||||
"cyc-b": &fakeOutbound{tag: "cyc-b", outboundTyp: C.TypeVLESS, detour: "cyc-a"},
|
|
||||||
}}
|
|
||||||
|
|
||||||
cases := []struct {
|
|
||||||
name string
|
|
||||||
start string
|
|
||||||
wantEmpty bool
|
|
||||||
wantTag string
|
|
||||||
}{
|
|
||||||
{"direct wireguard", "wg-out", false, "wg-out"},
|
|
||||||
{"transitive via vless", "vless-hop", false, "wg-deep"},
|
|
||||||
{"no wireguard in chain", "vless-leaf", true, ""},
|
|
||||||
{"selector in the middle is skipped", "sel", true, ""},
|
|
||||||
{"cyclic chain terminates", "cyc-a", true, ""},
|
|
||||||
{"unknown tag", "nope", true, ""},
|
|
||||||
}
|
|
||||||
for _, tc := range cases {
|
|
||||||
t.Run(tc.name, func(t *testing.T) {
|
|
||||||
got := awgDetourChainReachesWireGuard(mgr, tc.start, make(map[string]bool))
|
|
||||||
if tc.wantEmpty {
|
|
||||||
if got != "" {
|
|
||||||
t.Fatalf("expected no wireguard in chain, got %q", got)
|
|
||||||
}
|
|
||||||
return
|
|
||||||
}
|
|
||||||
if got != tc.wantTag {
|
|
||||||
t.Fatalf("expected blocked-by %q, got %q", tc.wantTag, got)
|
|
||||||
}
|
|
||||||
})
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
// lx:end awg
|
|
||||||
+20
-118
@@ -4,7 +4,6 @@ import (
|
|||||||
"context"
|
"context"
|
||||||
"net"
|
"net"
|
||||||
"net/netip"
|
"net/netip"
|
||||||
"strconv"
|
|
||||||
"sync"
|
"sync"
|
||||||
"sync/atomic"
|
"sync/atomic"
|
||||||
"time"
|
"time"
|
||||||
@@ -45,28 +44,14 @@ type Endpoint struct {
|
|||||||
localAddresses []netip.Prefix
|
localAddresses []netip.Prefix
|
||||||
endpoint *wireguard.Endpoint
|
endpoint *wireguard.Endpoint
|
||||||
started atomic.Bool
|
started atomic.Bool
|
||||||
// lx:begin awg
|
|
||||||
// awgActive marks this endpoint as running AmneziaWG (AmneziaWGOptions.IsSet());
|
|
||||||
// detour is its configured upstream tag. Start uses them to refuse to bring up
|
|
||||||
// an AmneziaWG-over-WireGuard chain, which hangs the kernel on Android — see
|
|
||||||
// awgDetourChainReachesWireGuard. The ledger lives here (not just in the dialer
|
|
||||||
// guard) because the hang happens synchronously in Start, before any dial.
|
|
||||||
awgActive bool
|
|
||||||
detour string
|
|
||||||
// awgChainBlocked is set by Start when the AmneziaWG-over-WireGuard guard
|
|
||||||
// fires: the device is left unstarted (started stays false) so no junk
|
|
||||||
// handshake runs and the kernel cannot hang, while the rest of the instance
|
|
||||||
// comes up. PostStart then skips this endpoint too.
|
|
||||||
awgChainBlocked bool
|
|
||||||
// lx:end awg
|
|
||||||
// lx:begin idle-suspend
|
// lx:begin idle-suspend
|
||||||
// SPEC 020 idle-suspend state. lastActivity is the unix-nano timestamp of the
|
// SPEC 020 idle-suspend state. lastActivity is the unix-nano timestamp of the
|
||||||
// last dial through this endpoint, stamped at PostStart and on every dial entry.
|
// last dial through this endpoint, stamped at PostStart and on every dial entry.
|
||||||
// idleAsleep is true while the endpoint is Down due to idle-suspend (distinct
|
// idleAsleep is true while the endpoint is Down due to idle-suspend (distinct
|
||||||
// from a guard-suspend, which sets started=false and clears idleAsleep, so a
|
// from a deliberately-stopped endpoint, which has started=false and
|
||||||
// guard-suspended endpoint fast-paths out of resumeOnDial and is never
|
// idleAsleep=false, so it fast-paths out of resumeOnDial and is never
|
||||||
// idle-woken). resumeMu serialises the idle tick's suspend decision, a dial's
|
// idle-woken). resumeMu serialises the idle tick's suspend decision against a
|
||||||
// wake, and the AmneziaWG guard-suspend against one another.
|
// dial's wake.
|
||||||
lastActivity atomic.Int64
|
lastActivity atomic.Int64
|
||||||
idleAsleep atomic.Bool
|
idleAsleep atomic.Bool
|
||||||
resumeMu sync.Mutex
|
resumeMu sync.Mutex
|
||||||
@@ -74,6 +59,16 @@ type Endpoint struct {
|
|||||||
}
|
}
|
||||||
|
|
||||||
func NewEndpoint(ctx context.Context, router adapter.Router, logger log.ContextLogger, tag string, options option.WireGuardEndpointOptions) (adapter.Endpoint, error) {
|
func NewEndpoint(ctx context.Context, router adapter.Router, logger log.ContextLogger, tag string, options option.WireGuardEndpointOptions) (adapter.Endpoint, error) {
|
||||||
|
// lx: allow OS-level fragmentation of the OUTER UDP socket by default, the
|
||||||
|
// same opt-out direct/hysteria/hysteria2/tuic already take. Without it the
|
||||||
|
// dialer sets DF (IP_MTU_DISCOVER=IP_PMTUDISC_DO on linux), and an outer
|
||||||
|
// datagram over the path MTU — routine once anything is encapsulated: WG's
|
||||||
|
// own ~32 B header, AmneziaWG s4 transport junk, or this endpoint carrying a
|
||||||
|
// nested tunnel — is dropped by the kernel ("message too long") instead of
|
||||||
|
// fragmented, so the tunnel comes up and then carries nothing. An explicit
|
||||||
|
// `udp_fragment: false` on the node still restores DF (UDPFragment wins over
|
||||||
|
// UDPFragmentDefault in common/dialer).
|
||||||
|
options.UDPFragmentDefault = true
|
||||||
ep := &Endpoint{
|
ep := &Endpoint{
|
||||||
Adapter: endpoint.NewAdapterWithDialerOptions(C.TypeWireGuard, tag, []string{N.NetworkTCP, N.NetworkUDP, N.NetworkICMP}, options.DialerOptions),
|
Adapter: endpoint.NewAdapterWithDialerOptions(C.TypeWireGuard, tag, []string{N.NetworkTCP, N.NetworkUDP, N.NetworkICMP}, options.DialerOptions),
|
||||||
ctx: ctx,
|
ctx: ctx,
|
||||||
@@ -81,10 +76,6 @@ func NewEndpoint(ctx context.Context, router adapter.Router, logger log.ContextL
|
|||||||
dnsRouter: service.FromContext[adapter.DNSRouter](ctx),
|
dnsRouter: service.FromContext[adapter.DNSRouter](ctx),
|
||||||
logger: logger,
|
logger: logger,
|
||||||
localAddresses: options.Address,
|
localAddresses: options.Address,
|
||||||
// lx:begin awg
|
|
||||||
awgActive: options.AmneziaWGOptions.IsSet(),
|
|
||||||
detour: options.Detour,
|
|
||||||
// lx:end awg
|
|
||||||
}
|
}
|
||||||
if options.Detour != "" && options.ListenPort != 0 {
|
if options.Detour != "" && options.ListenPort != 0 {
|
||||||
return nil, E.New("`listen_port` is conflict with `detour`")
|
return nil, E.New("`listen_port` is conflict with `detour`")
|
||||||
@@ -116,7 +107,8 @@ func NewEndpoint(ctx context.Context, router adapter.Router, logger log.ContextL
|
|||||||
Dialer: outboundDialer,
|
Dialer: outboundDialer,
|
||||||
CreateDialer: func(interfaceName string) N.Dialer {
|
CreateDialer: func(interfaceName string) N.Dialer {
|
||||||
return common.Must1(dialer.NewDefault(ctx, option.DialerOptions{
|
return common.Must1(dialer.NewDefault(ctx, option.DialerOptions{
|
||||||
BindInterface: interfaceName,
|
BindInterface: interfaceName,
|
||||||
|
UDPFragmentDefault: true, // lx: same reason as above — this is the bind-to-interface twin of the outer socket
|
||||||
}))
|
}))
|
||||||
},
|
},
|
||||||
Name: options.Name,
|
Name: options.Name,
|
||||||
@@ -157,33 +149,6 @@ func NewEndpoint(ctx context.Context, router adapter.Router, logger log.ContextL
|
|||||||
}
|
}
|
||||||
|
|
||||||
func (w *Endpoint) Start(stage adapter.StartStage) error {
|
func (w *Endpoint) Start(stage adapter.StartStage) error {
|
||||||
// lx:begin awg
|
|
||||||
// Refuse to bring up an AmneziaWG endpoint whose detour chain reaches a
|
|
||||||
// WireGuard-based endpoint: encapsulating AWG (junk handshake) inside a
|
|
||||||
// WireGuard tunnel hangs the kernel on Android. The hang happens here, in the
|
|
||||||
// synchronous Start path (peer-domain resolution over the detour, then the
|
|
||||||
// device's junk handshake) — before any dial — so the lazy DetourDialer guard
|
|
||||||
// never gets a chance to fire. We must catch it at Start instead.
|
|
||||||
//
|
|
||||||
// Behaviour is "variant B": do NOT return an error (that would abort the whole
|
|
||||||
// instance start). Instead log, skip device startup, and leave started=false
|
|
||||||
// so the rest of the config comes up and every dial through this endpoint
|
|
||||||
// fails cleanly with "WireGuard is not ready yet". A selector/urltest in the
|
|
||||||
// middle hides the real target at start time, so the chain walk stops at a
|
|
||||||
// group and that case is left to the lazy DetourDialer guard at dial time.
|
|
||||||
if stage == adapter.StartStateStart && w.awgActive && w.detour != "" {
|
|
||||||
if outboundManager := service.FromContext[adapter.OutboundManager](w.ctx); outboundManager != nil {
|
|
||||||
if blockedBy := awgDetourChainReachesWireGuard(outboundManager, w.detour, make(map[string]bool)); blockedBy != "" {
|
|
||||||
w.awgChainBlocked = true
|
|
||||||
w.logger.Error("amneziawg endpoint will not start: its detour chain reaches wireguard-based endpoint ", strconv.Quote(blockedBy), " — amneziawg over wireguard is not supported. Use a non-wireguard detour (e.g. vless).")
|
|
||||||
return nil
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
if w.awgChainBlocked {
|
|
||||||
return nil
|
|
||||||
}
|
|
||||||
// lx:end awg
|
|
||||||
switch stage {
|
switch stage {
|
||||||
case adapter.StartStateStart:
|
case adapter.StartStateStart:
|
||||||
return w.endpoint.Start(false)
|
return w.endpoint.Start(false)
|
||||||
@@ -200,69 +165,6 @@ func (w *Endpoint) Start(stage adapter.StartStage) error {
|
|||||||
return nil
|
return nil
|
||||||
}
|
}
|
||||||
|
|
||||||
// lx:begin awg
|
|
||||||
// awgDetourChainReachesWireGuard walks the transitive detour chain starting at
|
|
||||||
// tag and returns the tag of the first WireGuard-based outbound it reaches
|
|
||||||
// (type "wireguard", covering plain WireGuard and AmneziaWG), or "" if none. It
|
|
||||||
// follows each outbound's detour dependency; it deliberately does NOT expand
|
|
||||||
// selector/urltest groups, whose chosen member is only known at runtime — that
|
|
||||||
// case is handled lazily by the DetourDialer guard. visited guards against cyclic
|
|
||||||
// detour configs. All outbounds are registered before any Start, so every tag in
|
|
||||||
// the chain is resolvable here even though some may not have started yet.
|
|
||||||
func awgDetourChainReachesWireGuard(outboundManager adapter.OutboundManager, tag string, visited map[string]bool) string {
|
|
||||||
if tag == "" || visited[tag] {
|
|
||||||
return ""
|
|
||||||
}
|
|
||||||
visited[tag] = true
|
|
||||||
outbound, loaded := outboundManager.Outbound(tag)
|
|
||||||
if !loaded {
|
|
||||||
return ""
|
|
||||||
}
|
|
||||||
if outbound.Type() == C.TypeWireGuard {
|
|
||||||
return tag
|
|
||||||
}
|
|
||||||
if _, isGroup := outbound.(adapter.OutboundGroup); isGroup {
|
|
||||||
// Runtime-resolved target — leave it to the lazy DetourDialer guard.
|
|
||||||
return ""
|
|
||||||
}
|
|
||||||
for _, dependency := range outbound.Dependencies() {
|
|
||||||
if blockedBy := awgDetourChainReachesWireGuard(outboundManager, dependency, visited); blockedBy != "" {
|
|
||||||
return blockedBy
|
|
||||||
}
|
|
||||||
}
|
|
||||||
return ""
|
|
||||||
}
|
|
||||||
|
|
||||||
// IsAmneziaWG reports whether this endpoint runs AmneziaWG. Implements
|
|
||||||
// adapter.AmneziaWGSuspendable.
|
|
||||||
func (w *Endpoint) IsAmneziaWG() bool {
|
|
||||||
return w.awgActive
|
|
||||||
}
|
|
||||||
|
|
||||||
// SuspendAmneziaWG brings the device down and marks the endpoint not-ready, so a
|
|
||||||
// junk handshake is never sent and every dial fails with "WireGuard is not ready
|
|
||||||
// yet". Called by the selector guard when a group this endpoint detours through
|
|
||||||
// switches to a WireGuard member (AmneziaWG over WireGuard hangs the kernel on
|
|
||||||
// Android). Idempotent. Implements adapter.AmneziaWGSuspendable.
|
|
||||||
func (w *Endpoint) SuspendAmneziaWG() {
|
|
||||||
// Take resumeMu so this is ordered against resumeOnDial/SuspendIfIdle: without
|
|
||||||
// it, a dial that already passed resumeOnDial's idleAsleep checks could wake
|
|
||||||
// the endpoint back up right after we clear the flag, defeating the guard.
|
|
||||||
w.resumeMu.Lock()
|
|
||||||
defer w.resumeMu.Unlock()
|
|
||||||
if w.started.CompareAndSwap(true, false) {
|
|
||||||
w.logger.Error("amneziawg endpoint suspended: a selector in its detour chain switched to a wireguard-based member — amneziawg over wireguard is not supported")
|
|
||||||
}
|
|
||||||
// Clear any idle-suspend state so resumeOnDial does not resurrect a
|
|
||||||
// guard-suspended endpoint: if it was idle-asleep first, idleAsleep would still
|
|
||||||
// be true and the next dial would wake it (SPEC 022 #2). With idleAsleep=false
|
|
||||||
// resumeOnDial's fast path returns started (now false) and the endpoint stays down.
|
|
||||||
w.idleAsleep.Store(false)
|
|
||||||
w.endpoint.Suspend()
|
|
||||||
}
|
|
||||||
|
|
||||||
// lx:end awg
|
|
||||||
|
|
||||||
// lx:begin idle-suspend
|
// lx:begin idle-suspend
|
||||||
|
|
||||||
// stampActivity records the current time as the last dial through this endpoint.
|
// stampActivity records the current time as the last dial through this endpoint.
|
||||||
@@ -286,8 +188,8 @@ func (w *Endpoint) IdleSince() time.Duration {
|
|||||||
// holder — when it is unreachable from the active routing tree AND has been idle
|
// holder — when it is unreachable from the active routing tree AND has been idle
|
||||||
// past the threshold. Silent on every non-transition (edge-triggered logging).
|
// past the threshold. Silent on every non-transition (edge-triggered logging).
|
||||||
//
|
//
|
||||||
// It never touches a guard-suspended endpoint: that one already has
|
// It never touches a deliberately-stopped endpoint: that one already has
|
||||||
// started==false but idleAsleep==false, and the `!started` guard below short-
|
// started==false but idleAsleep==false, and the `!started` check below short-
|
||||||
// circuits before the CAS. resumeMu mutually excludes this against resumeOnDial.
|
// circuits before the CAS. resumeMu mutually excludes this against resumeOnDial.
|
||||||
func (w *Endpoint) SuspendIfIdle(reachable bool, threshold time.Duration) {
|
func (w *Endpoint) SuspendIfIdle(reachable bool, threshold time.Duration) {
|
||||||
w.resumeMu.Lock()
|
w.resumeMu.Lock()
|
||||||
@@ -296,7 +198,7 @@ func (w *Endpoint) SuspendIfIdle(reachable bool, threshold time.Duration) {
|
|||||||
return
|
return
|
||||||
}
|
}
|
||||||
if !w.started.Load() {
|
if !w.started.Load() {
|
||||||
// Already down some other way (guard-suspend, awg-chain-blocked, closed).
|
// Already down some other way (deliberately stopped, closed).
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
if w.idleAsleep.CompareAndSwap(false, true) {
|
if w.idleAsleep.CompareAndSwap(false, true) {
|
||||||
@@ -313,7 +215,7 @@ func (w *Endpoint) SuspendIfIdle(reachable bool, threshold time.Duration) {
|
|||||||
// session); that cost is on the first packet, as for any cold WG dial.
|
// session); that cost is on the first packet, as for any cold WG dial.
|
||||||
//
|
//
|
||||||
// Returns true if the endpoint is dialable (awake), false if it must stay down
|
// Returns true if the endpoint is dialable (awake), false if it must stay down
|
||||||
// (guard-suspend / chain-blocked — not an idle-suspend, so we do not resurrect it).
|
// (deliberately stopped / closed — not an idle-suspend, so we do not resurrect it).
|
||||||
func (w *Endpoint) resumeOnDial() bool {
|
func (w *Endpoint) resumeOnDial() bool {
|
||||||
w.stampActivity()
|
w.stampActivity()
|
||||||
if !w.idleAsleep.Load() {
|
if !w.idleAsleep.Load() {
|
||||||
|
|||||||
@@ -103,21 +103,21 @@ func TestSuspendIfIdle_idempotentCAS(t *testing.T) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
// TestSuspendIfIdle_guardSuspendedNotTouched is the §8 invariant verified live on
|
// TestSuspendIfIdle_stoppedNotTouched is the §8 invariant: a deliberately-stopped
|
||||||
// an AWG-over-WG endpoint (wg-3 in the prod run): a guard-suspended endpoint has
|
// endpoint (Close, or a start that never completed) has started=false WITHOUT
|
||||||
// started=false WITHOUT idleAsleep. The idle tick must early-return on !started and
|
// idleAsleep. The idle tick must early-return on !started and NOT flip idleAsleep
|
||||||
// NOT flip idleAsleep — otherwise a later resumeOnDial would idle-wake it and
|
// — otherwise a later resumeOnDial would idle-wake a device that was
|
||||||
// re-trigger the AWG-over-WG kernel hang the guard exists to prevent.
|
// intentionally down.
|
||||||
func TestSuspendIfIdle_guardSuspendedNotTouched(t *testing.T) {
|
func TestSuspendIfIdle_stoppedNotTouched(t *testing.T) {
|
||||||
w := newIdleTestEndpoint()
|
w := newIdleTestEndpoint()
|
||||||
w.started.Store(false) // guard-suspend (device.Down at Start), idleAsleep stays false
|
w.started.Store(false) // stopped, idleAsleep stays false
|
||||||
w.lastActivity.Store(time.Now().Add(-time.Hour).UnixNano())
|
w.lastActivity.Store(time.Now().Add(-time.Hour).UnixNano())
|
||||||
w.SuspendIfIdle(false, 30*time.Second)
|
w.SuspendIfIdle(false, 30*time.Second)
|
||||||
if w.idleAsleep.Load() {
|
if w.idleAsleep.Load() {
|
||||||
t.Fatal("a guard-suspended endpoint must NOT be flagged idleAsleep by the tick")
|
t.Fatal("a stopped endpoint must NOT be flagged idleAsleep by the tick")
|
||||||
}
|
}
|
||||||
if w.started.Load() {
|
if w.started.Load() {
|
||||||
t.Fatal("the tick must not change started for a guard-suspended endpoint")
|
t.Fatal("the tick must not change started for a stopped endpoint")
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -152,17 +152,17 @@ func TestResumeOnDial_dialBeforeTickRace(t *testing.T) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
func TestResumeOnDial_guardSuspendedNotWoken(t *testing.T) {
|
func TestResumeOnDial_stoppedNotWoken(t *testing.T) {
|
||||||
// A guard-suspended endpoint has started=false but idleAsleep=false.
|
// A deliberately-stopped endpoint (Close / failed start) has started=false but
|
||||||
// resumeOnDial must NOT wake it (returns started, i.e. false).
|
// idleAsleep=false. resumeOnDial must NOT wake it (returns started, i.e. false).
|
||||||
w := newIdleTestEndpoint()
|
w := newIdleTestEndpoint()
|
||||||
w.started.Store(false) // simulate guard/awg-chain suspend (not idle)
|
w.started.Store(false) // stopped, not idle-suspended
|
||||||
ok := w.resumeOnDial()
|
ok := w.resumeOnDial()
|
||||||
if ok {
|
if ok {
|
||||||
t.Fatal("resumeOnDial must not resurrect a guard-suspended (non-idle) endpoint")
|
t.Fatal("resumeOnDial must not resurrect a stopped (non-idle) endpoint")
|
||||||
}
|
}
|
||||||
if w.idleAsleep.Load() {
|
if w.idleAsleep.Load() {
|
||||||
t.Fatal("guard-suspended endpoint must not be flagged idleAsleep")
|
t.Fatal("stopped endpoint must not be flagged idleAsleep")
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
@@ -0,0 +1,88 @@
|
|||||||
|
#!/usr/bin/env bash
|
||||||
|
#
|
||||||
|
# run-panel-tests.sh — the admin-panel half of the release test gate.
|
||||||
|
#
|
||||||
|
# panel/package.json has declared a `test` script since the SPA was scaffolded
|
||||||
|
# and nothing had ever called it: not release.yml, not build-shaterd.sh (which
|
||||||
|
# only runs `npm ci` + `npm run build`). This script is what CI calls, and it
|
||||||
|
# does the one thing `npm test` alone cannot: prove that tests actually RAN.
|
||||||
|
#
|
||||||
|
# `node --test src/*.test.ts` with no matching file leaves the glob unexpanded;
|
||||||
|
# node then reports `pass 0` and exits 0 — a green CI step that ran nothing,
|
||||||
|
# which is the exact failure class this whole change is about. So: the test
|
||||||
|
# files are counted BEFORE the run (a rename to *.spec.ts is named as such
|
||||||
|
# rather than showing up as a mystery), and the pass count is asserted > 0 and
|
||||||
|
# the fail count 0 after it.
|
||||||
|
#
|
||||||
|
# KNOWN LIMIT: node --test counts a *.test.ts file that declares no cases at
|
||||||
|
# all as one passing "test" (the module loaded). So an emptied-out file still
|
||||||
|
# reads as pass 1 here. Deleting, renaming or breaking the file is caught;
|
||||||
|
# gutting its contents while keeping the name is not.
|
||||||
|
#
|
||||||
|
# NODE VERSION: >= 22.6. The tests are TypeScript executed directly by
|
||||||
|
# `node --test`; type stripping does not exist before then, so on node 20 the
|
||||||
|
# run dies with a syntax error. CI pins node 24 for this step (the SPA *build*
|
||||||
|
# still uses node 20 — that one goes through vite/tsc and does not care).
|
||||||
|
#
|
||||||
|
# Usage: scripts/run-panel-tests.sh
|
||||||
|
set -euo pipefail
|
||||||
|
|
||||||
|
SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
|
||||||
|
REPO="$(cd "$SCRIPT_DIR/.." && pwd)"
|
||||||
|
cd "$REPO/panel"
|
||||||
|
|
||||||
|
node_major="$(node -p 'process.versions.node.split(".")[0]')"
|
||||||
|
node_minor="$(node -p 'process.versions.node.split(".")[1]')"
|
||||||
|
if [ "$node_major" -lt 22 ] || { [ "$node_major" -eq 22 ] && [ "$node_minor" -lt 6 ]; }; then
|
||||||
|
echo " ERROR: panel tests are TypeScript under \`node --test\` and need node >= 22.6" >&2
|
||||||
|
echo " (got $(node --version)). Type stripping does not exist before that." >&2
|
||||||
|
exit 1
|
||||||
|
fi
|
||||||
|
|
||||||
|
# The glob `npm test` itself uses. Counted here so "somebody renamed the tests"
|
||||||
|
# is reported as that, instead of as a suspiciously fast green step.
|
||||||
|
shopt -s nullglob
|
||||||
|
files=(src/*.test.ts)
|
||||||
|
shopt -u nullglob
|
||||||
|
if [ "${#files[@]}" -eq 0 ]; then
|
||||||
|
echo " ERROR: no panel/src/*.test.ts — panel/package.json's \`test\` script" >&2
|
||||||
|
echo " would match nothing and still exit 0. Fix the glob or the files." >&2
|
||||||
|
exit 1
|
||||||
|
fi
|
||||||
|
|
||||||
|
echo "== panel tests (node $(node --version)), ${#files[@]} file(s) =="
|
||||||
|
if [ ! -d node_modules ]; then
|
||||||
|
npm ci
|
||||||
|
fi
|
||||||
|
|
||||||
|
out=""
|
||||||
|
rc=0
|
||||||
|
set +e
|
||||||
|
out="$(npm test --silent 2>&1)"
|
||||||
|
rc=$?
|
||||||
|
set -e
|
||||||
|
sed 's/^/ /' <<<"$out"
|
||||||
|
|
||||||
|
if [ "$rc" -ne 0 ]; then
|
||||||
|
echo " FAILED: npm test exited $rc" >&2
|
||||||
|
exit 1
|
||||||
|
fi
|
||||||
|
|
||||||
|
# node --test's summary is `ℹ pass N` (spec reporter) or `# pass N` (tap).
|
||||||
|
passed="$(sed -n 's/.*[[:space:]]pass[[:space:]]\{1,\}\([0-9]\{1,\}\).*/\1/p' <<<"$out" | tail -1)"
|
||||||
|
failed="$(sed -n 's/.*[[:space:]]fail[[:space:]]\{1,\}\([0-9]\{1,\}\).*/\1/p' <<<"$out" | tail -1)"
|
||||||
|
if [ -z "$passed" ]; then
|
||||||
|
echo " FAILED: could not find a pass count in node --test output — the gate" >&2
|
||||||
|
echo " cannot tell a green run from an empty one." >&2
|
||||||
|
exit 1
|
||||||
|
fi
|
||||||
|
if [ "$passed" -lt 1 ]; then
|
||||||
|
echo " FAILED: 0 panel tests ran. \`npm test\` returned success having done nothing." >&2
|
||||||
|
exit 1
|
||||||
|
fi
|
||||||
|
if [ -n "$failed" ] && [ "$failed" -gt 0 ]; then
|
||||||
|
echo " FAILED: $failed panel test(s) failed." >&2
|
||||||
|
exit 1
|
||||||
|
fi
|
||||||
|
|
||||||
|
echo " OK: $passed panel test(s) passed."
|
||||||
@@ -0,0 +1,249 @@
|
|||||||
|
#!/usr/bin/env bash
|
||||||
|
#
|
||||||
|
# run-tests.sh — THE test gate of the release tract.
|
||||||
|
#
|
||||||
|
# WHY THIS EXISTS (2026-07-26)
|
||||||
|
# Until now the release tract ran almost no tests. The only `go test` calls in
|
||||||
|
# the whole publishing path were scripts/build-shaterd.sh's one-package
|
||||||
|
# buildtags check and the three named tests scripts/check-router-tags.sh runs.
|
||||||
|
# The upstream .github/workflows/test.yml triggers on `stable`/`testing`/
|
||||||
|
# `unstable` — branches this fork does not have — and Gitea does not read
|
||||||
|
# .github/workflows at all once .gitea/workflows exists. Net effect: 115 of the
|
||||||
|
# 116 test files under shater/** had never executed in CI, and
|
||||||
|
# TestDNSFilterRemoteBlocklistHTTPClient shipped red through two releases
|
||||||
|
# before anyone ran it by hand.
|
||||||
|
#
|
||||||
|
# WHAT IT GUARANTEES
|
||||||
|
# 1. The suite runs under the SHIPPED build tags (scripts/router-tags.sh), not
|
||||||
|
# under some CI-local tag set. This is not cosmetic: the AmneziaWG tests in
|
||||||
|
# transport/wireguard are `//go:build with_awg` — 1 test file compiles
|
||||||
|
# without the tag set, 7 with it. The 2026-07-25 WireGuard outage was
|
||||||
|
# exactly a "built with X, verified with Y" gap.
|
||||||
|
# 2. It runs on linux. shater/generate has 44 test files on linux against 32 on
|
||||||
|
# windows/darwin; the linux-only half is where the routing, ruleset, DNS and
|
||||||
|
# health tests live.
|
||||||
|
# 3. Nothing is skipped SILENTLY. Two machine checks:
|
||||||
|
# - the tag set may only ADD test files, never hide them (a test behind
|
||||||
|
# `//go:build !with_awg` would vanish from the gate — this fails first);
|
||||||
|
# - every package that has tests must report `ok` by name; a suite that
|
||||||
|
# compiles down to "no test files" fails the gate instead of passing it.
|
||||||
|
# A guard that silently runs nothing is worse than no guard (same rule as
|
||||||
|
# scripts/check-router-tags.sh).
|
||||||
|
#
|
||||||
|
# Usage:
|
||||||
|
# scripts/run-tests.sh # full gate (~3 min warm on the runner)
|
||||||
|
# scripts/run-tests.sh --no-race # skip the -race pass (faster; local loop)
|
||||||
|
#
|
||||||
|
# Env:
|
||||||
|
# SHATER_GO_IMAGE docker image used to reach linux from a non-linux host
|
||||||
|
# (default golang:1.26 — keep it >= go.mod's toolchain).
|
||||||
|
# SHATER_NO_DOCKER=1 fail instead of falling back to docker.
|
||||||
|
set -euo pipefail
|
||||||
|
|
||||||
|
SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
|
||||||
|
REPO="$(cd "$SCRIPT_DIR/.." && pwd)"
|
||||||
|
cd "$REPO"
|
||||||
|
|
||||||
|
RACE=1
|
||||||
|
for a in "$@"; do
|
||||||
|
case "$a" in
|
||||||
|
--no-race) RACE=0 ;;
|
||||||
|
-h|--help) sed -n '2,41p' "$0" | sed 's/^# \{0,1\}//'; exit 0 ;;
|
||||||
|
*) echo "run-tests: unknown flag: $a" >&2; exit 2 ;;
|
||||||
|
esac
|
||||||
|
done
|
||||||
|
|
||||||
|
# shellcheck source=router-tags.sh
|
||||||
|
. "$SCRIPT_DIR/router-tags.sh"
|
||||||
|
|
||||||
|
# The fork's own trees plus the upstream trees the fork edits (adapter/, route/,
|
||||||
|
# option/, dns/ all carry shater changes). ROOTS, not a hand-kept package list: a
|
||||||
|
# new package with tests joins the gate the moment it is created, which is the
|
||||||
|
# whole point.
|
||||||
|
ROOTS=(./shater/... ./protocol/... ./transport/... ./adapter/... ./route/... ./option/... ./dns/...)
|
||||||
|
|
||||||
|
# common/ is mostly upstream, but common/tls, common/tlsfragment, common/sniff
|
||||||
|
# and common/urltest carry fork behaviour (D13 DPI-bypass, urltest health), so it
|
||||||
|
# is in — minus the privileged integration tests below.
|
||||||
|
ROOTS_COMMON=(./common/...)
|
||||||
|
# SKIP, WITH REASON: common/tlsspoof's TestIntegration* enter TCP_REPAIR and
|
||||||
|
# need CAP_NET_ADMIN. The act_runner job container runs as root but WITHOUT
|
||||||
|
# that capability, so they do not skip — they FAIL. Excluded by name so the
|
||||||
|
# rest of common/ can be a real gate instead of a permanently red one. (On
|
||||||
|
# linux every tlsspoof test is a TestIntegration*, so that package is
|
||||||
|
# effectively uncovered here; it is covered by the VM runs.)
|
||||||
|
SKIP_COMMON='^TestIntegration'
|
||||||
|
|
||||||
|
# SKIP, WITH REASON: the first -race run over this tree (2026-07-26 — nobody
|
||||||
|
# had ever run one) turned up two failures. One was a REAL product race:
|
||||||
|
# ClientBind.connect() touched its fields from both the Send() path and
|
||||||
|
# RoutineReceiveIncoming() with no lock, caught by
|
||||||
|
# transport/wireguard.TestAwgDetourClientBindDelivers; that one has since been
|
||||||
|
# fixed in client_bind.go and is NOT skipped — it is exactly what this pass is
|
||||||
|
# for. What is left:
|
||||||
|
# - shater/alert.TestExpiryDedupWithinDay — the test's own closure
|
||||||
|
# (expiry_test.go:78) reads a variable the test body writes at :85 while
|
||||||
|
# Notifier.dispatch's goroutine is still delivering. A test-side bug, ~one
|
||||||
|
# mutex to fix, but it lives in shater/ and is nobody's blocker to ship.
|
||||||
|
# Naming it here keeps the gate a gate from day one. It is skipped ONLY in the
|
||||||
|
# -race pass — it still runs, and still has to pass, in the main pass below.
|
||||||
|
# DELETE THE ENTRY THE MOMENT THE RACE IS FIXED.
|
||||||
|
RACE_SKIP='^TestExpiryDedupWithinDay$'
|
||||||
|
|
||||||
|
echo "== shater test gate =="
|
||||||
|
echo " tags : $SHATER_ROUTER_TAGS"
|
||||||
|
echo " ldflags: $SHATER_ROUTER_LDFLAGS"
|
||||||
|
echo " race : $([ "$RACE" -eq 1 ] && echo yes || echo no)"
|
||||||
|
echo
|
||||||
|
|
||||||
|
# --- linux, or re-exec on linux ---------------------------------------------
|
||||||
|
# The linux-only half of the suite is the half worth running (see header). From a
|
||||||
|
# non-linux host, re-exec inside a golang container rather than quietly testing
|
||||||
|
# 32 of shater/generate's 44 files — a partial gate reads exactly like a passing
|
||||||
|
# one.
|
||||||
|
if [ "$(go env GOOS)" != "linux" ] && [ "${SHATER_TESTS_IN_DOCKER:-0}" != "1" ]; then
|
||||||
|
if [ "${SHATER_NO_DOCKER:-0}" = "1" ] || ! command -v docker >/dev/null 2>&1; then
|
||||||
|
echo " ERROR: the gate needs linux (GOOS=$(go env GOOS)) and docker is unavailable/disabled." >&2
|
||||||
|
echo " Run it on the linux CI runner or the OpenWrt VM." >&2
|
||||||
|
exit 1
|
||||||
|
fi
|
||||||
|
image="${SHATER_GO_IMAGE:-golang:1.26}"
|
||||||
|
echo "== re-exec on linux via docker ($image) =="
|
||||||
|
host_repo="$REPO"
|
||||||
|
command -v cygpath >/dev/null 2>&1 && host_repo="$(cygpath -w "$REPO")"
|
||||||
|
MSYS2_ARG_CONV_EXCL='*' MSYS_NO_PATHCONV=1 docker run --rm \
|
||||||
|
-v "$host_repo":/src \
|
||||||
|
-v shater-tagcheck-gomod:/go/pkg/mod \
|
||||||
|
-v shater-tagcheck-gocache:/root/.cache/go-build \
|
||||||
|
-w /src \
|
||||||
|
-e SHATER_TESTS_IN_DOCKER=1 \
|
||||||
|
"$image" bash scripts/run-tests.sh "$@"
|
||||||
|
exit $?
|
||||||
|
fi
|
||||||
|
|
||||||
|
ALL_ROOTS=("${ROOTS[@]}" "${ROOTS_COMMON[@]}")
|
||||||
|
|
||||||
|
# --- [1/4] the tag set may only ADD test files, never hide them --------------
|
||||||
|
# `go list` counts the test files the compiler would actually take. If adding the
|
||||||
|
# shipped tags REMOVES a test file from any package, that test exists but the
|
||||||
|
# gate would never see it — which is the failure mode this whole script is about,
|
||||||
|
# just pointed the other way.
|
||||||
|
echo "== [1/4] no test file is hidden by the shipped tag set =="
|
||||||
|
LISTFMT='{{.ImportPath}} {{len .TestGoFiles}} {{len .XTestGoFiles}}'
|
||||||
|
plain="$(go list -f "$LISTFMT" "${ALL_ROOTS[@]}")"
|
||||||
|
tagged="$(go list -tags "$SHATER_ROUTER_TAGS" -f "$LISTFMT" "${ALL_ROOTS[@]}")"
|
||||||
|
hidden=0
|
||||||
|
while read -r pkg t x; do
|
||||||
|
[ -n "${pkg:-}" ] || continue
|
||||||
|
n_plain=$((t + x))
|
||||||
|
[ "$n_plain" -gt 0 ] || continue
|
||||||
|
line="$(awk -v p="$pkg" '$1 == p { print; exit }' <<<"$tagged")"
|
||||||
|
if [ -z "$line" ]; then
|
||||||
|
echo " HIDDEN: $pkg has $n_plain test file(s) untagged but no package at all under the shipped tags" >&2
|
||||||
|
hidden=1
|
||||||
|
continue
|
||||||
|
fi
|
||||||
|
read -r _ tt tx <<<"$line"
|
||||||
|
n_tagged=$((tt + tx))
|
||||||
|
if [ "$n_tagged" -lt "$n_plain" ]; then
|
||||||
|
echo " HIDDEN: $pkg — $n_plain test file(s) untagged, only $n_tagged under the shipped tags" >&2
|
||||||
|
hidden=1
|
||||||
|
elif [ "$n_tagged" -gt "$n_plain" ]; then
|
||||||
|
echo " +$((n_tagged - n_plain)) tag-gated test file(s): $pkg ($n_plain -> $n_tagged)"
|
||||||
|
fi
|
||||||
|
done <<<"$plain"
|
||||||
|
if [ "$hidden" -ne 0 ]; then
|
||||||
|
echo >&2
|
||||||
|
echo " FAILED: a test file is invisible to the tag set we ship. Either the" >&2
|
||||||
|
echo " constraint is wrong or the tag set is — do not paper over it" >&2
|
||||||
|
echo " by testing with different tags than we build with." >&2
|
||||||
|
exit 1
|
||||||
|
fi
|
||||||
|
echo
|
||||||
|
|
||||||
|
# --- the runner --------------------------------------------------------------
|
||||||
|
# Runs one suite and then PROVES it ran: every package `go list` says has tests
|
||||||
|
# must appear as `ok <pkg>` in the output. `go test` over a package whose tests
|
||||||
|
# all vanished behind a build constraint prints "[no test files]" and exits 0 —
|
||||||
|
# a green run that verified nothing.
|
||||||
|
LOG="$(mktemp)"
|
||||||
|
trap 'rm -f "$LOG"' EXIT
|
||||||
|
FAILED=0
|
||||||
|
|
||||||
|
run_suite() { # $1=label $2=extra go-test flags (may be empty) $3..=packages
|
||||||
|
local label="$1" extra="$2"
|
||||||
|
shift 2
|
||||||
|
local pkgs=("$@") rc=0 expect missing=0 pkg
|
||||||
|
|
||||||
|
expect="$(go list -tags "$SHATER_ROUTER_TAGS" \
|
||||||
|
-f '{{if or .TestGoFiles .XTestGoFiles}}{{.ImportPath}}{{end}}' \
|
||||||
|
"${pkgs[@]}" | grep -v '^$' || true)"
|
||||||
|
if [ -z "$expect" ]; then
|
||||||
|
echo " FAILED [$label]: go list reports no package with tests here — the gate" >&2
|
||||||
|
echo " would have run nothing and passed." >&2
|
||||||
|
FAILED=1
|
||||||
|
return
|
||||||
|
fi
|
||||||
|
echo " packages with tests: $(wc -l <<<"$expect" | tr -d ' ')"
|
||||||
|
|
||||||
|
set +e
|
||||||
|
# shellcheck disable=SC2086 # $extra is a deliberate word-split flag list
|
||||||
|
go test -count=1 $extra \
|
||||||
|
-tags "$SHATER_ROUTER_TAGS" -ldflags "$SHATER_ROUTER_LDFLAGS" \
|
||||||
|
"${pkgs[@]}" >"$LOG" 2>&1
|
||||||
|
rc=$?
|
||||||
|
set -e
|
||||||
|
sed 's/^/ /' "$LOG"
|
||||||
|
|
||||||
|
if [ "$rc" -ne 0 ]; then
|
||||||
|
echo " FAILED [$label]: go test exited $rc" >&2
|
||||||
|
FAILED=1
|
||||||
|
return
|
||||||
|
fi
|
||||||
|
|
||||||
|
while read -r pkg; do
|
||||||
|
[ -n "$pkg" ] || continue
|
||||||
|
grep -qE "^ok[[:space:]]+$pkg([[:space:]]|\$)" "$LOG" || {
|
||||||
|
echo " DID NOT RUN [$label]: $pkg" >&2
|
||||||
|
missing=1
|
||||||
|
}
|
||||||
|
done <<<"$expect"
|
||||||
|
if [ "$missing" -ne 0 ]; then
|
||||||
|
echo " FAILED [$label]: package(s) above have test files but produced no 'ok'" >&2
|
||||||
|
echo " line. Build-constraint or file-name drift emptied them." >&2
|
||||||
|
FAILED=1
|
||||||
|
return
|
||||||
|
fi
|
||||||
|
echo " OK [$label]"
|
||||||
|
}
|
||||||
|
|
||||||
|
# --- [2/4] the fork's trees, shipped tags, linux -----------------------------
|
||||||
|
echo "== [2/4] go test — the fork's trees (shipped tags, linux) =="
|
||||||
|
run_suite main "" "${ROOTS[@]}"
|
||||||
|
echo
|
||||||
|
|
||||||
|
# --- [3/4] common/, minus the tests that need CAP_NET_ADMIN ------------------
|
||||||
|
echo "== [3/4] go test — common/ (minus the CAP_NET_ADMIN integration tests) =="
|
||||||
|
run_suite common "-skip $SKIP_COMMON" "${ROOTS_COMMON[@]}"
|
||||||
|
echo
|
||||||
|
|
||||||
|
# --- [4/4] -race over the same trees -----------------------------------------
|
||||||
|
# Everything, not a subset: shater/netplane alone is ~110 s under -race and it is
|
||||||
|
# the single most concurrency-critical package we own (the nft data plane), so
|
||||||
|
# once it is in, adding the rest costs ~40 s more. common/ is left out — it is
|
||||||
|
# upstream code exercised by upstream CI.
|
||||||
|
if [ "$RACE" -eq 1 ]; then
|
||||||
|
echo "== [4/4] go test -race — the fork's trees =="
|
||||||
|
echo " known-red under -race, skipped BY NAME (fix it and delete from RACE_SKIP):"
|
||||||
|
echo " TestExpiryDedupWithinDay shater/alert (test-side race, expiry_test.go:78/85)"
|
||||||
|
run_suite race "-race -skip $RACE_SKIP" "${ROOTS[@]}"
|
||||||
|
else
|
||||||
|
echo "== [4/4] -race pass skipped (--no-race) =="
|
||||||
|
fi
|
||||||
|
echo
|
||||||
|
|
||||||
|
if [ "$FAILED" -ne 0 ]; then
|
||||||
|
echo "== TEST GATE FAILED — nothing may be published from this run. ==" >&2
|
||||||
|
exit 1
|
||||||
|
fi
|
||||||
|
echo "== OK: the shipped tag set, on linux, passes every test we own. =="
|
||||||
+307
-109
@@ -418,7 +418,7 @@ func (a *Applier) applyLocked(m *model.Model) (bool, error) {
|
|||||||
|
|
||||||
// (2) engine swap. On error the old engine keeps running and we do NOT touch
|
// (2) engine swap. On error the old engine keeps running and we do NOT touch
|
||||||
// netplane — abort and surface the error.
|
// netplane — abort and surface the error.
|
||||||
changed, err := a.eng.Apply(opts)
|
changed, err := engineApply(a, opts)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
// ...unless the engine is not running AT ALL. A failed apply that left the
|
// ...unless the engine is not running AT ALL. A failed apply that left the
|
||||||
// PREVIOUS engine running still has a valid, loaded data plane — replacing
|
// PREVIOUS engine running still has a valid, loaded data plane — replacing
|
||||||
@@ -434,99 +434,19 @@ func (a *Applier) applyLocked(m *model.Model) (bool, error) {
|
|||||||
|
|
||||||
// (3) netplane, fail-closed: on any failure return the error WITHOUT tearing
|
// (3) netplane, fail-closed: on any failure return the error WITHOUT tearing
|
||||||
// the engine/table down (kill-switch/table stay up; the watchdog decides).
|
// the engine/table down (kill-switch/table stay up; the watchdog decides).
|
||||||
//
|
plane, perr := applyDataPlane(a, m, opts, now)
|
||||||
// Render FIRST, then decide whether anything needs re-asserting. Gating the
|
if perr != nil {
|
||||||
// whole data plane on the engine's `changed` flag alone was wrong: the engine
|
// The engine is ALREADY running the new config at this point, and the data
|
||||||
// hash covers option.Options only, so a purely netplane-visible change (the
|
// plane is not — so nothing the previous apply published is true any more.
|
||||||
// kill-switch flipping open->closed, DNS force-intercept, a rule gaining an
|
// Publishing is what this branch used to skip entirely; see abortAfterSwap.
|
||||||
// iface:/zone: source, an inbound device rename) hashed identical, took the
|
return changed, a.abortAfterSwap(m, plane, warnings, configWarnings, perr)
|
||||||
// fast-path, and left the OLD ruleset loaded while the panel reported success.
|
|
||||||
// A stale-but-loaded fail-open table is exactly the leak this audit is about.
|
|
||||||
// Resolve WHERE the untunnelable-protocol drop applies before rendering. The
|
|
||||||
// engine is already running by this point (step 2), so its loaded rule-sets can
|
|
||||||
// supply the addresses a geoip list contributes; a stopped engine or a list that
|
|
||||||
// has not downloaded yet yields no plan and the conservative blanket drop.
|
|
||||||
//
|
|
||||||
// The resolved addresses are rendered INTO the ruleset text, so the idempotence
|
|
||||||
// check below sees them: a refreshed geoip list changes the text and triggers a
|
|
||||||
// real reload, rather than leaving a stale plan loaded (the D3 trap).
|
|
||||||
untunPlan := a.untunnelablePlanFor(m, opts)
|
|
||||||
ruleset, nftWarnings, err := netplane.RenderNftPlanAt(m, untunPlan, now)
|
|
||||||
if err != nil {
|
|
||||||
// Includes the refusal on an unusable interface name with a closed
|
|
||||||
// kill-switch: the previous ruleset stays loaded and keeps protecting the
|
|
||||||
// LAN while the operator fixes the name.
|
|
||||||
return changed, err
|
|
||||||
}
|
|
||||||
nftCurrent := ruleset == a.lastNft && netplane.TableExists()
|
|
||||||
if !nftCurrent {
|
|
||||||
if err := netplane.ApplyNft(ruleset); err != nil {
|
|
||||||
// The engine may be up, but with no table loaded nothing is diverted into
|
|
||||||
// it — LAN traffic goes straight out the WAN. That is the same silent
|
|
||||||
// fail-open as a dead engine, so it gets the same answer: if there is no
|
|
||||||
// table at all, hold the line rather than leave the LAN exposed.
|
|
||||||
if !netplane.TableExists() {
|
|
||||||
a.holdLocked(m, err)
|
|
||||||
}
|
|
||||||
return changed, err
|
|
||||||
}
|
|
||||||
a.lastNft = ruleset
|
|
||||||
}
|
|
||||||
// ApplyRouting is idempotent by del-then-add, which means it opens a brief
|
|
||||||
// window with NO fwmark rule installed — during it, diverted packets miss the
|
|
||||||
// `local default dev lo` table. Harmless on a real change (the plane is being
|
|
||||||
// rebuilt anyway), but pointless churn on a no-op reconcile, so skip it when
|
|
||||||
// the ruleset is unchanged AND the rule is verifiably still installed.
|
|
||||||
// routeWarnings carries an egress that was BUILT but cannot route (no nexthop on a
|
|
||||||
// non-point-to-point device). It is deliberately part of the status warning set:
|
|
||||||
// such an egress looks applied everywhere in the UI while being unable to reach
|
|
||||||
// anything off its own subnet. On the fast path nothing was rebuilt, so there is
|
|
||||||
// nothing new to report and the previous set stands.
|
|
||||||
var routeWarnings []string
|
|
||||||
if !nftCurrent || !netplane.RoutingPresent(m.Globals) {
|
|
||||||
var rerr error
|
|
||||||
routeWarnings, rerr = netplane.ApplyRoutingWithWarnings(m)
|
|
||||||
if rerr != nil {
|
|
||||||
return changed, rerr
|
|
||||||
}
|
|
||||||
}
|
|
||||||
if err := netplane.ApplySysctl(); err != nil {
|
|
||||||
return changed, err
|
|
||||||
}
|
|
||||||
// Per-diverted-ingress-iface knobs (accept_local/rp_filter): the static
|
|
||||||
// ApplySysctl above cannot know the LAN device names, and without
|
|
||||||
// accept_local=1 on the ingress iface the tproxied packet never reaches the
|
|
||||||
// engine socket — it escapes to the fail-closed forward drop and the LAN goes
|
|
||||||
// dark. Re-asserted on EVERY apply, never fast-pathed: netifd recreating a
|
|
||||||
// bridge (`ifup lan`, a VLAN change) hands back a device with the kernel
|
|
||||||
// defaults, silently un-setting accept_local behind our back. Fail-closed like
|
|
||||||
// the other netplane steps: return without teardown.
|
|
||||||
if err := netplane.ApplyIfaceSysctlsAt(m, now); err != nil {
|
|
||||||
return changed, err
|
|
||||||
}
|
|
||||||
|
|
||||||
// The plane is COMPLETE only here: table + policy routing + sysctls. ApplyNft
|
|
||||||
// already flushed the DNS conntrack when it loaded the ruleset, but that is
|
|
||||||
// one step too early — ApplyRouting is idempotent BY del-then-add, so it opens
|
|
||||||
// a window in which the fwmark rule is momentarily absent, and any DNS flow
|
|
||||||
// that crosses that window is tracked against a plane that is still being
|
|
||||||
// assembled. Flushing once more now that every piece is in place is what makes
|
|
||||||
// "no entry survives the transition" actually true. Only on a real change (the
|
|
||||||
// fast path assembled nothing), best-effort, and cheap: the :53 entry count is
|
|
||||||
// bounded by the number of clients.
|
|
||||||
if !nftCurrent {
|
|
||||||
if n, ferr := netplane.FlushDNSConntrack(); ferr != nil {
|
|
||||||
a.log.Debug("flush DNS conntrack after plane change: ", ferr)
|
|
||||||
} else if n > 0 {
|
|
||||||
a.log.Debug("plane changed: dropped ", n, " stale DNS conntrack entries")
|
|
||||||
}
|
|
||||||
}
|
}
|
||||||
|
|
||||||
// (4) success. Bump the effective-state generation ONLY when something really
|
// (4) success. Bump the effective-state generation ONLY when something really
|
||||||
// moved: a no-op reconcile must not invalidate an armed commit-confirm window
|
// moved: a no-op reconcile must not invalidate an armed commit-confirm window
|
||||||
// (cron reconciles every minute — counting those would cancel every rollback
|
// (cron reconciles every minute — counting those would cancel every rollback
|
||||||
// that commit-confirm exists to guarantee).
|
// that commit-confirm exists to guarantee).
|
||||||
if changed || !nftCurrent {
|
if changed || plane.changed {
|
||||||
a.stateGen.Add(1)
|
a.stateGen.Add(1)
|
||||||
}
|
}
|
||||||
a.setHolding(false)
|
a.setHolding(false)
|
||||||
@@ -544,7 +464,8 @@ func (a *Applier) applyLocked(m *model.Model) (bool, error) {
|
|||||||
// correctly so here: an egress that cannot reach off its own subnet is a configured
|
// correctly so here: an egress that cannot reach off its own subnet is a configured
|
||||||
// path that silently carries nothing, exactly the class of fault that channel exists
|
// path that silently carries nothing, exactly the class of fault that channel exists
|
||||||
// for.
|
// for.
|
||||||
ws := collectWarnings(m.Globals, warnings, append(nftWarnings, routeWarnings...), configWarnings, untunPlan.Notes()...)
|
ws := collectWarnings(m.Globals, warnings,
|
||||||
|
append(plane.nftWarnings, plane.routeWarnings...), configWarnings, plane.planNotes...)
|
||||||
a.setWarnings(ws)
|
a.setWarnings(ws)
|
||||||
// The log only hears about a CHANGE. Status above always carries the full set;
|
// The log only hears about a CHANGE. Status above always carries the full set;
|
||||||
// reprinting it on every no-op reconcile (cron, once a minute, plus every
|
// reprinting it on every no-op reconcile (cron, once a minute, plus every
|
||||||
@@ -559,6 +480,187 @@ func (a *Applier) applyLocked(m *model.Model) (bool, error) {
|
|||||||
return changed, nil
|
return changed, nil
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// planeOutcome is what the netplane half of an apply did, carried back to
|
||||||
|
// applyLocked so the SAME facts can be published whether it succeeded or failed.
|
||||||
|
// Before it existed, everything the netplane stage learned — the warnings it
|
||||||
|
// rendered, the notes the untunnelable plan produced — was thrown away on any
|
||||||
|
// error, which is why a failed apply left the previous config's verdict standing.
|
||||||
|
type planeOutcome struct {
|
||||||
|
// changed is true when the nft ruleset was actually (re)loaded, i.e. the data
|
||||||
|
// plane moved. Distinct from the engine's own `changed`: the engine hash covers
|
||||||
|
// option.Options only, so a purely netplane-visible change hashes identical.
|
||||||
|
changed bool
|
||||||
|
// stage names the netplane step that failed, in operator words, or "" on
|
||||||
|
// success. It is what the failure warning is addressed to.
|
||||||
|
stage string
|
||||||
|
planNotes []string
|
||||||
|
nftWarnings []string
|
||||||
|
routeWarnings []string
|
||||||
|
}
|
||||||
|
|
||||||
|
// engineApply and applyDataPlane are the two heavy halves of applyLocked, behind
|
||||||
|
// package-level seams for exactly one reason: the PUBLISHING behaviour around
|
||||||
|
// them (what Status says after a stage fails) is the thing this file gets wrong
|
||||||
|
// most easily and can otherwise only be tested on a router, with root, a real
|
||||||
|
// sing-box instance and a real nft binary — i.e. never, in the gate. Production
|
||||||
|
// always runs the real methods; a test substitutes a stage that fails and asserts
|
||||||
|
// what the operator is then told. Same seam pattern as applyHoldNft.
|
||||||
|
var (
|
||||||
|
engineApply = func(a *Applier, opts option.Options) (bool, error) { return a.eng.Apply(opts) }
|
||||||
|
applyDataPlane = (*Applier).applyDataPlaneLocked
|
||||||
|
)
|
||||||
|
|
||||||
|
// applyDataPlaneLocked is step (3) of the pipeline: nft ruleset, policy routing
|
||||||
|
// and sysctls, in that order, fail-closed. Caller holds a.mu and has ALREADY
|
||||||
|
// swapped the engine, so every failure here leaves the router in a mixed state —
|
||||||
|
// which is why the outcome is returned even on error.
|
||||||
|
func (a *Applier) applyDataPlaneLocked(m *model.Model, opts option.Options, now time.Time) (planeOutcome, error) {
|
||||||
|
// Render FIRST, then decide whether anything needs re-asserting. Gating the
|
||||||
|
// whole data plane on the engine's `changed` flag alone was wrong: the engine
|
||||||
|
// hash covers option.Options only, so a purely netplane-visible change (the
|
||||||
|
// kill-switch flipping open->closed, DNS force-intercept, a rule gaining an
|
||||||
|
// iface:/zone: source, an inbound device rename) hashed identical, took the
|
||||||
|
// fast-path, and left the OLD ruleset loaded while the panel reported success.
|
||||||
|
// A stale-but-loaded fail-open table is exactly the leak this audit is about.
|
||||||
|
// Resolve WHERE the untunnelable-protocol drop applies before rendering. The
|
||||||
|
// engine is already running by this point (step 2), so its loaded rule-sets can
|
||||||
|
// supply the addresses a geoip list contributes; a stopped engine or a list that
|
||||||
|
// has not downloaded yet yields no plan and the conservative blanket drop.
|
||||||
|
//
|
||||||
|
// The resolved addresses are rendered INTO the ruleset text, so the idempotence
|
||||||
|
// check below sees them: a refreshed geoip list changes the text and triggers a
|
||||||
|
// real reload, rather than leaving a stale plan loaded (the D3 trap).
|
||||||
|
untunPlan := a.untunnelablePlanFor(m, opts)
|
||||||
|
out := planeOutcome{planNotes: untunPlan.Notes()}
|
||||||
|
|
||||||
|
ruleset, nftWarnings, err := netplane.RenderNftPlanAt(m, untunPlan, now)
|
||||||
|
out.nftWarnings = nftWarnings
|
||||||
|
if err != nil {
|
||||||
|
// Includes the refusal on an unusable interface name with a closed
|
||||||
|
// kill-switch: the previous ruleset stays loaded and keeps protecting the
|
||||||
|
// LAN while the operator fixes the name.
|
||||||
|
out.stage = "rendering the nft ruleset"
|
||||||
|
return out, err
|
||||||
|
}
|
||||||
|
nftCurrent := ruleset == a.lastNft && netplane.TableExists()
|
||||||
|
if !nftCurrent {
|
||||||
|
if err := netplane.ApplyNft(ruleset); err != nil {
|
||||||
|
// The engine may be up, but with no table loaded nothing is diverted into
|
||||||
|
// it — LAN traffic goes straight out the WAN. That is the same silent
|
||||||
|
// fail-open as a dead engine, so it gets the same answer: if there is no
|
||||||
|
// table at all, hold the line rather than leave the LAN exposed.
|
||||||
|
if !netplane.TableExists() {
|
||||||
|
a.holdLocked(m, err)
|
||||||
|
}
|
||||||
|
out.stage = "loading the nft ruleset"
|
||||||
|
return out, err
|
||||||
|
}
|
||||||
|
a.lastNft = ruleset
|
||||||
|
out.changed = true
|
||||||
|
}
|
||||||
|
// ApplyRouting is idempotent by del-then-add, which means it opens a brief
|
||||||
|
// window with NO fwmark rule installed — during it, diverted packets miss the
|
||||||
|
// `local default dev lo` table. Harmless on a real change (the plane is being
|
||||||
|
// rebuilt anyway), but pointless churn on a no-op reconcile, so skip it when
|
||||||
|
// the ruleset is unchanged AND the rule is verifiably still installed.
|
||||||
|
// routeWarnings carries an egress that was BUILT but cannot route (no nexthop on a
|
||||||
|
// non-point-to-point device). It is deliberately part of the status warning set:
|
||||||
|
// such an egress looks applied everywhere in the UI while being unable to reach
|
||||||
|
// anything off its own subnet. On the fast path nothing was rebuilt, so there is
|
||||||
|
// nothing new to report and the previous set stands.
|
||||||
|
if !nftCurrent || !netplane.RoutingPresent(m.Globals) {
|
||||||
|
routeWarnings, rerr := netplane.ApplyRoutingWithWarnings(m)
|
||||||
|
out.routeWarnings = routeWarnings
|
||||||
|
if rerr != nil {
|
||||||
|
out.stage = "installing the policy routing"
|
||||||
|
return out, rerr
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if err := netplane.ApplySysctl(); err != nil {
|
||||||
|
out.stage = "setting the kernel sysctls"
|
||||||
|
return out, err
|
||||||
|
}
|
||||||
|
// Per-diverted-ingress-iface knobs (accept_local/rp_filter): the static
|
||||||
|
// ApplySysctl above cannot know the LAN device names, and without
|
||||||
|
// accept_local=1 on the ingress iface the tproxied packet never reaches the
|
||||||
|
// engine socket — it escapes to the fail-closed forward drop and the LAN goes
|
||||||
|
// dark. Re-asserted on EVERY apply, never fast-pathed: netifd recreating a
|
||||||
|
// bridge (`ifup lan`, a VLAN change) hands back a device with the kernel
|
||||||
|
// defaults, silently un-setting accept_local behind our back. Fail-closed like
|
||||||
|
// the other netplane steps: return without teardown.
|
||||||
|
if err := netplane.ApplyIfaceSysctlsAt(m, now); err != nil {
|
||||||
|
out.stage = "setting the per-interface sysctls"
|
||||||
|
return out, err
|
||||||
|
}
|
||||||
|
|
||||||
|
// The plane is COMPLETE only here: table + policy routing + sysctls. ApplyNft
|
||||||
|
// already flushed the DNS conntrack when it loaded the ruleset, but that is
|
||||||
|
// one step too early — ApplyRouting is idempotent BY del-then-add, so it opens
|
||||||
|
// a window in which the fwmark rule is momentarily absent, and any DNS flow
|
||||||
|
// that crosses that window is tracked against a plane that is still being
|
||||||
|
// assembled. Flushing once more now that every piece is in place is what makes
|
||||||
|
// "no entry survives the transition" actually true. Only on a real change (the
|
||||||
|
// fast path assembled nothing), best-effort, and cheap: the :53 entry count is
|
||||||
|
// bounded by the number of clients.
|
||||||
|
if out.changed {
|
||||||
|
if n, ferr := netplane.FlushDNSConntrack(); ferr != nil {
|
||||||
|
a.log.Debug("flush DNS conntrack after plane change: ", ferr)
|
||||||
|
} else if n > 0 {
|
||||||
|
a.log.Debug("plane changed: dropped ", n, " stale DNS conntrack entries")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return out, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// abortAfterSwap publishes an honest status for an apply that got PAST the engine
|
||||||
|
// swap and then failed in the data plane, and returns the cause unchanged.
|
||||||
|
//
|
||||||
|
// The defect it exists for: every netplane failure used to `return changed, err`
|
||||||
|
// before setTraffic/setWarnings, so Status kept serving the verdict and the
|
||||||
|
// findings of the PREVIOUS config while the engine was already running the new
|
||||||
|
// one. Worse, it did not self-heal — the next cron reconcile hashes identical,
|
||||||
|
// fails at the same stage, and returns at the same place, so the stale verdict
|
||||||
|
// stood for as long as the fault did. A green "Protected" over a half-installed
|
||||||
|
// plane is the exact inversion this audit is about: not an error shown when
|
||||||
|
// things are fine, but calm shown when they are not.
|
||||||
|
//
|
||||||
|
// What it publishes:
|
||||||
|
//
|
||||||
|
// - Traffic goes back to UNKNOWN (the zero value). It is tempting to publish
|
||||||
|
// TrafficOf(opts) here, since the ENGINE really is running those options — but
|
||||||
|
// the verdict describes where the LAN's traffic ends up, and that is decided by
|
||||||
|
// the engine and the data plane together. With one of them from this config and
|
||||||
|
// the other from the last one, the honest answer is that we do not know; the
|
||||||
|
// panel renders unknown, and is required never to render it as protected.
|
||||||
|
// - The warning set is replaced by THIS config's warnings, with a critical entry
|
||||||
|
// naming the stage that failed at the front. The operator gets the news about
|
||||||
|
// the config that is actually loaded, plus the fact that it is only half loaded.
|
||||||
|
//
|
||||||
|
// Caller holds a.mu.
|
||||||
|
func (a *Applier) abortAfterSwap(m *model.Model, plane planeOutcome, generateWarnings []string, configWarnings []model.Warning, cause error) error {
|
||||||
|
a.setTraffic(generate.Traffic{})
|
||||||
|
|
||||||
|
stage := plane.stage
|
||||||
|
if stage == "" {
|
||||||
|
stage = "installing the data plane"
|
||||||
|
}
|
||||||
|
ws := gatherWarnings(m.Globals, generateWarnings,
|
||||||
|
append(plane.nftWarnings, plane.routeWarnings...), configWarnings, plane.planNotes...)
|
||||||
|
ws = finalizeWarnings(append([]Warning{{
|
||||||
|
Severity: SeverityCritical,
|
||||||
|
Section: "netplane",
|
||||||
|
Name: stage,
|
||||||
|
Message: fmt.Sprintf("the engine was switched to this configuration but the data plane could NOT be "+
|
||||||
|
"completed — %s failed: %v. What the kernel holds is part of this configuration and part of the "+
|
||||||
|
"previous one, so where your traffic goes is UNKNOWN: treat this router as unprotected until a "+
|
||||||
|
"reconcile succeeds. It is retried every minute; if it keeps failing, fix the cause or roll back.",
|
||||||
|
stage, cause),
|
||||||
|
}}, ws...))
|
||||||
|
a.setWarnings(ws)
|
||||||
|
a.logWarningsIfChanged(ws)
|
||||||
|
return cause
|
||||||
|
}
|
||||||
|
|
||||||
// holdLocked installs the fail-closed HOLDING PLANE when the engine is not
|
// holdLocked installs the fail-closed HOLDING PLANE when the engine is not
|
||||||
// running and the kill-switch is closed. Caller holds a.mu.
|
// running and the kill-switch is closed. Caller holds a.mu.
|
||||||
//
|
//
|
||||||
@@ -940,12 +1042,28 @@ func (a *Applier) Rollback() error {
|
|||||||
defer release()
|
defer release()
|
||||||
a.mu.Lock()
|
a.mu.Lock()
|
||||||
defer a.mu.Unlock()
|
defer a.mu.Unlock()
|
||||||
if err := a.eng.Rollback(); err != nil {
|
m, err := rollbackEngineAndPlane(a)
|
||||||
|
if err != nil {
|
||||||
return err
|
return err
|
||||||
}
|
}
|
||||||
|
a.publishEngineRollback(m)
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// rollbackEngineAndPlane is the ACTION half of the no-snapshot rollback: drive the
|
||||||
|
// engine back to its predecessor config and re-assert the data plane from current
|
||||||
|
// UCI. It returns the model the plane was rebuilt from. Caller holds a.mu.
|
||||||
|
//
|
||||||
|
// It is a variable for the same reason as engineApply/applyDataPlane: what this
|
||||||
|
// rollback PUBLISHES afterwards is the part that was wrong, and it cannot be
|
||||||
|
// exercised at all without two real engine generations and a real nft binary.
|
||||||
|
var rollbackEngineAndPlane = func(a *Applier) (*model.Model, error) {
|
||||||
|
if err := a.eng.Rollback(); err != nil {
|
||||||
|
return nil, err
|
||||||
|
}
|
||||||
m, err := model.ReadUCI()
|
m, err := model.ReadUCI()
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return err
|
return nil, err
|
||||||
}
|
}
|
||||||
// One clock for the whole re-assert, same as applyLocked: the ruleset's
|
// One clock for the whole re-assert, same as applyLocked: the ruleset's
|
||||||
// divert set and the iface sysctls below must agree on the profile-effective
|
// divert set and the iface sysctls below must agree on the profile-effective
|
||||||
@@ -957,14 +1075,14 @@ func (a *Applier) Rollback() error {
|
|||||||
a.log.Warn("netplane: ", w)
|
a.log.Warn("netplane: ", w)
|
||||||
}
|
}
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return err
|
return nil, err
|
||||||
}
|
}
|
||||||
if err := netplane.ApplyNft(ruleset); err != nil {
|
if err := netplane.ApplyNft(ruleset); err != nil {
|
||||||
return err
|
return nil, err
|
||||||
}
|
}
|
||||||
a.lastNft = ruleset
|
a.lastNft = ruleset
|
||||||
if err := netplane.ApplyRouting(m); err != nil {
|
if err := netplane.ApplyRouting(m); err != nil {
|
||||||
return err
|
return nil, err
|
||||||
}
|
}
|
||||||
// The sysctl half must be re-asserted here too. It used to be missing: a
|
// The sysctl half must be re-asserted here too. It used to be missing: a
|
||||||
// rollback that changes the set of diverted ingress devices (a different
|
// rollback that changes the set of diverted ingress devices (a different
|
||||||
@@ -973,9 +1091,60 @@ func (a *Applier) Rollback() error {
|
|||||||
// then never reaches the engine socket and that network goes dark after a
|
// then never reaches the engine socket and that network goes dark after a
|
||||||
// rollback, which is precisely when the operator can least afford it.
|
// rollback, which is precisely when the operator can least afford it.
|
||||||
if err := netplane.ApplySysctl(); err != nil {
|
if err := netplane.ApplySysctl(); err != nil {
|
||||||
return err
|
return nil, err
|
||||||
}
|
}
|
||||||
return netplane.ApplyIfaceSysctlsAt(m, now)
|
if err := netplane.ApplyIfaceSysctlsAt(m, now); err != nil {
|
||||||
|
return nil, err
|
||||||
|
}
|
||||||
|
return m, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// publishEngineRollback makes Status describe the router the no-snapshot rollback
|
||||||
|
// just produced, instead of the one it rolled away FROM. Caller holds a.mu.
|
||||||
|
//
|
||||||
|
// The defect: this path touched none of the publishers. Apply a tunnel config,
|
||||||
|
// Confirm it (which consumes the snapshot), then roll back later — the engine goes
|
||||||
|
// to its predecessor, which may well be the `default -> direct` config, and the
|
||||||
|
// panel keeps showing the tunnel verdict and the tunnel config's warnings, in
|
||||||
|
// green, indefinitely. The whole LAN is on the plain WAN with its real address and
|
||||||
|
// the UI says Protected. Nothing else corrects it: the verdict is only ever
|
||||||
|
// rewritten by a successful apply, and a rollback is not one.
|
||||||
|
//
|
||||||
|
// The verdict published is UNKNOWN, not a computed one, and that is the honest
|
||||||
|
// answer rather than a lazy one: engine.Rollback re-applies option.Options that
|
||||||
|
// this process no longer holds (the engine keeps them, apply does not), so there
|
||||||
|
// is nothing here to run generate.TrafficOf over. Guessing from current UCI would
|
||||||
|
// be worse than saying nothing — UCI is the config we rolled AWAY from. Unknown is
|
||||||
|
// rendered as unknown by the panel and never as protected, and the next reconcile
|
||||||
|
// (cron, within a minute) replaces it with the truth.
|
||||||
|
func (a *Applier) publishEngineRollback(m *model.Model) {
|
||||||
|
a.setTraffic(generate.Traffic{})
|
||||||
|
a.setHolding(false) // a full plane was just loaded; whatever hold there was is over
|
||||||
|
// lastGood is the teardown/rollback target, and the data plane was just built
|
||||||
|
// from m — so m is what a later Teardown must know about to remove the right
|
||||||
|
// marks and routing tables.
|
||||||
|
a.lastGood = m
|
||||||
|
// The observatory's reachability plan was built from the config we rolled away
|
||||||
|
// from: left running it probes outbounds that may no longer exist and files the
|
||||||
|
// results against tags the running box does not have. There is no plan to
|
||||||
|
// replace it with (see above), so stop probing until the next apply installs one.
|
||||||
|
if a.eng != nil {
|
||||||
|
a.eng.StopObservatory()
|
||||||
|
}
|
||||||
|
ws := finalizeWarnings([]Warning{{
|
||||||
|
Severity: SeverityCritical,
|
||||||
|
Section: "engine",
|
||||||
|
Name: "rollback",
|
||||||
|
Message: "the engine was rolled back to the configuration that ran before the current one. " +
|
||||||
|
"That configuration is not the one on disk, so where your traffic goes and what was left " +
|
||||||
|
"un-applied are both UNKNOWN until the next reconcile (within a minute) re-applies the " +
|
||||||
|
"saved config and reports on it. Do not read this router as protected in the meantime.",
|
||||||
|
}})
|
||||||
|
a.setWarnings(ws)
|
||||||
|
a.logWarningsIfChanged(ws)
|
||||||
|
// The plane moved, so an armed commit-confirm watcher must see a changed
|
||||||
|
// generation and stand down rather than clobber what we just restored.
|
||||||
|
a.stateGen.Add(1)
|
||||||
}
|
}
|
||||||
|
|
||||||
// canRollback reports whether Rollback would actually revert something: an armed
|
// canRollback reports whether Rollback would actually revert something: an armed
|
||||||
@@ -1019,11 +1188,30 @@ func ActiveFlagPresent() bool {
|
|||||||
|
|
||||||
// Status is the read-side snapshot printed by `shaterd status` as JSON.
|
// Status is the read-side snapshot printed by `shaterd status` as JSON.
|
||||||
//
|
//
|
||||||
// running the daemon process is up (a live socket reply => true; the offline
|
// running SHATER IS RUNNING: the daemon answered AND its engine has a started
|
||||||
// stub reports false). Whether the ENGINE is intercepting is carried by
|
// sing-box instance carrying a config. false therefore covers every way
|
||||||
// active/table/hash, not by running — a daemon can be up but inert.
|
// of not proxying — daemon down, daemon up with a dead engine, disabled,
|
||||||
|
// torn down — and the fields below say which.
|
||||||
|
//
|
||||||
|
// It used to be the literal `true`, on the reasoning that Status() is
|
||||||
|
// only ever called from inside the live daemon. That was true and it was
|
||||||
|
// useless: a constant cannot report anything, and the panel built its
|
||||||
|
// headline on `running && active`, so the "not running" branch was
|
||||||
|
// physically unreachable and an engine that never started showed green.
|
||||||
|
// A field whose only possible value is the reassuring one is worse than
|
||||||
|
// no field: it is a promise the code cannot break.
|
||||||
|
//
|
||||||
|
// "Is the daemon process alive?" is a different question and is answered
|
||||||
|
// by whether the status call returned at all (plus uptime_seconds, which
|
||||||
|
// only a live daemon can produce).
|
||||||
// enabled globals.enabled in UCI.
|
// enabled globals.enabled in UCI.
|
||||||
// active ACTIVE_FLAG present (a successful enabled apply raised it).
|
// active ACTIVE_FLAG present. This is the "the service is meant to be running"
|
||||||
|
// latch that gates hotplug and cron, NOT a health signal: it is raised by
|
||||||
|
// a successful enabled apply and cleared only by teardown, so it stays up
|
||||||
|
// while the engine is down and the fail-closed holding plane is blocking
|
||||||
|
// the LAN — deliberately, because clearing it would switch off the very
|
||||||
|
// cron reconcile that brings the engine back. Never render it as "we are
|
||||||
|
// proxying"; that is what running/plane/traffic are for.
|
||||||
// table the `inet shater` nft table is loaded.
|
// table the `inet shater` nft table is loaded.
|
||||||
// hash the running engine's config hash ("" when the engine is not started).
|
// hash the running engine's config hash ("" when the engine is not started).
|
||||||
// kill_switch globals.kill_switch in UCI (closed = fail-closed, open = leaky).
|
// kill_switch globals.kill_switch in UCI (closed = fail-closed, open = leaky).
|
||||||
@@ -1044,9 +1232,14 @@ type Status struct {
|
|||||||
PanelPort int `json:"panel_port"`
|
PanelPort int `json:"panel_port"`
|
||||||
CanRollback bool `json:"can_rollback"`
|
CanRollback bool `json:"can_rollback"`
|
||||||
|
|
||||||
// EngineRunning is whether a sing-box instance is actually started. It is the
|
// EngineRunning is whether a sing-box instance is actually started.
|
||||||
// honest answer to "are we proxying?", which running/active/table each only
|
//
|
||||||
// approximate.
|
// It was added as the honest field to stand beside a `running` that was hard-wired
|
||||||
|
// true, and no consumer ever read it. Now that running carries the same fact it is
|
||||||
|
// kept as its explicit, unambiguous name — the two are equal by construction from
|
||||||
|
// the daemon — because it is already in the published API and reading
|
||||||
|
// `engine_running` in a client is self-documenting where `running` needs this
|
||||||
|
// comment.
|
||||||
EngineRunning bool `json:"engine_running"`
|
EngineRunning bool `json:"engine_running"`
|
||||||
|
|
||||||
// Plane describes what is loaded in the kernel RIGHT NOW:
|
// Plane describes what is loaded in the kernel RIGHT NOW:
|
||||||
@@ -1141,18 +1334,23 @@ func processUptime(now time.Time) (startedUnix, uptimeSeconds int64) {
|
|||||||
return now.Unix() - uptimeSeconds, uptimeSeconds
|
return now.Unix() - uptimeSeconds, uptimeSeconds
|
||||||
}
|
}
|
||||||
|
|
||||||
// Status returns the live status from this daemon's engine + kernel state. It is
|
// Status returns the live status from this daemon's engine + kernel state.
|
||||||
// only ever called from within the running daemon, so running=true; engine state
|
//
|
||||||
// is reflected by Active/Table/Hash (an inert daemon reports running=true but
|
// It is only ever called from within the running daemon, which is exactly why
|
||||||
// active=false/table=false/hash="").
|
// `running` is read off the engine rather than set to true: from in here the
|
||||||
|
// daemon's own liveness is a tautology, and the only thing left worth reporting
|
||||||
|
// under that name is whether shater is carrying any traffic. A daemon that is up
|
||||||
|
// with a dead engine reports running=false, plane="hold"/"none" and an unknown
|
||||||
|
// traffic verdict — which is the state this field exists to make expressible.
|
||||||
func (a *Applier) Status() Status {
|
func (a *Applier) Status() Status {
|
||||||
|
engineUp := a.eng != nil && a.eng.Running()
|
||||||
s := Status{
|
s := Status{
|
||||||
Running: true,
|
Running: engineUp,
|
||||||
Active: ActiveFlagPresent(),
|
Active: ActiveFlagPresent(),
|
||||||
Table: netplane.TableExists(),
|
Table: netplane.TableExists(),
|
||||||
Hash: a.eng.Hash(),
|
Hash: a.eng.Hash(),
|
||||||
CanRollback: a.canRollback(),
|
CanRollback: a.canRollback(),
|
||||||
EngineRunning: a.eng.Running(),
|
EngineRunning: engineUp,
|
||||||
Traffic: a.Traffic(),
|
Traffic: a.Traffic(),
|
||||||
Warnings: a.Warnings(),
|
Warnings: a.Warnings(),
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -0,0 +1,178 @@
|
|||||||
|
package apply
|
||||||
|
|
||||||
|
// Regression tests for the three ways this package used to report calm over a
|
||||||
|
// router that was not doing what its config said. Each of them is the INVERTED
|
||||||
|
// failure — not an error shown when things are fine, but green shown when they
|
||||||
|
// are not — which is the only kind that gets someone hurt.
|
||||||
|
|
||||||
|
import (
|
||||||
|
"errors"
|
||||||
|
"strings"
|
||||||
|
"testing"
|
||||||
|
"time"
|
||||||
|
|
||||||
|
"github.com/sagernet/sing-box/option"
|
||||||
|
"github.com/sagernet/sing-box/shater/engine"
|
||||||
|
"github.com/sagernet/sing-box/shater/generate"
|
||||||
|
"github.com/sagernet/sing-box/shater/model"
|
||||||
|
)
|
||||||
|
|
||||||
|
// TestStatusRunningReportsTheEngine pins the contract the panel headline is built
|
||||||
|
// on.
|
||||||
|
//
|
||||||
|
// `running` used to be the literal `true` in the only code path that produces it,
|
||||||
|
// so `running && active` — what the panel reads — could not go false however dead
|
||||||
|
// the engine was, and the "not running" branch was unreachable code. A status
|
||||||
|
// field that can only ever hold the reassuring value is not a weak signal, it is
|
||||||
|
// an unfalsifiable claim.
|
||||||
|
func TestStatusRunningReportsTheEngine(t *testing.T) {
|
||||||
|
a := New(engine.New(), nil)
|
||||||
|
|
||||||
|
// A fresh applier's engine has never started: nothing is being proxied, and
|
||||||
|
// the status must be able to say so.
|
||||||
|
s := a.Status()
|
||||||
|
if s.Running {
|
||||||
|
t.Errorf("Status().Running = true with a stopped engine — the field is a constant again")
|
||||||
|
}
|
||||||
|
if s.Running != s.EngineRunning {
|
||||||
|
t.Errorf("running (%v) and engine_running (%v) must agree: they are the same fact",
|
||||||
|
s.Running, s.EngineRunning)
|
||||||
|
}
|
||||||
|
|
||||||
|
// And the hold state — engine down, LAN blocked — must not read as running
|
||||||
|
// either. This is the three-green-lamps case: holding does not clear the
|
||||||
|
// ACTIVE flag (cron needs it to keep retrying), so `active` alone cannot say it.
|
||||||
|
g := model.DefaultGlobals()
|
||||||
|
g.KillSwitch = "open" // the early-return branch of holdLocked
|
||||||
|
a.mu.Lock()
|
||||||
|
a.holdLocked(&model.Model{Globals: g}, errors.New("engine start failed"))
|
||||||
|
a.mu.Unlock()
|
||||||
|
if a.Status().Running {
|
||||||
|
t.Errorf("Status().Running = true while the engine is down and the plane is held")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestApplyFailingAfterEngineSwapDropsTheOldVerdict is the defect-3 regression.
|
||||||
|
//
|
||||||
|
// Everything after the engine swap used to `return changed, err` before
|
||||||
|
// setTraffic/setWarnings, so a netplane failure left Status serving the VERDICT
|
||||||
|
// and the FINDINGS of the configuration that no longer runs. It did not self-heal
|
||||||
|
// either: the next cron reconcile hashes identical, fails at the same stage and
|
||||||
|
// returns at the same place, so the stale green stood for as long as the fault.
|
||||||
|
func TestApplyFailingAfterEngineSwapDropsTheOldVerdict(t *testing.T) {
|
||||||
|
a := New(engine.New(), nil)
|
||||||
|
|
||||||
|
// What the previous, fully successful apply published.
|
||||||
|
a.setTraffic(generate.Traffic{Verdict: generate.VerdictTunnel, Default: "auto", TunnelRules: 3})
|
||||||
|
a.setWarnings([]Warning{{
|
||||||
|
Severity: SeverityCritical, Section: "ruleset", Name: "stale",
|
||||||
|
Message: "a finding of the configuration that is no longer running",
|
||||||
|
}})
|
||||||
|
|
||||||
|
// The engine swap succeeds; the data plane does not.
|
||||||
|
restore := stubApplyStages(t,
|
||||||
|
func(a *Applier, opts option.Options) (bool, error) { return true, nil },
|
||||||
|
func(a *Applier, m *model.Model, opts option.Options, now time.Time) (planeOutcome, error) {
|
||||||
|
return planeOutcome{stage: "loading the nft ruleset"}, errors.New("nft: permission denied")
|
||||||
|
})
|
||||||
|
defer restore()
|
||||||
|
|
||||||
|
a.mu.Lock()
|
||||||
|
_, err := a.applyLocked(holdModel("closed"))
|
||||||
|
a.mu.Unlock()
|
||||||
|
if err == nil {
|
||||||
|
t.Fatalf("applyLocked must surface the netplane failure")
|
||||||
|
}
|
||||||
|
|
||||||
|
s := a.Status()
|
||||||
|
if s.Traffic.Verdict != "" {
|
||||||
|
t.Errorf("Traffic.Verdict = %q after a half-installed plane, want \"\" (unknown): "+
|
||||||
|
"the engine runs the new config and the kernel does not, so nobody knows where traffic goes",
|
||||||
|
s.Traffic.Verdict)
|
||||||
|
}
|
||||||
|
var sawAbort bool
|
||||||
|
for _, w := range s.Warnings {
|
||||||
|
if strings.Contains(w.Message, "a finding of the configuration that is no longer running") {
|
||||||
|
t.Errorf("the previous config's findings are still published: %+v", w)
|
||||||
|
}
|
||||||
|
if w.Section == "netplane" && strings.Contains(w.Message, "could NOT be completed") {
|
||||||
|
sawAbort = true
|
||||||
|
if w.Severity != SeverityCritical {
|
||||||
|
t.Errorf("an incomplete data plane is critical, got %q", w.Severity)
|
||||||
|
}
|
||||||
|
if !strings.Contains(w.Message, "loading the nft ruleset") {
|
||||||
|
t.Errorf("the warning must name the stage that failed: %q", w.Message)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if !sawAbort {
|
||||||
|
t.Errorf("no warning says the data plane is incomplete; warnings = %+v", s.Warnings)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestRollbackWithoutSnapshotRepublishes is the defect-2 regression.
|
||||||
|
//
|
||||||
|
// Apply a tunnel config, Confirm it (which consumes the commit-confirm snapshot),
|
||||||
|
// then roll back later. The no-snapshot branch drives engine.Rollback and rebuilds
|
||||||
|
// the plane — and used to touch none of the publishers, so the panel kept showing
|
||||||
|
// the tunnel verdict and the tunnel config's warnings in green while the engine
|
||||||
|
// had gone back to a predecessor that may route `default -> direct`. The entire
|
||||||
|
// LAN on the plain WAN, under a green "Protected", indefinitely.
|
||||||
|
func TestRollbackWithoutSnapshotRepublishes(t *testing.T) {
|
||||||
|
a := New(engine.New(), nil)
|
||||||
|
a.setTraffic(generate.Traffic{Verdict: generate.VerdictTunnel, Default: "auto", TunnelRules: 2})
|
||||||
|
a.setWarnings([]Warning{{
|
||||||
|
Severity: SeverityWarning, Section: "chain", Name: "hop",
|
||||||
|
Message: "a finding of the configuration we are rolling away from",
|
||||||
|
}})
|
||||||
|
|
||||||
|
m := holdModel("closed")
|
||||||
|
orig := rollbackEngineAndPlane
|
||||||
|
rollbackEngineAndPlane = func(*Applier) (*model.Model, error) { return m, nil }
|
||||||
|
defer func() { rollbackEngineAndPlane = orig }()
|
||||||
|
|
||||||
|
before := a.stateGen.Load()
|
||||||
|
if err := a.Rollback(); err != nil {
|
||||||
|
t.Fatalf("Rollback: %v", err)
|
||||||
|
}
|
||||||
|
|
||||||
|
s := a.Status()
|
||||||
|
if s.Traffic.Verdict != "" {
|
||||||
|
t.Errorf("Traffic.Verdict = %q after an engine rollback, want \"\" (unknown): the running "+
|
||||||
|
"config is one this process cannot describe", s.Traffic.Verdict)
|
||||||
|
}
|
||||||
|
var sawRollback bool
|
||||||
|
for _, w := range s.Warnings {
|
||||||
|
if strings.Contains(w.Message, "rolling away from") {
|
||||||
|
t.Errorf("the pre-rollback findings are still published: %+v", w)
|
||||||
|
}
|
||||||
|
if w.Section == "engine" && w.Name == "rollback" {
|
||||||
|
sawRollback = true
|
||||||
|
if w.Severity != SeverityCritical {
|
||||||
|
t.Errorf("an undescribable running config is critical, got %q", w.Severity)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if !sawRollback {
|
||||||
|
t.Errorf("nothing says the router is running a rolled-back config; warnings = %+v", s.Warnings)
|
||||||
|
}
|
||||||
|
if a.LastGood() != m {
|
||||||
|
t.Errorf("last-good must become the model the data plane was rebuilt from")
|
||||||
|
}
|
||||||
|
if a.stateGen.Load() == before {
|
||||||
|
t.Errorf("the plane moved but stateGen did not: an armed commit-confirm watcher " +
|
||||||
|
"would clobber the config we just restored")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// stubApplyStages replaces the two heavy halves of applyLocked for the duration of
|
||||||
|
// a test and returns the restore func.
|
||||||
|
func stubApplyStages(t *testing.T,
|
||||||
|
eng func(*Applier, option.Options) (bool, error),
|
||||||
|
plane func(*Applier, *model.Model, option.Options, time.Time) (planeOutcome, error),
|
||||||
|
) func() {
|
||||||
|
t.Helper()
|
||||||
|
origEngine, origPlane := engineApply, applyDataPlane
|
||||||
|
engineApply, applyDataPlane = eng, plane
|
||||||
|
return func() { engineApply, applyDataPlane = origEngine, origPlane }
|
||||||
|
}
|
||||||
@@ -1,13 +1,19 @@
|
|||||||
package apply
|
package apply
|
||||||
|
|
||||||
// Deciding WHERE the untunnelable-protocol drop applies.
|
// Deciding WHERE the untunnelable-protocol drop applies, for the ONE policy that
|
||||||
|
// asks: `icmp`.
|
||||||
//
|
//
|
||||||
// The data plane cannot tunnel anything that is not TCP or UDP (kernel TPROXY
|
// The data plane cannot tunnel anything that is not TCP or UDP (kernel TPROXY
|
||||||
// needs a socket; the proxy protocols carry TCP streams and UDP datagrams). The
|
// needs a socket; the proxy protocols carry TCP streams and UDP datagrams). Where
|
||||||
// drop that follows from that is only justified for destinations the routing
|
// a rule routes direct, the client's real address already reaches that
|
||||||
// rules actually send THROUGH the tunnel: where a rule routes direct, the
|
// destination over TCP, so dropping its ICMP hides nothing and merely breaks
|
||||||
// client's real address already reaches that destination over TCP, so dropping
|
// diagnostics.
|
||||||
// its ICMP hides nothing and merely breaks diagnostics.
|
//
|
||||||
|
// That reasoning used to govern `block` as well, which made `block` identical to
|
||||||
|
// `direct` under the ordinary "tunnel the blocked list, send the rest direct"
|
||||||
|
// configuration — see the essay in netplane/untunnelable.go. `block` now drops
|
||||||
|
// unconditionally and `direct` allows unconditionally; neither reads this file,
|
||||||
|
// and untunnelablePlanFor no longer builds a plan for them at all.
|
||||||
//
|
//
|
||||||
// This file computes the difference, by walking the FULLY RESOLVED routing rules
|
// This file computes the difference, by walking the FULLY RESOLVED routing rules
|
||||||
// that generate produced. Using generate's output rather than the raw model is
|
// that generate produced. Using generate's output rather than the raw model is
|
||||||
@@ -239,9 +245,31 @@ func buildUntunnelablePlan(opts option.Options, lookup ruleSetCIDRs) *netplane.U
|
|||||||
}
|
}
|
||||||
mt.AnyDst = !hasDstMatcher
|
mt.AnyDst = !hasDstMatcher
|
||||||
mt.Dst4, mt.Dst6 = netplane.PrefixStrings(dst)
|
mt.Dst4, mt.Dst6 = netplane.PrefixStrings(dst)
|
||||||
|
if !mt.AnyDst && len(mt.Dst4) == 0 && len(mt.Dst6) == 0 {
|
||||||
|
// The rule names addresses and not one of them survived into a form the
|
||||||
|
// data plane can express — unparseable, or IPv4-mapped IPv6, which
|
||||||
|
// PrefixStrings drops because nftables has no set type for it. The step
|
||||||
|
// would render no line at all, so the walk would silently step OVER a
|
||||||
|
// rule that CAN claim this traffic and let a later rule decide in its
|
||||||
|
// place. That is the over-permissive mistake this file exists to avoid,
|
||||||
|
// so it is undecidable rather than skippable.
|
||||||
|
plan.Warnings = append(plan.Warnings,
|
||||||
|
"a routing rule's addresses cannot be expressed by the firewall, so ping/IPTV/"+
|
||||||
|
"VPN-passthrough traffic is blocked from that rule onwards")
|
||||||
|
return plan
|
||||||
|
}
|
||||||
|
|
||||||
// Source predicate, when the rule is scoped to particular clients.
|
// Source predicate, when the rule is scoped to particular clients. Same
|
||||||
|
// reasoning as above, and here the consequence is worse than a skipped step:
|
||||||
|
// an empty source pair reads as "any source", so an ALLOW scoped to three lab
|
||||||
|
// machines would render as an allow for the whole LAN.
|
||||||
mt.Src4, mt.Src6 = netplane.PrefixStrings(parsePrefixes(d.SourceIPCIDR))
|
mt.Src4, mt.Src6 = netplane.PrefixStrings(parsePrefixes(d.SourceIPCIDR))
|
||||||
|
if len(d.SourceIPCIDR) > 0 && len(mt.Src4) == 0 && len(mt.Src6) == 0 {
|
||||||
|
plan.Warnings = append(plan.Warnings,
|
||||||
|
"a routing rule's source addresses cannot be expressed by the firewall, so ping/IPTV/"+
|
||||||
|
"VPN-passthrough traffic is blocked from that rule onwards")
|
||||||
|
return plan
|
||||||
|
}
|
||||||
|
|
||||||
plan.Matches = append(plan.Matches, mt)
|
plan.Matches = append(plan.Matches, mt)
|
||||||
|
|
||||||
@@ -419,10 +447,17 @@ func parsePrefixes(in []string) []netip.Prefix {
|
|||||||
// rule-set addresses against the RUNNING engine. A nil engine (or a stopped one)
|
// rule-set addresses against the RUNNING engine. A nil engine (or a stopped one)
|
||||||
// yields lookups that always report "not loaded", so the plan degrades to the
|
// yields lookups that always report "not loaded", so the plan degrades to the
|
||||||
// conservative blanket drop on its own.
|
// conservative blanket drop on its own.
|
||||||
|
//
|
||||||
|
// ONLY `icmp` has a use for it. The other two rungs are unconditional — `direct`
|
||||||
|
// allows every untunnelable protocol wherever it was going, `block` allows none —
|
||||||
|
// and netplane.untunnelableRules refuses to consult the plan for either. Building
|
||||||
|
// one anyway would push the routing rules' whole address space into kernel memory
|
||||||
|
// (geoip-us alone is ~159 000 prefixes, ~20 MB) to answer a question nothing asks;
|
||||||
|
// worse, it would render the sets into the ruleset text, so a geoip refresh would
|
||||||
|
// churn the data plane for a policy that cannot use it. Since `block` is the
|
||||||
|
// DEFAULT, this is the stock install's path.
|
||||||
func (a *Applier) untunnelablePlanFor(m *model.Model, opts option.Options) *netplane.UntunnelablePlan {
|
func (a *Applier) untunnelablePlanFor(m *model.Model, opts option.Options) *netplane.UntunnelablePlan {
|
||||||
if netplane.EffectiveUntunnelable(m.Globals) == netplane.UntunnelableDirect {
|
if netplane.EffectiveUntunnelable(m.Globals) != netplane.UntunnelableICMP {
|
||||||
// Everything untunnelable is allowed regardless of destination; computing
|
|
||||||
// (and loading into the kernel) thousands of prefixes would change nothing.
|
|
||||||
return nil
|
return nil
|
||||||
}
|
}
|
||||||
lookup := func(tag string) ([]netip.Prefix, bool) { return nil, false }
|
lookup := func(tag string) ([]netip.Prefix, bool) { return nil, false }
|
||||||
|
|||||||
@@ -63,11 +63,31 @@ func renderPlan(t *testing.T, m *model.Model, plan *netplane.UntunnelablePlan) s
|
|||||||
return rs
|
return rs
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// icmpPolicy puts the model on the ONE policy that consults the destination plan.
|
||||||
|
//
|
||||||
|
// Every assertion about the WALK's rendered form has to be made under it, and
|
||||||
|
// that is a change of contract rather than test bookkeeping: `block` now drops
|
||||||
|
// every untunnelable protocol unconditionally and `direct` accepts every one of
|
||||||
|
// them unconditionally, so netplane.untunnelableRules refuses to read the plan for
|
||||||
|
// either. A render-level test left on the default policy would be asserting
|
||||||
|
// against a section the renderer no longer writes — which is exactly how the
|
||||||
|
// defect survived: it was `block` rendering `direct`, under a test that read the
|
||||||
|
// resulting accept as the feature working.
|
||||||
|
func icmpPolicy(m *model.Model) *model.Model {
|
||||||
|
m.Globals.Untunnelable = netplane.UntunnelableICMP
|
||||||
|
return m
|
||||||
|
}
|
||||||
|
|
||||||
// TestOnlyPinnedAddressIsTunnelled is the first scenario from the brief: a rule
|
// TestOnlyPinnedAddressIsTunnelled is the first scenario from the brief: a rule
|
||||||
// sends ONLY 8.8.8.8/32 through the tunnel and everything else goes direct, so
|
// sends ONLY 8.8.8.8/32 through the tunnel and everything else goes direct, so
|
||||||
// only 8.8.8.8 may be un-pingable and the rest of the internet must answer.
|
// only 8.8.8.8 may be un-pingable and the rest of the internet must answer.
|
||||||
|
//
|
||||||
|
// The PLAN half is unchanged — the walk still resolves the pinned address to a
|
||||||
|
// deny and everything else to the routing default. Only the rendering moved to
|
||||||
|
// `icmp` (see icmpPolicy). What `block` renders for this same configuration is
|
||||||
|
// TestBlockDropsEvenWhenEverythingRoutesDirect, and it is nothing at all.
|
||||||
func TestOnlyPinnedAddressIsTunnelled(t *testing.T) {
|
func TestOnlyPinnedAddressIsTunnelled(t *testing.T) {
|
||||||
m := tunnelModel()
|
m := icmpPolicy(tunnelModel())
|
||||||
m.Rulesets = []model.Ruleset{pinnedIPSet("pin", "8.8.8.8/32")}
|
m.Rulesets = []model.Ruleset{pinnedIPSet("pin", "8.8.8.8/32")}
|
||||||
m.Rules = []model.Rule{
|
m.Rules = []model.Rule{
|
||||||
{Name: "pin", Enabled: true, Order: 10, DstRuleset: []string{"pin"}, Target: "group:auto"},
|
{Name: "pin", Enabled: true, Order: 10, DstRuleset: []string{"pin"}, Target: "group:auto"},
|
||||||
@@ -106,10 +126,128 @@ func TestOnlyPinnedAddressIsTunnelled(t *testing.T) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// TestBlockDropsEvenWhenEverythingRoutesDirect is the defect, in the exact
|
||||||
|
// configuration that makes it bite.
|
||||||
|
//
|
||||||
|
// "Tunnel the pinned address, send the rest direct" leaves the ROUTING DEFAULT
|
||||||
|
// direct, so the plan's DefaultAllow is true — and `block` used to walk the plan
|
||||||
|
// like `icmp` and inherit that default as a blanket
|
||||||
|
// `meta l4proto != { tcp, udp } accept`. ICMP, ESP, AH, GRE, IGMP and SCTP all
|
||||||
|
// left with the client's real address. That includes a client-run IPsec or PPTP
|
||||||
|
// tunnel: a standing second tunnel beside ours, carrying arbitrary traffic under a
|
||||||
|
// peer we neither route nor filter, for as long as it stays up — which is the very
|
||||||
|
// thing the middle rung exists to keep out of "I just want ping". Meanwhile the
|
||||||
|
// panel promised "Nothing leaves except through the tunnel."
|
||||||
|
//
|
||||||
|
// `block` is the DEFAULT policy, so this was the stock install.
|
||||||
|
//
|
||||||
|
// RED BEFORE THE FIX: with the old untunnelableRules the rendered forward chain
|
||||||
|
// carries `... meta l4proto != { tcp, udp } accept`, emitted from plan.DefaultAllow.
|
||||||
|
func TestBlockDropsEvenWhenEverythingRoutesDirect(t *testing.T) {
|
||||||
|
m := tunnelModel()
|
||||||
|
m.Globals.IPv6 = true
|
||||||
|
m.Globals.Untunnelable = netplane.UntunnelableBlock // the default, spelled out: it is the subject
|
||||||
|
m.Rulesets = []model.Ruleset{pinnedIPSet("pin", "8.8.8.8/32")}
|
||||||
|
m.Rules = []model.Rule{
|
||||||
|
{Name: "pin", Enabled: true, Order: 10, DstRuleset: []string{"pin"}, Target: "group:auto"},
|
||||||
|
{Name: "rest", Enabled: true, Order: 99, Target: "direct"},
|
||||||
|
}
|
||||||
|
plan := planFor(t, m, map[string][]string{"rs-pin": {"8.8.8.8/32"}})
|
||||||
|
if !plan.DefaultAllow {
|
||||||
|
t.Fatalf("precondition: this config must yield a plan whose default is ALLOW, or the "+
|
||||||
|
"test is not exercising the defect at all; plan=%+v", plan)
|
||||||
|
}
|
||||||
|
|
||||||
|
fwd := renderPlan(t, m, plan)
|
||||||
|
for _, line := range strings.Split(fwd, "\n") {
|
||||||
|
if !strings.Contains(line, netplane.UntunnelableFilterExpr()) &&
|
||||||
|
!strings.Contains(line, "echo-request") {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
t.Errorf("block emitted an untunnelable exception although its whole promise is that "+
|
||||||
|
"there are none: %q", strings.TrimSpace(line))
|
||||||
|
}
|
||||||
|
// The drops now carry the entire policy, so losing one would be silent.
|
||||||
|
if !strings.Contains(fwd, "meta nfproto ipv4 drop") ||
|
||||||
|
!strings.Contains(fwd, "meta nfproto ipv6 drop") {
|
||||||
|
t.Fatalf("block lost a fail-closed drop, so nothing enforces it:\n%s", fwd)
|
||||||
|
}
|
||||||
|
|
||||||
|
// And the applier must not even BUILD a plan for block: nothing reads it, and
|
||||||
|
// building one pushes the routing rules' whole address space into kernel memory
|
||||||
|
// and into the ruleset text (a geoip refresh would then churn the data plane
|
||||||
|
// for a policy that cannot use it).
|
||||||
|
opts, _, err := generate.GenerateWithWarnings(m)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("generate: %v", err)
|
||||||
|
}
|
||||||
|
var a Applier
|
||||||
|
if got := a.untunnelablePlanFor(m, opts); got != nil {
|
||||||
|
t.Errorf("block must build no destination plan at all, got %d step(s) / defaultAllow=%v",
|
||||||
|
len(got.Matches), got.DefaultAllow)
|
||||||
|
}
|
||||||
|
m.Globals.Untunnelable = netplane.UntunnelableDirect
|
||||||
|
if got := a.untunnelablePlanFor(m, opts); got != nil {
|
||||||
|
t.Errorf("direct must build no destination plan either, got %+v", got)
|
||||||
|
}
|
||||||
|
if got := a.untunnelablePlanFor(icmpPolicy(m), opts); got == nil {
|
||||||
|
t.Errorf("icmp is the policy that needs the plan; it must still get one")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestSourceScopedRuleDoesNotWidenTheOtherFamily: a step scoped to IPv6 clients
|
||||||
|
// must emit no IPv4 line at all.
|
||||||
|
//
|
||||||
|
// emit() only wrote `ip saddr @set` when THAT family had prefixes, so a rule whose
|
||||||
|
// source_ip_cidr held only IPv6 prefixes rendered an IPv4 line with no source
|
||||||
|
// clause whatsoever — an accept for every IPv4 host on the LAN, out of a rule the
|
||||||
|
// operator scoped to a handful of v6 addresses. The destination half had always
|
||||||
|
// skipped in that situation (`!AnyDst && len(dst) == 0`); the source half was the
|
||||||
|
// asymmetry, and the catch-all collapse in buildUntunnelablePlan reads "scoped"
|
||||||
|
// as the union of both families, so the two disagreed.
|
||||||
|
//
|
||||||
|
// RED BEFORE THE FIX: `... ip daddr @unt_d4_0 accept`, with no `ip saddr`.
|
||||||
|
func TestSourceScopedRuleDoesNotWidenTheOtherFamily(t *testing.T) {
|
||||||
|
m := icmpPolicy(tunnelModel())
|
||||||
|
m.Globals.IPv6 = true
|
||||||
|
m.Rulesets = []model.Ruleset{pinnedIPSet("lab", "198.51.100.0/24", "2001:db8:70::/48")}
|
||||||
|
m.Rules = []model.Rule{
|
||||||
|
{Name: "lab", Enabled: true, Order: 10, Src: []string{"2001:db8:9::/48"},
|
||||||
|
DstRuleset: []string{"lab"}, Target: "direct"},
|
||||||
|
{Name: "dflt", Enabled: true, Order: 99, Target: "group:auto"},
|
||||||
|
}
|
||||||
|
plan := planFor(t, m, map[string][]string{"rs-lab": {"198.51.100.0/24", "2001:db8:70::/48"}})
|
||||||
|
if len(plan.Matches) != 1 {
|
||||||
|
t.Fatalf("expected one step, got %+v", plan.Matches)
|
||||||
|
}
|
||||||
|
step := plan.Matches[0]
|
||||||
|
if len(step.Src4) != 0 || len(step.Src6) != 1 {
|
||||||
|
t.Fatalf("the step must carry v6 sources only: src4=%v src6=%v", step.Src4, step.Src6)
|
||||||
|
}
|
||||||
|
if len(step.Dst4) != 1 || len(step.Dst6) != 1 {
|
||||||
|
t.Fatalf("the destination list is dual-family: dst4=%v dst6=%v", step.Dst4, step.Dst6)
|
||||||
|
}
|
||||||
|
|
||||||
|
fwd := renderPlan(t, m, plan)
|
||||||
|
for _, line := range strings.Split(fwd, "\n") {
|
||||||
|
if !strings.Contains(line, "@unt_d4_0") {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
t.Errorf("a rule scoped to IPv6 sources emitted an IPv4 line; with no source clause on "+
|
||||||
|
"it that is an accept for the whole LAN: %q", strings.TrimSpace(line))
|
||||||
|
}
|
||||||
|
// ...and the family the rule really does scope must survive, or the guard
|
||||||
|
// over-corrected into dropping the step entirely.
|
||||||
|
if !strings.Contains(fwd, "ip6 saddr @unt_s6_0") ||
|
||||||
|
!strings.Contains(fwd, "ip6 daddr @unt_d6_0 accept") {
|
||||||
|
t.Errorf("the v6 half of the step was lost:\n%s", fwd)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
// TestCatchAllTunnelDropsEverything is the mirror case: a catch-all rule into the
|
// TestCatchAllTunnelDropsEverything is the mirror case: a catch-all rule into the
|
||||||
// tunnel means nothing is provably direct, so everything untunnelable is dropped.
|
// tunnel means nothing is provably direct, so everything untunnelable is dropped.
|
||||||
func TestCatchAllTunnelDropsEverything(t *testing.T) {
|
func TestCatchAllTunnelDropsEverything(t *testing.T) {
|
||||||
m := tunnelModel()
|
m := icmpPolicy(tunnelModel())
|
||||||
m.Rules = []model.Rule{{Name: "all", Enabled: true, Order: 99, Target: "group:auto"}}
|
m.Rules = []model.Rule{{Name: "all", Enabled: true, Order: 99, Target: "group:auto"}}
|
||||||
plan := planFor(t, m, nil)
|
plan := planFor(t, m, nil)
|
||||||
|
|
||||||
@@ -132,7 +270,7 @@ func TestCatchAllTunnelDropsEverything(t *testing.T) {
|
|||||||
// `ru-direct` routes a geoip list direct while everything else is tunnelled, so
|
// `ru-direct` routes a geoip list direct while everything else is tunnelled, so
|
||||||
// exactly those addresses become pingable.
|
// exactly those addresses become pingable.
|
||||||
func TestGeoIPRulesetResolvesToPingableAddresses(t *testing.T) {
|
func TestGeoIPRulesetResolvesToPingableAddresses(t *testing.T) {
|
||||||
m := tunnelModel()
|
m := icmpPolicy(tunnelModel())
|
||||||
m.Globals.IPv6 = true
|
m.Globals.IPv6 = true
|
||||||
m.Rulesets = []model.Ruleset{{
|
m.Rulesets = []model.Ruleset{{
|
||||||
Name: "ru", Type: "ipcidr", Source: "geoip", Categories: []string{"ru"},
|
Name: "ru", Type: "ipcidr", Source: "geoip", Categories: []string{"ru"},
|
||||||
@@ -173,7 +311,7 @@ func TestGeoIPRulesetResolvesToPingableAddresses(t *testing.T) {
|
|||||||
// NOT read as "contains no addresses". That would let later rules decide and
|
// NOT read as "contains no addresses". That would let later rules decide and
|
||||||
// could allow traffic the plan cannot actually account for.
|
// could allow traffic the plan cannot actually account for.
|
||||||
func TestUnloadedRuleSetStaysConservative(t *testing.T) {
|
func TestUnloadedRuleSetStaysConservative(t *testing.T) {
|
||||||
m := tunnelModel()
|
m := icmpPolicy(tunnelModel())
|
||||||
m.Rulesets = []model.Ruleset{{
|
m.Rulesets = []model.Ruleset{{
|
||||||
Name: "ru", Type: "ipcidr", Source: "geoip", Categories: []string{"ru"},
|
Name: "ru", Type: "ipcidr", Source: "geoip", Categories: []string{"ru"},
|
||||||
}}
|
}}
|
||||||
@@ -268,7 +406,7 @@ func TestMigratedDomainAndIPRuleStaysOutOfTheUntunnelablePlan(t *testing.T) {
|
|||||||
for _, target := range []string{"direct", "block", "group:auto"} {
|
for _, target := range []string{"direct", "block", "group:auto"} {
|
||||||
for _, engine := range []string{"up", "down"} {
|
for _, engine := range []string{"up", "down"} {
|
||||||
t.Run(target+"/engine-"+engine, func(t *testing.T) {
|
t.Run(target+"/engine-"+engine, func(t *testing.T) {
|
||||||
m := migratedRule(target)
|
m := icmpPolicy(migratedRule(target))
|
||||||
var loaded map[string][]string
|
var loaded map[string][]string
|
||||||
if engine == "up" {
|
if engine == "up" {
|
||||||
loaded = map[string][]string{"rs-rule-x": {}, "rs-rule-x-ip": {"203.0.113.0/24"}}
|
loaded = map[string][]string{"rs-rule-x": {}, "rs-rule-x-ip": {"203.0.113.0/24"}}
|
||||||
@@ -395,7 +533,7 @@ func TestHugePlanLoadsInFullAndReportsItsSize(t *testing.T) {
|
|||||||
addr := netip.AddrFrom4([4]byte{10, byte(i >> 16), byte(i >> 8), byte(i)})
|
addr := netip.AddrFrom4([4]byte{10, byte(i >> 16), byte(i >> 8), byte(i)})
|
||||||
huge = append(huge, netip.PrefixFrom(addr, 32).String())
|
huge = append(huge, netip.PrefixFrom(addr, 32).String())
|
||||||
}
|
}
|
||||||
m := tunnelModel()
|
m := icmpPolicy(tunnelModel())
|
||||||
m.Rulesets = []model.Ruleset{{
|
m.Rulesets = []model.Ruleset{{
|
||||||
Name: "big", Type: "ipcidr", Source: "geoip", Categories: []string{"us"},
|
Name: "big", Type: "ipcidr", Source: "geoip", Categories: []string{"us"},
|
||||||
}}
|
}}
|
||||||
@@ -532,7 +670,14 @@ func TestPlanNeverAcceptsTCPOrUDP(t *testing.T) {
|
|||||||
strings.TrimSpace(line))
|
strings.TrimSpace(line))
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
if policyLines == 0 {
|
// `block` is the exception, and now it is the point: it emits NO
|
||||||
|
// exception line whatsoever. That silence IS the policy — everything
|
||||||
|
// untunnelable falls through to the fail-closed drops below.
|
||||||
|
if policy == netplane.UntunnelableBlock {
|
||||||
|
if policyLines != 0 {
|
||||||
|
t.Errorf("block must emit no untunnelable exception at all:\n%s", fwd)
|
||||||
|
}
|
||||||
|
} else if policyLines == 0 {
|
||||||
t.Errorf("policy %q emitted no lines at all:\n%s", policy, fwd)
|
t.Errorf("policy %q emitted no lines at all:\n%s", policy, fwd)
|
||||||
}
|
}
|
||||||
if !strings.Contains(fwd, "meta nfproto ipv4 drop") ||
|
if !strings.Contains(fwd, "meta nfproto ipv4 drop") ||
|
||||||
@@ -569,7 +714,7 @@ func TestLocalPlaneSurvivesEveryPlan(t *testing.T) {
|
|||||||
// TestSourceScopedRuleNarrowsTheAllow: a rule scoped to particular clients must
|
// TestSourceScopedRuleNarrowsTheAllow: a rule scoped to particular clients must
|
||||||
// only grant those clients, not everyone.
|
// only grant those clients, not everyone.
|
||||||
func TestSourceScopedRuleNarrowsTheAllow(t *testing.T) {
|
func TestSourceScopedRuleNarrowsTheAllow(t *testing.T) {
|
||||||
m := tunnelModel()
|
m := icmpPolicy(tunnelModel())
|
||||||
m.Rulesets = []model.Ruleset{pinnedIPSet("lab", "198.51.100.0/24")}
|
m.Rulesets = []model.Ruleset{pinnedIPSet("lab", "198.51.100.0/24")}
|
||||||
m.Rules = []model.Rule{
|
m.Rules = []model.Rule{
|
||||||
{Name: "lab", Enabled: true, Order: 10, Src: []string{"192.168.9.0/24"},
|
{Name: "lab", Enabled: true, Order: 10, Src: []string{"192.168.9.0/24"},
|
||||||
|
|||||||
+175
-25
@@ -74,25 +74,76 @@ type Warning struct {
|
|||||||
Message string `json:"message"`
|
Message string `json:"message"`
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// notAppliedTags are the SCREAMING-KEBAB prefixes generate stamps on the one
|
||||||
|
// class of warning that means "you configured this protection and it is NOT in
|
||||||
|
// force right now". They exist precisely so the condition is greppable and
|
||||||
|
// machine-recognisable (generate/ruleset.go, generate/dnsfilter.go say so where
|
||||||
|
// they emit them), which makes them a STRUCTURAL signal rather than a guess at
|
||||||
|
// wording — so they decide severity outright, before anything else is consulted.
|
||||||
|
//
|
||||||
|
// This is the fix for the defect that made this whole classifier untrustworthy:
|
||||||
|
// the tagged texts say "NOT ACTIVE"/"unreachable right now" in words that matched
|
||||||
|
// none of the old markers ("UNREACHABLE" upper-case against "unreachable"
|
||||||
|
// lower-case, "is NOT applied" against "is configured but NOT ACTIVE"), and the
|
||||||
|
// tag also breaks entityRe below, so a blocklist that failed to download — the
|
||||||
|
// single most common real-world fault on this router, and the one the panel has
|
||||||
|
// no other way to show — was published as a plain `warning` under section
|
||||||
|
// "generate" with no name. Meanwhile `ruleset "x": url source with empty url`
|
||||||
|
// parsed cleanly and was graded critical. Severity was, in effect, inverted:
|
||||||
|
// a typo shouted, a network outage whispered.
|
||||||
|
var notAppliedTags = []string{
|
||||||
|
"RULESET-NOT-APPLIED", // generate/ruleset.go:821,997,1242
|
||||||
|
"DNS-FILTER-NOT-APPLIED", // generate/dnsfilter.go:132
|
||||||
|
}
|
||||||
|
|
||||||
// criticalMarkers are substrings that identify a warning as "protection you
|
// criticalMarkers are substrings that identify a warning as "protection you
|
||||||
// configured is not in effect".
|
// configured is not in effect", for the texts that carry neither a tag above nor
|
||||||
|
// a protection section below.
|
||||||
//
|
//
|
||||||
// This is a heuristic over free text, and it is one on purpose: generate emits
|
// This is a heuristic over free text, and it is one on purpose: generate emits
|
||||||
// plain strings today, and inventing a parallel structured warning API across a
|
// plain strings today, and inventing a parallel structured warning API across a
|
||||||
// package boundary owned by another agent would be a far larger change than the
|
// package boundary owned by another agent would be a far larger change than the
|
||||||
// problem warrants. The markers below are taken verbatim from the actual warning
|
// problem warrants. Every entry below is quoted from a warning that a producer
|
||||||
// texts, so they are exact rather than speculative. If generate ever emits its
|
// ACTUALLY emits, with the file it comes from — because the previous list had
|
||||||
// own severity, this list becomes dead code and the conversion simplifies.
|
// drifted into fiction: five of its nine entries matched no living text at all.
|
||||||
|
// Three of those five ("not covered", "fail-closed", "REJECTED") described ONE
|
||||||
|
// netplane message (netplane/nft.go:606), which reaches us on the netplane
|
||||||
|
// channel and is graded critical wholesale before classify() ever runs; one
|
||||||
|
// ("left un-blocked") named a message generate/doh.go:178 records as deleted;
|
||||||
|
// one ("UNREACHABLE") was upper-case against a lower-case text. A marker with no
|
||||||
|
// producer is not harmless: it reads as coverage, and it is what let the real
|
||||||
|
// texts go ungraded for as long as they did.
|
||||||
|
//
|
||||||
|
// If generate ever emits its own severity, this list becomes dead code and the
|
||||||
|
// conversion simplifies.
|
||||||
var criticalMarkers = []string{
|
var criticalMarkers = []string{
|
||||||
"UNREACHABLE", // remote rule-set/blocklist not applied
|
// The DoH NXDOMAIN rules were not built (generate/dns.go:41,67), and a routing
|
||||||
"is NOT applied", // ''
|
// rule whose sources the engine cannot see is not built either
|
||||||
"NOT emitted", // block_doh NXDOMAIN rules missing
|
// (generate/route.go:113) — in both cases the operator's block simply is not there.
|
||||||
"left un-blocked", // block_doh: upstream resolver excluded
|
"NOT emitted",
|
||||||
"inert", // dns_filter / per-device DNS configured but not working
|
// dns_filter / per-device DNS / dns_intercept configured but not working
|
||||||
"has NO effect", // dns_mode=fakeip with no fakeip resolver
|
// (generate/dns.go:32,35,38,58,61,64) — the filter is on in the UI and filtering nothing.
|
||||||
"not covered", // an interface outside the fail-closed guard
|
"inert",
|
||||||
"fail-closed", // ''
|
// A dns_rule that survived parsing but matches nothing (generate/dns.go:987).
|
||||||
"REJECTED", // an unusable interface name
|
"has NO effect",
|
||||||
|
// A routing rule that was emitted but whose target is never reached
|
||||||
|
// (generate/route.go:113,115): the traffic the operator sent through a tunnel
|
||||||
|
// follows the rules below it and the default instead. Present tense on purpose —
|
||||||
|
// "never applied" (past) is warnUnreachableRules' wording, which is graded by
|
||||||
|
// consequence a few lines below, not swept in here.
|
||||||
|
"never applies",
|
||||||
|
// A rule scoped to one source that now matches the WHOLE network
|
||||||
|
// (generate/route.go:407, generate/dns.go:992). Whatever the rule does — send a
|
||||||
|
// device direct, point it at another resolver — it now does it to every client,
|
||||||
|
// and nothing else in the UI shows that the scope collapsed.
|
||||||
|
"applies to EVERY client on the router",
|
||||||
|
"apply to ALL clients",
|
||||||
|
// DNS that leaves the router in plaintext to the provider while the UI shows a
|
||||||
|
// configured resolver (generate/dns.go:38,64,410) and node hostnames resolved
|
||||||
|
// direct from the real address (generate/dns.go:469). These are leaks of exactly
|
||||||
|
// the kind the tunnel exists to prevent.
|
||||||
|
"in the clear",
|
||||||
|
"your provider sees",
|
||||||
// A condition-less rule retired by a later condition-less rule whose target is
|
// A condition-less rule retired by a later condition-less rule whose target is
|
||||||
// `direct` (generate/route.go warnUnreachableRules): the operator's default
|
// `direct` (generate/route.go warnUnreachableRules): the operator's default
|
||||||
// policy — a tunnel, or a block — is not the one the router uses, so everything
|
// policy — a tunnel, or a block — is not the one the router uses, so everything
|
||||||
@@ -117,6 +168,22 @@ var protectionSections = map[string]bool{
|
|||||||
"allowlist": true,
|
"allowlist": true,
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// degradedProtectionMarkers are the exceptions to the section rule above: texts
|
||||||
|
// about a protection list that is STILL IN EFFECT.
|
||||||
|
//
|
||||||
|
// Grading these critical is the same defect pointed the other way. The panel's
|
||||||
|
// alarm banner lights on critical and on nothing else, so every critical that
|
||||||
|
// turns out to be cosmetic teaches the operator that the banner means nothing —
|
||||||
|
// and the next one, the one about the blocklist that really did not load, is the
|
||||||
|
// one they will not read. In particular the refresh failure is the NORMAL state
|
||||||
|
// of a Russian router for minutes at a time: the list is served from the copy
|
||||||
|
// compiled earlier and keeps blocking, which is a degradation, not a gap.
|
||||||
|
var degradedProtectionMarkers = []string{
|
||||||
|
"continuing with the copy compiled earlier", // generate/ruleset.go:819 — stale but blocking
|
||||||
|
"is IGNORED", // generate/ruleset.go:445,472 — a redundant field, the list loads
|
||||||
|
"bad update_interval", // generate/ruleset.go:862,1010 — falls back to the default interval
|
||||||
|
}
|
||||||
|
|
||||||
// infoMarkers identify operational notes that are not protection gaps.
|
// infoMarkers identify operational notes that are not protection gaps.
|
||||||
var infoMarkers = []string{"cache:"}
|
var infoMarkers = []string{"cache:"}
|
||||||
|
|
||||||
@@ -124,26 +191,67 @@ var infoMarkers = []string{"cache:"}
|
|||||||
// consistently, so Section/Name can be recovered from a plain string.
|
// consistently, so Section/Name can be recovered from a plain string.
|
||||||
var entityRe = regexp.MustCompile(`^([a-z_]+) "([^"]*)": (.*)$`)
|
var entityRe = regexp.MustCompile(`^([a-z_]+) "([^"]*)": (.*)$`)
|
||||||
|
|
||||||
|
// tagRe matches the SCREAMING-KEBAB prefix of a tagged warning (see notAppliedTags).
|
||||||
|
var tagRe = regexp.MustCompile(`^([A-Z][A-Z0-9-]*): `)
|
||||||
|
|
||||||
|
// taggedEntityRe recovers the entity from a TAGGED warning, whose shape is
|
||||||
|
//
|
||||||
|
// RULESET-NOT-APPLIED: ruleset "ads" is configured but NOT ACTIVE: ...
|
||||||
|
//
|
||||||
|
// i.e. the tag sits where entityRe expects the kind, and the entity is followed by
|
||||||
|
// prose rather than by ": ". Without this the single most important warning on the
|
||||||
|
// router arrived with Section "generate" and no Name, so the panel could neither
|
||||||
|
// group it nor link to the list it is about.
|
||||||
|
var taggedEntityRe = regexp.MustCompile(`^[A-Z][A-Z0-9-]*: ([a-z_]+) "([^"]*)"`)
|
||||||
|
|
||||||
// warningFromText normalises one free-text warning. defaultSection is used when
|
// warningFromText normalises one free-text warning. defaultSection is used when
|
||||||
// the text carries no `kind "name":` prefix.
|
// the text carries no `kind "name":` prefix.
|
||||||
func warningFromText(text, defaultSection, severity string) Warning {
|
func warningFromText(text, defaultSection, severity string) Warning {
|
||||||
w := Warning{Severity: severity, Section: defaultSection, Message: strings.TrimSpace(text)}
|
w := Warning{Severity: severity, Section: defaultSection, Message: strings.TrimSpace(text)}
|
||||||
if m := entityRe.FindStringSubmatch(w.Message); m != nil {
|
if m := entityRe.FindStringSubmatch(w.Message); m != nil {
|
||||||
w.Section, w.Name, w.Message = m[1], m[2], m[3]
|
w.Section, w.Name, w.Message = m[1], m[2], m[3]
|
||||||
|
return w
|
||||||
|
}
|
||||||
|
if m := taggedEntityRe.FindStringSubmatch(w.Message); m != nil {
|
||||||
|
// Attribution only — the message is deliberately left WHOLE. The tag is the
|
||||||
|
// operator's grep handle into `logread` (it is documented as such where it is
|
||||||
|
// emitted), so stripping it to save one repetition of the list's name would
|
||||||
|
// cost the one thing the tag exists for.
|
||||||
|
w.Section, w.Name = m[1], m[2]
|
||||||
}
|
}
|
||||||
return w
|
return w
|
||||||
}
|
}
|
||||||
|
|
||||||
// classify picks a severity from the parsed section plus the message text.
|
// classify picks a severity from the parsed section plus the message text.
|
||||||
// Section wins where it is decisive (see protectionSections); the markers then
|
//
|
||||||
// catch the global warnings that carry no entity prefix at all.
|
// Order is the whole design:
|
||||||
|
//
|
||||||
|
// 1. a not-applied TAG is structural and decides outright — it is the producer
|
||||||
|
// saying "this protection is off", not us guessing from prose;
|
||||||
|
// 2. info markers, so a cache relocation never reads as a fault;
|
||||||
|
// 3. the section, for the entity kinds whose entire purpose is to block
|
||||||
|
// something — minus the handful of texts that say the list still works;
|
||||||
|
// 4. the free-text markers, which catch the global warnings that carry no entity
|
||||||
|
// prefix at all.
|
||||||
func classify(section, text string) string {
|
func classify(section, text string) string {
|
||||||
|
if m := tagRe.FindStringSubmatch(text); m != nil {
|
||||||
|
for _, tag := range notAppliedTags {
|
||||||
|
if m[1] == tag {
|
||||||
|
return SeverityCritical
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
for _, m := range infoMarkers {
|
for _, m := range infoMarkers {
|
||||||
if strings.Contains(text, m) {
|
if strings.Contains(text, m) {
|
||||||
return SeverityInfo
|
return SeverityInfo
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
if protectionSections[section] {
|
if protectionSections[section] {
|
||||||
|
for _, m := range degradedProtectionMarkers {
|
||||||
|
if strings.Contains(text, m) {
|
||||||
|
return SeverityWarning
|
||||||
|
}
|
||||||
|
}
|
||||||
return SeverityCritical
|
return SeverityCritical
|
||||||
}
|
}
|
||||||
for _, m := range criticalMarkers {
|
for _, m := range criticalMarkers {
|
||||||
@@ -163,6 +271,15 @@ func classify(section, text string) string {
|
|||||||
// fail-closed guard does not cover that interface — always critical.
|
// fail-closed guard does not cover that interface — always critical.
|
||||||
// - configWarnings come from model.Validate (already structured).
|
// - configWarnings come from model.Validate (already structured).
|
||||||
func collectWarnings(g model.Globals, generateWarnings, netplaneWarnings []string, configWarnings []model.Warning, planWarnings ...string) []Warning {
|
func collectWarnings(g model.Globals, generateWarnings, netplaneWarnings []string, configWarnings []model.Warning, planWarnings ...string) []Warning {
|
||||||
|
return finalizeWarnings(gatherWarnings(g, generateWarnings, netplaneWarnings, configWarnings, planWarnings...))
|
||||||
|
}
|
||||||
|
|
||||||
|
// gatherWarnings is collectWarnings without the sort and the cap, so a caller
|
||||||
|
// that must FOLD IN a warning of its own (applyLocked's post-swap failure, which
|
||||||
|
// has to say that the data plane is incomplete) can do so and then finalize once.
|
||||||
|
// Sorting and capping a list twice is not equivalent: the second pass would drop
|
||||||
|
// the "N further warning(s) suppressed" disclosure the first pass appended.
|
||||||
|
func gatherWarnings(g model.Globals, generateWarnings, netplaneWarnings []string, configWarnings []model.Warning, planWarnings ...string) []Warning {
|
||||||
out := make([]Warning, 0, len(generateWarnings)+len(netplaneWarnings)+len(configWarnings)+1)
|
out := make([]Warning, 0, len(generateWarnings)+len(netplaneWarnings)+len(configWarnings)+1)
|
||||||
// The untunnelable-protocol policy always reports what it costs the user; it is
|
// The untunnelable-protocol policy always reports what it costs the user; it is
|
||||||
// the only one of these that describes correct behaviour rather than a fault.
|
// the only one of these that describes correct behaviour rather than a fault.
|
||||||
@@ -186,7 +303,12 @@ func collectWarnings(g model.Globals, generateWarnings, netplaneWarnings []strin
|
|||||||
Message: cw.Message,
|
Message: cw.Message,
|
||||||
})
|
})
|
||||||
}
|
}
|
||||||
|
return out
|
||||||
|
}
|
||||||
|
|
||||||
|
// finalizeWarnings sorts critical-first and applies the cap. Call it exactly once
|
||||||
|
// per published set.
|
||||||
|
func finalizeWarnings(out []Warning) []Warning {
|
||||||
// Stable sort by descending severity so the cap can only drop the least
|
// Stable sort by descending severity so the cap can only drop the least
|
||||||
// important entries, and the panel gets the worst news first.
|
// important entries, and the panel gets the worst news first.
|
||||||
sort.SliceStable(out, func(i, j int) bool {
|
sort.SliceStable(out, func(i, j int) bool {
|
||||||
@@ -241,33 +363,61 @@ func untunnelablePolicyWarnings(g model.Globals, planNotes []string) []Warning {
|
|||||||
// With the kill switch open the forward chain has no drops at all, so nothing
|
// With the kill switch open the forward chain has no drops at all, so nothing
|
||||||
// is restricted whatever the policy says. Saying that is more useful than
|
// is restricted whatever the policy says. Saying that is more useful than
|
||||||
// repeating a promise which is not being kept.
|
// repeating a promise which is not being kept.
|
||||||
|
//
|
||||||
|
// Neither this note nor the `direct` one below may claim IPTV, for the same
|
||||||
|
// reason the `block` note disclaims it: multicast does not cross this router
|
||||||
|
// under ANY of the three settings. The stream itself is WAN-side inbound and
|
||||||
|
// these rules never match it, and a client's outbound multicast UDP is dropped
|
||||||
|
// by the fail-closed guard regardless of the policy. Promising it here would be
|
||||||
|
// the identical lie to the one just removed from `block`, only in the branch
|
||||||
|
// where the operator is least likely to go looking for the cause.
|
||||||
if !killSwitchClosed(g) {
|
if !killSwitchClosed(g) {
|
||||||
if policy == netplane.UntunnelableDirect {
|
if policy == netplane.UntunnelableDirect {
|
||||||
return out
|
return out
|
||||||
}
|
}
|
||||||
return note(policy,
|
return note(policy,
|
||||||
"This setting has no effect while the kill switch is open: with the kill switch open "+
|
"This setting has no effect while the kill switch is open: with the kill switch open the "+
|
||||||
"nothing is blocked, so ping, IPTV and VPN passthrough all work — and all of them "+
|
"forward chain has no drops at all, so ping, traceroute and raw VPN passthrough "+
|
||||||
"reach the internet with your real IP address.")
|
"(IPsec ESP/AH, PPTP/GRE) all work — and every one of them reaches the internet with "+
|
||||||
|
"your real IP address. IPTV is not part of that: multicast does not pass this router "+
|
||||||
|
"on any setting, which is a separate matter from this one.")
|
||||||
}
|
}
|
||||||
|
|
||||||
switch policy {
|
switch policy {
|
||||||
case netplane.UntunnelableDirect:
|
case netplane.UntunnelableDirect:
|
||||||
return note(policy,
|
return note(policy,
|
||||||
"Ping, IPTV and VPN passthrough (IPsec/PPTP) work everywhere, but they go straight out "+
|
"Ping and traceroute work everywhere, and so does raw VPN passthrough (IPsec ESP/AH, "+
|
||||||
"with your real IP address instead of through the tunnel — they are the kinds of "+
|
"PPTP/GRE) — but all of it goes straight out with your real IP address instead of "+
|
||||||
"traffic a tunnel cannot carry.")
|
"through the tunnel, because a tunnel cannot carry this kind of traffic. VPNs that "+
|
||||||
|
"run over UDP (WireGuard, OpenVPN-UDP, IPsec through NAT) are ordinary tunnelled "+
|
||||||
|
"traffic and are unaffected either way. IPTV is not covered by this setting at all: "+
|
||||||
|
"multicast does not pass this router on any of the three, so switching to `direct` "+
|
||||||
|
"will not bring it back.")
|
||||||
case netplane.UntunnelableICMP:
|
case netplane.UntunnelableICMP:
|
||||||
return note(policy,
|
return note(policy,
|
||||||
"Ping and traceroute work everywhere, including addresses you send through the tunnel; "+
|
"Ping and traceroute work everywhere, including addresses you send through the tunnel; "+
|
||||||
"the host you ping sees your real IP address. IPTV and VPN passthrough (IPsec/PPTP) "+
|
"the host you ping sees your real IP address. IPTV and VPN passthrough (IPsec/PPTP) "+
|
||||||
"work only toward addresses your rules route directly.")
|
"work only toward addresses your rules route directly.")
|
||||||
default:
|
default:
|
||||||
|
// This text used to say these things "work only toward addresses your rules
|
||||||
|
// route directly". That was written when a `direct` route final made the
|
||||||
|
// untunnelable drop degenerate into a blanket accept — i.e. when the note was
|
||||||
|
// describing the bug rather than the policy. netplane now blocks what it says
|
||||||
|
// it blocks, so the honest sentence is that none of it works at all, and the
|
||||||
|
// note has to name what is and is NOT affected: "ping does not work" sends an
|
||||||
|
// operator hunting a fault, and the difference between raw ESP and IPsec
|
||||||
|
// through NAT is the difference between "my VPN broke" and "my VPN is fine".
|
||||||
return note(netplane.UntunnelableBlock,
|
return note(netplane.UntunnelableBlock,
|
||||||
"Ping, traceroute, IPTV and VPN passthrough work only toward addresses your rules route "+
|
"Ping, traceroute, IPsec/PPTP VPN passthrough and IPTV do not work from your devices at "+
|
||||||
"directly — those already see your real IP address anyway. Toward addresses you send "+
|
"all — not even toward addresses your rules route directly. None of this traffic can "+
|
||||||
"through the tunnel they will not work, because a tunnel cannot carry them and they "+
|
"travel through a tunnel, so rather than let it out with your real IP address it is "+
|
||||||
"would otherwise leak your real IP address.")
|
"dropped. Concretely: ping and Windows tracert fail (on Linux and macOS traceroute "+
|
||||||
|
"sends UDP probes instead, which ARE tunnelled — the hops it prints are the tunnel's "+
|
||||||
|
"path, not your own), and so do raw IPsec (ESP/AH) and PPTP/GRE — a PPTP session will "+
|
||||||
|
"even look connected, because its control channel is TCP and only the payload is "+
|
||||||
|
"dropped. VPNs that run over UDP are NOT affected: WireGuard, OpenVPN-UDP and IPsec "+
|
||||||
|
"through NAT (IKE on UDP 500, NAT-T on UDP 4500) keep working normally. Multicast "+
|
||||||
|
"IPTV does not cross this router under any setting; that one is not this policy.")
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
@@ -17,7 +17,7 @@ import (
|
|||||||
func TestCollectWarningsAttributesEntities(t *testing.T) {
|
func TestCollectWarningsAttributesEntities(t *testing.T) {
|
||||||
got := collectWarnings(blockGlobals(),
|
got := collectWarnings(blockGlobals(),
|
||||||
[]string{
|
[]string{
|
||||||
`ruleset "ads": remote list "https://x/y.srs" is UNREACHABLE right now, so it is NOT applied`,
|
ruleSetNotApplied,
|
||||||
`device "kids-tablet": no current IP (ip unset and MAC "aa:bb" not leased), skipped`,
|
`device "kids-tablet": no current IP (ip unset and MAC "aa:bb" not leased), skipped`,
|
||||||
`chain "hop": has no hops, target skipped`,
|
`chain "hop": has no hops, target skipped`,
|
||||||
`dns_filter enabled but no resolvers configured; filter inert`,
|
`dns_filter enabled but no resolvers configured; filter inert`,
|
||||||
@@ -38,14 +38,23 @@ func TestCollectWarningsAttributesEntities(t *testing.T) {
|
|||||||
if w.Severity != SeverityCritical {
|
if w.Severity != SeverityCritical {
|
||||||
t.Errorf("an unapplied blocklist is a protection gap; severity = %q, want critical", w.Severity)
|
t.Errorf("an unapplied blocklist is a protection gap; severity = %q, want critical", w.Severity)
|
||||||
}
|
}
|
||||||
if strings.Contains(w.Message, `ruleset "ads":`) {
|
// The RULESET-NOT-APPLIED tag stays in the message on purpose: generate
|
||||||
t.Errorf("the entity prefix must move into Section/Name, not stay in Message: %q", w.Message)
|
// documents it as the operator's grep handle into logread.
|
||||||
|
if !strings.HasPrefix(w.Message, "RULESET-NOT-APPLIED:") {
|
||||||
|
t.Errorf("a tagged warning must keep its greppable tag in the message: %q", w.Message)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
if w, ok := byName["device/kids-tablet"]; !ok {
|
if w, ok := byName["device/kids-tablet"]; !ok {
|
||||||
t.Errorf("device warning not attributed; got %+v", got)
|
t.Errorf("device warning not attributed; got %+v", got)
|
||||||
} else if w.Severity != SeverityWarning {
|
} else {
|
||||||
t.Errorf("a skipped device is not a protection gap; severity = %q, want warning", w.Severity)
|
if w.Severity != SeverityWarning {
|
||||||
|
t.Errorf("a skipped device is not a protection gap; severity = %q, want warning", w.Severity)
|
||||||
|
}
|
||||||
|
// The plain `kind "name": message` prefix, by contrast, MOVES into
|
||||||
|
// Section/Name — it carries no information the fields do not.
|
||||||
|
if strings.Contains(w.Message, `device "kids-tablet":`) {
|
||||||
|
t.Errorf("the entity prefix must move into Section/Name, not stay in Message: %q", w.Message)
|
||||||
|
}
|
||||||
}
|
}
|
||||||
if w, ok := byName["chain/hop"]; !ok || w.Severity != SeverityWarning {
|
if w, ok := byName["chain/hop"]; !ok || w.Severity != SeverityWarning {
|
||||||
t.Errorf("chain warning: got %+v", w)
|
t.Errorf("chain warning: got %+v", w)
|
||||||
@@ -115,7 +124,7 @@ func TestCollectWarningsCapKeepsCriticals(t *testing.T) {
|
|||||||
for i := 0; i < 200; i++ {
|
for i := 0; i < 200; i++ {
|
||||||
noisy = append(noisy, `chain "c": has no hops, target skipped`)
|
noisy = append(noisy, `chain "c": has no hops, target skipped`)
|
||||||
}
|
}
|
||||||
noisy = append(noisy, `ruleset "ads": remote list is UNREACHABLE right now, so it is NOT applied`)
|
noisy = append(noisy, ruleSetNotApplied)
|
||||||
|
|
||||||
got := collectWarnings(blockGlobals(), noisy, nil, nil)
|
got := collectWarnings(blockGlobals(), noisy, nil, nil)
|
||||||
if len(got) != maxStatusWarnings {
|
if len(got) != maxStatusWarnings {
|
||||||
@@ -232,6 +241,170 @@ func TestWarningsAgainstRealGenerateOutput(t *testing.T) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// ruleSetNotApplied is the message generate/ruleset.go:997 actually produces when
|
||||||
|
// a remote blocklist cannot be fetched — the single most common real fault on a
|
||||||
|
// router in Russia, and the one the panel has no other way to show (an omitted
|
||||||
|
// rule-set produces no row in GET /api/ruleset/status, so the UI is
|
||||||
|
// indistinguishable from "not configured").
|
||||||
|
//
|
||||||
|
// Quoted verbatim, tag and all, because the classifier is a heuristic over free
|
||||||
|
// text and a test written against invented text proves nothing about it. The
|
||||||
|
// version this replaced asserted on `... is UNREACHABLE ... is NOT applied`, a
|
||||||
|
// sentence no producer has ever emitted; it passed for as long as the real
|
||||||
|
// sentence was being graded a plain `warning` under section "generate" with no
|
||||||
|
// name at all.
|
||||||
|
const ruleSetNotApplied = `RULESET-NOT-APPLIED: ruleset "ads" is configured but NOT ACTIVE: ` +
|
||||||
|
`its source "https://big.oisd.nl/domainswild" is unreachable right now, so rule-set "ads" was ` +
|
||||||
|
`omitted and matches NOTHING until it loads (a blocklist blocks nothing; a routing rule is skipped). ` +
|
||||||
|
`Handing an unusable list to the engine would abort engine start and take the LAN down instead. ` +
|
||||||
|
`Retried automatically on the next reconcile (~1 min) — no action needed unless this persists.`
|
||||||
|
|
||||||
|
// TestClassifyRealGenerateTexts grades the sentences generate REALLY emits, by
|
||||||
|
// consequence.
|
||||||
|
//
|
||||||
|
// The defect this pins: severity was inverted. A blocklist that could not be
|
||||||
|
// downloaded — protection the operator configured, not in force — came out
|
||||||
|
// `warning`, because its tag broke the entity regexp and its wording matched no
|
||||||
|
// marker. A typo in the same list's URL came out `critical`, because that one
|
||||||
|
// parsed cleanly into section "ruleset". The panel's alarm banner lights on
|
||||||
|
// critical and on nothing else, so the router shouted about the typo and stayed
|
||||||
|
// quiet about the outage.
|
||||||
|
func TestClassifyRealGenerateTexts(t *testing.T) {
|
||||||
|
cases := []struct {
|
||||||
|
name string
|
||||||
|
text string
|
||||||
|
want string
|
||||||
|
// section/entity attribution, when the panel must be able to deep-link.
|
||||||
|
section, entity string
|
||||||
|
}{{
|
||||||
|
name: "remote blocklist could not be fetched",
|
||||||
|
text: ruleSetNotApplied,
|
||||||
|
want: SeverityCritical,
|
||||||
|
section: "ruleset", entity: "ads",
|
||||||
|
}, {
|
||||||
|
name: "not one DNS filter list could be built",
|
||||||
|
text: "DNS-FILTER-NOT-APPLIED: dns_filter is ON and lists are enabled, but NOT ONE of them " +
|
||||||
|
"could be built right now — nothing is being filtered or allowed. See the per-list " +
|
||||||
|
"warnings above for why. Rebuilt on the next reconcile (~1 min).",
|
||||||
|
want: SeverityCritical,
|
||||||
|
section: "generate",
|
||||||
|
}, {
|
||||||
|
// generate/ruleset.go:1214 — the counterpart the outage used to be graded
|
||||||
|
// BELOW. It stays critical: the consequence is identical (the list is not
|
||||||
|
// loaded and matches nothing), which is the whole point — the inversion is
|
||||||
|
// fixed by lifting the outage, not by lowering the typo.
|
||||||
|
name: "url blocklist with an empty url",
|
||||||
|
text: `ruleset "ads": url source with empty url, skipped`,
|
||||||
|
want: SeverityCritical,
|
||||||
|
section: "ruleset", entity: "ads",
|
||||||
|
}, {
|
||||||
|
// generate/ruleset.go:819 — the list IS still in force, from the copy
|
||||||
|
// compiled earlier. Grading this critical would light the alarm banner on
|
||||||
|
// most reconciles of a healthy router behind a flaky link, and a banner that
|
||||||
|
// is always on is a banner nobody reads when it finally matters.
|
||||||
|
name: "blocklist refresh failed but the compiled copy still blocks",
|
||||||
|
text: `ruleset "ads": could not refresh the list from "https://big.oisd.nl/domainswild" ` +
|
||||||
|
`(dial tcp: i/o timeout); continuing with the copy compiled earlier. Retried on the next reconcile.`,
|
||||||
|
want: SeverityWarning,
|
||||||
|
section: "ruleset", entity: "ads",
|
||||||
|
}, {
|
||||||
|
// generate/route.go:407 — a rule the operator scoped to one device now
|
||||||
|
// applies to the entire network. Whatever it does, it now does to everyone.
|
||||||
|
name: "a rule's source scope collapsed to the whole LAN",
|
||||||
|
text: `rule "kids": none of its source entries can be matched by the engine (interface/zone/MAC ` +
|
||||||
|
`selectors and invalid addresses are dropped), so the rule now applies to EVERY client on ` +
|
||||||
|
`the router instead of that source — check it is still what you want, and use IP ` +
|
||||||
|
`addresses/subnets as the source`,
|
||||||
|
want: SeverityCritical,
|
||||||
|
section: "rule", entity: "kids",
|
||||||
|
}, {
|
||||||
|
// generate/route.go:115 — the rule exists in the UI and routes nothing.
|
||||||
|
name: "a rule whose target is never reached",
|
||||||
|
text: `rule "work": no matcher the engine can evaluate, skipped — its target "group:auto" ` +
|
||||||
|
`never applies and the traffic follows the rules below it and the default`,
|
||||||
|
want: SeverityCritical,
|
||||||
|
section: "rule", entity: "work",
|
||||||
|
}, {
|
||||||
|
// generate/ruleset.go:445 — a redundant field on a list that loads fine.
|
||||||
|
name: "a redundant field on a working blocklist",
|
||||||
|
text: `ruleset "ads": format "binary" is IGNORED for source=geosite — the category decides. Remove it to avoid confusion.`,
|
||||||
|
want: SeverityWarning,
|
||||||
|
section: "ruleset", entity: "ads",
|
||||||
|
}, {
|
||||||
|
// generate/chain.go:108 — a configured path that does not resolve. Its
|
||||||
|
// traffic is blocked fail-closed, so no protection claim is broken.
|
||||||
|
name: "a chain with no hops",
|
||||||
|
text: `chain "hop": has no hops, target skipped`,
|
||||||
|
want: SeverityWarning,
|
||||||
|
section: "chain", entity: "hop",
|
||||||
|
}, {
|
||||||
|
// generate/cache.go:92 — operational, the lists still compile.
|
||||||
|
name: "the compiled lists moved to tmpfs",
|
||||||
|
text: "cache: only 3 MiB free on /overlay (need 8 MiB), using tmpfs /tmp/shater instead — " +
|
||||||
|
"remote rule-sets will be re-downloaded after every reboot, so free some space",
|
||||||
|
want: SeverityInfo,
|
||||||
|
section: "generate",
|
||||||
|
}}
|
||||||
|
|
||||||
|
for _, tc := range cases {
|
||||||
|
t.Run(tc.name, func(t *testing.T) {
|
||||||
|
var got *Warning
|
||||||
|
for _, w := range collectWarnings(blockGlobals(), []string{tc.text}, nil, nil) {
|
||||||
|
if w.Section == "untunnelable" {
|
||||||
|
continue // the always-present policy notice
|
||||||
|
}
|
||||||
|
w := w
|
||||||
|
got = &w
|
||||||
|
}
|
||||||
|
if got == nil {
|
||||||
|
t.Fatalf("the warning was dropped entirely")
|
||||||
|
}
|
||||||
|
if got.Severity != tc.want {
|
||||||
|
t.Errorf("severity = %q, want %q\n text: %s", got.Severity, tc.want, tc.text)
|
||||||
|
}
|
||||||
|
if got.Section != tc.section || got.Name != tc.entity {
|
||||||
|
t.Errorf("attribution = %q/%q, want %q/%q — the panel deep-links on these",
|
||||||
|
got.Section, got.Name, tc.section, tc.entity)
|
||||||
|
}
|
||||||
|
})
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestUnreachableRuleSetFromRealGenerate drives the ACTUAL producer, so this stays
|
||||||
|
// correct if generate rewords or re-tags the message. A url rule-set pointing at a
|
||||||
|
// closed local port fails the reachability probe exactly the way an unreachable
|
||||||
|
// public blocklist does, with no network needed.
|
||||||
|
func TestUnreachableRuleSetFromRealGenerate(t *testing.T) {
|
||||||
|
m := holdModel("closed")
|
||||||
|
m.Rulesets = []model.Ruleset{
|
||||||
|
{Name: "ads", Type: "domain", Source: "url", URL: "http://127.0.0.1:1/blocklist.srs", Format: "binary"},
|
||||||
|
}
|
||||||
|
m.Rules = []model.Rule{
|
||||||
|
{Name: "blockads", Enabled: true, Order: 10, DstRuleset: []string{"ads"}, Target: "block"},
|
||||||
|
}
|
||||||
|
|
||||||
|
_, genWarnings, err := generate.GenerateWithWarnings(m)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("GenerateWithWarnings: %v", err)
|
||||||
|
}
|
||||||
|
t.Logf("real generate warnings: %q", genWarnings)
|
||||||
|
|
||||||
|
var found *Warning
|
||||||
|
for _, w := range collectWarnings(blockGlobals(), genWarnings, nil, nil) {
|
||||||
|
if w.Section == "ruleset" && w.Name == "ads" {
|
||||||
|
w := w
|
||||||
|
found = &w
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if found == nil {
|
||||||
|
t.Fatalf("an unfetchable blocklist produced no warning attributed to it: %q", genWarnings)
|
||||||
|
}
|
||||||
|
if found.Severity != SeverityCritical {
|
||||||
|
t.Errorf("a blocklist that did not load is a protection gap the operator cannot otherwise "+
|
||||||
|
"see; severity = %q, want critical (message: %q)", found.Severity, found.Message)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
// blockGlobals is the default policy fixture: kill-switch closed, untunnelable
|
// blockGlobals is the default policy fixture: kill-switch closed, untunnelable
|
||||||
// traffic blocked — i.e. what a stock install runs.
|
// traffic blocked — i.e. what a stock install runs.
|
||||||
func blockGlobals() model.Globals {
|
func blockGlobals() model.Globals {
|
||||||
|
|||||||
@@ -329,6 +329,13 @@ func cmdRun() int {
|
|||||||
if addr, ok := panelAddr(globals); ok {
|
if addr, ok := panelAddr(globals); ok {
|
||||||
panelSrv = panel.NewServer(applier, panel.Options{Addr: addr, Logger: logger})
|
panelSrv = panel.NewServer(applier, panel.Options{Addr: addr, Logger: logger})
|
||||||
panelSrv.SetStats(statsAgg)
|
panelSrv.SetStats(statsAgg)
|
||||||
|
// The log download reads the sink's FILE, and the sink writes to it from
|
||||||
|
// its own goroutine — so without a barrier a download can miss the last
|
||||||
|
// milliseconds of lines, which are the ones the operator came for. Hand
|
||||||
|
// the panel the barrier only (not the sink): it is a consumer of the log,
|
||||||
|
// not its owner. Bounded on the panel side, so a stuck writer cannot turn
|
||||||
|
// /api/log into the hang the async sink was built to prevent.
|
||||||
|
panelSrv.SetLogSync(sink.Sync)
|
||||||
go func() {
|
go func() {
|
||||||
if err := panelSrv.Start(); err != nil && err != http.ErrServerClosed {
|
if err := panelSrv.Start(); err != nil && err != http.ErrServerClosed {
|
||||||
logger.Warn("panel server unavailable (daemon continues): ", err)
|
logger.Warn("panel server unavailable (daemon continues): ", err)
|
||||||
|
|||||||
@@ -33,8 +33,47 @@ var newStatsStore = stats.NewStore
|
|||||||
var _ stats.StatsStore = (*statsHolder)(nil)
|
var _ stats.StatsStore = (*statsHolder)(nil)
|
||||||
|
|
||||||
// statsHolder delegates every StatsStore call to the store currently installed.
|
// statsHolder delegates every StatsStore call to the store currently installed.
|
||||||
// Reads take a read lock so a swap never blocks concurrent panel queries for
|
//
|
||||||
// longer than the pointer exchange.
|
// The read lock is held for the WHOLE delegated call, not just long enough to
|
||||||
|
// pick the pointer up. That distinction is the entire safety property of this
|
||||||
|
// type, and getting it wrong is invisible in every test that does not race a
|
||||||
|
// reader against a swap:
|
||||||
|
//
|
||||||
|
// // WRONG — this is what it used to do.
|
||||||
|
// func (h *statsHolder) get() stats.StatsStore {
|
||||||
|
// h.mu.RLock(); defer h.mu.RUnlock(); return h.inner
|
||||||
|
// }
|
||||||
|
// func (h *statsHolder) Snapshot() stats.Snapshot { return h.get().Snapshot() }
|
||||||
|
//
|
||||||
|
// The lock is gone by the time Snapshot runs. A reader that has just taken the
|
||||||
|
// pointer is holding nothing: the swap acquires the write lock unopposed, closes
|
||||||
|
// that very store, and the reader then queries a closed one. For the persistent
|
||||||
|
// backend that is not an error the panel can see — a closed bolt ring answers
|
||||||
|
// every read with errRingUnavailable, i.e. an EMPTY PAGE, which the panel renders
|
||||||
|
// as "you have no traffic". A lie, presented as data, from the one component
|
||||||
|
// whose whole job is to report what is happening.
|
||||||
|
//
|
||||||
|
// Ordering the close correctly (see Reconfigure) protects the FILE. It cannot
|
||||||
|
// protect a caller that already holds the pointer; only the lock can.
|
||||||
|
//
|
||||||
|
// The cost is bounded and lands where it should. Readers do not block each other
|
||||||
|
// — RWMutex admits them all concurrently — so the panel's own load is unchanged.
|
||||||
|
// What now waits is the SWAP: it must let the in-flight reads finish, which is
|
||||||
|
// one page query (at most stats.MaxLogLimit rows) plus the old store's close. A
|
||||||
|
// swap happens when a human changes a setting.
|
||||||
|
//
|
||||||
|
// The two alternatives were weighed and rejected:
|
||||||
|
//
|
||||||
|
// - Reference-count the store and close it when the last reader leaves. It
|
||||||
|
// lets the swap return sooner, but it moves the close — flush, final bbolt
|
||||||
|
// commit, and any error it reports — onto whichever panel goroutine happens
|
||||||
|
// to drop the last reference. That is real lifecycle machinery, and error
|
||||||
|
// reporting from an arbitrary place, bought to avoid a wait measured in
|
||||||
|
// milliseconds on a once-per-Apply operation.
|
||||||
|
// - Make a closed store safe to read (serve its last snapshot). It cannot be
|
||||||
|
// done for the part that matters: the query and connection logs live in the
|
||||||
|
// bolt file, and a closed ring has nothing to serve but emptiness. That is
|
||||||
|
// the lie itself, not a defence against it.
|
||||||
type statsHolder struct {
|
type statsHolder struct {
|
||||||
mu sync.RWMutex
|
mu sync.RWMutex
|
||||||
inner stats.StatsStore
|
inner stats.StatsStore
|
||||||
@@ -56,23 +95,60 @@ func newStatsHolder(eng *engine.Engine, logger log.ContextLogger, backend string
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
func (h *statsHolder) get() stats.StatsStore {
|
// --- stats.StatsStore -------------------------------------------------------
|
||||||
|
//
|
||||||
|
// Every one of these runs the inner call INSIDE the read lock. There is
|
||||||
|
// deliberately no `get()` accessor any more: a helper that hands the pointer out
|
||||||
|
// is a helper that hands out an unguarded reference, and the whole defect above
|
||||||
|
// was one such call site.
|
||||||
|
|
||||||
|
func (h *statsHolder) Start() {
|
||||||
h.mu.RLock()
|
h.mu.RLock()
|
||||||
defer h.mu.RUnlock()
|
defer h.mu.RUnlock()
|
||||||
return h.inner
|
h.inner.Start()
|
||||||
}
|
}
|
||||||
|
|
||||||
// --- stats.StatsStore -------------------------------------------------------
|
func (h *statsHolder) Close() error {
|
||||||
|
h.mu.RLock()
|
||||||
|
defer h.mu.RUnlock()
|
||||||
|
return h.inner.Close()
|
||||||
|
}
|
||||||
|
|
||||||
func (h *statsHolder) Start() { h.get().Start() }
|
func (h *statsHolder) Resubscribe() {
|
||||||
func (h *statsHolder) Close() error { return h.get().Close() }
|
h.mu.RLock()
|
||||||
func (h *statsHolder) Resubscribe() { h.get().Resubscribe() }
|
defer h.mu.RUnlock()
|
||||||
|
h.inner.Resubscribe()
|
||||||
|
}
|
||||||
|
|
||||||
func (h *statsHolder) Queries(q stats.LogQuery) []stats.LogEntry { return h.get().Queries(q) }
|
func (h *statsHolder) Queries(q stats.LogQuery) []stats.LogEntry {
|
||||||
func (h *statsHolder) Conns(q stats.LogQuery) []stats.ConnLogEntry { return h.get().Conns(q) }
|
h.mu.RLock()
|
||||||
func (h *statsHolder) QueriesPage(q stats.LogQuery) stats.LogPage { return h.get().QueriesPage(q) }
|
defer h.mu.RUnlock()
|
||||||
func (h *statsHolder) ConnsPage(q stats.LogQuery) stats.ConnPage { return h.get().ConnsPage(q) }
|
return h.inner.Queries(q)
|
||||||
func (h *statsHolder) Snapshot() stats.Snapshot { return h.get().Snapshot() }
|
}
|
||||||
|
|
||||||
|
func (h *statsHolder) Conns(q stats.LogQuery) []stats.ConnLogEntry {
|
||||||
|
h.mu.RLock()
|
||||||
|
defer h.mu.RUnlock()
|
||||||
|
return h.inner.Conns(q)
|
||||||
|
}
|
||||||
|
|
||||||
|
func (h *statsHolder) QueriesPage(q stats.LogQuery) stats.LogPage {
|
||||||
|
h.mu.RLock()
|
||||||
|
defer h.mu.RUnlock()
|
||||||
|
return h.inner.QueriesPage(q)
|
||||||
|
}
|
||||||
|
|
||||||
|
func (h *statsHolder) ConnsPage(q stats.LogQuery) stats.ConnPage {
|
||||||
|
h.mu.RLock()
|
||||||
|
defer h.mu.RUnlock()
|
||||||
|
return h.inner.ConnsPage(q)
|
||||||
|
}
|
||||||
|
|
||||||
|
func (h *statsHolder) Snapshot() stats.Snapshot {
|
||||||
|
h.mu.RLock()
|
||||||
|
defer h.mu.RUnlock()
|
||||||
|
return h.inner.Snapshot()
|
||||||
|
}
|
||||||
|
|
||||||
// --- swapping ---------------------------------------------------------------
|
// --- swapping ---------------------------------------------------------------
|
||||||
|
|
||||||
@@ -90,17 +166,35 @@ func statsSpecFrom(g model.Globals) (string, stats.Config) {
|
|||||||
// Reconfigure swaps in a new store when backend/cfg differ from what is running.
|
// Reconfigure swaps in a new store when backend/cfg differ from what is running.
|
||||||
// It reports whether a swap happened.
|
// It reports whether a swap happened.
|
||||||
//
|
//
|
||||||
// Ordering is deliberate: the NEW store is built and started BEFORE the old one
|
// Ordering is deliberate, and it is the OPPOSITE of what it was: the old store is
|
||||||
// is closed, so a failure to construct it leaves the running store untouched and
|
// CLOSED FIRST, and only then is the new one built.
|
||||||
// stats keep working. The old store is then closed, which for the persistent
|
//
|
||||||
// backend flushes the pending write buffer into its final bbolt commit — dropping
|
// Building first looked safer — a construction failure would leave the running
|
||||||
// it without Close would lose the not-yet-committed tail of the log.
|
// store untouched — but the two stores of a persistent-backend swap resolve the
|
||||||
|
// SAME on-disk path, and the second bbolt.Open there cannot win a file the first
|
||||||
|
// one still holds. It sat on the flock for a second, timed out, and the old ring's
|
||||||
|
// "an unopenable file must be a corrupt file" recovery deleted the live database
|
||||||
|
// out from under its own writer (unlink of an open file succeeds on Linux, so
|
||||||
|
// nothing complained): months of query log gone, replaced by an empty DB whose
|
||||||
|
// Snapshot still reported the same backend. Changing ANY of the five stats knobs
|
||||||
|
// was enough. The overlap was the whole defect, so the overlap is what had to go —
|
||||||
|
// the classification bug is fixed alongside it in stats.newBoltRing, but a swap
|
||||||
|
// that hands the file over cleanly does not depend on that fix being right.
|
||||||
|
//
|
||||||
|
// Closing first also flushes the old store's pending write buffer into its final
|
||||||
|
// bbolt commit; dropping it without Close would lose the not-yet-committed tail.
|
||||||
|
//
|
||||||
|
// The whole exchange runs under the write lock, so a concurrent /api/stats poll
|
||||||
|
// sees either the OLD OPEN store or the new one — never the closed one in between,
|
||||||
|
// which would answer an honest-looking empty page. Readers pay one flush+commit of
|
||||||
|
// blocking on a knob change; a torn read costs the panel its trust.
|
||||||
//
|
//
|
||||||
// stats.NewStore itself never returns nil and never panics: an unopenable stats
|
// stats.NewStore itself never returns nil and never panics: an unopenable stats
|
||||||
// DB degrades to the in-memory ring internally and reports Backend "memory" from
|
// DB degrades to the in-memory ring internally and reports Backend "memory" from
|
||||||
// Snapshot, so the panel shows what is actually running rather than what was
|
// Snapshot, so the panel shows what is actually running rather than what was
|
||||||
// asked for. That is the "degrade, don't die" requirement, and it means
|
// asked for. That is the "degrade, don't die" requirement, and it means
|
||||||
// Reconfigure has no error path of its own.
|
// Reconfigure has no error path of its own — and it is why closing first cannot
|
||||||
|
// strand the daemon without a store.
|
||||||
//
|
//
|
||||||
// Log cursor continuity is DELIBERATELY not preserved across a swap. Each store
|
// Log cursor continuity is DELIBERATELY not preserved across a swap. Each store
|
||||||
// owns its own seq counter (memRing counts from zero; boltRing seeds from its own
|
// owns its own seq counter (memRing counts from zero; boltRing seeds from its own
|
||||||
@@ -112,12 +206,19 @@ func statsSpecFrom(g model.Globals) (string, stats.Config) {
|
|||||||
// honest Backend in the snapshot — which it gets for free by delegating.
|
// honest Backend in the snapshot — which it gets for free by delegating.
|
||||||
func (h *statsHolder) Reconfigure(backend string, cfg stats.Config) bool {
|
func (h *statsHolder) Reconfigure(backend string, cfg stats.Config) bool {
|
||||||
h.mu.Lock()
|
h.mu.Lock()
|
||||||
|
defer h.mu.Unlock()
|
||||||
if backend == h.backend && cfg == h.cfg {
|
if backend == h.backend && cfg == h.cfg {
|
||||||
h.mu.Unlock()
|
|
||||||
return false
|
return false
|
||||||
}
|
}
|
||||||
old, oldBackend := h.inner, h.backend
|
old, oldBackend := h.inner, h.backend
|
||||||
|
|
||||||
|
// Hand the resources over BEFORE asking for them again: the old store releases
|
||||||
|
// its DB file (and its subscriptions) here, so the new one opens a path nobody
|
||||||
|
// holds. See the ordering note above — this line is the fix.
|
||||||
|
if err := old.Close(); err != nil {
|
||||||
|
h.log.Warn("stats: closing the previous ", oldBackend, " store: ", err)
|
||||||
|
}
|
||||||
|
|
||||||
next := newStatsStore(backend, h.eng, h.log, cfg)
|
next := newStatsStore(backend, h.eng, h.log, cfg)
|
||||||
next.Start()
|
next.Start()
|
||||||
// Point the fresh store at the running box's event managers. Start() subscribes,
|
// Point the fresh store at the running box's event managers. Start() subscribes,
|
||||||
@@ -130,13 +231,7 @@ func (h *statsHolder) Reconfigure(backend string, cfg stats.Config) bool {
|
|||||||
h.inner = next
|
h.inner = next
|
||||||
h.backend = backend
|
h.backend = backend
|
||||||
h.cfg = cfg
|
h.cfg = cfg
|
||||||
h.mu.Unlock()
|
|
||||||
|
|
||||||
// Close the old store OUTSIDE the lock: a persistent-store Close flushes and
|
|
||||||
// commits, which can take a moment, and no panel read should block behind it.
|
|
||||||
if err := old.Close(); err != nil {
|
|
||||||
h.log.Warn("stats: closing the previous ", oldBackend, " store: ", err)
|
|
||||||
}
|
|
||||||
h.log.Info("stats: backend ", oldBackend, " -> ", backend,
|
h.log.Info("stats: backend ", oldBackend, " -> ", backend,
|
||||||
" (log cursors restart; reload the panel to resume live tailing)")
|
" (log cursors restart; reload the panel to resume live tailing)")
|
||||||
return true
|
return true
|
||||||
|
|||||||
@@ -1,8 +1,10 @@
|
|||||||
package main
|
package main
|
||||||
|
|
||||||
import (
|
import (
|
||||||
|
"sync"
|
||||||
"sync/atomic"
|
"sync/atomic"
|
||||||
"testing"
|
"testing"
|
||||||
|
"time"
|
||||||
|
|
||||||
"github.com/sagernet/sing-box/log"
|
"github.com/sagernet/sing-box/log"
|
||||||
"github.com/sagernet/sing-box/shater/engine"
|
"github.com/sagernet/sing-box/shater/engine"
|
||||||
@@ -150,3 +152,234 @@ func TestStatsHolderDelegatesAfterSwap(t *testing.T) {
|
|||||||
t.Errorf("Close: %v", err)
|
t.Errorf("Close: %v", err)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// orderedStore is a fakeStore that timestamps its construction and its Close against a
|
||||||
|
// shared step counter, so a test can assert the ORDER of the swap choreography rather
|
||||||
|
// than only its outcome.
|
||||||
|
type orderedStore struct {
|
||||||
|
fakeStore
|
||||||
|
step *atomic.Int32
|
||||||
|
builtAt int32
|
||||||
|
closedAt int32
|
||||||
|
}
|
||||||
|
|
||||||
|
func (o *orderedStore) Close() error {
|
||||||
|
o.closedAt = o.step.Add(1)
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestStatsHolderClosesOldBeforeOpeningNew is the data-loss regression: changing ANY of
|
||||||
|
// the five stats knobs used to destroy the persistent log.
|
||||||
|
//
|
||||||
|
// The swap built the replacement store BEFORE closing the one it replaced, and both
|
||||||
|
// halves resolve the SAME on-disk DB path (stats.statsFilePath). The second bbolt.Open
|
||||||
|
// therefore lost the flock race against a database that was still open, timed out after a
|
||||||
|
// second, and the ring's "an unopenable file must be a corrupt file" recovery deleted it —
|
||||||
|
// on Linux, unlinking an open file succeeds, so the old store went on writing into a
|
||||||
|
// nameless inode while the panel was told nothing had changed. Months of query log, gone,
|
||||||
|
// for a ring-size edit.
|
||||||
|
//
|
||||||
|
// The store the holder builds in production owns an exclusive resource, so the ONLY safe
|
||||||
|
// order is release-then-acquire. That is what is pinned here; the constructor seam makes
|
||||||
|
// it assertable without standing up a real DB.
|
||||||
|
func TestStatsHolderClosesOldBeforeOpeningNew(t *testing.T) {
|
||||||
|
var step atomic.Int32
|
||||||
|
var built []*orderedStore
|
||||||
|
orig := newStatsStore
|
||||||
|
newStatsStore = func(backend string, _ *engine.Engine, _ log.ContextLogger, cfg ...stats.Config) stats.StatsStore {
|
||||||
|
s := &orderedStore{step: &step}
|
||||||
|
s.backend = backend
|
||||||
|
if len(cfg) > 0 {
|
||||||
|
s.cfg = cfg[0]
|
||||||
|
}
|
||||||
|
s.builtAt = step.Add(1)
|
||||||
|
built = append(built, s)
|
||||||
|
return s
|
||||||
|
}
|
||||||
|
t.Cleanup(func() { newStatsStore = orig })
|
||||||
|
|
||||||
|
h := newStatsHolderFrom(nil, log.StdLogger(), globalsWith("sqlite", 200))
|
||||||
|
h.Start()
|
||||||
|
if !h.ReconfigureFrom(globalsWith("sqlite", 5000)) {
|
||||||
|
t.Fatalf("changing the ring size must swap the store")
|
||||||
|
}
|
||||||
|
if len(built) != 2 {
|
||||||
|
t.Fatalf("built %d stores, want 2", len(built))
|
||||||
|
}
|
||||||
|
old, next := built[0], built[1]
|
||||||
|
|
||||||
|
if old.closedAt == 0 {
|
||||||
|
t.Fatal("the replaced store was never closed — its buffered rows are lost and its DB file stays locked")
|
||||||
|
}
|
||||||
|
if old.closedAt > next.builtAt {
|
||||||
|
t.Fatalf("the new store was constructed at step %d while the old one was still open (closed at step %d): "+
|
||||||
|
"both resolve the same stats DB path, so the second open loses the flock race and the loser's "+
|
||||||
|
"recovery path deletes the live database", next.builtAt, old.closedAt)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestStatsHolderReadsNeverSeeTheClosedStore is the second half of the swap defect,
|
||||||
|
// and it is about the READER rather than the file.
|
||||||
|
//
|
||||||
|
// Closing the outgoing store before opening its replacement is what stops the DB file
|
||||||
|
// from being deleted, but it does nothing for a caller that is already inside a call on
|
||||||
|
// that store. The holder used to take the read lock only long enough to copy the pointer
|
||||||
|
// out (`get()`), so the sequence below was ordinary, not exotic:
|
||||||
|
//
|
||||||
|
// 1. a panel poll takes the pointer to the running store and drops the lock;
|
||||||
|
// 2. Reconfigure takes the write lock — unopposed, nobody is holding the read side —
|
||||||
|
// and closes that store;
|
||||||
|
// 3. the poll now runs its query against a store that is shut.
|
||||||
|
//
|
||||||
|
// For the persistent backend step 3 is not a visible error: a closed bolt ring answers
|
||||||
|
// every read with errRingUnavailable, which reaches the panel as an empty page and is
|
||||||
|
// drawn as "no traffic". That is the same class of lie as everything else in this pass,
|
||||||
|
// arriving from the other direction — the component whose entire job is to report what
|
||||||
|
// happened, reporting that nothing did.
|
||||||
|
//
|
||||||
|
// The test drives the three steps EXPLICITLY rather than hoping a hammering loop lands in
|
||||||
|
// the window (it did, about one run in five, which is exactly the kind of failure that
|
||||||
|
// gets dismissed as a flake). The reader is parked inside the outgoing store's Snapshot;
|
||||||
|
// the swap is then given every chance to close it underneath.
|
||||||
|
func TestStatsHolderReadsNeverSeeTheClosedStore(t *testing.T) {
|
||||||
|
var stores []*gatedStore
|
||||||
|
orig := newStatsStore
|
||||||
|
newStatsStore = func(backend string, _ *engine.Engine, _ log.ContextLogger, cfg ...stats.Config) stats.StatsStore {
|
||||||
|
g := &gatedStore{entered: make(chan struct{}), release: make(chan struct{})}
|
||||||
|
g.backend = backend
|
||||||
|
if len(stores) > 0 {
|
||||||
|
close(g.release) // only the OUTGOING store parks its reader
|
||||||
|
}
|
||||||
|
stores = append(stores, g)
|
||||||
|
return g
|
||||||
|
}
|
||||||
|
t.Cleanup(func() { newStatsStore = orig })
|
||||||
|
|
||||||
|
h := newStatsHolderFrom(nil, log.StdLogger(), globalsWith("sqlite", 200))
|
||||||
|
h.Start()
|
||||||
|
outgoing := stores[0]
|
||||||
|
|
||||||
|
// 1. A panel poll is inside the outgoing store and cannot be hurried.
|
||||||
|
read := make(chan string, 1)
|
||||||
|
go func() { read <- h.Snapshot().Backend }()
|
||||||
|
select {
|
||||||
|
case <-outgoing.entered:
|
||||||
|
case <-time.After(10 * time.Second):
|
||||||
|
close(outgoing.release)
|
||||||
|
t.Fatal("the reader never reached the store")
|
||||||
|
}
|
||||||
|
|
||||||
|
// 2. The swap runs while that reader is still inside.
|
||||||
|
swapped := make(chan struct{})
|
||||||
|
go func() {
|
||||||
|
defer close(swapped)
|
||||||
|
h.ReconfigureFrom(globalsWith("sqlite", 5000))
|
||||||
|
}()
|
||||||
|
|
||||||
|
// Give it every opportunity to close the store out from under the parked reader.
|
||||||
|
// A correct holder cannot: the reader holds the read lock, so the swap is still
|
||||||
|
// waiting for the write lock when this deadline expires. A holder that only guards
|
||||||
|
// the pointer gets there in microseconds, and this loop ends the moment it does.
|
||||||
|
deadline := time.Now().Add(300 * time.Millisecond)
|
||||||
|
for !outgoing.shut.Load() && time.Now().Before(deadline) {
|
||||||
|
time.Sleep(time.Millisecond)
|
||||||
|
}
|
||||||
|
|
||||||
|
// 3. Let the poll finish and see what it was served.
|
||||||
|
close(outgoing.release)
|
||||||
|
got := <-read
|
||||||
|
select {
|
||||||
|
case <-swapped:
|
||||||
|
case <-time.After(10 * time.Second):
|
||||||
|
t.Fatal("the swap never completed after the reader left")
|
||||||
|
}
|
||||||
|
|
||||||
|
if got == closedStoreBackend {
|
||||||
|
t.Fatal("a panel read was served by the store that was being closed — an empty page presented as data")
|
||||||
|
}
|
||||||
|
if got != "sqlite" {
|
||||||
|
t.Fatalf("the read returned %q, want the running backend", got)
|
||||||
|
}
|
||||||
|
if len(stores) != 2 {
|
||||||
|
t.Fatalf("built %d stores, want 2", len(stores))
|
||||||
|
}
|
||||||
|
if !outgoing.shut.Load() {
|
||||||
|
t.Fatal("the outgoing store was never closed")
|
||||||
|
}
|
||||||
|
if got := h.Snapshot().Backend; got != "sqlite" {
|
||||||
|
t.Fatalf("after the swap Snapshot reports %q, want sqlite", got)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// closedStoreBackend is what gatedStore reports once it has been closed. No real store
|
||||||
|
// says this; it exists so that a read served by a shut store is unmistakable instead of
|
||||||
|
// merely empty — which is precisely what makes the real defect hard to see.
|
||||||
|
const closedStoreBackend = "closed"
|
||||||
|
|
||||||
|
// gatedStore parks the first caller of Snapshot until the test releases it, and reports
|
||||||
|
// closedStoreBackend from the moment Close has run.
|
||||||
|
type gatedStore struct {
|
||||||
|
fakeStore
|
||||||
|
entered chan struct{}
|
||||||
|
release chan struct{}
|
||||||
|
enterOnce sync.Once
|
||||||
|
shut atomic.Bool
|
||||||
|
}
|
||||||
|
|
||||||
|
func (s *gatedStore) Close() error {
|
||||||
|
s.shut.Store(true)
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
|
||||||
|
func (s *gatedStore) Snapshot() stats.Snapshot {
|
||||||
|
s.enterOnce.Do(func() { close(s.entered) })
|
||||||
|
<-s.release
|
||||||
|
if s.shut.Load() {
|
||||||
|
return stats.Snapshot{Backend: closedStoreBackend}
|
||||||
|
}
|
||||||
|
return stats.Snapshot{Backend: s.backend}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestStatsHolderSwapWaitsForInFlightReads states the cost of the above out loud, so a
|
||||||
|
// later reader of this file knows the wait is intended and not an oversight: a swap does
|
||||||
|
// not begin while a read is in progress.
|
||||||
|
func TestStatsHolderSwapWaitsForInFlightReads(t *testing.T) {
|
||||||
|
var stores []*gatedStore
|
||||||
|
orig := newStatsStore
|
||||||
|
newStatsStore = func(backend string, _ *engine.Engine, _ log.ContextLogger, cfg ...stats.Config) stats.StatsStore {
|
||||||
|
g := &gatedStore{entered: make(chan struct{}), release: make(chan struct{})}
|
||||||
|
g.backend = backend
|
||||||
|
if len(stores) > 0 {
|
||||||
|
close(g.release)
|
||||||
|
}
|
||||||
|
stores = append(stores, g)
|
||||||
|
return g
|
||||||
|
}
|
||||||
|
t.Cleanup(func() { newStatsStore = orig })
|
||||||
|
|
||||||
|
h := newStatsHolderFrom(nil, log.StdLogger(), globalsWith("sqlite", 200))
|
||||||
|
h.Start()
|
||||||
|
outgoing := stores[0]
|
||||||
|
|
||||||
|
go func() { _ = h.Snapshot() }()
|
||||||
|
<-outgoing.entered
|
||||||
|
|
||||||
|
swapped := make(chan struct{})
|
||||||
|
go func() {
|
||||||
|
defer close(swapped)
|
||||||
|
h.ReconfigureFrom(globalsWith("sqlite", 5000))
|
||||||
|
}()
|
||||||
|
|
||||||
|
select {
|
||||||
|
case <-swapped:
|
||||||
|
close(outgoing.release)
|
||||||
|
t.Fatal("the swap completed while a read was still inside the store it retired")
|
||||||
|
case <-time.After(200 * time.Millisecond):
|
||||||
|
}
|
||||||
|
close(outgoing.release)
|
||||||
|
select {
|
||||||
|
case <-swapped:
|
||||||
|
case <-time.After(10 * time.Second):
|
||||||
|
t.Fatal("the swap never completed after the read finished")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|||||||
+23
-2
@@ -159,7 +159,21 @@ func New(logger ...log.ContextLogger) *Engine {
|
|||||||
// URLTestHistory() below. The pointer is stable across Apply swaps, so health
|
// URLTestHistory() below. The pointer is stable across Apply swaps, so health
|
||||||
// history survives config changes instead of being reset on every apply.
|
// history survives config changes instead of being reset on every apply.
|
||||||
ctx = service.ContextWithPtr(ctx, urltest.NewHistoryStorage())
|
ctx = service.ContextWithPtr(ctx, urltest.NewHistoryStorage())
|
||||||
return &Engine{ctx: ctx, log: l}
|
// Pre-register the engine itself as the probe GATE, into the same shared
|
||||||
|
// registry and for the same reason as the two above: every urltest group
|
||||||
|
// built by every box.New reads it out of ctx (protocol/group/urltest.go), and
|
||||||
|
// the registration must survive Apply swaps because the question it answers
|
||||||
|
// outlives any single box.
|
||||||
|
//
|
||||||
|
// The gate answers "may this outbound run its own scheduled probe right
|
||||||
|
// now" — see Engine.ProbeAllowed. It is a live call, not a stored flag, so
|
||||||
|
// a chain hop that comes back resumes probing with no reapply. Registering
|
||||||
|
// the Engine here (rather than a snapshot) is what makes that possible;
|
||||||
|
// the struct is built first purely so ctx can point at it.
|
||||||
|
e := &Engine{log: l}
|
||||||
|
ctx = service.ContextWith[urltest.ProbeGate](ctx, e)
|
||||||
|
e.ctx = ctx
|
||||||
|
return e
|
||||||
}
|
}
|
||||||
|
|
||||||
// Apply installs opts as the running configuration.
|
// Apply installs opts as the running configuration.
|
||||||
@@ -200,9 +214,16 @@ func (e *Engine) applyLocked(opts option.Options) (bool, error) {
|
|||||||
// change that flips a group used<->unused is then a real change that
|
// change that flips a group used<->unused is then a real change that
|
||||||
// triggers a swap, and an unchanged config hashes identically on every
|
// triggers a swap, and an unchanged config hashes identically on every
|
||||||
// reconcile because the stand-down is deterministic.
|
// reconcile because the stand-down is deterministic.
|
||||||
if stood := standDownUnusedSelfCheck(opts); stood > 0 && e.log != nil {
|
stood, used := standDownUnusedSelfCheck(opts)
|
||||||
|
if stood > 0 && e.log != nil {
|
||||||
e.log.Info("apply: stood down self-check on ", stood, " unused urltest group(s); the observatory is their only prober")
|
e.log.Info("apply: stood down self-check on ", stood, " unused urltest group(s); the observatory is their only prober")
|
||||||
}
|
}
|
||||||
|
// Publish the same walk's used-set for ProbeWhenIdle BEFORE the new box is
|
||||||
|
// built, because the question is asked during that box's PostStart: a used
|
||||||
|
// group arms its probing ticker there rather than waiting for traffic. The
|
||||||
|
// observatory's own copy is published later, after a SUCCESSFUL swap, which
|
||||||
|
// is too late to be the answer.
|
||||||
|
e.publishKeepWarm(used)
|
||||||
|
|
||||||
newHash, err := e.hashOptions(opts)
|
newHash, err := e.hashOptions(opts)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
|
|||||||
@@ -372,6 +372,17 @@ type ChainHealth struct {
|
|||||||
// SELECTED member's observation, or the freshest ALIVE member's when the
|
// SELECTED member's observation, or the freshest ALIVE member's when the
|
||||||
// selection has no measurement of its own — the number shown must always be a
|
// selection has no measurement of its own — the number shown must always be a
|
||||||
// measurement somebody took, never an average nobody did.
|
// measurement somebody took, never an average nobody did.
|
||||||
|
//
|
||||||
|
// # The hop nobody reached: BlockedBy
|
||||||
|
//
|
||||||
|
// The prober walks a chain in wire order and stops at the first dead hop
|
||||||
|
// (observatory.go probeChainOrdered), because a probe of hop 3 dials THROUGH
|
||||||
|
// hop 2 and a hop 2 with no live member makes that probe a measurement of hop 2.
|
||||||
|
// Everything behind the break is therefore not dialled at all, and this struct
|
||||||
|
// says so: State is untested — no fresh knowledge, which is the truth — the
|
||||||
|
// counters collapse into Untested, and BlockedBy names the hop that stopped the
|
||||||
|
// walk. A row like that must never be read as a fault of its own; the fault is
|
||||||
|
// at BlockedBy.Index, and that is the hop to go and fix.
|
||||||
type ChainHopHealth struct {
|
type ChainHopHealth struct {
|
||||||
Index int `json:"index"` // 1-based position on the wire, L1..Ln
|
Index int `json:"index"` // 1-based position on the wire, L1..Ln
|
||||||
Tag string `json:"tag"` // "chain-<name>-h<i>" — the wrapper actually dialled
|
Tag string `json:"tag"` // "chain-<name>-h<i>" — the wrapper actually dialled
|
||||||
@@ -386,6 +397,24 @@ type ChainHopHealth struct {
|
|||||||
Alive int `json:"alive"`
|
Alive int `json:"alive"`
|
||||||
Dead int `json:"dead"`
|
Dead int `json:"dead"`
|
||||||
Untested int `json:"untested"`
|
Untested int `json:"untested"`
|
||||||
|
// BlockedBy is set ONLY on a hop the ordered walk never reached, and names the
|
||||||
|
// dead hop in front of it. Absent (omitted, never null) on every hop that was
|
||||||
|
// itself measured. When it is present State is always "untested" and the
|
||||||
|
// counters are always 0/0/0/Total — see the type comment.
|
||||||
|
BlockedBy *ChainHopBlock `json:"blocked_by,omitempty"`
|
||||||
|
}
|
||||||
|
|
||||||
|
// ChainHopBlock identifies the hop whose failure stopped a chain's ordered probe
|
||||||
|
// walk: its 1-based wire index and the engine outbound tag that was dialled.
|
||||||
|
//
|
||||||
|
// The index is the field that matters — it is what the operator acts on, and it
|
||||||
|
// lines up with ChainHopHealth.Index on the same chain, so a reader can point
|
||||||
|
// straight at the offending row. The tag is diagnostic detail (logs, tooltips)
|
||||||
|
// and is not a label to put in front of a person: "chain-ewan-wg-subs-h2" is not
|
||||||
|
// a name anybody chose.
|
||||||
|
type ChainHopBlock struct {
|
||||||
|
Index int `json:"index"`
|
||||||
|
Tag string `json:"tag"`
|
||||||
}
|
}
|
||||||
|
|
||||||
// ChainHealth reports the reachability (used/unused) of every named chain plus
|
// ChainHealth reports the reachability (used/unused) of every named chain plus
|
||||||
@@ -481,9 +510,46 @@ func chainHopHealthOf(pool []adapter.Outbound, name string, view HealthView) []C
|
|||||||
// internet. Marked after sorting so the flag cannot depend on pool order.
|
// internet. Marked after sorting so the flag cannot depend on pool order.
|
||||||
hops[len(hops)-1].Exit = true
|
hops[len(hops)-1].Exit = true
|
||||||
}
|
}
|
||||||
|
markBlockedHops(hops)
|
||||||
return hops
|
return hops
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// markBlockedHops applies the ordered walk's outcome to the projection: from the
|
||||||
|
// first hop that reads DEAD, every later hop is rewritten as not-reached.
|
||||||
|
//
|
||||||
|
// This is a rewrite and not an annotation on purpose. The prober stops at the
|
||||||
|
// first dead hop, so a hop behind the break has not been dialled since the break
|
||||||
|
// appeared — but its board records do not vanish, they age out on the TTL. For
|
||||||
|
// as long as that takes, the raw projection would keep publishing a verdict about
|
||||||
|
// a hop nothing has attempted, and the most damaging version of that is a stale
|
||||||
|
// "dead" pointing the operator at a hop that may be perfectly fine. Untested is
|
||||||
|
// the honest reading of "nothing current is known", and BlockedBy is why.
|
||||||
|
//
|
||||||
|
// The counters collapse the same way, into Untested, so the invariants the whole
|
||||||
|
// health surface promises still hold verbatim: Tested == Alive+Dead and
|
||||||
|
// Alive+Dead+Untested == Total.
|
||||||
|
//
|
||||||
|
// Selected SURVIVES. It is not a measurement — it is the member the wrapper would
|
||||||
|
// route through — and it stays true (and useful) whether or not anything reached
|
||||||
|
// this hop.
|
||||||
|
func markBlockedHops(hops []ChainHopHealth) {
|
||||||
|
for i := range hops {
|
||||||
|
if hops[i].State != HealthDead {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
blocker := &ChainHopBlock{Index: hops[i].Index, Tag: hops[i].Tag}
|
||||||
|
for j := i + 1; j < len(hops); j++ {
|
||||||
|
h := &hops[j]
|
||||||
|
h.State = HealthUntested
|
||||||
|
h.DelayMs, h.AgeSeconds = 0, -1
|
||||||
|
h.Alive, h.Dead, h.Tested = 0, 0, 0
|
||||||
|
h.Untested = h.Total
|
||||||
|
h.BlockedBy = blocker
|
||||||
|
}
|
||||||
|
return // the first break is the only one that can be observed
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
// chainNodeHop is the one-measurement hop: the wrapper itself was dialled by
|
// chainNodeHop is the one-measurement hop: the wrapper itself was dialled by
|
||||||
// the observatory, so its own board state IS the hop's health and the counters
|
// the observatory, so its own board state IS the hop's health and the counters
|
||||||
// degenerate to whichever bucket that state fills.
|
// degenerate to whichever bucket that state fills.
|
||||||
|
|||||||
@@ -326,6 +326,10 @@ func TestChainHealthNilUsedSet(t *testing.T) {
|
|||||||
// the counters must obey the GroupHealth invariants on every hop, and the exit
|
// the counters must obey the GroupHealth invariants on every hop, and the exit
|
||||||
// flag must sit on the largest index. This is the "which hop died" question the
|
// flag must sit on the largest index. This is the "which hop died" question the
|
||||||
// whole hop surface exists to answer.
|
// whole hop surface exists to answer.
|
||||||
|
//
|
||||||
|
// L3 additionally pins the ordered walk's half of the contract: it sits BEHIND
|
||||||
|
// the dead L2, so whatever its stale board records say, it is reported untested
|
||||||
|
// with L2 named — never as a hop of its own with a verdict nobody took.
|
||||||
func TestChainHopHealthProjection(t *testing.T) {
|
func TestChainHopHealthProjection(t *testing.T) {
|
||||||
hist := urltest.NewHistoryStorage()
|
hist := urltest.NewHistoryStorage()
|
||||||
now := time.Now()
|
now := time.Now()
|
||||||
@@ -396,16 +400,34 @@ func TestChainHopHealthProjection(t *testing.T) {
|
|||||||
t.Errorf("h2 delay/age = %d/%d, want 0/3 (the selected member's failure observation)", h2.DelayMs, h2.AgeSeconds)
|
t.Errorf("h2 delay/age = %d/%d, want 0/3 (the selected member's failure observation)", h2.DelayMs, h2.AgeSeconds)
|
||||||
}
|
}
|
||||||
|
|
||||||
// L3: alive (one member answers), one member honestly untested — never
|
// L3: BEHIND the break. Its board still holds an alive record for member "a"
|
||||||
// folded into dead. The selected member is the alive one, so its numbers show.
|
// (probed before L2 died), but the prober has not dialled this hop since, so
|
||||||
if h3.Kind != "group" || h3.State != HealthAlive {
|
// publishing that record would be a verdict nobody currently holds — and the
|
||||||
t.Fatalf("h3 = %+v, want an alive exit hop", h3)
|
// symmetric case, a stale FAILURE, would send the operator to fix a hop that
|
||||||
|
// is fine. It reads untested, counters collapsed into Untested, with L2 named.
|
||||||
|
if h3.Kind != "group" || h3.State != HealthUntested {
|
||||||
|
t.Fatalf("h3 = %+v, want an UNTESTED hop: the walk stopped at L2 and never dialled it", h3)
|
||||||
}
|
}
|
||||||
if h3.Total != 2 || h3.Tested != 1 || h3.Alive != 1 || h3.Dead != 0 || h3.Untested != 1 {
|
if h3.BlockedBy == nil {
|
||||||
t.Errorf("h3 counters = %+v, want 1 alive / 1 untested", h3)
|
t.Fatalf("h3 = %+v, want BlockedBy naming L2 — 'why is this empty' must be answered on the row", h3)
|
||||||
}
|
}
|
||||||
if h3.Selected != "a" || h3.DelayMs != 200 || h3.AgeSeconds != 7 {
|
if h3.BlockedBy.Index != 2 || h3.BlockedBy.Tag != "chain-c-h2" {
|
||||||
t.Errorf("h3 = %+v, want selection a with its 200ms/7s", h3)
|
t.Errorf("h3.BlockedBy = %+v, want the dead hop 2 / chain-c-h2", *h3.BlockedBy)
|
||||||
|
}
|
||||||
|
if h3.Total != 2 || h3.Tested != 0 || h3.Alive != 0 || h3.Dead != 0 || h3.Untested != 2 {
|
||||||
|
t.Errorf("h3 counters = %+v, want every member untested (nothing was measured at this hop)", h3)
|
||||||
|
}
|
||||||
|
if h3.DelayMs != 0 || h3.AgeSeconds != -1 {
|
||||||
|
t.Errorf("h3 delay/age = %d/%d, want 0/-1 — there is no observation to age", h3.DelayMs, h3.AgeSeconds)
|
||||||
|
}
|
||||||
|
// The selection is not a measurement: the wrapper really would route through
|
||||||
|
// "a", and that stays true whether or not anything reached this hop.
|
||||||
|
if h3.Selected != "a" {
|
||||||
|
t.Errorf("h3 Selected = %q, want a — a block hides measurements, not configuration", h3.Selected)
|
||||||
|
}
|
||||||
|
// Nothing in FRONT of the break may be touched by it.
|
||||||
|
if h1.BlockedBy != nil || h2.BlockedBy != nil {
|
||||||
|
t.Errorf("BlockedBy leaked onto a measured hop: h1=%+v h2=%+v", h1.BlockedBy, h2.BlockedBy)
|
||||||
}
|
}
|
||||||
|
|
||||||
// The invariants, on every hop, exactly as GroupHealth promises.
|
// The invariants, on every hop, exactly as GroupHealth promises.
|
||||||
|
|||||||
+104
-6
@@ -109,6 +109,14 @@ const (
|
|||||||
// the observatory probed the target's path after the button press and the
|
// the observatory probed the target's path after the button press and the
|
||||||
// path did not answer. An honest negative, not a missing measurement.
|
// path did not answer. An honest negative, not a missing measurement.
|
||||||
groupTestErrPathDead = "the observatory's probe through this path failed"
|
groupTestErrPathDead = "the observatory's probe through this path failed"
|
||||||
|
// groupTestErrChainBlockedFmt: a CHAIN whose exit was never dialled because
|
||||||
|
// an earlier hop was probed and did not answer. The prober walks a chain in
|
||||||
|
// order and stops at the first dead hop, so there is no end-to-end
|
||||||
|
// measurement to wait for and there never will be while that hop is down.
|
||||||
|
// Reported at once, naming the hop, instead of spending the full deadline to
|
||||||
|
// answer "not reached yet" about a path that is known to be broken — and
|
||||||
|
// naming it is the whole value: the operator's next action is at that hop.
|
||||||
|
groupTestErrChainBlockedFmt = "hop %d of this chain was probed and did not answer, so nothing reaches the exit through it — fix that hop first"
|
||||||
)
|
)
|
||||||
|
|
||||||
// GroupTestResult is one target's test outcome — a group's or a chain's (Group
|
// GroupTestResult is one target's test outcome — a group's or a chain's (Group
|
||||||
@@ -168,6 +176,13 @@ type groupTestTarget struct {
|
|||||||
name string
|
name string
|
||||||
ob adapter.Outbound
|
ob adapter.Outbound
|
||||||
sel func() string
|
sel func() string
|
||||||
|
// isChain marks a CHAIN target. A chain is the one target whose measurement
|
||||||
|
// can be legitimately absent while the target is known to be broken: the
|
||||||
|
// prober walks a chain in order and stops at the first dead hop, so a chain
|
||||||
|
// whose hop 2 died never has its exit dialled at all. Without this flag a run
|
||||||
|
// would sit out its whole deadline and then report "not reached yet" about a
|
||||||
|
// path it knows is down — see testOneTarget.
|
||||||
|
isChain bool
|
||||||
}
|
}
|
||||||
|
|
||||||
// chainTargetsFrom discovers each materialised chain in the running
|
// chainTargetsFrom discovers each materialised chain in the running
|
||||||
@@ -212,7 +227,7 @@ func chainTargetsFrom(pool []adapter.Outbound) []groupTestTarget {
|
|||||||
if !ok {
|
if !ok {
|
||||||
continue
|
continue
|
||||||
}
|
}
|
||||||
t := groupTestTarget{name: name, ob: ob}
|
t := groupTestTarget{name: name, ob: ob, isChain: true}
|
||||||
if hop := chainLastGroupHop(byTag, name); hop != nil {
|
if hop := chainLastGroupHop(byTag, name); hop != nil {
|
||||||
t.sel = func() string { return chainMemberName(hop.Now(), name) }
|
t.sel = func() string { return chainMemberName(hop.Now(), name) }
|
||||||
}
|
}
|
||||||
@@ -226,18 +241,38 @@ func chainTargetsFrom(pool []adapter.Outbound) []groupTestTarget {
|
|||||||
// "chain-<name>-h<digits>". Member copies ("…-h<i>-<member>") and non-chain tags
|
// "chain-<name>-h<digits>". Member copies ("…-h<i>-<member>") and non-chain tags
|
||||||
// parse false.
|
// parse false.
|
||||||
func parseChainExitTag(tag string) (string, bool) {
|
func parseChainExitTag(tag string) (string, bool) {
|
||||||
|
name, _, ok := parseChainHopTag(tag)
|
||||||
|
return name, ok
|
||||||
|
}
|
||||||
|
|
||||||
|
// parseChainHopTag is parseChainExitTag plus the number: it recovers the chain
|
||||||
|
// NAME and the 1-based HOP INDEX from a hop wrapper tag "chain-<name>-h<digits>"
|
||||||
|
// (generate/chain.go buildHopWrapper). Member copies ("…-h<i>-<member>") and
|
||||||
|
// non-chain tags parse false.
|
||||||
|
//
|
||||||
|
// The index is what lets the probe plan ORDER a chain's measurements. A chain is
|
||||||
|
// a series path — hop i is dialled through hops 1..i-1 — so the difference
|
||||||
|
// between "this hop failed" and "the hop in front of it failed" is the whole
|
||||||
|
// diagnosis, and it cannot be recovered from an unordered set of tags.
|
||||||
|
//
|
||||||
|
// The split takes the LAST "-h<digits>", so a chain literally named "a-h2" is
|
||||||
|
// read as chain "a-h2" hop 1 rather than chain "a" hop 2 — the same documented
|
||||||
|
// reserved-namespace edge case parseGroupCopyTag accepts, pinned by
|
||||||
|
// TestParseChainExitTag.
|
||||||
|
func parseChainHopTag(tag string) (name string, hop int, ok bool) {
|
||||||
rest, ok := strings.CutPrefix(tag, "chain-")
|
rest, ok := strings.CutPrefix(tag, "chain-")
|
||||||
if !ok {
|
if !ok {
|
||||||
return "", false
|
return "", 0, false
|
||||||
}
|
}
|
||||||
i := strings.LastIndex(rest, "-h")
|
i := strings.LastIndex(rest, "-h")
|
||||||
if i <= 0 {
|
if i <= 0 {
|
||||||
return "", false
|
return "", 0, false
|
||||||
}
|
}
|
||||||
if _, ok := parseAllDigits(rest[i+2:]); !ok {
|
hop, ok = parseAllDigits(rest[i+2:])
|
||||||
return "", false
|
if !ok {
|
||||||
|
return "", 0, false
|
||||||
}
|
}
|
||||||
return rest[:i], true
|
return rest[:i], hop, true
|
||||||
}
|
}
|
||||||
|
|
||||||
// chainLastGroupHop finds the chain's LAST group hop — the wrapper selector
|
// chainLastGroupHop finds the chain's LAST group hop — the wrapper selector
|
||||||
@@ -507,6 +542,18 @@ func (e *Engine) testOneTarget(t groupTestTarget, used map[string]bool, obsEnabl
|
|||||||
|
|
||||||
deadline := time.Now().Add(groupTestWaitDeadline)
|
deadline := time.Now().Add(groupTestWaitDeadline)
|
||||||
for {
|
for {
|
||||||
|
// Asked BEFORE the board read, and only for a chain. The exit of a blocked
|
||||||
|
// chain does carry a fresh dead verdict now — the observatory derives it
|
||||||
|
// from the broken hop (markChainExitDead) so selection cannot be tempted
|
||||||
|
// down a path that provably does not work — but "the probe through this
|
||||||
|
// path failed" would be the wrong sentence to show a person: no probe of
|
||||||
|
// the exit ran, and the actionable fact is WHICH hop stopped it. Same
|
||||||
|
// verdict, better answer.
|
||||||
|
if idx, ok := e.blockedChainHop(t, since); ok {
|
||||||
|
res.OK = false
|
||||||
|
res.Error = fmt.Sprintf(groupTestErrChainBlockedFmt, idx)
|
||||||
|
return res
|
||||||
|
}
|
||||||
if e.readFreshObservation(&res, t, since) {
|
if e.readFreshObservation(&res, t, since) {
|
||||||
if res.OK {
|
if res.OK {
|
||||||
// The exit address is best-effort by design: see GroupTestResult.
|
// The exit address is best-effort by design: see GroupTestResult.
|
||||||
@@ -524,6 +571,57 @@ func (e *Engine) testOneTarget(t groupTestTarget, used map[string]bool, obsEnabl
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// blockedChainHop reports the hop that stopped this CHAIN target's ordered walk
|
||||||
|
// before its exit, and only when the finding is FRESH enough to answer this run.
|
||||||
|
//
|
||||||
|
// Both halves matter. The projection (chainHopHealthOf) already computes the
|
||||||
|
// break exactly once, and reading it here rather than re-deriving it keeps the
|
||||||
|
// run and the Targets card from ever naming different hops. And the freshness
|
||||||
|
// check is the same watermark readFreshObservation uses: a run rewinds the
|
||||||
|
// observatory and then reports what it measures AFTERWARDS, so a break left over
|
||||||
|
// from before the button press must not short-circuit the wait — the forced pass
|
||||||
|
// may be about to find that hop alive again.
|
||||||
|
//
|
||||||
|
// Non-chain targets, unmaterialised chains and chains with no break all report
|
||||||
|
// false, and the caller goes on waiting exactly as before.
|
||||||
|
func (e *Engine) blockedChainHop(t groupTestTarget, since time.Time) (int, bool) {
|
||||||
|
if !t.isChain {
|
||||||
|
return 0, false
|
||||||
|
}
|
||||||
|
return blockingHopOf(chainHopHealthOf(e.runningPool(), t.name, e.HealthView()), time.Now(), since)
|
||||||
|
}
|
||||||
|
|
||||||
|
// blockingHopOf is the decision itself, split out so it is testable without a
|
||||||
|
// box: given one chain's hop readout, the hop whose failure kept the EXIT from
|
||||||
|
// being dialled, and only when that failure was observed at or after since.
|
||||||
|
//
|
||||||
|
// The exit's own BlockedBy is what is consulted, not "any dead hop": a chain
|
||||||
|
// whose exit was itself probed and failed has a real end-to-end measurement, and
|
||||||
|
// that measurement is the honest answer (groupTestErrPathDead). Only when the
|
||||||
|
// exit was never reached is there nothing to wait for.
|
||||||
|
func blockingHopOf(hops []ChainHopHealth, now, since time.Time) (int, bool) {
|
||||||
|
if len(hops) == 0 {
|
||||||
|
return 0, false
|
||||||
|
}
|
||||||
|
blocker := hops[len(hops)-1].BlockedBy
|
||||||
|
if blocker == nil {
|
||||||
|
return 0, false
|
||||||
|
}
|
||||||
|
for _, h := range hops {
|
||||||
|
if h.Index != blocker.Index {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
if h.AgeSeconds < 0 {
|
||||||
|
return 0, false
|
||||||
|
}
|
||||||
|
if now.Add(-time.Duration(h.AgeSeconds) * time.Second).Before(since) {
|
||||||
|
return 0, false // a break observed before this run started; keep waiting
|
||||||
|
}
|
||||||
|
return h.Index, true
|
||||||
|
}
|
||||||
|
return 0, false
|
||||||
|
}
|
||||||
|
|
||||||
// measuredTagOf is the tag the observatory actually probes for this target's
|
// measuredTagOf is the tag the observatory actually probes for this target's
|
||||||
// dial path — the tag whose board entry answers "how is this target doing":
|
// dial path — the tag whose board entry answers "how is this target doing":
|
||||||
//
|
//
|
||||||
|
|||||||
@@ -587,3 +587,57 @@ func TestGroupTestScopeDedupedAndSorted(t *testing.T) {
|
|||||||
t.Fatalf("scope = %v, want [alpha zeta]", scope)
|
t.Fatalf("scope = %v, want [alpha zeta]", scope)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// A manual run against a chain whose walk is blocked must say so, at once and by
|
||||||
|
// hop number. The prober stops at the first dead hop, so such a chain's exit is
|
||||||
|
// never dialled and no end-to-end observation is ever coming: waiting for one
|
||||||
|
// would burn the full 120s deadline and then answer "the observatory has not
|
||||||
|
// reached this target yet" — which reads like a timing artefact about a chain
|
||||||
|
// that is, in fact, down and whose broken hop is already known.
|
||||||
|
//
|
||||||
|
// The freshness rule is the same one readFreshObservation uses: a run rewinds
|
||||||
|
// the observatory and reports what it measures AFTERWARDS, so a break left over
|
||||||
|
// from before the button press must not short-circuit the wait — the forced pass
|
||||||
|
// may be about to find that hop alive again.
|
||||||
|
func TestBlockingHopOfChainWalk(t *testing.T) {
|
||||||
|
since := time.Now()
|
||||||
|
// A 3-hop readout as the projection builds it: hop 2 dead, hop 3 not reached.
|
||||||
|
blocked := func(ageOfBreak int64) []ChainHopHealth {
|
||||||
|
blocker := &ChainHopBlock{Index: 2, Tag: "chain-c-h2"}
|
||||||
|
return []ChainHopHealth{
|
||||||
|
{Index: 1, Tag: "chain-c-h1", State: HealthAlive, AgeSeconds: 1},
|
||||||
|
{Index: 2, Tag: "chain-c-h2", State: HealthDead, AgeSeconds: ageOfBreak},
|
||||||
|
{Index: 3, Tag: "chain-c-h3", State: HealthUntested, AgeSeconds: -1, Exit: true, BlockedBy: blocker},
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// Measured after the run started: answer now, naming hop 2.
|
||||||
|
if idx, ok := blockingHopOf(blocked(0), since.Add(time.Second), since); !ok || idx != 2 {
|
||||||
|
t.Errorf("fresh break = (%d,%v), want (2,true) — the run must name the hop instead of timing out", idx, ok)
|
||||||
|
}
|
||||||
|
|
||||||
|
// Measured BEFORE the run started: keep waiting. The forced pass has not
|
||||||
|
// re-probed that hop yet, and it may be about to answer.
|
||||||
|
if idx, ok := blockingHopOf(blocked(30), since.Add(time.Second), since); ok {
|
||||||
|
t.Errorf("stale break reported as this run's finding (hop %d); a run answers with what IT measured", idx)
|
||||||
|
}
|
||||||
|
|
||||||
|
// A chain whose exit WAS reached has a real end-to-end reading; the ordinary
|
||||||
|
// path reports it (alive, or groupTestErrPathDead). Nothing to short-circuit.
|
||||||
|
reached := []ChainHopHealth{
|
||||||
|
{Index: 1, Tag: "chain-c-h1", State: HealthAlive, AgeSeconds: 1},
|
||||||
|
{Index: 2, Tag: "chain-c-h2", State: HealthDead, AgeSeconds: 0, Exit: true},
|
||||||
|
}
|
||||||
|
if _, ok := blockingHopOf(reached, since.Add(time.Second), since); ok {
|
||||||
|
t.Error("a dead EXIT was reported as a block; its own probe is the answer")
|
||||||
|
}
|
||||||
|
if _, ok := blockingHopOf(nil, since, since); ok {
|
||||||
|
t.Error("an unmaterialised chain reported a block")
|
||||||
|
}
|
||||||
|
|
||||||
|
// A non-chain target never consults any of this.
|
||||||
|
e := New()
|
||||||
|
if _, ok := e.blockedChainHop(groupTestTarget{name: "auto"}, since); ok {
|
||||||
|
t.Error("a GROUP target was treated as a blocked chain")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|||||||
+495
-11
@@ -2,6 +2,7 @@ package engine
|
|||||||
|
|
||||||
import (
|
import (
|
||||||
"context"
|
"context"
|
||||||
|
"sort"
|
||||||
"sync"
|
"sync"
|
||||||
"time"
|
"time"
|
||||||
|
|
||||||
@@ -23,12 +24,40 @@ import (
|
|||||||
// MarkFailed. The strategies read the same board, so a death found here makes the
|
// MarkFailed. The strategies read the same board, so a death found here makes the
|
||||||
// node unselectable immediately — no private overlay, no panel-only truth.
|
// node unselectable immediately — no private overlay, no panel-only truth.
|
||||||
//
|
//
|
||||||
// It never replaces a group's own probing. An ACTIVE urltest group is its own
|
// # ONE DIALLER PER TARGET
|
||||||
// fast failure detector (failover switches on its 30s probe tick), and the
|
//
|
||||||
// freshness gate below skips any tag whose newest observation is younger than the
|
// It does not probe what a urltest group already probes — not "less often",
|
||||||
// global probe interval — which is precisely the set of tags an active group is
|
// not at all. A urltest group dials its own members on its own ticker
|
||||||
// already keeping current. The observatory's budget lands on what nobody else
|
// (protocol/group/urltest.go testNodes), and for a chain's group hop that
|
||||||
// probes: idle groups' members, selector candidates, chain exits.
|
// checker is strictly the better instrument: each member copy carries the
|
||||||
|
// previous hop as its Detour, so the group's probe travels the chain prefix by
|
||||||
|
// construction, which is the path the traffic takes. Two probers on one target
|
||||||
|
// is one measurement too many however carefully they take turns, so the plan
|
||||||
|
// marks those measurements SelfChecked (probeplan.go) and this loop skips them.
|
||||||
|
//
|
||||||
|
// What is left is what no group covers, and it is the observatory's whole job:
|
||||||
|
//
|
||||||
|
// - targets with no group behind them — a chain's NODE hop wrapper, the
|
||||||
|
// AmneziaWG endpoint at L1, a chain exit that is a plain outbound. Nothing
|
||||||
|
// else in the process will ever dial these;
|
||||||
|
// - members of a SELECTOR group, which has no checker at all
|
||||||
|
// (protocol/group/selector.go), and of a urltest group whose self-check was
|
||||||
|
// stood down;
|
||||||
|
// - egress-bound group copies, whose measurement carries the base-tag alias
|
||||||
|
// that feeds the per-node stats and that no group checker writes;
|
||||||
|
// - the GATE — who is allowed to dial at all (the used-set drives SelfCheck,
|
||||||
|
// the ordered walk drives ProbeAllowed) — and the reading of the board back
|
||||||
|
// out for the panel.
|
||||||
|
//
|
||||||
|
// The freshness gate below still skips any tag whose newest observation is
|
||||||
|
// younger than the global probe interval, which now matters for the remaining
|
||||||
|
// population rather than as the mechanism that kept two probers apart.
|
||||||
|
//
|
||||||
|
// The cost of the split: a used urltest group that carries no traffic runs its
|
||||||
|
// warm-up sweep at box start and then, once its idle timeout expires, stops —
|
||||||
|
// and the observatory no longer fills in behind it, so its members age out to
|
||||||
|
// "untested". That is honest (nothing measured them) rather than wrong, and it
|
||||||
|
// is the price of one dialler per target.
|
||||||
//
|
//
|
||||||
// # Schedule and cost
|
// # Schedule and cost
|
||||||
//
|
//
|
||||||
@@ -41,6 +70,21 @@ import (
|
|||||||
// first pass starts IMMEDIATELY after an apply (cold boot: used paths validated
|
// first pass starts IMMEDIATELY after an apply (cold boot: used paths validated
|
||||||
// in seconds).
|
// in seconds).
|
||||||
//
|
//
|
||||||
|
// # A chain is walked in order, and the walk stops at the first dead hop
|
||||||
|
//
|
||||||
|
// Most of the plan is independent measurements and is fired concurrently. A
|
||||||
|
// CHAIN is not: hop i is dialled through hops 1..i-1, so a probe of hop 3 across
|
||||||
|
// a hop 2 with no live member measures hop 2's failure and learns nothing about
|
||||||
|
// hop 3. Each chain's jobs therefore run hop by hop (probeChainOrdered), and the
|
||||||
|
// walk returns at the first hop the board reads dead — every hop behind it is
|
||||||
|
// left undialled for that pass and reported untested with the blocker named,
|
||||||
|
// which is both the honest answer and the cheap one.
|
||||||
|
//
|
||||||
|
// Nothing about a block is remembered. The gate is a fresh read of the health
|
||||||
|
// board at every hop of every batch, and each cycle restarts at the top of the
|
||||||
|
// plan, so the blocking hop is always re-probed and a recovery opens the rest of
|
||||||
|
// the chain in the same pass.
|
||||||
|
//
|
||||||
// # Reconfiguration must not restart the walk
|
// # Reconfiguration must not restart the walk
|
||||||
//
|
//
|
||||||
// ConfigureObservatory runs on every successful apply, and cron reconciles every
|
// ConfigureObservatory runs on every successful apply, and cron reconciles every
|
||||||
@@ -98,6 +142,36 @@ type observatoryState struct {
|
|||||||
used map[string]bool
|
used map[string]bool
|
||||||
cursor int
|
cursor int
|
||||||
cycles uint64
|
cycles uint64
|
||||||
|
// keepWarm is the set of tags the routing config REACHES, and it answers
|
||||||
|
// ProbeWhenIdle: a group in it keeps its own probing ticker running with no
|
||||||
|
// traffic flowing.
|
||||||
|
//
|
||||||
|
// It is deliberately NOT the same field as used above, even though both hold
|
||||||
|
// a used-set. used is published only after a SUCCESSFUL apply, because the
|
||||||
|
// panel's "unused" badge must describe the config that is actually running;
|
||||||
|
// this one is published BEFORE the new box is built, because a group asks the
|
||||||
|
// question during its own PostStart. A failed apply can therefore leave this
|
||||||
|
// describing a config that never started — worst case a group keeps a ticker
|
||||||
|
// it did not need until the next apply, which nobody can see and which costs
|
||||||
|
// one probe interval. Merging the two would force one of those two timings on
|
||||||
|
// the other, and each is right for its own reader.
|
||||||
|
keepWarm map[string]bool
|
||||||
|
// hops indexes the plan's chain measurements — chain name -> hops in wire
|
||||||
|
// order -> the dial tags measuring each hop. Rebuilt from the plan whenever
|
||||||
|
// the plan is, and read (never mutated) by a tick to decide whether a hop's
|
||||||
|
// jobs may run at all. Derived state, kept rather than recomputed per tick
|
||||||
|
// only because it is derived from something that changes once per apply.
|
||||||
|
hops chainPlanIndex
|
||||||
|
// probeFn overrides how ONE planned measurement is executed. nil — the
|
||||||
|
// production value — means probeJob: resolve the dial tag against the running
|
||||||
|
// box and dial it.
|
||||||
|
//
|
||||||
|
// It exists for one assertion that cannot be made any other way. The ordered
|
||||||
|
// walk's contract is that hops below a dead one are never DIALLED; the
|
||||||
|
// resulting health readout cannot prove that, because "not probed" and
|
||||||
|
// "probed, failed, then overridden" are the same untested row. Only counting
|
||||||
|
// attempts distinguishes them, so the tests substitute a counter here.
|
||||||
|
probeFn func(ProbeJob)
|
||||||
// busy guards against a slow tick overlapping the next one.
|
// busy guards against a slow tick overlapping the next one.
|
||||||
busy bool
|
busy bool
|
||||||
// force is the ONE-SHOT "check now" flag raised by RefreshObservatory: while
|
// force is the ONE-SHOT "check now" flag raised by RefreshObservatory: while
|
||||||
@@ -127,7 +201,10 @@ func (e *Engine) ConfigureObservatory(cfg ObservatoryConfig) {
|
|||||||
e.obs.interval = interval
|
e.obs.interval = interval
|
||||||
|
|
||||||
if !cfg.Enabled {
|
if !cfg.Enabled {
|
||||||
e.obs.plan, e.obs.used, e.obs.cursor = nil, nil, 0
|
e.obs.plan, e.obs.used, e.obs.cursor, e.obs.hops = nil, nil, 0, nil
|
||||||
|
// The master switch is off: nothing probes in the background, so no group
|
||||||
|
// keeps a ticker alive on the background prober's behalf either.
|
||||||
|
e.obs.keepWarm = nil
|
||||||
e.obs.force = false
|
e.obs.force = false
|
||||||
if e.obs.stop != nil {
|
if e.obs.stop != nil {
|
||||||
close(e.obs.stop)
|
close(e.obs.stop)
|
||||||
@@ -143,6 +220,10 @@ func (e *Engine) ConfigureObservatory(cfg ObservatoryConfig) {
|
|||||||
e.obs.plan = jobs
|
e.obs.plan = jobs
|
||||||
e.obs.cursor = 0
|
e.obs.cursor = 0
|
||||||
}
|
}
|
||||||
|
// Derived from whichever plan is now installed — the kept one when the
|
||||||
|
// reconcile was a no-op, so the index can never describe a plan the cursor is
|
||||||
|
// not walking.
|
||||||
|
e.obs.hops = indexChainHops(e.obs.plan)
|
||||||
|
|
||||||
if e.obs.stop == nil {
|
if e.obs.stop == nil {
|
||||||
stop, done := make(chan struct{}), make(chan struct{})
|
stop, done := make(chan struct{}), make(chan struct{})
|
||||||
@@ -171,7 +252,8 @@ func (e *Engine) StopObservatory() {
|
|||||||
stop, done := e.obs.stop, e.obs.done
|
stop, done := e.obs.stop, e.obs.done
|
||||||
e.obs.stop, e.obs.done, e.obs.nudge = nil, nil, nil
|
e.obs.stop, e.obs.done, e.obs.nudge = nil, nil, nil
|
||||||
e.obs.enabled = false
|
e.obs.enabled = false
|
||||||
e.obs.plan, e.obs.used, e.obs.cursor = nil, nil, 0
|
e.obs.plan, e.obs.used, e.obs.cursor, e.obs.hops = nil, nil, 0, nil
|
||||||
|
e.obs.keepWarm = nil
|
||||||
e.obs.force = false
|
e.obs.force = false
|
||||||
if stop != nil {
|
if stop != nil {
|
||||||
close(stop)
|
close(stop)
|
||||||
@@ -214,6 +296,17 @@ func (e *Engine) RefreshObservatory() {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// setProbeFn installs a stand-in for the per-job dial (observatoryState.probeFn).
|
||||||
|
// It exists for the tests: the ordered walk promises that hops behind a dead one
|
||||||
|
// are never DIALLED, and only an attempt counter can show that — the health
|
||||||
|
// readout afterwards looks the same whether a job was skipped or run and
|
||||||
|
// discarded. Passing nil restores the production path.
|
||||||
|
func (e *Engine) setProbeFn(f func(ProbeJob)) {
|
||||||
|
e.obsMu.Lock()
|
||||||
|
defer e.obsMu.Unlock()
|
||||||
|
e.obs.probeFn = f
|
||||||
|
}
|
||||||
|
|
||||||
// ObservatoryStatus reports whether the observatory is on, how many measurements
|
// ObservatoryStatus reports whether the observatory is on, how many measurements
|
||||||
// the current plan holds, and how many full passes have completed. Cheap; used by
|
// the current plan holds, and how many full passes have completed. Cheap; used by
|
||||||
// tests and available for diagnostics.
|
// tests and available for diagnostics.
|
||||||
@@ -314,6 +407,11 @@ func (e *Engine) observatoryTickOnce() {
|
|||||||
batch := append([]ProbeJob(nil), e.obs.plan[start:end]...)
|
batch := append([]ProbeJob(nil), e.obs.plan[start:end]...)
|
||||||
e.obs.cursor = end
|
e.obs.cursor = end
|
||||||
interval := e.obs.interval
|
interval := e.obs.interval
|
||||||
|
hops := e.obs.hops
|
||||||
|
probe := e.obs.probeFn
|
||||||
|
if probe == nil {
|
||||||
|
probe = e.probeJob
|
||||||
|
}
|
||||||
// A forced pass (RefreshObservatory) bypasses the freshness gate for its one
|
// A forced pass (RefreshObservatory) bypasses the freshness gate for its one
|
||||||
// walk of the plan. The flag is captured for THIS batch and cleared the
|
// walk of the plan. The flag is captured for THIS batch and cleared the
|
||||||
// moment the walk's last batch is scheduled: the remaining jobs of this very
|
// moment the walk's last batch is scheduled: the remaining jobs of this very
|
||||||
@@ -350,8 +448,13 @@ func (e *Engine) observatoryTickOnce() {
|
|||||||
now := time.Now()
|
now := time.Now()
|
||||||
sem := make(chan struct{}, observatoryConcurrency)
|
sem := make(chan struct{}, observatoryConcurrency)
|
||||||
var wg sync.WaitGroup
|
var wg sync.WaitGroup
|
||||||
for _, j := range batch {
|
|
||||||
if !force && !observatoryShouldProbe(hist, j, interval, now) {
|
// Two populations, two disciplines. A job that measures nothing on a chain is
|
||||||
|
// independent of every other job and runs as it always did: fire the lot,
|
||||||
|
// bounded by the semaphore. A chain's jobs are a SERIES and run hop by hop.
|
||||||
|
free, chains := splitChainBatch(batch)
|
||||||
|
for _, j := range free {
|
||||||
|
if j.SelfChecked || (!force && !observatoryShouldProbe(hist, j, interval, now)) {
|
||||||
continue
|
continue
|
||||||
}
|
}
|
||||||
wg.Add(1)
|
wg.Add(1)
|
||||||
@@ -359,12 +462,393 @@ func (e *Engine) observatoryTickOnce() {
|
|||||||
go func(j ProbeJob) {
|
go func(j ProbeJob) {
|
||||||
defer wg.Done()
|
defer wg.Done()
|
||||||
defer func() { <-sem }()
|
defer func() { <-sem }()
|
||||||
e.probeJob(j)
|
probe(j)
|
||||||
}(j)
|
}(j)
|
||||||
}
|
}
|
||||||
|
for _, walk := range chains {
|
||||||
|
wg.Add(1)
|
||||||
|
go func(walk chainBatch) {
|
||||||
|
defer wg.Done()
|
||||||
|
e.probeChainOrdered(walk, hops, hist, interval, now, force, sem, probe)
|
||||||
|
}(walk)
|
||||||
|
}
|
||||||
wg.Wait()
|
wg.Wait()
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// probeChainOrdered runs one chain's share of a batch hop by hop, in wire order,
|
||||||
|
// and RETURNS the moment a hop below a dead one comes up — leaving every
|
||||||
|
// remaining job of that chain undialled for this pass.
|
||||||
|
//
|
||||||
|
// # Why the walk stops rather than filtering
|
||||||
|
//
|
||||||
|
// A chain is a series path: the plan's job for hop i dials through hops 1..i-1.
|
||||||
|
// If hop 2 has no live member, the hop-3 probe fails INSIDE hop 2, learns nothing
|
||||||
|
// about hop 3, and costs a full probe timeout to learn it. Recording that as
|
||||||
|
// "hop 3 is dead" is the exact class of lie this codebase keeps removing: dead is
|
||||||
|
// a positive finding about the thing that was probed, and hop 3 was never
|
||||||
|
// reached. So the walk stops, the skipped hops keep no fresh observation, and
|
||||||
|
// they surface as untested WITH the blocking hop named (grouphealth.go
|
||||||
|
// ChainHopHealth.BlockedBy) rather than as faults of their own.
|
||||||
|
//
|
||||||
|
// # Why it cannot get stuck
|
||||||
|
//
|
||||||
|
// Nothing is remembered between passes. The gate is re-read from the health board
|
||||||
|
// at every hop of every batch (blockedBefore), and the cursor rewinds to the top
|
||||||
|
// of the plan each cycle, so the FIRST dead hop is always re-probed — it is never
|
||||||
|
// itself below a break. The moment it answers again the gate opens and the rest
|
||||||
|
// of the chain is probed in the same pass. There is no "blocked" flag anywhere to
|
||||||
|
// clear, which is the point: a cached block would outlive the failure that caused
|
||||||
|
// it.
|
||||||
|
//
|
||||||
|
// # Cost
|
||||||
|
//
|
||||||
|
// Hops within one batch are serialised, so a batch holding several hops of one
|
||||||
|
// chain can take hops x probeTimeout instead of one probeTimeout. The busy guard
|
||||||
|
// in observatoryTickOnce absorbs that (a slow tick makes the next a no-op), and
|
||||||
|
// the case it exists for — a dead hop — now costs LESS than before, because
|
||||||
|
// everything behind it is skipped entirely.
|
||||||
|
func (e *Engine) probeChainOrdered(
|
||||||
|
walk chainBatch,
|
||||||
|
idx chainPlanIndex,
|
||||||
|
hist *urltest.HistoryStorage,
|
||||||
|
interval time.Duration,
|
||||||
|
now time.Time,
|
||||||
|
force bool,
|
||||||
|
sem chan struct{},
|
||||||
|
probe func(ProbeJob),
|
||||||
|
) {
|
||||||
|
for _, hop := range walk.hops {
|
||||||
|
// Re-read the board here, not once up front: the hop before this one may
|
||||||
|
// have been probed moments ago by the loop's previous iteration, and its
|
||||||
|
// verdict is exactly what decides this one.
|
||||||
|
if blocker, blocked := idx.blockedBefore(walk.chain, hop.index, e.HealthView()); blocked {
|
||||||
|
e.markChainExitDead(walk.chain, idx, blocker)
|
||||||
|
return // hops are ascending, so everything left is behind the same break
|
||||||
|
}
|
||||||
|
var wg sync.WaitGroup
|
||||||
|
for _, j := range hop.jobs {
|
||||||
|
// A self-checked job is measured by its own group's checker, which is
|
||||||
|
// gated by ProbeAllowed on the same rule this loop applies. It is
|
||||||
|
// present here only so the hop shows up in the map the gate reads;
|
||||||
|
// dialling it would be the duplicate the flag exists to remove.
|
||||||
|
if j.SelfChecked || (!force && !observatoryShouldProbe(hist, j, interval, now)) {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
wg.Add(1)
|
||||||
|
sem <- struct{}{}
|
||||||
|
go func(j ProbeJob) {
|
||||||
|
defer wg.Done()
|
||||||
|
defer func() { <-sem }()
|
||||||
|
probe(j)
|
||||||
|
}(j)
|
||||||
|
}
|
||||||
|
// The hop's verdict is only complete when every member copy has answered,
|
||||||
|
// so the next hop's gate must not be read until they have.
|
||||||
|
wg.Wait()
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// markChainExitDead records a DEAD verdict for a chain's EXIT on the shared
|
||||||
|
// health board when the ordered walk stopped short of it.
|
||||||
|
//
|
||||||
|
// # Two questions, two answers, and why they differ
|
||||||
|
//
|
||||||
|
// "Is this hop's own node alive?" and "can this chain carry traffic?" are not the
|
||||||
|
// same question, and behind a break they have different answers:
|
||||||
|
//
|
||||||
|
// - the HOP's own health is genuinely UNKNOWN. Nothing dialled it, so its card
|
||||||
|
// reads untested with BlockedBy naming the blocker (markBlockedHops), and
|
||||||
|
// that stays exactly as it is;
|
||||||
|
// - the CHAIN's health is KNOWN, and the answer is no. A chain is a series
|
||||||
|
// path; hop k was probed and did not answer — a positive finding — and every
|
||||||
|
// dial through the chain crosses hop k. The exit cannot carry traffic. That
|
||||||
|
// is not absence of knowledge, it is a conclusion drawn from a measurement
|
||||||
|
// somebody took.
|
||||||
|
//
|
||||||
|
// # Why the board must say so
|
||||||
|
//
|
||||||
|
// The board is not only the panel's data source; the strategies select on it, and
|
||||||
|
// they rank UNTESTED ABOVE DEAD on purpose (protocol/group/urltest_health_lx.go
|
||||||
|
// selectExcluding: "an untested one is a better bet than a known-dead one"). So
|
||||||
|
// leaving a provably-broken path untested is not a neutral silence — it is an
|
||||||
|
// invitation to route traffic down it, which is worse than the mislabelling this
|
||||||
|
// whole change set exists to remove. Marking it keeps
|
||||||
|
// TestChainDeadExitMarkedByObservatory's older invariant intact too: a dead chain
|
||||||
|
// is marked dead, and a dead entry stays distinguishable from a never-measured
|
||||||
|
// one.
|
||||||
|
//
|
||||||
|
// # Why this is not a fabricated verdict
|
||||||
|
//
|
||||||
|
// Nothing here claims a probe happened. It records the CONSEQUENCE of a probe
|
||||||
|
// that did happen, one hop earlier, and it is re-derived from the board every
|
||||||
|
// pass — the instant the blocking hop answers again, the walk proceeds and the
|
||||||
|
// exit's next verdict is a real measurement of its own. The flip is logged with
|
||||||
|
// its cause so the trail never reads as an unexplained death.
|
||||||
|
//
|
||||||
|
// The exit's tags are the LAST hop's measurement tags: the wrapper itself for a
|
||||||
|
// node exit, the member copies for a group exit — in both cases exactly what a
|
||||||
|
// reader consults to ask whether the chain works.
|
||||||
|
func (e *Engine) markChainExitDead(chain string, idx chainPlanIndex, blocker chainPlanHop) {
|
||||||
|
hops := idx[chain]
|
||||||
|
if len(hops) == 0 {
|
||||||
|
return
|
||||||
|
}
|
||||||
|
exit := hops[len(hops)-1]
|
||||||
|
if exit.index <= blocker.index {
|
||||||
|
return // the blocker IS the exit: its own probe already said so
|
||||||
|
}
|
||||||
|
hist := e.URLTestHistory()
|
||||||
|
if hist == nil {
|
||||||
|
return
|
||||||
|
}
|
||||||
|
ttl := e.healthTTL()
|
||||||
|
for _, tag := range exit.tags {
|
||||||
|
if e.log != nil && hist.Verdict(tag, ttl) != urltest.VerdictDead {
|
||||||
|
e.log.Info("observatory: ", tag, ": -> dead (chain hop ", blocker.index,
|
||||||
|
" was probed and did not answer, so nothing reaches the exit through it)")
|
||||||
|
}
|
||||||
|
hist.MarkFailed(tag)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// chainBatch is one chain's share of one batch, grouped into hops in wire order.
|
||||||
|
type chainBatch struct {
|
||||||
|
chain string
|
||||||
|
hops []chainBatchHop
|
||||||
|
}
|
||||||
|
|
||||||
|
// chainBatchHop is the jobs of ONE hop that fell inside the current batch. It is
|
||||||
|
// a subset: a hop with more member copies than a batch holds is split across
|
||||||
|
// consecutive batches, which is harmless — the gate is computed from the whole
|
||||||
|
// plan (chainPlanIndex), never from what happens to be in hand.
|
||||||
|
type chainBatchHop struct {
|
||||||
|
index int
|
||||||
|
jobs []ProbeJob
|
||||||
|
}
|
||||||
|
|
||||||
|
// splitChainBatch separates a batch into the chain-less jobs and the per-chain
|
||||||
|
// series. The plan is already sorted so a chain's jobs are contiguous and
|
||||||
|
// ascending, but this does not rely on that: it groups explicitly and sorts each
|
||||||
|
// chain's hops, so a batch that arrived out of order still runs in path order.
|
||||||
|
func splitChainBatch(batch []ProbeJob) (free []ProbeJob, chains []chainBatch) {
|
||||||
|
at := map[string]int{}
|
||||||
|
for _, j := range batch {
|
||||||
|
if j.Chain == "" {
|
||||||
|
free = append(free, j)
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
i, ok := at[j.Chain]
|
||||||
|
if !ok {
|
||||||
|
i = len(chains)
|
||||||
|
at[j.Chain] = i
|
||||||
|
chains = append(chains, chainBatch{chain: j.Chain})
|
||||||
|
}
|
||||||
|
c := &chains[i]
|
||||||
|
h := -1
|
||||||
|
for k := range c.hops {
|
||||||
|
if c.hops[k].index == j.Hop {
|
||||||
|
h = k
|
||||||
|
break
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if h < 0 {
|
||||||
|
c.hops = append(c.hops, chainBatchHop{index: j.Hop})
|
||||||
|
h = len(c.hops) - 1
|
||||||
|
}
|
||||||
|
c.hops[h].jobs = append(c.hops[h].jobs, j)
|
||||||
|
}
|
||||||
|
for i := range chains {
|
||||||
|
hops := chains[i].hops
|
||||||
|
sort.Slice(hops, func(a, b int) bool { return hops[a].index < hops[b].index })
|
||||||
|
}
|
||||||
|
return free, chains
|
||||||
|
}
|
||||||
|
|
||||||
|
// chainPlanIndex is the whole plan's chain measurements: chain name -> hops in
|
||||||
|
// wire order, each carrying every DIAL TAG that measures it.
|
||||||
|
//
|
||||||
|
// It is the gate's source of truth, and it must come from the PLAN rather than
|
||||||
|
// from the running box, for the same reason probeJob resolves its dial tag at
|
||||||
|
// dial time: the plan is what this pass is walking. A hop's verdict rolled up
|
||||||
|
// from a set of tags the pass does not probe would gate the pass on somebody
|
||||||
|
// else's measurements.
|
||||||
|
type chainPlanIndex map[string][]chainPlanHop
|
||||||
|
|
||||||
|
type chainPlanHop struct {
|
||||||
|
index int
|
||||||
|
tags []string
|
||||||
|
}
|
||||||
|
|
||||||
|
// indexChainHops builds the index from a plan. nil for a plan with no chain in it
|
||||||
|
// — the overwhelmingly common case, and a nil map answers every lookup with "no
|
||||||
|
// hops, nothing blocks" at no cost.
|
||||||
|
func indexChainHops(plan []ProbeJob) chainPlanIndex {
|
||||||
|
var out chainPlanIndex
|
||||||
|
for _, j := range plan {
|
||||||
|
if j.Chain == "" {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
if out == nil {
|
||||||
|
out = chainPlanIndex{}
|
||||||
|
}
|
||||||
|
hops := out[j.Chain]
|
||||||
|
at := -1
|
||||||
|
for k := range hops {
|
||||||
|
if hops[k].index == j.Hop {
|
||||||
|
at = k
|
||||||
|
break
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if at < 0 {
|
||||||
|
hops = append(hops, chainPlanHop{index: j.Hop})
|
||||||
|
at = len(hops) - 1
|
||||||
|
}
|
||||||
|
hops[at].tags = append(hops[at].tags, j.Dial)
|
||||||
|
out[j.Chain] = hops
|
||||||
|
}
|
||||||
|
for name, hops := range out {
|
||||||
|
sort.Slice(hops, func(a, b int) bool { return hops[a].index < hops[b].index })
|
||||||
|
out[name] = hops
|
||||||
|
}
|
||||||
|
return out
|
||||||
|
}
|
||||||
|
|
||||||
|
// blockedBefore reports the LOWEST hop of chain, strictly before hop, that the
|
||||||
|
// health board currently reads dead — the hop that stops the walk. Only a
|
||||||
|
// positive dead finding blocks: an untested hop in front means nothing is known
|
||||||
|
// yet, and refusing to probe on "nothing is known" would make a cold start
|
||||||
|
// permanent.
|
||||||
|
func (idx chainPlanIndex) blockedBefore(chain string, hop int, view HealthView) (chainPlanHop, bool) {
|
||||||
|
for _, h := range idx[chain] {
|
||||||
|
if h.index >= hop {
|
||||||
|
break
|
||||||
|
}
|
||||||
|
if hopBoardState(h.tags, view) == HealthDead {
|
||||||
|
return h, true
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return chainPlanHop{}, false
|
||||||
|
}
|
||||||
|
|
||||||
|
// ProbeAllowed implements urltest.ProbeGate: it answers, for one outbound tag,
|
||||||
|
// whether that outbound's OWN probing schedule may dial right now.
|
||||||
|
//
|
||||||
|
// # What it is for
|
||||||
|
//
|
||||||
|
// The observatory is not the only prober. A urltest group probes its own
|
||||||
|
// members — a warm-up sweep at box start and a ticker while traffic touches it
|
||||||
|
// — and a chain's GROUP HOP is materialised as exactly such a group
|
||||||
|
// ("chain-<n>-h<i>", generate/chain.go). Every member of that group dials
|
||||||
|
// through the hop in front of it, so when THAT hop is dead the group's sweep
|
||||||
|
// spends one probe timeout per member to rediscover the same broken hop, and
|
||||||
|
// files each result against a hop nothing reached. The projection already
|
||||||
|
// refuses to publish those readings (markBlockedHops); this stops them being
|
||||||
|
// taken.
|
||||||
|
//
|
||||||
|
// # Why this is not the SelfCheck flag
|
||||||
|
//
|
||||||
|
// standDownUnusedSelfCheck answers a different question — "does any enabled
|
||||||
|
// rule reach this group at all" — decided once, written into the config that
|
||||||
|
// reaches box.New, and true for the whole life of that box. A chain hop wrapper
|
||||||
|
// IS reached by the rules and must keep its flag (TestStandDownNeverTouchesChainHopWrappers
|
||||||
|
// pins that). This answers "is the path in front of it working AT THIS MOMENT",
|
||||||
|
// which changes on its own and must therefore be asked, never stored. Merging
|
||||||
|
// the two would either put a live question in a static flag or take a used
|
||||||
|
// chain permanently off the board.
|
||||||
|
//
|
||||||
|
// # Why it cannot get stuck
|
||||||
|
//
|
||||||
|
// It holds nothing. Every call recomputes the block from the same plan index
|
||||||
|
// and the same health board the observatory's own walk uses, so the moment the
|
||||||
|
// blocking hop answers a probe this returns true again and the wrapper's next
|
||||||
|
// tick dials normally. The blocking hop is itself never behind a break, so the
|
||||||
|
// observatory re-probes it every cycle: there is always something re-asking the
|
||||||
|
// question that produced the block.
|
||||||
|
//
|
||||||
|
// It answers TRUE whenever it does not know — no plan, observatory disabled,
|
||||||
|
// not a chain hop tag, chain absent from the plan. Refusing on missing
|
||||||
|
// information would silence probing exactly when least is known, and nothing
|
||||||
|
// would ever measure its way back out.
|
||||||
|
func (e *Engine) ProbeAllowed(tag string) bool {
|
||||||
|
chain, hop, ok := parseChainHopTag(tag)
|
||||||
|
if !ok {
|
||||||
|
return true // not a chain hop wrapper: nothing sits in front of it
|
||||||
|
}
|
||||||
|
e.obsMu.Lock()
|
||||||
|
enabled, idx := e.obs.enabled, e.obs.hops
|
||||||
|
e.obsMu.Unlock()
|
||||||
|
if !enabled || idx == nil {
|
||||||
|
return true
|
||||||
|
}
|
||||||
|
// HealthView takes obsMu itself (healthTTL), so it must be called with the
|
||||||
|
// lock released.
|
||||||
|
_, blocked := idx.blockedBefore(chain, hop, e.HealthView())
|
||||||
|
return !blocked
|
||||||
|
}
|
||||||
|
|
||||||
|
// ProbeWhenIdle implements urltest.ProbeGate: it reports whether an outbound
|
||||||
|
// must keep measuring itself with no traffic flowing through it.
|
||||||
|
//
|
||||||
|
// TRUE for anything the routing configuration reaches. Such a group's health is
|
||||||
|
// a live question at all times — a rule matching a narrow domain list is in
|
||||||
|
// force whether or not it fired in the last half hour — and without this its own
|
||||||
|
// ticker would retire on the idle timeout and NOTHING would replace it: the
|
||||||
|
// observatory stands off a urltest group's members by design (ProbeJob.SelfChecked).
|
||||||
|
// The panel would then report "untested" about an armed rule, and the first
|
||||||
|
// request through it would pay a cold probe instead of picking a member already
|
||||||
|
// known to work. Keeping the group's own ticker running restores exactly what
|
||||||
|
// the observatory used to spend on those same targets, at the same interval, and
|
||||||
|
// keeps it to ONE dialler.
|
||||||
|
//
|
||||||
|
// FALSE whenever nothing is known — no config applied yet, unknown tag, master
|
||||||
|
// switch off. This method adds work rather than withholding it, so the direction
|
||||||
|
// of "don't know" is the opposite of ProbeAllowed's: claiming keep-warm on
|
||||||
|
// missing information would leave groups probing forever for no stated reason.
|
||||||
|
//
|
||||||
|
// It does NOT bypass any other refusal. A stood-down group never asks (the
|
||||||
|
// group's own keepWarm short-circuits on selfCheckDisabled), and a kept-warm
|
||||||
|
// group behind a dead hop still ticks WITHOUT dialling, because every tick goes
|
||||||
|
// through scheduledCheck -> ProbeAllowed. Warm means measured on schedule, never
|
||||||
|
// "hammering through a broken path".
|
||||||
|
func (e *Engine) ProbeWhenIdle(tag string) bool {
|
||||||
|
e.obsMu.Lock()
|
||||||
|
defer e.obsMu.Unlock()
|
||||||
|
return e.obs.keepWarm[tag]
|
||||||
|
}
|
||||||
|
|
||||||
|
// publishKeepWarm installs the set ProbeWhenIdle answers from. Called from
|
||||||
|
// applyLocked with the used-set of the config being applied, and cleared when
|
||||||
|
// background probing is switched off — see the keepWarm field for why this is
|
||||||
|
// published on a different schedule from the observatory's own used-set.
|
||||||
|
func (e *Engine) publishKeepWarm(used map[string]bool) {
|
||||||
|
e.obsMu.Lock()
|
||||||
|
e.obs.keepWarm = used
|
||||||
|
e.obsMu.Unlock()
|
||||||
|
}
|
||||||
|
|
||||||
|
// hopBoardState rolls a hop's measurement tags into one verdict, by exactly the
|
||||||
|
// rule chainGroupHop (grouphealth.go) uses for the same question — "can this hop
|
||||||
|
// carry the chain": alive as soon as anything answers, dead only when something
|
||||||
|
// was tested and nothing answers, untested when nothing was tested at all. The
|
||||||
|
// prober and the panel must agree about which hop broke a chain, so this is that
|
||||||
|
// rule written once more against tags instead of against a materialised group;
|
||||||
|
// TestChainHopGateMatchesProjection pins the two together.
|
||||||
|
func hopBoardState(tags []string, view HealthView) string {
|
||||||
|
tested := 0
|
||||||
|
for _, tag := range tags {
|
||||||
|
switch state, _, _ := view.State(tag); state {
|
||||||
|
case HealthAlive:
|
||||||
|
return HealthAlive
|
||||||
|
case HealthDead:
|
||||||
|
tested++
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if tested > 0 {
|
||||||
|
return HealthDead
|
||||||
|
}
|
||||||
|
return HealthUntested
|
||||||
|
}
|
||||||
|
|
||||||
// observatoryShouldProbe is the freshness gate: a job runs when ANY tag it
|
// observatoryShouldProbe is the freshness gate: a job runs when ANY tag it
|
||||||
// covers has no observation at all, or its newest observation — success OR
|
// covers has no observation at all, or its newest observation — success OR
|
||||||
// failure — is older than the global probe interval.
|
// failure — is older than the global probe interval.
|
||||||
|
|||||||
File diff suppressed because it is too large
Load Diff
+221
-25
@@ -51,6 +51,28 @@ import (
|
|||||||
// ever being dialled by the observatory. Chain copies are recorded only under
|
// ever being dialled by the observatory. Chain copies are recorded only under
|
||||||
// themselves: a chain prefix is nobody else's path.
|
// themselves: a chain prefix is nobody else's path.
|
||||||
//
|
//
|
||||||
|
// # Who dials
|
||||||
|
//
|
||||||
|
// Reaching a target does not make it OURS to dial. A member of a urltest group
|
||||||
|
// already has a prober — that group's own checker — and for a chain hop that
|
||||||
|
// checker measures the better path: the member copy carries the previous hop as
|
||||||
|
// its Detour, so its probe travels the chain prefix without anything having to
|
||||||
|
// be arranged. Those measurements are emitted with SelfChecked set and the
|
||||||
|
// observatory never dials them. What stays ours is what no group probes: node
|
||||||
|
// hop wrappers, endpoints, selector members, stood-down groups' members, and
|
||||||
|
// egress copies (whose base-tag alias no checker writes).
|
||||||
|
//
|
||||||
|
// # Order
|
||||||
|
//
|
||||||
|
// The plan is a LIST, not a set, and for a chain the order is load-bearing. Each
|
||||||
|
// job that measures a chain carries the chain name and its 1-based hop index
|
||||||
|
// (ProbeJob.Chain/Hop, threaded through the walk as planPos), and the emitted
|
||||||
|
// plan is sorted so a chain's jobs come out grouped and ascending by hop. The
|
||||||
|
// observatory walks them in that order and stops at the first dead hop —
|
||||||
|
// probing hop 3 through a hop 2 with no live member measures hop 2's failure and
|
||||||
|
// nothing else, so recording it against hop 3 names the wrong suspect and burns a
|
||||||
|
// probe timeout doing it.
|
||||||
|
//
|
||||||
// # Dedup
|
// # Dedup
|
||||||
//
|
//
|
||||||
// The probe URL is global (plan §5.D), so the dedup key is the dial path alone:
|
// The probe URL is global (plan §5.D), so the dedup key is the dial path alone:
|
||||||
@@ -69,6 +91,42 @@ type ProbeJob struct {
|
|||||||
Dial string
|
Dial string
|
||||||
URL string
|
URL string
|
||||||
Store []string
|
Store []string
|
||||||
|
// Chain and Hop place this measurement ON a chain's path: the chain NAME and
|
||||||
|
// the 1-based hop index it measures. Both zero ("" / 0) for every job that is
|
||||||
|
// not part of a chain, which is most of them.
|
||||||
|
//
|
||||||
|
// This is what makes the walk ORDERED (observatory.go). A chain is a series
|
||||||
|
// path, so the probe of hop 3 dials THROUGH hops 1 and 2; if hop 2 has no live
|
||||||
|
// member, that probe fails at hop 2 and says nothing whatsoever about hop 3.
|
||||||
|
// Recording it as a verdict about hop 3 was the defect: it pointed the
|
||||||
|
// operator at the wrong hop and spent a full probe timeout to do it. The
|
||||||
|
// prober therefore walks a chain h1 -> hN and stops at the first dead hop, and
|
||||||
|
// these two fields are the only thing that tells it which jobs form which hop
|
||||||
|
// of which path.
|
||||||
|
//
|
||||||
|
// A hop is usually SEVERAL jobs — a group hop's member copies are one
|
||||||
|
// measurement each — so Hop is a grouping key, never a job's identity.
|
||||||
|
Chain string
|
||||||
|
Hop int
|
||||||
|
// SelfChecked marks a measurement whose target already has a dialler: it is a
|
||||||
|
// member of a URLTEST group, and that group's own checker probes exactly this
|
||||||
|
// outbound, on exactly this dial path, on the same global interval
|
||||||
|
// (protocol/group/urltest.go testNodes). The observatory therefore does NOT
|
||||||
|
// dial it. One target, one dialler.
|
||||||
|
//
|
||||||
|
// The group's checker is the RIGHT dialler for these, not merely an equal
|
||||||
|
// one: a chain hop member carries the Detour of the hop in front of it, so
|
||||||
|
// the checker's probe travels the chain prefix by construction, which is the
|
||||||
|
// path the traffic takes. Nothing had to be arranged for that to be true.
|
||||||
|
//
|
||||||
|
// The job STAYS in the plan rather than being dropped, because the plan is
|
||||||
|
// two things at once: the observatory's dial list AND the map of which tags
|
||||||
|
// measure which hop of which chain. The ordered walk's gate reads the second
|
||||||
|
// (indexChainHops -> blockedBefore), and it must see every measurement of a
|
||||||
|
// hop, not only the ones this prober performs — a hop whose members are all
|
||||||
|
// self-checked would otherwise vanish from the map and stop being able to
|
||||||
|
// block anything.
|
||||||
|
SelfChecked bool
|
||||||
}
|
}
|
||||||
|
|
||||||
// BuildObservatoryPlan derives the observatory's probe plan from the applied
|
// BuildObservatoryPlan derives the observatory's probe plan from the applied
|
||||||
@@ -81,7 +139,7 @@ type ProbeJob struct {
|
|||||||
// probeURL is the global probe URL; "" lets urltest fall back to its gstatic
|
// probeURL is the global probe URL; "" lets urltest fall back to its gstatic
|
||||||
// default, exactly as the generator does.
|
// default, exactly as the generator does.
|
||||||
//
|
//
|
||||||
// The result is deterministic (sorted by Dial) so a cursor walking it across
|
// The result is deterministic (see probeJobLess) so a cursor walking it across
|
||||||
// ticks is stable, and plansEqual can tell a no-op reconcile from real change.
|
// ticks is stable, and plansEqual can tell a no-op reconcile from real change.
|
||||||
func BuildObservatoryPlan(opts option.Options, probeURL string) ([]ProbeJob, map[string]bool) {
|
func BuildObservatoryPlan(opts option.Options, probeURL string) ([]ProbeJob, map[string]bool) {
|
||||||
w := &planWalk{
|
w := &planWalk{
|
||||||
@@ -92,7 +150,9 @@ func BuildObservatoryPlan(opts option.Options, probeURL string) ([]ProbeJob, map
|
|||||||
targets: map[string]*planTarget{},
|
targets: map[string]*planTarget{},
|
||||||
}
|
}
|
||||||
for _, root := range planRoots(opts) {
|
for _, root := range planRoots(opts) {
|
||||||
w.visit(root)
|
// A root comes from a routing rule, not from a group, so nothing is
|
||||||
|
// checking it on our behalf.
|
||||||
|
w.visit(root, planPos{}, false)
|
||||||
}
|
}
|
||||||
|
|
||||||
jobs := make([]ProbeJob, 0, len(w.targets))
|
jobs := make([]ProbeJob, 0, len(w.targets))
|
||||||
@@ -102,18 +162,49 @@ func BuildObservatoryPlan(opts option.Options, probeURL string) ([]ProbeJob, map
|
|||||||
store = append(store, tag)
|
store = append(store, tag)
|
||||||
}
|
}
|
||||||
sort.Strings(store)
|
sort.Strings(store)
|
||||||
jobs = append(jobs, ProbeJob{Dial: t.dial, URL: probeURL, Store: store})
|
jobs = append(jobs, ProbeJob{
|
||||||
|
Dial: t.dial, URL: probeURL, Store: store,
|
||||||
|
Chain: t.pos.chain, Hop: t.pos.hop, SelfChecked: t.selfChecked,
|
||||||
|
})
|
||||||
}
|
}
|
||||||
sort.Slice(jobs, func(i, j int) bool { return jobs[i].Dial < jobs[j].Dial })
|
sort.Slice(jobs, func(i, j int) bool { return probeJobLess(jobs[i], jobs[j]) })
|
||||||
return jobs, w.used
|
return jobs, w.used
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// probeJobLess is the plan's order, and it is a CONTRACT rather than a
|
||||||
|
// presentation detail: a chain's jobs come out grouped by chain and ascending by
|
||||||
|
// hop, so a cursor walking the plan meets h1 before h2 before h3. The observatory
|
||||||
|
// relies on that to stop at the first dead hop (see probeChainOrdered); a plan
|
||||||
|
// sorted by dial tag alone would interleave the hops of one path and there would
|
||||||
|
// be no "first" to stop at.
|
||||||
|
//
|
||||||
|
// Chain-less jobs (Chain "") sort first, among themselves by dial tag, exactly as
|
||||||
|
// the whole plan used to. Ties inside one hop break on the dial tag, so the order
|
||||||
|
// is total and two builds of one config are byte-identical — the property
|
||||||
|
// ConfigureObservatory's cursor keeping depends on.
|
||||||
|
func probeJobLess(a, b ProbeJob) bool {
|
||||||
|
if a.Chain != b.Chain {
|
||||||
|
return a.Chain < b.Chain
|
||||||
|
}
|
||||||
|
if a.Hop != b.Hop {
|
||||||
|
return a.Hop < b.Hop
|
||||||
|
}
|
||||||
|
return a.Dial < b.Dial
|
||||||
|
}
|
||||||
|
|
||||||
// planEntry is the slice of one configured outbound/endpoint the planner needs.
|
// planEntry is the slice of one configured outbound/endpoint the planner needs.
|
||||||
type planEntry struct {
|
type planEntry struct {
|
||||||
typ string
|
typ string
|
||||||
group bool // selector/urltest — expanded, never dialled
|
group bool // selector/urltest — expanded, never dialled
|
||||||
members []string // group members, in config order
|
members []string // group members, in config order
|
||||||
detour string // DialerOptions.Detour ("" when none)
|
detour string // DialerOptions.Detour ("" when none)
|
||||||
|
// selfChecks is true for a group that PROBES ITS OWN MEMBERS: a urltest
|
||||||
|
// group whose self-check has not been stood down. A selector never does (it
|
||||||
|
// has no checker at all — protocol/group/selector.go), and a urltest group
|
||||||
|
// with SelfCheck=false has had its schedule taken away deliberately
|
||||||
|
// (standDownUnusedSelfCheck), so in both of those cases the observatory is
|
||||||
|
// the only possible dialler and must remain one.
|
||||||
|
selfChecks bool
|
||||||
}
|
}
|
||||||
|
|
||||||
// planIndex flattens opts into tag -> planEntry. Endpoints (wireguard/AmneziaWG)
|
// planIndex flattens opts into tag -> planEntry. Endpoints (wireguard/AmneziaWG)
|
||||||
@@ -123,10 +214,11 @@ func planIndex(opts option.Options) map[string]planEntry {
|
|||||||
byTag := make(map[string]planEntry, len(opts.Outbounds)+len(opts.Endpoints))
|
byTag := make(map[string]planEntry, len(opts.Outbounds)+len(opts.Endpoints))
|
||||||
for _, ob := range opts.Outbounds {
|
for _, ob := range opts.Outbounds {
|
||||||
byTag[ob.Tag] = planEntry{
|
byTag[ob.Tag] = planEntry{
|
||||||
typ: ob.Type,
|
typ: ob.Type,
|
||||||
group: ob.Type == C.TypeSelector || ob.Type == C.TypeURLTest,
|
group: ob.Type == C.TypeSelector || ob.Type == C.TypeURLTest,
|
||||||
members: groupMembersOf(ob.Options),
|
members: groupMembersOf(ob.Options),
|
||||||
detour: detourOf(ob.Options),
|
detour: detourOf(ob.Options),
|
||||||
|
selfChecks: ob.Type == C.TypeURLTest && urltestSelfChecks(ob.Options),
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
for _, ep := range opts.Endpoints {
|
for _, ep := range opts.Endpoints {
|
||||||
@@ -192,6 +284,23 @@ func groupMembersOf(options any) []string {
|
|||||||
return nil
|
return nil
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// urltestSelfChecks reports whether a urltest group still runs its OWN probing
|
||||||
|
// schedule. nil/absent SelfCheck means on — the documented compatibility default
|
||||||
|
// (option.URLTestOutboundOptions.SelfCheck) — and only an explicit false stands
|
||||||
|
// it down. An options struct of an unexpected shape is read as "on", the same
|
||||||
|
// direction standDownUnusedSelfCheck refuses to guess in: assuming a checker
|
||||||
|
// exists when it does not would leave a target with NO dialler, which is worse
|
||||||
|
// than one dialled twice.
|
||||||
|
func urltestSelfChecks(options any) bool {
|
||||||
|
switch o := options.(type) {
|
||||||
|
case *option.URLTestOutboundOptions:
|
||||||
|
return o.SelfCheck == nil || *o.SelfCheck
|
||||||
|
case option.URLTestOutboundOptions:
|
||||||
|
return o.SelfCheck == nil || *o.SelfCheck
|
||||||
|
}
|
||||||
|
return true
|
||||||
|
}
|
||||||
|
|
||||||
// detourOf reads DialerOptions.Detour out of any outbound/endpoint/dns-server
|
// detourOf reads DialerOptions.Detour out of any outbound/endpoint/dns-server
|
||||||
// option struct that carries dialer options (they all embed option.DialerOptions
|
// option struct that carries dialer options (they all embed option.DialerOptions
|
||||||
// and therefore satisfy DialerOptionsWrapper); "" for everything else.
|
// and therefore satisfy DialerOptionsWrapper); "" for everything else.
|
||||||
@@ -215,11 +324,40 @@ func probeablePlanTag(typ, tag string) bool {
|
|||||||
return !strings.HasPrefix(tag, "egress-")
|
return !strings.HasPrefix(tag, "egress-")
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// planPos is where a measurement sits on a chain: the chain name and the 1-based
|
||||||
|
// hop index. The zero value means "not on any chain", which is what every job
|
||||||
|
// outside a chain carries.
|
||||||
|
//
|
||||||
|
// It is threaded through the walk rather than parsed back out of each tag,
|
||||||
|
// because only the walk knows the STRUCTURE. A hop wrapper's tag names its own
|
||||||
|
// position ("chain-<n>-h<i>"), but a group hop's member copies
|
||||||
|
// ("chain-<n>-h<i>-<member>") end in free-form operator text: a node named
|
||||||
|
// "h2-eu" would make the tag ambiguous to any parser. Inheriting the position
|
||||||
|
// from the wrapper the walk arrived through has no such failure mode.
|
||||||
|
type planPos struct {
|
||||||
|
chain string
|
||||||
|
hop int
|
||||||
|
}
|
||||||
|
|
||||||
|
// chainPosOf reads a hop WRAPPER tag's own position, or the zero position for
|
||||||
|
// anything else.
|
||||||
|
func chainPosOf(tag string) planPos {
|
||||||
|
if name, hop, ok := parseChainHopTag(tag); ok {
|
||||||
|
return planPos{chain: name, hop: hop}
|
||||||
|
}
|
||||||
|
return planPos{}
|
||||||
|
}
|
||||||
|
|
||||||
// planTarget accumulates one dial path's job: the tag to dial (the smallest
|
// planTarget accumulates one dial path's job: the tag to dial (the smallest
|
||||||
// dialable tag, for determinism) and every tag the measurement is recorded under.
|
// dialable tag, for determinism), every tag the measurement is recorded under,
|
||||||
|
// and where on a chain it sits (planPos).
|
||||||
type planTarget struct {
|
type planTarget struct {
|
||||||
dial string
|
dial string
|
||||||
store map[string]bool
|
store map[string]bool
|
||||||
|
pos planPos
|
||||||
|
// selfChecked: some group's own checker already dials this path, so the
|
||||||
|
// observatory must not. See ProbeJob.SelfChecked.
|
||||||
|
selfChecked bool
|
||||||
}
|
}
|
||||||
|
|
||||||
// planWalk is the reachability expansion state.
|
// planWalk is the reachability expansion state.
|
||||||
@@ -234,13 +372,30 @@ type planWalk struct {
|
|||||||
// addTarget registers tag as a measurement under the dial-path key, recording it
|
// addTarget registers tag as a measurement under the dial-path key, recording it
|
||||||
// under itself plus aliases. When several dialable tags share one path (copies of
|
// under itself plus aliases. When several dialable tags share one path (copies of
|
||||||
// one node through one egress across groups) the smallest dials, deterministically.
|
// one node through one egress across groups) the smallest dials, deterministically.
|
||||||
func (w *planWalk) addTarget(key, tag string, aliases ...string) {
|
//
|
||||||
|
// pos is the chain position and is taken from the FIRST arrival only. Two
|
||||||
|
// arrivals at one dial-path key are copies of the same node through the same
|
||||||
|
// egress (see the file header), which live outside every chain and therefore
|
||||||
|
// carry the zero position anyway; a chain tag's dial key is its own tag, so it is
|
||||||
|
// reached once by construction. Never overwriting means a shared key cannot have
|
||||||
|
// its position depend on walk order.
|
||||||
|
//
|
||||||
|
// selfChecked, by contrast, is ANDed across arrivals, and the asymmetry is
|
||||||
|
// deliberate: it claims "somebody else already dials this", and one path that
|
||||||
|
// reaches the target WITHOUT a checker in front of it (a selector member, a
|
||||||
|
// stood-down group) falsifies the claim outright. Merging the pessimistic way
|
||||||
|
// can only cost a duplicate probe; merging the optimistic way would leave a
|
||||||
|
// target with no dialler at all.
|
||||||
|
func (w *planWalk) addTarget(key string, pos planPos, selfChecked bool, tag string, aliases ...string) {
|
||||||
t := w.targets[key]
|
t := w.targets[key]
|
||||||
if t == nil {
|
if t == nil {
|
||||||
t = &planTarget{dial: tag, store: map[string]bool{}}
|
t = &planTarget{dial: tag, store: map[string]bool{}, pos: pos, selfChecked: selfChecked}
|
||||||
w.targets[key] = t
|
w.targets[key] = t
|
||||||
} else if tag < t.dial {
|
} else {
|
||||||
t.dial = tag
|
if tag < t.dial {
|
||||||
|
t.dial = tag
|
||||||
|
}
|
||||||
|
t.selfChecked = t.selfChecked && selfChecked
|
||||||
}
|
}
|
||||||
t.store[tag] = true
|
t.store[tag] = true
|
||||||
for _, a := range aliases {
|
for _, a := range aliases {
|
||||||
@@ -251,7 +406,15 @@ func (w *planWalk) addTarget(key, tag string, aliases ...string) {
|
|||||||
// visit expands one root/member reference. Groups recurse into their members;
|
// visit expands one root/member reference. Groups recurse into their members;
|
||||||
// leaves become measurements; a leaf's Detour link is then walked to surface the
|
// leaves become measurements; a leaf's Detour link is then walked to surface the
|
||||||
// group hops of a chain (walkDetour).
|
// group hops of a chain (walkDetour).
|
||||||
func (w *planWalk) visit(tag string) {
|
//
|
||||||
|
// pos is the chain position INHERITED from the group this reference was expanded
|
||||||
|
// out of; a tag that names its own hop position (a chain hop wrapper, typically
|
||||||
|
// the exit reached straight from a rule) overrides it.
|
||||||
|
//
|
||||||
|
// selfChecked says the group this reference came out of probes its own members,
|
||||||
|
// so the leaf reached here already has a dialler. A ROOT is reached from a
|
||||||
|
// routing rule, which has no checker behind it, so roots start false.
|
||||||
|
func (w *planWalk) visit(tag string, pos planPos, selfChecked bool) {
|
||||||
if tag == "" || w.visited[tag] {
|
if tag == "" || w.visited[tag] {
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
@@ -261,14 +424,17 @@ func (w *planWalk) visit(tag string) {
|
|||||||
return // dangling reference; box.New would have refused it anyway
|
return // dangling reference; box.New would have refused it anyway
|
||||||
}
|
}
|
||||||
w.used[tag] = true
|
w.used[tag] = true
|
||||||
|
if own := chainPosOf(tag); own.chain != "" {
|
||||||
|
pos = own
|
||||||
|
}
|
||||||
if ent.group {
|
if ent.group {
|
||||||
for _, m := range ent.members {
|
for _, m := range ent.members {
|
||||||
w.visitMember(tag, m)
|
w.visitMember(tag, m, pos, ent.selfChecks)
|
||||||
}
|
}
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
if probeablePlanTag(ent.typ, tag) {
|
if probeablePlanTag(ent.typ, tag) {
|
||||||
w.addTarget(dialKeyTag(tag), tag)
|
w.addTarget(dialKeyTag(tag), pos, selfChecked, tag)
|
||||||
}
|
}
|
||||||
w.walkDetour(ent.detour)
|
w.walkDetour(ent.detour)
|
||||||
}
|
}
|
||||||
@@ -277,19 +443,30 @@ func (w *planWalk) visit(tag string) {
|
|||||||
// ("group-<group>-m<i>-<member>") is keyed by its (node, egress) dial path — so
|
// ("group-<group>-m<i>-<member>") is keyed by its (node, egress) dial path — so
|
||||||
// two groups bound to one egress share the measurement — and recorded under the
|
// two groups bound to one egress share the measurement — and recorded under the
|
||||||
// copy AND the base node tag (the store alias). Every other member is expanded
|
// copy AND the base node tag (the store alias). Every other member is expanded
|
||||||
// exactly like a root.
|
// exactly like a root, carrying the owning group's chain position down with it:
|
||||||
func (w *planWalk) visitMember(group, member string) {
|
// the members of a chain's exit group hop are measurements OF that hop.
|
||||||
|
//
|
||||||
|
// The EGRESS COPY branch is deliberately never marked self-checked, even when
|
||||||
|
// the owning group runs its own checker. Such a job carries a second store tag —
|
||||||
|
// the base node tag, the alias that keeps the per-node stats surface filled in
|
||||||
|
// without the base outbound (a different dial path) ever being dialled. The
|
||||||
|
// group's checker records under the copy tag ALONE (protocol/group/urltest.go
|
||||||
|
// stores under RealTag), so handing this measurement over would silently drop
|
||||||
|
// the alias and empty /api/stats node_health for every egress-bound node. That
|
||||||
|
// is a real duplicate probe knowingly kept, and the alternative is losing data
|
||||||
|
// no other writer produces.
|
||||||
|
func (w *planWalk) visitMember(group, member string, pos planPos, ownerSelfChecks bool) {
|
||||||
ent, ok := w.byTag[member]
|
ent, ok := w.byTag[member]
|
||||||
if !ok {
|
if !ok {
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
if node, isCopy := parseGroupCopyTag(group, member); isCopy && !ent.group {
|
if node, isCopy := parseGroupCopyTag(group, member); isCopy && !ent.group {
|
||||||
w.used[member] = true
|
w.used[member] = true
|
||||||
w.addTarget(dialKeyCopy(node, ent.detour), member, node)
|
w.addTarget(dialKeyCopy(node, ent.detour), pos, false, member, node)
|
||||||
w.walkDetour(ent.detour)
|
w.walkDetour(ent.detour)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
w.visit(member)
|
w.visit(member, pos, ownerSelfChecks)
|
||||||
}
|
}
|
||||||
|
|
||||||
// walkDetour follows a probed tag's Detour links toward the router. A group met
|
// walkDetour follows a probed tag's Detour links toward the router. A group met
|
||||||
@@ -319,9 +496,15 @@ func (w *planWalk) walkDetour(tag string) {
|
|||||||
return
|
return
|
||||||
}
|
}
|
||||||
w.used[tag] = true
|
w.used[tag] = true
|
||||||
|
// A tag met on a detour path names its own position or has none: the walk
|
||||||
|
// arrived here from BELOW (a later hop detouring back), so there is nothing
|
||||||
|
// sensible to inherit.
|
||||||
|
pos := chainPosOf(tag)
|
||||||
if !ent.group {
|
if !ent.group {
|
||||||
if _, isHopWrapper := parseChainExitTag(tag); isHopWrapper && probeablePlanTag(ent.typ, tag) {
|
// A NODE hop wrapper is nobody's member, so no checker exists for it and
|
||||||
w.addTarget(dialKeyTag(tag), tag)
|
// the observatory is its only possible dialler.
|
||||||
|
if pos.chain != "" && probeablePlanTag(ent.typ, tag) {
|
||||||
|
w.addTarget(dialKeyTag(tag), pos, false, tag)
|
||||||
}
|
}
|
||||||
w.walkDetour(ent.detour)
|
w.walkDetour(ent.detour)
|
||||||
return
|
return
|
||||||
@@ -334,7 +517,11 @@ func (w *planWalk) walkDetour(tag string) {
|
|||||||
}
|
}
|
||||||
w.used[m] = true
|
w.used[m] = true
|
||||||
if !ment.group && probeablePlanTag(ment.typ, m) {
|
if !ment.group && probeablePlanTag(ment.typ, m) {
|
||||||
w.addTarget(dialKeyTag(m), m)
|
// A GROUP hop's members are that wrapper's members: when the wrapper
|
||||||
|
// is a urltest group it probes them itself, and it probes them along
|
||||||
|
// the chain prefix, because each copy carries the previous hop as its
|
||||||
|
// Detour. A selector wrapper has no checker, so those stay ours.
|
||||||
|
w.addTarget(dialKeyTag(m), pos, ent.selfChecks, m)
|
||||||
}
|
}
|
||||||
if next == "" {
|
if next == "" {
|
||||||
next = ment.detour // every member detours into the same previous hop
|
next = ment.detour // every member detours into the same previous hop
|
||||||
@@ -361,8 +548,14 @@ func dialKeyCopy(node, egress string) string {
|
|||||||
func lenPrefixed(s string) string { return fmt.Sprintf("%d:%s", len(s), s) }
|
func lenPrefixed(s string) string { return fmt.Sprintf("%d:%s", len(s), s) }
|
||||||
|
|
||||||
// plansEqual reports whether two plans describe exactly the same measurements —
|
// plansEqual reports whether two plans describe exactly the same measurements —
|
||||||
// same jobs, same order, same dial tag, same URL, same store set. This is what
|
// same jobs, same order, same dial tag, same URL, same store set, same chain
|
||||||
// lets a no-op reconcile keep the observatory's cursor (see ConfigureObservatory).
|
// position. This is what lets a no-op reconcile keep the observatory's cursor
|
||||||
|
// (see ConfigureObservatory).
|
||||||
|
//
|
||||||
|
// The chain position is part of the comparison because it is part of the PLAN,
|
||||||
|
// not an annotation on it: it decides the order jobs run in and which of them the
|
||||||
|
// ordered walk may skip. A config that moves a node from hop 2 to hop 3 without
|
||||||
|
// changing any dial tag is a different plan and must restart the cursor.
|
||||||
func plansEqual(a, b []ProbeJob) bool {
|
func plansEqual(a, b []ProbeJob) bool {
|
||||||
if len(a) != len(b) {
|
if len(a) != len(b) {
|
||||||
return false
|
return false
|
||||||
@@ -371,6 +564,9 @@ func plansEqual(a, b []ProbeJob) bool {
|
|||||||
if a[i].Dial != b[i].Dial || a[i].URL != b[i].URL || len(a[i].Store) != len(b[i].Store) {
|
if a[i].Dial != b[i].Dial || a[i].URL != b[i].URL || len(a[i].Store) != len(b[i].Store) {
|
||||||
return false
|
return false
|
||||||
}
|
}
|
||||||
|
if a[i].Chain != b[i].Chain || a[i].Hop != b[i].Hop || a[i].SelfChecked != b[i].SelfChecked {
|
||||||
|
return false
|
||||||
|
}
|
||||||
for j := range a[i].Store {
|
for j := range a[i].Store {
|
||||||
if a[i].Store[j] != b[i].Store[j] {
|
if a[i].Store[j] != b[i].Store[j] {
|
||||||
return false
|
return false
|
||||||
|
|||||||
@@ -26,6 +26,29 @@ import (
|
|||||||
// EMITTED rules and DNS detours as a whole, and BuildObservatoryPlan is the one
|
// EMITTED rules and DNS detours as a whole, and BuildObservatoryPlan is the one
|
||||||
// place that reachability is computed. Recomputing it in the generator would be
|
// place that reachability is computed. Recomputing it in the generator would be
|
||||||
// a second copy of the same walk, and second copies drift.
|
// a second copy of the same walk, and second copies drift.
|
||||||
|
//
|
||||||
|
// # NOT the same thing as Engine.ProbeAllowed — do not merge them
|
||||||
|
//
|
||||||
|
// There are two reasons a group's own schedule stands down, and they are
|
||||||
|
// answered in two different places on purpose:
|
||||||
|
//
|
||||||
|
// THIS FILE — "no enabled rule reaches this group AT ALL". A property of the
|
||||||
|
// CONFIG. Decided once per apply, written into the option struct that
|
||||||
|
// reaches box.New, and unchangeable for the life of that box, because the
|
||||||
|
// rules cannot change while the box runs.
|
||||||
|
//
|
||||||
|
// Engine.ProbeAllowed — "the path in front of this group is dead RIGHT NOW".
|
||||||
|
// A property of the WORLD. It changes on its own, so it is asked at every
|
||||||
|
// scheduled check and never stored; that is the only reason a chain hop
|
||||||
|
// resumes probing by itself when the hop in front of it recovers.
|
||||||
|
//
|
||||||
|
// A CHAIN HOP WRAPPER is precisely where the difference bites. It is reached by
|
||||||
|
// the rules, so it is in the used-set and this function must leave its flag
|
||||||
|
// alone — TestStandDownNeverTouchesChainHopWrappers pins that, and it is not a
|
||||||
|
// technicality: standing a hop wrapper down here would take a working chain off
|
||||||
|
// its own board permanently, with nothing to turn it back on. If a hop wrapper
|
||||||
|
// should not probe because the hop in front of it is down, that is the gate's
|
||||||
|
// answer, not this one's, and it lasts exactly as long as the failure does.
|
||||||
|
|
||||||
// standDownUnusedSelfCheck flips SelfCheck to false on every urltest group in
|
// standDownUnusedSelfCheck flips SelfCheck to false on every urltest group in
|
||||||
// opts that the observatory's used-set does not cover, and returns how many it
|
// opts that the observatory's used-set does not cover, and returns how many it
|
||||||
@@ -48,7 +71,13 @@ import (
|
|||||||
// or something foreign) is skipped rather than guessed at: it did not come from
|
// or something foreign) is skipped rather than guessed at: it did not come from
|
||||||
// our generator, and silently rebuilding somebody else's option struct is worse
|
// our generator, and silently rebuilding somebody else's option struct is worse
|
||||||
// than leaving one group's self-check up.
|
// than leaving one group's self-check up.
|
||||||
func standDownUnusedSelfCheck(opts option.Options) int {
|
// It also RETURNS the used-set it computed. That set is needed before the new
|
||||||
|
// box starts — a group's PostStart asks whether it must keep measuring with no
|
||||||
|
// traffic (Engine.ProbeWhenIdle), and the observatory does not publish its own
|
||||||
|
// copy until after the swap has succeeded. Handing back the walk that has
|
||||||
|
// already run here is cheaper than a second one and, more to the point, cannot
|
||||||
|
// disagree with the flags just written from it.
|
||||||
|
func standDownUnusedSelfCheck(opts option.Options) (int, map[string]bool) {
|
||||||
_, used := BuildObservatoryPlan(opts, "")
|
_, used := BuildObservatoryPlan(opts, "")
|
||||||
stood := 0
|
stood := 0
|
||||||
for i := range opts.Outbounds {
|
for i := range opts.Outbounds {
|
||||||
@@ -66,5 +95,5 @@ func standDownUnusedSelfCheck(opts option.Options) int {
|
|||||||
utOpts.SelfCheck = &off
|
utOpts.SelfCheck = &off
|
||||||
stood++
|
stood++
|
||||||
}
|
}
|
||||||
return stood
|
return stood, used
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -29,10 +29,17 @@ func TestStandDownUnusedSelfCheck(t *testing.T) {
|
|||||||
Route: fixRoute("auto"),
|
Route: fixRoute("auto"),
|
||||||
}
|
}
|
||||||
|
|
||||||
stood := standDownUnusedSelfCheck(opts)
|
stood, used := standDownUnusedSelfCheck(opts)
|
||||||
if stood != 1 {
|
if stood != 1 {
|
||||||
t.Fatalf("stood down %d group(s), want exactly 1 (idle)", stood)
|
t.Fatalf("stood down %d group(s), want exactly 1 (idle)", stood)
|
||||||
}
|
}
|
||||||
|
// The returned used-set is the SAME walk the flags were written from — the
|
||||||
|
// engine publishes it for ProbeWhenIdle before the new box is built, so the
|
||||||
|
// two statements about one group ("keeps its self-check", "keeps it warm")
|
||||||
|
// can never be computed from two different reachability answers.
|
||||||
|
if !used["auto"] || used["idle"] {
|
||||||
|
t.Errorf("used-set = %v, want the reached group in and the unreached one out", used)
|
||||||
|
}
|
||||||
|
|
||||||
find := func(tag string) option.Outbound {
|
find := func(tag string) option.Outbound {
|
||||||
t.Helper()
|
t.Helper()
|
||||||
@@ -70,7 +77,7 @@ func TestStandDownUnusedSelfCheck(t *testing.T) {
|
|||||||
// still counted — the function reports plan membership, not novelty) and
|
// still counted — the function reports plan membership, not novelty) and
|
||||||
// changes nothing further. This is what keeps the apply hash stable across
|
// changes nothing further. This is what keeps the apply hash stable across
|
||||||
// the every-minute reconcile.
|
// the every-minute reconcile.
|
||||||
if again := standDownUnusedSelfCheck(opts); again != 1 {
|
if again, _ := standDownUnusedSelfCheck(opts); again != 1 {
|
||||||
t.Fatalf("second pass stood down %d, want 1 (deterministic on identical opts)", again)
|
t.Fatalf("second pass stood down %d, want 1 (deterministic on identical opts)", again)
|
||||||
}
|
}
|
||||||
if idle.SelfCheck == nil || *idle.SelfCheck {
|
if idle.SelfCheck == nil || *idle.SelfCheck {
|
||||||
@@ -144,10 +151,23 @@ func TestStandDownNeverTouchesChainHopWrappers(t *testing.T) {
|
|||||||
Route: fixRoute("", "chain-p-h4"),
|
Route: fixRoute("", "chain-p-h4"),
|
||||||
}
|
}
|
||||||
|
|
||||||
stood := standDownUnusedSelfCheck(opts)
|
stood, used := standDownUnusedSelfCheck(opts)
|
||||||
if stood != 3 {
|
if stood != 3 {
|
||||||
t.Fatalf("stood down %d group(s), want exactly the 3 base groups sub0/sub1/sub2", stood)
|
t.Fatalf("stood down %d group(s), want exactly the 3 base groups sub0/sub1/sub2", stood)
|
||||||
}
|
}
|
||||||
|
// The hop wrappers are also what ProbeWhenIdle must say yes to: the chain is
|
||||||
|
// reached by a rule, so its hops keep measuring whether or not traffic is
|
||||||
|
// crossing them. The base groups behind them are not, and must not.
|
||||||
|
for _, tag := range []string{"chain-p-h2", "chain-p-h3", "chain-p-h4"} {
|
||||||
|
if !used[tag] {
|
||||||
|
t.Errorf("used-set is missing hop wrapper %q; it would idle out with nothing to replace it", tag)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
for _, tag := range []string{"sub0", "sub1", "sub2"} {
|
||||||
|
if used[tag] {
|
||||||
|
t.Errorf("used-set claims base group %q, which no rule reaches", tag)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
utOptions := func(tag string) *option.URLTestOutboundOptions {
|
utOptions := func(tag string) *option.URLTestOutboundOptions {
|
||||||
t.Helper()
|
t.Helper()
|
||||||
|
|||||||
@@ -27,7 +27,7 @@ func outboundByTag(opts option.Options, tag string) *option.Outbound {
|
|||||||
|
|
||||||
func byedpiModel(port int) *model.Model {
|
func byedpiModel(port int) *model.Model {
|
||||||
return &model.Model{
|
return &model.Model{
|
||||||
Globals: model.DefaultGlobals(),
|
Globals: nonDNSGlobals(),
|
||||||
Egresses: []model.Egress{{Name: "bd", Type: "byedpi", Port: port}},
|
Egresses: []model.Egress{{Name: "bd", Type: "byedpi", Port: port}},
|
||||||
Rules: []model.Rule{
|
Rules: []model.Rule{
|
||||||
{Name: "desync", Enabled: true, Order: 10, Src: []string{"192.168.1.0/24"}, Target: "egress:bd"},
|
{Name: "desync", Enabled: true, Order: 10, Src: []string{"192.168.1.0/24"}, Target: "egress:bd"},
|
||||||
|
|||||||
@@ -47,7 +47,7 @@ func generalRouteOutbound(rt *option.RouteOptions) (string, bool) {
|
|||||||
|
|
||||||
func threeNodeChainModel() *model.Model {
|
func threeNodeChainModel() *model.Model {
|
||||||
return &model.Model{
|
return &model.Model{
|
||||||
Globals: model.DefaultGlobals(),
|
Globals: nonDNSGlobals(),
|
||||||
Nodes: []model.Node{
|
Nodes: []model.Node{
|
||||||
{Name: "a", Enabled: true, URI: "ss://aes-256-gcm:secret@203.0.113.1:8388#a"},
|
{Name: "a", Enabled: true, URI: "ss://aes-256-gcm:secret@203.0.113.1:8388#a"},
|
||||||
{Name: "b", Enabled: true, URI: "ss://aes-256-gcm:secret@203.0.113.2:8388#b"},
|
{Name: "b", Enabled: true, URI: "ss://aes-256-gcm:secret@203.0.113.2:8388#b"},
|
||||||
@@ -122,7 +122,7 @@ func TestChainMultiHopDetourWiring(t *testing.T) {
|
|||||||
// (no wrapper, no detour outbound).
|
// (no wrapper, no detour outbound).
|
||||||
func TestChainSingleHopResolvesToHop(t *testing.T) {
|
func TestChainSingleHopResolvesToHop(t *testing.T) {
|
||||||
m := &model.Model{
|
m := &model.Model{
|
||||||
Globals: model.DefaultGlobals(),
|
Globals: nonDNSGlobals(),
|
||||||
Nodes: []model.Node{
|
Nodes: []model.Node{
|
||||||
{Name: "a", Enabled: true, URI: "ss://aes-256-gcm:secret@203.0.113.1:8388#a"},
|
{Name: "a", Enabled: true, URI: "ss://aes-256-gcm:secret@203.0.113.1:8388#a"},
|
||||||
},
|
},
|
||||||
@@ -210,7 +210,7 @@ func TestChainEmptyHopsWarnsBlocks(t *testing.T) {
|
|||||||
// through the previous hop; the next hop detours into the group wrapper.
|
// through the previous hop; the next hop detours into the group wrapper.
|
||||||
func TestChainGroupHop(t *testing.T) {
|
func TestChainGroupHop(t *testing.T) {
|
||||||
m := &model.Model{
|
m := &model.Model{
|
||||||
Globals: model.DefaultGlobals(),
|
Globals: nonDNSGlobals(),
|
||||||
Nodes: []model.Node{
|
Nodes: []model.Node{
|
||||||
{Name: "a", Enabled: true, URI: "ss://aes-256-gcm:secret@203.0.113.1:8388#a"},
|
{Name: "a", Enabled: true, URI: "ss://aes-256-gcm:secret@203.0.113.1:8388#a"},
|
||||||
{Name: "b", Enabled: true, URI: "ss://aes-256-gcm:secret@203.0.113.2:8388#b"},
|
{Name: "b", Enabled: true, URI: "ss://aes-256-gcm:secret@203.0.113.2:8388#b"},
|
||||||
|
|||||||
@@ -129,6 +129,13 @@ func TestDevicePerDeviceDNS(t *testing.T) {
|
|||||||
|
|
||||||
// TestDeviceAllowOverridesBlock proves the ordering guarantee: a device's Allow
|
// TestDeviceAllowOverridesBlock proves the ordering guarantee: a device's Allow
|
||||||
// rule is emitted BEFORE its Block rule (allow wins, terminal DNS route action).
|
// rule is emitted BEFORE its Block rule (allow wins, terminal DNS route action).
|
||||||
|
//
|
||||||
|
// Since D24 (dns_intercept on by default) the device rules are no longer the
|
||||||
|
// FIRST rules in the plane: buildDNS prepends the .lan / private-PTR preservation
|
||||||
|
// rule ahead of everything, so that local names keep resolving through dnsmasq
|
||||||
|
// even for a device that carries a block list. That is the contract this test now
|
||||||
|
// states — rule 0 is the local-zone rule, and allow-before-block holds among the
|
||||||
|
// device rules that follow it.
|
||||||
func TestDeviceAllowOverridesBlock(t *testing.T) {
|
func TestDeviceAllowOverridesBlock(t *testing.T) {
|
||||||
m := baseDeviceModel(model.Device{
|
m := baseDeviceModel(model.Device{
|
||||||
Name: "kid", Enabled: true, IP: "192.168.1.70",
|
Name: "kid", Enabled: true, IP: "192.168.1.70",
|
||||||
@@ -139,15 +146,23 @@ func TestDeviceAllowOverridesBlock(t *testing.T) {
|
|||||||
if err != nil {
|
if err != nil {
|
||||||
t.Fatalf("Generate: %v", err)
|
t.Fatalf("Generate: %v", err)
|
||||||
}
|
}
|
||||||
if opts.DNS == nil || len(opts.DNS.Rules) < 2 {
|
if opts.DNS == nil || len(opts.DNS.Rules) == 0 {
|
||||||
t.Fatalf("expected >=2 device DNS rules (allow+block), got %+v", opts.DNS)
|
t.Fatalf("expected a DNS plane, got %+v", opts.DNS)
|
||||||
}
|
}
|
||||||
allow := opts.DNS.Rules[0].DefaultOptions
|
if !isLocalZoneRule(opts.DNS.Rules[0]) {
|
||||||
|
t.Fatalf("rule 0 must be the .lan/PTR preservation rule (it has to outrank per-device policy), got %+v",
|
||||||
|
opts.DNS.Rules[0].DefaultOptions)
|
||||||
|
}
|
||||||
|
rules := nonLocalDNSRules(opts)
|
||||||
|
if len(rules) < 2 {
|
||||||
|
t.Fatalf("expected >=2 device DNS rules (allow+block), got %+v", rules)
|
||||||
|
}
|
||||||
|
allow := rules[0].DefaultOptions
|
||||||
if allow.DNSRuleAction.Action != C.RuleActionTypeRoute ||
|
if allow.DNSRuleAction.Action != C.RuleActionTypeRoute ||
|
||||||
len(allow.DomainSuffix) != 1 || allow.DomainSuffix[0] != "good.example" {
|
len(allow.DomainSuffix) != 1 || allow.DomainSuffix[0] != "good.example" {
|
||||||
t.Fatalf("first device DNS rule must be the allow(good.example)->route, got %+v", allow)
|
t.Fatalf("first device DNS rule must be the allow(good.example)->route, got %+v", allow)
|
||||||
}
|
}
|
||||||
block := opts.DNS.Rules[1].DefaultOptions
|
block := rules[1].DefaultOptions
|
||||||
if block.DNSRuleAction.Action != C.RuleActionTypePredefined ||
|
if block.DNSRuleAction.Action != C.RuleActionTypePredefined ||
|
||||||
len(block.DomainSuffix) != 1 || block.DomainSuffix[0] != "ads.example" {
|
len(block.DomainSuffix) != 1 || block.DomainSuffix[0] != "ads.example" {
|
||||||
t.Fatalf("second device DNS rule must be the block(ads.example)->NXDOMAIN, got %+v", block)
|
t.Fatalf("second device DNS rule must be the block(ads.example)->NXDOMAIN, got %+v", block)
|
||||||
@@ -155,21 +170,35 @@ func TestDeviceAllowOverridesBlock(t *testing.T) {
|
|||||||
}
|
}
|
||||||
|
|
||||||
// TestDeviceNoneEmitsNothing proves the guard: no devices (or none enabled) adds
|
// TestDeviceNoneEmitsNothing proves the guard: no devices (or none enabled) adds
|
||||||
// no device route rules and no device DNS rules — identical to a pre-Phase-6 build.
|
// no device route rules and no DEVICE DNS rules.
|
||||||
|
//
|
||||||
|
// "No DNS rules at all" is no longer the right statement of that guard, and
|
||||||
|
// weakening it to "some rules" would test nothing. Since D24 every intercepting
|
||||||
|
// plane carries exactly one rule that is not ours to suppress — the .lan / PTR
|
||||||
|
// preservation rule — so the guard is now: the device codegen contributes NOTHING,
|
||||||
|
// and what remains is that single local-zone rule, unchanged by a device list that
|
||||||
|
// is empty or entirely disabled.
|
||||||
func TestDeviceNoneEmitsNothing(t *testing.T) {
|
func TestDeviceNoneEmitsNothing(t *testing.T) {
|
||||||
|
assertOnlyLocalZoneRule := func(t *testing.T, what string, opts option.Options) {
|
||||||
|
t.Helper()
|
||||||
|
if opts.Route == nil || len(opts.Route.Rules) != 2 {
|
||||||
|
t.Fatalf("%s must emit only sniff+hijackdns route rules, got %+v", what, opts.Route)
|
||||||
|
}
|
||||||
|
if devRules := nonLocalDNSRules(opts); len(devRules) != 0 {
|
||||||
|
t.Fatalf("%s must emit no device DNS rules, got %+v", what, devRules)
|
||||||
|
}
|
||||||
|
if opts.DNS == nil || len(opts.DNS.Rules) != 1 || !isLocalZoneRule(opts.DNS.Rules[0]) {
|
||||||
|
t.Fatalf("%s must keep exactly the .lan/PTR preservation rule, got %+v", what, opts.DNS)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
// No devices at all.
|
// No devices at all.
|
||||||
m := baseDeviceModel()
|
m := baseDeviceModel()
|
||||||
opts, _, err := GenerateWithWarnings(m)
|
opts, _, err := GenerateWithWarnings(m)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
t.Fatalf("Generate: %v", err)
|
t.Fatalf("Generate: %v", err)
|
||||||
}
|
}
|
||||||
// Only sniff + hijack-dns route rules remain (no device/general rules).
|
assertOnlyLocalZoneRule(t, "no-devices model", opts)
|
||||||
if opts.Route == nil || len(opts.Route.Rules) != 2 {
|
|
||||||
t.Fatalf("no-devices model must emit only sniff+hijackdns route rules, got %+v", opts.Route)
|
|
||||||
}
|
|
||||||
if opts.DNS != nil && len(opts.DNS.Rules) != 0 {
|
|
||||||
t.Fatalf("no-devices model must emit no DNS rules, got %+v", opts.DNS.Rules)
|
|
||||||
}
|
|
||||||
|
|
||||||
// A single DISABLED device: same result.
|
// A single DISABLED device: same result.
|
||||||
m2 := baseDeviceModel(model.Device{Name: "off", Enabled: false, IP: "192.168.1.80", Block: []string{"x.example"}})
|
m2 := baseDeviceModel(model.Device{Name: "off", Enabled: false, IP: "192.168.1.80", Block: []string{"x.example"}})
|
||||||
@@ -177,10 +206,5 @@ func TestDeviceNoneEmitsNothing(t *testing.T) {
|
|||||||
if err != nil {
|
if err != nil {
|
||||||
t.Fatalf("Generate: %v", err)
|
t.Fatalf("Generate: %v", err)
|
||||||
}
|
}
|
||||||
if len(opts2.Route.Rules) != 2 {
|
assertOnlyLocalZoneRule(t, "disabled-only model", opts2)
|
||||||
t.Fatalf("disabled-only model must emit only sniff+hijackdns route rules, got %+v", opts2.Route.Rules)
|
|
||||||
}
|
|
||||||
if opts2.DNS != nil && len(opts2.DNS.Rules) != 0 {
|
|
||||||
t.Fatalf("disabled-only model must emit no DNS rules, got %+v", opts2.DNS.Rules)
|
|
||||||
}
|
|
||||||
}
|
}
|
||||||
|
|||||||
+278
-25
@@ -18,8 +18,9 @@ import (
|
|||||||
// buildDNS assembles the typed option.DNSOptions from the model resolvers +
|
// buildDNS assembles the typed option.DNSOptions from the model resolvers +
|
||||||
// dns_rules (v1.14 removed the flat servers list and top-level fakeip). Each
|
// dns_rules (v1.14 removed the flat servers list and top-level fakeip). Each
|
||||||
// server carries its own Detour (anti-leak: DNS through the proxy/direct as
|
// server carries its own Detour (anti-leak: DNS through the proxy/direct as
|
||||||
// configured); a server whose detour cannot be resolved falls back to the
|
// configured); a server whose detour cannot be resolved is FAIL-CLOSED to `block`
|
||||||
// default outbound with a warning.
|
// with a warning, so it stops answering rather than falling out to the plain WAN
|
||||||
|
// (see dnsDetour).
|
||||||
//
|
//
|
||||||
// Returns nil when the model declares no resolvers (engine uses its built-ins).
|
// Returns nil when the model declares no resolvers (engine uses its built-ins).
|
||||||
func (b *builder) buildDNS() *option.DNSOptions {
|
func (b *builder) buildDNS() *option.DNSOptions {
|
||||||
@@ -34,7 +35,7 @@ func (b *builder) buildDNS() *option.DNSOptions {
|
|||||||
b.warnf("per-device DNS block/allow configured but no resolvers; inert (needs a resolver for the in-engine DNS plane)")
|
b.warnf("per-device DNS block/allow configured but no resolvers; inert (needs a resolver for the in-engine DNS plane)")
|
||||||
}
|
}
|
||||||
if b.m.Globals.DNSIntercept {
|
if b.m.Globals.DNSIntercept {
|
||||||
b.warnf("dns_intercept enabled but no resolvers configured; .lan/local DNS won't be preserved (engine falls back to built-in DNS)")
|
b.warnf("dns_intercept is ON but there is no `config resolver`: every intercepted query is answered by the router's own system resolver (the engine's built-in `local` transport reads /etc/resolv.conf, which on OpenWrt is dnsmasq -> your provider, in the clear). .lan and the private PTR zones keep working, but DNS filtering, per-device DNS rules and the DNS anti-leak detour are inert until you add a resolver.")
|
||||||
}
|
}
|
||||||
if b.m.Globals.BlockDoH {
|
if b.m.Globals.BlockDoH {
|
||||||
b.warnf("block_doh enabled but no resolvers configured; the DoH NXDOMAIN rules (incl. the Firefox use-application-dns.net canary) are NOT emitted — only the :443 route rejects apply")
|
b.warnf("block_doh enabled but no resolvers configured; the DoH NXDOMAIN rules (incl. the Firefox use-application-dns.net canary) are NOT emitted — only the :443 route rejects apply")
|
||||||
@@ -60,7 +61,7 @@ func (b *builder) buildDNS() *option.DNSOptions {
|
|||||||
b.warnf("per-device DNS block/allow configured but no valid resolvers built; inert")
|
b.warnf("per-device DNS block/allow configured but no valid resolvers built; inert")
|
||||||
}
|
}
|
||||||
if b.m.Globals.DNSIntercept {
|
if b.m.Globals.DNSIntercept {
|
||||||
b.warnf("dns_intercept enabled but no valid resolvers built; .lan/local DNS won't be preserved (engine falls back to built-in DNS)")
|
b.warnf("dns_intercept is ON but no `config resolver` could be built (check the warnings above): every intercepted query is answered by the router's own system resolver (the engine's built-in `local` transport reads /etc/resolv.conf, which on OpenWrt is dnsmasq -> your provider, in the clear). .lan and the private PTR zones keep working, but DNS filtering, per-device DNS rules and the DNS anti-leak detour are inert until a resolver builds.")
|
||||||
}
|
}
|
||||||
if b.m.Globals.BlockDoH {
|
if b.m.Globals.BlockDoH {
|
||||||
b.warnf("block_doh enabled but no valid resolvers built; the DoH NXDOMAIN rules (incl. the Firefox use-application-dns.net canary) are NOT emitted — only the :443 route rejects apply")
|
b.warnf("block_doh enabled but no valid resolvers built; the DoH NXDOMAIN rules (incl. the Firefox use-application-dns.net canary) are NOT emitted — only the :443 route rejects apply")
|
||||||
@@ -116,6 +117,14 @@ func (b *builder) buildDNS() *option.DNSOptions {
|
|||||||
servers = append(servers, *b.endpointResolverServer)
|
servers = append(servers, *b.endpointResolverServer)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// IMPLICIT default_domain_resolver. The DNS plane is complete at this point, so
|
||||||
|
// this is the first place that can count its transports — which is the only thing
|
||||||
|
// the engine's dialer branches on (see implicitBootstrapResolver). Skipped when the
|
||||||
|
// operator named an endpoint_resolver: that one already set the field and MUST win.
|
||||||
|
if b.endpointResolver() == "" && len(servers) >= 2 {
|
||||||
|
b.bootstrapResolverTag = b.implicitBootstrapResolver(final, localDNSTag, serverTags, &servers)
|
||||||
|
}
|
||||||
|
|
||||||
// DoH-block NXDOMAIN rules sit FIRST (BlockDoH): they answer known public DoH
|
// DoH-block NXDOMAIN rules sit FIRST (BlockDoH): they answer known public DoH
|
||||||
// hostnames + the Firefox canary with NXDOMAIN so clients fall back to plaintext
|
// hostnames + the Firefox canary with NXDOMAIN so clients fall back to plaintext
|
||||||
// :53. Empty when BlockDoH is off. The synthetic local-dns rule prepended at the
|
// :53. Empty when BlockDoH is off. The synthetic local-dns rule prepended at the
|
||||||
@@ -137,16 +146,29 @@ func (b *builder) buildDNS() *option.DNSOptions {
|
|||||||
// slice in arbitrary order, and the panel promises "lower Order runs earlier /
|
// slice in arbitrary order, and the panel promises "lower Order runs earlier /
|
||||||
// first match wins". Mirrors sortedRuleIndices (generate.go) for route rules:
|
// first match wins". Mirrors sortedRuleIndices (generate.go) for route rules:
|
||||||
// stable sort by (Order, original index).
|
// stable sort by (Order, original index).
|
||||||
|
//
|
||||||
|
// A rule naming a resolver that does not exist is NOT dropped — see
|
||||||
|
// dnsRuleUnresolvedAction for why a dropped rule is a leak and what replaces it.
|
||||||
for _, i := range sortedDNSRuleIndices(b.m.DNSRules) {
|
for _, i := range sortedDNSRuleIndices(b.m.DNSRules) {
|
||||||
dr := b.m.DNSRules[i]
|
dr := b.m.DNSRules[i]
|
||||||
if !serverTags[dr.Resolver] {
|
known := serverTags[dr.Resolver]
|
||||||
b.warnf("dns_rule -> resolver %q: unknown resolver, skipped", dr.Resolver)
|
if !known {
|
||||||
continue
|
b.warnf("dns_rule -> resolver %q: no such resolver — wrong name, or it was removed, or it was skipped for "+
|
||||||
|
"a bad address/type. This rule is FAIL-CLOSED: the names (and clients) it singles out now get NXDOMAIN "+
|
||||||
|
"instead of quietly falling through to the default resolver %q. Falling through is what this used to do, "+
|
||||||
|
"and it sent exactly the lookups you had singled out to a resolver you did not choose for them — over "+
|
||||||
|
"whatever path that one uses. Restore the resolver, or point the rule at one that exists.",
|
||||||
|
dr.Resolver, final)
|
||||||
}
|
}
|
||||||
b.warnUnrecognisedPrefixes(fmt.Sprintf("dns_rule %q", dr.Resolver), dr.MatchDomain)
|
b.warnUnrecognisedPrefixes(fmt.Sprintf("dns_rule %q", dr.Resolver), dr.MatchDomain)
|
||||||
if rule, ok := b.dnsRule(dr); ok {
|
rule, ok := b.dnsRule(dr)
|
||||||
rules = append(rules, rule)
|
if !ok {
|
||||||
|
continue
|
||||||
}
|
}
|
||||||
|
if !known {
|
||||||
|
rule.DefaultOptions.DNSRuleAction = dnsRuleUnresolvedAction()
|
||||||
|
}
|
||||||
|
rules = append(rules, rule)
|
||||||
}
|
}
|
||||||
|
|
||||||
// Globals.ResolverFallback — the real failover chain, APPENDED last so it only
|
// Globals.ResolverFallback — the real failover chain, APPENDED last so it only
|
||||||
@@ -272,19 +294,11 @@ func (b *builder) endpointResolver() string {
|
|||||||
|
|
||||||
// Clone under a dedicated tag with Detour forced direct (bootstrap-direct; see the
|
// Clone under a dedicated tag with Detour forced direct (bootstrap-direct; see the
|
||||||
// doc comment). Guard the tag against colliding with any real resolver server tag.
|
// doc comment). Guard the tag against colliding with any real resolver server tag.
|
||||||
clone := *res
|
|
||||||
clone.Detour = "direct"
|
|
||||||
tag := endpointResolverTagPrefix + res.Name
|
|
||||||
taken := make(map[string]bool, len(b.m.Resolvers))
|
taken := make(map[string]bool, len(b.m.Resolvers))
|
||||||
for i := range b.m.Resolvers {
|
for i := range b.m.Resolvers {
|
||||||
taken[b.m.Resolvers[i].Name] = true
|
taken[b.m.Resolvers[i].Name] = true
|
||||||
}
|
}
|
||||||
for taken[tag] {
|
srv, tag, ok := b.bootstrapDirectServer(*res, endpointResolverTagPrefix, taken)
|
||||||
tag += "-x"
|
|
||||||
}
|
|
||||||
clone.Name = tag
|
|
||||||
|
|
||||||
srv, ok := b.dnsServer(clone)
|
|
||||||
if !ok {
|
if !ok {
|
||||||
// dnsServer already warned (bad address/type); leave the resolver unset.
|
// dnsServer already warned (bad address/type); leave the resolver unset.
|
||||||
b.warnf("endpoint_resolver %q (%s): could not be built as a bootstrap-direct server; route.default_domain_resolver is left unset", name, src)
|
b.warnf("endpoint_resolver %q (%s): could not be built as a bootstrap-direct server; route.default_domain_resolver is left unset", name, src)
|
||||||
@@ -295,6 +309,167 @@ func (b *builder) endpointResolver() string {
|
|||||||
return tag
|
return tag
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// bootstrapDirectServer clones res into a DNS server whose detour is FORCED to
|
||||||
|
// "direct", under a tag built from prefix+res.Name and guarded against every name in
|
||||||
|
// taken. ok=false when the resolver cannot be built as a server at all (bad address,
|
||||||
|
// unknown type) — the caller then decides what to say and never emits a reference.
|
||||||
|
//
|
||||||
|
// Every warning dnsServer raises while building the clone is DISCARDED, deliberately.
|
||||||
|
// The clone differs from what the operator wrote in exactly one field — Detour, forced
|
||||||
|
// to a target that always resolves — so each diagnostic it can produce (an ignored
|
||||||
|
// pool, a missing address, a type that does not exist) was already produced verbatim
|
||||||
|
// for the client-facing copy of the same resolver. Emitting them twice would make one
|
||||||
|
// typo look like two faults, and would attribute one of them to a server tag the
|
||||||
|
// operator never wrote and cannot find in their config.
|
||||||
|
func (b *builder) bootstrapDirectServer(res model.Resolver, prefix string, taken map[string]bool) (option.DNSServerOptions, string, bool) {
|
||||||
|
clone := res
|
||||||
|
clone.Detour = tagDirect
|
||||||
|
tag := prefix + res.Name
|
||||||
|
for taken[tag] {
|
||||||
|
tag += "-x"
|
||||||
|
}
|
||||||
|
clone.Name = tag
|
||||||
|
|
||||||
|
mark := len(b.warnings)
|
||||||
|
srv, ok := b.dnsServer(clone)
|
||||||
|
b.warnings = b.warnings[:mark]
|
||||||
|
if !ok {
|
||||||
|
return option.DNSServerOptions{}, "", false
|
||||||
|
}
|
||||||
|
return srv, tag, true
|
||||||
|
}
|
||||||
|
|
||||||
|
// bootstrapResolverTagPrefix names the synthetic DNS server the IMPLICIT
|
||||||
|
// default_domain_resolver is cloned into. Deliberately distinct from
|
||||||
|
// endpointResolverTagPrefix: one of those tags is a server the operator asked for by
|
||||||
|
// name, the other is one the generator chose for them, and a log line or a panel row
|
||||||
|
// naming either has to be able to say which.
|
||||||
|
const bootstrapResolverTagPrefix = "shater-bootstrap-dns-"
|
||||||
|
|
||||||
|
// implicitBootstrapResolver emits the DNS server that route.default_domain_resolver
|
||||||
|
// points at when the operator named NO endpoint_resolver, and returns its tag. It
|
||||||
|
// returns "" when nothing usable could be built, in which case the field stays unset —
|
||||||
|
// a dangling default_domain_resolver aborts box.New, so it is never emitted on spec.
|
||||||
|
//
|
||||||
|
// # Why the field has to be set at all
|
||||||
|
//
|
||||||
|
// The engine's dialer resolves a domain SERVER ADDRESS — a proxy node written as a
|
||||||
|
// hostname, and any domain destination handed to the `direct` outbound — through
|
||||||
|
// whatever DNS it is told to use. common/dialer/dialer.go picks that at construction
|
||||||
|
// time, and switches behaviour on the NUMBER OF DNS TRANSPORTS:
|
||||||
|
//
|
||||||
|
// 1 transport, no default_domain_resolver -> dnsTransport.Default(), i.e. the
|
||||||
|
// `final` server, DNS rules bypassed
|
||||||
|
// 2+ transports, no default_domain_resolver -> the transport stays nil and
|
||||||
|
// dns/router.go drops into
|
||||||
|
// lookupWithRules — the full CLIENT
|
||||||
|
// DNS rule chain
|
||||||
|
//
|
||||||
|
// The second is not a milder version of the first, it is a different plane. Sending a
|
||||||
|
// node's own hostname down the client rule chain means a blocklist, a block_doh
|
||||||
|
// NXDOMAIN rule or any dns_rule can now answer it: one sloppy entry in an ad list, or
|
||||||
|
// a fake-IP default resolver, stops being an ad that got through and becomes a tunnel
|
||||||
|
// that never comes up. It is also what raises the engine's `missing-domain-resolver`
|
||||||
|
// deprecation.
|
||||||
|
//
|
||||||
|
// None of that is specific to dns_intercept — two `config resolver`s were always
|
||||||
|
// enough to reach it. dns_intercept only made it UNIVERSAL, because the synthetic
|
||||||
|
// `shater-local-dns` server it adds is a second transport by itself. So the trigger
|
||||||
|
// here is the condition the engine actually branches on (two or more transports), not
|
||||||
|
// the toggle that happened to expose it.
|
||||||
|
//
|
||||||
|
// # Why THIS server
|
||||||
|
//
|
||||||
|
// A clone of the effective default resolver with its detour forced to "direct" —
|
||||||
|
// exactly the shape endpointResolver builds for the explicit case, for the reason
|
||||||
|
// spelled out there: a proxy server's own domain CANNOT be resolved through the proxy
|
||||||
|
// that domain is needed to open. That is a bootstrap loop, and its symptom (tunnel
|
||||||
|
// never comes up, nothing resolves, no log line names either as the cause) is the
|
||||||
|
// worst outcome on offer here.
|
||||||
|
//
|
||||||
|
// `final` rather than something fabricated, because it is the upstream the operator
|
||||||
|
// already trusts with every query no rule claimed — and cloning preserves its TYPE, so
|
||||||
|
// a DoH default stays DoH: the bootstrap lookups remain encrypted and still go to the
|
||||||
|
// chosen provider. The single thing dropped is the tunnel hop, which is the one hop
|
||||||
|
// that cannot exist yet.
|
||||||
|
//
|
||||||
|
// Fall-back order, because a fake-IP server invents addresses and would hand the dialer
|
||||||
|
// a synthetic IP no tunnel can ever connect to:
|
||||||
|
//
|
||||||
|
// final -> first non-fakeip resolver that built -> the synthetic local server
|
||||||
|
// (dnsmasq, already direct) -> "" (field left unset)
|
||||||
|
func (b *builder) implicitBootstrapResolver(final, localDNSTag string, serverTags map[string]bool, servers *[]option.DNSServerOptions) string {
|
||||||
|
res, substituted := b.bootstrapSource(final, serverTags)
|
||||||
|
if res == nil {
|
||||||
|
// Every resolver that built is fake-IP. dnsmasq (the synthetic local server) is
|
||||||
|
// the only real, already-direct transport left; with dns_intercept off there is
|
||||||
|
// not even that, and the field stays unset. warnFakeIPDefault has already said
|
||||||
|
// what a fake-IP default costs for CLIENT queries — this says what it costs the
|
||||||
|
// router's own bootstrap, which is a different (and invisible) bill.
|
||||||
|
if localDNSTag != "" {
|
||||||
|
b.warnf("every configured resolver is fake-IP, and a fake-IP server invents addresses instead of asking anyone, so it cannot resolve a real proxy server's domain. Proxy server domains are therefore resolved by the router's own system resolver (dnsmasq -> your provider, in the clear) — add a normal resolver, or name one with globals.endpoint_resolver.")
|
||||||
|
}
|
||||||
|
return localDNSTag
|
||||||
|
}
|
||||||
|
srv, tag, ok := b.bootstrapDirectServer(*res, bootstrapResolverTagPrefix, serverTags)
|
||||||
|
if !ok {
|
||||||
|
return localDNSTag
|
||||||
|
}
|
||||||
|
*servers = append(*servers, srv)
|
||||||
|
if substituted {
|
||||||
|
b.warnf("resolver_default %q is a fake-IP resolver, which invents addresses and cannot resolve a real proxy server's domain, so proxy server domains are resolved through %q instead (forced direct). Name the one you want with globals.endpoint_resolver if that is not it.", final, res.Name)
|
||||||
|
}
|
||||||
|
b.warnBootstrapInTheClear(*res)
|
||||||
|
return tag
|
||||||
|
}
|
||||||
|
|
||||||
|
// bootstrapSource picks the resolver to clone for the implicit bootstrap server:
|
||||||
|
// `final` when it is usable, otherwise the first non-fakeip resolver that actually
|
||||||
|
// built. substituted reports that the pick is NOT the default resolver, so the caller
|
||||||
|
// can say so rather than quietly resolving node domains somewhere else.
|
||||||
|
func (b *builder) bootstrapSource(final string, serverTags map[string]bool) (res *model.Resolver, substituted bool) {
|
||||||
|
if !b.resolverIsFakeIP(final) {
|
||||||
|
for i := range b.m.Resolvers {
|
||||||
|
if b.m.Resolvers[i].Name == final {
|
||||||
|
return &b.m.Resolvers[i], false
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
for i := range b.m.Resolvers {
|
||||||
|
r := &b.m.Resolvers[i]
|
||||||
|
if serverTags[r.Name] && !b.resolverIsFakeIP(r.Name) {
|
||||||
|
return r, true
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return nil, false
|
||||||
|
}
|
||||||
|
|
||||||
|
// warnBootstrapInTheClear names the one cost the implicit bootstrap server has that
|
||||||
|
// the operator did not choose and cannot see.
|
||||||
|
//
|
||||||
|
// It fires only for a PLAINTEXT resolver (plain/udp/tcp) that the operator had
|
||||||
|
// deliberately routed through a proxy. For that config, and only that one, forcing the
|
||||||
|
// bootstrap copy direct means the hostnames of the operator's OWN nodes go to their
|
||||||
|
// provider in clear text from their real address — while every other lookup on the box
|
||||||
|
// is still tunnelled, which is exactly the sort of gap that looks like it isn't there.
|
||||||
|
//
|
||||||
|
// Silent for doh/dot (the clone is still encrypted to the same upstream, so nothing
|
||||||
|
// leaks but the fact that a DoH connection exists) and silent for a resolver that was
|
||||||
|
// never detoured through a tunnel in the first place: nothing changed for it.
|
||||||
|
func (b *builder) warnBootstrapInTheClear(res model.Resolver) {
|
||||||
|
switch strings.ToLower(strings.TrimSpace(res.Type)) {
|
||||||
|
case "plain", "udp", "", "tcp":
|
||||||
|
default:
|
||||||
|
return
|
||||||
|
}
|
||||||
|
detour := strings.TrimSpace(res.Detour)
|
||||||
|
if detour == "" || strings.EqualFold(detour, tagDirect) || strings.EqualFold(detour, tagBlock) {
|
||||||
|
return
|
||||||
|
}
|
||||||
|
b.warnf("resolver %q: its queries are routed through %q, but the ADDRESS of the proxy %q opens has to be resolved before that proxy exists — so those lookups are made directly, over the plain WAN. %q is a plaintext resolver, so your provider sees the hostnames of your own nodes, coming from your real address, while everything else stays tunnelled. Use a doh/dot resolver, or point globals.endpoint_resolver at one you are content to expose.",
|
||||||
|
res.Name, detour, detour, res.Name)
|
||||||
|
}
|
||||||
|
|
||||||
// resolverFallbackRules implements Globals.ResolverFallback: when the primary
|
// resolverFallbackRules implements Globals.ResolverFallback: when the primary
|
||||||
// resolver produces NO response, the query is retried against the fallback.
|
// resolver produces NO response, the query is retried against the fallback.
|
||||||
//
|
//
|
||||||
@@ -394,6 +569,35 @@ func (b *builder) resolverFallbackRules(final string, serverTags map[string]bool
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// dnsRuleUnresolvedAction is what a dns_rule does when the resolver it names does
|
||||||
|
// not exist. It is the DNS-plane twin of ruleKillFallback (route.go), and it
|
||||||
|
// exists for the identical reason.
|
||||||
|
//
|
||||||
|
// A dns_rule used to be DROPPED whole in that case, which reads as harmless and is
|
||||||
|
// not: the domains it named then fell through to whatever the default resolver is.
|
||||||
|
// The operator singled those names out to be resolved somewhere specific — through
|
||||||
|
// the tunnel, by the corporate resolver, by the one server that answers them
|
||||||
|
// correctly — and dropping the rule silently converted that policy into the
|
||||||
|
// default policy. route.go says it best about routing rules: it is "a silent leak
|
||||||
|
// of exactly the traffic the operator singled out", and a DNS lookup leaks the
|
||||||
|
// name itself, which is the part a provider actually collects.
|
||||||
|
//
|
||||||
|
// So the rule stays, with its matchers intact, and answers NXDOMAIN. That is
|
||||||
|
// visible (the name stops resolving, the warning says why), scoped (only the
|
||||||
|
// rule's own matchers are affected, everything else resolves normally) and safe
|
||||||
|
// (nothing is sent anywhere the operator did not ask for).
|
||||||
|
//
|
||||||
|
// A predefined NXDOMAIN rather than action=reject on two counts: it is the answer
|
||||||
|
// this codebase already uses to refuse a name (dnsfilter.go), and a reject action
|
||||||
|
// built in Go with an unset Method PANICS the engine at match time
|
||||||
|
// (route/rule/rule_action.go RuleActionReject.Error) — the normalisation that
|
||||||
|
// would fill it in only runs on the JSON unmarshal path.
|
||||||
|
//
|
||||||
|
// The rule is still skipped entirely when it has NO usable matcher (dnsRule
|
||||||
|
// returns ok=false): an unmatched-but-emitted rule would NXDOMAIN every query in
|
||||||
|
// the network, which is a far larger blast radius than the fault deserves.
|
||||||
|
func dnsRuleUnresolvedAction() option.DNSRuleAction { return predefinedNXDOMAIN() }
|
||||||
|
|
||||||
// resolverIsFakeIP reports whether the named model resolver is a fakeip server.
|
// resolverIsFakeIP reports whether the named model resolver is a fakeip server.
|
||||||
func (b *builder) resolverIsFakeIP(name string) bool {
|
func (b *builder) resolverIsFakeIP(name string) bool {
|
||||||
for _, r := range b.m.Resolvers {
|
for _, r := range b.m.Resolvers {
|
||||||
@@ -452,8 +656,8 @@ func (b *builder) resolverIsFakeIP(name string) bool {
|
|||||||
// hosts, dhcp, mdns, tailscale, legacy. Those are missing capabilities, not
|
// hosts, dhcp, mdns, tailscale, legacy. Those are missing capabilities, not
|
||||||
// defects; adding one is a model+panel change, not a generator change.
|
// defects; adding one is a model+panel change, not a generator change.
|
||||||
func (b *builder) dnsServer(r model.Resolver) (option.DNSServerOptions, bool) {
|
func (b *builder) dnsServer(r model.Resolver) (option.DNSServerOptions, bool) {
|
||||||
detour := b.dnsDetour(r)
|
|
||||||
kind := strings.ToLower(strings.TrimSpace(r.Type))
|
kind := strings.ToLower(strings.TrimSpace(r.Type))
|
||||||
|
detour := b.dnsDetour(r, kind)
|
||||||
b.warnIgnoredResolverFields(r, kind)
|
b.warnIgnoredResolverFields(r, kind)
|
||||||
|
|
||||||
// A remote transport with no address aborts box.New ("invalid server address")
|
// A remote transport with no address aborts box.New ("invalid server address")
|
||||||
@@ -643,16 +847,65 @@ func (b *builder) warnFakeIPDefault(final string) {
|
|||||||
"Make a normal resolver the default and point a dns_rule at %q for just the domains you want fake-IP to handle.", final, final)
|
"Make a normal resolver the default and point a dns_rule at %q for just the domains you want fake-IP to handle.", final, final)
|
||||||
}
|
}
|
||||||
|
|
||||||
// dnsDetour resolves the resolver's Detour target to an outbound tag; an empty
|
// dnsDetour resolves the resolver's Detour target to the outbound tag its queries
|
||||||
// or unresolved detour means "use the default outbound" (returns "").
|
// must be dialed through.
|
||||||
func (b *builder) dnsDetour(r model.Resolver) string {
|
//
|
||||||
if strings.TrimSpace(r.Detour) == "" {
|
// "" — no detour is configured. The server dials over the default route,
|
||||||
|
// which is what "no detour" has always meant and is not a fault.
|
||||||
|
// <tag> — the detour resolved; queries go through that outbound.
|
||||||
|
// tagBlock — the detour is DANGLING; this resolver's queries are BLOCKED.
|
||||||
|
//
|
||||||
|
// # Why a dangling detour blocks (FAIL-CLOSED)
|
||||||
|
//
|
||||||
|
// This used to return "" for an unresolvable detour and say it was "using default
|
||||||
|
// outbound". It was not: "" is not a default outbound, it is
|
||||||
|
// dialer.NewWithOptions' NewDefault branch — an ordinary system socket on the
|
||||||
|
// router's own WAN. So a resolver written as `detour node:vps` — written that way
|
||||||
|
// precisely so the ISP never sees which names this network looks up — silently
|
||||||
|
// became a plaintext UDP :53 query to 1.1.1.1 over the provider's link, from the
|
||||||
|
// router's real address, the moment the node it named went away. Switching a node
|
||||||
|
// off from the panel is enough to trigger it, and it happens WHILE the kill-switch
|
||||||
|
// is closed and the traffic itself is blocked: the connections stop, the list of
|
||||||
|
// domains keeps flowing to the ISP, and nothing in the UI says so.
|
||||||
|
//
|
||||||
|
// It was also the last place in this generator where a dangling reference degraded
|
||||||
|
// into a direct path. Every neighbour already fails closed and says why —
|
||||||
|
// egressDetourOrBlock ("there is deliberately no fail-open opt-out here"),
|
||||||
|
// ruleKillFallback, chain.go's L1 hop — and wgdedup's retargetDroppedTags rewrites
|
||||||
|
// THIS VERY FIELD (opts.DNS.Servers[i]'s detour) to `block` when it drops an
|
||||||
|
// endpoint. One field with two opposite policies, the outcome depending on which
|
||||||
|
// piece of code noticed the breakage first, is not a policy at all.
|
||||||
|
//
|
||||||
|
// # What the operator gets instead
|
||||||
|
//
|
||||||
|
// `block` is a real outbound (always emitted, outbound.go) whose dial returns
|
||||||
|
// EPERM, so the DNS transport fails on every exchange. Concretely: queries routed
|
||||||
|
// to this resolver get a transport error instead of an answer; if it is also
|
||||||
|
// globals.resolver_default, that is EVERY query no earlier rule claimed, i.e. name
|
||||||
|
// resolution stops network-wide until the detour is fixed. That is the correct
|
||||||
|
// fail-closed outcome — a router that resolves nothing is recoverable, a router
|
||||||
|
// that quietly hands the ISP its whole DNS history is not — and it is exactly why
|
||||||
|
// the warning below names the consequence instead of the mechanism. When
|
||||||
|
// globals.resolver_fallback is configured it still applies: the failover chain
|
||||||
|
// fires on "no response", which a blocked transport produces.
|
||||||
|
//
|
||||||
|
// fakeip is excluded because its detour is genuinely inert (a fake-IP server mints
|
||||||
|
// answers locally and dials nothing at all); blocking it would be a lie in the
|
||||||
|
// other direction, and warnIgnoredResolverFields already reports the ignored field.
|
||||||
|
func (b *builder) dnsDetour(r model.Resolver, kind string) string {
|
||||||
|
if kind == "fakeip" || strings.TrimSpace(r.Detour) == "" {
|
||||||
return ""
|
return ""
|
||||||
}
|
}
|
||||||
tag, ok := b.resolveTarget(r.Detour)
|
tag, ok := b.resolveTarget(r.Detour)
|
||||||
if !ok {
|
if !ok {
|
||||||
b.warnf("resolver %q: unresolved detour %q, using default outbound", r.Name, r.Detour)
|
b.warnf("resolver %q: its detour %q does not exist — wrong name, or the node/group/chain/egress it named was "+
|
||||||
return ""
|
"removed or could not be built. The detour is FAIL-CLOSED, so this resolver now answers NOTHING: its "+
|
||||||
|
"queries are blocked instead of being sent out over the plain WAN in clear text, which is the exact leak "+
|
||||||
|
"the detour exists to prevent (your provider would get the full list of names your network looks up, from "+
|
||||||
|
"your real address, while the kill switch blocks the traffic itself). If this is the default resolver, "+
|
||||||
|
"name resolution stops for the whole network until you restore %q or point the resolver somewhere else.",
|
||||||
|
r.Name, r.Detour, strings.TrimSpace(r.Detour))
|
||||||
|
return tagBlock
|
||||||
}
|
}
|
||||||
return tag
|
return tag
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -0,0 +1,195 @@
|
|||||||
|
package generate
|
||||||
|
|
||||||
|
// The DNS plane's two dangling-reference paths, both of which used to degrade into
|
||||||
|
// "resolve it anyway, somewhere else".
|
||||||
|
//
|
||||||
|
// A DNS leak is not a smaller version of a traffic leak. The traffic can be
|
||||||
|
// blocked by the kill switch and still leave the provider holding the complete
|
||||||
|
// list of names this network looks up, which is the part that identifies the
|
||||||
|
// household. So a broken DNS reference has to fail CLOSED for exactly the same
|
||||||
|
// reason a broken egress binding does.
|
||||||
|
|
||||||
|
import (
|
||||||
|
"strings"
|
||||||
|
"testing"
|
||||||
|
|
||||||
|
C "github.com/sagernet/sing-box/constant"
|
||||||
|
|
||||||
|
"github.com/sagernet/sing-box/shater/model"
|
||||||
|
)
|
||||||
|
|
||||||
|
// warnContaining returns the first warning containing all of the fragments.
|
||||||
|
func warnContaining(warns []string, fragments ...string) (string, bool) {
|
||||||
|
for _, w := range warns {
|
||||||
|
all := true
|
||||||
|
for _, f := range fragments {
|
||||||
|
if !strings.Contains(w, f) {
|
||||||
|
all = false
|
||||||
|
break
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if all {
|
||||||
|
return w, true
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return "", false
|
||||||
|
}
|
||||||
|
|
||||||
|
// DEFECT 1. A resolver whose detour cannot be resolved used to get Detour="" and a
|
||||||
|
// warning saying it was "using default outbound". "" is not a default outbound: it
|
||||||
|
// is an ordinary system socket on the router's own WAN. So switching off the node
|
||||||
|
// a resolver detoured through turned every lookup in the network into a plaintext
|
||||||
|
// :53 query to 1.1.1.1 across the provider's link, from the real address, while
|
||||||
|
// the kill switch was closed and the traffic itself was being blocked.
|
||||||
|
//
|
||||||
|
// The detour must fail closed to `block`, like every other dangling reference in
|
||||||
|
// this generator (egressDetourOrBlock, ruleKillFallback, chain L1, and
|
||||||
|
// wgdedup.retargetDroppedTags — which rewrites this same field to `block`).
|
||||||
|
func TestResolverDanglingDetourFailsClosed(t *testing.T) {
|
||||||
|
m := &model.Model{
|
||||||
|
Globals: model.Globals{ResolverDefault: "cf"},
|
||||||
|
Resolvers: []model.Resolver{
|
||||||
|
// node:vps existed until the operator switched the node off.
|
||||||
|
{Name: "cf", Type: "udp", Address: "1.1.1.1", Detour: "node:vps"},
|
||||||
|
},
|
||||||
|
}
|
||||||
|
opts, warns, err := GenerateWithWarnings(m)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("GenerateWithWarnings: %v", err)
|
||||||
|
}
|
||||||
|
srv, ok := dnsServerByTag(opts.DNS, "cf")
|
||||||
|
if !ok {
|
||||||
|
t.Fatalf("resolver %q must still be emitted; servers=%+v", "cf", opts.DNS)
|
||||||
|
}
|
||||||
|
if got := udpServerDetour(t, srv); got != tagBlock {
|
||||||
|
t.Fatalf("LEAK: resolver with a dangling detour has detour %q, want %q — "+
|
||||||
|
"an empty detour sends every query out over the plain WAN in clear text, "+
|
||||||
|
"handing the provider the full list of names this network resolves.", got, tagBlock)
|
||||||
|
}
|
||||||
|
// The warning has to name the CONSEQUENCE, not the mechanism: "fail-closed" is
|
||||||
|
// also what makes apply/warnings.go grade it critical.
|
||||||
|
w, ok := warnContaining(warns, `resolver "cf"`, "FAIL-CLOSED")
|
||||||
|
if !ok {
|
||||||
|
t.Fatalf("expected a fail-closed warning naming the resolver; warnings=%v", warns)
|
||||||
|
}
|
||||||
|
for _, want := range []string{"answers NOTHING", "plain WAN", "name resolution stops"} {
|
||||||
|
if !strings.Contains(w, want) {
|
||||||
|
t.Errorf("warning must state the consequence, missing %q:\n%s", want, w)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// A detour that RESOLVES must be untouched — the fix may only change the broken
|
||||||
|
// case.
|
||||||
|
func TestResolverLiveDetourUnchanged(t *testing.T) {
|
||||||
|
m := &model.Model{
|
||||||
|
Globals: model.Globals{ResolverDefault: "cf"},
|
||||||
|
Resolvers: []model.Resolver{{Name: "cf", Type: "udp", Address: "1.1.1.1", Detour: "direct"}},
|
||||||
|
}
|
||||||
|
opts, _, err := GenerateWithWarnings(m)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("GenerateWithWarnings: %v", err)
|
||||||
|
}
|
||||||
|
srv, ok := dnsServerByTag(opts.DNS, "cf")
|
||||||
|
if !ok {
|
||||||
|
t.Fatalf("resolver must be emitted; dns=%+v", opts.DNS)
|
||||||
|
}
|
||||||
|
if got := udpServerDetour(t, srv); got != tagDirect {
|
||||||
|
t.Fatalf("a live detour must survive: detour = %q, want %q", got, tagDirect)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// A fakeip resolver dials nothing at all, so its detour is genuinely inert. Failing
|
||||||
|
// it closed would be a lie in the other direction — and would contradict
|
||||||
|
// warnIgnoredResolverFields, which reports the same field as ignored.
|
||||||
|
func TestFakeIPResolverDanglingDetourIsNotFailedClosed(t *testing.T) {
|
||||||
|
m := &model.Model{
|
||||||
|
Globals: model.Globals{ResolverDefault: "real"},
|
||||||
|
Resolvers: []model.Resolver{
|
||||||
|
{Name: "real", Type: "udp", Address: "1.1.1.1"},
|
||||||
|
{Name: "fake", Type: "fakeip", Detour: "node:gone"},
|
||||||
|
},
|
||||||
|
}
|
||||||
|
_, warns, err := GenerateWithWarnings(m)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("GenerateWithWarnings: %v", err)
|
||||||
|
}
|
||||||
|
if _, bad := warnContaining(warns, `resolver "fake"`, "FAIL-CLOSED"); bad {
|
||||||
|
t.Errorf("a fakeip resolver's detour is inert; it must not be reported as fail-closed:\n%v", warns)
|
||||||
|
}
|
||||||
|
if _, ok := warnContaining(warns, `resolver "fake"`, "IGNORED"); !ok {
|
||||||
|
t.Errorf("a fakeip resolver's detour must still be reported as ignored:\n%v", warns)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// DEFECT 1, adjacent (dns.go's dns_rule loop). A rule naming a resolver that no
|
||||||
|
// longer exists used to be DROPPED, which sounds harmless and is not: the names it
|
||||||
|
// singled out then fell through to the default resolver — the silent leak of
|
||||||
|
// exactly the lookups the operator had singled out, which route.go already refuses
|
||||||
|
// for routing rules. The rule now stays and answers NXDOMAIN.
|
||||||
|
func TestDNSRuleUnknownResolverFailsClosed(t *testing.T) {
|
||||||
|
m := &model.Model{
|
||||||
|
Globals: model.Globals{ResolverDefault: "cf"},
|
||||||
|
Resolvers: []model.Resolver{{Name: "cf", Type: "udp", Address: "1.1.1.1"}},
|
||||||
|
DNSRules: []model.DNSRule{
|
||||||
|
{Resolver: "work-dns", MatchDomain: []string{"intranet.example"}},
|
||||||
|
},
|
||||||
|
}
|
||||||
|
opts, warns, err := GenerateWithWarnings(m)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("GenerateWithWarnings: %v", err)
|
||||||
|
}
|
||||||
|
if opts.DNS == nil {
|
||||||
|
t.Fatal("expected a DNS plane")
|
||||||
|
}
|
||||||
|
var found bool
|
||||||
|
for _, r := range opts.DNS.Rules {
|
||||||
|
d := r.DefaultOptions
|
||||||
|
if len(d.Domain) != 1 || d.Domain[0] != "intranet.example" {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
found = true
|
||||||
|
if d.DNSRuleAction.Action != C.RuleActionTypePredefined {
|
||||||
|
t.Fatalf("a dns_rule whose resolver is gone must answer with a predefined action, got %+v",
|
||||||
|
d.DNSRuleAction)
|
||||||
|
}
|
||||||
|
if d.DNSRuleAction.PredefinedOptions.Rcode == nil {
|
||||||
|
t.Fatalf("expected an NXDOMAIN rcode, got %+v", d.DNSRuleAction.PredefinedOptions)
|
||||||
|
}
|
||||||
|
if d.DNSRuleAction.RouteOptions.Server != "" {
|
||||||
|
t.Fatalf("the rule must not route to any server, got %q", d.DNSRuleAction.RouteOptions.Server)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if !found {
|
||||||
|
t.Fatalf("LEAK: the dns_rule was dropped, so %q now resolves through the default resolver — "+
|
||||||
|
"exactly the lookup the operator singled out, sent somewhere they did not choose. rules=%+v",
|
||||||
|
"intranet.example", opts.DNS.Rules)
|
||||||
|
}
|
||||||
|
if _, ok := warnContaining(warns, `resolver "work-dns"`, "FAIL-CLOSED", "NXDOMAIN"); !ok {
|
||||||
|
t.Fatalf("expected a fail-closed warning for the dangling dns_rule; warnings=%v", warns)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// The guard that must survive: a dns_rule with NO usable matcher is still skipped
|
||||||
|
// entirely. Emitting it with an NXDOMAIN action would answer EVERY query in the
|
||||||
|
// network with "no such name".
|
||||||
|
func TestDNSRuleUnknownResolverWithNoMatcherIsSkipped(t *testing.T) {
|
||||||
|
m := &model.Model{
|
||||||
|
Globals: model.Globals{ResolverDefault: "cf"},
|
||||||
|
Resolvers: []model.Resolver{{Name: "cf", Type: "udp", Address: "1.1.1.1"}},
|
||||||
|
DNSRules: []model.DNSRule{{Resolver: "gone"}},
|
||||||
|
}
|
||||||
|
opts, _, err := GenerateWithWarnings(m)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("GenerateWithWarnings: %v", err)
|
||||||
|
}
|
||||||
|
for _, r := range opts.DNS.Rules {
|
||||||
|
d := r.DefaultOptions
|
||||||
|
if d.DNSRuleAction.Action != C.RuleActionTypePredefined {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
if len(d.Domain)+len(d.DomainSuffix)+len(d.DomainKeyword)+len(d.SourceIPCIDR)+len(d.RuleSet) == 0 {
|
||||||
|
t.Fatalf("a matcher-less NXDOMAIN rule would blackhole the whole network: %+v", d)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -80,19 +80,26 @@ func TestDNSFilterInlineBlockAllowValidates(t *testing.T) {
|
|||||||
t.Fatalf("expected rule-sets bl-ads, bl-zeros and al-safe, got %+v", opts.Route.RuleSet)
|
t.Fatalf("expected rule-sets bl-ads, bl-zeros and al-safe, got %+v", opts.Route.RuleSet)
|
||||||
}
|
}
|
||||||
|
|
||||||
// DNS rules: allow FIRST (route->cf), then nxdomain block, then zero block.
|
// DNS rules: the .lan/PTR preservation rule dns_intercept prepends comes first
|
||||||
if opts.DNS == nil || len(opts.DNS.Rules) < 3 {
|
// (D24 — local names outrank the filter), then allow (route->cf), then the
|
||||||
t.Fatalf("expected >=3 DNS rules (allow+nxdomain+zero), got %+v", opts.DNS)
|
// nxdomain block, then the zero block. The FILTER's own ordering is what this
|
||||||
|
// test is about, so it is asserted over the rules that are the filter's.
|
||||||
|
if opts.DNS == nil || len(opts.DNS.Rules) == 0 || !isLocalZoneRule(opts.DNS.Rules[0]) {
|
||||||
|
t.Fatalf("rule 0 must be the .lan/PTR preservation rule, got %+v", opts.DNS)
|
||||||
|
}
|
||||||
|
filterRules := nonLocalDNSRules(opts)
|
||||||
|
if len(filterRules) < 3 {
|
||||||
|
t.Fatalf("expected >=3 filter DNS rules (allow+nxdomain+zero), got %+v", filterRules)
|
||||||
}
|
}
|
||||||
|
|
||||||
allow := opts.DNS.Rules[0].DefaultOptions
|
allow := filterRules[0].DefaultOptions
|
||||||
if allow.RuleSet[0] != "al-safe" || allow.DNSRuleAction.Action != C.RuleActionTypeRoute ||
|
if allow.RuleSet[0] != "al-safe" || allow.DNSRuleAction.Action != C.RuleActionTypeRoute ||
|
||||||
allow.DNSRuleAction.RouteOptions.Server != "cf" {
|
allow.DNSRuleAction.RouteOptions.Server != "cf" {
|
||||||
t.Fatalf("first DNS rule must be allow(al-safe)->route:cf, got %+v", allow)
|
t.Fatalf("first DNS rule must be allow(al-safe)->route:cf, got %+v", allow)
|
||||||
}
|
}
|
||||||
|
|
||||||
// nxdomain block: predefined action, Rcode NXDOMAIN, no answer records.
|
// nxdomain block: predefined action, Rcode NXDOMAIN, no answer records.
|
||||||
nx := opts.DNS.Rules[1].DefaultOptions
|
nx := filterRules[1].DefaultOptions
|
||||||
if nx.RuleSet[0] != "bl-ads" || nx.DNSRuleAction.Action != C.RuleActionTypePredefined {
|
if nx.RuleSet[0] != "bl-ads" || nx.DNSRuleAction.Action != C.RuleActionTypePredefined {
|
||||||
t.Fatalf("second DNS rule must be block(bl-ads)->predefined, got %+v", nx)
|
t.Fatalf("second DNS rule must be block(bl-ads)->predefined, got %+v", nx)
|
||||||
}
|
}
|
||||||
@@ -105,7 +112,7 @@ func TestDNSFilterInlineBlockAllowValidates(t *testing.T) {
|
|||||||
}
|
}
|
||||||
|
|
||||||
// zero block: predefined action, A 0.0.0.0 + AAAA :: (IPv6 on), no Rcode.
|
// zero block: predefined action, A 0.0.0.0 + AAAA :: (IPv6 on), no Rcode.
|
||||||
zero := opts.DNS.Rules[2].DefaultOptions
|
zero := filterRules[2].DefaultOptions
|
||||||
if zero.RuleSet[0] != "bl-zeros" || zero.DNSRuleAction.Action != C.RuleActionTypePredefined {
|
if zero.RuleSet[0] != "bl-zeros" || zero.DNSRuleAction.Action != C.RuleActionTypePredefined {
|
||||||
t.Fatalf("third DNS rule must be block(bl-zeros)->predefined, got %+v", zero)
|
t.Fatalf("third DNS rule must be block(bl-zeros)->predefined, got %+v", zero)
|
||||||
}
|
}
|
||||||
@@ -149,10 +156,13 @@ func TestDNSFilterZeroNoIPv6(t *testing.T) {
|
|||||||
if !changed {
|
if !changed {
|
||||||
t.Fatalf("expected changed==true (warnings: %v)", warns)
|
t.Fatalf("expected changed==true (warnings: %v)", warns)
|
||||||
}
|
}
|
||||||
if opts.DNS == nil || len(opts.DNS.Rules) < 1 {
|
// Index past the .lan/PTR preservation rule dns_intercept prepends (D24): the
|
||||||
|
// subject here is the zero-block ANSWER, not where it sits in the plane.
|
||||||
|
filterRules := nonLocalDNSRules(opts)
|
||||||
|
if len(filterRules) < 1 {
|
||||||
t.Fatalf("expected a block DNS rule, got %+v", opts.DNS)
|
t.Fatalf("expected a block DNS rule, got %+v", opts.DNS)
|
||||||
}
|
}
|
||||||
ans := opts.DNS.Rules[0].DefaultOptions.DNSRuleAction.PredefinedOptions.Answer
|
ans := filterRules[0].DefaultOptions.DNSRuleAction.PredefinedOptions.Answer
|
||||||
if len(ans) != 1 {
|
if len(ans) != 1 {
|
||||||
t.Fatalf("zero block (IPv6 off) must carry ONLY the A record, got %+v", ans)
|
t.Fatalf("zero block (IPv6 off) must carry ONLY the A record, got %+v", ans)
|
||||||
}
|
}
|
||||||
@@ -338,8 +348,12 @@ func TestDNSFilterOffEmitsNothing(t *testing.T) {
|
|||||||
if len(opts.Route.RuleSet) != 0 {
|
if len(opts.Route.RuleSet) != 0 {
|
||||||
t.Fatalf("filter off must emit no rule-sets, got %+v", opts.Route.RuleSet)
|
t.Fatalf("filter off must emit no rule-sets, got %+v", opts.Route.RuleSet)
|
||||||
}
|
}
|
||||||
if opts.DNS != nil && len(opts.DNS.Rules) != 0 {
|
// "No DNS rules at all" stopped being the right statement of this guard when
|
||||||
t.Fatalf("filter off must emit no DNS rules, got %+v", opts.DNS.Rules)
|
// dns_intercept became the default (D24): an intercepting plane always carries
|
||||||
|
// the .lan/PTR preservation rule, which is not the filter's. The guard is that
|
||||||
|
// the FILTER contributes nothing.
|
||||||
|
if fr := nonLocalDNSRules(opts); len(fr) != 0 {
|
||||||
|
t.Fatalf("filter off must emit no filter DNS rules, got %+v", fr)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
@@ -35,7 +35,7 @@ func routeActionFor(opts option.Options, tag string) *option.RouteActionOptions
|
|||||||
|
|
||||||
func dpiModel(dpi string) *model.Model {
|
func dpiModel(dpi string) *model.Model {
|
||||||
return &model.Model{
|
return &model.Model{
|
||||||
Globals: model.DefaultGlobals(),
|
Globals: nonDNSGlobals(),
|
||||||
Egresses: []model.Egress{{Name: "frag", Type: "direct", DPI: dpi}},
|
Egresses: []model.Egress{{Name: "frag", Type: "direct", DPI: dpi}},
|
||||||
Rules: []model.Rule{
|
Rules: []model.Rule{
|
||||||
{Name: "desync", Enabled: true, Order: 10, Src: []string{"192.168.1.0/24"}, Target: "egress:frag"},
|
{Name: "desync", Enabled: true, Order: 10, Src: []string{"192.168.1.0/24"}, Target: "egress:frag"},
|
||||||
|
|||||||
@@ -13,7 +13,7 @@ import (
|
|||||||
// the egress tag (multi-WAN). No copies: the binding lands on the node itself.
|
// the egress tag (multi-WAN). No copies: the binding lands on the node itself.
|
||||||
func TestNodeEgressBindsOutbound(t *testing.T) {
|
func TestNodeEgressBindsOutbound(t *testing.T) {
|
||||||
m := &model.Model{
|
m := &model.Model{
|
||||||
Globals: model.DefaultGlobals(),
|
Globals: nonDNSGlobals(),
|
||||||
Egresses: []model.Egress{{Name: "wan2", Type: "direct"}},
|
Egresses: []model.Egress{{Name: "wan2", Type: "direct"}},
|
||||||
Nodes: []model.Node{
|
Nodes: []model.Node{
|
||||||
{Name: "a", Enabled: true, URI: "ss://aes-256-gcm:secret@203.0.113.1:8388#a", Egress: "wan2"},
|
{Name: "a", Enabled: true, URI: "ss://aes-256-gcm:secret@203.0.113.1:8388#a", Egress: "wan2"},
|
||||||
@@ -80,7 +80,7 @@ func TestNodeEgressMissingIsFailClosed(t *testing.T) {
|
|||||||
// outbound (instead of dialing directly from the router).
|
// outbound (instead of dialing directly from the router).
|
||||||
func TestChainEgressEntryHop(t *testing.T) {
|
func TestChainEgressEntryHop(t *testing.T) {
|
||||||
m := &model.Model{
|
m := &model.Model{
|
||||||
Globals: model.DefaultGlobals(),
|
Globals: nonDNSGlobals(),
|
||||||
Egresses: []model.Egress{{Name: "wan2", Type: "direct"}},
|
Egresses: []model.Egress{{Name: "wan2", Type: "direct"}},
|
||||||
Nodes: []model.Node{
|
Nodes: []model.Node{
|
||||||
{Name: "x", Enabled: true, URI: "ss://aes-256-gcm:secret@203.0.113.1:8388#x"},
|
{Name: "x", Enabled: true, URI: "ss://aes-256-gcm:secret@203.0.113.1:8388#x"},
|
||||||
|
|||||||
@@ -4,6 +4,7 @@ import (
|
|||||||
"strings"
|
"strings"
|
||||||
"testing"
|
"testing"
|
||||||
|
|
||||||
|
C "github.com/sagernet/sing-box/constant"
|
||||||
"github.com/sagernet/sing-box/option"
|
"github.com/sagernet/sing-box/option"
|
||||||
|
|
||||||
"github.com/sagernet/sing-box/shater/model"
|
"github.com/sagernet/sing-box/shater/model"
|
||||||
@@ -210,11 +211,16 @@ func TestEndpointResolverIsAlwaysDirect(t *testing.T) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
// TestEndpointResolverAbsentUnset: with no endpoint_resolver configured the field
|
// TestEndpointResolverAbsentUnset: one transport and no endpoint_resolver leaves the
|
||||||
// stays unset (prior behaviour preserved, engine uses its built-in resolver).
|
// field unset. That is not merely "the feature is off": with a single DNS transport
|
||||||
|
// the engine's dialer already resolves a domain server address through
|
||||||
|
// dnsTransport.Default() (common/dialer/dialer.go), so there is nothing to fix and
|
||||||
|
// nothing to say. dns_intercept is explicitly OFF here — it would add the synthetic
|
||||||
|
// local server and make this a TWO-transport plane, which is the other case entirely
|
||||||
|
// (TestBootstrapResolverSetWhenPlaneHasTwoTransports).
|
||||||
func TestEndpointResolverAbsentUnset(t *testing.T) {
|
func TestEndpointResolverAbsentUnset(t *testing.T) {
|
||||||
m := &model.Model{
|
m := &model.Model{
|
||||||
Globals: model.Globals{ResolverDefault: "cf"},
|
Globals: model.Globals{ResolverDefault: "cf"}, // zero value => DNSIntercept off
|
||||||
Resolvers: []model.Resolver{
|
Resolvers: []model.Resolver{
|
||||||
{Name: "cf", Type: "udp", Address: "1.1.1.1", Detour: "direct"},
|
{Name: "cf", Type: "udp", Address: "1.1.1.1", Detour: "direct"},
|
||||||
},
|
},
|
||||||
@@ -227,3 +233,160 @@ func TestEndpointResolverAbsentUnset(t *testing.T) {
|
|||||||
t.Fatalf("default_domain_resolver must be unset when the feature is off, got %+v", opts.Route.DefaultDomainResolver)
|
t.Fatalf("default_domain_resolver must be unset when the feature is off, got %+v", opts.Route.DefaultDomainResolver)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// bootstrapServerTag mirrors dns.go's bootstrapResolverTagPrefix + resolver name.
|
||||||
|
func bootstrapServerTag(resolver string) string {
|
||||||
|
return bootstrapResolverTagPrefix + resolver
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestBootstrapResolverNotThroughTheTunnel is the regression this file exists for
|
||||||
|
// since dns_intercept became the default.
|
||||||
|
//
|
||||||
|
// The config is the one we RECOMMEND against leaks: a single resolver, deliberately
|
||||||
|
// detoured through the proxy node. The node itself is written as a DOMAIN, so its
|
||||||
|
// address has to be resolved before the tunnel it provides can exist.
|
||||||
|
//
|
||||||
|
// With dns_intercept on, the DNS plane carries two transports (the resolver plus the
|
||||||
|
// synthetic `shater-local-dns`), and at two transports common/dialer/dialer.go stops
|
||||||
|
// using dnsTransport.Default() and leaves the transport nil — dns/router.go then
|
||||||
|
// resolves the node's own hostname through lookupWithRules, i.e. the CLIENT DNS plane,
|
||||||
|
// whose default server is the very resolver that dials through this node. A bootstrap
|
||||||
|
// loop, in the one configuration written specifically to be safe.
|
||||||
|
//
|
||||||
|
// So: default_domain_resolver must be set, and the server it names must be DIRECT.
|
||||||
|
func TestBootstrapResolverNotThroughTheTunnel(t *testing.T) {
|
||||||
|
g := model.DefaultGlobals() // dns_intercept ON (D24) => two DNS transports
|
||||||
|
g.ResolverDefault = "cf"
|
||||||
|
m := &model.Model{
|
||||||
|
Globals: g,
|
||||||
|
Nodes: []model.Node{
|
||||||
|
// A DOMAIN server address — the whole point: it must be resolved to dial.
|
||||||
|
{Name: "vps", Enabled: true, URI: "ss://aes-256-gcm:secret@vps.example.com:8388#vps"},
|
||||||
|
},
|
||||||
|
Resolvers: []model.Resolver{
|
||||||
|
{Name: "cf", Type: "udp", Address: "1.1.1.1", Detour: "node:vps"},
|
||||||
|
},
|
||||||
|
}
|
||||||
|
opts, warns, err := GenerateWithWarnings(m)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("GenerateWithWarnings: %v", err)
|
||||||
|
}
|
||||||
|
if opts.Route == nil || opts.Route.DefaultDomainResolver == nil {
|
||||||
|
t.Fatalf("default_domain_resolver must be set: without it the node's own hostname is "+
|
||||||
|
"resolved through the client DNS plane, whose default server dials through that node; warns=%v", warns)
|
||||||
|
}
|
||||||
|
want := bootstrapServerTag("cf")
|
||||||
|
if got := opts.Route.DefaultDomainResolver.Server; got != want {
|
||||||
|
t.Fatalf("default_domain_resolver.server = %q, want the bootstrap clone %q", got, want)
|
||||||
|
}
|
||||||
|
srv, ok := dnsServerByTag(opts.DNS, want)
|
||||||
|
if !ok {
|
||||||
|
t.Fatalf("bootstrap server %q must exist among the DNS servers (a dangling reference aborts box.New); servers=%+v", want, opts.DNS.Servers)
|
||||||
|
}
|
||||||
|
if d := udpServerDetour(t, srv); d != "direct" {
|
||||||
|
t.Fatalf("the server that resolves node addresses MUST NOT go through the tunnel those nodes open: detour = %q, want %q", d, "direct")
|
||||||
|
}
|
||||||
|
// The client-facing resolver keeps its anti-leak detour: the bootstrap copy is an
|
||||||
|
// addition, not a downgrade of the client DNS plane.
|
||||||
|
client, ok := dnsServerByTag(opts.DNS, "cf")
|
||||||
|
if !ok {
|
||||||
|
t.Fatalf("client resolver %q must still be emitted", "cf")
|
||||||
|
}
|
||||||
|
if d := udpServerDetour(t, client); d != "vps" {
|
||||||
|
t.Fatalf("client resolver detour = %q, want the proxy node %q", d, "vps")
|
||||||
|
}
|
||||||
|
// A plaintext resolver forced direct for bootstrap hands the node's hostname to the
|
||||||
|
// provider in the clear. That is the one cost the operator did not choose, so it is
|
||||||
|
// never silent.
|
||||||
|
if !warnsHaveSub(warns, "hostnames of your own nodes") {
|
||||||
|
t.Fatalf("a plaintext bootstrap lookup must be warned about, got %v", warns)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestBootstrapResolverDoesNotOverrideExplicit: an operator-configured
|
||||||
|
// endpoint_resolver keeps ownership of route.default_domain_resolver. The implicit
|
||||||
|
// bootstrap server must not be built at all — a second synthetic server would be a
|
||||||
|
// third transport nobody asked for and a tag nobody can find in their config.
|
||||||
|
func TestBootstrapResolverDoesNotOverrideExplicit(t *testing.T) {
|
||||||
|
g := model.DefaultGlobals() // dns_intercept ON => the implicit path is live
|
||||||
|
g.ResolverDefault = "cf"
|
||||||
|
g.EndpointResolver = "q9"
|
||||||
|
m := &model.Model{
|
||||||
|
Globals: g,
|
||||||
|
Resolvers: []model.Resolver{
|
||||||
|
{Name: "cf", Type: "udp", Address: "1.1.1.1", Detour: "direct"},
|
||||||
|
{Name: "q9", Type: "udp", Address: "9.9.9.9", Detour: "direct"},
|
||||||
|
},
|
||||||
|
}
|
||||||
|
opts, _, err := GenerateWithWarnings(m)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("GenerateWithWarnings: %v", err)
|
||||||
|
}
|
||||||
|
if opts.Route == nil || opts.Route.DefaultDomainResolver == nil {
|
||||||
|
t.Fatalf("default_domain_resolver must be set; route=%+v", opts.Route)
|
||||||
|
}
|
||||||
|
want := endpointServerTag("q9")
|
||||||
|
if got := opts.Route.DefaultDomainResolver.Server; got != want {
|
||||||
|
t.Fatalf("the explicit endpoint_resolver must win: default_domain_resolver.server = %q, want %q", got, want)
|
||||||
|
}
|
||||||
|
for _, tag := range []string{bootstrapServerTag("cf"), bootstrapServerTag("q9")} {
|
||||||
|
if _, ok := dnsServerByTag(opts.DNS, tag); ok {
|
||||||
|
t.Fatalf("no implicit bootstrap server may be emitted when endpoint_resolver is set, found %q", tag)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestBootstrapResolverSetWhenPlaneHasTwoTransports pins the TRIGGER: two or more DNS
|
||||||
|
// transports, which is the condition common/dialer/dialer.go actually branches on —
|
||||||
|
// not the dns_intercept toggle that happened to expose it. Two plain resolvers with
|
||||||
|
// intercept OFF reach it just as well, and did so long before D24.
|
||||||
|
//
|
||||||
|
// It also pins the shipped default's shape: a single DoH resolver plus intercept is a
|
||||||
|
// two-transport plane, so the field is set, the clone stays DoH (encrypted to the same
|
||||||
|
// upstream — only the tunnel hop is dropped) and nothing is warned about.
|
||||||
|
func TestBootstrapResolverSetWhenPlaneHasTwoTransports(t *testing.T) {
|
||||||
|
m := &model.Model{
|
||||||
|
Globals: model.Globals{ResolverDefault: "cf"}, // intercept OFF
|
||||||
|
Resolvers: []model.Resolver{
|
||||||
|
{Name: "cf", Type: "udp", Address: "1.1.1.1", Detour: "direct"},
|
||||||
|
{Name: "q9", Type: "udp", Address: "9.9.9.9", Detour: "direct"},
|
||||||
|
},
|
||||||
|
}
|
||||||
|
opts, warns, err := GenerateWithWarnings(m)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("GenerateWithWarnings: %v", err)
|
||||||
|
}
|
||||||
|
if opts.Route == nil || opts.Route.DefaultDomainResolver == nil {
|
||||||
|
t.Fatalf("two transports without dns_intercept must still set default_domain_resolver; warns=%v", warns)
|
||||||
|
}
|
||||||
|
if got, want := opts.Route.DefaultDomainResolver.Server, bootstrapServerTag("cf"); got != want {
|
||||||
|
t.Fatalf("default_domain_resolver.server = %q, want a direct clone of the default resolver %q", got, want)
|
||||||
|
}
|
||||||
|
|
||||||
|
// The shipped default: one DoH resolver + intercept. The clone must still be DoH.
|
||||||
|
g := model.DefaultGlobals()
|
||||||
|
g.ResolverDefault = "cf"
|
||||||
|
m2 := &model.Model{
|
||||||
|
Globals: g,
|
||||||
|
Resolvers: []model.Resolver{
|
||||||
|
{Name: "cf", Type: "doh", Address: "https://1.1.1.1/dns-query", Detour: "direct"},
|
||||||
|
},
|
||||||
|
}
|
||||||
|
opts2, warns2, err := GenerateWithWarnings(m2)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("GenerateWithWarnings: %v", err)
|
||||||
|
}
|
||||||
|
if opts2.Route == nil || opts2.Route.DefaultDomainResolver == nil {
|
||||||
|
t.Fatalf("dns_intercept adds a second transport, so the field must be set; warns=%v", warns2)
|
||||||
|
}
|
||||||
|
srv, ok := dnsServerByTag(opts2.DNS, bootstrapServerTag("cf"))
|
||||||
|
if !ok {
|
||||||
|
t.Fatalf("bootstrap clone missing; servers=%+v", opts2.DNS.Servers)
|
||||||
|
}
|
||||||
|
if srv.Type != C.DNSTypeHTTPS {
|
||||||
|
t.Fatalf("the clone must keep the resolver's TYPE (DoH stays DoH), got %q", srv.Type)
|
||||||
|
}
|
||||||
|
if len(warns2) != 0 {
|
||||||
|
t.Fatalf("the shipped default shape must generate no warnings, got %v", warns2)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|||||||
@@ -43,7 +43,7 @@ func warnMatching(warns []string, subs ...string) []string {
|
|||||||
// carrying the given strategy.
|
// carrying the given strategy.
|
||||||
func twoNodeGroupModel(strategy string) *model.Model {
|
func twoNodeGroupModel(strategy string) *model.Model {
|
||||||
return &model.Model{
|
return &model.Model{
|
||||||
Globals: model.DefaultGlobals(),
|
Globals: nonDNSGlobals(),
|
||||||
Nodes: []model.Node{
|
Nodes: []model.Node{
|
||||||
{Name: "n1", Enabled: true, URI: ss("203.0.113.1")},
|
{Name: "n1", Enabled: true, URI: ss("203.0.113.1")},
|
||||||
{Name: "n2", Enabled: true, URI: ss("203.0.113.2")},
|
{Name: "n2", Enabled: true, URI: ss("203.0.113.2")},
|
||||||
@@ -313,7 +313,7 @@ func TestFailoverWarnsOnceAcrossChainCopy(t *testing.T) {
|
|||||||
|
|
||||||
func egressModel(egType string) *model.Model {
|
func egressModel(egType string) *model.Model {
|
||||||
return &model.Model{
|
return &model.Model{
|
||||||
Globals: model.DefaultGlobals(),
|
Globals: nonDNSGlobals(),
|
||||||
Egresses: []model.Egress{{Name: "e1", Type: egType, Interface: "eth1"}},
|
Egresses: []model.Egress{{Name: "e1", Type: egType, Interface: "eth1"}},
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -375,7 +375,7 @@ func TestEgressTypeUnknownWarns(t *testing.T) {
|
|||||||
// egress, but the native tls_* flags are never applied to one.
|
// egress, but the native tls_* flags are never applied to one.
|
||||||
func TestByedpiEgressDPIWarnsIgnored(t *testing.T) {
|
func TestByedpiEgressDPIWarnsIgnored(t *testing.T) {
|
||||||
m := &model.Model{
|
m := &model.Model{
|
||||||
Globals: model.DefaultGlobals(),
|
Globals: nonDNSGlobals(),
|
||||||
Egresses: []model.Egress{{Name: "bd", Type: "byedpi", Port: 1080, DPI: "fragment"}},
|
Egresses: []model.Egress{{Name: "bd", Type: "byedpi", Port: 1080, DPI: "fragment"}},
|
||||||
}
|
}
|
||||||
_, warns, err := GenerateWithWarnings(m)
|
_, warns, err := GenerateWithWarnings(m)
|
||||||
@@ -387,7 +387,7 @@ func TestByedpiEgressDPIWarnsIgnored(t *testing.T) {
|
|||||||
}
|
}
|
||||||
for _, d := range []string{"", "off"} {
|
for _, d := range []string{"", "off"} {
|
||||||
m2 := &model.Model{
|
m2 := &model.Model{
|
||||||
Globals: model.DefaultGlobals(),
|
Globals: nonDNSGlobals(),
|
||||||
Egresses: []model.Egress{{Name: "bd", Type: "byedpi", DPI: d}},
|
Egresses: []model.Egress{{Name: "bd", Type: "byedpi", DPI: d}},
|
||||||
}
|
}
|
||||||
if _, w, _ := GenerateWithWarnings(m2); len(w) != 0 {
|
if _, w, _ := GenerateWithWarnings(m2); len(w) != 0 {
|
||||||
@@ -413,7 +413,7 @@ func TestEgressPortIgnoredOnNonByedpi(t *testing.T) {
|
|||||||
}
|
}
|
||||||
// byedpi uses it, so no warning.
|
// byedpi uses it, so no warning.
|
||||||
m := &model.Model{
|
m := &model.Model{
|
||||||
Globals: model.DefaultGlobals(),
|
Globals: nonDNSGlobals(),
|
||||||
Egresses: []model.Egress{{Name: "e1", Type: "byedpi", Port: 9050}},
|
Egresses: []model.Egress{{Name: "e1", Type: "byedpi", Port: 9050}},
|
||||||
}
|
}
|
||||||
if _, warns, _ := GenerateWithWarnings(m); len(warns) != 0 {
|
if _, warns, _ := GenerateWithWarnings(m); len(warns) != 0 {
|
||||||
@@ -564,7 +564,7 @@ func findChainOutbound(b *builder, tag string) *option.Outbound {
|
|||||||
// resolves must still produce a plain detour to the egress, with no warning.
|
// resolves must still produce a plain detour to the egress, with no warning.
|
||||||
func TestEgressBindingValidIsUnaffected(t *testing.T) {
|
func TestEgressBindingValidIsUnaffected(t *testing.T) {
|
||||||
m := &model.Model{
|
m := &model.Model{
|
||||||
Globals: model.DefaultGlobals(),
|
Globals: nonDNSGlobals(),
|
||||||
Egresses: []model.Egress{{Name: "wan2", Type: "direct"}},
|
Egresses: []model.Egress{{Name: "wan2", Type: "direct"}},
|
||||||
Nodes: []model.Node{
|
Nodes: []model.Node{
|
||||||
{Name: "n1", Enabled: true, URI: ss("203.0.113.1"), Egress: "wan2"},
|
{Name: "n1", Enabled: true, URI: ss("203.0.113.1"), Egress: "wan2"},
|
||||||
|
|||||||
@@ -0,0 +1,79 @@
|
|||||||
|
package generate
|
||||||
|
|
||||||
|
import (
|
||||||
|
"strings"
|
||||||
|
|
||||||
|
C "github.com/sagernet/sing-box/constant"
|
||||||
|
"github.com/sagernet/sing-box/option"
|
||||||
|
|
||||||
|
"github.com/sagernet/sing-box/shater/model"
|
||||||
|
)
|
||||||
|
|
||||||
|
// nonDNSGlobals is model.DefaultGlobals for a fixture whose subject is NOT the
|
||||||
|
// DNS plane: egress binding, chain flattening, rule-set compilation, byedpi
|
||||||
|
// wiring.
|
||||||
|
//
|
||||||
|
// D24 made `dns_intercept` ON by default, so a model built straight from
|
||||||
|
// DefaultGlobals now declares "force every LAN :53 into the engine, including the
|
||||||
|
// queries addressed to the router". buildDNS answers that declaration honestly on
|
||||||
|
// both sides: a model that intercepts DNS while configuring no `config resolver`
|
||||||
|
// earns a warning (there is no resolver plane to intercept INTO — the queries end
|
||||||
|
// up at the system resolver), and a model that does have one gets the .lan / PTR
|
||||||
|
// preservation rule prepended ahead of everything else.
|
||||||
|
//
|
||||||
|
// Both are correct, and both are noise in a test about SOCKS outbounds. So the
|
||||||
|
// fixtures that use this helper state plainly that they do not intercept DNS,
|
||||||
|
// rather than keeping an "unexpected warnings" assertion that would silently
|
||||||
|
// become a DNS assertion. The tests that ARE about the DNS plane —
|
||||||
|
// dns_intercept_test.go, dnsfilter_test.go, doh_test.go, devices_test.go — keep
|
||||||
|
// the real default: that is where the intercept contract belongs and is pinned.
|
||||||
|
func nonDNSGlobals() model.Globals {
|
||||||
|
g := model.DefaultGlobals()
|
||||||
|
g.DNSIntercept = false
|
||||||
|
return g
|
||||||
|
}
|
||||||
|
|
||||||
|
// isLocalZoneRule reports whether a DNS rule is the .lan / private-PTR
|
||||||
|
// preservation rule that dns_intercept prepends (buildDNS): a domain_suffix rule
|
||||||
|
// routing those zones to the synthetic `shater-local-dns` server, which points at
|
||||||
|
// dnsmasq on 127.0.0.1:53.
|
||||||
|
//
|
||||||
|
// Since D24 that rule is at index 0 of EVERY generated DNS plane that has a
|
||||||
|
// resolver, which is why the assertions below stopped being able to say
|
||||||
|
// "opts.DNS.Rules[0] is my rule". Matching on the server tag (a PREFIX — buildDNS
|
||||||
|
// appends "-x" if the operator already used that name) keeps those tests about
|
||||||
|
// their own subject instead of about an offset.
|
||||||
|
func isLocalZoneRule(r option.DNSRule) bool {
|
||||||
|
a := r.DefaultOptions.DNSRuleAction
|
||||||
|
return a.Action == C.RuleActionTypeRoute && strings.HasPrefix(a.RouteOptions.Server, "shater-local-dns")
|
||||||
|
}
|
||||||
|
|
||||||
|
// nonLocalDNSRules is opts.DNS.Rules without the local-zone preservation rule —
|
||||||
|
// in order, so index-based "allow first, then block" assertions keep working and
|
||||||
|
// keep meaning what they say.
|
||||||
|
func nonLocalDNSRules(opts option.Options) []option.DNSRule {
|
||||||
|
if opts.DNS == nil {
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
var out []option.DNSRule
|
||||||
|
for _, r := range opts.DNS.Rules {
|
||||||
|
if isLocalZoneRule(r) {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
out = append(out, r)
|
||||||
|
}
|
||||||
|
return out
|
||||||
|
}
|
||||||
|
|
||||||
|
// hasDNSServer reports whether the emitted DNS plane carries a server with tag.
|
||||||
|
func hasDNSServer(opts option.Options, tag string) bool {
|
||||||
|
if opts.DNS == nil {
|
||||||
|
return false
|
||||||
|
}
|
||||||
|
for _, s := range opts.DNS.Servers {
|
||||||
|
if s.Tag == tag {
|
||||||
|
return true
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return false
|
||||||
|
}
|
||||||
@@ -199,6 +199,14 @@ type builder struct {
|
|||||||
endpointResolverServer *option.DNSServerOptions
|
endpointResolverServer *option.DNSServerOptions
|
||||||
endpointResolverComputed bool
|
endpointResolverComputed bool
|
||||||
|
|
||||||
|
// bootstrapResolverTag is the IMPLICIT default_domain_resolver: the tag buildDNS
|
||||||
|
// emits when the operator configured no endpoint_resolver but the DNS plane ends
|
||||||
|
// up carrying two or more transports (see implicitBootstrapResolver). "" when the
|
||||||
|
// explicit endpoint resolver is in play, when a single transport makes the field
|
||||||
|
// unnecessary, or when nothing usable could be built. Consumed once, after
|
||||||
|
// buildRoute AND buildDNS have both run, by GenerateWithWarningsAt.
|
||||||
|
bootstrapResolverTag string
|
||||||
|
|
||||||
warnings []string
|
warnings []string
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -276,6 +284,18 @@ func GenerateWithWarningsAt(m *model.Model, now time.Time) (option.Options, []st
|
|||||||
route := b.buildRoute()
|
route := b.buildRoute()
|
||||||
dns := b.buildDNS()
|
dns := b.buildDNS()
|
||||||
|
|
||||||
|
// route.default_domain_resolver, second half. buildRoute already set it from an
|
||||||
|
// operator-configured endpoint_resolver (route.go), which always wins. What it
|
||||||
|
// could NOT decide is the implicit case: that depends on how many DNS transports
|
||||||
|
// the plane ends up with, and only buildDNS knows, because it runs second and
|
||||||
|
// because dns_intercept's synthetic local server is counted too. So the implicit
|
||||||
|
// tag is wired here, where both halves have run, and only into a field the
|
||||||
|
// explicit path left empty. See implicitBootstrapResolver for why the field must
|
||||||
|
// be set at all and why it points where it does.
|
||||||
|
if route != nil && route.DefaultDomainResolver == nil && b.bootstrapResolverTag != "" {
|
||||||
|
route.DefaultDomainResolver = &option.DomainResolveOptions{Server: b.bootstrapResolverTag}
|
||||||
|
}
|
||||||
|
|
||||||
// Fold in the multi-hop chain outbounds/endpoints materialised by resolveChain
|
// Fold in the multi-hop chain outbounds/endpoints materialised by resolveChain
|
||||||
// during buildRoute/buildDNS (a `chain:<name>` target reached from a rule,
|
// during buildRoute/buildDNS (a `chain:<name>` target reached from a rule,
|
||||||
// device or dns detour). Done AFTER every resolveTarget call so all referenced
|
// device or dns detour). Done AFTER every resolveTarget call so all referenced
|
||||||
|
|||||||
@@ -93,8 +93,11 @@ func TestShadowsocksTproxyKillSwitchClosed(t *testing.T) {
|
|||||||
if !hasRouteToOutbound(opts, "ss1") {
|
if !hasRouteToOutbound(opts, "ss1") {
|
||||||
t.Fatalf("expected a route rule with action route->ss1")
|
t.Fatalf("expected a route rule with action route->ss1")
|
||||||
}
|
}
|
||||||
// DoH resolver present and default.
|
// DoH resolver present and default. The server COUNT is no longer 1: since D24
|
||||||
if opts.DNS == nil || len(opts.DNS.Servers) != 1 || opts.DNS.Final != "cf-doh" {
|
// dns_intercept is on by default, so buildDNS also emits the synthetic
|
||||||
|
// `shater-local-dns` that keeps .lan resolving through dnsmasq. What this test
|
||||||
|
// is about is that the operator's DoH server is there and is the catch-all.
|
||||||
|
if opts.DNS == nil || opts.DNS.Final != "cf-doh" || !hasDNSServer(opts, "cf-doh") {
|
||||||
t.Fatalf("expected doh resolver 'cf-doh' as DNS final, got %+v", opts.DNS)
|
t.Fatalf("expected doh resolver 'cf-doh' as DNS final, got %+v", opts.DNS)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -219,7 +222,12 @@ func TestAllReachableProtocols(t *testing.T) {
|
|||||||
const vmessLink = "vmess://eyJ2IjoiMiIsInBzIjoidm1lc3MtdyIsImFkZCI6ImV4YW1wbGUubmV0IiwicG9ydCI6IjQ0MyIsImlkIjoiMzMzMzMzMzMtMzMzMy0zMzMzLTMzMzMtMzMzMzMzMzMzMzMzIiwiYWlkIjoiMCIsInNjeSI6ImF1dG8iLCJuZXQiOiJ3cyIsImhvc3QiOiJleGFtcGxlLm5ldCIsInBhdGgiOiIvd3MiLCJ0bHMiOiJ0bHMifQ=="
|
const vmessLink = "vmess://eyJ2IjoiMiIsInBzIjoidm1lc3MtdyIsImFkZCI6ImV4YW1wbGUubmV0IiwicG9ydCI6IjQ0MyIsImlkIjoiMzMzMzMzMzMtMzMzMy0zMzMzLTMzMzMtMzMzMzMzMzMzMzMzIiwiYWlkIjoiMCIsInNjeSI6ImF1dG8iLCJuZXQiOiJ3cyIsImhvc3QiOiJleGFtcGxlLm5ldCIsInBhdGgiOiIvd3MiLCJ0bHMiOiJ0bHMifQ=="
|
||||||
|
|
||||||
m := &model.Model{
|
m := &model.Model{
|
||||||
Globals: model.DefaultGlobals(),
|
// nonDNSGlobals, not DefaultGlobals: the subject here is "every share-link
|
||||||
|
// protocol reaches box.New", and this model configures no `config resolver`,
|
||||||
|
// so the D24 default (dns_intercept ON) would earn its own honest warning
|
||||||
|
// about an intercept with nothing to intercept into — turning the
|
||||||
|
// no-warnings assertion below into a DNS assertion by accident.
|
||||||
|
Globals: nonDNSGlobals(),
|
||||||
Inbounds: []model.Inbound{{Name: "lan", Enabled: true, Type: "tproxy", TproxyPort: 12363}},
|
Inbounds: []model.Inbound{{Name: "lan", Enabled: true, Type: "tproxy", TproxyPort: 12363}},
|
||||||
Nodes: []model.Node{
|
Nodes: []model.Node{
|
||||||
{Name: "vless-ws", Enabled: true, URI: "vless://11111111-1111-1111-1111-111111111111@example.com:443?type=ws&security=tls&path=/vl&host=cdn.example.com&sni=cdn.example.com#vless-ws"},
|
{Name: "vless-ws", Enabled: true, URI: "vless://11111111-1111-1111-1111-111111111111@example.com:443?type=ws&security=tls&path=/vl&host=cdn.example.com&sni=cdn.example.com#vless-ws"},
|
||||||
|
|||||||
@@ -56,7 +56,7 @@ func anyDetour(t *testing.T, opts option.Options, tag string) string {
|
|||||||
// and one deliberately not.
|
// and one deliberately not.
|
||||||
func twoGroupsOneSubModel() *model.Model {
|
func twoGroupsOneSubModel() *model.Model {
|
||||||
return &model.Model{
|
return &model.Model{
|
||||||
Globals: model.DefaultGlobals(),
|
Globals: nonDNSGlobals(),
|
||||||
Egresses: []model.Egress{{Name: "awg", Type: "direct"}},
|
Egresses: []model.Egress{{Name: "awg", Type: "direct"}},
|
||||||
Nodes: []model.Node{
|
Nodes: []model.Node{
|
||||||
{Name: "n1", Enabled: true, URI: ss("203.0.113.1"), FromSub: "qomar"},
|
{Name: "n1", Enabled: true, URI: ss("203.0.113.1"), FromSub: "qomar"},
|
||||||
|
|||||||
@@ -737,7 +737,7 @@ func TestChainCyclesTerminateFailClosed(t *testing.T) {
|
|||||||
// chainFlattenModel is a model with three usable nodes and a closed kill-switch,
|
// chainFlattenModel is a model with three usable nodes and a closed kill-switch,
|
||||||
// ready for the chain-flatten cases below to attach chains + one targeting rule.
|
// ready for the chain-flatten cases below to attach chains + one targeting rule.
|
||||||
func chainFlattenModel() *model.Model {
|
func chainFlattenModel() *model.Model {
|
||||||
g := model.DefaultGlobals()
|
g := nonDNSGlobals()
|
||||||
g.KillSwitch = "closed"
|
g.KillSwitch = "closed"
|
||||||
return &model.Model{
|
return &model.Model{
|
||||||
Globals: g,
|
Globals: g,
|
||||||
|
|||||||
@@ -85,7 +85,7 @@ func contains(list []string, want string) bool {
|
|||||||
// so it runs on every platform.
|
// so it runs on every platform.
|
||||||
func TestRoutingRuleSetInlineDomain(t *testing.T) {
|
func TestRoutingRuleSetInlineDomain(t *testing.T) {
|
||||||
m := &model.Model{
|
m := &model.Model{
|
||||||
Globals: model.DefaultGlobals(),
|
Globals: nonDNSGlobals(),
|
||||||
Rulesets: []model.Ruleset{
|
Rulesets: []model.Ruleset{
|
||||||
{Name: "ads", Type: "domain", Source: "inline", Entries: []string{"ads.example", "doubleclick.net"}},
|
{Name: "ads", Type: "domain", Source: "inline", Entries: []string{"ads.example", "doubleclick.net"}},
|
||||||
},
|
},
|
||||||
@@ -134,7 +134,7 @@ func TestRoutingRuleSetInlineDomain(t *testing.T) {
|
|||||||
// looks configured and matches nothing.
|
// looks configured and matches nothing.
|
||||||
func TestRoutingRuleSetInlineDomainRegex(t *testing.T) {
|
func TestRoutingRuleSetInlineDomainRegex(t *testing.T) {
|
||||||
m := &model.Model{
|
m := &model.Model{
|
||||||
Globals: model.DefaultGlobals(),
|
Globals: nonDNSGlobals(),
|
||||||
Rulesets: []model.Ruleset{
|
Rulesets: []model.Ruleset{
|
||||||
{Name: "ads", Type: "domain", Source: "inline", Entries: []string{`regexp:^ads\.`}},
|
{Name: "ads", Type: "domain", Source: "inline", Entries: []string{`regexp:^ads\.`}},
|
||||||
},
|
},
|
||||||
@@ -173,7 +173,7 @@ func TestRoutingRuleSetInlineDomainRegex(t *testing.T) {
|
|||||||
// and the route rule references it.
|
// and the route rule references it.
|
||||||
func TestRoutingRuleSetInlineIPCIDR(t *testing.T) {
|
func TestRoutingRuleSetInlineIPCIDR(t *testing.T) {
|
||||||
m := &model.Model{
|
m := &model.Model{
|
||||||
Globals: model.DefaultGlobals(),
|
Globals: nonDNSGlobals(),
|
||||||
Rulesets: []model.Ruleset{
|
Rulesets: []model.Ruleset{
|
||||||
{Name: "cn", Type: "ipcidr", Source: "inline", Entries: []string{"10.0.0.0/8", "192.168.0.0/16"}},
|
{Name: "cn", Type: "ipcidr", Source: "inline", Entries: []string{"10.0.0.0/8", "192.168.0.0/16"}},
|
||||||
},
|
},
|
||||||
@@ -259,7 +259,7 @@ func TestRoutingRuleSetUnusedNotEmitted(t *testing.T) {
|
|||||||
// The single-category tag now carries the category suffix (rs-<name>-<category>).
|
// The single-category tag now carries the category suffix (rs-<name>-<category>).
|
||||||
func TestRoutingRuleSetGeositeCategory(t *testing.T) {
|
func TestRoutingRuleSetGeositeCategory(t *testing.T) {
|
||||||
m := &model.Model{
|
m := &model.Model{
|
||||||
Globals: model.DefaultGlobals(),
|
Globals: nonDNSGlobals(),
|
||||||
Rulesets: []model.Ruleset{
|
Rulesets: []model.Ruleset{
|
||||||
{Name: "yt", Source: "geosite", Categories: []string{"youtube"}},
|
{Name: "yt", Source: "geosite", Categories: []string{"youtube"}},
|
||||||
},
|
},
|
||||||
@@ -302,7 +302,7 @@ func TestRoutingRuleSetGeositeCategory(t *testing.T) {
|
|||||||
// references the ruleset matches BOTH tags.
|
// references the ruleset matches BOTH tags.
|
||||||
func TestRoutingRuleSetGeositeMultiCategory(t *testing.T) {
|
func TestRoutingRuleSetGeositeMultiCategory(t *testing.T) {
|
||||||
m := &model.Model{
|
m := &model.Model{
|
||||||
Globals: model.DefaultGlobals(),
|
Globals: nonDNSGlobals(),
|
||||||
Rulesets: []model.Ruleset{
|
Rulesets: []model.Ruleset{
|
||||||
{Name: "social", Source: "geosite", Categories: []string{"youtube", "telegram"}},
|
{Name: "social", Source: "geosite", Categories: []string{"youtube", "telegram"}},
|
||||||
},
|
},
|
||||||
@@ -346,7 +346,7 @@ func TestRoutingRuleSetGeositeMultiCategory(t *testing.T) {
|
|||||||
// lower-cased (rs-ru-ru).
|
// lower-cased (rs-ru-ru).
|
||||||
func TestRoutingRuleSetGeoipCountryLowercased(t *testing.T) {
|
func TestRoutingRuleSetGeoipCountryLowercased(t *testing.T) {
|
||||||
m := &model.Model{
|
m := &model.Model{
|
||||||
Globals: model.DefaultGlobals(),
|
Globals: nonDNSGlobals(),
|
||||||
Rulesets: []model.Ruleset{
|
Rulesets: []model.Ruleset{
|
||||||
{Name: "ru", Source: "geoip", Categories: []string{"RU"}},
|
{Name: "ru", Source: "geoip", Categories: []string{"RU"}},
|
||||||
},
|
},
|
||||||
@@ -1185,7 +1185,7 @@ func TestFileSourceExistingPathEmitted(t *testing.T) {
|
|||||||
t.Fatalf("write: %v", err)
|
t.Fatalf("write: %v", err)
|
||||||
}
|
}
|
||||||
m := &model.Model{
|
m := &model.Model{
|
||||||
Globals: model.DefaultGlobals(),
|
Globals: nonDNSGlobals(),
|
||||||
Rulesets: []model.Ruleset{
|
Rulesets: []model.Ruleset{
|
||||||
{Name: "j", Source: "file", Path: jsonPath},
|
{Name: "j", Source: "file", Path: jsonPath},
|
||||||
{Name: "s", Source: "file", Path: srsPath},
|
{Name: "s", Source: "file", Path: srsPath},
|
||||||
|
|||||||
@@ -165,7 +165,11 @@ func TestShippedTagSetConstructsDeclaredProtocols(t *testing.T) {
|
|||||||
// No inbound on purpose: this test is about protocol construction, and a
|
// No inbound on purpose: this test is about protocol construction, and a
|
||||||
// tproxy listener would demand CAP_NET_ADMIN from every runner. The tproxy
|
// tproxy listener would demand CAP_NET_ADMIN from every runner. The tproxy
|
||||||
// path is covered by the rest of the suite.
|
// path is covered by the rest of the suite.
|
||||||
m := &model.Model{Globals: model.DefaultGlobals(), Nodes: nodes}
|
// nonDNSGlobals, not DefaultGlobals: this test is about PROTOCOL construction
|
||||||
|
// under the shipped tag set, and it asserts zero generator warnings. With the
|
||||||
|
// default dns_intercept (D24) a model with no `config resolver` earns a DNS
|
||||||
|
// warning that has nothing to say about whether vless or hysteria2 compiled in.
|
||||||
|
m := &model.Model{Globals: nonDNSGlobals(), Nodes: nodes}
|
||||||
|
|
||||||
opts, warns, changed := applyAndClose(t, m)
|
opts, warns, changed := applyAndClose(t, m)
|
||||||
if !changed {
|
if !changed {
|
||||||
|
|||||||
+317
-71
@@ -57,6 +57,7 @@ import (
|
|||||||
"os"
|
"os"
|
||||||
"path/filepath"
|
"path/filepath"
|
||||||
"sync"
|
"sync"
|
||||||
|
"sync/atomic"
|
||||||
"time"
|
"time"
|
||||||
|
|
||||||
"github.com/sagernet/sing-box/shater/model"
|
"github.com/sagernet/sing-box/shater/model"
|
||||||
@@ -123,6 +124,29 @@ const (
|
|||||||
// an evicted entry that had swallowed copies prints its summary on the way
|
// an evicted entry that had swallowed copies prints its summary on the way
|
||||||
// out (marked "repeat table full"), so a counter is never simply dropped.
|
// out (marked "repeat table full"), so a counter is never simply dropped.
|
||||||
maxRepeatKeys = 256
|
maxRepeatKeys = 256
|
||||||
|
|
||||||
|
// outQueueCap bounds the hand-off between the goroutines that LOG and the one
|
||||||
|
// goroutine that WRITES. See the Sink type comment for why the hand-off exists
|
||||||
|
// at all. The queue is the entire budget for "the destination has stopped
|
||||||
|
// accepting bytes": past it, lines are dropped and counted, never waited on.
|
||||||
|
//
|
||||||
|
// Why 4096: one line is ~150 B, so a full queue is well under a megabyte on a
|
||||||
|
// 512 MB router, and at the observed flood rate (~7 lines/s) it is ten minutes
|
||||||
|
// of backlog — far longer than any transient stall worth buffering, and short
|
||||||
|
// enough that a permanently wedged reader costs a bounded amount of memory
|
||||||
|
// instead of the daemon.
|
||||||
|
outQueueCap = 4096
|
||||||
|
|
||||||
|
// outControlTimeout bounds how long a CONTROL op (a Reconfigure, a Sync) waits
|
||||||
|
// for the writer. Control ops may not be dropped — a dropped Reconfigure would
|
||||||
|
// leave the writer on the old destination for good — so they block, but never
|
||||||
|
// indefinitely: a wedged writer must cost the caller a bounded delay, which is
|
||||||
|
// the whole point of this file's change.
|
||||||
|
outControlTimeout = 2 * time.Second
|
||||||
|
|
||||||
|
// closeDrainTimeout bounds Close's wait for the queue to drain. Shutdown must
|
||||||
|
// complete even when the log destination is the thing that is stuck.
|
||||||
|
closeDrainTimeout = 3 * time.Second
|
||||||
)
|
)
|
||||||
|
|
||||||
// FilePath returns the log-file location for the given persistence choice.
|
// FilePath returns the log-file location for the given persistence choice.
|
||||||
@@ -185,22 +209,52 @@ func ConfigFromGlobals(g model.Globals) Config {
|
|||||||
|
|
||||||
// Sink is the shared line sink. Create with New; write producers' raw log bytes
|
// Sink is the shared line sink. Create with New; write producers' raw log bytes
|
||||||
// via Write (io.Writer); flip settings with Reconfigure; Close on shutdown.
|
// via Write (io.Writer); flip settings with Reconfigure; Close on shutdown.
|
||||||
|
//
|
||||||
|
// # Two halves, and why
|
||||||
|
//
|
||||||
|
// The sink is split down the middle: a FRONT that every logging goroutine runs
|
||||||
|
// (line assembly and repeat suppression, under mu) and a BACK that exactly one
|
||||||
|
// goroutine runs — the actual writes to stderr and to the file. They are joined
|
||||||
|
// by a bounded queue, and the front NEVER blocks on it: a full queue drops the
|
||||||
|
// line and counts it.
|
||||||
|
//
|
||||||
|
// That split is not tidiness, it is the difference between a degraded log and an
|
||||||
|
// unreachable router. Under procd the daemon's stderr is a PIPE, read by
|
||||||
|
// logd/logread. A pipe holds 64 KiB; once it is full and the reader has stopped
|
||||||
|
// draining it — logd restarting, a `logread -f` that went away, a stopped
|
||||||
|
// consumer — a write to it blocks forever. The sink used to perform that write
|
||||||
|
// while holding mu, and mu is taken by Write, i.e. by every line the process
|
||||||
|
// logs. One stuck pipe therefore froze the engine, every panel handler, the apply
|
||||||
|
// path and the signal loop at once: a daemon alive to `kill -0` and answering
|
||||||
|
// nothing, which is precisely the unresponsiveness cmd/shaterd/ctlclient.go's
|
||||||
|
// timeouts were written around.
|
||||||
|
//
|
||||||
|
// With the split, the only thing a wedged destination can stop is the single
|
||||||
|
// drain goroutine. Everything else keeps running and loses log lines — the
|
||||||
|
// correct thing to sacrifice, and the same principle Write already stated: a
|
||||||
|
// logging destination that fails must degrade, never propagate its failure back
|
||||||
|
// into the code that was merely trying to log.
|
||||||
|
//
|
||||||
|
// ORDER is preserved end to end. Lines and control operations travel in one
|
||||||
|
// stream, so a Reconfigure takes effect exactly at its place in it and the lines
|
||||||
|
// written before it still reach the destination they were written for. Sync and
|
||||||
|
// Close are the barriers that turn "queued" back into "written".
|
||||||
type Sink struct {
|
type Sink struct {
|
||||||
|
// outDropped counts lines discarded because the output queue was full. First
|
||||||
|
// field, and an atomic type, so it needs no lock and stays 64-bit aligned on
|
||||||
|
// the 32-bit router targets.
|
||||||
|
outDropped atomic.Uint64
|
||||||
|
|
||||||
mu sync.Mutex
|
mu sync.Mutex
|
||||||
cfg Config
|
cfg Config
|
||||||
|
|
||||||
// stderr is the syslog half's destination — the REAL os.Stderr in the
|
|
||||||
// daemon (procd relays it to logread). Injected so tests can observe it.
|
|
||||||
stderr io.Writer
|
|
||||||
|
|
||||||
buf []byte // un-terminated tail awaiting its '\n'
|
buf []byte // un-terminated tail awaiting its '\n'
|
||||||
|
|
||||||
file *os.File // active segment; nil until first file write (lazy open)
|
// outCh carries lines and control ops to the drain goroutine. Close nils it
|
||||||
fileSize int64 // bytes in the active segment
|
// under mu, which is what makes the subsequent close(ch) safe: no sender can
|
||||||
fileErr bool // a standing open/write failure was already warned about
|
// still be holding it.
|
||||||
|
outCh chan outOp
|
||||||
suspended bool // persistent-path disk guard tripped
|
drainDone chan struct{} // closed when the drain goroutine has finished
|
||||||
lastProbe time.Time // last disk-free probe
|
|
||||||
|
|
||||||
// repeat suppression (emitLocked and the repeatSeries helpers below).
|
// repeat suppression (emitLocked and the repeatSeries helpers below).
|
||||||
// series is the table of messages seen inside their window, keyed by
|
// series is the table of messages seen inside their window, keyed by
|
||||||
@@ -211,12 +265,46 @@ type Sink struct {
|
|||||||
repeatTimer *time.Timer // armed at the earliest series deadline, if any
|
repeatTimer *time.Timer // armed at the earliest series deadline, if any
|
||||||
closed bool // Close ran: the timer must not write any more
|
closed bool // Close ran: the timer must not write any more
|
||||||
|
|
||||||
// test seams
|
// --- the back half: owned by the drain goroutine, and touched by nothing
|
||||||
|
// else once New has returned ---
|
||||||
|
|
||||||
|
// stderr is the syslog half's destination — the REAL os.Stderr in the
|
||||||
|
// daemon (procd relays it to logread). Injected so tests can observe it.
|
||||||
|
stderr io.Writer
|
||||||
|
|
||||||
|
outCfg Config // the configuration the WRITER is currently applying
|
||||||
|
outReported uint64 // outDropped as of the last drop notice printed
|
||||||
|
|
||||||
|
file *os.File // active segment; nil until first file write (lazy open)
|
||||||
|
fileSize int64 // bytes in the active segment
|
||||||
|
fileErr bool // a standing open/write failure was already warned about
|
||||||
|
|
||||||
|
suspended bool // persistent-path disk guard tripped
|
||||||
|
lastProbe time.Time // last disk-free probe
|
||||||
|
|
||||||
|
closeErr error // the file-close result; published by closing drainDone
|
||||||
|
|
||||||
|
// test seams. Written only before the first Write (by New, or by a test on a
|
||||||
|
// fresh sink); read by both halves from then on.
|
||||||
now func() time.Time
|
now func() time.Time
|
||||||
free func(string) (uint64, bool)
|
free func(string) (uint64, bool)
|
||||||
window time.Duration // repeatWindow, overridable in tests
|
window time.Duration // repeatWindow, overridable in tests
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// outOp is one item on the queue between the halves. Exactly one role is
|
||||||
|
// populated per op, and they share one stream so their ORDER is the order they
|
||||||
|
// were produced in:
|
||||||
|
//
|
||||||
|
// line != nil a complete line to fan out, with the timestamp it was written at
|
||||||
|
// cfg != nil install this destination configuration (from Reconfigure)
|
||||||
|
// ack != nil close this channel once everything queued before it is written
|
||||||
|
type outOp struct {
|
||||||
|
ts time.Time
|
||||||
|
line []byte
|
||||||
|
cfg *Config
|
||||||
|
ack chan struct{}
|
||||||
|
}
|
||||||
|
|
||||||
// New returns a Sink fanning to stderr (the daemon passes os.Stderr) under cfg.
|
// New returns a Sink fanning to stderr (the daemon passes os.Stderr) under cfg.
|
||||||
// The file is opened lazily on the first line that needs it. A cfg with the
|
// The file is opened lazily on the first line that needs it. A cfg with the
|
||||||
// file OFF erases leftover segments right away — a daemon booting with the
|
// file OFF erases leftover segments right away — a daemon booting with the
|
||||||
@@ -226,15 +314,23 @@ func New(stderr io.Writer, cfg Config) *Sink {
|
|||||||
if !cfg.ToFile {
|
if !cfg.ToFile {
|
||||||
purgeLogFiles(cfg.path(), PersistPath, TmpfsPath)
|
purgeLogFiles(cfg.path(), PersistPath, TmpfsPath)
|
||||||
}
|
}
|
||||||
return &Sink{
|
s := &Sink{
|
||||||
cfg: cfg,
|
cfg: cfg,
|
||||||
|
outCfg: cfg,
|
||||||
stderr: stderr,
|
stderr: stderr,
|
||||||
series: make(map[string]*repeatSeries),
|
series: make(map[string]*repeatSeries),
|
||||||
seriesLRU: list.New(),
|
seriesLRU: list.New(),
|
||||||
|
outCh: make(chan outOp, outQueueCap),
|
||||||
|
drainDone: make(chan struct{}),
|
||||||
now: time.Now,
|
now: time.Now,
|
||||||
free: freeBytes,
|
free: freeBytes,
|
||||||
window: repeatWindow,
|
window: repeatWindow,
|
||||||
}
|
}
|
||||||
|
// The channel is passed in rather than read from the field: Close NILS the
|
||||||
|
// field, and a goroutine that read it at its own leisure would race that write
|
||||||
|
// (and, if it lost, range over nil forever).
|
||||||
|
go s.drainLoop(s.outCh)
|
||||||
|
return s
|
||||||
}
|
}
|
||||||
|
|
||||||
// Write implements io.Writer for the log producers. It never returns an error
|
// Write implements io.Writer for the log producers. It never returns an error
|
||||||
@@ -268,23 +364,27 @@ func (s *Sink) Write(p []byte) (int, error) {
|
|||||||
func (s *Sink) Reconfigure(cfg Config) {
|
func (s *Sink) Reconfigure(cfg Config) {
|
||||||
cfg = cfg.normalized()
|
cfg = cfg.normalized()
|
||||||
s.mu.Lock()
|
s.mu.Lock()
|
||||||
defer s.mu.Unlock()
|
|
||||||
if cfg == s.cfg {
|
if cfg == s.cfg {
|
||||||
|
s.mu.Unlock()
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
// Settle every series in progress under the OLD configuration: their
|
// Settle every series in progress under the OLD configuration: their
|
||||||
// summaries belong to the destination the swallowed lines were headed for.
|
// summaries belong to the destination the swallowed lines were headed for.
|
||||||
s.flushAllSeriesLocked()
|
s.flushAllSeriesLocked()
|
||||||
if cfg.path() != s.cfg.path() || !cfg.ToFile {
|
|
||||||
s.closeFileLocked()
|
|
||||||
}
|
|
||||||
if !cfg.ToFile {
|
|
||||||
purgeLogFiles(s.cfg.path(), cfg.path(), PersistPath, TmpfsPath)
|
|
||||||
}
|
|
||||||
s.suspended = false
|
|
||||||
s.lastProbe = time.Time{}
|
|
||||||
s.fileErr = false
|
|
||||||
s.cfg = cfg
|
s.cfg = cfg
|
||||||
|
// The file side of the swap — closing the old segment, erasing what a
|
||||||
|
// switched-off file must not leave behind, resetting the disk verdict — happens
|
||||||
|
// in the writer, IN ORDER, after every line queued under the old configuration.
|
||||||
|
// Otherwise a line written a microsecond ago would land in the new file, or in
|
||||||
|
// one that is about to be deleted.
|
||||||
|
queued := s.enqueueControlLocked(outOp{cfg: &cfg})
|
||||||
|
s.mu.Unlock()
|
||||||
|
// Wait for it to take effect, so a caller that reconfigures and then looks at
|
||||||
|
// the files sees the swap done. Bounded: a wedged writer costs the caller a
|
||||||
|
// delay, never the caller itself.
|
||||||
|
if queued {
|
||||||
|
s.Sync()
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
// Close flushes a pending partial line, emits the summaries of every still-open
|
// Close flushes a pending partial line, emits the summaries of every still-open
|
||||||
@@ -293,20 +393,51 @@ func (s *Sink) Reconfigure(cfg Config) {
|
|||||||
// fires after Close is a no-op.
|
// fires after Close is a no-op.
|
||||||
func (s *Sink) Close() error {
|
func (s *Sink) Close() error {
|
||||||
s.mu.Lock()
|
s.mu.Lock()
|
||||||
defer s.mu.Unlock()
|
if s.closed {
|
||||||
|
s.mu.Unlock()
|
||||||
|
return nil
|
||||||
|
}
|
||||||
if len(s.buf) > 0 {
|
if len(s.buf) > 0 {
|
||||||
s.emitLocked(s.buf)
|
s.emitLocked(s.buf)
|
||||||
s.buf = nil
|
s.buf = nil
|
||||||
}
|
}
|
||||||
s.flushAllSeriesLocked()
|
s.flushAllSeriesLocked()
|
||||||
s.closed = true
|
s.closed = true
|
||||||
if s.file != nil {
|
s.stopRepeatTimerLocked()
|
||||||
err := s.file.Close()
|
ch := s.outCh
|
||||||
s.file = nil
|
s.outCh = nil // no sender can hold it any more, so closing it below is safe
|
||||||
s.fileSize = 0
|
s.mu.Unlock()
|
||||||
return err
|
|
||||||
|
if ch == nil {
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
close(ch)
|
||||||
|
select {
|
||||||
|
case <-s.drainDone:
|
||||||
|
case <-time.After(closeDrainTimeout):
|
||||||
|
// Shutdown must complete even when the log destination is the thing that
|
||||||
|
// is stuck. The queued tail is lost; saying so is the honest ending.
|
||||||
|
return fmt.Errorf("logsink: output blocked at shutdown; %d queued line(s) were not written", len(ch))
|
||||||
|
}
|
||||||
|
return s.closeErr
|
||||||
|
}
|
||||||
|
|
||||||
|
// Sync blocks until every line written so far has reached its destinations. It is
|
||||||
|
// the barrier that makes the asynchronous writer observable — for a caller about
|
||||||
|
// to read the log file, and for the tests. Bounded by outControlTimeout: a wedged
|
||||||
|
// destination delays the caller, it does not capture it.
|
||||||
|
func (s *Sink) Sync() {
|
||||||
|
s.mu.Lock()
|
||||||
|
ack := make(chan struct{})
|
||||||
|
queued := s.enqueueControlLocked(outOp{ack: ack})
|
||||||
|
s.mu.Unlock()
|
||||||
|
if !queued {
|
||||||
|
return
|
||||||
|
}
|
||||||
|
select {
|
||||||
|
case <-ack:
|
||||||
|
case <-time.After(outControlTimeout):
|
||||||
}
|
}
|
||||||
return nil
|
|
||||||
}
|
}
|
||||||
|
|
||||||
// --- internals (caller holds s.mu) -------------------------------------------
|
// --- internals (caller holds s.mu) -------------------------------------------
|
||||||
@@ -557,13 +688,128 @@ func repeatKey(line []byte) (string, bool) {
|
|||||||
return string(b), false
|
return string(b), false
|
||||||
}
|
}
|
||||||
|
|
||||||
// writeLineLocked stamps one line with the UTC wall clock and fans it out.
|
// writeLineLocked hands one finished line to the writer. It stamps the line with
|
||||||
|
// the wall clock HERE, not in the writer, so the timestamp is the moment the line
|
||||||
|
// was produced rather than the moment the queue happened to reach it — a backlog
|
||||||
|
// must not rewrite history.
|
||||||
|
//
|
||||||
|
// The line is COPIED: the caller's slice usually points into s.buf, which the very
|
||||||
|
// next Write reuses.
|
||||||
func (s *Sink) writeLineLocked(line []byte) {
|
func (s *Sink) writeLineLocked(line []byte) {
|
||||||
if !s.cfg.ToSyslog && !s.cfg.ToFile {
|
if !s.cfg.ToSyslog && !s.cfg.ToFile {
|
||||||
return // fully off: the line is dropped, nowhere else to go
|
return // fully off: the line is dropped, nowhere else to go
|
||||||
}
|
}
|
||||||
ts := s.now().UTC().Format(time.RFC3339)
|
s.enqueueLocked(outOp{ts: s.now(), line: append([]byte(nil), line...)})
|
||||||
if s.cfg.ToSyslog && s.stderr != nil {
|
}
|
||||||
|
|
||||||
|
// enqueueLocked offers one op to the writer WITHOUT ever blocking. A full queue
|
||||||
|
// means the destination has stopped accepting bytes; the line is dropped and
|
||||||
|
// counted, and the writer reports the count as soon as it moves again. Dropping
|
||||||
|
// log lines is the price of a daemon that keeps answering — see the type comment.
|
||||||
|
func (s *Sink) enqueueLocked(op outOp) {
|
||||||
|
if s.outCh == nil {
|
||||||
|
return // closed
|
||||||
|
}
|
||||||
|
select {
|
||||||
|
case s.outCh <- op:
|
||||||
|
default:
|
||||||
|
s.outDropped.Add(1)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// enqueueControlLocked posts a control op (a configuration swap, a barrier). These
|
||||||
|
// may NOT be dropped — a lost Reconfigure would leave the writer on the old
|
||||||
|
// destination for good — so it blocks, but only up to outControlTimeout. It
|
||||||
|
// reports whether the op made it onto the queue.
|
||||||
|
func (s *Sink) enqueueControlLocked(op outOp) bool {
|
||||||
|
if s.outCh == nil {
|
||||||
|
return false
|
||||||
|
}
|
||||||
|
select {
|
||||||
|
case s.outCh <- op:
|
||||||
|
return true
|
||||||
|
default:
|
||||||
|
}
|
||||||
|
timer := time.NewTimer(outControlTimeout)
|
||||||
|
defer timer.Stop()
|
||||||
|
select {
|
||||||
|
case s.outCh <- op:
|
||||||
|
return true
|
||||||
|
case <-timer.C:
|
||||||
|
return false
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// --- the writer (the drain goroutine owns everything below) ------------------
|
||||||
|
|
||||||
|
// drainLoop is the single goroutine that touches the destinations. It is the only
|
||||||
|
// place in the package allowed to block on a write, and blocking here costs
|
||||||
|
// nothing but log lines.
|
||||||
|
func (s *Sink) drainLoop(ch chan outOp) {
|
||||||
|
defer close(s.drainDone)
|
||||||
|
for op := range ch {
|
||||||
|
s.handleOut(op)
|
||||||
|
}
|
||||||
|
if s.file != nil {
|
||||||
|
s.closeErr = s.file.Close()
|
||||||
|
s.file = nil
|
||||||
|
s.fileSize = 0
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// handleOut performs one op. The three roles are handled in this order so that a
|
||||||
|
// single op could carry more than one of them if a future caller needs it.
|
||||||
|
func (s *Sink) handleOut(op outOp) {
|
||||||
|
if op.cfg != nil {
|
||||||
|
s.applyOutConfig(*op.cfg)
|
||||||
|
}
|
||||||
|
if op.line != nil {
|
||||||
|
s.reportDropsOut(op.ts)
|
||||||
|
s.fanOut(op.ts, op.line)
|
||||||
|
}
|
||||||
|
if op.ack != nil {
|
||||||
|
close(op.ack)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// applyOutConfig installs a new destination configuration at its place in the
|
||||||
|
// stream: the old segment is closed, a switched-off file is erased at both
|
||||||
|
// standard locations and the old one, and the disk verdict starts fresh (a
|
||||||
|
// re-enabled or re-pointed file must not inherit a stale "suspended").
|
||||||
|
func (s *Sink) applyOutConfig(cfg Config) {
|
||||||
|
if cfg.path() != s.outCfg.path() || !cfg.ToFile {
|
||||||
|
s.closeFileOut()
|
||||||
|
}
|
||||||
|
if !cfg.ToFile {
|
||||||
|
purgeLogFiles(s.outCfg.path(), cfg.path(), PersistPath, TmpfsPath)
|
||||||
|
}
|
||||||
|
s.suspended = false
|
||||||
|
s.lastProbe = time.Time{}
|
||||||
|
s.fileErr = false
|
||||||
|
s.outCfg = cfg
|
||||||
|
}
|
||||||
|
|
||||||
|
// reportDropsOut prints one notice per drop episode, the moment the writer is able
|
||||||
|
// to write again. Silently losing lines is exactly the sort of dishonesty the
|
||||||
|
// repeat-summary machinery above exists to avoid, so the gap is named.
|
||||||
|
func (s *Sink) reportDropsOut(ts time.Time) {
|
||||||
|
dropped := s.outDropped.Load()
|
||||||
|
if dropped <= s.outReported {
|
||||||
|
return
|
||||||
|
}
|
||||||
|
lost := dropped - s.outReported
|
||||||
|
s.outReported = dropped
|
||||||
|
s.fanOut(ts, []byte(fmt.Sprintf("WARN logsink: dropped %d log line(s) — the log destination stopped accepting writes (total %d)", lost, dropped)))
|
||||||
|
}
|
||||||
|
|
||||||
|
// fanOut stamps one line with the UTC wall clock and writes it to the enabled
|
||||||
|
// destinations. This is where the process may block; nothing else waits on it.
|
||||||
|
func (s *Sink) fanOut(at time.Time, line []byte) {
|
||||||
|
if !s.outCfg.ToSyslog && !s.outCfg.ToFile {
|
||||||
|
return // fully off: the line is dropped, nowhere else to go
|
||||||
|
}
|
||||||
|
ts := at.UTC().Format(time.RFC3339)
|
||||||
|
if s.outCfg.ToSyslog && s.stderr != nil {
|
||||||
out := make([]byte, 0, len(ts)+1+len(line)+1)
|
out := make([]byte, 0, len(ts)+1+len(line)+1)
|
||||||
out = append(out, ts...)
|
out = append(out, ts...)
|
||||||
out = append(out, ' ')
|
out = append(out, ' ')
|
||||||
@@ -571,7 +817,7 @@ func (s *Sink) writeLineLocked(line []byte) {
|
|||||||
out = append(out, '\n')
|
out = append(out, '\n')
|
||||||
_, _ = s.stderr.Write(out)
|
_, _ = s.stderr.Write(out)
|
||||||
}
|
}
|
||||||
if s.cfg.ToFile {
|
if s.outCfg.ToFile {
|
||||||
// The file gets the line with ANSI colour codes stripped: the producers
|
// The file gets the line with ANSI colour codes stripped: the producers
|
||||||
// colour for a terminal, and a downloaded .log full of escape bytes is
|
// colour for a terminal, and a downloaded .log full of escape bytes is
|
||||||
// broken UX. The syslog copy is passed through untouched (unchanged
|
// broken UX. The syslog copy is passed through untouched (unchanged
|
||||||
@@ -582,22 +828,22 @@ func (s *Sink) writeLineLocked(line []byte) {
|
|||||||
out = append(out, ' ')
|
out = append(out, ' ')
|
||||||
out = append(out, clean...)
|
out = append(out, clean...)
|
||||||
out = append(out, '\n')
|
out = append(out, '\n')
|
||||||
s.fileWriteLocked(out)
|
s.fileWriteOut(out)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
// fileWriteLocked appends one stamped line to the active segment, rotating
|
// fileWriteOut appends one stamped line to the active segment, rotating when the
|
||||||
// when the segment budget (MaxKB/2) would be exceeded.
|
// segment budget (MaxKB/2) would be exceeded.
|
||||||
func (s *Sink) fileWriteLocked(out []byte) {
|
func (s *Sink) fileWriteOut(out []byte) {
|
||||||
if s.cfg.Persist && !s.diskOKLocked() {
|
if s.outCfg.Persist && !s.diskOKOut() {
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
if s.file == nil && !s.openLocked() {
|
if s.file == nil && !s.openOut() {
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
segCap := int64(s.cfg.MaxKB) * 1024 / 2
|
segCap := int64(s.outCfg.MaxKB) * 1024 / 2
|
||||||
if s.fileSize > 0 && s.fileSize+int64(len(out)) > segCap {
|
if s.fileSize > 0 && s.fileSize+int64(len(out)) > segCap {
|
||||||
s.rotateLocked()
|
s.rotateOut()
|
||||||
if s.file == nil {
|
if s.file == nil {
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
@@ -607,24 +853,24 @@ func (s *Sink) fileWriteLocked(out []byte) {
|
|||||||
if err != nil {
|
if err != nil {
|
||||||
if !s.fileErr {
|
if !s.fileErr {
|
||||||
s.fileErr = true
|
s.fileErr = true
|
||||||
s.warnLocked("write " + s.cfg.path() + ": " + err.Error() + " — file logging degraded")
|
s.warnOut("write " + s.outCfg.path() + ": " + err.Error() + " — file logging degraded")
|
||||||
}
|
}
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
s.fileErr = false
|
s.fileErr = false
|
||||||
}
|
}
|
||||||
|
|
||||||
// openLocked opens (creating if needed) the active segment for append and
|
// openOut opens (creating if needed) the active segment for append and learns its
|
||||||
// learns its current size so the rotation budget survives a daemon restart.
|
// current size so the rotation budget survives a daemon restart.
|
||||||
func (s *Sink) openLocked() bool {
|
func (s *Sink) openOut() bool {
|
||||||
path := s.cfg.path()
|
path := s.outCfg.path()
|
||||||
if err := os.MkdirAll(filepath.Dir(path), 0o755); err != nil {
|
if err := os.MkdirAll(filepath.Dir(path), 0o755); err != nil {
|
||||||
s.warnOpenFailLocked(err)
|
s.warnOpenFailOut(err)
|
||||||
return false
|
return false
|
||||||
}
|
}
|
||||||
f, err := os.OpenFile(path, os.O_APPEND|os.O_CREATE|os.O_WRONLY, 0o644)
|
f, err := os.OpenFile(path, os.O_APPEND|os.O_CREATE|os.O_WRONLY, 0o644)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
s.warnOpenFailLocked(err)
|
s.warnOpenFailOut(err)
|
||||||
return false
|
return false
|
||||||
}
|
}
|
||||||
var size int64
|
var size int64
|
||||||
@@ -637,20 +883,20 @@ func (s *Sink) openLocked() bool {
|
|||||||
return true
|
return true
|
||||||
}
|
}
|
||||||
|
|
||||||
func (s *Sink) warnOpenFailLocked(err error) {
|
func (s *Sink) warnOpenFailOut(err error) {
|
||||||
if s.fileErr {
|
if s.fileErr {
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
s.fileErr = true
|
s.fileErr = true
|
||||||
s.warnLocked("open " + s.cfg.path() + ": " + err.Error() + " — file logging unavailable")
|
s.warnOut("open " + s.outCfg.path() + ": " + err.Error() + " — file logging unavailable")
|
||||||
}
|
}
|
||||||
|
|
||||||
// rotateLocked replaces "<path>.1" with the full active segment and starts a
|
// rotateOut replaces "<path>.1" with the full active segment and starts a fresh
|
||||||
// fresh one. If the rename is impossible the active segment is truncated in
|
// one. If the rename is impossible the active segment is truncated in place —
|
||||||
// place — losing the older half is strictly better than growing past the cap.
|
// losing the older half is strictly better than growing past the cap.
|
||||||
func (s *Sink) rotateLocked() {
|
func (s *Sink) rotateOut() {
|
||||||
path := s.cfg.path()
|
path := s.outCfg.path()
|
||||||
s.closeFileLocked()
|
s.closeFileOut()
|
||||||
// os.Rename replaces an existing target on POSIX but not on Windows (dev
|
// os.Rename replaces an existing target on POSIX but not on Windows (dev
|
||||||
// host / tests), so drop the old .1 explicitly first.
|
// host / tests), so drop the old .1 explicitly first.
|
||||||
_ = os.Remove(path + ".1")
|
_ = os.Remove(path + ".1")
|
||||||
@@ -658,16 +904,16 @@ func (s *Sink) rotateLocked() {
|
|||||||
if f, terr := os.OpenFile(path, os.O_TRUNC|os.O_CREATE|os.O_WRONLY, 0o644); terr == nil {
|
if f, terr := os.OpenFile(path, os.O_TRUNC|os.O_CREATE|os.O_WRONLY, 0o644); terr == nil {
|
||||||
s.file = f
|
s.file = f
|
||||||
s.fileSize = 0
|
s.fileSize = 0
|
||||||
s.warnLocked("rotate " + path + ": " + err.Error() + " — truncated the active segment instead")
|
s.warnOut("rotate " + path + ": " + err.Error() + " — truncated the active segment instead")
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
s.warnLocked("rotate " + path + ": " + err.Error() + " — file logging unavailable")
|
s.warnOut("rotate " + path + ": " + err.Error() + " — file logging unavailable")
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
s.openLocked()
|
s.openOut()
|
||||||
}
|
}
|
||||||
|
|
||||||
func (s *Sink) closeFileLocked() {
|
func (s *Sink) closeFileOut() {
|
||||||
if s.file != nil {
|
if s.file != nil {
|
||||||
_ = s.file.Close()
|
_ = s.file.Close()
|
||||||
s.file = nil
|
s.file = nil
|
||||||
@@ -691,7 +937,7 @@ func purgeLogFiles(paths ...string) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
// diskOKLocked is the persistent-path guard: require the full log cap PLUS a
|
// diskOKOut is the persistent-path guard: require the full log cap PLUS a
|
||||||
// fixed floor to be free before writing flash, re-probing at most once a
|
// fixed floor to be free before writing flash, re-probing at most once a
|
||||||
// minute. An undeterminable free count fails OPEN for logging (the file is the
|
// minute. An undeterminable free count fails OPEN for logging (the file is the
|
||||||
// thing being asked for; refusing it on a stat error helps nobody) — the cap
|
// thing being asked for; refusing it on a stat error helps nobody) — the cap
|
||||||
@@ -699,40 +945,40 @@ func purgeLogFiles(paths ...string) {
|
|||||||
// stats/diskfree_unix.go: on a ~33 MB-free rootfs, filling the disk takes down
|
// stats/diskfree_unix.go: on a ~33 MB-free rootfs, filling the disk takes down
|
||||||
// far more than logging, so under pressure the log yields, warns once, and the
|
// far more than logging, so under pressure the log yields, warns once, and the
|
||||||
// daemon lives.
|
// daemon lives.
|
||||||
func (s *Sink) diskOKLocked() bool {
|
func (s *Sink) diskOKOut() bool {
|
||||||
nowT := s.now()
|
nowT := s.now()
|
||||||
if !s.lastProbe.IsZero() && nowT.Sub(s.lastProbe) < diskProbeEvery {
|
if !s.lastProbe.IsZero() && nowT.Sub(s.lastProbe) < diskProbeEvery {
|
||||||
return !s.suspended
|
return !s.suspended
|
||||||
}
|
}
|
||||||
s.lastProbe = nowT
|
s.lastProbe = nowT
|
||||||
freeB, ok := s.free(filepath.Dir(s.cfg.path()))
|
freeB, ok := s.free(filepath.Dir(s.outCfg.path()))
|
||||||
if !ok {
|
if !ok {
|
||||||
s.suspended = false
|
s.suspended = false
|
||||||
return true
|
return true
|
||||||
}
|
}
|
||||||
need := uint64(s.cfg.MaxKB)*1024 + diskFloorBytes
|
need := uint64(s.outCfg.MaxKB)*1024 + diskFloorBytes
|
||||||
if freeB < need {
|
if freeB < need {
|
||||||
if !s.suspended {
|
if !s.suspended {
|
||||||
s.suspended = true
|
s.suspended = true
|
||||||
s.closeFileLocked()
|
s.closeFileOut()
|
||||||
s.warnLocked(fmt.Sprintf("only %d KiB free at %s (need %d KiB) — suspending persistent file logging",
|
s.warnOut(fmt.Sprintf("only %d KiB free at %s (need %d KiB) — suspending persistent file logging",
|
||||||
freeB/1024, filepath.Dir(s.cfg.path()), need/1024))
|
freeB/1024, filepath.Dir(s.outCfg.path()), need/1024))
|
||||||
}
|
}
|
||||||
return false
|
return false
|
||||||
}
|
}
|
||||||
if s.suspended {
|
if s.suspended {
|
||||||
s.suspended = false
|
s.suspended = false
|
||||||
s.warnLocked("disk space recovered — resuming persistent file logging")
|
s.warnOut("disk space recovered — resuming persistent file logging")
|
||||||
}
|
}
|
||||||
return true
|
return true
|
||||||
}
|
}
|
||||||
|
|
||||||
// warnLocked reports a sink-internal problem on the syslog half only (the file
|
// warnOut reports a sink-internal problem on the syslog half only (the file
|
||||||
// half is, in every caller, the thing that is failing). With ToSyslog off there
|
// half is, in every caller, the thing that is failing). With ToSyslog off there
|
||||||
// is nowhere left the operator allowed us to speak — the warning is dropped,
|
// is nowhere left the operator allowed us to speak — the warning is dropped,
|
||||||
// which is exactly what "fully silent" means.
|
// which is exactly what "fully silent" means.
|
||||||
func (s *Sink) warnLocked(msg string) {
|
func (s *Sink) warnOut(msg string) {
|
||||||
if !s.cfg.ToSyslog || s.stderr == nil {
|
if !s.outCfg.ToSyslog || s.stderr == nil {
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
ts := s.now().UTC().Format(time.RFC3339)
|
ts := s.now().UTC().Format(time.RFC3339)
|
||||||
|
|||||||
@@ -3,6 +3,7 @@ package logsink
|
|||||||
import (
|
import (
|
||||||
"bytes"
|
"bytes"
|
||||||
"fmt"
|
"fmt"
|
||||||
|
"io"
|
||||||
"os"
|
"os"
|
||||||
"path/filepath"
|
"path/filepath"
|
||||||
"regexp"
|
"regexp"
|
||||||
@@ -214,6 +215,8 @@ func TestReconfigureSwitchesPathAndToggles(t *testing.T) {
|
|||||||
_, _ = s.Write([]byte("goes to a\n"))
|
_, _ = s.Write([]byte("goes to a\n"))
|
||||||
s.Reconfigure(cfgB)
|
s.Reconfigure(cfgB)
|
||||||
_, _ = s.Write([]byte("goes to b\n"))
|
_, _ = s.Write([]byte("goes to b\n"))
|
||||||
|
// The writer is asynchronous; Sync is the barrier that makes it observable.
|
||||||
|
s.Sync()
|
||||||
if got := readFile(t, cfgB.Path); !strings.Contains(got, "goes to b") || strings.Contains(got, "goes nowhere") {
|
if got := readFile(t, cfgB.Path); !strings.Contains(got, "goes to b") || strings.Contains(got, "goes nowhere") {
|
||||||
t.Fatalf("b.log content wrong: %q", got)
|
t.Fatalf("b.log content wrong: %q", got)
|
||||||
}
|
}
|
||||||
@@ -252,6 +255,7 @@ func TestDiskGuardSuspendsAndResumes(t *testing.T) {
|
|||||||
}
|
}
|
||||||
|
|
||||||
_, _ = s.Write([]byte("while low 1\nwhile low 2\n"))
|
_, _ = s.Write([]byte("while low 1\nwhile low 2\n"))
|
||||||
|
s.Sync()
|
||||||
if _, err := os.Stat(cfg.Path); !os.IsNotExist(err) {
|
if _, err := os.Stat(cfg.Path); !os.IsNotExist(err) {
|
||||||
t.Fatalf("file written despite low disk (err=%v)", err)
|
t.Fatalf("file written despite low disk (err=%v)", err)
|
||||||
}
|
}
|
||||||
@@ -637,6 +641,7 @@ func TestRepeatTableOverflowReportsEviction(t *testing.T) {
|
|||||||
_, _ = s.Write([]byte(engLine(999, "ERROR", "one key too many")))
|
_, _ = s.Write([]byte(engLine(999, "ERROR", "one key too many")))
|
||||||
|
|
||||||
want := "repeated 1 time (repeat table full): ERROR chain-000 is down"
|
want := "repeated 1 time (repeat table full): ERROR chain-000 is down"
|
||||||
|
s.Sync()
|
||||||
got := payloads(t, stderr.String())
|
got := payloads(t, stderr.String())
|
||||||
found := false
|
found := false
|
||||||
for _, l := range got {
|
for _, l := range got {
|
||||||
@@ -784,6 +789,7 @@ func TestRepeatFatalNeverSuppressed(t *testing.T) {
|
|||||||
_, _ = s2.Write([]byte(engLine(sec, "ERROR", "about to die")))
|
_, _ = s2.Write([]byte(engLine(sec, "ERROR", "about to die")))
|
||||||
}
|
}
|
||||||
_, _ = s2.Write([]byte(engLine(4, "FATAL", "engine is gone")))
|
_, _ = s2.Write([]byte(engLine(4, "FATAL", "engine is gone")))
|
||||||
|
s2.Sync()
|
||||||
got := payloads(t, stderr2.String())
|
got := payloads(t, stderr2.String())
|
||||||
want := []string{
|
want := []string{
|
||||||
strings.TrimSuffix(engLine(1, "ERROR", "about to die"), "\n"),
|
strings.TrimSuffix(engLine(1, "ERROR", "about to die"), "\n"),
|
||||||
@@ -916,3 +922,189 @@ func TestNewFileOffPurgesLeftovers(t *testing.T) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// --- a wedged destination must not freeze the process ------------------------
|
||||||
|
|
||||||
|
// blockingWriter is the procd stderr pipe once logd stopped reading it: the first
|
||||||
|
// write enters and never returns until the test lets it. Nothing about this is
|
||||||
|
// exotic — a pipe holds 64 KiB and then blocks its writer, forever, with no error
|
||||||
|
// and no deadline.
|
||||||
|
type blockingWriter struct {
|
||||||
|
entered chan struct{}
|
||||||
|
release chan struct{}
|
||||||
|
enterOnce sync.Once
|
||||||
|
relOnce sync.Once
|
||||||
|
}
|
||||||
|
|
||||||
|
func newBlockingWriter() *blockingWriter {
|
||||||
|
return &blockingWriter{entered: make(chan struct{}), release: make(chan struct{})}
|
||||||
|
}
|
||||||
|
|
||||||
|
func (w *blockingWriter) Write(p []byte) (int, error) {
|
||||||
|
w.enterOnce.Do(func() { close(w.entered) })
|
||||||
|
<-w.release
|
||||||
|
return len(p), nil
|
||||||
|
}
|
||||||
|
|
||||||
|
func (w *blockingWriter) unblock() { w.relOnce.Do(func() { close(w.release) }) }
|
||||||
|
|
||||||
|
// TestWedgedDestinationNeverBlocksLogging is the freeze regression.
|
||||||
|
//
|
||||||
|
// The sink used to take s.mu on entry to Write and still hold it while writing to
|
||||||
|
// stderr. Under procd that write is a write to a pipe, so a reader that stops
|
||||||
|
// draining it stops the writer — permanently, since the daemon has no deadline on
|
||||||
|
// fd 2. Every goroutine that logged then queued on s.mu: the engine, every panel
|
||||||
|
// handler, the apply path and the signal loop. The daemon stayed alive to
|
||||||
|
// `kill -0` and answered nothing, which is exactly what
|
||||||
|
// cmd/shaterd/ctlclient.go's timeouts describe running into.
|
||||||
|
//
|
||||||
|
// The contract this pins: with the destination wedged, logging still RETURNS.
|
||||||
|
// Lines are lost and counted — that is the intended sacrifice — but nothing that
|
||||||
|
// merely wanted to log is captured by the log.
|
||||||
|
func TestWedgedDestinationNeverBlocksLogging(t *testing.T) {
|
||||||
|
w := newBlockingWriter()
|
||||||
|
cfg := fileCfg(t, true, true, 0)
|
||||||
|
s := New(w, cfg)
|
||||||
|
// Releasing the destination is not enough: the writer then drains its backlog,
|
||||||
|
// which RE-CREATES the log file. Registered as a cleanup — and therefore run
|
||||||
|
// BEFORE t.TempDir's own, which was registered first and so runs last — so the
|
||||||
|
// writer is finished before the directory is removed, on the failing paths too.
|
||||||
|
t.Cleanup(func() {
|
||||||
|
w.unblock()
|
||||||
|
select {
|
||||||
|
case <-s.drainDone:
|
||||||
|
case <-time.After(10 * time.Second):
|
||||||
|
t.Error("the writer never finished after the destination recovered")
|
||||||
|
}
|
||||||
|
})
|
||||||
|
|
||||||
|
// Wedge it. The write goes through a goroutine of its own because the whole
|
||||||
|
// point of the defect is that this call did not come back: in the broken sink
|
||||||
|
// it parks inside stderr.Write while holding the mutex every other logger
|
||||||
|
// needs, so a test that made this call from the main flow would simply hang
|
||||||
|
// here instead of reporting what it found.
|
||||||
|
go func() { _, _ = s.Write([]byte("the line that wedges the pipe\n")) }()
|
||||||
|
select {
|
||||||
|
case <-w.entered:
|
||||||
|
case <-time.After(10 * time.Second):
|
||||||
|
t.Fatal("the sink never wrote to stderr at all")
|
||||||
|
}
|
||||||
|
|
||||||
|
// Everything else in the process keeps logging. Enough lines to overrun the
|
||||||
|
// queue several times over, from several goroutines, exactly as the daemon
|
||||||
|
// does it.
|
||||||
|
const writers = 4
|
||||||
|
done := make(chan struct{})
|
||||||
|
go func() {
|
||||||
|
defer close(done)
|
||||||
|
var wg sync.WaitGroup
|
||||||
|
for g := 0; g < writers; g++ {
|
||||||
|
wg.Add(1)
|
||||||
|
go func(g int) {
|
||||||
|
defer wg.Done()
|
||||||
|
for i := 0; i < outQueueCap; i++ {
|
||||||
|
_, _ = s.Write([]byte(engLine(g*outQueueCap+i, "ERROR", fmt.Sprintf("worker %d line %d", g, i))))
|
||||||
|
}
|
||||||
|
}(g)
|
||||||
|
}
|
||||||
|
wg.Wait()
|
||||||
|
}()
|
||||||
|
select {
|
||||||
|
case <-done:
|
||||||
|
case <-time.After(20 * time.Second):
|
||||||
|
t.Fatal("a logging goroutine blocked on a wedged log destination — under procd this is the whole daemon freezing")
|
||||||
|
}
|
||||||
|
|
||||||
|
// The overrun was real, and it is accounted for rather than silent.
|
||||||
|
if dropped := s.outDropped.Load(); dropped == 0 {
|
||||||
|
t.Fatal("nothing was dropped although the queue was overrun many times — the front must be bounded")
|
||||||
|
}
|
||||||
|
|
||||||
|
// Shutdown completes even while the destination is still wedged: Close is
|
||||||
|
// bounded, and it says what it could not write.
|
||||||
|
closed := make(chan error, 1)
|
||||||
|
go func() { closed <- s.Close() }()
|
||||||
|
select {
|
||||||
|
case err := <-closed:
|
||||||
|
if err == nil {
|
||||||
|
t.Fatal("Close reported success although the queued tail was never written")
|
||||||
|
}
|
||||||
|
if !strings.Contains(err.Error(), "output blocked") {
|
||||||
|
t.Fatalf("Close error = %v, want it to name the blocked output", err)
|
||||||
|
}
|
||||||
|
case <-time.After(closeDrainTimeout + 10*time.Second):
|
||||||
|
t.Fatal("Close hung on a wedged log destination — shutdown must not depend on the log being drainable")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestWedgedDestinationRecoveryReportsTheGap: when the destination starts
|
||||||
|
// accepting again, the writer says how many lines the gap swallowed. A silent
|
||||||
|
// hole in the log is the same dishonesty the repeat summaries exist to avoid.
|
||||||
|
func TestWedgedDestinationRecoveryReportsTheGap(t *testing.T) {
|
||||||
|
w := newBlockingWriter()
|
||||||
|
var recorded syncBuffer
|
||||||
|
cfg := fileCfg(t, true, false, 0)
|
||||||
|
s := New(&chainWriter{first: w, then: &recorded}, cfg)
|
||||||
|
// Releasing the destination is not enough: the writer then drains its backlog,
|
||||||
|
// which RE-CREATES the log file. Registered as a cleanup — and therefore run
|
||||||
|
// BEFORE t.TempDir's own, which was registered first and so runs last — so the
|
||||||
|
// writer is finished before the directory is removed, on the failing paths too.
|
||||||
|
t.Cleanup(func() {
|
||||||
|
w.unblock()
|
||||||
|
select {
|
||||||
|
case <-s.drainDone:
|
||||||
|
case <-time.After(10 * time.Second):
|
||||||
|
t.Error("the writer never finished after the destination recovered")
|
||||||
|
}
|
||||||
|
})
|
||||||
|
|
||||||
|
go func() { _, _ = s.Write([]byte(engLine(0, "ERROR", "wedge"))) }()
|
||||||
|
select {
|
||||||
|
case <-w.entered:
|
||||||
|
case <-time.After(10 * time.Second):
|
||||||
|
t.Fatal("the sink never wrote to stderr at all")
|
||||||
|
}
|
||||||
|
for i := 1; i <= 2*outQueueCap; i++ {
|
||||||
|
_, _ = s.Write([]byte(engLine(i, "ERROR", fmt.Sprintf("line %d", i))))
|
||||||
|
}
|
||||||
|
if s.outDropped.Load() == 0 {
|
||||||
|
t.Fatal("the queue was not overrun; the test proves nothing")
|
||||||
|
}
|
||||||
|
|
||||||
|
w.unblock()
|
||||||
|
s.Sync()
|
||||||
|
if got := recorded.String(); !strings.Contains(got, "dropped") {
|
||||||
|
t.Fatalf("recovery did not report the gap; stderr tail: %q", lastLines(got, 3))
|
||||||
|
}
|
||||||
|
_ = s.Close()
|
||||||
|
}
|
||||||
|
|
||||||
|
// chainWriter sends the first write to one writer and everything after it to
|
||||||
|
// another, so a test can wedge the destination and still observe what comes out
|
||||||
|
// once it recovers.
|
||||||
|
type chainWriter struct {
|
||||||
|
first io.Writer
|
||||||
|
then io.Writer
|
||||||
|
once sync.Once
|
||||||
|
used bool
|
||||||
|
mu sync.Mutex
|
||||||
|
}
|
||||||
|
|
||||||
|
func (c *chainWriter) Write(p []byte) (int, error) {
|
||||||
|
c.mu.Lock()
|
||||||
|
first := !c.used
|
||||||
|
c.used = true
|
||||||
|
c.mu.Unlock()
|
||||||
|
if first {
|
||||||
|
return c.first.Write(p)
|
||||||
|
}
|
||||||
|
return c.then.Write(p)
|
||||||
|
}
|
||||||
|
|
||||||
|
func lastLines(s string, n int) string {
|
||||||
|
lines := strings.Split(strings.TrimSuffix(s, "\n"), "\n")
|
||||||
|
if len(lines) > n {
|
||||||
|
lines = lines[len(lines)-n:]
|
||||||
|
}
|
||||||
|
return strings.Join(lines, "\n")
|
||||||
|
}
|
||||||
|
|||||||
@@ -0,0 +1,105 @@
|
|||||||
|
package model
|
||||||
|
|
||||||
|
// Tests for the `dns_intercept` default flip (D24): forcing ALL LAN plaintext
|
||||||
|
// :53 — including the queries a client sends to the ROUTER, which is what DHCP
|
||||||
|
// hands out — into the engine is now the DEFAULT, not an opt-in.
|
||||||
|
//
|
||||||
|
// Two properties are pinned here, and they pull in opposite directions:
|
||||||
|
//
|
||||||
|
// 1. a config that never mentions the option (every install written before it
|
||||||
|
// existed, and the state a hand-edited file is usually in) must come back ON;
|
||||||
|
// 2. an EXPLICIT `option dns_intercept '0'` must stay OFF — including across a
|
||||||
|
// WriteUCI->ReadUCI round-trip, where a bool omitted at false would be re-read
|
||||||
|
// as the seed and silently flip the operator's decision back on.
|
||||||
|
//
|
||||||
|
// (2) is the reason render.go emits booleans always. It is the exact trap that
|
||||||
|
// makes a default-true bool different from a default-false one, so it gets a test
|
||||||
|
// of its own rather than riding on the generic round-trip fixture.
|
||||||
|
|
||||||
|
import (
|
||||||
|
"os"
|
||||||
|
"path/filepath"
|
||||||
|
"strings"
|
||||||
|
"testing"
|
||||||
|
)
|
||||||
|
|
||||||
|
// TestDNSInterceptDefaultOn: the seed and an option-less config both say ON.
|
||||||
|
func TestDNSInterceptDefaultOn(t *testing.T) {
|
||||||
|
if !DefaultGlobals().DNSIntercept {
|
||||||
|
t.Fatal("DefaultGlobals().DNSIntercept = false, want true (D24: intercept is opt-out)")
|
||||||
|
}
|
||||||
|
m, err := ParseUCIExport("package shater\n\nconfig globals 'globals'\n\toption enabled '1'\n")
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
if !m.Globals.DNSIntercept {
|
||||||
|
t.Fatal("a config with no dns_intercept option parsed as OFF; an absent option must fall back to the ON seed")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestDNSInterceptExplicitOffPreserved: the operator's opt-out survives both the
|
||||||
|
// parse (over an ON seed) and the render->parse round-trip.
|
||||||
|
func TestDNSInterceptExplicitOffPreserved(t *testing.T) {
|
||||||
|
m, err := ParseUCIExport("package shater\n\nconfig globals 'globals'\n" +
|
||||||
|
"\toption enabled '1'\n\toption dns_intercept '0'\n")
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
if m.Globals.DNSIntercept {
|
||||||
|
t.Fatal("explicit dns_intercept '0' was overwritten by the ON seed")
|
||||||
|
}
|
||||||
|
|
||||||
|
rendered := RenderUCIExport(m)
|
||||||
|
if !strings.Contains(rendered, "option dns_intercept '0'") {
|
||||||
|
t.Fatalf("render must emit the explicit '0' (an omitted bool would be re-read as ON):\n%s", rendered)
|
||||||
|
}
|
||||||
|
back, err := ParseUCIExport(rendered)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
if back.Globals.DNSIntercept {
|
||||||
|
t.Fatal("dns_intercept '0' did not survive the WriteUCI->ReadUCI round-trip — the opt-out would be silently re-enabled on the next save")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestDNSInterceptExplicitOnParses: the ON spelling is read as ON too (a seed
|
||||||
|
// that happens to agree must not be the only reason the value is true).
|
||||||
|
func TestDNSInterceptExplicitOnParses(t *testing.T) {
|
||||||
|
m, err := ParseUCIExport("package shater\n\nconfig globals 'globals'\n\toption dns_intercept '1'\n")
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
if !m.Globals.DNSIntercept {
|
||||||
|
t.Fatal("explicit dns_intercept '1' parsed as OFF")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestShippedConfigEnablesDNSIntercept reads the file the package actually
|
||||||
|
// installs. /etc/config/shater is a CONFFILE: it is written once, on first
|
||||||
|
// install, and never replaced on upgrade — so a fresh install's DNS posture is
|
||||||
|
// decided by this file's literal text, not by DefaultGlobals. The two must agree,
|
||||||
|
// and only a test that reads the shipped bytes can say that they do.
|
||||||
|
//
|
||||||
|
// It also pins the inert shipping posture (enabled '0') the file's own header
|
||||||
|
// promises, since both facts live in the same section.
|
||||||
|
func TestShippedConfigEnablesDNSIntercept(t *testing.T) {
|
||||||
|
path := filepath.Join("..", "..", "openwrt", "shater-core", "files", "etc", "config", "shater")
|
||||||
|
raw, err := os.ReadFile(path)
|
||||||
|
if err != nil {
|
||||||
|
t.Skipf("shipped config not readable from this checkout (%v)", err)
|
||||||
|
}
|
||||||
|
text := string(raw)
|
||||||
|
if !strings.Contains(text, "option dns_intercept '1'") {
|
||||||
|
t.Error("the shipped /etc/config/shater must set dns_intercept '1' explicitly: a fresh install reads this file, and an operator who later opts out must see the option they are flipping")
|
||||||
|
}
|
||||||
|
m, err := ParseUCIExport(text)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("shipped config does not parse: %v", err)
|
||||||
|
}
|
||||||
|
if !m.Globals.DNSIntercept {
|
||||||
|
t.Error("shipped config parses with DNSIntercept off")
|
||||||
|
}
|
||||||
|
if m.Globals.Enabled {
|
||||||
|
t.Error("shipped config must stay inert (enabled '0'): a fresh install may not touch connectivity")
|
||||||
|
}
|
||||||
|
}
|
||||||
+36
-2
@@ -212,8 +212,34 @@ type Globals struct {
|
|||||||
GeositeIndexURL string // GitHub git-trees URL used to suggest categories; "" = no suggestions
|
GeositeIndexURL string // GitHub git-trees URL used to suggest categories; "" = no suggestions
|
||||||
GeoipIndexURL string
|
GeoipIndexURL string
|
||||||
DNSFilter bool // master enable for the DNS bl/allow-list filter (D15); default false (opt-in)
|
DNSFilter bool // master enable for the DNS bl/allow-list filter (D15); default false (opt-in)
|
||||||
DNSIntercept bool // force ALL LAN plaintext DNS (:53) through the engine, incl. router-addressed queries; default false (opt-in)
|
// DNSIntercept forces ALL LAN plaintext DNS (:53) through the engine, INCLUDING
|
||||||
BlockDoH bool // block known public DoH resolvers (:443) so clients fall back to plaintext :53 (which the engine catches); default false (opt-in)
|
// queries addressed to the router itself. Default TRUE (opt-out) — see D24.
|
||||||
|
//
|
||||||
|
// It has to be the default, because the alternative is inverted: without it the
|
||||||
|
// nft prerouting :53 divert is gated (netplane/nft.go) and a router-addressed
|
||||||
|
// query falls through `fib daddr type local accept` into dnsmasq and out to the
|
||||||
|
// ISP in the clear — so the client with STANDARD settings (DNS = the router, as
|
||||||
|
// DHCP hands it out) leaked past the filter, the blocklists, the per-device
|
||||||
|
// rules and the resolver detour, while the client that hard-coded 8.8.8.8 "to
|
||||||
|
// bypass the router" was caught by the ordinary tproxy catch-all. Obedient
|
||||||
|
// clients leaked; evaders did not.
|
||||||
|
//
|
||||||
|
// Consequences of ON, all of them intended:
|
||||||
|
// - .lan and the RFC6303 private reverse zones are preserved by a synthetic
|
||||||
|
// server pointed at dnsmasq on 127.0.0.1:53 plus a prepended domain_suffix
|
||||||
|
// rule (generate/dns.go). That preservation needs at least one `config
|
||||||
|
// resolver`; with none, buildDNS emits no DNS plane at all and the engine
|
||||||
|
// falls back to its built-in `local` transport, which reads /etc/resolv.conf
|
||||||
|
// (127.0.0.1 = dnsmasq) — local names still resolve, but nothing is filtered.
|
||||||
|
// - while the engine is DOWN, router-addressed :53 is NOT blacked out: the
|
||||||
|
// holding plane hooks `forward` only and dnsmasq keeps answering (unfiltered,
|
||||||
|
// plaintext to the ISP). Queries aimed at an EXTERNAL resolver are dropped
|
||||||
|
// with the rest of the LAN's forwarded traffic. See netplane/nft.go.
|
||||||
|
// - an EXPLICIT `option dns_intercept '0'` still wins: render.go always emits
|
||||||
|
// the bool, so an operator's opt-out survives the WriteUCI->ReadUCI round-trip
|
||||||
|
// and is never silently re-enabled by this seed.
|
||||||
|
DNSIntercept bool
|
||||||
|
BlockDoH bool // block known public DoH resolvers (:443) so clients fall back to plaintext :53 (which the engine catches); default false (opt-in)
|
||||||
// GroupHealth is the master-switch of OUR background group health-probing (the
|
// GroupHealth is the master-switch of OUR background group health-probing (the
|
||||||
// scheduled probing that keeps a group's member delays fresh, see
|
// scheduled probing that keeps a group's member delays fresh, see
|
||||||
// apply.configureSweep); default TRUE (opt-out). It does NOT touch sing-box's own
|
// apply.configureSweep); default TRUE (opt-out). It does NOT touch sing-box's own
|
||||||
@@ -294,6 +320,13 @@ const (
|
|||||||
// The operational-log knobs are seeded to syslog ON + file ON + tmpfs + 2048 KiB: a
|
// The operational-log knobs are seeded to syslog ON + file ON + tmpfs + 2048 KiB: a
|
||||||
// config written before they existed keeps today's behaviour (lines reach logread) and
|
// config written before they existed keeps today's behaviour (lines reach logread) and
|
||||||
// additionally gains the bounded tmpfs file the /api/log download reads.
|
// additionally gains the bounded tmpfs file the /api/log download reads.
|
||||||
|
//
|
||||||
|
// DNSIntercept is seeded ON (opt-out, D24): a config that never mentions
|
||||||
|
// `dns_intercept` — every install predating the option — starts routing
|
||||||
|
// router-addressed :53 into the engine, because the opt-in default protected only
|
||||||
|
// the clients that were trying to evade the router (see the field comment). An
|
||||||
|
// EXPLICIT `option dns_intercept '0'` parses over this seed and is preserved:
|
||||||
|
// render.go always emits the bool, so opting out is a decision the config keeps.
|
||||||
func DefaultGlobals() Globals {
|
func DefaultGlobals() Globals {
|
||||||
return Globals{
|
return Globals{
|
||||||
Enabled: true,
|
Enabled: true,
|
||||||
@@ -307,6 +340,7 @@ func DefaultGlobals() Globals {
|
|||||||
FwmarkBase: 0x2000,
|
FwmarkBase: 0x2000,
|
||||||
TableBase: 0x2000,
|
TableBase: 0x2000,
|
||||||
GroupHealth: true,
|
GroupHealth: true,
|
||||||
|
DNSIntercept: true,
|
||||||
StatsBackend: "memory",
|
StatsBackend: "memory",
|
||||||
StatsRingSize: 200,
|
StatsRingSize: 200,
|
||||||
StatsTimelineMinutes: 60,
|
StatsTimelineMinutes: 60,
|
||||||
|
|||||||
+249
-19
@@ -12,6 +12,7 @@ import (
|
|||||||
"net/netip"
|
"net/netip"
|
||||||
"os/exec"
|
"os/exec"
|
||||||
"strings"
|
"strings"
|
||||||
|
"sync"
|
||||||
"time"
|
"time"
|
||||||
|
|
||||||
"github.com/sagernet/sing-box/shater/model"
|
"github.com/sagernet/sing-box/shater/model"
|
||||||
@@ -134,6 +135,23 @@ func addRouting(mark, table uint32, ipv6 bool) error {
|
|||||||
fams = append(fams, "-6")
|
fams = append(fams, "-6")
|
||||||
}
|
}
|
||||||
for _, fam := range fams {
|
for _, fam := range fams {
|
||||||
|
// Already intact? Then do NOTHING for this family.
|
||||||
|
//
|
||||||
|
// The reconciliation below is idempotent BY del-then-add, which means it
|
||||||
|
// opens a window with no fwmark rule installed, and any packet diverted
|
||||||
|
// during it misses the `local default dev lo` table and is eaten by the
|
||||||
|
// fail-closed forward drop. That is an acceptable cost when something
|
||||||
|
// actually has to be rebuilt — and pure damage when nothing does.
|
||||||
|
//
|
||||||
|
// It matters more now that RoutingPresent also fails on a broken per-egress
|
||||||
|
// binding (see below): an egress that cannot be installed at all — a wg
|
||||||
|
// interface that is down and stays down — makes every one-minute reconcile
|
||||||
|
// call back into here. Without this guard each of those would tear the MAIN
|
||||||
|
// tproxy rule down and put it back, once a minute, forever, for a fault that
|
||||||
|
// has nothing to do with it.
|
||||||
|
if mainRoutingPresentFam(fam, mark, table) {
|
||||||
|
continue
|
||||||
|
}
|
||||||
run(fam, "rule", "del", "fwmark", fmt.Sprintf("0x%x", mark), "lookup", fmt.Sprintf("%d", table))
|
run(fam, "rule", "del", "fwmark", fmt.Sprintf("0x%x", mark), "lookup", fmt.Sprintf("%d", table))
|
||||||
if out, err := execCommand("ip", fam, "rule", "add", "fwmark",
|
if out, err := execCommand("ip", fam, "rule", "add", "fwmark",
|
||||||
fmt.Sprintf("0x%x", mark), "lookup", fmt.Sprintf("%d", table)).CombinedOutput(); err != nil {
|
fmt.Sprintf("0x%x", mark), "lookup", fmt.Sprintf("%d", table)).CombinedOutput(); err != nil {
|
||||||
@@ -156,12 +174,100 @@ func removeRouting(mark, table uint32) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// egressBinding is one (family, mark, table) policy-routing binding that
|
||||||
|
// addEgressRouting INTENDED to install for a named egress. See
|
||||||
|
// rememberEgressBindings for why the intent, and not the outcome, is what gets
|
||||||
|
// recorded.
|
||||||
|
type egressBinding struct {
|
||||||
|
fam string // "-4" / "-6"
|
||||||
|
mark uint32
|
||||||
|
table uint32
|
||||||
|
}
|
||||||
|
|
||||||
|
// egressBindings is what the last addEgressRouting run set out to install, and it
|
||||||
|
// is the ONLY thing that tells RoutingPresent which per-egress rules and tables
|
||||||
|
// have to exist right now.
|
||||||
|
//
|
||||||
|
// It has to be remembered rather than derived: RoutingPresent is handed a
|
||||||
|
// model.Globals, and the number of egresses — the one input that decides which
|
||||||
|
// marks and tables exist — lives on model.Model, not on Globals. Recomputing it
|
||||||
|
// from the live system is not possible either: an egress whose rule has been
|
||||||
|
// deleted leaves nothing behind to count.
|
||||||
|
//
|
||||||
|
// The record is always populated by the time its answer can matter: applyLocked
|
||||||
|
// evaluates `!nftCurrent || !RoutingPresent(...)`, and nftCurrent can only be true
|
||||||
|
// after an apply in THIS process has already run ApplyRoutingWithWarnings.
|
||||||
|
var (
|
||||||
|
egressBindingsMu sync.Mutex
|
||||||
|
egressBindings []egressBinding
|
||||||
|
)
|
||||||
|
|
||||||
|
// rememberEgressBindings records what a run of addEgressRouting was supposed to
|
||||||
|
// install, whether or not each individual `ip` call succeeded.
|
||||||
|
//
|
||||||
|
// INTENT, not outcome, on purpose. A binding that failed to install (the egress
|
||||||
|
// device does not exist yet — `ifdown wg0`, a tunnel that comes up late) is
|
||||||
|
// exactly the one that must keep RoutingPresent false, so the next reconcile
|
||||||
|
// tries again and the egress repairs itself the moment its interface returns.
|
||||||
|
// Recording only the successes would make a failed egress permanently invisible
|
||||||
|
// to the presence check, which is the shape of the defect this fixes.
|
||||||
|
func rememberEgressBindings(b []egressBinding) {
|
||||||
|
egressBindingsMu.Lock()
|
||||||
|
egressBindings = b
|
||||||
|
egressBindingsMu.Unlock()
|
||||||
|
}
|
||||||
|
|
||||||
|
// snapshotEgressBindings returns a copy of the recorded bindings.
|
||||||
|
func snapshotEgressBindings() []egressBinding {
|
||||||
|
egressBindingsMu.Lock()
|
||||||
|
defer egressBindingsMu.Unlock()
|
||||||
|
return append([]egressBinding(nil), egressBindings...)
|
||||||
|
}
|
||||||
|
|
||||||
|
// runIP performs one `ip` mutation and returns a reason (command + stderr) when it
|
||||||
|
// fails. Used for the mutations whose failure is a routing hole; the del/flush
|
||||||
|
// half of the del-then-add idiom keeps using the ignore-errors form, because
|
||||||
|
// "there was nothing to delete" is the normal case there and not a fault.
|
||||||
|
func runIP(args ...string) error {
|
||||||
|
out, err := execCommand("ip", args...).CombinedOutput()
|
||||||
|
if err != nil {
|
||||||
|
if txt := strings.TrimSpace(string(out)); txt != "" {
|
||||||
|
return fmt.Errorf("`ip %s` failed: %v: %s", strings.Join(args, " "), err, txt)
|
||||||
|
}
|
||||||
|
return fmt.Errorf("`ip %s` failed: %v", strings.Join(args, " "), err)
|
||||||
|
}
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
|
||||||
// addEgressRouting realises the policy-routing half of an interface/tunnel egress
|
// addEgressRouting realises the policy-routing half of an interface/tunnel egress
|
||||||
// (the generator emits the SO_BINDTODEVICE+SO_MARK outbound; this binds the mark
|
// (the generator emits the SO_BINDTODEVICE+SO_MARK outbound; this binds the mark
|
||||||
// to a table whose default route leaves via the egress device). Idempotent.
|
// to a table whose default route leaves via the egress device). Idempotent.
|
||||||
|
//
|
||||||
|
// # Every failing `ip` call is reported
|
||||||
|
//
|
||||||
|
// This used to run ALL of its mutations through a `_ = ...Run()` helper, so an
|
||||||
|
// egress bound to an interface that did not exist at apply time — `ifdown wg0`, a
|
||||||
|
// tunnel that has not come up yet, a renamed device — produced a completely
|
||||||
|
// successful apply: no error, no warning, a green egress in the panel, and every
|
||||||
|
// node, group and rule bound to it marked for a routing table that was never
|
||||||
|
// created. The mark then finds no table of its own, falls through to the main
|
||||||
|
// table and leaves over the ordinary WAN with the router's real address. That is
|
||||||
|
// the leak the egress exists to prevent, delivered silently.
|
||||||
|
//
|
||||||
|
// The rule/route ADDs are therefore checked and warned about individually, with
|
||||||
|
// the egress name and the reason `ip` gave. The del/flush calls stay
|
||||||
|
// ignore-errors: they are the "make this idempotent" half and they legitimately
|
||||||
|
// fail when there is nothing there yet.
|
||||||
|
//
|
||||||
|
// A failure is a WARNING and not a returned error deliberately. Returning would
|
||||||
|
// abort the whole apply (the caller propagates it), leaving the sysctl and
|
||||||
|
// conntrack steps that follow un-run over one broken uplink — the household-wide
|
||||||
|
// punishment for a scoped fault that this codebase refuses everywhere else. The
|
||||||
|
// warning is worded so apply/warnings.go grades it critical, which is what it is.
|
||||||
func addEgressRouting(m *model.Model) ([]string, error) {
|
func addEgressRouting(m *model.Model) ([]string, error) {
|
||||||
run := func(args ...string) { _ = execCommand("ip", args...).Run() }
|
run := func(args ...string) { _ = execCommand("ip", args...).Run() }
|
||||||
var warnings []string
|
var warnings []string
|
||||||
|
var bindings []egressBinding
|
||||||
for i, eg := range m.Egresses {
|
for i, eg := range m.Egresses {
|
||||||
t := strings.ToLower(eg.Type)
|
t := strings.ToLower(eg.Type)
|
||||||
if (t != "interface" && t != "tunnel") || eg.Interface == "" {
|
if (t != "interface" && t != "tunnel") || eg.Interface == "" {
|
||||||
@@ -175,24 +281,50 @@ func addEgressRouting(m *model.Model) ([]string, error) {
|
|||||||
fams = append(fams, "-6")
|
fams = append(fams, "-6")
|
||||||
}
|
}
|
||||||
for _, fam := range fams {
|
for _, fam := range fams {
|
||||||
run(fam, "rule", "del", "fwmark", fmt.Sprintf("0x%x", mark), "lookup", fmt.Sprintf("%d", table))
|
bindings = append(bindings, egressBinding{fam: fam, mark: mark, table: table})
|
||||||
run(fam, "rule", "add", "fwmark", fmt.Sprintf("0x%x", mark), "lookup", fmt.Sprintf("%d", table))
|
markArg, tableArg := fmt.Sprintf("0x%x", mark), fmt.Sprintf("%d", table)
|
||||||
run(fam, "route", "flush", "table", fmt.Sprintf("%d", table))
|
|
||||||
|
run(fam, "rule", "del", "fwmark", markArg, "lookup", tableArg)
|
||||||
|
if err := runIP(fam, "rule", "add", "fwmark", markArg, "lookup", tableArg); err != nil {
|
||||||
|
// The `kind "name": message` prefix is the shape apply's warning
|
||||||
|
// normaliser parses (entityRe), so this lands as an egress-scoped,
|
||||||
|
// named warning in the panel rather than an anonymous line.
|
||||||
|
warnings = append(warnings, fmt.Sprintf(
|
||||||
|
"egress %q: its %s policy rule (fwmark %s -> table %d) could not be installed, so this egress "+
|
||||||
|
"is NOT applied: every node, group and rule bound to it is still marked, finds no routing "+
|
||||||
|
"table of its own, falls through to the main table and leaves over the plain WAN with your "+
|
||||||
|
"real IP address. Reason: %v",
|
||||||
|
eg.Name, famLabel(fam), markArg, table, err))
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
run(fam, "route", "flush", "table", tableArg)
|
||||||
|
|
||||||
// A gateway'd interface (WAN) needs `via <gw>`; a point-to-point tunnel
|
// A gateway'd interface (WAN) needs `via <gw>`; a point-to-point tunnel
|
||||||
// (wg/awg) has no gateway and routes straight out the device.
|
// (wg/awg) has no gateway and routes straight out the device.
|
||||||
if gw := egressGateway(fam, eg.Interface, dev); gw != "" {
|
gw := egressGateway(fam, eg.Interface, dev)
|
||||||
run(fam, "route", "add", "default", "via", gw, "dev", dev, "table", fmt.Sprintf("%d", table))
|
var rerr error
|
||||||
|
if gw != "" {
|
||||||
|
rerr = runIP(fam, "route", "add", "default", "via", gw, "dev", dev, "table", tableArg)
|
||||||
|
} else {
|
||||||
|
rerr = runIP(fam, "route", "add", "default", "dev", dev, "table", tableArg)
|
||||||
|
}
|
||||||
|
if rerr != nil {
|
||||||
|
warnings = append(warnings, fmt.Sprintf(
|
||||||
|
"egress %q: its %s default route could not be installed into table %d, so that table is EMPTY "+
|
||||||
|
"and this egress is NOT applied: every node, group and rule bound to it is marked for an "+
|
||||||
|
"empty table, falls through to the main table and leaves over the plain WAN with your real "+
|
||||||
|
"IP address. Check that interface %q (device %s) exists and is up. Reason: %v",
|
||||||
|
eg.Name, famLabel(fam), table, eg.Interface, dev, rerr))
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
if gw != "" {
|
||||||
continue
|
continue
|
||||||
}
|
}
|
||||||
run(fam, "route", "add", "default", "dev", dev, "table", fmt.Sprintf("%d", table))
|
|
||||||
// No nexthop AND not a point-to-point device: the route just installed says
|
// No nexthop AND not a point-to-point device: the route just installed says
|
||||||
// "everything is on the local segment", which for any address outside this
|
// "everything is on the local segment", which for any address outside this
|
||||||
// subnet is a guaranteed failure. The egress is configured, reports as
|
// subnet is a guaranteed failure. The egress is configured, reports as
|
||||||
// applied, and cannot carry a single packet off-link — say so.
|
// applied, and cannot carry a single packet off-link — say so.
|
||||||
if !isPointToPoint(dev) {
|
if !isPointToPoint(dev) {
|
||||||
// The `kind "name": message` prefix is the shape apply's warning
|
|
||||||
// normaliser parses (entityRe), so this lands as an egress-scoped,
|
|
||||||
// named warning in the panel rather than an anonymous line.
|
|
||||||
warnings = append(warnings, fmt.Sprintf(
|
warnings = append(warnings, fmt.Sprintf(
|
||||||
"egress %q: no %s gateway could be found for interface %q (device %s) "+
|
"egress %q: no %s gateway could be found for interface %q (device %s) "+
|
||||||
"by any means, so its routing table sends traffic straight onto the "+
|
"by any means, so its routing table sends traffic straight onto the "+
|
||||||
@@ -204,6 +336,7 @@ func addEgressRouting(m *model.Model) ([]string, error) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
rememberEgressBindings(bindings)
|
||||||
return warnings, nil
|
return warnings, nil
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -227,6 +360,10 @@ func uciGatewayOption(fam string) string {
|
|||||||
// block (safe when nothing is set up).
|
// block (safe when nothing is set up).
|
||||||
func removeEgressRouting(m *model.Model) {
|
func removeEgressRouting(m *model.Model) {
|
||||||
run := func(args ...string) { _ = execCommand("ip", args...).Run() }
|
run := func(args ...string) { _ = execCommand("ip", args...).Run() }
|
||||||
|
// Nothing is installed any more, so nothing may be claimed as installed: a
|
||||||
|
// stale record would make RoutingPresent report a torn-down plane as broken
|
||||||
|
// rather than as absent, and the main-pair check answers that question already.
|
||||||
|
rememberEgressBindings(nil)
|
||||||
for i := range m.Egresses {
|
for i := range m.Egresses {
|
||||||
mark := EgressMark(m.Globals, i)
|
mark := EgressMark(m.Globals, i)
|
||||||
table := EgressTable(m.Globals, i)
|
table := EgressTable(m.Globals, i)
|
||||||
@@ -403,30 +540,123 @@ func isPointToPoint(dev string) bool {
|
|||||||
// reconcile cannot repair, because a reconcile is exactly what consults this
|
// reconcile cannot repair, because a reconcile is exactly what consults this
|
||||||
// function. A presence check must cover everything its Apply counterpart
|
// function. A presence check must cover everything its Apply counterpart
|
||||||
// installs, or the idempotent fast-path becomes a trap.
|
// installs, or the idempotent fast-path becomes a trap.
|
||||||
|
//
|
||||||
|
// # The per-egress bindings are checked too
|
||||||
|
//
|
||||||
|
// The same sentence applies to the OTHER half of ApplyRouting, which this check
|
||||||
|
// used to be completely blind to: addEgressRouting installs a `fwmark -> table`
|
||||||
|
// rule plus a default route in that table for EVERY interface/tunnel egress, and
|
||||||
|
// neither EgressMark nor EgressTable was mentioned here at all.
|
||||||
|
//
|
||||||
|
// That blindness was a permanent leak, not a cosmetic gap. An egress rides an
|
||||||
|
// interface (a second WAN, a wg/awg tunnel), and `ifdown wg0` or a `network
|
||||||
|
// restart` makes the kernel delete every route through that device — including
|
||||||
|
// the default route in the egress's own table. The ip RULE survives, so the marked
|
||||||
|
// traffic still gets sent to that table, finds it empty, falls through to the main
|
||||||
|
// table and leaves over the ordinary WAN with the router's real address. The
|
||||||
|
// one-minute reconcile then asked this function whether routing was present, was
|
||||||
|
// told yes on the strength of the main tproxy pair alone, and skipped
|
||||||
|
// ApplyRouting — so nothing ever restored the route. The box reported plane=full
|
||||||
|
// with zero warnings while the tunnel it was built for was bypassed, and only a
|
||||||
|
// config edit or a daemon restart could clear it.
|
||||||
|
//
|
||||||
|
// Which bindings must exist comes from the record addEgressRouting keeps (see
|
||||||
|
// egressBindings): Globals alone cannot say how many egresses there are, and a
|
||||||
|
// deleted rule leaves nothing on the system to count.
|
||||||
func RoutingPresent(g model.Globals) bool {
|
func RoutingPresent(g model.Globals) bool {
|
||||||
wantRule := fmt.Sprintf("fwmark 0x%x", effFwmark(g))
|
mark, table := effFwmark(g), effTable(g)
|
||||||
table := fmt.Sprintf("%d", effTable(g))
|
|
||||||
fams := []string{"-4"}
|
fams := []string{"-4"}
|
||||||
if g.IPv6 {
|
if g.IPv6 {
|
||||||
fams = append(fams, "-6")
|
fams = append(fams, "-6")
|
||||||
}
|
}
|
||||||
for _, fam := range fams {
|
for _, fam := range fams {
|
||||||
out, err := execCommand("ip", fam, "rule", "show").Output()
|
if !mainRoutingPresentFam(fam, mark, table) {
|
||||||
if err != nil || !strings.Contains(string(out), wantRule) {
|
|
||||||
return false
|
return false
|
||||||
}
|
}
|
||||||
// `ip -4 route show table N` prints "local default dev lo scope host";
|
}
|
||||||
// the v6 form is "local default dev lo metric 1024 pref medium". Matching
|
return egressRoutingPresent()
|
||||||
// the route TYPE + destination covers both without pinning the trailing
|
}
|
||||||
// attributes, which differ by family and iproute2 version.
|
|
||||||
rout, rerr := execCommand("ip", fam, "route", "show", "table", table).Output()
|
// mainRoutingPresentFam reports whether BOTH halves of the main tproxy plane are
|
||||||
if rerr != nil || !strings.Contains(string(rout), "local default") {
|
// installed for one family: the fwmark rule and the `local default dev lo` route
|
||||||
|
// inside the table it points at.
|
||||||
|
func mainRoutingPresentFam(fam string, mark, table uint32) bool {
|
||||||
|
out, err := execCommand("ip", fam, "rule", "show").Output()
|
||||||
|
if err != nil || !fwmarkRulePresent(string(out), mark) {
|
||||||
|
return false
|
||||||
|
}
|
||||||
|
// `ip -4 route show table N` prints "local default dev lo scope host";
|
||||||
|
// the v6 form is "local default dev lo metric 1024 pref medium". Matching
|
||||||
|
// the route TYPE + destination covers both without pinning the trailing
|
||||||
|
// attributes, which differ by family and iproute2 version.
|
||||||
|
rout, rerr := execCommand("ip", fam, "route", "show", "table", fmt.Sprintf("%d", table)).Output()
|
||||||
|
return rerr == nil && strings.Contains(string(rout), "local default")
|
||||||
|
}
|
||||||
|
|
||||||
|
// egressRoutingPresent reports whether every per-egress binding the last apply set
|
||||||
|
// out to install is still there: the rule AND a default route in its table.
|
||||||
|
//
|
||||||
|
// The route half is the one that matters most — see RoutingPresent's essay — but
|
||||||
|
// both are checked, for the same reason the main pair checks both: they are
|
||||||
|
// removed by independent commands and either can go missing on its own.
|
||||||
|
func egressRoutingPresent() bool {
|
||||||
|
bindings := snapshotEgressBindings()
|
||||||
|
if len(bindings) == 0 {
|
||||||
|
return true // no interface/tunnel egress configured: nothing else to install
|
||||||
|
}
|
||||||
|
rules := map[string]string{} // family -> `ip -N rule show` output, read once
|
||||||
|
for _, b := range bindings {
|
||||||
|
out, cached := rules[b.fam]
|
||||||
|
if !cached {
|
||||||
|
raw, err := execCommand("ip", b.fam, "rule", "show").Output()
|
||||||
|
if err != nil {
|
||||||
|
return false
|
||||||
|
}
|
||||||
|
out = string(raw)
|
||||||
|
rules[b.fam] = out
|
||||||
|
}
|
||||||
|
if !fwmarkRulePresent(out, b.mark) {
|
||||||
|
return false
|
||||||
|
}
|
||||||
|
// Any default route will do: `default via <gw> dev X` for a gateway'd
|
||||||
|
// uplink, `default dev X` for a point-to-point tunnel. What must never pass
|
||||||
|
// is an EMPTY table, which is what an ifdown leaves behind.
|
||||||
|
rout, err := execCommand("ip", b.fam, "route", "show", "table", fmt.Sprintf("%d", b.table)).Output()
|
||||||
|
if err != nil || !strings.Contains(string(rout), "default") {
|
||||||
return false
|
return false
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
return true
|
return true
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// fwmarkRulePresent reports whether `ip rule show` output carries a rule for
|
||||||
|
// EXACTLY this mark.
|
||||||
|
//
|
||||||
|
// The match is anchored on the right because a plain substring test confuses
|
||||||
|
// marks that share a prefix: with a small fwmark_base the main divert mark (say
|
||||||
|
// 0x1) is a prefix of the first egress mark (0x101), so "the main rule is
|
||||||
|
// installed" could be answered by an egress rule that has nothing to do with it.
|
||||||
|
// `ip` always prints the mark as a whole token — "fwmark 0x101 lookup 8210", or
|
||||||
|
// "fwmark 0x101/0xff" when a mask is in play (never ours) — so requiring a
|
||||||
|
// separator (or end of line) after it is sufficient and version-independent.
|
||||||
|
func fwmarkRulePresent(out string, mark uint32) bool {
|
||||||
|
want := fmt.Sprintf("fwmark 0x%x", mark)
|
||||||
|
for _, line := range strings.Split(out, "\n") {
|
||||||
|
for i := 0; ; {
|
||||||
|
j := strings.Index(line[i:], want)
|
||||||
|
if j < 0 {
|
||||||
|
break
|
||||||
|
}
|
||||||
|
rest := line[i+j+len(want):]
|
||||||
|
if rest == "" || rest[0] == ' ' || rest[0] == '\t' || rest[0] == '\r' {
|
||||||
|
return true
|
||||||
|
}
|
||||||
|
i += j + len(want)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return false
|
||||||
|
}
|
||||||
|
|
||||||
// IfaceDevice resolves a UCI interface name (e.g. "lan") to its actual L3 device
|
// IfaceDevice resolves a UCI interface name (e.g. "lan") to its actual L3 device
|
||||||
// (e.g. "br-lan") for nft iifname matching / SO_BINDTODEVICE — bridges/VLANs have
|
// (e.g. "br-lan") for nft iifname matching / SO_BINDTODEVICE — bridges/VLANs have
|
||||||
// device != name, so matching the raw UCI name would silently intercept nothing.
|
// device != name, so matching the raw UCI name would silently intercept nothing.
|
||||||
|
|||||||
@@ -24,6 +24,10 @@ func (f *fakeNet) install(t *testing.T) {
|
|||||||
t.Helper()
|
t.Helper()
|
||||||
orig := execCommand
|
orig := execCommand
|
||||||
t.Cleanup(func() { execCommand = orig })
|
t.Cleanup(func() { execCommand = orig })
|
||||||
|
// addEgressRouting records what it installed for RoutingPresent (see
|
||||||
|
// egressBindings in apply.go). That record is package state, so it must not
|
||||||
|
// leak out of a test into the RoutingPresent cases in the other files.
|
||||||
|
t.Cleanup(func() { rememberEgressBindings(nil) })
|
||||||
execCommand = func(name string, arg ...string) *exec.Cmd {
|
execCommand = func(name string, arg ...string) *exec.Cmd {
|
||||||
out := ""
|
out := ""
|
||||||
switch {
|
switch {
|
||||||
|
|||||||
@@ -0,0 +1,287 @@
|
|||||||
|
package netplane
|
||||||
|
|
||||||
|
// Per-egress policy routing: it must be VERIFIED by the presence check, and its
|
||||||
|
// installation must be REPORTED when it fails.
|
||||||
|
//
|
||||||
|
// Both defects here produce the same end state and it is the worst one this
|
||||||
|
// project has: an egress that the panel shows as configured and applied, while
|
||||||
|
// the traffic bound to it is marked for a routing table that is empty or absent,
|
||||||
|
// falls through to the main table, and leaves over the ordinary WAN with the
|
||||||
|
// router's real address. Neither state raised anything — no error, no warning —
|
||||||
|
// and neither could be repaired by a reconcile.
|
||||||
|
|
||||||
|
import (
|
||||||
|
"fmt"
|
||||||
|
"os"
|
||||||
|
"os/exec"
|
||||||
|
"strings"
|
||||||
|
"testing"
|
||||||
|
|
||||||
|
"github.com/sagernet/sing-box/shater/model"
|
||||||
|
)
|
||||||
|
|
||||||
|
// egressPlaneNet is a canned view of the box's routing state plus a switch for
|
||||||
|
// making chosen `ip` mutations fail, so a test can drive both the presence check
|
||||||
|
// and the failure reporting without touching the real system.
|
||||||
|
type egressPlaneNet struct {
|
||||||
|
rules map[string]string // family ("-4"/"-6") -> `ip -N rule show` output
|
||||||
|
routes map[string]string // "<family> <table>" -> `ip -N route show table N` output
|
||||||
|
link map[string]string // device -> `ip link show dev D` output
|
||||||
|
ubus map[string]string // interface -> `ubus call ... status` JSON
|
||||||
|
failIf func(name string, arg []string) bool
|
||||||
|
calls []string // every intercepted invocation, joined
|
||||||
|
}
|
||||||
|
|
||||||
|
func (f *egressPlaneNet) install(t *testing.T) {
|
||||||
|
t.Helper()
|
||||||
|
orig := execCommand
|
||||||
|
t.Cleanup(func() { execCommand = orig })
|
||||||
|
// addEgressRouting records what it installed in package state; never let that
|
||||||
|
// leak into another test's RoutingPresent.
|
||||||
|
rememberEgressBindings(nil)
|
||||||
|
t.Cleanup(func() { rememberEgressBindings(nil) })
|
||||||
|
|
||||||
|
execCommand = func(name string, arg ...string) *exec.Cmd {
|
||||||
|
f.calls = append(f.calls, name+" "+strings.Join(arg, " "))
|
||||||
|
out := ""
|
||||||
|
switch {
|
||||||
|
case name == "ubus" && len(arg) >= 2 && arg[0] == "call":
|
||||||
|
out = f.ubus[strings.TrimPrefix(arg[1], "network.interface.")]
|
||||||
|
case name == "ip" && len(arg) >= 3 && arg[0] == "link" && arg[1] == "show":
|
||||||
|
out = f.link[arg[len(arg)-1]]
|
||||||
|
case name == "ip" && len(arg) >= 2 && arg[1] == "rule" && arg[2] == "show":
|
||||||
|
out = f.rules[arg[0]]
|
||||||
|
case name == "ip" && len(arg) >= 5 && arg[1] == "route" && arg[2] == "show" && arg[3] == "table":
|
||||||
|
out = f.routes[arg[0]+" "+arg[4]]
|
||||||
|
}
|
||||||
|
fail := f.failIf != nil && f.failIf(name, arg)
|
||||||
|
cs := append([]string{"-test.run=TestEgressPlaneHelperProcess", "--", name}, arg...)
|
||||||
|
cmd := exec.Command(os.Args[0], cs...)
|
||||||
|
env := append(os.Environ(), "GO_WANT_HELPER_PROCESS=1", "GO_HELPER_STDOUT="+out)
|
||||||
|
if fail {
|
||||||
|
env = append(env, "GO_HELPER_FAIL=1")
|
||||||
|
}
|
||||||
|
cmd.Env = env
|
||||||
|
return cmd
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestEgressPlaneHelperProcess is the canned-output / canned-failure child.
|
||||||
|
func TestEgressPlaneHelperProcess(t *testing.T) {
|
||||||
|
if os.Getenv("GO_WANT_HELPER_PROCESS") != "1" {
|
||||||
|
return
|
||||||
|
}
|
||||||
|
if os.Getenv("GO_HELPER_FAIL") == "1" {
|
||||||
|
os.Stderr.WriteString("Cannot find device \"wg0\"\n")
|
||||||
|
os.Exit(2)
|
||||||
|
}
|
||||||
|
os.Stdout.WriteString(os.Getenv("GO_HELPER_STDOUT"))
|
||||||
|
os.Exit(0)
|
||||||
|
}
|
||||||
|
|
||||||
|
// egressPlaneGlobals pins the marks/tables the cases below hard-code.
|
||||||
|
func egressPlaneGlobals() model.Globals {
|
||||||
|
return model.Globals{FwmarkBase: 0x2000, TableBase: 0x2000}
|
||||||
|
}
|
||||||
|
|
||||||
|
// wgEgressModel is one point-to-point tunnel egress on wg0 — the shape that
|
||||||
|
// produces no gateway warning of its own, so anything a test sees is the fault it
|
||||||
|
// is testing.
|
||||||
|
func wgEgressModel() *model.Model {
|
||||||
|
return &model.Model{
|
||||||
|
Globals: egressPlaneGlobals(),
|
||||||
|
Egresses: []model.Egress{{Name: "vpn", Type: "tunnel", Interface: "wgvpn"}},
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func wgEgressNet() *egressPlaneNet {
|
||||||
|
return &egressPlaneNet{
|
||||||
|
ubus: map[string]string{"wgvpn": `{"up":true,"l3_device":"wg0","route":[]}`},
|
||||||
|
link: map[string]string{"wg0": "5: wg0: <POINTOPOINT,NOARP,UP,LOWER_UP> mtu 1420"},
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// DEFECT 2. RoutingPresent decided the whole policy-routing plane was intact from
|
||||||
|
// the MAIN tproxy pair alone: EgressMark and EgressTable were not mentioned in it
|
||||||
|
// even once. So `ifdown wg0` (or a `network restart`), which makes the kernel drop
|
||||||
|
// every route through that device — including the default route inside the
|
||||||
|
// egress's own table — left a state where:
|
||||||
|
//
|
||||||
|
// - the ip rule still sends the egress mark to table 8208,
|
||||||
|
// - table 8208 is empty,
|
||||||
|
// - so the marked traffic falls through to the main table and leaves over the
|
||||||
|
// plain WAN with the router's real address,
|
||||||
|
// - and the one-minute reconcile asked this function, was told "present", and
|
||||||
|
// never re-ran ApplyRouting. Permanent, at plane=full, with zero warnings.
|
||||||
|
//
|
||||||
|
// The main pair is deliberately healthy in every case below, so only the per-egress
|
||||||
|
// half can decide the answer.
|
||||||
|
func TestRoutingPresentSeesEgressTables(t *testing.T) {
|
||||||
|
const (
|
||||||
|
mainRule = "32765:\tfrom all fwmark 0x2000 lookup 8192"
|
||||||
|
egressRule = "32764:\tfrom all fwmark 0x2100 lookup 8208"
|
||||||
|
mainRoute = "local default dev lo scope host"
|
||||||
|
egressRte = "default dev wg0 scope link"
|
||||||
|
)
|
||||||
|
cases := []struct {
|
||||||
|
name string
|
||||||
|
rules string // `ip -4 rule show`
|
||||||
|
egressTbl string // `ip -4 route show table 8208`
|
||||||
|
wantPresen bool
|
||||||
|
}{
|
||||||
|
{"complete", mainRule + "\n" + egressRule, egressRte, true},
|
||||||
|
// THE REGRESSION: ifdown wg0 emptied the egress table, the rule survived.
|
||||||
|
{"egress table emptied by ifdown", mainRule + "\n" + egressRule, "", false},
|
||||||
|
// The other half: the rule itself went away.
|
||||||
|
{"egress rule gone", mainRule, egressRte, false},
|
||||||
|
}
|
||||||
|
|
||||||
|
for _, c := range cases {
|
||||||
|
t.Run(c.name, func(t *testing.T) {
|
||||||
|
f := wgEgressNet()
|
||||||
|
f.install(t)
|
||||||
|
|
||||||
|
// Record the bindings the way production does: through a real apply.
|
||||||
|
if _, err := addEgressRouting(wgEgressModel()); err != nil {
|
||||||
|
t.Fatalf("addEgressRouting: %v", err)
|
||||||
|
}
|
||||||
|
|
||||||
|
f.rules = map[string]string{"-4": c.rules}
|
||||||
|
f.routes = map[string]string{
|
||||||
|
"-4 8192": mainRoute,
|
||||||
|
"-4 8208": c.egressTbl,
|
||||||
|
}
|
||||||
|
if got := RoutingPresent(egressPlaneGlobals()); got != c.wantPresen {
|
||||||
|
t.Fatalf("RoutingPresent = %v, want %v.\n"+
|
||||||
|
"A presence check blind to the per-egress table lets a reconcile skip ApplyRouting "+
|
||||||
|
"forever while the egress-marked traffic leaves over the plain WAN.",
|
||||||
|
got, c.wantPresen)
|
||||||
|
}
|
||||||
|
})
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// A model with no interface/tunnel egress must still report present on a healthy
|
||||||
|
// main pair — the egress half may only ever ADD a reason to re-apply.
|
||||||
|
func TestRoutingPresentUnaffectedWithoutEgresses(t *testing.T) {
|
||||||
|
f := &egressPlaneNet{
|
||||||
|
rules: map[string]string{"-4": "32765:\tfrom all fwmark 0x2000 lookup 8192"},
|
||||||
|
routes: map[string]string{"-4 8192": "local default dev lo scope host"},
|
||||||
|
}
|
||||||
|
f.install(t)
|
||||||
|
if _, err := addEgressRouting(&model.Model{Globals: egressPlaneGlobals()}); err != nil {
|
||||||
|
t.Fatalf("addEgressRouting: %v", err)
|
||||||
|
}
|
||||||
|
if !RoutingPresent(egressPlaneGlobals()) {
|
||||||
|
t.Fatal("a healthy main pair with no egresses must report present")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// fwmarkRulePresent must not confuse marks that share a prefix. With a small
|
||||||
|
// fwmark_base the main divert mark is a literal prefix of the first egress mark,
|
||||||
|
// and a plain strings.Contains would answer "the main rule is installed" with an
|
||||||
|
// egress rule that has nothing to do with it.
|
||||||
|
func TestFwmarkRulePresentIsExact(t *testing.T) {
|
||||||
|
const out = "32764:\tfrom all fwmark 0x101 lookup 273"
|
||||||
|
if fwmarkRulePresent(out, 0x1) {
|
||||||
|
t.Error("fwmark 0x1 must NOT be matched by a rule for fwmark 0x101")
|
||||||
|
}
|
||||||
|
if !fwmarkRulePresent(out, 0x101) {
|
||||||
|
t.Error("fwmark 0x101 must be matched by its own rule")
|
||||||
|
}
|
||||||
|
if fwmarkRulePresent("32764:\tfrom all fwmark 0x101/0xff lookup 273", 0x101) {
|
||||||
|
t.Error("a masked rule is a different rule and must not match")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// DEFECT 3. addEgressRouting ran every mutation through `_ = exec...Run()`, so an
|
||||||
|
// egress bound to a device that does not exist at apply time — `ifdown wg0`, a
|
||||||
|
// tunnel that has not come up yet, a renamed interface — reported a completely
|
||||||
|
// successful apply: no error, no warning, a green egress in the panel, and an
|
||||||
|
// empty routing table underneath it. Everything bound to that egress then leaves
|
||||||
|
// over the ordinary WAN with the router's real address.
|
||||||
|
func TestEgressRouteAddFailureIsReported(t *testing.T) {
|
||||||
|
f := wgEgressNet()
|
||||||
|
f.failIf = func(name string, arg []string) bool {
|
||||||
|
return name == "ip" && len(arg) >= 3 && arg[1] == "route" && arg[2] == "add"
|
||||||
|
}
|
||||||
|
f.install(t)
|
||||||
|
|
||||||
|
warns, err := addEgressRouting(wgEgressModel())
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("a broken egress must not abort the whole apply: %v", err)
|
||||||
|
}
|
||||||
|
if len(warns) != 1 {
|
||||||
|
t.Fatalf("a failing `ip route add` must produce exactly one operator warning, got %d: %v",
|
||||||
|
len(warns), warns)
|
||||||
|
}
|
||||||
|
w := warns[0]
|
||||||
|
// The `kind "name":` prefix is what apply/warnings.go parses into Section/Name,
|
||||||
|
// and the consequence clauses are what grade it critical there.
|
||||||
|
for _, want := range []string{
|
||||||
|
`egress "vpn":`,
|
||||||
|
"is NOT applied",
|
||||||
|
"leaves over the plain WAN with your real IP address",
|
||||||
|
"wgvpn",
|
||||||
|
"wg0",
|
||||||
|
"Cannot find device", // the reason `ip` itself gave
|
||||||
|
} {
|
||||||
|
if !strings.Contains(w, want) {
|
||||||
|
t.Errorf("warning is missing %q:\n%s", want, w)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// The same contract for the OTHER mutation: without its ip rule the egress mark is
|
||||||
|
// not bound to any table at all.
|
||||||
|
func TestEgressRuleAddFailureIsReported(t *testing.T) {
|
||||||
|
f := wgEgressNet()
|
||||||
|
f.failIf = func(name string, arg []string) bool {
|
||||||
|
return name == "ip" && len(arg) >= 3 && arg[1] == "rule" && arg[2] == "add"
|
||||||
|
}
|
||||||
|
f.install(t)
|
||||||
|
|
||||||
|
warns, err := addEgressRouting(wgEgressModel())
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("a broken egress must not abort the whole apply: %v", err)
|
||||||
|
}
|
||||||
|
if len(warns) != 1 {
|
||||||
|
t.Fatalf("a failing `ip rule add` must produce exactly one operator warning, got %d: %v",
|
||||||
|
len(warns), warns)
|
||||||
|
}
|
||||||
|
for _, want := range []string{`egress "vpn":`, "is NOT applied", "leaves over the plain WAN with your real IP address"} {
|
||||||
|
if !strings.Contains(warns[0], want) {
|
||||||
|
t.Errorf("warning is missing %q:\n%s", want, warns[0])
|
||||||
|
}
|
||||||
|
}
|
||||||
|
// A rule that could not be installed means the table binding does not exist;
|
||||||
|
// pushing a default route into an unreachable table would only hide that.
|
||||||
|
for _, c := range f.calls {
|
||||||
|
if strings.Contains(c, "route add default") {
|
||||||
|
t.Errorf("no default route may be installed once the rule add failed: %q", c)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// A failed install must keep RoutingPresent FALSE, so the next reconcile tries
|
||||||
|
// again and the egress repairs itself the moment its interface comes back. This is
|
||||||
|
// why the binding record holds INTENT and not outcome.
|
||||||
|
func TestFailedEgressKeepsRoutingAbsent(t *testing.T) {
|
||||||
|
f := wgEgressNet()
|
||||||
|
f.failIf = func(name string, arg []string) bool {
|
||||||
|
return name == "ip" && len(arg) >= 3 && arg[1] == "route" && arg[2] == "add"
|
||||||
|
}
|
||||||
|
f.install(t)
|
||||||
|
if _, err := addEgressRouting(wgEgressModel()); err != nil {
|
||||||
|
t.Fatalf("addEgressRouting: %v", err)
|
||||||
|
}
|
||||||
|
|
||||||
|
f.failIf = nil
|
||||||
|
f.rules = map[string]string{"-4": "32765:\tfrom all fwmark 0x2000 lookup 8192\n" +
|
||||||
|
fmt.Sprintf("32764:\tfrom all fwmark 0x%x lookup 8208", EgressMark(egressPlaneGlobals(), 0))}
|
||||||
|
f.routes = map[string]string{"-4 8192": "local default dev lo scope host", "-4 8208": ""}
|
||||||
|
if RoutingPresent(egressPlaneGlobals()) {
|
||||||
|
t.Fatal("an egress whose route could not be installed must keep the plane reported ABSENT, " +
|
||||||
|
"so the next reconcile retries it")
|
||||||
|
}
|
||||||
|
}
|
||||||
+107
-28
@@ -11,18 +11,36 @@ package netplane
|
|||||||
// the tunnel at all. The forward chain therefore has to decide what happens to
|
// the tunnel at all. The forward chain therefore has to decide what happens to
|
||||||
// them, and until now the answer was a blanket drop.
|
// them, and until now the answer was a blanket drop.
|
||||||
//
|
//
|
||||||
// # Why the blanket drop was wrong
|
// # `block` blocks. It did not, and that was the bug.
|
||||||
//
|
//
|
||||||
// The drop is justified by "we cannot tunnel this packet and we will not let it
|
// There is an argument that the drop is only justified for destinations the
|
||||||
// out with the client's real address". That justification only holds for
|
// routing rules actually send THROUGH the tunnel: when a rule routes a
|
||||||
// destinations the routing rules actually send THROUGH the tunnel. When a rule
|
// destination DIRECT, the client's real address already reaches it over TCP, so
|
||||||
// routes a destination DIRECT, the client's real address already reaches it over
|
// dropping ICMP to that same destination hides nothing and only breaks
|
||||||
// TCP — so dropping ICMP to that same destination hides nothing and only breaks
|
// diagnostics. That argument is sound about ICMP echo. It was applied to `block`
|
||||||
// diagnostics. The old behaviour over-blocked: with a `ru-direct` rule sending
|
// as a whole, and there it was wrong twice over:
|
||||||
// Russian addresses direct, those addresses still could not be pinged.
|
|
||||||
//
|
//
|
||||||
// So: drop untunnelable traffic exactly where an equivalent TCP flow would have
|
// - IT DEGENERATED. The commonest configuration on this router is "tunnel what
|
||||||
// been tunnelled, and let it out where that flow would have gone direct.
|
// is blocked, send the rest direct", which makes the routing DEFAULT direct —
|
||||||
|
// so EVERY destination is direct, so `block` allowed everything, everywhere.
|
||||||
|
// The most private of the three rungs was bit-for-bit identical to the most
|
||||||
|
// permissive one, and only the daemon log said so.
|
||||||
|
// - IT IS NOT THE SAME DISCLOSURE. "That host already sees your address over
|
||||||
|
// TCP" holds for a ping and does not hold for ESP, AH or GRE. A client-run
|
||||||
|
// IPsec or PPTP tunnel to a directly-routed server is a STANDING second
|
||||||
|
// tunnel, carrying arbitrary traffic past ours for as long as it stays up,
|
||||||
|
// under a peer we neither route nor filter. Letting that out is the exact
|
||||||
|
// exposure the middle rung was created to separate from "answering a ping",
|
||||||
|
// and `block` was granting it silently.
|
||||||
|
//
|
||||||
|
// So `block` now drops every untunnelable protocol, wherever it was going, with
|
||||||
|
// no reference to the routing rules at all. Its promise no longer depends on a
|
||||||
|
// configuration the operator can change without noticing. The price is real and
|
||||||
|
// is listed on the constants below; it is a price the owner chose knowingly, on
|
||||||
|
// the grounds that this router does not do ping and does not do IPTV.
|
||||||
|
//
|
||||||
|
// The destination walk below survives and serves `icmp`, where the ICMP half of
|
||||||
|
// the argument does hold.
|
||||||
//
|
//
|
||||||
// # What "exactly" can and cannot mean
|
// # What "exactly" can and cannot mean
|
||||||
//
|
//
|
||||||
@@ -48,17 +66,49 @@ import (
|
|||||||
// The three form a ladder, and each rung is a distinct, explicable amount of
|
// The three form a ladder, and each rung is a distinct, explicable amount of
|
||||||
// disclosure rather than an arbitrary midpoint:
|
// disclosure rather than an arbitrary midpoint:
|
||||||
//
|
//
|
||||||
// block — untunnelable traffic is allowed ONLY to destinations the routing
|
// block — EVERY untunnelable protocol is dropped, wherever it was going. The
|
||||||
// rules send direct anyway. No destination learns your address that
|
// default, and the only rung whose promise does not depend on how the
|
||||||
// would not have learned it from ordinary TCP. The default.
|
// routing rules happen to be written.
|
||||||
// icmp — additionally allows ICMP echo to TUNNELLED destinations, so ping and
|
// icmp — block, plus ICMP echo to every destination, so ping and traceroute
|
||||||
// traceroute work everywhere. The host you ping learns your real
|
// work. The host pinged learns the real address; nothing else does, and
|
||||||
// address; nothing else does, and nothing persistent is established.
|
// nothing persistent is established. The remaining protocols are also
|
||||||
|
// let out toward destinations the routing rules send DIRECT — see the
|
||||||
|
// plan walk below — because that host already has the address from
|
||||||
|
// ordinary TCP.
|
||||||
// direct — allows every untunnelable protocol everywhere, including ESP/AH/GRE
|
// direct — allows every untunnelable protocol everywhere, including ESP/AH/GRE
|
||||||
// to tunnelled destinations. That is a client-run VPN operating in
|
// to tunnelled destinations. That is a client-run VPN operating in
|
||||||
// parallel to ours with the real address, carrying arbitrary traffic
|
// parallel to ours with the real address, carrying arbitrary traffic
|
||||||
// for as long as it stays up — a different kind of exposure from
|
// for as long as it stays up — a different kind of exposure from
|
||||||
// answering a ping, which is why it is not folded into `icmp`.
|
// answering a ping, which is why it is not folded into `icmp`.
|
||||||
|
//
|
||||||
|
// # What `block` costs the LAN, exactly
|
||||||
|
//
|
||||||
|
// Everything below is LAN-ingress traffic bound for a PUBLIC address; the local
|
||||||
|
// plane (LAN-to-LAN, link-local, ICMPv6 ND/RA, multicast to the router) is
|
||||||
|
// accepted earlier in the forward chain and is untouched by any of this.
|
||||||
|
//
|
||||||
|
// - ICMPv4 and ICMPv6 ECHO: `ping` and Windows `tracert` from a LAN device stop
|
||||||
|
// answering for every public address. Linux/macOS `traceroute` defaults to UDP
|
||||||
|
// probes, which are tunnelled — so it still prints hops, but they are the
|
||||||
|
// tunnel's path, not the client's.
|
||||||
|
// - ICMP ERRORS raised BY a LAN host toward a public peer, including
|
||||||
|
// fragmentation-needed. Path-MTU discovery for tunnelled flows is unaffected
|
||||||
|
// (those terminate on the router, which raises its own errors from the output
|
||||||
|
// hook); only a LAN host's errors on a directly-routed flow are lost.
|
||||||
|
// - ESP (50) and AH (51): a device's own IPsec VPN, when it runs RAW ESP. IKE
|
||||||
|
// (UDP 500) and NAT-T ESP (UDP 4500) are UDP and go through the tunnel like
|
||||||
|
// anything else, so an IPsec client behind NAT — the usual case — is not
|
||||||
|
// affected by this at all. WireGuard and OpenVPN-UDP are UDP likewise.
|
||||||
|
// - GRE (47): PPTP's data plane, and plain GRE tunnels. PPTP's control channel
|
||||||
|
// is TCP 1723 and is tunnelled, so the connection appears to establish and
|
||||||
|
// then carries nothing.
|
||||||
|
// - SCTP (132): rare from a home LAN. WebRTC data channels are SCTP over
|
||||||
|
// DTLS/UDP and are therefore unaffected.
|
||||||
|
// - IGMP (2): a LAN client's multicast group joins, where they are routed
|
||||||
|
// rather than answered by the router itself. Note the multicast STREAM is
|
||||||
|
// WAN-ingress and never matched by these rules, and multicast UDP from the LAN
|
||||||
|
// is already dropped by the fail-closed guard whatever this policy says — so
|
||||||
|
// IPTV was not working through this router before the change either.
|
||||||
const (
|
const (
|
||||||
UntunnelableBlock = "block"
|
UntunnelableBlock = "block"
|
||||||
UntunnelableICMP = "icmp"
|
UntunnelableICMP = "icmp"
|
||||||
@@ -247,9 +297,13 @@ func untunnelableSets(plan *UntunnelablePlan) string {
|
|||||||
// They must sit AFTER the local-plane accepts (LAN-to-LAN, link-local, ICMPv6 ND)
|
// They must sit AFTER the local-plane accepts (LAN-to-LAN, link-local, ICMPv6 ND)
|
||||||
// and BEFORE the fail-closed drops.
|
// and BEFORE the fail-closed drops.
|
||||||
//
|
//
|
||||||
// plan may be nil, which means "no destination knowledge" — every destination is
|
// plan is read by `icmp` and by nothing else — the other two rungs are
|
||||||
// then treated as tunnelled, which is the conservative reading and reproduces the
|
// unconditional, and reading it there is exactly the bug this function used to
|
||||||
// behaviour from before destinations were taken into account.
|
// have. A nil plan means "no destination knowledge" (the holding plane, or an
|
||||||
|
// engine that has not resolved its lists yet): every destination is then treated
|
||||||
|
// as tunnelled, which is the conservative reading, so `icmp` degrades to plain
|
||||||
|
// echo-only. apply.untunnelablePlanFor therefore builds no plan at all except for
|
||||||
|
// `icmp`, and this function must stay correct if one is handed to it anyway.
|
||||||
func untunnelableRules(g model.Globals, iif string, plan *UntunnelablePlan) string {
|
func untunnelableRules(g model.Globals, iif string, plan *UntunnelablePlan) string {
|
||||||
if iif == "" {
|
if iif == "" {
|
||||||
return ""
|
return ""
|
||||||
@@ -257,19 +311,31 @@ func untunnelableRules(g model.Globals, iif string, plan *UntunnelablePlan) stri
|
|||||||
policy := EffectiveUntunnelable(g)
|
policy := EffectiveUntunnelable(g)
|
||||||
var sb strings.Builder
|
var sb strings.Builder
|
||||||
|
|
||||||
// `direct` short-circuits the whole plan: everything untunnelable leaves,
|
// Both ends of the ladder short-circuit the plan, in opposite directions, and
|
||||||
// wherever it was going.
|
// neither may consult it.
|
||||||
if policy == UntunnelableDirect {
|
switch policy {
|
||||||
|
case UntunnelableDirect:
|
||||||
|
// Everything untunnelable leaves, wherever it was going.
|
||||||
sb.WriteString("\t\t" + iif + " " + untunnelableFilter + " accept\n")
|
sb.WriteString("\t\t" + iif + " " + untunnelableFilter + " accept\n")
|
||||||
return sb.String()
|
return sb.String()
|
||||||
|
case UntunnelableBlock:
|
||||||
|
// Nothing leaves. Emitting NO line is how that is said: every untunnelable
|
||||||
|
// packet falls through to the fail-closed drops the caller writes next.
|
||||||
|
//
|
||||||
|
// This is the whole fix. `block` used to walk the plan like `icmp` does and
|
||||||
|
// therefore inherited its default — which, for the ordinary "tunnel the
|
||||||
|
// blocked list, send the rest direct" configuration, is ALLOW. The most
|
||||||
|
// private rung was rendering the same `l4proto != { tcp, udp } accept` as
|
||||||
|
// the most permissive one. There is deliberately no `plan` in scope for
|
||||||
|
// this branch: the promise is unconditional, so no routing knowledge may
|
||||||
|
// enter it.
|
||||||
|
return ""
|
||||||
}
|
}
|
||||||
|
|
||||||
// `icmp` lets echo out ahead of the destination walk, so ping and traceroute
|
// `icmp` lets echo out ahead of the destination walk, so ping and traceroute
|
||||||
// work even toward tunnelled destinations.
|
// work even toward tunnelled destinations.
|
||||||
if policy == UntunnelableICMP {
|
sb.WriteString("\t\t" + iif + " ip protocol icmp icmp type { echo-request, echo-reply } accept\n")
|
||||||
sb.WriteString("\t\t" + iif + " ip protocol icmp icmp type { echo-request, echo-reply } accept\n")
|
sb.WriteString("\t\t" + iif + " icmpv6 type { echo-request, echo-reply } accept\n")
|
||||||
sb.WriteString("\t\t" + iif + " icmpv6 type { echo-request, echo-reply } accept\n")
|
|
||||||
}
|
|
||||||
|
|
||||||
if plan == nil {
|
if plan == nil {
|
||||||
// No knowledge of where anything routes: nothing is provably direct, so
|
// No knowledge of where anything routes: nothing is provably direct, so
|
||||||
@@ -284,12 +350,25 @@ func untunnelableRules(g model.Globals, iif string, plan *UntunnelablePlan) stri
|
|||||||
if mt.Allow {
|
if mt.Allow {
|
||||||
verb = "accept"
|
verb = "accept"
|
||||||
}
|
}
|
||||||
|
// A step is emitted PER FAMILY, and a step that named addresses only in the
|
||||||
|
// other family is not applicable to this one. Both halves of the predicate
|
||||||
|
// have to obey that, or the missing clause silently widens the line:
|
||||||
|
//
|
||||||
|
// dst — skip; without `ip daddr @set` the line would match every destination.
|
||||||
|
// src — skip; without `ip saddr @set` the line would match every SOURCE. A
|
||||||
|
// rule whose source_ip_cidr held only IPv6 prefixes used to render an
|
||||||
|
// IPv4 line with no saddr clause at all, i.e. an accept for the whole
|
||||||
|
// LAN from a rule scoped to a handful of v6 clients. "Scoped at all"
|
||||||
|
// is the union of the two families, exactly as the catch-all collapse
|
||||||
|
// in apply/untunnelable.go reads it.
|
||||||
|
srcScoped := len(mt.Src4) > 0 || len(mt.Src6) > 0
|
||||||
emit := func(fam int, dstSet, srcSet string, dst, src []string) {
|
emit := func(fam int, dstSet, srcSet string, dst, src []string) {
|
||||||
// A step that named destinations for this family but has none is not
|
|
||||||
// applicable to it; skip rather than widening to "any".
|
|
||||||
if !mt.AnyDst && len(dst) == 0 {
|
if !mt.AnyDst && len(dst) == 0 {
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
|
if srcScoped && len(src) == 0 {
|
||||||
|
return
|
||||||
|
}
|
||||||
var b strings.Builder
|
var b strings.Builder
|
||||||
b.WriteString("\t\t" + iif + " " + untunnelableFilter + " ")
|
b.WriteString("\t\t" + iif + " " + untunnelableFilter + " ")
|
||||||
if len(src) > 0 {
|
if len(src) > 0 {
|
||||||
|
|||||||
+65
-1
@@ -384,8 +384,50 @@ var (
|
|||||||
cmd.Stdout = w
|
cmd.Stdout = w
|
||||||
return cmd.Run()
|
return cmd.Run()
|
||||||
}
|
}
|
||||||
|
// logSyncTimeout bounds how long a download waits for the log sink's barrier
|
||||||
|
// (Server.SetLogSync). See syncLogSink for why the wait is bounded at all.
|
||||||
|
//
|
||||||
|
// Why 2s: it is the sink's OWN control budget (logsink.outControlTimeout) —
|
||||||
|
// past one such budget the barrier has given up internally anyway, so waiting
|
||||||
|
// longer buys nothing. A healthy writer drains the queue in microseconds, so
|
||||||
|
// this is a ceiling nobody pays, not a delay added to every download.
|
||||||
|
logSyncTimeout = 2 * time.Second
|
||||||
)
|
)
|
||||||
|
|
||||||
|
// syncLogSink waits for the daemon's log sink to have WRITTEN everything it has
|
||||||
|
// accepted so far, so the download below sees the lines the operator just caused.
|
||||||
|
// A no-op when no sink is wired (SetLogSync unset).
|
||||||
|
//
|
||||||
|
// The wait is bounded, and that bound is the point. The sink was made
|
||||||
|
// asynchronous precisely so a wedged log destination (a full stderr pipe whose
|
||||||
|
// reader stopped draining) can no longer freeze everything that logs. A barrier
|
||||||
|
// that waited indefinitely would hand that same power back to it, one HTTP
|
||||||
|
// handler further along: /api/log would become the new place the daemon hangs.
|
||||||
|
// So a stuck writer costs the download a bounded delay and it then streams what
|
||||||
|
// IS on disk — degraded, like every other honest fallback in this handler.
|
||||||
|
//
|
||||||
|
// The barrier runs in its own goroutine so that giving up is really giving up:
|
||||||
|
// abandoning a call we cannot cancel. logsink.Sink.Sync is itself bounded, so the
|
||||||
|
// goroutine ends on its own; a seam that never returns would leak one goroutine
|
||||||
|
// per request, which is a contract for test seams to respect, not a lock to take.
|
||||||
|
func (s *Server) syncLogSink() {
|
||||||
|
fn := s.logSync
|
||||||
|
if fn == nil {
|
||||||
|
return
|
||||||
|
}
|
||||||
|
done := make(chan struct{})
|
||||||
|
go func() {
|
||||||
|
defer close(done)
|
||||||
|
fn()
|
||||||
|
}()
|
||||||
|
timer := time.NewTimer(logSyncTimeout)
|
||||||
|
defer timer.Stop()
|
||||||
|
select {
|
||||||
|
case <-done:
|
||||||
|
case <-timer.C:
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
// handleLog → GET /api/log?range=1d|3d|all: download shaterd's OPERATIONAL log
|
// handleLog → GET /api/log?range=1d|3d|all: download shaterd's OPERATIONAL log
|
||||||
// (the daemon's own lines + the embedded engine's, as written by shater/logsink)
|
// (the daemon's own lines + the embedded engine's, as written by shater/logsink)
|
||||||
// as a text/plain attachment. CONTRACT (the panel is built against this):
|
// as a text/plain attachment. CONTRACT (the panel is built against this):
|
||||||
@@ -398,6 +440,11 @@ var (
|
|||||||
// Content-Disposition: attachment; filename="shaterd-<range>-<YYYYMMDD>.log"
|
// Content-Disposition: attachment; filename="shaterd-<range>-<YYYYMMDD>.log"
|
||||||
// (date in UTC). The body is STREAMED line by line — never buffered whole —
|
// (date in UTC). The body is STREAMED line by line — never buffered whole —
|
||||||
// from the sink's segments, oldest first (<path>.1 then <path>).
|
// from the sink's segments, oldest first (<path>.1 then <path>).
|
||||||
|
// - the answer contains every line the daemon had ACCEPTED into the log when
|
||||||
|
// the request arrived: the sink writes asynchronously, so the handler first
|
||||||
|
// runs its barrier (Server.SetLogSync) to flush what is still queued. The
|
||||||
|
// barrier is bounded — a wedged log destination costs a short delay, never
|
||||||
|
// the request.
|
||||||
// - honesty fallbacks: turning file logging OFF made the sink DELETE the
|
// - honesty fallbacks: turning file logging OFF made the sink DELETE the
|
||||||
// saved segments (see shater/logsink), so with the toggle off and syslog on
|
// saved segments (see shater/logsink), so with the toggle off and syslog on
|
||||||
// the best effort is a `logread -e shater` scrape prefixed with an
|
// the best effort is a `logread -e shater` scrape prefixed with an
|
||||||
@@ -434,6 +481,14 @@ func (s *Server) handleLog(w http.ResponseWriter, r *http.Request) {
|
|||||||
g = m.Globals
|
g = m.Globals
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// The sink writes asynchronously, so the newest lines — the ones this
|
||||||
|
// download exists for — may still be queued. Flush them to their destination
|
||||||
|
// before reading it. Bounded; see syncLogSink. Skipped when logging is fully
|
||||||
|
// off: there is then nothing queued and nothing to read.
|
||||||
|
if g.LogToFile || g.LogToSyslog {
|
||||||
|
s.syncLogSink()
|
||||||
|
}
|
||||||
|
|
||||||
w.Header().Set("Content-Type", "text/plain; charset=utf-8")
|
w.Header().Set("Content-Type", "text/plain; charset=utf-8")
|
||||||
w.Header().Set("Cache-Control", "no-store")
|
w.Header().Set("Cache-Control", "no-store")
|
||||||
w.Header().Set("Content-Disposition",
|
w.Header().Set("Content-Disposition",
|
||||||
@@ -1491,13 +1546,22 @@ type groupHealthResponse struct {
|
|||||||
// box materialised:
|
// box materialised:
|
||||||
//
|
//
|
||||||
// chains[].hops = [{index,tag,kind,exit,state,delay_ms,age_seconds,selected,
|
// chains[].hops = [{index,tag,kind,exit,state,delay_ms,age_seconds,selected,
|
||||||
// total,tested,alive,dead,untested}...]
|
// total,tested,alive,dead,untested,blocked_by?}...]
|
||||||
//
|
//
|
||||||
// - hops is L1..Ln in wire order (index is 1-based); the entry with exit=true
|
// - hops is L1..Ln in wire order (index is 1-based); the entry with exit=true
|
||||||
// is the last hop, where traffic leaves to the internet. Each hop's numbers
|
// is the last hop, where traffic leaves to the internet. Each hop's numbers
|
||||||
// measure the chain PREFIX up to and including that hop — the observatory
|
// measure the chain PREFIX up to and including that hop — the observatory
|
||||||
// dials the hop wrappers — so a dead hop N with alive hops 1..N-1 localises
|
// dials the hop wrappers — so a dead hop N with alive hops 1..N-1 localises
|
||||||
// the failure to hop N's own leg.
|
// the failure to hop N's own leg.
|
||||||
|
// - the walk is ORDERED and stops at the first dead hop. Hop N is dialled
|
||||||
|
// THROUGH hops 1..N-1, so probing it across a dead hop 2 would measure hop 2
|
||||||
|
// again and say nothing about hop N. Everything behind the first dead hop is
|
||||||
|
// therefore left undialled and reported state="untested" with
|
||||||
|
// blocked_by={index,tag} naming the hop that stopped the walk; its counters
|
||||||
|
// are 0/0/0/total and its delay_ms/age_seconds are 0/-1. blocked_by is
|
||||||
|
// OMITTED on every hop that was actually measured, and a hop carrying it is
|
||||||
|
// never a fault of its own — the fault is at blocked_by.index. It follows
|
||||||
|
// that a dead hop with a LIVE hop behind it can no longer be reported.
|
||||||
// - kind is "node" (one measurement: total=1, selected empty) or "group" (a
|
// - kind is "node" (one measurement: total=1, selected empty) or "group" (a
|
||||||
// roll-up of the hop's member copies, with the same counter invariants as a
|
// roll-up of the hop's member copies, with the same counter invariants as a
|
||||||
// group: tested == alive+dead, alive+dead+untested == total; selected is
|
// group: tested == alive+dead, alive+dead+untested == total; selected is
|
||||||
|
|||||||
@@ -14,6 +14,7 @@ import (
|
|||||||
"testing"
|
"testing"
|
||||||
"time"
|
"time"
|
||||||
|
|
||||||
|
"github.com/sagernet/sing-box/shater/logsink"
|
||||||
"github.com/sagernet/sing-box/shater/model"
|
"github.com/sagernet/sing-box/shater/model"
|
||||||
)
|
)
|
||||||
|
|
||||||
@@ -211,3 +212,115 @@ func TestLogFileOffIgnoresLeftoverSegments(t *testing.T) {
|
|||||||
t.Fatalf("scrape output missing:\n%s", body)
|
t.Fatalf("scrape output missing:\n%s", body)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// --- the sink barrier (Server.SetLogSync) ------------------------------------
|
||||||
|
|
||||||
|
// slowWriter stands in for the destination the sink's writer goroutine owns
|
||||||
|
// (under procd: the stderr pipe procd relays to logread). Every write takes
|
||||||
|
// `d`, which is what puts the just-written line demonstrably IN FLIGHT while the
|
||||||
|
// download runs — the sink hands lines to that goroutine and returns at once.
|
||||||
|
type slowWriter struct{ d time.Duration }
|
||||||
|
|
||||||
|
func (w slowWriter) Write(p []byte) (int, error) {
|
||||||
|
time.Sleep(w.d)
|
||||||
|
return len(p), nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// A line logged immediately before the request must be IN the download. The sink
|
||||||
|
// writes asynchronously, so without the barrier the download opens the file
|
||||||
|
// while the line is still queued and serves a truncated log — silently, with a
|
||||||
|
// 200 — losing exactly the last moments the operator downloaded the log to read.
|
||||||
|
func TestLogDownloadIncludesJustQueuedLines(t *testing.T) {
|
||||||
|
g := model.DefaultGlobals() // file on, syslog on
|
||||||
|
path := withLogSeams(t, g, "")
|
||||||
|
|
||||||
|
// A sink whose writer goroutine is busy for 300ms on the syslog half — long
|
||||||
|
// enough that the file half provably has not happened yet when the request
|
||||||
|
// arrives, short enough to stay well inside logSyncTimeout.
|
||||||
|
sink := logsink.New(slowWriter{d: 300 * time.Millisecond}, logsink.Config{
|
||||||
|
ToSyslog: true, ToFile: true, Path: path,
|
||||||
|
})
|
||||||
|
t.Cleanup(func() { _ = sink.Close() })
|
||||||
|
|
||||||
|
s := newTestServer(t)
|
||||||
|
s.SetLogSync(sink.Sync)
|
||||||
|
srv := httptest.NewServer(s.Handler())
|
||||||
|
defer srv.Close()
|
||||||
|
cookie := login(t, srv, s) // before the write: the handshake must not absorb the delay
|
||||||
|
|
||||||
|
if _, err := io.WriteString(sink, "INFO[0001] the line the operator came for\n"); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
resp := getDaemonLog(t, srv, cookie, "?range=all")
|
||||||
|
body, _ := io.ReadAll(resp.Body)
|
||||||
|
resp.Body.Close()
|
||||||
|
if resp.StatusCode != http.StatusOK {
|
||||||
|
t.Fatalf("got %d, want 200 (%s)", resp.StatusCode, body)
|
||||||
|
}
|
||||||
|
if !strings.Contains(string(body), "the line the operator came for") {
|
||||||
|
t.Fatalf("the download lost the line queued just before it:\n%s", body)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// A wedged log destination must cost the download a bounded delay, not the
|
||||||
|
// download. The whole point of the asynchronous sink is that a stuck writer
|
||||||
|
// stops nothing but its own goroutine; a barrier that waited forever would just
|
||||||
|
// move the hang into /api/log.
|
||||||
|
func TestLogDownloadSurvivesWedgedSink(t *testing.T) {
|
||||||
|
g := model.DefaultGlobals()
|
||||||
|
withLogSeams(t, g, "2026-07-23T08:00:00Z INFO already on disk\n")
|
||||||
|
|
||||||
|
orig := logSyncTimeout
|
||||||
|
logSyncTimeout = 150 * time.Millisecond
|
||||||
|
t.Cleanup(func() { logSyncTimeout = orig })
|
||||||
|
|
||||||
|
// A barrier that never returns on its own, exactly like a Sync behind a
|
||||||
|
// writer blocked in write(2) on a full pipe nobody drains.
|
||||||
|
release := make(chan struct{})
|
||||||
|
t.Cleanup(func() { close(release) })
|
||||||
|
|
||||||
|
s := newTestServer(t)
|
||||||
|
s.SetLogSync(func() { <-release })
|
||||||
|
srv := httptest.NewServer(s.Handler())
|
||||||
|
defer srv.Close()
|
||||||
|
cookie := login(t, srv, s)
|
||||||
|
|
||||||
|
start := time.Now()
|
||||||
|
resp := getDaemonLog(t, srv, cookie, "?range=all")
|
||||||
|
body, _ := io.ReadAll(resp.Body)
|
||||||
|
resp.Body.Close()
|
||||||
|
elapsed := time.Since(start)
|
||||||
|
|
||||||
|
if resp.StatusCode != http.StatusOK {
|
||||||
|
t.Fatalf("got %d, want 200 (%s)", resp.StatusCode, body)
|
||||||
|
}
|
||||||
|
if elapsed > 2*time.Second {
|
||||||
|
t.Fatalf("download waited %v on a wedged sink; the barrier must give up after %v", elapsed, logSyncTimeout)
|
||||||
|
}
|
||||||
|
// Giving up means serving what IS on disk, not serving nothing.
|
||||||
|
if !strings.Contains(string(body), "already on disk") {
|
||||||
|
t.Fatalf("after giving up the barrier must still stream the file:\n%s", body)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// No sink wired (tests, an embedder, a panel brought up on its own): the
|
||||||
|
// download must behave exactly as it did before the barrier existed.
|
||||||
|
func TestLogDownloadWithoutSyncSeam(t *testing.T) {
|
||||||
|
g := model.DefaultGlobals()
|
||||||
|
withLogSeams(t, g, "2026-07-23T08:00:00Z INFO line without a sink\n")
|
||||||
|
|
||||||
|
s := newTestServer(t) // SetLogSync deliberately never called
|
||||||
|
srv := httptest.NewServer(s.Handler())
|
||||||
|
defer srv.Close()
|
||||||
|
cookie := login(t, srv, s)
|
||||||
|
|
||||||
|
resp := getDaemonLog(t, srv, cookie, "?range=all")
|
||||||
|
body, _ := io.ReadAll(resp.Body)
|
||||||
|
resp.Body.Close()
|
||||||
|
if resp.StatusCode != http.StatusOK {
|
||||||
|
t.Fatalf("got %d, want 200 (%s)", resp.StatusCode, body)
|
||||||
|
}
|
||||||
|
if !strings.Contains(string(body), "line without a sink") {
|
||||||
|
t.Fatalf("body missing the file content:\n%s", body)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|||||||
@@ -85,6 +85,10 @@ type Server struct {
|
|||||||
opts Options
|
opts Options
|
||||||
log log.ContextLogger
|
log log.ContextLogger
|
||||||
|
|
||||||
|
// logSync is the daemon's log-sink barrier (logsink.Sink.Sync), installed by
|
||||||
|
// SetLogSync. nil => the panel runs without a sink and reads the file as-is.
|
||||||
|
logSync func()
|
||||||
|
|
||||||
srv *http.Server
|
srv *http.Server
|
||||||
handler http.Handler
|
handler http.Handler
|
||||||
|
|
||||||
@@ -125,6 +129,22 @@ func NewServer(a *apply.Applier, opts Options) *Server {
|
|||||||
// Safe to leave unset — the endpoints then report empty aggregates.
|
// Safe to leave unset — the endpoints then report empty aggregates.
|
||||||
func (s *Server) SetStats(agg stats.StatsStore) { s.stats = agg }
|
func (s *Server) SetStats(agg stats.StatsStore) { s.stats = agg }
|
||||||
|
|
||||||
|
// SetLogSync installs the operational-log barrier — in the daemon,
|
||||||
|
// logsink.Sink.Sync — that turns "queued" into "written" for GET /api/log.
|
||||||
|
//
|
||||||
|
// The sink writes ASYNCHRONOUSLY (a bounded queue between the goroutines that log
|
||||||
|
// and the one goroutine that owns stderr and the file; see shater/logsink), so
|
||||||
|
// the newest lines can still be in flight when a download opens the file by path.
|
||||||
|
// Those are exactly the lines an operator downloads the log FOR — they are
|
||||||
|
// debugging what just happened — and losing them would be silent: the file is
|
||||||
|
// served fine, just short of its tail.
|
||||||
|
//
|
||||||
|
// The panel is the sink's CONSUMER, not its owner: it gets a barrier function,
|
||||||
|
// never the *Sink. Wired once at startup, like SetStats. Leaving it unset is
|
||||||
|
// valid — /api/log then reads whatever is on disk, which is the pre-async
|
||||||
|
// behaviour and all a panel without a sink (tests, an embedder) can promise.
|
||||||
|
func (s *Server) SetLogSync(fn func()) { s.logSync = fn }
|
||||||
|
|
||||||
// Handler exposes the fully-wired http.Handler (auth middleware + API + SPA) so
|
// Handler exposes the fully-wired http.Handler (auth middleware + API + SPA) so
|
||||||
// tests can drive it via httptest without opening a socket.
|
// tests can drive it via httptest without opening a socket.
|
||||||
func (s *Server) Handler() http.Handler { return s.handler }
|
func (s *Server) Handler() http.Handler { return s.handler }
|
||||||
|
|||||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user