Compare commits
5
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
6476722372 | ||
|
|
4078334d85 | ||
|
|
0a6689b29e | ||
|
|
ef22167b1a | ||
|
|
cbda0fee0a |
@@ -0,0 +1,141 @@
|
||||
// lx:begin health-board
|
||||
|
||||
package urltest
|
||||
|
||||
import (
|
||||
"strconv"
|
||||
"strings"
|
||||
"sync"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"github.com/sagernet/sing-box/adapter"
|
||||
)
|
||||
|
||||
// captureEvictions swaps the eviction notice sink for the duration of a test and
|
||||
// returns a func that reads back everything reported.
|
||||
func captureEvictions(t *testing.T) func() []string {
|
||||
t.Helper()
|
||||
var (
|
||||
mu sync.Mutex
|
||||
msgs []string
|
||||
)
|
||||
orig := boardEvictionLog
|
||||
boardEvictionLog = func(m string) {
|
||||
mu.Lock()
|
||||
msgs = append(msgs, m)
|
||||
mu.Unlock()
|
||||
}
|
||||
t.Cleanup(func() { boardEvictionLog = orig })
|
||||
return func() []string {
|
||||
mu.Lock()
|
||||
defer mu.Unlock()
|
||||
return append([]string(nil), msgs...)
|
||||
}
|
||||
}
|
||||
|
||||
// TestBoardHoldsAGenerationWithoutEvicting is the "what it holds" half of the
|
||||
// bound. A live generation on this box is ~1200 tags (≈380 nodes plus their
|
||||
// per-group egress copies and chain hops); the board must carry that — and a
|
||||
// second generation's worth of overlap during a subscription rename — with no
|
||||
// eviction at all, or the ceiling would be silently degrading real health data.
|
||||
func TestBoardHoldsAGenerationWithoutEvicting(t *testing.T) {
|
||||
read := captureEvictions(t)
|
||||
s := NewHistoryStorage()
|
||||
|
||||
const generation = 1200
|
||||
for gen := 0; gen < 2; gen++ {
|
||||
for i := 0; i < generation; i++ {
|
||||
s.StoreURLTestHistory("gen"+strconv.Itoa(gen)+"-node-"+strconv.Itoa(i),
|
||||
&adapter.URLTestHistory{LastOK: time.Now(), Delay: 20})
|
||||
}
|
||||
}
|
||||
if got := s.Evicted(); got != 0 {
|
||||
t.Fatalf("two full generations (%d tags) evicted %d entries; the board must hold them",
|
||||
2*generation, got)
|
||||
}
|
||||
if msgs := read(); len(msgs) != 0 {
|
||||
t.Fatalf("unexpected eviction notices: %v", msgs)
|
||||
}
|
||||
// Everything is still readable.
|
||||
if s.LoadURLTestHistory("gen0-node-0") == nil {
|
||||
t.Fatalf("the first tag of the first generation was lost without an eviction")
|
||||
}
|
||||
}
|
||||
|
||||
// TestBoardEvictsOldestAndSaysSo is the "what happens when it overflows" half.
|
||||
// Overflow must (a) actually bound the map, (b) drop the LEAST RECENTLY MEASURED
|
||||
// tags — on this box, exactly the ones no config names any more — and (c) be
|
||||
// audible: a silent eviction is a health board quietly forgetting nodes it is
|
||||
// still being asked about.
|
||||
func TestBoardEvictsOldestAndSaysSo(t *testing.T) {
|
||||
read := captureEvictions(t)
|
||||
s := NewHistoryStorage()
|
||||
|
||||
base := time.Now().Add(-24 * time.Hour)
|
||||
// Stale generation first: measured a day ago, nothing since.
|
||||
const stale = 1500
|
||||
for i := 0; i < stale; i++ {
|
||||
s.StoreURLTestHistory("stale-"+strconv.Itoa(i),
|
||||
&adapter.URLTestHistory{LastOK: base.Add(time.Duration(i) * time.Millisecond), Delay: 30})
|
||||
}
|
||||
if s.Evicted() != 0 {
|
||||
t.Fatalf("evicted before the ceiling was reached")
|
||||
}
|
||||
// Now push past the ceiling with fresh measurements.
|
||||
for i := 0; i <= maxBoardEntries; i++ {
|
||||
s.StoreURLTestHistory("fresh-"+strconv.Itoa(i),
|
||||
&adapter.URLTestHistory{LastOK: time.Now(), Delay: 15})
|
||||
}
|
||||
|
||||
if got := s.Evicted(); got == 0 {
|
||||
t.Fatalf("board grew past %d entries without evicting anything — it is still unbounded", maxBoardEntries)
|
||||
}
|
||||
s.access.RLock()
|
||||
size := len(s.delayHistory)
|
||||
s.access.RUnlock()
|
||||
if size > maxBoardEntries {
|
||||
t.Fatalf("board holds %d entries, above the %d ceiling", size, maxBoardEntries)
|
||||
}
|
||||
|
||||
// The day-old generation is what went, not the fresh one.
|
||||
if s.LoadURLTestHistory("stale-0") != nil {
|
||||
t.Fatalf("the oldest observation survived while newer ones were dropped")
|
||||
}
|
||||
if s.LoadURLTestHistory("fresh-"+strconv.Itoa(maxBoardEntries)) == nil {
|
||||
t.Fatalf("the newest measurement was evicted")
|
||||
}
|
||||
|
||||
msgs := read()
|
||||
if len(msgs) == 0 {
|
||||
t.Fatalf("entries were evicted with no notice — eviction must never be silent")
|
||||
}
|
||||
m := msgs[0]
|
||||
for _, want := range []string{"health board full", "evicted", "re-probed"} {
|
||||
if !strings.Contains(m, want) {
|
||||
t.Fatalf("eviction notice %q does not say %q", m, want)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestBoardEvictionThroughMarkFailed pins the OTHER write path. MarkFailed is how
|
||||
// a dead node is recorded, and a flood of dead renamed nodes is exactly the shape
|
||||
// of the leak — so it has to prune too, not just the success path.
|
||||
func TestBoardEvictionThroughMarkFailed(t *testing.T) {
|
||||
captureEvictions(t)
|
||||
s := NewHistoryStorage()
|
||||
for i := 0; i <= maxBoardEntries; i++ {
|
||||
s.MarkFailed("dead-" + strconv.Itoa(i))
|
||||
}
|
||||
s.access.RLock()
|
||||
size := len(s.delayHistory)
|
||||
s.access.RUnlock()
|
||||
if size > maxBoardEntries {
|
||||
t.Fatalf("MarkFailed grew the board to %d, above the %d ceiling", size, maxBoardEntries)
|
||||
}
|
||||
if s.Evicted() == 0 {
|
||||
t.Fatalf("MarkFailed never prunes — the failure path is still unbounded")
|
||||
}
|
||||
}
|
||||
|
||||
// lx:end health-board
|
||||
@@ -10,11 +10,128 @@
|
||||
package urltest
|
||||
|
||||
import (
|
||||
"sort"
|
||||
"strconv"
|
||||
"sync"
|
||||
"time"
|
||||
|
||||
"github.com/sagernet/sing-box/adapter"
|
||||
"github.com/sagernet/sing-box/log"
|
||||
)
|
||||
|
||||
// --- board capacity ---------------------------------------------------------
|
||||
//
|
||||
// The board is the one structure in the daemon whose key space is chosen by
|
||||
// somebody else. Its keys are outbound TAGS, and on this box a tag is a node
|
||||
// NAME straight out of the subscription — plus the derived per-group egress
|
||||
// copies ("group-<g>-m<i>-<node>") and per-chain hop copies the probe planner
|
||||
// creates for the same nodes. Providers rename their nodes freely, so a daily
|
||||
// subscription refresh introduces a whole new generation of keys, while the
|
||||
// store itself is pinned to the ENGINE's context (shater/engine.New) and so
|
||||
// outlives every generation and every Apply — by design, so health survives a
|
||||
// config change.
|
||||
//
|
||||
// Nothing ever removed a key. DeleteURLTestHistory exists but no shater path
|
||||
// calls it (only daemon/ and clashapi/, which this fork does not run), so the
|
||||
// map was strictly append-only for the life of the process — and the process is
|
||||
// expected to live for months.
|
||||
//
|
||||
// The arithmetic: ~380 nodes, and a config with a couple of egress-bound groups
|
||||
// plus a handful of chains puts a LIVE generation at roughly 380 base tags +
|
||||
// 2x380 group copies + ~100 chain copies ≈ 1200 keys. One new generation per day
|
||||
// is ~440k keys a year, at ~200 B per entry (map bucket + a tag string that is
|
||||
// routinely 30-50 B with flag emoji, + a 56 B URLTestHistory) ≈ 88 MB of a
|
||||
// 512 MB box — spent entirely on nodes that no longer exist.
|
||||
const (
|
||||
// maxBoardEntries is the hard ceiling. 4096 is ~3.4 live generations, so the
|
||||
// board comfortably holds the current config plus the overlap while a
|
||||
// subscription refresh swaps names, and still costs under a megabyte. A tighter
|
||||
// bound would start evicting tags the running config actually uses; a looser one
|
||||
// would stop being a bound in any useful sense.
|
||||
maxBoardEntries = 4096
|
||||
// keepBoardEntries is the prune target: drop a quarter at a time so the
|
||||
// O(n log n) selection is amortised over ~1024 inserts instead of running on
|
||||
// every probe once the board is full.
|
||||
keepBoardEntries = 3072
|
||||
)
|
||||
|
||||
// boardEvictionLog reports an eviction. A package var so tests can capture it;
|
||||
// production leaves it writing to the process log, which under procd is the same
|
||||
// syslog/logsink stream every other daemon line lands in.
|
||||
//
|
||||
// Eviction is NEVER silent. It is not free either: an evicted tag reverts to
|
||||
// "untested" and its next probe re-measures it, so a board that evicts entries
|
||||
// belonging to the LIVE config is a board whose ceiling is too low — and the only
|
||||
// way anyone finds that out is this line.
|
||||
var boardEvictionLog = func(msg string) { boardLogger().Warn(msg) }
|
||||
|
||||
// pruneLocked drops the least-recently-OBSERVED entries when the board exceeds
|
||||
// maxBoardEntries. "Least recently observed" is max(LastOK, LastFail): the entry
|
||||
// nothing has measured for the longest is, on this box, precisely a tag that no
|
||||
// longer exists in any config — a renamed node, a removed group copy, a retired
|
||||
// chain hop. Caller holds access.
|
||||
func (s *HistoryStorage) pruneLocked() {
|
||||
if len(s.delayHistory) <= maxBoardEntries {
|
||||
return
|
||||
}
|
||||
type kv struct {
|
||||
tag string
|
||||
seen time.Time
|
||||
}
|
||||
all := make([]kv, 0, len(s.delayHistory))
|
||||
for tag, h := range s.delayHistory {
|
||||
seen := h.LastOK
|
||||
if h.LastFail.After(seen) {
|
||||
seen = h.LastFail
|
||||
}
|
||||
all = append(all, kv{tag, seen})
|
||||
}
|
||||
sort.Slice(all, func(i, j int) bool { return all[i].seen.Before(all[j].seen) })
|
||||
drop := len(all) - keepBoardEntries
|
||||
var oldest time.Time
|
||||
for i := 0; i < drop; i++ {
|
||||
if i == 0 {
|
||||
oldest = all[i].seen
|
||||
}
|
||||
delete(s.delayHistory, all[i].tag)
|
||||
}
|
||||
s.evicted += uint64(drop)
|
||||
|
||||
msg := "urltest: health board full (" + strconv.Itoa(maxBoardEntries) + " tags) — evicted " +
|
||||
strconv.Itoa(drop) + " least-recently-measured entries (" + strconv.FormatUint(s.evicted, 10) +
|
||||
" total since start); they revert to untested and will be re-probed"
|
||||
if !oldest.IsZero() {
|
||||
msg += "; oldest observation was " + time.Since(oldest).Truncate(time.Second).String() + " ago"
|
||||
}
|
||||
boardEvictionLog(msg)
|
||||
}
|
||||
|
||||
// Evicted reports how many entries the capacity bound has dropped since the store
|
||||
// was created. Nonzero means the board reached maxBoardEntries at least once.
|
||||
func (s *HistoryStorage) Evicted() uint64 {
|
||||
if s == nil {
|
||||
return 0
|
||||
}
|
||||
s.access.RLock()
|
||||
defer s.access.RUnlock()
|
||||
return s.evicted
|
||||
}
|
||||
|
||||
// boardLogger is the process-wide fallback logger for eviction notices. The store
|
||||
// is built from a plain constructor with no logger in sight (box.New, the daemon,
|
||||
// shater/engine all call NewHistoryStorage()), so rather than change that
|
||||
// signature everywhere the notice goes to the standard logger — which on the
|
||||
// router is the daemon's own stderr, i.e. the same sink logsink owns.
|
||||
var (
|
||||
boardLogOnce sync.Once
|
||||
boardLog log.ContextLogger
|
||||
)
|
||||
|
||||
func boardLogger() log.ContextLogger {
|
||||
boardLogOnce.Do(func() { boardLog = log.StdLogger() })
|
||||
return boardLog
|
||||
}
|
||||
|
||||
// HealthVerdict classifies a stored history entry at read time.
|
||||
type HealthVerdict int
|
||||
|
||||
@@ -54,6 +171,7 @@ func (s *HistoryStorage) MarkFailed(tag string) {
|
||||
updated.Delay = previous.Delay
|
||||
}
|
||||
s.delayHistory[tag] = updated
|
||||
s.pruneLocked()
|
||||
s.notifyUpdated()
|
||||
s.access.Unlock()
|
||||
}
|
||||
|
||||
@@ -21,6 +21,10 @@ type HistoryStorage struct {
|
||||
access sync.RWMutex
|
||||
delayHistory map[string]*adapter.URLTestHistory
|
||||
updateHooks []*observable.Subscriber[struct{}]
|
||||
// evicted counts entries dropped by the capacity bound (board_lx.go). The map
|
||||
// is keyed by outbound tags chosen by a subscription provider, so it needs a
|
||||
// ceiling; see the comment on maxBoardEntries.
|
||||
evicted uint64
|
||||
}
|
||||
|
||||
func NewHistoryStorage() *HistoryStorage {
|
||||
@@ -71,6 +75,11 @@ func (s *HistoryStorage) StoreURLTestHistory(tag string, history *adapter.URLTes
|
||||
}
|
||||
// lx:end health-board
|
||||
s.delayHistory[tag] = history
|
||||
// lx:begin health-board — the map is keyed by provider-chosen tags and the
|
||||
// store outlives every engine generation, so it must bound itself here: no
|
||||
// shater path ever calls DeleteURLTestHistory. See maxBoardEntries.
|
||||
s.pruneLocked()
|
||||
// lx:end health-board
|
||||
s.notifyUpdated()
|
||||
s.access.Unlock()
|
||||
}
|
||||
|
||||
@@ -126,6 +126,12 @@ func (t *HTTP3Transport) newTransport() *http3.Transport {
|
||||
conn.Close()
|
||||
return nil, dialErr
|
||||
}
|
||||
// quic-go does not take ownership of the packet conn passed to
|
||||
// DialEarly: when the connection ends it only stops reading.
|
||||
go func() {
|
||||
<-quicConn.Context().Done()
|
||||
conn.Close()
|
||||
}()
|
||||
return quicConn, nil
|
||||
},
|
||||
TLSClientConfig: t.tlsConfig,
|
||||
|
||||
@@ -0,0 +1,351 @@
|
||||
package quic
|
||||
|
||||
import (
|
||||
"context"
|
||||
"crypto/tls"
|
||||
"net"
|
||||
"net/http"
|
||||
"net/url"
|
||||
"sync"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"github.com/sagernet/quic-go"
|
||||
"github.com/sagernet/quic-go/http3"
|
||||
sbTLS "github.com/sagernet/sing-box/common/tls"
|
||||
C "github.com/sagernet/sing-box/constant"
|
||||
"github.com/sagernet/sing-box/dns"
|
||||
"github.com/sagernet/sing-box/dns/transport"
|
||||
"github.com/sagernet/sing-box/option"
|
||||
"github.com/sagernet/sing/common"
|
||||
"github.com/sagernet/sing/common/logger"
|
||||
M "github.com/sagernet/sing/common/metadata"
|
||||
N "github.com/sagernet/sing/common/network"
|
||||
|
||||
mDNS "github.com/miekg/dns"
|
||||
)
|
||||
|
||||
var _ N.Dialer = (*trackingDialer)(nil)
|
||||
|
||||
// These tests pin down who owns the UDP socket handed to quic-go.
|
||||
//
|
||||
// quic-go's Dial/DialEarly take a net.PacketConn but do NOT take ownership of
|
||||
// it: quic.setupTransport() builds a Transport with createdConn=false, and
|
||||
// Transport.Close() then only calls conn.SetReadDeadline(time.Now()) instead of
|
||||
// conn.Close(). So every QUIC connection torn down here — idle timeout, a
|
||||
// retryable error, an engine reload calling Reset() — used to strand the UDP
|
||||
// socket that carried it for the rest of the process's life. On a router that
|
||||
// resolves through DoQ/DoH3 for months that is an unbounded fd leak.
|
||||
//
|
||||
// Both tests reconnect once and assert the socket from the FIRST connection is
|
||||
// actually closed. Without the `<-conn.Context().Done() -> rawConn.Close()`
|
||||
// watchdogs in quic.go / http3.go they fail on that assertion.
|
||||
|
||||
type trackedConn struct {
|
||||
net.Conn
|
||||
closeOnce sync.Once
|
||||
closed chan struct{}
|
||||
}
|
||||
|
||||
func (c *trackedConn) Close() error {
|
||||
c.closeOnce.Do(func() { close(c.closed) })
|
||||
return c.Conn.Close()
|
||||
}
|
||||
|
||||
// trackingDialer hands out real UDP sockets and remembers every one of them.
|
||||
type trackingDialer struct {
|
||||
access sync.Mutex
|
||||
conns []*trackedConn
|
||||
}
|
||||
|
||||
func (d *trackingDialer) DialContext(ctx context.Context, network string, destination M.Socksaddr) (net.Conn, error) {
|
||||
conn, err := (&net.Dialer{}).DialContext(ctx, network, destination.String())
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
tracked := &trackedConn{Conn: conn, closed: make(chan struct{})}
|
||||
d.access.Lock()
|
||||
d.conns = append(d.conns, tracked)
|
||||
d.access.Unlock()
|
||||
return tracked, nil
|
||||
}
|
||||
|
||||
func (d *trackingDialer) ListenPacket(ctx context.Context, destination M.Socksaddr) (net.PacketConn, error) {
|
||||
return net.ListenUDP("udp", nil)
|
||||
}
|
||||
|
||||
func (d *trackingDialer) count() int {
|
||||
d.access.Lock()
|
||||
defer d.access.Unlock()
|
||||
return len(d.conns)
|
||||
}
|
||||
|
||||
func (d *trackingDialer) at(index int) *trackedConn {
|
||||
d.access.Lock()
|
||||
defer d.access.Unlock()
|
||||
return d.conns[index]
|
||||
}
|
||||
|
||||
func (d *trackingDialer) closeAll() {
|
||||
d.access.Lock()
|
||||
defer d.access.Unlock()
|
||||
for _, conn := range d.conns {
|
||||
conn.Close()
|
||||
}
|
||||
}
|
||||
|
||||
func requireClosed(t *testing.T, conn *trackedConn, what string) {
|
||||
t.Helper()
|
||||
select {
|
||||
case <-conn.closed:
|
||||
case <-time.After(5 * time.Second):
|
||||
t.Fatalf("%s: the UDP socket of the retired QUIC connection was never closed — quic-go does not own it, we must", what)
|
||||
}
|
||||
}
|
||||
|
||||
func requireDialed(t *testing.T, dialer *trackingDialer, want int) {
|
||||
t.Helper()
|
||||
deadline := time.Now().Add(5 * time.Second)
|
||||
for time.Now().Before(deadline) {
|
||||
if dialer.count() >= want {
|
||||
return
|
||||
}
|
||||
time.Sleep(10 * time.Millisecond)
|
||||
}
|
||||
t.Fatalf("expected at least %d dial(s), got %d", want, dialer.count())
|
||||
}
|
||||
|
||||
func testServerTLSConfig(t *testing.T, nextProtos []string) *tls.Config {
|
||||
t.Helper()
|
||||
certificate, err := sbTLS.GenerateKeyPair(nil, nil, nil, "localhost")
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
return &tls.Config{
|
||||
Certificates: []tls.Certificate{*certificate},
|
||||
NextProtos: nextProtos,
|
||||
MinVersion: tls.VersionTLS13,
|
||||
}
|
||||
}
|
||||
|
||||
func testClientTLSConfig(t *testing.T, nextProtos []string) sbTLS.Config {
|
||||
t.Helper()
|
||||
config, err := sbTLS.NewClient(context.Background(), logger.NOP(), "localhost", option.OutboundTLSOptions{
|
||||
Enabled: true,
|
||||
Insecure: true,
|
||||
ServerName: "localhost",
|
||||
})
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
config.SetNextProtos(nextProtos)
|
||||
return config
|
||||
}
|
||||
|
||||
// startDoQServer serves a minimal DoQ responder and returns its address.
|
||||
func startDoQServer(t *testing.T) M.Socksaddr {
|
||||
t.Helper()
|
||||
listener, err := quic.ListenAddr("127.0.0.1:0", testServerTLSConfig(t, []string{"doq"}), nil)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
ctx, cancel := context.WithCancel(context.Background())
|
||||
t.Cleanup(func() {
|
||||
cancel()
|
||||
listener.Close()
|
||||
})
|
||||
go func() {
|
||||
for {
|
||||
conn, acceptErr := listener.Accept(ctx)
|
||||
if acceptErr != nil {
|
||||
return
|
||||
}
|
||||
go func(conn *quic.Conn) {
|
||||
for {
|
||||
stream, streamErr := conn.AcceptStream(ctx)
|
||||
if streamErr != nil {
|
||||
return
|
||||
}
|
||||
go func(stream *quic.Stream) {
|
||||
defer stream.Close()
|
||||
request, readErr := transport.ReadMessage(stream)
|
||||
if readErr != nil {
|
||||
return
|
||||
}
|
||||
response := new(mDNS.Msg)
|
||||
response.SetReply(request)
|
||||
transport.WriteMessage(stream, 0, response)
|
||||
}(stream)
|
||||
}
|
||||
}(conn)
|
||||
}
|
||||
}()
|
||||
return M.ParseSocksaddr(listener.Addr().String())
|
||||
}
|
||||
|
||||
func testQuery() *mDNS.Msg {
|
||||
message := new(mDNS.Msg)
|
||||
message.SetQuestion("example.com.", mDNS.TypeA)
|
||||
return message
|
||||
}
|
||||
|
||||
func TestQUICTransportClosesPacketConnOnReconnect(t *testing.T) {
|
||||
t.Parallel()
|
||||
serverAddr := startDoQServer(t)
|
||||
dialer := &trackingDialer{}
|
||||
t.Cleanup(dialer.closeAll)
|
||||
|
||||
dnsTransport := &Transport{
|
||||
TransportAdapter: dns.NewTransportAdapter(C.DNSTypeQUIC, "test-doq", nil),
|
||||
dialer: dialer,
|
||||
serverAddr: serverAddr,
|
||||
tlsConfig: testClientTLSConfig(t, []string{"doq"}),
|
||||
connection: transport.NewConnPool(transport.ConnPoolOptions[*quic.Conn]{
|
||||
Mode: transport.ConnPoolSingle,
|
||||
IsAlive: func(conn *quic.Conn) bool {
|
||||
return conn != nil && !common.Done(conn.Context())
|
||||
},
|
||||
Close: func(conn *quic.Conn, _ error) {
|
||||
conn.CloseWithError(0, "")
|
||||
},
|
||||
}),
|
||||
}
|
||||
t.Cleanup(func() { dnsTransport.Close() })
|
||||
|
||||
ctx, cancel := context.WithTimeout(context.Background(), 15*time.Second)
|
||||
defer cancel()
|
||||
|
||||
if _, err := dnsTransport.Exchange(ctx, testQuery()); err != nil {
|
||||
t.Fatal("first exchange: ", err)
|
||||
}
|
||||
requireDialed(t, dialer, 1)
|
||||
first := dialer.at(0)
|
||||
|
||||
// Retire the connection the way a retryable error or an engine reload does.
|
||||
dnsTransport.Reset()
|
||||
requireClosed(t, first, "Reset()")
|
||||
|
||||
// The reconnect must still work, on a fresh socket.
|
||||
if _, err := dnsTransport.Exchange(ctx, testQuery()); err != nil {
|
||||
t.Fatal("second exchange: ", err)
|
||||
}
|
||||
requireDialed(t, dialer, 2)
|
||||
second := dialer.at(1)
|
||||
if second == first {
|
||||
t.Fatal("expected a new UDP socket for the reconnect")
|
||||
}
|
||||
|
||||
if err := dnsTransport.Close(); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
requireClosed(t, second, "Close()")
|
||||
}
|
||||
|
||||
func TestHTTP3TransportClosesPacketConnOnReconnect(t *testing.T) {
|
||||
t.Parallel()
|
||||
mux := http.NewServeMux()
|
||||
mux.HandleFunc("/dns-query", func(writer http.ResponseWriter, request *http.Request) {
|
||||
message, err := readRequestMessage(request)
|
||||
if err != nil {
|
||||
writer.WriteHeader(http.StatusBadRequest)
|
||||
return
|
||||
}
|
||||
response := new(mDNS.Msg)
|
||||
response.SetReply(message)
|
||||
rawResponse, err := response.Pack()
|
||||
if err != nil {
|
||||
writer.WriteHeader(http.StatusInternalServerError)
|
||||
return
|
||||
}
|
||||
writer.Header().Set("Content-Type", transport.MimeType)
|
||||
writer.Write(rawResponse)
|
||||
})
|
||||
listener, err := quic.ListenAddrEarly("127.0.0.1:0", testServerTLSConfig(t, []string{http3.NextProtoH3}), nil)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
server := &http3.Server{Handler: mux}
|
||||
go server.ServeListener(listener)
|
||||
t.Cleanup(func() {
|
||||
server.Close()
|
||||
listener.Close()
|
||||
})
|
||||
serverAddr := M.ParseSocksaddr(listener.Addr().String())
|
||||
|
||||
dialer := &trackingDialer{}
|
||||
t.Cleanup(dialer.closeAll)
|
||||
|
||||
stdConfig := &tls.Config{
|
||||
InsecureSkipVerify: true,
|
||||
ServerName: "localhost",
|
||||
NextProtos: []string{http3.NextProtoH3},
|
||||
MinVersion: tls.VersionTLS13,
|
||||
}
|
||||
dnsTransport := &HTTP3Transport{
|
||||
TransportAdapter: dns.NewTransportAdapter(C.DNSTypeHTTP3, "test-doh3", nil),
|
||||
logger: logger.NOP(),
|
||||
dialer: dialer,
|
||||
destination: &url.URL{Scheme: "https", Host: "localhost", Path: "/dns-query"},
|
||||
headers: http.Header{},
|
||||
serverAddr: serverAddr,
|
||||
tlsConfig: stdConfig,
|
||||
}
|
||||
dnsTransport.transport = dnsTransport.newTransport()
|
||||
t.Cleanup(func() { dnsTransport.Close() })
|
||||
|
||||
ctx, cancel := context.WithTimeout(context.Background(), 15*time.Second)
|
||||
defer cancel()
|
||||
|
||||
if _, err = dnsTransport.Exchange(ctx, testQuery()); err != nil {
|
||||
t.Fatal("first exchange: ", err)
|
||||
}
|
||||
requireDialed(t, dialer, 1)
|
||||
first := dialer.at(0)
|
||||
|
||||
dnsTransport.Reset()
|
||||
requireClosed(t, first, "Reset()")
|
||||
|
||||
if _, err = dnsTransport.Exchange(ctx, testQuery()); err != nil {
|
||||
t.Fatal("second exchange: ", err)
|
||||
}
|
||||
requireDialed(t, dialer, 2)
|
||||
second := dialer.at(1)
|
||||
if second == first {
|
||||
t.Fatal("expected a new UDP socket for the reconnect")
|
||||
}
|
||||
|
||||
if err = dnsTransport.Close(); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
requireClosed(t, second, "Close()")
|
||||
}
|
||||
|
||||
func readRequestMessage(request *http.Request) (*mDNS.Msg, error) {
|
||||
defer request.Body.Close()
|
||||
rawMessage := make([]byte, 4096)
|
||||
n, err := readFull(request.Body, rawMessage)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
var message mDNS.Msg
|
||||
err = message.Unpack(rawMessage[:n])
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
return &message, nil
|
||||
}
|
||||
|
||||
func readFull(reader interface{ Read([]byte) (int, error) }, buffer []byte) (int, error) {
|
||||
var total int
|
||||
for total < len(buffer) {
|
||||
n, err := reader.Read(buffer[total:])
|
||||
total += n
|
||||
if err != nil {
|
||||
if total > 0 {
|
||||
return total, nil
|
||||
}
|
||||
return total, err
|
||||
}
|
||||
}
|
||||
return total, nil
|
||||
}
|
||||
@@ -4,6 +4,7 @@ import (
|
||||
"context"
|
||||
"errors"
|
||||
"os"
|
||||
"time"
|
||||
|
||||
"github.com/sagernet/quic-go"
|
||||
"github.com/sagernet/sing-box/adapter"
|
||||
@@ -117,6 +118,12 @@ func (t *Transport) Exchange(ctx context.Context, message *mDNS.Msg) (*mDNS.Msg,
|
||||
rawConn.Close()
|
||||
return nil, E.Cause(err, "establish QUIC connection")
|
||||
}
|
||||
// quic-go does not take ownership of the packet conn passed to
|
||||
// DialEarly: when the connection ends it only stops reading.
|
||||
go func() {
|
||||
<-earlyConnection.Context().Done()
|
||||
rawConn.Close()
|
||||
}()
|
||||
return earlyConnection, nil
|
||||
})
|
||||
if err != nil {
|
||||
@@ -144,6 +151,11 @@ func (t *Transport) exchange(ctx context.Context, message *mDNS.Msg, conn *quic.
|
||||
return nil, E.Cause(err, "open stream")
|
||||
}
|
||||
defer stream.CancelRead(0)
|
||||
stopWatch := context.AfterFunc(ctx, func() {
|
||||
stream.CancelRead(0)
|
||||
_ = stream.SetWriteDeadline(time.Now())
|
||||
})
|
||||
defer stopWatch()
|
||||
err = transport.WriteMessage(stream, 0, message)
|
||||
if err != nil {
|
||||
stream.Close()
|
||||
|
||||
@@ -80,6 +80,10 @@ define Package/shater-core/install
|
||||
$(INSTALL_DIR) $(1)/etc/init.d
|
||||
$(INSTALL_BIN) ./files/etc/init.d/shater $(1)/etc/init.d/shater
|
||||
$(INSTALL_BIN) ./files/etc/init.d/shater-cron $(1)/etc/init.d/shater-cron
|
||||
# START=21 one-shot that loads the persisted fail-closed plane before fw4's
|
||||
# `lan -> wan ACCEPT` can be the only thing on the box (the main init is
|
||||
# START=99, i.e. seconds of plaintext forwarding on every boot).
|
||||
$(INSTALL_BIN) ./files/etc/init.d/shater-armor $(1)/etc/init.d/shater-armor
|
||||
|
||||
$(INSTALL_DIR) $(1)/etc/hotplug.d/iface
|
||||
$(INSTALL_BIN) ./files/etc/hotplug.d/iface/99-shater $(1)/etc/hotplug.d/iface/99-shater
|
||||
|
||||
@@ -33,6 +33,24 @@
|
||||
# be running. `start` raises ACTIVE_FLAG, `stop` clears it; hotplug/cron
|
||||
# reconcile ONLY while the flag is up, so an admin `stop` STICKS — no
|
||||
# background actor may resurrect interception behind a stopped daemon.
|
||||
# * BEING REPLACED IS NOT BEING SWITCHED OFF. `restart`, `reload` (which is
|
||||
# stop+start, i.e. every LuCI Save & Apply) and every package upgrade all run
|
||||
# through `stop`, and the daemon's SIGTERM teardown removes the fail-closed
|
||||
# table unconditionally — it does not consult kill_switch at all. Between that
|
||||
# teardown and the successor's first apply the init GUARANTEES a gap: it waits
|
||||
# for the old process to exit (shater_wait_stopped), then runs `shaterd
|
||||
# migrate`, then starts a daemon that still has to build an engine. So a
|
||||
# restart is announced with RESTART_FLAG, which tells the outgoing daemon to
|
||||
# leave the fail-closed holding plane behind instead of bare routing. A real
|
||||
# `stop` raises no flag and therefore still means what it says.
|
||||
# * The FAIL-CLOSED PLANE MUST ALSO EXIST BEFORE THIS SCRIPT DOES. START=99 is
|
||||
# after fw4 (19) and netifd (20), so at every boot the LAN forwards to the WAN
|
||||
# in the clear for as long as it takes procd to decompress the daemon off
|
||||
# flash and get an engine up. /etc/init.d/shater-armor (START=21) loads
|
||||
# BOOT_ARMOR — a copy of the holding plane the daemon persists on every apply
|
||||
# — to close that window. This script owns the DISARM half: a deliberate
|
||||
# `stop`, or a missing daemon binary, removes the armor so it cannot outlive
|
||||
# the product it protects.
|
||||
# * The engine must never be permanently abandoned while interception stands:
|
||||
# respawn retries are infinite (procd never gives up); a sustained-dead
|
||||
# daemon is additionally escalated by the shater-cron watchdog.
|
||||
@@ -50,11 +68,36 @@ ACTIVE_FLAG=/var/run/shater.active
|
||||
# Written by `shaterd run`; the single-owner token this init waits on so a
|
||||
# restart never overlaps a new data plane with the previous one's teardown.
|
||||
PIDFILE=/var/run/shaterd.pid
|
||||
# Raised around a restart/reload, read by the OUTGOING `shaterd run` at SIGTERM:
|
||||
# present => "you are being replaced, leave the fail-closed plane standing";
|
||||
# absent => "you are being switched off, take everything down". tmpfs, so a
|
||||
# power cut can never make the next boot look like a restart.
|
||||
RESTART_FLAG=/var/run/shater.restarting
|
||||
# The persisted fail-closed holding plane. Written by the daemon on every apply,
|
||||
# loaded by /etc/init.d/shater-armor at boot. Its PRESENCE is the arm token, so
|
||||
# removing it here is how a deliberate stop stops the next boot from blocking.
|
||||
BOOT_ARMOR=/etc/shater/boot.nft
|
||||
# Seconds `start` will wait for a predecessor to finish its teardown. Must be
|
||||
# >= term_timeout below (procd's hard cap on a predecessor's life after SIGTERM)
|
||||
# so we never give up while procd is still letting it shut down cleanly.
|
||||
STOP_WAIT_SECS=40
|
||||
|
||||
# WHICH ACTION rc.common was invoked with, frozen at source time.
|
||||
#
|
||||
# rc.common sets `action=${2:-help}` before it sources this file, and every action
|
||||
# then runs as a function in THAT SAME shell — so `stop_service` can see whether it
|
||||
# was reached by `stop` or as the first half of `restart`/`reload`. That is the one
|
||||
# distinction procd itself does not expose (`restart` is literally `stop; start`,
|
||||
# and stop_service is called identically by both).
|
||||
#
|
||||
# Frozen into our own variable because `action` is a short, generic name that other
|
||||
# framework helpers also use as a local; a snapshot taken before any function runs
|
||||
# cannot be shadowed later. An EMPTY or unexpected value degrades to "real stop",
|
||||
# which is the pre-existing behaviour and the safe direction to be wrong in: it
|
||||
# costs a plaintext window on restart, where the other default would leave a
|
||||
# deliberately stopped router blocked.
|
||||
SHATER_RC_ACTION="$action"
|
||||
|
||||
# --- helpers ---------------------------------------------------------------
|
||||
|
||||
# True only when the stack is explicitly enabled in UCI.
|
||||
@@ -73,6 +116,21 @@ _slog() {
|
||||
[ "$(uci -q get shater.globals.log_syslog)" = "0" ] || logger -t shater "$@"
|
||||
}
|
||||
|
||||
# Announce/withdraw "this daemon is being replaced, not switched off". Read by
|
||||
# `shaterd run` when it receives SIGTERM.
|
||||
shater_mark_restart() {
|
||||
mkdir -p "$(dirname "$RESTART_FLAG")" 2>/dev/null
|
||||
: > "$RESTART_FLAG"
|
||||
}
|
||||
shater_clear_restart() { rm -f "$RESTART_FLAG"; }
|
||||
|
||||
# Remove the persisted boot armor, so the LAN is NOT blocked at the next boot
|
||||
# before the daemon starts. Called when the operator stops the service and when
|
||||
# the daemon binary is gone — in both cases nothing is going to come along and
|
||||
# replace the armor with a real data plane, and a kill switch with nothing behind
|
||||
# it is just a brick.
|
||||
shater_disarm_boot() { rm -f "$BOOT_ARMOR"; }
|
||||
|
||||
# Echo the pid of a LIVE `shaterd run`, or fail. The pidfile is written by the
|
||||
# daemon itself and removed only by the daemon that owns it, AFTER its teardown
|
||||
# has completed — so "pidfile names a live process" is precisely "the previous
|
||||
@@ -136,9 +194,20 @@ start_service() {
|
||||
# Guard: never claim to run without the daemon binary. A half-removed/failed
|
||||
# shaterd upgrade must degrade to "plugin off", not to a box that thinks
|
||||
# interception is live with nothing behind it.
|
||||
#
|
||||
# "Plugin off" now has to include DISARMING. With the boot armor in play, a
|
||||
# missing binary is the one case where the fail-closed plane could stand
|
||||
# forever with nothing able to replace it: the armor loads at START=21, the
|
||||
# daemon never starts, and every later boot repeats it. The product being gone
|
||||
# is not a security event — it is an uninstall — so the plane comes down and
|
||||
# the LAN returns to plain routing, loudly.
|
||||
if [ ! -x "$PROG" ]; then
|
||||
shater_clear_restart
|
||||
shater_disarm_boot
|
||||
rm -f "$ACTIVE_FLAG"
|
||||
nft delete table inet shater 2>/dev/null
|
||||
_slog -p daemon.err \
|
||||
"shaterd binary missing/not executable at $PROG — refusing to start (LAN stays on plain routing)"
|
||||
"shaterd binary missing/not executable at $PROG — refusing to start; the fail-closed plane and its boot armor have been REMOVED (LAN back to plain routing, unprotected). Reinstall shaterd."
|
||||
return 0
|
||||
fi
|
||||
|
||||
@@ -150,6 +219,11 @@ start_service() {
|
||||
# running, which is the boot case.
|
||||
shater_wait_stopped
|
||||
|
||||
# The predecessor is gone and has already consumed the flag (it reads it in its
|
||||
# SIGTERM handler). Withdraw it now, so a LATER `stop` is unambiguous even if
|
||||
# this start fails further down.
|
||||
shater_clear_restart
|
||||
|
||||
# Bring the UCI schema forward before the daemon reads it (idempotent;
|
||||
# refuses a newer schema) so an upgraded package never applies a stale config.
|
||||
#
|
||||
@@ -214,6 +288,33 @@ start_service() {
|
||||
}
|
||||
|
||||
stop_service() {
|
||||
# Say WHY we are stopping before procd sends the signal, because the daemon
|
||||
# cannot tell from the signal alone and the answer changes what it leaves in
|
||||
# the kernel:
|
||||
#
|
||||
# restart / reload -> a successor is coming. Raise RESTART_FLAG so the
|
||||
# outgoing daemon replaces its data plane with the
|
||||
# fail-closed HOLDING plane instead of removing it. The
|
||||
# gap until the successor applies is not a moment: this
|
||||
# script waits out the old process, runs `shaterd
|
||||
# migrate`, then starts a daemon that must build an
|
||||
# engine — all of it, until now, with `lan -> wan
|
||||
# ACCEPT` and nothing else.
|
||||
# anything else -> a deliberate `stop`. Everything comes down, and the
|
||||
# boot armor goes with it so the next boot does not
|
||||
# quietly reinstate what the operator just switched off.
|
||||
# An admin `stop` has to STICK; that is the same rule
|
||||
# ACTIVE_FLAG has always enforced for hotplug/cron.
|
||||
case "$SHATER_RC_ACTION" in
|
||||
restart|reload)
|
||||
shater_mark_restart
|
||||
;;
|
||||
*)
|
||||
shater_clear_restart
|
||||
shater_disarm_boot
|
||||
;;
|
||||
esac
|
||||
|
||||
# Drop the live-flag FIRST so a concurrent hotplug/cron tick cannot rebuild
|
||||
# what we are about to tear down. procd then sends SIGTERM to `shaterd run`,
|
||||
# which runs its OWN honest teardown (engine.Close + netplane restore) — we
|
||||
@@ -233,6 +334,11 @@ reload_service() {
|
||||
# disabled, `start` is a no-op, so a disable+apply cleanly tears everything
|
||||
# down. Because the wait lives in start_service, this path gets the same
|
||||
# stop-then-start ordering guarantee as `restart`.
|
||||
#
|
||||
# Marked EXPLICITLY as well as via SHATER_RC_ACTION: this is the path a routine
|
||||
# Save & Apply takes, so it is the one that must not depend on reading an
|
||||
# rc.common variable correctly. Belt and braces, one line.
|
||||
shater_mark_restart
|
||||
stop
|
||||
start
|
||||
}
|
||||
|
||||
@@ -0,0 +1,144 @@
|
||||
#!/bin/sh /etc/rc.common
|
||||
# /etc/init.d/shater-armor — the fail-closed plane, before the daemon exists.
|
||||
#
|
||||
# WHAT THIS CLOSES
|
||||
#
|
||||
# /etc/init.d/shater is START=99. By then fw4 (START=19) has long since loaded
|
||||
# `lan -> wan ACCEPT` and netifd (START=20) has brought the LAN bridge up, so the
|
||||
# router forwards LAN traffic to the WAN in the clear from the moment the link
|
||||
# comes up until `shaterd run` has been decompressed off flash, has waited out any
|
||||
# predecessor, has migrated UCI, has read the config and has installed its first
|
||||
# table. On router-class hardware with a UPX-packed binary that is seconds — and
|
||||
# they are exactly the seconds in which Wi-Fi finishes associating and every
|
||||
# client on the network reconnects and starts talking. `kill_switch=closed` was
|
||||
# configured the whole time and covered none of it.
|
||||
#
|
||||
# There was nothing in the package that could cover it either: no /etc/nftables.d
|
||||
# include, no `nft -f` in uci-defaults. Protection existed only inside a Go
|
||||
# process that had not started yet.
|
||||
#
|
||||
# HOW
|
||||
#
|
||||
# The daemon persists a copy of its fail-closed HOLDING plane (the same ruleset it
|
||||
# installs when the engine is down: one forward chain, LAN-to-LAN and router
|
||||
# traffic accepted, everything else from the diverted devices dropped) to
|
||||
# $ARMOR on every apply. This script loads it early. When the daemon comes up it
|
||||
# replaces the table atomically — the ruleset begins with `delete table` and adds
|
||||
# its own in one netlink transaction — so there is never a moment with no table.
|
||||
#
|
||||
# `iifname` matches by NAME at packet time, not by ifindex at load time, so
|
||||
# loading this before netifd has created br-lan is fine: the rules simply start
|
||||
# matching when the device appears. That is why START can sit here rather than
|
||||
# racing netifd.
|
||||
#
|
||||
# START=21: after fw4 (19) and netifd (20), because fw4's own start tears its
|
||||
# table down and rebuilds it and we do not want to be in the middle of that, and
|
||||
# because there is nothing to protect before the LAN device is being created. The
|
||||
# residual exposure is the fraction of a second between netifd's `ifup` and this
|
||||
# script, against seconds-to-a-minute before.
|
||||
#
|
||||
# THE ESCAPE HATCHES (a kill switch that cannot be switched off is a brick)
|
||||
#
|
||||
# * $ARMOR only exists while the daemon's last applied config was BOTH enabled
|
||||
# and fail-closed. `globals.enabled=0`, `kill_switch=open` and a deliberate
|
||||
# `/etc/init.d/shater stop` each remove it.
|
||||
# * We refuse to arm when the main service is disabled in rc.d, or when the
|
||||
# daemon binary is gone — in either case nothing would ever come along to
|
||||
# replace the armor with a real data plane.
|
||||
# * We refuse to arm when UCI can be read AND says the stack is disabled. A
|
||||
# config that cannot be read is NOT a refusal: that case is precisely why the
|
||||
# armor is a file rather than a query.
|
||||
# * The chain hooks `forward` only, so SSH, LuCI and the admin panel (all input
|
||||
# hook, to the router's own addresses) stay reachable. The operator can always
|
||||
# get in and undo this.
|
||||
#
|
||||
# busybox ash only — no bashisms.
|
||||
|
||||
START=21 # after firewall (19) and network (20), long before shater (99)
|
||||
STOP=89
|
||||
|
||||
ARMOR=/etc/shater/boot.nft
|
||||
PROG=/usr/bin/shaterd
|
||||
|
||||
# Syslog line that honors globals.log_syslog, like the other two inits. An
|
||||
# unreadable UCI leaves the option empty => ON, which is what we want here: the
|
||||
# one boot where the config cannot be read is the boot worth logging.
|
||||
_slog() {
|
||||
[ "$(uci -q get shater.globals.log_syslog)" = "0" ] || logger -t shater-armor "$@"
|
||||
}
|
||||
|
||||
# Is the MAIN service enabled at boot? Answered by looking for its rc.d symlink
|
||||
# rather than by running `/etc/init.d/shater enabled`: that is a USE_PROCD script,
|
||||
# so every action of it sources procd.sh, which takes a blocking flock — and this
|
||||
# runs at START=21, in the middle of boot, for a question a glob answers exactly
|
||||
# as well. The START number is not hardcoded; any S<NN>shater counts.
|
||||
shater_service_enabled() {
|
||||
local f
|
||||
for f in /etc/rc.d/S[0-9][0-9]shater; do
|
||||
[ -e "$f" ] && return 0
|
||||
done
|
||||
return 1
|
||||
}
|
||||
|
||||
start() {
|
||||
# No saved plane => the stack has never applied an enabled, fail-closed config
|
||||
# (or it was explicitly switched off). Nothing to do, and nothing to say.
|
||||
[ -f "$ARMOR" ] || return 0
|
||||
[ -s "$ARMOR" ] || {
|
||||
_slog -p daemon.err "$ARMOR is empty — NOT arming; the LAN is unprotected until shaterd starts"
|
||||
return 0
|
||||
}
|
||||
|
||||
# Never arm something nothing can disarm.
|
||||
[ -x "$PROG" ] || {
|
||||
_slog -p daemon.err \
|
||||
"$PROG is missing — NOT arming (nothing would replace the block with a working data plane); the LAN stays on plain routing"
|
||||
return 0
|
||||
}
|
||||
shater_service_enabled || {
|
||||
_slog -p daemon.warn \
|
||||
"the shater service is disabled in rc.d — NOT arming (nothing would replace the block with a working data plane); the LAN stays on plain routing"
|
||||
return 0
|
||||
}
|
||||
|
||||
# A READABLE config that says "off" wins over the saved plane (it means the
|
||||
# daemon was stopped before it could disarm). An UNREADABLE config does not:
|
||||
# that is the case this whole mechanism exists for.
|
||||
en=$(uci -q get shater.globals.enabled 2>/dev/null)
|
||||
if [ -n "$en" ] && [ "$en" != "1" ]; then
|
||||
rm -f "$ARMOR"
|
||||
_slog -p daemon.info "globals.enabled=$en — boot armor removed, not arming"
|
||||
return 0
|
||||
fi
|
||||
|
||||
command -v nft >/dev/null 2>&1 || {
|
||||
_slog -p daemon.err "nft is not installed — cannot arm; the LAN is unprotected until shaterd starts"
|
||||
return 0
|
||||
}
|
||||
|
||||
# Validate before loading: a truncated/incompatible snapshot must not leave a
|
||||
# half-built table behind on the one boot it is needed.
|
||||
if ! nft -c -f "$ARMOR" >/dev/null 2>&1; then
|
||||
_slog -p daemon.err \
|
||||
"$ARMOR did not validate (nft -c) — NOT arming; the LAN is unprotected until shaterd starts"
|
||||
return 0
|
||||
fi
|
||||
if nft -f "$ARMOR" >/dev/null 2>&1; then
|
||||
_slog -p daemon.warn \
|
||||
"fail-closed plane armed from $ARMOR: LAN->WAN forwarding is BLOCKED until shaterd applies. SSH, LuCI and the admin panel stay reachable."
|
||||
else
|
||||
_slog -p daemon.err \
|
||||
"could not load $ARMOR — the LAN is unprotected until shaterd starts"
|
||||
fi
|
||||
return 0
|
||||
}
|
||||
|
||||
stop() {
|
||||
# Deliberately a NO-OP. By the time anything stops this service the daemon owns
|
||||
# `inet shater`, and deleting the table here would dismantle a LIVE data plane
|
||||
# on the strength of a service that only ever ran for one second at boot. The
|
||||
# disarm paths that matter live where the decision is actually made:
|
||||
# /etc/init.d/shater stop (operator switched it off) and the daemon itself
|
||||
# (globals.enabled=0 / kill_switch=open).
|
||||
return 0
|
||||
}
|
||||
@@ -110,6 +110,14 @@ SHATER_BRINGUP='
|
||||
done
|
||||
[ -x /etc/init.d/shater ] && /etc/init.d/shater enable
|
||||
[ -x /etc/init.d/shater-cron ] && /etc/init.d/shater-cron enable
|
||||
# The boot-time fail-closed armor. `enable` only — it is a one-shot that loads
|
||||
# the persisted holding plane at START=21, and running it NOW would install a
|
||||
# block on a live box moments before the daemon replaces it anyway. It has to
|
||||
# be enabled here regardless of whether the stack is on: the file it loads only
|
||||
# exists while the daemon wants it to, so an enabled-but-unarmed service is a
|
||||
# no-op, and enabling it later would mean the first boot after an upgrade is
|
||||
# the one boot still exposed.
|
||||
[ -x /etc/init.d/shater-armor ] && /etc/init.d/shater-armor enable
|
||||
[ -x /etc/init.d/shater ] && /etc/init.d/shater restart
|
||||
[ -x /etc/init.d/shater-cron ] && /etc/init.d/shater-cron restart
|
||||
exit 0
|
||||
|
||||
@@ -262,6 +262,65 @@
|
||||
}
|
||||
}
|
||||
|
||||
/* ---- fixture band (dev builds only; see App.tsx MockBanner) ----
|
||||
Deliberately outside the crit/amber vocabulary: nothing is wrong with the
|
||||
router, there is no router. The hazard hatch is the service-sticker language a
|
||||
piece of network hardware already uses for "this unit is not in service". */
|
||||
.mock-band {
|
||||
display: flex;
|
||||
align-items: center;
|
||||
gap: calc(var(--u, 8px) * 1.5);
|
||||
margin-top: calc(var(--u, 8px) * 2);
|
||||
padding: 10px 14px;
|
||||
border: 1px dashed var(--faint);
|
||||
border-radius: 9px;
|
||||
background: repeating-linear-gradient(
|
||||
-45deg,
|
||||
var(--sink),
|
||||
var(--sink) 9px,
|
||||
var(--panel) 9px,
|
||||
var(--panel) 18px
|
||||
);
|
||||
}
|
||||
.mock-band-tag {
|
||||
flex-shrink: 0;
|
||||
align-self: flex-start;
|
||||
padding: 3px 7px;
|
||||
border: 1px solid var(--faint);
|
||||
border-radius: 4px;
|
||||
background: var(--raised);
|
||||
font-family: var(--font-mono);
|
||||
font-size: 10px;
|
||||
font-weight: 700;
|
||||
letter-spacing: 0.14em;
|
||||
color: var(--dim);
|
||||
}
|
||||
.mock-band-copy {
|
||||
flex: 1;
|
||||
min-width: 0;
|
||||
display: flex;
|
||||
flex-direction: column;
|
||||
gap: 2px;
|
||||
}
|
||||
.mock-band-headline {
|
||||
font-family: var(--font-mono);
|
||||
font-size: 12.5px;
|
||||
font-weight: 700;
|
||||
letter-spacing: 0.02em;
|
||||
color: var(--ink);
|
||||
}
|
||||
.mock-band-detail {
|
||||
font-size: 12.5px;
|
||||
line-height: 1.5;
|
||||
color: var(--dim);
|
||||
max-width: 76ch;
|
||||
}
|
||||
.mock-band-detail code {
|
||||
font-family: var(--font-mono);
|
||||
font-size: 11.5px;
|
||||
color: var(--ink);
|
||||
}
|
||||
|
||||
/* ---- commit-confirm band (every page except Apply, which has the full panel) ----
|
||||
Same plate as the protection banner so the two read as one family; the seconds
|
||||
are the loud element because they are the only thing that is running out. */
|
||||
@@ -397,6 +456,13 @@
|
||||
.finding--warning {
|
||||
border-color: color-mix(in srgb, var(--amber) 40%, var(--groove));
|
||||
}
|
||||
/* The daemon's "the list is capped" disclosure. Dashed, because the row is about
|
||||
what ISN'T here — it must not read as one more finding to work through. */
|
||||
.finding--truncated {
|
||||
border-style: dashed;
|
||||
border-color: color-mix(in srgb, var(--amber) 40%, var(--groove));
|
||||
background: var(--panel);
|
||||
}
|
||||
.finding-copy {
|
||||
flex: 1;
|
||||
min-width: 0;
|
||||
|
||||
+31
-1
@@ -8,6 +8,7 @@ import { usePendingConfirm } from './pendingConfirm'
|
||||
import { bootstrapSession } from './session'
|
||||
import { ROUTES, navigate, useRoute } from './router'
|
||||
import { engineState, protectionState } from './planeState'
|
||||
import { truncationNote } from './findings'
|
||||
import type { Route } from './router'
|
||||
import { Overview, Placeholder, Nodes, Routing, Apply, DNS, Devices, Targets, Settings, Profiles, Insights, Networks } from './pages'
|
||||
|
||||
@@ -100,6 +101,7 @@ export function App() {
|
||||
footer={<StatusBar status={status} />}
|
||||
>
|
||||
<Nav route={route} />
|
||||
<MockBanner />
|
||||
<PlaneBanner status={status} route={route} />
|
||||
<ConfirmBand route={route} onChanged={() => void refreshStatus()} />
|
||||
<Page route={route} status={status} onStatusChange={() => void refreshStatus()} />
|
||||
@@ -170,6 +172,31 @@ function ConfirmBand({ route, onChanged }: { route: Route; onChanged: () => void
|
||||
)
|
||||
}
|
||||
|
||||
/**
|
||||
* Says, on every page, that nothing on screen came from a router.
|
||||
*
|
||||
* Only a DEV build can ever render this — the fixtures are not in a production
|
||||
* bundle (api.ts initMockBackend), so an operator cannot reach this state at all.
|
||||
* It is here for the person who CAN: a footer line reading "DEMO DATA" is easy to
|
||||
* work past for an afternoon and then screenshot into a bug report, and every
|
||||
* number above it is invented.
|
||||
*/
|
||||
function MockBanner() {
|
||||
if (!MOCK) return null
|
||||
return (
|
||||
<div className="mock-band" role="status">
|
||||
<span className="mock-band-tag">FIXTURES</span>
|
||||
<div className="mock-band-copy">
|
||||
<span className="mock-band-headline">No router is being read</span>
|
||||
<span className="mock-band-detail">
|
||||
Every reading on this page is invented by <code>src/mock.ts</code> for offline
|
||||
development. Drop <code>?mock</code> from the address to talk to a daemon.
|
||||
</span>
|
||||
</div>
|
||||
</div>
|
||||
)
|
||||
}
|
||||
|
||||
/**
|
||||
* The protection state, pinned under the nav on every page EXCEPT Overview
|
||||
* (which shows the same state as its own headline readout — see planeState.ts).
|
||||
@@ -195,12 +222,15 @@ function PlaneBanner({ status, route }: { status: Status | null; route: Route })
|
||||
|
||||
const criticals = (status.warnings ?? []).filter((w) => w.severity === 'critical').length
|
||||
const state = protectionState(status)
|
||||
// The published list is capped at 50, so with a note attached the count is a
|
||||
// floor. Say "at least" rather than quoting a total the daemon didn't send.
|
||||
const atLeast = truncationNote(status.warnings) ? 'At least ' : ''
|
||||
|
||||
// Wording comes from the shared source of truth so the banner and Overview can
|
||||
// never describe the same router differently.
|
||||
const headline = state.alarm
|
||||
? state.headline
|
||||
: `${criticals} protection ${criticals === 1 ? 'gap' : 'gaps'} from the last apply`
|
||||
: `${atLeast}${criticals} protection ${criticals === 1 ? 'gap' : 'gaps'} from the last apply`
|
||||
const detail = state.alarm
|
||||
? state.detail
|
||||
: 'Something you configured isn’t in effect. Review the findings before relying on it.'
|
||||
|
||||
+78
-27
@@ -13,8 +13,12 @@
|
||||
// serves in-memory fixtures instead of hitting the network, so `npm run dev`
|
||||
// and screenshot runs render without a live backend. A real backend in dev is
|
||||
// reachable instead via the Vite proxy in vite.config.ts (no flag ⇒ real fetch).
|
||||
//
|
||||
// THE FIXTURES ARE A DEV-BUILD-ONLY ARTEFACT — see initMockBackend below. They
|
||||
// used to be a plain static import, decided at RUNTIME off `location.search`, so
|
||||
// the invented router shipped inside the binary that goes on real hardware and a
|
||||
// link ending in `?dev` painted a healthy appliance without making one request.
|
||||
|
||||
import * as mock from './mock'
|
||||
import { armPendingConfirm, clearPendingConfirm, noteConfirmTimeout } from './pendingConfirm'
|
||||
|
||||
// --- error type -------------------------------------------------------------
|
||||
@@ -1120,12 +1124,59 @@ export interface Model {
|
||||
|
||||
// --- transport --------------------------------------------------------------
|
||||
|
||||
/** True when the URL asks for the offline fixture backend (?mock or ?dev). */
|
||||
export const MOCK: boolean = (() => {
|
||||
// --- the offline fixture backend (dev builds only) ---------------------------
|
||||
|
||||
/**
|
||||
* True when the in-memory fixtures are serving this session instead of the
|
||||
* daemon. ALWAYS false in a production build — see {@link initMockBackend}.
|
||||
*
|
||||
* A live binding, not a constant: it is decided once during boot, before the
|
||||
* first render, and every importer sees the same value for the whole session.
|
||||
*/
|
||||
export let MOCK = false
|
||||
|
||||
/** The loaded fixture module. `null` unless a dev build was asked for `?mock`. */
|
||||
let fixtures: typeof import('./mock') | null = null
|
||||
|
||||
/**
|
||||
* Load the fixture backend, if this build has one and the URL asks for it.
|
||||
* Call ONCE from the entry point and await it before the first render — the
|
||||
* pages read {@link MOCK} while they render, so flipping it afterwards would
|
||||
* leave a half-mocked screen.
|
||||
*
|
||||
* Two gates, and the order matters. `import.meta.env.DEV` is folded to a literal
|
||||
* `false` by Vite at build time, so in a production build the whole body is
|
||||
* unreachable, `import('./mock')` is tree-shaken out of the module graph, and the
|
||||
* fixtures are not in the emitted bundle AT ALL — not lazily, not behind a flag.
|
||||
* `vite.config.ts` fails the build if that ever stops being true.
|
||||
*
|
||||
* This is deliberately stronger than "hide the mock behind a query flag". The
|
||||
* flag was the bug: `?dev` on a production URL rendered an invented healthy
|
||||
* router — 119 of 122 nodes alive, "Protected" — with no request made and one
|
||||
* line of small print in the footer to say so. A person cannot audit a bundle;
|
||||
* the only honest guarantee is that the invented data is not in it.
|
||||
*/
|
||||
export async function initMockBackend(): Promise<boolean> {
|
||||
if (import.meta.env.DEV && mockRequested()) {
|
||||
fixtures = await import('./mock')
|
||||
MOCK = true
|
||||
}
|
||||
return MOCK
|
||||
}
|
||||
|
||||
/** Does the URL ask for the offline fixture backend (`?mock` or `?dev`)? */
|
||||
function mockRequested(): boolean {
|
||||
if (typeof location === 'undefined') return false
|
||||
const q = new URLSearchParams(location.search)
|
||||
return q.has('mock') || q.has('dev')
|
||||
})()
|
||||
}
|
||||
|
||||
/** The fixture backend, for the `MOCK ? … : …` branches below. Throws rather
|
||||
* than inventing data if it is ever reached without having been loaded. */
|
||||
function mock(): NonNullable<typeof fixtures> {
|
||||
if (!fixtures) throw new Error('mock backend not loaded — call initMockBackend() first')
|
||||
return fixtures
|
||||
}
|
||||
|
||||
/** A decoded response plus the raw Headers, for endpoints whose contract puts
|
||||
* pagination metadata outside the JSON body (see the stats log endpoints). */
|
||||
@@ -1174,11 +1225,11 @@ async function req<T>(path: string, init?: RequestInit): Promise<T> {
|
||||
// --- endpoints --------------------------------------------------------------
|
||||
|
||||
export function getStatus(): Promise<Status> {
|
||||
return MOCK ? mock.getStatus() : req<Status>('api/status')
|
||||
return MOCK ? mock().getStatus() : req<Status>('api/status')
|
||||
}
|
||||
|
||||
export async function getConfig(): Promise<Model> {
|
||||
const m = await (MOCK ? mock.getConfig() : req<Model>('api/config'))
|
||||
const m = await (MOCK ? mock().getConfig() : req<Model>('api/config'))
|
||||
// Every page reads the config, and the commit-confirm window's length is the
|
||||
// only thing needed to arm a countdown — so it is captured here once instead of
|
||||
// being threaded through eight pages. See pendingConfirm.ts.
|
||||
@@ -1188,7 +1239,7 @@ export async function getConfig(): Promise<Model> {
|
||||
|
||||
export function putConfig(m: Model): Promise<{ ok: boolean; applied: boolean }> {
|
||||
return MOCK
|
||||
? mock.putConfig(m)
|
||||
? mock().putConfig(m)
|
||||
: req('api/config', { method: 'PUT', body: JSON.stringify(m) })
|
||||
}
|
||||
|
||||
@@ -1203,25 +1254,25 @@ export function putConfig(m: Model): Promise<{ ok: boolean; applied: boolean }>
|
||||
* state. See pendingConfirm.ts.
|
||||
*/
|
||||
export async function apply(): Promise<ApplyResult> {
|
||||
const r = await (MOCK ? mock.apply() : req<ApplyResult>('api/apply', { method: 'POST' }))
|
||||
const r = await (MOCK ? mock().apply() : req<ApplyResult>('api/apply', { method: 'POST' }))
|
||||
if (!r.error && r.changed) armPendingConfirm()
|
||||
return r
|
||||
}
|
||||
|
||||
export async function confirm(): Promise<ApplyResult> {
|
||||
const r = await (MOCK ? mock.confirm() : req<ApplyResult>('api/confirm', { method: 'POST' }))
|
||||
const r = await (MOCK ? mock().confirm() : req<ApplyResult>('api/confirm', { method: 'POST' }))
|
||||
if (!r.error) clearPendingConfirm()
|
||||
return r
|
||||
}
|
||||
|
||||
export async function rollback(): Promise<ApplyResult> {
|
||||
const r = await (MOCK ? mock.rollback() : req<ApplyResult>('api/rollback', { method: 'POST' }))
|
||||
const r = await (MOCK ? mock().rollback() : req<ApplyResult>('api/rollback', { method: 'POST' }))
|
||||
if (!r.error) clearPendingConfirm()
|
||||
return r
|
||||
}
|
||||
|
||||
export function getStats(): Promise<Stats> {
|
||||
return MOCK ? mock.getStats() : req<Stats>('api/stats')
|
||||
return MOCK ? mock().getStats() : req<Stats>('api/stats')
|
||||
}
|
||||
|
||||
// --- daemon log download ------------------------------------------------------
|
||||
@@ -1273,7 +1324,7 @@ function saveBlob(blob: Blob, filename: string): void {
|
||||
*/
|
||||
export async function downloadLog(range: LogRange): Promise<void> {
|
||||
if (MOCK) {
|
||||
saveBlob(new Blob([mock.getLogText(range)], { type: 'text/plain' }), `shater-log-${range}.txt`)
|
||||
saveBlob(new Blob([mock().getLogText(range)], { type: 'text/plain' }), `shater-log-${range}.txt`)
|
||||
return
|
||||
}
|
||||
let res: Response
|
||||
@@ -1352,14 +1403,14 @@ function logPage<T>(env: { body: T[] | null; headers: Headers }): StatsLogPage<T
|
||||
/** GET /api/stats/log — one page of the DNS query log with its cursor metadata. */
|
||||
export function getStatsLogPage(q: StatsLogQuery = {}): Promise<StatsLogPage<QueryLogEntry>> {
|
||||
return MOCK
|
||||
? mock.getStatsLogPage(q)
|
||||
? mock().getStatsLogPage(q)
|
||||
: reqFull<QueryLogEntry[] | null>(`api/stats/log${statsLogQS(q)}`).then(logPage)
|
||||
}
|
||||
|
||||
/** GET /api/stats/conns — one page of the connection log with its cursor metadata. */
|
||||
export function getStatsConnsPage(q: StatsLogQuery = {}): Promise<StatsLogPage<ConnLogEntry>> {
|
||||
return MOCK
|
||||
? mock.getStatsConnsPage(q)
|
||||
? mock().getStatsConnsPage(q)
|
||||
: reqFull<ConnLogEntry[] | null>(`api/stats/conns${statsLogQS(q)}`).then(logPage)
|
||||
}
|
||||
|
||||
@@ -1367,13 +1418,13 @@ export function getStatsConnsPage(q: StatsLogQuery = {}): Promise<StatsLogPage<C
|
||||
* Rows only; callers that tail the stream want {@link getStatsLogPage} instead. */
|
||||
export function getStatsLog(q: number | StatsLogQuery = {}): Promise<QueryLogEntry[]> {
|
||||
const o: StatsLogQuery = typeof q === 'number' ? { limit: q } : q
|
||||
return MOCK ? mock.getStatsLog(o) : req<QueryLogEntry[]>(`api/stats/log${statsLogQS(o)}`)
|
||||
return MOCK ? mock().getStatsLog(o) : req<QueryLogEntry[]>(`api/stats/log${statsLogQS(o)}`)
|
||||
}
|
||||
|
||||
/** GET /api/stats/conns — the live connection-event log (device→dest), newest first. */
|
||||
export function getStatsConns(q: number | StatsLogQuery = {}): Promise<ConnLogEntry[]> {
|
||||
const o: StatsLogQuery = typeof q === 'number' ? { limit: q } : q
|
||||
return MOCK ? mock.getStatsConns(o) : req<ConnLogEntry[]>(`api/stats/conns${statsLogQS(o)}`)
|
||||
return MOCK ? mock().getStatsConns(o) : req<ConnLogEntry[]>(`api/stats/conns${statsLogQS(o)}`)
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -1428,12 +1479,12 @@ export interface RulesReachability {
|
||||
|
||||
/** GET /api/rules/reachability — which routing rules can never fire, and why. */
|
||||
export function getRulesReachability(): Promise<RulesReachability> {
|
||||
return MOCK ? mock.getRulesReachability() : req<RulesReachability>('api/rules/reachability')
|
||||
return MOCK ? mock().getRulesReachability() : req<RulesReachability>('api/rules/reachability')
|
||||
}
|
||||
|
||||
/** GET /api/ruleset/status — remote rule-set / blocklist freshness + rule counts. */
|
||||
export function getRulesetStatus(): Promise<RulesetStatus[]> {
|
||||
return MOCK ? mock.getRulesetStatus() : req<RulesetStatus[]>('api/ruleset/status')
|
||||
return MOCK ? mock().getRulesetStatus() : req<RulesetStatus[]>('api/ruleset/status')
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -1444,7 +1495,7 @@ export function getRulesetStatus(): Promise<RulesetStatus[]> {
|
||||
*/
|
||||
export function updateRuleset(tag: string): Promise<RulesetStatus | { ok: boolean }> {
|
||||
return MOCK
|
||||
? mock.updateRuleset(tag)
|
||||
? mock().updateRuleset(tag)
|
||||
: req('api/ruleset/update', { method: 'POST', body: JSON.stringify({ tag }) })
|
||||
}
|
||||
|
||||
@@ -1472,7 +1523,7 @@ export interface RulesetCheck {
|
||||
*/
|
||||
export function checkRulesetCategory(source: string, category: string): Promise<RulesetCheck> {
|
||||
return MOCK
|
||||
? mock.checkRulesetCategory(source, category)
|
||||
? mock().checkRulesetCategory(source, category)
|
||||
: req<RulesetCheck>('api/ruleset/check', {
|
||||
method: 'POST',
|
||||
body: JSON.stringify({ source, category }),
|
||||
@@ -1501,18 +1552,18 @@ export interface RulesetCategories {
|
||||
*/
|
||||
export function getRulesetCategories(source: string): Promise<RulesetCategories> {
|
||||
return MOCK
|
||||
? mock.getRulesetCategories(source)
|
||||
? mock().getRulesetCategories(source)
|
||||
: req<RulesetCategories>(`api/ruleset/categories?source=${encodeURIComponent(source)}`)
|
||||
}
|
||||
|
||||
/** GET /api/devices — discovered LAN clients merged with per-device config. */
|
||||
export function getDevices(): Promise<DiscoveredDevice[]> {
|
||||
return MOCK ? mock.getDevices() : req<DiscoveredDevice[]>('api/devices')
|
||||
return MOCK ? mock().getDevices() : req<DiscoveredDevice[]>('api/devices')
|
||||
}
|
||||
|
||||
/** GET /api/interfaces — the router's UCI network interfaces for the egress picker. */
|
||||
export function getInterfaces(): Promise<Interface[]> {
|
||||
return MOCK ? mock.getInterfaces() : req<Interface[]>('api/interfaces')
|
||||
return MOCK ? mock().getInterfaces() : req<Interface[]>('api/interfaces')
|
||||
}
|
||||
|
||||
/** POST /api/session — exchange a single-use handoff token for a session cookie. */
|
||||
@@ -1548,7 +1599,7 @@ export function importWg(conf: string): Promise<{ uri: string; name: string }> {
|
||||
*/
|
||||
export function updateSubscription(name: string): Promise<{ added: number }> {
|
||||
return MOCK
|
||||
? mock.updateSubscription(name)
|
||||
? mock().updateSubscription(name)
|
||||
: req('api/subscription/update', { method: 'POST', body: JSON.stringify({ name }) })
|
||||
}
|
||||
|
||||
@@ -1568,7 +1619,7 @@ export function updateSubscription(name: string): Promise<{ added: number }> {
|
||||
export function getGroupsHealth(
|
||||
opts: { group?: string; members?: boolean } = {},
|
||||
): Promise<GroupsHealth> {
|
||||
if (MOCK) return mock.getGroupsHealth(opts)
|
||||
if (MOCK) return mock().getGroupsHealth(opts)
|
||||
const p = new URLSearchParams()
|
||||
if (opts.group) p.set('group', opts.group)
|
||||
if (opts.members) p.set('members', '1')
|
||||
@@ -1683,11 +1734,11 @@ export interface GroupTestStart {
|
||||
*/
|
||||
export function postGroupsTest(name = ''): Promise<GroupTestStart> {
|
||||
return MOCK
|
||||
? mock.postGroupsTest(name)
|
||||
? mock().postGroupsTest(name)
|
||||
: req<GroupTestStart>('api/groups/test', { method: 'POST', body: JSON.stringify({ name }) })
|
||||
}
|
||||
|
||||
/** GET /api/groups/test — progress + results of the current/last group test. */
|
||||
export function getGroupsTest(): Promise<GroupTestStatus> {
|
||||
return MOCK ? mock.getGroupsTest() : req<GroupTestStatus>('api/groups/test')
|
||||
return MOCK ? mock().getGroupsTest() : req<GroupTestStatus>('api/groups/test')
|
||||
}
|
||||
|
||||
@@ -0,0 +1,127 @@
|
||||
// findings.ts — which apply-time finding is shown where.
|
||||
//
|
||||
// Run with `npm test` (node's built-in test runner + native TypeScript
|
||||
// stripping; no test dependency is added to the SPA, which ships inside the
|
||||
// daemon binary).
|
||||
//
|
||||
// Two defects are pinned here.
|
||||
//
|
||||
// 1. THE TRUNCATION NOTE WAS UNREACHABLE. The daemon caps Status.warnings at 50
|
||||
// and overwrites the last slot with an `info` note counting what it dropped.
|
||||
// Overview filtered `info` away wholesale, and the settings-page route keys on
|
||||
// a section (`generate`) that no page owns — so the single line telling the
|
||||
// operator "you are not seeing all of it" reached no screen at all.
|
||||
//
|
||||
// 2. FINDINGS ABOUT AN ENTITY NEVER REACHED THAT ENTITY'S PAGE. The generator
|
||||
// drops a node it cannot build and names it; the Nodes page rendered that node
|
||||
// as an ordinary row with a green toggle, because it never read the findings.
|
||||
|
||||
import { test } from 'node:test'
|
||||
import assert from 'node:assert/strict'
|
||||
|
||||
import {
|
||||
attentionFindings,
|
||||
entityFindings,
|
||||
findingsByName,
|
||||
sectionNotes,
|
||||
truncationNote,
|
||||
worstSeverity,
|
||||
} from './findings.ts'
|
||||
import type { StatusWarning } from './api.ts'
|
||||
|
||||
const crit = (section: string, name: string, message = 'broken'): StatusWarning => ({
|
||||
severity: 'critical',
|
||||
section,
|
||||
name,
|
||||
message,
|
||||
})
|
||||
const warn = (section: string, name: string, message = 'degraded'): StatusWarning => ({
|
||||
severity: 'warning',
|
||||
section,
|
||||
name,
|
||||
message,
|
||||
})
|
||||
const info = (section: string, name: string, message: string): StatusWarning => ({
|
||||
severity: 'info',
|
||||
section,
|
||||
name,
|
||||
message,
|
||||
})
|
||||
|
||||
/** Verbatim from apply/warnings.go finalizeWarnings. */
|
||||
const SUPPRESSED = info(
|
||||
'generate',
|
||||
'',
|
||||
'7 further warning(s) suppressed; run `logread -e shater` for the full list',
|
||||
)
|
||||
|
||||
// --- the truncation note ----------------------------------------------------
|
||||
|
||||
test('the truncation note is found, whatever else is in the list', () => {
|
||||
const note = truncationNote([crit('rule', 'a'), warn('node', 'b'), SUPPRESSED])
|
||||
assert.notEqual(note, null)
|
||||
assert.match(note!.message, /7 further warning/)
|
||||
})
|
||||
|
||||
test('a whole list has no truncation note', () => {
|
||||
assert.equal(truncationNote([crit('rule', 'a'), warn('node', 'b')]), null)
|
||||
assert.equal(truncationNote([]), null)
|
||||
assert.equal(truncationNote(undefined), null)
|
||||
})
|
||||
|
||||
test('an ordinary info note is not mistaken for the truncation note', () => {
|
||||
const notes = [info('untunnelable', 'block', 'Ping and traceroute do not work…')]
|
||||
assert.equal(truncationNote(notes), null)
|
||||
})
|
||||
|
||||
test('the truncation note is kept out of the settings-page notes it would pollute', () => {
|
||||
const all = [info('generate', '', 'cache: moved to /overlay'), SUPPRESSED]
|
||||
const notes = sectionNotes(all, 'generate')
|
||||
assert.equal(notes.length, 1)
|
||||
assert.match(notes[0].message, /cache:/)
|
||||
})
|
||||
|
||||
test('the attention list still carries only critical and warning', () => {
|
||||
const all = [crit('rule', 'a'), warn('node', 'b'), info('untunnelable', 'block', 'x'), SUPPRESSED]
|
||||
const attention = attentionFindings(all)
|
||||
assert.equal(attention.length, 2)
|
||||
assert.ok(attention.every((w) => w.severity !== 'info'))
|
||||
})
|
||||
|
||||
// --- per-entity findings ----------------------------------------------------
|
||||
|
||||
test('a page takes only the sections it owns', () => {
|
||||
const all = [
|
||||
crit('node', 'tokyo-01', 'parse share-link: bad scheme (skipped)'),
|
||||
warn('subscription', 'qomar', 'fetch failed'),
|
||||
crit('rule', 'default', 'never applies'),
|
||||
info('generate', '', 'cache: x'),
|
||||
]
|
||||
const mine = entityFindings(all, ['node', 'subscription'])
|
||||
assert.deepEqual(
|
||||
mine.map((w) => w.name),
|
||||
['tokyo-01', 'qomar'],
|
||||
)
|
||||
})
|
||||
|
||||
test('entity findings never include info notes', () => {
|
||||
const all = [info('node', 'tokyo-01', 'just a note'), SUPPRESSED]
|
||||
assert.equal(entityFindings(all, ['node', 'generate']).length, 0)
|
||||
})
|
||||
|
||||
test('findings index by name, and global (unnamed) ones are left out', () => {
|
||||
const all = [
|
||||
crit('node', 'tokyo-01', 'first'),
|
||||
warn('node', 'tokyo-01', 'second'),
|
||||
crit('node', '', 'global to the section'),
|
||||
]
|
||||
const byName = findingsByName(entityFindings(all, ['node']))
|
||||
assert.equal(byName.size, 1)
|
||||
assert.equal(byName.get('tokyo-01')!.length, 2)
|
||||
})
|
||||
|
||||
test('one lamp per row takes the loudest severity', () => {
|
||||
assert.equal(worstSeverity([warn('node', 'a'), crit('node', 'a')]), 'critical')
|
||||
assert.equal(worstSeverity([warn('node', 'a')]), 'warning')
|
||||
assert.equal(worstSeverity([]), null)
|
||||
})
|
||||
+91
-2
@@ -6,7 +6,8 @@
|
||||
//
|
||||
// critical / warning — something needs attention: a protection promise is
|
||||
// broken, or something you configured isn't in effect. These belong on
|
||||
// Overview, where the operator looks first.
|
||||
// Overview, where the operator looks first — and, when they name an entity,
|
||||
// ALSO on the page that owns that entity (see `entityFindings`).
|
||||
//
|
||||
// info — a statement ABOUT the configuration, not a problem. It never clears,
|
||||
// because nothing is wrong: it is simply describing a choice that was made.
|
||||
@@ -16,9 +17,49 @@
|
||||
// page that never goes away and never asks for anything trains people to skim
|
||||
// the list — which is exactly how a real critical finding gets missed. Anything
|
||||
// standing in the findings list should be something you could act on.
|
||||
//
|
||||
// The one exception is carved out below: the daemon's own note that it dropped
|
||||
// findings to fit the cap. It is `info` by severity and unactionable by nature,
|
||||
// and it is the single most important line in the list, because it is the list
|
||||
// telling you it is not the whole list.
|
||||
|
||||
import type { StatusWarning } from './api'
|
||||
|
||||
/**
|
||||
* The daemon's truncation disclosure, verbatim from apply/warnings.go
|
||||
* finalizeWarnings:
|
||||
*
|
||||
* "%d further warning(s) suppressed; run `logread -e shater` for the full list"
|
||||
*
|
||||
* Matched on the stable clause rather than the whole sentence so a reworded tail
|
||||
* still registers. If this ever stops matching, the failure mode is a list that
|
||||
* silently claims to be complete — which is why `truncationNote` is tested.
|
||||
*/
|
||||
const SUPPRESSED_RE = /further warning\(s\) suppressed/
|
||||
|
||||
/**
|
||||
* The daemon's "this list is incomplete" note, or null when the list is whole.
|
||||
*
|
||||
* Status.warnings is capped at 50, sorted critical-first, and the last slot is
|
||||
* REPLACED by an `info` note counting what was dropped. That note therefore
|
||||
* arrives on the one channel the panel filtered away wholesale: `info` never
|
||||
* reached Overview, and the settings-page route (`sectionNotes`) keys on
|
||||
* section `generate`, which no page owns. So the single line saying "there are
|
||||
* findings you are not being shown" was the only one guaranteed to be invisible.
|
||||
*
|
||||
* Callers must render this WITH the attention list, not instead of it.
|
||||
*/
|
||||
export function truncationNote(warnings: StatusWarning[] | undefined): StatusWarning | null {
|
||||
return (
|
||||
(warnings ?? []).find((w) => w.severity === 'info' && SUPPRESSED_RE.test(w.message)) ?? null
|
||||
)
|
||||
}
|
||||
|
||||
/** Is this the truncation disclosure rather than an ordinary note? */
|
||||
function isTruncationNote(w: StatusWarning): boolean {
|
||||
return w.severity === 'info' && SUPPRESSED_RE.test(w.message)
|
||||
}
|
||||
|
||||
/** Findings that need attention — the Overview list. Info notes are excluded. */
|
||||
export function attentionFindings(warnings: StatusWarning[] | undefined): StatusWarning[] {
|
||||
return (warnings ?? []).filter((w) => w.severity === 'critical' || w.severity === 'warning')
|
||||
@@ -29,10 +70,58 @@ export function attentionFindings(warnings: StatusWarning[] | undefined): Status
|
||||
* (e.g. `untunnelable` → the Networks page's "Other traffic" section). Only info:
|
||||
* a critical/warning is an attention item and stays on Overview, so it can't be
|
||||
* quietly buried on a settings page instead.
|
||||
*
|
||||
* The truncation note is excluded: it is about the LIST, not about any section,
|
||||
* and it has its own home beside the list ({@link truncationNote}).
|
||||
*/
|
||||
export function sectionNotes(
|
||||
warnings: StatusWarning[] | undefined,
|
||||
section: string,
|
||||
): StatusWarning[] {
|
||||
return (warnings ?? []).filter((w) => w.severity === 'info' && w.section === section)
|
||||
return (warnings ?? []).filter(
|
||||
(w) => w.severity === 'info' && w.section === section && !isTruncationNote(w),
|
||||
)
|
||||
}
|
||||
|
||||
/**
|
||||
* The attention findings about entities ONE page owns — for that page to show
|
||||
* beside the entities themselves.
|
||||
*
|
||||
* Overview is where you look when you already suspect something; a page like
|
||||
* Nodes is where you look when you don't. The generator drops a node it cannot
|
||||
* build — an unparseable share link, a WireGuard key materialised twice — and
|
||||
* says so by name ("node \"x\": parse share-link: … (skipped)"), yet that node
|
||||
* kept rendering as an ordinary row with a green toggle, because the page never
|
||||
* read the findings at all. The switch says on; the engine has no such outbound.
|
||||
*
|
||||
* This does NOT move anything off Overview: the same finding appears in both
|
||||
* places, which is correct — one list is "what is wrong with this router", the
|
||||
* other is "what is wrong with this node".
|
||||
*/
|
||||
export function entityFindings(
|
||||
warnings: StatusWarning[] | undefined,
|
||||
sections: readonly string[],
|
||||
): StatusWarning[] {
|
||||
const want = new Set(sections)
|
||||
return attentionFindings(warnings).filter((w) => want.has(w.section))
|
||||
}
|
||||
|
||||
/** Index attention findings by entity name, for badging a row directly. Entries
|
||||
* with an empty `name` are global to their section and are left out. */
|
||||
export function findingsByName(findings: StatusWarning[]): Map<string, StatusWarning[]> {
|
||||
const out = new Map<string, StatusWarning[]>()
|
||||
for (const f of findings) {
|
||||
if (!f.name) continue
|
||||
const list = out.get(f.name)
|
||||
if (list) list.push(f)
|
||||
else out.set(f.name, [f])
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
/** The loudest severity in a set — for a row badge that has room for one lamp. */
|
||||
export function worstSeverity(findings: StatusWarning[]): 'critical' | 'warning' | null {
|
||||
if (findings.some((f) => f.severity === 'critical')) return 'critical'
|
||||
if (findings.length > 0) return 'warning'
|
||||
return null
|
||||
}
|
||||
|
||||
+25
-9
@@ -2,17 +2,33 @@ import { StrictMode } from 'react'
|
||||
import { createRoot } from 'react-dom/client'
|
||||
import './tokens.css'
|
||||
import { App } from './App'
|
||||
import { initMockBackend } from './api'
|
||||
import { ConfirmProvider } from './components'
|
||||
|
||||
const rootEl = document.getElementById('root')
|
||||
if (!rootEl) throw new Error('#root not found')
|
||||
|
||||
// ConfirmProvider sits ABOVE <App> so it survives App's early returns (the
|
||||
// unauth / no-link plates) — useConfirm() can never find itself without a host.
|
||||
createRoot(rootEl).render(
|
||||
<StrictMode>
|
||||
<ConfirmProvider>
|
||||
<App />
|
||||
</ConfirmProvider>
|
||||
</StrictMode>,
|
||||
)
|
||||
// Settle the fixture question BEFORE the first render: pages read `MOCK` while
|
||||
// they render, so a backend that arrives afterwards would paint half a screen
|
||||
// from the daemon and half from fixtures. In a production build this resolves
|
||||
// immediately and to `false` — the fixtures are not in the bundle to load (see
|
||||
// api.ts initMockBackend and the assertNoMockFixtures plugin in vite.config.ts).
|
||||
function mount() {
|
||||
// ConfirmProvider sits ABOVE <App> so it survives App's early returns (the
|
||||
// unauth / no-link plates) — useConfirm() can never find itself without a host.
|
||||
createRoot(rootEl!).render(
|
||||
<StrictMode>
|
||||
<ConfirmProvider>
|
||||
<App />
|
||||
</ConfirmProvider>
|
||||
</StrictMode>,
|
||||
)
|
||||
}
|
||||
|
||||
// A fixture module that fails to load is a broken dev checkout, not a reason to
|
||||
// hand the operator a blank plate — mount anyway and let the shell report that it
|
||||
// cannot reach a daemon, which by then is the truth.
|
||||
void initMockBackend().then(mount, (e) => {
|
||||
console.error('mock backend failed to load; continuing against the real API', e)
|
||||
mount()
|
||||
})
|
||||
|
||||
+52
-4
@@ -500,7 +500,15 @@ export async function getRulesetCategories(source: string): Promise<RulesetCateg
|
||||
// the field case the readout used to call "Protected" (one
|
||||
// rule, `default → direct`); `unknown` is a daemon too old to
|
||||
// report. Default: tunnel.
|
||||
function mockPlane(): { plane: 'full' | 'hold' | 'none'; engine: boolean; killSwitch: string } {
|
||||
// ?mock&plane=unreported → a daemon that sends NO `plane` field. The panel then
|
||||
// knows nothing about what is installed, which is the state
|
||||
// the Kill-switch module used to render as a green "ARMED"
|
||||
// (`undefined !== 'none'` is true).
|
||||
function mockPlane(): {
|
||||
plane: 'full' | 'hold' | 'none' | undefined
|
||||
engine: boolean
|
||||
killSwitch: string
|
||||
} {
|
||||
const q = typeof location === 'undefined' ? '' : location.search
|
||||
const params = new URLSearchParams(q)
|
||||
const killSwitch = params.get('ks') === 'open' ? 'open' : 'closed'
|
||||
@@ -508,6 +516,7 @@ function mockPlane(): { plane: 'full' | 'hold' | 'none'; engine: boolean; killSw
|
||||
if (p === 'hold') return { plane: 'hold', engine: false, killSwitch: 'closed' }
|
||||
if (p === 'none') return { plane: 'none', engine: false, killSwitch }
|
||||
if (p === 'open') return { plane: 'none', engine: false, killSwitch: 'open' }
|
||||
if (p === 'unreported') return { plane: undefined, engine: true, killSwitch }
|
||||
return { plane: 'full', engine: true, killSwitch }
|
||||
}
|
||||
|
||||
@@ -515,7 +524,7 @@ function mockPlane(): { plane: 'full' | 'hold' | 'none'; engine: boolean; killSw
|
||||
// meaningful with the plane installed: with the engine down there is no running
|
||||
// config to judge, and the daemon reports the unknown/zero value — so do the same
|
||||
// here rather than leaving a stale "tunnel" behind a dead engine.
|
||||
function mockTraffic(plane: 'full' | 'hold' | 'none'): Traffic | undefined {
|
||||
function mockTraffic(plane: 'full' | 'hold' | 'none' | undefined): Traffic | undefined {
|
||||
if (plane !== 'full') return { verdict: '', default: '', tunnel_rules: 0 }
|
||||
const params = new URLSearchParams(typeof location === 'undefined' ? '' : location.search)
|
||||
switch (params.get('traffic')) {
|
||||
@@ -560,6 +569,29 @@ const MOCK_WARNINGS: StatusWarning[] = [
|
||||
name: 'fakeip-pool',
|
||||
message: 'fake-IP resolver cannot be used as a fallback; the failover chain was not built',
|
||||
},
|
||||
// Two findings the generator attributes to a NODE by name — the class that the
|
||||
// Nodes page never showed, leaving a node the engine threw away rendered as an
|
||||
// ordinary row with a green toggle. Both name real fixture nodes so the row
|
||||
// badge, the collapsed-bucket "N flagged" count and the per-row strip all fire.
|
||||
{
|
||||
severity: 'warning',
|
||||
section: 'node',
|
||||
name: 'fi-trojan',
|
||||
message: 'parse share-link: unsupported scheme "trojan+ws" (skipped)',
|
||||
},
|
||||
{
|
||||
severity: 'warning',
|
||||
section: 'node',
|
||||
name: 'home-wg',
|
||||
message:
|
||||
'this WireGuard node is materialised twice in the engine config — as "home-wg" and as "group-stealth-m1-home-wg" — and traffic can reach both. A WireGuard peer keeps ONE session per public key, so two devices built from one private key evict each other continuously and NEITHER tunnel passes traffic. Only "home-wg" is kept; everything that routed through "group-stealth-m1-home-wg" is fail-closed (blocked) instead of leaving over the plain WAN',
|
||||
},
|
||||
{
|
||||
severity: 'warning',
|
||||
section: 'subscription',
|
||||
name: 'backup',
|
||||
message: 'fetch failed: dial tcp 203.0.113.9:443: i/o timeout — serving the nodes cached earlier',
|
||||
},
|
||||
{
|
||||
severity: 'info',
|
||||
section: 'generate',
|
||||
@@ -568,6 +600,19 @@ const MOCK_WARNINGS: StatusWarning[] = [
|
||||
},
|
||||
]
|
||||
|
||||
/**
|
||||
* The daemon's truncation disclosure, exactly as apply/warnings.go writes it when
|
||||
* the published set overflows the 50-entry cap. Served under `?mock&trunc` so the
|
||||
* "this list is incomplete" rendering is exercisable — it used to be dropped
|
||||
* wholesale by the panel's `info` filter and reached no screen at all.
|
||||
*/
|
||||
const MOCK_TRUNCATION: StatusWarning = {
|
||||
severity: 'info',
|
||||
section: 'generate',
|
||||
name: '',
|
||||
message: '7 further warning(s) suppressed; run `logread -e shater` for the full list',
|
||||
}
|
||||
|
||||
/**
|
||||
* The standing `untunnelable` note the daemon reports. It is INFO, never a
|
||||
* problem: it states a correct, chosen configuration. Two shapes, mirroring the
|
||||
@@ -614,9 +659,12 @@ function mockWarnings(killSwitch: string): StatusWarning[] {
|
||||
const params = new URLSearchParams(q)
|
||||
const mode = (CONFIG.Globals as { Untunnelable?: string }).Untunnelable ?? 'block'
|
||||
const notes = untunnelableNote(mode, killSwitch)
|
||||
// `?trunc` adds the daemon's "the published list is capped" disclosure, which
|
||||
// it appends IN PLACE OF the last entry it had room for.
|
||||
const trunc = params.has('trunc') ? [{ ...MOCK_TRUNCATION }] : []
|
||||
// A degraded plane always comes with the findings that explain it.
|
||||
if (params.has('warn') || params.get('plane')) {
|
||||
return [...MOCK_WARNINGS.map((w) => ({ ...w })), ...notes]
|
||||
if (params.has('warn') || params.get('plane') || trunc.length > 0) {
|
||||
return [...MOCK_WARNINGS.map((w) => ({ ...w })), ...notes, ...trunc]
|
||||
}
|
||||
return notes
|
||||
}
|
||||
|
||||
@@ -0,0 +1,443 @@
|
||||
/* Alerts section (rendered on Settings) — inherits the Faceplate tokens and the
|
||||
* shared page chrome from App.css (.toast, .mono). Every rule below is a
|
||||
* one-to-one copy of the DNS.css rule the markup used before the section moved
|
||||
* here, renamed `dns-*` → `alr-*` so nothing collides. Orange stays an accent. */
|
||||
|
||||
/* ---- section shell (matches the Settings group plates one-to-one) ---- */
|
||||
.alr-section {
|
||||
margin-top: calc(var(--u, 8px) * 3.5);
|
||||
}
|
||||
.alr-sec-hd {
|
||||
display: flex;
|
||||
align-items: baseline;
|
||||
gap: 12px;
|
||||
padding-bottom: 10px;
|
||||
border-bottom: 1px solid var(--groove);
|
||||
}
|
||||
.alr-sec-title {
|
||||
margin: 0;
|
||||
font-family: var(--font-mono);
|
||||
font-size: 13px;
|
||||
font-weight: 700;
|
||||
letter-spacing: var(--track-label, 0.18em);
|
||||
text-transform: uppercase;
|
||||
color: var(--dim);
|
||||
}
|
||||
.alr-sec-count {
|
||||
font-size: 11px;
|
||||
letter-spacing: 0.06em;
|
||||
color: var(--faint);
|
||||
}
|
||||
.alr-sec-note {
|
||||
margin: 10px 2px 0;
|
||||
font-family: var(--font-sans);
|
||||
font-size: 12.5px;
|
||||
line-height: 1.55;
|
||||
color: var(--dim);
|
||||
max-width: 56ch;
|
||||
}
|
||||
|
||||
/* ---- add form ---- */
|
||||
.alr-add {
|
||||
display: flex;
|
||||
flex-direction: column;
|
||||
gap: 10px;
|
||||
margin-top: calc(var(--u, 8px) * 2);
|
||||
}
|
||||
.alr-add-top {
|
||||
display: flex;
|
||||
flex-wrap: wrap;
|
||||
align-items: center;
|
||||
gap: 10px;
|
||||
}
|
||||
.alr-input {
|
||||
min-width: 0;
|
||||
padding: 9px 12px;
|
||||
border: 1px solid var(--groove);
|
||||
border-radius: 7px;
|
||||
background: var(--sink);
|
||||
color: var(--ink);
|
||||
font-family: var(--font-mono);
|
||||
font-size: 12.5px;
|
||||
letter-spacing: 0.02em;
|
||||
box-shadow: 0 1px 2px var(--shadow) inset;
|
||||
transition: border-color 0.15s, box-shadow 0.15s;
|
||||
}
|
||||
.alr-input::placeholder {
|
||||
color: var(--faint);
|
||||
}
|
||||
.alr-input:focus-visible {
|
||||
border-color: var(--accent);
|
||||
outline: 2px solid var(--accent);
|
||||
outline-offset: 1px;
|
||||
}
|
||||
.alr-input:disabled {
|
||||
opacity: 0.55;
|
||||
}
|
||||
.alr-input--name {
|
||||
flex: 0 1 14rem;
|
||||
}
|
||||
|
||||
/* segmented type picker */
|
||||
.alr-seg {
|
||||
display: inline-flex;
|
||||
border: 1px solid var(--groove);
|
||||
border-radius: 7px;
|
||||
overflow: hidden;
|
||||
background: var(--sink);
|
||||
}
|
||||
.alr-seg-btn {
|
||||
padding: 8px 14px;
|
||||
border: 0;
|
||||
background: transparent;
|
||||
color: var(--dim);
|
||||
font-family: var(--font-mono);
|
||||
font-size: 11px;
|
||||
letter-spacing: 0.08em;
|
||||
text-transform: uppercase;
|
||||
cursor: pointer;
|
||||
transition: background 0.15s, color 0.15s;
|
||||
}
|
||||
.alr-seg-btn + .alr-seg-btn {
|
||||
border-left: 1px solid var(--groove);
|
||||
}
|
||||
.alr-seg-btn.on {
|
||||
background: var(--accent);
|
||||
color: #fff;
|
||||
}
|
||||
.alr-seg-btn:focus-visible {
|
||||
outline: 2px solid var(--accent);
|
||||
outline-offset: -2px;
|
||||
}
|
||||
|
||||
.alr-resp {
|
||||
display: inline-flex;
|
||||
align-items: center;
|
||||
gap: 8px;
|
||||
}
|
||||
.alr-resp-label {
|
||||
font-size: 10px;
|
||||
letter-spacing: var(--track-label, 0.18em);
|
||||
text-transform: uppercase;
|
||||
color: var(--faint);
|
||||
}
|
||||
.alr-select {
|
||||
padding: 8px 10px;
|
||||
border: 1px solid var(--groove);
|
||||
border-radius: 7px;
|
||||
background: var(--sink);
|
||||
color: var(--ink);
|
||||
font-family: var(--font-mono);
|
||||
font-size: 11.5px;
|
||||
letter-spacing: 0.04em;
|
||||
cursor: pointer;
|
||||
}
|
||||
.alr-select:focus-visible {
|
||||
border-color: var(--accent);
|
||||
outline: 2px solid var(--accent);
|
||||
outline-offset: 1px;
|
||||
}
|
||||
|
||||
.alr-add-actions {
|
||||
display: flex;
|
||||
align-items: center;
|
||||
justify-content: flex-end;
|
||||
gap: 14px;
|
||||
flex-wrap: wrap;
|
||||
}
|
||||
.alr-field-err {
|
||||
flex: 1;
|
||||
min-width: 0;
|
||||
margin: 0;
|
||||
font-family: var(--font-mono);
|
||||
font-size: 11.5px;
|
||||
line-height: 1.5;
|
||||
color: var(--crit);
|
||||
}
|
||||
|
||||
/* ---- rows ---- */
|
||||
.alr-rows {
|
||||
list-style: none;
|
||||
margin: calc(var(--u, 8px) * 2) 0 0;
|
||||
padding: 0;
|
||||
display: flex;
|
||||
flex-direction: column;
|
||||
gap: 8px;
|
||||
}
|
||||
.alr-row {
|
||||
display: flex;
|
||||
align-items: center;
|
||||
gap: calc(var(--u, 8px) * 1.5);
|
||||
padding: 12px 14px;
|
||||
border: 1px solid var(--groove);
|
||||
border-radius: 8px;
|
||||
background: linear-gradient(
|
||||
180deg,
|
||||
var(--raised),
|
||||
color-mix(in srgb, var(--raised) 82%, var(--panel))
|
||||
);
|
||||
box-shadow: 0 1px 0 var(--edge) inset;
|
||||
}
|
||||
.alr-row-main {
|
||||
flex: 1;
|
||||
min-width: 0;
|
||||
display: flex;
|
||||
flex-direction: column;
|
||||
gap: 4px;
|
||||
}
|
||||
.alr-row-l1 {
|
||||
display: flex;
|
||||
align-items: center;
|
||||
gap: 8px;
|
||||
flex-wrap: wrap;
|
||||
}
|
||||
.alr-row-name {
|
||||
font-family: var(--font-mono);
|
||||
font-size: 13px;
|
||||
font-weight: 600;
|
||||
letter-spacing: 0.01em;
|
||||
color: var(--ink);
|
||||
overflow: hidden;
|
||||
text-overflow: ellipsis;
|
||||
white-space: nowrap;
|
||||
max-width: 24ch;
|
||||
}
|
||||
.alr-row-l2 {
|
||||
display: flex;
|
||||
align-items: center;
|
||||
gap: 10px;
|
||||
flex-wrap: wrap;
|
||||
font-size: 11.5px;
|
||||
letter-spacing: 0.02em;
|
||||
}
|
||||
.alr-row-detail {
|
||||
color: var(--dim);
|
||||
overflow: hidden;
|
||||
text-overflow: ellipsis;
|
||||
white-space: nowrap;
|
||||
max-width: 40ch;
|
||||
}
|
||||
|
||||
/* badge — groove-bordered, not orange (accent stays reserved) */
|
||||
.alr-badge {
|
||||
display: inline-block;
|
||||
padding: 2px 7px;
|
||||
border: 1px solid var(--groove);
|
||||
border-radius: 5px;
|
||||
background: color-mix(in srgb, var(--sink) 60%, transparent);
|
||||
font-family: var(--font-mono);
|
||||
font-size: 10px;
|
||||
font-weight: 600;
|
||||
letter-spacing: 0.1em;
|
||||
text-transform: uppercase;
|
||||
color: var(--dim);
|
||||
white-space: nowrap;
|
||||
}
|
||||
.alr-badge--accent {
|
||||
border-color: color-mix(in srgb, var(--accent) 55%, var(--groove));
|
||||
color: var(--accent);
|
||||
}
|
||||
|
||||
.alr-masked {
|
||||
font-family: var(--font-mono);
|
||||
font-size: 10px;
|
||||
letter-spacing: 0.08em;
|
||||
color: var(--faint);
|
||||
text-transform: uppercase;
|
||||
cursor: help;
|
||||
}
|
||||
|
||||
.alr-del {
|
||||
flex: none;
|
||||
padding: 6px 12px;
|
||||
font-size: 10.5px;
|
||||
}
|
||||
|
||||
/* ---- empty plate ---- */
|
||||
.alr-empty {
|
||||
margin-top: calc(var(--u, 8px) * 2);
|
||||
padding: calc(var(--u, 8px) * 3);
|
||||
border: 1px dashed var(--groove);
|
||||
border-radius: 9px;
|
||||
background: color-mix(in srgb, var(--raised) 55%, transparent);
|
||||
text-align: center;
|
||||
}
|
||||
.alr-empty-title {
|
||||
display: block;
|
||||
font-size: 13px;
|
||||
font-weight: 700;
|
||||
letter-spacing: 0.06em;
|
||||
color: var(--dim);
|
||||
}
|
||||
.alr-empty-body {
|
||||
margin: 8px auto 0;
|
||||
max-width: 48ch;
|
||||
font-family: var(--font-sans);
|
||||
font-size: 13px;
|
||||
line-height: 1.55;
|
||||
color: var(--dim);
|
||||
}
|
||||
|
||||
/* ---- loading skeleton ---- */
|
||||
.alr-skel {
|
||||
height: 62px;
|
||||
border: 1px solid var(--groove);
|
||||
border-radius: 8px;
|
||||
background: linear-gradient(90deg, var(--raised), var(--sink), var(--raised));
|
||||
background-size: 200% 100%;
|
||||
animation: alr-skel-shift 1.4s ease-in-out infinite;
|
||||
}
|
||||
@keyframes alr-skel-shift {
|
||||
from {
|
||||
background-position: 200% 0;
|
||||
}
|
||||
to {
|
||||
background-position: -200% 0;
|
||||
}
|
||||
}
|
||||
|
||||
/* the per-row delivery picker sits inline in the row */
|
||||
.alr-detour {
|
||||
flex: none;
|
||||
display: flex;
|
||||
flex-direction: column;
|
||||
gap: 5px;
|
||||
min-width: 0;
|
||||
}
|
||||
.alr-detour-label {
|
||||
font-size: 10px;
|
||||
letter-spacing: var(--track-label, 0.18em);
|
||||
text-transform: uppercase;
|
||||
color: var(--faint);
|
||||
}
|
||||
.alr-detour-select {
|
||||
max-width: 22rem;
|
||||
}
|
||||
|
||||
/* current delivery-path readout on the row */
|
||||
.alr-path {
|
||||
color: var(--faint);
|
||||
white-space: nowrap;
|
||||
}
|
||||
.alr-path[data-active='on'] {
|
||||
color: var(--dim);
|
||||
}
|
||||
.alr-path-name {
|
||||
color: var(--led-on);
|
||||
font-weight: 600;
|
||||
}
|
||||
.alr-path[data-missing='y'] .alr-path-name {
|
||||
color: var(--amber);
|
||||
}
|
||||
.alr-path-flag {
|
||||
color: var(--amber);
|
||||
}
|
||||
|
||||
/* alert delivery: deliver-via picker + fallback toggle + caution note */
|
||||
.alr-delivery {
|
||||
display: flex;
|
||||
flex-wrap: wrap;
|
||||
align-items: center;
|
||||
gap: 10px 20px;
|
||||
}
|
||||
.alr-fallback {
|
||||
display: inline-flex;
|
||||
align-items: center;
|
||||
gap: 8px;
|
||||
cursor: pointer;
|
||||
}
|
||||
.alr-fallback-label {
|
||||
font-size: 10px;
|
||||
letter-spacing: var(--track-label, 0.18em);
|
||||
text-transform: uppercase;
|
||||
color: var(--faint);
|
||||
}
|
||||
/* the per-row delivery controls sit inline in the row (like .alr-detour) */
|
||||
.alr-ctl {
|
||||
flex: none;
|
||||
display: flex;
|
||||
flex-direction: column;
|
||||
gap: 8px;
|
||||
min-width: 0;
|
||||
}
|
||||
.alr-note {
|
||||
margin: 0;
|
||||
font-family: var(--font-sans);
|
||||
font-size: 11.5px;
|
||||
line-height: 1.5;
|
||||
color: var(--amber);
|
||||
max-width: 56ch;
|
||||
}
|
||||
.alr-note--row {
|
||||
margin-top: 2px;
|
||||
}
|
||||
|
||||
/* alert event checkboxes */
|
||||
.alr-events {
|
||||
display: flex;
|
||||
flex-wrap: wrap;
|
||||
gap: 8px 16px;
|
||||
margin: 0;
|
||||
padding: 0;
|
||||
border: 0;
|
||||
}
|
||||
.alr-event {
|
||||
display: inline-flex;
|
||||
align-items: center;
|
||||
gap: 6px;
|
||||
font-size: 13px;
|
||||
color: var(--fp-text, inherit);
|
||||
cursor: pointer;
|
||||
}
|
||||
.alr-event input {
|
||||
accent-color: var(--fp-accent, currentColor);
|
||||
}
|
||||
|
||||
/* ---- responsive ---- */
|
||||
@media (max-width: 640px) {
|
||||
.alr-row {
|
||||
flex-wrap: wrap;
|
||||
}
|
||||
.alr-row-main {
|
||||
flex-basis: calc(100% - 90px);
|
||||
}
|
||||
.alr-del {
|
||||
margin-left: auto;
|
||||
}
|
||||
.alr-input--name {
|
||||
flex-basis: 100%;
|
||||
}
|
||||
.alr-detour {
|
||||
flex-basis: 100%;
|
||||
order: 3;
|
||||
flex-wrap: wrap;
|
||||
}
|
||||
/* A <select> won't shrink below its widest option unless it's allowed to:
|
||||
without min-width:0 the long detour labels push the page into a horizontal
|
||||
scroll at 390px. Let them fill the row and clip instead. */
|
||||
.alr-detour-select,
|
||||
.alr-resp .alr-select {
|
||||
max-width: 100%;
|
||||
width: 100%;
|
||||
min-width: 0;
|
||||
}
|
||||
.alr-resp {
|
||||
display: flex;
|
||||
flex-wrap: wrap;
|
||||
max-width: 100%;
|
||||
}
|
||||
.alr-ctl {
|
||||
flex-basis: 100%;
|
||||
order: 3;
|
||||
}
|
||||
}
|
||||
|
||||
@media (prefers-reduced-motion: reduce) {
|
||||
.alr-skel {
|
||||
animation: none;
|
||||
}
|
||||
.alr-input,
|
||||
.alr-seg-btn {
|
||||
transition: none;
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,690 @@
|
||||
import './Alerts.css'
|
||||
import { useCallback, useMemo, useState } from 'react'
|
||||
import { Button, Toggle, useConfirm } from '../components'
|
||||
import type { Alert, Model } from '../api'
|
||||
|
||||
// The Alerts section — out-of-band notifications (Telegram bot / webhook) for
|
||||
// kill-switch trips, apply failures, new devices and subscription expiry. It
|
||||
// lived at the bottom of the DNS page, which is the last place an operator
|
||||
// looking for "tell me when the tunnel dies" would think to look; it now renders
|
||||
// as a group on Settings. The component owns no I/O: every mutation goes through
|
||||
// the `onSave` prop so Settings keeps a single dirty banner and a single toast.
|
||||
//
|
||||
// NOTE on duplication: the detour helpers below (DetourCatalog, canonDetour,
|
||||
// detourValues, describeDetour, DetourSelect) plus asArray / uniqueName /
|
||||
// maskUrl / EmptyPlate are deliberate copies of the ones in DNS.tsx. DNS keeps
|
||||
// its own for resolvers and DNS rules; extracting a shared module would couple
|
||||
// two pages that otherwise share nothing, and that refactor is out of scope
|
||||
// here. If a third consumer ever appears, promote them then.
|
||||
|
||||
// ---- local Model extension --------------------------------------------------
|
||||
|
||||
/** The Model with the Alerts slice surfaced (index-signature passthrough). */
|
||||
type AlertsModel = Model & { Alerts?: Alert[] | null }
|
||||
|
||||
/**
|
||||
* Every event the daemon actually sends. A retired health-probe event was left
|
||||
* out on purpose: nothing ever fired it, so a channel that subscribed to it would
|
||||
* just stay quiet forever — the one failure mode an alert must not have. Only
|
||||
* events with a live firing path are offered here.
|
||||
*/
|
||||
const ALERT_EVENTS: ReadonlyArray<{ id: string; label: string }> = [
|
||||
{ id: 'killswitch', label: 'Kill-switch' },
|
||||
{ id: 'apply_fail', label: 'Apply failure' },
|
||||
{ id: 'new_device', label: 'New device' },
|
||||
{ id: 'sub_expiry', label: 'Subscription expiring' },
|
||||
]
|
||||
|
||||
// Shown when an alert routes through a detour with no direct fallback — the exact
|
||||
// case where a tunnel-down alert could fail to send. The user asked for this.
|
||||
const VIA_NO_FALLBACK_NOTE =
|
||||
'A kill-switch/tunnel-down alert may not send if it routes through the affected tunnel — enable fallback.'
|
||||
|
||||
// ---- helpers (copies of DNS.tsx — see the header note) ----------------------
|
||||
|
||||
const asArray = <T,>(a: T[] | null | undefined): T[] => (a ? a : [])
|
||||
|
||||
const HTTP_RE = /^https?:\/\//i
|
||||
|
||||
/** A remote URL often carries a token in its query/path — show host only. */
|
||||
function maskUrl(url: string): { host: string; masked: boolean } {
|
||||
try {
|
||||
const u = new URL(url)
|
||||
return { host: u.host, masked: u.search !== '' || u.pathname.replace(/\/+$/, '') !== '' }
|
||||
} catch {
|
||||
return { host: url || '—', masked: false }
|
||||
}
|
||||
}
|
||||
|
||||
function uniqueName(base: string, taken: Set<string>): string {
|
||||
const seed = base.trim() || 'alert'
|
||||
if (!taken.has(seed)) return seed
|
||||
let i = 2
|
||||
while (taken.has(`${seed}-${i}`)) i++
|
||||
return `${seed}-${i}`
|
||||
}
|
||||
|
||||
/** The live targets an alert's delivery can be pinned to (the picker). */
|
||||
interface DetourCatalog {
|
||||
groups: string[]
|
||||
chains: string[]
|
||||
egresses: { name: string; type: string }[]
|
||||
nodes: string[]
|
||||
}
|
||||
|
||||
/**
|
||||
* Normalise a stored `Via` to a picker option value. Empty/`direct` ⇒
|
||||
* `direct`; already-prefixed values (`group:`/`chain:`/`egress:`/`node:`) pass
|
||||
* through; a bare legacy name is resolved against the catalog so a still-valid
|
||||
* setup isn't mislabelled; anything unresolved is kept verbatim (shown stale).
|
||||
*/
|
||||
function canonDetour(raw: string | undefined, cat: DetourCatalog): string {
|
||||
const d = (raw ?? '').trim()
|
||||
if (!d || d.toLowerCase() === 'direct') return 'direct'
|
||||
if (/^(node|group|chain|egress):/i.test(d)) return d
|
||||
if (cat.egresses.some((e) => e.name === d)) return `egress:${d}`
|
||||
if (cat.groups.includes(d)) return `group:${d}`
|
||||
if (cat.chains.includes(d)) return `chain:${d}`
|
||||
if (cat.nodes.includes(d)) return `node:${d}`
|
||||
return d
|
||||
}
|
||||
|
||||
/** Every valid option value for a catalog, including `direct`. */
|
||||
function detourValues(cat: DetourCatalog): Set<string> {
|
||||
const s = new Set<string>(['direct'])
|
||||
for (const g of cat.groups) s.add(`group:${g}`)
|
||||
for (const c of cat.chains) s.add(`chain:${c}`)
|
||||
for (const e of cat.egresses) s.add(`egress:${e.name}`)
|
||||
for (const n of cat.nodes) s.add(`node:${n}`)
|
||||
return s
|
||||
}
|
||||
|
||||
/** Describe a canonical detour value for the row readout. */
|
||||
function describeDetour(
|
||||
canon: string,
|
||||
cat: DetourCatalog,
|
||||
valid: Set<string>,
|
||||
): { direct: boolean; prefix: string; name: string; missing: boolean } {
|
||||
if (canon === 'direct') return { direct: true, prefix: '', name: '', missing: false }
|
||||
const i = canon.indexOf(':')
|
||||
const kind = i === -1 ? '' : canon.slice(0, i)
|
||||
const name = i === -1 ? canon : canon.slice(i + 1)
|
||||
const missing = !valid.has(canon)
|
||||
let prefix = 'via'
|
||||
if (kind === 'group') prefix = 'via group'
|
||||
else if (kind === 'chain') prefix = 'via chain'
|
||||
else if (kind === 'node') prefix = 'via node'
|
||||
else if (kind === 'egress') {
|
||||
const eg = cat.egresses.find((e) => e.name === name)
|
||||
prefix = eg?.type === 'interface' ? 'via interface' : 'via egress'
|
||||
}
|
||||
return { direct: false, prefix, name, missing }
|
||||
}
|
||||
|
||||
// ---- section ----------------------------------------------------------------
|
||||
|
||||
export function AlertsSection({
|
||||
config,
|
||||
busy,
|
||||
loading,
|
||||
onSave,
|
||||
}: {
|
||||
/** Full desired-state model; null until it has loaded. */
|
||||
config: Model | null
|
||||
/** A save/apply is in flight — controls lock. */
|
||||
busy: boolean
|
||||
/** The config is still loading — show a skeleton row. */
|
||||
loading: boolean
|
||||
/** Persist the whole next model; resolves true on success (Settings' `save`). */
|
||||
onSave: (next: Model, okMsg: string) => Promise<boolean>
|
||||
}): JSX.Element {
|
||||
const confirm = useConfirm()
|
||||
const model = config as AlertsModel | null
|
||||
const alerts = useMemo<Alert[]>(() => asArray(model?.Alerts), [model])
|
||||
|
||||
// Alerts route through Direct/group/node/egress only (no chains) — the contract
|
||||
// vocabulary for Alert.Via. Built straight from the Model with chains dropped.
|
||||
const alertCatalog = useMemo<DetourCatalog>(
|
||||
() => ({
|
||||
groups: asArray(config?.Groups).map((g) => g.Name),
|
||||
chains: [],
|
||||
egresses: asArray(config?.Egresses).map((e) => ({ name: e.Name, type: e.Type })),
|
||||
nodes: asArray(config?.Nodes).map((n) => n.Name),
|
||||
}),
|
||||
[config],
|
||||
)
|
||||
const alertValid = useMemo(() => detourValues(alertCatalog), [alertCatalog])
|
||||
|
||||
const alertNames = useMemo(() => new Set(alerts.map((a) => a.Name)), [alerts])
|
||||
const alertsOn = alerts.filter((a) => a.Enabled).length
|
||||
|
||||
// ---- mutations — all writes go through onSave -----------------------------
|
||||
const addAlert = useCallback(
|
||||
(draft: Alert): Promise<boolean> => {
|
||||
if (!model) return Promise.resolve(false)
|
||||
const taken = new Set(alerts.map((a) => a.Name))
|
||||
const a: Alert = { ...draft, Name: uniqueName(draft.Name, taken) }
|
||||
return onSave({ ...model, Alerts: [...alerts, a] }, `Added ${a.Name}`)
|
||||
},
|
||||
[model, alerts, onSave],
|
||||
)
|
||||
|
||||
const toggleAlert = useCallback(
|
||||
(idx: number, on: boolean) => {
|
||||
if (!model) return
|
||||
const next = alerts.map((a, i) => (i === idx ? { ...a, Enabled: on } : a))
|
||||
void onSave({ ...model, Alerts: next }, `${next[idx].Name} ${on ? 'enabled' : 'disabled'}`)
|
||||
},
|
||||
[model, alerts, onSave],
|
||||
)
|
||||
|
||||
const removeAlert = useCallback(
|
||||
async (idx: number) => {
|
||||
if (!model) return
|
||||
const target = alerts[idx]
|
||||
const ok = await confirm({
|
||||
label: 'Delete alert',
|
||||
title: `Delete alert “${target.Name}”?`,
|
||||
body: 'This removes it from the config.',
|
||||
})
|
||||
if (!ok) return
|
||||
const next = alerts.filter((_, i) => i !== idx)
|
||||
void onSave({ ...model, Alerts: next }, `Deleted ${target.Name}`)
|
||||
},
|
||||
[model, alerts, onSave, confirm],
|
||||
)
|
||||
|
||||
const setAlertVia = useCallback(
|
||||
(idx: number, v: string) => {
|
||||
if (!model) return
|
||||
const via = v === 'direct' ? '' : v
|
||||
const next = alerts.map((a, i) => (i === idx ? { ...a, Via: via || undefined } : a))
|
||||
void onSave(
|
||||
{ ...model, Alerts: next },
|
||||
via ? `${next[idx].Name} delivers via ${via}` : `${next[idx].Name} delivers direct`,
|
||||
)
|
||||
},
|
||||
[model, alerts, onSave],
|
||||
)
|
||||
|
||||
const setAlertFallback = useCallback(
|
||||
(idx: number, on: boolean) => {
|
||||
if (!model) return
|
||||
const next = alerts.map((a, i) => (i === idx ? { ...a, Fallback: on || undefined } : a))
|
||||
void onSave(
|
||||
{ ...model, Alerts: next },
|
||||
`${next[idx].Name} direct fallback ${on ? 'on' : 'off'}`,
|
||||
)
|
||||
},
|
||||
[model, alerts, onSave],
|
||||
)
|
||||
|
||||
return (
|
||||
<div className="alr-section" aria-label="Alerts">
|
||||
<header className="alr-sec-hd">
|
||||
<h2 className="alr-sec-title">Alerts</h2>
|
||||
<span className="alr-sec-count mono">
|
||||
{alertsOn} / {alerts.length} on
|
||||
</span>
|
||||
</header>
|
||||
<p className="alr-sec-note">
|
||||
Out-of-band notifications. Delivered <strong>direct to the internet</strong> by default — so a
|
||||
kill-switch or engine-down alert still reaches you when the proxy is down. You can route one
|
||||
through a group, node or egress instead, with a direct fallback if that detour fails.
|
||||
</p>
|
||||
|
||||
<AddAlertForm
|
||||
busy={busy}
|
||||
disabled={!config}
|
||||
taken={alertNames}
|
||||
catalog={alertCatalog}
|
||||
valid={alertValid}
|
||||
onAdd={addAlert}
|
||||
/>
|
||||
|
||||
{loading ? (
|
||||
<ul className="alr-rows" aria-hidden="true">
|
||||
<li className="alr-skel" />
|
||||
</ul>
|
||||
) : alerts.length === 0 ? (
|
||||
<EmptyPlate
|
||||
title="No alerts"
|
||||
body="Add a Telegram bot or a webhook above to get notified when the kill-switch trips, a new device joins, or an apply fails."
|
||||
/>
|
||||
) : (
|
||||
<ul className="alr-rows">
|
||||
{alerts.map((a, i) => (
|
||||
<AlertRow
|
||||
key={`${a.Name}-${i}`}
|
||||
alert={a}
|
||||
busy={busy}
|
||||
catalog={alertCatalog}
|
||||
valid={alertValid}
|
||||
onToggle={(on) => toggleAlert(i, on)}
|
||||
onVia={(v) => setAlertVia(i, v)}
|
||||
onFallback={(on) => setAlertFallback(i, on)}
|
||||
onDelete={() => removeAlert(i)}
|
||||
/>
|
||||
))}
|
||||
</ul>
|
||||
)}
|
||||
</div>
|
||||
)
|
||||
}
|
||||
|
||||
// ---- alert add form + row ----------------------------------------------------
|
||||
|
||||
function AddAlertForm({
|
||||
busy,
|
||||
disabled,
|
||||
taken,
|
||||
catalog,
|
||||
valid,
|
||||
onAdd,
|
||||
}: {
|
||||
busy: boolean
|
||||
disabled: boolean
|
||||
taken: Set<string>
|
||||
catalog: DetourCatalog
|
||||
valid: Set<string>
|
||||
onAdd: (a: Alert) => Promise<boolean>
|
||||
}) {
|
||||
const [name, setName] = useState('')
|
||||
const [type, setType] = useState<'telegram' | 'webhook'>('telegram')
|
||||
const [token, setToken] = useState('')
|
||||
const [chatId, setChatId] = useState('')
|
||||
const [url, setUrl] = useState('')
|
||||
const [events, setEvents] = useState<string[]>(['killswitch'])
|
||||
const [via, setVia] = useState('direct')
|
||||
const [fallback, setFallback] = useState(false)
|
||||
const [err, setErr] = useState<string | null>(null)
|
||||
|
||||
const reset = () => {
|
||||
setName('')
|
||||
setType('telegram')
|
||||
setToken('')
|
||||
setChatId('')
|
||||
setUrl('')
|
||||
setEvents(['killswitch'])
|
||||
setVia('direct')
|
||||
setFallback(false)
|
||||
}
|
||||
|
||||
const toggleEvent = (id: string) =>
|
||||
setEvents((prev) => (prev.includes(id) ? prev.filter((e) => e !== id) : [...prev, id]))
|
||||
|
||||
const submit = async () => {
|
||||
const nm = name.trim()
|
||||
if (!nm) {
|
||||
setErr('Give the alert a name.')
|
||||
return
|
||||
}
|
||||
if (taken.has(nm)) {
|
||||
setErr(`An alert named “${nm}” already exists.`)
|
||||
return
|
||||
}
|
||||
if (type === 'telegram') {
|
||||
if (!token.trim() || !chatId.trim()) {
|
||||
setErr('Telegram needs a bot token and a chat ID.')
|
||||
return
|
||||
}
|
||||
} else if (!HTTP_RE.test(url.trim())) {
|
||||
setErr('Enter an http(s):// webhook URL.')
|
||||
return
|
||||
}
|
||||
if (events.length === 0) {
|
||||
setErr('Pick at least one event to notify on.')
|
||||
return
|
||||
}
|
||||
setErr(null)
|
||||
const routed = via !== 'direct'
|
||||
const routing = { Via: routed ? via : undefined, Fallback: routed && fallback ? true : undefined }
|
||||
const draft: Alert =
|
||||
type === 'telegram'
|
||||
? { Name: nm, Enabled: true, Type: 'telegram', Token: token.trim(), ChatID: chatId.trim(), Events: events, ...routing }
|
||||
: { Name: nm, Enabled: true, Type: 'webhook', URL: url.trim(), Events: events, ...routing }
|
||||
const ok = await onAdd(draft)
|
||||
if (ok) reset()
|
||||
}
|
||||
|
||||
const routed = via !== 'direct'
|
||||
|
||||
return (
|
||||
<form
|
||||
className="alr-add"
|
||||
onSubmit={(e) => {
|
||||
e.preventDefault()
|
||||
void submit()
|
||||
}}
|
||||
>
|
||||
<div className="alr-add-top">
|
||||
<input
|
||||
className="alr-input alr-input--name"
|
||||
type="text"
|
||||
spellCheck={false}
|
||||
autoComplete="off"
|
||||
placeholder="Alert name"
|
||||
aria-label="Alert name"
|
||||
value={name}
|
||||
onChange={(e) => {
|
||||
setName(e.target.value)
|
||||
if (err) setErr(null)
|
||||
}}
|
||||
disabled={busy || disabled}
|
||||
/>
|
||||
<div className="alr-seg" role="group" aria-label="Alert type">
|
||||
<button
|
||||
type="button"
|
||||
className={type === 'telegram' ? 'alr-seg-btn on' : 'alr-seg-btn'}
|
||||
aria-pressed={type === 'telegram'}
|
||||
onClick={() => setType('telegram')}
|
||||
disabled={busy || disabled}
|
||||
>
|
||||
Telegram
|
||||
</button>
|
||||
<button
|
||||
type="button"
|
||||
className={type === 'webhook' ? 'alr-seg-btn on' : 'alr-seg-btn'}
|
||||
aria-pressed={type === 'webhook'}
|
||||
onClick={() => setType('webhook')}
|
||||
disabled={busy || disabled}
|
||||
>
|
||||
Webhook
|
||||
</button>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
{type === 'telegram' ? (
|
||||
<>
|
||||
<input
|
||||
className="alr-input"
|
||||
type="password"
|
||||
spellCheck={false}
|
||||
autoComplete="off"
|
||||
placeholder="Bot token (kept secret)"
|
||||
aria-label="Telegram bot token"
|
||||
value={token}
|
||||
onChange={(e) => {
|
||||
setToken(e.target.value)
|
||||
if (err) setErr(null)
|
||||
}}
|
||||
disabled={busy || disabled}
|
||||
/>
|
||||
<input
|
||||
className="alr-input"
|
||||
type="text"
|
||||
spellCheck={false}
|
||||
autoComplete="off"
|
||||
placeholder="Chat ID (e.g. -1001234567890)"
|
||||
aria-label="Telegram chat ID"
|
||||
value={chatId}
|
||||
onChange={(e) => {
|
||||
setChatId(e.target.value)
|
||||
if (err) setErr(null)
|
||||
}}
|
||||
disabled={busy || disabled}
|
||||
/>
|
||||
</>
|
||||
) : (
|
||||
<input
|
||||
className="alr-input"
|
||||
type="text"
|
||||
inputMode="url"
|
||||
spellCheck={false}
|
||||
autoComplete="off"
|
||||
placeholder="https://hooks.example.com/…"
|
||||
aria-label="Webhook URL"
|
||||
value={url}
|
||||
onChange={(e) => {
|
||||
setUrl(e.target.value)
|
||||
if (err) setErr(null)
|
||||
}}
|
||||
disabled={busy || disabled}
|
||||
/>
|
||||
)}
|
||||
|
||||
<fieldset className="alr-events" aria-label="Events to notify on">
|
||||
{ALERT_EVENTS.map((ev) => (
|
||||
<label key={ev.id} className="alr-event">
|
||||
<input
|
||||
type="checkbox"
|
||||
checked={events.includes(ev.id)}
|
||||
onChange={() => toggleEvent(ev.id)}
|
||||
disabled={busy || disabled}
|
||||
/>
|
||||
<span>{ev.label}</span>
|
||||
</label>
|
||||
))}
|
||||
</fieldset>
|
||||
|
||||
<div className="alr-delivery">
|
||||
<label className="alr-resp">
|
||||
<span className="alr-resp-label mono">Deliver via</span>
|
||||
<DetourSelect
|
||||
value={via}
|
||||
catalog={catalog}
|
||||
valid={valid}
|
||||
busy={busy}
|
||||
disabled={disabled}
|
||||
ariaLabel="Deliver alert via"
|
||||
onChange={setVia}
|
||||
directLabel="Direct (default)"
|
||||
/>
|
||||
</label>
|
||||
<label className="alr-fallback">
|
||||
<Toggle
|
||||
pressed={fallback}
|
||||
onChange={setFallback}
|
||||
label={fallback ? 'Disable direct fallback' : 'Enable direct fallback'}
|
||||
disabled={busy || disabled || !routed}
|
||||
/>
|
||||
<span className="alr-fallback-label mono">Fallback to direct</span>
|
||||
</label>
|
||||
</div>
|
||||
{routed && !fallback && (
|
||||
<p className="alr-note" role="note">
|
||||
{VIA_NO_FALLBACK_NOTE}
|
||||
</p>
|
||||
)}
|
||||
|
||||
<div className="alr-add-actions">
|
||||
{err && (
|
||||
<p className="alr-field-err" role="alert">
|
||||
{err}
|
||||
</p>
|
||||
)}
|
||||
<Button type="submit" variant="primary" disabled={busy || disabled}>
|
||||
{busy ? 'Saving…' : 'Add alert'}
|
||||
</Button>
|
||||
</div>
|
||||
</form>
|
||||
)
|
||||
}
|
||||
|
||||
function AlertRow({
|
||||
alert,
|
||||
busy,
|
||||
catalog,
|
||||
valid,
|
||||
onToggle,
|
||||
onVia,
|
||||
onFallback,
|
||||
onDelete,
|
||||
}: {
|
||||
alert: Alert
|
||||
busy: boolean
|
||||
catalog: DetourCatalog
|
||||
valid: Set<string>
|
||||
onToggle: (on: boolean) => void
|
||||
onVia: (v: string) => void
|
||||
onFallback: (on: boolean) => void
|
||||
onDelete: () => void
|
||||
}) {
|
||||
// Never render the token/URL in clear — show a masked descriptor only.
|
||||
const detail = useMemo(() => {
|
||||
if (alert.Type === 'telegram') {
|
||||
return { text: `chat ${alert.ChatID || '—'}`, masked: !!alert.Token }
|
||||
}
|
||||
const { host, masked } = maskUrl(alert.URL ?? '')
|
||||
return { text: host, masked: masked || !!alert.URL }
|
||||
}, [alert.Type, alert.ChatID, alert.Token, alert.URL])
|
||||
|
||||
const events = asArray(alert.Events)
|
||||
const canon = useMemo(() => canonDetour(alert.Via, catalog), [alert.Via, catalog])
|
||||
const route = useMemo(() => describeDetour(canon, catalog, valid), [canon, catalog, valid])
|
||||
const routed = canon !== 'direct'
|
||||
const fallback = alert.Fallback ?? false
|
||||
|
||||
return (
|
||||
<li className="alr-row">
|
||||
<Toggle
|
||||
pressed={alert.Enabled}
|
||||
onChange={onToggle}
|
||||
label={`${alert.Enabled ? 'Disable' : 'Enable'} alert ${alert.Name}`}
|
||||
disabled={busy}
|
||||
/>
|
||||
<div className="alr-row-main">
|
||||
<div className="alr-row-l1">
|
||||
<span className="alr-row-name">{alert.Name}</span>
|
||||
<span className="alr-badge">{alert.Type}</span>
|
||||
{events.map((e) => (
|
||||
<span key={e} className="alr-badge alr-badge--accent">
|
||||
{e}
|
||||
</span>
|
||||
))}
|
||||
</div>
|
||||
<div className="alr-row-l2 mono">
|
||||
<span className="alr-row-detail">{detail.text}</span>
|
||||
{detail.masked && (
|
||||
<span className="alr-masked" title="Secret is stored but hidden here">
|
||||
secret hidden
|
||||
</span>
|
||||
)}
|
||||
{route.direct ? (
|
||||
<span className="alr-path">direct</span>
|
||||
) : (
|
||||
<span className="alr-path" data-active="on" data-missing={route.missing ? 'y' : undefined}>
|
||||
{route.prefix} <strong className="alr-path-name">{route.name}</strong>
|
||||
{route.missing && <span className="alr-path-flag"> (missing)</span>}
|
||||
{fallback ? ' · +direct fallback' : ' · no fallback'}
|
||||
</span>
|
||||
)}
|
||||
</div>
|
||||
{routed && !fallback && <p className="alr-note alr-note--row">{VIA_NO_FALLBACK_NOTE}</p>}
|
||||
</div>
|
||||
<div className="alr-ctl">
|
||||
<label className="alr-detour">
|
||||
<span className="alr-detour-label mono">Deliver via</span>
|
||||
<DetourSelect
|
||||
value={canon}
|
||||
catalog={catalog}
|
||||
valid={valid}
|
||||
busy={busy}
|
||||
disabled={false}
|
||||
ariaLabel={`Deliver alert ${alert.Name} via`}
|
||||
onChange={onVia}
|
||||
directLabel="Direct (default)"
|
||||
/>
|
||||
</label>
|
||||
<label className="alr-fallback">
|
||||
<Toggle
|
||||
pressed={fallback}
|
||||
onChange={onFallback}
|
||||
label={`${fallback ? 'Disable' : 'Enable'} direct fallback for ${alert.Name}`}
|
||||
disabled={busy || !routed}
|
||||
/>
|
||||
<span className="alr-fallback-label mono">Fallback to direct</span>
|
||||
</label>
|
||||
</div>
|
||||
<Button
|
||||
className="alr-del"
|
||||
onClick={onDelete}
|
||||
disabled={busy}
|
||||
aria-label={`Delete alert ${alert.Name}`}
|
||||
>
|
||||
Delete
|
||||
</Button>
|
||||
</li>
|
||||
)
|
||||
}
|
||||
|
||||
/** The live delivery picker: option list built from the Model's targets. */
|
||||
function DetourSelect({
|
||||
value,
|
||||
catalog,
|
||||
valid,
|
||||
busy,
|
||||
disabled,
|
||||
ariaLabel,
|
||||
onChange,
|
||||
directLabel = 'Direct (no proxy)',
|
||||
}: {
|
||||
value: string // canonical value
|
||||
catalog: DetourCatalog
|
||||
valid: Set<string>
|
||||
busy: boolean
|
||||
disabled: boolean
|
||||
ariaLabel: string
|
||||
onChange: (v: string) => void
|
||||
directLabel?: string
|
||||
}) {
|
||||
const missing = value !== 'direct' && !valid.has(value)
|
||||
return (
|
||||
<select
|
||||
className="alr-select alr-detour-select"
|
||||
value={value}
|
||||
onChange={(e) => onChange(e.target.value)}
|
||||
disabled={busy || disabled}
|
||||
aria-label={ariaLabel}
|
||||
>
|
||||
<option value="direct">{directLabel}</option>
|
||||
{catalog.groups.length > 0 && (
|
||||
<optgroup label="Groups">
|
||||
{catalog.groups.map((g) => (
|
||||
<option key={g} value={`group:${g}`}>
|
||||
Group {g} (balancer)
|
||||
</option>
|
||||
))}
|
||||
</optgroup>
|
||||
)}
|
||||
{catalog.chains.length > 0 && (
|
||||
<optgroup label="Chains">
|
||||
{catalog.chains.map((c) => (
|
||||
<option key={c} value={`chain:${c}`}>
|
||||
Chain {c}
|
||||
</option>
|
||||
))}
|
||||
</optgroup>
|
||||
)}
|
||||
{catalog.egresses.length > 0 && (
|
||||
<optgroup label="Interfaces / egresses">
|
||||
{catalog.egresses.map((e) => (
|
||||
<option key={e.name} value={`egress:${e.name}`}>
|
||||
Interface/egress {e.name}
|
||||
{e.type ? ` (${e.type})` : ''}
|
||||
</option>
|
||||
))}
|
||||
</optgroup>
|
||||
)}
|
||||
{catalog.nodes.length > 0 && (
|
||||
<optgroup label="Nodes">
|
||||
{catalog.nodes.map((n) => (
|
||||
<option key={n} value={`node:${n}`}>
|
||||
Node {n}
|
||||
</option>
|
||||
))}
|
||||
</optgroup>
|
||||
)}
|
||||
{missing && <option value={value}>{value} (missing)</option>}
|
||||
</select>
|
||||
)
|
||||
}
|
||||
|
||||
function EmptyPlate({ title, body }: { title: string; body: string }) {
|
||||
return (
|
||||
<div className="alr-empty">
|
||||
<span className="alr-empty-title mono">{title}</span>
|
||||
<p className="alr-empty-body">{body}</p>
|
||||
</div>
|
||||
)
|
||||
}
|
||||
+76
-32
@@ -11,7 +11,7 @@ import {
|
||||
ApiError,
|
||||
} from '../api'
|
||||
import type { Globals, Status } from '../api'
|
||||
import { engineReadout } from '../planeState'
|
||||
import { engineReadout, killSwitchReadout } from '../planeState'
|
||||
import { onPendingConfirmExpire, usePendingConfirm } from '../pendingConfirm'
|
||||
|
||||
// Short, readable config hash — drops the "sha256:" prefix like the footer does.
|
||||
@@ -103,30 +103,64 @@ export default function Apply() {
|
||||
void loadConfig()
|
||||
}, [loadConfig])
|
||||
|
||||
// The window running out is the daemon reverting on its own — observe it and
|
||||
// say so. The countdown itself ticks inside usePendingConfirm; this only reacts
|
||||
// to the end of it, and the store makes sure that fires exactly once even with
|
||||
// the app-wide band mounted alongside.
|
||||
// The window running out does NOT mean the daemon rolled back.
|
||||
//
|
||||
// apply.ArmRollback captures the data-plane generation when it arms, and on
|
||||
// expiry it compares. If anything re-applied the plane in between — another
|
||||
// panel apply, SIGHUP, a hotplug or the once-a-minute cron reconcile, the WAN
|
||||
// profile auto-switch — it disarms and KEEPS the running config, logging "NOT
|
||||
// rolling back" and nothing else. That is the common case on a production
|
||||
// router, and this page used to print "daemon auto-rolled back to last-good
|
||||
// config" for it: a confident report of an event that did not happen, with a
|
||||
// hash pair underneath that quietly said "unchanged".
|
||||
//
|
||||
// The panel cannot see which branch ran — the daemon says so only in its log.
|
||||
// So it reports the one thing it CAN observe, the live config hash, and waits
|
||||
// for the revert to land before reading it (a rollback is a full re-apply and
|
||||
// does not complete the instant the timer fires).
|
||||
const liveHashRef = useRef('')
|
||||
liveHashRef.current = status?.hash ?? ''
|
||||
useEffect(
|
||||
() =>
|
||||
onPendingConfirmExpire(() => {
|
||||
const before = liveHashRef.current
|
||||
flash('Auto-rolled back')
|
||||
void (async () => {
|
||||
const after = (await refreshStatus())?.hash ?? ''
|
||||
setResult({
|
||||
kind: 'expire',
|
||||
tone: 'warn',
|
||||
text: 'Confirm window elapsed — daemon auto-rolled back to last-good config.',
|
||||
before,
|
||||
after,
|
||||
})
|
||||
})()
|
||||
}),
|
||||
[flash, refreshStatus],
|
||||
)
|
||||
useEffect(() => {
|
||||
let cancelled = false
|
||||
const off = onPendingConfirmExpire(() => {
|
||||
const before = liveHashRef.current
|
||||
flash('Confirm window elapsed')
|
||||
setResult({
|
||||
kind: 'expire',
|
||||
tone: 'warn',
|
||||
text: 'Confirm window elapsed. Reading what the daemon did…',
|
||||
before,
|
||||
after: before,
|
||||
})
|
||||
void (async () => {
|
||||
let after = before
|
||||
for (let i = 0; i < 4 && !cancelled; i++) {
|
||||
await new Promise((r) => window.setTimeout(r, 1500))
|
||||
if (cancelled) return
|
||||
after = (await refreshStatus())?.hash ?? after
|
||||
if (after !== before) break
|
||||
}
|
||||
if (cancelled) return
|
||||
setResult({
|
||||
kind: 'expire',
|
||||
tone: 'warn',
|
||||
text:
|
||||
after !== before
|
||||
? 'Confirm window elapsed and the live config changed — the daemon reverted to its last-good config.'
|
||||
: 'Confirm window elapsed and the live config has not changed, so this config is still running. ' +
|
||||
'The daemon only reverts if nothing else re-applied the data plane while the window was open; ' +
|
||||
'otherwise it stands down and keeps what is live. Which one happened is in the daemon log — ' +
|
||||
'download it from Settings, or run `logread -e shater`.',
|
||||
before,
|
||||
after,
|
||||
})
|
||||
})()
|
||||
})
|
||||
return () => {
|
||||
cancelled = true
|
||||
off()
|
||||
}
|
||||
}, [flash, refreshStatus])
|
||||
|
||||
const confirmWindow = globals?.ConfirmTimeout ?? 0
|
||||
|
||||
@@ -241,6 +275,17 @@ export default function Apply() {
|
||||
// is a status readout, and the config on disk can already differ from what is
|
||||
// installed. Falls back to the config only while /api/status is unread.
|
||||
const killArmed = (status?.kill_switch ?? globals?.KillSwitch ?? 'closed') === 'closed'
|
||||
// Whether that setting is actually installed — same three-plus-unknown reading
|
||||
// as Overview, so the two pages cannot disagree about the same router.
|
||||
const kill = killSwitchReadout(status, globals?.KillSwitch)
|
||||
const killWord =
|
||||
kill.state === 'open'
|
||||
? 'open'
|
||||
: kill.state === 'armed'
|
||||
? 'fail-closed'
|
||||
: kill.state === 'inert'
|
||||
? 'closed · not in effect'
|
||||
: 'closed · not reported'
|
||||
// Every engine mark on this page comes from ONE reading, and that reading is
|
||||
// able to say "stopped" — see planeState.engineState for why `status.running`
|
||||
// could not. This page is where someone lands when the network is down; three
|
||||
@@ -282,11 +327,7 @@ export default function Apply() {
|
||||
variant={dataVariant}
|
||||
value={status?.table ? 'nft installed' : 'no table'}
|
||||
/>
|
||||
<StatusPip
|
||||
label="Kill-switch"
|
||||
variant={killArmed ? 'on' : 'amber'}
|
||||
value={killArmed ? 'fail-closed' : 'open'}
|
||||
/>
|
||||
<StatusPip label="Kill-switch" variant={kill.variant} value={killWord} />
|
||||
</div>
|
||||
|
||||
{statusError && (
|
||||
@@ -320,7 +361,7 @@ export default function Apply() {
|
||||
v: status?.table ? 'nft installed' : 'no table',
|
||||
hot: dataVariant === 'crit',
|
||||
},
|
||||
{ k: 'kill-switch', v: killArmed ? 'fail-closed' : 'open', hot: !killArmed },
|
||||
{ k: 'kill-switch', v: killWord, hot: kill.variant === 'crit' || !killArmed },
|
||||
]}
|
||||
/>
|
||||
<Module
|
||||
@@ -374,8 +415,9 @@ export default function Apply() {
|
||||
is what "live but not kept" means — so the readout survives a
|
||||
reload instead of depending on what this tab remembers. */}
|
||||
Applied config <span className="mono">{liveHash}</span> is live but not yet kept.
|
||||
Confirm to keep it — otherwise the daemon rolls back to the last-good config when
|
||||
the timer hits zero.
|
||||
Confirm to keep it. At zero the daemon rolls back to the last-good config — unless
|
||||
something else re-applies the data plane first, in which case it stands down and
|
||||
keeps whatever is live.
|
||||
</p>
|
||||
<div className="cc-bar" aria-hidden="true">
|
||||
<span className="cc-bar-fill" style={{ width: `${pct}%` }} />
|
||||
@@ -492,7 +534,9 @@ function labelFor(kind: ActionKind): string {
|
||||
case 'rollback':
|
||||
return 'Rollback'
|
||||
case 'expire':
|
||||
return 'Auto-rollback'
|
||||
// NOT "Auto-rollback": on expiry the daemon either reverts or stands down,
|
||||
// and this page cannot tell which. Name the event it did observe.
|
||||
return 'Window elapsed'
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -648,73 +648,6 @@
|
||||
}
|
||||
}
|
||||
|
||||
/* alert delivery: deliver-via picker + fallback toggle + caution note */
|
||||
.dns-alert-delivery {
|
||||
display: flex;
|
||||
flex-wrap: wrap;
|
||||
align-items: center;
|
||||
gap: 10px 20px;
|
||||
}
|
||||
.dns-fallback {
|
||||
display: inline-flex;
|
||||
align-items: center;
|
||||
gap: 8px;
|
||||
cursor: pointer;
|
||||
}
|
||||
.dns-fallback-label {
|
||||
font-size: 10px;
|
||||
letter-spacing: var(--track-label, 0.18em);
|
||||
text-transform: uppercase;
|
||||
color: var(--faint);
|
||||
}
|
||||
/* the per-alert-row delivery controls sit inline in the row (like .dns-detour) */
|
||||
.dns-alert-ctl {
|
||||
flex: none;
|
||||
display: flex;
|
||||
flex-direction: column;
|
||||
gap: 8px;
|
||||
min-width: 0;
|
||||
}
|
||||
.dns-alert-note {
|
||||
margin: 0;
|
||||
font-family: var(--font-sans);
|
||||
font-size: 11.5px;
|
||||
line-height: 1.5;
|
||||
color: var(--amber);
|
||||
max-width: 56ch;
|
||||
}
|
||||
.dns-alert-note--row {
|
||||
margin-top: 2px;
|
||||
}
|
||||
|
||||
@media (max-width: 640px) {
|
||||
.dns-alert-ctl {
|
||||
flex-basis: 100%;
|
||||
order: 3;
|
||||
}
|
||||
}
|
||||
|
||||
/* alert event checkboxes */
|
||||
.dns-events {
|
||||
display: flex;
|
||||
flex-wrap: wrap;
|
||||
gap: 8px 16px;
|
||||
margin: 0;
|
||||
padding: 0;
|
||||
border: 0;
|
||||
}
|
||||
.dns-event {
|
||||
display: inline-flex;
|
||||
align-items: center;
|
||||
gap: 6px;
|
||||
font-size: 13px;
|
||||
color: var(--fp-text, inherit);
|
||||
cursor: pointer;
|
||||
}
|
||||
.dns-event input {
|
||||
accent-color: var(--fp-accent, currentColor);
|
||||
}
|
||||
|
||||
@media (prefers-reduced-motion: reduce) {
|
||||
.dns-skel {
|
||||
animation: none;
|
||||
|
||||
+1
-480
@@ -9,7 +9,7 @@ import {
|
||||
updateRuleset as apiUpdateRuleset,
|
||||
ApiError,
|
||||
} from '../api'
|
||||
import type { Alert, DNSRule, Model, Resolver, RulesetStatus } from '../api'
|
||||
import type { DNSRule, Model, Resolver, RulesetStatus } from '../api'
|
||||
import { everyLabel, relFetch } from '../format'
|
||||
|
||||
// The DNS / Blocklists page is a thin editor over the desired-state Model —
|
||||
@@ -60,28 +60,9 @@ type GlobalsX = Model['Globals'] & { DNSFilter?: boolean }
|
||||
type DNSModel = Model & {
|
||||
Blocklists?: Blocklist[] | null
|
||||
Allowlists?: Allowlist[] | null
|
||||
Alerts?: Alert[] | null
|
||||
DNSRules?: DNSRule[] | null
|
||||
}
|
||||
|
||||
/**
|
||||
* Every event the daemon actually sends. A retired health-probe event was left
|
||||
* out on purpose: nothing ever fired it, so a channel that subscribed to it would
|
||||
* just stay quiet forever — the one failure mode an alert must not have. Only
|
||||
* events with a live firing path are offered here.
|
||||
*/
|
||||
const ALERT_EVENTS: ReadonlyArray<{ id: string; label: string }> = [
|
||||
{ id: 'killswitch', label: 'Kill-switch' },
|
||||
{ id: 'apply_fail', label: 'Apply failure' },
|
||||
{ id: 'new_device', label: 'New device' },
|
||||
{ id: 'sub_expiry', label: 'Subscription expiring' },
|
||||
]
|
||||
|
||||
// Shown when an alert routes through a detour with no direct fallback — the exact
|
||||
// case where a tunnel-down alert could fail to send. The user asked for this.
|
||||
const VIA_NO_FALLBACK_NOTE =
|
||||
'A kill-switch/tunnel-down alert may not send if it routes through the affected tunnel — enable fallback.'
|
||||
|
||||
// ---- helpers ---------------------------------------------------------------
|
||||
|
||||
const asArray = <T,>(a: T[] | null | undefined): T[] => (a ? a : [])
|
||||
@@ -310,7 +291,6 @@ export default function DNS() {
|
||||
const blocklists = useMemo(() => asArray(config?.Blocklists), [config])
|
||||
const allowlists = useMemo(() => asArray(config?.Allowlists), [config])
|
||||
const resolvers = useMemo<Resolver[]>(() => asArray(config?.Resolvers), [config])
|
||||
const alerts = useMemo<Alert[]>(() => asArray(config?.Alerts), [config])
|
||||
// Ascending Order — the engine evaluates DNS rules first-match, so the list is
|
||||
// shown and edited in the order it actually runs.
|
||||
const dnsRules = useMemo<DNSRule[]>(
|
||||
@@ -719,78 +699,6 @@ export default function DNS() {
|
||||
[config, dnsRules, save, confirm],
|
||||
)
|
||||
|
||||
// ---- alert mutations ------------------------------------------------------
|
||||
const addAlert = useCallback(
|
||||
(draft: Alert): Promise<boolean> => {
|
||||
if (!config) return Promise.resolve(false)
|
||||
const taken = new Set(alerts.map((a) => a.Name))
|
||||
const a: Alert = { ...draft, Name: uniqueName(draft.Name, taken) }
|
||||
return save({ ...config, Alerts: [...alerts, a] }, `Added ${a.Name}`)
|
||||
},
|
||||
[config, alerts, save],
|
||||
)
|
||||
|
||||
const toggleAlert = useCallback(
|
||||
(idx: number, on: boolean) => {
|
||||
if (!config) return
|
||||
const next = alerts.map((a, i) => (i === idx ? { ...a, Enabled: on } : a))
|
||||
void save({ ...config, Alerts: next }, `${next[idx].Name} ${on ? 'enabled' : 'disabled'}`)
|
||||
},
|
||||
[config, alerts, save],
|
||||
)
|
||||
|
||||
const removeAlert = useCallback(
|
||||
async (idx: number) => {
|
||||
if (!config) return
|
||||
const target = alerts[idx]
|
||||
const ok = await confirm({
|
||||
label: 'Delete alert',
|
||||
title: `Delete alert “${target.Name}”?`,
|
||||
body: 'This removes it from the config.',
|
||||
})
|
||||
if (!ok) return
|
||||
const next = alerts.filter((_, i) => i !== idx)
|
||||
void save({ ...config, Alerts: next }, `Deleted ${target.Name}`)
|
||||
},
|
||||
[config, alerts, save, confirm],
|
||||
)
|
||||
|
||||
const setAlertVia = useCallback(
|
||||
(idx: number, v: string) => {
|
||||
if (!config) return
|
||||
const via = v === 'direct' ? '' : v
|
||||
const next = alerts.map((a, i) => (i === idx ? { ...a, Via: via || undefined } : a))
|
||||
void save(
|
||||
{ ...config, Alerts: next },
|
||||
via ? `${next[idx].Name} delivers via ${via}` : `${next[idx].Name} delivers direct`,
|
||||
)
|
||||
},
|
||||
[config, alerts, save],
|
||||
)
|
||||
|
||||
const setAlertFallback = useCallback(
|
||||
(idx: number, on: boolean) => {
|
||||
if (!config) return
|
||||
const next = alerts.map((a, i) => (i === idx ? { ...a, Fallback: on || undefined } : a))
|
||||
void save(
|
||||
{ ...config, Alerts: next },
|
||||
`${next[idx].Name} direct fallback ${on ? 'on' : 'off'}`,
|
||||
)
|
||||
},
|
||||
[config, alerts, save],
|
||||
)
|
||||
|
||||
// Alerts route through Direct/group/node/egress only (no chains) — the contract
|
||||
// vocabulary for Alert.Via. Reuse the resolver detour catalog with chains dropped.
|
||||
const alertCatalog = useMemo<DetourCatalog>(
|
||||
() => ({ ...detourCatalog, chains: [] }),
|
||||
[detourCatalog],
|
||||
)
|
||||
const alertValid = useMemo(() => detourValues(alertCatalog), [alertCatalog])
|
||||
|
||||
const alertNames = useMemo(() => new Set(alerts.map((a) => a.Name)), [alerts])
|
||||
const alertsOn = alerts.filter((a) => a.Enabled).length
|
||||
|
||||
const loading = config === null && loadError === null
|
||||
|
||||
return (
|
||||
@@ -1211,57 +1119,6 @@ export default function DNS() {
|
||||
)}
|
||||
</div>
|
||||
|
||||
{/* ---- 5. ALERTS ---- */}
|
||||
<div className="dns-section" aria-label="Alerts">
|
||||
<header className="dns-sec-hd">
|
||||
<h2 className="dns-sec-title">Alerts</h2>
|
||||
<span className="dns-sec-count mono">
|
||||
{alertsOn} / {alerts.length} on
|
||||
</span>
|
||||
</header>
|
||||
<p className="dns-sec-note">
|
||||
Out-of-band notifications. Delivered <strong>direct to the internet</strong> by default — so a
|
||||
kill-switch or engine-down alert still reaches you when the proxy is down. You can route one
|
||||
through a group, node or egress instead, with a direct fallback if that detour fails.
|
||||
</p>
|
||||
|
||||
<AddAlertForm
|
||||
busy={busy}
|
||||
disabled={!config}
|
||||
taken={alertNames}
|
||||
catalog={alertCatalog}
|
||||
valid={alertValid}
|
||||
onAdd={addAlert}
|
||||
/>
|
||||
|
||||
{loading ? (
|
||||
<ul className="dns-rows" aria-hidden="true">
|
||||
<li className="dns-skel" />
|
||||
</ul>
|
||||
) : alerts.length === 0 ? (
|
||||
<EmptyPlate
|
||||
title="No alerts"
|
||||
body="Add a Telegram bot or a webhook above to get notified when the kill-switch trips, a new device joins, or an apply fails."
|
||||
/>
|
||||
) : (
|
||||
<ul className="dns-rows">
|
||||
{alerts.map((a, i) => (
|
||||
<AlertRow
|
||||
key={`${a.Name}-${i}`}
|
||||
alert={a}
|
||||
busy={busy}
|
||||
catalog={alertCatalog}
|
||||
valid={alertValid}
|
||||
onToggle={(on) => toggleAlert(i, on)}
|
||||
onVia={(v) => setAlertVia(i, v)}
|
||||
onFallback={(on) => setAlertFallback(i, on)}
|
||||
onDelete={() => removeAlert(i)}
|
||||
/>
|
||||
))}
|
||||
</ul>
|
||||
)}
|
||||
</div>
|
||||
|
||||
{toast && (
|
||||
<div className="toast" role="status">
|
||||
{toast}
|
||||
@@ -1271,342 +1128,6 @@ export default function DNS() {
|
||||
)
|
||||
}
|
||||
|
||||
// ---- alert add form + row --------------------------------------------------
|
||||
|
||||
function AddAlertForm({
|
||||
busy,
|
||||
disabled,
|
||||
taken,
|
||||
catalog,
|
||||
valid,
|
||||
onAdd,
|
||||
}: {
|
||||
busy: boolean
|
||||
disabled: boolean
|
||||
taken: Set<string>
|
||||
catalog: DetourCatalog
|
||||
valid: Set<string>
|
||||
onAdd: (a: Alert) => Promise<boolean>
|
||||
}) {
|
||||
const [name, setName] = useState('')
|
||||
const [type, setType] = useState<'telegram' | 'webhook'>('telegram')
|
||||
const [token, setToken] = useState('')
|
||||
const [chatId, setChatId] = useState('')
|
||||
const [url, setUrl] = useState('')
|
||||
const [events, setEvents] = useState<string[]>(['killswitch'])
|
||||
const [via, setVia] = useState('direct')
|
||||
const [fallback, setFallback] = useState(false)
|
||||
const [err, setErr] = useState<string | null>(null)
|
||||
|
||||
const reset = () => {
|
||||
setName('')
|
||||
setType('telegram')
|
||||
setToken('')
|
||||
setChatId('')
|
||||
setUrl('')
|
||||
setEvents(['killswitch'])
|
||||
setVia('direct')
|
||||
setFallback(false)
|
||||
}
|
||||
|
||||
const toggleEvent = (id: string) =>
|
||||
setEvents((prev) => (prev.includes(id) ? prev.filter((e) => e !== id) : [...prev, id]))
|
||||
|
||||
const submit = async () => {
|
||||
const nm = name.trim()
|
||||
if (!nm) {
|
||||
setErr('Give the alert a name.')
|
||||
return
|
||||
}
|
||||
if (taken.has(nm)) {
|
||||
setErr(`An alert named “${nm}” already exists.`)
|
||||
return
|
||||
}
|
||||
if (type === 'telegram') {
|
||||
if (!token.trim() || !chatId.trim()) {
|
||||
setErr('Telegram needs a bot token and a chat ID.')
|
||||
return
|
||||
}
|
||||
} else if (!HTTP_RE.test(url.trim())) {
|
||||
setErr('Enter an http(s):// webhook URL.')
|
||||
return
|
||||
}
|
||||
if (events.length === 0) {
|
||||
setErr('Pick at least one event to notify on.')
|
||||
return
|
||||
}
|
||||
setErr(null)
|
||||
const routed = via !== 'direct'
|
||||
const routing = { Via: routed ? via : undefined, Fallback: routed && fallback ? true : undefined }
|
||||
const draft: Alert =
|
||||
type === 'telegram'
|
||||
? { Name: nm, Enabled: true, Type: 'telegram', Token: token.trim(), ChatID: chatId.trim(), Events: events, ...routing }
|
||||
: { Name: nm, Enabled: true, Type: 'webhook', URL: url.trim(), Events: events, ...routing }
|
||||
const ok = await onAdd(draft)
|
||||
if (ok) reset()
|
||||
}
|
||||
|
||||
const routed = via !== 'direct'
|
||||
|
||||
return (
|
||||
<form
|
||||
className="dns-add"
|
||||
onSubmit={(e) => {
|
||||
e.preventDefault()
|
||||
void submit()
|
||||
}}
|
||||
>
|
||||
<div className="dns-add-top">
|
||||
<input
|
||||
className="dns-input dns-input--name"
|
||||
type="text"
|
||||
spellCheck={false}
|
||||
autoComplete="off"
|
||||
placeholder="Alert name"
|
||||
aria-label="Alert name"
|
||||
value={name}
|
||||
onChange={(e) => {
|
||||
setName(e.target.value)
|
||||
if (err) setErr(null)
|
||||
}}
|
||||
disabled={busy || disabled}
|
||||
/>
|
||||
<div className="dns-seg" role="group" aria-label="Alert type">
|
||||
<button
|
||||
type="button"
|
||||
className={type === 'telegram' ? 'dns-seg-btn on' : 'dns-seg-btn'}
|
||||
aria-pressed={type === 'telegram'}
|
||||
onClick={() => setType('telegram')}
|
||||
disabled={busy || disabled}
|
||||
>
|
||||
Telegram
|
||||
</button>
|
||||
<button
|
||||
type="button"
|
||||
className={type === 'webhook' ? 'dns-seg-btn on' : 'dns-seg-btn'}
|
||||
aria-pressed={type === 'webhook'}
|
||||
onClick={() => setType('webhook')}
|
||||
disabled={busy || disabled}
|
||||
>
|
||||
Webhook
|
||||
</button>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
{type === 'telegram' ? (
|
||||
<>
|
||||
<input
|
||||
className="dns-input"
|
||||
type="password"
|
||||
spellCheck={false}
|
||||
autoComplete="off"
|
||||
placeholder="Bot token (kept secret)"
|
||||
aria-label="Telegram bot token"
|
||||
value={token}
|
||||
onChange={(e) => {
|
||||
setToken(e.target.value)
|
||||
if (err) setErr(null)
|
||||
}}
|
||||
disabled={busy || disabled}
|
||||
/>
|
||||
<input
|
||||
className="dns-input"
|
||||
type="text"
|
||||
spellCheck={false}
|
||||
autoComplete="off"
|
||||
placeholder="Chat ID (e.g. -1001234567890)"
|
||||
aria-label="Telegram chat ID"
|
||||
value={chatId}
|
||||
onChange={(e) => {
|
||||
setChatId(e.target.value)
|
||||
if (err) setErr(null)
|
||||
}}
|
||||
disabled={busy || disabled}
|
||||
/>
|
||||
</>
|
||||
) : (
|
||||
<input
|
||||
className="dns-input"
|
||||
type="text"
|
||||
inputMode="url"
|
||||
spellCheck={false}
|
||||
autoComplete="off"
|
||||
placeholder="https://hooks.example.com/…"
|
||||
aria-label="Webhook URL"
|
||||
value={url}
|
||||
onChange={(e) => {
|
||||
setUrl(e.target.value)
|
||||
if (err) setErr(null)
|
||||
}}
|
||||
disabled={busy || disabled}
|
||||
/>
|
||||
)}
|
||||
|
||||
<fieldset className="dns-events" aria-label="Events to notify on">
|
||||
{ALERT_EVENTS.map((ev) => (
|
||||
<label key={ev.id} className="dns-event">
|
||||
<input
|
||||
type="checkbox"
|
||||
checked={events.includes(ev.id)}
|
||||
onChange={() => toggleEvent(ev.id)}
|
||||
disabled={busy || disabled}
|
||||
/>
|
||||
<span>{ev.label}</span>
|
||||
</label>
|
||||
))}
|
||||
</fieldset>
|
||||
|
||||
<div className="dns-alert-delivery">
|
||||
<label className="dns-resp">
|
||||
<span className="dns-resp-label mono">Deliver via</span>
|
||||
<DetourSelect
|
||||
value={via}
|
||||
catalog={catalog}
|
||||
valid={valid}
|
||||
busy={busy}
|
||||
disabled={disabled}
|
||||
ariaLabel="Deliver alert via"
|
||||
onChange={setVia}
|
||||
directLabel="Direct (default)"
|
||||
/>
|
||||
</label>
|
||||
<label className="dns-fallback">
|
||||
<Toggle
|
||||
pressed={fallback}
|
||||
onChange={setFallback}
|
||||
label={fallback ? 'Disable direct fallback' : 'Enable direct fallback'}
|
||||
disabled={busy || disabled || !routed}
|
||||
/>
|
||||
<span className="dns-fallback-label mono">Fallback to direct</span>
|
||||
</label>
|
||||
</div>
|
||||
{routed && !fallback && (
|
||||
<p className="dns-alert-note" role="note">
|
||||
{VIA_NO_FALLBACK_NOTE}
|
||||
</p>
|
||||
)}
|
||||
|
||||
<div className="dns-add-actions">
|
||||
{err && (
|
||||
<p className="dns-field-err" role="alert">
|
||||
{err}
|
||||
</p>
|
||||
)}
|
||||
<Button type="submit" variant="primary" disabled={busy || disabled}>
|
||||
{busy ? 'Saving…' : 'Add alert'}
|
||||
</Button>
|
||||
</div>
|
||||
</form>
|
||||
)
|
||||
}
|
||||
|
||||
function AlertRow({
|
||||
alert,
|
||||
busy,
|
||||
catalog,
|
||||
valid,
|
||||
onToggle,
|
||||
onVia,
|
||||
onFallback,
|
||||
onDelete,
|
||||
}: {
|
||||
alert: Alert
|
||||
busy: boolean
|
||||
catalog: DetourCatalog
|
||||
valid: Set<string>
|
||||
onToggle: (on: boolean) => void
|
||||
onVia: (v: string) => void
|
||||
onFallback: (on: boolean) => void
|
||||
onDelete: () => void
|
||||
}) {
|
||||
// Never render the token/URL in clear — show a masked descriptor only.
|
||||
const detail = useMemo(() => {
|
||||
if (alert.Type === 'telegram') {
|
||||
return { text: `chat ${alert.ChatID || '—'}`, masked: !!alert.Token }
|
||||
}
|
||||
const { host, masked } = maskUrl(alert.URL ?? '')
|
||||
return { text: host, masked: masked || !!alert.URL }
|
||||
}, [alert.Type, alert.ChatID, alert.Token, alert.URL])
|
||||
|
||||
const events = asArray(alert.Events)
|
||||
const canon = useMemo(() => canonDetour(alert.Via, catalog), [alert.Via, catalog])
|
||||
const route = useMemo(() => describeDetour(canon, catalog, valid), [canon, catalog, valid])
|
||||
const routed = canon !== 'direct'
|
||||
const fallback = alert.Fallback ?? false
|
||||
|
||||
return (
|
||||
<li className="dns-row">
|
||||
<Toggle
|
||||
pressed={alert.Enabled}
|
||||
onChange={onToggle}
|
||||
label={`${alert.Enabled ? 'Disable' : 'Enable'} alert ${alert.Name}`}
|
||||
disabled={busy}
|
||||
/>
|
||||
<div className="dns-row-main">
|
||||
<div className="dns-row-l1">
|
||||
<span className="dns-row-name">{alert.Name}</span>
|
||||
<span className="dns-badge">{alert.Type}</span>
|
||||
{events.map((e) => (
|
||||
<span key={e} className="dns-badge dns-badge--accent">
|
||||
{e}
|
||||
</span>
|
||||
))}
|
||||
</div>
|
||||
<div className="dns-row-l2 mono">
|
||||
<span className="dns-row-detail">{detail.text}</span>
|
||||
{detail.masked && (
|
||||
<span className="dns-masked" title="Secret is stored but hidden here">
|
||||
secret hidden
|
||||
</span>
|
||||
)}
|
||||
{route.direct ? (
|
||||
<span className="dns-path">direct</span>
|
||||
) : (
|
||||
<span className="dns-path" data-active="on" data-missing={route.missing ? 'y' : undefined}>
|
||||
{route.prefix} <strong className="dns-path-name">{route.name}</strong>
|
||||
{route.missing && <span className="dns-path-flag"> (missing)</span>}
|
||||
{fallback ? ' · +direct fallback' : ' · no fallback'}
|
||||
</span>
|
||||
)}
|
||||
</div>
|
||||
{routed && !fallback && <p className="dns-alert-note dns-alert-note--row">{VIA_NO_FALLBACK_NOTE}</p>}
|
||||
</div>
|
||||
<div className="dns-alert-ctl">
|
||||
<label className="dns-detour">
|
||||
<span className="dns-detour-label mono">Deliver via</span>
|
||||
<DetourSelect
|
||||
value={canon}
|
||||
catalog={catalog}
|
||||
valid={valid}
|
||||
busy={busy}
|
||||
disabled={false}
|
||||
ariaLabel={`Deliver alert ${alert.Name} via`}
|
||||
onChange={onVia}
|
||||
directLabel="Direct (default)"
|
||||
/>
|
||||
</label>
|
||||
<label className="dns-fallback">
|
||||
<Toggle
|
||||
pressed={fallback}
|
||||
onChange={onFallback}
|
||||
label={`${fallback ? 'Disable' : 'Enable'} direct fallback for ${alert.Name}`}
|
||||
disabled={busy || !routed}
|
||||
/>
|
||||
<span className="dns-fallback-label mono">Fallback to direct</span>
|
||||
</label>
|
||||
</div>
|
||||
<Button
|
||||
className="dns-del"
|
||||
onClick={onDelete}
|
||||
disabled={busy}
|
||||
aria-label={`Delete alert ${alert.Name}`}
|
||||
>
|
||||
Delete
|
||||
</Button>
|
||||
</li>
|
||||
)
|
||||
}
|
||||
|
||||
// ---- add form --------------------------------------------------------------
|
||||
|
||||
interface AddDraft {
|
||||
|
||||
@@ -196,6 +196,52 @@
|
||||
border-color: color-mix(in srgb, var(--amber) 55%, var(--groove));
|
||||
color: var(--amber);
|
||||
}
|
||||
/* "not built" — the saved switch says on and the engine has no such outbound. */
|
||||
.badge--crit {
|
||||
border-color: color-mix(in srgb, var(--crit) 55%, var(--groove));
|
||||
color: var(--crit);
|
||||
}
|
||||
|
||||
/* ---- last-apply findings, attached to the row they are about ----
|
||||
Sits under the row's own two lines, inside the row plate, so a node the
|
||||
generator threw away cannot read as an ordinary enabled node. Severity carries
|
||||
the colour; the accent stays reserved for controls. */
|
||||
.row-findings {
|
||||
margin: 6px 0 0;
|
||||
padding: 0;
|
||||
list-style: none;
|
||||
display: flex;
|
||||
flex-direction: column;
|
||||
gap: 5px;
|
||||
}
|
||||
.row-finding {
|
||||
display: flex;
|
||||
align-items: flex-start;
|
||||
gap: 8px;
|
||||
padding: 7px 9px;
|
||||
border: 1px solid color-mix(in srgb, var(--amber) 40%, var(--groove));
|
||||
border-radius: 6px;
|
||||
background: color-mix(in srgb, var(--sink) 35%, transparent);
|
||||
}
|
||||
.row-finding--critical {
|
||||
border-color: color-mix(in srgb, var(--crit) 45%, var(--groove));
|
||||
}
|
||||
.row-finding-msg {
|
||||
flex: 1;
|
||||
min-width: 0;
|
||||
font-size: 12px;
|
||||
line-height: 1.5;
|
||||
color: var(--ink);
|
||||
max-width: 82ch;
|
||||
overflow-wrap: anywhere;
|
||||
}
|
||||
/* Findings that belong to no single row (see Nodes.tsx globalFindings). */
|
||||
.node-findings {
|
||||
margin-bottom: calc(var(--u, 8px) * 2);
|
||||
}
|
||||
.node-findings .row-findings {
|
||||
margin-top: 0;
|
||||
}
|
||||
|
||||
/* masked-credential marker */
|
||||
.masked {
|
||||
@@ -641,6 +687,20 @@ select.fp-input {
|
||||
letter-spacing: 0.06em;
|
||||
color: var(--faint);
|
||||
}
|
||||
/* A collapsed bucket has to carry its own bad news: a 300-node subscription is
|
||||
closed by default, and the per-row findings inside it are otherwise unreachable
|
||||
without knowing to look. */
|
||||
.group-flagged {
|
||||
flex: none;
|
||||
display: inline-flex;
|
||||
align-items: center;
|
||||
gap: 6px;
|
||||
font-family: var(--font-mono);
|
||||
font-size: 10.5px;
|
||||
letter-spacing: 0.06em;
|
||||
text-transform: uppercase;
|
||||
color: var(--amber);
|
||||
}
|
||||
.group-rows {
|
||||
margin-top: 8px;
|
||||
}
|
||||
|
||||
+131
-2
@@ -6,12 +6,14 @@ import { Button, Led, Toggle, useConfirm } from '../components'
|
||||
import {
|
||||
apply as apiApply,
|
||||
getConfig,
|
||||
getStatus,
|
||||
putConfig,
|
||||
importWg,
|
||||
updateSubscription,
|
||||
ApiError,
|
||||
} from '../api'
|
||||
import type { Model, Node as NodeCfg, Subscription } from '../api'
|
||||
import type { Model, Node as NodeCfg, StatusWarning, Subscription } from '../api'
|
||||
import { entityFindings, findingsByName } from '../findings'
|
||||
import { fmtBytes, fmtDate, fmtUntil } from '../format'
|
||||
|
||||
// The whole page is a thin editor over the desired-state Model: every mutation
|
||||
@@ -519,6 +521,43 @@ export default function Nodes() {
|
||||
void loadConfig()
|
||||
}, [loadConfig])
|
||||
|
||||
// ---- what the last apply said about these nodes ---------------------------
|
||||
//
|
||||
// The generator drops a node it cannot build and names it: an unparseable
|
||||
// share link (generate/outbound.go), a name colliding with a reserved tag, a
|
||||
// WireGuard private key materialised twice (generate/wgdedup.go). Until now
|
||||
// this page never read /api/status, so a node the engine had thrown away
|
||||
// rendered as an ordinary row with a green toggle — the switch said on and
|
||||
// there was no such outbound anywhere in the running config.
|
||||
//
|
||||
// Findings are attached to the ROWS, not summarised at the top: a 300-node
|
||||
// subscription makes a list of names useless, and the row is where the false
|
||||
// reassurance was.
|
||||
const [findings, setFindings] = useState<StatusWarning[]>([])
|
||||
const loadFindings = useCallback(async () => {
|
||||
try {
|
||||
const s = await getStatus()
|
||||
setFindings(entityFindings(s.warnings, ['node', 'subscription']))
|
||||
} catch {
|
||||
// Status is a supplement here, not the page. Keep the last set rather than
|
||||
// clearing it — a dropped poll is not the same as "the problem is fixed".
|
||||
}
|
||||
}, [])
|
||||
useEffect(() => {
|
||||
void loadFindings()
|
||||
}, [loadFindings])
|
||||
|
||||
const nodeFindings = useMemo(
|
||||
() => findingsByName(findings.filter((w) => w.section === 'node')),
|
||||
[findings],
|
||||
)
|
||||
const subFindings = useMemo(
|
||||
() => findingsByName(findings.filter((w) => w.section === 'subscription')),
|
||||
[findings],
|
||||
)
|
||||
// Findings about nodes/subscriptions in general, which belong to no single row.
|
||||
const globalFindings = useMemo(() => findings.filter((w) => !w.name), [findings])
|
||||
|
||||
// ---- toast + persistent apply banner --------------------------------------
|
||||
const [toast, setToast] = useState<string | null>(null)
|
||||
const toastTimer = useRef<number | undefined>(undefined)
|
||||
@@ -573,8 +612,11 @@ export default function Nodes() {
|
||||
flash(`Apply failed — ${errText(e)}`)
|
||||
} finally {
|
||||
setApplying(false)
|
||||
// An apply is exactly what rewrites the findings — including clearing the
|
||||
// ones the operator just fixed.
|
||||
void loadFindings()
|
||||
}
|
||||
}, [flash, loadConfig])
|
||||
}, [flash, loadConfig, loadFindings])
|
||||
|
||||
// ---- node mutations -------------------------------------------------------
|
||||
const nodes = useMemo(() => asArray(config?.Nodes), [config])
|
||||
@@ -994,6 +1036,14 @@ export default function Nodes() {
|
||||
</div>
|
||||
)}
|
||||
|
||||
{/* Findings about nodes in general — no single row owns them, so they sit
|
||||
above the lists rather than being dropped for having no name. */}
|
||||
{globalFindings.length > 0 && (
|
||||
<div className="node-findings">
|
||||
<RowFindings findings={globalFindings} />
|
||||
</div>
|
||||
)}
|
||||
|
||||
{/* ---- NODES ---- */}
|
||||
<div className="node-section" aria-label="Nodes">
|
||||
<header className="sec-hd">
|
||||
@@ -1149,6 +1199,7 @@ export default function Nodes() {
|
||||
<NodeGroup
|
||||
key={g.key || '__manual__'}
|
||||
group={g}
|
||||
findings={nodeFindings}
|
||||
open={isGroupOpen(g)}
|
||||
busy={busy}
|
||||
egressNames={egressNames}
|
||||
@@ -1238,6 +1289,7 @@ export default function Nodes() {
|
||||
busy={busy}
|
||||
catalog={detourCatalog}
|
||||
valid={detourValid}
|
||||
findings={subFindings.get(s.Name) ?? EMPTY_FINDINGS}
|
||||
onToggle={(on) => toggleSub(i, on)}
|
||||
onDelete={() => removeSub(i)}
|
||||
onEdit={(patch) => editSub(i, patch)}
|
||||
@@ -1264,6 +1316,7 @@ function NodeGroup({
|
||||
open,
|
||||
busy,
|
||||
egressNames,
|
||||
findings,
|
||||
onToggle,
|
||||
onToggleNode,
|
||||
onRemoveNode,
|
||||
@@ -1274,6 +1327,8 @@ function NodeGroup({
|
||||
open: boolean
|
||||
busy: boolean
|
||||
egressNames: string[]
|
||||
/** Last-apply findings per node name (findings.ts findingsByName). */
|
||||
findings: Map<string, StatusWarning[]>
|
||||
onToggle: () => void
|
||||
onToggleNode: (idx: number, on: boolean) => void
|
||||
onRemoveNode: (idx: number) => void
|
||||
@@ -1283,6 +1338,13 @@ function NodeGroup({
|
||||
const panelId = `node-group-${group.key || 'manual'}`
|
||||
// The same inventory count as the section header, scoped to this bucket.
|
||||
const count = useMemo(() => fmtEnabled(group.items.map((i) => i.node)), [group.items])
|
||||
// How many nodes in this bucket the last apply had something to say about —
|
||||
// shown on the COLLAPSED header, because a subscription of 300 nodes is
|
||||
// collapsed by default and the row badge below would never be seen otherwise.
|
||||
const flagged = useMemo(
|
||||
() => group.items.filter(({ node }) => findings.has(node.Name)).length,
|
||||
[group.items, findings],
|
||||
)
|
||||
return (
|
||||
<section className={`node-group${open ? ' node-group--open' : ''}`}>
|
||||
<h3 className="group-hd-wrap">
|
||||
@@ -1296,6 +1358,12 @@ function NodeGroup({
|
||||
<span className="group-caret" aria-hidden="true" />
|
||||
<span className="group-name">{group.label}</span>
|
||||
<span className="group-count mono">{count}</span>
|
||||
{flagged > 0 && (
|
||||
<span className="group-flagged" title="Findings from the last apply">
|
||||
<Led variant="amber" />
|
||||
{flagged} flagged
|
||||
</span>
|
||||
)}
|
||||
</button>
|
||||
</h3>
|
||||
{open && group.key !== '' && (
|
||||
@@ -1312,6 +1380,7 @@ function NodeGroup({
|
||||
node={node}
|
||||
busy={busy}
|
||||
egressNames={egressNames}
|
||||
findings={findings.get(node.Name) ?? EMPTY_FINDINGS}
|
||||
onToggle={(on) => onToggleNode(idx, on)}
|
||||
onDelete={() => onRemoveNode(idx)}
|
||||
onRename={(name, onError) => onRenameNode(idx, name, onError)}
|
||||
@@ -1324,10 +1393,29 @@ function NodeGroup({
|
||||
)
|
||||
}
|
||||
|
||||
/** One shared empty array, so a clean row doesn't get a fresh identity per render. */
|
||||
const EMPTY_FINDINGS: StatusWarning[] = []
|
||||
|
||||
/**
|
||||
* Did the generator say it left this entity OUT of the engine config?
|
||||
*
|
||||
* The producers all end the sentence with the same word — "(skipped)" for an
|
||||
* unparseable share link or a bad WireGuard endpoint (generate/outbound.go),
|
||||
* "skipped" for a name colliding with a reserved tag — and wgdedup says only one
|
||||
* of the duplicates "is kept". Read the daemon's word rather than inventing a
|
||||
* verdict: a finding that does NOT say this may well be about a node that is
|
||||
* running perfectly, and badging it "not built" would be a new lie in place of
|
||||
* the old one.
|
||||
*/
|
||||
function skipped(findings: StatusWarning[]): boolean {
|
||||
return findings.some((f) => /\bskipped\b|\bis kept\b/i.test(f.message))
|
||||
}
|
||||
|
||||
function NodeRow({
|
||||
node,
|
||||
busy,
|
||||
egressNames,
|
||||
findings,
|
||||
onToggle,
|
||||
onDelete,
|
||||
onRename,
|
||||
@@ -1336,6 +1424,8 @@ function NodeRow({
|
||||
node: NodeCfg
|
||||
busy: boolean
|
||||
egressNames: string[]
|
||||
/** What the last apply said about THIS node; empty when it said nothing. */
|
||||
findings: StatusWarning[]
|
||||
onToggle: (on: boolean) => void
|
||||
onDelete: () => void
|
||||
onRename: (name: string, onError: (msg: string) => void) => Promise<boolean>
|
||||
@@ -1483,6 +1573,16 @@ function NodeRow({
|
||||
)}
|
||||
<span className="badge">{proto}</span>
|
||||
{node.Stale && <span className="badge badge--warn">stale</span>}
|
||||
{/* The toggle above is the SAVED state. When the last apply couldn't
|
||||
build this node the engine has no such outbound, and the two
|
||||
disagree — so the row says which, rather than leaving a green
|
||||
switch to imply the node is carrying traffic. The word is the
|
||||
daemon's own where it used one. */}
|
||||
{findings.length > 0 && (
|
||||
<span className={`badge badge--${skipped(findings) ? 'crit' : 'warn'}`}>
|
||||
{skipped(findings) ? 'not built' : 'flagged'}
|
||||
</span>
|
||||
)}
|
||||
</div>
|
||||
{renameErr && (
|
||||
<p className="row-err" role="alert">
|
||||
@@ -1506,6 +1606,7 @@ function NodeRow({
|
||||
</span>
|
||||
)}
|
||||
</div>
|
||||
<RowFindings findings={findings} />
|
||||
</div>
|
||||
<div className="row-actions">
|
||||
{canPin && (
|
||||
@@ -1625,6 +1726,7 @@ function SubRow({
|
||||
busy,
|
||||
catalog,
|
||||
valid,
|
||||
findings,
|
||||
onToggle,
|
||||
onDelete,
|
||||
onEdit,
|
||||
@@ -1635,6 +1737,8 @@ function SubRow({
|
||||
busy: boolean
|
||||
catalog: DetourCatalog
|
||||
valid: Set<string>
|
||||
/** What the last apply said about THIS subscription; empty when it said nothing. */
|
||||
findings: StatusWarning[]
|
||||
onToggle: (on: boolean) => void
|
||||
onDelete: () => void
|
||||
onEdit: (patch: Subscription) => Promise<boolean>
|
||||
@@ -1659,6 +1763,7 @@ function SubRow({
|
||||
<span className="row-name">{sub.Name}</span>
|
||||
{sub.Format && sub.Format !== 'auto' && <span className="badge">{sub.Format}</span>}
|
||||
{sub.FetchVia === 'proxy' && <span className="badge">via proxy</span>}
|
||||
{findings.length > 0 && <span className="badge badge--warn">flagged</span>}
|
||||
</div>
|
||||
<div className="row-line2 mono">
|
||||
<span className="row-host">{host}</span>
|
||||
@@ -1671,6 +1776,7 @@ function SubRow({
|
||||
every {interval} · {count} node{count === 1 ? '' : 's'}
|
||||
</span>
|
||||
</div>
|
||||
<RowFindings findings={findings} />
|
||||
</div>
|
||||
<div className="row-actions">
|
||||
<Button
|
||||
@@ -2142,6 +2248,29 @@ function HeaderRows({
|
||||
)
|
||||
}
|
||||
|
||||
/**
|
||||
* What the last apply said about THIS row, under the row it is about.
|
||||
*
|
||||
* Deliberately inside the row rather than in a list at the top of the page: the
|
||||
* failure being fixed is a node that looks fine, and a name in a summary three
|
||||
* screens up does not fix that. The wording is the daemon's own — these messages
|
||||
* already name the entity and say what was done about it ("(skipped)", "only X
|
||||
* is kept"), so paraphrasing them here would only invent a second vocabulary.
|
||||
*/
|
||||
function RowFindings({ findings }: { findings: StatusWarning[] }) {
|
||||
if (findings.length === 0) return null
|
||||
return (
|
||||
<ul className="row-findings" aria-label="Findings from the last apply">
|
||||
{findings.map((f, i) => (
|
||||
<li key={i} className={`row-finding row-finding--${f.severity}`}>
|
||||
<Led variant={f.severity === 'critical' ? 'crit' : 'amber'} />
|
||||
<span className="row-finding-msg">{f.message}</span>
|
||||
</li>
|
||||
))}
|
||||
</ul>
|
||||
)
|
||||
}
|
||||
|
||||
function EmptyPlate({ title, body }: { title: string; body: string }) {
|
||||
return (
|
||||
<div className="empty-plate">
|
||||
|
||||
@@ -14,8 +14,8 @@ import type { Model, Stats, Status, StatusWarning } from '../api'
|
||||
import { confirmTimeout } from '../pendingConfirm'
|
||||
import { navigate } from '../router'
|
||||
import type { Route } from '../router'
|
||||
import { attentionFindings } from '../findings'
|
||||
import { engineReadout, protectionState } from '../planeState'
|
||||
import { attentionFindings, truncationNote } from '../findings'
|
||||
import { engineReadout, killSwitchReadout, protectionState } from '../planeState'
|
||||
|
||||
// null-safe length for a Go slice that may arrive as null.
|
||||
const len = (a: unknown[] | null | undefined): number => (a ? a.length : 0)
|
||||
@@ -309,15 +309,19 @@ export function Overview({
|
||||
const engineVariant: LedVariant = engine.variant
|
||||
|
||||
const protection = protectionState(status)
|
||||
// Configured fail-closed AND actually enforcing it. `none` means nothing is
|
||||
// installed, so the setting is inert no matter what it says.
|
||||
const killInEffect = killArmed && status?.plane !== 'none'
|
||||
// Configured fail-closed, actually enforcing it, or not known — three answers,
|
||||
// and the third is not folded into the first. See planeState.killSwitchReadout.
|
||||
const kill = killSwitchReadout(status, g?.KillSwitch)
|
||||
|
||||
// Findings that need attention. `info` notes are statements about the config,
|
||||
// not problems, so they live beside the setting they describe (see findings.ts)
|
||||
// — keeping this list to things someone could actually act on.
|
||||
const warnings = attentionFindings(status?.warnings)
|
||||
const criticalCount = warnings.filter((w) => w.severity === 'critical').length
|
||||
// The daemon caps the published list at 50 and says so in an `info` note — the
|
||||
// one channel this page filters away. Carried separately so the list can admit
|
||||
// it is not the whole list. See findings.ts truncationNote.
|
||||
const truncated = truncationNote(status?.warnings)
|
||||
|
||||
return (
|
||||
<section className="page" aria-label="Overview">
|
||||
@@ -340,7 +344,7 @@ export function Overview({
|
||||
</p>
|
||||
)}
|
||||
|
||||
<Findings warnings={warnings} criticalCount={criticalCount} />
|
||||
<Findings warnings={warnings} criticalCount={criticalCount} truncated={truncated} />
|
||||
|
||||
<div className="grid">
|
||||
{/* Groups, not nodes: a group is where a dial path is defined, so it is the
|
||||
@@ -447,17 +451,17 @@ export function Overview({
|
||||
/>
|
||||
|
||||
{/* A kill-switch set to fail-closed is only ARMED if something is actually
|
||||
installed to enforce it. With no plane it is configured but inert, and
|
||||
saying "ARMED" there would be a false reassurance next to a readout
|
||||
that says nothing is protected. */}
|
||||
installed to enforce it, and "we haven't been told" is neither. With no
|
||||
plane it is configured but inert; with no reading the lamp stays unlit
|
||||
rather than joining the healthy branch by default. */}
|
||||
<Module
|
||||
name="Kill-switch"
|
||||
value={killInEffect ? 'ARMED' : killArmed ? 'NOT IN EFFECT' : 'OPEN'}
|
||||
led={{ variant: killInEffect ? 'on' : killArmed ? 'crit' : 'amber' }}
|
||||
value={kill.value}
|
||||
led={{ variant: kill.variant }}
|
||||
rows={[
|
||||
{ k: 'setting', v: killArmed ? 'fail-closed' : 'fail-open', hot: !killArmed },
|
||||
...(killArmed && !killInEffect
|
||||
? [{ k: 'blocking now', v: 'no — nothing installed', hot: true }]
|
||||
...(kill.blockingNow
|
||||
? [{ k: 'blocking now', v: kill.blockingNow, hot: kill.hot }]
|
||||
: [{ k: 'ipv6', v: g?.IPv6 ? 'covered' : 'off' }]),
|
||||
{ k: 'confirm', v: g?.ConfirmTimeout ? `${g.ConfirmTimeout}s window` : 'no auto-rollback' },
|
||||
]}
|
||||
@@ -536,11 +540,23 @@ const SECTION_ROUTE: Record<string, Route> = {
|
||||
rule: 'routing',
|
||||
ruleset: 'routing',
|
||||
blocklist: 'dns',
|
||||
allowlist: 'dns',
|
||||
resolver: 'dns',
|
||||
dns_rule: 'dns',
|
||||
device: 'devices',
|
||||
chain: 'targets',
|
||||
group: 'targets',
|
||||
// A node the generator dropped (unparseable share link, duplicate WireGuard
|
||||
// key, name colliding with a reserved tag) is reported under `node` — and had
|
||||
// nowhere to jump to, so the one page that could show it a green toggle was
|
||||
// also the one page the finding could not reach.
|
||||
node: 'nodes',
|
||||
subscription: 'nodes',
|
||||
egress: 'targets',
|
||||
inbound: 'networks',
|
||||
interface: 'networks',
|
||||
profile: 'profiles',
|
||||
alert: 'settings',
|
||||
// The standing note about non-TCP/UDP traffic — its control lives on Networks.
|
||||
untunnelable: 'networks',
|
||||
}
|
||||
@@ -558,11 +574,14 @@ const SECTION_ROUTE: Record<string, Route> = {
|
||||
function Findings({
|
||||
warnings,
|
||||
criticalCount,
|
||||
truncated,
|
||||
}: {
|
||||
warnings: StatusWarning[]
|
||||
criticalCount: number
|
||||
/** The daemon's "N further suppressed" note, when the list was capped. */
|
||||
truncated: StatusWarning | null
|
||||
}) {
|
||||
if (warnings.length === 0) return null
|
||||
if (warnings.length === 0 && !truncated) return null
|
||||
|
||||
const rank = { critical: 0, warning: 1, info: 2 } as const
|
||||
const sorted = [...warnings].sort((a, b) => rank[a.severity] - rank[b.severity])
|
||||
@@ -572,6 +591,10 @@ function Findings({
|
||||
<header className="findings-hd">
|
||||
<h2 className="findings-title">Last apply</h2>
|
||||
<span className="findings-count mono">
|
||||
{/* "at least" whenever the list was capped: the counts below it are a
|
||||
floor, not a total, and the cap drops the least severe FIRST — so
|
||||
on a config with fifty criticals the thing it drops is a critical. */}
|
||||
{truncated ? 'at least ' : ''}
|
||||
{criticalCount > 0
|
||||
? `${criticalCount} critical · ${warnings.length} total`
|
||||
: `${warnings.length} note${warnings.length === 1 ? '' : 's'}`}
|
||||
@@ -610,6 +633,20 @@ function Findings({
|
||||
</li>
|
||||
)
|
||||
})}
|
||||
{/* The list saying it is not the whole list. Last, because it is about
|
||||
everything above it — and never filtered out with the other `info`
|
||||
notes, which is where it used to disappear. */}
|
||||
{truncated && (
|
||||
<li className="finding finding--truncated">
|
||||
<Led variant="amber" />
|
||||
<div className="finding-copy">
|
||||
<span className="finding-where mono">list truncated</span>
|
||||
<span className="finding-msg">
|
||||
Some findings are missing from this list. {truncated.message}
|
||||
</span>
|
||||
</div>
|
||||
</li>
|
||||
)}
|
||||
</ul>
|
||||
</section>
|
||||
)
|
||||
|
||||
@@ -2,6 +2,7 @@ import './Settings.css'
|
||||
import { useCallback, useEffect, useRef, useState } from 'react'
|
||||
import type { ReactNode } from 'react'
|
||||
import { Button, Led, Select, Toggle, useConfirm } from '../components'
|
||||
import { AlertsSection } from './Alerts'
|
||||
import { apply as apiApply, downloadLog, getConfig, putConfig, ApiError } from '../api'
|
||||
import type { Globals, LogRange, Model } from '../api'
|
||||
|
||||
@@ -406,7 +407,7 @@ export default function Settings() {
|
||||
|
||||
<Field
|
||||
label="Log level"
|
||||
note="Verbosity of the daemon log. “none” silences the engine and drops the control-plane to panic-only — a turn-down, not a true off: even warnings and errors are hidden. The toggles below decide where whatever is emitted gets written; turning both off is the only full silence. Failures still raise alerts regardless of this level."
|
||||
note="Verbosity of the daemon log. “none” silences the engine and drops the control-plane to panic-only — a turn-down, not a true off: even warnings and errors are hidden. The toggles below decide where whatever is emitted gets written; turning both off is the only full silence. Failures still raise alerts regardless of this level — set up where they go in the Alerts section below."
|
||||
>
|
||||
<Select
|
||||
value={globals?.LogLevel || 'warning'}
|
||||
@@ -620,6 +621,13 @@ export default function Settings() {
|
||||
</Field>
|
||||
</Group>
|
||||
|
||||
{/* ---- ALERTS ---- */}
|
||||
{/* Extracted from the DNS page — out-of-band notifications belong with
|
||||
the appliance-wide knobs, next to the log level whose note points
|
||||
here. Renders its own section header (same plate as a Group); all
|
||||
writes go through `save`, so the dirty banner and toast stay one. */}
|
||||
<AlertsSection config={config} busy={busy} loading={loading} onSave={save} />
|
||||
|
||||
{/* ---- STATISTICS & LOGGING ---- */}
|
||||
<Group
|
||||
title="Statistics & logging"
|
||||
|
||||
@@ -15,7 +15,7 @@
|
||||
import { test } from 'node:test'
|
||||
import assert from 'node:assert/strict'
|
||||
|
||||
import { engineReadout, engineState, protectionState } from './planeState.ts'
|
||||
import { engineReadout, engineState, killSwitchReadout, protectionState } from './planeState.ts'
|
||||
import type { Status, Traffic } from './api.ts'
|
||||
|
||||
/** A healthy, fully-installed router; `traffic` is what each case varies. */
|
||||
@@ -189,3 +189,65 @@ test('no status at all is unknown, not down', () => {
|
||||
assert.equal(engineReadout(null).variant, 'off')
|
||||
assert.equal(engineReadout(null).word, 'checking…')
|
||||
})
|
||||
|
||||
// --- killSwitchReadout: "I don't know" is not "it's armed" -------------------
|
||||
//
|
||||
// The Overview module read `killArmed && status?.plane !== 'none'`, and
|
||||
// `undefined !== 'none'` is true — so a daemon that never reported `plane`, and
|
||||
// the seconds before the first status arrives, both lit a green lamp over the
|
||||
// word ARMED. These pin the fourth answer that expression could not express.
|
||||
|
||||
test('a daemon that does not report `plane` reads as not reported, never ARMED', () => {
|
||||
const { plane, ...noPlane } = status()
|
||||
void plane
|
||||
const k = killSwitchReadout(noPlane as Status)
|
||||
assert.equal(k.state, 'unknown')
|
||||
assert.notEqual(k.value, 'ARMED')
|
||||
assert.equal(k.variant, 'off')
|
||||
assert.notEqual(k.variant, 'on')
|
||||
assert.equal(k.blockingNow, 'not known')
|
||||
})
|
||||
|
||||
test('no status at all is unknown too, and says there is no reading', () => {
|
||||
const k = killSwitchReadout(null)
|
||||
assert.equal(k.state, 'unknown')
|
||||
assert.equal(k.variant, 'off')
|
||||
assert.equal(k.blockingNow, 'no reading yet')
|
||||
})
|
||||
|
||||
test('fail-closed with a plane installed is armed', () => {
|
||||
for (const plane of ['full', 'hold'] as const) {
|
||||
const k = killSwitchReadout(status({ plane }))
|
||||
assert.equal(k.state, 'armed')
|
||||
assert.equal(k.value, 'ARMED')
|
||||
assert.equal(k.variant, 'on')
|
||||
assert.equal(k.blockingNow, null)
|
||||
}
|
||||
})
|
||||
|
||||
test('fail-closed with no plane is configured but blocking nothing', () => {
|
||||
const k = killSwitchReadout(status({ plane: 'none', table: false }))
|
||||
assert.equal(k.state, 'inert')
|
||||
assert.equal(k.value, 'NOT IN EFFECT')
|
||||
assert.equal(k.variant, 'crit')
|
||||
assert.equal(k.hot, true)
|
||||
})
|
||||
|
||||
test('fail-open is the operator’s choice — amber, and never a plane question', () => {
|
||||
for (const plane of ['full', 'none', undefined] as const) {
|
||||
const k = killSwitchReadout(status({ kill_switch: 'open', plane }))
|
||||
assert.equal(k.state, 'open')
|
||||
assert.equal(k.value, 'OPEN')
|
||||
assert.equal(k.variant, 'amber')
|
||||
}
|
||||
})
|
||||
|
||||
test('the live kill_switch wins over the saved one; the saved one only fills a gap', () => {
|
||||
const { kill_switch, ...noKill } = status()
|
||||
void kill_switch
|
||||
// Live says open, config says closed → live wins.
|
||||
assert.equal(killSwitchReadout(status({ kill_switch: 'open' }), 'closed').state, 'open')
|
||||
// Nothing live → fall back to the saved policy.
|
||||
assert.equal(killSwitchReadout(noKill as Status, 'open').state, 'open')
|
||||
assert.equal(killSwitchReadout(noKill as Status, 'closed').state, 'armed')
|
||||
})
|
||||
|
||||
@@ -71,6 +71,85 @@ export function engineReadout(status: Status | null): { variant: LedVariant; wor
|
||||
}
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Is the kill-switch actually blocking anything?
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
/**
|
||||
* Four answers, and "unknown" is one of them.
|
||||
*
|
||||
* armed — configured fail-closed AND a data plane is installed to enforce it.
|
||||
* inert — configured fail-closed, but there is no plane. Nothing is blocking.
|
||||
* unknown — the daemon has not said how much plane is installed, so whether the
|
||||
* setting is in force is not known. NEVER paint this green.
|
||||
* open — configured fail-open. Nothing is meant to be blocked.
|
||||
*/
|
||||
export type KillSwitchState = 'armed' | 'inert' | 'unknown' | 'open'
|
||||
|
||||
export interface KillSwitchReadout {
|
||||
state: KillSwitchState
|
||||
/** The word the module puts in its readout. */
|
||||
value: string
|
||||
variant: LedVariant
|
||||
/** Is it blocking right now — the row under the readout. `null` ⇒ nothing to add. */
|
||||
blockingNow: string | null
|
||||
/** True when `blockingNow` is bad news and should be drawn hot. */
|
||||
hot: boolean
|
||||
}
|
||||
|
||||
/**
|
||||
* THE UNKNOWN BRANCH IS THE WHOLE POINT. This used to be
|
||||
*
|
||||
* killArmed && status?.plane !== 'none'
|
||||
*
|
||||
* and `undefined !== 'none'` is true — so a daemon that had not reported `plane`
|
||||
* at all, and a panel that had not yet received its first status, both landed in
|
||||
* the "ARMED" branch under a green lamp. Every other unknown in this file is an
|
||||
* unlit socket for exactly this reason (see engineReadout): the kill-switch is
|
||||
* the last thing standing between the LAN and the plain WAN, and "I don't know
|
||||
* whether it is installed" must never be dressed as "it is".
|
||||
*
|
||||
* `configured` is the SAVED policy from /api/config, used only while
|
||||
* /api/status has not reported one. The live value wins wherever it exists, as
|
||||
* everywhere else in the panel: this is a status readout, and the config on disk
|
||||
* can already differ from what is installed.
|
||||
*/
|
||||
export function killSwitchReadout(
|
||||
status: Status | null,
|
||||
configured?: string,
|
||||
): KillSwitchReadout {
|
||||
const closed = (status?.kill_switch ?? configured ?? 'closed') === 'closed'
|
||||
|
||||
if (!closed) {
|
||||
return { state: 'open', value: 'OPEN', variant: 'amber', blockingNow: null, hot: false }
|
||||
}
|
||||
|
||||
switch (status?.plane) {
|
||||
case 'full':
|
||||
case 'hold':
|
||||
// Something is installed, so the fail-closed guard is really in the path.
|
||||
return { state: 'armed', value: 'ARMED', variant: 'on', blockingNow: null, hot: false }
|
||||
case 'none':
|
||||
return {
|
||||
state: 'inert',
|
||||
value: 'NOT IN EFFECT',
|
||||
variant: 'crit',
|
||||
blockingNow: 'no — nothing installed',
|
||||
hot: true,
|
||||
}
|
||||
default:
|
||||
return {
|
||||
state: 'unknown',
|
||||
value: 'NOT REPORTED',
|
||||
variant: 'off',
|
||||
// Terse on purpose: this is a two-column readout row, and the long form
|
||||
// wrapped onto three lines beside a one-word key.
|
||||
blockingNow: status ? 'not known' : 'no reading yet',
|
||||
hot: false,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* `plane` + `engine_running` express the state more precisely than the three
|
||||
* booleans the old status strip exposed (engine active / config enabled / nft
|
||||
|
||||
+48
-1
@@ -1,15 +1,62 @@
|
||||
import { defineConfig } from 'vite'
|
||||
import type { Plugin } from 'vite'
|
||||
import react from '@vitejs/plugin-react'
|
||||
|
||||
// Minimal ambient for the dev-proxy target override — avoids pulling in @types/node
|
||||
// just for one env read. Vite runs this file under Node where `process` exists.
|
||||
declare const process: { env: Record<string, string | undefined> }
|
||||
|
||||
/** `src/mock.ts`, as the module graph spells it (POSIX-normalised for Windows). */
|
||||
const MOCK_MODULE = 'src/mock.ts'
|
||||
|
||||
/**
|
||||
* Refuse to emit a production bundle that contains the offline fixture backend.
|
||||
*
|
||||
* `src/mock.ts` describes an invented, healthy router: a full config, 122 nodes
|
||||
* with 119 of them alive, "Protected". It exists so `npm run dev` renders without
|
||||
* a daemon. It shipped inside the binary that goes on real hardware, switched on
|
||||
* by nothing more than a `?dev` on the end of the URL — so a link someone was
|
||||
* sent, or a bookmark they saved, showed an appliance in perfect health while
|
||||
* making no request to the appliance at all.
|
||||
*
|
||||
* api.ts now loads it behind `import.meta.env.DEV`, which Vite folds to a literal
|
||||
* `false` for a build, so Rollup drops the dynamic import and the module never
|
||||
* enters the graph. That is a property of a build tool's optimiser, and an
|
||||
* optimiser is not a promise: one refactor that makes the condition non-static
|
||||
* silently puts the fixtures back. So the property is CHECKED rather than
|
||||
* trusted — if `src/mock.ts` reaches any emitted chunk, the build fails here
|
||||
* instead of shipping.
|
||||
*/
|
||||
function assertNoMockFixtures(): Plugin {
|
||||
return {
|
||||
name: 'shater:assert-no-mock-fixtures',
|
||||
apply: 'build',
|
||||
generateBundle(_options, bundle) {
|
||||
const guilty: string[] = []
|
||||
for (const [file, output] of Object.entries(bundle)) {
|
||||
if (output.type !== 'chunk') continue
|
||||
for (const id of output.moduleIds) {
|
||||
if (id.replace(/\\/g, '/').endsWith(MOCK_MODULE)) guilty.push(`${file} ← ${id}`)
|
||||
}
|
||||
}
|
||||
if (guilty.length > 0) {
|
||||
this.error(
|
||||
`the offline fixture backend (${MOCK_MODULE}) reached the production bundle:\n ` +
|
||||
guilty.join('\n ') +
|
||||
`\nFixtures describe a router that does not exist. Keep every path to them behind ` +
|
||||
`\`import.meta.env.DEV\` so Rollup can drop them, and never gate them on a runtime ` +
|
||||
`flag such as a query parameter.`,
|
||||
)
|
||||
}
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
// The SPA is embedded in the forked sing-box binary and served by the daemon on
|
||||
// its own port. Relative base so it works under any mount path; single small
|
||||
// bundle (no code-splitting) keeps the embed simple and the flash budget low.
|
||||
export default defineConfig({
|
||||
plugins: [react()],
|
||||
plugins: [react(), assertNoMockFixtures()],
|
||||
base: './',
|
||||
build: {
|
||||
outDir: 'dist',
|
||||
|
||||
@@ -0,0 +1,194 @@
|
||||
package route
|
||||
|
||||
import (
|
||||
"context"
|
||||
"net"
|
||||
"net/netip"
|
||||
"testing"
|
||||
|
||||
"github.com/sagernet/sing-box/adapter"
|
||||
C "github.com/sagernet/sing-box/constant"
|
||||
"github.com/sagernet/sing-box/log"
|
||||
"github.com/sagernet/sing-box/option"
|
||||
R "github.com/sagernet/sing-box/route/rule"
|
||||
"github.com/sagernet/sing/common/json/badoption"
|
||||
M "github.com/sagernet/sing/common/metadata"
|
||||
N "github.com/sagernet/sing/common/network"
|
||||
|
||||
"github.com/stretchr/testify/require"
|
||||
)
|
||||
|
||||
// The pre-match path (adapter.JudgeFlow -> Router.PreMatch, used by the TUN/
|
||||
// WireGuard-endpoint flow dispatcher) used to prepare only fakeip and the IP
|
||||
// version. Everything a rule matches on that the inbound cannot know — the
|
||||
// connection owner and the neighbor behind the source address — was resolved in
|
||||
// matchRule only, so a `source_mac_address` / `source_hostname` rule silently
|
||||
// failed to match in pre-match and the flow fell through to the default
|
||||
// outbound. Upstream b911fb078 shares one prepareMatchMetadata between both
|
||||
// paths; these tests pin that.
|
||||
|
||||
// stubNeighborResolver answers for exactly one address.
|
||||
type stubNeighborResolver struct {
|
||||
address netip.Addr
|
||||
mac net.HardwareAddr
|
||||
hostname string
|
||||
}
|
||||
|
||||
func (r *stubNeighborResolver) LookupMAC(address netip.Addr) (net.HardwareAddr, bool) {
|
||||
if address != r.address || r.mac == nil {
|
||||
return nil, false
|
||||
}
|
||||
return r.mac, true
|
||||
}
|
||||
|
||||
func (r *stubNeighborResolver) LookupHostname(address netip.Addr) (string, bool) {
|
||||
if address != r.address || r.hostname == "" {
|
||||
return "", false
|
||||
}
|
||||
return r.hostname, true
|
||||
}
|
||||
|
||||
func (r *stubNeighborResolver) LookupAddresses(hostname string) []netip.Addr {
|
||||
if hostname != r.hostname {
|
||||
return nil
|
||||
}
|
||||
return []netip.Addr{r.address}
|
||||
}
|
||||
|
||||
func (r *stubNeighborResolver) Start() error { return nil }
|
||||
func (r *stubNeighborResolver) Close() error { return nil }
|
||||
|
||||
// stubDNSRouter / stubDNSTransportManager implement only what
|
||||
// prepareMatchMetadata reaches; every other method is left to the embedded nil
|
||||
// interface and would panic if it were ever called.
|
||||
type stubDNSRouter struct {
|
||||
adapter.DNSRouter
|
||||
}
|
||||
|
||||
func (s *stubDNSRouter) LookupReverseMapping(netip.Addr) (string, bool) { return "", false }
|
||||
|
||||
type stubDNSTransportManager struct {
|
||||
adapter.DNSTransportManager
|
||||
}
|
||||
|
||||
func (s *stubDNSTransportManager) FakeIP() adapter.FakeIPTransport { return nil }
|
||||
|
||||
// stubOutboundManager's default outbound supports no network at all, so a flow
|
||||
// that reaches preMatchFlow bails out with PreMatchContinue instead of nil-
|
||||
// dereferencing. That is exactly the pre-fix verdict we assert against.
|
||||
type stubOutboundManager struct {
|
||||
adapter.OutboundManager
|
||||
defaultOutbound adapter.Outbound
|
||||
}
|
||||
|
||||
func (s *stubOutboundManager) Default() adapter.Outbound { return s.defaultOutbound }
|
||||
|
||||
type stubNoNetworkOutbound struct {
|
||||
adapter.Outbound
|
||||
}
|
||||
|
||||
func (o *stubNoNetworkOutbound) Tag() string { return "stub" }
|
||||
func (o *stubNoNetworkOutbound) Type() string { return "direct" }
|
||||
func (o *stubNoNetworkOutbound) Network() []string { return nil }
|
||||
|
||||
func newPreMatchTestRouter(t *testing.T, resolver adapter.NeighborResolver, rules ...option.Rule) *Router {
|
||||
t.Helper()
|
||||
logger := log.NewNOPFactory().NewLogger("test")
|
||||
router := &Router{
|
||||
ctx: context.Background(),
|
||||
logger: logger,
|
||||
dns: &stubDNSRouter{},
|
||||
dnsTransport: &stubDNSTransportManager{},
|
||||
outbound: &stubOutboundManager{defaultOutbound: &stubNoNetworkOutbound{}},
|
||||
neighborResolver: resolver,
|
||||
needFindNeighbor: true,
|
||||
}
|
||||
for i, ruleOptions := range rules {
|
||||
rule, err := R.NewRule(router.ctx, logger, ruleOptions, false)
|
||||
require.NoError(t, err, "build rule[%d]", i)
|
||||
router.rules = append(router.rules, rule)
|
||||
}
|
||||
return router
|
||||
}
|
||||
|
||||
func rejectOnSourceMAC(macAddress string) option.Rule {
|
||||
return option.Rule{
|
||||
Type: C.RuleTypeDefault,
|
||||
DefaultOptions: option.DefaultRule{
|
||||
RawDefaultRule: option.RawDefaultRule{
|
||||
SourceMACAddress: badoption.Listable[string]{macAddress},
|
||||
},
|
||||
RuleAction: option.RuleAction{
|
||||
Action: C.RuleActionTypeReject,
|
||||
RejectOptions: option.RejectActionOptions{Method: C.RuleActionRejectMethodDefault},
|
||||
},
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
func rejectOnSourceHostname(hostname string) option.Rule {
|
||||
return option.Rule{
|
||||
Type: C.RuleTypeDefault,
|
||||
DefaultOptions: option.DefaultRule{
|
||||
RawDefaultRule: option.RawDefaultRule{
|
||||
SourceHostname: badoption.Listable[string]{hostname},
|
||||
},
|
||||
RuleAction: option.RuleAction{
|
||||
Action: C.RuleActionTypeReject,
|
||||
RejectOptions: option.RejectActionOptions{Method: C.RuleActionRejectMethodDefault},
|
||||
},
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
func preMatchMetadata() adapter.InboundContext {
|
||||
return adapter.InboundContext{
|
||||
Inbound: "tun-in",
|
||||
InboundType: C.TypeTun,
|
||||
Network: N.NetworkUDP,
|
||||
Source: M.ParseSocksaddr("192.168.1.5:41234"),
|
||||
Destination: M.ParseSocksaddr("1.1.1.1:443"),
|
||||
}
|
||||
}
|
||||
|
||||
func TestPreMatchResolvesNeighborMAC(t *testing.T) {
|
||||
t.Parallel()
|
||||
mac, err := net.ParseMAC("de:ad:be:ef:00:01")
|
||||
require.NoError(t, err)
|
||||
resolver := &stubNeighborResolver{
|
||||
address: netip.MustParseAddr("192.168.1.5"),
|
||||
mac: mac,
|
||||
hostname: "kitchen-tv",
|
||||
}
|
||||
router := newPreMatchTestRouter(t, resolver, rejectOnSourceMAC("de:ad:be:ef:00:01"))
|
||||
result := router.PreMatch(preMatchMetadata(), nil)
|
||||
require.Equal(t, adapter.PreMatchReject, result.Action,
|
||||
"source_mac_address rule must match in pre-match; the MAC has to be resolved there too")
|
||||
}
|
||||
|
||||
func TestPreMatchResolvesNeighborHostname(t *testing.T) {
|
||||
t.Parallel()
|
||||
resolver := &stubNeighborResolver{
|
||||
address: netip.MustParseAddr("192.168.1.5"),
|
||||
hostname: "kitchen-tv",
|
||||
}
|
||||
router := newPreMatchTestRouter(t, resolver, rejectOnSourceHostname("kitchen-tv"))
|
||||
result := router.PreMatch(preMatchMetadata(), nil)
|
||||
require.Equal(t, adapter.PreMatchReject, result.Action,
|
||||
"source_hostname rule must match in pre-match; the hostname has to be resolved there too")
|
||||
}
|
||||
|
||||
// A source the neighbor resolver does not know must still fall through, not
|
||||
// match on a half-filled metadata.
|
||||
func TestPreMatchNeighborMissDoesNotMatch(t *testing.T) {
|
||||
t.Parallel()
|
||||
mac, err := net.ParseMAC("de:ad:be:ef:00:01")
|
||||
require.NoError(t, err)
|
||||
resolver := &stubNeighborResolver{
|
||||
address: netip.MustParseAddr("192.168.1.9"),
|
||||
mac: mac,
|
||||
}
|
||||
router := newPreMatchTestRouter(t, resolver, rejectOnSourceMAC("de:ad:be:ef:00:01"))
|
||||
result := router.PreMatch(preMatchMetadata(), nil)
|
||||
require.Equal(t, adapter.PreMatchContinue, result.Action)
|
||||
}
|
||||
+28
-25
@@ -319,22 +319,14 @@ func (r *Router) PreMatch(metadata adapter.InboundContext, firstPacket []byte) a
|
||||
metadata.PreMatch = true
|
||||
continueResult := adapter.PreMatchResult{Action: adapter.PreMatchContinue}
|
||||
packetDestination := metadata.Destination
|
||||
if metadata.Destination.Addr.IsValid() && r.dnsTransport.FakeIP() != nil && r.dnsTransport.FakeIP().Store().Contains(metadata.Destination.Addr) {
|
||||
domain, loaded := r.dnsTransport.FakeIP().Store().Lookup(metadata.Destination.Addr)
|
||||
if !loaded || domain == "" {
|
||||
return continueResult
|
||||
}
|
||||
metadata.OriginDestination = metadata.Destination
|
||||
metadata.Destination = M.Socksaddr{
|
||||
Fqdn: domain,
|
||||
Port: metadata.Destination.Port,
|
||||
}
|
||||
metadata.FakeIP = true
|
||||
}
|
||||
if metadata.Destination.IsIPv4() {
|
||||
metadata.IPVersion = 4
|
||||
} else if metadata.Destination.IsIPv6() {
|
||||
metadata.IPVersion = 6
|
||||
// lx: pre-match used to prepare only fakeip + IP version, so process/neighbor
|
||||
// rule items (process_name, source_mac_address, source_hostname, …) never had
|
||||
// their metadata filled here and silently failed to match — they were resolved
|
||||
// in matchRule only. Both paths now share prepareMatchMetadata (upstream
|
||||
// b911fb078).
|
||||
err := r.prepareMatchMetadata(ctx, &metadata)
|
||||
if err != nil {
|
||||
return continueResult
|
||||
}
|
||||
for currentRuleIndex, currentRule := range r.rules {
|
||||
metadata.ResetRuleCache()
|
||||
@@ -540,13 +532,11 @@ func (r *Router) preMatchFlow(ctx context.Context, metadata *adapter.InboundCont
|
||||
return result
|
||||
}
|
||||
|
||||
func (r *Router) matchRule(
|
||||
ctx context.Context, metadata *adapter.InboundContext,
|
||||
inputConn net.Conn, inputPacketConn N.PacketConn,
|
||||
) (
|
||||
selectedRule adapter.Rule, selectedRuleIndex int,
|
||||
buffers []*buf.Buffer, packetBuffers []*N.PacketBuffer, fatalErr error,
|
||||
) {
|
||||
// prepareMatchMetadata fills in everything a rule may match on but the inbound
|
||||
// cannot know: the connection owner, the neighbor (MAC/hostname) behind the
|
||||
// source address, the fakeip / reverse-mapped domain and the IP version. Shared
|
||||
// by matchRule and PreMatch — see the note at the PreMatch call site.
|
||||
func (r *Router) prepareMatchMetadata(ctx context.Context, metadata *adapter.InboundContext) error {
|
||||
r.searchProcessInfo(ctx, metadata)
|
||||
if r.neighborResolver != nil && metadata.SourceMACAddress == nil && metadata.Source.Addr.IsValid() {
|
||||
mac, macFound := r.neighborResolver.LookupMAC(metadata.Source.Addr)
|
||||
@@ -568,8 +558,7 @@ func (r *Router) matchRule(
|
||||
if metadata.Destination.Addr.IsValid() && r.dnsTransport.FakeIP() != nil && r.dnsTransport.FakeIP().Store().Contains(metadata.Destination.Addr) {
|
||||
domain, loaded := r.dnsTransport.FakeIP().Store().Lookup(metadata.Destination.Addr)
|
||||
if !loaded {
|
||||
fatalErr = E.New("missing fakeip record, try enable `experimental.cache_file`")
|
||||
return
|
||||
return E.New("missing fakeip record, try enable `experimental.cache_file`")
|
||||
}
|
||||
if domain != "" {
|
||||
metadata.OriginDestination = metadata.Destination
|
||||
@@ -592,6 +581,20 @@ func (r *Router) matchRule(
|
||||
} else if metadata.Destination.IsIPv6() {
|
||||
metadata.IPVersion = 6
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
func (r *Router) matchRule(
|
||||
ctx context.Context, metadata *adapter.InboundContext,
|
||||
inputConn net.Conn, inputPacketConn N.PacketConn,
|
||||
) (
|
||||
selectedRule adapter.Rule, selectedRuleIndex int,
|
||||
buffers []*buf.Buffer, packetBuffers []*N.PacketBuffer, fatalErr error,
|
||||
) {
|
||||
fatalErr = r.prepareMatchMetadata(ctx, metadata)
|
||||
if fatalErr != nil {
|
||||
return
|
||||
}
|
||||
|
||||
match:
|
||||
for currentRuleIndex, currentRule := range r.rules {
|
||||
|
||||
+11
-11
@@ -80,15 +80,15 @@ SKIP_COMMON='^TestIntegration'
|
||||
# RoutineReceiveIncoming() with no lock, caught by
|
||||
# transport/wireguard.TestAwgDetourClientBindDelivers; that one has since been
|
||||
# fixed in client_bind.go and is NOT skipped — it is exactly what this pass is
|
||||
# for. What is left:
|
||||
# - shater/alert.TestExpiryDedupWithinDay — the test's own closure
|
||||
# (expiry_test.go:78) reads a variable the test body writes at :85 while
|
||||
# Notifier.dispatch's goroutine is still delivering. A test-side bug, ~one
|
||||
# mutex to fix, but it lives in shater/ and is nobody's blocker to ship.
|
||||
# Naming it here keeps the gate a gate from day one. It is skipped ONLY in the
|
||||
# -race pass — it still runs, and still has to pass, in the main pass below.
|
||||
# DELETE THE ENTRY THE MOMENT THE RACE IS FIXED.
|
||||
RACE_SKIP='^TestExpiryDedupWithinDay$'
|
||||
# for. What was left:
|
||||
# - shater/alert.TestExpiryDedupWithinDay — the test's own closure read a
|
||||
# variable the test body wrote while Notifier.dispatch's goroutine was
|
||||
# still delivering. FIXED 2026-07-26 (the simulated clock now has a mutex),
|
||||
# so the entry is gone and the -race pass covers the whole tree again.
|
||||
# Nothing is skipped under -race any more. Keep it that way: an entry here is a
|
||||
# hole in the gate, so add one only with a named reason and delete it the moment
|
||||
# the race is fixed.
|
||||
RACE_SKIP='^$'
|
||||
|
||||
echo "== shater test gate =="
|
||||
echo " tags : $SHATER_ROUTER_TAGS"
|
||||
@@ -234,8 +234,8 @@ echo
|
||||
# upstream code exercised by upstream CI.
|
||||
if [ "$RACE" -eq 1 ]; then
|
||||
echo "== [4/4] go test -race — the fork's trees =="
|
||||
echo " known-red under -race, skipped BY NAME (fix it and delete from RACE_SKIP):"
|
||||
echo " TestExpiryDedupWithinDay shater/alert (test-side race, expiry_test.go:78/85)"
|
||||
echo " nothing is skipped under -race"
|
||||
|
||||
run_suite race "-race -skip $RACE_SKIP" "${ROOTS[@]}"
|
||||
else
|
||||
echo "== [4/4] -race pass skipped (--no-race) =="
|
||||
|
||||
@@ -0,0 +1,141 @@
|
||||
package alert
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"strings"
|
||||
"sync"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"github.com/sagernet/sing-box/log"
|
||||
"github.com/sagernet/sing-box/shater/model"
|
||||
)
|
||||
|
||||
// warnLogger records Warn lines so a test can assert that an eviction was
|
||||
// actually announced. Everything else falls through to the standard logger.
|
||||
type warnLogger struct {
|
||||
log.ContextLogger
|
||||
mu sync.Mutex
|
||||
warns []string
|
||||
}
|
||||
|
||||
func newWarnLogger() *warnLogger { return &warnLogger{ContextLogger: log.StdLogger()} }
|
||||
|
||||
func (l *warnLogger) Warn(args ...any) {
|
||||
l.mu.Lock()
|
||||
l.warns = append(l.warns, fmt.Sprint(args...))
|
||||
l.mu.Unlock()
|
||||
}
|
||||
|
||||
func (l *warnLogger) lines() []string {
|
||||
l.mu.Lock()
|
||||
defer l.mu.Unlock()
|
||||
return append([]string(nil), l.warns...)
|
||||
}
|
||||
|
||||
// notifierAt builds a Notifier with no channels (so nothing is ever delivered —
|
||||
// only the dedup bookkeeping runs) and a clock the test drives.
|
||||
func notifierAt(t *testing.T, clock *time.Time) (*Notifier, *warnLogger) {
|
||||
t.Helper()
|
||||
lg := newWarnLogger()
|
||||
n := New([]model.Alert{}, lg)
|
||||
n.now = func() time.Time { return *clock }
|
||||
return n, lg
|
||||
}
|
||||
|
||||
func (n *Notifier) dedupSize() int {
|
||||
n.mu.Lock()
|
||||
defer n.mu.Unlock()
|
||||
return len(n.dedup)
|
||||
}
|
||||
|
||||
// TestDedupTableIsBoundedOverTime is the leak itself: a guest network whose
|
||||
// clients randomise their MAC produces an endless stream of distinct
|
||||
// "new_device:<MAC>" keys, and nothing ever deleted one. Spread over time — which
|
||||
// is how it actually happens — the table must stay small, and nothing may be
|
||||
// reported as evicted, because an entry past the dedup window could no longer
|
||||
// suppress anything anyway.
|
||||
func TestDedupTableIsBoundedOverTime(t *testing.T) {
|
||||
clock := time.Now()
|
||||
n, lg := notifierAt(t, &clock)
|
||||
|
||||
// Ten times the cap, at one incident per second: every key is long past the
|
||||
// 60s window by the time the next batch arrives.
|
||||
const fires = maxDedupKeys * 10
|
||||
for i := 0; i < fires; i++ {
|
||||
clock = clock.Add(time.Second)
|
||||
n.FireIncident(Incident{
|
||||
Events: []string{"new_device"},
|
||||
Title: "New device on the LAN",
|
||||
Key: fmt.Sprintf("new_device:02:00:00:%02x:%02x:%02x", i>>16&0xff, i>>8&0xff, i&0xff),
|
||||
})
|
||||
}
|
||||
|
||||
if got := n.dedupSize(); got > maxDedupKeys {
|
||||
t.Fatalf("dedup table holds %d entries after %d distinct incidents; cap is %d",
|
||||
got, fires, maxDedupKeys)
|
||||
}
|
||||
if got := n.DedupEvicted(); got != 0 {
|
||||
t.Fatalf("reported %d LIVE evictions; entries aged out of the window and losing them costs nothing", got)
|
||||
}
|
||||
if lines := lg.lines(); len(lines) != 0 {
|
||||
t.Fatalf("expiry sweep must be silent (it loses nothing), got: %v", lines)
|
||||
}
|
||||
}
|
||||
|
||||
// TestDedupTableEvictionIsAnnounced is the other half: when the cap genuinely
|
||||
// bites — more distinct incidents inside ONE dedup window than the table holds —
|
||||
// live suppression state is lost and repeats may notify twice. That must be said
|
||||
// out loud, not absorbed.
|
||||
func TestDedupTableEvictionIsAnnounced(t *testing.T) {
|
||||
clock := time.Now()
|
||||
n, lg := notifierAt(t, &clock)
|
||||
|
||||
// The clock does not move: every key stays inside its window.
|
||||
for i := 0; i < maxDedupKeys+10; i++ {
|
||||
clock = clock.Add(time.Millisecond) // still far inside dedupWindow
|
||||
n.FireIncident(Incident{
|
||||
Events: []string{"new_device"},
|
||||
Title: "New device on the LAN",
|
||||
Key: fmt.Sprintf("new_device:flood-%d", i),
|
||||
})
|
||||
}
|
||||
|
||||
if got := n.dedupSize(); got > maxDedupKeys {
|
||||
t.Fatalf("dedup table grew to %d, above the %d cap", got, maxDedupKeys)
|
||||
}
|
||||
if n.DedupEvicted() == 0 {
|
||||
t.Fatalf("a flood of %d in-window incidents evicted nothing — the cap is not enforced", maxDedupKeys+10)
|
||||
}
|
||||
lines := lg.lines()
|
||||
if len(lines) == 0 {
|
||||
t.Fatalf("live suppression entries were dropped with no notice")
|
||||
}
|
||||
if !strings.Contains(lines[0], "dedup table full") || !strings.Contains(lines[0], "notify twice") {
|
||||
t.Fatalf("eviction notice does not explain the consequence: %q", lines[0])
|
||||
}
|
||||
}
|
||||
|
||||
// TestDedupStillSuppressesWithinTheWindow guards the behaviour the bound must not
|
||||
// break: a repeat inside the window is still collapsed, and one outside it is not.
|
||||
func TestDedupStillSuppressesWithinTheWindow(t *testing.T) {
|
||||
clock := time.Now()
|
||||
n, _ := notifierAt(t, &clock)
|
||||
|
||||
fire := func() bool {
|
||||
n.mu.Lock()
|
||||
defer n.mu.Unlock()
|
||||
return n.suppressedLocked("killswitch\x00Kill-switch engaged")
|
||||
}
|
||||
if fire() {
|
||||
t.Fatalf("first fire was suppressed")
|
||||
}
|
||||
clock = clock.Add(dedupWindow / 2)
|
||||
if !fire() {
|
||||
t.Fatalf("a repeat inside the window was NOT suppressed")
|
||||
}
|
||||
clock = clock.Add(dedupWindow)
|
||||
if fire() {
|
||||
t.Fatalf("a repeat past the window was suppressed")
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,87 @@
|
||||
package alert
|
||||
|
||||
import (
|
||||
"net/http"
|
||||
"net/http/httptest"
|
||||
"sync/atomic"
|
||||
"testing"
|
||||
|
||||
"github.com/sagernet/sing-box/shater/model"
|
||||
)
|
||||
|
||||
// closingRT is an http.RoundTripper that also implements the CloseIdleConnections
|
||||
// hook http.Client forwards to, so a test can observe whether the client was ever
|
||||
// released. http.Transport implements the same hook — this stands in for it.
|
||||
type closingRT struct {
|
||||
rt http.RoundTripper
|
||||
closed atomic.Int32
|
||||
}
|
||||
|
||||
func (c *closingRT) RoundTrip(r *http.Request) (*http.Response, error) { return c.rt.RoundTrip(r) }
|
||||
func (c *closingRT) CloseIdleConnections() { c.closed.Add(1) }
|
||||
|
||||
// TestDetourClientIsClosedAfterDelivery: the detour factory (engine.HTTPClient)
|
||||
// mints a NEW http.Transport for every call, and that transport's idle connections
|
||||
// are live proxying sessions through an engine outbound whose object the dial
|
||||
// closure pins. The notifier used one per delivery and dropped it, so every alert
|
||||
// left a keep-alive session — and a reference to a possibly-retired engine
|
||||
// generation — behind for the whole idle timeout.
|
||||
func TestDetourClientIsClosedAfterDelivery(t *testing.T) {
|
||||
c, srv := newSink(t)
|
||||
n := New([]model.Alert{{
|
||||
Name: "hook", Enabled: true, Type: "webhook", URL: srv.URL,
|
||||
Events: []string{"killswitch"}, Via: "node:tunnel",
|
||||
}}, nil)
|
||||
|
||||
var made []*closingRT
|
||||
n.SetClientFactory(func(via string) (*http.Client, error) {
|
||||
rt := &closingRT{rt: http.DefaultTransport}
|
||||
made = append(made, rt)
|
||||
return &http.Client{Transport: rt}, nil
|
||||
})
|
||||
|
||||
n.Fire("killswitch", "Kill-switch engaged", "the tunnel is down")
|
||||
n.Wait()
|
||||
|
||||
if c.n() != 1 {
|
||||
t.Fatalf("delivery count = %d, want 1", c.n())
|
||||
}
|
||||
if len(made) != 1 {
|
||||
t.Fatalf("factory called %d times, want 1", len(made))
|
||||
}
|
||||
if got := made[0].closed.Load(); got == 0 {
|
||||
t.Fatalf("the per-delivery detour client was never closed — its idle connections " +
|
||||
"(and the engine outbound its dialer pins) outlive the alert")
|
||||
}
|
||||
}
|
||||
|
||||
// TestDetourClientIsClosedWhenTheSendFails covers the fallback path: a detour that
|
||||
// errors and falls back to direct still built a transport, and that one leaked too.
|
||||
func TestDetourClientIsClosedWhenTheSendFails(t *testing.T) {
|
||||
// A sink that rejects, so the detour send fails and Fallback kicks in.
|
||||
reject := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
||||
w.WriteHeader(http.StatusInternalServerError)
|
||||
}))
|
||||
defer reject.Close()
|
||||
|
||||
n := New([]model.Alert{{
|
||||
Name: "hook", Enabled: true, Type: "webhook", URL: reject.URL,
|
||||
Events: []string{"killswitch"}, Via: "node:tunnel", Fallback: true,
|
||||
}}, nil)
|
||||
|
||||
var rt *closingRT
|
||||
n.SetClientFactory(func(via string) (*http.Client, error) {
|
||||
rt = &closingRT{rt: http.DefaultTransport}
|
||||
return &http.Client{Transport: rt}, nil
|
||||
})
|
||||
|
||||
n.Fire("killswitch", "Kill-switch engaged", "the tunnel is down")
|
||||
n.Wait()
|
||||
|
||||
if rt == nil {
|
||||
t.Fatalf("factory was never called")
|
||||
}
|
||||
if got := rt.closed.Load(); got == 0 {
|
||||
t.Fatalf("a failed detour delivery still leaked its transport")
|
||||
}
|
||||
}
|
||||
@@ -5,6 +5,7 @@ import (
|
||||
"os"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
"sync"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
@@ -74,22 +75,37 @@ func TestExpiryDedupWithinDay(t *testing.T) {
|
||||
// test compresses 24 simulated hours into a few milliseconds of wall clock — so
|
||||
// without this the second delivery is (correctly) suppressed by that window and
|
||||
// the test would be measuring the fixture, not the behaviour.
|
||||
// The clock is shared with the notifier's DELIVERY goroutines (buildPayload
|
||||
// stamps the payload with n.now()), which are still in flight while this body
|
||||
// advances the simulated time — so it needs a lock, not a bare variable.
|
||||
var clockMu sync.Mutex
|
||||
simNow := now
|
||||
e.n.now = func() time.Time { return simNow }
|
||||
setNow := func(t time.Time) {
|
||||
clockMu.Lock()
|
||||
simNow = t
|
||||
clockMu.Unlock()
|
||||
}
|
||||
e.n.now = func() time.Time {
|
||||
clockMu.Lock()
|
||||
defer clockMu.Unlock()
|
||||
return simNow
|
||||
}
|
||||
|
||||
if got := e.Run(m, now); got != 1 {
|
||||
t.Fatalf("first Run fired %d, want 1", got)
|
||||
}
|
||||
// Simulate a minute-by-minute reconcile for the next 23 hours.
|
||||
for i := 1; i <= 23; i++ {
|
||||
simNow = now.Add(time.Duration(i) * time.Hour)
|
||||
if got := e.Run(m, simNow); got != 0 {
|
||||
at := now.Add(time.Duration(i) * time.Hour)
|
||||
setNow(at)
|
||||
if got := e.Run(m, at); got != 0 {
|
||||
t.Fatalf("Run at +%dh fired %d alerts, want 0 (dedup window is 24h)", i, got)
|
||||
}
|
||||
}
|
||||
// Just past the window it may speak again.
|
||||
simNow = now.Add(24*time.Hour + time.Minute)
|
||||
if got := e.Run(m, simNow); got != 1 {
|
||||
past := now.Add(24*time.Hour + time.Minute)
|
||||
setNow(past)
|
||||
if got := e.Run(m, past); got != 1 {
|
||||
t.Errorf("Run just past 24h fired %d, want 1", got)
|
||||
}
|
||||
e.n.Wait()
|
||||
|
||||
@@ -20,6 +20,7 @@ import (
|
||||
"fmt"
|
||||
"io"
|
||||
"net/http"
|
||||
"sort"
|
||||
"strings"
|
||||
"sync"
|
||||
"time"
|
||||
@@ -31,6 +32,29 @@ import (
|
||||
const (
|
||||
httpTimeout = 8 * time.Second
|
||||
dedupWindow = 60 * time.Second
|
||||
|
||||
// maxDedupKeys / keepDedupKeys bound the dedup table.
|
||||
//
|
||||
// Nothing ever deleted from it. Every key that had EVER fired stayed forever,
|
||||
// and its highest-cardinality producer is "new_device:<MAC>" — on a guest
|
||||
// network where clients randomise their MAC per association, that is a fresh
|
||||
// key per device per join, for the life of a daemon that runs for months.
|
||||
//
|
||||
// The real bound is not this cap, it is the window: an entry older than
|
||||
// dedupWindow (60s) can never suppress anything again, so it is pure garbage
|
||||
// and sweeping it costs nothing and changes no behaviour. compactDedupLocked
|
||||
// does that first. The cap only bites when 1024 DISTINCT incidents fired inside
|
||||
// one 60-second window — and there the eviction is a real loss of suppression
|
||||
// state, so it is reported rather than done quietly.
|
||||
//
|
||||
// Why 1024: the new_device watcher polls every ~45s, so a poll would have to
|
||||
// discover a thousand previously-unseen MACs at once to reach it — roughly 20x
|
||||
// the worst guest-network churn this box has seen. At ~100 B per entry (a
|
||||
// 28-byte "new_device:<MAC>" key plus a time.Time plus map overhead) the full
|
||||
// table is ~100 KB of a 512 MB router: cheap enough that a generous headroom
|
||||
// costs nothing.
|
||||
maxDedupKeys = 1024
|
||||
keepDedupKeys = 512
|
||||
)
|
||||
|
||||
// telegramAPIBase is the Telegram Bot API root. A package var so tests can point
|
||||
@@ -43,6 +67,10 @@ type Notifier struct {
|
||||
mu sync.Mutex
|
||||
alerts []model.Alert
|
||||
dedup map[string]time.Time // (event\x00title) -> last fire time
|
||||
// dedupEvicted counts LIVE dedup entries dropped by the capacity bound (see
|
||||
// compactDedupLocked). Expired entries swept out are NOT counted: they could no
|
||||
// longer suppress anything, so dropping them loses nothing.
|
||||
dedupEvicted uint64
|
||||
|
||||
client *http.Client
|
||||
log log.ContextLogger
|
||||
@@ -227,10 +255,69 @@ func (n *Notifier) suppressedLocked(key string) bool {
|
||||
if last, ok := n.dedup[key]; ok && now.Sub(last) < dedupWindow {
|
||||
return true
|
||||
}
|
||||
if len(n.dedup) >= maxDedupKeys {
|
||||
n.compactDedupLocked(now)
|
||||
}
|
||||
n.dedup[key] = now
|
||||
return false
|
||||
}
|
||||
|
||||
// compactDedupLocked bounds the dedup table. Caller holds n.mu.
|
||||
//
|
||||
// Two stages, deliberately distinct because only one of them loses anything:
|
||||
//
|
||||
// 1. Drop every entry older than dedupWindow. Such an entry cannot suppress a
|
||||
// future fire (suppressedLocked already ignores it), so this is garbage
|
||||
// collection, not eviction: no notification changes, and nothing is reported.
|
||||
// On any realistic traffic this stage alone keeps the table at "keys seen in
|
||||
// the last minute".
|
||||
// 2. If the table is STILL full, a thousand distinct incidents fired inside one
|
||||
// window. Now eviction is real — the oldest live entries go, and a repeat of
|
||||
// one of them within its window will notify a second time instead of being
|
||||
// collapsed. That is a visible change in behaviour, so it is logged, with the
|
||||
// running total, rather than silently absorbed.
|
||||
func (n *Notifier) compactDedupLocked(now time.Time) {
|
||||
for k, last := range n.dedup {
|
||||
if now.Sub(last) >= dedupWindow {
|
||||
delete(n.dedup, k)
|
||||
}
|
||||
}
|
||||
if len(n.dedup) < maxDedupKeys {
|
||||
return
|
||||
}
|
||||
|
||||
type kv struct {
|
||||
key string
|
||||
at time.Time
|
||||
}
|
||||
all := make([]kv, 0, len(n.dedup))
|
||||
for k, at := range n.dedup {
|
||||
all = append(all, kv{k, at})
|
||||
}
|
||||
sort.Slice(all, func(i, j int) bool { return all[i].at.Before(all[j].at) })
|
||||
drop := len(all) - keepDedupKeys
|
||||
for i := 0; i < drop; i++ {
|
||||
delete(n.dedup, all[i].key)
|
||||
}
|
||||
n.dedupEvicted += uint64(drop)
|
||||
n.log.Warn("alert: dedup table full (", maxDedupKeys,
|
||||
" incidents inside one ", dedupWindow, " window) — dropped ", drop,
|
||||
" live suppression entries (", n.dedupEvicted,
|
||||
" total); repeats of those incidents may notify twice")
|
||||
}
|
||||
|
||||
// DedupEvicted reports how many LIVE suppression entries the cap has dropped since
|
||||
// start (stage 2 of compactDedupLocked only — the expiry sweep is not counted,
|
||||
// because it loses nothing). Nonzero means alerts may have been delivered twice.
|
||||
func (n *Notifier) DedupEvicted() uint64 {
|
||||
if n == nil {
|
||||
return 0
|
||||
}
|
||||
n.mu.Lock()
|
||||
defer n.mu.Unlock()
|
||||
return n.dedupEvicted
|
||||
}
|
||||
|
||||
// dispatch delivers to a single alert in its own goroutine. Panics are recovered
|
||||
// and logged; a delivery error is logged. It never crashes the daemon.
|
||||
func (n *Notifier) dispatch(a model.Alert, event, title, body string) {
|
||||
@@ -276,6 +363,18 @@ func (n *Notifier) deliver(a model.Alert, event, title, body string) error {
|
||||
|
||||
if detour && factory != nil {
|
||||
client, cerr := factory(via)
|
||||
if cerr == nil && client != nil {
|
||||
// The factory (engine.HTTPClient) builds a BRAND NEW http.Transport per
|
||||
// call, with keep-alive and a 90s idle timeout, and we use it for exactly
|
||||
// one POST. Dropping it without this leaves the idle connection — a real
|
||||
// proxying session through an engine outbound, plus its read and write
|
||||
// loops — alive for the whole idle timeout. Worse, the transport's
|
||||
// DialContext closure captures that outbound object, so the idle
|
||||
// connection PINS a retired engine generation whose close budget is 5
|
||||
// seconds. One alert delivery per minute keeps a permanent rolling set of
|
||||
// them.
|
||||
defer client.CloseIdleConnections()
|
||||
}
|
||||
if cerr != nil {
|
||||
if a.Fallback {
|
||||
n.log.Warn("alert: ", a.Name, " detour ", via, " unavailable (", cerr, ") — falling back to direct")
|
||||
|
||||
+157
-18
@@ -61,10 +61,16 @@ type Applier struct {
|
||||
// silently skipped. "" until the first successful nft load.
|
||||
lastNft string
|
||||
|
||||
// holding is true while the fail-closed HOLDING PLANE is installed: the engine
|
||||
// holding is the LATCH half of the hold state: true while a fail-closed HOLDING
|
||||
// PLANE that THIS Applier installed (holdLocked) is in the kernel — the engine
|
||||
// is down and LAN->WAN forwarding is blocked. Surfaced in Status so the panel
|
||||
// can say "protected but not proxying" instead of showing a healthy-looking UI
|
||||
// over an unprotected router.
|
||||
//
|
||||
// It is deliberately NOT the whole answer: a plane installed by the boot armor
|
||||
// or left by a predecessor blocks the LAN just as hard and never touches this
|
||||
// field. Read it through Holding()/holdingWith(), which add the derived half;
|
||||
// writing an inference into this latch would create a claim nothing clears.
|
||||
// holding + lastWarnings live behind their OWN lock, NOT the apply mutex.
|
||||
// Status() reads them, and Status is polled by the LuCI dashboard, the cron
|
||||
// watchdog and hotplug; making those reads queue behind a running apply
|
||||
@@ -342,6 +348,19 @@ func (a *Applier) UpdateSubscription(name string) (added int, err error) {
|
||||
if err != nil {
|
||||
return 0, fmt.Errorf("subscription %q detour %q: %w", name, sub.FetchDetour, err)
|
||||
}
|
||||
// engine.HTTPClient builds a FRESH http.Transport per call, and its dialer
|
||||
// closes over the resolved outbound. Dropping the client on the floor leaves
|
||||
// that transport's idle keep-alives open — those are real proxied sessions
|
||||
// through the engine, a goroutine pair each — and the closure PINS the engine
|
||||
// generation they were dialled on. A generation's shutdown budget is five
|
||||
// seconds; an idle connection outlives it, which is how a retired box stays
|
||||
// alive with its WireGuard devices still held.
|
||||
//
|
||||
// Deferred rather than closed at the end on purpose: every exit from here is
|
||||
// covered, including the four error returns below, and those are the ones that
|
||||
// leak in the field — a subscription whose feed is broken is retried by cron
|
||||
// once a refresh interval, forever.
|
||||
defer client.CloseIdleConnections()
|
||||
}
|
||||
|
||||
// FetchWithInfo additionally parses the provider's `subscription-userinfo`
|
||||
@@ -542,14 +561,14 @@ func (a *Applier) applyDataPlaneLocked(m *model.Model, opts option.Options, now
|
||||
out.stage = "rendering the nft ruleset"
|
||||
return out, err
|
||||
}
|
||||
nftCurrent := ruleset == a.lastNft && netplane.TableExists()
|
||||
nftCurrent := ruleset == a.lastNft && tableExists()
|
||||
if !nftCurrent {
|
||||
if err := netplane.ApplyNft(ruleset); err != nil {
|
||||
// The engine may be up, but with no table loaded nothing is diverted into
|
||||
// it — LAN traffic goes straight out the WAN. That is the same silent
|
||||
// fail-open as a dead engine, so it gets the same answer: if there is no
|
||||
// table at all, hold the line rather than leave the LAN exposed.
|
||||
if !netplane.TableExists() {
|
||||
if !tableExists() {
|
||||
a.holdLocked(m, err)
|
||||
}
|
||||
out.stage = "loading the nft ruleset"
|
||||
@@ -726,10 +745,35 @@ func killSwitchClosed(g model.Globals) bool {
|
||||
// attempted, so the window between daemon start and a successful engine start is
|
||||
// protected rather than open. The daemon calls it at startup.
|
||||
//
|
||||
// It is a no-op when the plane is disabled, the kill-switch is open, or a table
|
||||
// is already loaded. A successful apply replaces the holding plane atomically
|
||||
// (the full ruleset is a single `delete table` + `table` transaction), so the
|
||||
// cost on a healthy boot is a few seconds of blocked forwarding — which is
|
||||
// It is a no-op when the plane is disabled or the kill-switch is open: fail-open
|
||||
// is the operator's documented choice and must not be quietly overridden.
|
||||
//
|
||||
// It is NO LONGER a no-op when a table is already loaded, and that is the change.
|
||||
// The early return on netplane.TableExists() cost two things:
|
||||
//
|
||||
// - It deferred to a table this process cannot inspect. Since the boot armor
|
||||
// landed, the table found here is usually /etc/init.d/shater-armor's snapshot
|
||||
// — which may have been rendered before an interface rename and therefore
|
||||
// protects a device that no longer exists. cmd/shaterd's own armorRender
|
||||
// preference says a fresh render beats a snapshot for exactly this reason.
|
||||
// - It skipped setHolding entirely, so with the boot armor loaded the daemon
|
||||
// reported holding=false over a LAN that really was cut off. Status then
|
||||
// published plane="full" and the apply-failure alert (cmd/shaterd
|
||||
// fireApplyFail) told the operator traffic was NOT being blocked at the very
|
||||
// moment it was — sending them to dismantle a protection that was working.
|
||||
//
|
||||
// Rather than adopt a table it cannot vouch for, ArmHold installs its own,
|
||||
// rendered from the model it was handed: what it then publishes is a fact about
|
||||
// something this process did, not a guess about something it found. That is also
|
||||
// why the fix is not simply "call setHolding(true) when a table is present" — see
|
||||
// foreignHold for the case where installing is impossible and a claim has to be
|
||||
// derived instead.
|
||||
//
|
||||
// Replacing a loaded table costs nothing in exposure: `nft -f` commits as ONE
|
||||
// netlink transaction (netplane.runNftStdin), so there is no window between the
|
||||
// delete and the new table, and a ruleset that fails to load leaves the previous
|
||||
// one in place. A successful apply then replaces the holding plane the same way,
|
||||
// so the cost on a healthy boot is a few seconds of blocked forwarding — which is
|
||||
// precisely what fail-closed is supposed to mean.
|
||||
func (a *Applier) ArmHold(m *model.Model) {
|
||||
if m == nil || !m.Globals.Enabled || !killSwitchClosed(m.Globals) {
|
||||
@@ -742,9 +786,6 @@ func (a *Applier) ArmHold(m *model.Model) {
|
||||
defer release()
|
||||
a.mu.Lock()
|
||||
defer a.mu.Unlock()
|
||||
if netplane.TableExists() {
|
||||
return
|
||||
}
|
||||
a.holdLocked(m, errEngineNotStartedYet)
|
||||
}
|
||||
|
||||
@@ -755,13 +796,103 @@ var errEngineNotStartedYet = errors.New("engine has not started yet")
|
||||
// running nft.
|
||||
var applyHoldNft = netplane.ApplyNft
|
||||
|
||||
// Holding reports whether the fail-closed holding plane is currently installed
|
||||
// (engine down, LAN->WAN blocked). Takes only the leaf lock, so it never waits
|
||||
// on a running apply.
|
||||
// tableExists and bootArmorPresent are the two OUTSIDE-WORLD facts this file's
|
||||
// honesty now rests on: is our table in the kernel, and does the persisted
|
||||
// fail-closed plane exist on flash. Seams for the same reason as engineApply and
|
||||
// applyDataPlane — what the daemon SAYS about a plane it did not install can
|
||||
// otherwise only be exercised on a router, with root, a real nft and a real
|
||||
// /etc/shater, i.e. never, in the gate. Production reads netplane.
|
||||
var (
|
||||
tableExists = netplane.TableExists
|
||||
bootArmorPresent = netplane.BootArmorPresent
|
||||
)
|
||||
|
||||
// Holding reports whether forwarded LAN traffic is currently being BLOCKED by a
|
||||
// fail-closed plane while the engine is down. It never waits on a running apply.
|
||||
//
|
||||
// It has two sources, and the second one is the whole point.
|
||||
//
|
||||
// The LATCH (a.holding) is written by holdLocked, i.e. when THIS process
|
||||
// installed the plane. It used to be the only source, which quietly made this a
|
||||
// record of what the process had DONE rather than a description of the router.
|
||||
// The boot armor (netplane/armor.go) opened the gap: /etc/init.d/shater-armor
|
||||
// loads the persisted holding plane at START=21, long before the daemon exists,
|
||||
// and the daemon's own unreadable-config path (cmd/shaterd armOnUnreadableConfig)
|
||||
// reinstates it without going anywhere near the apply pipeline. In both cases the
|
||||
// LAN really is cut off and the latch reads false. With an unreadable config
|
||||
// NOTHING ever corrects it, because the correction only happens on an apply and
|
||||
// no apply can run: Status publishes plane="full" over a blocked LAN, and
|
||||
// fireApplyFail words its incident as "traffic is NOT being blocked" at the exact
|
||||
// moment it is. That is the inverted lie — it does not hide a fault, it invents
|
||||
// one, and the obvious response to it is to tear down the protection that works.
|
||||
//
|
||||
// The DERIVED half (foreignHold) closes that, and it self-clears: it is computed
|
||||
// at read time from the kernel and the engine, so it goes false the instant
|
||||
// either fact changes, whereas a latch set from an inference would have to be
|
||||
// remembered to be cleared. Same reasoning as Warnings' live half.
|
||||
func (a *Applier) Holding() bool {
|
||||
return a.holdingWith(tableExists(), a.eng != nil && a.eng.Running())
|
||||
}
|
||||
|
||||
// holdingWith is Holding against facts the caller has already established, so
|
||||
// Status does not shell out to `nft list table` twice for one poll.
|
||||
func (a *Applier) holdingWith(tableLoaded, engineUp bool) bool {
|
||||
a.stateMu.RLock()
|
||||
defer a.stateMu.RUnlock()
|
||||
return a.holding
|
||||
latched := a.holding
|
||||
a.stateMu.RUnlock()
|
||||
return latched || foreignHold(tableLoaded, engineUp)
|
||||
}
|
||||
|
||||
// foreignHold answers: is a plane THIS PROCESS DID NOT INSTALL holding the LAN?
|
||||
//
|
||||
// It deliberately does not try to identify the loaded table, because it cannot.
|
||||
// netplane exposes no read-back of the loaded ruleset, `#` comments do not
|
||||
// survive `nft -f`, and the two candidates — our holding plane, or a full ruleset
|
||||
// left behind by a generation that died without tearing down — are not
|
||||
// distinguishable from here. Guessing which one it is would be a second lie
|
||||
// inside the fix for the first.
|
||||
//
|
||||
// It does not need to. What `holding` asserts is not "the holding plane is the
|
||||
// object in the kernel", it is "the engine is down and forwarded traffic is being
|
||||
// dropped" — and BOTH candidates do that, provided they were rendered from an
|
||||
// enabled, fail-closed config:
|
||||
//
|
||||
// - the holding plane drops by construction: netplane.RenderHoldNftAt ends in
|
||||
// `<iif> meta nfproto ipv4 drop` and its v6 twin;
|
||||
// - a leftover FULL ruleset drops too, by design. With no engine socket the
|
||||
// `tproxy` statement returns NFT_BREAK, which aborts its own rule before the
|
||||
// trailing `meta mark set ... accept`, so the packet reaches the forward chain
|
||||
// unmarked and meets the PRIMARY FAIL-CLOSED DROP there (netplane/nft.go).
|
||||
// That is the kill-switch leak this project fixed in the forward chain, and
|
||||
// netplane/armor.go restates the property in prose.
|
||||
//
|
||||
// So the question that actually decides the claim is whether the last config this
|
||||
// router applied was enabled AND fail-closed — and the boot armor's PRESENCE is
|
||||
// exactly that fact, by its own contract ("the file IS the arm token",
|
||||
// netplane/armor.go): written after every successful config read while enabled
|
||||
// with the kill switch closed, removed the moment either stops being true. A
|
||||
// leftover full plane from a kill_switch=open config — the one that really does
|
||||
// drop nothing — therefore reads FALSE here, which is the answer the operator
|
||||
// needs and the case condition (1) of this fix exists to protect.
|
||||
//
|
||||
// engineUp must be false. With the engine up, a loaded table is the working full
|
||||
// plane doing its job and nothing is being held; this is also what keeps a
|
||||
// successful apply's setHolding(false) from being undone one line later.
|
||||
//
|
||||
// Two residual gaps, named rather than papered over:
|
||||
//
|
||||
// - a full plane rendered under kill_switch=open, still loaded after the
|
||||
// operator closed the switch and the armor was rewritten, reads as holding
|
||||
// while it is not. Any reconcile closes it, and cron runs one a minute.
|
||||
// - a boot armor whose device set predates an interface rename protects the old
|
||||
// name. That is a property of the snapshot, not of this predicate, and it is
|
||||
// why ArmHold now replaces it with a fresh render the moment this daemon has a
|
||||
// readable model.
|
||||
func foreignHold(tableLoaded, engineUp bool) bool {
|
||||
if !tableLoaded || engineUp {
|
||||
return false
|
||||
}
|
||||
return bootArmorPresent()
|
||||
}
|
||||
|
||||
func (a *Applier) setHolding(v bool) {
|
||||
@@ -1344,10 +1475,14 @@ func processUptime(now time.Time) (startedUnix, uptimeSeconds int64) {
|
||||
// traffic verdict — which is the state this field exists to make expressible.
|
||||
func (a *Applier) Status() Status {
|
||||
engineUp := a.eng != nil && a.eng.Running()
|
||||
// Read the kernel ONCE and hand the same fact to both Table and the hold
|
||||
// verdict. Asking twice is not just an extra `nft` fork per poll: the two reads
|
||||
// could disagree across a teardown and publish table=false with plane="hold".
|
||||
tableLoaded := tableExists()
|
||||
s := Status{
|
||||
Running: engineUp,
|
||||
Active: ActiveFlagPresent(),
|
||||
Table: netplane.TableExists(),
|
||||
Table: tableLoaded,
|
||||
Hash: a.eng.Hash(),
|
||||
CanRollback: a.canRollback(),
|
||||
EngineRunning: engineUp,
|
||||
@@ -1356,9 +1491,13 @@ func (a *Applier) Status() Status {
|
||||
}
|
||||
s.StartedUnix, s.UptimeSeconds = processUptime(time.Now())
|
||||
switch {
|
||||
case !s.Table:
|
||||
case !tableLoaded:
|
||||
s.Plane = "none"
|
||||
case a.Holding():
|
||||
case a.holdingWith(tableLoaded, engineUp):
|
||||
// Includes the plane THIS PROCESS DID NOT INSTALL — the boot armor loaded by
|
||||
// /etc/init.d/shater-armor, or what a predecessor left behind. Without that
|
||||
// the branch below claimed "full" (documented as "traffic is diverted into a
|
||||
// RUNNING engine") over a dead engine and a blocked LAN.
|
||||
s.Plane = "hold"
|
||||
default:
|
||||
s.Plane = "full"
|
||||
|
||||
@@ -0,0 +1,270 @@
|
||||
package apply
|
||||
|
||||
// The hold state must describe THE ROUTER, not this process's memory of what it
|
||||
// did.
|
||||
//
|
||||
// The boot armor (netplane/armor.go) put a fail-closed plane in the kernel that
|
||||
// the Applier never installs: /etc/init.d/shater-armor loads it at START=21,
|
||||
// before the daemon exists, and cmd/shaterd reinstates it when the config cannot
|
||||
// be read. The latch behind Holding() knew nothing about either, so the daemon
|
||||
// reported `holding=false` and `plane="full"` over a LAN that was blocked — and
|
||||
// with an unreadable config that state is PERMANENT, because the only thing that
|
||||
// ever wrote the latch was an apply and no apply can run.
|
||||
//
|
||||
// What comes out the other end is an alert (cmd/shaterd fireApplyFail) that
|
||||
// chooses its wording from exactly this bool and tells the operator "Traffic is
|
||||
// NOT being blocked" while it is. That is the inverted failure: not a fault
|
||||
// hidden, but a fault invented — and the obvious response to it is to go and
|
||||
// dismantle the protection that is doing its job.
|
||||
|
||||
import (
|
||||
"errors"
|
||||
"os"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"github.com/sagernet/sing-box/option"
|
||||
"github.com/sagernet/sing-box/shater/engine"
|
||||
"github.com/sagernet/sing-box/shater/model"
|
||||
)
|
||||
|
||||
// stubPlaneFacts pins the two outside-world seams — is our table in the kernel,
|
||||
// is the persisted fail-closed plane on flash — for the duration of a test.
|
||||
func stubPlaneFacts(t *testing.T, table, armor bool) func() {
|
||||
t.Helper()
|
||||
origTable, origArmor := tableExists, bootArmorPresent
|
||||
tableExists = func() bool { return table }
|
||||
bootArmorPresent = func() bool { return armor }
|
||||
return func() { tableExists, bootArmorPresent = origTable, origArmor }
|
||||
}
|
||||
|
||||
// TestHoldingSeesAPlaneThisProcessDidNotInstall is the core regression.
|
||||
//
|
||||
// All three cases share the same latch value (false — this Applier has installed
|
||||
// nothing) and must still produce three different, correct answers.
|
||||
func TestHoldingSeesAPlaneThisProcessDidNotInstall(t *testing.T) {
|
||||
a := New(engine.New(), nil)
|
||||
|
||||
// (1) No table at all. Nothing is protecting the LAN and nothing may claim to.
|
||||
restore := stubPlaneFacts(t, false, true)
|
||||
if a.Holding() {
|
||||
t.Errorf("Holding() = true with no table loaded")
|
||||
}
|
||||
if got := a.Status().Plane; got != "none" {
|
||||
t.Errorf("Plane = %q with no table loaded, want \"none\"", got)
|
||||
}
|
||||
restore()
|
||||
|
||||
// (2) The boot armor's plane IS loaded and this engine never started. This is
|
||||
// what /etc/init.d/shater-armor leaves behind at every boot, and what the
|
||||
// daemon reinstates when /etc/config/shater cannot be read — the case where
|
||||
// nothing else will ever correct the answer.
|
||||
restore = stubPlaneFacts(t, true, true)
|
||||
if !a.Holding() {
|
||||
t.Errorf("Holding() = false while the boot armor's fail-closed plane is loaded and the " +
|
||||
"engine is down — the LAN is blocked and the daemon says it is not; the apply-failure " +
|
||||
"alert would send the operator to fix a protection that is working")
|
||||
}
|
||||
s := a.Status()
|
||||
if s.Plane != "hold" {
|
||||
t.Errorf("Plane = %q over a blocked LAN with a dead engine, want \"hold\" "+
|
||||
"(\"full\" means traffic is diverted into a RUNNING engine)", s.Plane)
|
||||
}
|
||||
if !s.Table {
|
||||
t.Errorf("Table = false although a table is loaded")
|
||||
}
|
||||
if s.Running || s.EngineRunning {
|
||||
t.Errorf("running/engine_running must stay false while holding: %+v", s)
|
||||
}
|
||||
restore()
|
||||
|
||||
// (3) A table is loaded but there is NO armor on flash. refreshBootArmor
|
||||
// removes that file exactly when the operator disables the stack or opens the
|
||||
// kill switch, so what is loaded here is a leftover that drops nothing.
|
||||
// Claiming a hold would be the new lie: it would tell someone who deliberately
|
||||
// chose fail-open that their LAN is cut off.
|
||||
restore = stubPlaneFacts(t, true, false)
|
||||
if a.Holding() {
|
||||
t.Errorf("Holding() = true with no fail-closed armor on flash — a leftover plane from a " +
|
||||
"kill_switch=open config blocks nothing, and saying otherwise is the same lie inverted")
|
||||
}
|
||||
restore()
|
||||
}
|
||||
|
||||
// TestSuccessfulApplyStillClearsTheHold: the self-healing path must survive the
|
||||
// derived half. A read-time inference that ignored the engine would re-assert the
|
||||
// hold one line after setHolding(false) and pin the router in "protected, not
|
||||
// proxying" forever — with the table loaded and the armor on flash, which is the
|
||||
// steady state of every healthy router.
|
||||
func TestSuccessfulApplyStillClearsTheHold(t *testing.T) {
|
||||
a := New(engine.New(), nil)
|
||||
t.Cleanup(func() { _ = a.eng.Close() })
|
||||
t.Cleanup(func() { _ = os.Remove(ActiveFlag) })
|
||||
|
||||
// A REAL started instance: the derived half asks the engine, so a stub would
|
||||
// not exercise the thing under test.
|
||||
if _, err := a.eng.Apply(mixedOn(18841)); err != nil {
|
||||
t.Fatalf("bring a real engine up: %v", err)
|
||||
}
|
||||
if !a.eng.Running() {
|
||||
t.Fatalf("precondition: the engine must be running")
|
||||
}
|
||||
|
||||
// The steady state of a healthy router: our table loaded, armor on flash.
|
||||
defer stubPlaneFacts(t, true, true)()
|
||||
|
||||
a.setHolding(true) // whatever put us on hold before this apply
|
||||
if !a.Holding() {
|
||||
t.Fatalf("precondition: the latch must read through")
|
||||
}
|
||||
|
||||
m := holdModel("closed")
|
||||
m.Globals.GroupHealth = false // no background probing from a unit test
|
||||
defer stubApplyStages(t,
|
||||
func(*Applier, option.Options) (bool, error) { return true, nil },
|
||||
func(*Applier, *model.Model, option.Options, time.Time) (planeOutcome, error) {
|
||||
return planeOutcome{changed: true}, nil
|
||||
})()
|
||||
|
||||
a.mu.Lock()
|
||||
_, err := a.applyLocked(m)
|
||||
a.mu.Unlock()
|
||||
if err != nil {
|
||||
t.Fatalf("applyLocked: %v", err)
|
||||
}
|
||||
|
||||
if a.Holding() {
|
||||
t.Fatalf("a successful apply did not clear the hold — the router is proxying and the " +
|
||||
"panel would still show \"protected, not proxying\"")
|
||||
}
|
||||
if got := a.Status().Plane; got != "full" {
|
||||
t.Errorf("Plane = %q after a successful apply with the engine up, want \"full\"", got)
|
||||
}
|
||||
}
|
||||
|
||||
// TestApplyFailureOverAForeignPlaneReportsBlocked pins the input cmd/shaterd's
|
||||
// fireApplyFail words its incident from.
|
||||
//
|
||||
// The shape is the real one: the engine will not start, so applyLocked calls
|
||||
// holdLocked — and holdLocked cannot install anything either (nft refuses, the
|
||||
// overlay is full). The latch therefore stays false. But the boot armor's plane
|
||||
// is still standing in the kernel, so forwarded traffic IS being dropped, and the
|
||||
// incident must say so. With only the latch, this is precisely where the daemon
|
||||
// said "Traffic is NOT being blocked (kill switch is open)" on a router whose
|
||||
// kill switch was closed and whose LAN was cut off.
|
||||
func TestApplyFailureOverAForeignPlaneReportsBlocked(t *testing.T) {
|
||||
a := New(engine.New(), nil)
|
||||
defer stubPlaneFacts(t, true, true)()
|
||||
|
||||
origHold := applyHoldNft
|
||||
applyHoldNft = func(string) error { return errors.New("nft -f (load) failed: no space left on device") }
|
||||
defer func() { applyHoldNft = origHold }()
|
||||
|
||||
defer stubApplyStages(t,
|
||||
func(*Applier, option.Options) (bool, error) {
|
||||
return false, errors.New("start rule-set[geosite]: connection refused")
|
||||
},
|
||||
func(*Applier, *model.Model, option.Options, time.Time) (planeOutcome, error) {
|
||||
t.Errorf("the data-plane stage must not run after the engine stage failed")
|
||||
return planeOutcome{}, nil
|
||||
})()
|
||||
|
||||
a.mu.Lock()
|
||||
_, err := a.applyLocked(holdModel("closed"))
|
||||
a.mu.Unlock()
|
||||
if err == nil {
|
||||
t.Fatalf("applyLocked must surface the engine failure")
|
||||
}
|
||||
|
||||
a.stateMu.RLock()
|
||||
latched := a.holding
|
||||
a.stateMu.RUnlock()
|
||||
if latched {
|
||||
t.Fatalf("precondition: holdLocked could not install a plane, so the latch must be false")
|
||||
}
|
||||
|
||||
// fireApplyFail(notifier, err, applier.Holding()) — this bool picks between
|
||||
// "forwarded LAN traffic is being dropped" and "Traffic is NOT being blocked".
|
||||
if !a.Holding() {
|
||||
t.Fatalf("Holding() = false while a fail-closed plane blocks the LAN: the incident would " +
|
||||
"read \"Traffic is NOT being blocked\" at the exact moment it is being blocked")
|
||||
}
|
||||
}
|
||||
|
||||
// TestArmHoldReplacesAPlaneItDidNotInstall: ArmHold used to return early on
|
||||
// TableExists() and publish nothing, which is how the boot armor's plane came to
|
||||
// be loaded with holding=false. It now installs its own — rendered from the model
|
||||
// it was handed, so the state it publishes is a fact about something this process
|
||||
// did rather than a guess about something it found (and the fresh render also
|
||||
// replaces a snapshot that may predate an interface rename).
|
||||
func TestArmHoldReplacesAPlaneItDidNotInstall(t *testing.T) {
|
||||
loaded := withHoldProbe(t)
|
||||
defer stubPlaneFacts(t, true, true)() // a table is ALREADY loaded
|
||||
a := New(engine.New(), nil)
|
||||
|
||||
a.ArmHold(holdModel("closed"))
|
||||
|
||||
if len(*loaded) != 1 {
|
||||
t.Fatalf("ArmHold loaded %d rulesets over an existing table, want 1 — it deferred to a "+
|
||||
"table it cannot inspect instead of installing one it can vouch for", len(*loaded))
|
||||
}
|
||||
if !a.Holding() {
|
||||
t.Errorf("ArmHold installed the holding plane but did not publish it")
|
||||
}
|
||||
if got := a.Status().Plane; got != "hold" {
|
||||
t.Errorf("Plane = %q after ArmHold, want \"hold\"", got)
|
||||
}
|
||||
}
|
||||
|
||||
// TestArmHoldStillRespectsFailOpen: replacing a foreign table must not become a
|
||||
// licence to install a plane the operator did not ask for. kill_switch=open and
|
||||
// globals.enabled=0 are explicit choices and ArmHold must keep obeying both.
|
||||
func TestArmHoldStillRespectsFailOpen(t *testing.T) {
|
||||
loaded := withHoldProbe(t)
|
||||
defer stubPlaneFacts(t, true, false)()
|
||||
a := New(engine.New(), nil)
|
||||
|
||||
a.ArmHold(holdModel("open"))
|
||||
if len(*loaded) != 0 {
|
||||
t.Errorf("kill_switch=open loaded %d rulesets, want 0", len(*loaded))
|
||||
}
|
||||
|
||||
disabled := holdModel("closed")
|
||||
disabled.Globals.Enabled = false
|
||||
a.ArmHold(disabled)
|
||||
if len(*loaded) != 0 {
|
||||
t.Errorf("globals.enabled=0 loaded %d rulesets, want 0", len(*loaded))
|
||||
}
|
||||
|
||||
a.ArmHold(nil)
|
||||
if len(*loaded) != 0 {
|
||||
t.Errorf("a nil model loaded %d rulesets, want 0", len(*loaded))
|
||||
}
|
||||
if a.Holding() {
|
||||
t.Errorf("nothing was installed, so nothing may be reported as holding")
|
||||
}
|
||||
}
|
||||
|
||||
// TestForeignHoldMatrix pins the predicate itself, including the two facts it
|
||||
// refuses to guess about.
|
||||
func TestForeignHoldMatrix(t *testing.T) {
|
||||
for _, tc := range []struct {
|
||||
name string
|
||||
table, engine, armor bool
|
||||
want bool
|
||||
}{
|
||||
{"no table", false, false, true, false},
|
||||
{"engine up: the table is the working full plane", true, true, true, false},
|
||||
{"armor on flash, engine down: blocked", true, false, true, true},
|
||||
{"no armor: the last config was disabled or fail-open", true, false, false, false},
|
||||
{"nothing at all", false, false, false, false},
|
||||
} {
|
||||
t.Run(tc.name, func(t *testing.T) {
|
||||
defer stubPlaneFacts(t, tc.table, tc.armor)()
|
||||
if got := foreignHold(tc.table, tc.engine); got != tc.want {
|
||||
t.Errorf("foreignHold(table=%v, engineUp=%v) = %v, want %v",
|
||||
tc.table, tc.engine, got, tc.want)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,259 @@
|
||||
// Keeping the fail-closed plane alive across the moments the daemon is not.
|
||||
//
|
||||
// The daemon owns the `inet shater` table, which means the table exists exactly
|
||||
// while the daemon does. Three of those moments are not covered by anything else,
|
||||
// and all three are the same defect wearing different clothes: the protection is
|
||||
// an in-process thing, and the process is not always there.
|
||||
//
|
||||
// BOOT /etc/init.d/shater is START=99. fw4 loaded `lan -> wan ACCEPT` at 19
|
||||
// and netifd brought the LAN up at 20; the clients that reconnect in
|
||||
// between are unprotected until the daemon has been decompressed off
|
||||
// flash, waited out any predecessor, migrated UCI and applied.
|
||||
// RESTART SIGTERM ran an unconditional Teardown — kill_switch was not so much
|
||||
// as consulted — and the successor cannot apply until the init's
|
||||
// shater_wait_stopped loop, `shaterd migrate` and engine start have all
|
||||
// finished. `reload_service` is stop+start, and so is every package
|
||||
// upgrade, so this ran on a routine `Save & Apply`.
|
||||
// NO CONFIG model.ReadUCI failing left the arming call unreached: it sat in the
|
||||
// else-branch of the successful read. Nothing recovered from it either
|
||||
// — Reconcile returns before any plane work, and the cron watchdog sees
|
||||
// a live pidof and its own `uci -q get` fails the same way.
|
||||
//
|
||||
// The answer to all three is one artifact: netplane's BOOT ARMOR, a persisted copy
|
||||
// of the fail-closed holding plane (netplane/armor.go). This file is the daemon's
|
||||
// half — it keeps that copy honest, and it reinstates it in the two cases the
|
||||
// daemon is the only one who can.
|
||||
//
|
||||
// Everything here is deliberately conservative in ONE direction: it never installs
|
||||
// a plane the operator did not ask for. `globals.enabled=0` or `kill_switch=open`
|
||||
// removes the armor and installs nothing, and a deliberate `/etc/init.d/shater
|
||||
// stop` is a handoff-free exit that leaves nothing behind. Fail-closed is a
|
||||
// policy, and a kill switch that outlives its own off switch is not one.
|
||||
package main
|
||||
|
||||
import (
|
||||
"os"
|
||||
|
||||
"github.com/sagernet/sing-box/log"
|
||||
"github.com/sagernet/sing-box/shater/model"
|
||||
"github.com/sagernet/sing-box/shater/netplane"
|
||||
)
|
||||
|
||||
// restartHandoffPath is raised by /etc/init.d/shater around a restart/reload and
|
||||
// cleared by its start (and by a real stop). Its presence at SIGTERM means "this
|
||||
// daemon is being REPLACED", as opposed to "this daemon is being switched off".
|
||||
//
|
||||
// tmpfs on purpose: a marker that survived a power cut would make the first boot
|
||||
// after it look like a restart.
|
||||
//
|
||||
// A var, not a const, only so tests can point it at a temp dir.
|
||||
var restartHandoffPath = "/var/run/shater.restarting"
|
||||
|
||||
// restartHandoffPending reports whether the init script announced a restart.
|
||||
//
|
||||
// Absent is read as "a real stop", which is the SAFE direction to be wrong in: it
|
||||
// degrades to exactly the behaviour that shipped before this file existed (full
|
||||
// teardown), whereas the other default would leave a stopped router blocked.
|
||||
func restartHandoffPending() bool {
|
||||
_, err := os.Stat(restartHandoffPath)
|
||||
return err == nil
|
||||
}
|
||||
|
||||
// armorPlan is what to do about the fail-closed plane at a decision point.
|
||||
type armorPlan int
|
||||
|
||||
const (
|
||||
// armorNothing: install nothing. Either the operator does not want a plane
|
||||
// (disabled / kill_switch=open), or there is nothing to install from.
|
||||
armorNothing armorPlan = iota
|
||||
// armorRender: build the holding plane from the model we just read. Preferred
|
||||
// whenever a model is readable — it reflects the CURRENT interface set, where a
|
||||
// snapshot may predate an interface rename.
|
||||
armorRender
|
||||
// armorSnapshot: reinstate the persisted boot armor. The only option when the
|
||||
// config cannot be read, which is precisely when it is needed.
|
||||
armorSnapshot
|
||||
)
|
||||
|
||||
// armorWanted reports whether m asks for a fail-closed plane at all: the stack is
|
||||
// enabled AND the kill switch is closed. Both halves are the operator's explicit
|
||||
// choice and neither may be second-guessed — `kill_switch=open` is a documented
|
||||
// decision to let traffic through when the engine is down, not an oversight.
|
||||
func armorWanted(m *model.Model) bool {
|
||||
return m != nil && m.Globals.Enabled && netplane.KillSwitchClosed(m.Globals)
|
||||
}
|
||||
|
||||
// planStartupArmor decides what to install when the daemon starts and could NOT
|
||||
// read its config.
|
||||
//
|
||||
// It is only ever consulted on the read-failure path: with a readable model the
|
||||
// applier's own ArmHold does this job (and does it better — it holds the apply
|
||||
// lock while it works). With no model there is nothing to render from, so the
|
||||
// persisted snapshot is the entire answer; with no snapshot either, nothing is
|
||||
// installed, because "this router has never applied an enabled, fail-closed
|
||||
// config" is then the most likely truth and blacking out a LAN on a guess is not
|
||||
// a recovery.
|
||||
func planStartupArmor(readErr error, snapshot bool) armorPlan {
|
||||
if readErr == nil {
|
||||
return armorNothing
|
||||
}
|
||||
if snapshot {
|
||||
return armorSnapshot
|
||||
}
|
||||
return armorNothing
|
||||
}
|
||||
|
||||
// planExitArmor decides what the daemon leaves behind when it is asked to exit.
|
||||
//
|
||||
// The handoff flag is the whole distinction the old code was missing. A restart,
|
||||
// a reload and a package upgrade all reach this point, and in all three the
|
||||
// operator has not asked for protection to end — only for this process to be
|
||||
// replaced. A `stop` has asked for exactly that, and must be obeyed: it is the
|
||||
// operator's escape hatch, and a kill switch that cannot be switched off is a
|
||||
// brick.
|
||||
func planExitArmor(handoff bool, m *model.Model, readErr error, snapshot bool) armorPlan {
|
||||
if !handoff {
|
||||
return armorNothing
|
||||
}
|
||||
if readErr != nil {
|
||||
// Being replaced with an unreadable config: the snapshot is the last thing
|
||||
// this router is known to have wanted, and it is still the honest answer.
|
||||
if snapshot {
|
||||
return armorSnapshot
|
||||
}
|
||||
return armorNothing
|
||||
}
|
||||
if !armorWanted(m) {
|
||||
return armorNothing
|
||||
}
|
||||
return armorRender
|
||||
}
|
||||
|
||||
// refreshBootArmor keeps the persisted holding plane in step with the desired
|
||||
// state. Called after every successful UCI read, so the snapshot on flash always
|
||||
// describes the config the router is actually running.
|
||||
//
|
||||
// Writing is content-gated inside netplane.SaveBootArmor (this runs once a minute
|
||||
// under cron; rewriting an identical file that often is how flash dies), and the
|
||||
// REMOVE half matters just as much as the write: turning the stack off, or opening
|
||||
// the kill switch, has to disarm the next boot too, or the operator's change would
|
||||
// silently come back after a power cut.
|
||||
func refreshBootArmor(m *model.Model, logger log.ContextLogger) {
|
||||
if !armorWanted(m) {
|
||||
if netplane.BootArmorPresent() {
|
||||
if err := netplane.RemoveBootArmor(); err != nil {
|
||||
logger.Warn("boot armor: could not remove ", netplane.BootArmorPath, ": ", err)
|
||||
} else {
|
||||
logger.Info("boot armor removed (the stack is disabled or the kill switch is open): ",
|
||||
"the LAN is no longer blocked at boot before the daemon starts")
|
||||
}
|
||||
}
|
||||
return
|
||||
}
|
||||
ruleset, err := netplane.RenderHoldNft(m)
|
||||
if err != nil {
|
||||
// A transient render failure must not disarm: a stale fail-closed plane is
|
||||
// recoverable (the daemon replaces it seconds into the next boot), an absent
|
||||
// one is a leak.
|
||||
logger.Warn("boot armor: could not render the fail-closed plane: ", err)
|
||||
return
|
||||
}
|
||||
if ruleset == "" {
|
||||
// No divert devices at all — this config intercepts nothing, so there is
|
||||
// nothing for a boot-time plane to protect. Blocking the LAN at boot on
|
||||
// behalf of a config that does not touch it would be a pure outage.
|
||||
if netplane.BootArmorPresent() {
|
||||
if rerr := netplane.RemoveBootArmor(); rerr != nil {
|
||||
logger.Warn("boot armor: could not remove ", netplane.BootArmorPath, ": ", rerr)
|
||||
}
|
||||
}
|
||||
return
|
||||
}
|
||||
changed, serr := netplane.SaveBootArmor(ruleset)
|
||||
switch {
|
||||
case serr != nil:
|
||||
logger.Warn("boot armor: could not write ", netplane.BootArmorPath, ": ", serr,
|
||||
" — the LAN will be unprotected between boot and this daemon's first apply")
|
||||
case changed:
|
||||
logger.Info("boot armor updated (", netplane.BootArmorPath,
|
||||
"): the LAN is fail-closed from early boot until the engine is up")
|
||||
}
|
||||
}
|
||||
|
||||
// armFromSnapshot reinstates the persisted holding plane. why is a short phrase
|
||||
// for the log ("config is unreadable", "restart handoff").
|
||||
func armFromSnapshot(why string, logger log.ContextLogger) {
|
||||
loaded, err := netplane.LoadBootArmor()
|
||||
switch {
|
||||
case err != nil:
|
||||
logger.Error("FAIL-CLOSED PLANE NOT INSTALLED (", why, "): the saved plane ",
|
||||
netplane.BootArmorPath, " could not be loaded: ", err,
|
||||
" — LAN traffic may be reaching the WAN unprotected")
|
||||
case loaded:
|
||||
logger.Error("fail-closed plane reinstated from ", netplane.BootArmorPath,
|
||||
" (", why, "): LAN->WAN forwarding is BLOCKED. ",
|
||||
"SSH, LuCI and the admin panel remain reachable.")
|
||||
default:
|
||||
logger.Warn("no saved fail-closed plane at ", netplane.BootArmorPath, " (", why,
|
||||
"): nothing was installed")
|
||||
}
|
||||
}
|
||||
|
||||
// armFromModel renders the holding plane for m and installs it.
|
||||
func armFromModel(m *model.Model, why string, logger log.ContextLogger) {
|
||||
ruleset, err := netplane.RenderHoldNft(m)
|
||||
if err != nil {
|
||||
logger.Error("FAIL-CLOSED PLANE NOT INSTALLED (", why, "): render failed: ", err)
|
||||
return
|
||||
}
|
||||
if ruleset == "" {
|
||||
// No divert devices: there is nothing this plane would protect.
|
||||
return
|
||||
}
|
||||
if err := netplane.ApplyNft(ruleset); err != nil {
|
||||
logger.Error("FAIL-CLOSED PLANE NOT INSTALLED (", why, "): ", err,
|
||||
" — LAN traffic may be reaching the WAN unprotected")
|
||||
return
|
||||
}
|
||||
logger.Info("fail-closed plane left in place (", why,
|
||||
"): LAN->WAN forwarding stays BLOCKED until the next daemon applies. ",
|
||||
"SSH, LuCI and the admin panel remain reachable.")
|
||||
}
|
||||
|
||||
// armOnUnreadableConfig is the window-3 answer: the daemon is up but cannot read
|
||||
// its own desired state, so it falls back to the last state it persisted.
|
||||
//
|
||||
// A table that is ALREADY loaded is left alone. This runs on every reconcile —
|
||||
// cron fires one a minute — and a `nft -f` is a delete-and-recreate of the whole
|
||||
// table plus a DNS conntrack flush, so re-installing an identical plane sixty
|
||||
// times an hour would be pure churn, and each replacement is itself a brief hole.
|
||||
// The question this path answers is "is there anything at all standing", and once
|
||||
// the answer is yes it stays yes until an apply succeeds and replaces it properly.
|
||||
func armOnUnreadableConfig(logger log.ContextLogger) {
|
||||
if netplane.TableExists() {
|
||||
return
|
||||
}
|
||||
if planStartupArmor(errUnreadableConfig, netplane.BootArmorPresent()) != armorSnapshot {
|
||||
logger.Warn("the config could not be read and there is no saved fail-closed plane at ",
|
||||
netplane.BootArmorPath, " — nothing is protecting the LAN; fix /etc/config/shater ",
|
||||
"(a full /overlay is the usual cause) and reconcile")
|
||||
return
|
||||
}
|
||||
armFromSnapshot("the config could not be read", logger)
|
||||
}
|
||||
|
||||
// errUnreadableConfig is a stand-in for "the read failed" in the call above,
|
||||
// where the concrete error has already been logged by the caller.
|
||||
var errUnreadableConfig = os.ErrInvalid
|
||||
|
||||
// armOnExit is the window-2 answer: what this daemon leaves in the kernel when it
|
||||
// is asked to go away. See planExitArmor for the policy.
|
||||
func armOnExit(handoff bool, logger log.ContextLogger) {
|
||||
m, err := model.ReadUCI()
|
||||
switch planExitArmor(handoff, m, err, netplane.BootArmorPresent()) {
|
||||
case armorRender:
|
||||
armFromModel(m, "restart handoff", logger)
|
||||
case armorSnapshot:
|
||||
armFromSnapshot("restart handoff, config unreadable", logger)
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,182 @@
|
||||
package main
|
||||
|
||||
// The daemon's half of the fail-closed armor: the two decisions that decide
|
||||
// whether the LAN is protected in the moments this process is not running.
|
||||
//
|
||||
// Both are pure functions on purpose. The behaviour they encode can otherwise
|
||||
// only be observed on a router, with root, by killing a daemon at the right
|
||||
// moment and reading `nft list ruleset` — i.e. never, in a gate.
|
||||
|
||||
import (
|
||||
"errors"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
"testing"
|
||||
|
||||
"github.com/sagernet/sing-box/log"
|
||||
"github.com/sagernet/sing-box/shater/model"
|
||||
"github.com/sagernet/sing-box/shater/netplane"
|
||||
)
|
||||
|
||||
func enabledClosed() *model.Model {
|
||||
return &model.Model{Globals: model.Globals{
|
||||
Enabled: true, KillSwitch: "closed", FwmarkBase: 0x2000, TableBase: 0x2000,
|
||||
}, Inbounds: []model.Inbound{{
|
||||
Name: "lan", Enabled: true, Type: "tproxy", Network: "lan",
|
||||
TproxyPort: 12345, TCP: true, UDP: true,
|
||||
}}}
|
||||
}
|
||||
|
||||
// TestPlanExitArmor is W2. SIGTERM used to run an unconditional Teardown — the
|
||||
// only place in this codebase that removes the fail-closed plane without so much
|
||||
// as reading kill_switch — and every `restart`, every `reload_service` (which is
|
||||
// what a LuCI Save & Apply runs) and every package upgrade went through it. The
|
||||
// gap that follows is guaranteed non-empty by the init script itself.
|
||||
//
|
||||
// So the exit has to know WHY it is exiting. What it must never do is confuse the
|
||||
// two directions: a restart that leaves nothing behind is a plaintext window, and
|
||||
// a stop that leaves a block behind is a router the operator cannot un-brick.
|
||||
//
|
||||
// RED BEFORE: there was no such decision — the daemon always tore everything down.
|
||||
func TestPlanExitArmor(t *testing.T) {
|
||||
open := enabledClosed()
|
||||
open.Globals.KillSwitch = "open"
|
||||
disabled := enabledClosed()
|
||||
disabled.Globals.Enabled = false
|
||||
|
||||
cases := []struct {
|
||||
name string
|
||||
handoff bool
|
||||
m *model.Model
|
||||
readErr error
|
||||
snapshot bool
|
||||
want armorPlan
|
||||
}{
|
||||
{"restart, enabled + fail-closed => leave the plane standing",
|
||||
true, enabledClosed(), nil, true, armorRender},
|
||||
{"restart, no snapshot on disk => still render from the live model",
|
||||
true, enabledClosed(), nil, false, armorRender},
|
||||
{"restart, kill_switch=open => the operator chose fail-open; install nothing",
|
||||
true, open, nil, true, armorNothing},
|
||||
{"restart, stack disabled => nothing to protect",
|
||||
true, disabled, nil, true, armorNothing},
|
||||
{"restart, config unreadable => the persisted plane is the last known truth",
|
||||
true, nil, errors.New("uci: no such file"), true, armorSnapshot},
|
||||
{"restart, config unreadable and nothing persisted => nothing to install",
|
||||
true, nil, errors.New("uci: no such file"), false, armorNothing},
|
||||
|
||||
// The escape hatch. A deliberate `/etc/init.d/shater stop` must mean what it
|
||||
// says in every one of these, or the kill switch has no off switch.
|
||||
{"stop, enabled + fail-closed => the plane goes away",
|
||||
false, enabledClosed(), nil, true, armorNothing},
|
||||
{"stop, config unreadable => still goes away",
|
||||
false, nil, errors.New("uci: no such file"), true, armorNothing},
|
||||
}
|
||||
for _, c := range cases {
|
||||
t.Run(c.name, func(t *testing.T) {
|
||||
if got := planExitArmor(c.handoff, c.m, c.readErr, c.snapshot); got != c.want {
|
||||
t.Errorf("planExitArmor(handoff=%v, readErr=%v, snapshot=%v) = %v, want %v",
|
||||
c.handoff, c.readErr, c.snapshot, got, c.want)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
// TestPlanStartupArmor is W3. An unreadable /etc/config/shater — a full /overlay
|
||||
// caught mid `uci commit` is the cause the init script itself documents — left the
|
||||
// arming call unreached, because it sat in the else-branch of the successful read.
|
||||
// Nothing recovered from that: Reconcile returns before any plane work and the
|
||||
// cron watchdog's own `uci -q get` fails identically, so the box sat there with a
|
||||
// live daemon, an answering panel and no table at all.
|
||||
//
|
||||
// RED BEFORE: no decision existed; the read-failure branch only logged.
|
||||
func TestPlanStartupArmor(t *testing.T) {
|
||||
readErr := errors.New("uci: cannot read /etc/config/shater")
|
||||
if got := planStartupArmor(readErr, true); got != armorSnapshot {
|
||||
t.Errorf("unreadable config with a persisted plane = %v, want armorSnapshot", got)
|
||||
}
|
||||
// Nothing persisted means this router has never applied an enabled,
|
||||
// fail-closed config. Blacking out a LAN on that guess is not a recovery.
|
||||
if got := planStartupArmor(readErr, false); got != armorNothing {
|
||||
t.Errorf("unreadable config with no persisted plane = %v, want armorNothing", got)
|
||||
}
|
||||
// A readable config is the applier's business (ArmHold), not this path's.
|
||||
if got := planStartupArmor(nil, true); got != armorNothing {
|
||||
t.Errorf("readable config = %v, want armorNothing", got)
|
||||
}
|
||||
}
|
||||
|
||||
// TestRefreshBootArmorTracksDesiredState is W1's durable half: the file
|
||||
// /etc/init.d/shater-armor loads at START=21 only exists while the operator wants
|
||||
// it to. Writing it is half the contract; REMOVING it when the stack is switched
|
||||
// off or the kill switch is opened is the other half, and the more dangerous one
|
||||
// to get wrong — a stale armor would reinstate, at the next power cut, a block the
|
||||
// operator had already turned off.
|
||||
//
|
||||
// RED BEFORE: neither the file nor this function existed.
|
||||
func TestRefreshBootArmorTracksDesiredState(t *testing.T) {
|
||||
orig := netplane.BootArmorPath
|
||||
netplane.BootArmorPath = filepath.Join(t.TempDir(), "shater", "boot.nft")
|
||||
defer func() { netplane.BootArmorPath = orig }()
|
||||
logger := log.StdLogger()
|
||||
|
||||
refreshBootArmor(enabledClosed(), logger)
|
||||
if !netplane.BootArmorPresent() {
|
||||
t.Fatalf("an enabled, fail-closed config must persist a boot armor")
|
||||
}
|
||||
b, err := os.ReadFile(netplane.BootArmorPath)
|
||||
if err != nil {
|
||||
t.Fatalf("read: %v", err)
|
||||
}
|
||||
// It must be the HOLDING plane — a forward chain that drops — and not the full
|
||||
// tproxy ruleset, which would reference an engine that is not running at boot.
|
||||
for _, must := range []string{"table inet shater", "hook forward", "drop"} {
|
||||
if !strings.Contains(string(b), must) {
|
||||
t.Errorf("the persisted armor must contain %q; got:\n%s", must, string(b))
|
||||
}
|
||||
}
|
||||
if strings.Contains(string(b), "tproxy") {
|
||||
t.Errorf("the persisted armor must NOT divert to an engine that is not running:\n%s", string(b))
|
||||
}
|
||||
|
||||
openKS := enabledClosed()
|
||||
openKS.Globals.KillSwitch = "open"
|
||||
refreshBootArmor(openKS, logger)
|
||||
if netplane.BootArmorPresent() {
|
||||
t.Errorf("kill_switch=open is a documented choice to let traffic through; " +
|
||||
"the boot armor must be removed, not left to block the next boot")
|
||||
}
|
||||
|
||||
refreshBootArmor(enabledClosed(), logger)
|
||||
if !netplane.BootArmorPresent() {
|
||||
t.Fatalf("re-arming after a disarm must work")
|
||||
}
|
||||
off := enabledClosed()
|
||||
off.Globals.Enabled = false
|
||||
refreshBootArmor(off, logger)
|
||||
if netplane.BootArmorPresent() {
|
||||
t.Errorf("globals.enabled=0 must remove the boot armor")
|
||||
}
|
||||
}
|
||||
|
||||
// TestRestartHandoffPending pins the marker's read side, including the default
|
||||
// that matters: an ABSENT marker means "a real stop". Defaulting the other way
|
||||
// would leave a stopped router blocked whenever the init script failed to write
|
||||
// the flag.
|
||||
func TestRestartHandoffPending(t *testing.T) {
|
||||
orig := restartHandoffPath
|
||||
defer func() { restartHandoffPath = orig }()
|
||||
dir := t.TempDir()
|
||||
restartHandoffPath = filepath.Join(dir, "shater.restarting")
|
||||
|
||||
if restartHandoffPending() {
|
||||
t.Errorf("an absent marker must read as a real stop")
|
||||
}
|
||||
if err := os.WriteFile(restartHandoffPath, nil, 0o644); err != nil {
|
||||
t.Fatalf("write marker: %v", err)
|
||||
}
|
||||
if !restartHandoffPending() {
|
||||
t.Errorf("a present marker must read as a restart handoff")
|
||||
}
|
||||
}
|
||||
@@ -289,10 +289,27 @@ func cmdRun() int {
|
||||
// reconcile. No-op when no iface-driven profiles are configured.
|
||||
go watchActiveProfile(applier, logger)
|
||||
|
||||
// Keep the persisted fail-closed plane in step with the config BEFORE anything
|
||||
// is attempted: it is what protects the LAN at the NEXT boot (and across the
|
||||
// next restart), and an engine start that hangs for a minute must not be what
|
||||
// stands between a config change and its armor being written.
|
||||
if readErr == nil {
|
||||
refreshBootArmor(m, logger)
|
||||
}
|
||||
|
||||
// Initial apply. A failed initial apply must NOT crash-loop the box into a
|
||||
// blackout: log it and stay up so a later SIGHUP/apply can fix the config.
|
||||
if readErr != nil {
|
||||
logger.Error("initial ReadUCI failed (staying up): ", readErr)
|
||||
// ...but STAYING UP IS NOT THE SAME AS BEING SAFE. This branch used to end
|
||||
// here, which meant an unreadable /etc/config/shater — a full /overlay caught
|
||||
// mid `uci commit` is the documented cause — left the router with no table at
|
||||
// all, permanently: nothing else installs one (Reconcile returns before any
|
||||
// plane work, and the cron watchdog's own `uci -q get` fails identically), and
|
||||
// the panel reported a live daemon the whole time. The persisted holding plane
|
||||
// is the last thing this router is KNOWN to have wanted, and it needs nothing
|
||||
// readable to be true.
|
||||
armOnUnreadableConfig(logger)
|
||||
} else if m.Globals.Enabled {
|
||||
// ARM FIRST, THEN TRY. Install the fail-closed holding plane BEFORE the
|
||||
// engine is attempted, so the gap between daemon start and a working engine
|
||||
@@ -369,9 +386,20 @@ func cmdRun() int {
|
||||
// learn whether the plane is meant to be up (globals.enabled) so a
|
||||
// reconcile error can raise the kill-switch alert.
|
||||
enabled := false
|
||||
if mm, e := model.ReadUCI(); e == nil {
|
||||
if mm, e := model.ReadUCI(); e != nil {
|
||||
// The live config just became unreadable. Same answer as at startup:
|
||||
// reinstate what this router last persisted, rather than run on with
|
||||
// whatever the kernel happens to hold.
|
||||
logger.Error("reconcile: could not read the config: ", e)
|
||||
armOnUnreadableConfig(logger)
|
||||
} else {
|
||||
notifier.Update(mm.Alerts)
|
||||
enabled = mm.Globals.Enabled
|
||||
// Keep the persisted fail-closed plane in step with the config the
|
||||
// operator just changed — including the disarm half, so turning the
|
||||
// stack off (or opening the kill switch) also stops the next boot from
|
||||
// blocking the LAN.
|
||||
refreshBootArmor(mm, logger)
|
||||
// Pick up a changed stats backend / sizing without a daemon restart.
|
||||
// A no-op when nothing changed, so a routine reconcile never churns
|
||||
// the store (and never restarts its log cursors).
|
||||
@@ -393,10 +421,27 @@ func cmdRun() int {
|
||||
// DNS-query manager). Re-point the aggregator at the current manager.
|
||||
statsAgg.Resubscribe()
|
||||
case syscall.SIGTERM, syscall.SIGINT:
|
||||
logger.Info("signal ", sig, ": honest teardown + exit")
|
||||
// Being REPLACED is not the same as being switched off, and until now
|
||||
// this path could not tell the difference: Teardown does not consult
|
||||
// kill_switch at all (compare holdLocked, which does), so `restart`,
|
||||
// `reload_service` — which is stop+start, i.e. every LuCI Save & Apply —
|
||||
// and every package upgrade dismantled the fail-closed plane and left the
|
||||
// LAN forwarding in the clear for as long as the successor needed to come
|
||||
// up. That interval is guaranteed non-empty by the init itself: it waits
|
||||
// for this process to exit, then runs `shaterd migrate`, then starts the
|
||||
// daemon, which then has to build an engine.
|
||||
//
|
||||
// The init script announces a restart with a tmpfs marker; absent it, this
|
||||
// is a deliberate stop and the plane goes away for good, which is the
|
||||
// operator's escape hatch and must keep working.
|
||||
handoff := restartHandoffPending()
|
||||
logger.Info("signal ", sig, ": honest teardown + exit (restart handoff: ", handoff, ")")
|
||||
if err := applier.Teardown(); err != nil {
|
||||
logger.Error("teardown: ", err)
|
||||
}
|
||||
// AFTER the teardown, never before: Teardown deletes the table, so a plane
|
||||
// installed first would simply be removed again.
|
||||
armOnExit(handoff, logger)
|
||||
return 0
|
||||
}
|
||||
}
|
||||
@@ -991,14 +1036,23 @@ func handleCtl(conn net.Conn, a *apply.Applier, ps *panel.Server, sa stats.Stats
|
||||
writeLine(conn, string(b))
|
||||
case "reconcile":
|
||||
ch, err := a.Reconcile()
|
||||
// The panel and the CLI reconcile over this socket, not by SIGHUP, so the
|
||||
// persisted fail-closed plane has to be refreshed here too — otherwise
|
||||
// enabling the stack from the panel left the next boot unarmed (and
|
||||
// DISABLING it left the next boot armed) until some unrelated SIGHUP
|
||||
// happened along.
|
||||
if mm, e := model.ReadUCI(); e == nil {
|
||||
refreshBootArmor(mm, l)
|
||||
}
|
||||
writeResult(conn, ch, err)
|
||||
case "apply":
|
||||
a.Snapshot()
|
||||
changed, err := a.Reconcile()
|
||||
if err == nil {
|
||||
if m, e := model.ReadUCI(); e == nil {
|
||||
if m, e := model.ReadUCI(); e == nil {
|
||||
if err == nil {
|
||||
a.ArmRollback(m.Globals.ConfirmTimeout)
|
||||
}
|
||||
refreshBootArmor(m, l)
|
||||
}
|
||||
writeResult(conn, changed, err)
|
||||
case "confirm":
|
||||
|
||||
@@ -104,6 +104,18 @@ func (e *Engine) HTTPClient(via string) (*http.Client, error) {
|
||||
// resolved one (the group test, grouptest.go) guarantee the request cannot end up
|
||||
// anywhere else — in particular not on the direct outbound, which would report the
|
||||
// ISP's address as the tunnel's exit address.
|
||||
//
|
||||
// # Every caller MUST call CloseIdleConnections on the returned client
|
||||
//
|
||||
// A fresh http.Transport is built per call and belongs to that call alone. Its idle
|
||||
// connections are not ordinary sockets: each is a live proxying session through an
|
||||
// engine outbound, with a read loop and a write loop of its own, and the DialContext
|
||||
// closure above captures the outbound OBJECT — so an idle connection keeps a whole
|
||||
// retired engine generation reachable long after Apply swapped it out and its 5s
|
||||
// close budget expired. Dropping the client without closing it therefore leaks far
|
||||
// more than a socket.
|
||||
//
|
||||
// grouptest.go and shater/generate/ruleset.go get this right; copy them.
|
||||
func httpClientVia(ob adapter.Outbound, timeout time.Duration) *http.Client {
|
||||
transport := &http.Transport{
|
||||
// Dial the underlying TCP connection through the selected outbound. The
|
||||
@@ -112,9 +124,16 @@ func httpClientVia(ob adapter.Outbound, timeout time.Duration) *http.Client {
|
||||
DialContext: func(ctx context.Context, network, addr string) (net.Conn, error) {
|
||||
return ob.DialContext(ctx, network, M.ParseSocksaddr(addr))
|
||||
},
|
||||
ForceAttemptHTTP2: true,
|
||||
MaxIdleConns: 8,
|
||||
IdleConnTimeout: 90 * time.Second,
|
||||
ForceAttemptHTTP2: true,
|
||||
MaxIdleConns: 8,
|
||||
// The backstop for a caller that forgets to close, not the intended
|
||||
// mechanism. 90s (the net/http default this used to carry) is eighteen times
|
||||
// the engine's 5s close budget, so a single forgotten client could pin a dead
|
||||
// generation through more than a minute and a half of it. 15s is still ample
|
||||
// for the reuse this actually buys — a redirect chain or the second request of
|
||||
// a subscription fetch, both within seconds — while bounding the damage of a
|
||||
// leak to about one apply cycle.
|
||||
IdleConnTimeout: 15 * time.Second,
|
||||
TLSHandshakeTimeout: 15 * time.Second,
|
||||
ExpectContinueTimeout: 1 * time.Second,
|
||||
}
|
||||
|
||||
@@ -116,6 +116,17 @@ func isWANIfaceName(name string) bool { return strings.HasPrefix(name, "wan") }
|
||||
// inbound connections — port forwards stop working, and nothing in the UI explains
|
||||
// why.
|
||||
//
|
||||
// THE REVERSE CHECK IS NOT HERE, ON PURPOSE. "This router has a LAN the config
|
||||
// never mentions" is the mirror image of this one and matters more (the factory
|
||||
// config ships a single `lan` inbound, and a VLAN added in LuCI is then proxied by
|
||||
// nothing and covered by no kill switch), but it cannot be answered from this
|
||||
// package: it needs the ENUMERATION of the router's interfaces and the fw4
|
||||
// forwarding graph, and this package is a stdlib-only leaf that netplane imports
|
||||
// — the dependency runs the wrong way, and NetClassifier deliberately answers one
|
||||
// question about one name rather than exposing an inventory. It lives in
|
||||
// netplane/coverage.go, where both the inventory and the divert set already are,
|
||||
// and reaches Status on the netplane warning channel.
|
||||
//
|
||||
// classify may be nil, in which case only the name heuristic applies.
|
||||
func ValidateRules(rules []Rule, classify NetClassifier) []Warning {
|
||||
if classify == nil {
|
||||
|
||||
@@ -10,6 +10,7 @@ import (
|
||||
"encoding/json"
|
||||
"fmt"
|
||||
"net/netip"
|
||||
"os"
|
||||
"os/exec"
|
||||
"strings"
|
||||
"sync"
|
||||
@@ -657,11 +658,47 @@ func fwmarkRulePresent(out string, mark uint32) bool {
|
||||
return false
|
||||
}
|
||||
|
||||
// sysClassNet is the kernel's list of network devices. A var only so a test can
|
||||
// point DeviceExists at a fixture directory.
|
||||
var sysClassNet = "/sys/class/net"
|
||||
|
||||
// DeviceExists reports whether the kernel actually has a network device by this
|
||||
// name.
|
||||
//
|
||||
// It exists because IfaceDevice's last resort is to hand back the name it was
|
||||
// given (see below), and nothing downstream could tell that apart from a real
|
||||
// device: nftValidDev checks the CHARACTERS of a name, not its existence, so
|
||||
// `lan` passes as readily as `br-lan` and produces rules that load cleanly and
|
||||
// match nothing. This is the one cheap question that separates the two, and it
|
||||
// asks the kernel rather than netifd so a device netifd does not manage still
|
||||
// counts. Used by coverage.go; deliberately not consulted by IfaceDevice itself,
|
||||
// which must keep returning a usable string for every caller.
|
||||
func DeviceExists(dev string) bool {
|
||||
dev = strings.TrimSpace(dev)
|
||||
if dev == "" || strings.ContainsAny(dev, "/\\") {
|
||||
return false
|
||||
}
|
||||
_, err := os.Stat(sysClassNet + "/" + dev)
|
||||
return err == nil
|
||||
}
|
||||
|
||||
// IfaceDevice resolves a UCI interface name (e.g. "lan") to its actual L3 device
|
||||
// (e.g. "br-lan") for nft iifname matching / SO_BINDTODEVICE — bridges/VLANs have
|
||||
// device != name, so matching the raw UCI name would silently intercept nothing.
|
||||
// Empty name defaults to br-lan; an unresolvable name is returned unchanged (it
|
||||
// may already be a device).
|
||||
//
|
||||
// THAT LAST FALLBACK IS A KNOWN TRAP, and it is deliberate. Returning the raw
|
||||
// name is right for the case it was written for — an operator who wrote a device
|
||||
// name where an interface name was expected — and it is a silent leak for the
|
||||
// case where the name resolves to nothing at all: the caller renders
|
||||
// `iifname "lan" ... drop`, the kernel accepts it, and that network is neither
|
||||
// diverted nor failed closed on. The fallback is kept (refusing here would strand
|
||||
// every legitimate device-name user) and the ambiguity is resolved one level up
|
||||
// instead: divertRef carries what the operator wrote alongside what it resolved
|
||||
// to, and coverage.go asks DeviceExists whether the result is real, reporting the
|
||||
// gap to the operator. Callers that need the distinction must go through there,
|
||||
// not re-derive it from the string.
|
||||
func IfaceDevice(name string) string {
|
||||
if name == "" {
|
||||
return "br-lan"
|
||||
|
||||
@@ -0,0 +1,170 @@
|
||||
package netplane
|
||||
|
||||
// The BOOT ARMOR: the fail-closed holding plane, persisted to flash so that
|
||||
// something OTHER than the running daemon can put it back.
|
||||
//
|
||||
// # Why a file exists at all
|
||||
//
|
||||
// Every protection this package installs lives inside one long-lived Go process.
|
||||
// That is fine while the process is running, and it is exactly nothing in the two
|
||||
// windows where it is not:
|
||||
//
|
||||
// - BOOT. /etc/init.d/shater is START=99, i.e. after fw4 (START=19) has already
|
||||
// loaded `lan -> wan ACCEPT` and after netifd (START=20) has brought the LAN
|
||||
// bridge up. Wi-Fi associates and every client reconnects in that window, and
|
||||
// the daemon only reaches applier.ArmHold after procd has decompressed a
|
||||
// UPX-packed binary off flash, waited out any predecessor, run `shaterd
|
||||
// migrate` and read UCI. On a small router that is seconds — every one of
|
||||
// them with `kill_switch=closed` configured and the LAN forwarding to the WAN
|
||||
// in the clear.
|
||||
// - AN UNREADABLE CONFIG. model.ReadUCI failing (a full /overlay caught mid
|
||||
// `uci commit` is the documented cause — see /etc/init.d/shater) used to mean
|
||||
// NO plane was ever installed: the arming call sat in the else-branch of the
|
||||
// successful read, Reconcile returns before any of it, and the cron watchdog
|
||||
// sees a live pidof and its own `uci -q get` fails too. Nothing in the system
|
||||
// could recover, and nothing said so.
|
||||
//
|
||||
// Both need a description of the fail-closed plane that survives the process. So
|
||||
// the daemon writes RenderHoldNft's output here on every apply, and two readers
|
||||
// use it: /etc/init.d/shater-armor at boot (before the daemon exists), and the
|
||||
// daemon itself when it cannot read its own config.
|
||||
//
|
||||
// # Why the HOLDING plane and not the full ruleset
|
||||
//
|
||||
// The full ruleset diverts to the engine's tproxy port. With no engine listening,
|
||||
// every `tproxy` statement returns NFT_BREAK and the packets fall through to the
|
||||
// forward chain, which does drop them — so it would "work". But it also needs
|
||||
// kmod-nft-tproxy loaded, declares dynamic sets and counters, and encodes a
|
||||
// tproxy port that may no longer be the configured one. The holding plane is a
|
||||
// single forward chain of accepts and two drops: it needs nothing but nf_tables,
|
||||
// it is what the daemon itself installs when the engine is down, and it says
|
||||
// exactly one thing — this device's forwarded traffic does not leave.
|
||||
//
|
||||
// # The file IS the arm token
|
||||
//
|
||||
// Its presence means "the last configuration this router applied was enabled and
|
||||
// fail-closed". Nothing else has to be readable for that to be true, which is the
|
||||
// whole point in the unreadable-config case. Correspondingly it is REMOVED the
|
||||
// moment that stops being true: globals.enabled=0, kill_switch=open, or a
|
||||
// deliberate `/etc/init.d/shater stop`. An operator who turned the stack off must
|
||||
// not find it reinstated by the next power cut.
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
|
||||
"github.com/sagernet/sing-box/shater/model"
|
||||
)
|
||||
|
||||
// BootArmorPath is where the persisted holding plane lives. It is a var, not a
|
||||
// const, ONLY so tests can point it at a temp dir; production never assigns it.
|
||||
//
|
||||
// /etc/shater, not /var/run: it has to survive the reboot it exists for. The
|
||||
// daemon already creates that directory for the engine's cache DB (D16).
|
||||
var BootArmorPath = "/etc/shater/boot.nft"
|
||||
|
||||
// KillSwitchClosed reports the EFFECTIVE kill-switch policy: closed (fail-closed)
|
||||
// unless the operator explicitly wrote "open". Exported so the daemon decides with
|
||||
// the same rule the ruleset renderer does — a second copy of this predicate is how
|
||||
// the plane and the policy end up disagreeing.
|
||||
func KillSwitchClosed(g model.Globals) bool { return genGlobalClosed(g) }
|
||||
|
||||
// BootArmorPresent reports whether a persisted holding plane exists.
|
||||
func BootArmorPresent() bool {
|
||||
st, err := os.Stat(BootArmorPath)
|
||||
return err == nil && st.Mode().IsRegular() && st.Size() > 0
|
||||
}
|
||||
|
||||
// SaveBootArmor persists ruleset as the boot armor and reports whether the file's
|
||||
// contents actually changed.
|
||||
//
|
||||
// The no-change fast path is not an optimisation, it is a durability requirement:
|
||||
// this is called on EVERY apply, cron reconciles once a minute, and the target is
|
||||
// raw flash on a router that is expected to run for years. Rewriting an identical
|
||||
// 700-byte file 525 600 times a year is how an eMMC/NAND block wears out for
|
||||
// nothing.
|
||||
//
|
||||
// The write is atomic (temp file in the same directory, then rename), so a power
|
||||
// cut mid-write leaves either the previous armor or none — never a half-written
|
||||
// ruleset that `nft -f` would reject at the next boot, which is precisely the
|
||||
// boot where it matters.
|
||||
func SaveBootArmor(ruleset string) (changed bool, err error) {
|
||||
if strings.TrimSpace(ruleset) == "" {
|
||||
return false, fmt.Errorf("refusing to save an empty boot armor")
|
||||
}
|
||||
if old, rerr := os.ReadFile(BootArmorPath); rerr == nil && string(old) == ruleset {
|
||||
return false, nil
|
||||
}
|
||||
dir := filepath.Dir(BootArmorPath)
|
||||
if err := os.MkdirAll(dir, 0o755); err != nil {
|
||||
return false, err
|
||||
}
|
||||
tmp, err := os.CreateTemp(dir, ".boot.nft.*")
|
||||
if err != nil {
|
||||
return false, err
|
||||
}
|
||||
name := tmp.Name()
|
||||
defer func() {
|
||||
if err != nil {
|
||||
_ = os.Remove(name)
|
||||
}
|
||||
}()
|
||||
if _, err = tmp.WriteString(ruleset); err != nil {
|
||||
_ = tmp.Close()
|
||||
return false, err
|
||||
}
|
||||
// Sync before the rename: the rename is what makes the file visible, and on
|
||||
// the flash filesystems this ships on an unsynced payload can survive the
|
||||
// rename as zeroes.
|
||||
if err = tmp.Sync(); err != nil {
|
||||
_ = tmp.Close()
|
||||
return false, err
|
||||
}
|
||||
if err = tmp.Close(); err != nil {
|
||||
return false, err
|
||||
}
|
||||
if err = os.Chmod(name, 0o644); err != nil {
|
||||
return false, err
|
||||
}
|
||||
if err = os.Rename(name, BootArmorPath); err != nil {
|
||||
return false, err
|
||||
}
|
||||
return true, nil
|
||||
}
|
||||
|
||||
// RemoveBootArmor deletes the persisted holding plane. Absent is success: this is
|
||||
// called on every apply that finds the stack disabled or fail-open, and "there was
|
||||
// nothing to remove" is the normal case, not a fault.
|
||||
func RemoveBootArmor() error {
|
||||
if err := os.Remove(BootArmorPath); err != nil && !os.IsNotExist(err) {
|
||||
return err
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// LoadBootArmor installs the persisted holding plane into the kernel and reports
|
||||
// whether there was one to install.
|
||||
//
|
||||
// It goes through ApplyNft rather than `nft -f <path>` so the snapshot is
|
||||
// VALIDATED (`nft -c`) before it is loaded and so a wedged nft is killed rather
|
||||
// than left blocking — the same treatment every other ruleset in this package
|
||||
// gets. A snapshot that no longer parses (an nft downgrade, a truncated file) is
|
||||
// therefore an error the caller can report, not a half-loaded table.
|
||||
func LoadBootArmor() (loaded bool, err error) {
|
||||
b, rerr := os.ReadFile(BootArmorPath)
|
||||
if rerr != nil {
|
||||
if os.IsNotExist(rerr) {
|
||||
return false, nil
|
||||
}
|
||||
return false, rerr
|
||||
}
|
||||
if strings.TrimSpace(string(b)) == "" {
|
||||
return false, fmt.Errorf("boot armor %s is empty", BootArmorPath)
|
||||
}
|
||||
if err := ApplyNft(string(b)); err != nil {
|
||||
return false, err
|
||||
}
|
||||
return true, nil
|
||||
}
|
||||
@@ -0,0 +1,178 @@
|
||||
package netplane
|
||||
|
||||
// The BOOT ARMOR's storage contract. It is the artifact three separate windows
|
||||
// depend on (early boot, restart handoff, unreadable config), so the properties
|
||||
// that matter are the durability ones: it must survive, it must not wear flash
|
||||
// out, and a power cut must never leave a half-written ruleset behind for the one
|
||||
// boot that reads it.
|
||||
|
||||
import (
|
||||
"os"
|
||||
"os/exec"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
"testing"
|
||||
|
||||
"github.com/sagernet/sing-box/shater/model"
|
||||
)
|
||||
|
||||
func armorTempPath(t *testing.T) string {
|
||||
t.Helper()
|
||||
p := filepath.Join(t.TempDir(), "shater", "boot.nft")
|
||||
orig := BootArmorPath
|
||||
BootArmorPath = p
|
||||
t.Cleanup(func() { BootArmorPath = orig })
|
||||
return p
|
||||
}
|
||||
|
||||
// TestBootArmorSaveIsContentGated: the daemon calls this on every apply and cron
|
||||
// reconciles once a minute, so an unconditional write is ~525 000 rewrites a year
|
||||
// of an identical file onto raw flash. An unchanged ruleset must not touch the
|
||||
// file at all.
|
||||
func TestBootArmorSaveIsContentGated(t *testing.T) {
|
||||
path := armorTempPath(t)
|
||||
|
||||
if BootArmorPresent() {
|
||||
t.Fatalf("BootArmorPresent must be false before anything is written")
|
||||
}
|
||||
changed, err := SaveBootArmor("table inet shater { }\n")
|
||||
if err != nil || !changed {
|
||||
t.Fatalf("first SaveBootArmor = (%v, %v), want (true, nil)", changed, err)
|
||||
}
|
||||
if !BootArmorPresent() {
|
||||
t.Fatalf("BootArmorPresent must be true after a save")
|
||||
}
|
||||
st1, err := os.Stat(path)
|
||||
if err != nil {
|
||||
t.Fatalf("stat: %v", err)
|
||||
}
|
||||
|
||||
changed, err = SaveBootArmor("table inet shater { }\n")
|
||||
if err != nil || changed {
|
||||
t.Fatalf("identical SaveBootArmor = (%v, %v), want (false, nil) — an unchanged "+
|
||||
"armor must not be rewritten", changed, err)
|
||||
}
|
||||
st2, err := os.Stat(path)
|
||||
if err != nil {
|
||||
t.Fatalf("stat: %v", err)
|
||||
}
|
||||
if !st1.ModTime().Equal(st2.ModTime()) {
|
||||
t.Errorf("the file was rewritten for identical content (mtime %v -> %v)",
|
||||
st1.ModTime(), st2.ModTime())
|
||||
}
|
||||
|
||||
if changed, err = SaveBootArmor("table inet shater { chain forward { } }\n"); err != nil || !changed {
|
||||
t.Fatalf("changed SaveBootArmor = (%v, %v), want (true, nil)", changed, err)
|
||||
}
|
||||
b, err := os.ReadFile(path)
|
||||
if err != nil || !strings.Contains(string(b), "chain forward") {
|
||||
t.Fatalf("file does not hold the new ruleset: %q (%v)", string(b), err)
|
||||
}
|
||||
|
||||
// An empty armor is worse than none: shater-armor would load a file that
|
||||
// blocks nothing while the log claims the LAN is protected.
|
||||
if _, err := SaveBootArmor(" \n"); err == nil {
|
||||
t.Errorf("SaveBootArmor must refuse an empty ruleset")
|
||||
}
|
||||
|
||||
if err := RemoveBootArmor(); err != nil {
|
||||
t.Fatalf("RemoveBootArmor: %v", err)
|
||||
}
|
||||
if BootArmorPresent() {
|
||||
t.Fatalf("BootArmorPresent must be false after removal")
|
||||
}
|
||||
// Removing what is not there is the normal case (every apply of a disabled
|
||||
// stack), not a fault.
|
||||
if err := RemoveBootArmor(); err != nil {
|
||||
t.Errorf("RemoveBootArmor on an absent file = %v, want nil", err)
|
||||
}
|
||||
}
|
||||
|
||||
// TestBootArmorSaveIsAtomic: nothing but the finished file may ever be visible at
|
||||
// BootArmorPath. The one boot that reads it is the boot after a power cut.
|
||||
func TestBootArmorSaveIsAtomic(t *testing.T) {
|
||||
path := armorTempPath(t)
|
||||
if _, err := SaveBootArmor("table inet shater { }\n"); err != nil {
|
||||
t.Fatalf("SaveBootArmor: %v", err)
|
||||
}
|
||||
entries, err := os.ReadDir(filepath.Dir(path))
|
||||
if err != nil {
|
||||
t.Fatalf("readdir: %v", err)
|
||||
}
|
||||
if len(entries) != 1 || entries[0].Name() != filepath.Base(path) {
|
||||
var names []string
|
||||
for _, e := range entries {
|
||||
names = append(names, e.Name())
|
||||
}
|
||||
t.Errorf("the save left temp files behind: %v", names)
|
||||
}
|
||||
}
|
||||
|
||||
// TestLoadBootArmorFeedsNft proves the reinstate path actually reaches the kernel
|
||||
// through the SAME validated loader every other ruleset uses (nft -c, then nft
|
||||
// -f), rather than a bare `nft -f <file>` that could half-load a truncated
|
||||
// snapshot.
|
||||
func TestLoadBootArmorFeedsNft(t *testing.T) {
|
||||
armorTempPath(t)
|
||||
|
||||
// Nothing saved: not an error, just nothing to do.
|
||||
loaded, err := LoadBootArmor()
|
||||
if err != nil || loaded {
|
||||
t.Fatalf("LoadBootArmor with no armor = (%v, %v), want (false, nil)", loaded, err)
|
||||
}
|
||||
|
||||
const ruleset = "table inet shater { chain forward { } }\n"
|
||||
if _, err := SaveBootArmor(ruleset); err != nil {
|
||||
t.Fatalf("SaveBootArmor: %v", err)
|
||||
}
|
||||
|
||||
var rec []string
|
||||
orig := execCommand
|
||||
defer func() { execCommand = orig }()
|
||||
execCommand = func(name string, arg ...string) *exec.Cmd {
|
||||
rec = append(rec, strings.Join(append([]string{name}, arg...), " "))
|
||||
cs := append([]string{"-test.run=TestNetplaneHelperProcess", "--", name}, arg...)
|
||||
cmd := exec.Command(os.Args[0], cs...)
|
||||
cmd.Env = append(os.Environ(), "GO_WANT_HELPER_PROCESS=1")
|
||||
return cmd
|
||||
}
|
||||
|
||||
loaded, err = LoadBootArmor()
|
||||
if err != nil || !loaded {
|
||||
t.Fatalf("LoadBootArmor = (%v, %v), want (true, nil)", loaded, err)
|
||||
}
|
||||
var sawCheck, sawLoad bool
|
||||
for _, c := range rec {
|
||||
switch c {
|
||||
case "nft -c -f -":
|
||||
sawCheck = true
|
||||
case "nft -f -":
|
||||
sawLoad = true
|
||||
}
|
||||
}
|
||||
if !sawCheck || !sawLoad {
|
||||
t.Errorf("LoadBootArmor must VALIDATE then load (nft -c -f -, nft -f -); commands: %v", rec)
|
||||
}
|
||||
}
|
||||
|
||||
// TestKillSwitchClosed pins that the daemon's arming decision and the renderer's
|
||||
// drop decision come from one predicate. Two copies of "is the kill switch
|
||||
// closed" is how an armed boot ends up protecting a router the operator asked to
|
||||
// fail open (or the reverse).
|
||||
func TestKillSwitchClosed(t *testing.T) {
|
||||
cases := []struct {
|
||||
val string
|
||||
want bool
|
||||
}{
|
||||
{"", true}, {"closed", true}, {"CLOSED", true}, {"nonsense", true},
|
||||
{"open", false}, {"OPEN", false}, {" open ", false},
|
||||
}
|
||||
for _, c := range cases {
|
||||
if got := KillSwitchClosed(model.Globals{KillSwitch: c.val}); got != c.want {
|
||||
t.Errorf("KillSwitchClosed(%q) = %v, want %v", c.val, got, c.want)
|
||||
}
|
||||
if got := genGlobalClosed(model.Globals{KillSwitch: c.val}); got != c.want {
|
||||
t.Errorf("genGlobalClosed(%q) = %v, want %v — the two must not diverge", c.val, got, c.want)
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,234 @@
|
||||
package netplane
|
||||
|
||||
// COVERAGE: the gap between the networks this router carries and the networks
|
||||
// this data plane actually touches.
|
||||
//
|
||||
// Everything else in this package answers "what does the plan say?". This file
|
||||
// answers the question nothing asked before: "what does the plan leave out?".
|
||||
// Two omissions, both silent, both leaks:
|
||||
//
|
||||
// 1. A DEVICE NAME THAT NAMES NOTHING. IfaceDevice returns the raw UCI name when
|
||||
// neither ubus nor uci can resolve it, and nftValidDev happily accepts "lan"
|
||||
// as a syntactically fine device name. The resulting `iifname "lan" ... drop`
|
||||
// loads into the kernel without a murmur and matches no packet ever, so that
|
||||
// network is neither diverted nor failed closed on. This is the failure mode
|
||||
// nftValidDev's own doc comment warns about at length — but the warning it
|
||||
// emits only fires for names it REJECTED, never for names it accepted that
|
||||
// resolve to nothing.
|
||||
//
|
||||
// 2. A CLIENT NETWORK NOBODY NAMED. The divert set is built exclusively from
|
||||
// what the operator listed: `config inbound` plus rules with `iface:`/`zone:`
|
||||
// sources. The factory config ships ONE inbound, `lan`. Add a VLAN or a guest
|
||||
// SSID in LuCI and its clients route to the internet through this router
|
||||
// without ever entering the engine — no proxy, and no kill switch either.
|
||||
// netplane.Interfaces() has been able to enumerate exactly this since the
|
||||
// egress picker was built; the result went to the panel and nowhere else.
|
||||
//
|
||||
// # Why these WARN and do not refuse
|
||||
//
|
||||
// The rejected-name path refuses the whole apply when the kill switch is closed,
|
||||
// and that is right there: a name we cannot express is a stable configuration
|
||||
// error, refusing leaves the PREVIOUS ruleset loaded and protecting, and the
|
||||
// operator has something concrete to fix.
|
||||
//
|
||||
// Neither of these has that shape.
|
||||
//
|
||||
// An unresolved name has no fail-closed action available at all — we do not know
|
||||
// the device, so there is nothing to write a drop against. Refusing would abort
|
||||
// the apply and leave the LAN on plain routing, i.e. it would produce the very
|
||||
// leak it is objecting to, plus an outage. And it is not necessarily stable: at
|
||||
// boot, netifd may not have answered yet.
|
||||
//
|
||||
// An uncovered network must not be closed silently for a much simpler reason:
|
||||
// this router does not know whether the operator MEANT to proxy it. A guest SSID
|
||||
// or an IoT VLAN deliberately kept off the tunnel is a legitimate configuration,
|
||||
// and dropping its forwarded traffic because our inbound list does not mention it
|
||||
// would cut a working network on a guess. So it is named, loudly, and the
|
||||
// decision stays with the person who built the network.
|
||||
//
|
||||
// Both travel on the netplane warning channel, which apply grades CRITICAL
|
||||
// wholesale — which is exactly the severity apply/warnings.go defines for "an
|
||||
// interface the kill-switch does not cover", in those words.
|
||||
//
|
||||
// # Why the whole file is silent off a router
|
||||
//
|
||||
// Every check here is gated on Interfaces() returning something. With no ubus
|
||||
// there is no inventory, and a coverage claim built on no inventory is a guess.
|
||||
// That also keeps the checks out of the unit suite's hair: on a build machine the
|
||||
// probe fails, the gate closes, and rendering behaves exactly as it did before.
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"strings"
|
||||
|
||||
"github.com/sagernet/sing-box/shater/model"
|
||||
)
|
||||
|
||||
// coverageWarnings reports what this plan does not cover. refs is the divert set
|
||||
// with provenance (nftDivertRefs); m supplies the egress list, which is the one
|
||||
// thing that tells a client network apart from an uplink we deliberately send
|
||||
// traffic OUT of.
|
||||
//
|
||||
// Returns nil when the router cannot be enumerated — see the package comment.
|
||||
func coverageWarnings(m *model.Model, refs []divertRef) []string {
|
||||
infos := Interfaces()
|
||||
if len(infos) == 0 {
|
||||
return nil
|
||||
}
|
||||
var out []string
|
||||
out = append(out, unresolvedDeviceWarnings(refs, infos)...)
|
||||
out = append(out, uncoveredNetworkWarnings(m, refs, infos)...)
|
||||
return out
|
||||
}
|
||||
|
||||
// unresolvedDeviceWarnings reports each diverted device that the kernel does not
|
||||
// have.
|
||||
//
|
||||
// "Does not have" is answered from two independent sources, and a hit in either
|
||||
// clears the device: the netifd inventory (which is how a normal interface is
|
||||
// resolved) and /sys/class/net (which is how a raw device pulled straight out of
|
||||
// a firewall zone's `list device`, or any device netifd does not manage, still
|
||||
// counts). Only a device unknown to BOTH is reported — an over-eager warning here
|
||||
// would be indistinguishable from the real one and would train the operator to
|
||||
// ignore both.
|
||||
func unresolvedDeviceWarnings(refs []divertRef, infos []IfaceInfo) []string {
|
||||
known := make(map[string]bool, len(infos))
|
||||
for _, i := range infos {
|
||||
if i.Device != "" {
|
||||
known[i.Device] = true
|
||||
}
|
||||
}
|
||||
var out []string
|
||||
for _, r := range refs {
|
||||
if known[r.Dev] || DeviceExists(r.Dev) {
|
||||
continue
|
||||
}
|
||||
out = append(out, fmt.Sprintf(
|
||||
"interface %q: this router has no network device called %q, and that is the name the "+
|
||||
"ruleset was written against — every rule scoped to it (the TPROXY divert, the "+
|
||||
"fail-closed drop, the accept_local ingress knob) loads into the kernel cleanly and "+
|
||||
"matches NOTHING. The traffic of this network is therefore neither proxied nor "+
|
||||
"blocked when the engine is down: it goes straight out the WAN in the clear. It "+
|
||||
"happens when a `config inbound`'s network (or a rule's iface:/zone: source) names "+
|
||||
"something netifd does not know — a typo, a deleted interface, or a device name used "+
|
||||
"where a UCI interface name belongs. Point it at an existing interface, or remove it.",
|
||||
r.Src, r.Dev))
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
// uncoveredNetworkWarnings reports every client network of this router whose
|
||||
// traffic reaches the internet without passing through the plan.
|
||||
//
|
||||
// # What counts as a client network
|
||||
//
|
||||
// Not "has a private address" — that catches tunnels and management links. The
|
||||
// test is the firewall's own: the interface sits in a zone that is FORWARDED TO A
|
||||
// WAN ZONE. That is precisely the statement "the devices on this network use this
|
||||
// router to reach the internet", written by the operator, in the file that
|
||||
// decides it. A zone with no such forwarding (a management VLAN, an egress tunnel
|
||||
// zone) reaches nothing through us and is not this check's business.
|
||||
//
|
||||
// On top of that, four exclusions, each removing a class of false positive:
|
||||
//
|
||||
// - the WAN zones themselves (traffic arriving there is not a client's);
|
||||
// - interfaces already in the divert set (the whole point);
|
||||
// - interfaces this model uses as an EGRESS — we send traffic out of those on
|
||||
// purpose, and their zone may well forward to wan;
|
||||
// - interfaces netifd reports as down, because they carry no clients right now
|
||||
// and the ifup hotplug reconciles the moment they do.
|
||||
func uncoveredNetworkWarnings(m *model.Model, refs []divertRef, infos []IfaceInfo) []string {
|
||||
forwarded := internetForwardedZones()
|
||||
if len(forwarded) == 0 {
|
||||
// Either there is no firewall config we can read, or nothing forwards to a
|
||||
// WAN zone at all. Both mean we cannot say a network reaches the internet
|
||||
// through us, and this check refuses to guess.
|
||||
return nil
|
||||
}
|
||||
covered := make(map[string]bool, len(refs))
|
||||
for _, r := range refs {
|
||||
covered[r.Dev] = true
|
||||
}
|
||||
egress := map[string]bool{}
|
||||
for _, eg := range m.Egresses {
|
||||
if n := strings.TrimSpace(eg.Interface); n != "" {
|
||||
egress[n] = true
|
||||
}
|
||||
}
|
||||
|
||||
var out []string
|
||||
for _, i := range infos {
|
||||
zone := strings.ToLower(strings.TrimSpace(i.Zone))
|
||||
switch {
|
||||
case covered[i.Device], !i.Up, zone == "", model.WANZoneNames[zone],
|
||||
!forwarded[zone], egress[i.Name]:
|
||||
continue
|
||||
}
|
||||
where := i.Device
|
||||
if i.Subnet != "" {
|
||||
where = fmt.Sprintf("%s, %s", i.Device, i.Subnet)
|
||||
}
|
||||
out = append(out, fmt.Sprintf(
|
||||
"interface %q: this is a client network of this router (%s; firewall zone %q, which is "+
|
||||
"forwarded to the internet) and NOTHING in this configuration touches it. Its devices "+
|
||||
"are not proxied, their DNS is not filtered, and the kill switch does not cover them "+
|
||||
"— when the engine is down their traffic keeps flowing to the WAN in the clear, "+
|
||||
"because the fail-closed drop is scoped to the interfaces named in the config and "+
|
||||
"this is not one of them. This is not closed automatically: doing so would cut a "+
|
||||
"network you may have meant to leave off the tunnel. Decide explicitly — add a "+
|
||||
"`config inbound` for it (or a rule with source iface:%s, which may still target "+
|
||||
"`direct`) to bring it under the plane, or move it into a firewall zone that is not "+
|
||||
"forwarded to the WAN.",
|
||||
i.Name, where, i.Zone, i.Name))
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
// internetForwardedZones returns the set of firewall zone names (lower-cased)
|
||||
// that are forwarded to a WAN zone, i.e. the zones whose members use this router
|
||||
// to reach the internet.
|
||||
//
|
||||
// It reads the SAME `uci -q export firewall` the zone map does — one more read of
|
||||
// a plane we already talk to, rather than a second notion of which zone is which.
|
||||
// Fail-open: an unreadable or zone-less firewall yields an empty set, and the
|
||||
// caller then reports nothing.
|
||||
func internetForwardedZones() map[string]bool {
|
||||
out, err := execCommand("uci", "-q", "export", "firewall").Output()
|
||||
if err != nil {
|
||||
return nil
|
||||
}
|
||||
return parseInternetForwardedZones(string(out))
|
||||
}
|
||||
|
||||
// parseInternetForwardedZones is internetForwardedZones' pure half.
|
||||
//
|
||||
// A zone is a WAN zone when its NAME is one of the conventional uplink names
|
||||
// (model.WANZoneNames — the same list rule-source validation uses). A `config
|
||||
// forwarding` whose dest is such a zone marks its src zone as internet-forwarded.
|
||||
func parseInternetForwardedZones(exportText string) map[string]bool {
|
||||
secs := scanUCISections(exportText)
|
||||
wan := map[string]bool{}
|
||||
for _, s := range secs {
|
||||
if s.typ != "zone" {
|
||||
continue
|
||||
}
|
||||
if n := strings.ToLower(strings.TrimSpace(s.options["name"])); n != "" && model.WANZoneNames[n] {
|
||||
wan[n] = true
|
||||
}
|
||||
}
|
||||
if len(wan) == 0 {
|
||||
return nil
|
||||
}
|
||||
fwd := map[string]bool{}
|
||||
for _, s := range secs {
|
||||
if s.typ != "forwarding" {
|
||||
continue
|
||||
}
|
||||
src := strings.ToLower(strings.TrimSpace(s.options["src"]))
|
||||
dest := strings.ToLower(strings.TrimSpace(s.options["dest"]))
|
||||
if src != "" && wan[dest] {
|
||||
fwd[src] = true
|
||||
}
|
||||
}
|
||||
return fwd
|
||||
}
|
||||
@@ -0,0 +1,283 @@
|
||||
package netplane
|
||||
|
||||
// W4 + the adjacent defect: what the plan does NOT cover has to reach the
|
||||
// operator, not just the panel's interface picker.
|
||||
//
|
||||
// Both tests drive the real RenderNftWithWarningsAt, because the warning channel
|
||||
// it returns is the one apply folds into Status — a check that only exercised the
|
||||
// helper would prove nothing about whether the news ever leaves netplane.
|
||||
|
||||
import (
|
||||
"os"
|
||||
"os/exec"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"github.com/sagernet/sing-box/shater/model"
|
||||
)
|
||||
|
||||
// A three-network router of exactly the shape this defect was found on: two LAN
|
||||
// bridges in the `lan` firewall zone, one uplink, and a forwarding that says the
|
||||
// lan zone reaches the internet through this box.
|
||||
const coverageFirewallExport = `package firewall
|
||||
|
||||
config defaults
|
||||
option input 'ACCEPT'
|
||||
|
||||
config zone
|
||||
option name 'lan'
|
||||
list network 'lan'
|
||||
list network 'sex'
|
||||
option input 'ACCEPT'
|
||||
|
||||
config zone
|
||||
option name 'wan'
|
||||
list network 'wan0'
|
||||
option masq '1'
|
||||
|
||||
config forwarding
|
||||
option src 'lan'
|
||||
option dest 'wan'
|
||||
`
|
||||
|
||||
const coverageIfaceDump = `{"interface":[
|
||||
{"interface":"loopback","up":true,"l3_device":"lo"},
|
||||
{"interface":"lan","up":true,"l3_device":"br-lan","ipv4-address":[{"address":"192.168.1.1","mask":24}]},
|
||||
{"interface":"sex","up":true,"l3_device":"br-sex","ipv4-address":[{"address":"10.67.0.1","mask":24}]},
|
||||
{"interface":"wan0","up":true,"l3_device":"eth1","ipv4-address":[{"address":"10.0.2.15","mask":24}]}
|
||||
]}`
|
||||
|
||||
// coverageExec answers the three probes the coverage checks make (the interface
|
||||
// dump, the firewall export, and IfaceDevice's per-interface status) and returns
|
||||
// empty output for everything else, which is what every other netplane test sees.
|
||||
func coverageExec(status map[string]string) func(string, ...string) *exec.Cmd {
|
||||
return func(name string, arg ...string) *exec.Cmd {
|
||||
joined := strings.Join(append([]string{name}, arg...), " ")
|
||||
out := ""
|
||||
switch {
|
||||
case joined == "ubus call network.interface dump":
|
||||
out = coverageIfaceDump
|
||||
case joined == "uci -q export firewall":
|
||||
out = coverageFirewallExport
|
||||
case strings.HasPrefix(joined, "ubus call network.interface.") &&
|
||||
strings.HasSuffix(joined, " status"):
|
||||
iface := strings.TrimSuffix(
|
||||
strings.TrimPrefix(joined, "ubus call network.interface."), " status")
|
||||
out = status[iface]
|
||||
}
|
||||
cs := append([]string{"-test.run=TestSysctlRevertHelperProcess", "--", name}, arg...)
|
||||
cmd := exec.Command(os.Args[0], cs...)
|
||||
cmd.Env = append(os.Environ(), "GO_WANT_HELPER_PROCESS=1", "GO_HELPER_STDOUT="+out)
|
||||
return cmd
|
||||
}
|
||||
}
|
||||
|
||||
// fakeSysClassNet points DeviceExists at a directory holding exactly the devices
|
||||
// the kernel is pretending to have.
|
||||
func fakeSysClassNet(t *testing.T, devs ...string) {
|
||||
t.Helper()
|
||||
dir := t.TempDir()
|
||||
for _, d := range devs {
|
||||
if err := os.MkdirAll(filepath.Join(dir, d), 0o755); err != nil {
|
||||
t.Fatalf("fixture %s: %v", d, err)
|
||||
}
|
||||
}
|
||||
orig := sysClassNet
|
||||
sysClassNet = dir
|
||||
t.Cleanup(func() { sysClassNet = orig })
|
||||
}
|
||||
|
||||
func coverageModel() *model.Model {
|
||||
return &model.Model{
|
||||
Globals: model.Globals{
|
||||
Enabled: true, KillSwitch: "closed", FwmarkBase: 0x2000, TableBase: 0x2000,
|
||||
},
|
||||
Inbounds: []model.Inbound{{
|
||||
Name: "lan", Enabled: true, Type: "tproxy", Network: "lan",
|
||||
TproxyPort: 12345, TCP: true, UDP: true,
|
||||
}},
|
||||
}
|
||||
}
|
||||
|
||||
func warningsAbout(ws []string, iface string) []string {
|
||||
var out []string
|
||||
for _, w := range ws {
|
||||
if strings.HasPrefix(w, "interface \""+iface+"\":") {
|
||||
out = append(out, w)
|
||||
}
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
// TestRenderNftWarnsUncoveredLANNetwork is W4. `sex` is a client network of this
|
||||
// router — it sits in the `lan` firewall zone, which is forwarded to `wan`, so its
|
||||
// devices reach the internet through this box — and NOTHING in the config mentions
|
||||
// it. Its traffic is neither proxied nor covered by the fail-closed drop, and
|
||||
// before this the only place that was visible was the panel's interface picker.
|
||||
//
|
||||
// RED BEFORE: renderNft produced warnings only for device names it had REJECTED,
|
||||
// so this returned an empty warning list.
|
||||
func TestRenderNftWarnsUncoveredLANNetwork(t *testing.T) {
|
||||
orig := execCommand
|
||||
defer func() { execCommand = orig }()
|
||||
execCommand = coverageExec(map[string]string{
|
||||
"lan": `{"l3_device":"br-lan"}`,
|
||||
"sex": `{"l3_device":"br-sex"}`,
|
||||
"wan0": `{"l3_device":"eth1"}`,
|
||||
})
|
||||
fakeSysClassNet(t, "br-lan", "br-sex", "eth1", "lo")
|
||||
|
||||
rs, warnings, err := RenderNftWithWarningsAt(coverageModel(), time.Now())
|
||||
if err != nil {
|
||||
t.Fatalf("RenderNftWithWarningsAt: %v", err)
|
||||
}
|
||||
if !strings.Contains(rs, `iifname "br-lan"`) {
|
||||
t.Fatalf("the plan should still divert br-lan; ruleset:\n%s", rs)
|
||||
}
|
||||
|
||||
got := warningsAbout(warnings, "sex")
|
||||
if len(got) != 1 {
|
||||
t.Fatalf("want exactly one warning about the uncovered network \"sex\", got %d: %v",
|
||||
len(got), warnings)
|
||||
}
|
||||
for _, must := range []string{"client network", "kill switch does not cover", "iface:sex"} {
|
||||
if !strings.Contains(got[0], must) {
|
||||
t.Errorf("the uncovered-network warning must say %q; got:\n%s", must, got[0])
|
||||
}
|
||||
}
|
||||
// The covered network and the uplink must NOT be reported: a check that also
|
||||
// fires on the interfaces it is fine with teaches the operator to ignore it.
|
||||
if w := warningsAbout(warnings, "lan"); len(w) != 0 {
|
||||
t.Errorf("the DIVERTED network must not be reported as uncovered: %v", w)
|
||||
}
|
||||
if w := warningsAbout(warnings, "wan0"); len(w) != 0 {
|
||||
t.Errorf("the WAN uplink is not a client network and must not be reported: %v", w)
|
||||
}
|
||||
}
|
||||
|
||||
// TestRenderNftWarnsUnresolvedDivertDevice is the adjacent defect. The inbound
|
||||
// names an interface that does not exist, IfaceDevice hands the raw UCI name back
|
||||
// (its documented last resort), nftValidDev accepts it as a syntactically fine
|
||||
// device name — and the ruleset loads with `iifname "lann"` matching nothing at
|
||||
// all. The divert does nothing, the fail-closed drop does nothing, and the apply
|
||||
// reports success.
|
||||
//
|
||||
// RED BEFORE: nftValidDev checks characters, not existence, so nothing anywhere
|
||||
// noticed. The warning must name what the operator wrote AND what it resolved to.
|
||||
func TestRenderNftWarnsUnresolvedDivertDevice(t *testing.T) {
|
||||
orig := execCommand
|
||||
defer func() { execCommand = orig }()
|
||||
// No status entry for "lann": ubus knows nothing about it, and neither does
|
||||
// `uci get network.lann.device` — the fallback-to-raw-name path.
|
||||
execCommand = coverageExec(map[string]string{
|
||||
"lan": `{"l3_device":"br-lan"}`,
|
||||
"sex": `{"l3_device":"br-sex"}`,
|
||||
"wan0": `{"l3_device":"eth1"}`,
|
||||
})
|
||||
fakeSysClassNet(t, "br-lan", "br-sex", "eth1", "lo")
|
||||
|
||||
m := coverageModel()
|
||||
m.Inbounds[0].Network = "lann"
|
||||
|
||||
rs, warnings, err := RenderNftWithWarningsAt(m, time.Now())
|
||||
if err != nil {
|
||||
t.Fatalf("RenderNftWithWarningsAt: %v", err)
|
||||
}
|
||||
// The point of the defect: this loads cleanly. The test pins that the plan is
|
||||
// still produced, so the only thing standing between the operator and a silent
|
||||
// leak is the warning.
|
||||
if !strings.Contains(rs, `iifname "lann"`) {
|
||||
t.Fatalf("expected the (useless) rule to still be rendered; ruleset:\n%s", rs)
|
||||
}
|
||||
|
||||
got := warningsAbout(warnings, "lann")
|
||||
if len(got) != 1 {
|
||||
t.Fatalf("want exactly one warning about the unresolved device, got %d: %v",
|
||||
len(got), warnings)
|
||||
}
|
||||
for _, must := range []string{"no network device called", `"lann"`, "matches NOTHING"} {
|
||||
if !strings.Contains(got[0], must) {
|
||||
t.Errorf("the unresolved-device warning must say %q; got:\n%s", must, got[0])
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestCoverageSilentWithoutInventory is the fail-open half of the contract: with
|
||||
// no ubus to enumerate the router (a build host, a stripped image, a boot before
|
||||
// netifd answers) a coverage claim would be a guess, so nothing is said at all.
|
||||
func TestCoverageSilentWithoutInventory(t *testing.T) {
|
||||
orig := execCommand
|
||||
defer func() { execCommand = orig }()
|
||||
var rec []string
|
||||
execCommand = fakeExec(&rec) // every probe succeeds with EMPTY output
|
||||
fakeSysClassNet(t) // and no device exists either
|
||||
|
||||
_, warnings, err := RenderNftWithWarningsAt(coverageModel(), time.Now())
|
||||
if err != nil {
|
||||
t.Fatalf("RenderNftWithWarningsAt: %v", err)
|
||||
}
|
||||
if len(warnings) != 0 {
|
||||
t.Fatalf("with no interface inventory the coverage checks must stay silent, got: %v", warnings)
|
||||
}
|
||||
}
|
||||
|
||||
// TestParseInternetForwardedZones pins what "a client network" means: a zone the
|
||||
// operator forwards to the uplink, read out of the firewall config rather than
|
||||
// guessed from addresses. A vpn zone that is not forwarded to wan is NOT one.
|
||||
func TestParseInternetForwardedZones(t *testing.T) {
|
||||
const export = `package firewall
|
||||
|
||||
config zone
|
||||
option name 'lan'
|
||||
list network 'lan'
|
||||
|
||||
config zone
|
||||
option name 'guest'
|
||||
list network 'guest'
|
||||
|
||||
config zone
|
||||
option name 'mgmt'
|
||||
list network 'mgmt'
|
||||
|
||||
config zone
|
||||
option name 'vpn'
|
||||
list network 'awg0'
|
||||
|
||||
config zone
|
||||
option name 'wan'
|
||||
list network 'wan0'
|
||||
|
||||
config forwarding
|
||||
option src 'lan'
|
||||
option dest 'wan'
|
||||
|
||||
config forwarding
|
||||
option src 'guest'
|
||||
option dest 'wan'
|
||||
|
||||
config forwarding
|
||||
option src 'lan'
|
||||
option dest 'vpn'
|
||||
`
|
||||
got := parseInternetForwardedZones(export)
|
||||
for _, want := range []string{"lan", "guest"} {
|
||||
if !got[want] {
|
||||
t.Errorf("zone %q forwards to wan and must count as a client zone; got %v", want, got)
|
||||
}
|
||||
}
|
||||
for _, notWant := range []string{"mgmt", "vpn", "wan"} {
|
||||
if got[notWant] {
|
||||
t.Errorf("zone %q does not forward to the uplink and must NOT count; got %v", notWant, got)
|
||||
}
|
||||
}
|
||||
|
||||
// No WAN zone at all => nothing can be said about reaching the internet.
|
||||
if m := parseInternetForwardedZones("package firewall\n\nconfig zone\n\toption name 'lan'\n"); len(m) != 0 {
|
||||
t.Errorf("with no wan zone the set must be empty, got %v", m)
|
||||
}
|
||||
if m := parseInternetForwardedZones(""); len(m) != 0 {
|
||||
t.Errorf("empty input must yield an empty set, got %v", m)
|
||||
}
|
||||
}
|
||||
+88
-16
@@ -358,11 +358,33 @@ func nftPlanRules(m *model.Model, _ time.Time) []model.Rule {
|
||||
return sortedRules(rules)
|
||||
}
|
||||
|
||||
// nftDivertDevs returns EVERY LAN-ingress device whose traffic this ruleset
|
||||
// diverts into the engine: the enabled tproxy inbounds' devices PLUS the devices
|
||||
// named by any enabled rule's `iface:`/`zone:` source. rules is the EFFECTIVE
|
||||
// rule list (profile overrides applied — nftPlanRules), never raw m.Rules: the
|
||||
// divert set must match what RenderNft actually emits per-rule.
|
||||
// divertRef is one device this ruleset diverts, together with WHAT THE OPERATOR
|
||||
// WROTE to bring it in — the inbound's `network`, an `iface:` source, a `zone:`
|
||||
// source.
|
||||
//
|
||||
// The provenance is not decoration. IfaceDevice returns the UCI name unchanged
|
||||
// when it cannot resolve it (see its doc comment), so a divert set is a list of
|
||||
// strings in which "br-lan" and "lan" are indistinguishable — one is a device,
|
||||
// the other is a name that will match nothing. Reporting that to an operator as
|
||||
// `device "lan" does not exist` is useless; reporting it as `interface "lan"
|
||||
// resolved to no device` is actionable, and only the reference knows which is
|
||||
// which. coverage.go is the only consumer.
|
||||
type divertRef struct {
|
||||
// Src is the operator-facing name: the inbound's UCI network, the bare name
|
||||
// from an `iface:` source, or "zone:<name>" for a device pulled out of a
|
||||
// firewall zone.
|
||||
Src string
|
||||
// Dev is what IfaceDevice/nftZoneDevices resolved Src to — the string that
|
||||
// actually appears in `iifname "..."`.
|
||||
Dev string
|
||||
}
|
||||
|
||||
// nftDivertRefs returns EVERY LAN-ingress device whose traffic this ruleset
|
||||
// diverts into the engine, with its provenance: the enabled tproxy inbounds'
|
||||
// devices PLUS the devices named by any enabled rule's `iface:`/`zone:` source.
|
||||
// rules is the EFFECTIVE rule list (profile overrides applied — nftPlanRules),
|
||||
// never raw m.Rules: the divert set must match what RenderNft actually emits
|
||||
// per-rule.
|
||||
//
|
||||
// The distinction matters for fail-closed correctness. RenderNft emits per-rule
|
||||
// tproxy lines for `iface:`/`zone:` sources, and those devices need NOT be tproxy
|
||||
@@ -374,15 +396,22 @@ func nftPlanRules(m *model.Model, _ time.Time) []model.Rule {
|
||||
// the tproxied packet to the engine socket at all.
|
||||
//
|
||||
// So: everything we divert, we must also be able to fail closed on and set the
|
||||
// ingress sysctls for. Result is deduped and order-stable (inbounds first).
|
||||
func nftDivertDevs(m *model.Model, rules []model.Rule) []string {
|
||||
devs := nftEnabledInboundDevs(m)
|
||||
if _, ok := nftPrimaryInbound(m); !ok {
|
||||
// No tproxy inbound => RenderNft emits no per-rule divert lines either.
|
||||
return devs
|
||||
// ingress sysctls for. Result is deduped BY DEVICE and order-stable (inbounds
|
||||
// first), so the first reference that produced a device is the one reported.
|
||||
func nftDivertRefs(m *model.Model, rules []model.Rule) []divertRef {
|
||||
var refs []divertRef
|
||||
for _, in := range m.Inbounds {
|
||||
if model.IsTproxyInbound(in) {
|
||||
refs = append(refs, divertRef{
|
||||
Src: orDefault(strings.TrimSpace(in.Network), "lan"),
|
||||
Dev: IfaceDevice(in.Network),
|
||||
})
|
||||
}
|
||||
}
|
||||
if len(devs) == 0 {
|
||||
return devs
|
||||
// No tproxy inbound => RenderNft emits no per-rule divert lines either; and
|
||||
// with no inbound device there is nothing for a rule source to add to.
|
||||
if _, ok := nftPrimaryInbound(m); !ok || len(refs) == 0 {
|
||||
return dedupRefs(refs)
|
||||
}
|
||||
for _, r := range rules {
|
||||
if !r.Enabled {
|
||||
@@ -392,12 +421,44 @@ func nftDivertDevs(m *model.Model, rules []model.Rule) []string {
|
||||
s := strings.TrimSpace(raw)
|
||||
switch {
|
||||
case strings.HasPrefix(s, "iface:"):
|
||||
devs = append(devs, IfaceDevice(strings.TrimPrefix(s, "iface:")))
|
||||
name := strings.TrimPrefix(s, "iface:")
|
||||
refs = append(refs, divertRef{Src: name, Dev: IfaceDevice(name)})
|
||||
case strings.HasPrefix(s, "zone:"):
|
||||
devs = append(devs, nftZoneDevices(strings.TrimPrefix(s, "zone:"))...)
|
||||
zone := strings.TrimPrefix(s, "zone:")
|
||||
for _, dev := range nftZoneDevices(zone) {
|
||||
refs = append(refs, divertRef{Src: "zone:" + zone, Dev: dev})
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
return dedupRefs(refs)
|
||||
}
|
||||
|
||||
// dedupRefs drops references whose device was already seen (and blank devices),
|
||||
// preserving order.
|
||||
func dedupRefs(in []divertRef) []divertRef {
|
||||
seen := map[string]bool{}
|
||||
var out []divertRef
|
||||
for _, r := range in {
|
||||
dev := strings.TrimSpace(r.Dev)
|
||||
if dev == "" || seen[dev] {
|
||||
continue
|
||||
}
|
||||
seen[dev] = true
|
||||
r.Dev = dev
|
||||
out = append(out, r)
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
// nftDivertDevs is nftDivertRefs projected onto the device names, which is what
|
||||
// every renderer and the sysctl/fail-closed scoping consume.
|
||||
func nftDivertDevs(m *model.Model, rules []model.Rule) []string {
|
||||
refs := nftDivertRefs(m, rules)
|
||||
devs := make([]string, 0, len(refs))
|
||||
for _, r := range refs {
|
||||
devs = append(devs, r.Dev)
|
||||
}
|
||||
return nftDedupStr(devs)
|
||||
}
|
||||
|
||||
@@ -598,7 +659,11 @@ func renderNft(m *model.Model, plan *UntunnelablePlan, now time.Time) (string, [
|
||||
// Validate every device the plan wants to divert BEFORE rendering a single
|
||||
// line, so a name we cannot express never reaches the kernel as a rule that
|
||||
// matches nothing. See the refusal contract in the doc comment.
|
||||
divertDevs := nftDivertDevs(m, planRules)
|
||||
divertRefs := nftDivertRefs(m, planRules)
|
||||
divertDevs := make([]string, 0, len(divertRefs))
|
||||
for _, r := range divertRefs {
|
||||
divertDevs = append(divertDevs, r.Dev)
|
||||
}
|
||||
validDevs, rejectedDevs := nftSplitDevs(divertDevs)
|
||||
var warnings []string
|
||||
for _, bad := range rejectedDevs {
|
||||
@@ -606,6 +671,13 @@ func renderNft(m *model.Model, plan *UntunnelablePlan, now time.Time) (string, [
|
||||
"interface name %q is not a usable device name and was REJECTED: it is not covered by the "+
|
||||
"fail-closed guard and its traffic is not diverted", bad))
|
||||
}
|
||||
// What this plan does NOT cover: a device name that resolves to nothing the
|
||||
// kernel has, and a client network of this router that the plan never mentions.
|
||||
// Both are silent leaks of exactly the kind the rejected-name check above
|
||||
// exists for, and both were previously visible only in the panel's interface
|
||||
// picker (or nowhere at all). Collected BEFORE the refusal below, so a refused
|
||||
// plan still tells the operator everything that is wrong with it.
|
||||
warnings = append(warnings, coverageWarnings(m, divertRefs)...)
|
||||
if len(rejectedDevs) > 0 && genGlobalClosed(m.Globals) {
|
||||
return "", warnings, fmt.Errorf(
|
||||
"refusing to apply: %d interface name(s) cannot be expressed in the ruleset (%s), so the "+
|
||||
|
||||
+65
-8
@@ -379,11 +379,26 @@ var (
|
||||
// the best-effort fallback when file logging is off but syslog is on. logread
|
||||
// has no wall-clock retention worth promising, hence "ranges approximate" in
|
||||
// the note the handler prepends.
|
||||
logSyslogScrape = func(w io.Writer) error {
|
||||
cmd := exec.Command("logread", "-e", "shater")
|
||||
//
|
||||
// The ctx is the REQUEST's, narrowed by logScrapeTimeout, and it is the only
|
||||
// thing that ever ends this child process. Without it, a browser tab that was
|
||||
// closed mid-download left `logread` running, its pipe open and the handler
|
||||
// goroutine blocked writing into a socket nobody reads — one leaked process and
|
||||
// two leaked descriptors per abandoned download, on a daemon that runs for
|
||||
// months. CommandContext kills the child on cancel, which closes the pipe, which
|
||||
// releases the copy goroutine Run is waiting on.
|
||||
logSyslogScrape = func(ctx context.Context, w io.Writer) error {
|
||||
cmd := exec.CommandContext(ctx, "logread", "-e", "shater")
|
||||
cmd.Stdout = w
|
||||
return cmd.Run()
|
||||
}
|
||||
// logScrapeTimeout hard-bounds the scrape even for a client that is still there.
|
||||
//
|
||||
// Why 30s: `logread` reads a ring buffer that is a few hundred KiB at most and
|
||||
// exits; on a healthy box it completes in milliseconds. 30s is therefore not a
|
||||
// budget anybody spends, it is the answer to "logread itself has wedged" — which
|
||||
// is exactly the situation an operator downloads the log during.
|
||||
logScrapeTimeout = 30 * time.Second
|
||||
// logSyncTimeout bounds how long a download waits for the log sink's barrier
|
||||
// (Server.SetLogSync). See syncLogSink for why the wait is bounded at all.
|
||||
//
|
||||
@@ -494,21 +509,30 @@ func (s *Server) handleLog(w http.ResponseWriter, r *http.Request) {
|
||||
w.Header().Set("Content-Disposition",
|
||||
`attachment; filename="shaterd-`+rng+`-`+now.Format("20060102")+`.log"`)
|
||||
|
||||
// The download is the one response that is legitimately big (the sink's file
|
||||
// budget tops out at model.LogMaxKBMax = 8 MiB) and so the one that cannot live
|
||||
// under a fixed WriteTimeout: a slow-but-real client would be cut off mid-file.
|
||||
// out renews the write deadline on every chunk instead, so progress buys time and
|
||||
// only a client that has genuinely STOPPED reading is dropped. See streamWriter.
|
||||
out := newStreamWriter(w)
|
||||
|
||||
switch {
|
||||
case !g.LogToFile && !g.LogToSyslog:
|
||||
_, _ = io.WriteString(w, "# logging disabled\n")
|
||||
_, _ = io.WriteString(out, "# logging disabled\n")
|
||||
case !g.LogToFile:
|
||||
_, _ = io.WriteString(w, "# note: file logging disabled; showing syslog ring only, ranges approximate\n")
|
||||
if err := logSyslogScrape(w); err != nil {
|
||||
_, _ = fmt.Fprintf(w, "# logread unavailable: %v\n", err)
|
||||
_, _ = io.WriteString(out, "# note: file logging disabled; showing syslog ring only, ranges approximate\n")
|
||||
ctx, cancel := context.WithTimeout(r.Context(), logScrapeTimeout)
|
||||
defer cancel()
|
||||
if err := logSyslogScrape(ctx, out); err != nil {
|
||||
_, _ = fmt.Fprintf(out, "# logread unavailable: %v\n", err)
|
||||
}
|
||||
default:
|
||||
segs := logsink.Segments(logSinkPath(g.LogPersist))
|
||||
if len(segs) == 0 {
|
||||
_, _ = io.WriteString(w, "# no log file yet\n")
|
||||
_, _ = io.WriteString(out, "# no log file yet\n")
|
||||
return
|
||||
}
|
||||
if err := logsink.CopyRange(w, segs, cutoff); err != nil {
|
||||
if err := logsink.CopyRange(out, segs, cutoff); err != nil {
|
||||
// Headers are long gone; nothing to do but note it server-side
|
||||
// (usually the client hung up mid-download).
|
||||
s.log.Debug("log download aborted: ", err)
|
||||
@@ -516,6 +540,39 @@ func (s *Server) handleLog(w http.ResponseWriter, r *http.Request) {
|
||||
}
|
||||
}
|
||||
|
||||
// streamChunkTimeout is how long ONE write of a streamed response may block before
|
||||
// the connection is torn down.
|
||||
//
|
||||
// Why 15s: a write blocks only while the socket send buffer is full, i.e. while the
|
||||
// peer is not reading. A client that is reading at any rate at all — even a few KiB/s
|
||||
// on a bad wireless link — drains the buffer far inside 15s and gets the deadline
|
||||
// pushed forward again, so a slow download is never truncated. A client that has
|
||||
// stopped (a suspended phone, a closed tab whose FIN was lost) is released in 15s
|
||||
// instead of holding the goroutine, socket and fd until the process restarts.
|
||||
const streamChunkTimeout = 15 * time.Second
|
||||
|
||||
// streamWriter renews the connection's write deadline before each chunk, turning a
|
||||
// fixed server-wide WriteTimeout into a per-chunk idle timeout for one response. It
|
||||
// is the only correct way to bound a large streamed body: the total is unknown, the
|
||||
// per-write stall is not.
|
||||
//
|
||||
// SetWriteDeadline returning http.ErrNotSupported (httptest.NewRecorder, a wrapped
|
||||
// ResponseWriter) is ignored on purpose — the writer then behaves exactly like the
|
||||
// bare one and the server-wide WriteTimeout still applies.
|
||||
type streamWriter struct {
|
||||
w io.Writer
|
||||
rc *http.ResponseController
|
||||
}
|
||||
|
||||
func newStreamWriter(w http.ResponseWriter) *streamWriter {
|
||||
return &streamWriter{w: w, rc: http.NewResponseController(w)}
|
||||
}
|
||||
|
||||
func (s *streamWriter) Write(p []byte) (int, error) {
|
||||
_ = s.rc.SetWriteDeadline(time.Now().Add(streamChunkTimeout))
|
||||
return s.w.Write(p)
|
||||
}
|
||||
|
||||
// handleStatsLog → GET /api/stats/log?limit=&before=&after=: the live DNS query log,
|
||||
// ALWAYS newest first, each row carrying a monotonic `seq` cursor. With no cursor it
|
||||
// returns the newest `limit` rows; before=<seq> returns the next OLDER page (seq<before);
|
||||
|
||||
@@ -109,9 +109,30 @@ type sessionRequest struct {
|
||||
Token string `json:"token"`
|
||||
}
|
||||
|
||||
// sessionBodyTimeout bounds how long the ONE unauthenticated route will wait for
|
||||
// its (tiny) body.
|
||||
//
|
||||
// The server-wide DefaultReadTimeout has to accommodate a 4 MiB config PUT, so it is
|
||||
// 60s. This route is different in kind: it is reachable with no credentials at all,
|
||||
// and its entire legitimate body is a 64-hex-char token inside a JSON object — under
|
||||
// 100 bytes, already delivered in the same TCP segment as the headers in every real
|
||||
// client. Anything that needs longer than 5 seconds to produce it is not a panel.
|
||||
//
|
||||
// Without this, "Content-Length: 4096" plus one byte an hour bought an anonymous peer
|
||||
// a goroutine, a socket and a file descriptor for as long as the read timeout allowed
|
||||
// — repeatable until the daemon runs out of descriptors.
|
||||
//
|
||||
// A var, not a const, purely so the timeout tests can dial it down; production never
|
||||
// assigns it.
|
||||
var sessionBodyTimeout = 5 * time.Second
|
||||
|
||||
// handleSession validates a minted token and, on success, sets the session cookie.
|
||||
// POST only. This is the panel's sole unauthenticated API route.
|
||||
func (s *Server) handleSession(w http.ResponseWriter, r *http.Request) {
|
||||
// Tighten the read deadline for this route only. ErrNotSupported is expected
|
||||
// under httptest.NewRecorder (no real connection) and is not a failure: the
|
||||
// server-wide ReadTimeout still applies.
|
||||
_ = http.NewResponseController(w).SetReadDeadline(time.Now().Add(sessionBodyTimeout))
|
||||
if r.Method != http.MethodPost {
|
||||
writeError(w, http.StatusMethodNotAllowed, "method not allowed")
|
||||
return
|
||||
|
||||
@@ -5,6 +5,7 @@ package panel
|
||||
// are all package-var seams; each test swaps them and restores via t.Cleanup.
|
||||
|
||||
import (
|
||||
"context"
|
||||
"io"
|
||||
"net/http"
|
||||
"net/http/httptest"
|
||||
@@ -131,7 +132,7 @@ func TestLogFileOffFallsBackToSyslogScrape(t *testing.T) {
|
||||
g.LogToFile = false // syslog stays on
|
||||
withLogSeams(t, g, "")
|
||||
origScrape := logSyslogScrape
|
||||
logSyslogScrape = func(w io.Writer) error {
|
||||
logSyslogScrape = func(_ context.Context, w io.Writer) error {
|
||||
_, err := io.WriteString(w, "ring line from logread\n")
|
||||
return err
|
||||
}
|
||||
@@ -188,7 +189,7 @@ func TestLogFileOffIgnoresLeftoverSegments(t *testing.T) {
|
||||
g.LogToFile = false // syslog stays on
|
||||
withLogSeams(t, g, "2026-07-23T08:00:00Z INFO leftover line\n")
|
||||
origScrape := logSyslogScrape
|
||||
logSyslogScrape = func(w io.Writer) error {
|
||||
logSyslogScrape = func(_ context.Context, w io.Writer) error {
|
||||
_, err := io.WriteString(w, "ring line from logread\n")
|
||||
return err
|
||||
}
|
||||
|
||||
+70
-1
@@ -39,6 +39,51 @@ const (
|
||||
DefaultTokenTTL = 60 * time.Second
|
||||
)
|
||||
|
||||
// Connection deadlines. Every one of these bounds a resource a REMOTE peer would
|
||||
// otherwise hold indefinitely: a goroutine, a socket and a file descriptor each.
|
||||
// On this box that matters more than on a server — a phone whose panel tab went to
|
||||
// sleep mid-request is the normal case, not an attack, and the daemon runs for
|
||||
// months without a restart, so "leaks one fd per stalled tab" is a countdown to the
|
||||
// process-wide fd limit.
|
||||
//
|
||||
// Only ReadHeaderTimeout existed before. It bounds the request LINE and HEADERS and
|
||||
// nothing else, so a peer that finished its headers, announced a Content-Length and
|
||||
// then stopped sending was parked forever — reachable without any authentication at
|
||||
// all through POST /api/session.
|
||||
const (
|
||||
// DefaultReadHeaderTimeout bounds the request line + headers. Unchanged value;
|
||||
// named so the whole set is visible in one place.
|
||||
DefaultReadHeaderTimeout = 10 * time.Second
|
||||
|
||||
// DefaultReadTimeout bounds reading an ENTIRE request, headers plus body.
|
||||
//
|
||||
// Why 60s: the largest legitimate body is a PUT /api/config at the 4 MiB cap
|
||||
// (maxConfigBytes) — a real 380-node config is ~10x smaller, but the cap is what
|
||||
// has to fit. 4 MiB in 60s is a 70 KB/s floor, which even a phone on a bad
|
||||
// corner of the LAN clears by an order of magnitude. It is also short enough
|
||||
// that a stalled connection is a bounded 60s cost instead of a permanent one.
|
||||
// The unauthenticated route gets a much tighter budget of its own — see
|
||||
// sessionBodyTimeout in auth.go.
|
||||
DefaultReadTimeout = 60 * time.Second
|
||||
|
||||
// DefaultWriteTimeout bounds writing a response.
|
||||
//
|
||||
// Why 60s: every JSON endpoint answers in milliseconds; the one legitimately
|
||||
// large response is the GET /api/log download, and that one renews its own
|
||||
// deadline per chunk while the client keeps reading (see streamWriter in
|
||||
// api.go), so this value is not the download's budget — it is the budget for a
|
||||
// client that has stopped reading altogether.
|
||||
DefaultWriteTimeout = 60 * time.Second
|
||||
|
||||
// DefaultIdleTimeout bounds an idle keep-alive connection between requests.
|
||||
//
|
||||
// Why 90s: the panel's own polling is the fastest legitimate reuse cadence and it
|
||||
// is seconds, not minutes, so 90s never costs a live tab its connection, while a
|
||||
// tab that was closed or suspended gives its fd back within a minute and a half
|
||||
// instead of never.
|
||||
DefaultIdleTimeout = 90 * time.Second
|
||||
)
|
||||
|
||||
// Options configures a Server. The zero value is valid: every field falls back to
|
||||
// the Default* above (or a safe built-in) in NewServer.
|
||||
type Options struct {
|
||||
@@ -56,6 +101,15 @@ type Options struct {
|
||||
CookieSecure bool
|
||||
// Logger is the daemon logger; nil => log.StdLogger().
|
||||
Logger log.ContextLogger
|
||||
|
||||
// Connection deadlines. Zero => the Default* above; they exist as knobs mainly
|
||||
// so tests can dial them down to milliseconds, but an embedder fronting the
|
||||
// panel with a slow link can raise them. See the const block for the reasoning
|
||||
// behind each default.
|
||||
ReadHeaderTimeout time.Duration
|
||||
ReadTimeout time.Duration
|
||||
WriteTimeout time.Duration
|
||||
IdleTimeout time.Duration
|
||||
}
|
||||
|
||||
func (o Options) withDefaults() Options {
|
||||
@@ -74,6 +128,18 @@ func (o Options) withDefaults() Options {
|
||||
if o.Logger == nil {
|
||||
o.Logger = log.StdLogger()
|
||||
}
|
||||
if o.ReadHeaderTimeout <= 0 {
|
||||
o.ReadHeaderTimeout = DefaultReadHeaderTimeout
|
||||
}
|
||||
if o.ReadTimeout <= 0 {
|
||||
o.ReadTimeout = DefaultReadTimeout
|
||||
}
|
||||
if o.WriteTimeout <= 0 {
|
||||
o.WriteTimeout = DefaultWriteTimeout
|
||||
}
|
||||
if o.IdleTimeout <= 0 {
|
||||
o.IdleTimeout = DefaultIdleTimeout
|
||||
}
|
||||
return o
|
||||
}
|
||||
|
||||
@@ -119,7 +185,10 @@ func NewServer(a *apply.Applier, opts Options) *Server {
|
||||
s.handler = s.buildRouter()
|
||||
s.srv = &http.Server{
|
||||
Handler: s.handler,
|
||||
ReadHeaderTimeout: 10 * time.Second,
|
||||
ReadHeaderTimeout: opts.ReadHeaderTimeout,
|
||||
ReadTimeout: opts.ReadTimeout,
|
||||
WriteTimeout: opts.WriteTimeout,
|
||||
IdleTimeout: opts.IdleTimeout,
|
||||
}
|
||||
return s
|
||||
}
|
||||
|
||||
@@ -0,0 +1,264 @@
|
||||
// Timeout coverage for the panel's own http.Server (server.go) and for the two
|
||||
// unbounded streaming paths in handleLog (api.go).
|
||||
//
|
||||
// These tests drive the REAL listener (Server.Start), not httptest over
|
||||
// Server.Handler(): the whole point is the *http.Server field set, which
|
||||
// httptest replaces with its own. Every deadline is dialled down through
|
||||
// Options so the suite stays sub-second.
|
||||
package panel
|
||||
|
||||
import (
|
||||
"bufio"
|
||||
"context"
|
||||
"io"
|
||||
"net"
|
||||
"net/http"
|
||||
"net/http/httptest"
|
||||
"strconv"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"github.com/sagernet/sing-box/shater/apply"
|
||||
"github.com/sagernet/sing-box/shater/engine"
|
||||
"github.com/sagernet/sing-box/shater/model"
|
||||
)
|
||||
|
||||
// startLive brings up a Server on a real loopback port with the given Options
|
||||
// overrides applied on top of the test defaults, and returns its address.
|
||||
func startLive(t *testing.T, opts Options) (*Server, string) {
|
||||
t.Helper()
|
||||
t.Setenv("SHATER_SUBS_DIR", t.TempDir())
|
||||
eng := engine.New()
|
||||
a := apply.New(eng, nil)
|
||||
opts.Addr = "127.0.0.1:0"
|
||||
if opts.SessionTTL == 0 {
|
||||
opts.SessionTTL = time.Hour
|
||||
}
|
||||
if opts.TokenTTL == 0 {
|
||||
opts.TokenTTL = time.Minute
|
||||
}
|
||||
s := NewServer(a, opts)
|
||||
|
||||
ln, err := net.Listen("tcp", opts.Addr)
|
||||
if err != nil {
|
||||
t.Fatalf("listen: %v", err)
|
||||
}
|
||||
s.lnMu.Lock()
|
||||
s.ln = ln
|
||||
s.lnMu.Unlock()
|
||||
go func() { _ = s.srv.Serve(ln) }()
|
||||
t.Cleanup(func() { _ = s.Close() })
|
||||
return s, ln.Addr().String()
|
||||
}
|
||||
|
||||
// TestSlowBodyDoesNotHoldConnectionForever is the unauthenticated slow-loris:
|
||||
// POST /api/session announcing a full 4 KiB body and then dribbling. Before the
|
||||
// read deadline existed, the goroutine + socket + fd stayed parked until the
|
||||
// client felt like finishing — i.e. forever.
|
||||
func TestSlowBodyDoesNotHoldConnectionForever(t *testing.T) {
|
||||
orig := sessionBodyTimeout
|
||||
sessionBodyTimeout = 200 * time.Millisecond
|
||||
t.Cleanup(func() { sessionBodyTimeout = orig })
|
||||
|
||||
_, addr := startLive(t, Options{
|
||||
ReadTimeout: 30 * time.Second, // deliberately generous: the route's OWN budget must bite
|
||||
WriteTimeout: 2 * time.Second,
|
||||
})
|
||||
|
||||
conn, err := net.Dial("tcp", addr)
|
||||
if err != nil {
|
||||
t.Fatalf("dial: %v", err)
|
||||
}
|
||||
defer conn.Close()
|
||||
|
||||
// Full headers (so ReadHeaderTimeout is satisfied), then one byte of a
|
||||
// promised 4096-byte body and nothing more.
|
||||
req := "POST /api/session HTTP/1.1\r\n" +
|
||||
"Host: panel\r\n" +
|
||||
"Content-Type: application/json\r\n" +
|
||||
"Content-Length: 4096\r\n\r\n{"
|
||||
if _, err := io.WriteString(conn, req); err != nil {
|
||||
t.Fatalf("write request: %v", err)
|
||||
}
|
||||
|
||||
// The server must give up on us. Either it closes (EOF/reset) or it answers
|
||||
// 400 and closes; both end the read. What must NOT happen is a hang.
|
||||
_ = conn.SetReadDeadline(time.Now().Add(5 * time.Second))
|
||||
done := make(chan error, 1)
|
||||
go func() {
|
||||
_, err := io.Copy(io.Discard, conn)
|
||||
done <- err
|
||||
}()
|
||||
select {
|
||||
case err := <-done:
|
||||
if ne, ok := err.(net.Error); ok && ne.Timeout() {
|
||||
t.Fatalf("connection still open 5s into a stalled 4 KiB body — the read deadline is not set")
|
||||
}
|
||||
case <-time.After(5 * time.Second):
|
||||
t.Fatalf("read never returned — the connection is held open by a stalled body")
|
||||
}
|
||||
}
|
||||
|
||||
// TestSlowBodyOnAuthenticatedRouteIsBounded covers the server-wide ReadTimeout —
|
||||
// the backstop for every route that is NOT /api/session (which has its own,
|
||||
// tighter budget). PUT /api/config accepts up to 4 MiB, so before ReadTimeout a
|
||||
// stalled upload parked a goroutine and an fd indefinitely.
|
||||
func TestSlowBodyOnAuthenticatedRouteIsBounded(t *testing.T) {
|
||||
s, addr := startLive(t, Options{
|
||||
ReadTimeout: 300 * time.Millisecond,
|
||||
WriteTimeout: 2 * time.Second,
|
||||
})
|
||||
tok := s.MintToken()
|
||||
|
||||
conn, err := net.Dial("tcp", addr)
|
||||
if err != nil {
|
||||
t.Fatalf("dial: %v", err)
|
||||
}
|
||||
defer conn.Close()
|
||||
|
||||
// Log in on this very connection, then start a PUT and stall it.
|
||||
body := `{"token":"` + tok + `"}`
|
||||
if _, err := io.WriteString(conn, "POST /api/session HTTP/1.1\r\nHost: panel\r\n"+
|
||||
"Content-Type: application/json\r\n"+
|
||||
"Content-Length: "+strconv.Itoa(len(body))+"\r\n\r\n"+body); err != nil {
|
||||
t.Fatalf("write session: %v", err)
|
||||
}
|
||||
br := bufio.NewReader(conn)
|
||||
resp, err := http.ReadResponse(br, nil)
|
||||
if err != nil {
|
||||
t.Fatalf("session response: %v", err)
|
||||
}
|
||||
_, _ = io.Copy(io.Discard, resp.Body)
|
||||
resp.Body.Close()
|
||||
var cookie string
|
||||
for _, c := range resp.Cookies() {
|
||||
if c.Name == DefaultCookieName {
|
||||
cookie = c.Name + "=" + c.Value
|
||||
}
|
||||
}
|
||||
if cookie == "" {
|
||||
t.Fatalf("no session cookie")
|
||||
}
|
||||
|
||||
if _, err := io.WriteString(conn, "PUT /api/config HTTP/1.1\r\nHost: panel\r\n"+
|
||||
"Cookie: "+cookie+"\r\n"+
|
||||
"Content-Type: application/json\r\n"+
|
||||
"Content-Length: 65536\r\n\r\n{"); err != nil {
|
||||
t.Fatalf("write PUT: %v", err)
|
||||
}
|
||||
|
||||
_ = conn.SetReadDeadline(time.Now().Add(5 * time.Second))
|
||||
if _, err := io.Copy(io.Discard, br); err != nil {
|
||||
if ne, ok := err.(net.Error); ok && ne.Timeout() {
|
||||
t.Fatalf("authenticated stalled body held the connection past ReadTimeout")
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestSlowHeaderStillBounded pins the pre-existing ReadHeaderTimeout so a later
|
||||
// refactor of the field set cannot drop it while adding the others.
|
||||
func TestSlowHeaderStillBounded(t *testing.T) {
|
||||
_, addr := startLive(t, Options{
|
||||
ReadHeaderTimeout: 200 * time.Millisecond,
|
||||
ReadTimeout: 5 * time.Second,
|
||||
WriteTimeout: 2 * time.Second,
|
||||
})
|
||||
conn, err := net.Dial("tcp", addr)
|
||||
if err != nil {
|
||||
t.Fatalf("dial: %v", err)
|
||||
}
|
||||
defer conn.Close()
|
||||
if _, err := io.WriteString(conn, "GET /api/status HTTP/1.1\r\nHost: panel\r\n"); err != nil {
|
||||
t.Fatalf("write: %v", err)
|
||||
}
|
||||
_ = conn.SetReadDeadline(time.Now().Add(3 * time.Second))
|
||||
if _, err := io.Copy(io.Discard, conn); err != nil {
|
||||
if ne, ok := err.(net.Error); ok && ne.Timeout() {
|
||||
t.Fatalf("header read was never bounded")
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestIdleKeepAliveIsBounded: a client that completes a request and then holds
|
||||
// the keep-alive connection open without ever sending another one must be
|
||||
// dropped. Without IdleTimeout the fd is that client's for as long as it wants.
|
||||
func TestIdleKeepAliveIsBounded(t *testing.T) {
|
||||
_, addr := startLive(t, Options{
|
||||
ReadTimeout: 2 * time.Second,
|
||||
WriteTimeout: 2 * time.Second,
|
||||
IdleTimeout: 300 * time.Millisecond,
|
||||
})
|
||||
conn, err := net.Dial("tcp", addr)
|
||||
if err != nil {
|
||||
t.Fatalf("dial: %v", err)
|
||||
}
|
||||
defer conn.Close()
|
||||
if _, err := io.WriteString(conn, "GET /api/status HTTP/1.1\r\nHost: panel\r\n\r\n"); err != nil {
|
||||
t.Fatalf("write: %v", err)
|
||||
}
|
||||
br := bufio.NewReader(conn)
|
||||
resp, err := http.ReadResponse(br, nil)
|
||||
if err != nil {
|
||||
t.Fatalf("read response: %v", err)
|
||||
}
|
||||
_, _ = io.Copy(io.Discard, resp.Body)
|
||||
resp.Body.Close()
|
||||
|
||||
// Now go quiet. The server must close the idle connection.
|
||||
_ = conn.SetReadDeadline(time.Now().Add(5 * time.Second))
|
||||
if _, err := br.ReadByte(); err != nil {
|
||||
if ne, ok := err.(net.Error); ok && ne.Timeout() {
|
||||
t.Fatalf("idle keep-alive connection still open after 5s — IdleTimeout is not set")
|
||||
}
|
||||
return // EOF: the server closed it, which is the point
|
||||
}
|
||||
t.Fatalf("unexpected bytes on an idle connection")
|
||||
}
|
||||
|
||||
// TestLogSyslogScrapeIsCancelledWithTheRequest: with file logging off, GET
|
||||
// /api/log shells `logread`. The exec had no context, so a client that walked
|
||||
// away left the child process, its pipe and the handler goroutine running.
|
||||
func TestLogSyslogScrapeIsCancelledWithTheRequest(t *testing.T) {
|
||||
g := model.DefaultGlobals()
|
||||
g.LogToFile = false // syslog stays on -> the scrape path
|
||||
withLogSeams(t, g, "")
|
||||
|
||||
entered := make(chan struct{})
|
||||
released := make(chan struct{})
|
||||
origScrape := logSyslogScrape
|
||||
logSyslogScrape = func(ctx context.Context, w io.Writer) error {
|
||||
close(entered)
|
||||
<-ctx.Done() // in production this is what kills the `logread` child
|
||||
close(released)
|
||||
return ctx.Err()
|
||||
}
|
||||
t.Cleanup(func() { logSyslogScrape = origScrape })
|
||||
|
||||
s := newTestServer(t)
|
||||
srv := httptest.NewServer(s.Handler())
|
||||
defer srv.Close()
|
||||
cookie := login(t, srv, s)
|
||||
|
||||
ctx, cancel := context.WithCancel(context.Background())
|
||||
req, _ := http.NewRequestWithContext(ctx, http.MethodGet, srv.URL+"/api/log?range=all", nil)
|
||||
req.AddCookie(cookie)
|
||||
go func() {
|
||||
r, err := http.DefaultClient.Do(req)
|
||||
if err == nil {
|
||||
_, _ = io.Copy(io.Discard, r.Body)
|
||||
r.Body.Close()
|
||||
}
|
||||
}()
|
||||
|
||||
select {
|
||||
case <-entered:
|
||||
case <-time.After(5 * time.Second):
|
||||
t.Fatalf("scrape never started")
|
||||
}
|
||||
cancel()
|
||||
select {
|
||||
case <-released:
|
||||
case <-time.After(5 * time.Second):
|
||||
t.Fatalf("the syslog scrape outlived the request it was serving — no context is threaded into it")
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,314 @@
|
||||
package stats
|
||||
|
||||
// Coverage for the retention bounds that were missing (the per-resolver and
|
||||
// per-outbound DNS maps), for the visibility of every eviction, and for the size of
|
||||
// Snapshot's critical section.
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"math/rand"
|
||||
"net/netip"
|
||||
"sort"
|
||||
"strings"
|
||||
"sync"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"github.com/sagernet/sing-box/common/dnstrack"
|
||||
"github.com/sagernet/sing-box/log"
|
||||
)
|
||||
|
||||
// warnLog records Warn lines so a test can assert an eviction was announced.
|
||||
type warnLog struct {
|
||||
log.ContextLogger
|
||||
mu sync.Mutex
|
||||
warns []string
|
||||
}
|
||||
|
||||
func newWarnLog() *warnLog { return &warnLog{ContextLogger: log.StdLogger()} }
|
||||
|
||||
func (l *warnLog) Warn(args ...any) {
|
||||
l.mu.Lock()
|
||||
l.warns = append(l.warns, fmt.Sprint(args...))
|
||||
l.mu.Unlock()
|
||||
}
|
||||
|
||||
func (l *warnLog) lines() []string {
|
||||
l.mu.Lock()
|
||||
defer l.mu.Unlock()
|
||||
return append([]string(nil), l.warns...)
|
||||
}
|
||||
|
||||
// dnsEvent builds a LAN-sourced query event carrying a resolver tag and an outbound.
|
||||
func dnsEvent(domain, server, outbound string) dnstrack.QueryEvent {
|
||||
return dnstrack.QueryEvent{
|
||||
Domain: domain,
|
||||
QueryType: 1,
|
||||
Client: netip.MustParseAddr("192.168.1.10"),
|
||||
DNSServer: server,
|
||||
Outbound: []string{outbound},
|
||||
}
|
||||
}
|
||||
|
||||
// TestOutboundMapHoldsAGenerationThenBoundsItself is the arithmetic behind
|
||||
// maxOutboundKeys made executable. A live generation on this box is ~1000 outbound
|
||||
// tags (≈380 nodes plus their per-group and per-chain copies); the map must carry
|
||||
// that plus a rename-day overlap untouched, and must NOT carry an unbounded number
|
||||
// of dead generations.
|
||||
func TestOutboundMapHoldsAGenerationThenBoundsItself(t *testing.T) {
|
||||
lg := newWarnLog()
|
||||
a := New(nil, lg)
|
||||
|
||||
// Two whole generations: nothing may be evicted.
|
||||
for gen := 0; gen < 2; gen++ {
|
||||
for i := 0; i < 1000; i++ {
|
||||
a.handleEvent(dnsEvent("a.example", "dns-local",
|
||||
fmt.Sprintf("gen%d-node-%d", gen, i)))
|
||||
}
|
||||
}
|
||||
if got := a.Snapshot().Dropped.Outbounds; got != 0 {
|
||||
t.Fatalf("two live generations (2000 tags) already evicted %d entries — the cap is too low", got)
|
||||
}
|
||||
if len(lg.lines()) != 0 {
|
||||
t.Fatalf("unexpected eviction notices while under the cap: %v", lg.lines())
|
||||
}
|
||||
|
||||
// Now blow past it.
|
||||
for i := 0; i < maxOutboundKeys; i++ {
|
||||
a.handleEvent(dnsEvent("a.example", "dns-local", fmt.Sprintf("gen9-node-%d", i)))
|
||||
}
|
||||
|
||||
a.mu.Lock()
|
||||
size := len(a.outbounds)
|
||||
a.mu.Unlock()
|
||||
if size > maxOutboundKeys {
|
||||
t.Fatalf("outbound map holds %d entries, above the %d cap — still unbounded", size, maxOutboundKeys)
|
||||
}
|
||||
snap := a.Snapshot()
|
||||
if snap.Dropped.Outbounds == 0 {
|
||||
t.Fatalf("map was capped but Snapshot reports nothing dropped — the eviction is invisible")
|
||||
}
|
||||
lines := lg.lines()
|
||||
if len(lines) == 0 {
|
||||
t.Fatalf("entries were evicted with no log line")
|
||||
}
|
||||
if !strings.Contains(lines[0], "outbounds") || !strings.Contains(lines[0], "sample, not a total") {
|
||||
t.Fatalf("eviction notice does not name the map or its consequence: %q", lines[0])
|
||||
}
|
||||
}
|
||||
|
||||
// TestServerMapIsBounded: the resolver-tag map is small in practice, but "small in
|
||||
// practice" was also true of every other map that turned out to grow forever.
|
||||
func TestServerMapIsBounded(t *testing.T) {
|
||||
a := New(nil, nil)
|
||||
for i := 0; i < maxServerKeys*3; i++ {
|
||||
a.handleEvent(dnsEvent("a.example", fmt.Sprintf("resolver-%d", i), ""))
|
||||
}
|
||||
a.mu.Lock()
|
||||
size := len(a.servers)
|
||||
a.mu.Unlock()
|
||||
if size > maxServerKeys {
|
||||
t.Fatalf("server map holds %d entries, above the %d cap", size, maxServerKeys)
|
||||
}
|
||||
if a.Snapshot().Dropped.Servers == 0 {
|
||||
t.Fatalf("server map was capped but nothing was reported dropped")
|
||||
}
|
||||
}
|
||||
|
||||
// TestPruneKeepsTheBusiestEntries: eviction is by count, so the tags that actually
|
||||
// carry traffic survive and the long tail goes. A cap that dropped the wrong ones
|
||||
// would quietly falsify the panel's "top outbounds".
|
||||
func TestPruneKeepsTheBusiestEntries(t *testing.T) {
|
||||
a := New(nil, nil)
|
||||
// One heavily-used tag...
|
||||
for i := 0; i < 50; i++ {
|
||||
a.handleEvent(dnsEvent("a.example", "dns-local", "busy-node"))
|
||||
}
|
||||
// ...and a long tail of one-hit wonders that overflows the cap.
|
||||
for i := 0; i < maxOutboundKeys*2; i++ {
|
||||
a.handleEvent(dnsEvent("a.example", "dns-local", fmt.Sprintf("tail-%d", i)))
|
||||
}
|
||||
a.mu.Lock()
|
||||
_, kept := a.outbounds["busy-node"]
|
||||
a.mu.Unlock()
|
||||
if !kept {
|
||||
t.Fatalf("the busiest outbound was evicted while single-hit entries survived")
|
||||
}
|
||||
}
|
||||
|
||||
// TestRetentionDisabledStillMeansUnlimited: the new caps must obey the existing
|
||||
// master switch, or a deployment that deliberately asked for unlimited retention
|
||||
// would silently get a bounded map instead.
|
||||
func TestRetentionDisabledStillMeansUnlimited(t *testing.T) {
|
||||
a := New(nil, nil, Config{RingSize: -1, TimelineMinutes: -1, MaxDomains: -1, RetentionDisabled: true})
|
||||
for i := 0; i < maxOutboundKeys+50; i++ {
|
||||
a.handleEvent(dnsEvent("a.example", fmt.Sprintf("resolver-%d", i), fmt.Sprintf("node-%d", i)))
|
||||
}
|
||||
a.mu.Lock()
|
||||
outs, servers := len(a.outbounds), len(a.servers)
|
||||
a.mu.Unlock()
|
||||
if outs <= maxOutboundKeys {
|
||||
t.Fatalf("RetentionDisabled still pruned the outbound map (%d entries)", outs)
|
||||
}
|
||||
if servers <= maxServerKeys {
|
||||
t.Fatalf("RetentionDisabled still pruned the server map (%d entries)", servers)
|
||||
}
|
||||
if d := a.Snapshot().Dropped; d != (DropStats{}) {
|
||||
t.Fatalf("RetentionDisabled reported drops: %+v", d)
|
||||
}
|
||||
}
|
||||
|
||||
// TestTopKMatchesAFullSort pins the refactor that shrank Snapshot's critical
|
||||
// section: the top-K selections must produce exactly what the old
|
||||
// materialise-everything-and-sort produced, order included. A projection that is
|
||||
// merely "close" would reorder the panel's charts at random.
|
||||
func TestTopKMatchesAFullSort(t *testing.T) {
|
||||
rng := rand.New(rand.NewSource(20260726))
|
||||
for trial := 0; trial < 200; trial++ {
|
||||
n := rng.Intn(60)
|
||||
m := make(map[string]uint64, n)
|
||||
for i := 0; i < n; i++ {
|
||||
m[fmt.Sprintf("name-%d", rng.Intn(80))] = uint64(rng.Intn(5))
|
||||
}
|
||||
for _, k := range []int{1, 3, 25, 1000} {
|
||||
got := mapToServerStats(m, k)
|
||||
|
||||
want := make([]ServerStat, 0, len(m))
|
||||
for name, c := range m {
|
||||
want = append(want, ServerStat{Name: name, Count: c})
|
||||
}
|
||||
sort.Slice(want, func(i, j int) bool {
|
||||
if want[i].Count != want[j].Count {
|
||||
return want[i].Count > want[j].Count
|
||||
}
|
||||
return want[i].Name < want[j].Name
|
||||
})
|
||||
if k > 0 && len(want) > k {
|
||||
want = want[:k]
|
||||
}
|
||||
if len(got) != len(want) {
|
||||
t.Fatalf("trial %d k=%d: got %d rows, want %d", trial, k, len(got), len(want))
|
||||
}
|
||||
for i := range got {
|
||||
if got[i] != want[i] {
|
||||
t.Fatalf("trial %d k=%d row %d: got %+v, want %+v", trial, k, i, got[i], want[i])
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestSnapshotResolvesNamesOutsideTheLock is item 4. Snapshot used to resolve every
|
||||
// device's DHCP hostname while holding the aggregator mutex, and that resolution can
|
||||
// read /tmp/dhcp.leases. While the mutex is held, handleEvent and handleConnEvent are
|
||||
// parked — and the 64-slot buffers feeding them drop on overflow with no counter and
|
||||
// no log, so the cost of a slow read is silently missing query-log rows.
|
||||
func TestSnapshotResolvesNamesOutsideTheLock(t *testing.T) {
|
||||
a := New(nil, nil)
|
||||
|
||||
// One real device row, so Snapshot has a name to resolve.
|
||||
a.mu.Lock()
|
||||
a.deviceDomains["192.168.1.50"] = map[string]uint64{"example.com": 3}
|
||||
a.mu.Unlock()
|
||||
|
||||
var once sync.Once
|
||||
entered := make(chan struct{})
|
||||
release := make(chan struct{})
|
||||
orig := parseLeases
|
||||
parseLeases = func(path string) map[string]string {
|
||||
once.Do(func() { close(entered) })
|
||||
<-release
|
||||
return map[string]string{}
|
||||
}
|
||||
t.Cleanup(func() { parseLeases = orig })
|
||||
|
||||
snapDone := make(chan struct{})
|
||||
go func() {
|
||||
a.Snapshot()
|
||||
close(snapDone)
|
||||
}()
|
||||
|
||||
select {
|
||||
case <-entered:
|
||||
case <-time.After(5 * time.Second):
|
||||
close(release)
|
||||
t.Fatalf("Snapshot never resolved a device name")
|
||||
}
|
||||
|
||||
// Snapshot is now parked inside the lease read. The DNS hot path must be free.
|
||||
// (A non-LAN client label needs no lease lookup of its own, so this measures the
|
||||
// aggregator mutex and nothing else.)
|
||||
folded := make(chan struct{})
|
||||
go func() {
|
||||
a.handleEvent(dnstrack.QueryEvent{
|
||||
Domain: "hot.example", QueryType: 1,
|
||||
Client: netip.MustParseAddr("198.51.100.7"),
|
||||
})
|
||||
close(folded)
|
||||
}()
|
||||
select {
|
||||
case <-folded:
|
||||
case <-time.After(5 * time.Second):
|
||||
close(release)
|
||||
t.Fatalf("handleEvent blocked behind Snapshot's lease-file read — the read is still under a.mu")
|
||||
}
|
||||
|
||||
close(release)
|
||||
<-snapDone
|
||||
}
|
||||
|
||||
// TestSnapshotDeviceDomainsStillNamedAndOrdered guards the output of the same
|
||||
// refactor: moving the work out of the lock must not change a single field.
|
||||
func TestSnapshotDeviceDomainsStillNamedAndOrdered(t *testing.T) {
|
||||
orig := parseLeases
|
||||
parseLeases = func(path string) map[string]string {
|
||||
return map[string]string{"192.168.1.50": "laptop"}
|
||||
}
|
||||
t.Cleanup(func() { parseLeases = orig })
|
||||
|
||||
a := New(nil, nil)
|
||||
a.mu.Lock()
|
||||
a.deviceDomains["192.168.1.50"] = map[string]uint64{"a.example": 1}
|
||||
a.deviceDomains["192.168.1.60"] = map[string]uint64{"b.example": 9}
|
||||
a.deviceDomains[routerDevice] = map[string]uint64{"c.example": 5}
|
||||
a.mu.Unlock()
|
||||
|
||||
rows := a.Snapshot().DeviceDomains
|
||||
if len(rows) != 3 {
|
||||
t.Fatalf("got %d device rows, want 3", len(rows))
|
||||
}
|
||||
// Busiest first: .60 (9), router (5), .50 (1).
|
||||
if rows[0].IP != "192.168.1.60" || rows[1].Name != routerDevice || rows[2].IP != "192.168.1.50" {
|
||||
t.Fatalf("rows are not ordered by hit count: %+v", rows)
|
||||
}
|
||||
if rows[2].Name != "laptop" {
|
||||
t.Fatalf("DHCP hostname was not resolved: %+v", rows[2])
|
||||
}
|
||||
if rows[1].IP != "" {
|
||||
t.Fatalf("the router pseudo-row must carry no IP: %+v", rows[1])
|
||||
}
|
||||
}
|
||||
|
||||
// TestDeviceDomainDropsAreReported: the per-device caps existed already but evicted
|
||||
// in silence. They are now counted, so an operator can tell "these are the top
|
||||
// domains" from "these are the top domains we still had room for".
|
||||
func TestDeviceDomainDropsAreReported(t *testing.T) {
|
||||
a := New(nil, nil)
|
||||
a.mu.Lock()
|
||||
for i := 0; i < maxDomainsPerDevice+50; i++ {
|
||||
a.foldDeviceDomainLocked("192.168.1.50", fmt.Sprintf("d%d.example", i))
|
||||
}
|
||||
// Overflow the client cap too.
|
||||
for i := 0; i < maxDeviceClients+10; i++ {
|
||||
a.foldDeviceDomainLocked(fmt.Sprintf("10.0.0.%d", i%256)+fmt.Sprintf("-%d", i), "x.example")
|
||||
}
|
||||
a.mu.Unlock()
|
||||
|
||||
d := a.Snapshot().Dropped
|
||||
if d.DeviceDomains == 0 {
|
||||
t.Fatalf("a device blew past the %d-domain cap and nothing was reported", maxDomainsPerDevice)
|
||||
}
|
||||
if d.DeviceClients == 0 {
|
||||
t.Fatalf("clients were refused at the %d-client cap and nothing was reported", maxDeviceClients)
|
||||
}
|
||||
}
|
||||
+40
-3
@@ -21,12 +21,30 @@ var leasesPath = "/tmp/dhcp.leases"
|
||||
|
||||
// leaseCache resolves a client IP to its DHCP hostname, caching the parsed lease
|
||||
// table and refreshing it at most once per ttl. Concurrency-safe.
|
||||
//
|
||||
// # Why the TTL refresh is asynchronous
|
||||
//
|
||||
// name() is called from paths that hold the AGGREGATOR mutex — handleEvent's
|
||||
// deviceLabel (the DNS hot path), leaseName, and Snapshot's per-device projection.
|
||||
// The cache has its own lock, so there is no deadlock, but a refresh is an
|
||||
// os.ReadFile of /tmp/dhcp.leases, and doing that inline meant a blocking disk
|
||||
// syscall ran while a.mu was held: every DNS event and every connection event in
|
||||
// the daemon parked behind it, and the bounded 64-slot event buffers they feed
|
||||
// from drop on overflow WITHOUT a counter. A file read on tmpfs is fast — until
|
||||
// the one time it isn't, and then the cost is silently missing query-log rows.
|
||||
//
|
||||
// So only the COLD load is synchronous (once per process, and in production it
|
||||
// happens on the poll loop's first tick, off any lock). Every later refresh runs
|
||||
// in its own goroutine while the caller is served from the previous table, which
|
||||
// is at most ttl stale — a DHCP hostname that is 30 seconds out of date is not a
|
||||
// defect, a stalled DNS pipeline is.
|
||||
type leaseCache struct {
|
||||
ttl time.Duration
|
||||
|
||||
mu sync.Mutex
|
||||
byIP map[string]string
|
||||
fetched time.Time
|
||||
loading bool // a refresh goroutine is in flight
|
||||
}
|
||||
|
||||
func newLeaseCache() *leaseCache {
|
||||
@@ -35,15 +53,20 @@ func newLeaseCache() *leaseCache {
|
||||
|
||||
// name returns a friendly device name for ip: the DHCP hostname when known, else
|
||||
// the bare IP (never empty). Best-effort — a missing/uparseable lease file just
|
||||
// yields the IP.
|
||||
// yields the IP. Never blocks on disk except on the very first call.
|
||||
func (c *leaseCache) name(ip string) string {
|
||||
if ip == "" {
|
||||
return ip
|
||||
}
|
||||
c.mu.Lock()
|
||||
if c.byIP == nil || time.Since(c.fetched) > c.ttl {
|
||||
switch {
|
||||
case c.byIP == nil:
|
||||
// Cold: nothing to serve, so this one read is unavoidable.
|
||||
c.byIP = parseLeases(leasesPath)
|
||||
c.fetched = time.Now()
|
||||
case time.Since(c.fetched) > c.ttl && !c.loading:
|
||||
c.loading = true
|
||||
go c.refresh()
|
||||
}
|
||||
host := c.byIP[ip]
|
||||
c.mu.Unlock()
|
||||
@@ -53,6 +76,17 @@ func (c *leaseCache) name(ip string) string {
|
||||
return ip
|
||||
}
|
||||
|
||||
// refresh re-reads the lease table off any caller's lock and publishes it.
|
||||
func (c *leaseCache) refresh() {
|
||||
table := parseLeases(leasesPath)
|
||||
now := time.Now()
|
||||
c.mu.Lock()
|
||||
c.byIP = table
|
||||
c.fetched = now
|
||||
c.loading = false
|
||||
c.mu.Unlock()
|
||||
}
|
||||
|
||||
// lanNetCache decides whether a connection SOURCE IP is a LAN client. It caches the
|
||||
// LAN-side subnets parsed from /etc/config/network (via shater/devices.LANNets, which
|
||||
// reuses the same UCI network parser the device-discovery page uses) and refreshes them
|
||||
@@ -118,7 +152,10 @@ func (c *lanNetCache) isLAN(addr netip.Addr) bool {
|
||||
|
||||
// parseLeases reads the dnsmasq lease table into an ip->hostname map. Any read or
|
||||
// parse failure yields an empty map (callers fall back to the raw IP).
|
||||
func parseLeases(path string) map[string]string {
|
||||
//
|
||||
// A package var so a test can make the read block and prove that no caller performs
|
||||
// it while holding the aggregator mutex; production never reassigns it.
|
||||
var parseLeases = func(path string) map[string]string {
|
||||
out := map[string]string{}
|
||||
data, err := os.ReadFile(path)
|
||||
if err != nil {
|
||||
|
||||
+106
-36
@@ -75,67 +75,137 @@ func classifyCounter(name string) (display, kind string) {
|
||||
}
|
||||
|
||||
// --- projection / sorting helpers ------------------------------------------
|
||||
//
|
||||
// Every projection below runs while the aggregator mutex is held, and while it is
|
||||
// held the DNS and connection folding paths are parked. Their event buffers are 64
|
||||
// slots deep and drop on overflow without a counter, so time spent here is paid in
|
||||
// missing query-log rows and undercounted stats — invisibly.
|
||||
//
|
||||
// That is why these are top-K SELECTIONS, not "materialise everything, sort it,
|
||||
// slice off 25". The domain and host maps hold up to 5000 entries and the answer is
|
||||
// 25 of them; the old shape allocated a 5000-element slice and ran a 5000-element
|
||||
// sort (~61k comparisons) under the lock to throw 99.5% of it away. topKInsert keeps
|
||||
// only the K survivors and rejects a candidate that cannot beat the weakest of them
|
||||
// in a single comparison, which is what almost every candidate does — so the work
|
||||
// drops to about one comparison per map entry and the allocation to K elements.
|
||||
//
|
||||
// The ORDER is unchanged: same comparator, same "highest count first, name as
|
||||
// tiebreak", so the output is byte-identical to what the sort produced.
|
||||
|
||||
// topKInsert inserts item into buf, a descending-ordered buffer holding at most k
|
||||
// items under the strict weak ordering `before`. It returns the updated buffer.
|
||||
// k <= 0 means unbounded, which degrades to insertion sort — callers with large
|
||||
// inputs must pass a positive k.
|
||||
func topKInsert[T any](buf []T, k int, item T, before func(a, b T) bool) []T {
|
||||
if k > 0 && len(buf) >= k && !before(item, buf[len(buf)-1]) {
|
||||
return buf // cannot displace even the weakest survivor: one comparison, done
|
||||
}
|
||||
i := sort.Search(len(buf), func(j int) bool { return before(item, buf[j]) })
|
||||
if k > 0 && len(buf) >= k {
|
||||
copy(buf[i+1:], buf[i:len(buf)-1]) // drop the weakest, shift right
|
||||
buf[i] = item
|
||||
return buf
|
||||
}
|
||||
buf = append(buf, item)
|
||||
copy(buf[i+1:], buf[i:len(buf)-1])
|
||||
buf[i] = item
|
||||
return buf
|
||||
}
|
||||
|
||||
// topKBuf allocates a buffer sized for a top-k selection over n candidates.
|
||||
func topKBuf[T any](k, n int) []T {
|
||||
if k > 0 && k < n {
|
||||
n = k
|
||||
}
|
||||
return make([]T, 0, n)
|
||||
}
|
||||
|
||||
// topDomainsLocked returns the n highest-count domains, most-popular first, with a
|
||||
// stable tiebreak on the domain name. Caller holds the aggregator mutex.
|
||||
func topDomainsLocked(m map[string]*domainAgg, n int) []DomainStat {
|
||||
all := make([]DomainStat, 0, len(m))
|
||||
for dom, ag := range m {
|
||||
all = append(all, DomainStat{Domain: dom, Count: ag.count, Blocked: ag.blocked})
|
||||
}
|
||||
sort.Slice(all, func(i, j int) bool {
|
||||
if all[i].Count != all[j].Count {
|
||||
return all[i].Count > all[j].Count
|
||||
before := func(a, b DomainStat) bool {
|
||||
if a.Count != b.Count {
|
||||
return a.Count > b.Count
|
||||
}
|
||||
return all[i].Domain < all[j].Domain
|
||||
})
|
||||
if n > 0 && len(all) > n {
|
||||
all = all[:n]
|
||||
return a.Domain < b.Domain
|
||||
}
|
||||
return all
|
||||
out := topKBuf[DomainStat](n, len(m))
|
||||
for dom, ag := range m {
|
||||
out = topKInsert(out, n, DomainStat{Domain: dom, Count: ag.count, Blocked: ag.blocked}, before)
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
// topHostsLocked returns the n highest-count destination hosts, most-popular first,
|
||||
// with a stable tiebreak on the host key. Caller holds the aggregator mutex.
|
||||
func topHostsLocked(m map[string]*hostAgg, n int) []HostStat {
|
||||
all := make([]HostStat, 0, len(m))
|
||||
before := func(a, b HostStat) bool {
|
||||
if a.Count != b.Count {
|
||||
return a.Count > b.Count
|
||||
}
|
||||
return a.Host < b.Host
|
||||
}
|
||||
out := topKBuf[HostStat](n, len(m))
|
||||
for _, h := range m {
|
||||
all = append(all, HostStat{
|
||||
out = topKInsert(out, n, HostStat{
|
||||
Host: h.host,
|
||||
IP: h.ip,
|
||||
Network: h.network,
|
||||
Proto: h.proto,
|
||||
Count: h.count,
|
||||
Bytes: h.bytes,
|
||||
})
|
||||
}, before)
|
||||
}
|
||||
sort.Slice(all, func(i, j int) bool {
|
||||
if all[i].Count != all[j].Count {
|
||||
return all[i].Count > all[j].Count
|
||||
}
|
||||
return all[i].Host < all[j].Host
|
||||
})
|
||||
if n > 0 && len(all) > n {
|
||||
all = all[:n]
|
||||
}
|
||||
return all
|
||||
return out
|
||||
}
|
||||
|
||||
// mapToServerStats projects a name->count map into a count-sorted slice.
|
||||
func mapToServerStats(m map[string]uint64) []ServerStat {
|
||||
out := make([]ServerStat, 0, len(m))
|
||||
for name, c := range m {
|
||||
out = append(out, ServerStat{Name: name, Count: c})
|
||||
}
|
||||
sort.Slice(out, func(i, j int) bool {
|
||||
if out[i].Count != out[j].Count {
|
||||
return out[i].Count > out[j].Count
|
||||
// mapToServerStats projects a name->count map into a count-sorted slice, keeping the
|
||||
// n highest counts (n <= 0 keeps everything).
|
||||
//
|
||||
// The limit is not cosmetic. a.outbounds is keyed by OUTBOUND TAG — a node name from
|
||||
// the subscription — so it is one of the two maps whose key space a provider chooses,
|
||||
// and this projection handed the whole thing to the panel on every /api/stats poll.
|
||||
// The panel already renders only the top 10 (Insights.tsx), so nothing on screen
|
||||
// changes; what changes is that a 2000-entry map stops being marshalled into JSON
|
||||
// several times a minute.
|
||||
func mapToServerStats(m map[string]uint64, n int) []ServerStat {
|
||||
before := func(a, b ServerStat) bool {
|
||||
if a.Count != b.Count {
|
||||
return a.Count > b.Count
|
||||
}
|
||||
return out[i].Name < out[j].Name
|
||||
})
|
||||
return a.Name < b.Name
|
||||
}
|
||||
out := topKBuf[ServerStat](n, len(m))
|
||||
for name, c := range m {
|
||||
out = topKInsert(out, n, ServerStat{Name: name, Count: c}, before)
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
// pruneCountMapLocked drops the lowest-count entries of a name->count map until keep
|
||||
// remain, and reports how many it dropped. Same amortised strategy as
|
||||
// pruneDomainsLocked: run only on overflow, so the sort is paid once per (max-keep)
|
||||
// inserts. Caller holds the aggregator mutex.
|
||||
func pruneCountMapLocked(m map[string]uint64, keep int) int {
|
||||
if len(m) <= keep {
|
||||
return 0
|
||||
}
|
||||
type kv struct {
|
||||
name string
|
||||
count uint64
|
||||
}
|
||||
all := make([]kv, 0, len(m))
|
||||
for name, c := range m {
|
||||
all = append(all, kv{name, c})
|
||||
}
|
||||
sort.Slice(all, func(i, j int) bool { return all[i].count < all[j].count })
|
||||
drop := len(all) - keep
|
||||
for i := 0; i < drop; i++ {
|
||||
delete(m, all[i].name)
|
||||
}
|
||||
return drop
|
||||
}
|
||||
|
||||
func sortDevicesByBytes(d []DeviceStat) {
|
||||
sort.Slice(d, func(i, j int) bool {
|
||||
if d[i].Bytes != d[j].Bytes {
|
||||
|
||||
+180
-43
@@ -79,6 +79,40 @@ const (
|
||||
maxDomainsPerDevice = 400 // per-client domain-map cap before it is pruned
|
||||
keepDomainsPerDevice = 200 // per-client prune target (kept by count)
|
||||
topDomainsPerDevice = 15 // domains Snapshot returns per device (highest count first)
|
||||
|
||||
// Caps for the two per-tag DNS maps. These were the last aggregates with no
|
||||
// bound of any kind: every distinct DNSServer and every distinct outbound tag
|
||||
// ever seen stayed in the map for the life of the process.
|
||||
//
|
||||
// a.servers is keyed by RESOLVER tag, which comes from the config — a handful
|
||||
// on any real box. maxServerKeys is therefore a backstop against churn (a tag
|
||||
// renamed on each of a few thousand applies), not against volume; 256 is ~25x
|
||||
// the largest resolver set this generator emits, and at ~80 B/entry the full
|
||||
// table is 20 KB.
|
||||
maxServerKeys = 256
|
||||
keepServerKeys = 128
|
||||
|
||||
// a.outbounds is keyed by OUTBOUND tag — on this box a node NAME chosen by the
|
||||
// subscription provider, plus the derived per-group and per-chain copies. With
|
||||
// ~380 nodes a live generation is on the order of 1000 distinct tags, and the
|
||||
// provider renames nodes on the daily refresh, so this map grew by a whole
|
||||
// generation per day forever. 2048 is two live generations, enough that a
|
||||
// rename-day overlap never evicts a tag still in use; at ~120 B/entry (tag
|
||||
// strings run 30-50 B) the full table is under 250 KB of a 512 MB box.
|
||||
maxOutboundKeys = 2048
|
||||
keepOutboundKeys = 1024
|
||||
|
||||
// topServersN is how many rows Snapshot returns for Servers / Outbounds. Mirrors
|
||||
// topDomainsN; the panel renders the top 10 of either (Insights.tsx), so 25 is
|
||||
// already generous.
|
||||
topServersN = 25
|
||||
|
||||
// dropLogInterval throttles the "aggregate pruned" notices. Every prune is
|
||||
// counted in Snapshot.Dropped and so is always visible in the API; the log line
|
||||
// exists for an operator who is watching logread, and once per aggregate per
|
||||
// five minutes is enough to make a standing condition obvious without becoming
|
||||
// the flood the log sink then has to collapse.
|
||||
dropLogInterval = 5 * time.Minute
|
||||
)
|
||||
|
||||
// Config holds the memory-retention tuning knobs, sourced from model.Globals (Task B,
|
||||
@@ -277,6 +311,35 @@ type ServerStat struct {
|
||||
Count uint64 `json:"count"`
|
||||
}
|
||||
|
||||
// DropStats reports what the retention caps have EVICTED since the daemon started —
|
||||
// one running total per bounded aggregate.
|
||||
//
|
||||
// It exists because eviction that nobody can see is indistinguishable from data that
|
||||
// never existed. Every one of these maps is bounded on purpose, and a bound that is
|
||||
// working is fine; a bound that is CONSTANTLY working means the box is seeing more
|
||||
// distinct domains / hosts / clients / outbound tags than the cap was sized for, and
|
||||
// the numbers in TopDomains, TopHosts and DeviceDomains are then a sample rather than
|
||||
// a total. That is a materially different claim, and the operator is entitled to
|
||||
// know which one they are reading.
|
||||
//
|
||||
// A zero DropStats — the normal state — says every aggregate above is complete.
|
||||
type DropStats struct {
|
||||
// Domains / Hosts count entries dropped from the network-wide domain and
|
||||
// destination-host maps by the maxDomains cap (lowest-count entries go first).
|
||||
Domains uint64 `json:"domains"`
|
||||
Hosts uint64 `json:"hosts"`
|
||||
// DeviceDomains counts per-device domain entries dropped by maxDomainsPerDevice,
|
||||
// summed over all devices. DeviceClients counts whole CLIENTS never tracked at
|
||||
// all because maxDeviceClients was already reached (a spoofed-source flood, or a
|
||||
// genuinely huge LAN).
|
||||
DeviceDomains uint64 `json:"device_domains"`
|
||||
DeviceClients uint64 `json:"device_clients"`
|
||||
// Servers / Outbounds count entries dropped from the per-resolver and
|
||||
// per-outbound-tag DNS maps (maxServerKeys / maxOutboundKeys).
|
||||
Servers uint64 `json:"servers"`
|
||||
Outbounds uint64 `json:"outbounds"`
|
||||
}
|
||||
|
||||
// NodeHealthStat is one node's REAL health as measured by the engine's internal
|
||||
// urltest probing (feedback #3 — replaces the panel's fake "enabled-count/total").
|
||||
// A node's tag equals its node name (the generator emits every leaf node with
|
||||
@@ -390,6 +453,11 @@ type Snapshot struct {
|
||||
// Recent is a small tail of the query log embedded for convenience; the full
|
||||
// log is served by GET /api/stats/log (RecentQueries).
|
||||
Recent []LogEntry `json:"recent"`
|
||||
// Dropped reports what the retention caps have evicted since start — see
|
||||
// DropStats. All zeros (the normal state) means every aggregate above is a
|
||||
// complete total rather than a sample. Additive field: a client that ignores it
|
||||
// sees exactly the pre-existing contract.
|
||||
Dropped DropStats `json:"dropped"`
|
||||
}
|
||||
|
||||
// Normalized returns a copy of the snapshot in which EVERY slice field is non-nil,
|
||||
@@ -521,6 +589,13 @@ type Aggregator struct {
|
||||
keepDomains int
|
||||
retentionDisabled bool
|
||||
|
||||
// dropped counts what the caps above have evicted, surfaced as Snapshot.Dropped.
|
||||
// Guarded by mu like every aggregate it describes. dropLogged remembers when each
|
||||
// kind last produced a log line so a standing overflow is reported at
|
||||
// dropLogInterval instead of on every prune.
|
||||
dropped DropStats
|
||||
dropLogged map[string]time.Time
|
||||
|
||||
// which manager each consume loop is currently attached to (pointer identity is how
|
||||
// Resubscribe detects a box-swap that replaced the manager).
|
||||
curManager *dnstrack.Manager
|
||||
@@ -603,6 +678,7 @@ func newAggregatorShared(eng *engine.Engine, logger log.ContextLogger, cfg ...Co
|
||||
outbounds: make(map[string]uint64),
|
||||
deviceDomains: make(map[string]map[string]uint64),
|
||||
hosts: make(map[string]*hostAgg),
|
||||
dropLogged: make(map[string]time.Time),
|
||||
timelineMinutes: tlMin,
|
||||
timelineUnlimited: tlUnlimited,
|
||||
maxDomains: maxDom,
|
||||
@@ -859,13 +935,26 @@ func (a *Aggregator) handleEvent(ev dnstrack.QueryEvent) {
|
||||
}
|
||||
}
|
||||
|
||||
// Per-server / per-outbound.
|
||||
// Per-server / per-outbound. Both maps are keyed by tags that come from OUTSIDE
|
||||
// this process (config resolver tags; subscription-chosen node names for the
|
||||
// outbounds), so both are bounded the same insert-then-prune way as the domain
|
||||
// map — and, like it, skipped when retention is deliberately disabled.
|
||||
if ev.DNSServer != "" {
|
||||
a.servers[ev.DNSServer]++
|
||||
if !a.retentionDisabled && len(a.servers) > maxServerKeys {
|
||||
n := pruneCountMapLocked(a.servers, keepServerKeys)
|
||||
a.dropped.Servers += uint64(n)
|
||||
a.noteDropLocked("servers", n, a.dropped.Servers, maxServerKeys)
|
||||
}
|
||||
}
|
||||
for _, ob := range ev.Outbound {
|
||||
if ob != "" {
|
||||
a.outbounds[ob]++
|
||||
if !a.retentionDisabled && len(a.outbounds) > maxOutboundKeys {
|
||||
n := pruneCountMapLocked(a.outbounds, keepOutboundKeys)
|
||||
a.dropped.Outbounds += uint64(n)
|
||||
a.noteDropLocked("outbounds", n, a.dropped.Outbounds, maxOutboundKeys)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -945,6 +1034,32 @@ func (a *Aggregator) pruneDomainsLocked() {
|
||||
for i := 0; i < drop; i++ {
|
||||
delete(a.domains, all[i].domain)
|
||||
}
|
||||
a.dropped.Domains += uint64(drop)
|
||||
a.noteDropLocked("domains", drop, a.dropped.Domains, a.maxDomains)
|
||||
}
|
||||
|
||||
// noteDropLocked records an eviction in the log, throttled per kind.
|
||||
//
|
||||
// Eviction is never silent here — the running totals go out on every /api/stats in
|
||||
// Snapshot.Dropped — but the API is a pull, and an operator staring at logread while
|
||||
// the numbers look wrong is exactly the person who needs to be told that TopDomains
|
||||
// is a sample and not a total. Throttling to dropLogInterval per kind keeps a
|
||||
// standing overflow visible without turning it into the flood logsink then has to
|
||||
// collapse. Caller holds mu.
|
||||
func (a *Aggregator) noteDropLocked(kind string, n int, total uint64, limit int) {
|
||||
if n <= 0 || a.log == nil {
|
||||
return
|
||||
}
|
||||
now := time.Now()
|
||||
if last, ok := a.dropLogged[kind]; ok && now.Sub(last) < dropLogInterval {
|
||||
return
|
||||
}
|
||||
a.dropLogged[kind] = now
|
||||
// Deliberately "dropped", not "evicted": most kinds here evict the lowest-count
|
||||
// entries already held, but device-clients drops the NEW arrival instead. One
|
||||
// word that is true of both beats a precise one that is false half the time.
|
||||
a.log.Warn("stats: ", kind, " at its ", limit, "-entry cap — dropped ", n,
|
||||
" (", total, " total since start); this aggregate is a sample, not a total")
|
||||
}
|
||||
|
||||
// --- connection consume loop (per-device domains) ---------------------------
|
||||
@@ -1206,6 +1321,8 @@ func (a *Aggregator) pruneHostsLocked() {
|
||||
for i := 0; i < drop; i++ {
|
||||
delete(a.hosts, all[i].key)
|
||||
}
|
||||
a.dropped.Hosts += uint64(drop)
|
||||
a.noteDropLocked("hosts", drop, a.dropped.Hosts, a.maxDomains)
|
||||
}
|
||||
|
||||
// leaseName resolves a source IP to its DHCP hostname, returning "" (not the bare IP)
|
||||
@@ -1255,6 +1372,8 @@ func (a *Aggregator) foldDeviceDomainLocked(ip, domain string) {
|
||||
// the outer map unbounded). When full, existing clients keep updating but new
|
||||
// ones are dropped. Skipped entirely when retention is disabled.
|
||||
if !a.retentionDisabled && len(a.deviceDomains) >= maxDeviceClients {
|
||||
a.dropped.DeviceClients++
|
||||
a.noteDropLocked("device-clients", 1, a.dropped.DeviceClients, maxDeviceClients)
|
||||
return
|
||||
}
|
||||
dm = make(map[string]uint64)
|
||||
@@ -1262,27 +1381,19 @@ func (a *Aggregator) foldDeviceDomainLocked(ip, domain string) {
|
||||
}
|
||||
dm[domain]++
|
||||
if !a.retentionDisabled && len(dm) > maxDomainsPerDevice {
|
||||
pruneDeviceDomainsLocked(dm, keepDomainsPerDevice)
|
||||
n := pruneDeviceDomainsLocked(dm, keepDomainsPerDevice)
|
||||
a.dropped.DeviceDomains += uint64(n)
|
||||
a.noteDropLocked("device-domains", n, a.dropped.DeviceDomains, maxDomainsPerDevice)
|
||||
}
|
||||
}
|
||||
|
||||
// pruneDeviceDomainsLocked trims one device's domain map back to keep entries, dropping
|
||||
// the lowest-count domains (same amortised strategy as pruneDomainsLocked). Caller
|
||||
// holds mu.
|
||||
func pruneDeviceDomainsLocked(dm map[string]uint64, keep int) {
|
||||
type kv struct {
|
||||
domain string
|
||||
count uint64
|
||||
}
|
||||
all := make([]kv, 0, len(dm))
|
||||
for dom, c := range dm {
|
||||
all = append(all, kv{dom, c})
|
||||
}
|
||||
sort.Slice(all, func(i, j int) bool { return all[i].count < all[j].count })
|
||||
drop := len(all) - keep
|
||||
for i := 0; i < drop; i++ {
|
||||
delete(dm, all[i].domain)
|
||||
}
|
||||
// It returns how many entries it dropped, so the caller can account for them in
|
||||
// Snapshot.Dropped.
|
||||
func pruneDeviceDomainsLocked(dm map[string]uint64, keep int) int {
|
||||
return pruneCountMapLocked(dm, keep)
|
||||
}
|
||||
|
||||
// --- nft poll loop ----------------------------------------------------------
|
||||
@@ -1403,8 +1514,6 @@ func (a *Aggregator) Snapshot() Snapshot {
|
||||
nodeHealth := a.collectNodeHealth()
|
||||
|
||||
a.mu.Lock()
|
||||
defer a.mu.Unlock()
|
||||
|
||||
snap := Snapshot{
|
||||
Backend: a.backend,
|
||||
EngineUp: a.eng != nil && a.eng.Running(),
|
||||
@@ -1415,13 +1524,23 @@ func (a *Aggregator) Snapshot() Snapshot {
|
||||
DeviceDomains: a.deviceDomainsLocked(),
|
||||
TopHosts: topHostsLocked(a.hosts, topHostsN),
|
||||
RuleTraffic: append([]RuleStat(nil), a.ruleTraffic...),
|
||||
Servers: mapToServerStats(a.servers),
|
||||
Outbounds: mapToServerStats(a.outbounds),
|
||||
Servers: mapToServerStats(a.servers, topServersN),
|
||||
Outbounds: mapToServerStats(a.outbounds, topServersN),
|
||||
NodeHealth: nodeHealth,
|
||||
Recent: a.recentLocked(20),
|
||||
Dropped: a.dropped,
|
||||
}
|
||||
if !a.lastPoll.IsZero() {
|
||||
snap.UpdatedAt = a.lastPoll.Format(time.RFC3339)
|
||||
lastPoll := a.lastPoll
|
||||
a.mu.Unlock()
|
||||
|
||||
// Naming and ordering happen OUTSIDE the lock. snap.DeviceDomains is already a
|
||||
// private copy at this point, and a.leases.name may miss its cache — a disk read
|
||||
// under a.mu parks the DNS and connection folding paths, whose 64-slot event
|
||||
// buffers then drop silently. See finishDeviceDomains.
|
||||
a.finishDeviceDomains(snap.DeviceDomains)
|
||||
|
||||
if !lastPoll.IsZero() {
|
||||
snap.UpdatedAt = lastPoll.Format(time.RFC3339)
|
||||
} else {
|
||||
snap.UpdatedAt = time.Now().Format(time.RFC3339)
|
||||
}
|
||||
@@ -1432,10 +1551,14 @@ func (a *Aggregator) Snapshot() Snapshot {
|
||||
}
|
||||
|
||||
// deviceDomainsLocked projects the per-device domain map into the serializable
|
||||
// []DeviceDomainStat, resolving each source IP to a DHCP hostname and returning each
|
||||
// device's top domains (highest count first). Devices are ordered by total hit count
|
||||
// (busiest first), tie-broken by IP for stability. Caller holds mu. (leaseCache has its
|
||||
// own lock, so name() here does not deadlock against a.mu.)
|
||||
// []DeviceDomainStat: one row per tracked source IP carrying that device's top
|
||||
// domains. Caller holds mu.
|
||||
//
|
||||
// It does the map walk and NOTHING else. Resolving DHCP hostnames and ordering the
|
||||
// rows are deliberately left to finishDeviceDomains, which runs unlocked: the row
|
||||
// slice is a private copy the moment this returns, so neither step needs the
|
||||
// aggregator mutex, and both used to be paid with the DNS hot path stopped behind
|
||||
// them — the name lookup potentially including an os.ReadFile of /tmp/dhcp.leases.
|
||||
func (a *Aggregator) deviceDomainsLocked() []DeviceDomainStat {
|
||||
out := make([]DeviceDomainStat, 0, len(a.deviceDomains))
|
||||
for ip, dm := range a.deviceDomains {
|
||||
@@ -1447,38 +1570,52 @@ func (a *Aggregator) deviceDomainsLocked() []DeviceDomainStat {
|
||||
stat.Name = routerDevice
|
||||
} else {
|
||||
stat.IP = ip
|
||||
stat.Name = a.leases.name(ip)
|
||||
}
|
||||
out = append(out, stat)
|
||||
}
|
||||
sort.Slice(out, func(i, j int) bool {
|
||||
ti, tj := totalDomainCount(out[i].Domains), totalDomainCount(out[j].Domains)
|
||||
return out
|
||||
}
|
||||
|
||||
// finishDeviceDomains resolves each row's DHCP hostname and orders the rows by total
|
||||
// hit count (busiest first, tie-broken by IP for stability) — the second half of
|
||||
// deviceDomainsLocked, run WITHOUT the aggregator mutex. The output is identical to
|
||||
// what the single locked pass produced.
|
||||
func (a *Aggregator) finishDeviceDomains(rows []DeviceDomainStat) {
|
||||
for i := range rows {
|
||||
if rows[i].IP != "" {
|
||||
rows[i].Name = a.leases.name(rows[i].IP)
|
||||
}
|
||||
}
|
||||
sort.Slice(rows, func(i, j int) bool {
|
||||
ti, tj := totalDomainCount(rows[i].Domains), totalDomainCount(rows[j].Domains)
|
||||
if ti != tj {
|
||||
return ti > tj
|
||||
}
|
||||
return out[i].IP < out[j].IP
|
||||
return rows[i].IP < rows[j].IP
|
||||
})
|
||||
return out
|
||||
}
|
||||
|
||||
// topDeviceDomains returns one device's n highest-count domains, most-popular first
|
||||
// with a stable name tiebreak. Note the returned Count is the (possibly-capped) live
|
||||
// count for domains that survived any prune.
|
||||
// This is the single hottest projection in Snapshot: it runs once per tracked
|
||||
// device, up to maxDeviceClients (512) of them, over maps of up to
|
||||
// maxDomainsPerDevice (400) entries, all under the aggregator mutex. Materialising
|
||||
// and fully sorting 400 entries per device to keep 15 was ~1.8M comparisons and
|
||||
// ~200k slice elements of allocation with the DNS hot path stopped; the top-K
|
||||
// selection rejects a candidate in one comparison and allocates 15 slots.
|
||||
func topDeviceDomains(dm map[string]uint64, n int) []DomainCount {
|
||||
all := make([]DomainCount, 0, len(dm))
|
||||
for dom, c := range dm {
|
||||
all = append(all, DomainCount{Domain: dom, Count: c})
|
||||
}
|
||||
sort.Slice(all, func(i, j int) bool {
|
||||
if all[i].Count != all[j].Count {
|
||||
return all[i].Count > all[j].Count
|
||||
before := func(a, b DomainCount) bool {
|
||||
if a.Count != b.Count {
|
||||
return a.Count > b.Count
|
||||
}
|
||||
return all[i].Domain < all[j].Domain
|
||||
})
|
||||
if n > 0 && len(all) > n {
|
||||
all = all[:n]
|
||||
return a.Domain < b.Domain
|
||||
}
|
||||
return all
|
||||
out := topKBuf[DomainCount](n, len(dm))
|
||||
for dom, c := range dm {
|
||||
out = topKInsert(out, n, DomainCount{Domain: dom, Count: c}, before)
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
// totalDomainCount sums a device's returned domain counts (used only to order devices).
|
||||
|
||||
@@ -87,22 +87,27 @@ func (c *Client) DialContext(ctx context.Context) (net.Conn, error) {
|
||||
request.Header.Set("Upgrade", "websocket")
|
||||
err = request.Write(conn)
|
||||
if err != nil {
|
||||
conn.Close()
|
||||
return nil, err
|
||||
}
|
||||
bufReader := std_bufio.NewReader(conn)
|
||||
response, err := http.ReadResponse(bufReader, request)
|
||||
if err != nil {
|
||||
conn.Close()
|
||||
return nil, err
|
||||
}
|
||||
if response.StatusCode != 101 ||
|
||||
!strings.EqualFold(response.Header.Get("Connection"), "upgrade") ||
|
||||
!strings.EqualFold(response.Header.Get("Upgrade"), "websocket") {
|
||||
conn.Close()
|
||||
response.Body.Close()
|
||||
return nil, E.New("v2ray-http-upgrade: unexpected status: ", response.Status)
|
||||
}
|
||||
if bufReader.Buffered() > 0 {
|
||||
buffer := buf.NewSize(bufReader.Buffered())
|
||||
_, err = buffer.ReadFullFrom(bufReader, buffer.Len())
|
||||
if err != nil {
|
||||
conn.Close()
|
||||
return nil, err
|
||||
}
|
||||
conn = bufio.NewCachedConn(conn, buffer)
|
||||
|
||||
@@ -0,0 +1,124 @@
|
||||
package v2rayhttpupgrade
|
||||
|
||||
import (
|
||||
"context"
|
||||
"net"
|
||||
"sync"
|
||||
"sync/atomic"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"github.com/sagernet/sing-box/option"
|
||||
M "github.com/sagernet/sing/common/metadata"
|
||||
|
||||
"github.com/stretchr/testify/require"
|
||||
)
|
||||
|
||||
// A failed http-upgrade handshake used to return the error while leaving the
|
||||
// dialed TCP connection open: nothing in the caller chain owns a conn that was
|
||||
// never returned. On a router that keeps ~380 nodes under a health checker, one
|
||||
// leaked descriptor per failed handshake is a slow death. Upstream 0f1763877.
|
||||
|
||||
type trackedConn struct {
|
||||
net.Conn
|
||||
closed atomic.Bool
|
||||
}
|
||||
|
||||
func (c *trackedConn) Close() error {
|
||||
c.closed.Store(true)
|
||||
return c.Conn.Close()
|
||||
}
|
||||
|
||||
// dialRecorder dials the real (test) listener and remembers every conn handed
|
||||
// out, so the test can assert on the exact conn the client was given.
|
||||
type dialRecorder struct {
|
||||
access sync.Mutex
|
||||
conns []*trackedConn
|
||||
}
|
||||
|
||||
func (d *dialRecorder) DialContext(ctx context.Context, network string, destination M.Socksaddr) (net.Conn, error) {
|
||||
conn, err := new(net.Dialer).DialContext(ctx, network, destination.String())
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
tracked := &trackedConn{Conn: conn}
|
||||
d.access.Lock()
|
||||
d.conns = append(d.conns, tracked)
|
||||
d.access.Unlock()
|
||||
return tracked, nil
|
||||
}
|
||||
|
||||
func (d *dialRecorder) ListenPacket(ctx context.Context, destination M.Socksaddr) (net.PacketConn, error) {
|
||||
return nil, net.ErrClosed
|
||||
}
|
||||
|
||||
func (d *dialRecorder) only(t *testing.T) *trackedConn {
|
||||
t.Helper()
|
||||
d.access.Lock()
|
||||
defer d.access.Unlock()
|
||||
require.Len(t, d.conns, 1, "client must have dialed exactly once")
|
||||
return d.conns[0]
|
||||
}
|
||||
|
||||
// serveOnce accepts one connection, drains the request and writes raw response
|
||||
// bytes back.
|
||||
func serveOnce(t *testing.T, response string) net.Listener {
|
||||
t.Helper()
|
||||
listener, err := net.Listen("tcp", "127.0.0.1:0")
|
||||
require.NoError(t, err)
|
||||
t.Cleanup(func() { listener.Close() })
|
||||
go func() {
|
||||
conn, acceptErr := listener.Accept()
|
||||
if acceptErr != nil {
|
||||
return
|
||||
}
|
||||
defer conn.Close()
|
||||
buffer := make([]byte, 4096)
|
||||
conn.SetReadDeadline(time.Now().Add(5 * time.Second))
|
||||
conn.Read(buffer)
|
||||
if response != "" {
|
||||
conn.Write([]byte(response))
|
||||
}
|
||||
}()
|
||||
return listener
|
||||
}
|
||||
|
||||
func newTestClient(t *testing.T, dialer *dialRecorder, listener net.Listener) *Client {
|
||||
t.Helper()
|
||||
client, err := NewClient(
|
||||
context.Background(),
|
||||
dialer,
|
||||
M.ParseSocksaddr(listener.Addr().String()),
|
||||
option.V2RayHTTPUpgradeOptions{Path: "/"},
|
||||
nil,
|
||||
)
|
||||
require.NoError(t, err)
|
||||
return client
|
||||
}
|
||||
|
||||
func TestClientClosesConnOnUnexpectedStatus(t *testing.T) {
|
||||
t.Parallel()
|
||||
listener := serveOnce(t, "HTTP/1.1 403 Forbidden\r\nContent-Length: 0\r\n\r\n")
|
||||
dialer := &dialRecorder{}
|
||||
client := newTestClient(t, dialer, listener)
|
||||
|
||||
conn, err := client.DialContext(context.Background())
|
||||
require.Error(t, err)
|
||||
require.Nil(t, conn)
|
||||
require.True(t, dialer.only(t).closed.Load(),
|
||||
"the dialed conn must be closed when the upgrade is refused")
|
||||
}
|
||||
|
||||
func TestClientClosesConnOnResponseFailure(t *testing.T) {
|
||||
t.Parallel()
|
||||
// The server hangs up without answering: http.ReadResponse fails.
|
||||
listener := serveOnce(t, "")
|
||||
dialer := &dialRecorder{}
|
||||
client := newTestClient(t, dialer, listener)
|
||||
|
||||
conn, err := client.DialContext(context.Background())
|
||||
require.Error(t, err)
|
||||
require.Nil(t, conn)
|
||||
require.True(t, dialer.only(t).closed.Load(),
|
||||
"the dialed conn must be closed when the response cannot be read")
|
||||
}
|
||||
@@ -78,6 +78,12 @@ func (c *Client) offerNew() (*quic.Conn, error) {
|
||||
packetConn.Close()
|
||||
return nil, err
|
||||
}
|
||||
// quic-go does not take ownership of the packet conn passed to Dial:
|
||||
// when the connection ends it only stops reading.
|
||||
go func() {
|
||||
<-quicConn.Context().Done()
|
||||
packetConn.Close()
|
||||
}()
|
||||
c.conn.Store(quicConn)
|
||||
c.rawConn = udpConn
|
||||
return quicConn, nil
|
||||
|
||||
@@ -0,0 +1,203 @@
|
||||
//go:build with_quic
|
||||
|
||||
package v2rayquic
|
||||
|
||||
import (
|
||||
"context"
|
||||
"net"
|
||||
"sync"
|
||||
"sync/atomic"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"github.com/sagernet/quic-go"
|
||||
"github.com/sagernet/sing-box/common/tls"
|
||||
"github.com/sagernet/sing-box/log"
|
||||
"github.com/sagernet/sing-box/option"
|
||||
qtls "github.com/sagernet/sing-quic"
|
||||
M "github.com/sagernet/sing/common/metadata"
|
||||
|
||||
"github.com/stretchr/testify/require"
|
||||
)
|
||||
|
||||
// quic-go does not take ownership of the packet conn handed to Dial: when the
|
||||
// QUIC connection ends it only stops reading from it. offerNew() then dials a
|
||||
// fresh one and overwrites c.rawConn, so the previous UDP socket was leaked for
|
||||
// the lifetime of the process — one per reconnect, on a box with 512 MB and a
|
||||
// health checker that reconnects constantly. Upstream 7067276170.
|
||||
|
||||
const testALPN = "shater-test"
|
||||
|
||||
type trackedConn struct {
|
||||
net.Conn
|
||||
closed atomic.Bool
|
||||
}
|
||||
|
||||
func (c *trackedConn) Close() error {
|
||||
c.closed.Store(true)
|
||||
return c.Conn.Close()
|
||||
}
|
||||
|
||||
type dialRecorder struct {
|
||||
access sync.Mutex
|
||||
conns []*trackedConn
|
||||
}
|
||||
|
||||
func (d *dialRecorder) DialContext(ctx context.Context, network string, destination M.Socksaddr) (net.Conn, error) {
|
||||
conn, err := new(net.Dialer).DialContext(ctx, network, destination.String())
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
tracked := &trackedConn{Conn: conn}
|
||||
d.access.Lock()
|
||||
d.conns = append(d.conns, tracked)
|
||||
d.access.Unlock()
|
||||
return tracked, nil
|
||||
}
|
||||
|
||||
func (d *dialRecorder) ListenPacket(ctx context.Context, destination M.Socksaddr) (net.PacketConn, error) {
|
||||
return nil, net.ErrClosed
|
||||
}
|
||||
|
||||
func (d *dialRecorder) only(t *testing.T) *trackedConn {
|
||||
t.Helper()
|
||||
d.access.Lock()
|
||||
defer d.access.Unlock()
|
||||
require.Len(t, d.conns, 1, "client must have dialed exactly once")
|
||||
return d.conns[0]
|
||||
}
|
||||
|
||||
// serveQUIC brings up a real QUIC listener on localhost with a self-signed
|
||||
// certificate, runs handler for every accepted connection, and returns its
|
||||
// address.
|
||||
func serveQUIC(t *testing.T, handler func(conn *quic.Conn)) M.Socksaddr {
|
||||
t.Helper()
|
||||
ctx := context.Background()
|
||||
logger := log.NewNOPFactory().NewLogger("test")
|
||||
|
||||
keyPem, certificatePem, err := tls.GenerateCertificate(nil, nil, time.Now, "localhost", time.Now().Add(time.Hour))
|
||||
require.NoError(t, err)
|
||||
serverTLSConfig, err := tls.NewSTDServer(ctx, logger, option.InboundTLSOptions{
|
||||
Enabled: true,
|
||||
ServerName: "localhost",
|
||||
ALPN: []string{testALPN},
|
||||
Certificate: []string{string(certificatePem)},
|
||||
Key: []string{string(keyPem)},
|
||||
})
|
||||
require.NoError(t, err)
|
||||
require.NoError(t, serverTLSConfig.Start())
|
||||
t.Cleanup(func() { serverTLSConfig.Close() })
|
||||
|
||||
packetConn, err := net.ListenPacket("udp", "127.0.0.1:0")
|
||||
require.NoError(t, err)
|
||||
t.Cleanup(func() { packetConn.Close() })
|
||||
|
||||
listener, err := qtls.Listen(packetConn, serverTLSConfig, &quic.Config{})
|
||||
require.NoError(t, err)
|
||||
t.Cleanup(func() { listener.Close() })
|
||||
|
||||
go func() {
|
||||
for {
|
||||
conn, acceptErr := listener.Accept(ctx)
|
||||
if acceptErr != nil {
|
||||
return
|
||||
}
|
||||
go handler(conn)
|
||||
}
|
||||
}()
|
||||
return M.ParseSocksaddr(packetConn.LocalAddr().String())
|
||||
}
|
||||
|
||||
func newTestClient(t *testing.T, dialer *dialRecorder, serverAddr M.Socksaddr) *Client {
|
||||
t.Helper()
|
||||
clientTLSConfig, err := tls.NewSTDClient(context.Background(), log.NewNOPFactory().NewLogger("test"), "localhost", option.OutboundTLSOptions{
|
||||
Enabled: true,
|
||||
Insecure: true,
|
||||
ServerName: "localhost",
|
||||
ALPN: []string{testALPN},
|
||||
})
|
||||
require.NoError(t, err)
|
||||
transport, err := NewClient(context.Background(), dialer, serverAddr, option.V2RayQUICOptions{}, clientTLSConfig)
|
||||
require.NoError(t, err)
|
||||
client, isClient := transport.(*Client)
|
||||
require.True(t, isClient)
|
||||
return client
|
||||
}
|
||||
|
||||
func TestClientClosesPacketConnWhenConnectionEnds(t *testing.T) {
|
||||
// Drop the connection right after the handshake: this is the server-side
|
||||
// reset / idle timeout the client must survive without leaking its socket.
|
||||
serverAddr := serveQUIC(t, func(conn *quic.Conn) {
|
||||
conn.CloseWithError(0, "bye")
|
||||
})
|
||||
dialer := &dialRecorder{}
|
||||
client := newTestClient(t, dialer, serverAddr)
|
||||
t.Cleanup(func() { client.Close() })
|
||||
|
||||
quicConn, err := client.offer()
|
||||
require.NoError(t, err)
|
||||
require.NotNil(t, quicConn)
|
||||
|
||||
// The server hangs up; the client keeps its Client alive (a health checker
|
||||
// would simply dial again later).
|
||||
select {
|
||||
case <-quicConn.Context().Done():
|
||||
case <-time.After(5 * time.Second):
|
||||
t.Fatal("server never closed the QUIC connection")
|
||||
}
|
||||
|
||||
tracked := dialer.only(t)
|
||||
require.Eventually(t, tracked.closed.Load, 5*time.Second, 10*time.Millisecond,
|
||||
"the UDP socket behind a dead QUIC connection must be closed, not leaked until Client.Close()")
|
||||
}
|
||||
|
||||
// quic-go's Stream.Close() does not unblock a Write parked on flow control. The
|
||||
// writer goroutine (for us: the copy loop of a proxied connection) then survives
|
||||
// its own connection forever. Closing has to push the write deadline into the
|
||||
// past as well.
|
||||
func TestStreamCloseUnblocksBlockedWrite(t *testing.T) {
|
||||
serverIsDone := make(chan struct{})
|
||||
t.Cleanup(func() { close(serverIsDone) })
|
||||
// Accept the stream but never read from it, so the client's writes fill the
|
||||
// receive window and block.
|
||||
serverAddr := serveQUIC(t, func(conn *quic.Conn) {
|
||||
_, err := conn.AcceptStream(context.Background())
|
||||
if err != nil {
|
||||
return
|
||||
}
|
||||
<-serverIsDone
|
||||
})
|
||||
dialer := &dialRecorder{}
|
||||
client := newTestClient(t, dialer, serverAddr)
|
||||
t.Cleanup(func() { client.Close() })
|
||||
|
||||
stream, err := client.DialContext(context.Background())
|
||||
require.NoError(t, err)
|
||||
|
||||
writeDone := make(chan error, 1)
|
||||
go func() {
|
||||
payload := make([]byte, 64*1024)
|
||||
// 32 MiB is far past any quic-go receive window, so this must park.
|
||||
for range 512 {
|
||||
_, writeErr := stream.Write(payload)
|
||||
if writeErr != nil {
|
||||
writeDone <- writeErr
|
||||
return
|
||||
}
|
||||
}
|
||||
writeDone <- nil
|
||||
}()
|
||||
|
||||
select {
|
||||
case err = <-writeDone:
|
||||
t.Fatal("the write never blocked, the test proves nothing: ", err)
|
||||
case <-time.After(time.Second):
|
||||
}
|
||||
|
||||
require.NoError(t, stream.Close())
|
||||
select {
|
||||
case <-writeDone:
|
||||
case <-time.After(5 * time.Second):
|
||||
t.Fatal("Write stayed blocked after Close: the writer goroutine is leaked")
|
||||
}
|
||||
}
|
||||
@@ -2,6 +2,7 @@ package v2rayquic
|
||||
|
||||
import (
|
||||
"net"
|
||||
"time"
|
||||
|
||||
"github.com/sagernet/quic-go"
|
||||
qtls "github.com/sagernet/sing-quic"
|
||||
@@ -37,5 +38,8 @@ func (s *StreamWrapper) Upstream() any {
|
||||
func (s *StreamWrapper) Close() error {
|
||||
s.CancelRead(0)
|
||||
s.Stream.Close()
|
||||
// quic-go's Stream.Close does not unblock a Write blocked on flow control,
|
||||
// but a past write deadline does; buffered data and the FIN are unaffected.
|
||||
s.Stream.SetWriteDeadline(time.Now())
|
||||
return nil
|
||||
}
|
||||
|
||||
@@ -93,12 +93,14 @@ func (c *Client) dialContext(ctx context.Context, requestURL *url.URL, headers h
|
||||
reader, _, err := ws.Dialer{Header: ws.HandshakeHeaderHTTP(headers), Protocols: protocols}.Upgrade(deadlineConn, requestURL)
|
||||
deadlineConn.SetDeadline(time.Time{})
|
||||
if err != nil {
|
||||
conn.Close()
|
||||
return nil, err
|
||||
}
|
||||
if reader != nil {
|
||||
buffer := buf.NewSize(reader.Buffered())
|
||||
_, err = buffer.ReadFullFrom(reader, buffer.Len())
|
||||
if err != nil {
|
||||
conn.Close()
|
||||
return nil, err
|
||||
}
|
||||
conn = bufio.NewCachedConn(conn, buffer)
|
||||
|
||||
@@ -0,0 +1,122 @@
|
||||
package v2raywebsocket
|
||||
|
||||
import (
|
||||
"context"
|
||||
"net"
|
||||
"sync"
|
||||
"sync/atomic"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"github.com/sagernet/sing-box/option"
|
||||
M "github.com/sagernet/sing/common/metadata"
|
||||
|
||||
"github.com/stretchr/testify/require"
|
||||
)
|
||||
|
||||
// A refused websocket upgrade used to return the error while leaving the dialed
|
||||
// TCP connection open — the conn was never returned, so nobody else could close
|
||||
// it. With ~380 nodes under a health checker every failing node leaks one
|
||||
// descriptor per probe. Upstream 0f1763877.
|
||||
|
||||
type trackedConn struct {
|
||||
net.Conn
|
||||
closed atomic.Bool
|
||||
}
|
||||
|
||||
func (c *trackedConn) Close() error {
|
||||
c.closed.Store(true)
|
||||
return c.Conn.Close()
|
||||
}
|
||||
|
||||
type dialRecorder struct {
|
||||
access sync.Mutex
|
||||
conns []*trackedConn
|
||||
}
|
||||
|
||||
func (d *dialRecorder) DialContext(ctx context.Context, network string, destination M.Socksaddr) (net.Conn, error) {
|
||||
conn, err := new(net.Dialer).DialContext(ctx, network, destination.String())
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
tracked := &trackedConn{Conn: conn}
|
||||
d.access.Lock()
|
||||
d.conns = append(d.conns, tracked)
|
||||
d.access.Unlock()
|
||||
return tracked, nil
|
||||
}
|
||||
|
||||
func (d *dialRecorder) ListenPacket(ctx context.Context, destination M.Socksaddr) (net.PacketConn, error) {
|
||||
return nil, net.ErrClosed
|
||||
}
|
||||
|
||||
func (d *dialRecorder) only(t *testing.T) *trackedConn {
|
||||
t.Helper()
|
||||
d.access.Lock()
|
||||
defer d.access.Unlock()
|
||||
require.Len(t, d.conns, 1, "client must have dialed exactly once")
|
||||
return d.conns[0]
|
||||
}
|
||||
|
||||
func serveOnce(t *testing.T, response string) net.Listener {
|
||||
t.Helper()
|
||||
listener, err := net.Listen("tcp", "127.0.0.1:0")
|
||||
require.NoError(t, err)
|
||||
t.Cleanup(func() { listener.Close() })
|
||||
go func() {
|
||||
conn, acceptErr := listener.Accept()
|
||||
if acceptErr != nil {
|
||||
return
|
||||
}
|
||||
defer conn.Close()
|
||||
buffer := make([]byte, 4096)
|
||||
conn.SetReadDeadline(time.Now().Add(5 * time.Second))
|
||||
conn.Read(buffer)
|
||||
if response != "" {
|
||||
conn.Write([]byte(response))
|
||||
}
|
||||
}()
|
||||
return listener
|
||||
}
|
||||
|
||||
func newTestClient(t *testing.T, dialer *dialRecorder, listener net.Listener) *Client {
|
||||
t.Helper()
|
||||
transport, err := NewClient(
|
||||
context.Background(),
|
||||
dialer,
|
||||
M.ParseSocksaddr(listener.Addr().String()),
|
||||
option.V2RayWebsocketOptions{Path: "/"},
|
||||
nil,
|
||||
)
|
||||
require.NoError(t, err)
|
||||
client, isClient := transport.(*Client)
|
||||
require.True(t, isClient)
|
||||
return client
|
||||
}
|
||||
|
||||
func TestClientClosesConnOnRefusedUpgrade(t *testing.T) {
|
||||
t.Parallel()
|
||||
listener := serveOnce(t, "HTTP/1.1 403 Forbidden\r\nContent-Length: 0\r\n\r\n")
|
||||
dialer := &dialRecorder{}
|
||||
client := newTestClient(t, dialer, listener)
|
||||
|
||||
conn, err := client.DialContext(context.Background())
|
||||
require.Error(t, err)
|
||||
require.Nil(t, conn)
|
||||
require.True(t, dialer.only(t).closed.Load(),
|
||||
"the dialed conn must be closed when the websocket upgrade is refused")
|
||||
}
|
||||
|
||||
func TestClientClosesConnOnHandshakeEOF(t *testing.T) {
|
||||
t.Parallel()
|
||||
// The server hangs up mid-handshake: ws.Dialer.Upgrade fails on read.
|
||||
listener := serveOnce(t, "")
|
||||
dialer := &dialRecorder{}
|
||||
client := newTestClient(t, dialer, listener)
|
||||
|
||||
conn, err := client.DialContext(context.Background())
|
||||
require.Error(t, err)
|
||||
require.Nil(t, conn)
|
||||
require.True(t, dialer.only(t).closed.Load(),
|
||||
"the dialed conn must be closed when the handshake cannot complete")
|
||||
}
|
||||
@@ -187,6 +187,7 @@ func (c *EarlyWebsocketConn) writeRequest(content []byte) error {
|
||||
if len(lateData) > 0 {
|
||||
_, err = conn.Write(lateData)
|
||||
if err != nil {
|
||||
conn.Close()
|
||||
return err
|
||||
}
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user