Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
1267d20fb8 | ||
|
|
35f697ed08 | ||
|
|
d0fb6befb1 | ||
|
|
81c96019b5 | ||
|
|
baed8ff8f2 | ||
|
|
f80fb4dd1b | ||
|
|
b71b793681 | ||
|
|
61c87ad1d9 | ||
|
|
4dee508e12 | ||
|
|
76da5134ef | ||
|
|
4c630c9a13 | ||
|
|
d8dbefcd07 | ||
|
|
974208fc05 | ||
|
|
2eb71e8244 | ||
|
|
668cccbf24 | ||
|
|
4ea4585402 | ||
|
|
dc6d102473 | ||
|
|
683afc0a47 | ||
|
|
2c3e20512e | ||
|
|
51b2f04672 | ||
|
|
f190c8251e | ||
|
|
1945404eaa |
@@ -0,0 +1,145 @@
|
||||
// lx:begin l3-honest-drop
|
||||
package adapter
|
||||
|
||||
import (
|
||||
"net/netip"
|
||||
"testing"
|
||||
|
||||
"github.com/sagernet/sing-tun"
|
||||
"github.com/sagernet/sing-tun/gtcpip/header"
|
||||
|
||||
"github.com/stretchr/testify/require"
|
||||
)
|
||||
|
||||
// judgeFlowRouter answers PreMatch with a canned verdict; JudgeFlow reads
|
||||
// nothing else off the Router.
|
||||
type judgeFlowRouter struct {
|
||||
Router
|
||||
result PreMatchResult
|
||||
}
|
||||
|
||||
func (r *judgeFlowRouter) PreMatch(InboundContext, []byte) PreMatchResult { return r.result }
|
||||
|
||||
// judgeFlowPort is the tun.Port half of a FlowOutbound. inet4 is what
|
||||
// PortAddresses reports for IPv4 — the one field the two ICMP consumers in
|
||||
// sing-tun disagree about (see the comment on
|
||||
// TestJudgeFlowICMPToBoundPortStaysAFlow).
|
||||
type judgeFlowPort struct {
|
||||
Outbound
|
||||
inet4 netip.Addr
|
||||
}
|
||||
|
||||
func (o *judgeFlowPort) Tag() string { return "wg-out" }
|
||||
func (o *judgeFlowPort) Type() string { return "wireguard" }
|
||||
func (o *judgeFlowPort) PortAddresses() (netip.Addr, netip.Addr) {
|
||||
return o.inet4, netip.Addr{}
|
||||
}
|
||||
func (o *judgeFlowPort) PortMTU() uint32 { return 1420 }
|
||||
func (o *judgeFlowPort) AttachReturn(tun.Return) error { return nil }
|
||||
func (o *judgeFlowPort) DetachReturn(tun.Return) error { return nil }
|
||||
func (o *judgeFlowPort) WritePackets(packets [][]byte) error { return nil }
|
||||
|
||||
// judgeFlowNonPort is a FlowOutbound-shaped result that is NOT a tun.Port — the
|
||||
// interface drift the second line of defense in JudgeFlow exists for.
|
||||
type judgeFlowNonPort struct {
|
||||
Outbound
|
||||
}
|
||||
|
||||
func (o *judgeFlowNonPort) Tag() string { return "drifted" }
|
||||
func (o *judgeFlowNonPort) Type() string { return "drifted" }
|
||||
|
||||
func judgeFlow(t *testing.T, protocol uint8, result PreMatchResult) tun.FlowVerdict {
|
||||
t.Helper()
|
||||
return JudgeFlow(
|
||||
&judgeFlowRouter{result: result},
|
||||
"l3-in", "tun", protocol,
|
||||
netip.MustParseAddrPort("192.168.1.2:1234"),
|
||||
netip.MustParseAddrPort("1.1.1.1:1234"),
|
||||
nil,
|
||||
)
|
||||
}
|
||||
|
||||
const (
|
||||
judgeFlowICMP = uint8(header.ICMPv4ProtocolNumber)
|
||||
judgeFlowTCP = uint8(header.TCPProtocolNumber)
|
||||
)
|
||||
|
||||
// TestJudgeFlowICMPToBoundPortStaysAFlow is the guard on the ONE fix that must
|
||||
// not be made here.
|
||||
//
|
||||
// sing-tun has two ICMP consumers with different requirements on the port:
|
||||
//
|
||||
// - ForwardDispatcher.createFlow (flow_dispatch.go) needs only a VALID port
|
||||
// address — it NATs the echo identifier and rewrites the source to that
|
||||
// address. This is the path every unfragmented LAN ping takes, and it is
|
||||
// what makes ping-through-WireGuard/AWG work at all.
|
||||
// - ICMPForwarder.installFlow (stack_gvisor_icmp.go) additionally requires the
|
||||
// address to be UNSPECIFIED, because it writes the packet to the port
|
||||
// unmodified. A WireGuard endpoint reports its concrete interface address
|
||||
// (transport/wireguard/port.go), so installFlow declines and HandlePacket
|
||||
// falls through to forging the echo reply.
|
||||
//
|
||||
// The tempting fix — "for ICMP, refuse ActionFlow when PortAddresses() is not
|
||||
// unspecified, so the verdict becomes a drop and the forgery is unreachable" —
|
||||
// is applied HERE, in the one function both consumers share, with byte-identical
|
||||
// arguments from either. It would therefore kill the working path too: every
|
||||
// ping through WireGuard/AWG, fragmented or not, would drop, and l3_tunnel would
|
||||
// carry nothing but `direct`. Keep this test failing loudly if anyone tries.
|
||||
func TestJudgeFlowICMPToBoundPortStaysAFlow(t *testing.T) {
|
||||
t.Parallel()
|
||||
port := &judgeFlowPort{inet4: netip.MustParseAddr("10.2.0.2")}
|
||||
verdict := judgeFlow(t, judgeFlowICMP, PreMatchResult{Action: PreMatchFlow, Outbound: port})
|
||||
require.Equal(t, tun.ActionFlow, verdict.Action,
|
||||
"ICMP to a WireGuard/AWG endpoint must stay a flow: the forward dispatcher NATs it by echo identifier and this is the whole point of l3_tunnel")
|
||||
require.Same(t, tun.Port(port), verdict.Port)
|
||||
}
|
||||
|
||||
// The `direct` shape: an unspecified port address. Both consumers accept it.
|
||||
func TestJudgeFlowICMPToUnspecifiedPortStaysAFlow(t *testing.T) {
|
||||
t.Parallel()
|
||||
port := &judgeFlowPort{inet4: netip.IPv4Unspecified()}
|
||||
verdict := judgeFlow(t, judgeFlowICMP, PreMatchResult{Action: PreMatchFlow, Outbound: port})
|
||||
require.Equal(t, tun.ActionFlow, verdict.Action)
|
||||
require.Same(t, tun.Port(port), verdict.Port)
|
||||
}
|
||||
|
||||
// PreMatchDrop is the honest verdict and must arrive as ActionDrop: it is the
|
||||
// only value (besides Reject) that stops ICMPForwarder.HandlePacket before the
|
||||
// Echo -> EchoReply rewrite.
|
||||
func TestJudgeFlowICMPDropReachesTheStackAsDrop(t *testing.T) {
|
||||
t.Parallel()
|
||||
verdict := judgeFlow(t, judgeFlowICMP, PreMatchResult{Action: PreMatchDrop})
|
||||
require.Equal(t, tun.ActionDrop, verdict.Action)
|
||||
}
|
||||
|
||||
// The second line of defense: a PreMatchFlow whose outbound is not a tun.Port
|
||||
// must not degrade ICMP to ActionAccept, because Accept is the forged reply.
|
||||
func TestJudgeFlowICMPNonPortOutboundDrops(t *testing.T) {
|
||||
t.Parallel()
|
||||
verdict := judgeFlow(t, judgeFlowICMP, PreMatchResult{Action: PreMatchFlow, Outbound: &judgeFlowNonPort{}})
|
||||
require.Equal(t, tun.ActionDrop, verdict.Action,
|
||||
"FlowOutbound and tun.Port are distinct interfaces; a drift between them must not silently re-enable the echo forger")
|
||||
}
|
||||
|
||||
func TestJudgeFlowTCPNonPortOutboundAccepts(t *testing.T) {
|
||||
t.Parallel()
|
||||
verdict := judgeFlow(t, judgeFlowTCP, PreMatchResult{Action: PreMatchFlow, Outbound: &judgeFlowNonPort{}})
|
||||
require.Equal(t, tun.ActionAccept, verdict.Action,
|
||||
"for TCP, falling back to Accept is upstream behaviour and must stay untouched")
|
||||
}
|
||||
|
||||
// TCP keeps every mapping it had, including the Continue -> Accept default that
|
||||
// is a forgery only for ICMP.
|
||||
func TestJudgeFlowTCPContinueStaysAccept(t *testing.T) {
|
||||
t.Parallel()
|
||||
verdict := judgeFlow(t, judgeFlowTCP, PreMatchResult{Action: PreMatchContinue})
|
||||
require.Equal(t, tun.ActionAccept, verdict.Action)
|
||||
}
|
||||
|
||||
func TestJudgeFlowTCPBypassStaysBypass(t *testing.T) {
|
||||
t.Parallel()
|
||||
verdict := judgeFlow(t, judgeFlowTCP, PreMatchResult{Action: PreMatchBypass})
|
||||
require.Equal(t, tun.ActionBypass, verdict.Action)
|
||||
}
|
||||
|
||||
// lx:end l3-honest-drop
|
||||
@@ -75,7 +75,18 @@ func JudgeFlow(router Router, inbound string, inboundType string, network uint8,
|
||||
case PreMatchFlow:
|
||||
port, isPort := result.Outbound.(tun.Port)
|
||||
if !isPort {
|
||||
// lx:begin l3-honest-drop
|
||||
// Second line of defense behind route.(*Router).preMatchFlow: a
|
||||
// PreMatchFlow result already implies the outbound is an
|
||||
// adapter.FlowOutbound, but FlowOutbound and tun.Port are distinct
|
||||
// interfaces, and a drift between them must not degrade ICMP to
|
||||
// ActionAccept — the TUN stack would then forge the echo reply
|
||||
// itself instead of admitting the tunnel cannot carry the packet.
|
||||
if networkName == N.NetworkICMP {
|
||||
return tun.FlowVerdict{Action: tun.ActionDrop}
|
||||
}
|
||||
return tun.FlowVerdict{Action: tun.ActionAccept}
|
||||
// lx:end l3-honest-drop
|
||||
}
|
||||
verdict := tun.FlowVerdict{Action: tun.ActionFlow, Port: port, UDPTimeout: result.UDPTimeout, NewTracker: result.NewTracker}
|
||||
if result.Destination.IsValid() {
|
||||
|
||||
@@ -12,6 +12,49 @@ as GitHub **pre-releases** and never become "Latest".
|
||||
|
||||
#### Unreleased (shater)
|
||||
|
||||
**`l3-honest-drop` — ICMP routed to an L4-only outbound is dropped, not
|
||||
forged** — ships with `shaterd` (part of the shater L3 ingress,
|
||||
`docs-shater/DECISIONS.md` D25), not as an lx release tag; recorded here because
|
||||
it edits two upstream files. Without it the TUN stack answers an unroutable echo
|
||||
ITSELF — sing-tun's `ICMPForwarder.HandlePacket` rewrites Echo→EchoReply
|
||||
whenever the flow judgment comes back Accept (`stack_gvisor_icmp.go`) — so a
|
||||
ping routed to vless/vmess/… would read as a working tunnel while the packet
|
||||
never left the router.
|
||||
|
||||
* **`route/route.go` (`PreMatch`)** — the pre-match walk was renamed to
|
||||
`preMatch` and the exported `PreMatch` became a thin FUNNEL that rewrites
|
||||
`PreMatchContinue` and `PreMatchBypass` to `PreMatchDrop` for
|
||||
`N.NetworkICMP`. An earlier version overrode `continueResult` inside
|
||||
`preMatchFlow` instead; that covered only the exits reaching that function and
|
||||
left three of the walk's own exits forging — the `prepareMatchMetadata` error
|
||||
return, the sniff bail-outs, and the `default:` arm of the rule-action switch
|
||||
(every action pre-match has no arm for: `hijack-dns`, `direct`, …). A guard on
|
||||
the single return value cannot be outgrown by a new exit. `PreMatchBypass` is
|
||||
folded in because sing-tun implements `ActionBypass` on the nfqueue plane only
|
||||
— on the TUN path it lands in the same `default:` arm as Accept, i.e. forges.
|
||||
* **`adapter/router.go` (`JudgeFlow`, the `!isPort` branch)** — ICMP returns
|
||||
`ActionDrop` where it fell through to `ActionAccept`. Second line of defense:
|
||||
`adapter.FlowOutbound` and `tun.Port` are distinct interfaces, and a drift
|
||||
between them must not quietly re-enable the forged reply.
|
||||
* **TCP/UDP behaviour is unchanged** — `PreMatchContinue` still means "take the
|
||||
ordinary connection route" for both, `PreMatchBypass` still means bypass, and
|
||||
the `!isPort` fallthrough still returns `ActionAccept` for them; pinned by
|
||||
`route/prematch_icmp_lx_test.go` and `adapter/judgeflow_icmp_lx_test.go`
|
||||
(both inside the marker), each ICMP case having an explicit TCP/UDP twin.
|
||||
* **NOT covered: a FRAGMENTED echo to a WireGuard/AWG outbound is still
|
||||
forged** — sing-tun's `ForwardDispatcher.Dispatch` returns before asking for a
|
||||
verdict at all when `parsed.fragment`, and the reassembled packet reaches
|
||||
`ICMPForwarder.HandlePacket`, whose `installFlow` demands an UNSPECIFIED port
|
||||
address that a WireGuard endpoint never has. Fixing it inside `JudgeFlow`
|
||||
is NOT possible — both consumers call it with identical arguments and the
|
||||
working path needs the concrete address. Full chain, the two viable fixes and
|
||||
the trap are in `docs-shater/DECISIONS.md` D25, "KNOWN HOLE".
|
||||
* **Rebase cost: two small marked blocks** (`lx:begin/end l3-honest-drop`, a
|
||||
wrapper function in `route/route.go` and one branch body in
|
||||
`adapter/router.go`) plus the two self-contained test files — carried across
|
||||
an upstream rebase by eye. Note that `PreMatch`'s own body now lives in
|
||||
`preMatch`, so an upstream change to the walk applies to that function.
|
||||
|
||||
**Fork-layer + control-plane rework of proxy health** — ships with `shaterd`
|
||||
(the shater router daemon), not as an lx release tag; recorded here because the
|
||||
load-bearing half lives in fork zones (`common/urltest`, `protocol/group`).
|
||||
|
||||
@@ -319,6 +319,10 @@ Three values, not two, because the leaks differ in *kind*: an ICMP echo is ephem
|
||||
user-initiated and reveals the address only to a host the user deliberately contacted,
|
||||
whereas ESP/GRE is a standing second tunnel carrying arbitrary traffic beside ours. A
|
||||
single toggle would make "I want ping to work" mean "I allow a parallel VPN bypass".
|
||||
*(Refined 2026-07-26 by D25: still true of TPROXY — but ICMP echo now has an
|
||||
opt-in data plane of its own, the dedicated L3 TUN, so the policy no longer
|
||||
speaks alone for ping; it keeps sole charge of ESP/GRE/IGMP and of the degraded
|
||||
paths.)*
|
||||
|
||||
**Fail-open degradations must be visible in the panel, not only in `logread`.** The
|
||||
audit deliberately converted many aborts into warn-and-continue (an unfetchable list,
|
||||
@@ -772,3 +776,429 @@ server, which restores exactly the pre-D24 behaviour and clears the notice. Unti
|
||||
that lands, an operator can get the same result by setting `endpoint_resolver` to a
|
||||
direct resolver. Note the hazard is **not** created by D24 — any config with two
|
||||
resolvers has it today; the default merely makes it universal.
|
||||
|
||||
## D25 — L3 ingress: LAN ICMP rides a dedicated TUN through the tunnel, not a policy verdict
|
||||
Decided 2026-07-26. D17 made everything TPROXY cannot divert an explicit policy
|
||||
(`Globals.Untunnelable` = block | icmp | direct) — and its premise still holds:
|
||||
kernel TPROXY delivers a packet by handing it to a listening SOCKET, and sockets
|
||||
exist for TCP and UDP only, so an ICMP echo has nothing to be handed to. But a
|
||||
policy can only choose between losing the packet and leaking it with the
|
||||
client's real source address; neither ever puts a ping THROUGH the tunnel. This
|
||||
decision adds the data plane D17 could not have: **`globals.l3_tunnel` (opt-in,
|
||||
default off; `model.Globals.L3Tunnel`) opens a second, dedicated ingress — a TUN
|
||||
device — and LAN ICMP enters the engine as raw IP packets**, where the ordinary
|
||||
route rules pick an outbound exactly as for any flow. The policy is refined, not
|
||||
repealed: it keeps sole charge of the protocols the engine cannot ingest at all,
|
||||
and of the degraded paths (both below).
|
||||
|
||||
**The whole mechanism is one mark, one rule, one device — the TPROXY plane is
|
||||
untouched.** The nft prerouting chain stamps `L3Mark` (= fwmark_base + 0x80,
|
||||
`netplane/nft.go` `l3MarkOffset`) on LAN `ip protocol icmp` / `meta l4proto
|
||||
ipv6-icmp` ONLY, and only after every local plane was already accepted
|
||||
(fib-local, RFC1918/link-local/multicast daddr sets) and — for v6 — after a
|
||||
unicast ND/NA carve-out, because one tunnelled neighbour probe is enough to take
|
||||
the LAN's v6 plane down (`renderNft`, the L3 block). `addL3Routing`
|
||||
(`netplane/apply.go`) binds that mark to a table (= table_base + 0x08) whose
|
||||
only content is `default dev shater-l3`; del-then-add idempotent, and a failed
|
||||
rule or route is a NAMED operator warning, never an apply abort. `generate`
|
||||
emits the synthetic `l3-in` TUN inbound bound to exactly `netplane.L3Device`,
|
||||
MTU 65535 (the largest IP datagram there can be, so the KERNEL can never
|
||||
fragment on the way in — see "the device MTU is not a tunnel budget" below),
|
||||
point-to-point /30 + /126 addresses from private space,
|
||||
the v6 one only when `globals.ipv6` is on — and only next to a tproxy inbound:
|
||||
the ingress rides the same LAN divert plane, and without one the TUN would sit
|
||||
dark while the config claims ICMP is tunnelled, so it is skipped with a warning
|
||||
(`generate/inbound.go`, `appendL3TunInbound`). `shater/registry` registers the
|
||||
`tun` inbound type; that costs no new build tag and no meaningful size because
|
||||
`with_wireguard` already requires `with_gvisor` (D23, `scripts/router-tags.sh`).
|
||||
|
||||
**`auto_route: false` is load-bearing, not a default we happened to keep.**
|
||||
sing-box's auto_route rewrites the router's MAIN routing table — it would drag
|
||||
everything the router itself sends (WAN traffic, DNS, the tunnel's own underlay)
|
||||
into this TUN. The fwmark rule + dedicated table above is deliberately the ONLY
|
||||
entrance, and disabling the feature can never strand a stale default route in
|
||||
main (`generate/inbound.go`; pinned by `TestL3TunnelEmitsTunInbound`).
|
||||
|
||||
**`stack: "gvisor"` is a deliberate choice, and the tempting reason for it is
|
||||
wrong.** It is TRUE that sing-tun's system stack answers an ICMP echo LOCALLY —
|
||||
`processIPv4ICMP` rewrites Echo→EchoReply in place and swaps the addresses
|
||||
(sing-tun `stack_system.go:648`; the v6 twin sits right under it). It is FALSE
|
||||
that this makes the system stack unusable here: `dispatchIPv4`
|
||||
(`stack_system.go:355-372`) hands the packet to the SAME `ForwardDispatcher`
|
||||
first and only falls through to that forger for packets addressed to the TUN
|
||||
itself, exactly as the gVisor filter does (`stack_gvisor_filter.go:52-113`).
|
||||
Both stacks would forward. gvisor is chosen because it is already linked —
|
||||
`with_wireguard` requires `with_gvisor` (D23), so it costs no tag and no new
|
||||
code path — and because it is the combination the integration test actually
|
||||
exercises. Do not re-derive this as "the system stack fakes ping": it fakes ping
|
||||
only where the dispatcher declined the packet.
|
||||
|
||||
**The ceiling is ICMP echo, and it is upstream's dispatcher — NOT the netstack.**
|
||||
This distinction matters because the netstack answer is the intuitive one and it
|
||||
is wrong. On the forward path a WireGuard/AWG endpoint never consults gVisor at
|
||||
all: `Endpoint.WritePackets` (`transport/wireguard/port.go:21-58`) reads the IP
|
||||
version and the destination address and hands the raw bytes to
|
||||
`wgDevice.InputPackets` — the protocol byte is never examined — and
|
||||
`returnDeviceWrapper.Write` (`:127-157`) offers every decrypted packet to
|
||||
`returnPath.ReturnPackets` before the stack sees it. WireGuard would carry ESP
|
||||
today if anything handed it one. What refuses is `ForwardDispatcher`: its parser
|
||||
sets `hasFlow` for TCP, UDP and ICMP echo alone (`flow_parse.go`,
|
||||
`parseTransport`, the echo identifier serving as the pseudo-port), and
|
||||
`createFlow` NATs through a port-shaped selector (`flow_dispatch.go:325`,
|
||||
`allocateSelector`) that ESP, AH and GRE do not have. So ESP/AH/GRE/IGMP/SCTP
|
||||
cannot enter the engine in ANY configuration and REMAIN on the D17 policy —
|
||||
or on the kernel egress of D26, which sidesteps the dispatcher entirely. The nft
|
||||
plane encodes the same boundary on purpose: it marks `icmp`/`ipv6-icmp` only,
|
||||
never `l4proto != { tcp, udp }`, because a marked ESP packet would enter the
|
||||
device and vanish — a black hole wearing a tunnel's name — instead of receiving
|
||||
the policy's honest verdict (`netplane/nft.go`, the prerouting L3 comment).
|
||||
|
||||
**What works and what does not, read off the upstream source.** ping v4/v6 —
|
||||
yes. Windows `tracert` — yes: the gVisor return path recognises
|
||||
`ICMPv4TimeExceeded` and `ICMPv4DstUnreachable` alongside EchoReply and NATs
|
||||
them back to the LAN client (`stack_gvisor_icmp.go:341+`, `returnPacket`). IPv6
|
||||
traceroute — intermediate hops stay invisible: the v6 branch of the same
|
||||
function accepts EchoReply only, so just the final destination answers. Several
|
||||
LAN clients behind the one tunnel address are already solved upstream:
|
||||
`ForwardDispatcher` NATs by echo identifier and rewrites the source to the
|
||||
outbound's port address (`flow_dispatch.go:325+`, `createFlow`; `icmpFlowKey`) —
|
||||
we wrote no NAT of our own.
|
||||
|
||||
**Which outbounds can carry it.** The contract is `adapter.FlowOutbound`
|
||||
(= `Outbound` + `tun.Port` + `PreMatchFlow`, `adapter/outbound.go`). In-tree
|
||||
implementors: the WireGuard/AWG endpoint (`protocol/wireguard`), `direct`
|
||||
(`protocol/direct`), `bridge` (`protocol/bridge`), `tailscale`
|
||||
(`protocol/tailscale`). Of those, the shaterd registry can construct only
|
||||
WireGuard/AWG and direct (`shater/registry/registry.go` — bridge and tailscale
|
||||
are not registered). Every proxy protocol — vless/vmess/trojan/shadowsocks/
|
||||
hysteria2/tuic/socks/http/shadowtls — is L4-only and cannot. Recorded as a known
|
||||
gap: `masque` is L3 by nature (CONNECT-IP; it builds a userspace gVisor stack
|
||||
per tunnel, `protocol/masque/outbound.go`) but implements no `tun.Port` and is
|
||||
not in the shater registry, so today it cannot carry the ingress. Wiring it up
|
||||
is possible future work, not a promise.
|
||||
|
||||
**ICMP to an L4-only outbound is DROPPED, and that took patching upstream files
|
||||
(the `lx:l3-honest-drop` delta — see `docs-lx/lx-changelog.md`).** In the gVisor
|
||||
stack the fallthrough verdict is a forgery: `ICMPForwarder.HandlePacket` answers
|
||||
the echo ITSELF (Echo→EchoReply + address swap) whenever the flow judgment comes
|
||||
back Accept (`stack_gvisor_icmp.go:120`), and upstream maps "no flow route" to
|
||||
exactly that Accept — so a ping routed to vless would read as tunnelled while
|
||||
the packet died on the router. Two small marked hunks make the truth observable:
|
||||
`route/route.go` wraps the whole pre-match walk — the walk itself became
|
||||
`preMatch`, and the exported `PreMatch` is now a FUNNEL that rewrites
|
||||
`PreMatchContinue` and `PreMatchBypass` to `PreMatchDrop` for `N.NetworkICMP` —
|
||||
and `adapter/router.go` (`JudgeFlow`, the `!isPort` branch) returns `ActionDrop`
|
||||
for ICMP where it fell through to `ActionAccept` — the second line of defense,
|
||||
because `FlowOutbound` and `tun.Port` are distinct interfaces and a drift
|
||||
between them must not quietly re-enable the forger. TCP/UDP verdicts are
|
||||
byte-identical; `route/prematch_icmp_lx_test.go` and
|
||||
`adapter/judgeflow_icmp_lx_test.go` pin both directions. The operator-facing
|
||||
text says the same out loud (`shater/apply/warnings.go`): proxy-routed addresses
|
||||
"cannot be pinged at all — deliberately".
|
||||
|
||||
> **Why a funnel and not an override inside the walk.** The first version of
|
||||
> this delta overrode the pre-declared `continueResult` inside `preMatchFlow`
|
||||
> and claimed to cover "every exit point of the function at once". It covered
|
||||
> every exit of THAT function; the walk above it has exits of its own that never
|
||||
> reach it — the `prepareMatchMetadata` error return (which arrived later, with
|
||||
> the shared-metadata refactor, upstream `b911fb078`), the sniff bail-outs, and
|
||||
> the `default:` arm of the rule-action switch, which catches every action
|
||||
> pre-match has no arm for (`hijack-dns`, `direct`, and whatever upstream adds
|
||||
> next). Each of those returned `PreMatchContinue`, i.e. `tun.ActionAccept`,
|
||||
> i.e. the forged reply. A guard on the single return value cannot be outgrown
|
||||
> by a new exit. `PreMatchBypass` joined the drop for the same reason: sing-tun
|
||||
> implements `ActionBypass` on the nfqueue plane only — the name appears nowhere
|
||||
> in `flow_dispatch.go` or `stack_gvisor_icmp.go` — so on the TUN path it lands
|
||||
> in the same `default:` arm as Accept and forges too. There is no honest bypass
|
||||
> for a packet that is already inside the engine's TUN.
|
||||
|
||||
**The device MTU is NOT a tunnel budget, and pretending it was manufactured
|
||||
forged replies.** `l3-in` is created with MTU **65535**, not the tunnel's 1420,
|
||||
and the maximum is the whole argument. This MTU governs exactly one thing:
|
||||
whether the KERNEL splits a packet on its way INTO the device. What the engine
|
||||
then puts into the tunnel is sized separately and correctly, against the
|
||||
OUTBOUND's MTU — `ForwardDispatcher.forwardToPort` (`flow_dispatch.go:445-481`)
|
||||
measures every forwarded packet against `Port.PortMTU()` and either fragments to
|
||||
it (no DF, `fragmentIPv4Packet`) or answers a well-formed `fragmentation needed`
|
||||
quoting it (DF, `buildFragmentationNeeded`, source = the far host, so PMTU
|
||||
discovery works end to end). That machinery was always there; it was simply
|
||||
never handed a whole packet.
|
||||
|
||||
At 1420 it wasn't. Anything above 1392 bytes of payload was fragmented by the
|
||||
kernel at this device, and a fragment is the one thing sing-tun will not judge:
|
||||
`Dispatch` (`flow_dispatch.go:176-177`) returns on `parsed.fragment` BEFORE
|
||||
calling `JudgeFlow` at all. The fragments fell through to the gVisor stack —
|
||||
promiscuous and spoofing (`stack_gvisor.go:219-223`) — which reassembled them
|
||||
and handed the echo to `ICMPForwarder.HandlePacket` (`stack_gvisor_icmp.go:105+`),
|
||||
whose `installFlow` (`:233-244`) writes to the port UNMODIFIED and therefore
|
||||
demands a port address that is valid **and UNSPECIFIED**. `direct` qualifies
|
||||
(`IPv4Unspecified()`); a WireGuard/AWG endpoint reports its concrete interface
|
||||
address (`transport/wireguard/port.go:13`) and does not. So it declined, and
|
||||
`HandlePacket` fell past the switch and FORGED the reply: `SetType(EchoReply)` +
|
||||
address swap. Net effect on the operator's bench: `ping -s 1392` honest,
|
||||
`ping -s 1393` a lie told by the router — and the lie was, of course, only for
|
||||
the outbounds this feature exists for. (Upstream applies the very same
|
||||
unspecified test and answers it honestly in the cloudflared ICMP handler,
|
||||
`protocol/cloudflare/inbound.go:163-167`: it drops. Only the TUN path forges.)
|
||||
|
||||
65535 rather than "big enough": no IP datagram can exceed it, so the kernel
|
||||
CANNOT fragment at this device, for any packet, ever. Any smaller value leaves
|
||||
a band of sizes open and re-opens the class. It is also sing-box's own default
|
||||
TUN MTU on Linux. Pinned by `TestL3TunnelMTULeavesNothingForTheKernelToFragment`
|
||||
and `TestL3TunnelMTUIsNotATunnelBudget` (`generate/l3mtu_test.go`), and — the
|
||||
assertion that matters — by the integration test reading the MTU back off the
|
||||
real kernel device, since a kernel that clamped it would restore the forgery
|
||||
without changing a generated byte.
|
||||
|
||||
Memory was MEASURED, not reasoned about: three paired runs of
|
||||
`TestIntegrationL3TunInboundStarts` under `-test.memprofilerate=1` (exact
|
||||
accounting, not sampled) allocate 5.41 / 5.48 / 5.47 MB at 65535 against
|
||||
5.76 / 5.46 / 5.70 MB at 1420, and a `-diff_base` profile attributes every
|
||||
difference to netlink interface enumeration. Nothing in the read path scales
|
||||
with the MTU: gVisor reads through `fdbased.BufConfig`, which sing-tun's `init`
|
||||
pins to a single 65535-byte view regardless of MTU, and `fdbased` keeps `mtu`
|
||||
only to return it from `MTU()`. Two adjacent facts, recorded because both are
|
||||
easy to derive wrongly: (a) `protocol/tun` computes
|
||||
`enableGSO = stack == gvisor && mtu < 49152`, so this MTU turns GSO off there —
|
||||
and then `StartStateStart` turns it back ON unconditionally because an
|
||||
`adapter.FlowOutbound` exists in the config, so the ~1.98 MB of TCP/UDP GRO
|
||||
scaffolding is present at BOTH MTUs and is priced by the flow-capable outbound,
|
||||
not by this number; (b) the `mtu_fix` on the `shater_l3` fw4 zone is now inert —
|
||||
only ICMP is ever marked into the device — and its uci-defaults comment still
|
||||
says "the tunnel MTU is 1420".
|
||||
|
||||
**What is still NOT covered, said plainly.**
|
||||
|
||||
1. **A big ping does not start WORKING — it starts FAILING HONESTLY.** Upstream's
|
||||
ICMP NAT is unfragmented-only in BOTH directions: `classifyReturn`
|
||||
(`flow_dispatch.go:703-710`) returns `returnPass` on `parsed.fragment` exactly
|
||||
as the forward path does. So a non-DF `ping -s 2000` now genuinely leaves the
|
||||
router (fragmented to the tunnel MTU by `forwardToPort`), the far host really
|
||||
answers, and the reply — fragmented by the peer to fit the tunnel — is not
|
||||
NAT'd back to the LAN client. The operator sees a timeout. That is the
|
||||
feature's promise ("travels or fails honestly"), not a capability claim.
|
||||
Carrying oversized ICMP end to end would need reassembly upstream does not
|
||||
have; it is not planned.
|
||||
2. **A client that puts fragments on the wire ITSELF.** The device MTU cannot
|
||||
un-fragment what already arrived fragmented, so such packets still reach the
|
||||
gVisor stack, still get reassembled there, and still receive a forged reply
|
||||
when the outbound is WireGuard/AWG. This is the residue the planned
|
||||
`ip frag-off & 0x3fff != 0` prerouting carve-out (`netplane/nft.go`) is for.
|
||||
**Whoever writes that rule must first check whether it can ever match:** fw4's
|
||||
ruleset uses conntrack, conntrack pulls in `nf_defrag_ipv4`/`nf_defrag_ipv6`,
|
||||
and defrag REASSEMBLES in PREROUTING before our marking rules run. Where
|
||||
defrag is active the case does not arise (the MTU covers it) and the rule is
|
||||
dead; where it is not, the rule is the only cover. Verify on the bench with
|
||||
`nft list ruleset | grep -c ct` and a fragment counter, do not assume.
|
||||
3. **The DF path changed hands and is untested on hardware.** It used to be the
|
||||
kernel that answered `fragmentation needed` (from the router's LAN address,
|
||||
MTU 1420); it is now the engine (from the far host's address, quoting
|
||||
`Port.PortMTU()`). Both are correct PMTUD; only the first has ever run on a
|
||||
real router.
|
||||
**fw4 has to be told about the device, and `list device` is the only spelling
|
||||
that works.** nftables runs EVERY table on every packet and a drop in any one of
|
||||
them wins — an accept in `inet shater` cannot override fw4, and fw4 WILL reject
|
||||
this forward: netifd never learns about a device the daemon creates at runtime,
|
||||
so `shater-l3` belongs to no zone and falls into fw4's zone-less defaults. Hence
|
||||
a real fw4 zone `shater_l3` + a lan→shater_l3 forwarding, seeded idempotently
|
||||
(NAMED sections) and unconditionally in uci-defaults
|
||||
(`openwrt/shater-core/files/etc/uci-defaults/30_shater-core`, `seed_l3_zone`),
|
||||
with `mtu_fix` set. That `mtu_fix` is now inert and should be read as such: it
|
||||
clamps forwarded TCP MSS to the route MTU, the device MTU is 65535, and nothing
|
||||
but ICMP is ever marked into this device — the uci-defaults comment still says
|
||||
"the tunnel MTU is 1420" and is stale. The device is attached
|
||||
via `list device`, deliberately NOT `list network`: fw4 resolves a zone's
|
||||
networks through netifd, which yields an EMPTY device set for a runtime-created
|
||||
TUN (a proto-none stub would have to be brought UP to contribute an l3_device,
|
||||
and nothing ever brings it up), while `list device` compiles to a plain
|
||||
iifname/oifname string match — valid before the TUN exists, matching from the
|
||||
moment shaterd creates it. `kmod-tun` joined DEPENDS so a slimmed image cannot
|
||||
lose `/dev/net/tun` (`openwrt/shater-core/Makefile`). Our own forward chain
|
||||
accepts both TUN legs ahead of the fail-closed drops — accepts that speak for
|
||||
OUR table only (`netplane/nft.go`, forward chain step 4).
|
||||
|
||||
**What the policy still owns, and the one combination that now warns.** With the
|
||||
ingress on, the mark is stamped in prerouting and the ROUTING decision carries
|
||||
echo into the TUN before the forward chain — where the policy's verdicts live —
|
||||
is ever consulted; that holds under every `untunnelable` value. The policy
|
||||
therefore governs exactly two things: the never-markable protocols above, and
|
||||
the fallback when the L3 rule/route did not come up (engine down, partial apply)
|
||||
— `block` turns that failure into an honest loss, `direct` into a silent leak
|
||||
with the real address. That is why `l3_tunnel` + `untunnelable=direct` draws a
|
||||
validation warning naming the safe choice (`model/validate.go`), why every
|
||||
rule/route failure surfaces as a named panel warning rather than an abort
|
||||
(`addL3Routing`), and why the D17 HOLDING plane never marks: the TUN is created
|
||||
BY the engine, and the holding plane exists precisely because the engine is not
|
||||
running — marking would dead-end ping in a device that does not exist
|
||||
(`netplane/nft.go`, hold comment).
|
||||
|
||||
- **Rejected: `auto_route` / letting the engine own the routing.** It rewrites
|
||||
the main table and intercepts the router's own WAN/DNS/underlay traffic; the
|
||||
blast radius of a toggle meant for LAN ping would be the whole router.
|
||||
- **Rejected: marking all `l4proto != { tcp, udp }` into the TUN.** ESP/AH/GRE/
|
||||
IGMP/SCTP cannot be parsed into flows upstream; they would vanish inside the
|
||||
device. A drop with a name (the policy's) beats a silent black hole.
|
||||
- **Rejected: keeping upstream's accept-and-forge for unroutable ICMP.** A ping
|
||||
that "works" without leaving the router is the inverted lie this project keeps
|
||||
deleting (D17's fiction purge, D23's dead WireGuard, D24's obedient-client
|
||||
leak).
|
||||
|
||||
**Proven, and not proven, said plainly.** The cold start is PROVEN, not assumed:
|
||||
`TestIntegrationL3TunInboundStarts`
|
||||
(`shater/generate/l3_integration_linux_test.go`, run as root with NET_ADMIN and
|
||||
`/dev/net/tun`, PASS) drives an `l3_tunnel=1` config through the SLIM registry
|
||||
(`registry.Context`, not upstream's `include.Context`) under the shipped router
|
||||
tag set: `box.New` + `Start` accept it, the kernel really ends up with the
|
||||
`shater-l3` device at the contract MTU 65535 — the assertion the value exists
|
||||
for, since a kernel that clamped it would silently restore the forged-reply
|
||||
band — and Close removes it; precisely
|
||||
the "built with X, verified with Y" gap class D23 exists for (a lost
|
||||
`tun.RegisterInbound` or a trimmed `with_gvisor` changes no generated byte and
|
||||
would otherwise surface only on the operator's router). Each layer contract is
|
||||
pinned besides (`generate` `TestL3Tunnel*`, `netplane` `TestL3Ingress*`,
|
||||
`route/prematch_icmp_lx_test.go`). Exactly two things remain UNVERIFIED:
|
||||
(a) the end-to-end path on live hardware — LAN client → prerouting mark →
|
||||
ip rule → TUN → WireGuard peer → reply back to the client — has not been
|
||||
exercised on a real router; (b) the steady-state memory cost of the second
|
||||
gVisor netstack (the `l3-in` TUN beside the WireGuard endpoint's own) is
|
||||
unmeasured on the target hardware. An indicative figure exists and is only
|
||||
that: on x86_64 in a container, idle and carrying no flows, peak RSS of a
|
||||
process that brought the same engine up went from ~26.0-26.8 MB without
|
||||
`l3_tunnel` to ~28.3-28.7 MB with it over three paired runs — about +2.2 MB.
|
||||
That was measured on a throwaway harness, not on aarch64, not under load, and
|
||||
with an empty ICMP NAT table, so it bounds nothing on the router. Neither
|
||||
item is folded into any claim above.
|
||||
|
||||
Consequence: a ping from the LAN either genuinely travels through the tunnel
|
||||
(WireGuard/AWG, direct) or fails honestly, at every size the router itself can
|
||||
put into the device — and a router that never opts in renders the pre-feature
|
||||
plane byte-for-byte (`TestL3IngressOptIn` pins the off-state render). Read
|
||||
"fails honestly" strictly: above the tunnel MTU a non-DF ping now leaves the
|
||||
router for real and then times out, because upstream's ICMP NAT does not carry
|
||||
fragments back either. The one qualifier left is item 2 above — a client that
|
||||
puts fragments on the wire ITSELF, on a router where conntrack defrag is not
|
||||
reassembling them first. This paragraph has been overclaimed twice already;
|
||||
extend it only against a bench result, never against a reading.
|
||||
|
||||
## D26 — What the engine cannot carry, the kernel carries: `untunnelable_egress`
|
||||
Decided 2026-07-26. D25 ended with ESP/AH/GRE/IGMP/SCTP still owned by the D17
|
||||
policy — that is, with a choice between dropping them and leaking them out the
|
||||
WAN, never a data plane. This decision gives them one, and deliberately NOT
|
||||
ours: **`globals.untunnelable_egress` (default empty;
|
||||
`model.Globals.UntunnelableEgress`) names an existing egress of type
|
||||
interface/tunnel, and LAN traffic that is neither TCP nor UDP is stamped in
|
||||
prerouting with that egress's own mark, so the KERNEL routes it out that
|
||||
egress's device with the kernel's own NAT.** No proxy, no engine, no userspace
|
||||
stack ever touches the packet — which is exactly why every protocol works.
|
||||
|
||||
**Where the engine's boundary actually is — recorded so nobody digs for it
|
||||
twice.** It is NOT the gVisor stack, and it is not WireGuard: on the forward
|
||||
path the WG/AWG endpoint never consults gVisor at all. `Endpoint.WritePackets`
|
||||
(`transport/wireguard/port.go:21-58`) takes the raw IP packet bytes, reads
|
||||
exactly the IP version and the destination address, and hands
|
||||
`device.InputPacketRef`s to `wgDevice.InputPackets` — the protocol byte is
|
||||
never read; on the way back (`port.go:127-157`) `returnDeviceWrapper.Write`
|
||||
offers every decrypted packet to `returnPath.ReturnPackets` first and only the
|
||||
unconsumed remainder falls through to the gVisor device. gVisor serves
|
||||
`DialContext`/`ListenPacket` — traffic the ENGINE originates — while forwarded
|
||||
traffic bypasses the stack in both directions, indifferent to protocol. The
|
||||
real ceiling sits one step earlier, in sing-tun's `ForwardDispatcher`:
|
||||
`parseTransport` (`flow_parse.go:106-153`) sets `hasFlow` for exactly TCP, UDP,
|
||||
ICMPv4 Echo/EchoReply and ICMPv6 EchoRequest/EchoReply — a packet of any other
|
||||
protocol is never dispatched as a flow — and `createFlow`
|
||||
(`flow_dispatch.go:325`) builds its NAT through
|
||||
`allocateSelector(packet.protocol, …, packet.source.Port())` (line 355), which
|
||||
needs a port-like selector that ESP/AH/GRE simply do not have (SCTP has ports,
|
||||
but the parser above never grants it a flow either). Tailscale documents the
|
||||
same frontier for its own userspace mode — "Any IP protocol other than TCP or
|
||||
UDP (such as SCTP) is not supported in userspace mode… All IP protocols are
|
||||
supported" in kernel mode
|
||||
(https://tailscale.com/docs/reference/kernel-vs-userspace-routers) — useful as
|
||||
external corroboration of where userspace data planes generally end, though OUR
|
||||
boundary is the dispatcher, not the stack. The kernel egress was therefore
|
||||
chosen not because userspace "cannot" in principle, but because the kernel
|
||||
delivers all protocols with zero new code on the hot path.
|
||||
|
||||
**The mechanism already existed; the feature is one binding and one marking
|
||||
step.** `addEgressRouting` (`netplane/apply.go`) has always installed, for
|
||||
every interface/tunnel egress, an `ip rule fwmark <EgressMark> lookup
|
||||
<EgressTable>` plus a `default dev <device>` route in that table — per-rule
|
||||
egress selection rides on it. The only missing piece was that nothing ever
|
||||
marked non-TCP/UDP traffic: `untunnelable=direct` merely ACCEPTED it in the
|
||||
forward chain, so it left over the main table, i.e. the WAN.
|
||||
`UntunnelableEgressBinding` (`netplane/nft.go`) resolves the option to the
|
||||
egress's index, its OWN mark and its OWN device — deliberately no third
|
||||
mark/table pair to keep coherent — and the prerouting chain stamps that mark on
|
||||
the untunnelable protocols. A name that does not resolve to an interface/tunnel
|
||||
egress with a device renders nothing and is reported: the D17 policy stays in
|
||||
sole charge, which is the fail-closed reading of a typo.
|
||||
|
||||
**Why `l4proto != { tcp, udp }` is safe here when D25 banned it.** D25 rejected
|
||||
the broad filter because the receiving side was the `ForwardDispatcher`, which
|
||||
classifies nothing beyond TCP/UDP/ICMP echo — a marked ESP packet would enter
|
||||
the TUN and vanish, a black hole wearing a tunnel's name. Here the receiving
|
||||
side is the kernel, which forwards ANY IP protocol and NATs what it has
|
||||
machinery for: SCTP carries ports and NATs like TCP/UDP; GRE is NATed only
|
||||
through the PPTP helper keyed on the call-id — the kernel's own comment calls
|
||||
GRE "generally not very suited for NAT, as it has no protocol-specific part as
|
||||
port numbers" (`net/netfilter/nf_conntrack_proto_gre.c`); ESP/AH pass as plain
|
||||
routed IP. Nothing on this path can silently swallow a protocol it does not
|
||||
understand, which was the entire objection.
|
||||
|
||||
**Order against D25: the L3 ingress claims ICMP first.** With `l3_tunnel` on,
|
||||
LAN ICMP is marked into the engine's TUN before the egress carrier is consulted
|
||||
— the engine path routes ping by the operator's rules, which a kernel egress
|
||||
cannot do — and only the remaining protocols go to the egress. With `l3_tunnel`
|
||||
off, ICMP goes to the egress with everything else. In both shapes marked
|
||||
traffic is settled by ROUTING before the forward chain speaks, so the D17
|
||||
policy now governs exactly the failure case — the rule or route that did not
|
||||
come up — the same division D25 already established for the L3 mark.
|
||||
|
||||
**What the feature refuses to promise — and the operator text refuses with it
|
||||
(`shater/apply/warnings.go`, the egress-carrier note).** (a) It is not a tunnel
|
||||
per se: the option accepts any interface/tunnel egress, and on the target
|
||||
routers a WireGuard device is the exception (`kmod-wireguard` is usually
|
||||
absent) while a second WAN is routine. Through a WireGuard egress this
|
||||
genuinely is a tunnel; through a second WAN it is simply another uplink, and
|
||||
the destination sees that uplink's real address. No text, comment or doc line
|
||||
may call it a tunnel unconditionally. (b) It does not revive IPTV: IGMP is
|
||||
LAN-side multicast group management, WireGuard is L3 point-to-point and carries
|
||||
no multicast, and multicast never crossed this router under any setting —
|
||||
routing IGMP out an egress restores nothing, and no wording may hint otherwise.
|
||||
(c) IPsec through NAT-T never needed it: RFC 3948 encapsulates ESP in UDP/4500,
|
||||
so a modern IPsec client behind NAT is ordinary UDP that already follows the
|
||||
routing rules; the raw-ESP case this feature carries is the no-NAT-T remainder.
|
||||
|
||||
- **Rejected: teaching the engine these protocols.** Extending `parseTransport`
|
||||
and the selector NAT upstream would be new hot-path code in an
|
||||
actively-maintained adversarial area, for protocols the kernel already
|
||||
forwards for free — and for ESP/AH/GRE there is no port-like selector to NAT
|
||||
by in the first place.
|
||||
- **Rejected 2026-07-26: carrying them through the userspace AWG endpoint
|
||||
site-to-site, with no NAT at all.** This is the alternative the "no port-like
|
||||
selector" line above does NOT dispose of, and it is written down because the
|
||||
obvious reading of that line — "impossible" — is wrong and would be
|
||||
re-derived. The endpoint is already protocol-blind in both directions
|
||||
(`transport/wireguard/port.go:21-58`, `:127-157`), so an ESP packet could be
|
||||
forwarded UNTOUCHED, keeping the LAN client's own source address, and the
|
||||
reply would come back addressed to that client and need only be written to the
|
||||
TUN. No selector, no NAT, every protocol. It needs two things we declined to
|
||||
take on: lx-owned code in the forward hot path, bypassing `ForwardDispatcher`
|
||||
on both legs — precisely the surface CONSTITUTION §2 exists to keep small on
|
||||
an actively-maintained upstream — and a SERVER-side prerequisite (our LAN
|
||||
prefix in the peer's `AllowedIPs`, plus a route back), which turns a router
|
||||
option into a deployment contract. The kernel egress above buys the same
|
||||
protocols with zero hot-path code, so this stays a design on file, not a gap.
|
||||
- **Rejected: a dedicated mark/table pair for the carrier.** `addEgressRouting`
|
||||
already binds `EgressMark`/`EgressTable` to the device; a third pair would be
|
||||
a second copy of the same route that could drift from the first.
|
||||
|
||||
**Not verified, said plainly.** The end-to-end path — LAN client → prerouting
|
||||
mark → ip rule → egress device → far end and back — has not been exercised with
|
||||
real ESP or GRE on live hardware. Nothing above claims it has.
|
||||
|
||||
Consequence: raw IPsec, PPTP/GRE, SCTP — and ICMP when the L3 ingress is off —
|
||||
leave through an egress the operator explicitly named, under kernel routing and
|
||||
kernel NAT, instead of being dropped or silently leaking out the WAN; and with
|
||||
the option empty (the default) the plane renders byte-for-byte as before, with
|
||||
the D17 policy in sole charge.
|
||||
|
||||
@@ -12,6 +12,28 @@ usable release, **[T1]** next, **[T2]** later. Phases refer to `ROADMAP.md`.
|
||||
## Transparent proxying & routing
|
||||
- **[MVP]** TPROXY transparent proxy for multiple LAN interfaces (TCP + UDP), SNI/
|
||||
Host/QUIC sniffing.
|
||||
- **[MVP]** **L3 ingress for ICMP** (`globals.l3_tunnel`, opt-in, default off):
|
||||
LAN ping travels THROUGH the tunnel instead of being dropped or answered by a
|
||||
forged local reply. The engine opens a dedicated TUN (`shater-l3`, gVisor
|
||||
stack, `auto_route` off); nft marks LAN icmp/icmpv6 only and a scoped
|
||||
`ip rule` routes it in — the TPROXY plane and the main routing table stay
|
||||
untouched (D25). Carried only by L3-capable egresses (WireGuard/AmneziaWG,
|
||||
direct); ICMP routed to vless/vmess/… is honestly dropped, never faked.
|
||||
Ceiling is upstream sing-tun's: ICMP echo only — Windows tracert works, IPv6
|
||||
traceroute shows just the destination; ESP/AH/GRE/IGMP stay with the
|
||||
`untunnelable` policy (D17) unless `untunnelable_egress` carries them (D26).
|
||||
- **[MVP]** **Kernel egress for untunnelable protocols**
|
||||
(`globals.untunnelable_egress`, opt-in, default empty): names an existing
|
||||
interface/tunnel egress, and IPsec (ESP/AH), PPTP/GRE, SCTP — everything that
|
||||
is neither TCP nor UDP, plus ICMP when the L3 ingress is off — is routed out
|
||||
that egress's device by the KERNEL with kernel NAT, reusing the egress's own
|
||||
fwmark/table from `addEgressRouting`; the proxy never sees a byte, which is
|
||||
why every protocol works (D26). What that buys depends on the device: a
|
||||
WireGuard interface really is a tunnel, a second WAN is just another uplink
|
||||
whose real address the destination sees. It does not revive multicast IPTV,
|
||||
and UDP-based VPNs (WireGuard, OpenVPN-UDP, IPsec NAT-T) never needed it —
|
||||
they follow the routing rules as before. The `untunnelable` policy (D17)
|
||||
keeps only the failure case: a route that did not come up.
|
||||
- **[MVP]** First-match routing rules by source (IP/CIDR/MAC/interface/zone),
|
||||
destination, port, proto → target (outbound/selector/chain/direct/block) + egress.
|
||||
A rule names its **destination through a rule-set only** — a reusable named list
|
||||
|
||||
@@ -40,6 +40,10 @@ define Package/shater-core
|
||||
# shaterd : the daemon our init supervises (`shaterd run`)
|
||||
# kmod-nft-tproxy : kernel TPROXY (shaterd emits the `inet shater` rules)
|
||||
# kmod-nft-socket : socket match used by the tproxy divert chain
|
||||
# kmod-tun : /dev/net/tun — the daemon opens the `shater-l3` TUN
|
||||
# for L3 ingress (globals.l3_tunnel); usually built-in
|
||||
# on stock images, but a slimmed image without it would
|
||||
# make the option fail with a cryptic open() error.
|
||||
# ip-full : `ip rule`/`ip route`/rt_tables for policy routing
|
||||
# nftables-json : shaterd shells out to `nft`, and netplane/stats.go
|
||||
# parses `nft -j list ...` — the JSON output only exists
|
||||
@@ -49,7 +53,7 @@ define Package/shater-core
|
||||
# ca-bundle : the daemon is CGO_ENABLED=0, so crypto/x509 has no
|
||||
# host cert fallback — without /etc/ssl/certs every
|
||||
# HTTPS subscription / .srs ruleset fetch fails.
|
||||
DEPENDS:=+shaterd +kmod-nft-tproxy +kmod-nft-socket +ip-full +nftables-json +ca-bundle
|
||||
DEPENDS:=+shaterd +kmod-nft-tproxy +kmod-nft-socket +kmod-tun +ip-full +nftables-json +ca-bundle
|
||||
PKGARCH:=all
|
||||
endef
|
||||
|
||||
|
||||
@@ -33,24 +33,30 @@
|
||||
# be running. `start` raises ACTIVE_FLAG, `stop` clears it; hotplug/cron
|
||||
# reconcile ONLY while the flag is up, so an admin `stop` STICKS — no
|
||||
# background actor may resurrect interception behind a stopped daemon.
|
||||
# * BEING REPLACED IS NOT BEING SWITCHED OFF. `restart`, `reload` (which is
|
||||
# stop+start, i.e. every LuCI Save & Apply) and every package upgrade all run
|
||||
# through `stop`, and the daemon's SIGTERM teardown removes the fail-closed
|
||||
# table unconditionally — it does not consult kill_switch at all. Between that
|
||||
# teardown and the successor's first apply the init GUARANTEES a gap: it waits
|
||||
# for the old process to exit (shater_wait_stopped), then runs `shaterd
|
||||
# migrate`, then starts a daemon that still has to build an engine. So a
|
||||
# restart is announced with RESTART_FLAG, which tells the outgoing daemon to
|
||||
# leave the fail-closed holding plane behind instead of bare routing. A real
|
||||
# `stop` raises no flag and therefore still means what it says.
|
||||
# * BEING REPLACED IS NOT BEING SWITCHED OFF. `restart` and `reload` (which is
|
||||
# stop+start, i.e. every LuCI Save & Apply) both run through `stop`, and the
|
||||
# daemon's SIGTERM teardown removes the fail-closed table unconditionally — it
|
||||
# does not consult kill_switch at all. Between that teardown and the
|
||||
# successor's first apply the init GUARANTEES a gap: it waits for the old
|
||||
# process to exit (shater_wait_stopped), then runs `shaterd migrate`, then
|
||||
# starts a daemon that still has to build an engine. So a restart is announced
|
||||
# with RESTART_FLAG, which tells the outgoing daemon to leave the fail-closed
|
||||
# holding plane STANDING — apply.TeardownExiting swaps it in with one nft
|
||||
# transaction and then skips the delete, so the table is never absent, not even
|
||||
# for the 80-90 ms the old arm-after-teardown order measured. A real `stop`
|
||||
# raises no flag and therefore still means what it says.
|
||||
# (A package UPGRADE does not come through here at all on apk v3: shater-core's
|
||||
# script table is post-install / pre-deinstall / post-upgrade, with no
|
||||
# pre-upgrade, so default_prerm — and its `stop` — runs only on REMOVAL.)
|
||||
# * The FAIL-CLOSED PLANE MUST ALSO EXIST BEFORE THIS SCRIPT DOES. START=99 is
|
||||
# after fw4 (19) and netifd (20), so at every boot the LAN forwards to the WAN
|
||||
# in the clear for as long as it takes procd to decompress the daemon off
|
||||
# flash and get an engine up. /etc/init.d/shater-armor (START=21) loads
|
||||
# BOOT_ARMOR — a copy of the holding plane the daemon persists on every apply
|
||||
# — to close that window. This script owns the DISARM half: a deliberate
|
||||
# `stop`, or a missing daemon binary, removes the armor so it cannot outlive
|
||||
# the product it protects.
|
||||
# — to close that window. This script owns the DISARM half, and it owns it
|
||||
# with a CLOSED LIST: an operator's `stop`, or a removal, and nothing else.
|
||||
# Powering the box down must not — `shutdown` reaches stop_service too, and it
|
||||
# is not a person switching the product off (see shater_stop_disarms).
|
||||
# * The engine must never be permanently abandoned while interception stands:
|
||||
# respawn retries are infinite (procd never gives up); a sustained-dead
|
||||
# daemon is additionally escalated by the shater-cron watchdog.
|
||||
@@ -84,20 +90,150 @@ STOP_WAIT_SECS=40
|
||||
|
||||
# WHICH ACTION rc.common was invoked with, frozen at source time.
|
||||
#
|
||||
# rc.common sets `action=${2:-help}` before it sources this file, and every action
|
||||
# then runs as a function in THAT SAME shell — so `stop_service` can see whether it
|
||||
# was reached by `stop` or as the first half of `restart`/`reload`. That is the one
|
||||
# distinction procd itself does not expose (`restart` is literally `stop; start`,
|
||||
# and stop_service is called identically by both).
|
||||
# rc.common does, in this order:
|
||||
# initscript=$1; action=${2:-help}; shift 2; ...; . "$initscript"; $action "$@"
|
||||
# so `action` is ALREADY assigned when this file is sourced, and every action then
|
||||
# runs as a function in THAT SAME shell. MEASURED on the target (ImmortalWrt
|
||||
# 25.12.1 r37978) with a throwaway probe init script, not read off documentation:
|
||||
#
|
||||
# /etc/init.d/X restart -> stop_service action=[restart], start_service [restart]
|
||||
# /etc/init.d/X stop -> stop_service action=[stop]
|
||||
# /etc/init.d/X reload -> reload_service action=[reload]
|
||||
# `reboot` -> stop_service action=[SHUTDOWN] <-- see below
|
||||
# the boot after it -> start_service action=[boot]
|
||||
#
|
||||
# A previous probe reported this variable EMPTY and the emptiness was written up as
|
||||
# the defect. It was the probe: `sh -x /etc/init.d/shater restart` bypasses the
|
||||
# `#!/bin/sh /etc/rc.common` shebang, so rc.common never runs, never assigns
|
||||
# `action`, and the variable reads empty no matter what this file does.
|
||||
#
|
||||
# Frozen into our own variable because `action` is a short, generic name that other
|
||||
# framework helpers also use as a local; a snapshot taken before any function runs
|
||||
# cannot be shadowed later. An EMPTY or unexpected value degrades to "real stop",
|
||||
# which is the pre-existing behaviour and the safe direction to be wrong in: it
|
||||
# costs a plaintext window on restart, where the other default would leave a
|
||||
# deliberately stopped router blocked.
|
||||
# cannot be shadowed later.
|
||||
SHATER_RC_ACTION="$action"
|
||||
|
||||
# --- what an action MEANS --------------------------------------------------
|
||||
#
|
||||
# THE BUG THESE TWO PREDICATES REPLACE (v0.2.17, measured on the live router).
|
||||
# The old stop_service was `case $action in restart|reload) keep;; *) DISARM;; esac`
|
||||
# — an open default that swept up every action nobody had enumerated. `reboot` is
|
||||
# one of them: procd runs the K-links with the action `shutdown`, so the shutdown
|
||||
# path deleted the arm token on the way down and the next boot had nothing to load.
|
||||
# The mechanism destroyed itself at exactly the moment it exists for. Instrument
|
||||
# reading from the router, one minute apart across a reboot:
|
||||
#
|
||||
# 13:28 /etc/shater/boot.nft present
|
||||
# ---- reboot (stop_service action=[shutdown] -> old `*` branch -> rm)
|
||||
# 18s at_S22: NO_TABLE armor_file=NO_FILE
|
||||
#
|
||||
# So both lists below are POSITIVE and CLOSED. An action nobody thought about —
|
||||
# `shutdown` above all, but also whatever a future procd invents — falls through
|
||||
# both and changes nothing. The default now fails in the recoverable direction: at
|
||||
# worst a boot arms when it need not have, which costs the second before the daemon
|
||||
# applies and is still gated by shater-armor's own four state refusals. The old
|
||||
# default failed in the direction of the plaintext window the feature was built to
|
||||
# close.
|
||||
#
|
||||
# They are predicates rather than an inline `case` so the test gate can execute the
|
||||
# real thing: it sources THIS FILE in /bin/sh and calls them with every action procd
|
||||
# actually uses (shater/cmd/shaterd/initscript_test.go). A comment claiming
|
||||
# `shutdown` is handled is what shipped last time.
|
||||
|
||||
# True only for the ONE action that means "the operator switched the product off".
|
||||
# Deliberately not `shutdown`: powering a router down is not turning a feature off.
|
||||
#
|
||||
# NOT sufficient on its own — see shater_stop_disarms. `stop` is also how the
|
||||
# package manager's plumbing reaches us, and a package manager is not a person.
|
||||
shater_action_disarms() {
|
||||
case "$1" in
|
||||
stop) return 0 ;;
|
||||
*) return 1 ;;
|
||||
esac
|
||||
}
|
||||
|
||||
# Is a package manager in the middle of a transaction RIGHT NOW?
|
||||
#
|
||||
# This is a state, read at the moment the decision is made, exactly like
|
||||
# shater-armor's four refusals — not a record of an event. The same question is
|
||||
# already asked (for the same reason: prerm/postinst plumbing is not a user
|
||||
# action) by the detached bring-up in /etc/uci-defaults/30_shater-core.
|
||||
shater_pkg_transaction() {
|
||||
pidof apk >/dev/null 2>&1 && return 0
|
||||
pidof opkg >/dev/null 2>&1 && return 0
|
||||
return 1
|
||||
}
|
||||
|
||||
# Is the main service still enabled at boot? Same glob, and for the same reason,
|
||||
# as shater-armor's own check: `/etc/init.d/shater enabled` would source procd.sh
|
||||
# and take a blocking flock, which is not something to do from inside a package
|
||||
# manager's transaction.
|
||||
shater_rc_enabled() {
|
||||
local f
|
||||
for f in /etc/rc.d/S[0-9][0-9]shater; do
|
||||
[ -e "$f" ] && return 0
|
||||
done
|
||||
return 1
|
||||
}
|
||||
|
||||
# THE ACTUAL DISARM DECISION.
|
||||
# $1 = action
|
||||
# $2 = 1 when a package transaction is in flight
|
||||
# $3 = 1 when the service is still enabled in rc.d
|
||||
# All three are passed in rather than read inside, so the gate can drive every
|
||||
# combination without a package manager or an /etc/rc.d.
|
||||
#
|
||||
# WHY IT IS NOT JUST THE ACTION. base-files' default_prerm runs, in this order:
|
||||
#
|
||||
# if [ "$PKG_UPGRADE" != "1" ]; then "$i" disable; fi
|
||||
# "$i" stop
|
||||
#
|
||||
# so a package manager reaches stop_service wearing the operator's clothes. Two
|
||||
# different intentions arrive as the same action, and the difference between them
|
||||
# is readable at the moment of the decision:
|
||||
#
|
||||
# REMOVAL — prerm has ALREADY run `disable`, so S99shater is gone. The product
|
||||
# is going away; the armor goes with it. (It is belt-and-braces even
|
||||
# so: shater-armor refuses to arm without that symlink, and the whole
|
||||
# init script is about to be deleted anyway.)
|
||||
# REPLACED — the service is still enabled, so something intends to bring it
|
||||
# back. That is not an operator switching anything off, and deleting
|
||||
# the armor here would leave the next boot unprotected. "The next
|
||||
# apply will rewrite it" is not an answer: the armor exists precisely
|
||||
# to cover a reboot, and a reboot between an update and the first
|
||||
# apply is how this product is deployed.
|
||||
#
|
||||
# MEASURED, because the paragraph above is about a path I got wrong once already.
|
||||
# On THIS target (apk-tools 3.0.5, ImmortalWrt 25.12.1) shater-core's script table
|
||||
# is post-install / pre-deinstall / post-upgrade, with NO pre-upgrade — so an apk
|
||||
# UPGRADE never executes default_prerm and never calls `stop` at all. Verified with
|
||||
# a real `apk fix --reinstall shater-core` while sampling the armor file: 245 625
|
||||
# samples, zero disappearances, even with this guard mutated off. The upgrade half
|
||||
# of this predicate is therefore defence-in-depth for a shape that is one
|
||||
# `pre-upgrade` script (or a returning opkg lane) away, NOT a fix for an observed
|
||||
# failure. The removal half is live today.
|
||||
shater_stop_disarms() {
|
||||
shater_action_disarms "$1" || return 1
|
||||
# No package manager involved => a person typed it. The escape hatch must work.
|
||||
[ "$2" = "1" ] || return 0
|
||||
# A package transaction that has NOT disabled the service is replacing it.
|
||||
[ "$3" = "1" ] && return 1
|
||||
return 0
|
||||
}
|
||||
|
||||
# True when a successor is coming, so the outgoing daemon should leave the
|
||||
# fail-closed holding plane standing instead of removing it.
|
||||
#
|
||||
# `shutdown` is deliberately NOT a handoff either: nothing is coming, and the
|
||||
# kernel that would hold the plane is going away with it. Leaving the flag down
|
||||
# there also keeps the marker's meaning exact — it says "you are being replaced",
|
||||
# and at shutdown nothing is.
|
||||
shater_action_handoff() {
|
||||
case "$1" in
|
||||
restart|reload) return 0 ;;
|
||||
*) return 1 ;;
|
||||
esac
|
||||
}
|
||||
|
||||
# --- helpers ---------------------------------------------------------------
|
||||
|
||||
# True only when the stack is explicitly enabled in UCI.
|
||||
@@ -125,10 +261,14 @@ shater_mark_restart() {
|
||||
shater_clear_restart() { rm -f "$RESTART_FLAG"; }
|
||||
|
||||
# Remove the persisted boot armor, so the LAN is NOT blocked at the next boot
|
||||
# before the daemon starts. Called when the operator stops the service and when
|
||||
# the daemon binary is gone — in both cases nothing is going to come along and
|
||||
# replace the armor with a real data plane, and a kill switch with nothing behind
|
||||
# it is just a brick.
|
||||
# before the daemon starts. Called from exactly two places, both of which are a
|
||||
# statement about the PRODUCT rather than about this process: an operator typing
|
||||
# `stop`, and a daemon binary that is no longer on the box. In neither case is
|
||||
# anything going to come along and replace the armor with a real data plane, and a
|
||||
# kill switch with nothing behind it is just a brick.
|
||||
#
|
||||
# NOT called on the shutdown path. That is the whole fix — see
|
||||
# shater_action_disarms.
|
||||
shater_disarm_boot() { rm -f "$BOOT_ARMOR"; }
|
||||
|
||||
# Echo the pid of a LIVE `shaterd run`, or fail. The pidfile is written by the
|
||||
@@ -290,30 +430,42 @@ start_service() {
|
||||
stop_service() {
|
||||
# Say WHY we are stopping before procd sends the signal, because the daemon
|
||||
# cannot tell from the signal alone and the answer changes what it leaves in
|
||||
# the kernel:
|
||||
# the kernel. Two INDEPENDENT questions, and the old code conflated them into
|
||||
# one two-armed `case` whose else-branch answered both wrongly for `shutdown`:
|
||||
#
|
||||
# restart / reload -> a successor is coming. Raise RESTART_FLAG so the
|
||||
# outgoing daemon replaces its data plane with the
|
||||
# fail-closed HOLDING plane instead of removing it. The
|
||||
# gap until the successor applies is not a moment: this
|
||||
# script waits out the old process, runs `shaterd
|
||||
# migrate`, then starts a daemon that must build an
|
||||
# engine — all of it, until now, with `lan -> wan
|
||||
# ACCEPT` and nothing else.
|
||||
# anything else -> a deliberate `stop`. Everything comes down, and the
|
||||
# boot armor goes with it so the next boot does not
|
||||
# quietly reinstate what the operator just switched off.
|
||||
# An admin `stop` has to STICK; that is the same rule
|
||||
# ACTIVE_FLAG has always enforced for hotplug/cron.
|
||||
case "$SHATER_RC_ACTION" in
|
||||
restart|reload)
|
||||
shater_mark_restart
|
||||
;;
|
||||
*)
|
||||
shater_clear_restart
|
||||
shater_disarm_boot
|
||||
;;
|
||||
esac
|
||||
# 1. IS A SUCCESSOR COMING (this process only)? restart / reload.
|
||||
# Raise RESTART_FLAG so the outgoing daemon replaces its data plane with
|
||||
# the fail-closed HOLDING plane instead of removing it. The gap until the
|
||||
# successor applies is not a moment: this script waits out the old
|
||||
# process, runs `shaterd migrate`, then starts a daemon that must build an
|
||||
# engine — all of it, before this flag existed, with `lan -> wan ACCEPT`
|
||||
# and nothing else.
|
||||
#
|
||||
# 2. IS THE PRODUCT BEING SWITCHED OFF (across boots)? `stop` — and only
|
||||
# `stop`, and only when a PERSON is behind it (shater_stop_disarms; the
|
||||
# package manager reaches us through `stop` too). Then the boot armor goes
|
||||
# with it, so the next boot does not quietly reinstate what the operator
|
||||
# just switched off — the same rule ACTIVE_FLAG has always enforced for
|
||||
# hotplug/cron.
|
||||
#
|
||||
# `shutdown` answers NO to both, which is the defect this replaced: a reboot is
|
||||
# not a successor and it is certainly not an operator switching the product off.
|
||||
# It is the boot the armor exists for. An upgrade answers NO to the second for
|
||||
# the same kind of reason.
|
||||
if shater_action_handoff "$SHATER_RC_ACTION"; then
|
||||
shater_mark_restart
|
||||
else
|
||||
shater_clear_restart
|
||||
fi
|
||||
local in_pkg=0 rc_en=0
|
||||
shater_pkg_transaction && in_pkg=1
|
||||
shater_rc_enabled && rc_en=1
|
||||
if shater_stop_disarms "$SHATER_RC_ACTION" "$in_pkg" "$rc_en"; then
|
||||
shater_disarm_boot
|
||||
elif [ "$in_pkg" = "1" ] && shater_action_disarms "$SHATER_RC_ACTION"; then
|
||||
_slog -p daemon.info \
|
||||
"stop came from a package transaction that left the service enabled — keeping the boot armor, so being replaced cannot leave the next boot unprotected"
|
||||
fi
|
||||
|
||||
# Drop the live-flag FIRST so a concurrent hotplug/cron tick cannot rebuild
|
||||
# what we are about to tear down. procd then sends SIGTERM to `shaterd run`,
|
||||
|
||||
@@ -39,12 +39,24 @@
|
||||
#
|
||||
# THE ESCAPE HATCHES (a kill switch that cannot be switched off is a brick)
|
||||
#
|
||||
# These are STATE checks, evaluated here, at the moment of arming — not a record
|
||||
# of something that happened on the way down. That distinction is the whole
|
||||
# lesson of v0.2.17: the arm token was deleted by an EVENT on the shutdown path
|
||||
# ("this looks like a stop"), and since `reboot` also runs the K-links, the
|
||||
# mechanism reliably erased itself on the one transition it was built for. An
|
||||
# event on the way down cannot be trusted to describe the world on the way up; a
|
||||
# question asked on the way up can be.
|
||||
#
|
||||
# * $ARMOR only exists while the daemon's last applied config was BOTH enabled
|
||||
# and fail-closed. `globals.enabled=0`, `kill_switch=open` and a deliberate
|
||||
# `/etc/init.d/shater stop` each remove it.
|
||||
# and fail-closed. `globals.enabled=0` and `kill_switch=open` each remove it
|
||||
# at the next apply, and an operator typing `/etc/init.d/shater stop` removes
|
||||
# it there and then. Powering the box off does NOT.
|
||||
# * We refuse to arm when the main service is disabled in rc.d, or when the
|
||||
# daemon binary is gone — in either case nothing would ever come along to
|
||||
# replace the armor with a real data plane.
|
||||
# replace the armor with a real data plane. These two are what makes a
|
||||
# genuinely uninstalled/disabled product safe REGARDLESS of what the file
|
||||
# says, which is why they are checked here rather than trusted to have been
|
||||
# acted on earlier.
|
||||
# * We refuse to arm when UCI can be read AND says the stack is disabled. A
|
||||
# config that cannot be read is NOT a refusal: that case is precisely why the
|
||||
# armor is a file rather than a query.
|
||||
@@ -52,6 +64,12 @@
|
||||
# hook, to the router's own addresses) stay reachable. The operator can always
|
||||
# get in and undo this.
|
||||
#
|
||||
# Note what a bare `/etc/init.d/shater stop` does NOT mean: it does not survive a
|
||||
# reboot, because S99shater is still linked and procd starts the daemon again. So
|
||||
# "stopped" is not a durable off-state and this script must not be designed as if
|
||||
# it were — the durable ones are `disable` (no S??shater) and `globals.enabled=0`,
|
||||
# and those are the two refusals above.
|
||||
#
|
||||
# busybox ash only — no bashisms.
|
||||
|
||||
START=21 # after firewall (19) and network (20), long before shater (99)
|
||||
|
||||
@@ -59,6 +59,68 @@ if uci -q get shater.globals >/dev/null 2>&1 || [ -f /etc/config/shater ]; then
|
||||
uci -q commit shater
|
||||
fi
|
||||
|
||||
# Introduce the daemon-created `shater-l3` TUN to fw4 (L3 ingress, D-L3). The
|
||||
# daemon policy-routes LAN ICMP into that device from OUR nft table
|
||||
# `inet shater`, but nftables runs EVERY table on every packet and a drop in
|
||||
# any one of them wins — an accept in `inet shater` cannot override fw4. And
|
||||
# fw4 WILL drop this forward: netifd knows nothing about a device the daemon
|
||||
# creates at runtime, so it belongs to no zone and falls into fw4's zone-less
|
||||
# defaults (REJECT). The device has to be declared to fw4 itself; it cannot be
|
||||
# fixed from our own table.
|
||||
#
|
||||
# Seeded UNCONDITIONALLY (not gated on globals.l3_tunnel): uci-defaults run
|
||||
# once, so gating on the option would require re-running this script when the
|
||||
# option is flipped later — which never happens. An idle zone is harmless: its
|
||||
# device match is a plain iifname/oifname STRING compare that simply never hits
|
||||
# while the TUN does not exist.
|
||||
#
|
||||
# Idempotency: `config zone`/`config forwarding` are normally ANONYMOUS
|
||||
# sections, and a naive `uci add firewall zone` would append a duplicate on
|
||||
# every re-run (uci-defaults re-run on package upgrade/reinstall). All sections
|
||||
# here are NAMED instead, guarded by an existence check — a re-run re-finds the
|
||||
# section and touches nothing.
|
||||
seed_l3_zone() {
|
||||
# No fw4 on this image (bare nftables build) => nothing drops the forward
|
||||
# on fw4's behalf and there is nothing to punch through.
|
||||
[ -f /etc/config/firewall ] || return 0
|
||||
if ! uci -q get firewall.shater_l3 >/dev/null; then
|
||||
uci set firewall.shater_l3=zone
|
||||
uci set firewall.shater_l3.name='shater_l3'
|
||||
uci set firewall.shater_l3.input='REJECT'
|
||||
uci set firewall.shater_l3.output='ACCEPT'
|
||||
uci set firewall.shater_l3.forward='REJECT'
|
||||
uci set firewall.shater_l3.masq='0'
|
||||
# INERT TODAY, kept for the day it is not. mtu_fix clamps forwarded TCP
|
||||
# MSS to the route MTU — but shater-l3 is 65535 (deliberately: at any
|
||||
# smaller value the kernel fragments into the device, and the flow
|
||||
# dispatcher refuses to judge a fragment and lets the stack forge the
|
||||
# echo reply — see l3MTU in shater/generate/inbound.go), so the clamp has
|
||||
# nothing to clamp to. And only ICMP is ever marked into this device, so
|
||||
# no TCP rides here to be clamped in the first place. It earns its keep
|
||||
# the moment either of those changes; removing it would make that day
|
||||
# silent.
|
||||
uci set firewall.shater_l3.mtu_fix='1'
|
||||
# `list device`, deliberately NOT the usual `list network`: fw4
|
||||
# resolves a zone's networks through netifd, and netifd never learns
|
||||
# about a device the daemon creates at runtime — a stub interface
|
||||
# (proto none) would need to be brought UP to contribute an l3_device,
|
||||
# and nothing ever brings it up, so `list network` resolves to an
|
||||
# EMPTY device set and fw4 keeps dropping the forward. `list device`
|
||||
# instead compiles to an iifname/oifname STRING match, valid before
|
||||
# the TUN exists and matching from the moment shaterd creates it —
|
||||
# no netifd involvement and no firewall reload at enable time. Do not
|
||||
# "normalize" this to `list network` in a refactor; it breaks silently.
|
||||
uci add_list firewall.shater_l3.device='shater-l3'
|
||||
fi
|
||||
if ! uci -q get firewall.shater_l3_fwd >/dev/null; then
|
||||
uci set firewall.shater_l3_fwd=forwarding
|
||||
uci set firewall.shater_l3_fwd.src='lan'
|
||||
uci set firewall.shater_l3_fwd.dest='shater_l3'
|
||||
fi
|
||||
uci -q commit firewall
|
||||
}
|
||||
seed_l3_zone
|
||||
|
||||
# Bring the UCI schema forward on upgrade (idempotent; refuses a newer schema).
|
||||
[ -x /usr/bin/shaterd ] && /usr/bin/shaterd migrate >/dev/null 2>&1
|
||||
|
||||
@@ -120,6 +182,17 @@ SHATER_BRINGUP='
|
||||
[ -x /etc/init.d/shater-armor ] && /etc/init.d/shater-armor enable
|
||||
[ -x /etc/init.d/shater ] && /etc/init.d/shater restart
|
||||
[ -x /etc/init.d/shater-cron ] && /etc/init.d/shater-cron restart
|
||||
# Fold the seeded shater_l3 zone into the LIVE ruleset — matters on a live
|
||||
# opkg/apk install only, where firewall started long before our commit and
|
||||
# nothing else would re-read it until the next reboot. Gated on the fw4
|
||||
# table actually being loaded: at FIRST boot this job can run before the
|
||||
# S19 firewall start, and an early reload would install a ruleset built
|
||||
# from a half-initialized netifd AND make the later start a no-op (fw4
|
||||
# start skips when its table already exists). No table => the pending S19
|
||||
# start reads the committed config by itself, no reload needed.
|
||||
if nft list tables 2>/dev/null | grep -q "inet fw4"; then
|
||||
[ -x /etc/init.d/firewall ] && /etc/init.d/firewall reload
|
||||
fi
|
||||
exit 0
|
||||
'
|
||||
SHATER_TMO=""
|
||||
|
||||
+51
-11
@@ -60,16 +60,31 @@ type RuleForce = {
|
||||
}
|
||||
|
||||
/**
|
||||
* Everything `Proto` can match, and nothing else. The engine understands two
|
||||
* transports and exactly ten application protocols its sniffers can name
|
||||
* (generate/route.go sniffedProtocols); a value outside this set builds a rule
|
||||
* that is perfectly valid and can never fire — so its traffic quietly falls
|
||||
* Everything `Proto` can match, and nothing else. A value outside this set builds
|
||||
* a rule that is perfectly valid and can never fire — so its traffic quietly falls
|
||||
* through to whatever rule sits below it. That is why this is a closed list and
|
||||
* not a text box.
|
||||
*
|
||||
* Split into two groups because they answer different questions: the transport is
|
||||
* known the moment a packet arrives, while an app protocol is only known once the
|
||||
* first bytes have been read and labelled.
|
||||
* Three groups, because the engine reads them through three different matchers
|
||||
* (generate/route.go, ruleMatchers) and they answer different questions:
|
||||
*
|
||||
* - Transport — the L4 network. Known the moment a packet arrives.
|
||||
* - Detected protocol — the L7 label a sniffer puts on a connection once its
|
||||
* first bytes have been read. This group, and ONLY this group, is the engine's
|
||||
* `sniffedProtocols` set; anything else routed into that matcher is inert.
|
||||
* - Layer 3 — ICMP. Not a sniffed label: it lands in the emitted rule's
|
||||
* `network`, never in `protocol` (the sniffers are skipped outright for an
|
||||
* ICMP flow, so they never report "icmp"). All three spellings are the SAME
|
||||
* one network; `icmpv4`/`icmpv6` additionally pin `ip_version`, which the
|
||||
* engine derives from the destination address.
|
||||
*
|
||||
* ICMP carries caveats the picker deliberately does not try to enforce, because
|
||||
* the daemon reports each one against the whole config on apply: it reaches the
|
||||
* engine only while globals l3_tunnel is on, it has no ports (a port matcher
|
||||
* beside it can never be satisfied), `icmpv6` also needs globals ipv6 on, and it
|
||||
* is DROPPED rather than falling through when routed at a target that cannot
|
||||
* carry layer 3 — i.e. every proxy protocol. Only wireguard/AmneziaWG nodes and
|
||||
* direct/interface egresses can carry a ping.
|
||||
*/
|
||||
const PROTO_TRANSPORT: ReadonlyArray<{ id: string; label: string }> = [
|
||||
{ id: 'tcp', label: 'TCP' },
|
||||
@@ -87,10 +102,27 @@ const PROTO_APP: ReadonlyArray<{ id: string; label: string }> = [
|
||||
{ id: 'rdp', label: 'RDP' },
|
||||
{ id: 'ntp', label: 'NTP' },
|
||||
]
|
||||
const PROTO_VALUES = new Set([...PROTO_TRANSPORT, ...PROTO_APP].map((p) => p.id))
|
||||
/**
|
||||
* The family-qualified spellings are offered next to plain `icmp` rather than
|
||||
* hidden behind it: the engine treats them as first-class and the difference is
|
||||
* observable (an `ip_version` item on the same rule), so hiding them would leave a
|
||||
* capability reachable only by hand-editing /etc/config/shater — and would mean
|
||||
* that anyone who edited such a rule here lost the narrowing on the next save.
|
||||
*/
|
||||
const PROTO_L3: ReadonlyArray<{ id: string; label: string }> = [
|
||||
{ id: 'icmp', label: 'ICMP (ping)' },
|
||||
{ id: 'icmpv4', label: 'ICMP — IPv4 only' },
|
||||
{ id: 'icmpv6', label: 'ICMP — IPv6 only' },
|
||||
]
|
||||
const PROTO_VALUES = new Set(
|
||||
[...PROTO_TRANSPORT, ...PROTO_APP, ...PROTO_L3].map((p) => p.id),
|
||||
)
|
||||
|
||||
/** The Proto picker's option list — shared by the inline add row and the editor. */
|
||||
function ProtoOptions({ value }: { value: string }) {
|
||||
// The engine lower-cases `Proto` before matching it, so a hand-written `ICMP`
|
||||
// is a working rule; judge it the same way and flag only what really is inert.
|
||||
const matches = PROTO_VALUES.has(value.trim().toLowerCase())
|
||||
return (
|
||||
<>
|
||||
<option value="">any</option>
|
||||
@@ -108,10 +140,18 @@ function ProtoOptions({ value }: { value: string }) {
|
||||
</option>
|
||||
))}
|
||||
</optgroup>
|
||||
{/* A stored value the engine can't detect is kept and flagged, never
|
||||
silently rewritten — the rule it belongs to is live right now. */}
|
||||
<optgroup label="Layer 3">
|
||||
{PROTO_L3.map((p) => (
|
||||
<option key={p.id} value={p.id}>
|
||||
{p.label}
|
||||
</option>
|
||||
))}
|
||||
</optgroup>
|
||||
{/* A stored value none of the groups spells verbatim is kept and offered as
|
||||
written, never silently rewritten — the rule it belongs to is live right
|
||||
now. It is flagged only when the engine cannot match it either. */}
|
||||
{value !== '' && !PROTO_VALUES.has(value) && (
|
||||
<option value={value}>{value} — never matches</option>
|
||||
<option value={value}>{matches ? value : `${value} — never matches`}</option>
|
||||
)}
|
||||
</>
|
||||
)
|
||||
|
||||
@@ -0,0 +1,268 @@
|
||||
// lx:begin l3-honest-drop
|
||||
package route
|
||||
|
||||
import (
|
||||
"context"
|
||||
"net/netip"
|
||||
"testing"
|
||||
|
||||
"github.com/sagernet/sing-box/adapter"
|
||||
C "github.com/sagernet/sing-box/constant"
|
||||
"github.com/sagernet/sing-box/log"
|
||||
"github.com/sagernet/sing-box/option"
|
||||
R "github.com/sagernet/sing-box/route/rule"
|
||||
"github.com/sagernet/sing/common/json/badoption"
|
||||
M "github.com/sagernet/sing/common/metadata"
|
||||
N "github.com/sagernet/sing/common/network"
|
||||
|
||||
"github.com/stretchr/testify/require"
|
||||
)
|
||||
|
||||
// The contract under test: PreMatch never answers "continue" (nor "bypass") for
|
||||
// an ICMP flow. adapter.JudgeFlow maps both to tun.ActionAccept, and the TUN
|
||||
// stack answers Accept by FORGING the echo reply itself
|
||||
// (sing-tun stack_gvisor_icmp.go ICMPForwarder.HandlePacket, the fallthrough
|
||||
// under the Flow/Reject/Drop switch). A verdict of "continue" therefore reads to
|
||||
// the operator as a working ping off a tunnel that never carried the packet.
|
||||
//
|
||||
// Every test below has a TCP/UDP twin: the honest drop must not leak into the
|
||||
// protocols where "continue" really does mean "take the ordinary connection
|
||||
// route".
|
||||
|
||||
// icmpL4Outbound is a minimal L4-only outbound (the vless/vmess/... shape): it
|
||||
// does NOT implement adapter.FlowOutbound, and Network() lists only TCP/UDP.
|
||||
// Unused Outbound methods come from the embedded nil interface and are never
|
||||
// called on the pre-match paths under test.
|
||||
type icmpL4Outbound struct {
|
||||
adapter.Outbound
|
||||
tag string
|
||||
}
|
||||
|
||||
func (o *icmpL4Outbound) Tag() string { return o.tag }
|
||||
func (o *icmpL4Outbound) Type() string { return "vless" }
|
||||
func (o *icmpL4Outbound) Network() []string { return []string{N.NetworkTCP, N.NetworkUDP} }
|
||||
|
||||
// icmpOutboundManager resolves tags from a fixed map and hands the same L4-only
|
||||
// outbound out as the default; the rest of the OutboundManager surface is never
|
||||
// touched by the pre-match walk.
|
||||
type icmpOutboundManager struct {
|
||||
adapter.OutboundManager
|
||||
defaultOutbound adapter.Outbound
|
||||
outbounds map[string]adapter.Outbound
|
||||
}
|
||||
|
||||
func (m *icmpOutboundManager) Default() adapter.Outbound { return m.defaultOutbound }
|
||||
|
||||
func (m *icmpOutboundManager) Outbound(tag string) (adapter.Outbound, bool) {
|
||||
outbound, loaded := m.outbounds[tag]
|
||||
return outbound, loaded
|
||||
}
|
||||
|
||||
// icmpDNSRouter / icmpDNSTransportManager implement only what
|
||||
// prepareMatchMetadata reaches. FakeIP returns nil unless a transport is
|
||||
// installed, which is how the "fakeip lookup failed" exit is driven below.
|
||||
type icmpDNSRouter struct {
|
||||
adapter.DNSRouter
|
||||
}
|
||||
|
||||
func (s *icmpDNSRouter) LookupReverseMapping(netip.Addr) (string, bool) { return "", false }
|
||||
|
||||
type icmpDNSTransportManager struct {
|
||||
adapter.DNSTransportManager
|
||||
fakeIP adapter.FakeIPTransport
|
||||
}
|
||||
|
||||
func (s *icmpDNSTransportManager) FakeIP() adapter.FakeIPTransport {
|
||||
if s.fakeIP == nil {
|
||||
return nil
|
||||
}
|
||||
return s.fakeIP
|
||||
}
|
||||
|
||||
// icmpMissingFakeIPTransport claims every address and then fails to look any of
|
||||
// them up — exactly the "missing fakeip record, try enable
|
||||
// `experimental.cache_file`" error prepareMatchMetadata returns.
|
||||
type icmpMissingFakeIPTransport struct {
|
||||
adapter.FakeIPTransport
|
||||
}
|
||||
|
||||
func (t *icmpMissingFakeIPTransport) Store() adapter.FakeIPStore {
|
||||
return &icmpMissingFakeIPStore{}
|
||||
}
|
||||
|
||||
type icmpMissingFakeIPStore struct {
|
||||
adapter.FakeIPStore
|
||||
}
|
||||
|
||||
func (s *icmpMissingFakeIPStore) Contains(netip.Addr) bool { return true }
|
||||
func (s *icmpMissingFakeIPStore) Lookup(netip.Addr) (string, bool) { return "", false }
|
||||
|
||||
type icmpRouterOptions struct {
|
||||
fakeIP adapter.FakeIPTransport
|
||||
rules []option.Rule
|
||||
}
|
||||
|
||||
func icmpTestRouter(t *testing.T, options icmpRouterOptions) *Router {
|
||||
t.Helper()
|
||||
logger := log.NewNOPFactory().NewLogger("test")
|
||||
defaultOutbound := &icmpL4Outbound{tag: "proxy-out"}
|
||||
router := &Router{
|
||||
ctx: context.Background(),
|
||||
logger: logger,
|
||||
dns: &icmpDNSRouter{},
|
||||
dnsTransport: &icmpDNSTransportManager{fakeIP: options.fakeIP},
|
||||
outbound: &icmpOutboundManager{
|
||||
defaultOutbound: defaultOutbound,
|
||||
outbounds: map[string]adapter.Outbound{defaultOutbound.Tag(): defaultOutbound},
|
||||
},
|
||||
}
|
||||
for i, ruleOptions := range options.rules {
|
||||
rule, err := R.NewRule(router.ctx, logger, ruleOptions, false)
|
||||
require.NoError(t, err, "build rule[%d]", i)
|
||||
router.rules = append(router.rules, rule)
|
||||
}
|
||||
return router
|
||||
}
|
||||
|
||||
func icmpTestMetadata(network string) adapter.InboundContext {
|
||||
return adapter.InboundContext{
|
||||
Inbound: "l3-in",
|
||||
InboundType: C.TypeTun,
|
||||
Network: network,
|
||||
Source: M.SocksaddrFrom(netip.MustParseAddr("192.168.1.2"), 0),
|
||||
Destination: M.SocksaddrFrom(netip.MustParseAddr("1.1.1.1"), 0),
|
||||
}
|
||||
}
|
||||
|
||||
// lanRuleWithAction matches every packet from the test source, so the action is
|
||||
// what the test is actually about.
|
||||
func lanRuleWithAction(action option.RuleAction) option.Rule {
|
||||
return option.Rule{
|
||||
Type: C.RuleTypeDefault,
|
||||
DefaultOptions: option.DefaultRule{
|
||||
RawDefaultRule: option.RawDefaultRule{
|
||||
SourceIPCIDR: badoption.Listable[string]{"192.168.1.0/24"},
|
||||
},
|
||||
RuleAction: action,
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
// --- exit 1: an outbound that cannot carry layer 3 --------------------------
|
||||
|
||||
func TestPreMatchICMPToL4OutboundDrops(t *testing.T) {
|
||||
t.Parallel()
|
||||
router := icmpTestRouter(t, icmpRouterOptions{})
|
||||
result := router.PreMatch(icmpTestMetadata(N.NetworkICMP), nil)
|
||||
require.Equal(t, adapter.PreMatchDrop, result.Action,
|
||||
"ICMP to an L4-only outbound fell through to the ordinary pre-match path: the TUN stack will forge the echo reply and ping will lie about a tunnel that never saw the packet")
|
||||
}
|
||||
|
||||
func TestPreMatchTCPToL4OutboundContinues(t *testing.T) {
|
||||
t.Parallel()
|
||||
router := icmpTestRouter(t, icmpRouterOptions{})
|
||||
result := router.PreMatch(icmpTestMetadata(N.NetworkTCP), nil)
|
||||
require.Equal(t, adapter.PreMatchContinue, result.Action,
|
||||
"TCP to an L4-only outbound must keep taking the ordinary connection route; the ICMP honest-drop must not leak into TCP/UDP pre-match")
|
||||
}
|
||||
|
||||
func TestPreMatchUDPToL4OutboundContinues(t *testing.T) {
|
||||
t.Parallel()
|
||||
router := icmpTestRouter(t, icmpRouterOptions{})
|
||||
result := router.PreMatch(icmpTestMetadata(N.NetworkUDP), nil)
|
||||
require.Equal(t, adapter.PreMatchContinue, result.Action,
|
||||
"UDP to an L4-only outbound must keep taking the ordinary connection route")
|
||||
}
|
||||
|
||||
// --- exit 2: prepareMatchMetadata failed before any rule was walked ---------
|
||||
|
||||
// This exit arrived with the shared prepareMatchMetadata refactor (upstream
|
||||
// b911fb078): it returns before the rule walk, so it never reaches preMatchFlow
|
||||
// where the ICMP override used to live.
|
||||
func TestPreMatchICMPMetadataErrorDrops(t *testing.T) {
|
||||
t.Parallel()
|
||||
router := icmpTestRouter(t, icmpRouterOptions{fakeIP: &icmpMissingFakeIPTransport{}})
|
||||
result := router.PreMatch(icmpTestMetadata(N.NetworkICMP), nil)
|
||||
require.Equal(t, adapter.PreMatchDrop, result.Action,
|
||||
"a fakeip record that cannot be resolved must not degrade ICMP to continue: continue is tun.ActionAccept, and Accept is a forged echo reply")
|
||||
}
|
||||
|
||||
func TestPreMatchTCPMetadataErrorContinues(t *testing.T) {
|
||||
t.Parallel()
|
||||
router := icmpTestRouter(t, icmpRouterOptions{fakeIP: &icmpMissingFakeIPTransport{}})
|
||||
result := router.PreMatch(icmpTestMetadata(N.NetworkTCP), nil)
|
||||
require.Equal(t, adapter.PreMatchContinue, result.Action,
|
||||
"for TCP the metadata-error exit must keep meaning `take the ordinary connection route`")
|
||||
}
|
||||
|
||||
// --- exit 3: a rule action the pre-match walk does not handle ---------------
|
||||
|
||||
// hijack-dns is one of the actions PreMatch's switch has no arm for, so it lands
|
||||
// in the default arm. Any future unhandled action lands there too — that is why
|
||||
// the guard is a funnel on the return value and not a per-arm override.
|
||||
func TestPreMatchICMPUnhandledRuleActionDrops(t *testing.T) {
|
||||
t.Parallel()
|
||||
router := icmpTestRouter(t, icmpRouterOptions{
|
||||
rules: []option.Rule{lanRuleWithAction(option.RuleAction{Action: C.RuleActionTypeHijackDNS})},
|
||||
})
|
||||
result := router.PreMatch(icmpTestMetadata(N.NetworkICMP), nil)
|
||||
require.Equal(t, adapter.PreMatchDrop, result.Action,
|
||||
"an unhandled rule action must not degrade ICMP to continue: continue is tun.ActionAccept, and Accept is a forged echo reply")
|
||||
}
|
||||
|
||||
func TestPreMatchTCPUnhandledRuleActionContinues(t *testing.T) {
|
||||
t.Parallel()
|
||||
router := icmpTestRouter(t, icmpRouterOptions{
|
||||
rules: []option.Rule{lanRuleWithAction(option.RuleAction{Action: C.RuleActionTypeHijackDNS})},
|
||||
})
|
||||
result := router.PreMatch(icmpTestMetadata(N.NetworkTCP), nil)
|
||||
require.Equal(t, adapter.PreMatchContinue, result.Action,
|
||||
"the unhandled-action exit must stay a continue for TCP")
|
||||
}
|
||||
|
||||
// --- exit 4: an explicit bypass ---------------------------------------------
|
||||
|
||||
// sing-tun implements ActionBypass on the nfqueue plane only; on the TUN path it
|
||||
// falls into the same default arm as Accept (flow_dispatch.go judgeAndInstall,
|
||||
// and the ICMP forwarder's switch has no Bypass case either), i.e. into the same
|
||||
// forgery. There is no honest bypass for a packet already inside the engine's
|
||||
// TUN.
|
||||
func TestPreMatchICMPBypassDrops(t *testing.T) {
|
||||
t.Parallel()
|
||||
router := icmpTestRouter(t, icmpRouterOptions{
|
||||
rules: []option.Rule{lanRuleWithAction(option.RuleAction{Action: C.RuleActionTypeBypass})},
|
||||
})
|
||||
result := router.PreMatch(icmpTestMetadata(N.NetworkICMP), nil)
|
||||
require.Equal(t, adapter.PreMatchDrop, result.Action,
|
||||
"bypass degrades to tun.ActionAccept on the TUN path, which is the forged echo reply again")
|
||||
}
|
||||
|
||||
func TestPreMatchTCPBypassIsStillBypass(t *testing.T) {
|
||||
t.Parallel()
|
||||
router := icmpTestRouter(t, icmpRouterOptions{
|
||||
rules: []option.Rule{lanRuleWithAction(option.RuleAction{Action: C.RuleActionTypeBypass})},
|
||||
})
|
||||
result := router.PreMatch(icmpTestMetadata(N.NetworkTCP), nil)
|
||||
require.Equal(t, adapter.PreMatchBypass, result.Action,
|
||||
"the ICMP honest-drop must not turn a TCP bypass rule into a drop")
|
||||
}
|
||||
|
||||
// --- the verdicts that must pass through untouched ---------------------------
|
||||
|
||||
// A reject rule already carries its own honest verdict; the funnel must not
|
||||
// rewrite it (a Reject sends an ICMP unreachable, which is information, not a
|
||||
// forged liveness signal).
|
||||
func TestPreMatchICMPRejectIsNotRewritten(t *testing.T) {
|
||||
t.Parallel()
|
||||
router := icmpTestRouter(t, icmpRouterOptions{
|
||||
rules: []option.Rule{lanRuleWithAction(option.RuleAction{
|
||||
Action: C.RuleActionTypeReject,
|
||||
RejectOptions: option.RejectActionOptions{Method: C.RuleActionRejectMethodDefault},
|
||||
})},
|
||||
})
|
||||
result := router.PreMatch(icmpTestMetadata(N.NetworkICMP), nil)
|
||||
require.Equal(t, adapter.PreMatchReject, result.Action,
|
||||
"the ICMP funnel must only rewrite continue/bypass, never an explicit reject")
|
||||
}
|
||||
|
||||
// lx:end l3-honest-drop
|
||||
@@ -314,7 +314,48 @@ func (r *Router) routePacketConnection(ctx context.Context, conn N.PacketConn, m
|
||||
return nil
|
||||
}
|
||||
|
||||
// lx:begin l3-honest-drop
|
||||
// PreMatch funnels every verdict of the pre-match walk through one ICMP check.
|
||||
//
|
||||
// An ICMP flow has no fallback path, so PreMatchContinue is not "try the
|
||||
// ordinary connection route" the way it is for TCP and UDP: the TUN stack takes
|
||||
// the packet back and answers the echo ITSELF (sing-tun stack_gvisor_icmp.go —
|
||||
// adapter.JudgeFlow maps Continue to tun.ActionAccept, and the ICMP forwarder
|
||||
// answers Accept by rewriting Echo into EchoReply and swapping the addresses).
|
||||
// A ping routed to an outbound that cannot carry layer 3 — every proxy
|
||||
// protocol; only adapter.FlowOutbound can — would therefore return a FORGED
|
||||
// reply, and the operator would read a working ping off a tunnel that never saw
|
||||
// the packet. Dropping instead reports the truth.
|
||||
//
|
||||
// PreMatchBypass is folded into the same drop because sing-tun implements
|
||||
// bypass for the nfqueue plane only (`ActionBypass` appears nowhere in
|
||||
// flow_dispatch.go / stack_gvisor_icmp.go): on the TUN path it degrades to the
|
||||
// same Accept, i.e. to the same forgery. There is no honest bypass for an ICMP
|
||||
// packet that is already inside the engine's TUN.
|
||||
//
|
||||
// This is a funnel and not an override inside the walk on purpose: the walk has
|
||||
// several independent exits that say "continue" (the prepareMatchMetadata error
|
||||
// return, the sniff bail-outs, the un-routable `bypass`, and the default arm of
|
||||
// the rule-action switch), and an earlier version of this delta guarded only
|
||||
// the ones that pass through preMatchFlow — leaving the others as narrow paths
|
||||
// to the forged reply. Guarding the single return value cannot be outgrown by a
|
||||
// new exit.
|
||||
func (r *Router) PreMatch(metadata adapter.InboundContext, firstPacket []byte) adapter.PreMatchResult {
|
||||
result := r.preMatch(metadata, firstPacket)
|
||||
if metadata.Network == N.NetworkICMP {
|
||||
switch result.Action {
|
||||
case adapter.PreMatchContinue, adapter.PreMatchBypass:
|
||||
return adapter.PreMatchResult{Action: adapter.PreMatchDrop}
|
||||
}
|
||||
}
|
||||
return result
|
||||
}
|
||||
|
||||
// preMatch is upstream's PreMatch body, unchanged; only the name moved, so that
|
||||
// the funnel above owns the exported entry point. An upstream change to the
|
||||
// pre-match walk applies to THIS function.
|
||||
func (r *Router) preMatch(metadata adapter.InboundContext, firstPacket []byte) adapter.PreMatchResult {
|
||||
// lx:end l3-honest-drop
|
||||
ctx := log.ContextWithNewID(r.ctx)
|
||||
metadata.PreMatch = true
|
||||
continueResult := adapter.PreMatchResult{Action: adapter.PreMatchContinue}
|
||||
@@ -440,6 +481,11 @@ func applyRouteOptionsOverride(metadata *adapter.InboundContext, routeOptions *R
|
||||
|
||||
func (r *Router) preMatchFlow(ctx context.Context, metadata *adapter.InboundContext, packetDestination M.Socksaddr, matchedRule adapter.Rule, outboundTag string) adapter.PreMatchResult {
|
||||
continueResult := adapter.PreMatchResult{Action: adapter.PreMatchContinue}
|
||||
// lx: ICMP does NOT get a local override here any more — the honest drop is
|
||||
// applied once, to the single return value of PreMatch (see the funnel
|
||||
// there, marker l3-honest-drop). Overriding continueResult in this function
|
||||
// covered only the exits that reach it and left the walk's own exits
|
||||
// forging.
|
||||
var outbound adapter.Outbound
|
||||
if outboundTag == "" {
|
||||
outbound = r.outbound.Default()
|
||||
|
||||
+180
-12
@@ -22,11 +22,16 @@
|
||||
# 2. It runs on linux. shater/generate has 44 test files on linux against 32 on
|
||||
# windows/darwin; the linux-only half is where the routing, ruleset, DNS and
|
||||
# health tests live.
|
||||
# 3. Nothing is skipped SILENTLY. Two machine checks:
|
||||
# 3. Nothing is skipped SILENTLY. Three machine checks:
|
||||
# - the tag set may only ADD test files, never hide them (a test behind
|
||||
# `//go:build !with_awg` would vanish from the gate — this fails first);
|
||||
# - every package that has tests must report `ok` by name; a suite that
|
||||
# compiles down to "no test files" fails the gate instead of passing it.
|
||||
# compiles down to "no test files" fails the gate instead of passing it;
|
||||
# - every ^TestIntegration under the fork's trees must produce a verdict
|
||||
# BY NAME ([5/5]). `ok <pkg>` is printed whether the privileged tests in
|
||||
# that package ran or called t.Skip, so the second check cannot see them
|
||||
# — and the gate would keep saying "passes every test we own" while the
|
||||
# tests that need a real kernel never executed.
|
||||
# A guard that silently runs nothing is worse than no guard (same rule as
|
||||
# scripts/check-router-tags.sh).
|
||||
#
|
||||
@@ -38,6 +43,10 @@
|
||||
# SHATER_GO_IMAGE docker image used to reach linux from a non-linux host
|
||||
# (default golang:1.26 — keep it >= go.mod's toolchain).
|
||||
# SHATER_NO_DOCKER=1 fail instead of falling back to docker.
|
||||
# SHATER_REQUIRE_PRIVILEGED=1
|
||||
# turn [5/5]'s "did not run here" report into a hard
|
||||
# failure. Use it on the OpenWrt VM or in any pre-release
|
||||
# run that must actually have exercised the kernel paths.
|
||||
set -euo pipefail
|
||||
|
||||
SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
|
||||
@@ -48,7 +57,7 @@ RACE=1
|
||||
for a in "$@"; do
|
||||
case "$a" in
|
||||
--no-race) RACE=0 ;;
|
||||
-h|--help) sed -n '2,41p' "$0" | sed 's/^# \{0,1\}//'; exit 0 ;;
|
||||
-h|--help) sed -n '2,49p' "$0" | sed 's/^# \{0,1\}//'; exit 0 ;;
|
||||
*) echo "run-tests: unknown flag: $a" >&2; exit 2 ;;
|
||||
esac
|
||||
done
|
||||
@@ -72,6 +81,12 @@ ROOTS_COMMON=(./common/...)
|
||||
# rest of common/ can be a real gate instead of a permanently red one. (On
|
||||
# linux every tlsspoof test is a TestIntegration*, so that package is
|
||||
# effectively uncovered here; it is covered by the VM runs.)
|
||||
# The ^TestIntegration prefix is the fork-wide marker for "needs capabilities
|
||||
# the ordinary gate lacks", and [5/5] below leans on the same convention to
|
||||
# catch privileged tests inside ROOTS, which are NOT name-filtered and would
|
||||
# otherwise skip behind a green `ok <pkg>`. ROOTS_COMMON stays out of [5/5]:
|
||||
# these fail rather than skip without the capability, and that is a decision
|
||||
# about upstream code, not about the fork's own coverage.
|
||||
SKIP_COMMON='^TestIntegration'
|
||||
|
||||
# SKIP, WITH REASON: the first -race run over this tree (2026-07-26 — nobody
|
||||
@@ -111,12 +126,30 @@ if [ "$(go env GOOS)" != "linux" ] && [ "${SHATER_TESTS_IN_DOCKER:-0}" != "1" ];
|
||||
echo "== re-exec on linux via docker ($image) =="
|
||||
host_repo="$REPO"
|
||||
command -v cygpath >/dev/null 2>&1 && host_repo="$(cygpath -w "$REPO")"
|
||||
# Hand the container CAP_NET_ADMIN and /dev/net/tun when this host's docker
|
||||
# can. shater/generate's ^TestIntegration tests open a real TUN and stand a
|
||||
# real engine on it; without the device they skip, and a dev running the gate
|
||||
# by hand would get a green result that never touched the kernel path the
|
||||
# branch is about. The dev host CAN give them (Docker Desktop's VM has the tun
|
||||
# module) — the CI runner cannot, which is what [5/5] exists to say out loud.
|
||||
# PROBED, never assumed: a docker whose kernel lacks tun refuses --device and
|
||||
# would take the whole gate down with it.
|
||||
priv_flags=()
|
||||
if MSYS2_ARG_CONV_EXCL='*' MSYS_NO_PATHCONV=1 docker run --rm \
|
||||
--cap-add NET_ADMIN --device /dev/net/tun "$image" true >/dev/null 2>&1; then
|
||||
priv_flags=(--cap-add NET_ADMIN --device /dev/net/tun)
|
||||
echo " CAP_NET_ADMIN + /dev/net/tun: available — the privileged tests will really run"
|
||||
else
|
||||
echo " CAP_NET_ADMIN + /dev/net/tun: NOT available from this docker — [5/5] will report the gap"
|
||||
fi
|
||||
MSYS2_ARG_CONV_EXCL='*' MSYS_NO_PATHCONV=1 docker run --rm \
|
||||
"${priv_flags[@]+"${priv_flags[@]}"}" \
|
||||
-v "$host_repo":/src \
|
||||
-v shater-tagcheck-gomod:/go/pkg/mod \
|
||||
-v shater-tagcheck-gocache:/root/.cache/go-build \
|
||||
-w /src \
|
||||
-e SHATER_TESTS_IN_DOCKER=1 \
|
||||
-e SHATER_REQUIRE_PRIVILEGED="${SHATER_REQUIRE_PRIVILEGED:-0}" \
|
||||
"$image" bash scripts/run-tests.sh "$@"
|
||||
exit $?
|
||||
fi
|
||||
@@ -128,7 +161,7 @@ ALL_ROOTS=("${ROOTS[@]}" "${ROOTS_COMMON[@]}")
|
||||
# shipped tags REMOVES a test file from any package, that test exists but the
|
||||
# gate would never see it — which is the failure mode this whole script is about,
|
||||
# just pointed the other way.
|
||||
echo "== [1/4] no test file is hidden by the shipped tag set =="
|
||||
echo "== [1/5] no test file is hidden by the shipped tag set =="
|
||||
LISTFMT='{{.ImportPath}} {{len .TestGoFiles}} {{len .XTestGoFiles}}'
|
||||
plain="$(go list -f "$LISTFMT" "${ALL_ROOTS[@]}")"
|
||||
tagged="$(go list -tags "$SHATER_ROUTER_TAGS" -f "$LISTFMT" "${ALL_ROOTS[@]}")"
|
||||
@@ -217,28 +250,140 @@ run_suite() { # $1=label $2=extra go-test flags (may be empty) $3..=packages
|
||||
echo " OK [$label]"
|
||||
}
|
||||
|
||||
# --- [2/4] the fork's trees, shipped tags, linux -----------------------------
|
||||
echo "== [2/4] go test — the fork's trees (shipped tags, linux) =="
|
||||
# --- [2/5] the fork's trees, shipped tags, linux -----------------------------
|
||||
echo "== [2/5] go test — the fork's trees (shipped tags, linux) =="
|
||||
run_suite main "" "${ROOTS[@]}"
|
||||
echo
|
||||
|
||||
# --- [3/4] common/, minus the tests that need CAP_NET_ADMIN ------------------
|
||||
echo "== [3/4] go test — common/ (minus the CAP_NET_ADMIN integration tests) =="
|
||||
# --- [3/5] common/, minus the tests that need CAP_NET_ADMIN ------------------
|
||||
echo "== [3/5] go test — common/ (minus the CAP_NET_ADMIN integration tests) =="
|
||||
run_suite common "-skip $SKIP_COMMON" "${ROOTS_COMMON[@]}"
|
||||
echo
|
||||
|
||||
# --- [4/4] -race over the same trees -----------------------------------------
|
||||
# --- [4/5] -race over the same trees -----------------------------------------
|
||||
# Everything, not a subset: shater/netplane alone is ~110 s under -race and it is
|
||||
# the single most concurrency-critical package we own (the nft data plane), so
|
||||
# once it is in, adding the rest costs ~40 s more. common/ is left out — it is
|
||||
# upstream code exercised by upstream CI.
|
||||
if [ "$RACE" -eq 1 ]; then
|
||||
echo "== [4/4] go test -race — the fork's trees =="
|
||||
echo "== [4/5] go test -race — the fork's trees =="
|
||||
echo " nothing is skipped under -race"
|
||||
|
||||
|
||||
run_suite race "-race -skip $RACE_SKIP" "${ROOTS[@]}"
|
||||
else
|
||||
echo "== [4/4] -race pass skipped (--no-race) =="
|
||||
echo "== [4/5] -race pass skipped (--no-race) =="
|
||||
fi
|
||||
echo
|
||||
|
||||
# --- [5/5] the privileged tests may not skip in silence ----------------------
|
||||
# THE HOLE THIS CLOSES. Some tests can only prove what they claim against a real
|
||||
# kernel: shater/generate's TestIntegrationL3TunInboundStarts opens /dev/net/tun
|
||||
# and stands a real engine on it, TestIntegrationL3EgressICMPIsAFlow binds a real
|
||||
# socket to a real device. Both guard themselves with t.Skip when root or the
|
||||
# device is missing — the honest thing for a test to do, and completely INVISIBLE
|
||||
# above: `go test` prints `ok <pkg>` whether they ran or skipped, so [2/5]'s
|
||||
# per-package `ok` check is satisfied either way and the gate closes by claiming
|
||||
# it "passes every test we own". That is precisely the failure this whole script
|
||||
# was written for (115 of 116 test files never running while CI stayed green),
|
||||
# one level down and harder to see.
|
||||
#
|
||||
# The list is DISCOVERED, not hand-kept — `go test -list` over the same ROOTS —
|
||||
# so a privileged test written next month joins this check on the day it is
|
||||
# named, with no edit here. It keys on the ^TestIntegration prefix, already this
|
||||
# fork's marker for "needs capabilities the ordinary gate lacks" (SKIP_COMMON
|
||||
# above excludes common/tlsspoof's TestIntegration* for exactly that reason).
|
||||
# Name a privileged test anything else and it is invisible again — so don't.
|
||||
#
|
||||
# Verdicts, per test, by name:
|
||||
# RAN — it executed here; printed so that is visible rather than assumed.
|
||||
# FAILED — fatal, like any other failure.
|
||||
# MISSING — `go test -list` named it and the run produced no verdict for it:
|
||||
# fatal. A test that vanished between listing and running is the
|
||||
# same class of hole as one hidden by a build tag.
|
||||
# SKIPPED while this environment HAS root and /dev/net/tun — fatal. The
|
||||
# capability guard cannot be what skipped it, so something else did
|
||||
# and only the test knows what.
|
||||
# SKIPPED because the environment genuinely cannot run it — reported loudly,
|
||||
# by name, and it REPLACES the closing banner, so the last line of
|
||||
# the gate can never claim coverage it does not have. Deliberately
|
||||
# not fatal by default: the act_runner is an LXC guest whose kernel
|
||||
# has no tun module at all (checked 2026-07-26 on 10.10.10.211 —
|
||||
# `modprobe tun` answers "Module tun not found", /dev/net does not
|
||||
# exist, and act_runner runs job containers with privileged:false
|
||||
# and no container.options), so the device cannot be handed down
|
||||
# without reconfiguring the Proxmox host. Making it fatal would
|
||||
# paint CI permanently red and teach everyone to ignore the gate.
|
||||
# SHATER_REQUIRE_PRIVILEGED=1 makes it fatal for the runs that can.
|
||||
echo "== [5/5] the privileged tests (^TestIntegration) produced a verdict by name =="
|
||||
PRIV_RE='^TestIntegration'
|
||||
PRIV_UNVERIFIED=""
|
||||
# -ldflags is NOT optional on the discovery call either: `go test -list` LINKS
|
||||
# each test binary before it can enumerate its tests, and without
|
||||
# -checklinkname=0 every package that pulls common/badtls fails to link. The
|
||||
# first cut of this step omitted it, swallowed the error with `2>/dev/null ||
|
||||
# true`, and reported "none declared" — a check against silent skipping that was
|
||||
# itself silently skipping. Hence also: the exit status is inspected, and an
|
||||
# empty list is only ever reported after a SUCCESSFUL enumeration.
|
||||
set +e
|
||||
priv_expect_raw="$(go test -list "$PRIV_RE" \
|
||||
-tags "$SHATER_ROUTER_TAGS" -ldflags "$SHATER_ROUTER_LDFLAGS" "${ROOTS[@]}" 2>&1)"
|
||||
priv_list_rc=$?
|
||||
set -e
|
||||
priv_expect="$(grep -E "$PRIV_RE" <<<"$priv_expect_raw" | sort -u || true)"
|
||||
if [ "$priv_list_rc" -ne 0 ]; then
|
||||
echo " FAILED [privileged]: could not enumerate the privileged tests (go test -list exited $priv_list_rc)." >&2
|
||||
echo " An unreadable list is NOT an empty list — this check refuses to" >&2
|
||||
echo " report 'nothing to verify' on the strength of a failed command." >&2
|
||||
sed 's/^/ /' <<<"$priv_expect_raw" | grep -vE '^\s+(ok|\?)\s' >&2 || true
|
||||
FAILED=1
|
||||
elif [ -z "$priv_expect" ]; then
|
||||
echo " none declared under the fork's trees — nothing to verify"
|
||||
else
|
||||
priv_capable=0
|
||||
if [ "$(id -u)" = "0" ] && [ -e /dev/net/tun ]; then
|
||||
priv_capable=1
|
||||
fi
|
||||
echo " declared: $(wc -l <<<"$priv_expect" | tr -d ' ')"
|
||||
echo " this environment: uid=$(id -u), /dev/net/tun $([ -e /dev/net/tun ] && echo present || echo MISSING) => can run them: $([ "$priv_capable" -eq 1 ] && echo yes || echo NO)"
|
||||
set +e
|
||||
go test -count=1 -v -run "$PRIV_RE" \
|
||||
-tags "$SHATER_ROUTER_TAGS" -ldflags "$SHATER_ROUTER_LDFLAGS" \
|
||||
"${ROOTS[@]}" >"$LOG" 2>&1
|
||||
priv_rc=$?
|
||||
set -e
|
||||
# The verdict lines plus whatever reason the test printed just before them,
|
||||
# so a skip is readable here and not just counted.
|
||||
grep -E '^(--- (PASS|SKIP|FAIL): |[[:space:]]+[^[:space:]]+\.go:[0-9]+: )' "$LOG" \
|
||||
| sed 's/^/ | /' || true
|
||||
priv_bad=0
|
||||
while read -r name; do
|
||||
[ -n "$name" ] || continue
|
||||
if grep -qE "^--- PASS: ${name}([[:space:]]|\$)" "$LOG"; then
|
||||
echo " RAN $name"
|
||||
elif grep -qE "^--- FAIL: ${name}([[:space:]]|\$)" "$LOG"; then
|
||||
echo " FAILED $name" >&2
|
||||
priv_bad=1
|
||||
elif grep -qE "^--- SKIP: ${name}([[:space:]]|\$)" "$LOG"; then
|
||||
if [ "$priv_capable" -eq 1 ]; then
|
||||
echo " SKIPPED $name — but this environment HAS root and /dev/net/tun, so the capability guard is NOT what skipped it" >&2
|
||||
priv_bad=1
|
||||
else
|
||||
echo " DID NOT RUN $name — skipped: no root and/or no /dev/net/tun here"
|
||||
PRIV_UNVERIFIED="$PRIV_UNVERIFIED $name"
|
||||
fi
|
||||
else
|
||||
echo " MISSING $name — go test -list named it, the run produced no verdict for it" >&2
|
||||
priv_bad=1
|
||||
fi
|
||||
done <<<"$priv_expect"
|
||||
if [ "$priv_rc" -ne 0 ] && [ "$priv_bad" -eq 0 ]; then
|
||||
echo " FAILED [privileged]: go test exited $priv_rc with every named test accounted for —" >&2
|
||||
echo " a build or package-level failure, see the log above." >&2
|
||||
priv_bad=1
|
||||
fi
|
||||
if [ "$priv_bad" -ne 0 ]; then
|
||||
FAILED=1
|
||||
fi
|
||||
fi
|
||||
echo
|
||||
|
||||
@@ -246,4 +391,27 @@ if [ "$FAILED" -ne 0 ]; then
|
||||
echo "== TEST GATE FAILED — nothing may be published from this run. ==" >&2
|
||||
exit 1
|
||||
fi
|
||||
if [ -n "$PRIV_UNVERIFIED" ]; then
|
||||
echo "== !! PASSED, BUT NOT FULLY VERIFIED !! =================================="
|
||||
echo " Every test that COULD run here passed. These did not run at all:"
|
||||
for t in $PRIV_UNVERIFIED; do
|
||||
echo " - $t"
|
||||
done
|
||||
echo
|
||||
echo " They need root + CAP_NET_ADMIN + /dev/net/tun, which this environment"
|
||||
echo " does not have. Nothing about the kernel paths they cover was verified"
|
||||
echo " by this run. To actually run them, from a host whose docker can:"
|
||||
echo " scripts/run-tests.sh # the re-exec hands the container both"
|
||||
echo " or directly:"
|
||||
echo " docker run --rm --cap-add NET_ADMIN --device /dev/net/tun \\"
|
||||
echo " -v \"\$PWD\":/src -w /src golang:1.26 bash scripts/run-tests.sh"
|
||||
echo " or on the OpenWrt VM. SHATER_REQUIRE_PRIVILEGED=1 makes this a hard"
|
||||
echo " failure instead of this notice."
|
||||
echo "=========================================================================="
|
||||
if [ "${SHATER_REQUIRE_PRIVILEGED:-0}" = "1" ]; then
|
||||
echo "== TEST GATE FAILED: SHATER_REQUIRE_PRIVILEGED=1 and the tests above did not run. ==" >&2
|
||||
exit 1
|
||||
fi
|
||||
exit 0
|
||||
fi
|
||||
echo "== OK: the shipped tag set, on linux, passes every test we own. =="
|
||||
|
||||
+62
-4
@@ -805,6 +805,10 @@ var applyHoldNft = netplane.ApplyNft
|
||||
var (
|
||||
tableExists = netplane.TableExists
|
||||
bootArmorPresent = netplane.BootArmorPresent
|
||||
// teardownNft is a seam for the same reason: whether the table is REMOVED or
|
||||
// REPLACED on the way out is the whole of the restart-gap fix, and it can
|
||||
// otherwise only be observed on a router with a real nft.
|
||||
teardownNft = netplane.TeardownNft
|
||||
)
|
||||
|
||||
// Holding reports whether forwarded LAN traffic is currently being BLOCKED by a
|
||||
@@ -967,7 +971,42 @@ func (a *Applier) Reconcile() (changed bool, err error) {
|
||||
|
||||
// Teardown is the honest teardown: engine.Close + netplane routing/nft teardown +
|
||||
// clear ACTIVE_FLAG, under the flock. Safe to call when nothing is up (idempotent).
|
||||
func (a *Applier) Teardown() error {
|
||||
//
|
||||
// This is the "everything goes" form — the operator disabled the stack or stopped
|
||||
// the service. A process that is being REPLACED wants TeardownExiting instead.
|
||||
func (a *Applier) Teardown() error { return a.teardown(nil) }
|
||||
|
||||
// TeardownExiting is Teardown for a daemon that is going away, with the one
|
||||
// ordering that never leaves the LAN uncovered.
|
||||
//
|
||||
// THE GAP THIS CLOSES (MEASURED on the stand: 80-90 ms, twice). The exit path used
|
||||
// to be `Teardown(); armOnExit()` — TeardownNft DELETED the table, and only then
|
||||
// was the fail-closed holding plane installed. Two nft transactions, and between
|
||||
// them the `inet shater` table does not exist at all, so fw4's `lan -> wan ACCEPT`
|
||||
// is the only policy on the box and the whole LAN forwards in the clear. That is
|
||||
// not a boot-time window: it is every `restart`, every `reload_service` (i.e.
|
||||
// every LuCI Save & Apply) and every package upgrade. The width is two `nft`
|
||||
// invocations, so it does NOT grow with the engine — eng.Close runs before the
|
||||
// table is touched — but it is the whole LAN, in the clear, every time.
|
||||
//
|
||||
// The comment that used to sit on the call site — "AFTER the teardown, never
|
||||
// before: Teardown deletes the table, so a plane installed first would simply be
|
||||
// removed again" — described the mechanism correctly and drew the wrong conclusion
|
||||
// from it: the answer is not to arm later, it is to stop deleting.
|
||||
//
|
||||
// So arm FIRST and then skip the delete. RenderHoldNft's output is a single
|
||||
// `nft -f` script that opens with `table inet shater` / `delete table inet shater`
|
||||
// / `table inet shater { ... }` — one netlink transaction, in which the table is
|
||||
// REPLACED rather than removed and re-added. The kernel never observes its
|
||||
// absence, so a sampler cannot either.
|
||||
//
|
||||
// arm reports whether it actually installed a plane. When it did NOT — a real
|
||||
// `stop`, or a handoff with kill_switch=open, where fail-open is the operator's
|
||||
// documented choice — the table is removed exactly as before. "Keep the table"
|
||||
// therefore follows from "a plane is standing", never from the caller's intent.
|
||||
func (a *Applier) TeardownExiting(arm func() bool) error { return a.teardown(arm) }
|
||||
|
||||
func (a *Applier) teardown(arm func() bool) error {
|
||||
release, err := lockExclusive()
|
||||
if err != nil {
|
||||
return err
|
||||
@@ -976,6 +1015,15 @@ func (a *Applier) Teardown() error {
|
||||
a.mu.Lock()
|
||||
defer a.mu.Unlock()
|
||||
|
||||
// Install the successor plane BEFORE anything is dismantled, and do it while
|
||||
// still holding the apply lock, so a concurrent apply cannot slip between the
|
||||
// swap and the teardown. arm must not call back into the Applier (cmd/shaterd's
|
||||
// armOnExit goes straight to model + netplane) or this deadlocks.
|
||||
kept := false
|
||||
if arm != nil {
|
||||
kept = arm()
|
||||
}
|
||||
|
||||
// Stop the observatory BEFORE closing the engine, and wait for an in-flight
|
||||
// tick: a teardown must not leave probe dials racing a box that is going away.
|
||||
a.eng.StopObservatory()
|
||||
@@ -998,8 +1046,14 @@ func (a *Applier) Teardown() error {
|
||||
if err := netplane.TeardownRouting(m); err != nil && firstErr == nil {
|
||||
firstErr = err
|
||||
}
|
||||
if err := netplane.TeardownNft(); err != nil && firstErr == nil {
|
||||
firstErr = err
|
||||
// The policy routing above is safe to remove either way: the holding plane is a
|
||||
// single `forward` chain of accepts and drops and consults no routing table, so
|
||||
// it keeps working with the ip rules gone. The TABLE is the one thing that must
|
||||
// not be removed out from under it.
|
||||
if !kept {
|
||||
if err := teardownNft(); err != nil && firstErr == nil {
|
||||
firstErr = err
|
||||
}
|
||||
}
|
||||
// Put the per-ingress-iface knobs back the way we found them. With the table
|
||||
// and the policy routing gone, a lingering accept_local=1 / rp_filter=0 on a
|
||||
@@ -1011,7 +1065,11 @@ func (a *Applier) Teardown() error {
|
||||
clearActiveFlag(a.log)
|
||||
a.lastGood = nil
|
||||
a.lastNft = ""
|
||||
a.setHolding(false)
|
||||
// Not a blanket false any more: with a holding plane standing, forwarded LAN
|
||||
// traffic really IS being blocked, and saying otherwise here is the inverted lie
|
||||
// Holding()'s doc comment is about — the process is exiting, but Status can
|
||||
// still be read over the control socket before it does.
|
||||
a.setHolding(kept)
|
||||
a.setTraffic(generate.Traffic{})
|
||||
a.setWarnings(nil)
|
||||
// The plane is gone, so the logged set no longer describes anything. Forget it,
|
||||
|
||||
@@ -0,0 +1,149 @@
|
||||
package apply
|
||||
|
||||
// The exit path must never leave the LAN uncovered, and "never" is an ORDER, not
|
||||
// an intention.
|
||||
//
|
||||
// WHAT THIS PINS. The daemon's SIGTERM path used to be:
|
||||
//
|
||||
// applier.Teardown() // netplane.TeardownNft() -> `nft delete table inet shater`
|
||||
// armOnExit(handoff) // then, separately, install the fail-closed holding plane
|
||||
//
|
||||
// Two nft transactions. Between them the `inet shater` table does not exist, so
|
||||
// fw4's `lan -> wan ACCEPT` is the only policy on the box and every forwarded LAN
|
||||
// packet leaves in the clear. MEASURED on the stand at 80-90 ms, reproduced twice
|
||||
// with a 35 000-sample run at ~1.3 ms resolution — and this is not a boot-time
|
||||
// window that heals itself: it is every `restart`, every `reload_service` (i.e.
|
||||
// every LuCI Save & Apply), and every package upgrade.
|
||||
//
|
||||
// The old call site even carried a comment explaining the mechanism — "AFTER the
|
||||
// teardown, never before: Teardown deletes the table, so a plane installed first
|
||||
// would simply be removed again" — and drew the wrong conclusion from a correct
|
||||
// observation. The fix is not to arm later, it is to stop deleting: arm first (a
|
||||
// single `nft -f` that opens with `delete table` and closes with the new one, so
|
||||
// the kernel replaces rather than removes), then skip the delete.
|
||||
//
|
||||
// So the property under test is a SEQUENCE, and the test records the order the
|
||||
// two seams are called in. A test that only asserted "the table still exists at
|
||||
// the end" would pass against the broken code.
|
||||
|
||||
import (
|
||||
"errors"
|
||||
"testing"
|
||||
|
||||
"github.com/sagernet/sing-box/shater/engine"
|
||||
)
|
||||
|
||||
// recordTeardownSeams captures the order in which the exit path touches the
|
||||
// kernel: "arm" when a holding plane is installed, "delete" when the table is
|
||||
// removed.
|
||||
func recordTeardownSeams(t *testing.T) (*[]string, func()) {
|
||||
t.Helper()
|
||||
var calls []string
|
||||
orig := teardownNft
|
||||
teardownNft = func() error {
|
||||
calls = append(calls, "delete")
|
||||
return nil
|
||||
}
|
||||
return &calls, func() { teardownNft = orig }
|
||||
}
|
||||
|
||||
// TestTeardownExitingReplacesThePlaneInsteadOfRemovingIt is the regression: with a
|
||||
// successor coming, the table must be swapped and NEVER deleted.
|
||||
func TestTeardownExitingReplacesThePlaneInsteadOfRemovingIt(t *testing.T) {
|
||||
calls, restore := recordTeardownSeams(t)
|
||||
defer restore()
|
||||
|
||||
a := New(engine.New(), nil)
|
||||
if err := a.TeardownExiting(func() bool {
|
||||
*calls = append(*calls, "arm")
|
||||
return true
|
||||
}); err != nil {
|
||||
t.Fatalf("TeardownExiting: %v", err)
|
||||
}
|
||||
|
||||
if len(*calls) != 1 || (*calls)[0] != "arm" {
|
||||
t.Fatalf("exit path did %v, want exactly [arm]: the holding plane must be installed "+
|
||||
"and the table must NOT be deleted — a `delete` here is the 80-90 ms window in which "+
|
||||
"fw4's lan->wan ACCEPT is the only policy on the box", *calls)
|
||||
}
|
||||
if !a.Holding() && tableExists == nil {
|
||||
t.Errorf("unreachable; keeps the linter honest about the seam")
|
||||
}
|
||||
}
|
||||
|
||||
// TestTeardownExitingArmsBeforeItTearsDown pins the ORDER even in the case where
|
||||
// the table does still get removed. Arming has to be the first thing that touches
|
||||
// the kernel; if it ran after the delete we would be back to the two-transaction
|
||||
// gap with extra steps.
|
||||
func TestTeardownExitingArmsBeforeItTearsDown(t *testing.T) {
|
||||
calls, restore := recordTeardownSeams(t)
|
||||
defer restore()
|
||||
|
||||
a := New(engine.New(), nil)
|
||||
// arm reports FALSE: nothing was installed (a render failure, or kill_switch=open
|
||||
// where fail-open is the operator's documented choice). The table must then come
|
||||
// down exactly as it always did.
|
||||
if err := a.TeardownExiting(func() bool {
|
||||
*calls = append(*calls, "arm")
|
||||
return false
|
||||
}); err != nil {
|
||||
t.Fatalf("TeardownExiting: %v", err)
|
||||
}
|
||||
|
||||
if len(*calls) != 2 || (*calls)[0] != "arm" || (*calls)[1] != "delete" {
|
||||
t.Fatalf("exit path did %v, want [arm delete]: arming must precede the delete, and a "+
|
||||
"plane that was NOT installed must not keep the table alive", *calls)
|
||||
}
|
||||
}
|
||||
|
||||
// TestTeardownStillRemovesEverything is the escape hatch. A deliberate `stop`, and
|
||||
// a Reconcile that finds the stack disabled, both come through the plain Teardown
|
||||
// and must dismantle the plane completely — a kill switch that cannot be switched
|
||||
// off is a brick.
|
||||
func TestTeardownStillRemovesEverything(t *testing.T) {
|
||||
calls, restore := recordTeardownSeams(t)
|
||||
defer restore()
|
||||
|
||||
a := New(engine.New(), nil)
|
||||
if err := a.Teardown(); err != nil {
|
||||
t.Fatalf("Teardown: %v", err)
|
||||
}
|
||||
if len(*calls) != 1 || (*calls)[0] != "delete" {
|
||||
t.Fatalf("Teardown did %v, want [delete]: an operator's stop must take the table with it", *calls)
|
||||
}
|
||||
}
|
||||
|
||||
// TestTeardownExitingReportsHolding pins the honesty half: with a plane left
|
||||
// standing the Applier must not go on saying it installed nothing. Status can
|
||||
// still be read over the control socket between the swap and the exit, and
|
||||
// "holding=false over a blocked LAN" is the inverted lie holdstate_test.go is
|
||||
// about, just reached down a different path.
|
||||
func TestTeardownExitingReportsHolding(t *testing.T) {
|
||||
_, restore := recordTeardownSeams(t)
|
||||
defer restore()
|
||||
restoreFacts := stubPlaneFacts(t, true, true)
|
||||
defer restoreFacts()
|
||||
|
||||
a := New(engine.New(), nil)
|
||||
if err := a.TeardownExiting(func() bool { return true }); err != nil {
|
||||
t.Fatalf("TeardownExiting: %v", err)
|
||||
}
|
||||
if !a.Holding() {
|
||||
t.Errorf("Holding() = false right after the exit path left a fail-closed plane standing")
|
||||
}
|
||||
}
|
||||
|
||||
// TestTeardownExitingSurvivesANilArm keeps the plain-Teardown contract explicit:
|
||||
// a nil arm is "nothing to install", not a panic.
|
||||
func TestTeardownExitingSurvivesANilArm(t *testing.T) {
|
||||
calls, restore := recordTeardownSeams(t)
|
||||
defer restore()
|
||||
|
||||
a := New(engine.New(), nil)
|
||||
if err := a.TeardownExiting(nil); err != nil && !errors.Is(err, nil) {
|
||||
t.Fatalf("TeardownExiting(nil): %v", err)
|
||||
}
|
||||
if len(*calls) != 1 || (*calls)[0] != "delete" {
|
||||
t.Fatalf("TeardownExiting(nil) did %v, want [delete]", *calls)
|
||||
}
|
||||
}
|
||||
@@ -360,6 +360,128 @@ func untunnelablePolicyWarnings(g model.Globals, planNotes []string) []Warning {
|
||||
return append(out, Warning{Severity: SeverityInfo, Section: section, Name: name, Message: msg})
|
||||
}
|
||||
|
||||
// The egress carrier owns the whole story the moment untunnelable_egress
|
||||
// names an egress: the carried protocols get that egress's own mark in
|
||||
// prerouting and the ROUTING decision sends them out its device with the
|
||||
// kernel's NAT — the forward chain, where the policy's verdicts live, no
|
||||
// longer decides their fate. So this branch sits above every other and
|
||||
// returns its own text; with the option empty, the notes below are
|
||||
// byte-for-byte what they were. Two phrasing rules here are load-bearing.
|
||||
// First, the egress is NEVER called a tunnel unconditionally: the option
|
||||
// accepts any interface/tunnel egress, and on the routers this ships to
|
||||
// that is at least as often a second WAN — another uplink, whose real
|
||||
// address the far end sees — as a WireGuard device. Second, the note must
|
||||
// say out loud that UDP-based VPNs are none of this option's business: an
|
||||
// operator who reads "VPN passthrough" and enables it for a WireGuard
|
||||
// client that already worked through the ordinary tunnel has been misled,
|
||||
// not helped. What the policy still owns is exactly the failure path — a
|
||||
// name that resolves to no interface/tunnel egress, a rule or route that
|
||||
// did not come up — and each tail below says what that failure looks like,
|
||||
// because under `direct` (or an open kill switch) it is a silent leak
|
||||
// through the normal uplink with the real address, and nothing anywhere
|
||||
// else would say so.
|
||||
if egressName := strings.TrimSpace(g.UntunnelableEgress); egressName != "" {
|
||||
var msg, failure string
|
||||
if netplane.L3Enabled(g) {
|
||||
msg = "Ping and Windows tracert keep travelling THROUGH the tunnel, toward every " +
|
||||
"address your rules send to an outbound that can carry plain IP " +
|
||||
"(WireGuard/AmneziaWG) — the L3 ingress claims ICMP before this option is " +
|
||||
"consulted, and addresses your rules send anywhere else " +
|
||||
"(vless/vmess/trojan/shadowsocks and the like) still cannot be pinged at " +
|
||||
"all, deliberately. Everything else the proxy cannot carry — IPsec " +
|
||||
"(ESP/AH), PPTP/GRE, SCTP and every other protocol that is neither TCP " +
|
||||
"nor UDP — now leaves through egress \"" + egressName + "\": the kernel " +
|
||||
"routes it out that interface with that interface's own NAT, and none of " +
|
||||
"it goes through the proxy or follows your routing rules. "
|
||||
failure = "when one of the routes is not in place — the L3 route for ping, the " +
|
||||
"egress route for the rest (a name that matches no interface/tunnel " +
|
||||
"egress, or a rule or route that failed to come up): "
|
||||
} else {
|
||||
msg = "Ping, Windows tracert, IPsec (ESP/AH), PPTP/GRE, SCTP and every other " +
|
||||
"protocol that is neither TCP nor UDP now leave through egress \"" +
|
||||
egressName + "\": the kernel routes them out that interface with that " +
|
||||
"interface's own NAT, and none of it goes through the proxy or follows " +
|
||||
"your routing rules — the hops tracert prints are that interface's path " +
|
||||
"(on Linux and macOS traceroute sends UDP probes instead, which still " +
|
||||
"follow your rules). "
|
||||
failure = "when the egress route is not in place (a name that matches no " +
|
||||
"interface/tunnel egress, or a rule or route that failed to come up): "
|
||||
}
|
||||
msg += "What that buys depends entirely on what the interface IS: a WireGuard " +
|
||||
"interface really is a tunnel, but a second WAN is not — it is just another " +
|
||||
"uplink, and the host on the far end sees that uplink's real address. Two " +
|
||||
"things this option does NOT do: multicast IPTV does not pass this router " +
|
||||
"under any setting, and carrying IGMP out an egress cannot change that; and " +
|
||||
"VPNs that run over UDP (WireGuard, OpenVPN-UDP, IPsec through NAT — IKE on " +
|
||||
"UDP 500, NAT-T on UDP 4500) never needed it: they are ordinary tunnelled " +
|
||||
"traffic, keep following your routing rules exactly as before, and gain " +
|
||||
"nothing from this option. The `untunnelable` policy no longer decides this " +
|
||||
"traffic's fate — routing settles it before the forward chain gets a say — " +
|
||||
"and answers only for failure, " + failure
|
||||
switch {
|
||||
case !killSwitchClosed(g):
|
||||
msg += "with the kill switch open nothing is dropped, so whatever loses its " +
|
||||
"route quietly leaves through your normal uplink with your real IP address."
|
||||
case policy == netplane.UntunnelableDirect:
|
||||
msg += "\"direct\" quietly lets it leave through your normal uplink with " +
|
||||
"your real IP address."
|
||||
case policy == netplane.UntunnelableICMP:
|
||||
msg += "\"icmp\" drops it, excepting only ping — which then quietly leaves " +
|
||||
"with your real IP address instead of failing."
|
||||
default:
|
||||
msg += "\"block\" drops it — an honest loss rather than a silent leak."
|
||||
}
|
||||
return note(egressName, msg)
|
||||
}
|
||||
|
||||
// The L3 ingress rewrites the ICMP half of every note below, so it gets one
|
||||
// text of its own rather than four patched variants: echo is marked in
|
||||
// prerouting and the ROUTING decision carries it into the engine's TUN before
|
||||
// the forward chain — where the policy accepts and the kill-switch drops
|
||||
// live — is ever consulted. That holds under all three policy values and
|
||||
// with the kill switch open alike, which is why this branch sits above the
|
||||
// kill-switch note: "reaches the internet with your real IP address" stops
|
||||
// being true for ping the moment the divert exists. What the policy still
|
||||
// owns is exactly two things, and both are said: the protocols the engine
|
||||
// cannot ingest at all (raw IPsec, PPTP/GRE), and the fallback path a marked
|
||||
// packet takes when the L3 route failed to install — under `direct` (or an
|
||||
// open kill switch) that failure is a SILENT leak with the real address,
|
||||
// under `block` an honest packet loss. The unpingable-through-proxy sentence
|
||||
// is deliberate too: those pings used to be answered by the router itself,
|
||||
// and a fake "alive" is worse than a truthful timeout.
|
||||
if netplane.L3Enabled(g) {
|
||||
msg := "Ping and Windows tracert work and travel THROUGH the tunnel, toward every " +
|
||||
"address your rules send to an outbound that can carry plain IP " +
|
||||
"(WireGuard/AmneziaWG). Addresses your rules send anywhere else " +
|
||||
"(vless/vmess/trojan/shadowsocks and the like) cannot be pinged at all — " +
|
||||
"deliberately: those pings used to be answered by the router itself, reporting " +
|
||||
"hosts alive it had never reached. The hops tracert prints are the tunnel's " +
|
||||
"path, not your own, and IPv6 traceroute shows only the destination, none of " +
|
||||
"the hops on the way. Raw VPN passthrough (IPsec ESP/AH, PPTP/GRE) cannot " +
|
||||
"enter the tunnel at all and stays with the untunnelable policy: "
|
||||
switch {
|
||||
case !killSwitchClosed(g):
|
||||
msg += "with the kill switch open none of it is dropped, so it leaves with your " +
|
||||
"real IP address — and if the L3 route ever fails to come up, ping quietly " +
|
||||
"does the same instead of failing."
|
||||
case policy == netplane.UntunnelableDirect:
|
||||
msg += "\"direct\" lets it out with your real IP address — and if the L3 route " +
|
||||
"ever fails to come up, ping quietly does the same instead of failing."
|
||||
case policy == netplane.UntunnelableICMP:
|
||||
msg += "\"icmp\" drops it, excepting only echo — which now rides the tunnel " +
|
||||
"anyway, so the exception matters just once: if the L3 route ever fails to " +
|
||||
"come up, it lets ping quietly leave with your real IP address instead of " +
|
||||
"failing."
|
||||
default:
|
||||
msg += "\"block\" drops it — and if the L3 route ever fails to come up, ping " +
|
||||
"fails outright rather than leaking."
|
||||
}
|
||||
msg += " VPNs that run over UDP (WireGuard, OpenVPN-UDP, IPsec through NAT) are " +
|
||||
"ordinary tunnelled traffic and are unaffected either way. Multicast IPTV does " +
|
||||
"not pass this router on any setting; the L3 ingress does not change that."
|
||||
return note(policy, msg)
|
||||
}
|
||||
|
||||
// With the kill switch open the forward chain has no drops at all, so nothing
|
||||
// is restricted whatever the policy says. Saying that is more useful than
|
||||
// repeating a promise which is not being kept.
|
||||
|
||||
+34
-12
@@ -180,44 +180,59 @@ func refreshBootArmor(m *model.Model, logger log.ContextLogger) {
|
||||
}
|
||||
}
|
||||
|
||||
// armFromSnapshot reinstates the persisted holding plane. why is a short phrase
|
||||
// for the log ("config is unreadable", "restart handoff").
|
||||
func armFromSnapshot(why string, logger log.ContextLogger) {
|
||||
// armFromSnapshot reinstates the persisted holding plane and reports whether a
|
||||
// plane is now standing. why is a short phrase for the log ("config is
|
||||
// unreadable", "restart handoff").
|
||||
func armFromSnapshot(why string, logger log.ContextLogger) bool {
|
||||
loaded, err := netplane.LoadBootArmor()
|
||||
switch {
|
||||
case err != nil:
|
||||
logger.Error("FAIL-CLOSED PLANE NOT INSTALLED (", why, "): the saved plane ",
|
||||
netplane.BootArmorPath, " could not be loaded: ", err,
|
||||
" — LAN traffic may be reaching the WAN unprotected")
|
||||
return false
|
||||
case loaded:
|
||||
logger.Error("fail-closed plane reinstated from ", netplane.BootArmorPath,
|
||||
" (", why, "): LAN->WAN forwarding is BLOCKED. ",
|
||||
"SSH, LuCI and the admin panel remain reachable.")
|
||||
return true
|
||||
default:
|
||||
logger.Warn("no saved fail-closed plane at ", netplane.BootArmorPath, " (", why,
|
||||
"): nothing was installed")
|
||||
return false
|
||||
}
|
||||
}
|
||||
|
||||
// armFromModel renders the holding plane for m and installs it.
|
||||
func armFromModel(m *model.Model, why string, logger log.ContextLogger) {
|
||||
// armFromModel renders the holding plane for m, installs it, and reports whether
|
||||
// a plane is now standing.
|
||||
//
|
||||
// The return value is load-bearing on the exit path: Applier.TeardownExiting keeps
|
||||
// the nft table only when a plane really was installed, so a render failure or a
|
||||
// fail-open config falls back to the old remove-everything behaviour instead of
|
||||
// leaving whatever the engine happened to have in the kernel.
|
||||
func armFromModel(m *model.Model, why string, logger log.ContextLogger) bool {
|
||||
ruleset, err := netplane.RenderHoldNft(m)
|
||||
if err != nil {
|
||||
logger.Error("FAIL-CLOSED PLANE NOT INSTALLED (", why, "): render failed: ", err)
|
||||
return
|
||||
return false
|
||||
}
|
||||
if ruleset == "" {
|
||||
// No divert devices: there is nothing this plane would protect.
|
||||
return
|
||||
return false
|
||||
}
|
||||
// One `nft -f` that opens with `delete table` and closes with the new table:
|
||||
// the swap is a single netlink transaction, so this REPLACES whatever plane is
|
||||
// loaded without the table ever being absent. That property is why the exit
|
||||
// path can arm before it tears down.
|
||||
if err := netplane.ApplyNft(ruleset); err != nil {
|
||||
logger.Error("FAIL-CLOSED PLANE NOT INSTALLED (", why, "): ", err,
|
||||
" — LAN traffic may be reaching the WAN unprotected")
|
||||
return
|
||||
return false
|
||||
}
|
||||
logger.Info("fail-closed plane left in place (", why,
|
||||
"): LAN->WAN forwarding stays BLOCKED until the next daemon applies. ",
|
||||
"SSH, LuCI and the admin panel remain reachable.")
|
||||
return true
|
||||
}
|
||||
|
||||
// armOnUnreadableConfig is the window-3 answer: the daemon is up but cannot read
|
||||
@@ -247,13 +262,20 @@ func armOnUnreadableConfig(logger log.ContextLogger) {
|
||||
var errUnreadableConfig = os.ErrInvalid
|
||||
|
||||
// armOnExit is the window-2 answer: what this daemon leaves in the kernel when it
|
||||
// is asked to go away. See planExitArmor for the policy.
|
||||
func armOnExit(handoff bool, logger log.ContextLogger) {
|
||||
// is asked to go away, and whether anything is now standing there. See
|
||||
// planExitArmor for the policy.
|
||||
//
|
||||
// It is called BY Applier.TeardownExiting, before the teardown and under the apply
|
||||
// lock, so that the plane is swapped rather than removed-then-rebuilt. It must
|
||||
// therefore never call back into the Applier — everything here goes straight to
|
||||
// model.ReadUCI and netplane.
|
||||
func armOnExit(handoff bool, logger log.ContextLogger) bool {
|
||||
m, err := model.ReadUCI()
|
||||
switch planExitArmor(handoff, m, err, netplane.BootArmorPresent()) {
|
||||
case armorRender:
|
||||
armFromModel(m, "restart handoff", logger)
|
||||
return armFromModel(m, "restart handoff", logger)
|
||||
case armorSnapshot:
|
||||
armFromSnapshot("restart handoff, config unreadable", logger)
|
||||
return armFromSnapshot("restart handoff, config unreadable", logger)
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
@@ -0,0 +1,169 @@
|
||||
// Executing the SHIPPED init script's action classification.
|
||||
//
|
||||
// WHY THIS TEST IS SHAPED LIKE THIS
|
||||
//
|
||||
// The boot-armor defect that shipped in v0.2.17 was not in any Go file. The Go
|
||||
// half was correct and fully covered: refreshBootArmor tracked desired state,
|
||||
// planExitArmor made the right call, LoadBootArmor validated before loading.
|
||||
// Every one of those tests was green while the feature did not work at all on
|
||||
// hardware, because the thing that broke it was one shell `case` in
|
||||
// /etc/init.d/shater whose default arm swept up procd's `shutdown` action — so
|
||||
// the arm token was deleted on the way down, every reboot, and the boot it
|
||||
// existed to protect always found no file.
|
||||
//
|
||||
// A unit test that cannot see the shell file cannot catch that, and a comment in
|
||||
// the shell file claiming `shutdown` is handled is precisely what shipped. So
|
||||
// this runs the real thing: it sources the actual packaged
|
||||
// openwrt/shater-core/files/etc/init.d/shater in /bin/sh and calls its two
|
||||
// classification predicates with every action procd actually uses.
|
||||
//
|
||||
// Sourcing the whole file is safe and deliberate — at top level it contains only
|
||||
// variable assignments and function definitions, nothing that touches the system —
|
||||
// and sourcing the WHOLE file is the point: a test that copy-pasted the `case`
|
||||
// would pass while the shipped script said something else.
|
||||
//
|
||||
// The action names are not invented. They were measured on the target
|
||||
// (ImmortalWrt 25.12.1 r37978) with a throwaway probe init script:
|
||||
//
|
||||
// /etc/init.d/X restart -> stop_service action=[restart]
|
||||
// /etc/init.d/X stop -> stop_service action=[stop]
|
||||
// /etc/init.d/X reload -> reload_service action=[reload]
|
||||
// `reboot` -> stop_service action=[shutdown]
|
||||
// the boot after it -> start_service action=[boot]
|
||||
package main
|
||||
|
||||
import (
|
||||
"os/exec"
|
||||
"path/filepath"
|
||||
"runtime"
|
||||
"strings"
|
||||
"testing"
|
||||
)
|
||||
|
||||
// initScriptPath is the packaged init script, relative to this package dir.
|
||||
const initScriptPath = "../../../openwrt/shater-core/files/etc/init.d/shater"
|
||||
|
||||
// askInitScript sources the init script in /bin/sh and reports whether fn
|
||||
// returns true for the given arguments.
|
||||
func askInitScript(t *testing.T, fn string, args ...string) bool {
|
||||
t.Helper()
|
||||
abs, err := filepath.Abs(initScriptPath)
|
||||
if err != nil {
|
||||
t.Fatalf("resolve %s: %v", initScriptPath, err)
|
||||
}
|
||||
// `. script` then call the predicate. `set -e` is deliberately NOT used: the
|
||||
// predicates report by exit status, and a false answer is not an error.
|
||||
script := `. "$1" || exit 3; shift; if ` + fn + ` "$@"; then echo yes; else echo no; fi`
|
||||
argv := append([]string{"-c", script, "sh", abs}, args...)
|
||||
out, err := exec.Command("/bin/sh", argv...).CombinedOutput()
|
||||
if err != nil {
|
||||
t.Fatalf("%s(%q): %v\n%s", fn, args, err, out)
|
||||
}
|
||||
switch strings.TrimSpace(string(out)) {
|
||||
case "yes":
|
||||
return true
|
||||
case "no":
|
||||
return false
|
||||
default:
|
||||
t.Fatalf("%s(%q): unreadable answer %q", fn, args, out)
|
||||
return false
|
||||
}
|
||||
}
|
||||
|
||||
// TestInitScriptActionClassification pins the two closed lists. The `shutdown`
|
||||
// rows are the regression: both must be false, because a reboot is neither a
|
||||
// handoff (nothing is coming) nor an operator switching the product off.
|
||||
func TestInitScriptActionClassification(t *testing.T) {
|
||||
if runtime.GOOS == "windows" {
|
||||
t.Skip("needs a POSIX /bin/sh; the gate runs on linux")
|
||||
}
|
||||
for _, tc := range []struct {
|
||||
action string
|
||||
disarms bool
|
||||
handoff bool
|
||||
why string
|
||||
}{
|
||||
{"stop", true, false, "the operator switched the product off"},
|
||||
{"shutdown", false, false, "REBOOT/POWEROFF — must not disarm; this is the boot the armor exists for"},
|
||||
{"restart", false, true, "a successor is coming"},
|
||||
{"reload", false, true, "Save & Apply is stop+start"},
|
||||
{"boot", false, false, "start side, never reaches stop_service"},
|
||||
{"start", false, false, "start side"},
|
||||
{"", false, false, "unknown/empty degrades to changing nothing"},
|
||||
{"enable", false, false, "not a lifecycle transition"},
|
||||
{"disable", false, false, "durable off, but handled by shater-armor's rc.d refusal, not here"},
|
||||
} {
|
||||
if got := askInitScript(t, "shater_action_disarms", tc.action); got != tc.disarms {
|
||||
t.Errorf("shater_action_disarms(%q) = %v, want %v (%s)", tc.action, got, tc.disarms, tc.why)
|
||||
}
|
||||
if got := askInitScript(t, "shater_action_handoff", tc.action); got != tc.handoff {
|
||||
t.Errorf("shater_action_handoff(%q) = %v, want %v (%s)", tc.action, got, tc.handoff, tc.why)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestInitScriptStopDisarmsOnlyForAPerson pins the rest of the decision: `stop`
|
||||
// disarms when a PERSON is behind it, or when the product is being removed — and
|
||||
// not when something is merely replacing it.
|
||||
//
|
||||
// base-files' default_prerm reaches stop_service as a plain `stop`:
|
||||
//
|
||||
// if [ "$PKG_UPGRADE" != "1" ]; then "$i" disable; fi
|
||||
// "$i" stop
|
||||
//
|
||||
// so two different intentions arrive as one action. The rc.d state separates them:
|
||||
// a removal has already run `disable`, a replacement has not.
|
||||
//
|
||||
// On this target an apk UPGRADE turns out never to run default_prerm at all
|
||||
// (no pre-upgrade script — verified with a real `apk fix --reinstall` while
|
||||
// sampling the armor file), so the upgrade rows below are defence-in-depth rather
|
||||
// than a reproduction. The removal rows are live behaviour.
|
||||
func TestInitScriptStopDisarmsOnlyForAPerson(t *testing.T) {
|
||||
if runtime.GOOS == "windows" {
|
||||
t.Skip("needs a POSIX /bin/sh; the gate runs on linux")
|
||||
}
|
||||
for _, tc := range []struct {
|
||||
action string
|
||||
inPkg string
|
||||
rcEnable string
|
||||
want bool
|
||||
why string
|
||||
}{
|
||||
{"stop", "0", "1", true, "an operator typed it — the escape hatch must keep working"},
|
||||
{"stop", "0", "0", true, "an operator typed it on an already-disabled service"},
|
||||
{"stop", "1", "1", false, "BEING REPLACED — prerm left the service enabled, so something is coming back"},
|
||||
{"stop", "1", "0", true, "REMOVAL — prerm already ran `disable`; the product is going away"},
|
||||
{"shutdown", "0", "1", false, "reboot never disarms"},
|
||||
{"shutdown", "1", "1", false, "reboot never disarms, package manager or not"},
|
||||
{"shutdown", "1", "0", false, "still a reboot; the action decides first"},
|
||||
{"restart", "0", "1", false, "a successor is coming"},
|
||||
{"restart", "1", "0", false, "a successor is coming; action decides before any state"},
|
||||
{"reload", "1", "1", false, "Save & Apply"},
|
||||
{"", "1", "0", false, "unknown action changes nothing"},
|
||||
} {
|
||||
got := askInitScript(t, "shater_stop_disarms", tc.action, tc.inPkg, tc.rcEnable)
|
||||
if got != tc.want {
|
||||
t.Errorf("shater_stop_disarms(%q, in_pkg=%s, rc_enabled=%s) = %v, want %v (%s)",
|
||||
tc.action, tc.inPkg, tc.rcEnable, got, tc.want, tc.why)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestInitScriptsParse is the cheapest possible guard against the class of bug
|
||||
// that no Go test can otherwise see: a shell file that ships syntactically
|
||||
// broken. An init script that fails to parse takes the whole service down and
|
||||
// `go build` is perfectly happy about it.
|
||||
func TestInitScriptsParse(t *testing.T) {
|
||||
if runtime.GOOS == "windows" {
|
||||
t.Skip("needs a POSIX /bin/sh; the gate runs on linux")
|
||||
}
|
||||
for _, name := range []string{"shater", "shater-armor", "shater-cron"} {
|
||||
p, err := filepath.Abs(filepath.Join(filepath.Dir(initScriptPath), name))
|
||||
if err != nil {
|
||||
t.Fatalf("resolve %s: %v", name, err)
|
||||
}
|
||||
if out, err := exec.Command("/bin/sh", "-n", p).CombinedOutput(); err != nil {
|
||||
t.Errorf("/etc/init.d/%s does not parse: %v\n%s", name, err, out)
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -436,12 +436,18 @@ func cmdRun() int {
|
||||
// operator's escape hatch and must keep working.
|
||||
handoff := restartHandoffPending()
|
||||
logger.Info("signal ", sig, ": honest teardown + exit (restart handoff: ", handoff, ")")
|
||||
if err := applier.Teardown(); err != nil {
|
||||
// BEFORE the teardown, not after. This used to read `Teardown(); armOnExit()`
|
||||
// on the reasoning that Teardown deletes the table so arming first would be
|
||||
// undone — true, and the wrong conclusion: it left a measured 80-90 ms window per
|
||||
// restart (80-90 ms measured) in which no `inet shater` table existed at all and fw4's
|
||||
// `lan -> wan ACCEPT` was the only policy on the box. TeardownExiting arms
|
||||
// first (one nft transaction that REPLACES the table) and then skips the
|
||||
// delete iff a plane really went in.
|
||||
if err := applier.TeardownExiting(func() bool {
|
||||
return armOnExit(handoff, logger)
|
||||
}); err != nil {
|
||||
logger.Error("teardown: ", err)
|
||||
}
|
||||
// AFTER the teardown, never before: Teardown deletes the table, so a plane
|
||||
// installed first would simply be removed again.
|
||||
armOnExit(handoff, logger)
|
||||
return 0
|
||||
}
|
||||
}
|
||||
|
||||
@@ -17,6 +17,8 @@
|
||||
// # What this package covers (Phase-2 MVP gate)
|
||||
//
|
||||
// - tproxy inbound (+ mixed/socks/dokodemo local listeners)
|
||||
// - the synthetic "l3-in" TUN inbound (globals l3_tunnel): the L3 ingress that
|
||||
// lets the engine carry ICMP, which kernel TPROXY cannot divert at all
|
||||
// - outbounds: vless, vmess, trojan, shadowsocks, hysteria2, tuic, shadowtls
|
||||
// with the shared TLS/Reality/uTLS container and ws/grpc/httpupgrade/http/
|
||||
// quic/xhttp transports, plus per-node multiplex
|
||||
|
||||
+145
-2
@@ -9,11 +9,15 @@ import (
|
||||
C "github.com/sagernet/sing-box/constant"
|
||||
"github.com/sagernet/sing-box/option"
|
||||
"github.com/sagernet/sing/common/auth"
|
||||
"github.com/sagernet/sing/common/json/badoption"
|
||||
|
||||
"github.com/sagernet/sing-box/shater/model"
|
||||
"github.com/sagernet/sing-box/shater/netplane"
|
||||
)
|
||||
|
||||
// buildInbounds maps every enabled model.Inbound to a typed sing-box inbound.
|
||||
// buildInbounds maps every enabled model.Inbound to a typed sing-box inbound,
|
||||
// then appends the synthetic L3-ingress TUN inbound when globals l3_tunnel
|
||||
// calls for one (see appendL3TunInbound).
|
||||
//
|
||||
// # What each Type really produces
|
||||
//
|
||||
@@ -87,7 +91,146 @@ func (b *builder) buildInbounds() []option.Inbound {
|
||||
seenTag[built.Tag] = true
|
||||
inbounds = append(inbounds, built)
|
||||
}
|
||||
return inbounds
|
||||
return b.appendL3TunInbound(inbounds, seenTag)
|
||||
}
|
||||
|
||||
// The L3-ingress TUN parameters below are a fixed contract with the netplane:
|
||||
// the nft prerouting chain stamps netplane.L3Mark on LAN traffic the tunnel can
|
||||
// only carry at layer 3, ApplyRouting points the L3 table's default route at
|
||||
// netplane.L3Device, and this package opens the engine's end of that device.
|
||||
// None of it is operator-tunable, which is why the inbound is synthesised from
|
||||
// globals instead of being a model.Inbound.
|
||||
const (
|
||||
// l3InboundTag is how route rules and diagnostics address the L3 ingress.
|
||||
l3InboundTag = "l3-in"
|
||||
// l3MTU is 65535 — the largest total length an IPv4 datagram can carry —
|
||||
// and that maximum IS the reason, not a round number.
|
||||
//
|
||||
// This MTU is NOT a tunnel budget. It decides exactly one thing: whether
|
||||
// the KERNEL fragments a packet on its way INTO the device. What the
|
||||
// engine then puts into the tunnel is sized against the OUTBOUND's own
|
||||
// MTU: sing-tun's ForwardDispatcher.forwardToPort measures every forwarded
|
||||
// packet against Port.PortMTU() — the WireGuard/AWG endpoint's MTU — and
|
||||
// either fragments to it (no DF) or answers a proper `fragmentation
|
||||
// needed` quoting it (DF). It already does that work correctly; it only
|
||||
// has to be handed a WHOLE packet to do it.
|
||||
//
|
||||
// The previous value, 1420, was the WireGuard payload budget copied one
|
||||
// layer too far out, and it was not merely useless — it was a forgery
|
||||
// generator. Anything bigger was fragmented by the kernel at this device,
|
||||
// and sing-tun's dispatcher returns on `parsed.fragment` BEFORE asking for
|
||||
// a routing verdict at all (flow_dispatch.go). The fragments then reached
|
||||
// the gVisor stack, which reassembled them and handed the echo to
|
||||
// ICMPForwarder.HandlePacket, whose installFlow demands an UNSPECIFIED
|
||||
// port address that a WireGuard endpoint never has — so it fell through
|
||||
// and ANSWERED THE ECHO ITSELF. Net effect: `ping -s 1392` was honest and
|
||||
// `ping -s 1393` was a lie told by the router. See D25.
|
||||
//
|
||||
// 65535 is chosen over any other large value because no IP datagram can
|
||||
// exceed it: the kernel therefore CANNOT fragment at this device, for any
|
||||
// packet, ever. Any smaller value leaves a band open and re-opens the bug.
|
||||
// It is also sing-box's own default TUN MTU on Linux
|
||||
// (protocol/tun/inbound.go), so it is a well-trodden value.
|
||||
//
|
||||
// It costs NO memory, and that was measured rather than assumed: three
|
||||
// paired runs of TestIntegrationL3TunInboundStarts under
|
||||
// -test.memprofilerate=1 (exact accounting, not sampled) allocate 5.41 /
|
||||
// 5.48 / 5.47 MB at 65535 against 5.76 / 5.46 / 5.70 MB at 1420, and a
|
||||
// -diff_base profile attributes every difference to netlink interface
|
||||
// enumeration, not to the MTU. Nothing in the read path scales with it:
|
||||
// gVisor reads through fdbased.BufConfig, which sing-tun's init pins to a
|
||||
// single 65535-byte view regardless of MTU, and fdbased keeps `mtu` only
|
||||
// to return it from MTU().
|
||||
//
|
||||
// Do NOT expect a saving from GSO either, tempting as the arithmetic is.
|
||||
// protocol/tun computes `enableGSO = stack == gvisor && mtu < 49152`, so
|
||||
// this MTU turns it off there — and then StartStateStart turns it back ON
|
||||
// unconditionally because an adapter.FlowOutbound exists in the config
|
||||
// (protocol/tun/inbound.go, the outbound/endpoint scan). The ~1.98 MB of
|
||||
// TCP/UDP GRO scaffolding is therefore present at BOTH MTUs; it is priced
|
||||
// by the presence of a flow-capable outbound, not by this number.
|
||||
l3MTU = 65535
|
||||
)
|
||||
|
||||
// l3Addr4/l3Addr6 are the device's point-to-point addresses — the kernel only
|
||||
// routes over an interface that has one. The prefixes are deliberately tiny
|
||||
// (/30, /126) and from private space no sane LAN uses, so they cannot shadow a
|
||||
// real subnet.
|
||||
var (
|
||||
l3Addr4 = netip.MustParsePrefix("172.19.242.1/30")
|
||||
l3Addr6 = netip.MustParsePrefix("fdfe:d3ad:b33f::1/126")
|
||||
)
|
||||
|
||||
// appendL3TunInbound appends the synthetic L3-ingress TUN inbound when globals
|
||||
// l3_tunnel opted in. Kernel TPROXY diverts nothing but TCP/UDP, so the rest of
|
||||
// the LAN's traffic (in practice: ICMP echo) can reach the engine only as raw
|
||||
// IP packets through a TUN device; the netplane policy-routes such packets into
|
||||
// netplane.L3Device and the engine picks them up here.
|
||||
//
|
||||
// There is deliberately no loop-guard RoutingMark: a TUN inbound is not a
|
||||
// socket (option.TunInboundOptions carries no ListenOptions), and the loop
|
||||
// guard already lives on every outbound dialer (DialerOptions.RoutingMark =
|
||||
// LoopMark), which is what keeps the engine's own egress out of the divert.
|
||||
//
|
||||
// Warnings carry the `icmp "tunnel"` entity prefix — the same entity the
|
||||
// netplane's L3 warnings use — so the panel groups everything about the L3
|
||||
// ingress under one section instead of scattering it.
|
||||
func (b *builder) appendL3TunInbound(inbounds []option.Inbound, seenTag map[string]bool) []option.Inbound {
|
||||
if !b.m.Globals.L3Tunnel {
|
||||
return inbounds
|
||||
}
|
||||
hasTproxy := false
|
||||
for _, in := range inbounds {
|
||||
if in.Type == C.TypeTProxy {
|
||||
hasTproxy = true
|
||||
break
|
||||
}
|
||||
}
|
||||
if !hasTproxy {
|
||||
// The L3 ingress rides the same LAN divert plane as tproxy: with no
|
||||
// tproxy inbound the netplane raises no divert chain and marks nothing,
|
||||
// so the TUN would sit dark while the config claims ICMP is tunnelled.
|
||||
b.warnf("icmp \"tunnel\": l3_tunnel is on but no tproxy inbound is enabled — the L3 ingress only receives LAN traffic the tproxy divert plane marks for it, so it would carry nothing; the %q inbound is not started (enable a tproxy inbound to restore it)", l3InboundTag)
|
||||
return inbounds
|
||||
}
|
||||
if seenTag[l3InboundTag] {
|
||||
// Unreachable today — inboundTag prefixes every model-derived tag with
|
||||
// "in-" — but kept on the same fail-degraded principle as the guards in
|
||||
// buildInbounds: losing the L3 ingress is strictly smaller damage than
|
||||
// two inbounds fighting over one tag.
|
||||
b.warnf("icmp \"tunnel\": inbound tag %q is already taken — the L3 ingress inbound is skipped so the existing listener keeps working (rename the clashing inbound to restore it)", l3InboundTag)
|
||||
return inbounds
|
||||
}
|
||||
addrs := badoption.Listable[netip.Prefix]{l3Addr4}
|
||||
if b.m.Globals.IPv6 {
|
||||
addrs = append(addrs, l3Addr6)
|
||||
}
|
||||
return append(inbounds, option.Inbound{
|
||||
Type: C.TypeTun,
|
||||
Tag: l3InboundTag,
|
||||
Options: &option.TunInboundOptions{
|
||||
InterfaceName: netplane.L3Device,
|
||||
MTU: l3MTU,
|
||||
Address: addrs,
|
||||
// AutoRoute MUST stay false. auto_route rewrites the router's MAIN
|
||||
// routing table and would drag everything the router itself sends —
|
||||
// WAN traffic, DNS, the tunnel's own underlay — into this TUN. The
|
||||
// netplane installs the scoped rule/route (fwmark L3Mark -> L3Table
|
||||
// -> L3Device) itself; that is the whole routing story.
|
||||
AutoRoute: false,
|
||||
// gvisor is a CHOICE, not a necessity, and the tempting reason for
|
||||
// it is wrong: BOTH sing-tun stacks forward ICMP through the same
|
||||
// ForwardDispatcher first, and both forge an echo reply only for
|
||||
// what that dispatcher declined (system: stack_system.go
|
||||
// dispatchIPv4 -> processIPv4ICMP; gvisor: stack_gvisor_filter.go
|
||||
// -> ICMPForwarder.HandlePacket). Do not re-derive this as "the
|
||||
// system stack fakes ping" — D25 says so explicitly. gvisor is
|
||||
// picked because it is already linked (with_wireguard requires
|
||||
// with_gvisor, D23), so it costs no build tag and no new code path,
|
||||
// and because it is the combination the integration test exercises.
|
||||
Stack: "gvisor",
|
||||
},
|
||||
})
|
||||
}
|
||||
|
||||
// listenKey returns the "addr:port" an inbound binds, for the clash guard. ok is
|
||||
|
||||
@@ -17,12 +17,20 @@ import (
|
||||
"github.com/sagernet/sing-box/option"
|
||||
|
||||
"github.com/sagernet/sing-box/shater/model"
|
||||
"github.com/sagernet/sing-box/shater/netplane"
|
||||
)
|
||||
|
||||
// buildInboundsOf runs buildInbounds over a single inbound and returns the
|
||||
// emitted inbounds plus the warnings.
|
||||
func buildInboundsOf(in model.Inbound) ([]option.Inbound, []string) {
|
||||
b := newBuilder(&model.Model{Globals: model.DefaultGlobals(), Inbounds: []model.Inbound{in}})
|
||||
return buildInboundsWith(model.DefaultGlobals(), in)
|
||||
}
|
||||
|
||||
// buildInboundsWith is buildInboundsOf with explicit globals and any number of
|
||||
// inbounds — for the knobs (l3_tunnel, ipv6) that decide WHICH inbounds exist
|
||||
// rather than how a single one is shaped.
|
||||
func buildInboundsWith(g model.Globals, ins ...model.Inbound) ([]option.Inbound, []string) {
|
||||
b := newBuilder(&model.Model{Globals: g, Inbounds: ins})
|
||||
return b.buildInbounds(), b.warnings
|
||||
}
|
||||
|
||||
@@ -305,3 +313,120 @@ func TestSniffIsNotAnInboundField(t *testing.T) {
|
||||
t.Fatal("legacy inbound sniff fields must stay empty (rejected by sing-box 1.13+)")
|
||||
}
|
||||
}
|
||||
|
||||
// tunInbounds filters the emitted inbounds down to the TUN ones. The synthetic
|
||||
// L3 ingress is their only source — TestInboundTypeMapping pins that a user
|
||||
// inbound of type "tun" stays rejected.
|
||||
func tunInbounds(ins []option.Inbound) []option.Inbound {
|
||||
var tuns []option.Inbound
|
||||
for _, in := range ins {
|
||||
if in.Type == C.TypeTun {
|
||||
tuns = append(tuns, in)
|
||||
}
|
||||
}
|
||||
return tuns
|
||||
}
|
||||
|
||||
// lanTproxy is the minimal enabled tproxy inbound the L3 tests pair with: the
|
||||
// L3 ingress only exists alongside the LAN divert plane.
|
||||
func lanTproxy() model.Inbound {
|
||||
return model.Inbound{Name: "lan", Enabled: true, Type: "tproxy", TproxyPort: 12345, TCP: true, UDP: true}
|
||||
}
|
||||
|
||||
// TestL3TunnelOffEmitsNoTunInbound: l3_tunnel is OPT-IN. The synthetic TUN
|
||||
// inbound brings a device, a gVisor netstack and new routing behaviour with it;
|
||||
// a router whose operator never asked must come through an upgrade unchanged.
|
||||
func TestL3TunnelOffEmitsNoTunInbound(t *testing.T) {
|
||||
ins, _ := buildInboundsWith(model.DefaultGlobals(), lanTproxy())
|
||||
if got := tunInbounds(ins); len(got) != 0 {
|
||||
t.Fatalf("l3_tunnel defaults to off, yet a TUN inbound was emitted — the L3 ingress grew out of an upgrade nobody opted into")
|
||||
}
|
||||
}
|
||||
|
||||
// TestL3TunnelEmitsTunInbound pins the data-plane contract of the L3 ingress.
|
||||
// Every field is load-bearing: the netplane points the L3 table's default
|
||||
// route at exactly netplane.L3Device, auto_route off keeps sing-box away from
|
||||
// the router's MAIN routing table, and only the gVisor stack actually forwards
|
||||
// ICMP — the system stack forges echo replies locally.
|
||||
func TestL3TunnelEmitsTunInbound(t *testing.T) {
|
||||
g := model.DefaultGlobals()
|
||||
g.L3Tunnel = true
|
||||
ins, warns := buildInboundsWith(g, lanTproxy())
|
||||
tuns := tunInbounds(ins)
|
||||
if len(tuns) != 1 {
|
||||
t.Fatalf("want exactly one L3 TUN inbound, got %d (warns=%v)", len(tuns), warns)
|
||||
}
|
||||
if tuns[0].Tag != "l3-in" {
|
||||
t.Fatalf("tag = %q, want %q — route rules address the L3 ingress by exactly this tag, so any other spelling detaches it from its routing", tuns[0].Tag, "l3-in")
|
||||
}
|
||||
to, ok := tuns[0].Options.(*option.TunInboundOptions)
|
||||
if !ok {
|
||||
t.Fatalf("options are %T, want *option.TunInboundOptions", tuns[0].Options)
|
||||
}
|
||||
if to.InterfaceName != netplane.L3Device {
|
||||
t.Fatalf("interface_name = %q, want netplane.L3Device (%q) — the netplane routes marked LAN traffic into that exact device, so any other name leaves a TUN nothing feeds", to.InterfaceName, netplane.L3Device)
|
||||
}
|
||||
if to.AutoRoute {
|
||||
t.Fatal("auto_route is on — sing-box would rewrite the router's MAIN routing table and drag the router's own WAN/DNS/underlay traffic into the tunnel; the netplane owns the scoped L3 rules")
|
||||
}
|
||||
if to.Stack != "gvisor" {
|
||||
t.Fatalf("stack = %q, want gvisor — the system stack answers ICMP echo locally instead of forwarding it, which is the exact forged reply l3_tunnel exists to remove", to.Stack)
|
||||
}
|
||||
if to.MTU != 65535 {
|
||||
t.Fatalf("mtu = %d, want 65535 — see TestL3TunnelMTULeavesNothingForTheKernelToFragment for why the number is the maximum and not a tunnel budget", to.MTU)
|
||||
}
|
||||
if len(to.Address) != 2 || to.Address[0].String() != "172.19.242.1/30" || to.Address[1].String() != "fdfe:d3ad:b33f::1/126" {
|
||||
t.Fatalf("address = %v, want [172.19.242.1/30 fdfe:d3ad:b33f::1/126] — the netplane's routes are built against exactly these prefixes", to.Address)
|
||||
}
|
||||
}
|
||||
|
||||
// TestL3TunnelIPv6OffKeepsV4Only: with globals ipv6 off the netplane installs
|
||||
// no v6 rules, so a v6 prefix on the TUN would advertise an ICMPv6 path that
|
||||
// dead-ends inside the device.
|
||||
func TestL3TunnelIPv6OffKeepsV4Only(t *testing.T) {
|
||||
g := model.DefaultGlobals()
|
||||
g.L3Tunnel = true
|
||||
g.IPv6 = false
|
||||
ins, _ := buildInboundsWith(g, lanTproxy())
|
||||
tuns := tunInbounds(ins)
|
||||
if len(tuns) != 1 {
|
||||
t.Fatalf("want the L3 TUN inbound, got %d", len(tuns))
|
||||
}
|
||||
to, ok := tuns[0].Options.(*option.TunInboundOptions)
|
||||
if !ok {
|
||||
t.Fatalf("options are %T, want *option.TunInboundOptions", tuns[0].Options)
|
||||
}
|
||||
if len(to.Address) != 1 || !to.Address[0].Addr().Is4() {
|
||||
t.Fatalf("address = %v — on an ipv6=off router only the v4 prefix may remain; a v6 address advertises an ICMPv6 path the netplane never routes", to.Address)
|
||||
}
|
||||
}
|
||||
|
||||
// TestL3TunnelWithoutTproxySkipped: the L3 ingress rides the LAN divert plane
|
||||
// that only exists alongside a tproxy inbound. Without one the TUN would sit
|
||||
// dark while the config claims ICMP is tunnelled — so it is skipped, and the
|
||||
// skip is said out loud.
|
||||
func TestL3TunnelWithoutTproxySkipped(t *testing.T) {
|
||||
g := model.DefaultGlobals()
|
||||
g.L3Tunnel = true
|
||||
ins, warns := buildInboundsWith(g, model.Inbound{
|
||||
Name: "sock", Enabled: true, Type: "socks", Listen: "127.0.0.1", Port: 1080, TCP: true, UDP: true,
|
||||
})
|
||||
if got := tunInbounds(ins); len(got) != 0 {
|
||||
t.Fatalf("no tproxy inbound is enabled, yet the L3 TUN inbound was emitted — it would carry nothing while claiming ICMP coverage")
|
||||
}
|
||||
if !warnsHave(warns, "no tproxy inbound") {
|
||||
t.Fatalf("expected a warning explaining the skipped L3 ingress, got %v", warns)
|
||||
}
|
||||
if !warnsHave(warns, `icmp "tunnel": `) {
|
||||
t.Fatalf("the L3 warning must carry the `icmp \"tunnel\"` entity prefix — without it the panel cannot group it with the data plane's L3 warnings and it lands nameless under a bare \"generate\" section; got %v", warns)
|
||||
}
|
||||
|
||||
// The same holds for a config with no inbounds at all (every one disabled).
|
||||
ins, warns = buildInboundsWith(g)
|
||||
if got := tunInbounds(ins); len(got) != 0 {
|
||||
t.Fatalf("an inbound-less config emitted the L3 TUN inbound — it would carry nothing while claiming ICMP coverage")
|
||||
}
|
||||
if !warnsHave(warns, "no tproxy inbound") {
|
||||
t.Fatalf("expected a warning explaining the skipped L3 ingress, got %v", warns)
|
||||
}
|
||||
}
|
||||
|
||||
@@ -0,0 +1,160 @@
|
||||
//go:build linux
|
||||
|
||||
// The egress half of the L3 ICMP story, live-engine part: not the SHAPE of the
|
||||
// config (l3_egress_test.go pins that portably) but what the running engine
|
||||
// BELIEVES about the two egress kinds. The belief is the whole feature:
|
||||
//
|
||||
// - An interface egress must come up as an adapter.FlowOutbound whose
|
||||
// PreMatchFlow answers Flow for ICMP. That answer exists only if
|
||||
// protocol/direct's constructor actually built its ping.Port, which it does
|
||||
// only when dialer.NewWithOptions hands back a *dialer.DefaultDialer for a
|
||||
// dialer that carries BindInterface + RoutingMark. Nothing in the portable
|
||||
// suite can see this: if that cast ever stops holding (an upstream bump
|
||||
// wrapping the bound dialer, say), the generated JSON stays byte-identical,
|
||||
// every codegen test stays green, and ping through every interface egress
|
||||
// silently degrades from "leaves via the second WAN" to "dropped".
|
||||
// - A byedpi egress must NOT look ICMP-capable: route.preMatchFlow
|
||||
// (l3-honest-drop) drops an ICMP flow whose outbound either lacks icmp in
|
||||
// Network() or is not a FlowOutbound, and SOCKS satisfies both refusals. If
|
||||
// it ever stops refusing, the drop stops happening — and the TUN stack's
|
||||
// alternative is forging the echo reply itself.
|
||||
//
|
||||
// Gating mirrors l3_integration_linux_test.go, whose comment carries the full
|
||||
// argument: the model opts into l3_tunnel, so Start opens /dev/net/tun and
|
||||
// needs root + CAP_NET_ADMIN, which the ordinary gate's containers do not
|
||||
// expose; the TestIntegration prefix and the honest skips below keep the gate
|
||||
// green while telling a human exactly how to run this for real.
|
||||
package generate
|
||||
|
||||
import (
|
||||
"net/netip"
|
||||
"os"
|
||||
"slices"
|
||||
"testing"
|
||||
|
||||
"github.com/sagernet/sing-box/adapter"
|
||||
N "github.com/sagernet/sing/common/network"
|
||||
|
||||
"github.com/sagernet/sing-box/shater/engine"
|
||||
"github.com/sagernet/sing-box/shater/model"
|
||||
"github.com/sagernet/sing-box/shater/netplane"
|
||||
)
|
||||
|
||||
// l3EgressICMPModel is the smallest l3_tunnel=1 config that carries both
|
||||
// egress kinds: the mandatory tproxy divert plane, a resolver, one interface
|
||||
// egress and one byedpi egress, each with a rule routing into it.
|
||||
//
|
||||
// The interface egress binds to `lo` — deliberately: BindInterface must name a
|
||||
// device that EXISTS on the runner (netplane.IfaceDevice passes an
|
||||
// unresolvable name through unchanged, and every kernel has lo), and this test
|
||||
// never dials, so nothing actually leaves through it. IPv6 is off for the same
|
||||
// reason as the sibling model: a disable_ipv6=1 host must not masquerade as
|
||||
// the regression this test hunts.
|
||||
func l3EgressICMPModel() *model.Model {
|
||||
g := model.DefaultGlobals()
|
||||
g.KillSwitch = "closed"
|
||||
g.ResolverDefault = "cf"
|
||||
g.L3Tunnel = true
|
||||
g.IPv6 = false
|
||||
return &model.Model{
|
||||
Globals: g,
|
||||
Inbounds: []model.Inbound{
|
||||
// 12404: next free port above the package's hand-allocated tproxy
|
||||
// band (12403 belongs to l3_integration_linux_test.go). Start
|
||||
// binds for real, so a clash with a sibling would fail this test
|
||||
// for reasons that have nothing to do with the egresses.
|
||||
{Name: "lan", Enabled: true, Type: "tproxy", TproxyPort: 12404, TCP: true, UDP: true},
|
||||
},
|
||||
Resolvers: []model.Resolver{
|
||||
{Name: "cf", Type: "doh", Address: "https://1.1.1.1/dns-query", Detour: "direct"},
|
||||
},
|
||||
Egresses: []model.Egress{
|
||||
{Name: "lo", Type: "interface", Interface: "lo"},
|
||||
{Name: "bd", Type: "byedpi", Port: 1080},
|
||||
},
|
||||
Rules: []model.Rule{
|
||||
{Name: "ping-lo", Enabled: true, Order: 10, Src: []string{"192.168.88.0/24"}, Target: "egress:lo"},
|
||||
{Name: "desync", Enabled: true, Order: 20, Src: []string{"192.168.89.0/24"}, Target: "egress:bd"},
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
// TestIntegrationL3EgressICMPIsAFlow proves the live engine's verdict on ICMP
|
||||
// through each egress kind, in failure-mode order:
|
||||
//
|
||||
// 1. the interface egress outbound is an adapter.FlowOutbound advertising
|
||||
// icmp — the two static gates route.preMatchFlow checks before it even
|
||||
// asks the outbound;
|
||||
// 2. its PreMatchFlow(icmp) answers Flow — the dynamic gate, true only when
|
||||
// the ping.Port was really constructed despite BindInterface+RoutingMark
|
||||
// on the dialer. THE assertion of this file: its failure mode is a ping
|
||||
// that silently turns into a drop with not one generated byte changed;
|
||||
// 3. the byedpi egress outbound fails at least one of the same static gates,
|
||||
// which is precisely what makes l3-honest-drop DROP a ping routed at it
|
||||
// instead of the TUN stack forging the echo reply.
|
||||
func TestIntegrationL3EgressICMPIsAFlow(t *testing.T) {
|
||||
if os.Geteuid() != 0 {
|
||||
t.Skipf("needs root to open and configure a TUN device (euid=%d) — run as root with CAP_NET_ADMIN and /dev/net/tun, e.g. on the OpenWrt VM or via `docker run --cap-add NET_ADMIN --device /dev/net/tun`", os.Geteuid())
|
||||
}
|
||||
if _, err := os.Stat("/dev/net/tun"); err != nil {
|
||||
t.Skipf("/dev/net/tun is not available (%v) — expose it (modprobe tun; in docker: --device /dev/net/tun --cap-add NET_ADMIN) and run as root", err)
|
||||
}
|
||||
|
||||
opts, warns, err := GenerateWithWarnings(l3EgressICMPModel())
|
||||
if err != nil {
|
||||
t.Fatalf("Generate: unexpected error: %v", err)
|
||||
}
|
||||
if len(warns) != 0 {
|
||||
t.Fatalf("generate degraded the model (warnings: %v) — the egress and L3 skip paths warn instead of failing, so a warning here usually means an egress outbound or the TUN inbound was silently dropped and the engine verdicts below would prove nothing", warns)
|
||||
}
|
||||
|
||||
e := engine.New()
|
||||
// Idempotent; covers every Fatalf below. The wait is NOT decoration: this
|
||||
// test opens netplane.L3Device, and the name is a singleton, so yielding
|
||||
// before the kernel has taken it back leaves the next test in this package
|
||||
// to meet `TUNSETIFF: device or resource busy` and fail for a reason that
|
||||
// has nothing to do with what it asserts.
|
||||
t.Cleanup(func() {
|
||||
_ = e.Close()
|
||||
l3WaitDeviceGone(t)
|
||||
})
|
||||
if _, err := e.Apply(opts); err != nil {
|
||||
t.Fatalf("engine.Apply (box.New validate + Start) rejected the config: %v\n%s", err, l3StartFailureHint(err))
|
||||
}
|
||||
om := e.Instance().Outbound()
|
||||
if om == nil {
|
||||
t.Fatalf("running box has no OutboundManager — nothing below could prove anything")
|
||||
}
|
||||
|
||||
// 1+2. The interface egress: the engine must consider it ICMP-capable, and
|
||||
// capable FOR REAL (the ping.Port exists), not just by interface shape.
|
||||
loTag := netplane.EgressOutboundTag("lo")
|
||||
loOb, ok := om.Outbound(loTag)
|
||||
if !ok {
|
||||
t.Fatalf("running box has no outbound %q — every rule bound to this egress resolved into nothing, so its traffic is fail-closed blocked and the second-WAN path this feature sells does not exist", loTag)
|
||||
}
|
||||
if !slices.Contains(loOb.Network(), N.NetworkICMP) {
|
||||
t.Fatalf("outbound %q Network() = %v, without %q — route.preMatchFlow refuses the flow at its first static gate, so every ping routed through an interface egress is dropped while TCP/UDP keep flowing", loTag, loOb.Network(), N.NetworkICMP)
|
||||
}
|
||||
flow, isFlow := loOb.(adapter.FlowOutbound)
|
||||
if !isFlow {
|
||||
t.Fatalf("outbound %q (%T) is not an adapter.FlowOutbound — route.preMatchFlow can then never answer Flow for it, so every ping routed through an interface egress is dropped while the config still claims the egress carries L3", loTag, loOb)
|
||||
}
|
||||
if got := flow.PreMatchFlow(N.NetworkICMP, netip.MustParseAddr("203.0.113.9")); got != adapter.PreMatchFlow {
|
||||
t.Fatalf("PreMatchFlow(icmp) = %v, want adapter.PreMatchFlow — the direct outbound started WITHOUT its ping.Port, i.e. dialer.NewWithOptions no longer yields a *dialer.DefaultDialer once BindInterface+RoutingMark are set (protocol/direct only builds icmpPort behind that cast); ping through every interface egress then silently turns into a drop with not one generated byte changed, so only this live check can catch it", got)
|
||||
}
|
||||
|
||||
// 3. The byedpi egress: at least one static gate must refuse it. Both
|
||||
// refusing is today's reality (SOCKS advertises no icmp and is no
|
||||
// FlowOutbound); the regression is BOTH passing, because then
|
||||
// l3-honest-drop stops dropping and the TUN stack answers the echo itself
|
||||
// — a forged reply from a desync hop that never saw the packet.
|
||||
bdTag := netplane.EgressOutboundTag("bd")
|
||||
bdOb, ok := om.Outbound(bdTag)
|
||||
if !ok {
|
||||
t.Fatalf("running box has no outbound %q — every rule bound to this egress resolved into nothing, so its domains lost the desync entirely", bdTag)
|
||||
}
|
||||
if _, isFlow := bdOb.(adapter.FlowOutbound); isFlow && slices.Contains(bdOb.Network(), N.NetworkICMP) {
|
||||
t.Fatalf("outbound %q (%T, networks %v) passes both of route.preMatchFlow's static gates for ICMP — l3-honest-drop then no longer drops a ping routed at the byedpi egress, and the TUN stack forges the echo reply locally: the operator reads a working ping off a SOCKS hop that cannot carry the packet", bdTag, bdOb, bdOb.Network())
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,123 @@
|
||||
// The egress half of the L3 ICMP story, portable codegen part: WHAT the
|
||||
// generator must emit so a ping routed to `egress:<name>` behaves honestly.
|
||||
//
|
||||
// The chain these tests pin (all upstream sing-box internals, none of them
|
||||
// visible in the generated JSON): protocol/direct is the ONLY proxy outbound
|
||||
// in the shipped registry that implements adapter.FlowOutbound — at
|
||||
// construction it wraps a ping.Port around the dialer's Control chain, and
|
||||
// common/dialer/default.go appends BindInterface + RoutingMark to exactly
|
||||
// that chain (dialer.Control -> DefaultDialer.dialer4 ->
|
||||
// DialerForICMPDestination -> the raw ICMP socket). So the SHAPE asserted
|
||||
// here — a direct outbound carrying the egress device and the deterministic
|
||||
// egress mark — is precisely what makes a tunnelled ping leave through the
|
||||
// right interface under the right policy table. A SOCKS outbound (byedpi)
|
||||
// sits on the other side of the same line: it cannot implement tun.Port, so
|
||||
// route.preMatchFlow's l3-honest-drop block DROPS ICMP routed at it instead
|
||||
// of letting the TUN stack forge an echo reply locally.
|
||||
//
|
||||
// Portable (no box.New): these assert on the generated option.Options only.
|
||||
// The live-engine proof that the direct outbound REALLY constructs its
|
||||
// icmpPort despite bind+mark lives in l3_egress_linux_test.go.
|
||||
package generate
|
||||
|
||||
import (
|
||||
"testing"
|
||||
|
||||
C "github.com/sagernet/sing-box/constant"
|
||||
"github.com/sagernet/sing-box/option"
|
||||
|
||||
"github.com/sagernet/sing-box/shater/model"
|
||||
"github.com/sagernet/sing-box/shater/netplane"
|
||||
)
|
||||
|
||||
// l3EgressOutbound finds the outbound emitted under tag. A local twin of
|
||||
// generate_test.go's findOutbound, which lives in the linux-gated engine
|
||||
// suite and does not exist on other platforms.
|
||||
func l3EgressOutbound(opts option.Options, tag string) *option.Outbound {
|
||||
for i := range opts.Outbounds {
|
||||
if opts.Outbounds[i].Tag == tag {
|
||||
return &opts.Outbounds[i]
|
||||
}
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// TestEgressInterfaceOutboundCarriesICMP pins the three fields that decide
|
||||
// whether a ping routed to an interface egress ACTUALLY leaves through that
|
||||
// interface:
|
||||
//
|
||||
// - Type direct — the one registered outbound type whose constructor builds
|
||||
// a ping.Port (adapter.FlowOutbound); any other type demotes ICMP through
|
||||
// this egress to the honest drop.
|
||||
// - BindInterface = the egress device — appended to dialer.Control, which
|
||||
// DefaultDialer.dialer4 carries and DialerForICMPDestination hands to the
|
||||
// raw ICMP socket.
|
||||
// - RoutingMark = the deterministic egress mark — same Control chain; it is
|
||||
// what the netplane's policy rule matches to steer the packet into the
|
||||
// egress table and past the tproxy divert.
|
||||
func TestEgressInterfaceOutboundCarriesICMP(t *testing.T) {
|
||||
m := &model.Model{
|
||||
Globals: model.DefaultGlobals(),
|
||||
Inbounds: []model.Inbound{
|
||||
{Name: "lan", Enabled: true, Type: "tproxy", TproxyPort: 12345, TCP: true, UDP: true},
|
||||
},
|
||||
Egresses: []model.Egress{{Name: "wan2", Type: "interface", Interface: "wan2"}},
|
||||
Rules: []model.Rule{
|
||||
{Name: "via-wan2", Enabled: true, Order: 10, Src: []string{"192.168.2.0/24"}, Target: "egress:wan2"},
|
||||
},
|
||||
}
|
||||
opts, warns, err := GenerateWithWarnings(m)
|
||||
if err != nil {
|
||||
t.Fatalf("Generate: unexpected error: %v", err)
|
||||
}
|
||||
tag := netplane.EgressOutboundTag("wan2")
|
||||
ob := l3EgressOutbound(opts, tag)
|
||||
if ob == nil {
|
||||
t.Fatalf("no outbound %q was emitted (warnings: %v) — every rule bound to this egress then resolves through the fail-closed path (egressDetourOrBlock) and the operator's second WAN silently carries nothing", tag, warns)
|
||||
}
|
||||
if ob.Type != C.TypeDirect {
|
||||
t.Fatalf("egress outbound type = %q, want %q — direct is the only proxy outbound in the shipped registry that implements adapter.FlowOutbound (its constructor builds the ping.Port), so any other type turns every ping routed through this egress into a drop while TCP/UDP keep flowing, and nobody can tell the second WAN's L3 path is dead", ob.Type, C.TypeDirect)
|
||||
}
|
||||
do, ok := ob.Options.(*option.DirectOutboundOptions)
|
||||
if !ok {
|
||||
t.Fatalf("egress outbound options are %T, want *option.DirectOutboundOptions — without the typed dialer options there is no BindInterface/RoutingMark to reach the ICMP socket's Control chain at all", ob.Options)
|
||||
}
|
||||
if want := netplane.IfaceDevice("wan2"); do.BindInterface != want {
|
||||
t.Fatalf("BindInterface = %q, want %q — this field is appended to dialer.Control (common/dialer/default.go), which lands in DefaultDialer.dialer4, whose Control DialerForICMPDestination hands to the ICMP socket; losing it sends the echo over the MAIN routing table, i.e. out the plain default WAN instead of this egress — a silent leak, not a visible failure", do.BindInterface, want)
|
||||
}
|
||||
if want := option.FwMark(netplane.EgressMark(m.Globals, 0)); do.RoutingMark != want {
|
||||
t.Fatalf("RoutingMark = %#x, want %#x — the mark rides the same Control chain into the ICMP socket, and it is what the netplane's per-egress policy rule matches; without it the egress's own packets are routed by the main table (leaking past the egress) or re-caught by the tproxy divert (a routing loop)", uint32(do.RoutingMark), uint32(want))
|
||||
}
|
||||
}
|
||||
|
||||
// TestEgressByeDPIOutboundCannotCarryICMP pins the OTHER side of the line: a
|
||||
// byedpi egress is a SOCKS5 hop into the local ciadpi desync proxy, and SOCKS
|
||||
// does not (and cannot) implement tun.Port, so route.preMatchFlow's
|
||||
// l3-honest-drop block DROPS ICMP routed at it. That drop is the feature: the
|
||||
// only alternative the TUN stack offers is answering the echo ITSELF
|
||||
// (stack_gvisor_icmp.go), i.e. a forged reply from a path that never saw the
|
||||
// packet. Pinning the SOCKS type here pins the reason the drop happens.
|
||||
func TestEgressByeDPIOutboundCannotCarryICMP(t *testing.T) {
|
||||
m := &model.Model{
|
||||
Globals: model.DefaultGlobals(),
|
||||
Inbounds: []model.Inbound{
|
||||
{Name: "lan", Enabled: true, Type: "tproxy", TproxyPort: 12345, TCP: true, UDP: true},
|
||||
},
|
||||
Egresses: []model.Egress{{Name: "bd", Type: "byedpi", Port: 1080}},
|
||||
Rules: []model.Rule{
|
||||
{Name: "desync", Enabled: true, Order: 10, Src: []string{"192.168.3.0/24"}, Target: "egress:bd"},
|
||||
},
|
||||
}
|
||||
opts, warns, err := GenerateWithWarnings(m)
|
||||
if err != nil {
|
||||
t.Fatalf("Generate: unexpected error: %v", err)
|
||||
}
|
||||
tag := netplane.EgressOutboundTag("bd")
|
||||
ob := l3EgressOutbound(opts, tag)
|
||||
if ob == nil {
|
||||
t.Fatalf("no outbound %q was emitted (warnings: %v) — every rule bound to this egress then resolves through the fail-closed path and the desync stops covering its domains", tag, warns)
|
||||
}
|
||||
if ob.Type != C.TypeSOCKS {
|
||||
t.Fatalf("byedpi egress outbound type = %q, want %q — the type is load-bearing twice over: only a SOCKS hop actually reaches the local ciadpi process (anything else skips the desync entirely), and its inability to implement tun.Port is exactly what makes the l3-honest-drop block DROP a ping routed here instead of the TUN stack forging an echo reply from a path that never carried the packet", ob.Type, C.TypeSOCKS)
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,217 @@
|
||||
//go:build linux
|
||||
|
||||
// The engine half of the L3-ingress proof — the one that needs a real kernel.
|
||||
//
|
||||
// Everything else about l3_tunnel is already pinned by unprivileged tests: the
|
||||
// portable codegen suite (inbound_test.go TestL3Tunnel*) fixes the SHAPE of the
|
||||
// synthetic "l3-in" TUN inbound (tag, netplane.L3Device name, auto_route off,
|
||||
// gvisor stack, MTU, addresses), and the netplane tests fix the nft marks and
|
||||
// the ip rule/route plan around it. What NONE of them prove is that the engine
|
||||
// actually accepts that config on the router build: shaterd swaps upstream's
|
||||
// include.Context for the slim shater/registry (engine.New wires
|
||||
// registry.Context), where tun.RegisterInbound is a deliberate hand-kept
|
||||
// entry, and the gVisor netstack the inbound demands (stack: gvisor) is only
|
||||
// compiled in because with_wireguard drags with_gvisor along
|
||||
// (scripts/router-tags.sh). Losing either — the registry entry deleted as
|
||||
// "unused", the tag trimmed for size — changes no generated byte, so every
|
||||
// codegen assertion stays green, and the failure surfaces as a dead engine on
|
||||
// the operator's router at the first l3_tunnel=1 apply (the 2026-07-25
|
||||
// WireGuard outage was exactly this "built with X, verified with Y" class).
|
||||
// This file closes that gap: box.New + Start must accept the generated config
|
||||
// under the slim registry, the kernel must end up with the shater-l3 device
|
||||
// the netplane routes into (at the contract MTU), and Close must remove it.
|
||||
//
|
||||
// Separate file, TestIntegration name, honest skips: opening /dev/net/tun and
|
||||
// configuring the device needs root + CAP_NET_ADMIN, which the ordinary gate
|
||||
// does not have — scripts/run-tests.sh reaches linux through containers (the
|
||||
// act_runner job container, or the docker re-exec from a dev host) that expose
|
||||
// no /dev/net/tun, and its SKIP_COMMON already excludes common/tlsspoof's
|
||||
// TestIntegration* for the same capability reason; the TestIntegration prefix
|
||||
// keeps this test inside that naming convention. The shater/... roots are not
|
||||
// name-filtered, so what keeps the ordinary gate green here are the guards
|
||||
// below: no root or no /dev/net/tun means a loud skip that says how to run it
|
||||
// for real (the OpenWrt VM, or
|
||||
// `docker run --cap-add NET_ADMIN --device /dev/net/tun`).
|
||||
package generate
|
||||
|
||||
import (
|
||||
"net"
|
||||
"os"
|
||||
"strings"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
C "github.com/sagernet/sing-box/constant"
|
||||
|
||||
"github.com/sagernet/sing-box/shater/engine"
|
||||
"github.com/sagernet/sing-box/shater/model"
|
||||
"github.com/sagernet/sing-box/shater/netplane"
|
||||
)
|
||||
|
||||
// l3GoneTimeout bounds how long assertion 4 waits for the kernel to drop the
|
||||
// device after Close. Removal is normally immediate with the last fd, but
|
||||
// unregister_netdevice may defer briefly under load; polling keeps the check
|
||||
// honest without a flaky fixed sleep.
|
||||
const l3GoneTimeout = 5 * time.Second
|
||||
|
||||
// l3IntegrationModel is the smallest realistic l3_tunnel=1 config: one enabled
|
||||
// tproxy inbound (the L3 ingress refuses to exist without the divert plane it
|
||||
// rides), one resolver, one shadowsocks node on TEST-NET and a rule into it —
|
||||
// the same skeleton as the other *_linux_test.go models, plus the L3 opt-in.
|
||||
//
|
||||
// IPv6 is off deliberately: the v6 address half of the TUN contract is pinned
|
||||
// by the portable codegen tests, and carrying it here would couple THIS proof
|
||||
// (the tun inbound starts under the slim registry) to the runner kernel's
|
||||
// ipv6 sysctls — a disable_ipv6=1 host would fail address configuration and
|
||||
// masquerade as the registry/tag regression this test hunts.
|
||||
func l3IntegrationModel() *model.Model {
|
||||
g := model.DefaultGlobals()
|
||||
g.KillSwitch = "closed"
|
||||
g.ResolverDefault = "cf"
|
||||
g.L3Tunnel = true
|
||||
g.IPv6 = false
|
||||
return &model.Model{
|
||||
Globals: g,
|
||||
Inbounds: []model.Inbound{
|
||||
// 12403: first port above the package's hand-allocated tproxy band
|
||||
// (currently topping out at 12402). Start binds for real, so a
|
||||
// clash with a sibling's port would fail this test for reasons
|
||||
// that have nothing to do with the TUN.
|
||||
{Name: "lan", Enabled: true, Type: "tproxy", TproxyPort: 12403, TCP: true, UDP: true},
|
||||
},
|
||||
Nodes: []model.Node{
|
||||
{Name: "exit", Enabled: true, URI: "ss://aes-256-gcm:secret@203.0.113.1:8388#exit"},
|
||||
},
|
||||
Resolvers: []model.Resolver{
|
||||
{Name: "cf", Type: "doh", Address: "https://1.1.1.1/dns-query", Detour: "direct"},
|
||||
},
|
||||
Rules: []model.Rule{
|
||||
{Name: "via-exit", Enabled: true, Order: 10, DstPort: "443", Target: "node:exit"},
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
// TestIntegrationL3TunInboundStarts proves the generated l3_tunnel config is
|
||||
// not merely well-formed but ALIVE: the engine (slim registry, compiled tag
|
||||
// set) accepts it, the kernel ends up with the device the netplane routes
|
||||
// into, and Close returns the name. Four assertions, in failure-mode order:
|
||||
//
|
||||
// 1. engine.Apply (box.New + Start) succeeds — a rejection here is the slim
|
||||
// registry losing tun.RegisterInbound or the build losing the gVisor
|
||||
// stack, see l3StartFailureHint;
|
||||
// 2. net.InterfaceByName(netplane.L3Device) finds the device — the L3 table's
|
||||
// default route points at exactly that name, so no device means marked
|
||||
// LAN ICMP blackholes while the config claims it is tunnelled;
|
||||
// 3. the kernel ACCEPTS the contract MTU 65535 on a TUN — the whole point of
|
||||
// the value is that no IP datagram can exceed it, so the kernel can never
|
||||
// fragment on the way in (D25); a kernel that clamped it would silently
|
||||
// restore the forged-reply band;
|
||||
// 4. after Close the device is GONE — a leak would jam every later apply
|
||||
// (each swap reopens the same name) until shaterd itself is restarted.
|
||||
func TestIntegrationL3TunInboundStarts(t *testing.T) {
|
||||
if os.Geteuid() != 0 {
|
||||
t.Skipf("needs root to open and configure a TUN device (euid=%d) — run as root with CAP_NET_ADMIN and /dev/net/tun, e.g. on the OpenWrt VM or via `docker run --cap-add NET_ADMIN --device /dev/net/tun`", os.Geteuid())
|
||||
}
|
||||
if _, err := os.Stat("/dev/net/tun"); err != nil {
|
||||
t.Skipf("/dev/net/tun is not available (%v) — expose it (modprobe tun; in docker: --device /dev/net/tun --cap-add NET_ADMIN) and run as root", err)
|
||||
}
|
||||
|
||||
opts, warns, err := GenerateWithWarnings(l3IntegrationModel())
|
||||
if err != nil {
|
||||
t.Fatalf("Generate: unexpected error: %v", err)
|
||||
}
|
||||
if len(warns) != 0 {
|
||||
t.Fatalf("generate degraded the model (warnings: %v) — the L3 skip paths warn instead of failing, so a warning here usually means the TUN inbound was silently dropped and the engine run below would prove nothing", warns)
|
||||
}
|
||||
// Precondition, not the point: the SHAPE of the tun inbound is
|
||||
// inbound_test.go's job. If it is missing here, fail with the right
|
||||
// address instead of a misleading "no such interface" three steps later.
|
||||
hasTun := false
|
||||
for _, in := range opts.Inbounds {
|
||||
if in.Type == C.TypeTun {
|
||||
hasTun = true
|
||||
break
|
||||
}
|
||||
}
|
||||
if !hasTun {
|
||||
t.Fatalf("generated options carry no tun inbound — a codegen regression (inbound_test.go TestL3TunnelEmitsTunInbound should be red too), not an engine one; nothing below could prove anything")
|
||||
}
|
||||
|
||||
// 1. The engine — slim registry, the tag set this binary was built with —
|
||||
// takes the config and starts it.
|
||||
e := engine.New()
|
||||
t.Cleanup(func() { _ = e.Close() }) // idempotent; covers every Fatalf below
|
||||
changed, err := e.Apply(opts)
|
||||
if err != nil {
|
||||
t.Fatalf("engine.Apply (box.New validate + Start) rejected the l3_tunnel config: %v\n%s", err, l3StartFailureHint(err))
|
||||
}
|
||||
if !changed {
|
||||
t.Fatalf("expected Apply changed==true on a fresh engine")
|
||||
}
|
||||
|
||||
// 2. The device is REAL. Start constructs the TUN synchronously
|
||||
// (protocol/tun StartStateStart -> tun.New), so no settling loop is
|
||||
// needed on this side.
|
||||
iface, err := net.InterfaceByName(netplane.L3Device)
|
||||
if err != nil {
|
||||
t.Fatalf("Start reported success but %q does not exist (%v) — the netplane's L3 table points its default route at exactly that name, so marked LAN ICMP would blackhole while the config claims it is tunnelled", netplane.L3Device, err)
|
||||
}
|
||||
|
||||
// 3. The kernel device carries the contract MTU. This is the assertion the
|
||||
// value was chosen for: at 65535 no IP datagram can exceed the device MTU,
|
||||
// so the kernel cannot fragment on the way in and sing-tun'''s dispatcher
|
||||
// always gets a whole packet to judge. A kernel that silently clamped this
|
||||
// (or a driver with a lower max_mtu) would put the forged-reply band back
|
||||
// without changing a single generated byte.
|
||||
if iface.MTU != 65535 {
|
||||
t.Fatalf("%s MTU = %d, want 65535 — the kernel did not take the contract MTU; anything smaller means the kernel fragments packets above it INTO this device, sing-tun'''s ForwardDispatcher declines fragments before asking for a verdict, and the gVisor ICMP forwarder forges the echo reply for WireGuard/AWG (D25)", netplane.L3Device, iface.MTU)
|
||||
}
|
||||
|
||||
// 4. Close returns the device name to the kernel.
|
||||
if err := e.Close(); err != nil {
|
||||
t.Fatalf("engine Close failed: %v — a box that does not stop keeps %s open, and every later apply that reopens the name starts dead", err, netplane.L3Device)
|
||||
}
|
||||
l3WaitDeviceGone(t)
|
||||
}
|
||||
|
||||
// l3WaitDeviceGone blocks until the L3 TUN is back in the kernel's hands.
|
||||
//
|
||||
// It is shared rather than inlined because the name is a SINGLETON: two tests
|
||||
// in this package each stand an engine up on netplane.L3Device, and the second
|
||||
// one meets `TUNSETIFF: device or resource busy` if the first only closed its
|
||||
// box and moved on. Removal is normally immediate with the last fd, but
|
||||
// unregister_netdevice may defer briefly under load, so every test that opens
|
||||
// the device MUST wait here before yielding it — polling rather than a fixed
|
||||
// sleep keeps that honest instead of flaky.
|
||||
func l3WaitDeviceGone(t *testing.T) {
|
||||
t.Helper()
|
||||
deadline := time.Now().Add(l3GoneTimeout)
|
||||
for {
|
||||
if _, err := net.InterfaceByName(netplane.L3Device); err != nil {
|
||||
return // gone: the kernel dropped the device with the engine's fd
|
||||
}
|
||||
if time.Now().After(deadline) {
|
||||
t.Fatalf("%s still exists %v after Close — the engine leaked the TUN; every subsequent apply swaps in a fresh box that must reopen this exact name, so from the first leak onward the L3 ingress comes up dead until shaterd is restarted (and a re-run of this suite would inherit the stale device)", netplane.L3Device, l3GoneTimeout)
|
||||
}
|
||||
time.Sleep(50 * time.Millisecond)
|
||||
}
|
||||
}
|
||||
|
||||
// l3StartFailureHint names the specific regression (or environment defect) an
|
||||
// Apply failure most likely is, so the gate output points at the fix instead
|
||||
// of a bare engine error. Substring matching is the same pragmatism engine.go
|
||||
// itself applies to swap conflicts (isAddrInUse/isCacheLockConflict): the
|
||||
// upstream errors carry no exported sentinels.
|
||||
func l3StartFailureHint(err error) string {
|
||||
msg := err.Error()
|
||||
switch {
|
||||
case strings.Contains(msg, "type not found: tun"):
|
||||
return "hint: the slim registry no longer registers the tun inbound (shater/registry InboundRegistry must call tun.RegisterInbound) — on the router, l3_tunnel=1 then leaves the engine DOWN, and with the fail-closed nft plane that blackholes the whole LAN, not just ICMP"
|
||||
case strings.Contains(msg, "not included in this build"):
|
||||
return "hint: the gVisor netstack is compiled out — the tag set lost with_gvisor (scripts/router-tags.sh keeps it via with_wireguard); this is the 2026-07-25 outage class: green codegen, dead engine at the first Start on the operator's router"
|
||||
case strings.Contains(msg, "operation not permitted"):
|
||||
return "hint: environment, not code — this runner has root and /dev/net/tun but the kernel refused the device (missing CAP_NET_ADMIN? seccomp?); rerun with --cap-add NET_ADMIN or on the OpenWrt VM"
|
||||
default:
|
||||
return "hint: whatever the cause, on the router this means applying l3_tunnel=1 leaves the engine down, and with the fail-closed nft plane that is a LAN-wide outage"
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,88 @@
|
||||
package generate
|
||||
|
||||
import (
|
||||
"testing"
|
||||
|
||||
"github.com/sagernet/sing-box/option"
|
||||
"github.com/sagernet/sing-box/shater/model"
|
||||
)
|
||||
|
||||
// maxIPv4Datagram is the largest value the 16-bit IPv4 Total Length field can
|
||||
// hold, i.e. the largest IP datagram that can exist on the wire at all. It is
|
||||
// spelled out here rather than written as a literal because it IS the argument:
|
||||
// a device whose MTU is this large cannot be fragmented into.
|
||||
const maxIPv4Datagram = 65535
|
||||
|
||||
// TestL3TunnelMTULeavesNothingForTheKernelToFragment is the reason the number
|
||||
// is the number. Read this before changing l3MTU.
|
||||
//
|
||||
// The MTU of `shater-l3` is NOT a tunnel budget. It governs exactly one thing:
|
||||
// whether the KERNEL splits a packet on its way INTO the device. What the
|
||||
// engine then puts into the tunnel is sized separately and correctly, against
|
||||
// the OUTBOUND's own MTU — sing-tun's ForwardDispatcher.forwardToPort measures
|
||||
// each forwarded packet against Port.PortMTU() and either fragments to it (no
|
||||
// DF) or answers a `fragmentation needed` quoting it (DF).
|
||||
//
|
||||
// The value used to be 1420, the WireGuard payload budget, copied one layer too
|
||||
// far out. That did not make pings fit the tunnel; it made the kernel fragment
|
||||
// everything above 1392 bytes of payload right here, and a fragment is the one
|
||||
// thing sing-tun's dispatcher will not judge: Dispatch returns on
|
||||
// `parsed.fragment` BEFORE calling JudgeFlow, the fragments fall through to the
|
||||
// gVisor stack, which reassembles them and hands the echo to
|
||||
// ICMPForwarder.HandlePacket — whose installFlow requires an UNSPECIFIED port
|
||||
// address that a WireGuard/AWG endpoint never has. It declines, and HandlePacket
|
||||
// FORGES the echo reply. So `ping -s 1392` was honest and `ping -s 1393` was a
|
||||
// lie told by the router (D25).
|
||||
//
|
||||
// Hence the invariant, not merely the constant: the MTU must be at least the
|
||||
// largest datagram that can exist, so that NO packet can ever be fragmented
|
||||
// into this device. Anything smaller re-opens a band of sizes where a ping
|
||||
// reads as tunnelled without leaving the router.
|
||||
func TestL3TunnelMTULeavesNothingForTheKernelToFragment(t *testing.T) {
|
||||
to := l3TunOptions(t)
|
||||
if to.MTU < maxIPv4Datagram {
|
||||
t.Fatalf("l3-in MTU = %d, must be >= %d (the largest IP datagram there can be).\n"+
|
||||
"Below that the kernel fragments packets between the MTU and the datagram size on their way INTO shater-l3, sing-tun's ForwardDispatcher declines fragments without ever asking for a routing verdict, and the gVisor ICMP forwarder answers the echo ITSELF for any outbound whose port address is not unspecified — every WireGuard/AWG endpoint.\n"+
|
||||
"The result is not a dropped ping, it is a FORGED reply: the operator reads a working tunnel off a packet that died on the router. Do not size this to the tunnel MTU — forwardToPort already sizes against Port.PortMTU().",
|
||||
to.MTU, maxIPv4Datagram)
|
||||
}
|
||||
// The concrete value, so that a change is a decision and not a drift. 65535
|
||||
// is also sing-box's own default TUN MTU on Linux (protocol/tun/inbound.go).
|
||||
if to.MTU != maxIPv4Datagram {
|
||||
t.Fatalf("l3-in MTU = %d, want exactly %d — larger is impossible on the wire and buys nothing; if you have a reason, put it in the l3MTU comment and change this test deliberately", to.MTU, maxIPv4Datagram)
|
||||
}
|
||||
}
|
||||
|
||||
// TestL3TunnelMTUIsNotATunnelBudget guards the specific regression: someone
|
||||
// reading "the tunnel is 1420" and "re-aligning" the device to it. The two
|
||||
// numbers are unrelated, and making them equal is what produced the forged
|
||||
// replies in the first place.
|
||||
func TestL3TunnelMTUIsNotATunnelBudget(t *testing.T) {
|
||||
to := l3TunOptions(t)
|
||||
// 1500 is the largest MTU a plain Ethernet LAN hands the router in one
|
||||
// piece; every plausible tunnel budget (1420 for WireGuard, 1280 for a
|
||||
// conservative v6 path, 1412 for AWG with headers) sits below it. An l3-in
|
||||
// MTU anywhere in that range means the kernel is fragmenting into the
|
||||
// device again.
|
||||
if to.MTU <= 1500 {
|
||||
t.Fatalf("l3-in MTU = %d — that is tunnel/LAN-sized, so the kernel will fragment into shater-l3 and the gVisor ICMP forwarder will forge echo replies for WireGuard/AWG. The device MTU and the tunnel MTU are NOT the same number; see l3MTU's comment and D25.", to.MTU)
|
||||
}
|
||||
}
|
||||
|
||||
// l3TunOptions builds the l3_tunnel-on config and returns the TUN inbound's
|
||||
// options, failing with a useful address if the inbound is missing entirely.
|
||||
func l3TunOptions(t *testing.T) *option.TunInboundOptions {
|
||||
t.Helper()
|
||||
g := model.DefaultGlobals()
|
||||
g.L3Tunnel = true
|
||||
ins, warns := buildInboundsWith(g, lanTproxy())
|
||||
tuns := tunInbounds(ins)
|
||||
if len(tuns) != 1 {
|
||||
t.Fatalf("want exactly one L3 TUN inbound, got %d (warns=%v)", len(tuns), warns)
|
||||
}
|
||||
to, ok := tuns[0].Options.(*option.TunInboundOptions)
|
||||
if !ok {
|
||||
t.Fatalf("options are %T, want *option.TunInboundOptions", tuns[0].Options)
|
||||
}
|
||||
return to
|
||||
}
|
||||
+301
-2
@@ -7,9 +7,11 @@ import (
|
||||
C "github.com/sagernet/sing-box/constant"
|
||||
"github.com/sagernet/sing-box/option"
|
||||
"github.com/sagernet/sing/common/json/badoption"
|
||||
N "github.com/sagernet/sing/common/network"
|
||||
|
||||
"github.com/sagernet/sing-box/shater/model"
|
||||
"github.com/sagernet/sing-box/shater/netplane"
|
||||
"github.com/sagernet/sing-box/shater/parse"
|
||||
)
|
||||
|
||||
// buildRoute assembles option.RouteOptions:
|
||||
@@ -117,6 +119,15 @@ func (b *builder) buildRoute() *option.RouteOptions {
|
||||
continue
|
||||
}
|
||||
|
||||
// An ICMP rule is the one shape whose traffic reaches the engine at layer 3,
|
||||
// and layer 3 has prerequisites the rule itself cannot state. Diagnose it
|
||||
// only when its target RESOLVED: an unresolved one already earned the much
|
||||
// louder ruleKillFallback warning above, and the kill fallback is a policy
|
||||
// decision about a broken reference, not about ICMP.
|
||||
if isICMPProto(r.Proto) && ok {
|
||||
b.warnICMPRule(r, want)
|
||||
}
|
||||
|
||||
route := option.RouteActionOptions{Outbound: target}
|
||||
b.applyDPI(&route, target, r.Name)
|
||||
general = append(general, option.Rule{
|
||||
@@ -378,6 +389,35 @@ func (b *builder) ruleMatchers(r model.Rule) (raw option.RawDefaultRule, matched
|
||||
switch p {
|
||||
case "tcp", "udp":
|
||||
raw.Network = badoption.Listable[string]{p}
|
||||
case protoICMP, protoICMPv4, protoICMPv6:
|
||||
// ICMP is a NETWORK, not a sniffed L7 label. Routed through the default
|
||||
// branch below it landed in RawDefaultRule.Protocol, where it is compared
|
||||
// against what the sniffers reported — and the sniffers never report
|
||||
// "icmp" (route.go's pre-match skips the sniff action for an ICMP flow
|
||||
// outright), so the rule was valid, warned about, and dead. The L3 ingress
|
||||
// therefore had no way to express its target at all: every ping fell to
|
||||
// whatever the catch-all resolved to, which on a real config is a group of
|
||||
// proxy nodes that cannot carry layer 3.
|
||||
//
|
||||
// NetworkItem.Match is a plain map lookup over metadata.Network, which
|
||||
// adapter.JudgeFlow sets to N.NetworkICMP for BOTH ICMPv4 and ICMPv6
|
||||
// (header.ICMPv4ProtocolNumber and header.ICMPv6ProtocolNumber share the
|
||||
// one case there). So there is exactly ONE network value here and `icmp`
|
||||
// covers both families — a separate `icmpv6` network would match nothing,
|
||||
// forever.
|
||||
//
|
||||
// The family is still expressible, and precisely: metadata.IPVersion is
|
||||
// derived from the destination address (route.go prepareMatchMetadata,
|
||||
// which PreMatch runs before the rules), and an ICMPv6 packet always
|
||||
// carries an IPv6 destination. `icmpv4`/`icmpv6` therefore narrow the SAME
|
||||
// network with an ip_version item rather than inventing a second network —
|
||||
// no false positives and no false negatives, unlike an inert Protocol
|
||||
// matcher (which would also silently widen to both families if we mapped
|
||||
// it onto plain `icmp`).
|
||||
raw.Network = badoption.Listable[string]{N.NetworkICMP}
|
||||
if v := icmpProtoIPVersion(p); v != 0 {
|
||||
raw.IPVersion = v
|
||||
}
|
||||
default:
|
||||
// Sniffed L7 protocol. route/rule.NewProtocolItem does NO validation —
|
||||
// it just compares the string against what the sniffers reported — so an
|
||||
@@ -386,7 +426,7 @@ func (b *builder) ruleMatchers(r model.Rule) (raw option.RawDefaultRule, matched
|
||||
// the rule to ALL traffic, which for a `direct` target is a leak), but
|
||||
// the operator is told it is inert.
|
||||
if !sniffedProtocols[p] {
|
||||
b.warnf("rule %q: proto %q is not something this engine can detect — the rule is kept but can NEVER match, so its traffic silently follows the rules below it. Use tcp, udp or one of: %s", r.Name, r.Proto, sniffedProtocolList())
|
||||
b.warnf("rule %q: proto %q is not something this engine can detect — the rule is kept but can NEVER match, so its traffic silently follows the rules below it. Use tcp, udp, icmp (icmpv4/icmpv6 narrow it to one family) or one of: %s", r.Name, r.Proto, sniffedProtocolList())
|
||||
}
|
||||
raw.Protocol = badoption.Listable[string]{p}
|
||||
}
|
||||
@@ -411,10 +451,269 @@ func (b *builder) ruleMatchers(r model.Rule) (raw option.RawDefaultRule, matched
|
||||
return raw, matched
|
||||
}
|
||||
|
||||
// The ICMP spellings `Rule.Proto` accepts. All three become the SAME engine
|
||||
// network (N.NetworkICMP); the two family-qualified ones additionally pin
|
||||
// RawDefaultRule.IPVersion. See the switch in ruleMatchers for why there is only
|
||||
// one network and why the family is an ip_version item rather than a second one.
|
||||
const (
|
||||
protoICMP = "icmp"
|
||||
protoICMPv4 = "icmpv4"
|
||||
protoICMPv6 = "icmpv6"
|
||||
)
|
||||
|
||||
// isICMPProto reports whether a Rule.Proto value asks for the L3 ingress.
|
||||
func isICMPProto(proto string) bool {
|
||||
switch strings.TrimSpace(strings.ToLower(proto)) {
|
||||
case protoICMP, protoICMPv4, protoICMPv6:
|
||||
return true
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
// icmpProtoIPVersion is the ip_version an ICMP proto narrows to; 0 = both
|
||||
// families (plain `icmp`) or not an ICMP proto at all.
|
||||
func icmpProtoIPVersion(proto string) int {
|
||||
switch strings.TrimSpace(strings.ToLower(proto)) {
|
||||
case protoICMPv4:
|
||||
return 4
|
||||
case protoICMPv6:
|
||||
return 6
|
||||
}
|
||||
return 0
|
||||
}
|
||||
|
||||
// warnICMPRule reports the ways an emitted `proto icmp` rule can still be a
|
||||
// no-op, none of which is visible anywhere else.
|
||||
//
|
||||
// A rule that matches nothing is normally cheap: its traffic falls through to
|
||||
// the rules below it. Not here. ICMP has no fall-through — the packet either
|
||||
// reaches an L3-capable outbound or is DROPPED (route.preMatchFlow's
|
||||
// l3-honest-drop block, which exists so the TUN stack cannot forge an echo reply
|
||||
// for a path that never carried the packet). So an ICMP rule that cannot fire is
|
||||
// not a dead setting, it is ping that stops working, with a rule in the UI that
|
||||
// says it should.
|
||||
//
|
||||
// Deliberately worded clear of shater/apply's criticalMarkers: a dropped ping is
|
||||
// not a protection gap (nothing leaks — the failure is fail-CLOSED), so these are
|
||||
// warnings, and phrases like "never applies" / "NOT emitted" would light the
|
||||
// panel's alarm banner for something that costs the operator ping and nothing else.
|
||||
func (b *builder) warnICMPRule(r model.Rule, want string) {
|
||||
if !b.m.Globals.L3Tunnel {
|
||||
// The prerequisite, and the only one whose absence makes the other checks
|
||||
// moot: with no L3 ingress the packet never enters the engine, so the target
|
||||
// is not consulted at all.
|
||||
b.warnf("rule %q: proto %s matches only traffic that reaches the engine at layer 3, and nothing does while globals l3_tunnel is off — kernel TPROXY diverts TCP and UDP and nothing else, so no ping ever enters the engine and this rule cannot fire. ICMP is governed by globals untunnelable (%s) instead, whatever this rule's target %q says. Turn l3_tunnel on to make the rule live", r.Name, strings.ToLower(strings.TrimSpace(r.Proto)), untunnelablePolicyName(b.m.Globals.Untunnelable), want)
|
||||
return
|
||||
}
|
||||
if icmpProtoIPVersion(r.Proto) == 6 && !b.m.Globals.IPv6 {
|
||||
// netplane/nft.go marks ipv6-icmp into the L3 TUN only under Globals.IPv6,
|
||||
// and generate/inbound.go gives that TUN an IPv6 address on the same
|
||||
// condition. With IPv6 off the v6 half of the ingress simply does not exist.
|
||||
b.warnf("rule %q: proto icmpv6 needs globals ipv6 on — the L3 ingress marks ipv6-icmp into the engine's TUN only then, and the TUN is given no IPv6 address either, so with IPv6 off this rule cannot fire. Use proto icmp (one value, both families) or turn ipv6 on", r.Name)
|
||||
return
|
||||
}
|
||||
// A port matcher and ICMP are mutually exclusive by construction:
|
||||
// adapter.JudgeFlow zeroes source and destination ports for an ICMP flow
|
||||
// before PreMatch runs, so a PortItem next to the network item can never be
|
||||
// satisfied. The rule is emitted (dropping the port would WIDEN it) but it is
|
||||
// dead.
|
||||
if single, ranges, _ := splitPorts(r.DstPort); len(single)+len(ranges) > 0 {
|
||||
b.warnf("rule %q: proto %s together with a port matcher can never match — an ICMP packet has no port, and the engine zeroes both ports on an ICMP flow before matching it (adapter.JudgeFlow). Drop the port from this rule, or move the port part into a rule of its own", r.Name, strings.ToLower(strings.TrimSpace(r.Proto)))
|
||||
}
|
||||
switch b.l3Target(want, 0) {
|
||||
case l3Drops:
|
||||
b.warnf("rule %q: proto %s is routed to %q, which cannot carry a layer-3 packet — only wireguard/AmneziaWG nodes and direct/interface egresses reach the engine's flow path; every proxy protocol (vless, vmess, trojan, shadowsocks, hysteria2, tuic, shadowtls) and the byedpi SOCKS egress cannot, because none of them can implement it. The engine DROPS a ping routed at such a target rather than let the TUN stack answer the echo itself from a path that never carried the packet, so this rule makes those pings fail instead of tunnelling them. Point it at a wireguard/AmneziaWG node, at an interface egress, or at a chain whose LAST hop is one of those", r.Name, strings.ToLower(strings.TrimSpace(r.Proto)), want)
|
||||
case l3Partial:
|
||||
b.warnf("rule %q: proto %s is routed to %q, whose members disagree about layer 3 — the ping is carried while the group's current member is a wireguard/AmneziaWG node and dropped while it is a proxy-protocol one, and nothing in the UI says which is in force right now. Point the rule at the L3-capable node itself (or at a group holding only those) if ping must behave the same from one minute to the next", r.Name, strings.ToLower(strings.TrimSpace(r.Proto)), want)
|
||||
}
|
||||
}
|
||||
|
||||
// untunnelablePolicyName spells the Untunnelable policy for a diagnostic, naming
|
||||
// the empty value as the "block" it means (model.Globals.Untunnelable).
|
||||
func untunnelablePolicyName(policy string) string {
|
||||
if p := strings.ToLower(strings.TrimSpace(policy)); p != "" {
|
||||
return p
|
||||
}
|
||||
return "block"
|
||||
}
|
||||
|
||||
// l3Verdict is what a route target does with a layer-3 packet.
|
||||
type l3Verdict int
|
||||
|
||||
const (
|
||||
// l3Unknown: undecidable from the model alone — say nothing rather than guess.
|
||||
l3Unknown l3Verdict = iota
|
||||
// l3Carries: the emitted outbound implements adapter.FlowOutbound.
|
||||
l3Carries
|
||||
// l3Drops: it does not, so route.preMatchFlow drops the packet.
|
||||
l3Drops
|
||||
// l3Blocks: `block`. It drops the packet too, but that IS the stated policy,
|
||||
// so it is never reported.
|
||||
l3Blocks
|
||||
// l3Partial: a group whose members disagree; the answer changes with the pick.
|
||||
l3Partial
|
||||
)
|
||||
|
||||
// l3Target answers, from the MODEL, whether a rule target ends up at an outbound
|
||||
// that can carry a layer-3 packet.
|
||||
//
|
||||
// # Why this is decidable here at all
|
||||
//
|
||||
// The capability is not a runtime property to be discovered: it is fixed by the
|
||||
// outbound TYPE this package is about to emit, and adapter registration makes the
|
||||
// list exhaustive. Only `direct` (protocol/direct/outbound.go:66), `wireguard`
|
||||
// (protocol/wireguard/endpoint.go:73), `tailscale` and `bridge` declare
|
||||
// N.NetworkICMP among their networks, and of those exactly two are reachable from
|
||||
// a shater model: a wireguard/AmneziaWG node, and the direct outbound behind
|
||||
// `direct` or an interface/direct egress. Everything else this generator emits —
|
||||
// vless, vmess, trojan, shadowsocks, hysteria2, tuic, shadowtls and the byedpi
|
||||
// SOCKS egress — cannot.
|
||||
//
|
||||
// # The predicate that MUST change in lockstep
|
||||
//
|
||||
// "Is this node an endpoint?" is decided in outbound.go by
|
||||
// `p.WG != nil || p.Protocol == "wireguard"`. l3Node repeats that test. If one
|
||||
// side learns a new L3-capable node kind and the other does not, this diagnosis
|
||||
// starts lying in whichever direction the drift went — a false alarm on a working
|
||||
// ping, or silence on a broken one.
|
||||
//
|
||||
// depth bounds the chain->chain recursion; expandHops already refuses cycles, so
|
||||
// it is a belt on top of a brace.
|
||||
func (b *builder) l3Target(want string, depth int) l3Verdict {
|
||||
if depth > 8 {
|
||||
return l3Unknown
|
||||
}
|
||||
// The kind switch mirrors resolveTarget exactly — the two must agree about what
|
||||
// a target string means, or this diagnoses a different outbound than the one the
|
||||
// rule is routed to.
|
||||
kind, name := model.SplitTarget(want)
|
||||
switch strings.ToLower(kind) {
|
||||
case tagDirect, "":
|
||||
if strings.EqualFold(want, tagBlock) {
|
||||
return l3Blocks
|
||||
}
|
||||
return l3Carries
|
||||
case tagBlock:
|
||||
return l3Blocks
|
||||
case "node":
|
||||
return b.l3Node(name)
|
||||
case "group":
|
||||
return b.l3Group(name)
|
||||
case "egress":
|
||||
return b.l3Egress(name)
|
||||
case "chain":
|
||||
return b.l3Chain(name, depth)
|
||||
default:
|
||||
// Bare name: node then group, same order resolveTarget uses.
|
||||
if b.nodeTags[want] {
|
||||
return b.l3Node(want)
|
||||
}
|
||||
if b.groupTags[want] {
|
||||
return b.l3Group(want)
|
||||
}
|
||||
return l3Unknown
|
||||
}
|
||||
}
|
||||
|
||||
// l3Node: a node carries layer 3 exactly when it is emitted as a WireGuard/
|
||||
// AmneziaWG ENDPOINT rather than a proxy outbound.
|
||||
func (b *builder) l3Node(name string) l3Verdict {
|
||||
for i := range b.m.Nodes {
|
||||
n := b.m.Nodes[i]
|
||||
if n.Name != name || !n.Enabled {
|
||||
continue
|
||||
}
|
||||
p, err := parse.ParseShareLink(n.URI)
|
||||
if err != nil {
|
||||
// Unreachable from warnICMPRule (an unparseable node has no tag, so the
|
||||
// target would not have resolved), and not ours to report twice anyway.
|
||||
return l3Unknown
|
||||
}
|
||||
if p.WG != nil || p.Protocol == "wireguard" {
|
||||
return l3Carries
|
||||
}
|
||||
return l3Drops
|
||||
}
|
||||
return l3Unknown
|
||||
}
|
||||
|
||||
// l3Group folds its members' verdicts. route.preMatchFlow unwraps a group through
|
||||
// group.Now() before testing the outbound, so the group's answer IS its current
|
||||
// member's — which is why a mixed group is reported as its own case rather than
|
||||
// rounded to either side.
|
||||
func (b *builder) l3Group(name string) l3Verdict {
|
||||
g, found := b.findGroup(name)
|
||||
if !found {
|
||||
return l3Unknown
|
||||
}
|
||||
var members []string
|
||||
// groupMembers re-emits the member diagnostics buildGroups already surfaced.
|
||||
b.withoutNewWarnings(func() { members = b.groupMembers(g) })
|
||||
carries, drops := 0, 0
|
||||
for _, member := range members {
|
||||
switch b.l3Node(member) {
|
||||
case l3Carries:
|
||||
carries++
|
||||
case l3Drops:
|
||||
drops++
|
||||
}
|
||||
}
|
||||
switch {
|
||||
case carries == 0 && drops == 0:
|
||||
return l3Unknown
|
||||
case carries == 0:
|
||||
return l3Drops
|
||||
case drops == 0:
|
||||
return l3Carries
|
||||
default:
|
||||
return l3Partial
|
||||
}
|
||||
}
|
||||
|
||||
// l3Egress: an interface/direct egress is a DIRECT outbound (the one proxy
|
||||
// outbound in the shipped registry that builds a ping.Port), byedpi is SOCKS, and
|
||||
// any other type emits no outbound at all — so it never reaches this diagnosis.
|
||||
func (b *builder) l3Egress(name string) l3Verdict {
|
||||
for _, eg := range b.m.Egresses {
|
||||
if eg.Name != name {
|
||||
continue
|
||||
}
|
||||
switch strings.ToLower(strings.TrimSpace(eg.Type)) {
|
||||
case "interface", "direct", "":
|
||||
return l3Carries
|
||||
default:
|
||||
return l3Drops
|
||||
}
|
||||
}
|
||||
return l3Unknown
|
||||
}
|
||||
|
||||
// l3Chain: a rule routed at a chain enters at the LAST hop's wrapper (chain.go —
|
||||
// Detour points backwards, so Ln is the outbound the rule is handed to and the
|
||||
// exit on the wire). That wrapper is a copy of Ln's own outbound, so Ln's
|
||||
// capability is the chain's.
|
||||
func (b *builder) l3Chain(name string, depth int) l3Verdict {
|
||||
var (
|
||||
hops []string
|
||||
baseDetour string
|
||||
expanded bool
|
||||
)
|
||||
// expandHops flattens sub-chains and reports cycles/undefined references;
|
||||
// resolveChain already surfaced all of that for this same chain.
|
||||
b.withoutNewWarnings(func() {
|
||||
expanded = b.expandHops(name, map[string]bool{}, &hops, &baseDetour, true)
|
||||
})
|
||||
if !expanded || len(hops) == 0 {
|
||||
return l3Unknown
|
||||
}
|
||||
return b.l3Target(hops[len(hops)-1], depth+1)
|
||||
}
|
||||
|
||||
// sniffedProtocols is exactly the set of L7 labels this engine's sniffers can
|
||||
// ever put on a connection (route/rule.RuleActionSniff.build + the default
|
||||
// stream/packet sniffer sets in route/route.go). A `proto` outside this set —
|
||||
// and outside tcp/udp — matches nothing, forever.
|
||||
// and outside the tcp/udp/icmp NETWORKS handled ahead of it in ruleMatchers —
|
||||
// matches nothing, forever.
|
||||
var sniffedProtocols = map[string]bool{
|
||||
C.ProtocolTLS: true,
|
||||
C.ProtocolHTTP: true,
|
||||
|
||||
@@ -0,0 +1,377 @@
|
||||
// The rule side of the L3 ingress: whether `proto icmp` can be SAID at all, and
|
||||
// whether saying it is honest.
|
||||
//
|
||||
// Before this, `icmp` fell through ruleMatchers' proto switch into
|
||||
// RawDefaultRule.Protocol — the sniffed-L7 field — where it was compared against
|
||||
// labels the sniffers report. They never report "icmp" (route.go's pre-match
|
||||
// skips the sniff action for an ICMP flow outright), so the rule was structurally
|
||||
// valid and permanently dead. The consequence was not a dead setting but a dead
|
||||
// FEATURE: with no way to write "ICMP goes here", every ping fell to whatever the
|
||||
// catch-all resolved to, and on the configuration this was built for that is a
|
||||
// group of VLESS nodes, which cannot carry layer 3 at all.
|
||||
//
|
||||
// Portable (no box.New): these assert on the generated option.Options and on the
|
||||
// warning texts only. The live-engine proofs that a direct/wireguard outbound
|
||||
// really carries the flow live in l3_egress_linux_test.go and
|
||||
// l3_integration_linux_test.go.
|
||||
package generate
|
||||
|
||||
import (
|
||||
"strings"
|
||||
"testing"
|
||||
|
||||
"github.com/sagernet/sing-box/option"
|
||||
|
||||
"github.com/sagernet/sing-box/shater/model"
|
||||
)
|
||||
|
||||
// --- fixture ----------------------------------------------------------------
|
||||
|
||||
// icmpModel builds a model carrying one of everything the L3 verdict has to tell
|
||||
// apart: a WireGuard node (an ENDPOINT — carries layer 3), a Shadowsocks node (a
|
||||
// proxy outbound — cannot), an interface egress (a direct outbound — carries), a
|
||||
// byedpi egress (SOCKS — cannot), three groups spanning all-proxy/all-L3/mixed,
|
||||
// and two chains that differ only in which kind of node they EXIT through.
|
||||
func icmpModel(l3Tunnel bool, rules ...model.Rule) *model.Model {
|
||||
g := model.DefaultGlobals()
|
||||
g.DNSIntercept = false // not the subject; keeps the DNS diagnostics out
|
||||
g.KillSwitch = "closed"
|
||||
g.L3Tunnel = l3Tunnel
|
||||
return &model.Model{
|
||||
Globals: g,
|
||||
Inbounds: []model.Inbound{
|
||||
{Name: "lan", Enabled: true, Type: "tproxy", TproxyPort: 12345, TCP: true, UDP: true},
|
||||
},
|
||||
Nodes: []model.Node{
|
||||
{Name: "wg1", Enabled: true, URI: wgDedupURI(wgDedupKey(1), wgDedupKey(2), "203.0.113.10", 51820)},
|
||||
{Name: "wg2", Enabled: true, URI: wgDedupURI(wgDedupKey(3), wgDedupKey(4), "203.0.113.11", 51820)},
|
||||
{Name: "ss1", Enabled: true, URI: "ss://aes-256-gcm:secret@203.0.113.1:8388#ss1"},
|
||||
},
|
||||
Egresses: []model.Egress{
|
||||
{Name: "wan2", Type: "interface", Interface: "wan2"},
|
||||
{Name: "bd", Type: "byedpi", Port: 1080},
|
||||
},
|
||||
Groups: []model.Group{
|
||||
{Name: "proxies", Source: "manual", Nodes: []string{"ss1"}},
|
||||
{Name: "l3only", Source: "manual", Nodes: []string{"wg1", "wg2"}},
|
||||
{Name: "mixed", Source: "manual", Nodes: []string{"ss1", "wg1"}},
|
||||
},
|
||||
Chains: []model.Chain{
|
||||
{Name: "exit-proxy", Hops: []string{"node:wg1", "node:ss1"}},
|
||||
{Name: "exit-wg", Hops: []string{"node:ss1", "node:wg1"}},
|
||||
},
|
||||
Rules: rules,
|
||||
}
|
||||
}
|
||||
|
||||
// genICMP generates the fixture and returns the model-derived route rules plus
|
||||
// the warnings.
|
||||
func genICMP(t *testing.T, l3Tunnel bool, rules ...model.Rule) ([]option.Rule, []string) {
|
||||
t.Helper()
|
||||
opts, warns, err := GenerateWithWarnings(icmpModel(l3Tunnel, rules...))
|
||||
if err != nil {
|
||||
t.Fatalf("Generate: %v", err)
|
||||
}
|
||||
return generalRules(opts.Route), warns
|
||||
}
|
||||
|
||||
// oneICMPRule generates a single-rule model and returns that rule's matchers.
|
||||
func oneICMPRule(t *testing.T, l3Tunnel bool, r model.Rule) (option.RawDefaultRule, []string) {
|
||||
t.Helper()
|
||||
rules, warns := genICMP(t, l3Tunnel, r)
|
||||
if len(rules) != 1 {
|
||||
t.Fatalf("want exactly 1 emitted route rule, got %d (warns=%v) — a rule the generator drops carries no target at all, and for ICMP that is not a fall-through but a drop", len(rules), warns)
|
||||
}
|
||||
return rules[0].DefaultOptions.RawDefaultRule, warns
|
||||
}
|
||||
|
||||
// icmpRule is a `proto <p>` rule scoped to one LAN source, pointed at target.
|
||||
func icmpRule(name, proto, target string) model.Rule {
|
||||
return model.Rule{
|
||||
Name: name, Enabled: true, Order: 10,
|
||||
Src: []string{"192.168.1.0/24"},
|
||||
Proto: proto,
|
||||
Target: target,
|
||||
}
|
||||
}
|
||||
|
||||
// --- what the matcher must become -------------------------------------------
|
||||
|
||||
// TestProtoICMPBecomesTheNetworkNotTheSniffedProtocol is the fix itself.
|
||||
//
|
||||
// route/rule.NewNetworkItem matches metadata.Network with a plain map lookup, and
|
||||
// adapter.JudgeFlow sets that to N.NetworkICMP ("icmp") for an ICMP flow — so the
|
||||
// value belongs in Network. RawDefaultRule.Protocol is the sniffed-L7 field
|
||||
// (NewProtocolItem compares against what the sniffers labelled the connection),
|
||||
// and nothing ever labels a flow "icmp": route.go's PreMatch skips the sniff
|
||||
// action for ICMP before it starts. A value there is a rule that cannot fire.
|
||||
func TestProtoICMPBecomesTheNetworkNotTheSniffedProtocol(t *testing.T) {
|
||||
raw, warns := oneICMPRule(t, true, icmpRule("ping", "icmp", "node:wg1"))
|
||||
if len(raw.Protocol) != 0 {
|
||||
t.Fatalf("proto icmp landed in RawDefaultRule.Protocol = %v — that field is matched against the SNIFFED protocol label, and no sniffer ever produces \"icmp\" (PreMatch skips sniffing for an ICMP flow), so the rule is valid, silent and permanently dead: every ping keeps falling to the catch-all instead of the target the operator wrote", raw.Protocol)
|
||||
}
|
||||
if len(raw.Network) != 1 || raw.Network[0] != "icmp" {
|
||||
t.Fatalf("RawDefaultRule.Network = %v, want [icmp] — NetworkItem.Match is a map lookup over metadata.Network, which adapter.JudgeFlow sets to exactly this string for an ICMP flow; anything else and the L3 ingress has no expressible target at all", raw.Network)
|
||||
}
|
||||
if raw.IPVersion != 0 {
|
||||
t.Fatalf("RawDefaultRule.IPVersion = %d, want 0 — plain `icmp` must cover BOTH families (JudgeFlow maps ICMPv4 and ICMPv6 to the one network), so narrowing it here would silently drop half the pings the operator asked for", raw.IPVersion)
|
||||
}
|
||||
if routeWarnsHave(warns, "is not something this engine can detect") {
|
||||
t.Fatalf("icmp was reported as an undetectable sniffed protocol: %v — it is a NETWORK, and the warning tells the operator to stop using the only spelling that works", warns)
|
||||
}
|
||||
}
|
||||
|
||||
// TestProtoICMPFamilySpellingsNarrowByIPVersion pins the answer to "is ICMPv6 a
|
||||
// second network?": it is not. adapter.JudgeFlow folds
|
||||
// header.ICMPv4ProtocolNumber and header.ICMPv6ProtocolNumber into ONE case and
|
||||
// sets N.NetworkICMP for both, so an `icmpv6` network value would match nothing,
|
||||
// forever. The family is still expressible, and exactly: metadata.IPVersion comes
|
||||
// from the destination address (prepareMatchMetadata, which PreMatch runs), and
|
||||
// an ICMPv6 packet always has an IPv6 destination. So the family spellings narrow
|
||||
// the same network with ip_version instead of inventing a second one.
|
||||
func TestProtoICMPFamilySpellingsNarrowByIPVersion(t *testing.T) {
|
||||
for _, tc := range []struct {
|
||||
proto string
|
||||
want int
|
||||
}{{"icmpv4", 4}, {"icmpv6", 6}} {
|
||||
raw, warns := oneICMPRule(t, true, icmpRule("ping", tc.proto, "node:wg1"))
|
||||
if len(raw.Network) != 1 || raw.Network[0] != "icmp" {
|
||||
t.Fatalf("proto %s: Network = %v, want [icmp] — there is no separate icmpv6 network in this engine; emitting one produces a rule that can never match", tc.proto, raw.Network)
|
||||
}
|
||||
if raw.IPVersion != tc.want {
|
||||
t.Fatalf("proto %s: IPVersion = %d, want %d — without it the rule silently covers the OTHER family too, which is a different rule than the one written", tc.proto, raw.IPVersion, tc.want)
|
||||
}
|
||||
if len(raw.Protocol) != 0 {
|
||||
t.Fatalf("proto %s: landed in Protocol = %v (warns=%v) — the sniffed-L7 field, where it can never match", tc.proto, raw.Protocol, warns)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestProtoTCPUDPAndSniffedAreUnchanged: the new branch must not move any value
|
||||
// that already worked. tcp/udp stay networks, a sniffed L7 label stays in
|
||||
// Protocol, and neither picks up an ip_version it never had.
|
||||
func TestProtoTCPUDPAndSniffedAreUnchanged(t *testing.T) {
|
||||
for _, network := range []string{"tcp", "udp"} {
|
||||
raw, _ := oneICMPRule(t, true, icmpRule("t", network, "node:ss1"))
|
||||
if len(raw.Network) != 1 || raw.Network[0] != network {
|
||||
t.Fatalf("proto %s: Network = %v, want [%s]", network, raw.Network, network)
|
||||
}
|
||||
if len(raw.Protocol) != 0 || raw.IPVersion != 0 {
|
||||
t.Fatalf("proto %s: Protocol = %v / IPVersion = %d, want empty/0 — the ICMP branch leaked into the transport branch", network, raw.Protocol, raw.IPVersion)
|
||||
}
|
||||
}
|
||||
raw, warns := oneICMPRule(t, true, icmpRule("t", "tls", "node:ss1"))
|
||||
if len(raw.Protocol) != 1 || raw.Protocol[0] != "tls" {
|
||||
t.Fatalf("proto tls: Protocol = %v, want [tls] — a sniffed L7 label belongs in the sniffed field", raw.Protocol)
|
||||
}
|
||||
if len(raw.Network) != 0 || raw.IPVersion != 0 {
|
||||
t.Fatalf("proto tls: Network = %v / IPVersion = %d, want empty/0", raw.Network, raw.IPVersion)
|
||||
}
|
||||
if routeWarnsHave(warns, "is not something this engine can detect") {
|
||||
t.Fatalf("tls is sniffable and was reported as not: %v", warns)
|
||||
}
|
||||
// The vocabulary the undetectable-proto warning offers must now name icmp —
|
||||
// it is the list an operator reads when their spelling was rejected, and
|
||||
// leaving icmp out of it points them away from the only working value.
|
||||
_, warns = oneICMPRule(t, true, icmpRule("t", "ping", "node:ss1"))
|
||||
if !routeWarnsHave(warns, "is not something this engine can detect") {
|
||||
t.Fatalf("proto \"ping\" is neither a network nor a sniffed label, yet nothing was reported: %v", warns)
|
||||
}
|
||||
if !routeWarnsHave(warns, "Use tcp, udp, icmp") {
|
||||
t.Fatalf("the undetectable-proto warning still offers only tcp/udp + sniffed labels: %v — an operator who wrote \"ping\" is told every value EXCEPT the one that would work", warns)
|
||||
}
|
||||
}
|
||||
|
||||
// --- the two prerequisites the rule cannot state itself ----------------------
|
||||
|
||||
// TestProtoICMPWithoutL3TunnelIsReported: with globals l3_tunnel off, kernel
|
||||
// TPROXY diverts TCP and UDP and nothing else, so no ICMP packet ever enters the
|
||||
// engine and the rule — perfectly well-formed — matches nothing at all. Silence
|
||||
// here is the worst kind: the panel shows a rule that says ping is tunnelled.
|
||||
func TestProtoICMPWithoutL3TunnelIsReported(t *testing.T) {
|
||||
_, warns := oneICMPRule(t, false, icmpRule("ping", "icmp", "node:wg1"))
|
||||
if !routeWarnsHave(warns, "globals l3_tunnel is off") {
|
||||
t.Fatalf("a proto icmp rule under l3_tunnel=0 was accepted in silence: %v — the L3 ingress is the ONLY path an ICMP packet has into the engine, so without it the rule is decoration", warns)
|
||||
}
|
||||
// And the opposite: turning the ingress on must silence it, or the warning is
|
||||
// noise that trains the operator to ignore the section.
|
||||
_, warns = oneICMPRule(t, true, icmpRule("ping", "icmp", "node:wg1"))
|
||||
if routeWarnsHave(warns, "globals l3_tunnel is off") {
|
||||
t.Fatalf("l3_tunnel is ON and the rule was still reported as dead: %v", warns)
|
||||
}
|
||||
}
|
||||
|
||||
// TestProtoICMPv6WithoutIPv6IsReported: the v6 half of the ingress is gated twice
|
||||
// on globals ipv6 — netplane/nft.go marks ipv6-icmp into the TUN only then, and
|
||||
// generate/inbound.go gives the TUN an IPv6 address on the same condition. An
|
||||
// icmpv6 rule with IPv6 off is therefore inert for a reason nothing else states.
|
||||
func TestProtoICMPv6WithoutIPv6IsReported(t *testing.T) {
|
||||
m := icmpModel(true, icmpRule("ping6", "icmpv6", "node:wg1"))
|
||||
m.Globals.IPv6 = false
|
||||
_, warns, err := GenerateWithWarnings(m)
|
||||
if err != nil {
|
||||
t.Fatalf("Generate: %v", err)
|
||||
}
|
||||
if !routeWarnsHave(warns, "proto icmpv6 needs globals ipv6 on") {
|
||||
t.Fatalf("an icmpv6 rule under ipv6=0 was accepted in silence: %v — neither the nft marking nor the TUN address exists in that configuration, so the rule cannot fire", warns)
|
||||
}
|
||||
_, warns = oneICMPRule(t, true, icmpRule("ping6", "icmpv6", "node:wg1"))
|
||||
if routeWarnsHave(warns, "proto icmpv6 needs globals ipv6 on") {
|
||||
t.Fatalf("ipv6 is on and the icmpv6 rule was still reported: %v", warns)
|
||||
}
|
||||
}
|
||||
|
||||
// TestProtoICMPWithPortMatcherIsReported: adapter.JudgeFlow zeroes both ports on
|
||||
// an ICMP flow before PreMatch runs, so a port item sitting next to the network
|
||||
// item can never be satisfied. The rule is still EMITTED — dropping the port
|
||||
// matcher would widen it — but it is dead, and only this says so.
|
||||
func TestProtoICMPWithPortMatcherIsReported(t *testing.T) {
|
||||
r := icmpRule("ping", "icmp", "node:wg1")
|
||||
r.DstPort = "443"
|
||||
_, warns := oneICMPRule(t, true, r)
|
||||
if !routeWarnsHave(warns, "an ICMP packet has no port") {
|
||||
t.Fatalf("proto icmp + dst_port was accepted in silence: %v — the two matchers are mutually exclusive by construction, so the rule can never fire", warns)
|
||||
}
|
||||
_, warns = oneICMPRule(t, true, icmpRule("ping", "icmp", "node:wg1"))
|
||||
if routeWarnsHave(warns, "an ICMP packet has no port") {
|
||||
t.Fatalf("a portless icmp rule was reported as having a port: %v", warns)
|
||||
}
|
||||
}
|
||||
|
||||
// --- does the TARGET carry layer 3? -----------------------------------------
|
||||
|
||||
// TestProtoICMPToAnL4TargetIsReported is the honesty half. An ICMP flow has no
|
||||
// fall-through: route.preMatchFlow's l3-honest-drop block DROPS a ping routed at
|
||||
// an outbound that cannot carry layer 3, precisely so the TUN stack cannot forge
|
||||
// an echo reply for a path that never saw the packet. So "ICMP -> group of VLESS
|
||||
// nodes" is not a dead setting, it is ping that stops working — and the operator
|
||||
// wrote the rule believing the opposite.
|
||||
//
|
||||
// Every target below is decidable from the model, because the capability is fixed
|
||||
// by the outbound TYPE this package emits: only direct (protocol/direct) and the
|
||||
// wireguard/AmneziaWG endpoints declare N.NetworkICMP.
|
||||
func TestProtoICMPToAnL4TargetIsReported(t *testing.T) {
|
||||
for _, target := range []string{
|
||||
"node:ss1", // a proxy outbound
|
||||
"group:proxies", // a group of nothing but proxy outbounds
|
||||
"egress:bd", // byedpi: a SOCKS hop, which cannot implement tun.Port
|
||||
"chain:exit-proxy", // the rule enters at the LAST hop, and that one is ss1
|
||||
} {
|
||||
_, warns := oneICMPRule(t, true, icmpRule("ping", "icmp", target))
|
||||
if !routeWarnsHave(warns, "cannot carry a layer-3 packet") {
|
||||
t.Fatalf("proto icmp -> %s was accepted in silence: %v — the engine DROPS that ping (route.preMatchFlow, l3-honest-drop) and nothing else in the UI says so; the operator reads a rule claiming ping is tunnelled and a ping that fails", target, warns)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestProtoICMPToAnL3TargetIsSilent: the same check must stay quiet for every
|
||||
// target that really does carry the packet, including `block` — a blocked ping is
|
||||
// a stated policy, not an accident — and a chain that merely PASSES THROUGH a
|
||||
// proxy hop on its way to a WireGuard exit.
|
||||
func TestProtoICMPToAnL3TargetIsSilent(t *testing.T) {
|
||||
for _, target := range []string{
|
||||
"node:wg1", // a wireguard endpoint
|
||||
"group:l3only", // a group of nothing else
|
||||
"egress:wan2", // an interface egress = a direct outbound
|
||||
"direct", // the baseline direct outbound
|
||||
"block", // dropping the ping IS the policy here
|
||||
"chain:exit-wg", // enters at the LAST hop, which is wg1
|
||||
} {
|
||||
_, warns := oneICMPRule(t, true, icmpRule("ping", "icmp", target))
|
||||
if routeWarnsHave(warns, "cannot carry a layer-3 packet") {
|
||||
t.Fatalf("proto icmp -> %s was reported as unable to carry layer 3: %v — a false alarm on a working path teaches the operator to ignore the one that is real", target, warns)
|
||||
}
|
||||
if routeWarnsHave(warns, "disagree about layer 3") {
|
||||
t.Fatalf("proto icmp -> %s was reported as a mixed group: %v", target, warns)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestProtoICMPToAMixedGroupIsReported: route.preMatchFlow unwraps a group
|
||||
// through group.Now() before testing the outbound, so a group holding both kinds
|
||||
// answers differently from one pick to the next — ping works, then does not, with
|
||||
// nothing in the UI to say which member is in force. That is its own report, not
|
||||
// a rounding of either side.
|
||||
func TestProtoICMPToAMixedGroupIsReported(t *testing.T) {
|
||||
_, warns := oneICMPRule(t, true, icmpRule("ping", "icmp", "group:mixed"))
|
||||
if !routeWarnsHave(warns, "disagree about layer 3") {
|
||||
t.Fatalf("proto icmp -> a group of one wireguard and one proxy node was accepted in silence: %v — the ping's fate follows the group's current pick", warns)
|
||||
}
|
||||
if routeWarnsHave(warns, "cannot carry a layer-3 packet") {
|
||||
t.Fatalf("a MIXED group was reported as unable to carry layer 3 at all: %v — it can, half the time, and telling the operator otherwise sends them to change a target that is only unstable", warns)
|
||||
}
|
||||
}
|
||||
|
||||
// TestProtoICMPDiagnosticsAreQuietForNonICMPRules: none of the above may fire on
|
||||
// the tcp/udp/sniffed rules that make up every existing config. A byedpi egress
|
||||
// carrying TCP is exactly right, and saying otherwise would flood the panel.
|
||||
func TestProtoICMPDiagnosticsAreQuietForNonICMPRules(t *testing.T) {
|
||||
_, warns := genICMP(t, true,
|
||||
icmpRule("a", "tcp", "egress:bd"),
|
||||
icmpRule("b", "udp", "group:proxies"),
|
||||
icmpRule("c", "tls", "node:ss1"),
|
||||
)
|
||||
for _, marker := range []string{
|
||||
"cannot carry a layer-3 packet",
|
||||
"disagree about layer 3",
|
||||
"globals l3_tunnel is off",
|
||||
"an ICMP packet has no port",
|
||||
} {
|
||||
if routeWarnsHave(warns, marker) {
|
||||
t.Fatalf("a non-ICMP rule drew the ICMP diagnostic %q: %v", marker, warns)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestProtoICMPUnresolvedTargetIsNotDoubleReported: a rule whose target does not
|
||||
// resolve already earns ruleKillFallback's much louder warning and is routed to
|
||||
// block. A capability verdict on the target it never reaches would compete with
|
||||
// it — the operator's problem is the broken reference, and it is the one sentence
|
||||
// that must be read first.
|
||||
//
|
||||
// The shape below is the one where the two actually collide: a group named after
|
||||
// a node is SKIPPED by buildGroups (the outbound manager last-wins on a duplicate
|
||||
// tag), so `group:ss1` does not resolve — while the group DEFINITION is still in
|
||||
// the model, holding a proxy member, so a verdict is perfectly computable for it.
|
||||
func TestProtoICMPUnresolvedTargetIsNotDoubleReported(t *testing.T) {
|
||||
m := icmpModel(true, icmpRule("ping", "icmp", "group:ss1"))
|
||||
m.Groups = append(m.Groups, model.Group{Name: "ss1", Source: "manual", Nodes: []string{"ss1"}})
|
||||
_, warns, err := GenerateWithWarnings(m)
|
||||
if err != nil {
|
||||
t.Fatalf("Generate: %v", err)
|
||||
}
|
||||
if !routeWarnsHave(warns, "unresolved target") {
|
||||
t.Fatalf("a rule pointing at a non-existent node lost its unresolved-target warning: %v", warns)
|
||||
}
|
||||
if routeWarnsHave(warns, "cannot carry a layer-3 packet") {
|
||||
t.Fatalf("the unresolved target was ALSO reported as unable to carry layer 3: %v — the operator's problem is the missing node, and the second sentence competes with the first", warns)
|
||||
}
|
||||
}
|
||||
|
||||
// TestProtoICMPWarningsStayBelowTheAlarmThreshold: shater/apply grades generate's
|
||||
// free text, and a handful of substrings ("never applies", "NOT emitted", …) light
|
||||
// the panel's alarm banner. A ping that fails is fail-CLOSED — nothing leaks — so
|
||||
// these belong under `warning`, and the wording must not drift into the markers.
|
||||
// Checked here rather than in shater/apply because that package imports this one.
|
||||
func TestProtoICMPWarningsStayBelowTheAlarmThreshold(t *testing.T) {
|
||||
var texts []string
|
||||
_, w := oneICMPRule(t, false, icmpRule("a", "icmp", "node:wg1"))
|
||||
texts = append(texts, w...)
|
||||
_, w = oneICMPRule(t, true, icmpRule("b", "icmp", "node:ss1"))
|
||||
texts = append(texts, w...)
|
||||
_, w = oneICMPRule(t, true, icmpRule("c", "icmp", "group:mixed"))
|
||||
texts = append(texts, w...)
|
||||
|
||||
// The exact substrings shater/apply/warnings.go greps for.
|
||||
markers := []string{"NOT emitted", "inert", "has NO effect", "never applies", "in the clear"}
|
||||
for _, text := range texts {
|
||||
if !strings.Contains(text, "layer 3") && !strings.Contains(text, "layer-3") && !strings.Contains(text, "l3_tunnel") {
|
||||
continue // somebody else's warning
|
||||
}
|
||||
for _, m := range markers {
|
||||
if strings.Contains(text, m) {
|
||||
t.Fatalf("ICMP warning %q contains the critical marker %q — it would light the panel's alarm banner for a fail-CLOSED ping failure, and every cosmetic critical teaches the operator to skip the real one", text, m)
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,152 @@
|
||||
package model
|
||||
|
||||
// fwmark_base / table_base: two free-form hex fields in the panel's "Advanced"
|
||||
// section that nothing had ever validated, and whose DERIVED values are not
|
||||
// visible from the number typed.
|
||||
//
|
||||
// The panel offers them with the note "only change these if another app on the
|
||||
// router already uses the same range", which reads like a courtesy. It is not:
|
||||
// every mark and every routing table the data plane owns is computed from these
|
||||
// two numbers by adding a fixed offset, and some of the results land on things
|
||||
// the kernel or the engine already owns. Nothing downstream refuses them,
|
||||
// nothing warns, and the failures they produce (the router losing its uplink,
|
||||
// the main routing table being flushed) point nowhere near a number in a
|
||||
// collapsed section.
|
||||
|
||||
import (
|
||||
"strings"
|
||||
"testing"
|
||||
)
|
||||
|
||||
func eg(n int) []Egress {
|
||||
out := make([]Egress, n)
|
||||
for i := range out {
|
||||
out[i] = Egress{Name: "e", Type: "interface", Interface: "eth1"}
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
func joined(ws []Warning) string {
|
||||
var b strings.Builder
|
||||
for _, w := range ws {
|
||||
b.WriteString(w.Error())
|
||||
b.WriteString("\n")
|
||||
}
|
||||
return b.String()
|
||||
}
|
||||
|
||||
// TestValidateMarkBasesFwmark.
|
||||
//
|
||||
// 0x7f is the case the L3 ingress added and the reason this check exists: the L3
|
||||
// mark is fwmark_base + 0x80, so 0x7f puts it exactly on 0xff — the loop-guard
|
||||
// mark the engine stamps on its OWN outgoing traffic. `ip rule fwmark 0xff
|
||||
// lookup <l3 table>` then captures everything the engine sends and routes it
|
||||
// into the engine's own TUN. The router loses the internet the moment l3_tunnel
|
||||
// is switched on, and nothing connects that to a hex field.
|
||||
//
|
||||
// 0xff is the same failure by the older route (the divert mark itself lands on
|
||||
// the loop guard) and predates the L3 offset; adding an offset simply added a
|
||||
// second base that does it, which is exactly why the check is written as "derive
|
||||
// every value and look for collisions" rather than as a blacklist of two numbers.
|
||||
func TestValidateMarkBasesFwmark(t *testing.T) {
|
||||
cases := []struct {
|
||||
name string
|
||||
base uint32
|
||||
egress int
|
||||
wantBad bool
|
||||
want string // substring the message must carry
|
||||
}{
|
||||
{"unset means the 0x2000 default", 0, 2, false, ""},
|
||||
{"the shipped default", 0x2000, 8, false, ""},
|
||||
{"a sane custom base", 0x30000, 4, false, ""},
|
||||
// The L3 offset lands the derived mark on the engine's own loop guard.
|
||||
{"0x7f: L3 mark becomes 0xff", 0x7f, 0, true, "fwmark_base + 0x80"},
|
||||
{"0x7f with egresses too", 0x7f, 3, true, "0xff"},
|
||||
// The divert mark itself is the loop guard.
|
||||
{"0xff: the divert mark IS the loop guard", 0xff, 0, true, "fwmark_base itself"},
|
||||
// Past the top of a 32-bit mark the arithmetic wraps onto another mark.
|
||||
{"wraps past 32 bits", 0xffffff80, 4, true, "wraps around"},
|
||||
}
|
||||
for _, c := range cases {
|
||||
t.Run(c.name, func(t *testing.T) {
|
||||
ws := ValidateMarkBases(Globals{FwmarkBase: c.base}, eg(c.egress))
|
||||
if got := len(ws) > 0; got != c.wantBad {
|
||||
t.Fatalf("fwmark_base 0x%x with %d egresses: warned = %v, want %v.\n%s",
|
||||
c.base, c.egress, got, c.wantBad, joined(ws))
|
||||
}
|
||||
if c.want != "" && !strings.Contains(joined(ws), c.want) {
|
||||
t.Errorf("the warning must name the derived value that collides (%q):\n%s", c.want, joined(ws))
|
||||
}
|
||||
for _, w := range ws {
|
||||
if w.Section != "globals" {
|
||||
t.Errorf("warnings must be scoped to globals, got %q", w.Section)
|
||||
}
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
// TestValidateMarkBasesTable: the same arithmetic decides routing table ids, and
|
||||
// there the reserved values are the kernel's own. Table 254 is `main` — and this
|
||||
// codebase does `ip route flush table <n>` on teardown, so a base whose derived
|
||||
// table lands on it does not merely collide, it takes the router off the network.
|
||||
//
|
||||
// The egress count matters, which is why this check is not in ValidateGlobals:
|
||||
// egress #i uses table_base + 0x10 + i, so whether a base is safe depends on how
|
||||
// many egresses are configured. table_base 239 is fine on a router with no
|
||||
// egresses and flushes the kernel's `local` table on one with a single egress.
|
||||
func TestValidateMarkBasesTable(t *testing.T) {
|
||||
cases := []struct {
|
||||
name string
|
||||
base uint32
|
||||
egress int
|
||||
wantBad bool
|
||||
want string
|
||||
}{
|
||||
{"unset", 0, 2, false, ""},
|
||||
{"the shipped default", 0x2000, 8, false, ""},
|
||||
{"the divert table IS main", 254, 0, true, "MAIN routing table"},
|
||||
{"the L3 table becomes main", 246, 0, true, "table_base + 0x8"},
|
||||
{"the L3 table becomes `default`", 245, 0, true, "253"},
|
||||
// Count-sensitive: harmless with no egresses, fatal with one.
|
||||
{"no egress, no derived egress table", 239, 0, false, ""},
|
||||
{"one egress lands on `local`", 239, 1, true, "local"},
|
||||
{"two egresses reach main", 238, 2, true, "MAIN routing table"},
|
||||
}
|
||||
for _, c := range cases {
|
||||
t.Run(c.name, func(t *testing.T) {
|
||||
ws := ValidateMarkBases(Globals{TableBase: c.base}, eg(c.egress))
|
||||
if got := len(ws) > 0; got != c.wantBad {
|
||||
t.Fatalf("table_base %d with %d egresses: warned = %v, want %v.\n%s",
|
||||
c.base, c.egress, got, c.wantBad, joined(ws))
|
||||
}
|
||||
if c.want != "" && !strings.Contains(joined(ws), c.want) {
|
||||
t.Errorf("the warning must say WHICH kernel table is being taken over (%q):\n%s",
|
||||
c.want, joined(ws))
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
// The check has to be reachable from the one entry point the daemon and the
|
||||
// panel actually call, or it is a function nobody runs.
|
||||
func TestValidateSurfacesMarkBaseCollisions(t *testing.T) {
|
||||
m := &Model{
|
||||
Globals: Globals{FwmarkBase: 0x7f, TableBase: 0x2000},
|
||||
Egresses: []Egress{{Name: "wan2", Type: "interface", Interface: "eth1"}},
|
||||
}
|
||||
ws := m.Validate(nil)
|
||||
if !strings.Contains(joined(ws), "0xff") {
|
||||
t.Fatalf("Model.Validate does not run ValidateMarkBases, so the panel and the log never see it:\n%s",
|
||||
joined(ws))
|
||||
}
|
||||
}
|
||||
|
||||
// A base whose derived values are all distinct and all clear of the reserved
|
||||
// ones must stay silent even with a lot of egresses — the check must not become
|
||||
// noise that gets ignored.
|
||||
func TestValidateMarkBasesQuietOnTheDefault(t *testing.T) {
|
||||
if ws := ValidateMarkBases(Globals{FwmarkBase: 0x2000, TableBase: 0x2000}, eg(64)); len(ws) != 0 {
|
||||
t.Fatalf("the shipped defaults with 64 egresses must warn about nothing:\n%s", joined(ws))
|
||||
}
|
||||
}
|
||||
@@ -260,6 +260,48 @@ type Globals struct {
|
||||
// current behaviour exactly.
|
||||
Untunnelable string
|
||||
|
||||
// L3Tunnel opts the router into the L3 ingress: the engine opens a TUN device
|
||||
// (netplane.L3Device) and the nft prerouting chain policy-routes LAN traffic
|
||||
// the tunnel can only carry at layer 3 into it. There the engine's ordinary
|
||||
// route rules decide the outbound, and an L3-capable one (wireguard/AWG —
|
||||
// anything implementing adapter.FlowOutbound) carries the packet for real,
|
||||
// NAT'd onto the tunnel's own address.
|
||||
//
|
||||
// TODAY THAT MEANS ICMP ECHO AND NOTHING ELSE, and the reason is worth stating
|
||||
// precisely because the obvious one is wrong. It is NOT the gVisor stack: a
|
||||
// WireGuard/AWG endpoint forwards at layer 3 straight past it — WritePackets
|
||||
// reads only the IP version and the destination address before handing the raw
|
||||
// bytes to the device (transport/wireguard/port.go:21-58), and the return path
|
||||
// offers every decrypted packet back before the stack ever sees it (:127-157).
|
||||
// The limit is sing-tun's dispatcher: flow_parse.go sets hasFlow for TCP, UDP
|
||||
// and ICMP echo alone, so nothing else is dispatched, and flow_dispatch.go's
|
||||
// createFlow NATs through a port-shaped selector that ESP, AH and GRE do not
|
||||
// have. Those stay governed by Untunnelable and UntunnelableEgress below. Ping
|
||||
// and Windows `tracert` through the tunnel are the whole of what this buys.
|
||||
//
|
||||
// Default false: without it nothing changes, and Untunnelable alone decides
|
||||
// what happens to non-TCP/UDP traffic.
|
||||
L3Tunnel bool
|
||||
|
||||
// UntunnelableEgress names an egress of type interface/tunnel that carries the
|
||||
// protocols the engine will not dispatch: ESP, AH, GRE, IGMP, SCTP. The
|
||||
// blocker is sing-tun's flow dispatcher (see L3Tunnel above), not the tunnel
|
||||
// itself, but the effect is the same — L3Tunnel cannot help them.
|
||||
//
|
||||
// It is not a tunnel of ours. The kernel simply routes those packets out that
|
||||
// egress's device, with the kernel's own NAT, and every protocol works because
|
||||
// nothing in the path has to understand any of them. What that BUYS depends
|
||||
// entirely on what the device is: a WireGuard interface really is a tunnel, a
|
||||
// second WAN is just a different uplink and the destination sees that uplink's
|
||||
// real address. The panel says which, at apply time, from whether the device is
|
||||
// point-to-point.
|
||||
//
|
||||
// Empty (the default) leaves Untunnelable above in sole charge, exactly as
|
||||
// before. A name that does not resolve to an interface/tunnel egress is
|
||||
// reported and ignored — the policy applies unchanged, which is the
|
||||
// fail-closed reading.
|
||||
UntunnelableEgress string
|
||||
|
||||
// StatsBackend selects the statistics storage backend (pluggable stats store).
|
||||
// Values: "off" | "memory" | "sqlite" (default "memory"). Contract: stats.NewStore.
|
||||
// off => no-op store: nothing is subscribed/aggregated (zero per-query overhead),
|
||||
|
||||
@@ -89,6 +89,8 @@ func RenderUCIExport(m *Model) string {
|
||||
w.boolOpt("block_doh", g.BlockDoH)
|
||||
w.boolOpt("group_health", g.GroupHealth)
|
||||
w.strOpt("untunnelable", g.Untunnelable)
|
||||
w.boolOpt("l3_tunnel", g.L3Tunnel)
|
||||
w.strOpt("untunnelable_egress", g.UntunnelableEgress)
|
||||
w.strOpt("stats_backend", g.StatsBackend)
|
||||
// The three stats-size knobs use intOptAlways (NOT the omit-zero intOpt) because 0
|
||||
// means UNLIMITED, a meaningful value that MUST survive the round-trip. If they were
|
||||
|
||||
+70
-19
@@ -17,25 +17,27 @@ import (
|
||||
func richModel() *Model {
|
||||
return &Model{
|
||||
Globals: Globals{
|
||||
Enabled: false, // exercise false via always-emit bool
|
||||
LogLevel: "debug",
|
||||
KillSwitch: "open",
|
||||
Untunnelable: "direct",
|
||||
IPv6: false,
|
||||
FwmarkBase: 0x2000,
|
||||
TableBase: 0x3000,
|
||||
ConfirmTimeout: 30,
|
||||
ResolverDefault: "cf",
|
||||
ResolverFallback: "fake",
|
||||
ProbeURL: "http://gstatic.com/generate_204",
|
||||
ProbeInterval: "60s",
|
||||
SchemaVersion: 1,
|
||||
ActiveProfile: "home",
|
||||
PanelPort: 8090,
|
||||
DNSFilter: true,
|
||||
DNSIntercept: true,
|
||||
BlockDoH: true,
|
||||
StatsBackend: "off",
|
||||
Enabled: false, // exercise false via always-emit bool
|
||||
LogLevel: "debug",
|
||||
KillSwitch: "open",
|
||||
Untunnelable: "direct",
|
||||
L3Tunnel: true, // opt-in; exercise the non-default via round-trip
|
||||
UntunnelableEgress: "frag", // exercise the non-default via round-trip
|
||||
IPv6: false,
|
||||
FwmarkBase: 0x2000,
|
||||
TableBase: 0x3000,
|
||||
ConfirmTimeout: 30,
|
||||
ResolverDefault: "cf",
|
||||
ResolverFallback: "fake",
|
||||
ProbeURL: "http://gstatic.com/generate_204",
|
||||
ProbeInterval: "60s",
|
||||
SchemaVersion: 1,
|
||||
ActiveProfile: "home",
|
||||
PanelPort: 8090,
|
||||
DNSFilter: true,
|
||||
DNSIntercept: true,
|
||||
BlockDoH: true,
|
||||
StatsBackend: "off",
|
||||
},
|
||||
Inbounds: []Inbound{
|
||||
{
|
||||
@@ -312,6 +314,55 @@ func TestGroupHealthRoundTrip(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
// TestL3TunnelRoundTrip pins the l3_tunnel opt-in: an EXPLICIT true (L3 ingress
|
||||
// on) survives WriteUCI->ReadUCI, and an ABSENT option stays false (opt-in,
|
||||
// never accidentally on).
|
||||
func TestL3TunnelRoundTrip(t *testing.T) {
|
||||
for _, v := range []bool{true, false} {
|
||||
g := DefaultGlobals()
|
||||
g.L3Tunnel = v
|
||||
got, err := ParseUCIExport(RenderUCIExport(&Model{Globals: g}))
|
||||
if err != nil {
|
||||
t.Fatalf("v=%v parse: %v", v, err)
|
||||
}
|
||||
if got.Globals.L3Tunnel != v {
|
||||
t.Fatalf("v=%v round-trip: l3_tunnel=%v, want %v", v, got.Globals.L3Tunnel, v)
|
||||
}
|
||||
}
|
||||
// Absent option => off (opt-in default, the Go zero value).
|
||||
got, err := ParseUCIExport("package shater\n")
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if got.Globals.L3Tunnel {
|
||||
t.Fatalf("absent l3_tunnel = %v, want false (default, opt-in)", got.Globals.L3Tunnel)
|
||||
}
|
||||
}
|
||||
|
||||
// TestUntunnelableEgressRoundTrip pins the untunnelable_egress option: an
|
||||
// EXPLICIT name survives WriteUCI->ReadUCI (else the kernel carrier for
|
||||
// ESP/AH/GRE/IGMP/SCTP silently switches off on the next re-render), and an
|
||||
// ABSENT option stays "" so the Untunnelable policy remains in sole charge.
|
||||
func TestUntunnelableEgressRoundTrip(t *testing.T) {
|
||||
g := DefaultGlobals()
|
||||
g.UntunnelableEgress = "wan2"
|
||||
got, err := ParseUCIExport(RenderUCIExport(&Model{Globals: g}))
|
||||
if err != nil {
|
||||
t.Fatalf("parse: %v", err)
|
||||
}
|
||||
if got.Globals.UntunnelableEgress != "wan2" {
|
||||
t.Fatalf("round-trip: untunnelable_egress=%q, want %q", got.Globals.UntunnelableEgress, "wan2")
|
||||
}
|
||||
// Absent option => "" (default: no kernel carrier, policy alone decides).
|
||||
got, err = ParseUCIExport("package shater\n")
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if got.Globals.UntunnelableEgress != "" {
|
||||
t.Fatalf("absent untunnelable_egress = %q, want \"\" (default, opt-in)", got.Globals.UntunnelableEgress)
|
||||
}
|
||||
}
|
||||
|
||||
// TestNodeEgressRoundTrip pins the node-level egress binding (multi-WAN): a
|
||||
// node's `egress` option survives WriteUCI->ReadUCI so the generator can bind the
|
||||
// node's own upstream to that egress outbound. An absent option parses to "".
|
||||
|
||||
@@ -346,6 +346,13 @@ func applyGlobals(g *Globals, s uciSection) {
|
||||
// stays ON (opt-out); an EXPLICIT "0" disables our background probing.
|
||||
g.GroupHealth = s.optBool("group_health", g.GroupHealth)
|
||||
g.Untunnelable = s.optOr("untunnelable", g.Untunnelable)
|
||||
// Opt-in, and the default is the Go zero value (DefaultGlobals does not seed
|
||||
// it): an ABSENT option keeps the L3 ingress OFF; only an EXPLICIT "1" opens it.
|
||||
g.L3Tunnel = s.optBool("l3_tunnel", g.L3Tunnel)
|
||||
// Which egress carries the protocols the engine cannot (ESP/AH/GRE/IGMP/SCTP).
|
||||
// Default "" (DefaultGlobals does not seed it): the Untunnelable policy above
|
||||
// stays in sole charge, exactly as before the option existed.
|
||||
g.UntunnelableEgress = s.optOr("untunnelable_egress", g.UntunnelableEgress)
|
||||
g.StatsBackend = s.optOr("stats_backend", g.StatsBackend)
|
||||
// Stats-size knobs: 0 = UNLIMITED, N = limit. The default (g.*) is the DefaultGlobals
|
||||
// seed (200/60/5000), so an ABSENT option stays bounded; an EXPLICIT "0" parses to 0
|
||||
|
||||
@@ -320,6 +320,28 @@ func ValidateGlobals(g Globals) []Warning {
|
||||
"(non-TCP/UDP traffic is dropped); valid values are block/icmp/direct.", g.Untunnelable))
|
||||
}
|
||||
|
||||
// l3_tunnel + untunnelable=direct is a conflict of intent, not a typo. The L3
|
||||
// mark is stamped in prerouting and the routing decision carries echo into the
|
||||
// engine's TUN before the forward chain — where the `direct` accept lives — is
|
||||
// ever consulted, so `direct` no longer buys the operator anything for ping.
|
||||
// What it still does is exactly what l3_tunnel was turned on to stop: it lets
|
||||
// the never-markable protocols (raw IPsec ESP/AH, PPTP/GRE) out with the real
|
||||
// address, and it turns any failure to install the L3 route (interface down,
|
||||
// partial apply) into a SILENT leak — the marked echo falls through to the
|
||||
// main table, reaches forward, and `direct` waves it out — where `block`
|
||||
// makes the same failure an honest packet loss. Both values are individually
|
||||
// valid, which is exactly why nothing else says a word about the combination.
|
||||
// Warn only — like every check here, the config is never edited behind the
|
||||
// operator's back. The normalisation mirrors netplane.EffectiveUntunnelable
|
||||
// (unimportable from here: netplane imports model).
|
||||
if g.L3Tunnel && strings.ToLower(strings.TrimSpace(g.Untunnelable)) == "direct" {
|
||||
add("l3_tunnel is on, so ping already travels through the tunnel whatever this policy " +
|
||||
"says — \"direct\" no longer buys you anything for ping, but it still sends raw VPN " +
|
||||
"passthrough (IPsec ESP/AH, PPTP/GRE) out with your real IP address, and if the L3 " +
|
||||
"route ever fails to come up your pings silently fall back to leaking directly " +
|
||||
"instead of failing. With \"block\" the same failure is honest packet loss.")
|
||||
}
|
||||
|
||||
switch strings.ToLower(strings.TrimSpace(g.StatsBackend)) {
|
||||
case "", "off", "memory", "sqlite":
|
||||
default:
|
||||
@@ -404,6 +426,206 @@ func ValidateGroups(groups []Group) []Warning {
|
||||
return out
|
||||
}
|
||||
|
||||
// ValidateUntunnelableEgress reports an untunnelable_egress option that cannot
|
||||
// do what it says. It is a cross-section check: the option lives in globals but
|
||||
// only means something when it resolves to an interface/tunnel egress with a
|
||||
// device, so neither ValidateGlobals (which sees no egresses) nor a per-egress
|
||||
// check (which sees no globals) can catch it.
|
||||
//
|
||||
// The stakes are higher than a cosmetic typo. The data plane resolves the name
|
||||
// with netplane.UntunnelableEgressBinding, which fails CLOSED: an unresolvable
|
||||
// name renders nothing and the plain Untunnelable policy stays in sole charge.
|
||||
// The operator who set the option believes ESP/AH/GRE/IGMP/SCTP now leave
|
||||
// through the named egress; in reality they are still governed by the policy —
|
||||
// dropped, or let out directly — and nothing else says so. The resolution here
|
||||
// mirrors that binding (unimportable from here: netplane imports model), except
|
||||
// the device lookup: only a statically empty Interface is checkable in this
|
||||
// leaf package, a name that fails to resolve at apply time is netplane's to
|
||||
// report.
|
||||
func ValidateUntunnelableEgress(g Globals, egresses []Egress) []Warning {
|
||||
name := strings.TrimSpace(g.UntunnelableEgress)
|
||||
if name == "" {
|
||||
return nil
|
||||
}
|
||||
add := func(msg string) []Warning {
|
||||
return []Warning{{Section: "globals", Message: msg}}
|
||||
}
|
||||
// DUPLICATE OF netplane.UntunnelableEgressBinding by necessity, and it MUST
|
||||
// change in lockstep with it: if the two resolutions diverge, this warning
|
||||
// lies exactly in the configs where it matters most.
|
||||
for _, eg := range egresses {
|
||||
if !strings.EqualFold(strings.TrimSpace(eg.Name), name) {
|
||||
continue
|
||||
}
|
||||
if t := strings.ToLower(strings.TrimSpace(eg.Type)); t != "interface" && t != "tunnel" {
|
||||
return add(fmt.Sprintf("untunnelable_egress %q is an egress of type %q, not "+
|
||||
"interface/tunnel: it has no device and no routing table of its own, so the "+
|
||||
"kernel has nowhere to route ESP/AH/GRE/IGMP/SCTP. The option is ignored and "+
|
||||
"those protocols stay on the untunnelable policy.", g.UntunnelableEgress, eg.Type))
|
||||
}
|
||||
if strings.TrimSpace(eg.Interface) == "" {
|
||||
return add(fmt.Sprintf("untunnelable_egress %q names an egress with no interface, "+
|
||||
"so there is no device to route out of. The option is ignored and "+
|
||||
"ESP/AH/GRE/IGMP/SCTP stay on the untunnelable policy.", g.UntunnelableEgress))
|
||||
}
|
||||
return nil
|
||||
}
|
||||
return add(fmt.Sprintf("untunnelable_egress %q does not match any configured egress, so "+
|
||||
"the option is ignored: ESP/AH/GRE/IGMP/SCTP stay on the untunnelable policy instead "+
|
||||
"of leaving through that egress.", g.UntunnelableEgress))
|
||||
}
|
||||
|
||||
// The mark/table layout the netplane derives from fwmark_base and table_base.
|
||||
//
|
||||
// DUPLICATED FROM shater/netplane (unimportable from here: netplane imports
|
||||
// model) and MUST change in lockstep with it. Unlike a duplicated string
|
||||
// comparison, a drift here is silent in both directions — the validator would
|
||||
// bless a base the data plane cannot use, or condemn one it can. netplane's
|
||||
// TestMarkLayoutConstantsLockstep asserts every constant below against the
|
||||
// originals, so the drift is a build-time failure and not a field report.
|
||||
const (
|
||||
// MarkLoopGuard is the fixed mark on the engine's OWN traffic. It is not
|
||||
// derived from fwmark_base and cannot move, so every derived mark has to
|
||||
// stay off it.
|
||||
MarkLoopGuard = 0xff
|
||||
// MarkL3Offset / MarkEgressOffset: L3 mark = base+0x80, egress #i = base+0x100+i.
|
||||
MarkL3Offset = 0x80
|
||||
MarkEgressOffset = 0x100
|
||||
// TableL3Offset / TableEgressOffset: L3 table = base+0x08, egress #i = base+0x10+i.
|
||||
TableL3Offset = 0x08
|
||||
TableEgressOffset = 0x10
|
||||
// maxFwmark is the width of an nft/SO_MARK mark and of a routing table id.
|
||||
maxFwmark = 0xffffffff
|
||||
)
|
||||
|
||||
// reservedTables are the routing table ids the kernel owns. Writing our default
|
||||
// route into one of them is not a collision, it is an amputation: table 254 is
|
||||
// `main`, and this codebase does `ip route flush table <n>` on teardown.
|
||||
var reservedTables = map[uint64]string{
|
||||
0: "unspec (not a usable table)",
|
||||
253: "the kernel's `default` table",
|
||||
254: "the kernel's MAIN routing table — flushing it takes the whole router off the network",
|
||||
255: "the kernel's `local` table — flushing it breaks the router's own addresses",
|
||||
}
|
||||
|
||||
// derivedValue is one number the mark/table layout computes from a base, paired
|
||||
// with an operator-facing description of where it came from.
|
||||
type derivedValue struct {
|
||||
what string
|
||||
v uint64
|
||||
}
|
||||
|
||||
// ValidateMarkBases reports a fwmark_base or table_base whose DERIVED values
|
||||
// collide with something that already exists.
|
||||
//
|
||||
// The panel offers both under "Advanced", as free hex fields with no bounds at
|
||||
// all, and neither the model, the netplane nor the applier has ever checked
|
||||
// them. What makes that more than a footgun is that the derived values are not
|
||||
// obvious from the number typed: fwmark_base 0x7f looks harmless and is not —
|
||||
// the L3 mark is base+0x80, so it lands exactly on 0xff, the loop-guard mark the
|
||||
// engine stamps on its OWN traffic, and `ip rule fwmark 0xff lookup <l3 table>`
|
||||
// then routes every packet the engine sends into the engine's own TUN. The
|
||||
// router loses the internet the moment l3_tunnel is switched on, for a reason
|
||||
// nothing on screen connects to a number in a collapsed "Advanced" section.
|
||||
// fwmark_base 0xff had produced the same class of failure since long before the
|
||||
// L3 offset existed; adding an offset simply added a second value that does it.
|
||||
//
|
||||
// It takes the egresses because the derived range depends on how many there are
|
||||
// (egress #i uses base+0x100+i), which Globals alone cannot say.
|
||||
//
|
||||
// The collision test is deliberately written as "build every value this layout
|
||||
// derives, then look for duplicates and reserved ids" rather than as a list of
|
||||
// known-bad bases. A future offset added to the layout is then covered by
|
||||
// construction — it only has to be listed in the two loops below, and
|
||||
// netplane's constant-lockstep test is what makes sure it is.
|
||||
func ValidateMarkBases(g Globals, egresses []Egress) []Warning {
|
||||
var out []Warning
|
||||
add := func(msg string) { out = append(out, Warning{Section: "globals", Message: msg}) }
|
||||
|
||||
// 0 means "unset" for both fields (netplane's effFwmark/effTable substitute
|
||||
// the 0x2000 contract default), so it is not a value to judge.
|
||||
if b := uint64(g.FwmarkBase); b != 0 {
|
||||
marks := []derivedValue{
|
||||
{"the divert mark (fwmark_base itself)", b},
|
||||
{"the L3-tunnel mark (fwmark_base + 0x80)", b + MarkL3Offset},
|
||||
}
|
||||
for i := range egresses {
|
||||
marks = append(marks, derivedValue{
|
||||
fmt.Sprintf("egress %q's mark (fwmark_base + 0x100 + %d)", egresses[i].Name, i),
|
||||
b + MarkEgressOffset + uint64(i),
|
||||
})
|
||||
}
|
||||
for _, d := range marks {
|
||||
if d.v > maxFwmark {
|
||||
add(fmt.Sprintf("fwmark_base 0x%x is too large: %s would be 0x%x, past the 32-bit "+
|
||||
"limit of a firewall mark, so it wraps around onto another mark. Use a base below "+
|
||||
"0x%x.", g.FwmarkBase, d.what, d.v, uint64(maxFwmark)-MarkEgressOffset-uint64(len(egresses))))
|
||||
break
|
||||
}
|
||||
if d.v == MarkLoopGuard {
|
||||
add(fmt.Sprintf("fwmark_base 0x%x is unusable: %s comes out as 0x%x, which is the mark "+
|
||||
"the engine stamps on its OWN outgoing traffic. The policy rule for that mark would "+
|
||||
"then capture everything the engine sends and route it back into shater's own tables — "+
|
||||
"the router loses its internet connection and the panel cannot explain why. Pick "+
|
||||
"another base (the default 0x2000 is clear of everything).",
|
||||
g.FwmarkBase, d.what, d.v))
|
||||
}
|
||||
}
|
||||
if a, b2, dup := firstDuplicate(marks); dup {
|
||||
add(fmt.Sprintf("fwmark_base 0x%x makes two different things share one mark: %s and %s are "+
|
||||
"both 0x%x, so the policy rule for one of them routes the other's traffic as well.",
|
||||
g.FwmarkBase, a.what, b2.what, a.v))
|
||||
}
|
||||
}
|
||||
|
||||
if b := uint64(g.TableBase); b != 0 {
|
||||
tables := []derivedValue{
|
||||
{"the divert table (table_base itself)", b},
|
||||
{"the L3-tunnel table (table_base + 0x8)", b + TableL3Offset},
|
||||
}
|
||||
for i := range egresses {
|
||||
tables = append(tables, derivedValue{
|
||||
fmt.Sprintf("egress %q's table (table_base + 0x10 + %d)", egresses[i].Name, i),
|
||||
b + TableEgressOffset + uint64(i),
|
||||
})
|
||||
}
|
||||
for _, d := range tables {
|
||||
if d.v > maxFwmark {
|
||||
add(fmt.Sprintf("table_base 0x%x is too large: %s would be %d, past the 32-bit limit of a "+
|
||||
"routing table id.", g.TableBase, d.what, d.v))
|
||||
break
|
||||
}
|
||||
if why, bad := reservedTables[d.v]; bad {
|
||||
add(fmt.Sprintf("table_base 0x%x is unusable: %s comes out as table %d, which is %s. "+
|
||||
"shater owns the tables it derives — it writes a default route into them and flushes "+
|
||||
"them on teardown — so this one must be a table nothing else uses. Pick another base "+
|
||||
"(the default 0x2000 is clear of everything).", g.TableBase, d.what, d.v, why))
|
||||
}
|
||||
}
|
||||
if a, b2, dup := firstDuplicate(tables); dup {
|
||||
add(fmt.Sprintf("table_base 0x%x makes two different things share one routing table: %s and %s "+
|
||||
"are both table %d, so whichever is applied last overwrites the other's default route.",
|
||||
g.TableBase, a.what, b2.what, a.v))
|
||||
}
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
// firstDuplicate reports the first pair of derived values that collide. The
|
||||
// current layout cannot produce one without overflowing first; it is checked
|
||||
// anyway because "the offsets do not overlap" is a property of the arithmetic
|
||||
// that a future offset can quietly break, and this is where that would surface.
|
||||
func firstDuplicate(vals []derivedValue) (derivedValue, derivedValue, bool) {
|
||||
seen := map[uint64]derivedValue{}
|
||||
for _, d := range vals {
|
||||
if prev, ok := seen[d.v]; ok {
|
||||
return prev, d, true
|
||||
}
|
||||
seen[d.v] = d
|
||||
}
|
||||
return derivedValue{}, derivedValue{}, false
|
||||
}
|
||||
|
||||
// Validate runs every model-level check and returns the combined warnings. It
|
||||
// never fails: the result is advisory, for the log and the panel.
|
||||
func (m *Model) Validate(classify NetClassifier) []Warning {
|
||||
@@ -412,6 +634,8 @@ func (m *Model) Validate(classify NetClassifier) []Warning {
|
||||
}
|
||||
out := ValidateRules(m.Rules, classify)
|
||||
out = append(out, ValidateGlobals(m.Globals)...)
|
||||
out = append(out, ValidateUntunnelableEgress(m.Globals, m.Egresses)...)
|
||||
out = append(out, ValidateMarkBases(m.Globals, m.Egresses)...)
|
||||
out = append(out, ValidateSubscriptions(m.Subscriptions)...)
|
||||
out = append(out, ValidateGroups(m.Groups)...)
|
||||
out = append(out, ValidateAlerts(m.Alerts)...)
|
||||
|
||||
@@ -222,6 +222,93 @@ func TestValidateGlobalsUnknownEnums(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
// TestValidateGlobalsL3TunnelDirectConflict: two individually-valid options whose
|
||||
// combination quietly downgrades the L3 ingress — `direct` keeps letting raw
|
||||
// IPsec/GRE out with the real address and turns an L3-route failure into a silent
|
||||
// ping leak instead of an honest loss. The validator must say so, and must say
|
||||
// nothing for either option alone or for l3_tunnel with the safe policies.
|
||||
func TestValidateGlobalsL3TunnelDirectConflict(t *testing.T) {
|
||||
ws := ValidateGlobals(Globals{L3Tunnel: true, Untunnelable: "direct"})
|
||||
if !hasWarning(ws, "globals", "l3_tunnel") {
|
||||
t.Fatalf("l3_tunnel + untunnelable=direct must warn, got %v", ws)
|
||||
}
|
||||
// The message has to name the safe alternative, or it sends the user hunting.
|
||||
if !hasWarning(ws, "globals", "\"block\"") {
|
||||
t.Errorf("the warning must point at \"block\" as the safe policy, got %v", ws)
|
||||
}
|
||||
// Normalisation mirrors netplane.EffectiveUntunnelable: case and whitespace
|
||||
// must not hide the conflict.
|
||||
if ws := ValidateGlobals(Globals{L3Tunnel: true, Untunnelable: " Direct "}); !hasWarning(ws, "globals", "l3_tunnel") {
|
||||
t.Errorf("normalised \" Direct \" must still warn, got %v", ws)
|
||||
}
|
||||
// Only the combination warns: each option alone, and l3_tunnel with the
|
||||
// policies that keep the failure mode honest, stay quiet.
|
||||
for _, g := range []Globals{
|
||||
{Untunnelable: "direct"},
|
||||
{L3Tunnel: true},
|
||||
{L3Tunnel: true, Untunnelable: "block"},
|
||||
{L3Tunnel: true, Untunnelable: "icmp"},
|
||||
} {
|
||||
if ws := ValidateGlobals(g); len(ws) != 0 {
|
||||
t.Errorf("globals %+v must not warn, got %v", g, ws)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestValidateUntunnelableEgress: the option fails CLOSED in the data plane —
|
||||
// netplane.UntunnelableEgressBinding renders nothing for a name that does not
|
||||
// resolve to an interface/tunnel egress with a device, and the plain
|
||||
// untunnelable policy quietly stays in charge. The operator believes
|
||||
// ESP/AH/GRE/IGMP/SCTP now leave through the named egress; without this warning
|
||||
// nothing tells them the option did nothing. Each of the three ways to
|
||||
// mis-point the option must be named, and a correctly-pointed option must stay
|
||||
// silent — a false warning on a working config teaches operators to ignore the
|
||||
// real ones.
|
||||
func TestValidateUntunnelableEgress(t *testing.T) {
|
||||
egresses := []Egress{
|
||||
{Name: "wan2", Type: "interface", Interface: "wan2"},
|
||||
{Name: "dpi", Type: "byedpi", Port: 1080},
|
||||
{Name: "nodev", Type: "tunnel"},
|
||||
}
|
||||
|
||||
// The name does not match any egress: the option is dropped on the floor.
|
||||
ws := ValidateUntunnelableEgress(Globals{UntunnelableEgress: "wan3"}, egresses)
|
||||
if !hasWarning(ws, "globals", "does not match any configured egress") {
|
||||
t.Fatalf("an unresolvable name must warn (the option silently does nothing), got %v", ws)
|
||||
}
|
||||
|
||||
// The name matches, but the egress type has no device or routing table.
|
||||
ws = ValidateUntunnelableEgress(Globals{UntunnelableEgress: "dpi"}, egresses)
|
||||
if !hasWarning(ws, "globals", "not interface/tunnel") {
|
||||
t.Fatalf("a byedpi/direct egress must warn (nowhere to route to), got %v", ws)
|
||||
}
|
||||
// The message has to name the offending type, or the operator cannot see
|
||||
// which of their egresses they mis-picked.
|
||||
if !hasWarning(ws, "globals", "\"byedpi\"") {
|
||||
t.Errorf("the warning must name the egress's actual type, got %v", ws)
|
||||
}
|
||||
|
||||
// Right type, but no interface: still no device to route out of.
|
||||
ws = ValidateUntunnelableEgress(Globals{UntunnelableEgress: "nodev"}, egresses)
|
||||
if !hasWarning(ws, "globals", "no interface") {
|
||||
t.Fatalf("an interface-less egress must warn (no device to route out of), got %v", ws)
|
||||
}
|
||||
|
||||
// Resolution mirrors netplane.UntunnelableEgressBinding: case and whitespace
|
||||
// must not hide a warning the apply stage would act on.
|
||||
if ws := ValidateUntunnelableEgress(Globals{UntunnelableEgress: " DPI "}, egresses); !hasWarning(ws, "globals", "not interface/tunnel") {
|
||||
t.Errorf("normalised \" DPI \" must still warn, got %v", ws)
|
||||
}
|
||||
|
||||
// The quiet cases: an unset option, and one pointing at a well-formed
|
||||
// interface egress (any casing), produce no warnings at all.
|
||||
for _, name := range []string{"", "wan2", " WAN2 "} {
|
||||
if ws := ValidateUntunnelableEgress(Globals{UntunnelableEgress: name}, egresses); len(ws) != 0 {
|
||||
t.Errorf("untunnelable_egress=%q must not warn, got %v", name, ws)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestDeadKnobsAreGoneNotDocumented is the deletion regression. dns_mode and the
|
||||
// two xudp_* node options were confirmed to have no reader anywhere, so they were
|
||||
// REMOVED rather than left with a "not used" comment: a value that survives a
|
||||
|
||||
+216
-15
@@ -109,23 +109,33 @@ func ApplyRouting(m *model.Model) error {
|
||||
|
||||
// ApplyRoutingWithWarnings reconciles ip rule/route: fwmark(FwmarkBase) ->
|
||||
// table(TableBase) with `local default dev lo`, so tproxy-marked packets are delivered
|
||||
// locally, plus the per-egress mark->table->device bindings. Idempotent (del-then-add).
|
||||
// locally, plus the per-egress mark->table->device bindings and the L3-ingress
|
||||
// binding (L3Mark -> L3Table -> L3Device). Idempotent (del-then-add).
|
||||
//
|
||||
// It mirrors RenderNftWithWarnings: the warnings are operator-facing statements about an
|
||||
// egress that WAS built but cannot carry traffic, which apply folds into the status
|
||||
// It mirrors RenderNftWithWarnings: the warnings are operator-facing statements about a
|
||||
// binding that WAS built but cannot carry traffic, which apply folds into the status
|
||||
// warning list so the panel shows them.
|
||||
func ApplyRoutingWithWarnings(m *model.Model) ([]string, error) {
|
||||
if err := addRouting(effFwmark(m.Globals), effTable(m.Globals), m.Globals.IPv6); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
return addEgressRouting(m)
|
||||
warnings, err := addEgressRouting(m)
|
||||
if err != nil {
|
||||
return warnings, err
|
||||
}
|
||||
l3Warnings, err := addL3Routing(m)
|
||||
return append(warnings, l3Warnings...), err
|
||||
}
|
||||
|
||||
// TeardownRouting removes the main tproxy rule/table and every reserved per-egress
|
||||
// rule/table. Safe when nothing is set up.
|
||||
// TeardownRouting removes the main tproxy rule/table, every reserved per-egress
|
||||
// rule/table, and the L3-ingress rule/table. Safe when nothing is set up. The L3
|
||||
// pair is removed UNCONDITIONALLY (not gated on L3Enabled) for the same reason
|
||||
// the egress block is swept wholesale: teardown must clean up what a PREVIOUS
|
||||
// config installed, and the current model cannot testify about the past.
|
||||
func TeardownRouting(m *model.Model) error {
|
||||
removeRouting(effFwmark(m.Globals), effTable(m.Globals))
|
||||
removeEgressRouting(m)
|
||||
removeRouting(L3Mark(m.Globals), L3Table(m.Globals))
|
||||
return nil
|
||||
}
|
||||
|
||||
@@ -240,6 +250,51 @@ func runIP(args ...string) error {
|
||||
return nil
|
||||
}
|
||||
|
||||
// unreachableFloorMetric is the metric of the `unreachable default` route every
|
||||
// mark-driven table gets as its last resort. Routes to the same prefix are
|
||||
// ordered by metric, so any real default route (the kernel gives them metric 0
|
||||
// unless asked otherwise) always wins while it exists; this one is only ever
|
||||
// consulted once the real one is gone. It is deliberately absurd rather than
|
||||
// merely large, so nothing a person might plausibly configure can outrank it.
|
||||
const unreachableFloorMetric = "4294967295"
|
||||
|
||||
// addUnreachableFloor installs the fail-closed floor of a mark-driven routing
|
||||
// table: a metric-maximal `unreachable default`.
|
||||
//
|
||||
// # Why a fallthrough is not a safety net
|
||||
//
|
||||
// `ip rule fwmark X lookup N` does not send the packet to table N. It sends the
|
||||
// LOOKUP to table N, and a lookup that finds nothing there simply continues down
|
||||
// the rule list — to `main`. So an empty table N is not "this traffic is stuck",
|
||||
// it is "this traffic is routed exactly as if it had never been marked": out the
|
||||
// default WAN, with the router's real address, wearing a mark the firewall was
|
||||
// told to trust. Every leak in this file's history is a variation on that one
|
||||
// sentence.
|
||||
//
|
||||
// The table empties for entirely ordinary reasons and, crucially, WITHOUT
|
||||
// anything we control changing: an egress rides an interface, `ifdown wg0` or a
|
||||
// `network restart` happens, and the kernel garbage-collects every route through
|
||||
// that device. The L3 table empties the same way when the engine is restarted
|
||||
// and its TUN is destroyed and recreated. The nft ruleset is byte-identical
|
||||
// across all of that, so the applier's idempotence check sees nothing to do.
|
||||
//
|
||||
// An `unreachable` route has no device, so the kernel has no reason to remove it
|
||||
// when a device dies — which is exactly the property needed. Once it is in the
|
||||
// table the lookup can no longer FAIL, only succeed with "unreachable", and the
|
||||
// fallthrough to main becomes structurally impossible instead of merely
|
||||
// unlikely. That is worth more than the warning it replaces: warnings depend on
|
||||
// somebody reading them, at the moment the interface goes down, which is not the
|
||||
// moment anybody is reading anything.
|
||||
//
|
||||
// This is deliberately unconditional — not gated on the kill-switch. The
|
||||
// kill-switch decides whether traffic may escape the TUNNEL; an egress binding
|
||||
// is a statement about WHICH UPLINK, and silently substituting a different one
|
||||
// is never what "fail open" was meant to permit.
|
||||
func addUnreachableFloor(fam, tableArg string) error {
|
||||
return runIP(fam, "route", "add", "unreachable", "default",
|
||||
"metric", unreachableFloorMetric, "table", tableArg)
|
||||
}
|
||||
|
||||
// addEgressRouting realises the policy-routing half of an interface/tunnel egress
|
||||
// (the generator emits the SO_BINDTODEVICE+SO_MARK outbound; this binds the mark
|
||||
// to a table whose default route leaves via the egress device). Idempotent.
|
||||
@@ -270,11 +325,15 @@ func addEgressRouting(m *model.Model) ([]string, error) {
|
||||
var warnings []string
|
||||
var bindings []egressBinding
|
||||
for i, eg := range m.Egresses {
|
||||
t := strings.ToLower(eg.Type)
|
||||
if (t != "interface" && t != "tunnel") || eg.Interface == "" {
|
||||
// ONE resolution, shared with the prerouting marking and the forward
|
||||
// accept (see EgressDevice): an egress this skips is an egress whose mark
|
||||
// is neither stamped nor accepted anywhere, and vice versa. When these
|
||||
// drifted apart, the difference was a marked packet with no table.
|
||||
dev := EgressDevice(eg)
|
||||
if dev == "" {
|
||||
continue
|
||||
}
|
||||
dev := IfaceDevice(eg.Interface)
|
||||
iface := strings.TrimSpace(eg.Interface)
|
||||
mark := EgressMark(m.Globals, i)
|
||||
table := EgressTable(m.Globals, i)
|
||||
fams := []string{"-4"}
|
||||
@@ -299,10 +358,18 @@ func addEgressRouting(m *model.Model) ([]string, error) {
|
||||
continue
|
||||
}
|
||||
run(fam, "route", "flush", "table", tableArg)
|
||||
if err := addUnreachableFloor(fam, tableArg); err != nil {
|
||||
warnings = append(warnings, fmt.Sprintf(
|
||||
"egress %q: its %s routing table %d could not be given a fail-closed floor (%v), so if "+
|
||||
"interface %q (device %s) ever goes down the kernel deletes the route in that table, the "+
|
||||
"lookup falls through to the main table, and traffic bound to this egress leaves over the "+
|
||||
"plain WAN with your real IP address instead of failing.",
|
||||
eg.Name, famLabel(fam), table, err, iface, dev))
|
||||
}
|
||||
|
||||
// A gateway'd interface (WAN) needs `via <gw>`; a point-to-point tunnel
|
||||
// (wg/awg) has no gateway and routes straight out the device.
|
||||
gw := egressGateway(fam, eg.Interface, dev)
|
||||
gw := egressGateway(fam, iface, dev)
|
||||
var rerr error
|
||||
if gw != "" {
|
||||
rerr = runIP(fam, "route", "add", "default", "via", gw, "dev", dev, "table", tableArg)
|
||||
@@ -315,7 +382,7 @@ func addEgressRouting(m *model.Model) ([]string, error) {
|
||||
"and this egress is NOT applied: every node, group and rule bound to it is marked for an "+
|
||||
"empty table, falls through to the main table and leaves over the plain WAN with your real "+
|
||||
"IP address. Check that interface %q (device %s) exists and is up. Reason: %v",
|
||||
eg.Name, famLabel(fam), table, eg.Interface, dev, rerr))
|
||||
eg.Name, famLabel(fam), table, iface, dev, rerr))
|
||||
continue
|
||||
}
|
||||
if gw != "" {
|
||||
@@ -333,7 +400,7 @@ func addEgressRouting(m *model.Model) ([]string, error) {
|
||||
"CANNOT REACH ANYTHING outside its own subnet — every node, group and "+
|
||||
"rule bound to it will fail to connect. Check that the interface is up "+
|
||||
"and has a lease, or set a static nexthop (network.%s.%s).",
|
||||
eg.Name, famLabel(fam), eg.Interface, dev, eg.Interface, uciGatewayOption(fam)))
|
||||
eg.Name, famLabel(fam), iface, dev, iface, uciGatewayOption(fam)))
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -341,6 +408,66 @@ func addEgressRouting(m *model.Model) ([]string, error) {
|
||||
return warnings, nil
|
||||
}
|
||||
|
||||
// addL3Routing realises the policy-routing half of the L3 ingress (Globals.
|
||||
// L3Tunnel): the nft prerouting chain stamps L3Mark on LAN ICMP/ICMPv6, and this
|
||||
// binds that mark to a table whose only content is `default dev shater-l3`. That
|
||||
// rule+table pair is deliberately the ONLY path into the TUN — the engine opens
|
||||
// it with auto_route off, so the main routing table is never touched and
|
||||
// disabling the feature can never strand a stale default route there. Idempotent
|
||||
// (del-then-add), same shape as addEgressRouting above.
|
||||
//
|
||||
// Failures are WARNINGS, not errors, for addEgressRouting's documented reason:
|
||||
// returning would abort the whole apply — sysctls, conntrack flush and all —
|
||||
// over one broken binding. The `icmp "tunnel":` prefix is the `kind "name":`
|
||||
// shape apply's warning normaliser parses, so this lands in the panel as a named
|
||||
// warning (the kind says what the operator loses: ICMP), and the wording leads
|
||||
// with the observable consequence rather than the mechanics.
|
||||
func addL3Routing(m *model.Model) ([]string, error) {
|
||||
if !L3Enabled(m.Globals) {
|
||||
return nil, nil
|
||||
}
|
||||
run := func(args ...string) { _ = execCommand("ip", args...).Run() }
|
||||
var warnings []string
|
||||
markArg := fmt.Sprintf("0x%x", L3Mark(m.Globals))
|
||||
tableArg := fmt.Sprintf("%d", L3Table(m.Globals))
|
||||
fams := []string{"-4"}
|
||||
if m.Globals.IPv6 {
|
||||
fams = append(fams, "-6")
|
||||
}
|
||||
for _, fam := range fams {
|
||||
run(fam, "rule", "del", "fwmark", markArg, "lookup", tableArg)
|
||||
if err := runIP(fam, "rule", "add", "fwmark", markArg, "lookup", tableArg); err != nil {
|
||||
warnings = append(warnings, fmt.Sprintf(
|
||||
"icmp \"tunnel\": %s ping from LAN clients will NOT work: their ICMP is marked for the L3 "+
|
||||
"tunnel, but the policy rule (fwmark %s -> table %s) could not be installed, so the marked "+
|
||||
"packets fall through to the main routing table and the kill-switch treats them like any "+
|
||||
"escaped traffic — dropped when closed, out the plain WAN with your real IP address when "+
|
||||
"open. Reason: %v",
|
||||
famLabel(fam), markArg, tableArg, err))
|
||||
continue
|
||||
}
|
||||
run(fam, "route", "flush", "table", tableArg)
|
||||
if err := addUnreachableFloor(fam, tableArg); err != nil {
|
||||
warnings = append(warnings, fmt.Sprintf(
|
||||
"icmp \"tunnel\": %s table %s could not be given a fail-closed floor (%v), so whenever the "+
|
||||
"engine restarts and its %s device is destroyed, the kernel empties that table and the "+
|
||||
"marked ICMP falls through to the main routing table — dropped when the kill-switch is "+
|
||||
"closed, out the plain WAN with your real IP address when it is open.",
|
||||
famLabel(fam), tableArg, err, L3Device))
|
||||
}
|
||||
if err := runIP(fam, "route", "add", "default", "dev", L3Device, "table", tableArg); err != nil {
|
||||
warnings = append(warnings, fmt.Sprintf(
|
||||
"icmp \"tunnel\": %s ping from LAN clients will NOT work: their ICMP is marked for the L3 "+
|
||||
"tunnel, but table %s got no route into %s (is the engine running and its TUN up?), so the "+
|
||||
"marked packets fall through to the main routing table and the kill-switch treats them like "+
|
||||
"any escaped traffic — dropped when closed, out the plain WAN with your real IP address when "+
|
||||
"open. Reason: %v",
|
||||
famLabel(fam), tableArg, L3Device, err))
|
||||
}
|
||||
}
|
||||
return warnings, nil
|
||||
}
|
||||
|
||||
// famLabel renders an `ip` family flag for an operator-facing message.
|
||||
func famLabel(fam string) string {
|
||||
if fam == "-6" {
|
||||
@@ -564,6 +691,23 @@ func isPointToPoint(dev string) bool {
|
||||
// Which bindings must exist comes from the record addEgressRouting keeps (see
|
||||
// egressBindings): Globals alone cannot say how many egresses there are, and a
|
||||
// deleted rule leaves nothing on the system to count.
|
||||
//
|
||||
// # And the L3 ingress, for the third time
|
||||
//
|
||||
// addL3Routing installs a third `fwmark -> table` + default-route pair
|
||||
// (Globals.L3Tunnel), and it was added without being added here — the exact
|
||||
// defect the two paragraphs above describe as already caught twice, committed a
|
||||
// third time. Its trigger is not even an interface going down: the engine
|
||||
// recreates the shater-l3 TUN on any config change that restarts it (a node URI
|
||||
// edited, a subscription refreshed), the kernel deletes `default dev shater-l3`
|
||||
// with the old device and does not restore it with the new one, and the nft text
|
||||
// is unchanged so the fast-path skips ApplyRouting forever. LAN ping then stays
|
||||
// dead until somebody restarts the daemon, and nothing on screen says why.
|
||||
//
|
||||
// Unlike the egress bindings this one needs no record: L3Mark, L3Table and
|
||||
// L3Enabled are all derivable from Globals, which is what this function is
|
||||
// handed. TestEveryStampedMarkIsRoutedAndVerified is what stops the FOURTH mark
|
||||
// from being added without a line here.
|
||||
func RoutingPresent(g model.Globals) bool {
|
||||
mark, table := effFwmark(g), effTable(g)
|
||||
fams := []string{"-4"}
|
||||
@@ -575,9 +719,63 @@ func RoutingPresent(g model.Globals) bool {
|
||||
return false
|
||||
}
|
||||
}
|
||||
if !l3RoutingPresent(g, fams) {
|
||||
return false
|
||||
}
|
||||
return egressRoutingPresent()
|
||||
}
|
||||
|
||||
// l3RoutingPresent reports whether the L3-ingress binding is installed for every
|
||||
// family the model asks for: the `fwmark -> table` rule AND a real default route
|
||||
// inside that table. With Globals.L3Tunnel off nothing is installed by design,
|
||||
// so the answer is trivially yes.
|
||||
//
|
||||
// A false answer while the engine's TUN is genuinely absent (the engine is
|
||||
// down, or coming up) is intended, not a wart: it makes every reconcile re-run
|
||||
// ApplyRouting, so the route reappears by itself the moment the device does.
|
||||
// That is the same self-repair contract the per-egress bindings have, and the
|
||||
// main tproxy pair is guarded against the churn by addRouting's own presence
|
||||
// check.
|
||||
func l3RoutingPresent(g model.Globals, fams []string) bool {
|
||||
if !L3Enabled(g) {
|
||||
return true
|
||||
}
|
||||
mark, table := L3Mark(g), L3Table(g)
|
||||
for _, fam := range fams {
|
||||
out, err := execCommand("ip", fam, "rule", "show").Output()
|
||||
if err != nil || !fwmarkRulePresent(string(out), mark) {
|
||||
return false
|
||||
}
|
||||
rout, rerr := execCommand("ip", fam, "route", "show", "table", fmt.Sprintf("%d", table)).Output()
|
||||
if rerr != nil || !realDefaultRoutePresent(string(rout)) {
|
||||
return false
|
||||
}
|
||||
}
|
||||
return true
|
||||
}
|
||||
|
||||
// realDefaultRoutePresent reports whether a table holds a default route that
|
||||
// actually FORWARDS something: `default via <gw> dev X`, `default dev X`.
|
||||
//
|
||||
// It exists because addUnreachableFloor puts an `unreachable default` in every
|
||||
// mark-driven table, and a substring test for "default" would be satisfied by
|
||||
// that floor alone — turning the safety net into a blindfold, and reporting the
|
||||
// precise state it was installed to survive (device gone, real route deleted) as
|
||||
// a healthy plane that needs no repair. The floor keeps the packet from leaking;
|
||||
// this keeps the reconcile from giving up on fixing it.
|
||||
//
|
||||
// `ip route show` prints the route type as the FIRST token for every non-unicast
|
||||
// type (`unreachable default ...`, `blackhole default ...`) and omits it for
|
||||
// unicast, so a line that starts with "default" is a real one.
|
||||
func realDefaultRoutePresent(out string) bool {
|
||||
for _, line := range strings.Split(out, "\n") {
|
||||
if strings.HasPrefix(strings.TrimSpace(line), "default") {
|
||||
return true
|
||||
}
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
// mainRoutingPresentFam reports whether BOTH halves of the main tproxy plane are
|
||||
// installed for one family: the fwmark rule and the `local default dev lo` route
|
||||
// inside the table it points at.
|
||||
@@ -619,11 +817,14 @@ func egressRoutingPresent() bool {
|
||||
if !fwmarkRulePresent(out, b.mark) {
|
||||
return false
|
||||
}
|
||||
// Any default route will do: `default via <gw> dev X` for a gateway'd
|
||||
// Any REAL default route will do: `default via <gw> dev X` for a gateway'd
|
||||
// uplink, `default dev X` for a point-to-point tunnel. What must never pass
|
||||
// is an EMPTY table, which is what an ifdown leaves behind.
|
||||
// is an EMPTY table — nor a table holding only addUnreachableFloor's
|
||||
// `unreachable default`, which is the shape an ifdown leaves behind now
|
||||
// that the floor is installed, and which means the very thing this check
|
||||
// exists to catch.
|
||||
rout, err := execCommand("ip", b.fam, "route", "show", "table", fmt.Sprintf("%d", b.table)).Output()
|
||||
if err != nil || !strings.Contains(string(rout), "default") {
|
||||
if err != nil || !realDefaultRoutePresent(string(rout)) {
|
||||
return false
|
||||
}
|
||||
}
|
||||
|
||||
@@ -131,6 +131,18 @@ func SaveBootArmor(ruleset string) (changed bool, err error) {
|
||||
if err = os.Rename(name, BootArmorPath); err != nil {
|
||||
return false, err
|
||||
}
|
||||
// And fsync the DIRECTORY. Syncing the file only guarantees its contents; the
|
||||
// rename that publishes them is a directory operation, and on the flash
|
||||
// filesystems this ships on (jffs2/f2fs/ubifs, and ext4 on the x86 images) an
|
||||
// unsynced rename can be lost across a power cut while the file's data is not.
|
||||
// The result would be a boot with no armor and no error anywhere — i.e. exactly
|
||||
// the failure this file exists to prevent, on exactly the boot it exists for.
|
||||
// A best-effort sync: a filesystem that will not open its own directory is not
|
||||
// a reason to report a write that did happen as failed.
|
||||
if d, derr := os.Open(dir); derr == nil {
|
||||
_ = d.Sync()
|
||||
_ = d.Close()
|
||||
}
|
||||
return true, nil
|
||||
}
|
||||
|
||||
|
||||
@@ -202,8 +202,11 @@ func TestFwmarkRulePresentIsExact(t *testing.T) {
|
||||
// over the ordinary WAN with the router's real address.
|
||||
func TestEgressRouteAddFailureIsReported(t *testing.T) {
|
||||
f := wgEgressNet()
|
||||
// Only the REAL default route fails; `route add unreachable default` (the
|
||||
// table's fail-closed floor) is left working, so the single warning asserted
|
||||
// below is the one this test is about.
|
||||
f.failIf = func(name string, arg []string) bool {
|
||||
return name == "ip" && len(arg) >= 3 && arg[1] == "route" && arg[2] == "add"
|
||||
return name == "ip" && len(arg) >= 4 && arg[1] == "route" && arg[2] == "add" && arg[3] == "default"
|
||||
}
|
||||
f.install(t)
|
||||
|
||||
@@ -269,7 +272,7 @@ func TestEgressRuleAddFailureIsReported(t *testing.T) {
|
||||
func TestFailedEgressKeepsRoutingAbsent(t *testing.T) {
|
||||
f := wgEgressNet()
|
||||
f.failIf = func(name string, arg []string) bool {
|
||||
return name == "ip" && len(arg) >= 3 && arg[1] == "route" && arg[2] == "add"
|
||||
return name == "ip" && len(arg) >= 4 && arg[1] == "route" && arg[2] == "add" && arg[3] == "default"
|
||||
}
|
||||
f.install(t)
|
||||
if _, err := addEgressRouting(wgEgressModel()); err != nil {
|
||||
|
||||
@@ -0,0 +1,602 @@
|
||||
package netplane
|
||||
|
||||
// The mark plane: every fwmark this ruleset STAMPS must be routed somewhere, and
|
||||
// every place that trusts a mark must be unable to trust it once the routing
|
||||
// stopped happening.
|
||||
//
|
||||
// All four defects below share one shape. `ip rule fwmark X lookup N` does not
|
||||
// deliver a packet to table N — it delivers the LOOKUP to table N, and a lookup
|
||||
// that finds nothing there continues down the rule list to `main`. So an empty
|
||||
// table N does not mean "this traffic is stuck", it means "this traffic is
|
||||
// routed as if it had never been marked": out the default WAN, with the router's
|
||||
// real address, still carrying a mark the firewall was told to trust. Tables go
|
||||
// empty for entirely ordinary reasons and without anything we render changing —
|
||||
// an ifdown, an engine restart destroying its TUN — so neither the nft
|
||||
// idempotence check nor the operator has any reason to notice.
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"regexp"
|
||||
"sort"
|
||||
"strings"
|
||||
"testing"
|
||||
|
||||
"github.com/sagernet/sing-box/shater/model"
|
||||
)
|
||||
|
||||
// --- defect 1: a dead egress leaked non-TCP/UDP past a CLOSED kill-switch ----
|
||||
|
||||
// TestUntunnelableEgressAcceptIsBoundToItsDevice is the security regression.
|
||||
//
|
||||
// SCENARIO (reproduced by the auditor against a live kernel, then reduced to
|
||||
// this): untunnelable_egress='wwan', untunnelable='block', kill_switch='closed'.
|
||||
// The interface goes down. The kernel deletes every route through its device,
|
||||
// including `default dev wwan0 table 8208`. Nothing we render changes, so no
|
||||
// apply runs. nft keeps stamping 0x2100 on every non-TCP/UDP packet from the
|
||||
// LAN; `ip rule fwmark 0x2100 lookup 8208` now finds an empty table and falls
|
||||
// through to main; the packet leaves out the default WAN — and the forward chain
|
||||
// waved it there itself, because its accept looked only at the mark, and that
|
||||
// accept sits ABOVE the fail-closed drop.
|
||||
//
|
||||
// The fix is the conjunction: the mark says where the packet was SENT, oifname
|
||||
// says where it is actually GOING, and only both together mean the routing did
|
||||
// what the mark asked. `meta mark X oifname "dev"` is strictly narrower than the
|
||||
// mark alone, so it can only ever remove leaks, never add one.
|
||||
//
|
||||
// (The comment this replaced argued, correctly, that an oifname accept ALONE
|
||||
// would be too loose — it would bless whatever the routing table pushed at that
|
||||
// device, from wherever. That is why the mark stays. It is not an argument for
|
||||
// dropping oifname, which is the conclusion that was drawn from it.)
|
||||
func TestUntunnelableEgressAcceptIsBoundToItsDevice(t *testing.T) {
|
||||
m := realisticModel() // kill_switch=closed, egress "wwan" on wwan0 at idx 0
|
||||
m.Globals.UntunnelableEgress = "wwan"
|
||||
m.Globals.Untunnelable = "block"
|
||||
rs, err := RenderNft(m)
|
||||
if err != nil {
|
||||
t.Fatalf("RenderNft: %v", err)
|
||||
}
|
||||
|
||||
// Precondition: prerouting really is stamping the mark, or the rest is moot.
|
||||
if pre := preroutingChain(t, rs); !strings.Contains(pre, "meta mark set 0x2100 accept") {
|
||||
t.Fatalf("prerouting no longer stamps the egress mark; this test is testing nothing:\n%s", pre)
|
||||
}
|
||||
|
||||
fwd := forwardChain(t, rs)
|
||||
for _, line := range strings.Split(fwd, "\n") {
|
||||
line = strings.TrimSpace(line)
|
||||
if !strings.Contains(line, "meta mark 0x2100") || !strings.HasSuffix(line, "accept") {
|
||||
continue
|
||||
}
|
||||
if !strings.Contains(line, `oifname "wwan0"`) {
|
||||
t.Errorf("the forward chain accepts the untunnelable-egress mark on the mark ALONE:\n"+
|
||||
" %s\n"+
|
||||
"An ifdown empties table 8208, the fwmark lookup falls through to main, and this line "+
|
||||
"then accepts the packet on its way out the DEFAULT WAN — above the fail-closed drop, "+
|
||||
"with the kill-switch closed. Bind the accept to the egress device as well: "+
|
||||
"`meta mark 0x2100 oifname \"wwan0\" accept`.", line)
|
||||
}
|
||||
}
|
||||
// And the accept must still exist and still precede the drop, or the feature
|
||||
// is dead instead of leaky.
|
||||
acc := strings.Index(fwd, `meta mark 0x2100 oifname "wwan0" accept`)
|
||||
drop := strings.Index(fwd, "meta nfproto ipv4 drop")
|
||||
if acc < 0 {
|
||||
t.Fatalf("the device-bound egress accept is missing entirely — marked ESP/GRE is now dropped "+
|
||||
"by the kill-switch on its way out the egress:\n%s", fwd)
|
||||
}
|
||||
if drop >= 0 && acc > drop {
|
||||
t.Errorf("the egress accept must precede the fail-closed drop or it never matches:\n%s", fwd)
|
||||
}
|
||||
}
|
||||
|
||||
// TestEgressTableGetsFailClosedFloor is the other half of the same defect, and
|
||||
// the half that survives an OPEN kill-switch (where there is no forward drop to
|
||||
// be saved by).
|
||||
//
|
||||
// The nft conjunction above stops the leaked packet from being ACCEPTED; this
|
||||
// stops it from being ROUTED at all. An `unreachable default` at the maximum
|
||||
// metric loses to any real default route while one exists, has no device so the
|
||||
// kernel never garbage-collects it, and turns "lookup failed, try main" into
|
||||
// "lookup succeeded: unreachable". The fallthrough stops being a matter of
|
||||
// noticing a warning in time and becomes structurally impossible.
|
||||
func TestEgressTableGetsFailClosedFloor(t *testing.T) {
|
||||
f := wgEgressNet()
|
||||
f.install(t)
|
||||
if _, err := addEgressRouting(wgEgressModel()); err != nil {
|
||||
t.Fatalf("addEgressRouting: %v", err)
|
||||
}
|
||||
|
||||
var floor, real string
|
||||
for _, c := range f.calls {
|
||||
if strings.Contains(c, "route add unreachable default") && strings.HasSuffix(c, "table 8208") {
|
||||
floor = c
|
||||
}
|
||||
if strings.Contains(c, "route add default") && strings.HasSuffix(c, "table 8208") {
|
||||
real = c
|
||||
}
|
||||
}
|
||||
if floor == "" {
|
||||
t.Fatalf("no `ip route add unreachable default ... table 8208` was issued.\n"+
|
||||
"Without it, an ifdown empties the egress table, the fwmark lookup falls through to the "+
|
||||
"MAIN table, and everything bound to this egress leaves over the plain WAN with the "+
|
||||
"router's real address. Calls were:\n %s", strings.Join(f.calls, "\n "))
|
||||
}
|
||||
if !strings.Contains(floor, "metric "+unreachableFloorMetric) {
|
||||
t.Errorf("the floor must carry the maximal metric or it can outrank the real default route: %q", floor)
|
||||
}
|
||||
if real == "" {
|
||||
t.Fatalf("the real default route disappeared along with the fix:\n %s", strings.Join(f.calls, "\n "))
|
||||
}
|
||||
}
|
||||
|
||||
// TestFloorDoesNotMaskAnEmptyTable: the floor keeps the packet from leaking, and
|
||||
// must not therefore keep the RECONCILE from repairing the egress. A table
|
||||
// holding only `unreachable default` is the exact state an ifdown now leaves
|
||||
// behind, i.e. the state the presence check exists to catch — a substring test
|
||||
// for "default" would read it as healthy and turn the safety net into a
|
||||
// blindfold.
|
||||
func TestFloorDoesNotMaskAnEmptyTable(t *testing.T) {
|
||||
const floorOnly = "unreachable default metric 4294967295"
|
||||
f := wgEgressNet()
|
||||
f.install(t)
|
||||
if _, err := addEgressRouting(wgEgressModel()); err != nil {
|
||||
t.Fatalf("addEgressRouting: %v", err)
|
||||
}
|
||||
f.rules = map[string]string{"-4": "32765:\tfrom all fwmark 0x2000 lookup 8192\n" +
|
||||
"32764:\tfrom all fwmark 0x2100 lookup 8208"}
|
||||
f.routes = map[string]string{
|
||||
"-4 8192": "local default dev lo scope host",
|
||||
"-4 8208": floorOnly,
|
||||
}
|
||||
if RoutingPresent(egressPlaneGlobals()) {
|
||||
t.Fatal("a table holding ONLY the unreachable floor is an egress whose real route is gone; " +
|
||||
"reporting it present makes the reconcile skip ApplyRouting and the egress never comes back")
|
||||
}
|
||||
// Sanity: the same table WITH a real route is present again.
|
||||
f.routes["-4 8208"] = "default dev wg0 scope link\n" + floorOnly
|
||||
if !RoutingPresent(egressPlaneGlobals()) {
|
||||
t.Fatal("a healthy egress table must still report present with the floor installed")
|
||||
}
|
||||
}
|
||||
|
||||
// --- defect 2: RoutingPresent had never heard of the L3 ingress -------------
|
||||
|
||||
// l3Globals pins the marks/tables the L3 cases hard-code: mark 0x2080, table 8200.
|
||||
func l3Globals() model.Globals {
|
||||
return model.Globals{FwmarkBase: 0x2000, TableBase: 0x2000, L3Tunnel: true}
|
||||
}
|
||||
|
||||
// TestRoutingPresentSeesL3Table.
|
||||
//
|
||||
// addL3Routing installs a THIRD `fwmark -> table` + default-route pair and
|
||||
// RoutingPresent was never told about it — it checked the main tproxy pair and
|
||||
// the per-egress bindings and returned. This is the same defect that had already
|
||||
// been found twice (main table, then egress tables) and whose own comment in
|
||||
// apply.go says "a presence check must cover everything its Apply counterpart
|
||||
// installs, or the idempotent fast-path becomes a trap".
|
||||
//
|
||||
// Its trigger does not even need an interface to go down. Edit a node's URI: the
|
||||
// engine is recreated, the kernel destroys the shater-l3 device and with it
|
||||
// `default dev shater-l3 table 8200`, and the new device arrives without it. The
|
||||
// rendered nft text is unchanged, so nftCurrent stays true; RoutingPresent said
|
||||
// true on the strength of main+egress; applyLocked therefore skipped
|
||||
// ApplyRoutingWithWarnings and NOTHING ever reinstalled the route. LAN ping was
|
||||
// dead until somebody restarted the daemon, permanently, at plane=full, with no
|
||||
// warning — and with untunnelable=direct or an open kill-switch it was not dead
|
||||
// but leaking.
|
||||
func TestRoutingPresentSeesL3Table(t *testing.T) {
|
||||
const (
|
||||
mainRule = "32765:\tfrom all fwmark 0x2000 lookup 8192"
|
||||
l3Rule = "32763:\tfrom all fwmark 0x2080 lookup 8200"
|
||||
mainRoute = "local default dev lo scope host"
|
||||
l3Route = "default dev shater-l3 scope link"
|
||||
floor = "unreachable default metric 4294967295"
|
||||
)
|
||||
cases := []struct {
|
||||
name string
|
||||
rules string
|
||||
l3Table string
|
||||
want bool
|
||||
}{
|
||||
{"complete", mainRule + "\n" + l3Rule, l3Route + "\n" + floor, true},
|
||||
// THE REGRESSION: the engine restarted, the TUN was destroyed and
|
||||
// recreated, the kernel took the route with it. The rule survived.
|
||||
{"l3 table emptied by an engine restart", mainRule + "\n" + l3Rule, floor, false},
|
||||
// The other half: the rule itself went away (a `network reload` flushes
|
||||
// `ip rule` wholesale).
|
||||
{"l3 rule gone", mainRule, l3Route, false},
|
||||
}
|
||||
for _, c := range cases {
|
||||
t.Run(c.name, func(t *testing.T) {
|
||||
f := &egressPlaneNet{
|
||||
rules: map[string]string{"-4": c.rules},
|
||||
routes: map[string]string{"-4 8192": mainRoute, "-4 8200": c.l3Table},
|
||||
}
|
||||
f.install(t)
|
||||
if got := RoutingPresent(l3Globals()); got != c.want {
|
||||
t.Fatalf("RoutingPresent = %v, want %v.\n"+
|
||||
"A presence check blind to the L3 pair lets the fast-path skip ApplyRouting forever "+
|
||||
"while LAN ICMP is marked for a table that no longer routes it.", got, c.want)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
// With l3_tunnel OFF nothing is installed by design, so its absence must not be
|
||||
// read as a broken plane — the L3 half may only ever ADD a reason to re-apply.
|
||||
func TestRoutingPresentUnaffectedWithoutL3(t *testing.T) {
|
||||
f := &egressPlaneNet{
|
||||
rules: map[string]string{"-4": "32765:\tfrom all fwmark 0x2000 lookup 8192"},
|
||||
routes: map[string]string{"-4 8192": "local default dev lo scope host"},
|
||||
}
|
||||
f.install(t)
|
||||
g := l3Globals()
|
||||
g.L3Tunnel = false
|
||||
if !RoutingPresent(g) {
|
||||
t.Fatal("a healthy main pair with l3_tunnel off must report present")
|
||||
}
|
||||
}
|
||||
|
||||
// Both families, for the reason the main pair checks both: `ip -6 rule` is
|
||||
// flushed independently of `ip -4 rule`, so a v6 half can go missing on its own
|
||||
// and a v4-only check would call the plane intact.
|
||||
func TestRoutingPresentChecksL3PerFamily(t *testing.T) {
|
||||
f := &egressPlaneNet{
|
||||
rules: map[string]string{
|
||||
"-4": "32765:\tfrom all fwmark 0x2000 lookup 8192\n32763:\tfrom all fwmark 0x2080 lookup 8200",
|
||||
"-6": "32765:\tfrom all fwmark 0x2000 lookup 8192", // the v6 L3 rule is gone
|
||||
},
|
||||
routes: map[string]string{
|
||||
"-4 8192": "local default dev lo scope host",
|
||||
"-6 8192": "local default dev lo metric 1024 pref medium",
|
||||
"-4 8200": "default dev shater-l3 scope link",
|
||||
"-6 8200": "default dev shater-l3 metric 1024 pref medium",
|
||||
},
|
||||
}
|
||||
f.install(t)
|
||||
g := l3Globals()
|
||||
g.IPv6 = true
|
||||
if RoutingPresent(g) {
|
||||
t.Fatal("the v6 L3 rule is missing; reporting present leaves v6 ICMP marked with nowhere to go")
|
||||
}
|
||||
}
|
||||
|
||||
// The L3 table gets the same fail-closed floor as the egress tables — the engine
|
||||
// restarting is a far more frequent event than an interface going down.
|
||||
func TestL3TableGetsFailClosedFloor(t *testing.T) {
|
||||
f := &egressPlaneNet{}
|
||||
f.install(t)
|
||||
warns, err := addL3Routing(&model.Model{Globals: l3Globals()})
|
||||
if err != nil {
|
||||
t.Fatalf("addL3Routing: %v", err)
|
||||
}
|
||||
if len(warns) != 0 {
|
||||
t.Fatalf("a healthy install must warn about nothing: %v", warns)
|
||||
}
|
||||
var floor string
|
||||
for _, c := range f.calls {
|
||||
if strings.Contains(c, "route add unreachable default") && strings.HasSuffix(c, "table 8200") {
|
||||
floor = c
|
||||
}
|
||||
}
|
||||
if floor == "" {
|
||||
t.Fatalf("no fail-closed floor in the L3 table: when the engine recreates its TUN the table is "+
|
||||
"empty, the fwmark lookup falls through to main, and marked ICMP is dropped by a closed "+
|
||||
"kill-switch or leaks past an open one. Calls were:\n %s", strings.Join(f.calls, "\n "))
|
||||
}
|
||||
}
|
||||
|
||||
// --- defect 3: the validator and the data plane disagreed about a name ------
|
||||
|
||||
// TestUntunnelableEgressResolutionLockstep.
|
||||
//
|
||||
// model.ValidateUntunnelableEgress duplicates netplane.UntunnelableEgressBinding
|
||||
// because the import can only run one way (netplane imports model). The
|
||||
// duplication is unavoidable; the DIVERGENCE was not, and a comment saying "MUST
|
||||
// change in lockstep" had already failed to prevent one: the validator rejected
|
||||
// an egress on `strings.TrimSpace(Interface) == ""` and the binding on
|
||||
// `Interface == ""`, so `option interface ' '` produced a panel saying "the
|
||||
// option is ignored" over a data plane that was stamping the mark — straight
|
||||
// into defect 1 above, with the operator explicitly told it could not happen.
|
||||
//
|
||||
// One table, both functions, one verdict. A future divergence is a test failure
|
||||
// rather than a warning that lies exactly where it matters most.
|
||||
func TestUntunnelableEgressResolutionLockstep(t *testing.T) {
|
||||
cases := []struct {
|
||||
name string
|
||||
option string
|
||||
egresses []model.Egress
|
||||
resolves bool
|
||||
}{
|
||||
{"empty option", "", []model.Egress{{Name: "wan2", Type: "interface", Interface: "eth1"}}, false},
|
||||
{"blank option", " ", []model.Egress{{Name: "wan2", Type: "interface", Interface: "eth1"}}, false},
|
||||
{"plain match", "wan2", []model.Egress{{Name: "wan2", Type: "interface", Interface: "eth1"}}, true},
|
||||
{"tunnel type", "vpn", []model.Egress{{Name: "vpn", Type: "tunnel", Interface: "wg0"}}, true},
|
||||
{"case-insensitive name", "WAN2", []model.Egress{{Name: "wan2", Type: "interface", Interface: "eth1"}}, true},
|
||||
{"padded option", " wan2 ", []model.Egress{{Name: "wan2", Type: "interface", Interface: "eth1"}}, true},
|
||||
{"padded egress name", "wan2", []model.Egress{{Name: " wan2 ", Type: "interface", Interface: "eth1"}}, true},
|
||||
{"padded type", "wan2", []model.Egress{{Name: "wan2", Type: " Interface ", Interface: "eth1"}}, true},
|
||||
{"no such egress", "wan9", []model.Egress{{Name: "wan2", Type: "interface", Interface: "eth1"}}, false},
|
||||
{"wrong type (direct)", "d", []model.Egress{{Name: "d", Type: "direct"}}, false},
|
||||
{"wrong type (byedpi)", "d", []model.Egress{{Name: "d", Type: "byedpi"}}, false},
|
||||
{"no interface", "hole", []model.Egress{{Name: "hole", Type: "interface"}}, false},
|
||||
// THE DIVERGENCE. IfaceDevice(" ") hands back " ", which is neither
|
||||
// empty nor a device: the binding used to say ok, the validator used to
|
||||
// say "ignored", and the packets were marked for a table nobody built.
|
||||
{"whitespace interface", "hole", []model.Egress{{Name: "hole", Type: "interface", Interface: " "}}, false},
|
||||
{"whitespace interface, tunnel", "hole", []model.Egress{{Name: "hole", Type: "tunnel", Interface: "\t"}}, false},
|
||||
// A padded but real interface name is a name, not an absence.
|
||||
{"padded interface", "wan2", []model.Egress{{Name: "wan2", Type: "interface", Interface: " eth1 "}}, true},
|
||||
// First name match decides, in both.
|
||||
{"second egress matches", "vpn", []model.Egress{
|
||||
{Name: "wan2", Type: "interface", Interface: "eth1"},
|
||||
{Name: "vpn", Type: "tunnel", Interface: "wg0"},
|
||||
}, true},
|
||||
}
|
||||
for _, c := range cases {
|
||||
t.Run(c.name, func(t *testing.T) {
|
||||
g := model.Globals{FwmarkBase: 0x2000, TableBase: 0x2000, UntunnelableEgress: c.option}
|
||||
_, _, dev, ok := UntunnelableEgressBinding(&model.Model{Globals: g, Egresses: c.egresses})
|
||||
warns := model.ValidateUntunnelableEgress(g, c.egresses)
|
||||
// The validator is silent when the option is empty (nothing was
|
||||
// asked for) and when it resolves; it warns when it was asked for
|
||||
// and cannot be honoured.
|
||||
validatorAccepts := len(warns) == 0
|
||||
wantValidatorAccepts := c.resolves || strings.TrimSpace(c.option) == ""
|
||||
|
||||
if ok != c.resolves {
|
||||
t.Errorf("netplane.UntunnelableEgressBinding ok = %v (dev %q), want %v", ok, dev, c.resolves)
|
||||
}
|
||||
if validatorAccepts != wantValidatorAccepts {
|
||||
t.Errorf("model.ValidateUntunnelableEgress silent = %v, want %v: %v",
|
||||
validatorAccepts, wantValidatorAccepts, warns)
|
||||
}
|
||||
// The lockstep assertion proper: for a non-empty option the two must
|
||||
// agree. This is the one that fires when they drift.
|
||||
if strings.TrimSpace(c.option) != "" && ok != validatorAccepts {
|
||||
t.Errorf("VERDICTS DIVERGE for option %q: the data plane %s the egress while the "+
|
||||
"validator %s it. Whichever way round, the panel is telling the operator the "+
|
||||
"opposite of what the packets do — and the marked-but-unrouted direction is a "+
|
||||
"kill-switch bypass (see TestUntunnelableEgressAcceptIsBoundToItsDevice).",
|
||||
c.option, resolvedWord(ok), resolvedWord(validatorAccepts))
|
||||
}
|
||||
if ok && strings.TrimSpace(dev) == "" {
|
||||
t.Errorf("a resolved binding must yield a usable device, got %q", dev)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func resolvedWord(ok bool) string {
|
||||
if ok {
|
||||
return "ACCEPTS"
|
||||
}
|
||||
return "REJECTS"
|
||||
}
|
||||
|
||||
// --- the general rule the three defects above are instances of --------------
|
||||
|
||||
var stampedMarkRe = regexp.MustCompile(`meta mark set 0x([0-9a-fA-F]+)`)
|
||||
var ruleAddRe = regexp.MustCompile(`^ip (-[46]) rule add fwmark 0x([0-9a-fA-F]+) lookup (\d+)$`)
|
||||
|
||||
// TestEveryStampedMarkIsRoutedAndVerified is the checklist, executable.
|
||||
//
|
||||
// Both blocking defects were the same omission twice: a new mark was added to
|
||||
// prerouting and one of the three things a mark needs was not added with it.
|
||||
// Rather than trusting the next author to remember, this test derives the marks
|
||||
// from the RENDERED RULESET and demands all three of each:
|
||||
//
|
||||
// (a) it is ROUTED — ApplyRouting issues an `ip rule fwmark <M> lookup <T>`;
|
||||
// (b) its table has a FAIL-CLOSED FLOOR, so an emptied table cannot fall
|
||||
// through to main;
|
||||
// (c) the forward chain, if it trusts the mark at all, trusts it only in
|
||||
// conjunction with an oifname;
|
||||
// (d) RoutingPresent NOTICES when that table loses its real route, so the
|
||||
// reconcile repairs it instead of skipping forever.
|
||||
//
|
||||
// A mark added to prerouting without any of these fails here without anyone
|
||||
// having to think of writing a test for it.
|
||||
func TestEveryStampedMarkIsRoutedAndVerified(t *testing.T) {
|
||||
// Everything on at once, so every mark this plane can stamp is stamped.
|
||||
m := realisticModel()
|
||||
m.Globals.IPv6 = true
|
||||
m.Globals.L3Tunnel = true
|
||||
m.Globals.DNSIntercept = true
|
||||
m.Globals.UntunnelableEgress = "wwan"
|
||||
|
||||
rs, err := RenderNft(m)
|
||||
if err != nil {
|
||||
t.Fatalf("RenderNft: %v", err)
|
||||
}
|
||||
pre, fwd := preroutingChain(t, rs), forwardChain(t, rs)
|
||||
|
||||
stamped := map[uint64]bool{}
|
||||
for _, mt := range stampedMarkRe.FindAllStringSubmatch(pre, -1) {
|
||||
var v uint64
|
||||
if _, err := fmt.Sscanf(mt[1], "%x", &v); err == nil {
|
||||
stamped[v] = true
|
||||
}
|
||||
}
|
||||
// The divert mark is the one documented exception: its "routing" is the main
|
||||
// tproxy pair (mark -> table -> `local default dev lo`), which RoutingPresent
|
||||
// checks separately and which cannot go stale the way a device-backed table
|
||||
// can, because `lo` never goes down. It is also never accepted in forward.
|
||||
// The exception is verified, not asserted — see the end of this test.
|
||||
divert := uint64(effFwmark(m.Globals))
|
||||
delete(stamped, divert)
|
||||
if len(stamped) < 2 {
|
||||
t.Fatalf("expected at least the L3 and untunnelable-egress marks to be stamped, found %v.\n"+
|
||||
"If the marks moved, move this test with them; if they stopped being stamped, this "+
|
||||
"whole file is testing nothing.", stamped)
|
||||
}
|
||||
|
||||
f := &egressPlaneNet{}
|
||||
f.install(t)
|
||||
if _, err := ApplyRoutingWithWarnings(m); err != nil {
|
||||
t.Fatalf("ApplyRoutingWithWarnings: %v", err)
|
||||
}
|
||||
|
||||
// (a) every stamped mark is bound to a table, per family.
|
||||
routed := map[uint64][]markBindingView{}
|
||||
for _, c := range f.calls {
|
||||
mt := ruleAddRe.FindStringSubmatch(c)
|
||||
if mt == nil {
|
||||
continue
|
||||
}
|
||||
var v uint64
|
||||
if _, err := fmt.Sscanf(mt[2], "%x", &v); err != nil {
|
||||
continue
|
||||
}
|
||||
routed[v] = append(routed[v], markBindingView{mt[1], mt[3]})
|
||||
}
|
||||
for _, mark := range sortedMarks(stamped) {
|
||||
bs := routed[mark]
|
||||
if len(bs) == 0 {
|
||||
t.Errorf("(a) mark 0x%x is STAMPED in prerouting but no `ip rule fwmark 0x%x lookup ...` is "+
|
||||
"ever installed. A mark with no rule is not a route, it is a main-table fallthrough: "+
|
||||
"the packet leaves out the default WAN carrying a mark the firewall trusts.", mark, mark)
|
||||
continue
|
||||
}
|
||||
for _, b := range bs {
|
||||
// (b) the table it points at has a fail-closed floor.
|
||||
want := fmt.Sprintf("ip %s route add unreachable default metric %s table %s",
|
||||
b.fam, unreachableFloorMetric, b.table)
|
||||
if !containsCall(f.calls, want) {
|
||||
t.Errorf("(b) table %s (mark 0x%x, %s) gets no `unreachable default` floor. When the "+
|
||||
"device behind it disappears the kernel empties the table, the fwmark lookup "+
|
||||
"fails over to main, and the traffic leaves over the plain WAN. Expected call: %q",
|
||||
b.table, mark, b.fam, want)
|
||||
}
|
||||
}
|
||||
// (c) forward may trust the mark only together with an oifname.
|
||||
for _, line := range strings.Split(fwd, "\n") {
|
||||
line = strings.TrimSpace(line)
|
||||
if !strings.Contains(line, fmt.Sprintf("meta mark 0x%x", mark)) || !strings.HasSuffix(line, "accept") {
|
||||
continue
|
||||
}
|
||||
if !strings.Contains(line, "oifname") {
|
||||
t.Errorf("(c) the forward chain accepts mark 0x%x on the mark alone:\n %s\n"+
|
||||
"The mark says where the packet was sent; only oifname says where it went. "+
|
||||
"Once the mark's table is empty those differ, and this line waves the packet "+
|
||||
"out the default WAN above the fail-closed drop.", mark, line)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// (d) RoutingPresent must notice each table losing its real route.
|
||||
healthy := healthyPlaneView(routed, uint64(effTable(m.Globals)))
|
||||
f.rules, f.routes = healthy.rules, healthy.routes
|
||||
if !RoutingPresent(m.Globals) {
|
||||
t.Fatalf("the synthetic healthy plane must report present, or every case below is vacuous.\n"+
|
||||
"rules: %v\nroutes: %v", f.rules, f.routes)
|
||||
}
|
||||
for _, mark := range sortedMarks(stamped) {
|
||||
for _, b := range routed[mark] {
|
||||
key := b.fam + " " + b.table
|
||||
saved := f.routes[key]
|
||||
f.routes[key] = "unreachable default metric " + unreachableFloorMetric // real route gone
|
||||
if RoutingPresent(m.Globals) {
|
||||
t.Errorf("(d) table %s (mark 0x%x, %s) lost its real route and RoutingPresent still says "+
|
||||
"present. The fast-path then skips ApplyRouting forever and nothing ever "+
|
||||
"reinstalls it — the failure is permanent and a reconcile cannot repair it.",
|
||||
b.table, mark, b.fam)
|
||||
}
|
||||
f.routes[key] = saved
|
||||
}
|
||||
}
|
||||
// The exception is checked too: the main tproxy table is covered by
|
||||
// mainRoutingPresentFam, so "skip the divert mark" is a delegation and not a
|
||||
// hole in the sweep above.
|
||||
mainKey := fmt.Sprintf("-4 %d", effTable(m.Globals))
|
||||
saved := f.routes[mainKey]
|
||||
f.routes[mainKey] = ""
|
||||
if RoutingPresent(m.Globals) {
|
||||
t.Errorf("the main tproxy table (mark 0x%x) is the one mark this test skips; it must be "+
|
||||
"covered by mainRoutingPresentFam instead, and it is not", divert)
|
||||
}
|
||||
f.routes[mainKey] = saved
|
||||
}
|
||||
|
||||
// markBindingView is one `ip <fam> rule add fwmark <M> lookup <T>` the apply issued.
|
||||
type markBindingView struct{ fam, table string }
|
||||
|
||||
type planeView struct {
|
||||
rules map[string]string
|
||||
routes map[string]string
|
||||
}
|
||||
|
||||
// healthyPlaneView synthesises the `ip rule show` / `ip route show table N`
|
||||
// output of a fully installed plane from the rule-adds the apply issued.
|
||||
func healthyPlaneView(routed map[uint64][]markBindingView, mainTable uint64) planeView {
|
||||
v := planeView{rules: map[string]string{}, routes: map[string]string{}}
|
||||
pref := 32765
|
||||
for _, mark := range sortedMarks(markSet(routed)) {
|
||||
for _, b := range routed[mark] {
|
||||
v.rules[b.fam] += fmt.Sprintf("%d:\tfrom all fwmark 0x%x lookup %s\n", pref, mark, b.table)
|
||||
pref--
|
||||
if b.table == fmt.Sprintf("%d", mainTable) {
|
||||
v.routes[b.fam+" "+b.table] = "local default dev lo scope host"
|
||||
continue
|
||||
}
|
||||
v.routes[b.fam+" "+b.table] = "default dev somedev scope link\n" +
|
||||
"unreachable default metric " + unreachableFloorMetric
|
||||
}
|
||||
}
|
||||
return v
|
||||
}
|
||||
|
||||
func markSet(routed map[uint64][]markBindingView) map[uint64]bool {
|
||||
out := map[uint64]bool{}
|
||||
for k := range routed {
|
||||
out[k] = true
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
func sortedMarks(set map[uint64]bool) []uint64 {
|
||||
out := make([]uint64, 0, len(set))
|
||||
for k := range set {
|
||||
out = append(out, k)
|
||||
}
|
||||
sort.Slice(out, func(i, j int) bool { return out[i] < out[j] })
|
||||
return out
|
||||
}
|
||||
|
||||
func containsCall(calls []string, want string) bool {
|
||||
for _, c := range calls {
|
||||
if c == want {
|
||||
return true
|
||||
}
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
// --- defect 4's other half: the duplicated layout constants -----------------
|
||||
|
||||
// TestMarkLayoutConstantsLockstep pins model's copy of the mark/table layout to
|
||||
// the netplane functions that actually compute it. model.ValidateMarkBases has
|
||||
// to know the offsets to tell an operator that fwmark_base 0x7f puts the L3 mark
|
||||
// on 0xff; it cannot import netplane (the dependency runs the other way), so it
|
||||
// carries its own copy. A drift would make the validator bless a base the data
|
||||
// plane cannot use, or condemn one it can — silently, in both directions.
|
||||
func TestMarkLayoutConstantsLockstep(t *testing.T) {
|
||||
g := model.Globals{FwmarkBase: 0x4000, TableBase: 0x5000}
|
||||
checks := []struct {
|
||||
what string
|
||||
derived, model uint32
|
||||
}{
|
||||
{"loop-guard mark", loopMark, model.MarkLoopGuard},
|
||||
{"L3 mark offset", L3Mark(g) - g.FwmarkBase, model.MarkL3Offset},
|
||||
{"egress mark offset", EgressMark(g, 0) - g.FwmarkBase, model.MarkEgressOffset},
|
||||
{"egress mark stride", EgressMark(g, 7) - EgressMark(g, 0), 7},
|
||||
{"L3 table offset", L3Table(g) - g.TableBase, model.TableL3Offset},
|
||||
{"egress table offset", EgressTable(g, 0) - g.TableBase, model.TableEgressOffset},
|
||||
{"egress table stride", EgressTable(g, 7) - EgressTable(g, 0), 7},
|
||||
}
|
||||
for _, c := range checks {
|
||||
if c.derived != c.model {
|
||||
t.Errorf("%s: netplane computes 0x%x, model.ValidateMarkBases assumes 0x%x — the validator "+
|
||||
"is now judging a layout that does not exist", c.what, c.derived, c.model)
|
||||
}
|
||||
}
|
||||
}
|
||||
+267
-7
@@ -45,6 +45,12 @@ const (
|
||||
egEgressMarkOffset = 0x100
|
||||
egEgressTableOffset = 0x10
|
||||
|
||||
// l3MarkOffset / l3TableOffset carve the L3-ingress mark and routing table
|
||||
// out of the same reserved block, BELOW the per-egress range so no number of
|
||||
// egresses can ever collide with them.
|
||||
l3MarkOffset = 0x80
|
||||
l3TableOffset = 0x08
|
||||
|
||||
// defaultFwmark / defaultTable are the model contract defaults when a raw
|
||||
// Globals arrives with the fields unset (0).
|
||||
defaultFwmark = 0x2000
|
||||
@@ -81,6 +87,90 @@ func EgressTable(g model.Globals, idx int) uint32 {
|
||||
return effTable(g) + egEgressTableOffset + uint32(idx)
|
||||
}
|
||||
|
||||
// L3Device is the TUN interface the engine opens for the L3 ingress. Non-TCP/UDP
|
||||
// LAN traffic (today: ICMP echo) is policy-routed into it so the engine can carry
|
||||
// it through an L3-capable outbound. The generate stage emits an inbound bound to
|
||||
// this exact name; ApplyRouting points the L3 table's default route at it.
|
||||
const L3Device = "shater-l3"
|
||||
|
||||
// L3Mark is the fwmark the nft prerouting chain stamps on LAN traffic the tunnel
|
||||
// can only carry at layer 3, and which ApplyRouting binds to L3Table.
|
||||
func L3Mark(g model.Globals) uint32 {
|
||||
return effFwmark(g) + l3MarkOffset
|
||||
}
|
||||
|
||||
// L3Table is the routing table whose default route leaves via L3Device.
|
||||
func L3Table(g model.Globals) uint32 {
|
||||
return effTable(g) + l3TableOffset
|
||||
}
|
||||
|
||||
// L3Enabled reports whether the L3 ingress is configured. It is opt-in: without
|
||||
// it the untunnelable policy behaves exactly as before.
|
||||
func L3Enabled(g model.Globals) bool { return g.L3Tunnel }
|
||||
|
||||
// EgressDevice resolves an egress to the L3 device its mark is routed out of, or
|
||||
// "" when this egress has no device at all (wrong type, or no interface).
|
||||
//
|
||||
// It is the SINGLE resolution that everything touching an egress mark must
|
||||
// agree on, because those things are not independent: the prerouting marking
|
||||
// (UntunnelableEgressBinding), the forward-chain accept that lets that mark past
|
||||
// the kill-switch, and addEgressRouting, which installs the mark's `ip rule` and
|
||||
// routing table. A mark that one of them stamps and another does not route is
|
||||
// not a cosmetic inconsistency — it is a packet that falls through to the main
|
||||
// table and leaves over the plain WAN with the router's real address.
|
||||
//
|
||||
// They HAD already diverged: the binding rejected an egress on `Interface == ""`
|
||||
// while model.ValidateUntunnelableEgress rejected it on
|
||||
// `strings.TrimSpace(Interface) == ""`, so `option interface ' '` produced a
|
||||
// panel that said "the option is ignored" over a data plane that was marking
|
||||
// packets for a table nobody built. Trimming here is what the operator meant and
|
||||
// what the validator already assumed; TestUntunnelableEgressResolutionLockstep
|
||||
// pins the two verdicts together so the next drift is a test failure and not a
|
||||
// leak.
|
||||
func EgressDevice(eg model.Egress) string {
|
||||
if t := strings.ToLower(strings.TrimSpace(eg.Type)); t != "interface" && t != "tunnel" {
|
||||
return ""
|
||||
}
|
||||
// IfaceDevice("") falls back to br-lan — the right default for an INBOUND
|
||||
// with no network, catastrophic here: the LAN bridge would impersonate the
|
||||
// egress device while addEgressRouting never installs the mark's routing, so
|
||||
// marked packets would fall through to the main table and leave over the
|
||||
// plain WAN. An egress without an interface has no device.
|
||||
iface := strings.TrimSpace(eg.Interface)
|
||||
if iface == "" {
|
||||
return ""
|
||||
}
|
||||
return IfaceDevice(iface)
|
||||
}
|
||||
|
||||
// UntunnelableEgressBinding resolves Globals.UntunnelableEgress to the egress's
|
||||
// index, its nft mark and its device. ok is false when the option is empty or
|
||||
// names something that is not an interface/tunnel egress with a device — the
|
||||
// caller then renders nothing and the Untunnelable policy stays in sole charge,
|
||||
// which is the fail-closed reading of a typo.
|
||||
//
|
||||
// The mark and table are the egress's OWN (EgressMark/EgressTable), not a third
|
||||
// pair: addEgressRouting already binds them to that device, so routing the
|
||||
// untunnelable protocols there is a matter of stamping the existing mark in
|
||||
// prerouting and nothing else.
|
||||
func UntunnelableEgressBinding(m *model.Model) (idx int, mark uint32, device string, ok bool) {
|
||||
name := strings.TrimSpace(m.Globals.UntunnelableEgress)
|
||||
if name == "" {
|
||||
return 0, 0, "", false
|
||||
}
|
||||
for i, eg := range m.Egresses {
|
||||
if !strings.EqualFold(strings.TrimSpace(eg.Name), name) {
|
||||
continue
|
||||
}
|
||||
dev := EgressDevice(eg)
|
||||
if dev == "" {
|
||||
return 0, 0, "", false
|
||||
}
|
||||
return i, EgressMark(m.Globals, i), dev, true
|
||||
}
|
||||
return 0, 0, "", false
|
||||
}
|
||||
|
||||
// EgressOutboundTag is the outbound tag the generator emits for an
|
||||
// interface/tunnel egress (coordination contract with the generate stage).
|
||||
func EgressOutboundTag(name string) string { return "egress-" + name }
|
||||
@@ -563,12 +653,20 @@ func RenderHoldNftAt(m *model.Model, now time.Time) (string, error) {
|
||||
sb.WriteString("\t\ttype filter hook forward priority filter; policy accept;\n")
|
||||
sb.WriteString("\t\t# HOLDING PLANE: the engine is down; LAN->WAN is blocked (fail-closed).\n")
|
||||
// Router-origin / egress-marked traffic survives, so the daemon can still
|
||||
// reach the network and recover on its own.
|
||||
// reach the network and recover on its own. The egress accepts are the same
|
||||
// mark+oifname conjunction the full plane uses (see the essay there): the
|
||||
// mark says where a packet was sent, oifname says where it actually went,
|
||||
// and only both together mean the egress's routing table did its job. Here
|
||||
// the difference is theory rather than practice — this plane stamps no marks
|
||||
// at all — but a bare mark accept in the plane whose entire purpose is
|
||||
// failing closed would be exactly the wrong thing to leave lying around.
|
||||
sb.WriteString(fmt.Sprintf("\t\tmeta mark 0x%x accept\n", loopMark))
|
||||
for i, eg := range m.Egresses {
|
||||
if t := strings.ToLower(eg.Type); t == "interface" || t == "tunnel" {
|
||||
sb.WriteString(fmt.Sprintf("\t\tmeta mark 0x%x accept\n", EgressMark(m.Globals, i)))
|
||||
dev := EgressDevice(eg)
|
||||
if dev == "" {
|
||||
continue
|
||||
}
|
||||
sb.WriteString(fmt.Sprintf("\t\tmeta mark 0x%x oifname %q accept\n", EgressMark(m.Globals, i), dev))
|
||||
}
|
||||
// Keep the LAN itself usable.
|
||||
sb.WriteString("\t\t" + iif + " ip daddr { 10.0.0.0/8, 172.16.0.0/12, 192.168.0.0/16, 127.0.0.0/8, 169.254.0.0/16 } accept\n")
|
||||
@@ -582,6 +680,28 @@ func RenderHoldNftAt(m *model.Model, now time.Time) (string, error) {
|
||||
// every destination is treated as tunnelled. `direct` still lets everything out
|
||||
// (it needs no knowledge); `block` and `icmp` degrade to their conservative form.
|
||||
sb.WriteString(untunnelableRules(m.Globals, iif, nil))
|
||||
// Deliberately NO L3-ingress marking here (Globals.L3Tunnel): the TUN it
|
||||
// routes into is created BY the engine, and this plane exists precisely
|
||||
// because the engine is not running. Marking ICMP at a device that does not
|
||||
// exist would not keep ping alive — it would dead-end the packets in an
|
||||
// empty routing table while LOOKING like a feature. While holding, the
|
||||
// honest verdicts are the policy above and the drops below.
|
||||
//
|
||||
// The untunnelable-egress marking (Globals.UntunnelableEgress) is absent
|
||||
// for a harder reason than the TUN's: the egress DEVICE may well exist
|
||||
// while the engine is down, but this plane is installed precisely when an
|
||||
// apply aborted before the netplane stage — possibly before
|
||||
// addEgressRouting ever installed the `ip rule fwmark -> table` pair
|
||||
// (first boot after a power cut, engine choking on an undownloaded
|
||||
// rule-set, is the canonical case). A packet marked without that pair
|
||||
// falls through to the MAIN table, and the egress-mark accept above would
|
||||
// wave it out the DEFAULT WAN — a kill-switch bypass wearing the egress's
|
||||
// name. It cannot be done from here anyway: this plane is one forward
|
||||
// chain by design, and a forward-hook mark cannot re-route a packet whose
|
||||
// routing decision already happened. While holding, non-TCP/UDP traffic
|
||||
// gets the Untunnelable policy verdict above — the same honest
|
||||
// degradation the L3 ingress chose.
|
||||
//
|
||||
// Everything else from the protected devices is dropped, both families —
|
||||
// IPv6 unconditionally, because with no engine there is no v6 divert either.
|
||||
sb.WriteString("\t\t" + iif + " meta nfproto ipv4 drop\n")
|
||||
@@ -848,6 +968,11 @@ func renderNft(m *model.Model, plan *UntunnelablePlan, now time.Time) (string, [
|
||||
sb.WriteString(fmt.Sprintf("\t\tmeta mark 0x%x accept\n", EgressMark(m.Globals, i)))
|
||||
}
|
||||
}
|
||||
// An L3-marked packet must never be re-diverted or re-marked either
|
||||
// (Globals.L3Tunnel; same contract as the loop-guard/egress marks above).
|
||||
if L3Enabled(m.Globals) {
|
||||
sb.WriteString(fmt.Sprintf("\t\tmeta mark 0x%x accept\n", L3Mark(m.Globals)))
|
||||
}
|
||||
// --- DNS force-intercept (Globals.DNSIntercept) ---
|
||||
// Divert ALL LAN plaintext :53 into the engine, INCLUDING queries addressed to
|
||||
// the router itself, which the fib-local bypass below would otherwise hand to
|
||||
@@ -866,6 +991,83 @@ func renderNft(m *model.Model, plan *UntunnelablePlan, now time.Time) (string, [
|
||||
sb.WriteString("\t\tfib daddr type local accept\n")
|
||||
sb.WriteString("\t\tip daddr { 10.0.0.0/8, 172.16.0.0/12, 192.168.0.0/16, 127.0.0.0/8, 169.254.0.0/16, 224.0.0.0/4, 255.255.255.255 } accept\n")
|
||||
sb.WriteString("\t\tip6 daddr { ::1, fc00::/7, fe80::/10, ff00::/8 } accept\n")
|
||||
// --- L3 ingress (Globals.L3Tunnel): route what TPROXY cannot carry ---
|
||||
// Kernel TPROXY needs a socket to hand the packet to, so everything below
|
||||
// moves TCP and UDP only. With the L3 ingress enabled, LAN ICMP/ICMPv6 bound
|
||||
// for a PUBLIC destination (every local plane was accepted above: router
|
||||
// itself via fib-local, RFC1918/link-local/multicast via the daddr sets) is
|
||||
// fwmark-routed into the engine's TUN instead of falling to the forward-chain
|
||||
// verdicts. The mark plus addL3Routing's `ip rule` are the ONLY way into that
|
||||
// device: the TUN runs with auto_route off and the main routing table is
|
||||
// never touched, by design — see L3Device.
|
||||
//
|
||||
// Deliberately ONLY icmp/ipv6-icmp, never `l4proto != { tcp, udp }`. The
|
||||
// boundary is sing-tun's ForwardDispatcher, not the outbound: flow_parse.go
|
||||
// classifies TCP, UDP and ICMP echo and nothing else, and flow_dispatch.go
|
||||
// builds its NAT flows from port-like selectors, so a marked
|
||||
// ESP/AH/GRE/IGMP/SCTP packet would enter the device and vanish — a black
|
||||
// hole wearing a tunnel's name — instead of receiving the untunnelable
|
||||
// policy's honest forward-chain verdict (drop, or direct-out if the
|
||||
// operator chose to disclose that much). That policy — or the
|
||||
// untunnelable-egress marking below, whose receiving side is the kernel
|
||||
// and has no such boundary — stays in charge of everything this ingress
|
||||
// cannot carry.
|
||||
if L3Enabled(m.Globals) && dnsIif != "" {
|
||||
sb.WriteString(fmt.Sprintf("\t\t%s ip protocol icmp meta mark set 0x%x accept\n", dnsIif, L3Mark(m.Globals)))
|
||||
if m.Globals.IPv6 {
|
||||
// ND/RA hold the LAN's v6 plane together and mean nothing off-link.
|
||||
// Most are already out of reach (multicast daddr and router-addressed
|
||||
// unicast were accepted above), but a unicast NS/NA between global LAN
|
||||
// addresses is not, and one tunnelled neighbour probe is enough to
|
||||
// take v6 down — so accept them BEFORE the icmpv6 mark. MLD is
|
||||
// multicast-addressed and already covered by the ff00::/8 accept.
|
||||
sb.WriteString("\t\t" + dnsIif + " icmpv6 type { nd-router-solicit, nd-router-advert, nd-neighbor-solicit, nd-neighbor-advert } accept\n")
|
||||
sb.WriteString(fmt.Sprintf("\t\t%s meta l4proto ipv6-icmp meta mark set 0x%x accept\n", dnsIif, L3Mark(m.Globals)))
|
||||
}
|
||||
}
|
||||
// --- untunnelable egress (Globals.UntunnelableEgress): the kernel carries the rest ---
|
||||
// Everything the L3 ingress above leaves alone — ESP, AH, GRE, IGMP, SCTP,
|
||||
// and ICMP too when l3_tunnel is off — is stamped with the named egress's
|
||||
// OWN mark and accepted; the `ip rule fwmark -> table` pair that
|
||||
// addEgressRouting already binds to that egress then routes it out the
|
||||
// egress device. No new mark, no new table, no engine in the path.
|
||||
//
|
||||
// The filter is the WIDE negated set while the L3 block above matches ICMP
|
||||
// only, and the asymmetry is the receiving side. The L3 mark delivers into
|
||||
// the engine's TUN, where sing-tun's ForwardDispatcher classifies nothing
|
||||
// beyond TCP/UDP/ICMP echo — wide there is a black hole. This mark
|
||||
// delivers to the kernel's own forwarding path, which routes ANY IP
|
||||
// protocol — narrow here would just amputate the feature.
|
||||
//
|
||||
// The order is load-bearing twice. AFTER the L3 block: when l3_tunnel is
|
||||
// on, ICMP must be claimed by the mark that routes it through the engine's
|
||||
// own rules, and this line may only sweep up what that block did not take
|
||||
// (first match wins). AFTER the local-plane accepts (fib-local, RFC1918/
|
||||
// link-local/multicast, ND/RA): pinging the gateway, LAN-to-LAN traffic
|
||||
// and v6 neighbour discovery must never leave through an uplink. Nothing
|
||||
// is re-accepted here: the per-egress mgmt bypass at the top of this chain
|
||||
// and its forward-chain mirror already pass this mark.
|
||||
if _, uMark, _, uOK := UntunnelableEgressBinding(m); uOK && dnsIif != "" {
|
||||
if m.Globals.IPv6 {
|
||||
if !L3Enabled(m.Globals) {
|
||||
// The guard the L3 block plants before its ICMPv6 mark, needed
|
||||
// here whenever that block is off: a unicast NS/NA between
|
||||
// global LAN addresses is not covered by the daddr accepts
|
||||
// above, and one neighbour probe sent out an uplink is enough
|
||||
// to take LAN IPv6 down. With l3_tunnel on it already ran.
|
||||
sb.WriteString("\t\t" + dnsIif + " icmpv6 type { nd-router-solicit, nd-router-advert, nd-neighbor-solicit, nd-neighbor-advert } accept\n")
|
||||
}
|
||||
sb.WriteString(fmt.Sprintf("\t\t%s meta l4proto != { tcp, udp } meta mark set 0x%x accept\n", dnsIif, uMark))
|
||||
} else {
|
||||
// addEgressRouting installs the -6 rule/table pair only when
|
||||
// Globals.IPv6 is on. Marked v6 without that pair does not go to
|
||||
// the egress — it falls through to the MAIN v6 table carrying a
|
||||
// mark the forward chain accepts, i.e. out the default WAN past
|
||||
// the kill-switch. Scope the mark to v4 and leave v6 to the
|
||||
// forward chain's existing verdicts instead.
|
||||
sb.WriteString(fmt.Sprintf("\t\t%s meta nfproto ipv4 meta l4proto != { tcp, udp } meta mark set 0x%x accept\n", dnsIif, uMark))
|
||||
}
|
||||
}
|
||||
// --- DNS anti-leak: only :853 is carved out; plain :53 is DIVERTED (D14) ---
|
||||
if dnsIif != "" {
|
||||
// Keep encrypted-DNS bypass ports (DoT :853 tcp, DoQ :853 udp) OUT of
|
||||
@@ -910,11 +1112,52 @@ func renderNft(m *model.Model, plan *UntunnelablePlan, now time.Time) (string, [
|
||||
|
||||
// 1) mgmt / interface+tunnel egress marks: router-origin and egress
|
||||
// traffic must never be dropped (mirror of the prerouting bypass).
|
||||
// These same accepts are what let untunnelable-egress traffic
|
||||
// (Globals.UntunnelableEgress) past the kill-switch: prerouting
|
||||
// stamps the egress's own mark on non-TCP/UDP LAN ingress, and it
|
||||
// is accepted HERE.
|
||||
//
|
||||
// THE MARK ALONE IS NOT ENOUGH, and believing it was is what made
|
||||
// this the leakiest line in the file. The mark says where the
|
||||
// packet was SENT; it does not say where it WENT. `ip rule fwmark
|
||||
// 0x2100 lookup 8208` only diverts the lookup — when that table is
|
||||
// empty the lookup FAILS OVER to the main table, silently, and the
|
||||
// packet leaves out the default WAN still carrying the mark this
|
||||
// line accepts, straight past a closed kill-switch. The table is
|
||||
// empty for the most ordinary reason there is: the egress rides an
|
||||
// interface, the interface goes down, and the kernel deletes every
|
||||
// route through that device. Nothing in this ruleset changed, so
|
||||
// the applier's idempotence check sees no work, and the leak is
|
||||
// permanent.
|
||||
//
|
||||
// Ordinary egress traffic never had this hole because the engine
|
||||
// binds those sockets to the device (SO_BINDTODEVICE, see
|
||||
// generate/outbound.go) and a dead device fails the socket. The
|
||||
// untunnelable-egress path has no such second opinion: a mark is
|
||||
// all it is made of. So the accept carries the second opinion
|
||||
// instead — `meta mark X oifname "dev"` is the conjunction of "we
|
||||
// sent it there" AND "it is actually going there", which is
|
||||
// strictly narrower than either half and true only when the
|
||||
// routing did what the mark asked for.
|
||||
//
|
||||
// (An oifname accept ALONE would be the loose one — it would bless
|
||||
// everything the routing table ever pushes at that device, whatever
|
||||
// put it there — which is why the mark stays. The comment this
|
||||
// replaces argued that point correctly and then drew the wrong
|
||||
// conclusion from it: it justified dropping oifname INSTEAD OF the
|
||||
// mark, which nobody proposed, rather than using both.)
|
||||
//
|
||||
// An egress with no device (EgressDevice == "") gets no accept at
|
||||
// all: addEgressRouting installs no rule and no table for it and
|
||||
// prerouting stamps nothing, so an accept for its mark could only
|
||||
// ever wave through something that has no business here.
|
||||
sb.WriteString(fmt.Sprintf("\t\tmeta mark 0x%x accept\n", loopMark))
|
||||
for i, eg := range m.Egresses {
|
||||
if t := strings.ToLower(eg.Type); t == "interface" || t == "tunnel" {
|
||||
sb.WriteString(fmt.Sprintf("\t\tmeta mark 0x%x accept\n", EgressMark(m.Globals, i)))
|
||||
dev := EgressDevice(eg)
|
||||
if dev == "" {
|
||||
continue
|
||||
}
|
||||
sb.WriteString(fmt.Sprintf("\t\tmeta mark 0x%x oifname %q accept\n", EgressMark(m.Globals, i), dev))
|
||||
}
|
||||
// 2) LAN-to-LAN / link-local (v4): the LAN itself must keep working
|
||||
// (same private-range bypass sets the prerouting chain uses).
|
||||
@@ -932,11 +1175,28 @@ func renderNft(m *model.Model, plan *UntunnelablePlan, now time.Time) (string, [
|
||||
sb.WriteString("\t\t" + dnsIif + " ip6 daddr fe80::/10 accept\n")
|
||||
sb.WriteString("\t\t" + dnsIif + " ip6 daddr ff00::/8 accept\n")
|
||||
}
|
||||
// 4) Untunnelable protocols (never TCP/UDP, so never divertible): the
|
||||
// 4) L3 ingress legs (Globals.L3Tunnel). The LAN->TUN leg still has a
|
||||
// LAN iifname, so the fail-closed drop below would eat exactly the
|
||||
// packets prerouting just marked and routed — accept by the TUN
|
||||
// oifname first. The TUN->LAN leg (iifname) is the engine answering
|
||||
// LAN clients. These accepts speak for OUR table only: fw4's
|
||||
// forward chain runs independently and a drop there still wins —
|
||||
// the shater-l3 firewall zone that keeps fw4 out of the way is
|
||||
// provisioned in uci-defaults, not here.
|
||||
if L3Enabled(m.Globals) {
|
||||
sb.WriteString(fmt.Sprintf("\t\toifname \"%s\" accept\n", L3Device))
|
||||
sb.WriteString(fmt.Sprintf("\t\tiifname \"%s\" accept\n", L3Device))
|
||||
}
|
||||
// 5) Untunnelable protocols (never TCP/UDP, so never divertible): the
|
||||
// configured policy decides whether they leave directly or are dropped
|
||||
// with everything else. Must precede the drops to have any effect.
|
||||
// With an untunnelable egress bound, the traffic prerouting marked
|
||||
// never reaches these lines (accepted by mark in 1); the policy
|
||||
// keeps judging what the marking skipped — v6 while Globals.IPv6
|
||||
// is off — and everything, as before, when the option is empty
|
||||
// or does not resolve.
|
||||
sb.WriteString(untunnelableRules(m.Globals, dnsIif, plan))
|
||||
// 5) Fail-closed drops for public-bound LAN-ingress traffic.
|
||||
// 6) Fail-closed drops for public-bound LAN-ingress traffic.
|
||||
sb.WriteString("\t\t" + dnsIif + " meta nfproto ipv4 drop\n")
|
||||
sb.WriteString("\t\t" + dnsIif + " meta nfproto ipv6 drop\n")
|
||||
case ipv6Off:
|
||||
|
||||
+463
-3
@@ -204,10 +204,20 @@ func TestRenderNftForwardFailClosed(t *testing.T) {
|
||||
if !strings.Contains(fwd, "meta mark 0xff accept") {
|
||||
t.Errorf("closed mode: forward chain missing the mgmt loop-guard accept:\n%s", fwd)
|
||||
}
|
||||
// Egress-marked traffic (interface egress at idx 0 => 0x2100) is bypassed too.
|
||||
if !strings.Contains(fwd, "meta mark 0x2100 accept") {
|
||||
// Egress-marked traffic (interface egress at idx 0 => 0x2100) is bypassed too,
|
||||
// but ONLY in conjunction with the egress DEVICE. The mark records where the
|
||||
// packet was sent; oifname records where it actually went. They disagree
|
||||
// exactly when the egress's routing table has been emptied (ifdown, network
|
||||
// restart) and the lookup fell through to main — i.e. when the packet is on
|
||||
// its way out the default WAN wearing a mark this chain was told to trust.
|
||||
if !strings.Contains(fwd, `meta mark 0x2100 oifname "wwan0" accept`) {
|
||||
t.Errorf("closed mode: forward chain missing the egress-mark bypass:\n%s", fwd)
|
||||
}
|
||||
if strings.Contains(fwd, "meta mark 0x2100 accept") {
|
||||
t.Errorf("closed mode: the egress-mark accept must be qualified by oifname — a mark-only "+
|
||||
"accept passes the same packet after its table went empty and it fell back to the "+
|
||||
"default WAN:\n%s", fwd)
|
||||
}
|
||||
|
||||
// Open: no v4 fail-closed drop in the forward chain.
|
||||
mo := realisticModel()
|
||||
@@ -640,7 +650,7 @@ func TestRenderHoldNft(t *testing.T) {
|
||||
{"atomic replace", "delete table inet shater"},
|
||||
{"forward hook", "type filter hook forward priority filter;"},
|
||||
{"router-origin survives", "meta mark 0xff accept"},
|
||||
{"egress mark survives", "meta mark 0x2100 accept"},
|
||||
{"egress mark survives", `meta mark 0x2100 oifname "wwan0" accept`},
|
||||
{"LAN-to-LAN survives", "ip daddr { 10.0.0.0/8, 172.16.0.0/12, 192.168.0.0/16"},
|
||||
{"ICMPv6 ND survives", "nd-neighbor-solicit"},
|
||||
{"v4 blocked", "meta nfproto ipv4 drop"},
|
||||
@@ -973,3 +983,453 @@ func TestUntunnelableInHoldingPlane(t *testing.T) {
|
||||
t.Errorf("block policy must emit no relaxation in the holding plane:\n%s", rs2)
|
||||
}
|
||||
}
|
||||
|
||||
// --- L3 ingress: LAN ICMP into the engine's TUN (Globals.L3Tunnel) -----------
|
||||
|
||||
// preroutingChain extracts the `chain prerouting { ... }` block, mirroring
|
||||
// forwardChain, so an ordering assert cannot be satisfied by a similar line
|
||||
// living in the wrong hook.
|
||||
func preroutingChain(t *testing.T, rs string) string {
|
||||
t.Helper()
|
||||
start := strings.Index(rs, "chain prerouting {")
|
||||
if start < 0 {
|
||||
t.Fatalf("no prerouting chain in ruleset:\n%s", rs)
|
||||
}
|
||||
rest := rs[start:]
|
||||
end := strings.Index(rest, "\n\t}\n")
|
||||
if end < 0 {
|
||||
t.Fatalf("unterminated prerouting chain:\n%s", rest)
|
||||
}
|
||||
return rest[:end]
|
||||
}
|
||||
|
||||
// TestL3IngressOptIn: l3_tunnel is opt-in, and with it unset the render must be
|
||||
// byte-for-byte what it was before the feature existed — no TUN device, no L3
|
||||
// mark, no extra ND accept. Every deployed router upgrades through this default;
|
||||
// a single changed line here is an unreviewed behaviour change on all of them.
|
||||
func TestL3IngressOptIn(t *testing.T) {
|
||||
m := realisticModel() // L3Tunnel deliberately left false
|
||||
rs, err := RenderNft(m)
|
||||
if err != nil {
|
||||
t.Fatalf("RenderNft: %v", err)
|
||||
}
|
||||
// 0x2080 = FwmarkBase 0x2000 + the 0x80 L3 offset; it appears in NO other
|
||||
// mark or table, so its absence covers every gated line that stamps or
|
||||
// bypasses the mark.
|
||||
for _, s := range []string{L3Device, "0x2080"} {
|
||||
if strings.Contains(rs, s) {
|
||||
t.Errorf("L3Tunnel=false must leave the render untouched, found %q:\n%s", s, rs)
|
||||
}
|
||||
}
|
||||
// With IPv6 on and L3 off there is no ND/RA accept anywhere: the forward
|
||||
// chain's ND line exists only in the IPv6-OFF branch. Finding one would mean
|
||||
// the L3 prerouting block leaked past its gate.
|
||||
m6 := realisticModel()
|
||||
m6.Globals.IPv6 = true
|
||||
rs6, err := RenderNft(m6)
|
||||
if err != nil {
|
||||
t.Fatalf("RenderNft(ipv6): %v", err)
|
||||
}
|
||||
if strings.Contains(rs6, "nd-router-solicit") {
|
||||
t.Errorf("L3Tunnel=false leaked the prerouting ND/RA accept:\n%s", rs6)
|
||||
}
|
||||
}
|
||||
|
||||
// TestL3IngressDivertsICMP: with l3_tunnel=1, LAN ICMP is marked for the L3
|
||||
// table in prerouting AND both TUN legs are let through the forward chain.
|
||||
// Losing the mark means ping is back to the engine's faked echo replies; losing
|
||||
// the forward legs means the marked packet is eaten by the very fail-closed
|
||||
// drop the mark exists to route around.
|
||||
func TestL3IngressDivertsICMP(t *testing.T) {
|
||||
m := realisticModel()
|
||||
m.Globals.L3Tunnel = true
|
||||
rs, err := RenderNft(m)
|
||||
if err != nil {
|
||||
t.Fatalf("RenderNft: %v", err)
|
||||
}
|
||||
pre := preroutingChain(t, rs)
|
||||
|
||||
mark := strings.Index(pre, "ip protocol icmp meta mark set 0x2080 accept")
|
||||
if mark < 0 {
|
||||
t.Fatalf("prerouting does not mark LAN ICMP for the L3 table:\n%s", pre)
|
||||
}
|
||||
// The marking must be ingress-scoped, or WAN-ingress ICMP gets marked too.
|
||||
line := pre[strings.LastIndex(pre[:mark], "\n")+1 : mark]
|
||||
if !strings.Contains(line, "iifname") {
|
||||
t.Errorf("L3 mark is unscoped (would also mark WAN ingress): %q", strings.TrimSpace(line))
|
||||
}
|
||||
// Router-local and LAN-to-LAN ICMP must be accepted BEFORE the mark:
|
||||
// pinging the gateway or a LAN neighbour must never enter the tunnel.
|
||||
if fibLocal := strings.Index(pre, "fib daddr type local accept"); fibLocal < 0 || fibLocal > mark {
|
||||
t.Errorf("router-local accept must precede the L3 mark or pinging the gateway breaks:\n%s", pre)
|
||||
}
|
||||
if lanV4 := strings.Index(pre, "ip daddr { 10.0.0.0/8"); lanV4 < 0 || lanV4 > mark {
|
||||
t.Errorf("private-range accept must precede the L3 mark or LAN-to-LAN ping breaks:\n%s", pre)
|
||||
}
|
||||
// The mgmt bypass keeps an already-marked packet from being re-diverted.
|
||||
if !strings.Contains(pre, "meta mark 0x2080 accept") {
|
||||
t.Errorf("prerouting missing the L3-mark mgmt bypass:\n%s", pre)
|
||||
}
|
||||
|
||||
fwd := forwardChain(t, rs)
|
||||
oif := strings.Index(fwd, "oifname \"shater-l3\" accept")
|
||||
if oif < 0 {
|
||||
t.Fatalf("forward chain missing the LAN->TUN leg (oifname): the marked packet is dropped fail-closed:\n%s", fwd)
|
||||
}
|
||||
if !strings.Contains(fwd, "iifname \"shater-l3\" accept") {
|
||||
t.Errorf("forward chain missing the TUN->LAN leg (iifname): the engine's replies are droppable:\n%s", fwd)
|
||||
}
|
||||
if drop := strings.Index(fwd, "meta nfproto ipv4 drop"); drop >= 0 && oif > drop {
|
||||
t.Errorf("TUN accept must precede the fail-closed drop or it never matches:\n%s", fwd)
|
||||
}
|
||||
}
|
||||
|
||||
// TestL3IngressNDBeforeMark: the ND/RA accept must STRICTLY precede the ICMPv6
|
||||
// mark. One tunnelled neighbour solicitation is enough to break address
|
||||
// resolution, which takes IPv6 down for the whole LAN while every dashboard
|
||||
// stays green.
|
||||
func TestL3IngressNDBeforeMark(t *testing.T) {
|
||||
m := realisticModel()
|
||||
m.Globals.L3Tunnel = true
|
||||
m.Globals.IPv6 = true
|
||||
rs, err := RenderNft(m)
|
||||
if err != nil {
|
||||
t.Fatalf("RenderNft: %v", err)
|
||||
}
|
||||
pre := preroutingChain(t, rs)
|
||||
|
||||
nd := strings.Index(pre, "icmpv6 type { nd-router-solicit, nd-router-advert, nd-neighbor-solicit, nd-neighbor-advert } accept")
|
||||
v6mark := strings.Index(pre, "meta l4proto ipv6-icmp meta mark set 0x2080 accept")
|
||||
if v6mark < 0 {
|
||||
t.Fatalf("IPv6 on: prerouting does not mark ICMPv6 for the L3 table:\n%s", pre)
|
||||
}
|
||||
if nd < 0 {
|
||||
t.Fatalf("IPv6 on: prerouting has no ND/RA accept guarding the ICMPv6 mark:\n%s", pre)
|
||||
}
|
||||
if nd > v6mark {
|
||||
t.Errorf("ND/RA accept sits AFTER the ICMPv6 mark: neighbour discovery gets tunnelled and LAN IPv6 dies:\n%s", pre)
|
||||
}
|
||||
}
|
||||
|
||||
// TestL3IngressLeavesUntunnelablePolicyInCharge: the L3 ingress carries ICMP and
|
||||
// NOTHING else. ESP/AH/GRE/IGMP/SCTP cannot enter the engine — sing-tun's
|
||||
// ForwardDispatcher classifies only TCP/UDP/ICMP echo and builds its NAT flows
|
||||
// from port-like selectors, so other protocols are never dispatched — and
|
||||
// marking them would trade the untunnelable policy's honest verdict (drop, or
|
||||
// direct-out by explicit operator choice) for a silent black hole inside the
|
||||
// TUN.
|
||||
func TestL3IngressLeavesUntunnelablePolicyInCharge(t *testing.T) {
|
||||
m := realisticModel()
|
||||
m.Globals.L3Tunnel = true
|
||||
m.Globals.IPv6 = true
|
||||
rs, err := RenderNft(m)
|
||||
if err != nil {
|
||||
t.Fatalf("RenderNft: %v", err)
|
||||
}
|
||||
// Every line that stamps the L3 mark must match ICMP explicitly — never the
|
||||
// negated set `!= { tcp, udp }` that would sweep ESP/GRE along.
|
||||
for _, line := range strings.Split(rs, "\n") {
|
||||
if !strings.Contains(line, "meta mark set 0x2080") {
|
||||
continue
|
||||
}
|
||||
if !strings.Contains(line, "ip protocol icmp") && !strings.Contains(line, "ipv6-icmp") {
|
||||
t.Errorf("non-ICMP traffic marked into the TUN (black hole, not tunnel): %q", strings.TrimSpace(line))
|
||||
}
|
||||
}
|
||||
|
||||
// The policy machinery itself survives L3: with `direct` the negated-set
|
||||
// accept is still emitted, and the TUN legs precede it, so ICMP tunnels
|
||||
// while ESP/GRE still leave by the operator's explicit choice.
|
||||
m.Globals.Untunnelable = "direct"
|
||||
rsDirect, err := RenderNft(m)
|
||||
if err != nil {
|
||||
t.Fatalf("RenderNft(direct): %v", err)
|
||||
}
|
||||
fwd := forwardChain(t, rsDirect)
|
||||
policy := strings.Index(fwd, "meta l4proto != { tcp, udp } accept")
|
||||
if policy < 0 {
|
||||
t.Fatalf("untunnelable=direct lost its accept with L3 on:\n%s", fwd)
|
||||
}
|
||||
if oif := strings.Index(fwd, "oifname \"shater-l3\" accept"); oif < 0 || oif > policy {
|
||||
t.Errorf("TUN legs must precede the untunnelable accept — L3 owns what it can carry:\n%s", fwd)
|
||||
}
|
||||
|
||||
// Under the default `block`, ESP/GRE are still dropped by policy: no
|
||||
// untunnelable accept appears, and the fail-closed drops remain.
|
||||
m.Globals.Untunnelable = ""
|
||||
rsBlock, err := RenderNft(m)
|
||||
if err != nil {
|
||||
t.Fatalf("RenderNft(block): %v", err)
|
||||
}
|
||||
fwdBlock := forwardChain(t, rsBlock)
|
||||
if strings.Contains(fwdBlock, "l4proto != { tcp, udp }") {
|
||||
t.Errorf("block policy must emit no untunnelable accept, L3 on or off:\n%s", fwdBlock)
|
||||
}
|
||||
if !strings.Contains(fwdBlock, "meta nfproto ipv4 drop") {
|
||||
t.Errorf("block policy lost the fail-closed drop with L3 on:\n%s", fwdBlock)
|
||||
}
|
||||
}
|
||||
|
||||
// TestL3IngressAbsentFromHoldingPlane: while the engine is down its TUN does not
|
||||
// exist, so the holding plane must not send anything at shater-l3. Marking ICMP
|
||||
// at a nonexistent device is not "ping keeps working" — it is a black hole that
|
||||
// LOOKS like a feature while the engine is in exactly the state the holding
|
||||
// plane exists to be honest about.
|
||||
func TestL3IngressAbsentFromHoldingPlane(t *testing.T) {
|
||||
m := realisticModel()
|
||||
m.Globals.L3Tunnel = true
|
||||
m.Globals.IPv6 = true
|
||||
rs, err := RenderHoldNft(m)
|
||||
if err != nil {
|
||||
t.Fatalf("RenderHoldNft: %v", err)
|
||||
}
|
||||
for _, s := range []string{L3Device, "0x2080"} {
|
||||
if strings.Contains(rs, s) {
|
||||
t.Errorf("holding plane must not reference the L3 ingress (found %q) — the engine is down and the TUN is gone:\n%s", s, rs)
|
||||
}
|
||||
}
|
||||
// The hold still fails closed for what the L3 ingress would have carried.
|
||||
if !strings.Contains(rs, "meta nfproto ipv4 drop") {
|
||||
t.Errorf("holding plane lost its fail-closed drop:\n%s", rs)
|
||||
}
|
||||
}
|
||||
|
||||
// --- untunnelable egress: the kernel carries what nothing else can ----------
|
||||
|
||||
// TestUntunnelableEgressOptIn: untunnelable_egress is opt-in, and with it unset
|
||||
// the render must be byte-for-byte what it was before the feature existed.
|
||||
// Every deployed router upgrades through this default; a marking line appearing
|
||||
// here would silently re-route ESP/GRE/ICMP on all of them without anyone
|
||||
// having asked for it.
|
||||
func TestUntunnelableEgressOptIn(t *testing.T) {
|
||||
m := realisticModel() // UntunnelableEgress deliberately left empty
|
||||
rs, err := RenderNft(m)
|
||||
if err != nil {
|
||||
t.Fatalf("RenderNft: %v", err)
|
||||
}
|
||||
// 0x2100 itself legitimately appears (the wwan egress's mgmt bypass); what
|
||||
// must not exist is anything STAMPING it, or any negated-set match in
|
||||
// prerouting — the policy's own negated set lives in the forward chain.
|
||||
if strings.Contains(rs, "meta mark set 0x2100") {
|
||||
t.Errorf("empty untunnelable_egress must not stamp the egress mark anywhere:\n%s", rs)
|
||||
}
|
||||
if pre := preroutingChain(t, rs); strings.Contains(pre, "l4proto != { tcp, udp }") {
|
||||
t.Errorf("empty untunnelable_egress must leave prerouting free of the negated set:\n%s", pre)
|
||||
}
|
||||
// Same under a permissive policy: `direct` emits its negated-set accept in
|
||||
// the FORWARD chain; prerouting marking must still not appear.
|
||||
m.Globals.Untunnelable = "direct"
|
||||
rsDirect, err := RenderNft(m)
|
||||
if err != nil {
|
||||
t.Fatalf("RenderNft(direct): %v", err)
|
||||
}
|
||||
if pre := preroutingChain(t, rsDirect); strings.Contains(pre, "l4proto != { tcp, udp }") {
|
||||
t.Errorf("the policy's accept belongs to forward; prerouting marking appeared without the option:\n%s", pre)
|
||||
}
|
||||
}
|
||||
|
||||
// TestUntunnelableEgressCarriesAllProtocols: with the option bound to an
|
||||
// interface egress, prerouting stamps the egress's OWN mark on the whole
|
||||
// non-TCP/UDP remainder and the forward chain accepts that mark ahead of the
|
||||
// fail-closed drops. Losing the prerouting line quietly returns ESP/GRE to the
|
||||
// untunnelable policy's verdict (typically: dropped); losing the forward accept
|
||||
// means the kernel routes the marked packet at the egress and the kill-switch
|
||||
// eats it on the way — the option reads configured and carries nothing.
|
||||
func TestUntunnelableEgressCarriesAllProtocols(t *testing.T) {
|
||||
m := realisticModel() // IPv6 off: the marking must stay v4-scoped
|
||||
m.Globals.UntunnelableEgress = "wwan"
|
||||
rs, err := RenderNft(m)
|
||||
if err != nil {
|
||||
t.Fatalf("RenderNft: %v", err)
|
||||
}
|
||||
pre := preroutingChain(t, rs)
|
||||
|
||||
mark := strings.Index(pre, "meta nfproto ipv4 meta l4proto != { tcp, udp } meta mark set 0x2100 accept")
|
||||
if mark < 0 {
|
||||
t.Fatalf("prerouting does not mark the non-TCP/UDP remainder for the egress — or is not v4-scoped "+
|
||||
"while IPv6 is off, in which case marked v6 finds no -6 rule, falls through to the main table "+
|
||||
"and leaves over the default WAN:\n%s", pre)
|
||||
}
|
||||
// Ingress-scoped, or WAN-ingress ESP/GRE gets marked and re-routed too.
|
||||
line := pre[strings.LastIndex(pre[:mark], "\n")+1 : mark]
|
||||
if !strings.Contains(line, "iifname") {
|
||||
t.Errorf("egress marking is unscoped (would also mark WAN ingress): %q", strings.TrimSpace(line))
|
||||
}
|
||||
// The mgmt bypass keeps an already-marked packet from being re-diverted.
|
||||
if !strings.Contains(pre, "meta mark 0x2100 accept") {
|
||||
t.Errorf("prerouting missing the egress-mark mgmt bypass:\n%s", pre)
|
||||
}
|
||||
|
||||
fwd := forwardChain(t, rs)
|
||||
acc := strings.Index(fwd, `meta mark 0x2100 oifname "wwan0" accept`)
|
||||
drop := strings.Index(fwd, "meta nfproto ipv4 drop")
|
||||
if acc < 0 {
|
||||
t.Fatalf("forward chain does not accept the egress mark: the kernel routes marked ESP/GRE at the egress and the kill-switch drops it on the way:\n%s", fwd)
|
||||
}
|
||||
if drop >= 0 && acc > drop {
|
||||
t.Errorf("the egress-mark accept must precede the fail-closed drop or it never matches:\n%s", fwd)
|
||||
}
|
||||
|
||||
// IPv6 on: the -6 rule/table half exists too, so the marking widens to the
|
||||
// family-neutral set.
|
||||
m.Globals.IPv6 = true
|
||||
rs6, err := RenderNft(m)
|
||||
if err != nil {
|
||||
t.Fatalf("RenderNft(ipv6): %v", err)
|
||||
}
|
||||
pre6 := preroutingChain(t, rs6)
|
||||
if !strings.Contains(pre6, " meta l4proto != { tcp, udp } meta mark set 0x2100 accept") ||
|
||||
strings.Contains(pre6, "meta nfproto ipv4 meta l4proto != { tcp, udp } meta mark set 0x2100") {
|
||||
t.Errorf("IPv6 on: the marking must cover both families — v6 ESP/GRE has a -6 rule and table to land in:\n%s", pre6)
|
||||
}
|
||||
|
||||
// The holding plane stays marking-free either way: while the engine is down
|
||||
// nothing guarantees the mark's `ip rule` was ever installed, and a mark
|
||||
// with no rule is a main-table fallthrough that the hold's own egress-mark
|
||||
// accept would wave out the default WAN.
|
||||
hold, err := RenderHoldNft(m)
|
||||
if err != nil {
|
||||
t.Fatalf("RenderHoldNft: %v", err)
|
||||
}
|
||||
if strings.Contains(hold, "meta mark set") {
|
||||
t.Errorf("holding plane must not stamp any mark — a mark without its routing is a kill-switch bypass:\n%s", hold)
|
||||
}
|
||||
}
|
||||
|
||||
// TestUntunnelableEgressFailClosedOnTypo: a name that resolves to no egress, or
|
||||
// to one the kernel cannot route by mark (byedpi/direct egresses have no device
|
||||
// and no mark routing), must change NOTHING — the untunnelable policy stays in
|
||||
// sole charge. Rendering the marking anyway would stamp packets with a mark no
|
||||
// `ip rule` serves: they fall through to the main table and leave over the
|
||||
// default WAN — a typo turned kill-switch bypass.
|
||||
func TestUntunnelableEgressFailClosedOnTypo(t *testing.T) {
|
||||
build := func(name string) *model.Model {
|
||||
m := realisticModel()
|
||||
m.Egresses = append(m.Egresses,
|
||||
model.Egress{Name: "dpi", Type: "byedpi"},
|
||||
model.Egress{Name: "straight", Type: "direct"},
|
||||
)
|
||||
m.Globals.UntunnelableEgress = name
|
||||
return m
|
||||
}
|
||||
baseline, err := RenderNft(build(""))
|
||||
if err != nil {
|
||||
t.Fatalf("RenderNft(baseline): %v", err)
|
||||
}
|
||||
for _, name := range []string{"wan2", "dpi", "straight"} {
|
||||
rs, err := RenderNft(build(name))
|
||||
if err != nil {
|
||||
t.Fatalf("RenderNft(%q): %v", name, err)
|
||||
}
|
||||
if rs != baseline {
|
||||
t.Errorf("untunnelable_egress=%q names no routable egress and must render byte-identical to the empty option:\n%s", name, rs)
|
||||
}
|
||||
}
|
||||
if fwd := forwardChain(t, baseline); !strings.Contains(fwd, "meta nfproto ipv4 drop") {
|
||||
t.Errorf("the fail-closed drop must survive an unresolvable option:\n%s", fwd)
|
||||
}
|
||||
}
|
||||
|
||||
// TestUntunnelableEgressEmptyInterfaceIsNotBrLan: an interface egress with NO
|
||||
// interface must not resolve. IfaceDevice("") falls back to br-lan, so without
|
||||
// the binding's own guard the LAN bridge would impersonate the egress device:
|
||||
// the marking renders, addEgressRouting (which skips on the same emptiness
|
||||
// check) never installs the mark's rule or table, marked ESP/GRE falls through
|
||||
// to the main table, and the forward chain accepts it by mark — out the default
|
||||
// WAN past a closed kill-switch.
|
||||
func TestUntunnelableEgressEmptyInterfaceIsNotBrLan(t *testing.T) {
|
||||
build := func(name string) *model.Model {
|
||||
m := realisticModel()
|
||||
m.Egresses = append(m.Egresses, model.Egress{Name: "hole", Type: "interface", Interface: ""})
|
||||
m.Globals.UntunnelableEgress = name
|
||||
return m
|
||||
}
|
||||
baseline, err := RenderNft(build(""))
|
||||
if err != nil {
|
||||
t.Fatalf("RenderNft(baseline): %v", err)
|
||||
}
|
||||
rs, err := RenderNft(build("hole"))
|
||||
if err != nil {
|
||||
t.Fatalf("RenderNft: %v", err)
|
||||
}
|
||||
if rs != baseline {
|
||||
t.Errorf("an egress without an interface has no device; rendering its marking routes ESP/GRE into the main table and out the default WAN:\n%s", rs)
|
||||
}
|
||||
}
|
||||
|
||||
// TestUntunnelableEgressYieldsICMPToL3: with both features on, the L3 ICMP mark
|
||||
// must STRICTLY precede the egress marking — nft evaluates first-match, so the
|
||||
// reverse order sends ICMP out the uplink instead of through the engine's
|
||||
// tunnel, and the L3 ingress is silently dead while both options read enabled.
|
||||
func TestUntunnelableEgressYieldsICMPToL3(t *testing.T) {
|
||||
m := realisticModel()
|
||||
m.Globals.L3Tunnel = true
|
||||
m.Globals.UntunnelableEgress = "wwan"
|
||||
rs, err := RenderNft(m)
|
||||
if err != nil {
|
||||
t.Fatalf("RenderNft: %v", err)
|
||||
}
|
||||
pre := preroutingChain(t, rs)
|
||||
icmp := strings.Index(pre, "ip protocol icmp meta mark set 0x2080 accept")
|
||||
wide := strings.Index(pre, "meta l4proto != { tcp, udp } meta mark set 0x2100 accept")
|
||||
if icmp < 0 {
|
||||
t.Fatalf("L3 lost its ICMP mark with the egress option on:\n%s", pre)
|
||||
}
|
||||
if wide < 0 {
|
||||
t.Fatalf("egress marking lost with L3 on:\n%s", pre)
|
||||
}
|
||||
if icmp > wide {
|
||||
t.Errorf("egress marking precedes the L3 ICMP mark: ICMP rides the uplink and the L3 ingress silently dies:\n%s", pre)
|
||||
}
|
||||
|
||||
// Same discipline for the ICMPv6 leg when IPv6 is on.
|
||||
m.Globals.IPv6 = true
|
||||
rs6, err := RenderNft(m)
|
||||
if err != nil {
|
||||
t.Fatalf("RenderNft(ipv6): %v", err)
|
||||
}
|
||||
pre6 := preroutingChain(t, rs6)
|
||||
v6 := strings.Index(pre6, "meta l4proto ipv6-icmp meta mark set 0x2080 accept")
|
||||
wide6 := strings.Index(pre6, "meta l4proto != { tcp, udp } meta mark set 0x2100 accept")
|
||||
if v6 < 0 || wide6 < 0 {
|
||||
t.Fatalf("IPv6 on: missing the ICMPv6 mark (%d) or the egress marking (%d):\n%s", v6, wide6, pre6)
|
||||
}
|
||||
if v6 > wide6 {
|
||||
t.Errorf("egress marking precedes the L3 ICMPv6 mark: ICMPv6 rides the uplink and the v6 half of the L3 ingress dies:\n%s", pre6)
|
||||
}
|
||||
}
|
||||
|
||||
// TestUntunnelableEgressExceptionsFirst: the local-plane accepts — router-local
|
||||
// (fib), private/link-local ranges, and the v6 ND/RA guard — must all precede
|
||||
// the egress marking. After it they are dead lines: pinging the gateway leaves
|
||||
// through the uplink, LAN-to-LAN traffic is re-routed out of the LAN, and one
|
||||
// re-routed neighbour solicitation takes LAN IPv6 down.
|
||||
func TestUntunnelableEgressExceptionsFirst(t *testing.T) {
|
||||
m := realisticModel()
|
||||
m.Globals.IPv6 = true // L3 off: the ND/RA guard must come from the egress block itself
|
||||
m.Globals.UntunnelableEgress = "wwan"
|
||||
rs, err := RenderNft(m)
|
||||
if err != nil {
|
||||
t.Fatalf("RenderNft: %v", err)
|
||||
}
|
||||
pre := preroutingChain(t, rs)
|
||||
wide := strings.Index(pre, "meta l4proto != { tcp, udp } meta mark set 0x2100 accept")
|
||||
if wide < 0 {
|
||||
t.Fatalf("egress marking missing:\n%s", pre)
|
||||
}
|
||||
for what, needle := range map[string]string{
|
||||
"router-local (fib) accept": "fib daddr type local accept",
|
||||
"private v4 ranges accept": "ip daddr { 10.0.0.0/8",
|
||||
"v6 local ranges accept": "ip6 daddr { ::1, fc00::/7, fe80::/10, ff00::/8 } accept",
|
||||
"ND/RA guard": "icmpv6 type { nd-router-solicit, nd-router-advert, nd-neighbor-solicit, nd-neighbor-advert } accept",
|
||||
} {
|
||||
at := strings.Index(pre, needle)
|
||||
if at < 0 {
|
||||
t.Errorf("%s is missing from prerouting, so that plane now leaves through the uplink:\n%s", what, pre)
|
||||
continue
|
||||
}
|
||||
if at > wide {
|
||||
t.Errorf("%s sits after the egress marking and is dead — that plane leaves through the uplink:\n%s", what, pre)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -12,7 +12,13 @@
|
||||
// CONTRACT: every type registered here is a type shater/generate emits
|
||||
// (see generate.go's package comment for the authoritative list):
|
||||
//
|
||||
// - inbounds: tproxy, redirect, direct (dokodemo), socks, http, mixed
|
||||
// - inbounds: tproxy, redirect, direct (dokodemo), socks, http, mixed, and
|
||||
// tun — the synthetic "l3-in" L3 ingress generate emits under globals
|
||||
// l3_tunnel, so non-TCP/UDP LAN traffic (ICMP echo) can reach an
|
||||
// L3-capable outbound instead of being faked or dropped. tun keeps the
|
||||
// slim promise: its gVisor stack is already linked because with_wireguard
|
||||
// requires with_gvisor (scripts/router-tags.sh), so registering it adds no
|
||||
// new build tag and no meaningful size.
|
||||
// - outbounds: direct, block, selector, urltest, socks, http, shadowsocks,
|
||||
// vmess, trojan, vless, shadowtls (+ hysteria2/tuic behind with_quic)
|
||||
// - endpoints: wireguard/AWG (behind with_wireguard)
|
||||
@@ -51,6 +57,7 @@ import (
|
||||
"github.com/sagernet/sing-box/protocol/shadowtls"
|
||||
"github.com/sagernet/sing-box/protocol/socks"
|
||||
"github.com/sagernet/sing-box/protocol/trojan"
|
||||
"github.com/sagernet/sing-box/protocol/tun"
|
||||
"github.com/sagernet/sing-box/protocol/vless"
|
||||
"github.com/sagernet/sing-box/protocol/vmess"
|
||||
)
|
||||
@@ -64,6 +71,7 @@ func Context(ctx context.Context) context.Context {
|
||||
func InboundRegistry() *inbound.Registry {
|
||||
registry := inbound.NewRegistry()
|
||||
|
||||
tun.RegisterInbound(registry)
|
||||
redirect.RegisterRedirect(registry)
|
||||
redirect.RegisterTProxy(registry)
|
||||
direct.RegisterInbound(registry)
|
||||
|
||||
Reference in New Issue
Block a user