Compare commits
10
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
3c7536dba0 | ||
|
|
3c92e1cbfd | ||
|
|
bcc9df9282 | ||
|
|
ee3641fe45 | ||
|
|
439f62238f | ||
|
|
d971eb85ee | ||
|
|
a0f6083e28 | ||
|
|
77369aedfe | ||
|
|
515ae6d1b7 | ||
|
|
a2ffbb1292 |
@@ -18,6 +18,32 @@ type URLTestOutboundOptions struct {
|
||||
// lx: SPEC 019 v2 — load-balancing.
|
||||
Mode string `json:"mode,omitempty"` // least_test (default) | round_robin
|
||||
Balancer *URLTestBalancerOptions `json:"balancer,omitempty"`
|
||||
// lx: health board §5.C — SelfCheck stands the group's OWN background
|
||||
// health-check up or down. nil/absent == true, so every existing config keeps
|
||||
// today's behaviour.
|
||||
//
|
||||
// Why this exists at all: a urltest group probes its members BY ITSELF — a
|
||||
// warm-up sweep at PostStart and a ticker for as long as traffic keeps
|
||||
// touching it — and it dials the members' outbounds DIRECTLY, from the
|
||||
// router, over whatever the default WAN route is. For a group that traffic
|
||||
// actually flows through, that is exactly right: the probe travels the same
|
||||
// path the connections do. But for a group NO routing rule reaches, that
|
||||
// same probe measures a path nothing uses — and it stores the result under
|
||||
// the members' BASE tags, which every health consumer then reads as "the
|
||||
// node's health". A node that is blocked on the direct WAN and perfectly
|
||||
// alive behind a tunnel therefore reads "dead" the moment such a group
|
||||
// probes it; the reading is not merely stale, it is FALSE, and it poisons
|
||||
// the shared board for everyone (selection, the panel, the observatory's
|
||||
// freshness gate). SelfCheck=false is how the control plane stands such a
|
||||
// group's own schedule down: the shater engine computes which groups the
|
||||
// applied rules actually reach (the observatory's used-set) and disables
|
||||
// the self-check on the rest, so the ONLY prober left is the observatory —
|
||||
// which probes along the real dial paths and nothing else.
|
||||
//
|
||||
// The flag suppresses only the group's own SCHEDULE (the PostStart warm-up
|
||||
// and the Touch ticker). An EXPLICIT CheckOutbounds/URLTest call — the
|
||||
// adapter interface a human or an API invokes on purpose — still works.
|
||||
SelfCheck *bool `json:"self_check,omitempty"`
|
||||
}
|
||||
|
||||
// URLTestBalancerOptions configures round_robin: a fixed-size pool of live nodes, lazily
|
||||
|
||||
+136
-16
@@ -390,19 +390,92 @@ export interface GroupHealth {
|
||||
* any more and nothing to report here beyond the groups themselves.
|
||||
*/
|
||||
|
||||
/**
|
||||
* One hop of one chain, measured where that hop actually sits in the path.
|
||||
*
|
||||
* This is the reading the daemon always took and never showed. A chain is not a
|
||||
* target with a single health — it is an ordered series of them, and the only
|
||||
* question an operator ever asks about a broken chain is WHICH hop broke. The
|
||||
* end-to-end exit reading cannot answer that: it says "the path is dead" for a
|
||||
* four-hop chain and leaves the person to guess between four suspects.
|
||||
*
|
||||
* WIRE ORDER. `index` is 1-based and counts hops in the order the router dials
|
||||
* them: hop 1 is the first physical hop, and each later hop is dialled THROUGH
|
||||
* the ones before it. The hop carrying `exit: true` — always the largest index —
|
||||
* is where traffic leaves for the internet. A leading `egress:` in the chain's
|
||||
* configured Hops is NOT a numbered hop: the daemon lifts it into the entry
|
||||
* detour of hop 1, so a chain written `egress:ewan → node:awgout → group:sub0`
|
||||
* reports two hops, not three. Anything zipping this against the model's Hops
|
||||
* must drop that leading egress first and give up on labelling entirely if the
|
||||
* counts still disagree — a chain that splices sub-chains gets flattened here,
|
||||
* and a confidently WRONG hop name is worse than no name.
|
||||
*
|
||||
* `tag` is the engine-side outbound (`chain-<name>-h2`). Debugging and tooltips
|
||||
* only; it is never a label to put in front of a person.
|
||||
*
|
||||
* NODE HOP vs GROUP HOP. For `kind: "node"` the hop IS the measurement: `total`
|
||||
* is 1, the counters follow its own state, and `selected` is ''. For
|
||||
* `kind: "group"` the counters roll up that hop's per-hop member COPIES — the
|
||||
* copies dialled through the hops in front of it, which is exactly why they can
|
||||
* read alive here while the same group's standalone card reads dead. Both
|
||||
* readings are true; they measure different dial paths. `selected` is the node
|
||||
* NAME the hop routes through right now, and `delay_ms` / `age_seconds` belong
|
||||
* to that selected member (or the freshest alive one).
|
||||
*
|
||||
* Invariants the daemon guarantees — never re-derive them, just read them:
|
||||
* `tested === alive + dead` and `alive + dead + untested === total`.
|
||||
*
|
||||
* `state` is a closed set. `untested` is NEVER "dead" and never "healthy": it
|
||||
* means nothing fresh enough is known, which for a used chain is seconds away
|
||||
* from resolving on its own. `age_seconds: -1` means the age is unknown.
|
||||
*/
|
||||
export interface ChainHopHealth {
|
||||
/** 1-based WIRE order. Hop 1 is dialled first; see the note above. */
|
||||
index: number
|
||||
/** Engine outbound tag (`chain-<name>-h2`) — tooltips/debugging, never a label. */
|
||||
tag: string
|
||||
/** `node` ⇒ the hop is the measurement. `group` ⇒ the counters roll up members. */
|
||||
kind: 'node' | 'group'
|
||||
/** This hop is where traffic leaves for the internet. Always the largest index. */
|
||||
exit: boolean
|
||||
/** Closed set — switch on it exhaustively. `untested` is never "dead". */
|
||||
state: 'alive' | 'dead' | 'untested'
|
||||
/** RTT of the selected/freshest alive member; 0 (meaningless) when not alive. */
|
||||
delay_ms: number
|
||||
/** Age of that measurement in seconds; -1 when unknown. */
|
||||
age_seconds: number
|
||||
/** Node name this GROUP hop routes through right now; '' for a node hop. */
|
||||
selected: string
|
||||
total: number
|
||||
tested: number
|
||||
alive: number
|
||||
dead: number
|
||||
untested: number
|
||||
}
|
||||
|
||||
/** Per-chain reachability, the chain analogue of {@link GroupHealth}.used (plan
|
||||
* §5.E): a chain no enabled routing rule routes through is outside the
|
||||
* observatory's plan, so its exit is never probed and the Targets card renders it
|
||||
* "unused" instead of an exit-test readout. A chain has no membership counters —
|
||||
* it is a fixed path, and its end-to-end health is the exit test's job. */
|
||||
* observatory's plan, so nothing probes it and the Targets card says so instead
|
||||
* of rendering a health reading. A chain has no membership counters of its own —
|
||||
* it is a fixed path, and its health lives on its {@link ChainHopHealth} hops. */
|
||||
export interface ChainHealth {
|
||||
name: string
|
||||
/** An enabled routing rule (the Final target, a DNS-resolver detour, a device
|
||||
* target, …) reaches this chain, so the observatory probes its exit in the
|
||||
* target, …) reaches this chain, so the observatory probes its hops in the
|
||||
* background. false ⇒ nothing routes through the chain: it is skipped by the
|
||||
* background probing and its end-to-end health stays untested. That is an
|
||||
* "unused" note about the ROUTING CONFIG, never a health problem. */
|
||||
* background probing and its health stays untested. That is an "unused" note
|
||||
* about the ROUTING CONFIG, never a health problem. */
|
||||
used: boolean
|
||||
/**
|
||||
* Per-hop health in wire order (see {@link ChainHopHealth}).
|
||||
*
|
||||
* MAY BE ABSENT, and absent does not mean "this chain has no hops". It means
|
||||
* the engine never materialised per-hop outbounds for it: the chain is unused,
|
||||
* or it collapses to a single hop and the daemon points traffic straight at
|
||||
* that target instead of building a copy of it. Read a missing key as "nothing
|
||||
* measured per hop", never as an empty path or as a fault.
|
||||
*/
|
||||
hops?: ChainHopHealth[]
|
||||
}
|
||||
|
||||
export interface GroupsHealth {
|
||||
@@ -1262,6 +1335,17 @@ export interface RuleReach {
|
||||
shadowed_by_order?: number
|
||||
/** Operator-facing sentence; absent when `unreachable` is false. */
|
||||
reason?: string
|
||||
/**
|
||||
* Whether the rule is IN FORCE right now — `Rule.Enabled` after the active WAN
|
||||
* profile's overrides. This is NOT `GET /api/config`'s `Enabled`: that one is
|
||||
* the desired state the page PUTs back, and on a router with profiles the two
|
||||
* legitimately disagree. Draw rows from this; keep the switch on the other.
|
||||
*/
|
||||
effective_enabled: boolean
|
||||
/** The active profile that CHANGED this rule's state; absent when none did. */
|
||||
overridden_by?: string
|
||||
/** Which way it went. Absent together with `overridden_by`. */
|
||||
override?: 'enabled' | 'disabled'
|
||||
}
|
||||
|
||||
/** GET /api/rules/reachability. `rules` is ALWAYS an array, one entry per rule in
|
||||
@@ -1421,8 +1505,21 @@ export function getGroupsHealth(
|
||||
}
|
||||
|
||||
/**
|
||||
* One group's (or chain's) last test: which member the balancer picked, how fast
|
||||
* it answered, and what the internet saw as the source address.
|
||||
* What the OBSERVATORY measured for one group or chain — not a dial the panel
|
||||
* made.
|
||||
*
|
||||
* This shape used to come from a fresh connection opened on demand, straight at
|
||||
* the target. That was a lie on any router whose proxies are blocked when dialled
|
||||
* directly and work only as a hop behind a tunnel: the card reported dead for a
|
||||
* path that carries traffic all day. The daemon now has exactly one thing that
|
||||
* measures — the background observatory, which probes along the REAL dial path,
|
||||
* per-hop copies and all — and this endpoint reports what it found. There is no
|
||||
* second measurement anywhere, and the panel never opens a connection of its own.
|
||||
*
|
||||
* So read the fields as a READ, not as a test run: `ok` and `delay_ms` are the
|
||||
* observatory's verdict for the path traffic actually takes, and `tested_unix`
|
||||
* (router clock, seconds) is when the OBSERVATORY took that measurement — which
|
||||
* can be a few seconds before the refresh was asked for.
|
||||
*
|
||||
* `ok:true` with an EMPTY `exit_ip`/`exit_country` is a valid, successful result,
|
||||
* not a partial failure: the delay was measured but the exit address could not be
|
||||
@@ -1431,10 +1528,23 @@ export function getGroupsHealth(
|
||||
*
|
||||
* Chains ride the same endpoint. For a chain row, `group` carries the CHAIN's
|
||||
* name and `selected` the node its last group hop picked ('' when the exit hop
|
||||
* isn't a group). Everything else reads the same way.
|
||||
* isn't a group). Per-hop detail is a different read: {@link ChainHopHealth}.
|
||||
*
|
||||
* `ok:false` ⇒ the test failed and `error` carries the human reason; every other
|
||||
* field is meaningless. `tested_unix` is the router's clock, in seconds.
|
||||
* `ok:false` ⇒ there is no usable measurement and `error` carries the human
|
||||
* reason; every other field is meaningless. Four of those reasons are about the
|
||||
* observatory rather than the path, and must not be rendered as "your target is
|
||||
* broken":
|
||||
*
|
||||
* "not routed by any enabled rule, so nothing measures it — the observatory
|
||||
* only probes paths the rules use"
|
||||
* "the observatory has not reached this target yet — it refreshes on the
|
||||
* global probe interval"
|
||||
* "background probing is disabled, so there is nothing to measure this target
|
||||
* with"
|
||||
* "the observatory's probe through this path failed"
|
||||
*
|
||||
* Only the last one is a health finding. The first three say the measurement
|
||||
* does not exist, which is a different thing and a different fix.
|
||||
*/
|
||||
export interface GroupTestResult {
|
||||
group: string // group name — or a chain name for a chain row
|
||||
@@ -1451,7 +1561,11 @@ export interface GroupTestResult {
|
||||
* GET /api/groups/test — progress plus every result so far. `results` is ALWAYS
|
||||
* an array (never null); `done`/`total` count finished vs targeted groups and
|
||||
* chains while `running` is true. Idle reads `{running:false}` with the last
|
||||
* run's results still attached, so a reload after a test still shows what it found.
|
||||
* run's results still attached, so a reload still shows what was last read.
|
||||
*
|
||||
* "Running" means the observatory is working through an out-of-turn refresh pass
|
||||
* over the named targets and this endpoint is collecting what it measures. It is
|
||||
* not the panel dialling anything.
|
||||
*/
|
||||
export interface GroupTestStatus {
|
||||
running: boolean
|
||||
@@ -1484,10 +1598,16 @@ export interface GroupTestStart {
|
||||
}
|
||||
|
||||
/**
|
||||
* POST /api/groups/test — measure a target's delay and exit address. Pass a
|
||||
* group or chain name to test one; pass nothing (or '') to test every group
|
||||
* and every chain. Singleton: a second call while a run is in flight resolves
|
||||
* to `{started:false, reason:'already running'}` rather than failing.
|
||||
* POST /api/groups/test — ask the observatory for an out-of-turn refresh pass,
|
||||
* then report what it measured. Pass a group or chain name to refresh one; pass
|
||||
* nothing (or '') for every group and every chain.
|
||||
*
|
||||
* It does NOT dial. The observatory is the only thing in the daemon that
|
||||
* measures anything, and it measures along the real dial path — so this is the
|
||||
* "don't wait for the next probe interval" button, not a second opinion. The
|
||||
* numbers it returns are the same numbers the cards are already showing, just
|
||||
* fresher. Singleton: a second call while a pass is in flight resolves to
|
||||
* `{started:false, reason:'already running'}` rather than failing.
|
||||
*/
|
||||
export function postGroupsTest(name = ''): Promise<GroupTestStart> {
|
||||
return MOCK
|
||||
|
||||
+170
-33
@@ -6,7 +6,7 @@
|
||||
// state mutates in-memory so the Apply / Confirm / Rollback flow is exercisable.
|
||||
//
|
||||
// Type-only imports from api.ts (erased at build) keep this free of a runtime cycle.
|
||||
import type { ApplyResult, ChainHealth, ConnLogEntry, DiscoveredDevice, GroupHealth, GroupMemberHealth, GroupsHealth, GroupTestResult, GroupTestStart, GroupTestStatus, Interface, Model, QueryLogEntry, RuleReach, RulesReachability, RulesetCategories, RulesetCheck, RulesetStatus, Stats, StatsLogPage, StatsLogQuery, Status, StatusWarning, Traffic } from './api'
|
||||
import type { ApplyResult, ChainHealth, ChainHopHealth, ConnLogEntry, DiscoveredDevice, GroupHealth, GroupMemberHealth, GroupsHealth, GroupTestResult, GroupTestStart, GroupTestStatus, Interface, Model, Profile, QueryLogEntry, RuleReach, RulesReachability, RulesetCategories, RulesetCheck, RulesetStatus, Stats, StatsLogPage, StatsLogQuery, Status, StatusWarning, Traffic } from './api'
|
||||
|
||||
let armed = false // a pending commit-confirm auto-rollback
|
||||
let hasLastGood = false // a predecessor config exists to roll back to (post-apply)
|
||||
@@ -129,9 +129,25 @@ const CONFIG: Model = {
|
||||
{ Name: 'via-tunnel', Source: 'subscription', Subscription: 'primary', Strategy: 'leastping', Egress: 'awg' },
|
||||
{ Name: 'fallback', Source: 'subscription', Subscription: 'backup', Strategy: 'roundrobin', Egress: '' },
|
||||
],
|
||||
// One multi-hop chain so `?mock` exercises the chain card's Test button and
|
||||
// its result readout: enters through the awg tunnel, exits via the auto group.
|
||||
Chains: [{ Name: 'relay', Hops: ['egress:awg', 'group:auto'] }],
|
||||
// Three chains, one per state the hop readout has to render.
|
||||
Chains: [
|
||||
// The owner's real production shape: leave through a WAN interface, cross an
|
||||
// AmneziaWG node, then three subscription groups in series. The leading
|
||||
// `egress:` is NOT a numbered hop — the daemon lifts it into hop 1's entry
|
||||
// detour — so this reports FOUR hops, and hop 3 is dead while its neighbours
|
||||
// answer. That single red notch in the middle of a live path is the entire
|
||||
// reason per-hop health exists, so `?mock` must show it at a glance.
|
||||
{
|
||||
Name: 'ewan-wg-subs',
|
||||
Hops: ['egress:wan', 'node:home-wg', 'group:auto', 'group:stealth', 'group:via-tunnel'],
|
||||
},
|
||||
// Used, but the observatory hasn't come round yet — every hop untested. Not
|
||||
// dead and not healthy: the state the panel most easily renders as a fault.
|
||||
{ Name: 'sub-fresh', Hops: ['node:home-wg', 'group:fallback'] },
|
||||
// No enabled rule targets it, so the observatory skips it entirely and the
|
||||
// daemon never materialises its hops: `used:false` and NO `hops` key.
|
||||
{ Name: 'relay', Hops: ['egress:awg', 'group:auto'] },
|
||||
],
|
||||
Egresses: [
|
||||
{ Name: 'wan', Type: 'interface', Interface: 'wan' },
|
||||
// An AmneziaWG tunnel — the whole point of a group-level egress binding.
|
||||
@@ -145,6 +161,13 @@ const CONFIG: Model = {
|
||||
Rules: [
|
||||
{ Name: 'block-ads', Enabled: true, Order: 10, DstRuleset: ['ad-hosts'], Target: 'block' },
|
||||
{ Name: 'ru-bypass', Enabled: true, Order: 20, DstRuleset: ['ru-inside'], Target: 'direct' },
|
||||
// These two are what make the chains USED — the observatory probes only the
|
||||
// paths an enabled rule can reach, so without them every chain card would
|
||||
// read "not routed" and the hop rail would never appear in `?mock`. Kept
|
||||
// ABOVE the condition-less rule at Order 40, which would otherwise swallow
|
||||
// everything below it and mark them "never applies".
|
||||
{ Name: 'media-via-chain', Enabled: true, Order: 22, DstRuleset: ['yt-geosite'], Target: 'chain:ewan-wg-subs' },
|
||||
{ Name: 'spare-via-chain', Enabled: true, Order: 24, DstRuleset: ['ad-hosts'], Target: 'chain:sub-fresh' },
|
||||
{ Name: 'private-direct', Enabled: true, Order: 30, DstRuleset: ['private-nets'], Target: 'direct' },
|
||||
// A SECOND condition-less rule, above the real default. It reads like a working
|
||||
// rule and does nothing: a rule with no conditions becomes the router's default,
|
||||
@@ -305,16 +328,61 @@ const RULESET_STATUS: RulesetStatus[] = [
|
||||
/** GET /api/rules/reachability. Mirrors the daemon's analysis over CONFIG.Rules:
|
||||
* a rule with no conditions is the router's default, and the LAST such rule by
|
||||
* Order wins — every earlier one can never apply. It reads the live CONFIG so
|
||||
* edits made in `?mock` keep the badge honest. */
|
||||
* edits made in `?mock` keep the badge honest.
|
||||
*
|
||||
* It also mirrors model.ResolveActiveProfile + ApplyProfileRuleOverrides, because
|
||||
* `effective_enabled` is the whole point of the endpoint: CONFIG's `mobile-uplink`
|
||||
* is active and both enables and disables rules, so `?mock` shows the same
|
||||
* desired-vs-effective split the field config does. */
|
||||
export async function getRulesReachability(): Promise<RulesReachability> {
|
||||
await wait(60)
|
||||
const rules = CONFIG.Rules ?? []
|
||||
|
||||
// A pin naming an existing, ENABLED profile wins outright. Otherwise auto-select:
|
||||
// highest Priority among enabled profiles, ties by Name, skipping any with an
|
||||
// iface condition (the WAN watcher owns those and expresses its verdict as the pin).
|
||||
const profiles = CONFIG.Profiles ?? []
|
||||
const pinned = String(CONFIG.Globals?.ActiveProfile ?? '').trim()
|
||||
let prof: Profile | null = profiles.find((p) => p.Enabled && p.Name === pinned) ?? null
|
||||
if (!prof) {
|
||||
for (const p of profiles) {
|
||||
if (!p.Enabled || (p.MatchIface ?? []).length > 0) continue
|
||||
const pp = p.Priority ?? 0
|
||||
const bp = prof?.Priority ?? 0
|
||||
if (!prof || pp > bp || (pp === bp && p.Name < prof.Name)) prof = p
|
||||
}
|
||||
}
|
||||
// Enable first, then Disable, so a name in both ends up disabled (Disable wins).
|
||||
const effective = rules.map((r) => Boolean(r.Enabled))
|
||||
if (prof) {
|
||||
const force = (names: string[] | null | undefined, on: boolean) => {
|
||||
for (const raw of names ?? []) {
|
||||
const n = raw.trim()
|
||||
rules.forEach((r, i) => {
|
||||
if (r.Name === n) effective[i] = on
|
||||
})
|
||||
}
|
||||
}
|
||||
force(prof.EnableRules, true)
|
||||
force(prof.DisableRules, false)
|
||||
}
|
||||
const activeProfile = prof
|
||||
|
||||
const out: RuleReach[] = rules.map((r, index) => ({
|
||||
index,
|
||||
name: String(r.Name ?? ''),
|
||||
order: Number(r.Order ?? 0),
|
||||
unreachable: false,
|
||||
shadowed_by_index: -1,
|
||||
effective_enabled: effective[index],
|
||||
// Annotate only where the profile actually FLIPPED the outcome — a profile that
|
||||
// disables an already-off rule has overridden nothing the operator can see.
|
||||
...(activeProfile && effective[index] !== Boolean(r.Enabled)
|
||||
? {
|
||||
overridden_by: activeProfile.Name,
|
||||
override: effective[index] ? ('enabled' as const) : ('disabled' as const),
|
||||
}
|
||||
: {}),
|
||||
}))
|
||||
const conditionless = (r: (typeof rules)[number]): boolean =>
|
||||
!(r.Src ?? []).length &&
|
||||
@@ -325,7 +393,10 @@ export async function getRulesReachability(): Promise<RulesReachability> {
|
||||
String(r.Target ?? '').trim() || (r.Egress ? `egress:${String(r.Egress).trim()}` : '')
|
||||
const defaults = rules
|
||||
.map((r, index) => ({ r, index }))
|
||||
.filter(({ r }) => r.Enabled && conditionless(r) && target(r))
|
||||
// The EFFECTIVE flag, not the configured one: a rule the active profile
|
||||
// switched off is not in force and cannot retire anything (model's
|
||||
// RuleReachability runs over the effective set for the same reason).
|
||||
.filter(({ r, index }) => effective[index] && conditionless(r) && target(r))
|
||||
.sort((a, b) => Number(a.r.Order ?? 0) - Number(b.r.Order ?? 0) || a.index - b.index)
|
||||
const winner = defaults[defaults.length - 1]
|
||||
if (winner) {
|
||||
@@ -1124,14 +1195,47 @@ function healthList(): GroupHealth[] {
|
||||
return (CONFIG.Groups ?? []).map((g) => summarise(g.Name, GROUP_MEMBERS.get(g.Name) ?? []))
|
||||
}
|
||||
|
||||
/** Per-chain reachability for the Targets page's "unused" badge (plan §5.E) — the
|
||||
* chain analogue of healthList's `used` field. The mock's single chain `relay` is
|
||||
* NOT referenced by any rule in CONFIG.Rules (they target group:auto / block /
|
||||
* direct), so it reads used=false and its card renders "unused" — exactly the case
|
||||
* the badge exists to surface. A stopped engine reports no chains. */
|
||||
/**
|
||||
* Per-hop health, keyed by chain name — what the observatory measured at each
|
||||
* position of the path, in WIRE order.
|
||||
*
|
||||
* `ewan-wg-subs` is the fixture that matters. Its hop 1 is the AmneziaWG node and
|
||||
* answers; hops 2 and 4 are subscription groups that answer THROUGH it; hop 3 is
|
||||
* a group whose members all time out at that position. Note that hop 4 rolls up
|
||||
* `via-tunnel`, the same group whose standalone card reads 0 of 6 alive — alive
|
||||
* as a hop, dead on its own, both true, because they measure different dial
|
||||
* paths. That contradiction is the whole point of measuring per hop.
|
||||
*
|
||||
* `sub-fresh` is used but never yet reached: every hop untested, nothing dead.
|
||||
* `relay` is absent from this map on purpose — an unused chain is never
|
||||
* materialised, so the daemon sends no `hops` key at all, which is "nothing
|
||||
* measured", not "no hops".
|
||||
*/
|
||||
const CHAIN_HOPS: Record<string, ChainHopHealth[]> = {
|
||||
'ewan-wg-subs': [
|
||||
{ index: 1, tag: 'chain-ewan-wg-subs-h1', kind: 'node', exit: false, state: 'alive', delay_ms: 41, age_seconds: 22, selected: '', total: 1, tested: 1, alive: 1, dead: 0, untested: 0 },
|
||||
{ index: 2, tag: 'chain-ewan-wg-subs-h2', kind: 'group', exit: false, state: 'alive', delay_ms: 96, age_seconds: 18, selected: '🇳🇱 Amsterdam-01', total: 298, tested: 122, alive: 119, dead: 3, untested: 176 },
|
||||
{ index: 3, tag: 'chain-ewan-wg-subs-h3', kind: 'group', exit: false, state: 'dead', delay_ms: 0, age_seconds: 15, selected: '', total: 24, tested: 24, alive: 0, dead: 24, untested: 0 },
|
||||
{ index: 4, tag: 'chain-ewan-wg-subs-h4', kind: 'group', exit: true, state: 'alive', delay_ms: 148, age_seconds: 19, selected: '🇸🇬 Singapore-09', total: 6, tested: 6, alive: 6, dead: 0, untested: 0 },
|
||||
],
|
||||
'sub-fresh': [
|
||||
{ index: 1, tag: 'chain-sub-fresh-h1', kind: 'node', exit: false, state: 'untested', delay_ms: 0, age_seconds: -1, selected: '', total: 1, tested: 0, alive: 0, dead: 0, untested: 1 },
|
||||
{ index: 2, tag: 'chain-sub-fresh-h2', kind: 'group', exit: true, state: 'untested', delay_ms: 0, age_seconds: -1, selected: '', total: 24, tested: 0, alive: 0, dead: 0, untested: 24 },
|
||||
],
|
||||
}
|
||||
|
||||
/** Per-chain reachability plus per-hop health for the Targets page. `used` is the
|
||||
* chain analogue of healthList's field; `hops` is OMITTED (never null, never []),
|
||||
* exactly like the daemon, for a chain the engine never materialised. A stopped
|
||||
* engine reports no chains at all. */
|
||||
function chainHealthList(): ChainHealth[] {
|
||||
if (!mockPlane().engine) return []
|
||||
return (CONFIG.Chains ?? []).map((c) => ({ name: c.Name, used: chainUsed(c.Name) }))
|
||||
return (CONFIG.Chains ?? []).map((c) => {
|
||||
const hops = CHAIN_HOPS[c.Name]
|
||||
const h: ChainHealth = { name: c.Name, used: chainUsed(c.Name) }
|
||||
if (hops) h.hops = hops.map((x) => ({ ...x }))
|
||||
return h
|
||||
})
|
||||
}
|
||||
|
||||
/** A chain is "used" when some enabled routing rule (or Final, or a DNS detour)
|
||||
@@ -1189,15 +1293,23 @@ class ApiErrorLike extends Error {
|
||||
}
|
||||
}
|
||||
|
||||
// Mock group/chain test. Deliberately covers every state the UI has to render,
|
||||
// one per target, so a single offline run exercises all of them:
|
||||
// auto → ok WITH an exit address
|
||||
// stealth → ok WITHOUT one (delay measured, address undeterminable) — a
|
||||
// SUCCESS, and the case the UI most easily gets wrong
|
||||
// relay → the chain: same wire shape, `group` carries the CHAIN's name and
|
||||
// `selected` the node its exit group picked
|
||||
// fallback → a failure carrying a human reason
|
||||
// Results land one per GET poll, so the running/progress state is visible too.
|
||||
// Mock refresh results. The endpoint no longer dials anything: it asks the
|
||||
// observatory to measure out of turn and reports what the observatory found, so
|
||||
// every row here is a READ of a background measurement. Deliberately covers every
|
||||
// state the UI has to render, one per target, so a single offline run exercises
|
||||
// all of them:
|
||||
// auto → ok WITH an exit address
|
||||
// stealth → ok WITHOUT one (delay measured, address undeterminable) — a
|
||||
// SUCCESS, and the case the UI most easily gets wrong
|
||||
// ewan-wg-subs → the chain: same wire shape, `group` carries the CHAIN's name
|
||||
// and `selected` the node its exit hop picked
|
||||
// via-tunnel → the one honest health FAILURE: a probe that ran and failed
|
||||
// fallback,
|
||||
// relay → not routed at all, so no measurement exists to report
|
||||
// sub-fresh → routed, but the observatory hasn't come round yet
|
||||
// The last three are absence of measurement, not a broken target, and the copy
|
||||
// has to keep them apart. Results land one per GET poll, so the running/progress
|
||||
// state is visible too.
|
||||
const GROUP_TEST_SHAPE: Record<string, Omit<GroupTestResult, 'group' | 'tested_unix'>> = {
|
||||
auto: {
|
||||
selected: 'nl-reality-2',
|
||||
@@ -1215,32 +1327,57 @@ const GROUP_TEST_SHAPE: Record<string, Omit<GroupTestResult, 'group' | 'tested_u
|
||||
ok: true,
|
||||
error: '',
|
||||
},
|
||||
// The chain — Selected is the node the chain's exit group (auto) picked.
|
||||
relay: {
|
||||
selected: 'nl-reality-2',
|
||||
delay_ms: 61,
|
||||
exit_ip: '185.12.34.56',
|
||||
exit_country: 'NL',
|
||||
ok: true,
|
||||
error: '',
|
||||
// The chain, and the pairing that makes the whole feature worth building. A
|
||||
// chain is one series path, so with hop 3 dead the end-to-end probe CANNOT
|
||||
// succeed — this row and the hop rail have to tell one story, not two. The
|
||||
// row says the path is down; the rail says which of the four hops did it,
|
||||
// which is the part nobody could see before.
|
||||
'ewan-wg-subs': {
|
||||
selected: '',
|
||||
delay_ms: 0,
|
||||
exit_ip: '',
|
||||
exit_country: '',
|
||||
ok: false,
|
||||
error: 'the observatory’s probe through this path failed',
|
||||
},
|
||||
// Dead through its tunnel, exactly as its membership health says — the exit
|
||||
// test and the member health tell the same story about the same group.
|
||||
// The one real health failure in the fixture: the observatory's probe ran along
|
||||
// this path and did not come back.
|
||||
'via-tunnel': {
|
||||
selected: '',
|
||||
delay_ms: 0,
|
||||
exit_ip: '',
|
||||
exit_country: '',
|
||||
ok: false,
|
||||
error: 'no member answered through egress awg (6 of 6 timed out)',
|
||||
error: 'the observatory’s probe through this path failed',
|
||||
},
|
||||
// Not a health verdict — nothing routes here, so no measurement of it exists.
|
||||
fallback: {
|
||||
selected: '',
|
||||
delay_ms: 0,
|
||||
exit_ip: '',
|
||||
exit_country: '',
|
||||
ok: false,
|
||||
error: 'no reachable node in the group (all 3 members timed out)',
|
||||
error:
|
||||
'not routed by any enabled rule, so nothing measures it — the observatory only probes paths the rules use',
|
||||
},
|
||||
relay: {
|
||||
selected: '',
|
||||
delay_ms: 0,
|
||||
exit_ip: '',
|
||||
exit_country: '',
|
||||
ok: false,
|
||||
error:
|
||||
'not routed by any enabled rule, so nothing measures it — the observatory only probes paths the rules use',
|
||||
},
|
||||
// Routed, materialised, simply not reached yet. Untested is not dead.
|
||||
'sub-fresh': {
|
||||
selected: '',
|
||||
delay_ms: 0,
|
||||
exit_ip: '',
|
||||
exit_country: '',
|
||||
ok: false,
|
||||
error:
|
||||
'the observatory has not reached this target yet — it refreshes on the global probe interval',
|
||||
},
|
||||
}
|
||||
|
||||
|
||||
@@ -58,6 +58,45 @@
|
||||
color: var(--ink);
|
||||
}
|
||||
|
||||
/* ---- active-profile banner ----
|
||||
*
|
||||
* Deliberately NOT the accent plate the save→apply bar wears above. Orange is
|
||||
* "there is something for you to do" on this faceplate, and an active WAN profile
|
||||
* is a standing condition, not a pending action. A quiet plate with an amber tag
|
||||
* reads as "note the state" — and it is the SAME amber the overridden rows below
|
||||
* carry, so the banner and its rows are visibly one story rather than two
|
||||
* unrelated oddities. */
|
||||
.rt-prof-banner {
|
||||
display: flex;
|
||||
align-items: flex-start;
|
||||
gap: 10px;
|
||||
margin: 0 0 calc(var(--u, 8px) * 2.5);
|
||||
padding: 10px 14px;
|
||||
border: 1px solid color-mix(in srgb, var(--amber) 35%, var(--groove));
|
||||
border-radius: 8px;
|
||||
background: color-mix(in srgb, var(--amber) 7%, transparent);
|
||||
font-family: var(--font-sans);
|
||||
font-size: 12px;
|
||||
line-height: 1.55;
|
||||
color: var(--dim);
|
||||
}
|
||||
.rt-prof-banner strong {
|
||||
color: var(--ink);
|
||||
font-weight: 600;
|
||||
}
|
||||
.rt-prof-tag {
|
||||
flex: none;
|
||||
margin-top: 1px;
|
||||
padding: 2px 7px;
|
||||
border: 1px solid color-mix(in srgb, var(--amber) 55%, var(--groove));
|
||||
border-radius: 999px;
|
||||
background: color-mix(in srgb, var(--amber) 12%, transparent);
|
||||
font-size: 9px;
|
||||
letter-spacing: var(--track-label);
|
||||
text-transform: uppercase;
|
||||
color: var(--amber);
|
||||
}
|
||||
|
||||
/* ---- empty state ---- */
|
||||
.rt-empty {
|
||||
padding: calc(var(--u, 8px) * 4) 0 calc(var(--u, 8px) * 3);
|
||||
@@ -289,6 +328,51 @@
|
||||
opacity: 0.62;
|
||||
}
|
||||
|
||||
/* ---- a rule the active WAN profile overrides ----
|
||||
*
|
||||
* The row itself needs no new paint: an overridden-off rule already wears `.off`
|
||||
* (it is off, whatever its switch says) and an overridden-on rule wears nothing
|
||||
* (it is on). What was missing was never colour — it was the sentence naming who
|
||||
* decided. So this is the per-row twin of the banner and borrows .rt-dead-note's
|
||||
* type wholesale: same voice, same size, one <p> margin to reset. */
|
||||
/* The same pill as .rt-badge.dead, so the two override states read as one pair,
|
||||
* but in accent — a rule the profile forces ON is active, and active is orange on
|
||||
* this faceplate. The pill is also what keeps it from running into the plain
|
||||
* "default route · final" badge beside it, where "final on · by profile" read as
|
||||
* one phrase. */
|
||||
.rt-badge.prof-on {
|
||||
padding: 1px 7px;
|
||||
border: 1px solid var(--accent-soft);
|
||||
border-radius: 999px;
|
||||
background: color-mix(in srgb, var(--accent) 10%, transparent);
|
||||
}
|
||||
|
||||
.rt-prof-note {
|
||||
margin: 0;
|
||||
}
|
||||
.rt-prof-note strong {
|
||||
color: var(--ink);
|
||||
font-weight: 600;
|
||||
}
|
||||
|
||||
/* Switch + its legend. The caption shows ONLY while a profile overrides the rule,
|
||||
* and it is what keeps the control honest: the plate says what the router is
|
||||
* doing, this says the switch is about the saved setting. A legend under the
|
||||
* control it names is the faceplate's own idiom. */
|
||||
.rt-switch {
|
||||
display: inline-flex;
|
||||
flex-direction: column;
|
||||
align-items: center;
|
||||
gap: 3px;
|
||||
}
|
||||
.rt-switch-note {
|
||||
font-family: var(--font-mono);
|
||||
font-size: 8.5px;
|
||||
letter-spacing: var(--track-label);
|
||||
text-transform: uppercase;
|
||||
color: var(--faint);
|
||||
}
|
||||
|
||||
/* ---- target chip (styled like the artifact's group:auto mono chips) ---- */
|
||||
.rt-target {
|
||||
display: inline-flex;
|
||||
|
||||
+142
-25
@@ -43,6 +43,21 @@ type RRule = Rule & {
|
||||
LegacyDst?: string[] | null
|
||||
}
|
||||
|
||||
/**
|
||||
* Whether a rule is in force, kept strictly apart from whether it is switched on.
|
||||
*
|
||||
* `on` is the EFFECTIVE state — what the router is actually doing — and every mark
|
||||
* on the row is drawn from it. `profile`/`dir` are set only when the active WAN
|
||||
* profile is the reason the two differ, so the row can name who overrode the
|
||||
* saved setting instead of leaving the operator to guess why a switch that reads
|
||||
* "on" routes nothing.
|
||||
*/
|
||||
type RuleForce = {
|
||||
on: boolean
|
||||
profile: string | null
|
||||
dir: 'enabled' | 'disabled' | null
|
||||
}
|
||||
|
||||
/**
|
||||
* Everything `Proto` can match, and nothing else. The engine understands two
|
||||
* transports and exactly ten application protocols its sniffers can name
|
||||
@@ -484,7 +499,7 @@ export default function Routing() {
|
||||
}, [config])
|
||||
|
||||
/**
|
||||
* The verdict for one rule, or null when it can fire.
|
||||
* The daemon's verdict for one rule, or null when we have none that describes it.
|
||||
*
|
||||
* Verdicts are fetched separately from the config, so between an optimistic edit
|
||||
* and the refetch they can describe the PREVIOUS rule list. Re-checking the
|
||||
@@ -492,18 +507,53 @@ export default function Routing() {
|
||||
* badge on a working rule: a mismatch means the verdict is not about this row,
|
||||
* and no badge is the honest answer.
|
||||
*/
|
||||
const shadowOf = useCallback(
|
||||
(r: RRule): { by: string; byOrder: number; reason: string } | null => {
|
||||
const verdictOf = useCallback(
|
||||
(r: RRule): RuleReach | null => {
|
||||
const i = modelIndex.get(r)
|
||||
if (i === undefined) return null
|
||||
const v = reach.get(i)
|
||||
if (!v || !v.unreachable || !v.shadowed_by) return null
|
||||
if (v.name !== r.Name || v.order !== r.Order) return null
|
||||
return { by: v.shadowed_by, byOrder: v.shadowed_by_order ?? 0, reason: v.reason ?? '' }
|
||||
if (!v || v.name !== r.Name || v.order !== r.Order) return null
|
||||
return v
|
||||
},
|
||||
[modelIndex, reach],
|
||||
)
|
||||
|
||||
const shadowOf = useCallback(
|
||||
(r: RRule): { by: string; byOrder: number; reason: string } | null => {
|
||||
const v = verdictOf(r)
|
||||
if (!v || !v.unreachable || !v.shadowed_by) return null
|
||||
return { by: v.shadowed_by, byOrder: v.shadowed_by_order ?? 0, reason: v.reason ?? '' }
|
||||
},
|
||||
[verdictOf],
|
||||
)
|
||||
|
||||
/**
|
||||
* Whether a rule is IN FORCE, and who decided that — the two states this page
|
||||
* used to conflate.
|
||||
*
|
||||
* `Rule.Enabled` from /api/config is the DESIRED state: what the operator saved,
|
||||
* what the switch edits, what gets PUT back. The active WAN profile can override
|
||||
* it in either direction, and then the desired state is no longer what the router
|
||||
* is doing. Drawing the row from `Enabled` is what let a config with two rules
|
||||
* `enabled '1'` show two live switches while the engine ran one chain.
|
||||
*
|
||||
* With no verdict — an older daemon, a stopped one, or one still describing the
|
||||
* previous config — the desired state is all we know, so the row falls back to it
|
||||
* and claims no profile rather than inventing one. The `typeof` guard is for the
|
||||
* older daemon specifically: it answers without `effective_enabled` at all, and
|
||||
* reading `undefined` as false would gray out every rule on the page.
|
||||
*/
|
||||
const forceOf = useCallback(
|
||||
(r: RRule): RuleForce => {
|
||||
const v = verdictOf(r)
|
||||
if (!v || typeof v.effective_enabled !== 'boolean') {
|
||||
return { on: !!r.Enabled, profile: null, dir: null }
|
||||
}
|
||||
return { on: v.effective_enabled, profile: v.overridden_by ?? null, dir: v.override ?? null }
|
||||
},
|
||||
[verdictOf],
|
||||
)
|
||||
|
||||
// Rulesets are named domain/IP lists rules match against (rule.DstRuleset).
|
||||
const rulesets = useMemo<Ruleset[]>(
|
||||
() => [...((config?.Rulesets as Ruleset[] | null | undefined) ?? [])],
|
||||
@@ -815,7 +865,14 @@ export default function Routing() {
|
||||
)
|
||||
}
|
||||
|
||||
const enabledCount = rules.filter((r) => r.Enabled).length
|
||||
// One force verdict per displayed rule, computed once and handed down — the row,
|
||||
// the counter and the banner must all be reading the SAME answer.
|
||||
const force = rules.map((r) => forceOf(r))
|
||||
// EFFECTIVE, not configured. A counter that added up saved switches said "2 / 2
|
||||
// active" for a config the router was running one rule of.
|
||||
const enabledCount = force.filter((f) => f.on).length
|
||||
const overridden = force.filter((f) => f.profile !== null)
|
||||
const overrideProfile = overridden[0]?.profile ?? null
|
||||
|
||||
return (
|
||||
<section className="page" aria-label="Routing rules">
|
||||
@@ -824,12 +881,28 @@ export default function Routing() {
|
||||
Rules run top to bottom on the bus — the <strong>first match wins</strong>. Traffic that
|
||||
reaches the bottom follows the default route.
|
||||
</p>
|
||||
<span className="rt-count mono" aria-label={`${enabledCount} of ${rules.length} rules active`}>
|
||||
<span className="rt-count mono" aria-label={`${enabledCount} of ${rules.length} rules in force`}>
|
||||
{enabledCount}
|
||||
<small> / {rules.length} active</small>
|
||||
<small> / {rules.length} in force</small>
|
||||
</span>
|
||||
</div>
|
||||
|
||||
{/* Said once at the top, so the per-row badges below read as consequences of
|
||||
one thing rather than as N unrelated oddities. Only shown when a profile
|
||||
actually changed something: a router that uses no profiles, or one whose
|
||||
profile agrees with every saved switch, gets no banner at all. */}
|
||||
{overrideProfile && (
|
||||
<p className="rt-prof-banner" role="status">
|
||||
<span className="rt-prof-tag mono">profile</span>
|
||||
<span>
|
||||
<strong className="mono">{overrideProfile}</strong> is the active WAN profile and is
|
||||
overriding {overridden.length === 1 ? '1 rule' : `${overridden.length} rules`} below. The
|
||||
switches keep showing what you saved; the rows show what the router is running. Change
|
||||
which rules a profile forces on the <strong>Profiles</strong> page.
|
||||
</span>
|
||||
</p>
|
||||
)}
|
||||
|
||||
{actionError && (
|
||||
<p className="page-error" role="alert">
|
||||
{actionError}
|
||||
@@ -877,6 +950,7 @@ export default function Routing() {
|
||||
busy={saving}
|
||||
editingOther={editingRule !== null}
|
||||
shadow={shadowOf(r)}
|
||||
force={force[i]}
|
||||
onEdit={onEditRule}
|
||||
onToggle={onToggle}
|
||||
onMove={onMove}
|
||||
@@ -927,6 +1001,7 @@ function RuleRow({
|
||||
busy,
|
||||
editingOther,
|
||||
shadow,
|
||||
force,
|
||||
onEdit,
|
||||
onToggle,
|
||||
onMove,
|
||||
@@ -939,6 +1014,9 @@ function RuleRow({
|
||||
editingOther: boolean
|
||||
/** Set when the daemon reports this rule can never fire; null when it can. */
|
||||
shadow: { by: string; byOrder: number; reason: string } | null
|
||||
/** Whether the rule is IN FORCE, and which profile decided that (see RuleForce).
|
||||
* Every mark on this row comes from here; `rule.Enabled` drives only the switch. */
|
||||
force: RuleForce
|
||||
onEdit: (name: string) => void
|
||||
onToggle: (name: string) => void
|
||||
onMove: (name: string, dir: 'up' | 'down') => void
|
||||
@@ -966,14 +1044,14 @@ function RuleRow({
|
||||
const isDefault = isCatchAll(rule) && !dead
|
||||
const target = effectiveTarget(rule)
|
||||
const tone = targetTone(target)
|
||||
const cls = [
|
||||
'rt-rule',
|
||||
rule.Enabled ? '' : 'off',
|
||||
isDefault ? 'final' : '',
|
||||
inert ? 'dead' : '',
|
||||
]
|
||||
// Dimmed by the EFFECTIVE state, never by the saved one. A rule the active
|
||||
// profile switched off is not in force, and the row has to read that way even
|
||||
// though its switch — which edits the saved setting — is still on.
|
||||
const cls = ['rt-rule', force.on ? '' : 'off', isDefault ? 'final' : '', inert ? 'dead' : '']
|
||||
.filter(Boolean)
|
||||
.join(' ')
|
||||
// What the switch says, spelled out, for the moment the two disagree.
|
||||
const savedState = rule.Enabled ? 'on' : 'off'
|
||||
|
||||
return (
|
||||
<li className={cls}>
|
||||
@@ -1007,6 +1085,11 @@ function RuleRow({
|
||||
{isDefault && <span className="rt-badge">default route · final</span>}
|
||||
{unmigrated && <span className="rt-badge dead">held off · not migrated</span>}
|
||||
{dead && <span className="rt-badge dead">never applies</span>}
|
||||
{/* Amber for the rule the profile switched OFF (warn semantics: wired but
|
||||
not connected), accent for the one it switched ON — orange is the
|
||||
faceplate's active state, and a force-enabled rule is exactly that. */}
|
||||
{force.dir === 'disabled' && <span className="rt-badge dead">off · by profile</span>}
|
||||
{force.dir === 'enabled' && <span className="rt-badge prof-on">on · by profile</span>}
|
||||
</div>
|
||||
<div className="rt-match">
|
||||
{unmigrated ? (
|
||||
@@ -1034,6 +1117,22 @@ function RuleRow({
|
||||
<Matchers rule={rule} />
|
||||
)}
|
||||
</div>
|
||||
{/* Added BELOW the matchers, not instead of them: the rule's conditions are
|
||||
still worth reading — the operator is deciding whether to change the
|
||||
profile or the rule.
|
||||
Two clauses only. The banner at the top of the page already carries the
|
||||
general explanation and the way to change it, and a profile that
|
||||
overrides several rules would otherwise repeat that paragraph on every
|
||||
one of them. What is left is the part only this row can say: whether it
|
||||
is in force, and what its own switch is showing instead. */}
|
||||
{force.profile && (
|
||||
<p className="rt-dead-note rt-prof-note">
|
||||
{force.dir === 'disabled' ? 'Not in force' : 'In force'} — profile{' '}
|
||||
<strong className="mono">{force.profile}</strong> switches this rule{' '}
|
||||
{force.dir === 'disabled' ? 'off' : 'on'}. The switch still reads{' '}
|
||||
<strong>{savedState}</strong>: that is the saved setting.
|
||||
</p>
|
||||
)}
|
||||
</div>
|
||||
|
||||
<div className={`rt-target ${tone}`} title={`target: ${target}`}>
|
||||
@@ -1059,16 +1158,34 @@ function RuleRow({
|
||||
emits neither), so enabling it here would save a live rule with no
|
||||
destination left at all — the catch-all this tripwire exists to
|
||||
prevent. `shaterd migrate` clears LegacyDst and the switch comes back. */}
|
||||
<Toggle
|
||||
pressed={rule.Enabled}
|
||||
onChange={() => onToggle(rule.Name)}
|
||||
label={
|
||||
unmigrated
|
||||
? `Rule ${rule.Name} is held disabled until the config is migrated`
|
||||
: `${rule.Enabled ? 'Disable' : 'Enable'} rule ${rule.Name}`
|
||||
}
|
||||
disabled={frozen || unmigrated}
|
||||
/>
|
||||
{/* The switch edits the SAVED setting and nothing else, so it keeps showing
|
||||
rule.Enabled even while the active profile forces the opposite. Mirroring
|
||||
the effective state here would be worse than the bug it replaces: the
|
||||
operator would flip a switch that was never theirs, and the PUT would
|
||||
write the profile's decision into UCI as if they had chosen it. The row
|
||||
above says what the router is doing; the "saved" caption says what this
|
||||
control is for. */}
|
||||
<span className="rt-switch">
|
||||
<Toggle
|
||||
pressed={rule.Enabled}
|
||||
onChange={() => onToggle(rule.Name)}
|
||||
label={
|
||||
unmigrated
|
||||
? `Rule ${rule.Name} is held disabled until the config is migrated`
|
||||
: force.profile
|
||||
? `Saved setting for rule ${rule.Name} is ${savedState}; profile ${force.profile} is forcing it ${
|
||||
force.dir === 'disabled' ? 'off' : 'on'
|
||||
}. This switch changes the saved setting only.`
|
||||
: `${rule.Enabled ? 'Disable' : 'Enable'} rule ${rule.Name}`
|
||||
}
|
||||
disabled={frozen || unmigrated}
|
||||
/>
|
||||
{force.profile && (
|
||||
<span className="rt-switch-note" aria-hidden="true">
|
||||
saved
|
||||
</span>
|
||||
)}
|
||||
</span>
|
||||
<button
|
||||
type="button"
|
||||
className="rt-del"
|
||||
|
||||
@@ -704,6 +704,13 @@
|
||||
.tg-test--bad .tg-test-msg {
|
||||
color: var(--crit);
|
||||
}
|
||||
/* "Nothing measured this" is not a failure and must never be dressed as one: an
|
||||
unlit lamp and the faintest text on the card, the same register the group
|
||||
readout uses for its unmeasured state. */
|
||||
.tg-test--none .tg-test-msg {
|
||||
font-family: var(--font-sans);
|
||||
color: var(--faint);
|
||||
}
|
||||
.tg-test--wait .tg-test-msg {
|
||||
color: var(--amber);
|
||||
}
|
||||
@@ -1038,6 +1045,221 @@
|
||||
}
|
||||
|
||||
/* ---- responsive ---- */
|
||||
/* ---- chain hop rail ----
|
||||
* The chain section's signature, and the one place this card spends any
|
||||
* boldness: the path is drawn as a CONDUCTOR with a numbered lamp at each hop,
|
||||
* and the conductor is SEVERED below the first hop that was probed and did not
|
||||
* answer. A chain is a single series path, so the question is never "how many
|
||||
* hops are green", it is "where does my traffic stop" — and a broken line answers
|
||||
* that before a word has been read.
|
||||
*
|
||||
* The two marks carry two different facts and must not be conflated:
|
||||
* - the LAMP is that hop's own measurement (good / warn / crit / unlit), and a
|
||||
* hop below the break that answered keeps its green, because it did answer;
|
||||
* - the CONDUCTOR is reachability through the path, which really does stop.
|
||||
*
|
||||
* Orange is untouched here. Semantics carry every colour, and everything that is
|
||||
* not a lamp is groove-grey. No transitions and no animation anywhere in the
|
||||
* rail, so there is nothing for reduced-motion to switch off. */
|
||||
.ch-rail {
|
||||
gap: 8px;
|
||||
}
|
||||
.ch-eyebrow {
|
||||
font-family: var(--font-mono);
|
||||
font-size: 9px;
|
||||
letter-spacing: var(--track-label);
|
||||
text-transform: uppercase;
|
||||
color: var(--faint);
|
||||
}
|
||||
.ch-hops {
|
||||
--ch-num: 1.8ch; /* the engraved hop number's gutter */
|
||||
--ch-gap: 8px;
|
||||
--ch-led: 10px; /* must match .led's width */
|
||||
--ch-lampy: 14px; /* row top → lamp centre; the conductor's anchor */
|
||||
/* x of the conductor: the number gutter, one gap, then the lamp's centre */
|
||||
--ch-spine: calc(var(--ch-num) + var(--ch-gap) + var(--ch-led) / 2);
|
||||
list-style: none;
|
||||
margin: 0;
|
||||
padding: 0;
|
||||
}
|
||||
.ch-hop {
|
||||
position: relative;
|
||||
display: grid;
|
||||
grid-template-columns: var(--ch-num) var(--ch-led) minmax(0, 1fr);
|
||||
column-gap: var(--ch-gap);
|
||||
align-items: start;
|
||||
}
|
||||
|
||||
/* the conductor — two halves per row, so a break lands on one link only */
|
||||
.ch-hop::before,
|
||||
.ch-hop::after {
|
||||
content: '';
|
||||
position: absolute;
|
||||
left: var(--ch-spine);
|
||||
width: 2px;
|
||||
margin-left: -1px;
|
||||
/* Brighter than a plain groove: this line IS the readout, and at groove
|
||||
strength it disappeared into the panel and took the whole idea with it. */
|
||||
background: color-mix(in srgb, var(--dim) 55%, var(--groove));
|
||||
}
|
||||
.ch-hop::before {
|
||||
top: 0;
|
||||
height: calc(var(--ch-lampy) - var(--ch-led) / 2 - 3px);
|
||||
}
|
||||
.ch-hop::after {
|
||||
top: calc(var(--ch-lampy) + var(--ch-led) / 2 + 3px);
|
||||
bottom: 0;
|
||||
}
|
||||
/* Nothing feeds hop 1 from above, and nothing leaves the exit downward — the
|
||||
path starts and ends inside this rail. */
|
||||
.ch-hop--first::before {
|
||||
display: none;
|
||||
}
|
||||
.ch-hop--exit::after {
|
||||
bottom: auto;
|
||||
height: 9px;
|
||||
}
|
||||
/* …the exit ends on a crossbar instead of trailing off: end of line. */
|
||||
.ch-hop--exit .ch-socket {
|
||||
position: relative;
|
||||
}
|
||||
.ch-hop--exit .ch-socket::after {
|
||||
content: '';
|
||||
position: absolute;
|
||||
left: 50%;
|
||||
transform: translateX(-50%);
|
||||
top: calc(var(--ch-lampy) + var(--ch-led) / 2 + 12px);
|
||||
width: 11px;
|
||||
height: 2px;
|
||||
background: color-mix(in srgb, var(--dim) 55%, var(--groove));
|
||||
}
|
||||
|
||||
/* THE SEVER. Everything from the dead hop's outgoing link downward is drawn as a
|
||||
broken conductor: unmistakably not-a-line at a glance, and unmistakably not a
|
||||
colour, because a colour here would compete with the lamps that carry health. */
|
||||
.ch-hop--dead::after,
|
||||
.ch-hop--severed::before,
|
||||
.ch-hop--severed::after {
|
||||
background: repeating-linear-gradient(
|
||||
to bottom,
|
||||
color-mix(in srgb, var(--dim) 45%, var(--groove)) 0 3px,
|
||||
transparent 3px 7px
|
||||
);
|
||||
}
|
||||
|
||||
.ch-num {
|
||||
font-size: 10px;
|
||||
line-height: calc(var(--ch-lampy) * 2);
|
||||
text-align: right;
|
||||
color: var(--faint);
|
||||
}
|
||||
.ch-socket {
|
||||
display: flex;
|
||||
align-items: center;
|
||||
height: calc(var(--ch-lampy) * 2);
|
||||
}
|
||||
.ch-body {
|
||||
min-width: 0;
|
||||
/* Separates one hop from the next. The conductor runs through this space, so
|
||||
too little of it and two hops read as one wrapped row. */
|
||||
padding-bottom: 8px;
|
||||
}
|
||||
.ch-l1 {
|
||||
display: flex;
|
||||
align-items: center;
|
||||
flex-wrap: wrap;
|
||||
gap: 8px;
|
||||
min-height: calc(var(--ch-lampy) * 2);
|
||||
}
|
||||
.ch-name {
|
||||
font-size: 12px;
|
||||
color: var(--ink);
|
||||
overflow-wrap: anywhere;
|
||||
}
|
||||
.ch-hop--untested .ch-name {
|
||||
color: var(--dim);
|
||||
}
|
||||
/* The exit marker is NEUTRAL on purpose. The config path above this rail tags its
|
||||
exit green, which is free there — but in here green means "answering", and a
|
||||
green badge on the last hop would read as a health claim about it. */
|
||||
.ch-tag {
|
||||
font-family: var(--font-mono);
|
||||
font-size: 8.5px;
|
||||
letter-spacing: 0.14em;
|
||||
text-transform: uppercase;
|
||||
color: var(--faint);
|
||||
}
|
||||
.ch-delay {
|
||||
font-size: 11.5px;
|
||||
font-weight: 700;
|
||||
color: var(--ink);
|
||||
}
|
||||
.ch-quiet {
|
||||
font-family: var(--font-sans);
|
||||
font-size: 12px;
|
||||
color: var(--faint);
|
||||
}
|
||||
.ch-age {
|
||||
margin-left: auto;
|
||||
font-size: 10.5px;
|
||||
color: var(--faint);
|
||||
white-space: nowrap;
|
||||
}
|
||||
|
||||
/* The counters read exactly as they do on a group card — alive out of TESTED,
|
||||
with the untested remainder as a quiet aside only when there is one. Same
|
||||
register, same weights, deliberately not a second dialect. */
|
||||
.ch-l2 {
|
||||
display: flex;
|
||||
align-items: center;
|
||||
flex-wrap: wrap;
|
||||
gap: 8px;
|
||||
margin-top: 1px;
|
||||
font-size: 11.5px;
|
||||
letter-spacing: 0.02em;
|
||||
color: var(--dim);
|
||||
}
|
||||
.ch-count {
|
||||
font-size: 12px;
|
||||
color: var(--dim);
|
||||
white-space: nowrap;
|
||||
}
|
||||
.ch-count b {
|
||||
font-size: 14px;
|
||||
font-weight: 700;
|
||||
color: var(--ink);
|
||||
}
|
||||
.ch-hop--dead .ch-count b {
|
||||
color: var(--crit);
|
||||
}
|
||||
.ch-word {
|
||||
font-size: 10.5px;
|
||||
letter-spacing: 0.12em;
|
||||
text-transform: uppercase;
|
||||
color: var(--faint);
|
||||
}
|
||||
.ch-dead {
|
||||
padding: 1px 6px;
|
||||
border-radius: 4px;
|
||||
background: color-mix(in srgb, var(--crit) 12%, transparent);
|
||||
font-size: 10.5px;
|
||||
color: var(--crit);
|
||||
white-space: nowrap;
|
||||
cursor: help;
|
||||
}
|
||||
.ch-rest {
|
||||
font-size: 10.5px;
|
||||
color: var(--faint);
|
||||
}
|
||||
.ch-now {
|
||||
max-width: 28ch;
|
||||
overflow: hidden;
|
||||
text-overflow: ellipsis;
|
||||
white-space: nowrap;
|
||||
font-size: 10.5px;
|
||||
color: var(--dim);
|
||||
}
|
||||
|
||||
@media (max-width: 640px) {
|
||||
.tg-sec-hd {
|
||||
flex-wrap: wrap;
|
||||
@@ -1060,6 +1282,20 @@
|
||||
.gh-now {
|
||||
max-width: 100%;
|
||||
}
|
||||
/* On a phone the age stamp stops being pushed to a lonely right edge and just
|
||||
joins the end of the hop's line; the selected node gets the full width
|
||||
instead of an ellipsis it doesn't need. */
|
||||
.ch-age {
|
||||
margin-left: 0;
|
||||
}
|
||||
.ch-now {
|
||||
max-width: 100%;
|
||||
}
|
||||
/* Every field of a hop wraps onto its own line at this width, so the gap
|
||||
between hops has to grow with them or the rail reads as one block of text. */
|
||||
.ch-body {
|
||||
padding-bottom: 12px;
|
||||
}
|
||||
.gh-mems {
|
||||
max-height: 260px;
|
||||
}
|
||||
|
||||
+375
-83
@@ -22,6 +22,8 @@ import type {
|
||||
GroupTestResult,
|
||||
GroupTestStatus,
|
||||
Chain,
|
||||
ChainHealth,
|
||||
ChainHopHealth,
|
||||
Egress,
|
||||
Interface,
|
||||
Node,
|
||||
@@ -436,14 +438,14 @@ const normalizeTest = (st: GroupTestStatus): GroupTestStatus => ({
|
||||
})
|
||||
|
||||
/**
|
||||
* How the header names the reach of a running exit test. The name matters more
|
||||
* How the header names the reach of a running refresh pass. The name matters more
|
||||
* than the number when there is only one: "auto" tells the operator which button
|
||||
* they pressed; "1 target" tells them nothing they didn't already know.
|
||||
*/
|
||||
function scopeLabel(scope: string[], targetCount: number): string {
|
||||
if (scope.length === 1) return scope[0]
|
||||
if (scope.length === 0) return 'exits' // pre-scope daemon — say nothing false
|
||||
return scope.length >= targetCount ? 'every exit' : `${scope.length} exits`
|
||||
if (scope.length === 0) return 'targets' // pre-scope daemon — say nothing false
|
||||
return scope.length >= targetCount ? 'every target' : `${scope.length} targets`
|
||||
}
|
||||
|
||||
/** Which editor (add or edit-by-name) is open within a section. */
|
||||
@@ -590,10 +592,12 @@ export default function Targets() {
|
||||
[health],
|
||||
)
|
||||
|
||||
// ---- group/chain exit test: how fast, through which node, out which address --
|
||||
// The POST only kicks a run off, and a 2 s poll of the GET carries progress
|
||||
// plus every result so far. One endpoint covers groups and chains alike:
|
||||
// POST with a group or chain name tests that one; an empty name tests them all.
|
||||
// ---- out-of-turn refresh: how fast, through which node, out which address ----
|
||||
// The POST does NOT dial. It asks the observatory — the only thing in the daemon
|
||||
// that measures anything, and it measures along the real dial path — to come
|
||||
// round out of turn; a 2 s poll of the GET carries progress plus every reading
|
||||
// so far. One endpoint covers groups and chains alike: POST with a name refreshes
|
||||
// that one, an empty name refreshes them all.
|
||||
const [gtest, setGtest] = useState<GroupTestStatus>(IDLE_TEST)
|
||||
const [gtestErr, setGtestErr] = useState<string | null>(null)
|
||||
const [polling, setPolling] = useState(false)
|
||||
@@ -636,7 +640,7 @@ export default function Targets() {
|
||||
void readTest().then((st) => {
|
||||
if (!alive || !st || st.running) return
|
||||
setPolling(false)
|
||||
flash('Group test complete')
|
||||
flash('Readings refreshed')
|
||||
})
|
||||
}, 2000)
|
||||
return () => {
|
||||
@@ -652,16 +656,16 @@ export default function Targets() {
|
||||
if (r.started) {
|
||||
setGtestErr(null)
|
||||
setPolling(true)
|
||||
flash(name ? `Testing ${name}…` : 'Testing every exit…')
|
||||
flash(name ? `Refreshing ${name}…` : 'Refreshing every reading…')
|
||||
void readTest()
|
||||
} else if (r.reason === 'already running') {
|
||||
setPolling(true) // pick up the run someone else started
|
||||
flash('A group test is already running')
|
||||
setPolling(true) // pick up the pass someone else started
|
||||
flash('The prober is already refreshing')
|
||||
} else {
|
||||
flash(`Couldn’t start the test — ${r.reason || 'the daemon refused it'}`)
|
||||
flash(`Couldn’t ask for a refresh — ${r.reason || 'the daemon refused it'}`)
|
||||
}
|
||||
} catch (e) {
|
||||
flash(`Couldn’t start the test — ${errText(e)}`)
|
||||
flash(`Couldn’t ask for a refresh — ${errText(e)}`)
|
||||
}
|
||||
},
|
||||
[flash, readTest],
|
||||
@@ -910,10 +914,11 @@ export default function Targets() {
|
||||
<h2 className="tg-sec-title">Groups</h2>
|
||||
<span className="tg-sec-count mono">{groups.length} configured</span>
|
||||
{/* The observatory's background probing is invisible by design — it
|
||||
keeps every used group's and chain's numbers fresh on its own. The
|
||||
one manual run left is the exit test: it is scoped to the groups
|
||||
and chains it names, so its progress says WHICH, and its badge
|
||||
lands only on those cards. */}
|
||||
keeps every used group's and chain's numbers fresh on its own, along
|
||||
the path traffic actually takes. The one manual control left does
|
||||
not measure anything itself: it asks that prober to come round out
|
||||
of turn. It is scoped to the groups and chains it names, so its
|
||||
progress says WHICH, and its badge lands only on those cards. */}
|
||||
<div className="tg-sec-ctl">
|
||||
{groupHealthOn && (
|
||||
<>
|
||||
@@ -921,11 +926,11 @@ export default function Targets() {
|
||||
<span
|
||||
className="tg-run tg-run--exit"
|
||||
role="status"
|
||||
title="An exit test sends one connection through each group or chain it covers and reports the delay and the address the internet sees."
|
||||
title="The background prober is measuring the targets this refresh covers, along the path each one's traffic really takes."
|
||||
>
|
||||
<Led variant="amber" pulse />
|
||||
<span className="tg-run-what">
|
||||
exit test · {scopeLabel(asArray(gtest.scope), groups.length + chains.length)}
|
||||
refreshing · {scopeLabel(asArray(gtest.scope), groups.length + chains.length)}
|
||||
</span>
|
||||
<span className="tg-run-n mono">
|
||||
{gtest.done}/{gtest.total}
|
||||
@@ -935,9 +940,9 @@ export default function Targets() {
|
||||
<Button
|
||||
onClick={() => void runTest()}
|
||||
disabled={busy || !config || (groups.length === 0 && chains.length === 0) || gtest.running}
|
||||
title="Send one connection through each group and chain and report the delay and the exit address the internet sees"
|
||||
title="Ask the background prober to measure every group and chain out of turn, then show what it measured. The panel opens no connection of its own."
|
||||
>
|
||||
{gtest.running ? 'Testing…' : 'Test every exit'}
|
||||
{gtest.running ? 'Refreshing…' : 'Refresh every reading'}
|
||||
</Button>
|
||||
</>
|
||||
)}
|
||||
@@ -957,6 +962,13 @@ export default function Targets() {
|
||||
dials out through a tunnel measures them through that tunnel, so the same node can be alive
|
||||
in one group and dead in another.
|
||||
</p>
|
||||
<p className="tg-sec-note">
|
||||
One thing measures, and the panel is not it. A background prober walks every path your
|
||||
rules use — hop by hop, exactly as traffic goes — and every number on this page is a read
|
||||
of what it found. <strong>Refresh every reading</strong> asks it to come round out of turn
|
||||
instead of waiting for the next pass; it opens no connection of its own, so a target no
|
||||
rule routes through has nothing to report and says so.
|
||||
</p>
|
||||
|
||||
{groupHealthOn && healthErr && (
|
||||
<p className="tg-test-err" role="alert">
|
||||
@@ -967,7 +979,7 @@ export default function Targets() {
|
||||
|
||||
{groupHealthOn && gtestErr && (
|
||||
<p className="tg-test-err" role="alert">
|
||||
Couldn’t read the test results — {gtestErr}.{' '}
|
||||
Couldn’t read the refreshed numbers — {gtestErr}.{' '}
|
||||
<button className="linkish" onClick={() => void readTest()}>
|
||||
Retry
|
||||
</button>
|
||||
@@ -1110,7 +1122,9 @@ export default function Targets() {
|
||||
chain={c}
|
||||
busy={busy}
|
||||
showHealth={groupHealthOn}
|
||||
used={healthByChain.get(c.Name)?.used}
|
||||
// The whole chain health record, not just `.used` — the card
|
||||
// renders the observatory's per-hop measurements from it.
|
||||
health={healthByChain.get(c.Name)}
|
||||
test={testByGroup.get(c.Name)}
|
||||
// The badge is this card's business only when the run names it.
|
||||
testing={gtest.running && testScope.has(c.Name)}
|
||||
@@ -1232,7 +1246,7 @@ function GroupRow({
|
||||
group: Group
|
||||
busy: boolean
|
||||
/** Group health checks are on (Settings). When false, the card drops its health
|
||||
* readout, its exit-test readout and its Test button — it is config only. */
|
||||
* readout, its end-to-end reading and its Refresh button — it is config only. */
|
||||
showHealth: boolean
|
||||
/** This group's membership health, or undefined when the engine hasn't built
|
||||
* it (not applied yet, or dropped for having no usable members). */
|
||||
@@ -1242,14 +1256,14 @@ function GroupRow({
|
||||
healthKnown: boolean
|
||||
test?: GroupTestResult
|
||||
/**
|
||||
* A group exit test covering THIS group is in flight.
|
||||
* A refresh pass covering THIS group is in flight.
|
||||
*
|
||||
* Deliberately not "a test is running": the caller resolves it against the run's
|
||||
* scope. There is no per-card equivalent for the health run — that one measures
|
||||
* every group at once and is reported once, in the section header.
|
||||
*/
|
||||
testing: boolean
|
||||
/** Any exit test is in flight; the daemon runs one at a time. */
|
||||
/** Any refresh pass is in flight; the daemon runs one at a time. */
|
||||
testBusy: boolean
|
||||
onTest: () => void
|
||||
onEdit: () => void
|
||||
@@ -1309,7 +1323,11 @@ function GroupRow({
|
||||
health={health}
|
||||
healthKnown={healthKnown}
|
||||
/>
|
||||
<GroupTestReadout test={test} pending={testing && !test} />
|
||||
<GroupTestReadout
|
||||
test={test}
|
||||
pending={testing && !test}
|
||||
hideAbsence={health?.used === false}
|
||||
/>
|
||||
</>
|
||||
)}
|
||||
</div>
|
||||
@@ -1320,7 +1338,7 @@ function GroupRow({
|
||||
editLabel={`Edit group ${group.Name}`}
|
||||
deleteLabel={`Delete group ${group.Name}`}
|
||||
onTest={showHealth ? onTest : undefined}
|
||||
testLabel={showHealth ? `Test the exit of group ${group.Name}` : undefined}
|
||||
testLabel={showHealth ? `Refresh the reading for group ${group.Name}` : undefined}
|
||||
testDisabled={testBusy}
|
||||
/>
|
||||
</li>
|
||||
@@ -1379,21 +1397,7 @@ function GroupHealthReadout({
|
||||
// its members would stay "untested" forever. That is a fact about the ROUTING
|
||||
// CONFIG, not about the members — so instead of counters that could only ever
|
||||
// read as a permanent unknown, the card says so, quietly: unused, not unwell.
|
||||
if (!health.used) {
|
||||
return (
|
||||
<div className="gh gh--unused">
|
||||
<div className="gh-line">
|
||||
<span
|
||||
className="gh-unused"
|
||||
title="No enabled rule routes through this group, so its members are not probed. Add it to a rule to see health."
|
||||
>
|
||||
unused
|
||||
</span>
|
||||
<span className="gh-quiet">not probed — no enabled rule routes through this group</span>
|
||||
</div>
|
||||
</div>
|
||||
)
|
||||
}
|
||||
if (!health.used) return <NotRoutedNote kind="group" />
|
||||
|
||||
const v = verdictOf(health)
|
||||
const { total, tested, alive, dead, untested } = health
|
||||
@@ -1556,6 +1560,56 @@ function GroupHealthReadout({
|
||||
)
|
||||
}
|
||||
|
||||
/**
|
||||
* The card's answer when NOTHING ROUTES THROUGH THIS TARGET. Shared by the group
|
||||
* card and the chain card, because it is the same misunderstanding on both.
|
||||
*
|
||||
* It has to carry two statements, and the old one-liner ("not probed — no enabled
|
||||
* rule routes through this group") only carried the first. Read fast it still
|
||||
* landed as a verdict: a card that normally shows health and today shows a grey
|
||||
* pill reads as "the health is bad". So the two meanings are now separated, on
|
||||
* purpose and in this order:
|
||||
*
|
||||
* 1. the ROUTING FACT — nothing routes here, so nothing measures it;
|
||||
* 2. the NON-FACT — this is not a health reading at all. Absent numbers are
|
||||
* absence of measurement, never failure.
|
||||
*
|
||||
* For a GROUP there is a third line, and it is the confusion this whole change
|
||||
* exists to end: a group used only as a hop inside a chain is never routed to
|
||||
* DIRECTLY, so it correctly reads unused here while carrying real traffic as a
|
||||
* hop. Its health is measured at that hop, on the chain's card.
|
||||
*
|
||||
* Unused is neutral — groove-grey, never amber, never crit. It is a state of the
|
||||
* config, and the config is not sick.
|
||||
*/
|
||||
function NotRoutedNote({ kind }: { kind: 'group' | 'chain' }) {
|
||||
return (
|
||||
<div className="gh gh--unused">
|
||||
<div className="gh-line">
|
||||
<span className="gh-unused">unused</span>
|
||||
<span className="gh-quiet">
|
||||
No enabled rule routes through this {kind}, so the observatory never probes it.
|
||||
</span>
|
||||
</div>
|
||||
<p className="gh-say">
|
||||
That is a routing fact, not a health reading. There are no numbers here because nothing
|
||||
measured this {kind} — not because it failed.
|
||||
</p>
|
||||
{kind === 'group' ? (
|
||||
<p className="gh-say">
|
||||
A group used only as a hop inside a chain reads unused here on purpose: the rules point at
|
||||
the chain, not at the group. Its members are measured at that hop, so its real health is on
|
||||
that chain’s card, hop by hop.
|
||||
</p>
|
||||
) : (
|
||||
<p className="gh-say">
|
||||
Point a rule at this chain and the observatory starts measuring every hop within seconds.
|
||||
</p>
|
||||
)}
|
||||
</div>
|
||||
)
|
||||
}
|
||||
|
||||
/**
|
||||
* One group's member rows, fetched on demand.
|
||||
*
|
||||
@@ -1661,7 +1715,9 @@ function MemberRow({ member }: { member: GroupMemberHealth }) {
|
||||
}
|
||||
|
||||
/**
|
||||
* What a group test found, in the four states it actually has.
|
||||
* What the OBSERVATORY measured for this target end to end, in the four states it
|
||||
* actually has. Nothing here was dialled by the panel — it is a read of the
|
||||
* background prober's own measurement along the real path.
|
||||
*
|
||||
* The one worth spelling out: `ok` with an EMPTY `exit_ip` is a SUCCESS. The
|
||||
* delay was measured; only the address lookup came back empty. Rendering that as
|
||||
@@ -1669,15 +1725,45 @@ function MemberRow({ member }: { member: GroupMemberHealth }) {
|
||||
* traffic, so it reads as a result with the address slot marked unknown — dim,
|
||||
* not red, and the LED stays green.
|
||||
*/
|
||||
function GroupTestReadout({ test, pending }: { test?: GroupTestResult; pending: boolean }) {
|
||||
/**
|
||||
* Errors that mean NO MEASUREMENT EXISTS, as opposed to "this target is broken".
|
||||
*
|
||||
* Three of the observatory's four failure reasons are about the observatory, not
|
||||
* about the path: nothing routes here, nothing has reached it yet, or background
|
||||
* probing is switched off. Painting those crit-red — which is what `ok:false`
|
||||
* used to buy you — reports a fault that nobody has found, on a target that may
|
||||
* be carrying traffic perfectly. Only "the observatory's probe through this path
|
||||
* failed" is a health finding, and it is deliberately NOT in this list.
|
||||
*
|
||||
* Matched on a stable fragment rather than the whole sentence, so a daemon that
|
||||
* rewords the tail still classifies. An error we don't recognise stays red: an
|
||||
* unknown failure is likelier to be real than not, and that is the safe default.
|
||||
*/
|
||||
const NO_MEASUREMENT = [
|
||||
'not routed by any enabled rule',
|
||||
'has not reached this target yet',
|
||||
'background probing is disabled',
|
||||
]
|
||||
const isAbsence = (err: string): boolean => NO_MEASUREMENT.some((frag) => err.includes(frag))
|
||||
|
||||
function GroupTestReadout({
|
||||
test,
|
||||
pending,
|
||||
hideAbsence,
|
||||
}: {
|
||||
test?: GroupTestResult
|
||||
pending: boolean
|
||||
/** The card already explains why nothing measures this target (the unused
|
||||
* note), so an absence error here would just say it a second time. */
|
||||
hideAbsence?: boolean
|
||||
}) {
|
||||
if (pending) {
|
||||
// "testing", never "measuring": the health run owns that word and covers every
|
||||
// group at once. Two runs that read the same on a card is how one group's test
|
||||
// came to look like all four were busy.
|
||||
// Names who is working and on what: the prober, on this target. The badge is
|
||||
// scoped to the cards the run covers, so it can say "this one" honestly.
|
||||
return (
|
||||
<div className="tg-test tg-test--wait" role="status">
|
||||
<Led variant="amber" pulse />
|
||||
<span className="tg-test-msg">testing this exit…</span>
|
||||
<span className="tg-test-msg">waiting for the prober to measure this…</span>
|
||||
</div>
|
||||
)
|
||||
}
|
||||
@@ -1686,10 +1772,22 @@ function GroupTestReadout({ test, pending }: { test?: GroupTestResult; pending:
|
||||
const at = test.tested_unix ? fmtClock(test.tested_unix) : ''
|
||||
|
||||
if (!test.ok) {
|
||||
// No measurement exists. Unlit lamp, quiet text: this panel's way of saying
|
||||
// "no verdict", which is precisely the state — never a red one.
|
||||
if (isAbsence(test.error)) {
|
||||
if (hideAbsence) return null
|
||||
return (
|
||||
<div className="tg-test tg-test--none" role="status">
|
||||
<Led variant="off" />
|
||||
<span className="tg-test-msg">{test.error}</span>
|
||||
{at && <span className="tg-test-at mono">{at}</span>}
|
||||
</div>
|
||||
)
|
||||
}
|
||||
return (
|
||||
<div className="tg-test tg-test--bad" role="status">
|
||||
<Led variant="crit" />
|
||||
<span className="tg-test-msg">{test.error || 'the test failed'}</span>
|
||||
<span className="tg-test-msg">{test.error || 'the probe failed'}</span>
|
||||
{at && <span className="tg-test-at mono">{at}</span>}
|
||||
</div>
|
||||
)
|
||||
@@ -2114,7 +2212,7 @@ function ChainRow({
|
||||
chain,
|
||||
busy,
|
||||
showHealth,
|
||||
used,
|
||||
health,
|
||||
test,
|
||||
testing,
|
||||
testBusy,
|
||||
@@ -2125,19 +2223,19 @@ function ChainRow({
|
||||
chain: Chain
|
||||
busy: boolean
|
||||
/** Group health checks are on (Settings). When false, the card drops its
|
||||
* exit-test readout and Test button — it is config only. */
|
||||
* health readout and Refresh button — it is config only. */
|
||||
showHealth: boolean
|
||||
/** This chain's reachability (GroupHealth.Used's chain analogue, plan §5.E).
|
||||
* undefined ⇒ the health endpoint hasn't reported this chain (not applied yet, or
|
||||
* a daemon version without chains): no badge. false ⇒ no enabled rule routes
|
||||
* through the chain, so the observatory never probes it and the card renders
|
||||
* "unused" instead of an exit-test readout. */
|
||||
used?: boolean
|
||||
/** Everything the observatory knows about this chain: whether any enabled rule
|
||||
* routes through it, and the per-hop measurements along it.
|
||||
* undefined ⇒ the health endpoint hasn't reported this chain at all (not
|
||||
* applied yet, or a daemon version without chains): the card says nothing
|
||||
* rather than guessing. */
|
||||
health?: ChainHealth
|
||||
test?: GroupTestResult
|
||||
/** An exit test covering THIS chain is in flight (the caller resolves it
|
||||
/** A refresh pass covering THIS chain is in flight (the caller resolves it
|
||||
* against the run's scope, exactly as for a group card). */
|
||||
testing: boolean
|
||||
/** Any exit test is in flight; the daemon runs one at a time. */
|
||||
/** Any refresh pass is in flight; the daemon runs one at a time. */
|
||||
testBusy: boolean
|
||||
onTest: () => void
|
||||
onEdit: () => void
|
||||
@@ -2181,25 +2279,21 @@ function ChainRow({
|
||||
{showHealth && (
|
||||
<>
|
||||
{/* A chain no enabled rule routes through is never probed (the
|
||||
observatory walks only reachable paths), so instead of an exit-test
|
||||
readout the card says so, quietly — the same "unused" pattern the
|
||||
group card uses (GroupHealthReadout), not a new design. `used` is
|
||||
undefined until the health endpoint reports this chain (or from a
|
||||
daemon version without chains): no badge then. */}
|
||||
{used === false && (
|
||||
<div className="gh gh--unused">
|
||||
<div className="gh-line">
|
||||
<span
|
||||
className="gh-unused"
|
||||
title="No enabled rule routes through this chain, so its exit is not probed. Add it to a rule to see health."
|
||||
>
|
||||
unused
|
||||
</span>
|
||||
<span className="gh-quiet">not probed — no enabled rule routes through this chain</span>
|
||||
</div>
|
||||
</div>
|
||||
)}
|
||||
<GroupTestReadout test={test} pending={testing && !test} />
|
||||
observatory walks only reachable paths), so instead of a health
|
||||
readout the card says so — the same "unused" note the group card
|
||||
uses, not a new design. `health` is undefined until the endpoint
|
||||
reports this chain (or on a daemon without chains): say nothing
|
||||
then rather than guess. */}
|
||||
{health?.used === false ? (
|
||||
<NotRoutedNote kind="chain" />
|
||||
) : health?.used ? (
|
||||
<ChainHopRail chain={chain.Name} defs={hops} hops={health.hops} />
|
||||
) : null}
|
||||
<GroupTestReadout
|
||||
test={test}
|
||||
pending={testing && !test}
|
||||
hideAbsence={health?.used === false}
|
||||
/>
|
||||
</>
|
||||
)}
|
||||
</div>
|
||||
@@ -2210,13 +2304,210 @@ function ChainRow({
|
||||
editLabel={`Edit chain ${chain.Name}`}
|
||||
deleteLabel={`Delete chain ${chain.Name}`}
|
||||
onTest={showHealth ? onTest : undefined}
|
||||
testLabel={showHealth ? `Test the exit of chain ${chain.Name}` : undefined}
|
||||
testLabel={showHealth ? `Refresh the reading for chain ${chain.Name}` : undefined}
|
||||
testDisabled={testBusy}
|
||||
/>
|
||||
</li>
|
||||
)
|
||||
}
|
||||
|
||||
// ---- chain hop rail --------------------------------------------------------
|
||||
|
||||
/**
|
||||
* The API gives hops an index and no name. The page already knows the names — the
|
||||
* model's own `Hops` strings ("egress:ewan", "node:awgout", "group:sub0") — so
|
||||
* zip the two by POSITION.
|
||||
*
|
||||
* Two things make that safe rather than clever. A LEADING `egress:` is not a
|
||||
* numbered hop: the daemon lifts it into hop 1's entry detour, so it is dropped
|
||||
* before counting. And if the counts still disagree — a chain that splices
|
||||
* sub-chains gets FLATTENED by the daemon, producing more wire hops than the
|
||||
* config lists — every label is dropped. A hop labelled with its neighbour's name
|
||||
* is worse than a hop with no name at all: it would send someone to fix the wrong
|
||||
* target.
|
||||
*/
|
||||
function hopLabels(defs: string[], hops: ChainHopHealth[]): (string | undefined)[] {
|
||||
const numbered = defs.length > 0 && defs[0].startsWith('egress:') ? defs.slice(1) : defs
|
||||
if (numbered.length !== hops.length) return hops.map(() => undefined)
|
||||
return hops.map((h) => (h.index >= 1 && h.index <= numbered.length ? numbered[h.index - 1] : undefined))
|
||||
}
|
||||
|
||||
/**
|
||||
* One hop's lamp.
|
||||
*
|
||||
* `dead` is crit and `untested` is an UNLIT socket — never red, because nothing
|
||||
* has been measured and an unlit lamp is this panel's way of saying "no verdict".
|
||||
* The fourth case is the page's existing house reading, applied here for
|
||||
* consistency rather than invented: a group hop that is carrying traffic but has
|
||||
* confirmed failures on its board is amber. `state` stays the daemon's word for
|
||||
* "can this hop carry traffic"; the amber only qualifies HOW WELL.
|
||||
*/
|
||||
function hopLed(h: ChainHopHealth): LedVariant {
|
||||
if (h.state === 'dead') return 'crit'
|
||||
if (h.state === 'untested') return 'off'
|
||||
return h.dead > 0 ? 'amber' : 'on'
|
||||
}
|
||||
|
||||
/**
|
||||
* What the observatory measured at each position of a chain — the reading the
|
||||
* daemon always took and the panel never showed.
|
||||
*
|
||||
* THE DESIGN RISK, and the one place this card spends any boldness: the rail
|
||||
* draws the CONDUCTOR as well as the lamps, and severs it below the first dead
|
||||
* hop. A chain is a single series path, so the operator's real question is never
|
||||
* "how many hops are green" — it is "where does my traffic stop". Four lamps in a
|
||||
* column answer the first question and leave the second to arithmetic. A broken
|
||||
* conductor answers the second one before you have read a single word, which is
|
||||
* the whole reason this feature exists.
|
||||
*
|
||||
* It stays honest by keeping the two facts on two different marks. Each LAMP is
|
||||
* that hop's own measurement and never changes because of a hop in front of it —
|
||||
* a hop after the break that answers still shows green, because it really did
|
||||
* answer. The CONDUCTOR is reachability through the path, and that genuinely does
|
||||
* stop at the break. Nothing here is re-derived from the daemon's counters; the
|
||||
* only thing the panel adds is what a chain structurally is.
|
||||
*
|
||||
* Everything around the rail is deliberately quiet: no colour but the semantic
|
||||
* lamps, no motion at all, the orange accent untouched.
|
||||
*/
|
||||
function ChainHopRail({
|
||||
chain,
|
||||
defs,
|
||||
hops,
|
||||
}: {
|
||||
chain: string
|
||||
/** The chain's configured hops, straight off the model — the only source of names. */
|
||||
defs: string[]
|
||||
/** Absent ⇒ the engine never materialised per-hop outbounds. NOT "no hops". */
|
||||
hops?: ChainHopHealth[]
|
||||
}) {
|
||||
const ordered = useMemo(() => [...asArray(hops)].sort((a, b) => a.index - b.index), [hops])
|
||||
const labels = useMemo(() => hopLabels(defs, ordered), [defs, ordered])
|
||||
|
||||
// The first hop that was probed and did not answer. Everything after it is
|
||||
// unreachable THROUGH THIS CHAIN, whatever its own lamp says. `untested` is
|
||||
// never a break: nothing was measured, so nothing is known to be severed.
|
||||
const breakAt = ordered.findIndex((h) => h.state === 'dead')
|
||||
|
||||
if (ordered.length === 0) {
|
||||
// Say why, in one line, instead of an empty rail. The daemon collapses a
|
||||
// single-target chain into a plain alias and never builds copies to measure,
|
||||
// so we can tell the two absences apart from the config alone.
|
||||
const numbered = defs.filter((d, i) => !(i === 0 && d.startsWith('egress:')))
|
||||
return (
|
||||
<div className="gh gh--absent">
|
||||
<span className="gh-absent-msg">
|
||||
{numbered.length <= 1
|
||||
? 'This chain has a single hop, so the engine points traffic straight at that target instead of building a path to measure. Its health is on that target’s own card.'
|
||||
: 'The engine hasn’t built this chain’s hops yet, so there is nothing measured per hop. They appear once it is running with this config applied.'}
|
||||
</span>
|
||||
</div>
|
||||
)
|
||||
}
|
||||
|
||||
return (
|
||||
<div className="gh ch-rail">
|
||||
<span className="ch-eyebrow">measured, hop by hop</span>
|
||||
<ol className="ch-hops">
|
||||
{ordered.map((h, i) => {
|
||||
const label = labels[i]
|
||||
const severed = breakAt >= 0 && i > breakAt
|
||||
const cls = [
|
||||
'ch-hop',
|
||||
`ch-hop--${h.state}`,
|
||||
severed ? 'ch-hop--severed' : '',
|
||||
h.exit ? 'ch-hop--exit' : '',
|
||||
i === 0 ? 'ch-hop--first' : '',
|
||||
]
|
||||
.filter(Boolean)
|
||||
.join(' ')
|
||||
const age = fmtAge(h.age_seconds)
|
||||
return (
|
||||
<li key={h.tag || h.index} className={cls}>
|
||||
<span className="ch-num mono" aria-hidden="true">
|
||||
{h.index}
|
||||
</span>
|
||||
<span className="ch-socket">
|
||||
<Led variant={hopLed(h)} />
|
||||
</span>
|
||||
<div className="ch-body">
|
||||
<div className="ch-l1">
|
||||
<span className="ch-name mono" title={`engine outbound ${h.tag}`}>
|
||||
{label ?? (h.kind === 'group' ? 'a group hop' : 'a node hop')}
|
||||
</span>
|
||||
{h.exit && <span className="ch-tag">exit</span>}
|
||||
{h.state === 'alive' && h.delay_ms > 0 && (
|
||||
<span className="ch-delay mono">{h.delay_ms} ms</span>
|
||||
)}
|
||||
{h.state === 'untested' && <span className="ch-quiet">not measured yet</span>}
|
||||
{age && <span className="ch-age mono">{age}</span>}
|
||||
</div>
|
||||
|
||||
{/* A node hop IS its own measurement (total 1), so counters would
|
||||
only restate the lamp. A group hop rolls up its per-hop member
|
||||
copies, and those read exactly as they do everywhere else in
|
||||
this app: alive out of TESTED, with the untested remainder as a
|
||||
quiet aside only when there is one. */}
|
||||
{h.kind === 'group' && h.total > 0 && (
|
||||
<div className="ch-l2">
|
||||
{h.tested === 0 ? (
|
||||
<span className="ch-rest mono">
|
||||
{h.total} member{h.total === 1 ? '' : 's'}, none measured
|
||||
</span>
|
||||
) : (
|
||||
<>
|
||||
<span className="ch-count mono">
|
||||
<b>{h.alive}</b> / {h.tested}
|
||||
</span>
|
||||
<span className="ch-word">alive</span>
|
||||
{h.dead > 0 && (
|
||||
<span
|
||||
className="ch-dead mono"
|
||||
title={`${h.dead} member${h.dead === 1 ? '' : 's'} were probed at this hop and did not answer`}
|
||||
>
|
||||
{h.dead} not answering
|
||||
</span>
|
||||
)}
|
||||
{h.untested > 0 && (
|
||||
<span className="ch-rest mono">
|
||||
tested {h.tested} of {h.total}
|
||||
</span>
|
||||
)}
|
||||
</>
|
||||
)}
|
||||
{h.selected && (
|
||||
<span
|
||||
className="ch-now mono"
|
||||
title={`Traffic crossing hop ${h.index} of “${chain}” is on ${h.selected}`}
|
||||
>
|
||||
now → {h.selected}
|
||||
</span>
|
||||
)}
|
||||
</div>
|
||||
)}
|
||||
</div>
|
||||
</li>
|
||||
)
|
||||
})}
|
||||
</ol>
|
||||
|
||||
{/* The sentence the rail's shape implies, written out — because the break is
|
||||
the answer someone came here for, and a graphic alone should never be the
|
||||
only place a finding exists. */}
|
||||
{breakAt >= 0 && (
|
||||
<p className="gh-say gh-say--bad">
|
||||
Hop {ordered[breakAt].index}
|
||||
{labels[breakAt] ? ` (${labels[breakAt]})` : ''} was probed and did not answer. A chain is
|
||||
one path, so traffic stops there
|
||||
{breakAt < ordered.length - 1
|
||||
? ' — the hops after it answer on their own, but nothing reaches them through this chain.'
|
||||
: '.'}
|
||||
</p>
|
||||
)}
|
||||
</div>
|
||||
)
|
||||
}
|
||||
|
||||
function ChainEditor({
|
||||
initial,
|
||||
hopOptions,
|
||||
@@ -2691,8 +2982,8 @@ function RowActions({
|
||||
busy: boolean
|
||||
editLabel: string
|
||||
deleteLabel: string
|
||||
// Only groups and chains can be tested, so the control is optional and absent
|
||||
// everywhere else rather than a disabled stub on every row.
|
||||
// Only groups and chains are probed, so the refresh control is optional and
|
||||
// absent everywhere else rather than a disabled stub on every row.
|
||||
onTest?: () => void
|
||||
testLabel?: string
|
||||
testDisabled?: boolean
|
||||
@@ -2705,8 +2996,9 @@ function RowActions({
|
||||
onClick={onTest}
|
||||
disabled={busy || testDisabled}
|
||||
aria-label={testLabel}
|
||||
title="Ask the background prober to measure this target out of turn. It does not open a connection from the panel."
|
||||
>
|
||||
Test
|
||||
Refresh
|
||||
</Button>
|
||||
)}
|
||||
<Button className="tg-act" onClick={onEdit} disabled={busy} aria-label={editLabel}>
|
||||
|
||||
@@ -44,6 +44,13 @@ type URLTest struct {
|
||||
group *URLTestGroup
|
||||
interruptExternalConnections bool
|
||||
balancer *balancer // lx: SPEC 019 — nil for least_test (default)
|
||||
// lx: health board §5.C — true when options.SelfCheck == false: the group's
|
||||
// OWN probing schedule (PostStart warm-up + Touch ticker) is stood down and
|
||||
// the observatory is the only thing that measures its members. Stored
|
||||
// INVERTED so the zero value keeps today's behaviour for every construction
|
||||
// path that does not go through NewURLTest (hand-built groups in tests).
|
||||
// See option.URLTestOutboundOptions.SelfCheck for the full reasoning.
|
||||
selfCheckDisabled bool
|
||||
}
|
||||
|
||||
func NewURLTest(ctx context.Context, router adapter.Router, logger log.ContextLogger, tag string, options option.URLTestOutboundOptions) (adapter.Outbound, error) {
|
||||
@@ -71,6 +78,9 @@ func NewURLTest(ctx context.Context, router adapter.Router, logger log.ContextLo
|
||||
idleTimeout: time.Duration(options.IdleTimeout),
|
||||
interruptExternalConnections: options.InterruptExistConnections,
|
||||
balancer: balancer,
|
||||
// nil/absent means true (self-check on) — the documented default, so a
|
||||
// config written before the flag existed behaves exactly as it always has.
|
||||
selfCheckDisabled: options.SelfCheck != nil && !*options.SelfCheck,
|
||||
}
|
||||
if len(outbound.tags) == 0 {
|
||||
return nil, E.New("missing tags")
|
||||
@@ -92,6 +102,9 @@ func (s *URLTest) Start() error {
|
||||
return err
|
||||
}
|
||||
group.balancer = s.balancer // lx: SPEC 019 v2 — health-check drives the pool through it
|
||||
// lx: health board §5.C — carry the stand-down flag onto the group the same
|
||||
// way the balancer travels: set after construction, immutable from then on.
|
||||
group.selfCheckDisabled = s.selfCheckDisabled
|
||||
if s.balancer != nil {
|
||||
// lx: health board §5.B — slot liveness reads through the board verdict, so a
|
||||
// death recorded by any prober or a failed dial takes effect on the next pick,
|
||||
@@ -319,6 +332,12 @@ type URLTestGroup struct {
|
||||
lastActive common.TypedValue[time.Time]
|
||||
lastSelected common.TypedValue[string] // lx: SPEC 019 — Now() in balanced modes
|
||||
balancer *balancer // lx: SPEC 019 v2 — round_robin pool; nil for least_test
|
||||
// lx: health board §5.C — mirrors URLTest.selfCheckDisabled (set by Start,
|
||||
// immutable afterwards, zero value = probing on). Guards ONLY the group's
|
||||
// own schedule: the PostStart warm-up sweep and the Touch ticker. An
|
||||
// explicit CheckOutbounds/URLTest call is untouched — the flag stands down
|
||||
// the schedule, not the capability.
|
||||
selfCheckDisabled bool
|
||||
}
|
||||
|
||||
func NewURLTestGroup(ctx context.Context, outboundManager adapter.OutboundManager, logger log.Logger, outbounds []adapter.Outbound, link string, interval time.Duration, tolerance uint16, idleTimeout time.Duration, interruptExternalConnections bool) (*URLTestGroup, error) {
|
||||
@@ -362,14 +381,35 @@ func (g *URLTestGroup) PostStart() {
|
||||
g.lastActive.Store(time.Now())
|
||||
// lx: SPEC 019 v2 — seed the pool so round_robin can route from the first connection,
|
||||
// before the first health-check completes (history-warm nodes first, else config order).
|
||||
// The seed only READS the board, so it runs even with the self-check stood down.
|
||||
g.seedPool()
|
||||
go g.CheckOutbounds(false)
|
||||
// lx: health board §5.C — the warm-up sweep is the first half of the group's
|
||||
// own probing schedule, and it fires for EVERY group at box start, including
|
||||
// groups no routing rule reaches. For those, the sweep dials every member
|
||||
// directly from the router — a path nothing uses — and records the outcome
|
||||
// under the members' base tags, forging the board reading the observatory
|
||||
// exists to keep honest. A stood-down group therefore skips it entirely; the
|
||||
// observatory (or nothing, for a truly unused group) is what measures its
|
||||
// members.
|
||||
if !g.selfCheckDisabled {
|
||||
go g.CheckOutbounds(false)
|
||||
}
|
||||
}
|
||||
|
||||
func (g *URLTestGroup) Touch() {
|
||||
if !g.started {
|
||||
return
|
||||
}
|
||||
// lx: health board §5.C — Touch's only job is to keep the group's OWN
|
||||
// probing ticker alive while traffic flows. With the self-check stood down
|
||||
// there is deliberately no ticker to start or feed: the observatory owns the
|
||||
// schedule, and a stray dial through an unused group (a stale rule cache, a
|
||||
// manual pin) must not arm 30 minutes of direct probing under the members'
|
||||
// base tags. Checked before the lock because the flag is immutable after
|
||||
// Start, exactly like the started fast-path above.
|
||||
if g.selfCheckDisabled {
|
||||
return
|
||||
}
|
||||
g.access.Lock()
|
||||
defer g.access.Unlock()
|
||||
if g.ticker != nil {
|
||||
|
||||
@@ -0,0 +1,116 @@
|
||||
package group
|
||||
|
||||
// lx: health board §5.C tests — SelfCheck stands the group's OWN probing
|
||||
// schedule down: no PostStart warm-up sweep, no Touch ticker. The explicit
|
||||
// CheckOutbounds path stays available, and the nil default keeps probing.
|
||||
|
||||
import (
|
||||
"context"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"github.com/sagernet/sing-box/common/urltest"
|
||||
"github.com/sagernet/sing-box/log"
|
||||
"github.com/sagernet/sing-box/option"
|
||||
)
|
||||
|
||||
// waitForHistory polls until the store holds an entry for tag or the deadline
|
||||
// passes; reports whether it appeared. PostStart's sweep runs on its own
|
||||
// goroutine, so both directions of the assertion need a bounded wait.
|
||||
func waitForHistory(hist *urltest.HistoryStorage, tag string, deadline time.Duration) bool {
|
||||
stop := time.Now().Add(deadline)
|
||||
for time.Now().Before(stop) {
|
||||
if hist.LoadURLTestHistory(tag) != nil {
|
||||
return true
|
||||
}
|
||||
time.Sleep(5 * time.Millisecond)
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
// A group with the self-check stood down writes NOTHING to the history storage
|
||||
// on PostStart: the warm-up sweep — which would dial the member directly from
|
||||
// the router and mark the failure under its base tag — must not fire. And
|
||||
// Touch, the other half of the schedule, must not start a ticker either.
|
||||
func TestSelfCheckDisabledPostStartWritesNothing(t *testing.T) {
|
||||
hist := urltest.NewHistoryStorage()
|
||||
a := &healthNode{tag: "a", fail: true}
|
||||
manager := managerOf(a)
|
||||
g := healthTestGroup(hist, manager, a)
|
||||
g.selfCheckDisabled = true
|
||||
|
||||
g.PostStart()
|
||||
// The absence of a write is the assertion, so give the (non-existent) sweep
|
||||
// real time to have happened before declaring victory.
|
||||
if waitForHistory(hist, "a", 150*time.Millisecond) {
|
||||
t.Fatal("a stood-down group's PostStart wrote to the board; the warm-up sweep must not fire")
|
||||
}
|
||||
|
||||
g.Touch()
|
||||
g.access.Lock()
|
||||
ticker := g.ticker
|
||||
g.access.Unlock()
|
||||
if ticker != nil {
|
||||
t.Fatal("Touch armed the probing ticker on a stood-down group")
|
||||
}
|
||||
}
|
||||
|
||||
// The default (SelfCheck nil, i.e. the zero-value field on a hand-built group)
|
||||
// keeps today's behaviour: PostStart's warm-up sweep runs and records the
|
||||
// failing member on the board.
|
||||
func TestSelfCheckDefaultStillProbesOnPostStart(t *testing.T) {
|
||||
hist := urltest.NewHistoryStorage()
|
||||
a := &healthNode{tag: "a", fail: true}
|
||||
manager := managerOf(a)
|
||||
g := healthTestGroup(hist, manager, a)
|
||||
|
||||
g.PostStart()
|
||||
if !waitForHistory(hist, "a", 5*time.Second) {
|
||||
t.Fatal("default group's PostStart never probed; the self-check must stay on unless stood down")
|
||||
}
|
||||
if v := hist.Verdict("a", 10*time.Minute); v != urltest.VerdictDead {
|
||||
t.Fatalf("verdict(a) = %v, want dead from the warm-up sweep", v)
|
||||
}
|
||||
}
|
||||
|
||||
// An EXPLICIT CheckOutbounds still probes a stood-down group: the flag
|
||||
// suppresses the group's own schedule, never a deliberate request (the adapter
|
||||
// interface a human or an API invokes on purpose).
|
||||
func TestSelfCheckDisabledExplicitCheckStillProbes(t *testing.T) {
|
||||
hist := urltest.NewHistoryStorage()
|
||||
a := &healthNode{tag: "a", fail: true}
|
||||
manager := managerOf(a)
|
||||
g := healthTestGroup(hist, manager, a)
|
||||
g.selfCheckDisabled = true
|
||||
|
||||
g.CheckOutbounds(true)
|
||||
if hist.LoadURLTestHistory("a") == nil {
|
||||
t.Fatal("an explicit CheckOutbounds(true) did not probe; the flag must only stand down the schedule")
|
||||
}
|
||||
}
|
||||
|
||||
// The option → outbound plumbing: nil/absent means on, an explicit false means
|
||||
// stood down, an explicit true means on. NewURLTest is the only place the
|
||||
// option is read, so this is where a plumbing regression would hide.
|
||||
func TestSelfCheckOptionPlumbing(t *testing.T) {
|
||||
build := func(selfCheck *bool) *URLTest {
|
||||
t.Helper()
|
||||
opts := option.URLTestOutboundOptions{Outbounds: []string{"a"}}
|
||||
opts.SelfCheck = selfCheck
|
||||
ob, err := NewURLTest(context.Background(), nil, log.NewNOPFactory().Logger(), "t", opts)
|
||||
if err != nil {
|
||||
t.Fatalf("NewURLTest: %v", err)
|
||||
}
|
||||
return ob.(*URLTest)
|
||||
}
|
||||
if build(nil).selfCheckDisabled {
|
||||
t.Fatal("nil SelfCheck must keep the self-check ON (the compatibility default)")
|
||||
}
|
||||
on, off := true, false
|
||||
if build(&on).selfCheckDisabled {
|
||||
t.Fatal("SelfCheck=true must keep the self-check on")
|
||||
}
|
||||
if !build(&off).selfCheckDisabled {
|
||||
t.Fatal("SelfCheck=false must stand the self-check down")
|
||||
}
|
||||
}
|
||||
+28
-12
@@ -205,11 +205,15 @@ func (a *Applier) configureObservatory(m *model.Model, opts option.Options) {
|
||||
})
|
||||
}
|
||||
|
||||
// TestGroups launches the engine's one-shot exit test (delay + exit address, F2)
|
||||
// of the named groups/chains, returning started=false when a run is already in
|
||||
// flight or the engine is absent. names empty/nil = every group and every chain.
|
||||
// It takes NEITHER the apply mutex nor the flock, so kicking off a test never
|
||||
// blocks behind an Apply.
|
||||
// TestGroups launches the engine's one-shot group/chain test (F2): the engine
|
||||
// asks its observatory for an out-of-turn pass and reports what it measured,
|
||||
// plus the exit address for alive rule-routed targets — it no longer dials
|
||||
// health probes of its own. Returns started=false when a run is already in
|
||||
// flight or the engine is absent. names empty/nil = every group and every
|
||||
// chain. probeURL is passed through for signature stability and IGNORED by the
|
||||
// engine (the probe URL is a global observatory setting now). It takes NEITHER
|
||||
// the apply mutex nor the flock, so kicking off a test never blocks behind an
|
||||
// Apply.
|
||||
func (a *Applier) TestGroups(names []string, probeURL string) (started bool) {
|
||||
if a.eng == nil {
|
||||
return false
|
||||
@@ -686,18 +690,30 @@ func (a *Applier) setWarnings(ws []Warning) {
|
||||
a.stateMu.Unlock()
|
||||
}
|
||||
|
||||
// Warnings returns the normalised warning set from the last successful apply.
|
||||
// Never nil: an empty slice means "the last apply was clean", which the panel
|
||||
// must render differently from "no apply has run yet" (Active/Plane cover that).
|
||||
// Warnings returns the normalised warning set from the last successful apply,
|
||||
// PLUS whatever is wrong right now that no apply can describe. Never nil: an
|
||||
// empty slice means "the last apply was clean", which the panel must render
|
||||
// differently from "no apply has run yet" (Active/Plane cover that).
|
||||
//
|
||||
// The live half is currently the engine's abandoned generations
|
||||
// (engineTeardownWarnings). It is computed at READ time rather than folded into
|
||||
// lastWarnings on purpose: a superseded box that will not shut down is a
|
||||
// condition of the process, not a property of a config. Folding it in would make
|
||||
// it appear only after the NEXT successful apply and then stay published long
|
||||
// after the shutdown finally completed — reporting a leak that is over, and
|
||||
// staying silent about one that is not. Read-time means it shows up the instant
|
||||
// it happens and clears itself the instant it resolves.
|
||||
func (a *Applier) Warnings() []Warning {
|
||||
var out []Warning
|
||||
if a.eng != nil {
|
||||
out = engineTeardownWarnings(a.eng.PendingCloses())
|
||||
}
|
||||
a.stateMu.RLock()
|
||||
defer a.stateMu.RUnlock()
|
||||
if a.lastWarnings == nil {
|
||||
if out == nil && a.lastWarnings == nil {
|
||||
return []Warning{}
|
||||
}
|
||||
out := make([]Warning, len(a.lastWarnings))
|
||||
copy(out, a.lastWarnings)
|
||||
return out
|
||||
return append(out, a.lastWarnings...)
|
||||
}
|
||||
|
||||
// Reconcile re-reads UCI and either tears down (disabled) or re-applies (enabled).
|
||||
|
||||
@@ -0,0 +1,283 @@
|
||||
package apply
|
||||
|
||||
// Regression cover for the leaked-engine-generation defect.
|
||||
//
|
||||
// Observed on the router: one shaterd process was carrying up to FOUR sing-box
|
||||
// instances at once. sing-box stamps every log line with the elapsed seconds of
|
||||
// ITS OWN instance, so the same process printed `ERROR[2015]` and `ERROR[0129]`
|
||||
// in the same second — two engines half an hour apart in age, both alive, both
|
||||
// dialling, both holding WireGuard devices built from the same private keys. A
|
||||
// full daemon stop+start collapsed it back to one generation, which places the
|
||||
// leak squarely on the config re-apply path rather than on startup.
|
||||
//
|
||||
// The tests below pin the two halves of the fix:
|
||||
//
|
||||
// TestApplySwapsLeaveExactlyOneEngineGeneration — the healthy path really
|
||||
// retires the old instance (its listener is provably gone), N times in a row.
|
||||
// TestStuckEngineCloseDoesNotBlockTheApply — a shutdown that never returns
|
||||
// is bounded, does not stall the apply, and is REPORTED as a critical warning
|
||||
// for exactly as long as it is true.
|
||||
|
||||
import (
|
||||
"io"
|
||||
"net"
|
||||
"net/netip"
|
||||
"strconv"
|
||||
"strings"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
C "github.com/sagernet/sing-box/constant"
|
||||
"github.com/sagernet/sing-box/option"
|
||||
"github.com/sagernet/sing-box/shater/engine"
|
||||
"github.com/sagernet/sing/common/json/badoption"
|
||||
)
|
||||
|
||||
// mixedOn builds a minimal but REAL engine config: a mixed inbound bound to
|
||||
// 127.0.0.1:port plus a direct outbound. Two configs with different ports hash
|
||||
// differently, so each Apply is a genuine swap rather than a hash-gate no-op —
|
||||
// and the bound port is the observable that proves whether the old instance
|
||||
// actually died.
|
||||
func mixedOn(port uint16) option.Options {
|
||||
listen := badoption.Addr(netip.MustParseAddr("127.0.0.1"))
|
||||
return option.Options{
|
||||
Log: &option.LogOptions{Level: "error"},
|
||||
Inbounds: []option.Inbound{{
|
||||
Type: C.TypeMixed,
|
||||
Tag: "mixed-in",
|
||||
Options: &option.HTTPMixedInboundOptions{
|
||||
ListenOptions: option.ListenOptions{Listen: &listen, ListenPort: port},
|
||||
},
|
||||
}},
|
||||
Outbounds: []option.Outbound{{
|
||||
Type: C.TypeDirect,
|
||||
Tag: "direct-out",
|
||||
Options: &option.DirectOutboundOptions{},
|
||||
}},
|
||||
}
|
||||
}
|
||||
|
||||
// portFree reports whether 127.0.0.1:port can be bound right now — i.e. whether
|
||||
// the instance that used to listen there is really gone. Retried briefly because
|
||||
// a listener is released by Close, not by the return of Close's caller.
|
||||
func portFree(port uint16) bool {
|
||||
deadline := time.Now().Add(3 * time.Second)
|
||||
for {
|
||||
ln, err := net.Listen("tcp", net.JoinHostPort("127.0.0.1", strconv.Itoa(int(port))))
|
||||
if err == nil {
|
||||
_ = ln.Close()
|
||||
return true
|
||||
}
|
||||
if time.Now().After(deadline) {
|
||||
return false
|
||||
}
|
||||
time.Sleep(20 * time.Millisecond)
|
||||
}
|
||||
}
|
||||
|
||||
// TestApplySwapsLeaveExactlyOneEngineGeneration is the core regression: after N
|
||||
// sequential applies the process must be carrying ONE engine, not N.
|
||||
//
|
||||
// "Carrying" is checked two ways on purpose. Generations() is the engine's own
|
||||
// accounting (running instance + every retirement still in flight) and would
|
||||
// catch a retirement that silently never completes. The port check is
|
||||
// independent of that accounting: if the superseded instance were still alive it
|
||||
// would still hold its listener, and the bind would fail. A fix that only
|
||||
// reset a pointer would pass the first check and fail the second.
|
||||
func TestApplySwapsLeaveExactlyOneEngineGeneration(t *testing.T) {
|
||||
const (
|
||||
firstPort = 18801
|
||||
applies = 5
|
||||
)
|
||||
|
||||
a := New(engine.New(), nil)
|
||||
t.Cleanup(func() { _ = a.eng.Close() })
|
||||
|
||||
for i := 0; i < applies; i++ {
|
||||
port := uint16(firstPort + i)
|
||||
changed, err := a.eng.Apply(mixedOn(port))
|
||||
if err != nil {
|
||||
t.Fatalf("apply #%d (port %d): %v", i+1, port, err)
|
||||
}
|
||||
if !changed {
|
||||
t.Fatalf("apply #%d: every config here differs, so the swap must be real", i+1)
|
||||
}
|
||||
if got := a.eng.Generations(); got != 1 {
|
||||
t.Fatalf("after apply #%d the process carries %d engine instances, want exactly 1 — "+
|
||||
"a superseded generation is still alive (this is the four-generations-in-one-process leak)",
|
||||
i+1, got)
|
||||
}
|
||||
if stuck := a.eng.PendingCloses(); len(stuck) != 0 {
|
||||
t.Fatalf("after apply #%d: %d generation(s) abandoned, want none: %+v", i+1, len(stuck), stuck)
|
||||
}
|
||||
if i > 0 {
|
||||
prev := uint16(firstPort + i - 1)
|
||||
if !portFree(prev) {
|
||||
t.Fatalf("after apply #%d the PREVIOUS generation still holds 127.0.0.1:%d — "+
|
||||
"the old engine was replaced in the field but never actually stopped", i+1, prev)
|
||||
}
|
||||
}
|
||||
// The warning set must stay clean while teardown is healthy: a critical
|
||||
// warning that cries wolf on every apply is worse than none.
|
||||
for _, w := range a.Warnings() {
|
||||
if w.Section == "engine" {
|
||||
t.Fatalf("after apply #%d a healthy swap produced an engine warning: %+v", i+1, w)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if err := a.eng.Close(); err != nil {
|
||||
t.Fatalf("close: %v", err)
|
||||
}
|
||||
if got := a.eng.Generations(); got != 0 {
|
||||
t.Fatalf("after Close the process carries %d engine instances, want 0", got)
|
||||
}
|
||||
if !portFree(firstPort + applies - 1) {
|
||||
t.Fatalf("after Close the last generation still holds its listener")
|
||||
}
|
||||
}
|
||||
|
||||
// hangingCloser returns a box closer that BLOCKS the first close it is handed
|
||||
// until release() is called, and performs every later close normally. That is the
|
||||
// shape of the real fault: one subsystem of one generation (a WireGuard endpoint)
|
||||
// refuses to come down, while the rest of the process is fine.
|
||||
func hangingCloser() (closer func(io.Closer) error, release func()) {
|
||||
gate := make(chan struct{})
|
||||
first := make(chan struct{}, 1)
|
||||
first <- struct{}{}
|
||||
return func(c io.Closer) error {
|
||||
select {
|
||||
case <-first:
|
||||
<-gate // the stuck generation: never returns until released
|
||||
return c.Close() // ...and then really does close, so the port frees
|
||||
default:
|
||||
return c.Close()
|
||||
}
|
||||
}, func() {
|
||||
close(gate)
|
||||
}
|
||||
}
|
||||
|
||||
// TestStuckEngineCloseDoesNotBlockTheApply pins all four requirements of the
|
||||
// bounded teardown at once:
|
||||
//
|
||||
// 1. the apply COMPLETES — a shutdown that never returns must not hold the
|
||||
// control plane (and therefore the panel) hostage;
|
||||
// 2. the new engine is running afterwards — fail-closed semantics are unchanged,
|
||||
// the swap succeeded;
|
||||
// 3. the abandoned generation is REPORTED as a critical warning, by name, for as
|
||||
// long as it is still running — it is not silently swallowed;
|
||||
// 4. generations do not stack: a further apply while the leak persists leaves
|
||||
// one live instance plus the one abandoned one, not three.
|
||||
func TestStuckEngineCloseDoesNotBlockTheApply(t *testing.T) {
|
||||
const budget = 200 * time.Millisecond
|
||||
restoreBudget := engine.SetCloseBudget(budget)
|
||||
defer restoreBudget()
|
||||
|
||||
a := New(engine.New(), nil)
|
||||
|
||||
// Generation 1 comes up with the REAL closer still installed.
|
||||
if _, err := a.eng.Apply(mixedOn(18811)); err != nil {
|
||||
t.Fatalf("apply #1: %v", err)
|
||||
}
|
||||
|
||||
closer, release := hangingCloser()
|
||||
restoreCloser := engine.SetBoxCloser(closer)
|
||||
released := false
|
||||
defer func() {
|
||||
if !released {
|
||||
release()
|
||||
}
|
||||
restoreCloser()
|
||||
_ = a.eng.Close()
|
||||
}()
|
||||
|
||||
// (1) Generation 2: the retirement of generation 1 will never return.
|
||||
start := time.Now()
|
||||
changed, err := a.eng.Apply(mixedOn(18812))
|
||||
elapsed := time.Since(start)
|
||||
if err != nil {
|
||||
t.Fatalf("apply #2 must SUCCEED despite the stuck teardown: %v", err)
|
||||
}
|
||||
if !changed {
|
||||
t.Fatalf("apply #2: expected a real swap")
|
||||
}
|
||||
// Generously bounded: the budget plus box.New+Start. The point is that it
|
||||
// returned at all — before the fix this waited on Box.Close forever.
|
||||
if elapsed > budget+20*time.Second {
|
||||
t.Fatalf("apply #2 took %s: a stuck teardown must not stall the apply", elapsed)
|
||||
}
|
||||
|
||||
// (2) fail-closed semantics unchanged: the new engine really is up.
|
||||
if !a.eng.Running() {
|
||||
t.Fatalf("apply #2: the new engine must be running")
|
||||
}
|
||||
|
||||
// (3) the leak is visible, named, and critical.
|
||||
stuck := a.eng.PendingCloses()
|
||||
if len(stuck) != 1 {
|
||||
t.Fatalf("PendingCloses() = %+v, want exactly the one abandoned generation", stuck)
|
||||
}
|
||||
if stuck[0].Generation != 1 {
|
||||
t.Errorf("abandoned generation = %d, want 1", stuck[0].Generation)
|
||||
}
|
||||
ws := a.Status().Warnings // the exact set `shaterd status` and the panel read
|
||||
var found *Warning
|
||||
for i := range ws {
|
||||
if ws[i].Section == "engine" {
|
||||
found = &ws[i]
|
||||
break
|
||||
}
|
||||
}
|
||||
if found == nil {
|
||||
t.Fatalf("a superseded engine that will not shut down produced NO warning; "+
|
||||
"Status would show a healthy router: %+v", ws)
|
||||
}
|
||||
if found.Severity != SeverityCritical {
|
||||
t.Errorf("stuck-teardown warning severity = %q, want %q", found.Severity, SeverityCritical)
|
||||
}
|
||||
if !strings.Contains(found.Name, "generation 1") {
|
||||
t.Errorf("stuck-teardown warning must name the generation, got Name=%q", found.Name)
|
||||
}
|
||||
if !strings.Contains(found.Message, "STILL RUNNING") {
|
||||
t.Errorf("stuck-teardown warning must say the instance is still running, got %q", found.Message)
|
||||
}
|
||||
if got := a.eng.Generations(); got != 2 {
|
||||
t.Fatalf("Generations() = %d, want 2 (one live + one abandoned)", got)
|
||||
}
|
||||
|
||||
// (4) another apply while the leak persists must not add a THIRD generation:
|
||||
// with a generation abandoned the swap goes close-old-then-start-new, so the
|
||||
// process still holds one live instance plus the one that will not die.
|
||||
if _, err := a.eng.Apply(mixedOn(18813)); err != nil {
|
||||
t.Fatalf("apply #3: %v", err)
|
||||
}
|
||||
if got := a.eng.Generations(); got != 2 {
|
||||
t.Fatalf("Generations() = %d after a third apply, want 2 — generations are stacking, "+
|
||||
"which is exactly the four-live-engines fault", got)
|
||||
}
|
||||
if stuck := a.eng.PendingCloses(); len(stuck) != 1 || stuck[0].Generation != 1 {
|
||||
t.Fatalf("PendingCloses() = %+v, want only the original abandoned generation 1", stuck)
|
||||
}
|
||||
|
||||
// (5) and it CLEARS: when the shutdown finally completes the warning goes away
|
||||
// on its own. A leak report that outlives the leak trains the operator to
|
||||
// ignore the panel.
|
||||
release()
|
||||
released = true
|
||||
deadline := time.Now().Add(5 * time.Second)
|
||||
for len(a.eng.PendingCloses()) > 0 && time.Now().Before(deadline) {
|
||||
time.Sleep(10 * time.Millisecond)
|
||||
}
|
||||
if got := a.eng.PendingCloses(); len(got) != 0 {
|
||||
t.Fatalf("the finished shutdown is still reported as abandoned: %+v", got)
|
||||
}
|
||||
for _, w := range a.Warnings() {
|
||||
if w.Section == "engine" {
|
||||
t.Fatalf("the engine warning outlived the leak it describes: %+v", w)
|
||||
}
|
||||
}
|
||||
if got := a.eng.Generations(); got != 1 {
|
||||
t.Fatalf("Generations() = %d after the stuck shutdown completed, want 1", got)
|
||||
}
|
||||
}
|
||||
@@ -24,7 +24,9 @@ import (
|
||||
"sort"
|
||||
"strconv"
|
||||
"strings"
|
||||
"time"
|
||||
|
||||
"github.com/sagernet/sing-box/shater/engine"
|
||||
"github.com/sagernet/sing-box/shater/model"
|
||||
"github.com/sagernet/sing-box/shater/netplane"
|
||||
)
|
||||
@@ -269,6 +271,50 @@ func untunnelablePolicyWarnings(g model.Globals, planNotes []string) []Warning {
|
||||
}
|
||||
}
|
||||
|
||||
// engineTeardownWarnings turns the engine's ABANDONED generations — superseded
|
||||
// sing-box instances whose shutdown overran the hard close budget and are still
|
||||
// running inside this process — into operator-facing warnings.
|
||||
//
|
||||
// Critical, without hesitation. A leaked generation is not untidiness:
|
||||
//
|
||||
// - it still holds its WireGuard devices, and two devices built from the same
|
||||
// private key evict each other at the peer (one session per public key), so
|
||||
// the leak reproduces BETWEEN generations exactly the fault
|
||||
// generate/wgdedup.go removes WITHIN a config — the tunnel flaps and neither
|
||||
// end can say why;
|
||||
// - it still holds its outbound connections and keeps probing nodes, so the
|
||||
// log fills with errors attributed to a config that is no longer applied;
|
||||
// - on a 512 MiB router each one costs real memory that is never returned.
|
||||
//
|
||||
// The generation number is carried in Name so two consecutive status reads can
|
||||
// tell "the same stuck generation" from "another one just leaked", and the
|
||||
// elapsed time is in the message because a shutdown at 8s and one at 40 minutes
|
||||
// are different problems.
|
||||
func engineTeardownWarnings(stuck []engine.StuckClose) []Warning {
|
||||
if len(stuck) == 0 {
|
||||
return nil
|
||||
}
|
||||
out := make([]Warning, 0, len(stuck))
|
||||
for _, s := range stuck {
|
||||
config := "unknown config"
|
||||
if len(s.Hash) >= 12 {
|
||||
config = "config " + s.Hash[:12]
|
||||
}
|
||||
out = append(out, Warning{
|
||||
Severity: SeverityCritical,
|
||||
Section: "engine",
|
||||
Name: fmt.Sprintf("generation %d", s.Generation),
|
||||
Message: fmt.Sprintf(
|
||||
"a superseded engine instance (%s) has been shutting down for %s and is STILL RUNNING: "+
|
||||
"it keeps its outbound connections and its WireGuard devices, so it can evict the "+
|
||||
"live tunnel at the peer and it keeps writing to the log. The current configuration "+
|
||||
"is applied and running; restart shaterd if this does not clear.",
|
||||
config, s.Elapsed.Round(time.Second)),
|
||||
})
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
func severityRank(s string) int {
|
||||
switch s {
|
||||
case SeverityCritical:
|
||||
|
||||
+126
-31
@@ -61,6 +61,25 @@ type Engine struct {
|
||||
current option.Options // options the running Box was built from
|
||||
hash string // stable hash of current (hex sha256 of canonical JSON)
|
||||
|
||||
// instanceCancel cancels the context THIS box was built on — its own
|
||||
// cancellable child of e.ctx, one per box (see newBox). box.New does not
|
||||
// derive a cancellable context of its own, so without this every goroutine
|
||||
// inside a retired box that waits on ctx.Done() waits forever. Upstream's
|
||||
// runner does exactly this and calls cancel before Close
|
||||
// (cmd/sing-box/cmd_run.go). nil when nothing is running.
|
||||
instanceCancel context.CancelFunc
|
||||
// instanceGen is the 1-based sequence number of the running box, bumped on
|
||||
// every adopted swap. It is what names a generation in the log and in
|
||||
// PendingCloses when a shutdown overruns its budget.
|
||||
instanceGen uint64
|
||||
|
||||
// pending holds the retirements in flight (see teardown.go). Its OWN leaf
|
||||
// lock, deliberately not mu: PendingCloses is read by the status path, and
|
||||
// the moment that read matters most is while an apply is holding mu waiting
|
||||
// out a shutdown that will not finish.
|
||||
pendingMu sync.Mutex
|
||||
pending []*pendingClose
|
||||
|
||||
// defaultLogWriter, when non-nil, is handed to EVERY box.New this engine
|
||||
// performs (box.Options.DefaultLogWriter): the daemon points it at its
|
||||
// long-lived logsink once, and each Apply-swapped box then logs into that
|
||||
@@ -174,6 +193,17 @@ func (e *Engine) Apply(opts option.Options) (changed bool, err error) {
|
||||
}
|
||||
|
||||
func (e *Engine) applyLocked(opts option.Options) (bool, error) {
|
||||
// Stand down the self-check of urltest groups no rule reaches (see
|
||||
// selfcheck.go for the whole argument). This MUTATES opts in place — the
|
||||
// option structs are pointers behind an `any` — and it must run BEFORE the
|
||||
// hash below, so the hash describes the config that is really built: a rule
|
||||
// change that flips a group used<->unused is then a real change that
|
||||
// triggers a swap, and an unchanged config hashes identically on every
|
||||
// reconcile because the stand-down is deterministic.
|
||||
if stood := standDownUnusedSelfCheck(opts); stood > 0 && e.log != nil {
|
||||
e.log.Info("apply: stood down self-check on ", stood, " unused urltest group(s); the observatory is their only prober")
|
||||
}
|
||||
|
||||
newHash, err := e.hashOptions(opts)
|
||||
if err != nil {
|
||||
return false, E.Cause(err, "hash options")
|
||||
@@ -184,12 +214,18 @@ func (e *Engine) applyLocked(opts option.Options) (bool, error) {
|
||||
return false, nil
|
||||
}
|
||||
|
||||
// (2) build + validate. box.New constructs and validates every adapter.
|
||||
nb, err := box.New(box.Options{
|
||||
Context: e.ctx,
|
||||
Options: opts,
|
||||
DefaultLogWriter: e.defaultLogWriter,
|
||||
})
|
||||
// (1b) Do not stack generations. A superseded box whose shutdown overran its
|
||||
// budget is still holding its outbound connections and its WireGuard devices;
|
||||
// building another one on top of it is how one process ends up carrying four
|
||||
// live generations. Give any abandoned teardown one more budget to finish
|
||||
// BEFORE we create anything (see teardown.go awaitAbandonedLocked). Placed
|
||||
// after the hash gate on purpose: a no-op reconcile — cron, every minute —
|
||||
// must stay free.
|
||||
stuck := e.awaitAbandonedLocked()
|
||||
|
||||
// (2) build + validate on its OWN cancellable context. box.New constructs and
|
||||
// validates every adapter.
|
||||
nb, nbCancel, err := e.newBox(opts)
|
||||
if err != nil {
|
||||
// Validation failed: keep the running instance, do not swap.
|
||||
return false, E.Cause(err, "create instance")
|
||||
@@ -211,8 +247,15 @@ func (e *Engine) applyLocked(opts option.Options) (bool, error) {
|
||||
// close-old-then-start-new DIRECTLY — skipping the stall. (box.New above
|
||||
// already validated opts, so we never tear the old box down for a config
|
||||
// that would fail to build.)
|
||||
if e.instance != nil && sharesCacheFileLock(e.current, opts) {
|
||||
return e.closeOldThenStart(nb, opts, newHash)
|
||||
//
|
||||
// A generation that is STILL abandoned after the wait above forces the same
|
||||
// path for a different reason: start-new-first would put a second LIVE box
|
||||
// alongside a third that refuses to die, all three contending for the same
|
||||
// tproxy port, the same cache file and — the expensive one — the same
|
||||
// WireGuard private keys. Closing the current box first keeps the process to
|
||||
// at most one live instance plus the abandoned one.
|
||||
if e.instance != nil && (stuck > 0 || sharesCacheFileLock(e.current, opts)) {
|
||||
return e.closeOldThenStart(nb, nbCancel, opts, newHash)
|
||||
}
|
||||
|
||||
// (3b) start-new-first (zero-downtime when there is no resource conflict).
|
||||
@@ -223,25 +266,59 @@ func (e *Engine) applyLocked(opts option.Options) (bool, error) {
|
||||
// close-old-then-start-new then. Any OTHER Start failure keeps today's
|
||||
// behavior: discard the new box, keep the old one running.
|
||||
if e.instance == nil || !isSwapConflict(err) {
|
||||
nbCancel()
|
||||
_ = nb.Close()
|
||||
return false, E.Cause(err, "start instance")
|
||||
}
|
||||
return e.closeOldThenStart(nb, opts, newHash)
|
||||
return e.closeOldThenStart(nb, nbCancel, opts, newHash)
|
||||
}
|
||||
|
||||
// Swap succeeded. The previously running config becomes last-good.
|
||||
old := e.instance
|
||||
if old != nil {
|
||||
if e.instance != nil {
|
||||
e.lastGood = e.current
|
||||
e.hasLastGood = true
|
||||
_ = old.Close()
|
||||
// Retire the old generation under the hard budget. Its error — including
|
||||
// ErrCloseTimeout — is deliberately NOT returned: the new box is started
|
||||
// and carrying traffic, so this apply SUCCEEDED, and failing it here would
|
||||
// abort the caller's netplane stage and leave a stale ruleset loaded over a
|
||||
// perfectly healthy engine. An abandoned generation is surfaced through
|
||||
// PendingCloses() (critical warning in `shaterd status` and the panel) and
|
||||
// an ERROR line naming it — visible, but not mistaken for a failed apply.
|
||||
_ = e.retireLocked(e.instance, e.instanceCancel, e.instanceGen, e.hash)
|
||||
}
|
||||
e.instance = nb
|
||||
e.current = opts
|
||||
e.hash = newHash
|
||||
e.adoptLocked(nb, nbCancel, opts, newHash)
|
||||
return true, nil
|
||||
}
|
||||
|
||||
// newBox builds a box on its OWN cancellable child of the engine context and
|
||||
// returns the cancel alongside it. Every goroutine the box starts inherits that
|
||||
// context, so cancelling it is what unwinds the ones Close does not reach; see
|
||||
// teardown.go for why the shared, never-cancelled context was the defect.
|
||||
func (e *Engine) newBox(opts option.Options) (*box.Box, context.CancelFunc, error) {
|
||||
ctx, cancel := context.WithCancel(e.ctx)
|
||||
b, err := box.New(box.Options{
|
||||
Context: ctx,
|
||||
Options: opts,
|
||||
DefaultLogWriter: e.defaultLogWriter,
|
||||
})
|
||||
if err != nil {
|
||||
cancel()
|
||||
return nil, nil, err
|
||||
}
|
||||
return b, cancel, nil
|
||||
}
|
||||
|
||||
// adoptLocked installs a started box as THE running instance and gives it the
|
||||
// next generation number. Caller holds e.mu and has already retired whatever was
|
||||
// running before.
|
||||
func (e *Engine) adoptLocked(b *box.Box, cancel context.CancelFunc, opts option.Options, hash string) {
|
||||
e.instanceGen++
|
||||
e.instance = b
|
||||
e.instanceCancel = cancel
|
||||
e.current = opts
|
||||
e.hash = hash
|
||||
}
|
||||
|
||||
// closeOldThenStart is the close-old-then-start-new swap. It is taken both
|
||||
// proactively (the incoming config shares the running box's cache_file lock, so
|
||||
// start-new-first cannot work) and reactively (start-new-first hit a swap
|
||||
@@ -256,27 +333,36 @@ func (e *Engine) applyLocked(opts option.Options) (bool, error) {
|
||||
// already validated by the caller's box.New, so the rebuild below is expected to
|
||||
// succeed; the restore path guards the rare case it does not. The caller holds
|
||||
// e.mu.
|
||||
func (e *Engine) closeOldThenStart(discard *box.Box, opts option.Options, newHash string) (bool, error) {
|
||||
func (e *Engine) closeOldThenStart(discard *box.Box, discardCancel context.CancelFunc, opts option.Options, newHash string) (bool, error) {
|
||||
// (a) Drop the pre-built box (a Start-failed box cannot be restarted, and the
|
||||
// proactively-built one must not hold the cache_file lock while we rebuild).
|
||||
// It was never adopted, so it is not a generation — close it inline, but
|
||||
// cancel its context first exactly like a retired one.
|
||||
if discardCancel != nil {
|
||||
discardCancel()
|
||||
}
|
||||
if discard != nil {
|
||||
_ = discard.Close()
|
||||
}
|
||||
|
||||
// (b) Free the port + cache_file lock by closing the old instance. Remember
|
||||
// its config so we can restore it if the fresh box cannot come up.
|
||||
// (b) Free the port + cache_file lock by closing the old instance, under the
|
||||
// hard budget. Remember its config so we can restore it if the fresh box
|
||||
// cannot come up. An overrun here is reported by retireLocked and tracked in
|
||||
// PendingCloses; we still proceed, because the resources it was supposed to
|
||||
// free are exactly what the operator is waiting on.
|
||||
prevOpts := e.current
|
||||
prevHash := e.hash
|
||||
prevLastGood := e.lastGood
|
||||
prevHasLastGood := e.hasLastGood
|
||||
_ = e.instance.Close()
|
||||
e.instance = nil
|
||||
_ = e.retireLocked(e.instance, e.instanceCancel, e.instanceGen, prevHash)
|
||||
e.instance, e.instanceCancel = nil, nil
|
||||
|
||||
// (c) Build a FRESH box for opts (the discarded one cannot be reused).
|
||||
nb2, err := box.New(box.Options{Context: e.ctx, Options: opts, DefaultLogWriter: e.defaultLogWriter})
|
||||
nb2, cancel2, err := e.newBox(opts)
|
||||
if err == nil {
|
||||
err = nb2.Start()
|
||||
if err != nil {
|
||||
cancel2()
|
||||
_ = nb2.Close()
|
||||
}
|
||||
}
|
||||
@@ -284,28 +370,25 @@ func (e *Engine) closeOldThenStart(discard *box.Box, opts option.Options, newHas
|
||||
// (d) Success: the old config we just closed becomes last-good.
|
||||
e.lastGood = prevOpts
|
||||
e.hasLastGood = true
|
||||
e.instance = nb2
|
||||
e.current = opts
|
||||
e.hash = newHash
|
||||
e.adoptLocked(nb2, cancel2, opts, newHash)
|
||||
return true, nil
|
||||
}
|
||||
|
||||
// (e) The fresh box could not come up and the old one is already closed —
|
||||
// interception is currently down. Try to RESTORE the previous config so we
|
||||
// do not leave the tunnel dead.
|
||||
rb, rerr := box.New(box.Options{Context: e.ctx, Options: prevOpts, DefaultLogWriter: e.defaultLogWriter})
|
||||
rb, rcancel, rerr := e.newBox(prevOpts)
|
||||
if rerr == nil {
|
||||
rerr = rb.Start()
|
||||
if rerr != nil {
|
||||
rcancel()
|
||||
_ = rb.Close()
|
||||
}
|
||||
}
|
||||
if rerr == nil {
|
||||
// Old config restored: keep current/hash/last-good exactly as they were
|
||||
// (do NOT advance them). Report that opts was not applied.
|
||||
e.instance = rb
|
||||
e.current = prevOpts
|
||||
e.hash = prevHash
|
||||
e.adoptLocked(rb, rcancel, prevOpts, prevHash)
|
||||
e.lastGood = prevLastGood
|
||||
e.hasLastGood = prevHasLastGood
|
||||
return false, E.Cause(err, "start instance (config not applied; previous config restored)")
|
||||
@@ -314,7 +397,7 @@ func (e *Engine) closeOldThenStart(discard *box.Box, opts option.Options, newHas
|
||||
// Restore ALSO failed: the engine is now STOPPED. e.instance stays nil (never
|
||||
// pointing at a closed box). The fail-closed nft kill-switch keeps the LAN
|
||||
// safe (no unproxied leak) even though interception is down.
|
||||
e.instance = nil
|
||||
e.instance, e.instanceCancel = nil, nil
|
||||
e.hash = ""
|
||||
return false, E.Cause(E.Errors(err, rerr), "start instance failed and could not restore previous config; engine stopped")
|
||||
}
|
||||
@@ -410,14 +493,26 @@ func (e *Engine) HasLastGood() bool {
|
||||
}
|
||||
|
||||
// Close stops the running instance, if any. It is idempotent.
|
||||
//
|
||||
// Unlike the swap path, this one DOES return ErrCloseTimeout: here the shutdown
|
||||
// is the whole operation, so "it did not stop" is the result, not a footnote. The
|
||||
// caller (apply.Teardown) records it as the teardown's error while still
|
||||
// completing the netplane teardown — the data plane must come down even when a
|
||||
// box will not.
|
||||
func (e *Engine) Close() error {
|
||||
e.mu.Lock()
|
||||
defer e.mu.Unlock()
|
||||
if e.instance == nil {
|
||||
// Nothing running, but a previously abandoned generation may still be:
|
||||
// give it a last budget so a teardown followed by a restart does not carry
|
||||
// the leak across.
|
||||
if e.awaitAbandonedLocked() > 0 {
|
||||
return ErrCloseTimeout
|
||||
}
|
||||
return nil
|
||||
}
|
||||
err := e.instance.Close()
|
||||
e.instance = nil
|
||||
err := e.retireLocked(e.instance, e.instanceCancel, e.instanceGen, e.hash)
|
||||
e.instance, e.instanceCancel = nil, nil
|
||||
e.hash = ""
|
||||
return err
|
||||
}
|
||||
|
||||
+203
-14
@@ -1,6 +1,7 @@
|
||||
package engine
|
||||
|
||||
import (
|
||||
"sort"
|
||||
"strings"
|
||||
|
||||
"github.com/sagernet/sing-box/adapter"
|
||||
@@ -320,11 +321,15 @@ func parseGroupCopyTag(group, tag string) (member string, ok bool) {
|
||||
return member, true
|
||||
}
|
||||
|
||||
// ChainHealth is one configured chain's reachability, mirroring GroupHealth.Used
|
||||
// for the chain card (plan §5.E): a chain no enabled rule routes through is never
|
||||
// probed — the observatory walks only reachable paths — and the panel renders it
|
||||
// "unused" rather than as a health problem. A chain has no membership counters: it
|
||||
// is a fixed path, and its end-to-end health is the exit test's job, not a roll-up.
|
||||
// ChainHealth is one configured chain's reachability plus its PER-HOP health,
|
||||
// mirroring GroupHealth.Used for the chain card (plan §5.E): a chain no enabled
|
||||
// rule routes through is never probed — the observatory walks only reachable
|
||||
// paths — and the panel renders it "unused" rather than as a health problem.
|
||||
// A chain has no membership counters of its own: it is a fixed path, and its
|
||||
// end-to-end health is the exit probe's job. What it DOES have is hops, and
|
||||
// since the observatory now probes every hop wrapper (probeplan.go walkDetour),
|
||||
// each hop's health is on the board and is projected here so an operator can
|
||||
// see WHICH hop died instead of only that the chain did.
|
||||
type ChainHealth struct {
|
||||
// Name is the chain's model name (config chain "chain:<name>"), what the Targets
|
||||
// page lists and what a rule targets.
|
||||
@@ -336,34 +341,218 @@ type ChainHealth struct {
|
||||
// when the observatory is disabled or not yet configured: no badge is better
|
||||
// than a wrong one (the same rule as GroupHealth.Used).
|
||||
Used bool `json:"used"`
|
||||
// Hops is the per-hop health readout, L1..Ln in wire order (see ChainHopHealth).
|
||||
// It is EMPTY for a chain the running box never materialised: a chain no rule
|
||||
// references is resolved lazily and never built, and a 1-hop chain without an
|
||||
// egress entry resolves straight to its target with no wrapper — in both cases
|
||||
// there are no "chain-<name>-h…" outbounds to project. An absent "hops" key
|
||||
// therefore means "nothing materialised to report on", NEVER "this chain has
|
||||
// no hops" — the model, not this projection, knows how many hops were
|
||||
// configured.
|
||||
Hops []ChainHopHealth `json:"hops,omitempty"`
|
||||
}
|
||||
|
||||
// ChainHealth reports the reachability (used/unused) of every named chain, one row
|
||||
// per name, in the order given. It is the chain analogue of GroupHealth.Used: a
|
||||
// chain the observatory's used-set does not cover is reported Used=false so the
|
||||
// panel can mark it "unused" instead of running an exit test against a path nothing
|
||||
// routes through.
|
||||
// ChainHopHealth is one hop of one materialised chain, as the health board saw
|
||||
// it — a projection, like everything in this file: nothing here dials.
|
||||
//
|
||||
// A NODE hop is a single measurement: the observatory dials the hop wrapper
|
||||
// "chain-<name>-h<i>", which pulls exactly the path prefix up to and including
|
||||
// this hop, so Total=1, the counters follow the hop's own state, DelayMs and
|
||||
// AgeSeconds are its own observation, and Selected is "" (a fixed hop selects
|
||||
// nothing).
|
||||
//
|
||||
// A GROUP hop rolls up its member copies "chain-<name>-h<i>-<member>", each of
|
||||
// which the observatory probes through its own prefix of the chain. The
|
||||
// counters obey the same invariants as GroupHealth — Tested == Alive+Dead and
|
||||
// Alive+Dead+Untested == Total — so the panel needs no arithmetic of its own.
|
||||
// State summarises them: "alive" when at least one member is alive (the hop can
|
||||
// carry traffic), "dead" when at least one was tested and none is alive (a
|
||||
// positive finding of a dead hop), "untested" when nothing was tested. Selected
|
||||
// is the node NAME the wrapper currently picks; DelayMs/AgeSeconds are the
|
||||
// SELECTED member's observation, or the freshest ALIVE member's when the
|
||||
// selection has no measurement of its own — the number shown must always be a
|
||||
// measurement somebody took, never an average nobody did.
|
||||
type ChainHopHealth struct {
|
||||
Index int `json:"index"` // 1-based position on the wire, L1..Ln
|
||||
Tag string `json:"tag"` // "chain-<name>-h<i>" — the wrapper actually dialled
|
||||
Kind string `json:"kind"` // "node" | "group"
|
||||
Exit bool `json:"exit"` // the LAST hop: where traffic leaves to the internet
|
||||
State string `json:"state"` // "alive" | "dead" | "untested"
|
||||
DelayMs int `json:"delay_ms"`
|
||||
AgeSeconds int64 `json:"age_seconds"` // -1 when unknown
|
||||
Selected string `json:"selected"` // group hop: the node NAME it currently selects; "" otherwise
|
||||
Total int `json:"total"`
|
||||
Tested int `json:"tested"`
|
||||
Alive int `json:"alive"`
|
||||
Dead int `json:"dead"`
|
||||
Untested int `json:"untested"`
|
||||
}
|
||||
|
||||
// ChainHealth reports the reachability (used/unused) of every named chain plus
|
||||
// the per-hop health of each one the running box materialised, one row per
|
||||
// name, in the order given. The Used half is the chain analogue of
|
||||
// GroupHealth.Used: a chain the observatory's used-set does not cover is
|
||||
// reported Used=false so the panel can mark it "unused" instead of a health
|
||||
// readout against a path nothing routes through.
|
||||
//
|
||||
// names come from the desired-state model, NOT the running box: a chain no rule
|
||||
// references is never materialised (generate/chain.go resolveChain is lazy), so it
|
||||
// is invisible to a box-only enumeration — yet the panel lists it from the config
|
||||
// and must be able to badge it. The engine supplies the only fact a box read can
|
||||
// add here, the observatory's published used-set. Pure apart from that read; nil
|
||||
// names or a stopped engine (nil used-set) yield an empty/used-everything result.
|
||||
// and must be able to badge it. The engine supplies what only it can: the
|
||||
// observatory's published used-set, and the running box's outbound/endpoint
|
||||
// pool the hop projection reads. Still no dialling anywhere; nil names or a
|
||||
// stopped engine (nil used-set, empty pool) yield an empty/used-everything
|
||||
// result with no hops.
|
||||
func (e *Engine) ChainHealth(names []string) []ChainHealth {
|
||||
out := make([]ChainHealth, 0, len(names))
|
||||
used := e.observatoryUsed()
|
||||
usedChains := usedChainNames(used)
|
||||
pool := e.runningPool()
|
||||
view := e.HealthView()
|
||||
for _, name := range names {
|
||||
name = strings.TrimSpace(name)
|
||||
if name == "" {
|
||||
continue
|
||||
}
|
||||
out = append(out, ChainHealth{Name: name, Used: used == nil || usedChains[name]})
|
||||
out = append(out, ChainHealth{
|
||||
Name: name,
|
||||
Used: used == nil || usedChains[name],
|
||||
Hops: chainHopHealthOf(pool, name, view),
|
||||
})
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
// runningPool snapshots the running box's outbounds AND endpoints into one
|
||||
// list. The endpoints matter: a chain hop rebuilt from a wireguard/AmneziaWG
|
||||
// node is an ENDPOINT copy, invisible in Outbounds() — the same trap
|
||||
// groupTargets documents — and a hop projection that missed it would silently
|
||||
// drop the very hop this feature exists to localise (the production chain's
|
||||
// first hop IS an AWG endpoint). Empty (never nil-unsafe) on a stopped engine.
|
||||
func (e *Engine) runningPool() []adapter.Outbound {
|
||||
var pool []adapter.Outbound
|
||||
inst := e.Instance()
|
||||
if inst == nil {
|
||||
return pool
|
||||
}
|
||||
if om := inst.Outbound(); om != nil {
|
||||
pool = append(pool, om.Outbounds()...)
|
||||
}
|
||||
if em := inst.Endpoint(); em != nil {
|
||||
for _, ep := range em.Endpoints() {
|
||||
pool = append(pool, ep)
|
||||
}
|
||||
}
|
||||
return pool
|
||||
}
|
||||
|
||||
// chainHopHealthOf projects one chain's hop wrappers out of an outbound pool
|
||||
// against one health view. Pure apart from the view reads — the same
|
||||
// unit-testing contract as groupHealthOf: hand it a fake pool and a hand-built
|
||||
// store and every branch is reachable without a box.
|
||||
//
|
||||
// A pool entry belongs to chain <name> when its tag is exactly
|
||||
// "chain-<name>-h<digits>" — the hop WRAPPER the observatory dials. Member
|
||||
// copies ("chain-<name>-h<i>-<member>") are not hops themselves; they are
|
||||
// reached through the wrapper's own member list (adapter.OutboundGroup.All), so
|
||||
// the roll-up sees exactly what the wrapper can select, in its order.
|
||||
func chainHopHealthOf(pool []adapter.Outbound, name string, view HealthView) []ChainHopHealth {
|
||||
prefix := "chain-" + name + "-h"
|
||||
var hops []ChainHopHealth
|
||||
for _, ob := range pool {
|
||||
tag := ob.Tag()
|
||||
rest, ok := strings.CutPrefix(tag, prefix)
|
||||
if !ok {
|
||||
continue
|
||||
}
|
||||
idx, ok := parseAllDigits(rest)
|
||||
if !ok {
|
||||
continue // a member copy, or another chain sharing the prefix
|
||||
}
|
||||
if g, isGroup := ob.(adapter.OutboundGroup); isGroup {
|
||||
hops = append(hops, chainGroupHop(g, name, idx, view))
|
||||
} else {
|
||||
hops = append(hops, chainNodeHop(tag, idx, view))
|
||||
}
|
||||
}
|
||||
sort.Slice(hops, func(i, j int) bool { return hops[i].Index < hops[j].Index })
|
||||
if len(hops) > 0 {
|
||||
// The largest index is the exit — the wrapper whose probe leaves to the
|
||||
// internet. Marked after sorting so the flag cannot depend on pool order.
|
||||
hops[len(hops)-1].Exit = true
|
||||
}
|
||||
return hops
|
||||
}
|
||||
|
||||
// chainNodeHop is the one-measurement hop: the wrapper itself was dialled by
|
||||
// the observatory, so its own board state IS the hop's health and the counters
|
||||
// degenerate to whichever bucket that state fills.
|
||||
func chainNodeHop(tag string, idx int, view HealthView) ChainHopHealth {
|
||||
hop := ChainHopHealth{Index: idx, Tag: tag, Kind: "node", Total: 1}
|
||||
hop.State, hop.DelayMs, hop.AgeSeconds = view.State(tag)
|
||||
switch hop.State {
|
||||
case HealthAlive:
|
||||
hop.Alive = 1
|
||||
case HealthDead:
|
||||
hop.Dead = 1
|
||||
default:
|
||||
hop.Untested = 1
|
||||
}
|
||||
hop.Tested = hop.Alive + hop.Dead
|
||||
return hop
|
||||
}
|
||||
|
||||
// chainGroupHop rolls a group hop up over its member copies. The counters carry
|
||||
// the GroupHealth invariants; the summary State answers the only question a hop
|
||||
// row asks — "can this hop carry the chain": alive while anything answers,
|
||||
// dead only on a positive all-tested-dead finding, untested when nothing is
|
||||
// known (never dead-by-absence, the same honesty rule as everywhere else).
|
||||
func chainGroupHop(g adapter.OutboundGroup, chain string, idx int, view HealthView) ChainHopHealth {
|
||||
hop := ChainHopHealth{Index: idx, Tag: g.Tag(), Kind: "group", AgeSeconds: -1}
|
||||
selected := g.Now()
|
||||
if selected != "" {
|
||||
hop.Selected = chainMemberName(selected, chain)
|
||||
}
|
||||
|
||||
// The number a hop row shows must be a real observation: the selected
|
||||
// member's when it has one, else the freshest alive member's.
|
||||
freshDelay, freshAge := 0, int64(-1)
|
||||
selDelay, selAge := 0, int64(-1)
|
||||
for _, tag := range g.All() {
|
||||
state, delayMs, age := view.State(tag)
|
||||
switch state {
|
||||
case HealthAlive:
|
||||
hop.Alive++
|
||||
if age >= 0 && (freshAge < 0 || age < freshAge) {
|
||||
freshDelay, freshAge = delayMs, age
|
||||
}
|
||||
case HealthDead:
|
||||
hop.Dead++
|
||||
default:
|
||||
hop.Untested++
|
||||
}
|
||||
hop.Total++
|
||||
if tag == selected && age >= 0 {
|
||||
selDelay, selAge = delayMs, age
|
||||
}
|
||||
}
|
||||
hop.Tested = hop.Alive + hop.Dead
|
||||
switch {
|
||||
case hop.Alive > 0:
|
||||
hop.State = HealthAlive
|
||||
case hop.Tested > 0:
|
||||
hop.State = HealthDead
|
||||
default:
|
||||
hop.State = HealthUntested
|
||||
}
|
||||
if selAge >= 0 {
|
||||
hop.DelayMs, hop.AgeSeconds = selDelay, selAge
|
||||
} else {
|
||||
hop.DelayMs, hop.AgeSeconds = freshDelay, freshAge
|
||||
}
|
||||
return hop
|
||||
}
|
||||
|
||||
// usedChainNames recovers the set of chain NAMES the observatory's used-set covers.
|
||||
//
|
||||
// The used-set is keyed by outbound TAG (the generator's schema, materialised in
|
||||
|
||||
@@ -309,8 +309,145 @@ func TestChainHealthNilUsedSet(t *testing.T) {
|
||||
t.Fatalf("ChainHealth = %+v, want %+v (nil used-set ⇒ all used, blanks dropped)", got, want)
|
||||
}
|
||||
for i, w := range want {
|
||||
if got[i] != w {
|
||||
if got[i].Name != w.Name || got[i].Used != w.Used {
|
||||
t.Errorf("ChainHealth[%d] = %+v, want %+v", i, got[i], w)
|
||||
}
|
||||
// A stopped engine materialised nothing: hops must be empty, and the
|
||||
// contract says empty means "nothing materialised", not "no hops".
|
||||
if len(got[i].Hops) != 0 {
|
||||
t.Errorf("ChainHealth[%d].Hops = %+v, want empty on a stopped engine", i, got[i].Hops)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestChainHopHealthProjection drives the per-hop readout over a 3-hop chain —
|
||||
// node hop L1, group hop L2, group hop L3 (the exit) — with exactly ONE hop's
|
||||
// members dead. The dead state must land on THAT hop's index and nowhere else,
|
||||
// the counters must obey the GroupHealth invariants on every hop, and the exit
|
||||
// flag must sit on the largest index. This is the "which hop died" question the
|
||||
// whole hop surface exists to answer.
|
||||
func TestChainHopHealthProjection(t *testing.T) {
|
||||
hist := urltest.NewHistoryStorage()
|
||||
now := time.Now()
|
||||
|
||||
// L1 (node hop wrapper): alive — the prefix up to hop 1 works.
|
||||
hist.StoreURLTestHistory("chain-c-h1", &adapter.URLTestHistory{LastOK: now.Add(-5 * time.Second), Delay: 40})
|
||||
// L2 (group hop): BOTH member copies dead — this is the hop that died.
|
||||
hist.StoreURLTestHistory("chain-c-h2-x", &adapter.URLTestHistory{LastFail: now.Add(-3 * time.Second)})
|
||||
hist.StoreURLTestHistory("chain-c-h2-y", &adapter.URLTestHistory{LastFail: now.Add(-2 * time.Second)})
|
||||
// L3 (group hop, exit): one member alive, one never measured.
|
||||
hist.StoreURLTestHistory("chain-c-h3-a", &adapter.URLTestHistory{LastOK: now.Add(-7 * time.Second), Delay: 200})
|
||||
// chain-c-h3-b: nothing at all.
|
||||
|
||||
pool := []adapter.Outbound{
|
||||
&depOutbound{failingOutbound{tag: "chain-c-h1"}, nil},
|
||||
&depGroup{fakeGroup{
|
||||
tag: "chain-c-h2", kind: C.TypeSelector,
|
||||
all: []string{"chain-c-h2-x", "chain-c-h2-y"},
|
||||
now: "chain-c-h2-x",
|
||||
}, []string{"chain-c-h2-x", "chain-c-h2-y"}},
|
||||
&depGroup{fakeGroup{
|
||||
tag: "chain-c-h3", kind: C.TypeSelector,
|
||||
all: []string{"chain-c-h3-a", "chain-c-h3-b"},
|
||||
now: "chain-c-h3-a",
|
||||
}, []string{"chain-c-h3-a", "chain-c-h3-b"}},
|
||||
// Noise the projection must ignore: a member copy is not a hop, another
|
||||
// chain's wrapper is not this chain's.
|
||||
&depOutbound{failingOutbound{tag: "chain-c-h2-x"}, nil},
|
||||
&depOutbound{failingOutbound{tag: "chain-other-h1"}, nil},
|
||||
}
|
||||
|
||||
hops := chainHopHealthOf(pool, "c", newHealthView(hist, healthTTLFloor, now))
|
||||
if len(hops) != 3 {
|
||||
t.Fatalf("got %d hops, want 3: %+v", len(hops), hops)
|
||||
}
|
||||
|
||||
// Ordered by index, exit on the largest.
|
||||
for i, wantIdx := range []int{1, 2, 3} {
|
||||
if hops[i].Index != wantIdx {
|
||||
t.Fatalf("hops out of order: %+v", hops)
|
||||
}
|
||||
if got, want := hops[i].Exit, wantIdx == 3; got != want {
|
||||
t.Errorf("hop %d Exit = %v, want %v", wantIdx, got, want)
|
||||
}
|
||||
}
|
||||
|
||||
h1, h2, h3 := hops[0], hops[1], hops[2]
|
||||
|
||||
// L1: one measurement, its own numbers, no selection.
|
||||
if h1.Kind != "node" || h1.State != HealthAlive || h1.Total != 1 || h1.Alive != 1 ||
|
||||
h1.DelayMs != 40 || h1.AgeSeconds != 5 || h1.Selected != "" {
|
||||
t.Errorf("h1 = %+v, want an alive node hop with its own 40ms/5s and no selection", h1)
|
||||
}
|
||||
|
||||
// L2: the dead hop. Both members tested, none alive => a POSITIVE dead
|
||||
// finding on exactly this index. The selected member (x) is dead, and a
|
||||
// dead observation carries delay 0 with the failure's age.
|
||||
if h2.Kind != "group" || h2.State != HealthDead {
|
||||
t.Fatalf("h2 = %+v, want the DEAD group hop — this is the answer to 'which hop died'", h2)
|
||||
}
|
||||
if h2.Total != 2 || h2.Tested != 2 || h2.Alive != 0 || h2.Dead != 2 || h2.Untested != 0 {
|
||||
t.Errorf("h2 counters = %+v, want 2 tested / 2 dead", h2)
|
||||
}
|
||||
if h2.Selected != "x" {
|
||||
t.Errorf("h2 Selected = %q, want the node NAME x", h2.Selected)
|
||||
}
|
||||
if h2.DelayMs != 0 || h2.AgeSeconds != 3 {
|
||||
t.Errorf("h2 delay/age = %d/%d, want 0/3 (the selected member's failure observation)", h2.DelayMs, h2.AgeSeconds)
|
||||
}
|
||||
|
||||
// L3: alive (one member answers), one member honestly untested — never
|
||||
// folded into dead. The selected member is the alive one, so its numbers show.
|
||||
if h3.Kind != "group" || h3.State != HealthAlive {
|
||||
t.Fatalf("h3 = %+v, want an alive exit hop", h3)
|
||||
}
|
||||
if h3.Total != 2 || h3.Tested != 1 || h3.Alive != 1 || h3.Dead != 0 || h3.Untested != 1 {
|
||||
t.Errorf("h3 counters = %+v, want 1 alive / 1 untested", h3)
|
||||
}
|
||||
if h3.Selected != "a" || h3.DelayMs != 200 || h3.AgeSeconds != 7 {
|
||||
t.Errorf("h3 = %+v, want selection a with its 200ms/7s", h3)
|
||||
}
|
||||
|
||||
// The invariants, on every hop, exactly as GroupHealth promises.
|
||||
for _, h := range hops {
|
||||
if h.Tested != h.Alive+h.Dead {
|
||||
t.Errorf("hop %d: Tested = %d, want alive+dead = %d", h.Index, h.Tested, h.Alive+h.Dead)
|
||||
}
|
||||
if h.Alive+h.Dead+h.Untested != h.Total {
|
||||
t.Errorf("hop %d: alive+dead+untested = %d, want total = %d",
|
||||
h.Index, h.Alive+h.Dead+h.Untested, h.Total)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// A group hop whose SELECTION has no measurement shows the freshest ALIVE
|
||||
// member's numbers instead — the shown number must always be an observation
|
||||
// somebody took.
|
||||
func TestChainHopSelectionWithoutMeasurementFallsBack(t *testing.T) {
|
||||
hist := urltest.NewHistoryStorage()
|
||||
now := time.Now()
|
||||
hist.StoreURLTestHistory("chain-c-h1-a", &adapter.URLTestHistory{LastOK: now.Add(-30 * time.Second), Delay: 90})
|
||||
hist.StoreURLTestHistory("chain-c-h1-b", &adapter.URLTestHistory{LastOK: now.Add(-4 * time.Second), Delay: 150})
|
||||
// The selection points at c, which has no observation at all.
|
||||
pool := []adapter.Outbound{
|
||||
&depGroup{fakeGroup{
|
||||
tag: "chain-c-h1", kind: C.TypeSelector,
|
||||
all: []string{"chain-c-h1-a", "chain-c-h1-b", "chain-c-h1-c"},
|
||||
now: "chain-c-h1-c",
|
||||
}, nil},
|
||||
}
|
||||
hops := chainHopHealthOf(pool, "c", newHealthView(hist, healthTTLFloor, now))
|
||||
if len(hops) != 1 {
|
||||
t.Fatalf("got %d hops, want 1", len(hops))
|
||||
}
|
||||
h := hops[0]
|
||||
if h.Selected != "c" {
|
||||
t.Errorf("Selected = %q, want c (the selection is reported even unmeasured)", h.Selected)
|
||||
}
|
||||
if h.DelayMs != 150 || h.AgeSeconds != 4 {
|
||||
t.Errorf("delay/age = %d/%d, want the freshest ALIVE member's 150/4", h.DelayMs, h.AgeSeconds)
|
||||
}
|
||||
if h.State != HealthAlive || h.Alive != 2 || h.Untested != 1 {
|
||||
t.Errorf("hop = %+v, want alive with 2 alive / 1 untested", h)
|
||||
}
|
||||
}
|
||||
+228
-59
@@ -12,33 +12,46 @@ import (
|
||||
"time"
|
||||
|
||||
"github.com/sagernet/sing-box/adapter"
|
||||
"github.com/sagernet/sing-box/common/urltest"
|
||||
C "github.com/sagernet/sing-box/constant"
|
||||
)
|
||||
|
||||
// Exit test — "what am I actually going out through, and how fast" (F2, plan §5.E).
|
||||
// Group/chain test — "ask the observatory to refresh, then report what it
|
||||
// measured, plus the exit address" (F2, plan §5.E; reworked under the one-probe
|
||||
// rule).
|
||||
//
|
||||
// # Why this is not just another node probe
|
||||
// # This file no longer measures latency. On purpose.
|
||||
//
|
||||
// The observatory (observatory.go) answers "which nodes are alive". It cannot
|
||||
// answer the question an operator actually asks after switching a group: *through
|
||||
// which address am I leaving the country right now*. A group is an indirection —
|
||||
// a selector or urltest over many members — so the delay of the group and the
|
||||
// public address it exits from are properties of the CURRENT selection, not of
|
||||
// any node the panel can point at. A CHAIN is the same question over a longer
|
||||
// path: its exit wrapper "chain-<name>-hN" tunnels through every hop, so dialling
|
||||
// it measures the whole L1..Ln path end to end.
|
||||
// It used to: one urltest.URLTest per target, dialled right here. That was the
|
||||
// defect, not a feature. The observatory (observatory.go) probes every path the
|
||||
// routing rules actually use — per-chain member copies, egress-bound group
|
||||
// copies, chain exits — and it is the ONLY thing allowed to dial for health.
|
||||
// A second dial path from this file measured the WRONG thing: pressing "Test"
|
||||
// dialled the base groups directly from the router over the default WAN, a path
|
||||
// no rule routes through, and (worse) URLTest's DialContext touched the group,
|
||||
// arming its own 30-minute probing ticker, which kept writing those false
|
||||
// direct measurements under the members' base tags. For a node that is blocked
|
||||
// on the direct WAN and alive only behind a tunnel, that reading is not stale —
|
||||
// it is FALSE, and it poisoned selection and the panel alike.
|
||||
//
|
||||
// So one test per target (group or chain) measures two things:
|
||||
// So a manual test is now a READ with a refresh request in front of it:
|
||||
//
|
||||
// delay_ms — reusing urltest.URLTest, the SAME primitive the observatory uses
|
||||
// for a node. There is deliberately no second latency mechanism
|
||||
// here: the target is just an outbound, and probing it exercises
|
||||
// exactly the path traffic will take.
|
||||
// exit_ip — a real HTTP request THROUGH the target to a service that echoes
|
||||
// the client address back. Nothing else can produce this number: the
|
||||
// router cannot know its own public address, and the proxy protocol
|
||||
// does not report it.
|
||||
// delay_ms / ok — the observatory's OWN measurement of the target's dial
|
||||
// path. TestGroups rewinds the observatory (RefreshObservatory)
|
||||
// so the numbers are fresh — taken AFTER the button press —
|
||||
// then each target waits for an observation newer than the
|
||||
// run's start instant and reports it verbatim. tested_unix is
|
||||
// the instant that observation was taken, not "now".
|
||||
// exit_ip — a real HTTP request THROUGH the target to a service that
|
||||
// echoes the client address back. This is the ONE connection
|
||||
// this file still opens, and it stays because no probe can
|
||||
// answer it: the router cannot know its own public address,
|
||||
// and the proxy protocol does not report it. It is NOT a
|
||||
// second health probe — it travels the target's own routed
|
||||
// path (the same outbound object the rules dial), and it is
|
||||
// only ever issued for a target that (a) the observatory's
|
||||
// used-set covers and (b) just resolved ALIVE. A target
|
||||
// outside the rules, or one whose path is down, gets an empty
|
||||
// exit_ip — never a connection.
|
||||
//
|
||||
// # The direct-egress trap
|
||||
//
|
||||
@@ -51,16 +64,15 @@ import (
|
||||
// exit_ip instead of somebody else's number.
|
||||
|
||||
const (
|
||||
// groupTestDelayTimeout bounds the latency probe of one group.
|
||||
groupTestDelayTimeout = 5 * time.Second
|
||||
// groupTestExitTimeout bounds the whole exit-address lookup for one group
|
||||
// (connect through the tunnel + TLS + response). Short on purpose: this runs
|
||||
// while a human waits on a panel button, and a slow answer is worth less than a
|
||||
// prompt "could not determine".
|
||||
groupTestExitTimeout = 6 * time.Second
|
||||
// groupTestConcurrency bounds how many groups are tested in parallel. Small:
|
||||
// every one of these opens a real tunnelled connection, and a router with a
|
||||
// handful of groups is the normal case.
|
||||
// groupTestConcurrency bounds how many EXIT-ADDRESS lookups run in parallel.
|
||||
// Small: each one opens a real tunnelled connection, and a router with a
|
||||
// handful of groups is the normal case. The board polling itself is not
|
||||
// bounded by this — it is sleep-and-read, no I/O.
|
||||
groupTestConcurrency = 4
|
||||
// exitBodyLimit caps what is read from the exit-address service. These responses
|
||||
// are a few hundred bytes; anything larger is a hijacked/captive-portal answer
|
||||
@@ -68,19 +80,58 @@ const (
|
||||
exitBodyLimit = 4 << 10
|
||||
)
|
||||
|
||||
// groupTestWaitDeadline / groupTestPollEvery pace the wait for the observatory:
|
||||
// each covered target polls the health board once a second until its measured
|
||||
// tag carries an observation newer than the run's start, giving up after the
|
||||
// deadline. 120s covers a forced pass of a large plan (batches chain
|
||||
// back-to-back during a forced pass, see observatoryTickOnce) with room for the
|
||||
// probe timeouts of a mostly-dead population. Package variables, not constants,
|
||||
// for exactly one reason: the timeout tests must not take two minutes.
|
||||
var (
|
||||
groupTestWaitDeadline = 120 * time.Second
|
||||
groupTestPollEvery = time.Second
|
||||
)
|
||||
|
||||
// The fixed result texts for the targets that are never (or not yet) measured.
|
||||
// They are contract, not decoration — the panel shows them verbatim.
|
||||
const (
|
||||
// groupTestErrNotRouted: the target is outside the observatory's used-set,
|
||||
// so no rule routes through it and nothing measures it. Dialling it anyway
|
||||
// would recreate the false direct measurement this rework removed.
|
||||
groupTestErrNotRouted = "not routed by any enabled rule, so nothing measures it — the observatory only probes paths the rules use"
|
||||
// groupTestErrNotReached: the deadline passed without a fresh observation.
|
||||
groupTestErrNotReached = "the observatory has not reached this target yet — it refreshes on the global probe interval"
|
||||
// groupTestErrProbingOff: the observatory is disabled (the GroupHealth
|
||||
// master switch), so a refresh request has nothing to wake and waiting for
|
||||
// the deadline would just delay the same answer by two minutes.
|
||||
groupTestErrProbingOff = "background probing is disabled, so there is nothing to measure this target with"
|
||||
// groupTestErrPathDead: the fresh observation exists and it is a FAILURE —
|
||||
// the observatory probed the target's path after the button press and the
|
||||
// path did not answer. An honest negative, not a missing measurement.
|
||||
groupTestErrPathDead = "the observatory's probe through this path failed"
|
||||
)
|
||||
|
||||
// GroupTestResult is one target's test outcome — a group's or a chain's (Group
|
||||
// then carries the chain's model name). The JSON tags are the panel contract —
|
||||
// see the shater API docs for /api/groups/test.
|
||||
// see the shater API docs for /api/groups/test. The field names and types are
|
||||
// FROZEN; what changed in the rework is where the numbers come from.
|
||||
//
|
||||
// DelayMs and OK are a READ of the observatory's measurement of the target's
|
||||
// dial path — not a fresh dial performed by this file. OK is true exactly when
|
||||
// the health board's state for the measured tag is alive; DelayMs is that
|
||||
// observation's RTT; TestedUnix is the instant the OBSERVATION was taken (now
|
||||
// minus its age), so a result honestly says how old its number is instead of
|
||||
// stamping the poll time over it.
|
||||
//
|
||||
// Selected is the group's current pick (OutboundGroup.Now()); for a chain it is
|
||||
// the node NAME the chain's last group hop currently selects, "" when the chain
|
||||
// has no group hop (a fixed path selects nothing).
|
||||
//
|
||||
// OK reports whether the LATENCY measurement succeeded, which is the test's primary
|
||||
// question. A failed exit-address lookup deliberately does NOT clear it: knowing the
|
||||
// target is up and fast is useful on its own, and a probe service being unreachable
|
||||
// says nothing about the tunnel. In that case OK stays true and ExitIP is empty —
|
||||
// "not determined", never a guess and never somebody else's address.
|
||||
// A failed exit-address lookup deliberately does NOT clear OK: knowing the
|
||||
// target is up and fast is useful on its own, and a probe service being
|
||||
// unreachable says nothing about the tunnel. In that case OK stays true and
|
||||
// ExitIP is empty — "not determined", never a guess and never somebody else's
|
||||
// address.
|
||||
type GroupTestResult struct {
|
||||
Group string `json:"group"`
|
||||
Selected string `json:"selected"`
|
||||
@@ -259,19 +310,30 @@ func parseAllDigits(s string) (int, bool) {
|
||||
// and chains (empty/nil = every group and every chain in the running box),
|
||||
// returning started=false when a run is already in flight.
|
||||
//
|
||||
// It is a SINGLETON: a second request while a run is in flight is refused rather
|
||||
// than queued or run in parallel, because these runs open real tunnelled
|
||||
// connections and a panel that double-fires a button must not multiply the load
|
||||
// on the uplink. The observatory checks the same guard and skips its tick while
|
||||
// a run is in flight (observatory.go), so a manual test never competes with
|
||||
// background probing for the uplink.
|
||||
// It does NOT probe. It records the run's start instant, asks the observatory
|
||||
// for an out-of-turn full pass (RefreshObservatory), and then each target waits
|
||||
// for the health board to carry an observation NEWER than that instant on the
|
||||
// tag the observatory actually measures for it — see testOneTarget. The only
|
||||
// connection a run may still open is the exit-address lookup, and only for a
|
||||
// target that resolved alive on a rule-routed path.
|
||||
//
|
||||
// probeURL is the latency-probe URL; "" falls back to urltest's gstatic default.
|
||||
// It is a SINGLETON: a second request while a run is in flight is refused
|
||||
// rather than queued, because two overlapping runs would each rewind the
|
||||
// observatory's cursor and neither pass would ever complete — and the panel's
|
||||
// progress contract assumes one run's counters at a time anyway.
|
||||
//
|
||||
// Apply-swap safety: the target outbounds are snapshotted up front, so a config swap
|
||||
// mid-run cannot change what is being tested. A target torn down mid-run simply fails
|
||||
// its probe and is reported not-ok.
|
||||
// probeURL is accepted and IGNORED. The probe URL is a global observatory
|
||||
// setting now (ObservatoryConfig.ProbeURL, set at apply time); a per-run URL
|
||||
// would mean this run measures something different from what the board holds,
|
||||
// which is exactly the two-instruments split the rework removed. The parameter
|
||||
// stays so the callers (shater/apply, shater/panel) keep compiling and the
|
||||
// control-plane API shape does not churn.
|
||||
//
|
||||
// Apply-swap safety: the target outbounds are snapshotted up front, so a config
|
||||
// swap mid-run cannot change what is being tested. A target torn down mid-run
|
||||
// simply never receives a fresh observation and resolves on the deadline.
|
||||
func (e *Engine) TestGroups(names []string, probeURL string) (started bool) {
|
||||
_ = probeURL // ignored — see the doc comment above
|
||||
if !e.groupTestRunning.CompareAndSwap(false, true) {
|
||||
return false
|
||||
}
|
||||
@@ -285,6 +347,13 @@ func (e *Engine) TestGroups(names []string, probeURL string) (started bool) {
|
||||
// GroupTestStatus for why the scope is a set of names.
|
||||
e.setGroupTestScope(scopeOf(targets, missing))
|
||||
|
||||
// The freshness watermark: only an observation taken AFTER this instant may
|
||||
// answer this run. Recorded BEFORE the refresh request so a probe that lands
|
||||
// between the two can never be missed, only double-counted as fresh — the
|
||||
// harmless direction.
|
||||
t0 := time.Now()
|
||||
e.RefreshObservatory()
|
||||
|
||||
go func() {
|
||||
defer e.groupTestRunning.Store(false)
|
||||
|
||||
@@ -303,15 +372,22 @@ func (e *Engine) TestGroups(names []string, probeURL string) (started bool) {
|
||||
e.setGroupTestResults(results)
|
||||
e.groupTestDone.Add(int64(len(missing)))
|
||||
|
||||
// The used-set and the enabled bit are snapshotted once for the whole
|
||||
// run: they only change on an apply, and a run that straddles an apply
|
||||
// is already best-effort (see the swap-safety note above).
|
||||
used := e.observatoryUsed()
|
||||
obsEnabled, _, _ := e.ObservatoryStatus()
|
||||
|
||||
// One goroutine per target: they spend their life sleeping on the board
|
||||
// poll, so there is nothing to bound — the semaphore below bounds the
|
||||
// exit-address lookups, the only real connections left in a run.
|
||||
sem := make(chan struct{}, groupTestConcurrency)
|
||||
var wg sync.WaitGroup
|
||||
for i, tgt := range targets {
|
||||
wg.Add(1)
|
||||
sem <- struct{}{}
|
||||
go func(i int, tgt groupTestTarget) {
|
||||
defer wg.Done()
|
||||
defer func() { <-sem }()
|
||||
res := e.testOneTarget(tgt, probeURL)
|
||||
res := e.testOneTarget(tgt, used, obsEnabled, t0, sem)
|
||||
e.storeGroupTestResult(i, res)
|
||||
e.groupTestDone.Add(1)
|
||||
}(i, tgt)
|
||||
@@ -393,29 +469,122 @@ func (e *Engine) groupTargets(names []string) (targets []groupTestTarget, missin
|
||||
return targets, missing
|
||||
}
|
||||
|
||||
// testOneTarget measures one target: what it currently selects, the latency
|
||||
// through it, and the public address it exits from. For a chain the dialled
|
||||
// outbound is the exit wrapper, so the delay and the exit address are end-to-end
|
||||
// properties of the whole L1..Ln path.
|
||||
func (e *Engine) testOneTarget(t groupTestTarget, probeURL string) GroupTestResult {
|
||||
// testOneTarget resolves one target WITHOUT probing it: it decides whether the
|
||||
// observatory measures this target at all, and if so waits for a fresh
|
||||
// observation and reports it. The three ways out, in order:
|
||||
//
|
||||
// 1. the used-set does not cover the target — no enabled rule routes through
|
||||
// it, so nothing measures it and nothing SHOULD: resolved immediately with
|
||||
// groupTestErrNotRouted, never dialled, empty exit address. A nil used-set
|
||||
// means "unknown" (observatory not yet configured) and is treated as
|
||||
// covered — no refusal is better than a wrong one;
|
||||
// 2. the observatory is disabled — the refresh request went nowhere, so the
|
||||
// wait below could only ever end on its deadline: resolved immediately
|
||||
// with groupTestErrProbingOff instead of stalling the panel for two
|
||||
// minutes to say the same thing;
|
||||
// 3. covered and enabled — poll the health board once a second until the
|
||||
// MEASURED TAG carries an observation newer than since, then report that
|
||||
// observation verbatim (readFreshObservation). On the deadline:
|
||||
// groupTestErrNotReached.
|
||||
//
|
||||
// The exit-address lookup runs ONLY on the alive path of (3) — a rule-routed
|
||||
// target whose path just answered a probe — and through sem, so a run never
|
||||
// opens more than groupTestConcurrency tunnelled connections at once.
|
||||
func (e *Engine) testOneTarget(t groupTestTarget, used map[string]bool, obsEnabled bool, since time.Time, sem chan struct{}) GroupTestResult {
|
||||
res := GroupTestResult{Group: t.name, TestedUnix: time.Now().Unix()}
|
||||
if t.sel != nil {
|
||||
res.Selected = t.sel()
|
||||
}
|
||||
|
||||
ctx, cancel := context.WithTimeout(context.Background(), groupTestDelayTimeout)
|
||||
delay, err := urltest.URLTest(ctx, probeURL, t.ob)
|
||||
cancel()
|
||||
if err != nil {
|
||||
res.Error = err.Error()
|
||||
if used != nil && !used[t.ob.Tag()] {
|
||||
res.Error = groupTestErrNotRouted
|
||||
return res
|
||||
}
|
||||
if !obsEnabled {
|
||||
res.Error = groupTestErrProbingOff
|
||||
return res
|
||||
}
|
||||
res.OK = true
|
||||
res.DelayMs = int(delay)
|
||||
|
||||
// The exit address is best-effort by design: see GroupTestResult.OK.
|
||||
res.ExitIP, res.ExitCountry = e.exitAddress(t.ob)
|
||||
return res
|
||||
deadline := time.Now().Add(groupTestWaitDeadline)
|
||||
for {
|
||||
if e.readFreshObservation(&res, t, since) {
|
||||
if res.OK {
|
||||
// The exit address is best-effort by design: see GroupTestResult.
|
||||
sem <- struct{}{}
|
||||
res.ExitIP, res.ExitCountry = e.exitAddress(t.ob)
|
||||
<-sem
|
||||
}
|
||||
return res
|
||||
}
|
||||
if !time.Now().Before(deadline) {
|
||||
res.Error = groupTestErrNotReached
|
||||
return res
|
||||
}
|
||||
time.Sleep(groupTestPollEvery)
|
||||
}
|
||||
}
|
||||
|
||||
// measuredTagOf is the tag the observatory actually probes for this target's
|
||||
// dial path — the tag whose board entry answers "how is this target doing":
|
||||
//
|
||||
// - a GROUP target dials whatever it currently selects, so the group's health
|
||||
// IS its selection's health: OutboundGroup.Now(). For an egress-bound group
|
||||
// that is a per-group copy tag, which is exactly what the plan probes;
|
||||
// - a CHAIN whose exit wrapper is a PLAIN outbound is probed end-to-end under
|
||||
// that wrapper tag (probeplan.go: a chain exit is its own measurement);
|
||||
// - a CHAIN whose exit wrapper is a GROUP (the last hop is a group) has its
|
||||
// member copies probed instead of the wrapper, so the wrapper's Now() — the
|
||||
// member copy the chain currently dials through — is the measured tag. The
|
||||
// wrapper here IS the last group hop, so this is chainLastGroupHop's Now()
|
||||
// without a second lookup;
|
||||
// - fallback: the target's own tag, for a group that has not selected yet
|
||||
// (cold selector mid-swap). Its board entry is almost certainly empty, and
|
||||
// the caller then honestly reports "not reached" rather than inventing one.
|
||||
func measuredTagOf(t groupTestTarget) string {
|
||||
if g, ok := t.ob.(adapter.OutboundGroup); ok {
|
||||
if now := g.Now(); now != "" {
|
||||
return now
|
||||
}
|
||||
}
|
||||
return t.ob.Tag()
|
||||
}
|
||||
|
||||
// readFreshObservation reads the board once: if the target's measured tag holds
|
||||
// an observation taken at or after since, it is written into res (state, delay,
|
||||
// the observation's own timestamp, and the current selection so Selected and
|
||||
// the measurement describe the same pick) and true is returned. Otherwise res
|
||||
// is left for the next poll.
|
||||
//
|
||||
// Precision note: the board reports ages in whole seconds, so "at or after
|
||||
// since" is accurate to one second — an observation taken up to a second
|
||||
// BEFORE the refresh can slip through as fresh. That is the acceptable
|
||||
// direction: it is still a real measurement of the same path, at most a second
|
||||
// older than requested; the strict direction (discarding genuinely fresh
|
||||
// observations) would make every run one probe interval slower for nothing.
|
||||
func (e *Engine) readFreshObservation(res *GroupTestResult, t groupTestTarget, since time.Time) bool {
|
||||
// Re-read the selection at every poll: a forced observatory pass is exactly
|
||||
// the kind of event that makes a urltest group switch members, and the
|
||||
// measurement below is taken against the CURRENT pick.
|
||||
if t.sel != nil {
|
||||
res.Selected = t.sel()
|
||||
}
|
||||
tag := measuredTagOf(t)
|
||||
now := time.Now()
|
||||
state, delayMs, age := e.HealthView().State(tag)
|
||||
if age < 0 {
|
||||
return false // no observation at all (untested)
|
||||
}
|
||||
observedAt := now.Add(-time.Duration(age) * time.Second)
|
||||
if observedAt.Before(since) {
|
||||
return false // an old reading; the refresh has not reached this tag yet
|
||||
}
|
||||
res.OK = state == HealthAlive
|
||||
res.DelayMs = delayMs
|
||||
res.TestedUnix = observedAt.Unix()
|
||||
if !res.OK {
|
||||
res.Error = groupTestErrPathDead
|
||||
}
|
||||
return true
|
||||
}
|
||||
|
||||
// exitProbe is one exit-address service: a URL and the parser for its body.
|
||||
|
||||
+142
-15
@@ -3,6 +3,7 @@ package engine
|
||||
import (
|
||||
"context"
|
||||
"crypto/tls"
|
||||
"fmt"
|
||||
"net"
|
||||
"net/http"
|
||||
"net/http/httptest"
|
||||
@@ -143,10 +144,10 @@ func TestChainMemberName(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
// dialableOutbound routes every dial to a fixed local address — a stand-in for a
|
||||
// chain exit whose whole path is up. urltest.URLTest dials the probe URL's host
|
||||
// through the outbound, so pointing every dial at a local HTTP server makes the
|
||||
// latency probe succeed without any network.
|
||||
// dialableOutbound routes every dial to a fixed local address — the sink the
|
||||
// exit-address lookup lands in during tests. Since the rework nothing else in
|
||||
// this file dials at all; the local server refuses to be an exit-address
|
||||
// service (wrong status, no TLS), so the lookup honestly comes back empty.
|
||||
type dialableOutbound struct {
|
||||
failingOutbound
|
||||
addr string
|
||||
@@ -158,20 +159,42 @@ func (d *dialableOutbound) DialContext(ctx context.Context, network string, _ M.
|
||||
return (&net.Dialer{}).DialContext(ctx, network, d.addr)
|
||||
}
|
||||
|
||||
// TestChainExitTestMeasuresEndToEnd is the §6-S4 acceptance path for chains: the
|
||||
// exit test dials the chain's EXIT TAG, returns a measured delay, carries the
|
||||
// chain's model name (not the wrapper tag) as the result's Group, and reports the
|
||||
// last group hop's pick as Selected. The exit address is measured through the
|
||||
// same outbound (unreachable from a test => empty, never a guess).
|
||||
func TestChainExitTestMeasuresEndToEnd(t *testing.T) {
|
||||
probe := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
||||
// countingOutbound counts every DialContext/ListenPacket. It is the tripwire of
|
||||
// the rework: a target the observatory does not cover must NEVER be dialled,
|
||||
// and the counter is the proof.
|
||||
type countingOutbound struct {
|
||||
failingOutbound
|
||||
dials int
|
||||
}
|
||||
|
||||
func (c *countingOutbound) DialContext(ctx context.Context, network string, dst M.Socksaddr) (net.Conn, error) {
|
||||
c.dials++
|
||||
return nil, fmt.Errorf("dial refused (test)")
|
||||
}
|
||||
|
||||
func (c *countingOutbound) ListenPacket(ctx context.Context, dst M.Socksaddr) (net.PacketConn, error) {
|
||||
c.dials++
|
||||
return nil, fmt.Errorf("dial refused (test)")
|
||||
}
|
||||
|
||||
// groupTestSem is a fresh exit-lookup semaphore for direct testOneTarget calls.
|
||||
func groupTestSem() chan struct{} { return make(chan struct{}, groupTestConcurrency) }
|
||||
|
||||
// TestChainTestReportsObservatoryMeasurement is the §6-S4 acceptance path for
|
||||
// chains under the one-probe rule: the manual test does NOT dial the exit — it
|
||||
// reads the observatory's board entry for the exit tag, reports its delay and
|
||||
// ITS timestamp, carries the chain's model name as Group and the last group
|
||||
// hop's pick as Selected. The exit-address lookup still travels the exit
|
||||
// outbound (unreachable from a test => empty, never a guess).
|
||||
func TestChainTestReportsObservatoryMeasurement(t *testing.T) {
|
||||
sink := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
||||
w.WriteHeader(http.StatusNoContent)
|
||||
}))
|
||||
defer probe.Close()
|
||||
defer sink.Close()
|
||||
|
||||
exit := &dialableOutbound{
|
||||
failingOutbound: failingOutbound{tag: "chain-x-h2"},
|
||||
addr: probe.Listener.Addr().String(),
|
||||
addr: sink.Listener.Addr().String(),
|
||||
deps: []string{"chain-x-h1"},
|
||||
}
|
||||
pool := []adapter.Outbound{
|
||||
@@ -187,14 +210,31 @@ func TestChainExitTestMeasuresEndToEnd(t *testing.T) {
|
||||
if len(targets) != 1 || targets[0].ob.Tag() != "chain-x-h2" {
|
||||
t.Fatalf("targets = %+v, want the chain dialled via its exit tag", targets)
|
||||
}
|
||||
if got := measuredTagOf(targets[0]); got != "chain-x-h2" {
|
||||
t.Fatalf("measuredTagOf = %q, want the plain exit wrapper itself", got)
|
||||
}
|
||||
|
||||
e := New()
|
||||
res := e.testOneTarget(targets[0], probe.URL)
|
||||
// The observatory measured the exit 10 seconds ago, AFTER the run's start
|
||||
// instant below: that observation — delay and timestamp both — is the answer.
|
||||
observedAt := time.Now().Add(-10 * time.Second)
|
||||
e.URLTestHistory().StoreURLTestHistory("chain-x-h2", &adapter.URLTestHistory{LastOK: observedAt, Delay: 77})
|
||||
since := time.Now().Add(-time.Minute)
|
||||
|
||||
res := e.testOneTarget(targets[0], map[string]bool{"chain-x-h2": true}, true, since, groupTestSem())
|
||||
if res.Group != "x" {
|
||||
t.Errorf("Group = %q, want the chain's model name x", res.Group)
|
||||
}
|
||||
if !res.OK || res.Error != "" {
|
||||
t.Fatalf("result = %+v, want a successful measurement through the exit tag (a local roundtrip may legitimately read 0ms)", res)
|
||||
t.Fatalf("result = %+v, want the board's alive observation reported", res)
|
||||
}
|
||||
if res.DelayMs != 77 {
|
||||
t.Errorf("DelayMs = %d, want the observatory's 77 — this file measures nothing itself", res.DelayMs)
|
||||
}
|
||||
// tested_unix is the OBSERVATION's instant (whole-second precision), never
|
||||
// the poll's.
|
||||
if got, want := res.TestedUnix, observedAt.Unix(); got < want-1 || got > want+1 {
|
||||
t.Errorf("TestedUnix = %d, want the observation's instant ~%d", got, want)
|
||||
}
|
||||
if res.Selected != "relay" {
|
||||
t.Errorf("Selected = %q, want the last group hop's pick relay", res.Selected)
|
||||
@@ -204,6 +244,93 @@ func TestChainExitTestMeasuresEndToEnd(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
// TestTestOneTargetUnroutedNeverDialled is the rework's core promise: a target
|
||||
// outside the observatory's used-set is resolved immediately with the
|
||||
// documented explanation and ZERO dials — no latency probe, no exit-address
|
||||
// lookup, nothing. Dialling it would manufacture exactly the false direct
|
||||
// measurement the rework removed.
|
||||
func TestTestOneTargetUnroutedNeverDialled(t *testing.T) {
|
||||
e := New()
|
||||
ob := &countingOutbound{failingOutbound: failingOutbound{tag: "idle"}}
|
||||
tgt := groupTestTarget{name: "idle", ob: ob}
|
||||
|
||||
res := e.testOneTarget(tgt, map[string]bool{"something-else": true}, true, time.Now(), groupTestSem())
|
||||
if ob.dials != 0 {
|
||||
t.Fatalf("an unrouted target was dialled %d time(s); it must never be", ob.dials)
|
||||
}
|
||||
if res.OK {
|
||||
t.Fatal("an unrouted target reported ok=true")
|
||||
}
|
||||
if res.Error != groupTestErrNotRouted {
|
||||
t.Fatalf("Error = %q, want the not-routed explanation", res.Error)
|
||||
}
|
||||
if res.ExitIP != "" || res.ExitCountry != "" {
|
||||
t.Fatalf("an unrouted target must carry no exit address: %+v", res)
|
||||
}
|
||||
}
|
||||
|
||||
// A disabled observatory resolves covered targets immediately too: the refresh
|
||||
// request went nowhere, so waiting out the deadline would only delay the same
|
||||
// honest answer — and still, nothing is dialled.
|
||||
func TestTestOneTargetProbingOffNeverDialled(t *testing.T) {
|
||||
e := New()
|
||||
ob := &countingOutbound{failingOutbound: failingOutbound{tag: "auto"}}
|
||||
tgt := groupTestTarget{name: "auto", ob: ob}
|
||||
|
||||
res := e.testOneTarget(tgt, nil, false, time.Now(), groupTestSem())
|
||||
if ob.dials != 0 {
|
||||
t.Fatalf("target dialled %d time(s) with probing off; want 0", ob.dials)
|
||||
}
|
||||
if res.OK || res.Error != groupTestErrProbingOff {
|
||||
t.Fatalf("result = %+v, want ok=false with the probing-off explanation", res)
|
||||
}
|
||||
}
|
||||
|
||||
// The deadline path: a covered target whose measured tag never receives a fresh
|
||||
// observation resolves with the not-reached explanation — and, again, without a
|
||||
// single dial of its own.
|
||||
func TestTestOneTargetDeadlineWithoutObservation(t *testing.T) {
|
||||
// Shrink the wait so the test answers in milliseconds; restore afterwards.
|
||||
oldDeadline, oldPoll := groupTestWaitDeadline, groupTestPollEvery
|
||||
groupTestWaitDeadline, groupTestPollEvery = 30*time.Millisecond, 5*time.Millisecond
|
||||
defer func() { groupTestWaitDeadline, groupTestPollEvery = oldDeadline, oldPoll }()
|
||||
|
||||
e := New()
|
||||
ob := &countingOutbound{failingOutbound: failingOutbound{tag: "auto"}}
|
||||
tgt := groupTestTarget{name: "auto", ob: ob}
|
||||
|
||||
res := e.testOneTarget(tgt, map[string]bool{"auto": true}, true, time.Now(), groupTestSem())
|
||||
if ob.dials != 0 {
|
||||
t.Fatalf("target dialled %d time(s) while waiting on the board; want 0", ob.dials)
|
||||
}
|
||||
if res.OK || res.Error != groupTestErrNotReached {
|
||||
t.Fatalf("result = %+v, want ok=false with the not-reached explanation", res)
|
||||
}
|
||||
}
|
||||
|
||||
// A fresh DEAD observation is an answer, not a timeout: ok=false with the
|
||||
// path-dead explanation, delay 0, and no exit-address connection for a path
|
||||
// that just failed its probe.
|
||||
func TestTestOneTargetReportsFreshDeath(t *testing.T) {
|
||||
e := New()
|
||||
ob := &countingOutbound{failingOutbound: failingOutbound{tag: "auto"}}
|
||||
tgt := groupTestTarget{name: "auto", ob: ob}
|
||||
|
||||
since := time.Now().Add(-time.Minute)
|
||||
e.URLTestHistory().MarkFailed("auto")
|
||||
|
||||
res := e.testOneTarget(tgt, map[string]bool{"auto": true}, true, since, groupTestSem())
|
||||
if ob.dials != 0 {
|
||||
t.Fatalf("a dead target was dialled %d time(s); the exit lookup is for alive paths only", ob.dials)
|
||||
}
|
||||
if res.OK || res.Error != groupTestErrPathDead {
|
||||
t.Fatalf("result = %+v, want ok=false with the path-dead explanation", res)
|
||||
}
|
||||
if res.DelayMs != 0 || res.ExitIP != "" {
|
||||
t.Fatalf("a dead path must carry no delay and no exit address: %+v", res)
|
||||
}
|
||||
}
|
||||
|
||||
// TestGroupTestSingleton is the "no parallel runs" invariant: while a run holds the
|
||||
// guard, a second TestGroups must be refused rather than starting a concurrent run
|
||||
// (each of these opens real tunnelled connections).
|
||||
|
||||
@@ -100,6 +100,14 @@ type observatoryState struct {
|
||||
cycles uint64
|
||||
// busy guards against a slow tick overlapping the next one.
|
||||
busy bool
|
||||
// force is the ONE-SHOT "check now" flag raised by RefreshObservatory: while
|
||||
// it is up the freshness gate is bypassed, so a manual refresh really
|
||||
// RE-MEASURES everything on the plan instead of skipping whatever happens to
|
||||
// be fresh, and each finished batch immediately nudges the next one so the
|
||||
// pass runs back-to-back rather than on the 10s tick. It is cleared the
|
||||
// moment the forced pass schedules its last batch (cursor reaches the end of
|
||||
// the plan) — one full walk, then back to the polite steady state.
|
||||
force bool
|
||||
}
|
||||
|
||||
// ConfigureObservatory installs (or re-tunes, or stops) the observatory. It is
|
||||
@@ -120,6 +128,7 @@ func (e *Engine) ConfigureObservatory(cfg ObservatoryConfig) {
|
||||
|
||||
if !cfg.Enabled {
|
||||
e.obs.plan, e.obs.used, e.obs.cursor = nil, nil, 0
|
||||
e.obs.force = false
|
||||
if e.obs.stop != nil {
|
||||
close(e.obs.stop)
|
||||
e.obs.stop, e.obs.done, e.obs.nudge = nil, nil, nil
|
||||
@@ -163,6 +172,7 @@ func (e *Engine) StopObservatory() {
|
||||
e.obs.stop, e.obs.done, e.obs.nudge = nil, nil, nil
|
||||
e.obs.enabled = false
|
||||
e.obs.plan, e.obs.used, e.obs.cursor = nil, nil, 0
|
||||
e.obs.force = false
|
||||
if stop != nil {
|
||||
close(stop)
|
||||
}
|
||||
@@ -172,6 +182,38 @@ func (e *Engine) StopObservatory() {
|
||||
}
|
||||
}
|
||||
|
||||
// RefreshObservatory asks for an OUT-OF-TURN full pass of the probe plan: the
|
||||
// cursor is rewound to the top, the one-shot force flag is raised so the pass
|
||||
// bypasses the freshness gate (a "check now" that skipped everything fresh
|
||||
// would measure nothing and answer with yesterday's numbers), and the loop is
|
||||
// nudged so the first batch starts immediately instead of on the next 10s tick.
|
||||
//
|
||||
// This is the ONLY way the panel's "Test" button touches the network now: the
|
||||
// manual group test (grouptest.go) no longer dials anything itself — it calls
|
||||
// this, then watches the health board for observations newer than its start
|
||||
// instant. One prober, one set of dial paths, one truth.
|
||||
//
|
||||
// No-op when the observatory is disabled or has no plan: there is nothing to
|
||||
// walk, and raising force with no loop to consume it would only arm a stale
|
||||
// flag for a future enable. The caller (TestGroups) reports that situation
|
||||
// through its own results; this method stays silent about it on purpose —
|
||||
// it is a request, not a query.
|
||||
func (e *Engine) RefreshObservatory() {
|
||||
e.obsMu.Lock()
|
||||
defer e.obsMu.Unlock()
|
||||
if !e.obs.enabled || len(e.obs.plan) == 0 {
|
||||
return
|
||||
}
|
||||
e.obs.cursor = 0
|
||||
e.obs.force = true
|
||||
if e.obs.nudge != nil {
|
||||
select {
|
||||
case e.obs.nudge <- struct{}{}:
|
||||
default:
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// ObservatoryStatus reports whether the observatory is on, how many measurements
|
||||
// the current plan holds, and how many full passes have completed. Cheap; used by
|
||||
// tests and available for diagnostics.
|
||||
@@ -240,17 +282,18 @@ func (e *Engine) observatoryLoop(stop <-chan struct{}, done chan<- struct{}, nud
|
||||
// observatoryTickOnce runs one batch. It is deliberately conservative about when
|
||||
// it does nothing:
|
||||
//
|
||||
// - a manual exit test (grouptest.go) is in flight => SKIP. That run is the
|
||||
// human's explicit request; the observatory defers rather than competing for
|
||||
// the uplink. It checks the guard but never CLAIMS it, so a pressed Test
|
||||
// button can never be answered "already running" because of background work;
|
||||
// - the previous tick has not finished => SKIP, so slow probes never stack;
|
||||
// - the engine is stopped => jobs no longer resolve and are skipped (probeJob).
|
||||
//
|
||||
// There is deliberately NO deferral to a manual group test any more. The guard
|
||||
// that used to sit here ("skip the tick while e.groupTestRunning is up") made
|
||||
// sense when the manual run dialled the uplink itself and the two would have
|
||||
// competed for it; that run no longer probes ANYTHING — it asks for a forced
|
||||
// pass via RefreshObservatory and then reads the board. Keeping the guard would
|
||||
// therefore be worse than pointless: a manual run now DEPENDS on the
|
||||
// observatory ticking, and skipping ticks for its whole duration would deadlock
|
||||
// the very refresh it is waiting on (up to its full 120s deadline).
|
||||
func (e *Engine) observatoryTickOnce() {
|
||||
if e.groupTestRunning.Load() {
|
||||
return
|
||||
}
|
||||
|
||||
e.obsMu.Lock()
|
||||
if e.obs.busy || !e.obs.enabled || len(e.obs.plan) == 0 {
|
||||
e.obsMu.Unlock()
|
||||
@@ -258,7 +301,8 @@ func (e *Engine) observatoryTickOnce() {
|
||||
}
|
||||
if e.obs.cursor >= len(e.obs.plan) {
|
||||
// Cycle complete: count it and start the next pass from the top. This is
|
||||
// the ONE place the cursor is deliberately rewound.
|
||||
// the ONE place the cursor is rewound on the observatory's own schedule
|
||||
// (RefreshObservatory rewinds it too, but that is the caller's request).
|
||||
e.obs.cycles++
|
||||
e.obs.cursor = 0
|
||||
}
|
||||
@@ -270,12 +314,32 @@ func (e *Engine) observatoryTickOnce() {
|
||||
batch := append([]ProbeJob(nil), e.obs.plan[start:end]...)
|
||||
e.obs.cursor = end
|
||||
interval := e.obs.interval
|
||||
// A forced pass (RefreshObservatory) bypasses the freshness gate for its one
|
||||
// walk of the plan. The flag is captured for THIS batch and cleared the
|
||||
// moment the walk's last batch is scheduled: the remaining jobs of this very
|
||||
// batch still run unfiltered off the captured copy, and the next pass is an
|
||||
// ordinary polite one again.
|
||||
force := e.obs.force
|
||||
if force && end >= len(e.obs.plan) {
|
||||
e.obs.force = false
|
||||
}
|
||||
e.obs.busy = true
|
||||
e.obsMu.Unlock()
|
||||
|
||||
defer func() {
|
||||
e.obsMu.Lock()
|
||||
e.obs.busy = false
|
||||
// Mid-forced-pass: chain straight into the next batch instead of waiting
|
||||
// out the 10s tick. A human pressed "check now"; walking a subscription-
|
||||
// sized plan at one batch per tick would stretch the answer past the
|
||||
// manual run's deadline for no reason — the batch size still bounds how
|
||||
// much is in flight at once, which is the limit that actually matters.
|
||||
if e.obs.force && e.obs.nudge != nil {
|
||||
select {
|
||||
case e.obs.nudge <- struct{}{}:
|
||||
default:
|
||||
}
|
||||
}
|
||||
e.obsMu.Unlock()
|
||||
}()
|
||||
|
||||
@@ -287,7 +351,7 @@ func (e *Engine) observatoryTickOnce() {
|
||||
sem := make(chan struct{}, observatoryConcurrency)
|
||||
var wg sync.WaitGroup
|
||||
for _, j := range batch {
|
||||
if !observatoryShouldProbe(hist, j, interval, now) {
|
||||
if !force && !observatoryShouldProbe(hist, j, interval, now) {
|
||||
continue
|
||||
}
|
||||
wg.Add(1)
|
||||
|
||||
@@ -100,28 +100,89 @@ func TestObservatoryTickStoppedEngine(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
// The exit-test gate: while a manual group/chain test is in flight the tick is a
|
||||
// no-op — the observatory checks the guard but never claims it, so a pressed
|
||||
// Test button is never refused because of background work.
|
||||
func TestObservatoryDefersToExitTest(t *testing.T) {
|
||||
// The INVERSE of the old exit-test gate, pinned so it cannot come back: the
|
||||
// tick must RUN while a manual group test is in flight. The manual run no
|
||||
// longer probes anything — it asks for a forced pass and then waits on the
|
||||
// board — so a tick that deferred to it would deadlock the very refresh the
|
||||
// run is polling for, until its 120s deadline reported "not reached" about
|
||||
// paths nobody attempted.
|
||||
func TestObservatoryTicksDuringManualRun(t *testing.T) {
|
||||
e := New()
|
||||
defer e.StopObservatory()
|
||||
e.ConfigureObservatory(ObservatoryConfig{Enabled: true, Options: obsFixture(), ProbeInterval: time.Minute})
|
||||
|
||||
e.groupTestRunning.Store(true)
|
||||
e.observatoryTickOnce()
|
||||
if _, cursor, _ := obsState(e); cursor != 0 {
|
||||
t.Fatal("tick ran while an exit test was in flight")
|
||||
if _, cursor, _ := obsState(e); cursor == 0 {
|
||||
t.Fatal("tick deferred to an in-flight manual run; the run depends on the tick now")
|
||||
}
|
||||
e.groupTestRunning.Store(false)
|
||||
|
||||
// And the guard is free: a manual run can start immediately.
|
||||
// The observatory never claims the manual-run guard either way round: a
|
||||
// manual run can still start immediately.
|
||||
if !e.TestGroups(nil, "") {
|
||||
t.Fatal("a manual exit test was refused; the observatory must never hold groupTestRunning")
|
||||
t.Fatal("a manual test was refused; the observatory must never hold groupTestRunning")
|
||||
}
|
||||
waitGroupTestIdle(t, e)
|
||||
}
|
||||
|
||||
// RefreshObservatory is the manual run's whole probing story: it rewinds the
|
||||
// cursor, raises the one-shot force flag (bypassing the freshness gate for one
|
||||
// full walk), and the flag clears itself when the forced pass schedules its
|
||||
// last batch. Disabled or plan-less observatories ignore it entirely.
|
||||
func TestRefreshObservatoryForcesOnePass(t *testing.T) {
|
||||
e := New()
|
||||
defer e.StopObservatory()
|
||||
|
||||
// Disabled: a refresh request is a documented no-op.
|
||||
e.RefreshObservatory()
|
||||
e.obsMu.Lock()
|
||||
force := e.obs.force
|
||||
e.obsMu.Unlock()
|
||||
if force {
|
||||
t.Fatal("RefreshObservatory raised force on a disabled observatory")
|
||||
}
|
||||
|
||||
e.ConfigureObservatory(ObservatoryConfig{Enabled: true, Options: obsFixture(), ProbeInterval: time.Minute})
|
||||
|
||||
// Fresh observations on every planned tag: the polite gate would skip them
|
||||
// all, which is exactly what a forced pass must NOT do.
|
||||
hist := e.URLTestHistory()
|
||||
now := time.Now()
|
||||
hist.StoreURLTestHistory("n1", &adapter.URLTestHistory{LastOK: now, Delay: 5})
|
||||
hist.StoreURLTestHistory("n2", &adapter.URLTestHistory{LastOK: now, Delay: 5})
|
||||
|
||||
// Pretend a walk was mid-plan; the refresh must rewind it.
|
||||
e.obsMu.Lock()
|
||||
e.obs.cursor = 1
|
||||
e.obsMu.Unlock()
|
||||
e.RefreshObservatory()
|
||||
e.obsMu.Lock()
|
||||
cursor, force := e.obs.cursor, e.obs.force
|
||||
e.obsMu.Unlock()
|
||||
if cursor != 0 || !force {
|
||||
t.Fatalf("after refresh: cursor=%d force=%v, want 0/true", cursor, force)
|
||||
}
|
||||
|
||||
// One tick covers the whole 2-job plan (batch is 24), so the forced pass
|
||||
// completes and the flag clears itself — one full walk, not a permanent mode.
|
||||
e.observatoryTickOnce()
|
||||
e.obsMu.Lock()
|
||||
cursor, force = e.obs.cursor, e.obs.force
|
||||
e.obsMu.Unlock()
|
||||
if force {
|
||||
t.Fatal("force flag survived the forced pass; it must be one-shot")
|
||||
}
|
||||
if cursor != 2 {
|
||||
t.Fatalf("cursor after forced tick = %d, want 2 (the pass really walked the plan)", cursor)
|
||||
}
|
||||
// The stopped engine resolves no outbounds, so the probes were skipped —
|
||||
// but the fresh history proves the GATE was bypassed only if the jobs were
|
||||
// attempted; attempt-tracking lives in probeJob, which needs a box. What is
|
||||
// pinned here is the flag lifecycle and the rewind, the two halves
|
||||
// TestGroups depends on.
|
||||
}
|
||||
|
||||
// The freshness gate: a job whose every covered tag has an observation younger
|
||||
// than the global probe interval is skipped — that set is exactly what an ACTIVE
|
||||
// group is measuring itself, and the observatory must not duplicate or suppress
|
||||
|
||||
+39
-20
@@ -27,8 +27,12 @@ import (
|
||||
// - a CHAIN entry tag "chain-<n>-hN" (the exit wrapper, generate/chain.go) is
|
||||
// probed ITSELF: dialling the copy pulls the whole L1→…→Ln path through its
|
||||
// Detour links, so one probe is the chain's end-to-end health. Intermediate
|
||||
// NODE hops are not probed separately — their death is visible in the exit
|
||||
// probe and no selection depends on them;
|
||||
// NODE hop wrappers ("chain-<n>-h<i>" that are plain outbounds) are probed
|
||||
// TOO — not because selection needs them (it does not), but because an
|
||||
// operator does: dialling "chain-<n>-h1" measures exactly the prefix of the
|
||||
// path up to and including hop 1, so when the exit probe dies the per-hop
|
||||
// probes say WHICH hop died instead of only that the chain did (the
|
||||
// ChainHealth hops surface, grouphealth.go);
|
||||
// - a GROUP hop inside a chain ("chain-<n>-h<i>", a wrapper selector reached by
|
||||
// walking the exit's Detour links) additionally has its member copies
|
||||
// "chain-<n>-h<i>-<member>" probed: they are the wrapper's selection
|
||||
@@ -291,8 +295,20 @@ func (w *planWalk) visitMember(group, member string) {
|
||||
// walkDetour follows a probed tag's Detour links toward the router. A group met
|
||||
// on the way is a chain's GROUP HOP wrapper: its member copies are selection
|
||||
// candidates and are probed, each through its own prefix of the path. A plain
|
||||
// outbound met on the way is an intermediate node hop — marked used, never probed
|
||||
// on its own — and the walk continues through it.
|
||||
// outbound met on the way falls in two halves:
|
||||
//
|
||||
// - a chain NODE HOP wrapper ("chain-<n>-h<i>", parseChainExitTag) is probed
|
||||
// as a measurement of its own. Dialling it pulls exactly the prefix of the
|
||||
// chain up to and including hop i through the Detour links, so its verdict
|
||||
// LOCALISES a failure: the exit probe says the chain died, the hop probes
|
||||
// say where. It used to be skipped here on the argument that its death
|
||||
// shows up in the exit probe anyway — true for end-to-end health, useless
|
||||
// for an operator staring at a dead 4-hop chain. The measurement is stored
|
||||
// under the wrapper tag alone: a chain prefix is nobody else's dial path,
|
||||
// so no alias may borrow its verdict;
|
||||
// - anything else on the detour path — an "egress-…" interface outbound, an
|
||||
// ordinary node's egress detour — keeps today's behaviour: marked used,
|
||||
// never probed on its own, and the walk continues through it.
|
||||
func (w *planWalk) walkDetour(tag string) {
|
||||
if tag == "" || w.walked[tag] {
|
||||
return
|
||||
@@ -303,25 +319,28 @@ func (w *planWalk) walkDetour(tag string) {
|
||||
return
|
||||
}
|
||||
w.used[tag] = true
|
||||
if ent.group {
|
||||
next := ""
|
||||
for _, m := range ent.members {
|
||||
ment, ok := w.byTag[m]
|
||||
if !ok {
|
||||
continue
|
||||
}
|
||||
w.used[m] = true
|
||||
if !ment.group && probeablePlanTag(ment.typ, m) {
|
||||
w.addTarget(dialKeyTag(m), m)
|
||||
}
|
||||
if next == "" {
|
||||
next = ment.detour // every member detours into the same previous hop
|
||||
}
|
||||
if !ent.group {
|
||||
if _, isHopWrapper := parseChainExitTag(tag); isHopWrapper && probeablePlanTag(ent.typ, tag) {
|
||||
w.addTarget(dialKeyTag(tag), tag)
|
||||
}
|
||||
w.walkDetour(next)
|
||||
w.walkDetour(ent.detour)
|
||||
return
|
||||
}
|
||||
w.walkDetour(ent.detour)
|
||||
next := ""
|
||||
for _, m := range ent.members {
|
||||
ment, ok := w.byTag[m]
|
||||
if !ok {
|
||||
continue
|
||||
}
|
||||
w.used[m] = true
|
||||
if !ment.group && probeablePlanTag(ment.typ, m) {
|
||||
w.addTarget(dialKeyTag(m), m)
|
||||
}
|
||||
if next == "" {
|
||||
next = ment.detour // every member detours into the same previous hop
|
||||
}
|
||||
}
|
||||
w.walkDetour(next)
|
||||
}
|
||||
|
||||
// dialKeyTag / dialKeyCopy build the dedup key for one measurement target.
|
||||
|
||||
@@ -218,6 +218,64 @@ func TestObservatoryPlanChain(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
// A chain with a plain NODE hop wrapper on the way to the exit: the wrapper IS
|
||||
// a measurement of its own now (walkDetour) — dialling "chain-<c>-h1" measures
|
||||
// exactly the path prefix up to hop 1, which is what localises a dead hop. And
|
||||
// the other half of the honesty contract: a base node that appears ONLY as a
|
||||
// chain group-hop member is never measured under its BASE tag — the member
|
||||
// copy's observation stays on the copy, because the copy's dial path (through
|
||||
// the chain prefix) is not the base outbound's dial path, and no job may write
|
||||
// a direct-looking verdict onto a tag it did not dial.
|
||||
func TestObservatoryPlanChainHopWrappersProbed(t *testing.T) {
|
||||
opts := option.Options{
|
||||
Outbounds: []option.Outbound{
|
||||
// L1: a plain node hop wrapper.
|
||||
fixNode("chain-c2-h1", ""),
|
||||
// L2: a group hop over one member copy of base node N.
|
||||
fixNode("chain-c2-h2-N", "chain-c2-h1"),
|
||||
fixGroup(C.TypeSelector, "chain-c2-h2", "chain-c2-h2-N"),
|
||||
// L3: the exit, a plain node copy.
|
||||
fixNode("chain-c2-h3", "chain-c2-h2"),
|
||||
// The base node behind the L2 member copy: emitted, referenced by
|
||||
// NOTHING but the copy's name.
|
||||
fixNode("N", ""),
|
||||
},
|
||||
Route: fixRoute("", "chain-c2-h3"),
|
||||
}
|
||||
byDial, used := planOf(t, opts)
|
||||
|
||||
// The node hop wrapper is a job of its own, stored only under itself.
|
||||
j := assertPlanHas(t, byDial, "chain-c2-h1")
|
||||
if len(j.Store) != 1 || j.Store[0] != "chain-c2-h1" {
|
||||
t.Errorf("node hop wrapper store = %v, want only itself (a chain prefix is nobody else's path)", j.Store)
|
||||
}
|
||||
// The exit and the group-hop member copy are jobs, as before.
|
||||
assertPlanHas(t, byDial, "chain-c2-h3")
|
||||
assertPlanHas(t, byDial, "chain-c2-h2-N")
|
||||
|
||||
// NO job may record anything under the base tag N: not as a dial, not as a
|
||||
// store alias. A direct measurement can never land on it from this config.
|
||||
if _, ok := byDial["N"]; ok {
|
||||
t.Error("plan dials the base node N, which no rule reaches")
|
||||
}
|
||||
for dial, job := range byDial {
|
||||
for _, tag := range job.Store {
|
||||
if tag == "N" {
|
||||
t.Errorf("job %q stores under the base tag N; a chain member copy must never alias its base", dial)
|
||||
}
|
||||
}
|
||||
}
|
||||
if used["N"] {
|
||||
t.Error("used-set claims the base node N; only the chain copies are on the path")
|
||||
}
|
||||
// The wrappers themselves are used (they are the path).
|
||||
for _, u := range []string{"chain-c2-h1", "chain-c2-h2", "chain-c2-h3", "chain-c2-h2-N"} {
|
||||
if !used[u] {
|
||||
t.Errorf("used-set is missing %q", u)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// DNS server detours and route.Final are roots too — a group referenced only as a
|
||||
// resolver's detour is still a used, probed path.
|
||||
func TestObservatoryPlanDNSDetourRoot(t *testing.T) {
|
||||
|
||||
@@ -0,0 +1,70 @@
|
||||
package engine
|
||||
|
||||
import (
|
||||
C "github.com/sagernet/sing-box/constant"
|
||||
"github.com/sagernet/sing-box/option"
|
||||
)
|
||||
|
||||
// Standing down the self-check of unused urltest groups (health board §5.C).
|
||||
//
|
||||
// # Why the engine does this, and not the generator
|
||||
//
|
||||
// A urltest group probes its own members: a warm-up sweep at PostStart and a
|
||||
// ticker for as long as traffic touches it (protocol/group/urltest.go). For a
|
||||
// group the routing rules actually use, that probing travels the same path the
|
||||
// traffic does and is welcome. For a group NO rule reaches, it dials the
|
||||
// members' base outbounds directly from the router — a path nothing uses — and
|
||||
// stores the results under the base tags, where every health consumer reads
|
||||
// them as "the node's health". A node that only works behind a tunnel then
|
||||
// reads dead on the shared board because an idle group measured it over the
|
||||
// blocked direct WAN. The observatory's plan (probeplan.go) is the single
|
||||
// statement of what gets probed and along which paths; this file makes the
|
||||
// config that reaches box.New SAY so, by flipping SelfCheck off on every
|
||||
// urltest group outside the plan's used-set.
|
||||
//
|
||||
// The generator cannot do this: whether a group is used is a property of the
|
||||
// EMITTED rules and DNS detours as a whole, and BuildObservatoryPlan is the one
|
||||
// place that reachability is computed. Recomputing it in the generator would be
|
||||
// a second copy of the same walk, and second copies drift.
|
||||
|
||||
// standDownUnusedSelfCheck flips SelfCheck to false on every urltest group in
|
||||
// opts that the observatory's used-set does not cover, and returns how many it
|
||||
// stood down (for the apply log and the tests). Selector groups are left alone
|
||||
// — they have no self-check to stand down — and a group any rule, route.Final
|
||||
// or DNS detour reaches keeps its default (nil == on).
|
||||
//
|
||||
// It MUTATES the option structs the caller handed in: opts.Outbounds carries
|
||||
// the generator's option values as pointers behind an `any` (the generator
|
||||
// always emits *option.URLTestOutboundOptions), so writing through them changes
|
||||
// the caller's config. That is deliberate, not an accident to guard against:
|
||||
// the plan is the single source of what gets probed, and the config that
|
||||
// reaches box.New must already say so — a copy-on-write here would build a box
|
||||
// whose groups probe paths the hash and the plan claim nobody probes. It runs
|
||||
// BEFORE hashOptions in applyLocked for the same reason: the hash must describe
|
||||
// what is really built, so a rule change that flips a group used<->unused is a
|
||||
// real config change and triggers a swap.
|
||||
//
|
||||
// A urltest outbound whose Options is not the expected pointer type (a value,
|
||||
// or something foreign) is skipped rather than guessed at: it did not come from
|
||||
// our generator, and silently rebuilding somebody else's option struct is worse
|
||||
// than leaving one group's self-check up.
|
||||
func standDownUnusedSelfCheck(opts option.Options) int {
|
||||
_, used := BuildObservatoryPlan(opts, "")
|
||||
stood := 0
|
||||
for i := range opts.Outbounds {
|
||||
ob := &opts.Outbounds[i]
|
||||
if ob.Type != C.TypeURLTest || used[ob.Tag] {
|
||||
continue
|
||||
}
|
||||
utOpts, ok := ob.Options.(*option.URLTestOutboundOptions)
|
||||
if !ok {
|
||||
continue
|
||||
}
|
||||
// One fresh pointer per group, so no two option structs alias a shared
|
||||
// bool that a later caller could flip for both at once.
|
||||
off := false
|
||||
utOpts.SelfCheck = &off
|
||||
stood++
|
||||
}
|
||||
return stood
|
||||
}
|
||||
@@ -0,0 +1,204 @@
|
||||
package engine
|
||||
|
||||
import (
|
||||
"testing"
|
||||
|
||||
C "github.com/sagernet/sing-box/constant"
|
||||
"github.com/sagernet/sing-box/option"
|
||||
)
|
||||
|
||||
// standDownUnusedSelfCheck is the enforcement half of the one-probe rule: a
|
||||
// urltest group no rule reaches gets SelfCheck=false written into its option
|
||||
// struct (in place — the struct the box will be built from), a rule-reachable
|
||||
// one keeps its nil default, and selectors are never touched at all (they have
|
||||
// no self-check to stand down). The count it returns is what applyLocked logs.
|
||||
func TestStandDownUnusedSelfCheck(t *testing.T) {
|
||||
opts := option.Options{
|
||||
Outbounds: []option.Outbound{
|
||||
fixNode("n1", ""),
|
||||
fixNode("idle1", ""),
|
||||
fixNode("pin1", ""),
|
||||
// Reached by a rule: must keep probing itself.
|
||||
fixGroup(C.TypeURLTest, "auto", "n1"),
|
||||
// Reached by nothing: its self-check dials a path nobody uses and
|
||||
// must be stood down.
|
||||
fixGroup(C.TypeURLTest, "idle", "idle1"),
|
||||
// A selector reached by nothing: left alone — no self-check exists.
|
||||
fixGroup(C.TypeSelector, "pins", "pin1"),
|
||||
},
|
||||
Route: fixRoute("auto"),
|
||||
}
|
||||
|
||||
stood := standDownUnusedSelfCheck(opts)
|
||||
if stood != 1 {
|
||||
t.Fatalf("stood down %d group(s), want exactly 1 (idle)", stood)
|
||||
}
|
||||
|
||||
find := func(tag string) option.Outbound {
|
||||
t.Helper()
|
||||
for _, ob := range opts.Outbounds {
|
||||
if ob.Tag == tag {
|
||||
return ob
|
||||
}
|
||||
}
|
||||
t.Fatalf("outbound %q missing from opts", tag)
|
||||
return option.Outbound{}
|
||||
}
|
||||
|
||||
// The unused urltest group: SelfCheck is now an explicit false in the very
|
||||
// struct the caller handed in (the mutation is the point — the config that
|
||||
// reaches box.New must say what the plan says).
|
||||
idle := find("idle").Options.(*option.URLTestOutboundOptions)
|
||||
if idle.SelfCheck == nil || *idle.SelfCheck {
|
||||
t.Fatalf("idle group SelfCheck = %v, want an explicit false", idle.SelfCheck)
|
||||
}
|
||||
|
||||
// The used urltest group keeps the nil default (self-check on): its probing
|
||||
// travels the path traffic actually takes and is welcome.
|
||||
auto := find("auto").Options.(*option.URLTestOutboundOptions)
|
||||
if auto.SelfCheck != nil {
|
||||
t.Fatalf("used group SelfCheck = %v, want nil (untouched default)", *auto.SelfCheck)
|
||||
}
|
||||
|
||||
// The selector's option struct has no SelfCheck field at all; what is
|
||||
// pinned here is that stand-down neither counted it nor mangled its type.
|
||||
if _, ok := find("pins").Options.(*option.SelectorOutboundOptions); !ok {
|
||||
t.Fatal("selector options were rebuilt; stand-down must leave selectors alone")
|
||||
}
|
||||
|
||||
// Idempotence: a second pass finds the same one group (already-false is
|
||||
// still counted — the function reports plan membership, not novelty) and
|
||||
// changes nothing further. This is what keeps the apply hash stable across
|
||||
// the every-minute reconcile.
|
||||
if again := standDownUnusedSelfCheck(opts); again != 1 {
|
||||
t.Fatalf("second pass stood down %d, want 1 (deterministic on identical opts)", again)
|
||||
}
|
||||
if idle.SelfCheck == nil || *idle.SelfCheck {
|
||||
t.Fatal("second pass flipped the idle group's SelfCheck back")
|
||||
}
|
||||
}
|
||||
|
||||
// TestStandDownNeverTouchesChainHopWrappers pins the property the production
|
||||
// config lives or dies by, DIRECTLY rather than transitively through the plan
|
||||
// tests: a CHAIN HOP WRAPPER group ("chain-<name>-h<i>", type urltest) must
|
||||
// NEVER be stood down.
|
||||
//
|
||||
// The failure mode this guards against is subtle and silent. The owner's live
|
||||
// rule routes through egress:ewan -> node:awgout -> group:sub0 -> group:sub1
|
||||
// -> group:sub2 — a chain whose group hops are rebuilt as urltest wrappers
|
||||
// over per-chain member copies. Those wrappers are the ONLY probing that
|
||||
// travels the real path: their own schedule is the on-path measurement, and
|
||||
// the selection inside the chain (which member each hop dials) depends on the
|
||||
// verdicts it writes. Reachability of a wrapper is established by walkDetour
|
||||
// (probeplan.go) walking the exit's Detour links — a DIFFERENT code path from
|
||||
// the root/member expansion the other tests exercise. If a future change to
|
||||
// that walk dropped the wrappers from the used-set, standDownUnusedSelfCheck
|
||||
// would obediently flip their SelfCheck off, the chain would stop measuring
|
||||
// itself, and nothing else would ever measure it: we would have replaced the
|
||||
// false reading this rework removed with NO reading at all. Both existing
|
||||
// tests could stay green while that happened — the plan test asserts the
|
||||
// used-set, not the stand-down; the stand-down test above never mentions
|
||||
// chains. This one closes that gap by asserting the stand-down's OUTPUT on
|
||||
// the production topology.
|
||||
//
|
||||
// The fixture mirrors what the generator actually emits for that config, not
|
||||
// a toy: an exit wrapper chain-p-h4 (urltest over member copies) whose copies
|
||||
// detour into chain-p-h3 (urltest), whose copies detour into chain-p-h2
|
||||
// (urltest), whose copies detour into chain-p-h1 (the plain AmneziaWG hop
|
||||
// copy), which detours into egress-ewan — the detour lives on the member
|
||||
// COPIES, not on the wrappers, exactly as generate/chain.go builds it. The
|
||||
// base groups sub0/sub1/sub2 are ALSO emitted, over the base node tags,
|
||||
// reached by nothing — which is precisely what buildGroups does today and
|
||||
// precisely the false direct prober the stand-down exists to silence.
|
||||
func TestStandDownNeverTouchesChainHopWrappers(t *testing.T) {
|
||||
opts := option.Options{
|
||||
Outbounds: []option.Outbound{
|
||||
// The egress the whole chain leaves through.
|
||||
fixUtility(C.TypeDirect, "egress-ewan"),
|
||||
// L1: the plain AWG hop copy (in production an endpoint; a plain
|
||||
// outbound here exercises the same walk — walkDetour treats both as
|
||||
// non-group entries).
|
||||
fixNode("chain-p-h1", "egress-ewan"),
|
||||
// L2..L4: urltest group hop wrappers over per-chain member copies;
|
||||
// each COPY detours into the previous hop.
|
||||
fixNode("chain-p-h2-a", "chain-p-h1"),
|
||||
fixNode("chain-p-h2-b", "chain-p-h1"),
|
||||
fixGroup(C.TypeURLTest, "chain-p-h2", "chain-p-h2-a", "chain-p-h2-b"),
|
||||
fixNode("chain-p-h3-a", "chain-p-h2"),
|
||||
fixNode("chain-p-h3-b", "chain-p-h2"),
|
||||
fixGroup(C.TypeURLTest, "chain-p-h3", "chain-p-h3-a", "chain-p-h3-b"),
|
||||
fixNode("chain-p-h4-a", "chain-p-h3"),
|
||||
fixNode("chain-p-h4-b", "chain-p-h3"),
|
||||
fixGroup(C.TypeURLTest, "chain-p-h4", "chain-p-h4-a", "chain-p-h4-b"),
|
||||
// The base nodes and the base groups over them: emitted alongside
|
||||
// the chain, reached by no rule — the generator keeps emitting them
|
||||
// today, and they are exactly the direct-WAN probers that poisoned
|
||||
// the board.
|
||||
fixNode("a", ""),
|
||||
fixNode("b", ""),
|
||||
fixGroup(C.TypeURLTest, "sub0", "a", "b"),
|
||||
fixGroup(C.TypeURLTest, "sub1", "a", "b"),
|
||||
fixGroup(C.TypeURLTest, "sub2", "a", "b"),
|
||||
},
|
||||
// One rule, targeting the chain's exit wrapper — the production shape.
|
||||
Route: fixRoute("", "chain-p-h4"),
|
||||
}
|
||||
|
||||
stood := standDownUnusedSelfCheck(opts)
|
||||
if stood != 3 {
|
||||
t.Fatalf("stood down %d group(s), want exactly the 3 base groups sub0/sub1/sub2", stood)
|
||||
}
|
||||
|
||||
utOptions := func(tag string) *option.URLTestOutboundOptions {
|
||||
t.Helper()
|
||||
for _, ob := range opts.Outbounds {
|
||||
if ob.Tag == tag {
|
||||
o, ok := ob.Options.(*option.URLTestOutboundOptions)
|
||||
if !ok {
|
||||
t.Fatalf("outbound %q is not a urltest group", tag)
|
||||
}
|
||||
return o
|
||||
}
|
||||
}
|
||||
t.Fatalf("outbound %q missing from opts", tag)
|
||||
return nil
|
||||
}
|
||||
|
||||
// Every hop wrapper keeps its self-check: nil, the untouched default. An
|
||||
// explicit false on ANY of these is the chain going blind.
|
||||
for _, wrapper := range []string{"chain-p-h2", "chain-p-h3", "chain-p-h4"} {
|
||||
if sc := utOptions(wrapper).SelfCheck; sc != nil {
|
||||
t.Errorf("hop wrapper %q SelfCheck = %v, want nil — its own schedule IS the on-path measurement", wrapper, *sc)
|
||||
}
|
||||
}
|
||||
// Every base group is stood down: nothing routes through them, and their
|
||||
// probing would dial the base nodes over the direct WAN.
|
||||
for _, base := range []string{"sub0", "sub1", "sub2"} {
|
||||
if sc := utOptions(base).SelfCheck; sc == nil || *sc {
|
||||
t.Errorf("base group %q SelfCheck = %v, want an explicit false", base, sc)
|
||||
}
|
||||
}
|
||||
|
||||
// And the reason the stand-down of sub0/sub1/sub2 is SAFE in this config:
|
||||
// with the base groups silenced, the base node tags a/b have no measurement
|
||||
// path at all — no job dials them and no job stores under them (the chain's
|
||||
// member copies record only under themselves). Their board entries simply
|
||||
// stop being written, honestly untested, instead of carrying a false
|
||||
// direct-WAN verdict.
|
||||
jobs, used := BuildObservatoryPlan(opts, "")
|
||||
for _, baseTag := range []string{"a", "b"} {
|
||||
if used[baseTag] {
|
||||
t.Errorf("used-set claims base node %q; only the chain copies are on the path", baseTag)
|
||||
}
|
||||
for _, j := range jobs {
|
||||
if j.Dial == baseTag {
|
||||
t.Errorf("plan dials base node %q, which no rule reaches", baseTag)
|
||||
}
|
||||
for _, s := range j.Store {
|
||||
if s == baseTag {
|
||||
t.Errorf("job %q stores under base node %q; a chain copy must never alias its base", j.Dial, baseTag)
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,344 @@
|
||||
package engine
|
||||
|
||||
// Bounded, deterministic teardown of a SUPERSEDED sing-box instance.
|
||||
//
|
||||
// # The defect this file exists for
|
||||
//
|
||||
// sing-box's Box has no live reload, so every applied config change builds a
|
||||
// fresh Box and retires the old one (see engine.go). Retiring it used to be one
|
||||
// line — `_ = old.Close()` — and that line had three separate problems, all of
|
||||
// which had to be true at once for the observed failure:
|
||||
//
|
||||
// 1. NO CANCELLATION. Upstream's own runner gives every Box its OWN cancellable
|
||||
// context and calls cancel() BEFORE Close (cmd/sing-box/cmd_run.go:136 and
|
||||
// :188-190). This engine handed EVERY box.New the one shared, never-cancelled
|
||||
// context built in New (engine.go), and box.New does not derive a cancellable
|
||||
// child of what it is given (box.go: `ctx := options.Context` and nothing
|
||||
// else). So every goroutine inside a retired box that would have stopped on
|
||||
// context cancellation simply never stopped, and only what each adapter's
|
||||
// Close() explicitly tears down actually went away.
|
||||
//
|
||||
// 2. NO TIME BUDGET. Box.Close walks its subsystems SEQUENTIALLY and waits for
|
||||
// each one forever; the only thing watching the clock is a taskmonitor that
|
||||
// prints "close endpoint/wireguard[...] take too much time to finish!" after
|
||||
// C.StopTimeout and then keeps waiting anyway (common/taskmonitor/monitor.go).
|
||||
// A wireguard endpoint that will not come down therefore stalls the rest of
|
||||
// the close list, and every subsystem AFTER it in the walk is never reached.
|
||||
//
|
||||
// 3. NO EVIDENCE. The error was discarded (`_ =`) and the pointer to the old box
|
||||
// was overwritten in the same breath, so after the swap nothing in the process
|
||||
// could tell — or even ask — whether the previous generation had actually
|
||||
// died. On the router this accumulated: up to four generations logging side by
|
||||
// side inside one shaterd process, each with its own outbound connections and,
|
||||
// worse, its own WireGuard devices. Two devices sharing a private key evict
|
||||
// each other at the peer, which is exactly the fault generate/wgdedup.go
|
||||
// removes WITHIN a config — reintroduced here BETWEEN generations.
|
||||
//
|
||||
// # The contract now
|
||||
//
|
||||
// - Every box gets its own cancellable context, cancelled before Close.
|
||||
// - Close runs against a HARD budget (closeBudget). When it expires the apply
|
||||
// moves on: the new box is already started and serving, and blocking the
|
||||
// control plane on a shutdown that is not going to finish would only add an
|
||||
// unresponsive panel to the problem.
|
||||
// - An overrun is LOUD, not swallowed: an ERROR line naming the generation, and
|
||||
// a critical entry in the operator-facing warning set for as long as the
|
||||
// abandoned generation is still running (see PendingCloses).
|
||||
// - Generations do not stack: before an apply builds another instance it gives
|
||||
// any abandoned teardown one more budget to finish, and if one is still alive
|
||||
// the swap is forced onto the close-old-then-start-new path so the process
|
||||
// never holds two LIVE boxes on top of an abandoned one.
|
||||
|
||||
import (
|
||||
"errors"
|
||||
"io"
|
||||
"sync"
|
||||
"time"
|
||||
|
||||
box "github.com/sagernet/sing-box"
|
||||
)
|
||||
|
||||
// ErrCloseTimeout reports that a superseded instance did not finish shutting
|
||||
// down inside the close budget and has been ABANDONED (it may still be running).
|
||||
//
|
||||
// It is deliberately NOT propagated out of Apply. The swap it accompanies
|
||||
// succeeded — the new box is built, started and carrying traffic — and returning
|
||||
// a failure there would make the caller abort the netplane stage and, with no
|
||||
// engine fault to point at, leave a stale ruleset loaded over a healthy engine.
|
||||
// The fact is surfaced through PendingCloses() instead, which reaches
|
||||
// `shaterd status` and the panel and clears itself the moment the shutdown
|
||||
// finally completes. Close()/Teardown DO return it: there the failure to stop is
|
||||
// the whole point of the operation.
|
||||
var ErrCloseTimeout = errors.New("engine instance did not shut down within the close budget")
|
||||
|
||||
// defaultCloseBudget is the HARD wall-clock budget for retiring one superseded
|
||||
// box.
|
||||
//
|
||||
// Five seconds, for three reasons that all point at the same number:
|
||||
//
|
||||
// - it is exactly sing-box's own C.StopTimeout — the threshold at which the
|
||||
// engine itself declares a single lifecycle stop to be taking too long. A
|
||||
// close that blows past the budget upstream considers excessive is by
|
||||
// definition not a slow close, it is a stuck one;
|
||||
// - it stays under C.FatalStopTimeout (10s), which is when upstream's CLI gives
|
||||
// up and calls the process unclosable. We want to have already reacted by
|
||||
// then;
|
||||
// - it is paid AFTER the replacement box is started and serving, and only on an
|
||||
// apply that really changed something (the hash gate makes a no-op reconcile
|
||||
// free), so the worst case is five seconds added to one apply — not to the
|
||||
// every-minute cron reconcile, and never to a status poll.
|
||||
const defaultCloseBudget = 5 * time.Second
|
||||
|
||||
// closeBudget/closeBoxFn are package-wide seams. Production never touches them;
|
||||
// tests in this package and in shater/apply install a short budget and a
|
||||
// deliberately hung closer to exercise the abandoned-generation path without
|
||||
// standing up a box that really refuses to die.
|
||||
var (
|
||||
seamMu sync.RWMutex
|
||||
closeBudget = defaultCloseBudget
|
||||
closeBoxFn = func(c io.Closer) error { return c.Close() }
|
||||
)
|
||||
|
||||
// SetCloseBudget overrides the close budget and returns a function restoring the
|
||||
// previous value. TEST SEAM — production uses defaultCloseBudget.
|
||||
func SetCloseBudget(d time.Duration) (restore func()) {
|
||||
seamMu.Lock()
|
||||
prev := closeBudget
|
||||
closeBudget = d
|
||||
seamMu.Unlock()
|
||||
return func() {
|
||||
seamMu.Lock()
|
||||
closeBudget = prev
|
||||
seamMu.Unlock()
|
||||
}
|
||||
}
|
||||
|
||||
// SetBoxCloser overrides HOW a superseded instance is closed and returns a
|
||||
// function restoring the previous closer. TEST SEAM — production calls
|
||||
// Box.Close. A closer that never returns is how the abandoned-generation path is
|
||||
// tested.
|
||||
func SetBoxCloser(fn func(io.Closer) error) (restore func()) {
|
||||
seamMu.Lock()
|
||||
prev := closeBoxFn
|
||||
closeBoxFn = fn
|
||||
seamMu.Unlock()
|
||||
return func() {
|
||||
seamMu.Lock()
|
||||
closeBoxFn = prev
|
||||
seamMu.Unlock()
|
||||
}
|
||||
}
|
||||
|
||||
func currentCloseBudget() time.Duration {
|
||||
seamMu.RLock()
|
||||
defer seamMu.RUnlock()
|
||||
return closeBudget
|
||||
}
|
||||
|
||||
func currentBoxCloser() func(io.Closer) error {
|
||||
seamMu.RLock()
|
||||
defer seamMu.RUnlock()
|
||||
return closeBoxFn
|
||||
}
|
||||
|
||||
// pendingClose tracks ONE retirement in flight. It lives in Engine.pending from
|
||||
// the moment the retirement starts until Close returns, and `abandoned` records
|
||||
// whether the budget expired while it was still running — i.e. whether this
|
||||
// generation is a leak the operator must be told about, or merely a close that
|
||||
// is in progress under a lock the caller is holding anyway.
|
||||
type pendingClose struct {
|
||||
gen uint64
|
||||
hash string // config hash the retired box was built from
|
||||
since time.Time
|
||||
done chan struct{} // closed when the underlying Close finally returns
|
||||
abandoned bool // guarded by Engine.pendingMu
|
||||
finished bool // guarded by Engine.pendingMu
|
||||
}
|
||||
|
||||
// StuckClose is the read-side view of a superseded generation that overran the
|
||||
// close budget and is STILL running inside this process.
|
||||
type StuckClose struct {
|
||||
// Generation is the 1-based sequence number of the box that will not die.
|
||||
// It matches nothing in the sing-box log by itself, but it lets two status
|
||||
// reads tell "the same old generation" from "another one just leaked".
|
||||
Generation uint64
|
||||
// Hash is the config hash the leaked box was built from — the same value
|
||||
// `shaterd status` reported as `hash` while that config was the running one.
|
||||
// It is what connects "an engine is stuck" to WHICH configuration is stuck.
|
||||
Hash string
|
||||
// Since is when its shutdown was started.
|
||||
Since time.Time
|
||||
// Elapsed is how long it has been shutting down, as of the read.
|
||||
Elapsed time.Duration
|
||||
}
|
||||
|
||||
// retireLocked tears down a superseded box under the close budget. The caller
|
||||
// holds e.mu and must clear e.instance itself.
|
||||
//
|
||||
// Order matters and mirrors upstream: cancel the box's context FIRST so every
|
||||
// goroutine keyed on it unwinds, and only then call Close, which is what actually
|
||||
// releases the listeners, the WireGuard devices and the cache-file lock.
|
||||
//
|
||||
// Returns nil on a clean shutdown, ErrCloseTimeout when the budget expired (the
|
||||
// close keeps running in the background and is reaped when it finishes), or the
|
||||
// close's own error.
|
||||
func (e *Engine) retireLocked(b *box.Box, cancel func(), gen uint64, hash string) error {
|
||||
if cancel != nil {
|
||||
cancel()
|
||||
}
|
||||
if b == nil {
|
||||
return nil
|
||||
}
|
||||
|
||||
pc := &pendingClose{gen: gen, hash: hash, since: time.Now(), done: make(chan struct{})}
|
||||
e.pendingMu.Lock()
|
||||
e.pending = append(e.pending, pc)
|
||||
e.pendingMu.Unlock()
|
||||
|
||||
closeFn := currentBoxCloser()
|
||||
errc := make(chan error, 1)
|
||||
go func() {
|
||||
err := closeFn(b)
|
||||
e.pendingMu.Lock()
|
||||
pc.finished = true
|
||||
wasAbandoned := pc.abandoned
|
||||
e.removePendingLocked(pc)
|
||||
e.pendingMu.Unlock()
|
||||
if wasAbandoned && e.log != nil {
|
||||
// The counterpart of the ERROR below: an operator who saw the leak
|
||||
// reported must be able to see it resolve without restarting anything.
|
||||
e.log.Warn("engine generation ", gen, " finally finished shutting down after ",
|
||||
time.Since(pc.since).Round(time.Millisecond), "; it is no longer running")
|
||||
}
|
||||
close(pc.done)
|
||||
errc <- err
|
||||
}()
|
||||
|
||||
budget := currentCloseBudget()
|
||||
timer := time.NewTimer(budget)
|
||||
defer timer.Stop()
|
||||
select {
|
||||
case err := <-errc:
|
||||
return err
|
||||
case <-timer.C:
|
||||
}
|
||||
|
||||
// The budget expired. Claim the generation as abandoned — unless the close
|
||||
// happened to land in the same instant, in which case there is nothing to
|
||||
// report and we take its real result.
|
||||
e.pendingMu.Lock()
|
||||
abandoned := !pc.finished
|
||||
if abandoned {
|
||||
pc.abandoned = true
|
||||
}
|
||||
e.pendingMu.Unlock()
|
||||
if !abandoned {
|
||||
return <-errc
|
||||
}
|
||||
|
||||
if e.log != nil {
|
||||
e.log.Error("engine generation ", gen, " (config ", shortHash(hash), ") did NOT shut down within ", budget,
|
||||
" and has been ABANDONED: it may still hold its outbound connections and its ",
|
||||
"WireGuard devices (two devices with the same private key evict each other at ",
|
||||
"the peer), and it keeps writing to this log. The new configuration is running; ",
|
||||
"restart shaterd if this generation never clears.")
|
||||
}
|
||||
return ErrCloseTimeout
|
||||
}
|
||||
|
||||
// shortHash renders a config hash the way the log and the panel both need it:
|
||||
// enough to identify the configuration, short enough to read in a syslog line.
|
||||
func shortHash(h string) string {
|
||||
if len(h) < 12 {
|
||||
return "unknown"
|
||||
}
|
||||
return h[:12]
|
||||
}
|
||||
|
||||
// removePendingLocked drops pc from the pending list. Caller holds pendingMu.
|
||||
func (e *Engine) removePendingLocked(pc *pendingClose) {
|
||||
for i, p := range e.pending {
|
||||
if p == pc {
|
||||
e.pending = append(e.pending[:i], e.pending[i+1:]...)
|
||||
return
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// pendingSnapshot copies the in-flight retirements. Leaf lock only.
|
||||
func (e *Engine) pendingSnapshot() []*pendingClose {
|
||||
e.pendingMu.Lock()
|
||||
defer e.pendingMu.Unlock()
|
||||
return append([]*pendingClose(nil), e.pending...)
|
||||
}
|
||||
|
||||
// PendingCloses lists the superseded generations that overran their close budget
|
||||
// and are still running. Empty is the healthy answer.
|
||||
//
|
||||
// It takes ONLY the pending leaf lock — never e.mu — so the panel and
|
||||
// `shaterd status` can report a leaking teardown even while the apply that
|
||||
// produced it is still holding the engine mutex. That is deliberate: the one
|
||||
// moment this information matters most is while an apply is stalled behind a
|
||||
// shutdown that will not finish.
|
||||
func (e *Engine) PendingCloses() []StuckClose {
|
||||
now := time.Now()
|
||||
e.pendingMu.Lock()
|
||||
defer e.pendingMu.Unlock()
|
||||
out := make([]StuckClose, 0, len(e.pending))
|
||||
for _, p := range e.pending {
|
||||
if !p.abandoned {
|
||||
continue
|
||||
}
|
||||
out = append(out, StuckClose{
|
||||
Generation: p.gen,
|
||||
Hash: p.hash,
|
||||
Since: p.since,
|
||||
Elapsed: now.Sub(p.since),
|
||||
})
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
// Generations reports how many sing-box instances this process is carrying: the
|
||||
// running one (0 or 1) plus every superseded generation whose shutdown has not
|
||||
// finished. ONE is the healthy answer for a started engine; anything above it is
|
||||
// the leak this file exists to make impossible to hide.
|
||||
func (e *Engine) Generations() int {
|
||||
e.mu.Lock()
|
||||
live := 0
|
||||
if e.instance != nil {
|
||||
live = 1
|
||||
}
|
||||
e.mu.Unlock()
|
||||
e.pendingMu.Lock()
|
||||
defer e.pendingMu.Unlock()
|
||||
return live + len(e.pending)
|
||||
}
|
||||
|
||||
// awaitAbandonedLocked gives every already-abandoned generation ONE more close
|
||||
// budget to finish, and returns how many are still running afterwards. The caller
|
||||
// holds e.mu and is about to build another instance.
|
||||
//
|
||||
// This is what keeps generations from stacking. Without it, a box that will not
|
||||
// come down means the NEXT apply quietly adds a third instance to the process,
|
||||
// and the one after that a fourth — which is precisely the accumulation observed
|
||||
// on the router. The wait is bounded by one budget for the whole set (not per
|
||||
// entry), so a permanently stuck generation costs a single extra budget on an
|
||||
// apply that actually changes the config, and nothing at all on the every-minute
|
||||
// no-op reconcile, which never gets past the hash gate.
|
||||
func (e *Engine) awaitAbandonedLocked() int {
|
||||
pending := e.pendingSnapshot()
|
||||
if len(pending) == 0 {
|
||||
return 0
|
||||
}
|
||||
deadline := time.NewTimer(currentCloseBudget())
|
||||
defer deadline.Stop()
|
||||
for _, p := range pending {
|
||||
select {
|
||||
case <-p.done:
|
||||
case <-deadline.C:
|
||||
return len(e.PendingCloses())
|
||||
}
|
||||
}
|
||||
return len(e.PendingCloses())
|
||||
}
|
||||
@@ -0,0 +1,247 @@
|
||||
// lx: pins the invariant that a chain copy of an AmneziaWG node carries the
|
||||
// SAME obfuscation parameters as the base node, field for field.
|
||||
//
|
||||
// Context: on the router a node placed behind an egress hop
|
||||
// (egress:ewan -> node:awgout, tag chain-<name>-h1) handshook forever and never
|
||||
// passed a transport packet, while the same node standalone was healthy. One
|
||||
// candidate explanation was that rebuildNode loses AWG params on the copy — it
|
||||
// does not (both the base and the copy go through the same
|
||||
// builder.wireguardEndpoint / amneziaOptions, the copy differing only in Tag and
|
||||
// DialerOptions.Detour), and this test holds that line. The real cause was the
|
||||
// bind swap the detour triggers: a detour makes Endpoint.Start pick ClientBind
|
||||
// instead of conn.StdNetBind, and ClientBind unconditionally overwrote bytes 1-3
|
||||
// of every datagram, shredding the h4 magic header of transport packets. See
|
||||
// transport/wireguard/client_bind.go (hasReserved) and its regression tests.
|
||||
//
|
||||
// Honest note: this test passes both before and after that fix — it is a pin on
|
||||
// a path that was never broken, not the reproducer for the bug.
|
||||
//
|
||||
// Unlike generate_test.go this file is NOT Linux-gated: it stops at
|
||||
// GenerateWithWarnings and never calls engine.Apply / box.New, so it needs no
|
||||
// routing_mark validation and runs on every platform.
|
||||
package generate
|
||||
|
||||
import (
|
||||
"encoding/base64"
|
||||
"fmt"
|
||||
"net/url"
|
||||
"reflect"
|
||||
"testing"
|
||||
|
||||
"github.com/sagernet/sing-box/option"
|
||||
"github.com/sagernet/sing-box/shater/model"
|
||||
)
|
||||
|
||||
// awgKey returns a valid 32-byte base64 WireGuard key seeded by fill. Local to
|
||||
// this file so it does not depend on the Linux-only suite's helpers.
|
||||
func awgKey(fill byte) string {
|
||||
b := make([]byte, 32)
|
||||
for i := range b {
|
||||
b[i] = fill + byte(i)
|
||||
}
|
||||
return base64.StdEncoding.EncodeToString(b)
|
||||
}
|
||||
|
||||
// TestChainCopyPreservesAmneziaWGOptions builds a chain whose terminal hop is an
|
||||
// AmneziaWG node and asserts the chain copy's AmneziaWGOptions equals the base
|
||||
// endpoint's, comparing every field of the struct (so a field added later
|
||||
// without being mapped in amneziaOptions is caught here too).
|
||||
func TestChainCopyPreservesAmneziaWGOptions(t *testing.T) {
|
||||
priv := awgKey(1)
|
||||
pub := awgKey(9)
|
||||
// Every knob the option struct carries that the share-link can express:
|
||||
// jc/jmin/jmax, s1-s3, ranged h1-h4 (AWG 2.0), i1-i5. s4 is deliberately left
|
||||
// unset (0) — that is the real-world shape in which the transport magic lands
|
||||
// in bytes 0-3 and the ClientBind bug bit.
|
||||
uri := fmt.Sprintf(
|
||||
"awg://%s@203.0.113.10:51820?publickey=%s&address=10.13.13.2/32&allowedips=0.0.0.0/0"+
|
||||
"&jc=4&jmin=40&jmax=70&s1=86&s2=57&s3=13"+
|
||||
"&h1=1618116899-1618116949&h2=1795397486-1795397536&h3=3333333333&h4=1618116899-1618116949"+
|
||||
"&i1=%s&i2=%s&i3=%s&i4=%s&i5=%s#awgout",
|
||||
url.QueryEscape(priv), url.QueryEscape(pub),
|
||||
url.QueryEscape("<b 0xf0>"), url.QueryEscape("<c>"), url.QueryEscape("<t>"),
|
||||
url.QueryEscape("<r 10>"), url.QueryEscape("<b 0xab>"),
|
||||
)
|
||||
|
||||
node := model.Node{Name: "awgout", Enabled: true, URI: uri}
|
||||
|
||||
// Two SEPARATE configs, not one. wgdedup refuses to materialise a WireGuard
|
||||
// node twice from one private key (two devices sharing a key evict each other's
|
||||
// session), so a single config can hold either the base endpoint or the chain
|
||||
// copy — never both. That is also why on the router this node exists only as
|
||||
// "chain-<name>-h1". The invariant under test is therefore cross-config: the
|
||||
// same node, referenced directly vs referenced through a chain, must yield the
|
||||
// same AmneziaWG parameters.
|
||||
direct := &model.Model{
|
||||
Globals: model.DefaultGlobals(),
|
||||
Nodes: []model.Node{node},
|
||||
Rules: []model.Rule{
|
||||
{Name: "default", Enabled: true, Order: 100, Target: "node:awgout"},
|
||||
},
|
||||
}
|
||||
chained := &model.Model{
|
||||
Globals: model.DefaultGlobals(),
|
||||
Nodes: []model.Node{node},
|
||||
Egresses: []model.Egress{
|
||||
{Name: "ewan", Type: "interface", Interface: "wan"},
|
||||
},
|
||||
Chains: []model.Chain{
|
||||
{Name: "viaewan", Hops: []string{"egress:ewan", "node:awgout"}},
|
||||
},
|
||||
Rules: []model.Rule{
|
||||
{Name: "default", Enabled: true, Order: 100, Target: "chain:viaewan"},
|
||||
},
|
||||
}
|
||||
|
||||
directOpts, warns, err := GenerateWithWarnings(direct)
|
||||
if err != nil {
|
||||
t.Fatalf("GenerateWithWarnings(direct): %v (warnings: %v)", err, warns)
|
||||
}
|
||||
chainedOpts, chainWarns, err := GenerateWithWarnings(chained)
|
||||
if err != nil {
|
||||
t.Fatalf("GenerateWithWarnings(chained): %v (warnings: %v)", err, chainWarns)
|
||||
}
|
||||
|
||||
base, ok := awgOptionsForTag(directOpts, "awgout")
|
||||
if !ok {
|
||||
t.Fatalf("base endpoint %q not found; tags = %v (warnings: %v)",
|
||||
"awgout", endpointTags(directOpts), warns)
|
||||
}
|
||||
if !base.IsSet() {
|
||||
t.Fatalf("base endpoint has no AmneziaWG params: %+v", base)
|
||||
}
|
||||
|
||||
// Absolute expectation, not just parity. A "copy == base" assertion alone is
|
||||
// satisfied when BOTH lose a field (they share amneziaOptions), so pin the
|
||||
// literal values the share-link carries. Adding a field to
|
||||
// option.AmneziaWGOptions without mapping it in amneziaOptions fails the
|
||||
// exhaustiveness check below.
|
||||
want := option.AmneziaWGOptions{
|
||||
Jc: 4, Jmin: 40, Jmax: 70,
|
||||
S1: 86, S2: 57, S3: 13, S4: 0,
|
||||
H1: "1618116899-1618116949",
|
||||
H2: "1795397486-1795397536",
|
||||
H3: "3333333333",
|
||||
H4: "1618116899-1618116949",
|
||||
I1: "<b 0xf0>", I2: "<c>", I3: "<t>", I4: "<r 10>", I5: "<b 0xab>",
|
||||
}
|
||||
if base != want {
|
||||
t.Fatalf("base AmneziaWG params lost/garbled in parsing:\n got %+v\nwant %+v", base, want)
|
||||
}
|
||||
// Exhaustiveness: every field of the struct must be exercised above, so a
|
||||
// newly added knob cannot slip through unmapped and unnoticed.
|
||||
assertAWGFieldsCovered(t, want)
|
||||
|
||||
// The chain hop copy: buildHopWrapper tags it chain-<chain>-h<idx>. Find it as
|
||||
// "the wireguard endpoint in the chained config" rather than hardcoding the
|
||||
// index, so a change in hop numbering does not silently no-op this test.
|
||||
var (
|
||||
copyTag string
|
||||
copyOptions option.AmneziaWGOptions
|
||||
found bool
|
||||
)
|
||||
for _, endpoint := range chainedOpts.Endpoints {
|
||||
wg, isWG := endpoint.Options.(*option.WireGuardEndpointOptions)
|
||||
if !isWG {
|
||||
continue
|
||||
}
|
||||
copyTag, copyOptions, found = endpoint.Tag, wg.AmneziaWGOptions, true
|
||||
break
|
||||
}
|
||||
if !found {
|
||||
t.Fatalf("no chain copy endpoint emitted; endpoint tags = %v (warnings: %v)",
|
||||
endpointTags(chainedOpts), chainWarns)
|
||||
}
|
||||
if copyTag == "awgout" {
|
||||
t.Fatalf("expected a per-hop chain copy tag, got the base tag %q", copyTag)
|
||||
}
|
||||
|
||||
// Field-by-field, via reflection: any field of AmneziaWGOptions the copy fails
|
||||
// to carry is named explicitly rather than hidden behind one struct diff.
|
||||
baseValue := reflect.ValueOf(base)
|
||||
copyValue := reflect.ValueOf(copyOptions)
|
||||
for i := 0; i < baseValue.NumField(); i++ {
|
||||
field := baseValue.Type().Field(i)
|
||||
want := baseValue.Field(i).Interface()
|
||||
got := copyValue.Field(i).Interface()
|
||||
if !reflect.DeepEqual(want, got) {
|
||||
t.Errorf("chain copy %q lost AmneziaWG field %s: got %#v, want %#v",
|
||||
copyTag, field.Name, got, want)
|
||||
}
|
||||
}
|
||||
if !t.Failed() && base != copyOptions {
|
||||
t.Fatalf("chain copy %q AmneziaWGOptions differ from base: %+v vs %+v",
|
||||
copyTag, copyOptions, base)
|
||||
}
|
||||
|
||||
// The copy must additionally differ from the base in exactly the way the chain
|
||||
// intends: it detours through the egress hop, the base does not.
|
||||
baseDialer, okBase := dialerForTag(directOpts, "awgout")
|
||||
copyDialer, okCopy := dialerForTag(chainedOpts, copyTag)
|
||||
if !okBase || !okCopy {
|
||||
t.Fatalf("dialer options missing: base=%v copy=%v", okBase, okCopy)
|
||||
}
|
||||
if baseDialer.Detour != "" {
|
||||
t.Errorf("base endpoint must dial directly, got Detour=%q", baseDialer.Detour)
|
||||
}
|
||||
if copyDialer.Detour == "" {
|
||||
t.Errorf("chain copy %q must carry the egress hop as Detour, got empty", copyTag)
|
||||
}
|
||||
}
|
||||
|
||||
// assertAWGFieldsCovered fails if any field of option.AmneziaWGOptions is left
|
||||
// at its zero value in the expectation, other than the ones deliberately unset:
|
||||
//
|
||||
// S4 — kept 0 on purpose; that is the shape in which the transport magic
|
||||
// lands in bytes 0-3, i.e. the configuration that broke on the router.
|
||||
// Id/Ip/Ib — WireSock masquerade sugar, mutually exclusive with an explicit I1
|
||||
// (device_awg.go rejects the combination), and I1 is set here.
|
||||
//
|
||||
// The point is that adding a knob to the struct without extending this test
|
||||
// turns into a failure here rather than silent non-coverage.
|
||||
func assertAWGFieldsCovered(t *testing.T, want option.AmneziaWGOptions) {
|
||||
t.Helper()
|
||||
deliberatelyUnset := map[string]bool{"S4": true, "Id": true, "Ip": true, "Ib": true}
|
||||
value := reflect.ValueOf(want)
|
||||
for i := 0; i < value.NumField(); i++ {
|
||||
name := value.Type().Field(i).Name
|
||||
if deliberatelyUnset[name] {
|
||||
continue
|
||||
}
|
||||
if value.Field(i).IsZero() {
|
||||
t.Errorf("AmneziaWGOptions.%s is not exercised by this test "+
|
||||
"(add it to the share-link and to `want`, or to deliberatelyUnset)", name)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// awgOptionsForTag returns the AmneziaWGOptions of the wireguard endpoint tagged
|
||||
// tag.
|
||||
func awgOptionsForTag(opts option.Options, tag string) (option.AmneziaWGOptions, bool) {
|
||||
for _, endpoint := range opts.Endpoints {
|
||||
if endpoint.Tag != tag {
|
||||
continue
|
||||
}
|
||||
wg, ok := endpoint.Options.(*option.WireGuardEndpointOptions)
|
||||
if !ok {
|
||||
return option.AmneziaWGOptions{}, false
|
||||
}
|
||||
return wg.AmneziaWGOptions, true
|
||||
}
|
||||
return option.AmneziaWGOptions{}, false
|
||||
}
|
||||
|
||||
// dialerForTag returns the DialerOptions of the wireguard endpoint tagged tag.
|
||||
func dialerForTag(opts option.Options, tag string) (option.DialerOptions, bool) {
|
||||
for _, endpoint := range opts.Endpoints {
|
||||
if endpoint.Tag != tag {
|
||||
continue
|
||||
}
|
||||
wg, ok := endpoint.Options.(*option.WireGuardEndpointOptions)
|
||||
if !ok {
|
||||
return option.DialerOptions{}, false
|
||||
}
|
||||
return wg.DialerOptions, true
|
||||
}
|
||||
return option.DialerOptions{}, false
|
||||
}
|
||||
@@ -185,7 +185,15 @@ func TestDNSFilterRemoteBlocklistHTTPClient(t *testing.T) {
|
||||
{Name: "cf", Type: "doh", Address: "https://1.1.1.1/dns-query", Detour: "direct"},
|
||||
},
|
||||
Blocklists: []model.Blocklist{
|
||||
{Name: "remote-ads", Enabled: true, Source: "url", URL: srv.URL, Response: "nxdomain", UpdateInterval: "24h"},
|
||||
// The ".srs" suffix is LOAD-BEARING, not decoration: ruleSetURLIsEngineNative
|
||||
// (ruleset.go) decides remote-vs-compiled-local by URL EXTENSION alone, and
|
||||
// this test is about the REMOTE path — the engine fetching the compiled set
|
||||
// itself through the direct outbound. httptest.NewServer's bare
|
||||
// "http://127.0.0.1:<port>" has no extension, so it fell into the TEXT-list
|
||||
// path instead: the list was downloaded by generate's own listFetcher, parsed
|
||||
// as a hosts file and compiled into a LOCAL rule-set, which every assertion
|
||||
// below then contradicted. Do not trim the suffix.
|
||||
{Name: "remote-ads", Enabled: true, Source: "url", URL: srv.URL + "/blocklist.srs", Response: "nxdomain", UpdateInterval: "24h"},
|
||||
},
|
||||
}
|
||||
|
||||
|
||||
@@ -310,6 +310,19 @@ func GenerateWithWarningsAt(m *model.Model, now time.Time) (option.Options, []st
|
||||
Route: route,
|
||||
DNS: dns,
|
||||
}
|
||||
|
||||
// LAST, on the finished config: fold away duplicate WireGuard devices.
|
||||
//
|
||||
// A WG/AWG node may be materialised several times over — the always-emitted
|
||||
// base endpoint (outbound.go), a per-chain hop copy and a chain group hop's
|
||||
// per-member copy (chain.go) — and unlike a TCP proxy copy, each of those is a
|
||||
// real device holding the SAME private key. A WireGuard peer keeps one session
|
||||
// per key, so the copies evict each other and none of them passes traffic:
|
||||
// every chain containing a WG node was dead for exactly this reason. Running
|
||||
// on the assembled opts (rather than at each producer) is what makes the
|
||||
// guarantee hold for all three paths and any future fourth. See wgdedup.go.
|
||||
b.dedupWireGuardEndpoints(&opts)
|
||||
|
||||
return opts, b.warnings, nil
|
||||
}
|
||||
|
||||
|
||||
@@ -162,14 +162,23 @@ func TestObservatoryPlanFromGeneratedConfig(t *testing.T) {
|
||||
t.Errorf("chain exit %q store = %v, want only itself (a chain prefix is nobody else's path)", chainExit, store)
|
||||
}
|
||||
}
|
||||
// The chain's intermediate node hop n1 is used (the exit detours through it) but
|
||||
// NOT probed separately — its death is visible in the exit probe, and no
|
||||
// selection depends on it. n1 IS probed via the plain "auto" group, so the
|
||||
// assertion is "no SEPARATE chain-hop-h1 job", not "n1 never dialled".
|
||||
if jobDials(jobs, "chain-hop-h1") {
|
||||
t.Errorf("plan dials intermediate chain hop chain-hop-h1 — only the exit is probed end-to-end")
|
||||
// The chain's intermediate NODE hop wrapper IS probed now, as a measurement
|
||||
// of its own: dialling chain-hop-h1 measures exactly the prefix of the path
|
||||
// up to and including hop 1, which is what lets an operator see WHICH hop
|
||||
// died instead of only that the chain did (engine/probeplan.go walkDetour,
|
||||
// surfaced through ChainHealth.Hops). Its store is only itself — a chain
|
||||
// prefix is nobody else's dial path, so the base node n1 must never inherit
|
||||
// a verdict measured through the chain.
|
||||
chainHop := "chain-hop-h1"
|
||||
if !jobDials(jobs, chainHop) {
|
||||
t.Errorf("plan does not probe intermediate chain hop %q; jobs=%v", chainHop, dialsOfJobs(jobs))
|
||||
} else {
|
||||
store := jobStore(jobs, chainHop)
|
||||
if len(store) != 1 || store[0] != chainHop {
|
||||
t.Errorf("chain hop %q store = %v, want only itself (no base alias for a chain prefix)", chainHop, store)
|
||||
}
|
||||
}
|
||||
if !used[chainExit] || !used["chain-hop-h1"] {
|
||||
if !used[chainExit] || !used[chainHop] {
|
||||
t.Errorf("used-set is missing chain hop tags; used=%v", used)
|
||||
}
|
||||
// The used groups themselves are in the used-set but never dialled (a group is
|
||||
|
||||
@@ -0,0 +1,596 @@
|
||||
package generate
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"sort"
|
||||
"strings"
|
||||
|
||||
C "github.com/sagernet/sing-box/constant"
|
||||
"github.com/sagernet/sing-box/option"
|
||||
|
||||
"github.com/sagernet/sing-box/shater/netplane"
|
||||
"github.com/sagernet/sing-box/shater/parse"
|
||||
)
|
||||
|
||||
// WireGuard endpoint de-duplication — one private key, one device.
|
||||
//
|
||||
// # The defect this exists to prevent
|
||||
//
|
||||
// Everywhere else in this package a node may be COPIED freely: a per-chain hop
|
||||
// copy (chain.go buildHopWrapper/rebuildNode) and a per-group egress copy
|
||||
// (group.go) are rebuilt fresh from the share-link so each can carry its own
|
||||
// Detour without aliasing the shared base outbound. For a vless/ss/trojan node
|
||||
// that is exactly right — a copy is just another TCP client, and two of them cost
|
||||
// two connections.
|
||||
//
|
||||
// For a WireGuard/AmneziaWG node it is NOT. Each emitted endpoint is a real
|
||||
// device holding the node's private key, and a WireGuard PEER keeps exactly ONE
|
||||
// session per public key: the peer's endpoint address is rewritten by whichever
|
||||
// device most recently authenticated. Two devices built from the same key
|
||||
// therefore evict each other continuously — with persistent keepalive on both,
|
||||
// the eviction loop never settles, and NEITHER of them passes traffic. Observed
|
||||
// on the box as two UDP sockets from shaterd to the same peer port, the server's
|
||||
// peer endpoint flapping between them, zero bytes through the tunnel and every
|
||||
// hop of the chain failing with "context deadline exceeded".
|
||||
//
|
||||
// The multiplier is that buildOutboundsAndEndpoints (outbound.go) emits the BASE
|
||||
// endpoint for every enabled node unconditionally — whether or not anything
|
||||
// references it. So a WG node used only as a chain hop always produced two
|
||||
// devices: the unused base one and the chain copy. That is not an exotic
|
||||
// configuration, it is what any chain containing a WG node looks like, which made
|
||||
// EVERY such chain permanently dead.
|
||||
//
|
||||
// # Why this is one pass over the assembled option.Options
|
||||
//
|
||||
// There are three independent code paths that can materialise a WG device (the
|
||||
// base node, a chain hop copy, a chain GROUP hop's per-member copy) and nothing
|
||||
// stops a fourth from being added. Deduplicating at each producer would have to
|
||||
// be remembered at each producer. Doing it once, on the finished option.Options,
|
||||
// catches all of them by construction and needs no cooperation from the callers:
|
||||
// whatever put a second device in the config, it is gone before the config
|
||||
// reaches box.New.
|
||||
//
|
||||
// # What is left alone
|
||||
//
|
||||
// Non-wireguard outbounds are untouched (their copies are legitimate), groups
|
||||
// keep their semantics, and a node that appears exactly once — the overwhelmingly
|
||||
// common case, including "used only inside a chain" — produces no warning at all.
|
||||
|
||||
// wgDeviceKey returns the PHYSICAL DEVICE identity of an endpoint: two endpoints
|
||||
// with equal keys would be two devices fighting over one session at the peer.
|
||||
//
|
||||
// The identity is deliberately narrow — the private key plus the peer set (public
|
||||
// key + address:port), peers sorted so ordering cannot split a pair — because
|
||||
// collapsing too much is a routing change and collapsing too little leaves the
|
||||
// bug in place. Two nodes with DIFFERENT private keys are different devices at
|
||||
// the peer and must never be merged, however similar the rest of their config is;
|
||||
// two copies of the SAME key are the same device however different their MTU,
|
||||
// AllowedIPs or Detour happen to be.
|
||||
//
|
||||
// ok=false for anything that is not a wireguard endpoint, and for one with no
|
||||
// private key at all (it cannot be identified, and it would not come up anyway).
|
||||
func wgDeviceKey(ep option.Endpoint) (string, bool) {
|
||||
if ep.Type != C.TypeWireGuard {
|
||||
return "", false
|
||||
}
|
||||
wg, ok := ep.Options.(*option.WireGuardEndpointOptions)
|
||||
if !ok {
|
||||
return "", false
|
||||
}
|
||||
priv := strings.TrimSpace(wg.PrivateKey)
|
||||
if priv == "" {
|
||||
return "", false
|
||||
}
|
||||
peers := make([]string, 0, len(wg.Peers))
|
||||
for _, p := range wg.Peers {
|
||||
peers = append(peers, fmt.Sprintf("%s@%s:%d",
|
||||
strings.TrimSpace(p.PublicKey), strings.TrimSpace(p.Address), p.Port))
|
||||
}
|
||||
sort.Strings(peers)
|
||||
return priv + "|" + strings.Join(peers, ","), true
|
||||
}
|
||||
|
||||
// wgPrivateKeyOf recovers the private-key half of a wgDeviceKey.
|
||||
func wgPrivateKeyOf(deviceKey string) string {
|
||||
priv, _, _ := strings.Cut(deviceKey, "|")
|
||||
return priv
|
||||
}
|
||||
|
||||
// dedupWireGuardEndpoints enforces "one private key = at most one live device" on
|
||||
// the fully assembled config. Called from GenerateWithWarningsAt once opts is
|
||||
// complete (chain outbounds/endpoints folded in, route built), because only then
|
||||
// is every producer's output visible and the reachability walk meaningful.
|
||||
//
|
||||
// Per group of endpoints sharing a device identity:
|
||||
//
|
||||
// - endpoints nothing can route to are DELETED, silently. This is the normal
|
||||
// case (the always-emitted base endpoint of a node used only in a chain) and
|
||||
// warning about it would train the operator to ignore the warning list.
|
||||
// - if more than one REACHABLE copy remains, the config asks for something a
|
||||
// single key cannot express — e.g. the same WG node entered over two different
|
||||
// WANs in two chains, which is two devices by definition. The first copy in
|
||||
// tag order is kept (deterministic across runs), the rest are deleted and every
|
||||
// reference to them is rewritten to `block`, and each is reported as critical.
|
||||
// - exactly one left: nothing to say.
|
||||
//
|
||||
// Deleting rather than merely re-pointing is essential: box.New starts EVERY
|
||||
// endpoint in the config regardless of whether anything routes to it, so a
|
||||
// duplicate left in the list would still bring its device up and still fight for
|
||||
// the session. Re-pointing alone would have fixed nothing.
|
||||
func (b *builder) dedupWireGuardEndpoints(opts *option.Options) {
|
||||
if opts == nil || len(opts.Endpoints) < 2 {
|
||||
return
|
||||
}
|
||||
|
||||
tagsByKey := map[string][]string{}
|
||||
var keyOrder []string
|
||||
for i := range opts.Endpoints {
|
||||
key, ok := wgDeviceKey(opts.Endpoints[i])
|
||||
if !ok {
|
||||
continue
|
||||
}
|
||||
if _, seen := tagsByKey[key]; !seen {
|
||||
keyOrder = append(keyOrder, key)
|
||||
}
|
||||
tagsByKey[key] = append(tagsByKey[key], opts.Endpoints[i].Tag)
|
||||
}
|
||||
var dupKeys []string
|
||||
for _, key := range keyOrder {
|
||||
if len(tagsByKey[key]) > 1 {
|
||||
dupKeys = append(dupKeys, key)
|
||||
}
|
||||
}
|
||||
if len(dupKeys) == 0 {
|
||||
return // the common case: no key is materialised twice, nothing to do
|
||||
}
|
||||
|
||||
reachable := reachableOptionTags(opts, b.subscriptionDetourSeeds()...)
|
||||
var names map[string]string // private key -> node name, built only if we must name one
|
||||
drop := map[string]bool{}
|
||||
for _, key := range dupKeys {
|
||||
tags := append([]string(nil), tagsByKey[key]...)
|
||||
sort.Strings(tags) // deterministic survivor, independent of emission order
|
||||
var live []string
|
||||
for _, tag := range tags {
|
||||
if reachable[tag] {
|
||||
live = append(live, tag)
|
||||
continue
|
||||
}
|
||||
drop[tag] = true // dead weight: a device nothing routes to
|
||||
}
|
||||
if len(live) < 2 {
|
||||
continue
|
||||
}
|
||||
if names == nil {
|
||||
names = b.wgNodeNamesByKey()
|
||||
}
|
||||
name := names[wgPrivateKeyOf(key)]
|
||||
if name == "" {
|
||||
name = live[0] // unknown to the model (defensive): name it by its tag
|
||||
}
|
||||
keep := live[0]
|
||||
for _, tag := range live[1:] {
|
||||
drop[tag] = true
|
||||
b.warnf("node %q: this WireGuard node is materialised twice in the engine config — as %q and as %q — and traffic can reach both. A WireGuard peer keeps ONE session per public key, so two devices built from one private key evict each other continuously and NEITHER tunnel passes traffic. Only %q is kept; everything that routed through %q is fail-closed (blocked) instead of leaving over the plain WAN. Give the second path its own WireGuard node (its own key), or route both paths through the same one",
|
||||
name, keep, tag, keep, tag)
|
||||
}
|
||||
}
|
||||
if len(drop) == 0 {
|
||||
return
|
||||
}
|
||||
|
||||
// Rewrite first, delete second: the rewrite must still see the tags it is
|
||||
// replacing. Every dangling reference is pointed at `block`, never at `direct`
|
||||
// — a consumer whose tunnel just disappeared must stop, not fall out onto the
|
||||
// plain WAN with the router's real address.
|
||||
retargetDroppedTags(opts, drop, tagBlock)
|
||||
removeDroppedEndpoints(opts, drop)
|
||||
}
|
||||
|
||||
// wgNodeNamesByKey maps a WireGuard private key back to the model node name that
|
||||
// carries it, so a warning can name the NODE the operator configured rather than
|
||||
// the generated tag of a copy ("chain-ewan-wg-subs-h2" means nothing on the Nodes
|
||||
// page). Built by re-parsing the share-links, which keeps this file self-contained:
|
||||
// no producer has to remember to register its copies here.
|
||||
//
|
||||
// Built lazily (only when a critical duplicate is actually reported) because a
|
||||
// subscription can hold hundreds of nodes and this parses all of them.
|
||||
// First-wins on a key shared by two nodes: they are the same device anyway, so
|
||||
// either name identifies it for the operator.
|
||||
func (b *builder) wgNodeNamesByKey() map[string]string {
|
||||
names := map[string]string{}
|
||||
for i := range b.m.Nodes {
|
||||
n := b.m.Nodes[i]
|
||||
if !n.Enabled {
|
||||
continue
|
||||
}
|
||||
p, err := parse.ParseShareLink(n.URI)
|
||||
if err != nil || p.WG == nil {
|
||||
continue
|
||||
}
|
||||
key := strings.TrimSpace(p.WG.PrivateKey)
|
||||
if key == "" {
|
||||
continue
|
||||
}
|
||||
if _, seen := names[key]; !seen {
|
||||
names[key] = n.Name
|
||||
}
|
||||
}
|
||||
return names
|
||||
}
|
||||
|
||||
// --- reachability over option structures ------------------------------------
|
||||
|
||||
// optionNode is one outbound/endpoint reduced to what the reachability walk needs.
|
||||
type optionNode struct {
|
||||
group bool // selector/urltest: routes through its members
|
||||
members []string // group members (+ its default, if any)
|
||||
detour string // ordinary outbound/endpoint: its DialerOptions.Detour
|
||||
}
|
||||
|
||||
// reachableOptionTags is the option-layer twin of route.walkReachable
|
||||
// (route/reachability_lx.go): the set of outbound/endpoint tags traffic can
|
||||
// currently reach. That one walks live adapters inside a running box, which does
|
||||
// not exist yet at generate time, so the same question is answered over the
|
||||
// option structures.
|
||||
//
|
||||
// One deliberate difference: a group contributes ALL of its members, not just the
|
||||
// one it would select right now. The runtime walk can ask a selector what it has
|
||||
// chosen; here nothing has been chosen yet, and treating the unselected members as
|
||||
// unreachable would delete an endpoint the group is free to switch to a second
|
||||
// later. Over-approximating is the safe direction — the worst it costs is a
|
||||
// duplicate reported as critical instead of being removed silently.
|
||||
// extraSeeds are references that exist OUTSIDE option.Options — today the
|
||||
// subscription fetch detours (see subscriptionDetourSeeds). They are entry points
|
||||
// exactly like a route rule, so they are walked identically.
|
||||
func reachableOptionTags(opts *option.Options, extraSeeds ...string) map[string]bool {
|
||||
graph := optionGraph(opts)
|
||||
reachable := map[string]bool{}
|
||||
for _, seed := range optionSeedTags(opts) {
|
||||
walkOptionReachable(seed, graph, reachable)
|
||||
}
|
||||
for _, seed := range extraSeeds {
|
||||
walkOptionReachable(seed, graph, reachable)
|
||||
}
|
||||
return reachable
|
||||
}
|
||||
|
||||
// subscriptionDetourSeeds returns the outbound tags the SUBSCRIPTION fetcher dials
|
||||
// through — references that are just as real as a route rule's, but that live in
|
||||
// the model rather than in option.Options and so are invisible to the walk above.
|
||||
//
|
||||
// A subscription with fetch_via=proxy is pulled through the engine outbound named
|
||||
// by its fetch_detour: apply.UpdateSubscription (shater/apply/apply.go) hands that
|
||||
// string to engine.HTTPClient, which resolves it with engine.ViaToTag and looks the
|
||||
// tag up in the RUNNING box. So `fetch_detour=node:awg` is a direct, load-bearing
|
||||
// use of that node's BASE endpoint. Without this seed the base endpoint looked
|
||||
// unreferenced, was deleted as a duplicate of the node's chain copy, and updating
|
||||
// the subscription failed with "unknown outbound tag" — a working setup broken by
|
||||
// a pass that is supposed to fix one.
|
||||
//
|
||||
// Enabled is deliberately NOT consulted: UpdateSubscription looks a subscription up
|
||||
// by name and never checks it, so the panel can fetch a disabled one and the detour
|
||||
// must resolve when it does.
|
||||
//
|
||||
// Non-node forms come along for free rather than being filtered out: one mapping
|
||||
// (viaOutboundTag) covers them all, and seeding `egress-<x>` or a group tag costs
|
||||
// nothing — a tag that does not exist is ignored by the walk. Filtering would be
|
||||
// extra code that could only make the result less correct (a WG node that is a
|
||||
// member of a group used ONLY as a fetch detour would lose its endpoint).
|
||||
func (b *builder) subscriptionDetourSeeds() []string {
|
||||
var seeds []string
|
||||
for i := range b.m.Subscriptions {
|
||||
sub := b.m.Subscriptions[i]
|
||||
if !strings.EqualFold(strings.TrimSpace(sub.FetchVia), "proxy") {
|
||||
continue // a direct fetch dials no outbound at all
|
||||
}
|
||||
seeds = append(seeds, viaOutboundTag(sub.FetchDetour))
|
||||
}
|
||||
return seeds
|
||||
}
|
||||
|
||||
// viaOutboundTag mirrors engine.ViaToTag (shater/engine/httpclient.go) EXACTLY:
|
||||
// the shater `via` selector -> the box-internal outbound tag it names.
|
||||
//
|
||||
// ""/"direct" -> "direct"
|
||||
// "group:<X>" -> "<X>"
|
||||
// "node:<X>" -> "<X>"
|
||||
// "egress:<X>" -> "egress-<X>"
|
||||
// "chain:<X>" -> "<X>"
|
||||
// "<X>" -> "<X>"
|
||||
//
|
||||
// Duplicated rather than imported, following the same rule the engine side already
|
||||
// applies to the generator's tag formats (see engine/grouphealth_test.go copyTag):
|
||||
// shater/generate is the layer BELOW the runtime and must not pull the whole engine
|
||||
// in for one string function. TestViaOutboundTagMatchesEngine is the tripwire that
|
||||
// fails the moment the two drift apart.
|
||||
//
|
||||
// Note this is NOT b.resolveTarget: that one validates a target against the emitted
|
||||
// tag sets and MATERIALISES a chain as a side effect. This is a pure spelling of
|
||||
// what the runtime will look up, which is the only question the walk is asking.
|
||||
func viaOutboundTag(via string) string {
|
||||
via = strings.TrimSpace(via)
|
||||
if via == "" || strings.EqualFold(via, tagDirect) {
|
||||
return tagDirect
|
||||
}
|
||||
for _, prefix := range []string{"group:", "node:", "chain:"} {
|
||||
if rest, ok := cutPrefixFold(via, prefix); ok {
|
||||
return strings.TrimSpace(rest)
|
||||
}
|
||||
}
|
||||
if rest, ok := cutPrefixFold(via, "egress:"); ok {
|
||||
return netplane.EgressOutboundTag(strings.TrimSpace(rest))
|
||||
}
|
||||
return via
|
||||
}
|
||||
|
||||
// cutPrefixFold is strings.CutPrefix with a case-insensitive prefix match.
|
||||
func cutPrefixFold(s, prefix string) (string, bool) {
|
||||
if len(s) >= len(prefix) && strings.EqualFold(s[:len(prefix)], prefix) {
|
||||
return s[len(prefix):], true
|
||||
}
|
||||
return "", false
|
||||
}
|
||||
|
||||
// optionGraph indexes every outbound/endpoint tag to its outgoing references.
|
||||
func optionGraph(opts *option.Options) map[string]optionNode {
|
||||
graph := make(map[string]optionNode, len(opts.Outbounds)+len(opts.Endpoints))
|
||||
for i := range opts.Outbounds {
|
||||
ob := opts.Outbounds[i]
|
||||
if members, def, ok := groupMembersOf(ob.Options); ok {
|
||||
all := append([]string(nil), members...)
|
||||
if def != "" {
|
||||
all = append(all, def)
|
||||
}
|
||||
graph[ob.Tag] = optionNode{group: true, members: all}
|
||||
continue
|
||||
}
|
||||
graph[ob.Tag] = optionNode{detour: dialerDetour(ob.Options)}
|
||||
}
|
||||
for i := range opts.Endpoints {
|
||||
graph[opts.Endpoints[i].Tag] = optionNode{detour: dialerDetour(opts.Endpoints[i].Options)}
|
||||
}
|
||||
return graph
|
||||
}
|
||||
|
||||
// optionSeedTags collects every tag traffic can ENTER the outbound graph at: the
|
||||
// route's Final (the kill-switch backstop), each route rule's routed/bypassed
|
||||
// outbound, each DNS server's detour, and the download detours of the remote
|
||||
// rule-sets / geo sources / clash external UI (those dial through an outbound too,
|
||||
// so a tag one of them names is in use even if no user traffic reaches it).
|
||||
func optionSeedTags(opts *option.Options) []string {
|
||||
var seeds []string
|
||||
if rt := opts.Route; rt != nil {
|
||||
seeds = append(seeds, rt.Final)
|
||||
for i := range rt.Rules {
|
||||
seeds = append(seeds, ruleActionOutbound(rt.Rules[i]))
|
||||
}
|
||||
for i := range rt.RuleSet {
|
||||
if rt.RuleSet[i].Type == C.RuleSetTypeRemote {
|
||||
seeds = append(seeds, rt.RuleSet[i].RemoteOptions.DownloadDetour)
|
||||
}
|
||||
}
|
||||
if rt.GeoIP != nil {
|
||||
seeds = append(seeds, rt.GeoIP.DownloadDetour)
|
||||
}
|
||||
if rt.Geosite != nil {
|
||||
seeds = append(seeds, rt.Geosite.DownloadDetour)
|
||||
}
|
||||
}
|
||||
if opts.DNS != nil {
|
||||
for i := range opts.DNS.Servers {
|
||||
seeds = append(seeds, dialerDetour(opts.DNS.Servers[i].Options))
|
||||
}
|
||||
}
|
||||
if opts.Experimental != nil && opts.Experimental.ClashAPI != nil {
|
||||
seeds = append(seeds, opts.Experimental.ClashAPI.ExternalUIDownloadDetour)
|
||||
}
|
||||
return seeds
|
||||
}
|
||||
|
||||
// walkOptionReachable marks tag reachable and descends into everything it routes
|
||||
// through, transitively. The visited set doubles as the cycle guard.
|
||||
func walkOptionReachable(tag string, graph map[string]optionNode, reachable map[string]bool) {
|
||||
if tag == "" || reachable[tag] {
|
||||
return
|
||||
}
|
||||
reachable[tag] = true
|
||||
node, ok := graph[tag]
|
||||
if !ok {
|
||||
return // dangling reference; box.New reports it, this pass does not care
|
||||
}
|
||||
if node.group {
|
||||
for _, member := range node.members {
|
||||
walkOptionReachable(member, graph, reachable)
|
||||
}
|
||||
return
|
||||
}
|
||||
walkOptionReachable(node.detour, graph, reachable)
|
||||
}
|
||||
|
||||
// ruleActionOutbound returns the outbound tag a route rule's action sends traffic
|
||||
// to, or "" for the actions that route nowhere (reject/sniff/dns/hijack-dns/…).
|
||||
// An empty Action is the route action (see option.RuleAction.UnmarshalJSON), so it
|
||||
// is read the same way.
|
||||
func ruleActionOutbound(rule option.Rule) string {
|
||||
action := rule.DefaultOptions.RuleAction
|
||||
if rule.Type == C.RuleTypeLogical {
|
||||
action = rule.LogicalOptions.RuleAction
|
||||
}
|
||||
switch action.Action {
|
||||
case C.RuleActionTypeRoute, "":
|
||||
return action.RouteOptions.Outbound
|
||||
case C.RuleActionTypeBypass:
|
||||
return action.BypassOptions.Outbound
|
||||
}
|
||||
return ""
|
||||
}
|
||||
|
||||
// --- rewriting + removal ----------------------------------------------------
|
||||
|
||||
// dialerDetour reads DialerOptions.Detour off any typed options that carry a
|
||||
// dialer (every *OutboundOptions, WireGuardEndpointOptions and DNS server
|
||||
// transport embeds option.DialerOptions). "" for options that carry none.
|
||||
func dialerDetour(o any) string {
|
||||
w, ok := o.(option.DialerOptionsWrapper)
|
||||
if !ok {
|
||||
return ""
|
||||
}
|
||||
return w.TakeDialerOptions().Detour
|
||||
}
|
||||
|
||||
// groupMembersOf reports a group outbound's member list and its default member.
|
||||
// ok=false for anything that is not a selector/urltest.
|
||||
func groupMembersOf(o any) (members []string, def string, ok bool) {
|
||||
switch g := o.(type) {
|
||||
case *option.SelectorOutboundOptions:
|
||||
return g.Outbounds, g.Default, true
|
||||
case *option.URLTestOutboundOptions:
|
||||
return g.Outbounds, "", true
|
||||
}
|
||||
return nil, "", false
|
||||
}
|
||||
|
||||
// retargetDroppedTags repoints every reference to a dropped endpoint tag, so the
|
||||
// config stays internally consistent once the endpoint is gone. A dangling tag is
|
||||
// not merely untidy: box.New resolves group members and route targets eagerly and
|
||||
// refuses the whole config over one of them, which with the kill-switch closed is
|
||||
// the entire LAN offline.
|
||||
//
|
||||
// Detours, route targets and the route Final go to `to` (block) — the fail-closed
|
||||
// direction. Group MEMBER lists instead drop the tag and only fall back to a lone
|
||||
// block member when that empties the group: a group that still has other tunnels
|
||||
// to balance over should use them, and a block sitting in a urltest pool would
|
||||
// otherwise be probed and reported dead forever on the group health card.
|
||||
func retargetDroppedTags(opts *option.Options, drop map[string]bool, to string) {
|
||||
for i := range opts.Outbounds {
|
||||
if !retargetGroupMembers(opts.Outbounds[i].Options, drop, to) {
|
||||
retargetDetour(opts.Outbounds[i].Options, drop, to)
|
||||
}
|
||||
}
|
||||
for i := range opts.Endpoints {
|
||||
retargetDetour(opts.Endpoints[i].Options, drop, to)
|
||||
}
|
||||
if opts.DNS != nil {
|
||||
for i := range opts.DNS.Servers {
|
||||
retargetDetour(opts.DNS.Servers[i].Options, drop, to)
|
||||
}
|
||||
}
|
||||
if rt := opts.Route; rt != nil {
|
||||
if drop[rt.Final] {
|
||||
rt.Final = to
|
||||
}
|
||||
for i := range rt.Rules {
|
||||
retargetRuleAction(&rt.Rules[i], drop, to)
|
||||
}
|
||||
for i := range rt.RuleSet {
|
||||
if rt.RuleSet[i].Type == C.RuleSetTypeRemote && drop[rt.RuleSet[i].RemoteOptions.DownloadDetour] {
|
||||
rt.RuleSet[i].RemoteOptions.DownloadDetour = to
|
||||
}
|
||||
}
|
||||
if rt.GeoIP != nil && drop[rt.GeoIP.DownloadDetour] {
|
||||
rt.GeoIP.DownloadDetour = to
|
||||
}
|
||||
if rt.Geosite != nil && drop[rt.Geosite.DownloadDetour] {
|
||||
rt.Geosite.DownloadDetour = to
|
||||
}
|
||||
}
|
||||
if opts.Experimental != nil && opts.Experimental.ClashAPI != nil &&
|
||||
drop[opts.Experimental.ClashAPI.ExternalUIDownloadDetour] {
|
||||
opts.Experimental.ClashAPI.ExternalUIDownloadDetour = to
|
||||
}
|
||||
}
|
||||
|
||||
// retargetRuleAction points a route rule whose target was dropped at `to`. It is
|
||||
// the write half of ruleActionOutbound and must stay in step with it: a rule the
|
||||
// seed walk counted as a reference is a rule this has to be able to repoint.
|
||||
func retargetRuleAction(rule *option.Rule, drop map[string]bool, to string) {
|
||||
action := &rule.DefaultOptions.RuleAction
|
||||
if rule.Type == C.RuleTypeLogical {
|
||||
action = &rule.LogicalOptions.RuleAction
|
||||
}
|
||||
switch action.Action {
|
||||
case C.RuleActionTypeRoute, "":
|
||||
if drop[action.RouteOptions.Outbound] {
|
||||
action.RouteOptions.Outbound = to
|
||||
}
|
||||
case C.RuleActionTypeBypass:
|
||||
if drop[action.BypassOptions.Outbound] {
|
||||
action.BypassOptions.Outbound = to
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// retargetDetour points a dropped Detour at `to`, leaving any other value alone.
|
||||
func retargetDetour(o any, drop map[string]bool, to string) {
|
||||
w, ok := o.(option.DialerOptionsWrapper)
|
||||
if !ok {
|
||||
return
|
||||
}
|
||||
d := w.TakeDialerOptions()
|
||||
if !drop[d.Detour] {
|
||||
return
|
||||
}
|
||||
d.Detour = to
|
||||
w.ReplaceDialerOptions(d)
|
||||
}
|
||||
|
||||
// retargetGroupMembers prunes dropped members from a group, reporting whether the
|
||||
// options were a group at all (so the caller knows not to look for a detour).
|
||||
// A selector whose Default was dropped falls back to an empty Default, which the
|
||||
// engine reads as "the first member" — always a member that still exists.
|
||||
func retargetGroupMembers(o any, drop map[string]bool, to string) bool {
|
||||
switch g := o.(type) {
|
||||
case *option.SelectorOutboundOptions:
|
||||
g.Outbounds = pruneMembers(g.Outbounds, drop, to)
|
||||
if drop[g.Default] {
|
||||
g.Default = ""
|
||||
}
|
||||
return true
|
||||
case *option.URLTestOutboundOptions:
|
||||
g.Outbounds = pruneMembers(g.Outbounds, drop, to)
|
||||
return true
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
// pruneMembers removes dropped tags from a member list, substituting a single
|
||||
// fallback member when that would leave the group empty (an empty selector/urltest
|
||||
// is refused by box.New, and the fallback is `block` so the emptied group is
|
||||
// fail-closed rather than a hole).
|
||||
func pruneMembers(in []string, drop map[string]bool, fallback string) []string {
|
||||
hit := false
|
||||
for _, tag := range in {
|
||||
if drop[tag] {
|
||||
hit = true
|
||||
break
|
||||
}
|
||||
}
|
||||
if !hit {
|
||||
return in
|
||||
}
|
||||
out := make([]string, 0, len(in))
|
||||
for _, tag := range in {
|
||||
if drop[tag] {
|
||||
continue
|
||||
}
|
||||
out = append(out, tag)
|
||||
}
|
||||
if len(out) == 0 {
|
||||
return []string{fallback}
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
// removeDroppedEndpoints filters the dropped endpoints out of the config. This is
|
||||
// the step that actually stops the duplicate device from being created.
|
||||
func removeDroppedEndpoints(opts *option.Options, drop map[string]bool) {
|
||||
kept := opts.Endpoints[:0]
|
||||
for _, ep := range opts.Endpoints {
|
||||
if drop[ep.Tag] {
|
||||
continue
|
||||
}
|
||||
kept = append(kept, ep)
|
||||
}
|
||||
opts.Endpoints = kept
|
||||
}
|
||||
@@ -0,0 +1,416 @@
|
||||
package generate
|
||||
|
||||
import (
|
||||
"encoding/base64"
|
||||
"fmt"
|
||||
"net/url"
|
||||
"sort"
|
||||
"strings"
|
||||
"testing"
|
||||
|
||||
C "github.com/sagernet/sing-box/constant"
|
||||
"github.com/sagernet/sing-box/option"
|
||||
|
||||
"github.com/sagernet/sing-box/shater/engine"
|
||||
"github.com/sagernet/sing-box/shater/model"
|
||||
)
|
||||
|
||||
// These tests pin the one invariant a WireGuard node cannot survive without: the
|
||||
// generated config must never contain two devices built from one private key. See
|
||||
// wgdedup.go for why (a peer keeps a single session per key, so two devices evict
|
||||
// each other and neither passes traffic).
|
||||
//
|
||||
// They are deliberately NOT Linux-gated: nothing here goes through box.New, so the
|
||||
// `routing_mark` platform restriction that gates generate_test.go does not apply.
|
||||
|
||||
// wgDedupKey returns a valid 32-byte base64 WireGuard key seeded by fill. It is a
|
||||
// separate helper from generate_test.go's validKey on purpose: that file is
|
||||
// //go:build linux, and this suite must run on every dev platform.
|
||||
func wgDedupKey(fill byte) string {
|
||||
b := make([]byte, 32)
|
||||
for i := range b {
|
||||
b[i] = fill + byte(i)
|
||||
}
|
||||
return base64.StdEncoding.EncodeToString(b)
|
||||
}
|
||||
|
||||
// wgDedupURI builds a wireguard:// share-link for a node.
|
||||
func wgDedupURI(priv, pub, server string, port int) string {
|
||||
return fmt.Sprintf("wireguard://%s@%s:%d?publickey=%s&address=10.13.13.2/32&allowedips=0.0.0.0/0",
|
||||
url.QueryEscape(priv), server, port, url.QueryEscape(pub))
|
||||
}
|
||||
|
||||
// endpointTags lists the emitted endpoint tags, sorted for stable comparison.
|
||||
func endpointTags(opts option.Options) []string {
|
||||
out := make([]string, 0, len(opts.Endpoints))
|
||||
for i := range opts.Endpoints {
|
||||
out = append(out, opts.Endpoints[i].Tag)
|
||||
}
|
||||
sort.Strings(out)
|
||||
return out
|
||||
}
|
||||
|
||||
// routeRuleOutbounds lists the outbound tag of every route rule that routes.
|
||||
func routeRuleOutbounds(rt *option.RouteOptions) []string {
|
||||
if rt == nil {
|
||||
return nil
|
||||
}
|
||||
var out []string
|
||||
for i := range rt.Rules {
|
||||
if tag := ruleActionOutbound(rt.Rules[i]); tag != "" {
|
||||
out = append(out, tag)
|
||||
}
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
func warningsContaining(warns []string, needle string) []string {
|
||||
var out []string
|
||||
for _, w := range warns {
|
||||
if strings.Contains(w, needle) {
|
||||
out = append(out, w)
|
||||
}
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
// (а) A WG node used ONLY as a chain hop must leave exactly ONE wireguard
|
||||
// endpoint behind, and it must be the chain copy: the base endpoint that
|
||||
// buildOutboundsAndEndpoints emits for every enabled node is the second device
|
||||
// that killed the tunnel, and nothing routes to it. Silent — this is the normal
|
||||
// shape of any chain containing a WG node, not an operator error.
|
||||
//
|
||||
// (в) rides along here: the non-WG hop (ss1) keeps BOTH its base outbound and its
|
||||
// chain copy, because two TCP clients are not a conflict.
|
||||
func TestWGDedupChainOnlyNodeDropsBaseEndpoint(t *testing.T) {
|
||||
priv, pub := wgDedupKey(1), wgDedupKey(9)
|
||||
m := &model.Model{
|
||||
Globals: model.DefaultGlobals(),
|
||||
Nodes: []model.Node{
|
||||
{Name: "wg1", Enabled: true, URI: wgDedupURI(priv, pub, "203.0.113.10", 51820)},
|
||||
{Name: "ss1", Enabled: true, URI: "ss://aes-256-gcm:secret@203.0.113.2:8388#ss1"},
|
||||
},
|
||||
Chains: []model.Chain{
|
||||
{Name: "c", Hops: []string{"node:wg1", "node:ss1"}},
|
||||
},
|
||||
Rules: []model.Rule{
|
||||
{Name: "via-chain", Enabled: true, Order: 10, DstPort: "443", Target: "chain:c"},
|
||||
},
|
||||
}
|
||||
|
||||
opts, warns, err := GenerateWithWarnings(m)
|
||||
if err != nil {
|
||||
t.Fatalf("Generate: %v", err)
|
||||
}
|
||||
if got := endpointTags(opts); len(got) != 1 || got[0] != "chain-c-h1" {
|
||||
t.Fatalf("endpoints = %v, want exactly [chain-c-h1] (base wg1 removed)", got)
|
||||
}
|
||||
if opts.Endpoints[0].Type != C.TypeWireGuard {
|
||||
t.Fatalf("surviving endpoint type = %q, want %q", opts.Endpoints[0].Type, C.TypeWireGuard)
|
||||
}
|
||||
// The surviving copy must still be the chain's L1: dialled directly from the
|
||||
// router (no detour), with h2 detouring into it.
|
||||
if d := anyDetour(t, opts, "chain-c-h1"); d != "" {
|
||||
t.Fatalf("chain-c-h1 detour = %q, want \"\" (L1 dials directly)", d)
|
||||
}
|
||||
if d := anyDetour(t, opts, "chain-c-h2"); d != "chain-c-h1" {
|
||||
t.Fatalf("chain-c-h2 detour = %q, want chain-c-h1", d)
|
||||
}
|
||||
// (в) the non-WG node keeps its base outbound AND its chain copy.
|
||||
if obByTag(opts, "ss1") == nil {
|
||||
t.Fatalf("base outbound of the non-wireguard node was removed; outbounds=%v", outboundTags(opts))
|
||||
}
|
||||
if obByTag(opts, "chain-c-h2") == nil {
|
||||
t.Fatalf("chain copy of the non-wireguard node missing; outbounds=%v", outboundTags(opts))
|
||||
}
|
||||
// Silent: nothing was misconfigured.
|
||||
if got := warningsContaining(warns, "materialised twice"); len(got) != 0 {
|
||||
t.Fatalf("the ordinary chain case must not warn, got %v", got)
|
||||
}
|
||||
}
|
||||
|
||||
// (б) The same WG node entered over two DIFFERENT egresses in two chains asks for
|
||||
// two devices from one key, which the peer cannot give. One copy survives (first
|
||||
// in tag order, deterministic); the loser is deleted and its consumer is pointed
|
||||
// at `block` — never at direct, which would put that traffic on the plain WAN with
|
||||
// the router's real address. The operator is told, critically.
|
||||
func TestWGDedupTwoChainsDifferentEgressFailClosed(t *testing.T) {
|
||||
priv, pub := wgDedupKey(2), wgDedupKey(7)
|
||||
m := &model.Model{
|
||||
Globals: model.DefaultGlobals(),
|
||||
Nodes: []model.Node{
|
||||
{Name: "wg1", Enabled: true, URI: wgDedupURI(priv, pub, "203.0.113.10", 51820)},
|
||||
},
|
||||
Egresses: []model.Egress{
|
||||
{Name: "w1", Type: "interface", Interface: "eth0"},
|
||||
{Name: "w2", Type: "interface", Interface: "eth1"},
|
||||
},
|
||||
Chains: []model.Chain{
|
||||
{Name: "a", Hops: []string{"egress:w1", "node:wg1"}},
|
||||
{Name: "b", Hops: []string{"egress:w2", "node:wg1"}},
|
||||
},
|
||||
Rules: []model.Rule{
|
||||
{Name: "over-w1", Enabled: true, Order: 10, DstPort: "443", Target: "chain:a"},
|
||||
{Name: "over-w2", Enabled: true, Order: 20, DstPort: "8443", Target: "chain:b"},
|
||||
},
|
||||
}
|
||||
|
||||
opts, warns, err := GenerateWithWarnings(m)
|
||||
if err != nil {
|
||||
t.Fatalf("Generate: %v", err)
|
||||
}
|
||||
if got := endpointTags(opts); len(got) != 1 || got[0] != "chain-a-h1" {
|
||||
t.Fatalf("endpoints = %v, want exactly [chain-a-h1] (one device per key)", got)
|
||||
}
|
||||
// The survivor keeps its own entry egress; the loser's rule is fail-closed.
|
||||
if d := anyDetour(t, opts, "chain-a-h1"); d != "egress-w1" {
|
||||
t.Fatalf("chain-a-h1 detour = %q, want egress-w1", d)
|
||||
}
|
||||
targets := routeRuleOutbounds(opts.Route)
|
||||
var sawKept, sawBlocked bool
|
||||
for _, tag := range targets {
|
||||
switch tag {
|
||||
case "chain-a-h1":
|
||||
sawKept = true
|
||||
case tagBlock:
|
||||
sawBlocked = true
|
||||
case "chain-b-h1":
|
||||
t.Fatalf("a rule still routes to the deleted copy chain-b-h1: %v", targets)
|
||||
}
|
||||
}
|
||||
if !sawKept || !sawBlocked {
|
||||
t.Fatalf("route targets = %v, want the survivor kept and the loser fail-closed to %q", targets, tagBlock)
|
||||
}
|
||||
// Critical, and it must name the node plus BOTH copies so the operator can act.
|
||||
found := warningsContaining(warns, "materialised twice")
|
||||
if len(found) != 1 {
|
||||
t.Fatalf("want exactly one duplicate warning, got %v (all: %v)", found, warns)
|
||||
}
|
||||
w := found[0]
|
||||
for _, want := range []string{`node "wg1"`, "chain-a-h1", "chain-b-h1", "fail-closed"} {
|
||||
if !strings.Contains(w, want) {
|
||||
t.Fatalf("warning must contain %q, got: %s", want, w)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// (г) Two DIFFERENT WireGuard nodes are two different devices at two different
|
||||
// peers and must never be folded together, however alike the rest of their config
|
||||
// is. Both chain copies survive; both base endpoints go.
|
||||
func TestWGDedupDistinctNodesNotCollapsed(t *testing.T) {
|
||||
m := &model.Model{
|
||||
Globals: model.DefaultGlobals(),
|
||||
Nodes: []model.Node{
|
||||
{Name: "wgA", Enabled: true, URI: wgDedupURI(wgDedupKey(3), wgDedupKey(11), "203.0.113.10", 51820)},
|
||||
{Name: "wgB", Enabled: true, URI: wgDedupURI(wgDedupKey(4), wgDedupKey(12), "203.0.113.11", 51820)},
|
||||
},
|
||||
Chains: []model.Chain{
|
||||
{Name: "c", Hops: []string{"node:wgA", "node:wgB"}},
|
||||
},
|
||||
Rules: []model.Rule{
|
||||
{Name: "via-chain", Enabled: true, Order: 10, DstPort: "443", Target: "chain:c"},
|
||||
},
|
||||
}
|
||||
|
||||
opts, warns, err := GenerateWithWarnings(m)
|
||||
if err != nil {
|
||||
t.Fatalf("Generate: %v", err)
|
||||
}
|
||||
got := endpointTags(opts)
|
||||
if len(got) != 2 || got[0] != "chain-c-h1" || got[1] != "chain-c-h2" {
|
||||
t.Fatalf("endpoints = %v, want [chain-c-h1 chain-c-h2] (distinct keys kept apart)", got)
|
||||
}
|
||||
keys := map[string]bool{}
|
||||
for i := range opts.Endpoints {
|
||||
k, ok := wgDeviceKey(opts.Endpoints[i])
|
||||
if !ok {
|
||||
t.Fatalf("endpoint %q is not a keyed wireguard endpoint", opts.Endpoints[i].Tag)
|
||||
}
|
||||
keys[k] = true
|
||||
}
|
||||
if len(keys) != 2 {
|
||||
t.Fatalf("expected two distinct device identities, got %d", len(keys))
|
||||
}
|
||||
if got := warningsContaining(warns, "materialised twice"); len(got) != 0 {
|
||||
t.Fatalf("distinct nodes must not warn, got %v", got)
|
||||
}
|
||||
}
|
||||
|
||||
// (д) A WG node a rule targets DIRECTLY, with no chain copy anywhere, has exactly
|
||||
// one device already. The pass must leave it completely alone — it only ever
|
||||
// removes a duplicate, never the last copy of a node.
|
||||
func TestWGDedupDirectlyTargetedBaseEndpointKept(t *testing.T) {
|
||||
priv, pub := wgDedupKey(5), wgDedupKey(13)
|
||||
m := &model.Model{
|
||||
Globals: model.DefaultGlobals(),
|
||||
Nodes: []model.Node{
|
||||
{Name: "wg1", Enabled: true, URI: wgDedupURI(priv, pub, "203.0.113.10", 51820)},
|
||||
},
|
||||
Rules: []model.Rule{
|
||||
{Name: "direct-to-node", Enabled: true, Order: 10, DstPort: "443", Target: "node:wg1"},
|
||||
},
|
||||
}
|
||||
|
||||
opts, warns, err := GenerateWithWarnings(m)
|
||||
if err != nil {
|
||||
t.Fatalf("Generate: %v", err)
|
||||
}
|
||||
if got := endpointTags(opts); len(got) != 1 || got[0] != "wg1" {
|
||||
t.Fatalf("endpoints = %v, want exactly [wg1] (base endpoint untouched)", got)
|
||||
}
|
||||
if got := routeRuleOutbounds(opts.Route); len(got) != 1 || got[0] != "wg1" {
|
||||
t.Fatalf("route targets = %v, want [wg1] (rule not rewritten)", got)
|
||||
}
|
||||
if got := warningsContaining(warns, "materialised twice"); len(got) != 0 {
|
||||
t.Fatalf("a singly-used node must not warn, got %v", got)
|
||||
}
|
||||
}
|
||||
|
||||
// --- subscription fetch_detour is a real reference ---------------------------
|
||||
|
||||
// fetchDetourModel is the shared fixture for the three fetch_detour cases: one WG
|
||||
// node, optionally used as a chain hop, optionally named as a subscription's fetch
|
||||
// detour. Both switches independent, so the three combinations are exactly the
|
||||
// three outcomes the pass must produce.
|
||||
func fetchDetourModel(inChain bool, fetchDetour string) *model.Model {
|
||||
m := &model.Model{
|
||||
Globals: model.DefaultGlobals(),
|
||||
Nodes: []model.Node{
|
||||
{Name: "wg1", Enabled: true, URI: wgDedupURI(wgDedupKey(21), wgDedupKey(31), "203.0.113.10", 51820)},
|
||||
{Name: "ss1", Enabled: true, URI: "ss://aes-256-gcm:secret@203.0.113.2:8388#ss1"},
|
||||
},
|
||||
}
|
||||
if inChain {
|
||||
m.Chains = []model.Chain{{Name: "c", Hops: []string{"node:wg1", "node:ss1"}}}
|
||||
m.Rules = []model.Rule{{Name: "via-chain", Enabled: true, Order: 10, DstPort: "443", Target: "chain:c"}}
|
||||
} else {
|
||||
m.Rules = []model.Rule{{Name: "via-ss", Enabled: true, Order: 10, DstPort: "443", Target: "node:ss1"}}
|
||||
}
|
||||
if fetchDetour != "" {
|
||||
m.Subscriptions = []model.Subscription{{
|
||||
Name: "sub0", Enabled: true, URL: "https://example.net/sub",
|
||||
FetchVia: "proxy", FetchDetour: fetchDetour,
|
||||
}}
|
||||
}
|
||||
return m
|
||||
}
|
||||
|
||||
// Case 1 — node only in the chain, no fetch detour on it: the base endpoint is
|
||||
// genuinely unreferenced and goes, the chain works. (The pre-existing behaviour;
|
||||
// pinned here so the new seed cannot accidentally resurrect the base endpoint.)
|
||||
func TestWGDedupFetchDetourAbsentBaseStillDropped(t *testing.T) {
|
||||
opts, warns, err := GenerateWithWarnings(fetchDetourModel(true, ""))
|
||||
if err != nil {
|
||||
t.Fatalf("Generate: %v", err)
|
||||
}
|
||||
if got := endpointTags(opts); len(got) != 1 || got[0] != "chain-c-h1" {
|
||||
t.Fatalf("endpoints = %v, want exactly [chain-c-h1]", got)
|
||||
}
|
||||
if got := warningsContaining(warns, "materialised twice"); len(got) != 0 {
|
||||
t.Fatalf("must be silent, got %v", got)
|
||||
}
|
||||
}
|
||||
|
||||
// Case 2 — node in the chain AND named as a fetch detour: the config asks for the
|
||||
// node on two different dial paths, which one private key cannot provide. Both are
|
||||
// reachable, so one survives and the other is fail-closed with the critical
|
||||
// warning. Choosing silently for the operator is what we must NOT do.
|
||||
func TestWGDedupFetchDetourAndChainBothReachable(t *testing.T) {
|
||||
opts, warns, err := GenerateWithWarnings(fetchDetourModel(true, "node:wg1"))
|
||||
if err != nil {
|
||||
t.Fatalf("Generate: %v", err)
|
||||
}
|
||||
// Tag order decides: "chain-c-h1" < "wg1", so the chain copy is the survivor.
|
||||
if got := endpointTags(opts); len(got) != 1 || got[0] != "chain-c-h1" {
|
||||
t.Fatalf("endpoints = %v, want exactly [chain-c-h1] (one device per key)", got)
|
||||
}
|
||||
found := warningsContaining(warns, "materialised twice")
|
||||
if len(found) != 1 {
|
||||
t.Fatalf("want exactly one duplicate warning, got %v (all: %v)", found, warns)
|
||||
}
|
||||
for _, want := range []string{`node "wg1"`, "chain-c-h1", `"wg1"`, "fail-closed"} {
|
||||
if !strings.Contains(found[0], want) {
|
||||
t.Fatalf("warning must contain %q, got: %s", want, found[0])
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Case 3 — node ONLY named as a fetch detour, no chain copy: a single device that
|
||||
// something really does dial. It must survive untouched, or updating the
|
||||
// subscription fails with "unknown outbound tag" — the regression this seed exists
|
||||
// to prevent.
|
||||
func TestWGDedupFetchDetourOnlyKeepsBaseEndpoint(t *testing.T) {
|
||||
for _, via := range []string{"node:wg1", "wg1", "NODE:wg1", " node:wg1 "} {
|
||||
t.Run(via, func(t *testing.T) {
|
||||
opts, warns, err := GenerateWithWarnings(fetchDetourModel(false, via))
|
||||
if err != nil {
|
||||
t.Fatalf("Generate: %v", err)
|
||||
}
|
||||
if got := endpointTags(opts); len(got) != 1 || got[0] != "wg1" {
|
||||
t.Fatalf("endpoints = %v, want exactly [wg1] (the fetch detour's target)", got)
|
||||
}
|
||||
if got := warningsContaining(warns, "materialised twice"); len(got) != 0 {
|
||||
t.Fatalf("a single device must not warn, got %v", got)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
// A fetch_via=direct subscription dials no outbound at all, so its (ignored)
|
||||
// fetch_detour must NOT keep a duplicate alive.
|
||||
func TestWGDedupFetchDetourIgnoredWhenFetchViaDirect(t *testing.T) {
|
||||
m := fetchDetourModel(true, "node:wg1")
|
||||
m.Subscriptions[0].FetchVia = "direct"
|
||||
opts, warns, err := GenerateWithWarnings(m)
|
||||
if err != nil {
|
||||
t.Fatalf("Generate: %v", err)
|
||||
}
|
||||
if got := endpointTags(opts); len(got) != 1 || got[0] != "chain-c-h1" {
|
||||
t.Fatalf("endpoints = %v, want exactly [chain-c-h1] (direct fetch is not a reference)", got)
|
||||
}
|
||||
if got := warningsContaining(warns, "materialised twice"); len(got) != 0 {
|
||||
t.Fatalf("must be silent, got %v", got)
|
||||
}
|
||||
}
|
||||
|
||||
// TestViaOutboundTagMatchesEngine is the drift tripwire for the duplicated `via`
|
||||
// mapping. viaOutboundTag must agree with engine.ViaToTag on every form, because
|
||||
// the seed is only correct if it names the tag the RUNTIME will look up: if the
|
||||
// engine's mapping changes and this one does not, the pass starts deleting an
|
||||
// endpoint the subscription fetcher still dials.
|
||||
func TestViaOutboundTagMatchesEngine(t *testing.T) {
|
||||
for _, via := range []string{
|
||||
"", " ", "direct", "DIRECT",
|
||||
"node:wg1", "NODE:wg1", " node: wg1 ", "node:",
|
||||
"group:auto", "Group:auto",
|
||||
"egress:wan2", "EGRESS:wan2",
|
||||
"chain:ewan", "Chain:ewan",
|
||||
"wg1", " wg1 ", "weird:value",
|
||||
} {
|
||||
if got, want := viaOutboundTag(via), engine.ViaToTag(via); got != want {
|
||||
t.Errorf("viaOutboundTag(%q) = %q, engine.ViaToTag = %q — the mappings have drifted", via, got, want)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// An UNREFERENCED WG node is still a single device and stays: this pass is about
|
||||
// duplicates, not about pruning unused config. (Guards the >1 precondition — an
|
||||
// over-eager version would delete it and silently change what a later `selector`
|
||||
// or a hand-written detour could reach.)
|
||||
func TestWGDedupUnreferencedSingleNodeKept(t *testing.T) {
|
||||
m := &model.Model{
|
||||
Globals: model.DefaultGlobals(),
|
||||
Nodes: []model.Node{
|
||||
{Name: "wg1", Enabled: true, URI: wgDedupURI(wgDedupKey(6), wgDedupKey(14), "203.0.113.10", 51820)},
|
||||
{Name: "wg2", Enabled: true, URI: wgDedupURI(wgDedupKey(8), wgDedupKey(15), "203.0.113.11", 51820)},
|
||||
},
|
||||
}
|
||||
|
||||
opts, _, err := GenerateWithWarnings(m)
|
||||
if err != nil {
|
||||
t.Fatalf("Generate: %v", err)
|
||||
}
|
||||
if got := endpointTags(opts); len(got) != 2 {
|
||||
t.Fatalf("endpoints = %v, want both nodes kept", got)
|
||||
}
|
||||
}
|
||||
+323
-9
@@ -16,6 +16,15 @@
|
||||
// answer "give me the last day" — the prefix is what makes the download
|
||||
// ranges real. UTC only: the binary ships without tzdata, so any local-zone
|
||||
// rendering would be a fiction.
|
||||
// - collapses REPEATS of the same message into "repeated N times: <message>"
|
||||
// (see repeatKey / repeatWindow / maxRepeatKeys): a broken outbound makes
|
||||
// the engine repeat one line hundreds of times a minute, which evicts the
|
||||
// whole rest of the router's syslog ring buffer within minutes. The repeats
|
||||
// do NOT have to be adjacent — a flood usually interleaves a handful of
|
||||
// messages (one per broken chain), so the sink keeps a small table of the
|
||||
// messages seen recently instead of comparing with the previous line only.
|
||||
// Suppression applies to both halves identically; fatal/panic lines are
|
||||
// never suppressed.
|
||||
// - fans it out according to Config: to the REAL os.Stderr (procd relays fd2
|
||||
// to syslog/logread) when ToSyslog, and to a size-capped, 2-segment rotated
|
||||
// file when ToFile. Both off => the line is dropped — that IS the "fully
|
||||
@@ -42,6 +51,7 @@ package logsink
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"container/list"
|
||||
"fmt"
|
||||
"io"
|
||||
"os"
|
||||
@@ -74,6 +84,45 @@ const (
|
||||
// the log's own cap: the guard requires cap+floor available, so even a log
|
||||
// that grows to its full cap leaves the rootfs this much headroom.
|
||||
diskFloorBytes = 4 << 20
|
||||
|
||||
// repeatWindow bounds how long copies of one message may stay silent: the
|
||||
// first occurrence is printed and opens a window; every further copy of THAT
|
||||
// message inside the window is swallowed and reported by a
|
||||
// "repeated N times: <message>" summary when the window ends. A message that
|
||||
// is still flooding keeps its window (one summary per window, counter
|
||||
// restarting); a message that went quiet is forgotten, so it prints in full
|
||||
// the next time it happens.
|
||||
//
|
||||
// Why 5s: the observed flood (dead chains -> "WireGuard is not ready yet")
|
||||
// runs at ~7 lines/s across three chain tags — 436 lines in one minute —
|
||||
// while the router's syslog ring holds ~760 lines total: one faulty
|
||||
// subscription erases every other subsystem's history, including our own
|
||||
// startup lines, in under two minutes. A 5s window turns that into one
|
||||
// summary per message per window (~36 lines/min instead of 436) — short
|
||||
// enough that an operator tailing `logread -f` sees the fault acknowledged
|
||||
// within 5 seconds and keeps seeing it, so a standing problem never looks
|
||||
// like a frozen log. A longer window (30-60s) would hide the ongoing-ness; a
|
||||
// shorter one would not relieve the ring.
|
||||
repeatWindow = 5 * time.Second
|
||||
|
||||
// maxRepeatKeys bounds the table of recently seen messages. Suppression must
|
||||
// work across INTERLEAVED messages, so the sink cannot keep just the last
|
||||
// key — but the keys come from the engine and are as varied as its log, so
|
||||
// the table must not be allowed to grow with them.
|
||||
//
|
||||
// Why 256: the floods worth collapsing are per-outbound or per-chain, and a
|
||||
// pathological config on this box has a few hundred nodes of which only the
|
||||
// broken handful actually log; 256 distinct messages in flight covers that
|
||||
// with room to spare, while a table this size is trivial next to the
|
||||
// process — entries hold the log line itself, ~150 B typically (~40 KiB
|
||||
// total) and 8 KiB at the absolute worst (maxPartialLine), i.e. ~2 MiB even
|
||||
// in the case that cannot really happen.
|
||||
//
|
||||
// Eviction is least-recently-seen: the entry that has gone longest without a
|
||||
// copy is the one least likely to be flooding. Eviction is never silent —
|
||||
// an evicted entry that had swallowed copies prints its summary on the way
|
||||
// out (marked "repeat table full"), so a counter is never simply dropped.
|
||||
maxRepeatKeys = 256
|
||||
)
|
||||
|
||||
// FilePath returns the log-file location for the given persistence choice.
|
||||
@@ -153,9 +202,19 @@ type Sink struct {
|
||||
suspended bool // persistent-path disk guard tripped
|
||||
lastProbe time.Time // last disk-free probe
|
||||
|
||||
// repeat suppression (emitLocked and the repeatSeries helpers below).
|
||||
// series is the table of messages seen inside their window, keyed by
|
||||
// repeatKey; seriesLRU holds the same *repeatSeries values in
|
||||
// most-recently-seen-first order so eviction is O(1).
|
||||
series map[string]*repeatSeries
|
||||
seriesLRU *list.List
|
||||
repeatTimer *time.Timer // armed at the earliest series deadline, if any
|
||||
closed bool // Close ran: the timer must not write any more
|
||||
|
||||
// test seams
|
||||
now func() time.Time
|
||||
free func(string) (uint64, bool)
|
||||
now func() time.Time
|
||||
free func(string) (uint64, bool)
|
||||
window time.Duration // repeatWindow, overridable in tests
|
||||
}
|
||||
|
||||
// New returns a Sink fanning to stderr (the daemon passes os.Stderr) under cfg.
|
||||
@@ -168,10 +227,13 @@ func New(stderr io.Writer, cfg Config) *Sink {
|
||||
purgeLogFiles(cfg.path(), PersistPath, TmpfsPath)
|
||||
}
|
||||
return &Sink{
|
||||
cfg: cfg,
|
||||
stderr: stderr,
|
||||
now: time.Now,
|
||||
free: freeBytes,
|
||||
cfg: cfg,
|
||||
stderr: stderr,
|
||||
series: make(map[string]*repeatSeries),
|
||||
seriesLRU: list.New(),
|
||||
now: time.Now,
|
||||
free: freeBytes,
|
||||
window: repeatWindow,
|
||||
}
|
||||
}
|
||||
|
||||
@@ -210,6 +272,9 @@ func (s *Sink) Reconfigure(cfg Config) {
|
||||
if cfg == s.cfg {
|
||||
return
|
||||
}
|
||||
// Settle every series in progress under the OLD configuration: their
|
||||
// summaries belong to the destination the swallowed lines were headed for.
|
||||
s.flushAllSeriesLocked()
|
||||
if cfg.path() != s.cfg.path() || !cfg.ToFile {
|
||||
s.closeFileLocked()
|
||||
}
|
||||
@@ -222,8 +287,10 @@ func (s *Sink) Reconfigure(cfg Config) {
|
||||
s.cfg = cfg
|
||||
}
|
||||
|
||||
// Close flushes a pending partial line and closes the file segment. The sink
|
||||
// must not be written to afterwards.
|
||||
// Close flushes a pending partial line, emits the summaries of every still-open
|
||||
// series of repeats (no series may be lost at shutdown) and closes the file
|
||||
// segment. The sink must not be written to afterwards; a repeat timer that
|
||||
// fires after Close is a no-op.
|
||||
func (s *Sink) Close() error {
|
||||
s.mu.Lock()
|
||||
defer s.mu.Unlock()
|
||||
@@ -231,6 +298,8 @@ func (s *Sink) Close() error {
|
||||
s.emitLocked(s.buf)
|
||||
s.buf = nil
|
||||
}
|
||||
s.flushAllSeriesLocked()
|
||||
s.closed = true
|
||||
if s.file != nil {
|
||||
err := s.file.Close()
|
||||
s.file = nil
|
||||
@@ -242,12 +311,257 @@ func (s *Sink) Close() error {
|
||||
|
||||
// --- internals (caller holds s.mu) -------------------------------------------
|
||||
|
||||
// emitLocked stamps one complete line with the UTC wall clock and fans it out.
|
||||
// emitLocked runs one complete line through repeat suppression and, unless it
|
||||
// is swallowed as a copy of a message already printed inside its window, hands
|
||||
// it to writeLineLocked.
|
||||
func (s *Sink) emitLocked(line []byte) {
|
||||
if !s.cfg.ToSyslog && !s.cfg.ToFile {
|
||||
return // fully off: the line is dropped, nowhere else to go
|
||||
}
|
||||
line = bytes.TrimSuffix(line, []byte{'\r'})
|
||||
key, exempt := repeatKey(line)
|
||||
if exempt {
|
||||
// fatal/panic: always printed, and never opens a series — a dying
|
||||
// daemon must not have its last words counted instead of said. The
|
||||
// pending summaries go out FIRST: the process may not live long enough
|
||||
// to reach Close, and a swallowed count that is never reported is
|
||||
// exactly the failure this whole mechanism exists to avoid.
|
||||
s.flushAllSeriesLocked()
|
||||
s.writeLineLocked(line)
|
||||
return
|
||||
}
|
||||
now := s.now()
|
||||
if ser, ok := s.series[key]; ok {
|
||||
if now.Before(ser.deadline) {
|
||||
ser.count++
|
||||
s.seriesLRU.MoveToFront(ser.el)
|
||||
return // swallowed; the deadline did not move, so no re-arming
|
||||
}
|
||||
// The window elapsed without the timer having got to it (a coarse
|
||||
// timer, a frozen test clock, a burst racing the callback): settle it
|
||||
// here on exactly the same terms the sweep would have used.
|
||||
if s.expireSeriesLocked(ser, now) {
|
||||
// Still flooding: it stays suppressed, this copy opens the count of
|
||||
// the new window.
|
||||
ser.count = 1
|
||||
s.seriesLRU.MoveToFront(ser.el)
|
||||
s.armRepeatTimerLocked()
|
||||
return
|
||||
}
|
||||
}
|
||||
s.openSeriesLocked(key, now)
|
||||
s.writeLineLocked(line)
|
||||
s.armRepeatTimerLocked()
|
||||
}
|
||||
|
||||
// repeatSeries is one message inside its window: the identity that is being
|
||||
// collapsed, how many copies have been swallowed since the head line or the
|
||||
// last summary, and when the current window ends.
|
||||
type repeatSeries struct {
|
||||
key string
|
||||
count int
|
||||
deadline time.Time
|
||||
el *list.Element // this series' node in Sink.seriesLRU
|
||||
}
|
||||
|
||||
// windowLocked is the effective suppression window (tests override s.window).
|
||||
func (s *Sink) windowLocked() time.Duration {
|
||||
if s.window > 0 {
|
||||
return s.window
|
||||
}
|
||||
return repeatWindow
|
||||
}
|
||||
|
||||
// openSeriesLocked starts a window for key, evicting the least recently seen
|
||||
// series first if the table is full. An evicted series that had swallowed
|
||||
// copies reports them on the way out, so the table's size limit can shorten a
|
||||
// window but can never lose a count.
|
||||
func (s *Sink) openSeriesLocked(key string, now time.Time) {
|
||||
for len(s.series) >= maxRepeatKeys {
|
||||
back := s.seriesLRU.Back()
|
||||
if back == nil {
|
||||
break
|
||||
}
|
||||
ev := back.Value.(*repeatSeries)
|
||||
if ev.count > 0 {
|
||||
s.summariseSeriesLocked(ev, " (repeat table full)")
|
||||
}
|
||||
s.dropSeriesLocked(ev)
|
||||
}
|
||||
ser := &repeatSeries{key: key, deadline: now.Add(s.windowLocked())}
|
||||
ser.el = s.seriesLRU.PushFront(ser)
|
||||
s.series[key] = ser
|
||||
}
|
||||
|
||||
// dropSeriesLocked forgets a series entirely (its next copy prints in full).
|
||||
func (s *Sink) dropSeriesLocked(ser *repeatSeries) {
|
||||
s.seriesLRU.Remove(ser.el)
|
||||
delete(s.series, ser.key)
|
||||
}
|
||||
|
||||
// expireSeriesLocked settles a series whose window has ended and reports
|
||||
// whether it stays open. One that swallowed copies prints their summary and
|
||||
// keeps its slot for another window — an ongoing flood must be acknowledged
|
||||
// every window without re-printing its head line. One that swallowed nothing is
|
||||
// forgotten: the message occurs rarely enough that it needs no collapsing at
|
||||
// all, and holding its slot would only push a real flood out of the table.
|
||||
func (s *Sink) expireSeriesLocked(ser *repeatSeries, now time.Time) bool {
|
||||
if ser.count == 0 {
|
||||
s.dropSeriesLocked(ser)
|
||||
return false
|
||||
}
|
||||
s.summariseSeriesLocked(ser, "")
|
||||
ser.count = 0
|
||||
ser.deadline = now.Add(s.windowLocked())
|
||||
return true
|
||||
}
|
||||
|
||||
// summariseSeriesLocked prints one summary line. It NAMES the message it counts
|
||||
// (the key: the level plus the text, i.e. the line minus the uptime field that
|
||||
// repeatKey drops): with several messages collapsed at once, a bare "last
|
||||
// message repeated N times" would leave the reader unable to tell which line
|
||||
// the number belongs to — the very confusion that made the old adjacent-only
|
||||
// suppression useless in the field. note marks a summary that was forced out
|
||||
// early (table full) rather than by its window.
|
||||
func (s *Sink) summariseSeriesLocked(ser *repeatSeries, note string) {
|
||||
unit := "times"
|
||||
if ser.count == 1 {
|
||||
unit = "time"
|
||||
}
|
||||
s.writeLineLocked([]byte(fmt.Sprintf("repeated %d %s%s: %s", ser.count, unit, note, ser.key)))
|
||||
}
|
||||
|
||||
// onRepeatWindow is the timer callback: it settles every series whose window
|
||||
// has ended, so a flood is reported while it happens instead of only when it
|
||||
// stops or when the daemon closes.
|
||||
func (s *Sink) onRepeatWindow() {
|
||||
s.mu.Lock()
|
||||
defer s.mu.Unlock()
|
||||
if s.closed {
|
||||
return
|
||||
}
|
||||
now := s.now()
|
||||
for e := s.seriesLRU.Back(); e != nil; {
|
||||
prev := e.Prev() // taken before a possible removal of e
|
||||
ser := e.Value.(*repeatSeries)
|
||||
if !now.Before(ser.deadline) {
|
||||
s.expireSeriesLocked(ser, now)
|
||||
}
|
||||
e = prev
|
||||
}
|
||||
s.armRepeatTimerLocked()
|
||||
}
|
||||
|
||||
// armRepeatTimerLocked points the single repeat timer at the earliest deadline
|
||||
// in the table (and stops it when the table is empty). One timer for all series
|
||||
// keeps the cost at one goroutine wake per window, not one per message.
|
||||
func (s *Sink) armRepeatTimerLocked() {
|
||||
if s.closed {
|
||||
s.stopRepeatTimerLocked()
|
||||
return
|
||||
}
|
||||
var earliest time.Time
|
||||
for _, ser := range s.series {
|
||||
if earliest.IsZero() || ser.deadline.Before(earliest) {
|
||||
earliest = ser.deadline
|
||||
}
|
||||
}
|
||||
if earliest.IsZero() {
|
||||
s.stopRepeatTimerLocked()
|
||||
return
|
||||
}
|
||||
d := earliest.Sub(s.now())
|
||||
if d < 0 {
|
||||
d = 0
|
||||
}
|
||||
if s.repeatTimer == nil {
|
||||
s.repeatTimer = time.AfterFunc(d, s.onRepeatWindow)
|
||||
return
|
||||
}
|
||||
// A callback that already started is harmless: it blocks on s.mu and then
|
||||
// finds nothing expired.
|
||||
s.repeatTimer.Stop()
|
||||
s.repeatTimer.Reset(d)
|
||||
}
|
||||
|
||||
func (s *Sink) stopRepeatTimerLocked() {
|
||||
if s.repeatTimer != nil {
|
||||
s.repeatTimer.Stop()
|
||||
s.repeatTimer = nil
|
||||
}
|
||||
}
|
||||
|
||||
// flushAllSeriesLocked empties the table, printing a summary for every series
|
||||
// that had swallowed copies — oldest-seen first, so the summaries come out in
|
||||
// the order the messages last appeared. Used wherever the sink's state ends:
|
||||
// Close, Reconfigure (the counts belong to the destination they were headed
|
||||
// for) and a fatal/panic line.
|
||||
func (s *Sink) flushAllSeriesLocked() {
|
||||
s.stopRepeatTimerLocked()
|
||||
for e := s.seriesLRU.Back(); e != nil; e = e.Prev() {
|
||||
if ser := e.Value.(*repeatSeries); ser.count > 0 {
|
||||
s.summariseSeriesLocked(ser, "")
|
||||
}
|
||||
}
|
||||
s.seriesLRU.Init()
|
||||
s.series = make(map[string]*repeatSeries)
|
||||
}
|
||||
|
||||
// repeatKey reduces a raw producer line to the identity used for suppression,
|
||||
// and reports whether the line is EXEMPT from it.
|
||||
//
|
||||
// Both of the daemon's log producers (the control-plane formatter built in
|
||||
// cmd/shaterd/logsetup.go and the engine's own, both log.Formatter) render a
|
||||
// line as "<LEVEL>[<seconds since start>] <message>" — see log/format.go. The
|
||||
// bracketed uptime ticks every second, so comparing whole lines would suppress
|
||||
// nothing at all beyond a same-second burst; the key therefore drops exactly
|
||||
// that field and keeps everything else, LEVEL included (the same text at INFO
|
||||
// and at ERROR is not the same event).
|
||||
//
|
||||
// What the key deliberately does NOT drop is the per-connection "[id duration]"
|
||||
// group log.Formatter inserts for context-bound lines: those ids identify
|
||||
// distinct connections, and folding them together would turn "50 connections
|
||||
// failed" into one indistinguishable count. Such lines differ by id, so each
|
||||
// gets its own (single-copy, never summarised) series — which is correct, they
|
||||
// are not repeats. It also means they are the messages most likely to fill the
|
||||
// repeat table; that is what maxRepeatKeys and its least-recently-seen eviction
|
||||
// are for.
|
||||
//
|
||||
// The key doubles as the text a summary names itself with, which is why it
|
||||
// keeps the LEVEL and reads as a message on its own ("ERROR outbound/…: …").
|
||||
//
|
||||
// ANSI colour is stripped first so the key is identical for the coloured
|
||||
// (terminal) and plain (procd) renderings of the same message.
|
||||
//
|
||||
// FATAL/PANIC are exempt: they are emitted at most a handful of times, they are
|
||||
// the reason the operator is reading the log, and a summary line is a worse
|
||||
// thing to find than a duplicate.
|
||||
func repeatKey(line []byte) (string, bool) {
|
||||
b := stripANSI(line)
|
||||
i := 0
|
||||
for i < len(b) && b[i] >= 'A' && b[i] <= 'Z' {
|
||||
i++
|
||||
}
|
||||
if i > 0 && i < len(b) && b[i] == '[' {
|
||||
j := i + 1
|
||||
for j < len(b) && b[j] >= '0' && b[j] <= '9' {
|
||||
j++
|
||||
}
|
||||
if j > i+1 && j < len(b) && b[j] == ']' {
|
||||
level := string(b[:i])
|
||||
return level + string(b[j+1:]), level == "FATAL" || level == "PANIC"
|
||||
}
|
||||
}
|
||||
// Anything not in the daemon's format (a foreign writer, a bare line):
|
||||
// compare it whole.
|
||||
return string(b), false
|
||||
}
|
||||
|
||||
// writeLineLocked stamps one line with the UTC wall clock and fans it out.
|
||||
func (s *Sink) writeLineLocked(line []byte) {
|
||||
if !s.cfg.ToSyslog && !s.cfg.ToFile {
|
||||
return // fully off: the line is dropped, nowhere else to go
|
||||
}
|
||||
ts := s.now().UTC().Format(time.RFC3339)
|
||||
if s.cfg.ToSyslog && s.stderr != nil {
|
||||
out := make([]byte, 0, len(ts)+1+len(line)+1)
|
||||
|
||||
@@ -5,7 +5,10 @@ import (
|
||||
"fmt"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"regexp"
|
||||
"strconv"
|
||||
"strings"
|
||||
"sync"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
@@ -379,6 +382,521 @@ func TestReconfigureFileOffDeletesSegments(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
// --- repeat suppression -------------------------------------------------------
|
||||
|
||||
// syncBuffer is a bytes.Buffer safe for the concurrent access the repeat timer
|
||||
// goroutine and the test's reader make.
|
||||
type syncBuffer struct {
|
||||
mu sync.Mutex
|
||||
b bytes.Buffer
|
||||
}
|
||||
|
||||
func (s *syncBuffer) Write(p []byte) (int, error) {
|
||||
s.mu.Lock()
|
||||
defer s.mu.Unlock()
|
||||
return s.b.Write(p)
|
||||
}
|
||||
|
||||
func (s *syncBuffer) String() string {
|
||||
s.mu.Lock()
|
||||
defer s.mu.Unlock()
|
||||
return s.b.String()
|
||||
}
|
||||
|
||||
// engLine renders a line exactly as both of the daemon's log.Formatter
|
||||
// producers do: "<LEVEL>[<seconds since start>] <message>".
|
||||
func engLine(sec int, level, msg string) string {
|
||||
return fmt.Sprintf("%s[%04d] %s\n", level, sec, msg)
|
||||
}
|
||||
|
||||
// payloads strips the sink's RFC3339 prefix off every line of a rendered log.
|
||||
func payloads(t *testing.T, body string) []string {
|
||||
t.Helper()
|
||||
if body == "" {
|
||||
return nil
|
||||
}
|
||||
var out []string
|
||||
for _, line := range strings.Split(strings.TrimSuffix(body, "\n"), "\n") {
|
||||
if _, ok := LineTime([]byte(line)); !ok {
|
||||
t.Fatalf("line without RFC3339 prefix: %q", line)
|
||||
}
|
||||
out = append(out, line[strings.IndexByte(line, ' ')+1:])
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
// A summary names the message it counts: "repeated N time(s)[ (note)]: <msg>".
|
||||
var repeatSummaryRe = regexp.MustCompile(`^repeated (\d+) times?( \([^)]*\))?: (.*)$`)
|
||||
|
||||
// summaryByMessage returns how many suppressed copies the summaries in payloads
|
||||
// account for PER NAMED MESSAGE, and how many summary lines there were.
|
||||
func summaryByMessage(t *testing.T, lines []string) (map[string]int, int) {
|
||||
t.Helper()
|
||||
per := make(map[string]int)
|
||||
count := 0
|
||||
for _, l := range lines {
|
||||
m := repeatSummaryRe.FindStringSubmatch(l)
|
||||
if m == nil {
|
||||
continue
|
||||
}
|
||||
n, err := strconv.Atoi(m[1])
|
||||
if err != nil {
|
||||
t.Fatalf("unparseable summary %q: %v", l, err)
|
||||
}
|
||||
per[m[3]] += n
|
||||
count++
|
||||
}
|
||||
return per, count
|
||||
}
|
||||
|
||||
// summarySum returns how many suppressed lines the summaries in payloads
|
||||
// account for, and how many summary lines there were.
|
||||
func summarySum(t *testing.T, lines []string) (total, count int) {
|
||||
t.Helper()
|
||||
per, count := summaryByMessage(t, lines)
|
||||
for _, n := range per {
|
||||
total += n
|
||||
}
|
||||
return total, count
|
||||
}
|
||||
|
||||
// TestRepeatRunCollapsed: the simplest shape of the production symptom — one
|
||||
// broken chain repeating the same message back to back — leaves ONE copy of the
|
||||
// line plus ONE summary carrying the right count and naming the message, in
|
||||
// BOTH halves. The uptime field of the producer's prefix differs on every line:
|
||||
// that is exactly what repeatKey must ignore. An unrelated message in between
|
||||
// no longer ENDS the series (a series lives for its window, not until the next
|
||||
// distinct line) — it is simply printed, and the summary follows at Close.
|
||||
func TestRepeatRunCollapsed(t *testing.T) {
|
||||
var stderr syncBuffer
|
||||
cfg := fileCfg(t, true, true, 0)
|
||||
s := New(&stderr, cfg)
|
||||
s.window = time.Hour // no timer flush: this test is about series boundaries
|
||||
|
||||
const msg = "outbound/urltest[chain-ewan-wg-subs-h2]: WireGuard is not ready yet"
|
||||
for sec := 1; sec <= 6; sec++ {
|
||||
if _, err := s.Write([]byte(engLine(sec, "ERROR", msg))); err != nil {
|
||||
t.Fatalf("Write: %v", err)
|
||||
}
|
||||
}
|
||||
_, _ = s.Write([]byte(engLine(7, "INFO", "something else entirely")))
|
||||
if err := s.Close(); err != nil {
|
||||
t.Fatalf("Close: %v", err)
|
||||
}
|
||||
|
||||
for _, src := range []struct{ name, body string }{
|
||||
{"file", readFile(t, cfg.Path)},
|
||||
{"stderr", stderr.String()},
|
||||
} {
|
||||
got := payloads(t, src.body)
|
||||
want := []string{
|
||||
strings.TrimSuffix(engLine(1, "ERROR", msg), "\n"),
|
||||
strings.TrimSuffix(engLine(7, "INFO", "something else entirely"), "\n"),
|
||||
"repeated 5 times: ERROR " + msg,
|
||||
}
|
||||
if len(got) != len(want) {
|
||||
t.Fatalf("%s: got %d lines %q, want %d %q", src.name, len(got), got, len(want), want)
|
||||
}
|
||||
for i := range want {
|
||||
if got[i] != want[i] {
|
||||
t.Errorf("%s line %d = %q, want %q", src.name, i, got[i], want[i])
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestRepeatAlternatingSuppressedPerMessage is the reworked
|
||||
// TestRepeatAlternatingNotSuppressed. Its old contract — A B A B is four
|
||||
// distinct events, nothing may be swallowed — was the bug: on the router the
|
||||
// flood ALWAYS alternates (one message per broken chain), so adjacency-only
|
||||
// suppression collapsed nothing at all.
|
||||
//
|
||||
// The property that test really guarded, and that this one still guards, is
|
||||
// that distinct messages are never folded into one count: A and B each get
|
||||
// their OWN series, their own summary and their own N. What changed is that the
|
||||
// second copy of each is now suppressed instead of printed.
|
||||
func TestRepeatAlternatingSuppressedPerMessage(t *testing.T) {
|
||||
var stderr syncBuffer
|
||||
cfg := fileCfg(t, true, true, 0)
|
||||
s := New(&stderr, cfg)
|
||||
s.window = time.Hour
|
||||
|
||||
for sec := 1; sec <= 4; sec++ {
|
||||
msg := "alpha happened"
|
||||
if sec%2 == 0 {
|
||||
msg = "beta happened"
|
||||
}
|
||||
_, _ = s.Write([]byte(engLine(sec, "WARN", msg)))
|
||||
}
|
||||
_ = s.Close()
|
||||
|
||||
for _, src := range []struct{ name, body string }{
|
||||
{"file", readFile(t, cfg.Path)},
|
||||
{"stderr", stderr.String()},
|
||||
} {
|
||||
got := payloads(t, src.body)
|
||||
want := []string{
|
||||
strings.TrimSuffix(engLine(1, "WARN", "alpha happened"), "\n"),
|
||||
strings.TrimSuffix(engLine(2, "WARN", "beta happened"), "\n"),
|
||||
"repeated 1 time: WARN alpha happened",
|
||||
"repeated 1 time: WARN beta happened",
|
||||
}
|
||||
if len(got) != len(want) {
|
||||
t.Fatalf("%s: got %d lines %q, want %d %q", src.name, len(got), got, len(want), want)
|
||||
}
|
||||
for i := range want {
|
||||
if got[i] != want[i] {
|
||||
t.Errorf("%s line %d = %q, want %q", src.name, i, got[i], want[i])
|
||||
}
|
||||
}
|
||||
// The counts are per message — never merged into one "repeated 2".
|
||||
per, n := summaryByMessage(t, got)
|
||||
if n != 2 || per["WARN alpha happened"] != 1 || per["WARN beta happened"] != 1 {
|
||||
t.Errorf("%s: summaries %v (%d lines), want one per message with N=1", src.name, per, n)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestRepeatInterleavedFloodCollapsed is the field case that motivated the
|
||||
// table: the engine cycles the SAME message over three chain tags, so no two
|
||||
// identical lines are ever adjacent. 300 lines must leave 3 printed heads and
|
||||
// exactly 3 summaries, each naming its own message with its own count.
|
||||
func TestRepeatInterleavedFloodCollapsed(t *testing.T) {
|
||||
var stderr syncBuffer
|
||||
cfg := fileCfg(t, true, true, 0)
|
||||
s := New(&stderr, cfg)
|
||||
s.window = time.Hour // the summaries come from Close, deterministically
|
||||
|
||||
msgs := []string{
|
||||
"outbound/urltest[chain-ewan-wg-subs-h2]: WireGuard is not ready yet",
|
||||
"outbound/urltest[chain-ewan-wg-subs-h3]: WireGuard is not ready yet",
|
||||
"outbound/urltest[chain-ewan-wg-subs-h4]: WireGuard is not ready yet",
|
||||
}
|
||||
const copies = 100
|
||||
for i := 0; i < copies; i++ {
|
||||
for _, m := range msgs {
|
||||
if _, err := s.Write([]byte(engLine(i, "ERROR", m))); err != nil {
|
||||
t.Fatalf("Write: %v", err)
|
||||
}
|
||||
}
|
||||
}
|
||||
if err := s.Close(); err != nil {
|
||||
t.Fatalf("Close: %v", err)
|
||||
}
|
||||
|
||||
for _, src := range []struct{ name, body string }{
|
||||
{"file", readFile(t, cfg.Path)},
|
||||
{"stderr", stderr.String()},
|
||||
} {
|
||||
got := payloads(t, src.body)
|
||||
var want []string
|
||||
for _, m := range msgs { // the heads, in first-seen order
|
||||
want = append(want, strings.TrimSuffix(engLine(0, "ERROR", m), "\n"))
|
||||
}
|
||||
for _, m := range msgs { // the summaries, oldest-seen first
|
||||
want = append(want, fmt.Sprintf("repeated %d times: ERROR %s", copies-1, m))
|
||||
}
|
||||
if len(got) != len(want) {
|
||||
t.Fatalf("%s: got %d lines %q, want %d %q", src.name, len(got), got, len(want), want)
|
||||
}
|
||||
for i := range want {
|
||||
if got[i] != want[i] {
|
||||
t.Errorf("%s line %d = %q, want %q", src.name, i, got[i], want[i])
|
||||
}
|
||||
}
|
||||
per, n := summaryByMessage(t, got)
|
||||
if n != len(msgs) {
|
||||
t.Errorf("%s: %d summary lines, want %d (one per message)", src.name, n, len(msgs))
|
||||
}
|
||||
for _, m := range msgs {
|
||||
if per["ERROR "+m] != copies-1 {
|
||||
t.Errorf("%s: message %q summarised %d copies, want %d", src.name, m, per["ERROR "+m], copies-1)
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestRepeatTableOverflowReportsEviction: the table is bounded, and hitting the
|
||||
// bound never drops a count silently — the least-recently-seen series is
|
||||
// flushed with a summary that says WHY it was cut short, and the table stays at
|
||||
// its limit.
|
||||
func TestRepeatTableOverflowReportsEviction(t *testing.T) {
|
||||
var stderr syncBuffer
|
||||
cfg := fileCfg(t, true, false, 0)
|
||||
s := New(&stderr, cfg)
|
||||
s.window = time.Hour
|
||||
|
||||
// Fill the table, every key with exactly one swallowed copy to lose.
|
||||
for i := 0; i < maxRepeatKeys; i++ {
|
||||
msg := fmt.Sprintf("chain-%03d is down", i)
|
||||
_, _ = s.Write([]byte(engLine(i, "ERROR", msg)))
|
||||
_, _ = s.Write([]byte(engLine(i, "ERROR", msg)))
|
||||
}
|
||||
// One key too many: the oldest series (chain-000) must be evicted, and its
|
||||
// swallowed copy must be reported on the way out.
|
||||
_, _ = s.Write([]byte(engLine(999, "ERROR", "one key too many")))
|
||||
|
||||
want := "repeated 1 time (repeat table full): ERROR chain-000 is down"
|
||||
got := payloads(t, stderr.String())
|
||||
found := false
|
||||
for _, l := range got {
|
||||
if l == want {
|
||||
found = true
|
||||
}
|
||||
}
|
||||
if !found {
|
||||
tail := got
|
||||
if len(tail) > 4 {
|
||||
tail = tail[len(tail)-4:]
|
||||
}
|
||||
t.Fatalf("eviction was silent: no %q (tail of the output: %q)", want, tail)
|
||||
}
|
||||
s.mu.Lock()
|
||||
size, lru := len(s.series), s.seriesLRU.Len()
|
||||
s.mu.Unlock()
|
||||
if size != maxRepeatKeys || lru != maxRepeatKeys {
|
||||
t.Errorf("table holds %d entries (lru %d), want the cap %d", size, lru, maxRepeatKeys)
|
||||
}
|
||||
|
||||
_ = s.Close()
|
||||
// Nothing anywhere was lost: every written line is either printed or counted.
|
||||
got = payloads(t, stderr.String())
|
||||
suppressed, summaries := summarySum(t, got)
|
||||
printed := len(got) - summaries
|
||||
if wrote := 2*maxRepeatKeys + 1; printed+suppressed != wrote {
|
||||
t.Errorf("%d printed + %d suppressed = %d, want %d written", printed, suppressed, printed+suppressed, wrote)
|
||||
}
|
||||
}
|
||||
|
||||
// TestRepeatSeriesSurviveReconfigure: a live Reconfigure settles every open
|
||||
// series into the destination its swallowed copies were headed for, and starts
|
||||
// the next configuration with an empty table.
|
||||
func TestRepeatSeriesSurviveReconfigure(t *testing.T) {
|
||||
dir := t.TempDir()
|
||||
cfgA := Config{ToFile: true, Path: filepath.Join(dir, "a.log")}
|
||||
cfgB := Config{ToFile: true, Path: filepath.Join(dir, "b.log")}
|
||||
s := New(nil, cfgA)
|
||||
s.window = time.Hour
|
||||
|
||||
// Two interleaved series: alpha keeps 2 swallowed copies, beta keeps 1.
|
||||
for sec, msg := range []string{"alpha down", "beta down", "alpha down", "beta down", "alpha down"} {
|
||||
_, _ = s.Write([]byte(engLine(sec+1, "ERROR", msg)))
|
||||
}
|
||||
s.Reconfigure(cfgB)
|
||||
_, _ = s.Write([]byte(engLine(6, "ERROR", "alpha down")))
|
||||
if err := s.Close(); err != nil {
|
||||
t.Fatalf("Close: %v", err)
|
||||
}
|
||||
|
||||
a := payloads(t, readFile(t, cfgA.Path))
|
||||
perA, nA := summaryByMessage(t, a)
|
||||
if nA != 2 || perA["ERROR alpha down"] != 2 || perA["ERROR beta down"] != 1 {
|
||||
t.Errorf("a.log summaries %v (%d lines), want alpha=2 beta=1 flushed by Reconfigure: %q", perA, nA, a)
|
||||
}
|
||||
b := payloads(t, readFile(t, cfgB.Path))
|
||||
if _, nB := summaryByMessage(t, b); nB != 0 {
|
||||
t.Errorf("b.log carries summaries of lines written before the swap: %q", b)
|
||||
}
|
||||
// The new configuration starts fresh: the message prints in full again.
|
||||
if len(b) != 1 || b[0] != strings.TrimSuffix(engLine(6, "ERROR", "alpha down"), "\n") {
|
||||
t.Errorf("b.log = %q, want the re-printed head line only", b)
|
||||
}
|
||||
}
|
||||
|
||||
// TestRepeatLastRunSurvivesClose: a run still open at shutdown is summarised by
|
||||
// Close — the last series is never silently lost.
|
||||
func TestRepeatLastRunSurvivesClose(t *testing.T) {
|
||||
var stderr syncBuffer
|
||||
cfg := fileCfg(t, true, true, 0)
|
||||
s := New(&stderr, cfg)
|
||||
s.window = time.Hour
|
||||
|
||||
for sec := 1; sec <= 4; sec++ {
|
||||
_, _ = s.Write([]byte(engLine(sec, "ERROR", "dying in a loop")))
|
||||
}
|
||||
if err := s.Close(); err != nil {
|
||||
t.Fatalf("Close: %v", err)
|
||||
}
|
||||
|
||||
for _, src := range []struct{ name, body string }{
|
||||
{"file", readFile(t, cfg.Path)},
|
||||
{"stderr", stderr.String()},
|
||||
} {
|
||||
got := payloads(t, src.body)
|
||||
if len(got) != 2 {
|
||||
t.Fatalf("%s: got %q, want the line plus one summary", src.name, got)
|
||||
}
|
||||
if want := "repeated 3 times: ERROR dying in a loop"; got[1] != want {
|
||||
t.Errorf("%s: summary = %q, want %q", src.name, got[1], want)
|
||||
}
|
||||
}
|
||||
|
||||
// A run of exactly two lines is summarised with correct grammar.
|
||||
var stderr2 syncBuffer
|
||||
cfg2 := fileCfg(t, true, false, 0)
|
||||
s2 := New(&stderr2, cfg2)
|
||||
s2.window = time.Hour
|
||||
_, _ = s2.Write([]byte(engLine(1, "WARN", "twice only")))
|
||||
_, _ = s2.Write([]byte(engLine(2, "WARN", "twice only")))
|
||||
_ = s2.Close()
|
||||
want := "repeated 1 time: WARN twice only"
|
||||
if got := payloads(t, stderr2.String()); len(got) != 2 || got[1] != want {
|
||||
t.Errorf("two-line run rendered as %q, want the line plus %q", got, want)
|
||||
}
|
||||
}
|
||||
|
||||
// TestRepeatFatalNeverSuppressed: fatal/panic lines are exempt — every copy is
|
||||
// printed, and they do not become the head of a run either.
|
||||
func TestRepeatFatalNeverSuppressed(t *testing.T) {
|
||||
var stderr syncBuffer
|
||||
cfg := fileCfg(t, true, true, 0)
|
||||
s := New(&stderr, cfg)
|
||||
s.window = time.Hour
|
||||
|
||||
for sec := 1; sec <= 3; sec++ {
|
||||
_, _ = s.Write([]byte(engLine(sec, "FATAL", "engine is gone")))
|
||||
}
|
||||
for sec := 4; sec <= 6; sec++ {
|
||||
_, _ = s.Write([]byte(engLine(sec, "PANIC", "engine is gone")))
|
||||
}
|
||||
_ = s.Close()
|
||||
|
||||
for _, src := range []struct{ name, body string }{
|
||||
{"file", readFile(t, cfg.Path)},
|
||||
{"stderr", stderr.String()},
|
||||
} {
|
||||
got := payloads(t, src.body)
|
||||
if len(got) != 6 {
|
||||
t.Fatalf("%s: got %d lines %q, want all 6 fatal/panic copies", src.name, len(got), got)
|
||||
}
|
||||
if _, n := summarySum(t, got); n != 0 {
|
||||
t.Errorf("%s: fatal/panic run produced %d summaries: %q", src.name, n, got)
|
||||
}
|
||||
}
|
||||
|
||||
// A fatal line also settles everything the table was holding BEFORE it
|
||||
// speaks: the process may never reach Close, and a swallowed count that is
|
||||
// never reported is the failure this mechanism exists to prevent.
|
||||
var stderr2 syncBuffer
|
||||
s2 := New(&stderr2, fileCfg(t, true, false, 0))
|
||||
s2.window = time.Hour
|
||||
for sec := 1; sec <= 3; sec++ {
|
||||
_, _ = s2.Write([]byte(engLine(sec, "ERROR", "about to die")))
|
||||
}
|
||||
_, _ = s2.Write([]byte(engLine(4, "FATAL", "engine is gone")))
|
||||
got := payloads(t, stderr2.String())
|
||||
want := []string{
|
||||
strings.TrimSuffix(engLine(1, "ERROR", "about to die"), "\n"),
|
||||
"repeated 2 times: ERROR about to die",
|
||||
strings.TrimSuffix(engLine(4, "FATAL", "engine is gone"), "\n"),
|
||||
}
|
||||
if len(got) != len(want) {
|
||||
t.Fatalf("got %q, want %q", got, want)
|
||||
}
|
||||
for i := range want {
|
||||
if got[i] != want[i] {
|
||||
t.Errorf("line %d = %q, want %q", i, got[i], want[i])
|
||||
}
|
||||
}
|
||||
_ = s2.Close()
|
||||
}
|
||||
|
||||
// TestRepeatWindowFlushesOngoingRun: a flood that never stops still reports
|
||||
// itself — the window timer emits a summary without any further input, and the
|
||||
// counter restarts (so the NEXT summary counts only the lines after it).
|
||||
func TestRepeatWindowFlushesOngoingRun(t *testing.T) {
|
||||
var stderr syncBuffer
|
||||
cfg := fileCfg(t, true, false, 0)
|
||||
s := New(&stderr, cfg)
|
||||
s.window = 50 * time.Millisecond
|
||||
|
||||
for sec := 1; sec <= 5; sec++ {
|
||||
_, _ = s.Write([]byte(engLine(sec, "ERROR", "flooding")))
|
||||
}
|
||||
want := "repeated 4 times: ERROR flooding"
|
||||
deadline := time.Now().Add(5 * time.Second)
|
||||
for !strings.Contains(stderr.String(), "repeated ") && time.Now().Before(deadline) {
|
||||
time.Sleep(5 * time.Millisecond)
|
||||
}
|
||||
got := payloads(t, stderr.String())
|
||||
if len(got) != 2 || got[1] != want {
|
||||
t.Fatalf("timer flush produced %q, want the line plus %q", got, want)
|
||||
}
|
||||
|
||||
// The flood continues. Whether the second batch lands inside the window the
|
||||
// flush opened (swallowed) or after it lapsed (a fresh head line) is a race
|
||||
// with the timer, so what is asserted is what must hold either way: the
|
||||
// flood is reported AGAIN, every copy is accounted for exactly once, and
|
||||
// every summary names this one message.
|
||||
for sec := 6; sec <= 8; sec++ {
|
||||
_, _ = s.Write([]byte(engLine(sec, "ERROR", "flooding")))
|
||||
}
|
||||
_ = s.Close()
|
||||
got = payloads(t, stderr.String())
|
||||
per, summaries := summaryByMessage(t, got)
|
||||
suppressed, _ := summarySum(t, got)
|
||||
printed := len(got) - summaries
|
||||
if printed+suppressed != 8 {
|
||||
t.Errorf("%d printed + %d suppressed = %d, want the 8 written lines: %q", printed, suppressed, printed+suppressed, got)
|
||||
}
|
||||
if summaries < 2 {
|
||||
t.Errorf("the continuing flood was reported %d time(s), want a second summary: %q", summaries, got)
|
||||
}
|
||||
if len(per) != 1 || per["ERROR flooding"] != suppressed {
|
||||
t.Errorf("summaries %v, want all of them naming %q", per, "ERROR flooding")
|
||||
}
|
||||
}
|
||||
|
||||
// TestRepeatConcurrentWritesKeepCount: the sink is written from every goroutine
|
||||
// of the engine. Under -race, a hammering of identical lines must not lose or
|
||||
// double-count anything: printed copies + summarised copies == lines written.
|
||||
func TestRepeatConcurrentWritesKeepCount(t *testing.T) {
|
||||
var stderr syncBuffer
|
||||
cfg := fileCfg(t, true, true, 0)
|
||||
s := New(&stderr, cfg)
|
||||
s.window = 20 * time.Millisecond // let the timer race the writers on purpose
|
||||
|
||||
const (
|
||||
writers = 8
|
||||
perGo = 100
|
||||
expected = writers * perGo
|
||||
)
|
||||
var wg sync.WaitGroup
|
||||
for g := 0; g < writers; g++ {
|
||||
wg.Add(1)
|
||||
go func(g int) {
|
||||
defer wg.Done()
|
||||
for i := 0; i < perGo; i++ {
|
||||
_, _ = s.Write([]byte(engLine(g*perGo+i, "ERROR", "concurrent flood")))
|
||||
}
|
||||
}(g)
|
||||
}
|
||||
wg.Wait()
|
||||
if err := s.Close(); err != nil {
|
||||
t.Fatalf("Close: %v", err)
|
||||
}
|
||||
|
||||
for _, src := range []struct{ name, body string }{
|
||||
{"file", readFile(t, cfg.Path)},
|
||||
{"stderr", stderr.String()},
|
||||
} {
|
||||
got := payloads(t, src.body)
|
||||
suppressed, summaries := summarySum(t, got)
|
||||
printed := len(got) - summaries
|
||||
if printed+suppressed != expected {
|
||||
t.Errorf("%s: %d printed + %d suppressed = %d, want %d (lines: %q)",
|
||||
src.name, printed, suppressed, printed+suppressed, expected, got)
|
||||
}
|
||||
if printed < 1 {
|
||||
t.Errorf("%s: nothing printed at all — the first line must always show", src.name)
|
||||
}
|
||||
// Suppression really happened (the whole point).
|
||||
if printed > expected/10 {
|
||||
t.Errorf("%s: %d of %d lines printed — suppression did not engage", src.name, printed, expected)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestNewFileOffPurgesLeftovers: a sink CREATED with the file off (daemon boot
|
||||
// with the toggle already off) deletes segments left by a previous life.
|
||||
func TestNewFileOffPurgesLeftovers(t *testing.T) {
|
||||
|
||||
@@ -124,8 +124,36 @@ type RuleReach struct {
|
||||
ShadowedByOrder int `json:"shadowed_by_order,omitempty"`
|
||||
// Reason is the operator-facing sentence; "" when Unreachable is false.
|
||||
Reason string `json:"reason,omitempty"`
|
||||
|
||||
// EffectiveEnabled is whether the rule is IN FORCE right now — Rule.Enabled
|
||||
// after the active WAN profile's enable/disable overrides. It is NOT the flag
|
||||
// GET /api/config carries: that one is the desired state the panel PUTs back,
|
||||
// and on a router with profiles the two legitimately disagree.
|
||||
//
|
||||
// This field exists because the panel had no way to tell them apart and so drew
|
||||
// the desired state as if it were the truth: a config with `ewan-default` and
|
||||
// `swan-default` both `enabled '1'` in UCI, under a profile that enables the
|
||||
// first and disables the second, showed BOTH switches on while the engine ran
|
||||
// only one chain. Same defect class as the two-`default` shadowing above — the
|
||||
// interface claiming a setting is in force when it is not.
|
||||
EffectiveEnabled bool `json:"effective_enabled"`
|
||||
// OverriddenBy names the active profile that CHANGED this rule's state, and
|
||||
// Override says which way it went: "enabled" or "disabled". Both are empty
|
||||
// unless the profile actually flipped the outcome — a profile that disables a
|
||||
// rule already switched off in UCI has overridden nothing the operator can see,
|
||||
// and saying otherwise would put a profile's name on every row it merely
|
||||
// mentions.
|
||||
OverriddenBy string `json:"overridden_by,omitempty"`
|
||||
Override string `json:"override,omitempty"`
|
||||
}
|
||||
|
||||
// Override directions carried by RuleReach.Override. Values are part of the
|
||||
// /api/rules/reachability contract; the panel switches on them.
|
||||
const (
|
||||
RuleOverrideEnabled = "enabled"
|
||||
RuleOverrideDisabled = "disabled"
|
||||
)
|
||||
|
||||
// RuleReachability returns one verdict per rule, in the INPUT slice's order.
|
||||
//
|
||||
// The input must already be the EFFECTIVE rule set — profile enable/disable
|
||||
@@ -147,6 +175,10 @@ func RuleReachability(rules []Rule) []RuleReach {
|
||||
for i := range rules {
|
||||
out[i] = RuleReach{
|
||||
Index: i, Name: rules[i].Name, Order: rules[i].Order, ShadowedByIndex: -1,
|
||||
// The input IS the effective set (see the doc comment), so its Enabled flag
|
||||
// is the effective one. Who overrode it — if anyone — is not knowable from
|
||||
// this slice alone; Model.EffectiveRuleReachability fills that in.
|
||||
EffectiveEnabled: rules[i].Enabled,
|
||||
}
|
||||
}
|
||||
|
||||
@@ -208,3 +240,46 @@ func (m *Model) EffectiveRules() ([]Rule, []Warning) {
|
||||
rules, owarns := ApplyProfileRuleOverrides(m.Rules, prof)
|
||||
return rules, append(warns, owarns...)
|
||||
}
|
||||
|
||||
// EffectiveRuleReachability is RuleReachability over m's EFFECTIVE rules, with
|
||||
// each verdict annotated by the active profile's override. It returns one verdict
|
||||
// per rule in m.Rules, at the SAME index — ApplyProfileRuleOverrides returns a
|
||||
// same-length copy in the same order, and RuleReachability preserves its input's
|
||||
// order, so a client can zip the result with GET /api/config's `Rules`.
|
||||
//
|
||||
// The override annotation is a DIFF, not a second reading of the profile's name
|
||||
// lists: a verdict is marked only where the effective flag actually differs from
|
||||
// the desired-state one. That is deliberate on both counts —
|
||||
//
|
||||
// - it cannot drift from ApplyProfileRuleOverrides, because it observes that
|
||||
// function's output rather than re-deciding what it should have done (an
|
||||
// unmigrated rule the profile is forbidden to enable, for instance, produces no
|
||||
// diff and therefore no annotation, with nothing here having to know the rule);
|
||||
// - and it answers the question the operator is actually asking, which is not
|
||||
// "does the profile mention this rule" but "is this row's switch telling me the
|
||||
// truth".
|
||||
func (m *Model) EffectiveRuleReachability() ([]RuleReach, []Warning) {
|
||||
if m == nil {
|
||||
return []RuleReach{}, nil
|
||||
}
|
||||
prof, warns := ResolveActiveProfile(m)
|
||||
eff, owarns := ApplyProfileRuleOverrides(m.Rules, prof)
|
||||
warns = append(warns, owarns...)
|
||||
|
||||
out := RuleReachability(eff)
|
||||
if prof == nil {
|
||||
return out, warns
|
||||
}
|
||||
for i := range out {
|
||||
if i >= len(m.Rules) || m.Rules[i].Enabled == eff[i].Enabled {
|
||||
continue
|
||||
}
|
||||
out[i].OverriddenBy = prof.Name
|
||||
if eff[i].Enabled {
|
||||
out[i].Override = RuleOverrideEnabled
|
||||
} else {
|
||||
out[i].Override = RuleOverrideDisabled
|
||||
}
|
||||
}
|
||||
return out, warns
|
||||
}
|
||||
|
||||
+68
-11
@@ -119,8 +119,16 @@ func (s *Server) handleConfig(w http.ResponseWriter, r *http.Request) {
|
||||
}
|
||||
}
|
||||
|
||||
// configGetRead is handleConfigGet's test seam (same pattern as reachConfigRead /
|
||||
// logConfigRead). It exists so a test can serve GET /api/config and
|
||||
// GET /api/rules/reachability from ONE canned model and compare them: the promise
|
||||
// that the two responses line up by `index` is a promise about a pair of
|
||||
// endpoints, and asserting it against a slice the test holds itself would only
|
||||
// re-check the fixture.
|
||||
var configGetRead = model.ReadUCI
|
||||
|
||||
func (s *Server) handleConfigGet(w http.ResponseWriter, r *http.Request) {
|
||||
m, err := model.ReadUCI()
|
||||
m, err := configGetRead()
|
||||
if err != nil {
|
||||
writeError(w, http.StatusInternalServerError, "read config: "+err.Error())
|
||||
return
|
||||
@@ -153,11 +161,19 @@ var reachConfigRead = model.ReadUCI
|
||||
// it.
|
||||
//
|
||||
// The verdict is computed over the EFFECTIVE rules (active WAN profile's
|
||||
// enable/disable applied via model.EffectiveRules), because a rule the active
|
||||
// profile switched off is not in force and must not be blamed for retiring
|
||||
// enable/disable applied, via model.EffectiveRuleReachability), because a rule the
|
||||
// active profile switched off is not in force and must not be blamed for retiring
|
||||
// anything. Profile-resolution warnings are dropped here: this endpoint answers
|
||||
// one question, and the same warnings already reach the operator through
|
||||
// `shaterd status` on every apply.
|
||||
//
|
||||
// The same call also carries `effective_enabled` (+ `overridden_by`/`override`),
|
||||
// which is this endpoint's answer to "is the switch on that row telling the truth".
|
||||
// It belongs HERE and not on /api/config for the reason above: /api/config is the
|
||||
// desired state, PUT back verbatim, and a rule's Enabled flag there must keep
|
||||
// meaning "what the operator asked for" even while a profile is overriding it.
|
||||
// Both facts come from one read of the config, so they cannot describe different
|
||||
// configs.
|
||||
func (s *Server) handleRulesReachability(w http.ResponseWriter, r *http.Request) {
|
||||
if r.Method != http.MethodGet {
|
||||
writeError(w, http.StatusMethodNotAllowed, "method not allowed")
|
||||
@@ -168,8 +184,7 @@ func (s *Server) handleRulesReachability(w http.ResponseWriter, r *http.Request)
|
||||
writeError(w, http.StatusInternalServerError, "read config: "+err.Error())
|
||||
return
|
||||
}
|
||||
rules, _ := m.EffectiveRules()
|
||||
out := model.RuleReachability(rules)
|
||||
out, _ := m.EffectiveRuleReachability()
|
||||
if out == nil {
|
||||
out = []model.RuleReach{}
|
||||
}
|
||||
@@ -1344,9 +1359,24 @@ type groupTestStartResponse struct {
|
||||
Reason string `json:"reason,omitempty"`
|
||||
}
|
||||
|
||||
// handleGroupsTest → /api/groups/test: measure each target's (group's or chain's)
|
||||
// latency and the public address it currently exits from. A chain is dialled via
|
||||
// its exit wrapper, so the numbers are end-to-end properties of the whole path.
|
||||
// handleGroupsTest → /api/groups/test: report each target's (group's or chain's)
|
||||
// health as the OBSERVATORY just measured it, plus the public address it
|
||||
// currently exits from.
|
||||
//
|
||||
// POST no longer dials anything. It asks the engine's observatory for an
|
||||
// out-of-turn pass of its probe plan and then reports what that pass measured:
|
||||
// each target resolves off the health board the moment its measured path
|
||||
// carries an observation newer than the button press. The consequences the
|
||||
// frontend should expect (the response SHAPE is unchanged):
|
||||
// - delay_ms/ok/tested_unix describe the observatory's measurement of the
|
||||
// target's real routed path — tested_unix is when that observation was
|
||||
// taken, which may be a second or two before the poll that delivered it;
|
||||
// - a target no enabled rule routes through is never measured and comes back
|
||||
// ok=false with an error saying so, immediately — the observatory only
|
||||
// probes paths the rules use, and the panel should render that as a
|
||||
// routing fact, not a failure;
|
||||
// - a run can take up to the engine's wait deadline (120s) when the
|
||||
// observatory has a large plan to walk; poll GET for progress as before.
|
||||
//
|
||||
// Contract for the frontend:
|
||||
// - POST {"name":"auto"} → test that group or chain; an empty/absent name tests
|
||||
@@ -1380,9 +1410,11 @@ func (s *Server) handleGroupsTest(w http.ResponseWriter, r *http.Request) {
|
||||
if n := strings.TrimSpace(req.Name); n != "" {
|
||||
names = []string{n}
|
||||
}
|
||||
// Probe URL from the model (best-effort) — the global one, the only probe
|
||||
// instrument left; a UCI read failure falls back to the engine's built-in
|
||||
// default.
|
||||
// Probe URL from the model (best-effort), passed through for signature
|
||||
// stability only: the engine IGNORES it now. The probe URL is a global
|
||||
// observatory setting installed at apply time, and the manual test reads
|
||||
// the observatory's measurements rather than dialling with a URL of its
|
||||
// own — see engine.TestGroups.
|
||||
var probeURL string
|
||||
if m, err := model.ReadUCI(); err == nil {
|
||||
probeURL = strings.TrimSpace(m.Globals.ProbeURL)
|
||||
@@ -1454,6 +1486,31 @@ type groupHealthResponse struct {
|
||||
// was measured THROUGH the group's egress, so it is not comparable with the global
|
||||
// per-node numbers in /api/stats node_health. That is the entire point of this
|
||||
// endpoint — the same node in two groups with two egresses has two health states.
|
||||
//
|
||||
// chains[] additionally carries the PER-HOP readout for every chain the running
|
||||
// box materialised:
|
||||
//
|
||||
// chains[].hops = [{index,tag,kind,exit,state,delay_ms,age_seconds,selected,
|
||||
// total,tested,alive,dead,untested}...]
|
||||
//
|
||||
// - hops is L1..Ln in wire order (index is 1-based); the entry with exit=true
|
||||
// is the last hop, where traffic leaves to the internet. Each hop's numbers
|
||||
// measure the chain PREFIX up to and including that hop — the observatory
|
||||
// dials the hop wrappers — so a dead hop N with alive hops 1..N-1 localises
|
||||
// the failure to hop N's own leg.
|
||||
// - kind is "node" (one measurement: total=1, selected empty) or "group" (a
|
||||
// roll-up of the hop's member copies, with the same counter invariants as a
|
||||
// group: tested == alive+dead, alive+dead+untested == total; selected is
|
||||
// the node NAME the hop currently picks; delay_ms/age_seconds are the
|
||||
// selected member's observation, or the freshest alive member's when the
|
||||
// selection has none).
|
||||
// - state follows the same closed set and the same honesty rule as groups:
|
||||
// "dead" only on a positive finding, "untested" for no fresh data — and for
|
||||
// a group hop, "alive" as long as ANY member answers.
|
||||
// - an ABSENT/empty hops key means the running box never materialised the
|
||||
// chain (unused, or a 1-hop chain that resolves straight to its target) —
|
||||
// it must NOT be read as "this chain has no hops"; the config, not this
|
||||
// endpoint, knows the configured hop count.
|
||||
func (s *Server) handleGroupsHealth(w http.ResponseWriter, r *http.Request) {
|
||||
if r.Method != http.MethodGet {
|
||||
writeError(w, http.StatusMethodNotAllowed, "method not allowed")
|
||||
|
||||
@@ -95,6 +95,185 @@ func TestRulesReachabilityEmptyIsArray(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
// profileModel is the field config that made the effective-state fields
|
||||
// necessary: two rules both `enabled '1'` in UCI, and an active profile that
|
||||
// force-enables one and disables the other. The engine runs one chain; the panel
|
||||
// used to draw two live switches.
|
||||
func profileModel(activeProfile string) *model.Model {
|
||||
return &model.Model{
|
||||
Globals: model.Globals{ActiveProfile: activeProfile},
|
||||
Profiles: []model.Profile{{
|
||||
Name: "ethernet-uplink",
|
||||
Enabled: true,
|
||||
EnableRules: []string{"ewan-default"},
|
||||
DisableRules: []string{"swan-default"},
|
||||
}},
|
||||
Rules: []model.Rule{
|
||||
{Name: "ewan-default", Enabled: true, Order: 10, Target: "direct", DstRuleset: []string{"ru-inside"}},
|
||||
{Name: "swan-default", Enabled: true, Order: 20, Target: "group:auto", DstRuleset: []string{"ru-inside"}},
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
// TestRulesReachabilityProfileDisabled: the reported bug. `swan-default` is
|
||||
// `enabled '1'` in UCI, the active profile disables it, and the endpoint must say
|
||||
// so — otherwise the row keeps claiming a rule is in force that the engine never
|
||||
// loaded.
|
||||
func TestRulesReachabilityProfileDisabled(t *testing.T) {
|
||||
s := newTestServer(t)
|
||||
srv := httptest.NewServer(s.Handler())
|
||||
defer srv.Close()
|
||||
cookie := login(t, srv, s)
|
||||
|
||||
orig := reachConfigRead
|
||||
defer func() { reachConfigRead = orig }()
|
||||
reachConfigRead = func() (*model.Model, error) { return profileModel("ethernet-uplink"), nil }
|
||||
|
||||
code, out := getReach(t, srv, cookie)
|
||||
if code != http.StatusOK {
|
||||
t.Fatalf("got %d, want 200", code)
|
||||
}
|
||||
if len(out.Rules) != 2 {
|
||||
t.Fatalf("want one verdict per rule (2), got %d", len(out.Rules))
|
||||
}
|
||||
sw := out.Rules[1]
|
||||
if sw.Name != "swan-default" {
|
||||
t.Fatalf("verdicts must keep the model's rule order: %+v", out.Rules)
|
||||
}
|
||||
if sw.EffectiveEnabled {
|
||||
t.Fatalf("the active profile disables this rule, so it is NOT in force: %+v", sw)
|
||||
}
|
||||
if sw.OverriddenBy != "ethernet-uplink" {
|
||||
t.Fatalf("the row must be able to name who switched it off, got %q: %+v", sw.OverriddenBy, sw)
|
||||
}
|
||||
if sw.Override != model.RuleOverrideDisabled {
|
||||
t.Fatalf("override direction must be %q, got %q", model.RuleOverrideDisabled, sw.Override)
|
||||
}
|
||||
// Existing semantics untouched: neither rule is *unreachable* — both carry a
|
||||
// destination matcher, so the shadowing analysis has nothing to say.
|
||||
if sw.Unreachable || out.Rules[0].Unreachable {
|
||||
t.Fatalf("profile overrides must not be reported as shadowing: %+v", out.Rules)
|
||||
}
|
||||
}
|
||||
|
||||
// TestRulesReachabilityProfileEnabled: the symmetric case — a rule switched OFF in
|
||||
// UCI that the active profile forces on is in force, and says who did it.
|
||||
func TestRulesReachabilityProfileEnabled(t *testing.T) {
|
||||
s := newTestServer(t)
|
||||
srv := httptest.NewServer(s.Handler())
|
||||
defer srv.Close()
|
||||
cookie := login(t, srv, s)
|
||||
|
||||
m := profileModel("ethernet-uplink")
|
||||
m.Rules[0].Enabled = false // desired state says off; the profile says otherwise
|
||||
|
||||
orig := reachConfigRead
|
||||
defer func() { reachConfigRead = orig }()
|
||||
reachConfigRead = func() (*model.Model, error) { return m, nil }
|
||||
|
||||
_, out := getReach(t, srv, cookie)
|
||||
ew := out.Rules[0]
|
||||
if !ew.EffectiveEnabled {
|
||||
t.Fatalf("the active profile force-enables this rule, so it IS in force: %+v", ew)
|
||||
}
|
||||
if ew.OverriddenBy != "ethernet-uplink" || ew.Override != model.RuleOverrideEnabled {
|
||||
t.Fatalf("want enabled-by-ethernet-uplink, got %q/%q: %+v", ew.OverriddenBy, ew.Override, ew)
|
||||
}
|
||||
}
|
||||
|
||||
// TestRulesReachabilityNoProfileUnchanged: with no active profile the effective
|
||||
// flag is just the configured one and NOTHING claims an override — a page that
|
||||
// badges a profile name on a router that uses no profiles is the same lie in the
|
||||
// other direction.
|
||||
func TestRulesReachabilityNoProfileUnchanged(t *testing.T) {
|
||||
s := newTestServer(t)
|
||||
srv := httptest.NewServer(s.Handler())
|
||||
defer srv.Close()
|
||||
cookie := login(t, srv, s)
|
||||
|
||||
m := profileModel("ethernet-uplink")
|
||||
m.Profiles = nil // the pin names nothing that exists…
|
||||
m.Rules[1].Enabled = false // …and one rule is plainly switched off
|
||||
|
||||
orig := reachConfigRead
|
||||
defer func() { reachConfigRead = orig }()
|
||||
reachConfigRead = func() (*model.Model, error) { return m, nil }
|
||||
|
||||
_, out := getReach(t, srv, cookie)
|
||||
if !out.Rules[0].EffectiveEnabled || out.Rules[1].EffectiveEnabled {
|
||||
t.Fatalf("without a profile the effective flag is the configured one: %+v", out.Rules)
|
||||
}
|
||||
for _, r := range out.Rules {
|
||||
if r.OverriddenBy != "" || r.Override != "" {
|
||||
t.Fatalf("no active profile, so no row may name one: %+v", r)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestRulesReachabilityIndexMatchesConfig: the contract the panel zips on — one
|
||||
// verdict per rule, at the SAME position as GET /api/config's `Rules`, with
|
||||
// name/order echoed. Both endpoints are served from ONE canned model here, so this
|
||||
// checks the pair, not the fixture. A profile is active precisely because that is
|
||||
// when the two responses disagree about `Enabled` and the alignment is easiest to
|
||||
// get wrong.
|
||||
func TestRulesReachabilityIndexMatchesConfig(t *testing.T) {
|
||||
s := newTestServer(t)
|
||||
srv := httptest.NewServer(s.Handler())
|
||||
defer srv.Close()
|
||||
cookie := login(t, srv, s)
|
||||
|
||||
m := profileModel("ethernet-uplink")
|
||||
// A third rule, out of Order sequence, so a verdict list accidentally sorted by
|
||||
// Order (the panel's display order) would not line up.
|
||||
m.Rules = append(m.Rules, model.Rule{Name: "zz-first", Enabled: true, Order: 5, Target: "block", DstPort: "25"})
|
||||
|
||||
origReach, origCfg := reachConfigRead, configGetRead
|
||||
defer func() { reachConfigRead, configGetRead = origReach, origCfg }()
|
||||
reachConfigRead = func() (*model.Model, error) { return m, nil }
|
||||
configGetRead = func() (*model.Model, error) { return m, nil }
|
||||
|
||||
req, _ := http.NewRequest(http.MethodGet, srv.URL+"/api/config", nil)
|
||||
req.AddCookie(cookie)
|
||||
resp, err := http.DefaultClient.Do(req)
|
||||
if err != nil {
|
||||
t.Fatalf("GET /api/config: %v", err)
|
||||
}
|
||||
defer resp.Body.Close()
|
||||
var cfg struct {
|
||||
Rules []struct {
|
||||
Name string
|
||||
Order int
|
||||
Enabled bool
|
||||
}
|
||||
}
|
||||
if err := json.NewDecoder(resp.Body).Decode(&cfg); err != nil {
|
||||
t.Fatalf("decode /api/config: %v", err)
|
||||
}
|
||||
|
||||
_, out := getReach(t, srv, cookie)
|
||||
if len(out.Rules) != len(cfg.Rules) {
|
||||
t.Fatalf("one verdict per configured rule: %d verdicts for %d rules", len(out.Rules), len(cfg.Rules))
|
||||
}
|
||||
for i, v := range out.Rules {
|
||||
if v.Index != i {
|
||||
t.Fatalf("verdict %d carries index %d — the panel keys rows on it", i, v.Index)
|
||||
}
|
||||
if v.Name != cfg.Rules[i].Name || v.Order != cfg.Rules[i].Order {
|
||||
t.Fatalf("verdict %d is about %q/%d, /api/config's row %d is %q/%d",
|
||||
i, v.Name, v.Order, i, cfg.Rules[i].Name, cfg.Rules[i].Order)
|
||||
}
|
||||
}
|
||||
// And the whole point: /api/config still reports the DESIRED state (both
|
||||
// original rules on, as UCI has them) while the verdicts report the effective
|
||||
// one. If these ever agree, the desired-state contract has been broken.
|
||||
if !cfg.Rules[0].Enabled || !cfg.Rules[1].Enabled {
|
||||
t.Fatalf("/api/config must keep serving the desired state verbatim: %+v", cfg.Rules)
|
||||
}
|
||||
if !out.Rules[0].EffectiveEnabled || out.Rules[1].EffectiveEnabled {
|
||||
t.Fatalf("verdicts must report the effective state: %+v", out.Rules)
|
||||
}
|
||||
}
|
||||
|
||||
// TestRulesReachabilityMethodAndAuth: it is a read endpoint behind the session
|
||||
// cookie, like every other /api route except /api/session.
|
||||
func TestRulesReachabilityMethodAndAuth(t *testing.T) {
|
||||
|
||||
@@ -0,0 +1,380 @@
|
||||
//go:build with_awg
|
||||
|
||||
// lx: end-to-end regression for the AmneziaWG-over-detour ClientBind fix.
|
||||
//
|
||||
// When an AmneziaWG endpoint runs through a detour, the WireGuard bind is our
|
||||
// ClientBind (not conn.StdNetBind — that path is taken only by the DefaultDialer
|
||||
// / no-detour case, see endpoint.go). Upstream ClientBind unconditionally
|
||||
// stripped bytes 1-3 of every datagram (clear(b[1:4]) on receive, copy on
|
||||
// Send) — those are the Cloudflare WARP "reserved" bytes. AmneziaWG 2.0 ranged
|
||||
// magic headers (h1-h4) instead put a full uint32 into bytes 0-3
|
||||
// (send.go RoutineEncryption: LittleEndian.PutUint32(header[0:4], magic)).
|
||||
// Zeroing bytes 1-3 collapses that magic to val <= 255, which falls outside the
|
||||
// configured range, so DeterminePacketTypeAndPadding returns MessageUnknownType
|
||||
// and the packet is dropped — the AWG tunnel never comes up at all.
|
||||
//
|
||||
// This test wires two REAL wireguard-go Devices together through ClientBind on
|
||||
// both ends (loopback UDP, no StdNetBind), configures ranged h1-h4 + s4 + junk
|
||||
// on both, then pushes an inner IP packet through and asserts it is delivered to
|
||||
// the peer's TUN within the deadline. With the pre-fix unconditional clear the
|
||||
// handshake magic (h1/h2) is destroyed and delivery never happens; with the fix
|
||||
// the reserved bytes are left alone for a plain-AWG (reserved == [0,0,0])
|
||||
// endpoint and the packet arrives.
|
||||
package wireguard
|
||||
|
||||
import (
|
||||
"context"
|
||||
"crypto/rand"
|
||||
"encoding/hex"
|
||||
"fmt"
|
||||
"net"
|
||||
"net/netip"
|
||||
"os"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"github.com/sagernet/sing/common/logger"
|
||||
M "github.com/sagernet/sing/common/metadata"
|
||||
N "github.com/sagernet/sing/common/network"
|
||||
"github.com/sagernet/sing/service/pause"
|
||||
"github.com/sagernet/wireguard-go/device"
|
||||
"github.com/sagernet/wireguard-go/tun"
|
||||
|
||||
"golang.org/x/crypto/curve25519"
|
||||
)
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// loopback UDP dialer: replaces N.Dialer so ClientBind.connect() gets a real
|
||||
// UDP socket bound to 127.0.0.1, connected to the peer's loopback listener.
|
||||
// This is the "detour" stand-in — the point is only that the bind is ClientBind
|
||||
// and NOT conn.StdNetBind.
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
// loopbackDialer binds ClientBind's listen socket to a FIXED loopback port so
|
||||
// the two binds can address each other by a known port. We use the non-connect
|
||||
// ClientBind path (isConnect == false), which routes outbound via
|
||||
// PacketConn.WriteTo to the wireguard-configured endpoint addr:port — matching
|
||||
// each bind's fixed listen port. (The connect path dials from an ephemeral
|
||||
// source port, so neither side would ever bind the reserved ports.)
|
||||
type loopbackDialer struct {
|
||||
listenPort int
|
||||
}
|
||||
|
||||
func (loopbackDialer) DialContext(ctx context.Context, network string, destination M.Socksaddr) (net.Conn, error) {
|
||||
var d net.Dialer
|
||||
d.LocalAddr = &net.UDPAddr{IP: net.IPv4(127, 0, 0, 1)}
|
||||
return d.DialContext(ctx, "udp", destination.String())
|
||||
}
|
||||
|
||||
func (d loopbackDialer) ListenPacket(ctx context.Context, destination M.Socksaddr) (net.PacketConn, error) {
|
||||
var lc net.ListenConfig
|
||||
return lc.ListenPacket(ctx, "udp", fmt.Sprintf("127.0.0.1:%d", d.listenPort))
|
||||
}
|
||||
|
||||
var _ N.Dialer = loopbackDialer{}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// channelTUN: a minimal tun.Device. Read() blocks handing out inbound IP
|
||||
// packets (packets we inject to be encrypted and sent to the peer); Write()
|
||||
// captures decrypted inner packets the Device delivers — that is the delivery
|
||||
// signal the test waits on. A leading `offset` region is reserved exactly like
|
||||
// a real TUN, which the Device uses for its own headers.
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
type channelTUN struct {
|
||||
name string
|
||||
mtu int
|
||||
inbound chan []byte // packets to hand to the Device via Read (to encrypt+send)
|
||||
outbound chan []byte // packets the Device delivered via Write (decrypted inner)
|
||||
events chan tun.Event
|
||||
closed chan struct{}
|
||||
}
|
||||
|
||||
func newChannelTUN(name string, mtu int) *channelTUN {
|
||||
t := &channelTUN{
|
||||
name: name,
|
||||
mtu: mtu,
|
||||
inbound: make(chan []byte, 16),
|
||||
outbound: make(chan []byte, 16),
|
||||
events: make(chan tun.Event, 4),
|
||||
closed: make(chan struct{}),
|
||||
}
|
||||
t.events <- tun.EventUp
|
||||
return t
|
||||
}
|
||||
|
||||
func (t *channelTUN) File() *os.File { return nil }
|
||||
|
||||
func (t *channelTUN) Read(bufs [][]byte, sizes []int, offset int) (int, error) {
|
||||
select {
|
||||
case <-t.closed:
|
||||
return 0, os.ErrClosed
|
||||
case pkt := <-t.inbound:
|
||||
n := copy(bufs[0][offset:], pkt)
|
||||
sizes[0] = n
|
||||
return 1, nil
|
||||
}
|
||||
}
|
||||
|
||||
func (t *channelTUN) Write(bufs [][]byte, offset int) (int, error) {
|
||||
for _, b := range bufs {
|
||||
if len(b) <= offset {
|
||||
continue
|
||||
}
|
||||
pkt := make([]byte, len(b)-offset)
|
||||
copy(pkt, b[offset:])
|
||||
select {
|
||||
case t.outbound <- pkt:
|
||||
case <-t.closed:
|
||||
return 0, os.ErrClosed
|
||||
default:
|
||||
}
|
||||
}
|
||||
return len(bufs), nil
|
||||
}
|
||||
|
||||
func (t *channelTUN) MTU() (int, error) { return t.mtu, nil }
|
||||
func (t *channelTUN) Name() (string, error) { return t.name, nil }
|
||||
func (t *channelTUN) Events() <-chan tun.Event { return t.events }
|
||||
func (t *channelTUN) BatchSize() int { return 1 }
|
||||
|
||||
func (t *channelTUN) Close() error {
|
||||
select {
|
||||
case <-t.closed:
|
||||
default:
|
||||
close(t.closed)
|
||||
close(t.events)
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
var _ tun.Device = (*channelTUN)(nil)
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// key material
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
type wgKeypair struct {
|
||||
private [32]byte
|
||||
public [32]byte
|
||||
}
|
||||
|
||||
func genKeypair(t *testing.T) wgKeypair {
|
||||
t.Helper()
|
||||
var kp wgKeypair
|
||||
if _, err := rand.Read(kp.private[:]); err != nil {
|
||||
t.Fatalf("read random: %v", err)
|
||||
}
|
||||
// curve25519 clamping (as WireGuard does for private keys).
|
||||
kp.private[0] &= 248
|
||||
kp.private[31] &= 127
|
||||
kp.private[31] |= 64
|
||||
pub, err := curve25519.X25519(kp.private[:], curve25519.Basepoint)
|
||||
if err != nil {
|
||||
t.Fatalf("derive public key: %v", err)
|
||||
}
|
||||
copy(kp.public[:], pub)
|
||||
return kp
|
||||
}
|
||||
|
||||
// awgObfLines are the AmneziaWG 2.0 obfuscation knobs, identical on both ends so
|
||||
// the handshake magic ranges line up. h1-h4 are ranged (AWG 2.0) so the magic
|
||||
// occupies the full uint32 in bytes 0-3 — exactly the bytes the buggy
|
||||
// ClientBind used to zero.
|
||||
const awgObfLines = "" +
|
||||
"\njc=4" +
|
||||
"\njmin=8" +
|
||||
"\njmax=80" +
|
||||
"\ns4=12" +
|
||||
"\nh1=1888111000-1888111100" +
|
||||
"\nh2=1888122000-1888122100" +
|
||||
"\nh3=1888133000-1888133100" +
|
||||
"\nh4=1888222333-1888222444"
|
||||
|
||||
// buildDevice constructs a real wireguard-go Device driven by ClientBind (the
|
||||
// detour path). It listens on a loopback UDP port and dials the peer's port.
|
||||
func buildDevice(t *testing.T, ctx context.Context, name string, tunDev tun.Device, self wgKeypair, peerPub [32]byte, listenPort int, peerAddr netip.AddrPort) (*device.Device, *ClientBind) {
|
||||
t.Helper()
|
||||
|
||||
// ClientBind (the detour bind), NOT conn.StdNetBind. Non-connect mode so the
|
||||
// bind listens on a fixed loopback port and both ends can reach each other.
|
||||
// reserved stays [0,0,0]: this is plain AmneziaWG, NOT WARP — exactly the case
|
||||
// the pre-fix unconditional clear(b[1:4]) corrupted.
|
||||
bind := NewClientBind(ctx, logger.NOP(), loopbackDialer{listenPort: listenPort}, false, peerAddr, [3]uint8{})
|
||||
|
||||
dev := device.NewDevice(ctx, tunDev, bind, &device.Logger{
|
||||
Verbosef: func(string, ...any) {},
|
||||
Errorf: func(format string, args ...any) { t.Logf("["+name+"] "+format, args...) },
|
||||
}, 0)
|
||||
|
||||
ipc := "private_key=" + hex.EncodeToString(self.private[:]) +
|
||||
fmt.Sprintf("\nlisten_port=%d", listenPort) +
|
||||
awgObfLines +
|
||||
"\npublic_key=" + hex.EncodeToString(peerPub[:]) +
|
||||
"\nendpoint=" + peerAddr.String() +
|
||||
"\npersistent_keepalive_interval=1" +
|
||||
"\nallowed_ip=0.0.0.0/0"
|
||||
|
||||
if err := dev.IpcSet(ipc); err != nil {
|
||||
t.Fatalf("[%s] IpcSet: %v", name, err)
|
||||
}
|
||||
if err := dev.Up(); err != nil {
|
||||
t.Fatalf("[%s] device up: %v", name, err)
|
||||
}
|
||||
return dev, bind
|
||||
}
|
||||
|
||||
// TestAwgDetourClientBindDelivers is the red/green e2e. It stands up two AWG
|
||||
// Devices linked through ClientBind (loopback UDP), pushes an inner IP packet,
|
||||
// and asserts delivery within 20s. See the file header for why the pre-fix
|
||||
// unconditional clear(b[1:4]) makes this impossible.
|
||||
func TestAwgDetourClientBindDelivers(t *testing.T) {
|
||||
// A pause manager must be in the context: ClientBind.receive/Send and the
|
||||
// Device both pull it via service.FromContext and call WaitActive().
|
||||
ctx := pause.ContextWithDefaultManager(context.Background())
|
||||
|
||||
// Two loopback UDP listeners just to reserve ports; ClientBind opens its own
|
||||
// sockets via the dialer, so we only need the port numbers to be free and
|
||||
// wire each side to the other's port.
|
||||
portA := reserveLoopbackUDPPort(t)
|
||||
portB := reserveLoopbackUDPPort(t)
|
||||
addrA := netip.AddrPortFrom(netip.MustParseAddr("127.0.0.1"), uint16(portA))
|
||||
addrB := netip.AddrPortFrom(netip.MustParseAddr("127.0.0.1"), uint16(portB))
|
||||
|
||||
kpA := genKeypair(t)
|
||||
kpB := genKeypair(t)
|
||||
|
||||
const mtu = 1420
|
||||
tunA := newChannelTUN("wgA", mtu)
|
||||
tunB := newChannelTUN("wgB", mtu)
|
||||
|
||||
// A: 10.0.0.1, B: 10.0.0.2 (allowed_ip 0.0.0.0/0 on both, so routing is trivial).
|
||||
devA, _ := buildDevice(t, ctx, "A", tunA, kpA, kpB.public, portA, addrB)
|
||||
devB, _ := buildDevice(t, ctx, "B", tunB, kpB, kpA.public, portB, addrA)
|
||||
defer devA.Close()
|
||||
defer devB.Close()
|
||||
defer tunA.Close()
|
||||
defer tunB.Close()
|
||||
|
||||
// Craft an inner IPv4/UDP packet from 10.0.0.1 -> 10.0.0.2 carrying a marker.
|
||||
marker := []byte("LX-AWG-DETOUR-E2E")
|
||||
pkt := buildIPv4UDP(
|
||||
netip.MustParseAddr("10.0.0.1"), netip.MustParseAddr("10.0.0.2"),
|
||||
4711, 4712, marker,
|
||||
)
|
||||
|
||||
// Feed A's TUN so the Device encrypts it and sends it (over ClientBind) to B.
|
||||
// Resend periodically: the first datagrams race the handshake, and until the
|
||||
// handshake completes there is no keypair to encrypt transport data.
|
||||
sendDone := make(chan struct{})
|
||||
go func() {
|
||||
ticker := time.NewTicker(200 * time.Millisecond)
|
||||
defer ticker.Stop()
|
||||
for {
|
||||
select {
|
||||
case <-sendDone:
|
||||
return
|
||||
default:
|
||||
}
|
||||
buf := make([]byte, len(pkt))
|
||||
copy(buf, pkt)
|
||||
select {
|
||||
case tunA.inbound <- buf:
|
||||
case <-sendDone:
|
||||
return
|
||||
}
|
||||
select {
|
||||
case <-ticker.C:
|
||||
case <-sendDone:
|
||||
return
|
||||
}
|
||||
}
|
||||
}()
|
||||
defer close(sendDone)
|
||||
|
||||
deadline := time.After(20 * time.Second)
|
||||
for {
|
||||
select {
|
||||
case got := <-tunB.outbound:
|
||||
if containsMarker(got, marker) {
|
||||
return // GREEN: inner packet delivered end-to-end through ClientBind
|
||||
}
|
||||
// Ignore non-marker traffic (keepalives never reach TUN, but be safe).
|
||||
case <-deadline:
|
||||
t.Fatal("timeout: inner AWG packet was not delivered to peer TUN within 20s " +
|
||||
"(pre-fix ClientBind zeroes bytes 1-3, collapsing the ranged h1-h4 magic " +
|
||||
"below its range so the handshake/transport packets are dropped)")
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// reserveLoopbackUDPPort grabs a free UDP port on loopback and releases it, so
|
||||
// ClientBind's own listen socket can claim it. There is a tiny race window, but
|
||||
// on loopback in a test it is not a practical problem.
|
||||
func reserveLoopbackUDPPort(t *testing.T) int {
|
||||
t.Helper()
|
||||
c, err := net.ListenUDP("udp", &net.UDPAddr{IP: net.IPv4(127, 0, 0, 1), Port: 0})
|
||||
if err != nil {
|
||||
t.Fatalf("reserve udp port: %v", err)
|
||||
}
|
||||
port := c.LocalAddr().(*net.UDPAddr).Port
|
||||
_ = c.Close()
|
||||
return port
|
||||
}
|
||||
|
||||
// buildIPv4UDP assembles a minimal IPv4 + UDP datagram (with checksums) carrying
|
||||
// payload, so a real WireGuard Device routes and delivers it as a valid IP packet.
|
||||
func buildIPv4UDP(src, dst netip.Addr, srcPort, dstPort uint16, payload []byte) []byte {
|
||||
udpLen := 8 + len(payload)
|
||||
totalLen := 20 + udpLen
|
||||
b := make([]byte, totalLen)
|
||||
|
||||
// IPv4 header
|
||||
b[0] = 0x45 // version 4, IHL 5
|
||||
b[1] = 0x00
|
||||
putU16(b[2:], uint16(totalLen))
|
||||
putU16(b[4:], 0) // id
|
||||
putU16(b[6:], 0) // flags/frag
|
||||
b[8] = 64 // TTL
|
||||
b[9] = 17 // protocol UDP
|
||||
putU16(b[10:], 0) // checksum (fill below)
|
||||
copy(b[12:16], src.AsSlice())
|
||||
copy(b[16:20], dst.AsSlice())
|
||||
putU16(b[10:], ipChecksum(b[:20]))
|
||||
|
||||
// UDP header
|
||||
putU16(b[20:], srcPort)
|
||||
putU16(b[22:], dstPort)
|
||||
putU16(b[24:], uint16(udpLen))
|
||||
putU16(b[26:], 0) // checksum optional for IPv4, leave 0
|
||||
copy(b[28:], payload)
|
||||
return b
|
||||
}
|
||||
|
||||
func putU16(b []byte, v uint16) {
|
||||
b[0] = byte(v >> 8)
|
||||
b[1] = byte(v)
|
||||
}
|
||||
|
||||
func ipChecksum(h []byte) uint16 {
|
||||
var sum uint32
|
||||
for i := 0; i+1 < len(h); i += 2 {
|
||||
sum += uint32(h[i])<<8 | uint32(h[i+1])
|
||||
}
|
||||
for sum>>16 != 0 {
|
||||
sum = (sum & 0xffff) + (sum >> 16)
|
||||
}
|
||||
return ^uint16(sum)
|
||||
}
|
||||
|
||||
func containsMarker(pkt, marker []byte) bool {
|
||||
if len(pkt) < len(marker) {
|
||||
return false
|
||||
}
|
||||
for i := 0; i+len(marker) <= len(pkt); i++ {
|
||||
if string(pkt[i:i+len(marker)]) == string(marker) {
|
||||
return true
|
||||
}
|
||||
}
|
||||
return false
|
||||
}
|
||||
@@ -50,6 +50,53 @@ func NewClientBind(ctx context.Context, logger logger.Logger, dialer N.Dialer, i
|
||||
}
|
||||
}
|
||||
|
||||
// hasReserved reports whether any Cloudflare "reserved" value is set. The
|
||||
// receive path must only zero bytes 1-3 when a reserved value exists (WARP);
|
||||
// otherwise an AmneziaWG magic header that lands in bytes 1-3 (small s1/s2/s4
|
||||
// padding) would be corrupted and the packet dropped. The send path stamps the
|
||||
// bytes under the same condition.
|
||||
//
|
||||
// This is the twin of StdNetBind.hasReserved (see the vendored
|
||||
// submodules/wireguard-go/conn/bind_std.go, hasReserved + its callers in
|
||||
// receiveIP and Send). The two binds implement ONE contract — "never touch
|
||||
// bytes 1-3 unless WARP is configured" — and must be read, and fixed, as a
|
||||
// pair. They were not: bind_std.go got the gate first and ClientBind kept the
|
||||
// unconditional clear, which is exactly how the bug below survived.
|
||||
//
|
||||
// Why the bug is detour-only. transport/wireguard/endpoint.go (Endpoint.Start)
|
||||
// picks the bind with common.Cast[dialer.WireGuardListener](e.options.Dialer):
|
||||
//
|
||||
// - no detour -> the dialer is *dialer.DefaultDialer, the only type with a
|
||||
// WireGuardControl method, so the cast succeeds -> conn.StdNetBind (gated,
|
||||
// healthy);
|
||||
// - detour set -> dialer.NewWithOptions builds a *dialer.DetourDialer, whose
|
||||
// Upstream() is the detour outbound (e.g. *direct.Outbound). Neither
|
||||
// implements WireGuardControl, so common.Cast walks the upstream chain and
|
||||
// fails -> ClientBind (this file). An AmneziaWG node therefore works
|
||||
// standalone and dies the moment it is placed behind an egress/chain hop.
|
||||
//
|
||||
// Why handshakes survived and transport did not. AmneziaWG writes the magic as
|
||||
// a little-endian uint32 at packet[padding:], where padding is s1 (initiation),
|
||||
// s2 (response) or s4 (transport) — see device.DeterminePacketTypeAndPadding.
|
||||
// With the usual s1/s2 > 0 the magic sits well past byte 3, so clearing bytes
|
||||
// 1-3 only scribbles on the random junk prefix and the handshake completes. s4
|
||||
// is 0 by default, so a transport packet puts the magic in bytes 0-3: zeroing
|
||||
// 1-3 collapses it to its low byte (<= 255), which falls outside every h4
|
||||
// range, the peer classifies it MessageUnknownType and drops it silently. The
|
||||
// session looks established locally, every data packet vanishes, and the peer
|
||||
// re-handshakes forever.
|
||||
func (c *ClientBind) hasReserved() bool {
|
||||
if c.reserved != [3]uint8{} {
|
||||
return true
|
||||
}
|
||||
for _, reserved := range c.reservedForEndpoint {
|
||||
if reserved != [3]uint8{} {
|
||||
return true
|
||||
}
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
func (c *ClientBind) connect() (*wireConn, error) {
|
||||
serverConn := c.conn
|
||||
if serverConn != nil {
|
||||
@@ -134,7 +181,12 @@ func (c *ClientBind) receive(packets [][]byte, sizes []int, eps []conn.Endpoint)
|
||||
return
|
||||
}
|
||||
sizes[0] = n
|
||||
if n > 3 {
|
||||
// lx: only strip the Cloudflare "reserved" bytes when a reserved value is
|
||||
// actually configured (WARP). AmneziaWG writes a full uint32 magic header
|
||||
// into bytes 0-3 of a transport packet (s4 defaults to 0); unconditionally
|
||||
// clearing 1-3, as upstream did, destroys it and the endpoint drops every
|
||||
// inbound data packet. StdNetBind gates the same clear on hasReserved().
|
||||
if n > 3 && c.hasReserved() {
|
||||
b := packets[0]
|
||||
clear(b[1:4])
|
||||
}
|
||||
@@ -179,7 +231,14 @@ func (c *ClientBind) Send(bufs [][]byte, ep conn.Endpoint, offset int) error {
|
||||
if !loaded {
|
||||
reserved = c.reserved
|
||||
}
|
||||
copy(buf[1:4], reserved[:])
|
||||
// lx: only stamp the reserved bytes when non-zero (WARP). For a
|
||||
// plain WG / AmneziaWG endpoint reserved is [0,0,0]; overwriting
|
||||
// bytes 1-3 zeroes the upper bytes of the AWG magic header and the
|
||||
// peer drops the packet. See the matching guard in receive() and
|
||||
// the hasReserved() comment for why this bites only on detour.
|
||||
if reserved != [3]uint8{} {
|
||||
copy(buf[1:4], reserved[:])
|
||||
}
|
||||
}
|
||||
_, err = udpConn.WriteToUDPAddrPort(buf, destination)
|
||||
if err != nil {
|
||||
|
||||
@@ -0,0 +1,260 @@
|
||||
// lx: guards the AmneziaWG-vs-reserved fix in ClientBind — bytes 1-3 (the
|
||||
// Cloudflare "reserved" field) must only be touched when a reserved value is
|
||||
// configured (WARP). For a plain WG / AmneziaWG endpoint they carry the upper
|
||||
// bytes of a magic header, and clearing them breaks the tunnel.
|
||||
//
|
||||
// These tests drive the REAL ClientBind.Send / ClientBind.receive over loopback
|
||||
// UDP, so they observe what actually lands on the wire rather than restating
|
||||
// the condition. Before the hasReserved() gate they are red:
|
||||
//
|
||||
// Send: magic corrupted on the wire: got 40 (0x00000028), want 1618116904
|
||||
// receive: magic corrupted on receive: got 40 (0x00000028), want 1618116904
|
||||
//
|
||||
// 0x28 is the low byte of the magic — exactly the failure seen on the router,
|
||||
// where a node with h4=1618116899-1618116949 put 56 (0x38, inside the range's
|
||||
// low-byte window 0x23..0x55) on the wire and the peer dropped every packet.
|
||||
package wireguard
|
||||
|
||||
import (
|
||||
"context"
|
||||
"encoding/binary"
|
||||
"net"
|
||||
"net/netip"
|
||||
"os"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"github.com/sagernet/sing/common/logger"
|
||||
M "github.com/sagernet/sing/common/metadata"
|
||||
N "github.com/sagernet/sing/common/network"
|
||||
"github.com/sagernet/wireguard-go/conn"
|
||||
)
|
||||
|
||||
// awgTransportMagic is a value from a real AmneziaWG h4 range
|
||||
// (1618116899-1618116949 = 0x60728123..0x60728155). It uses bytes 1-3, so any
|
||||
// unconditional reserved-clear collapses it to 0x28 = 40.
|
||||
const awgTransportMagic uint32 = 1618116904 // 0x60728128
|
||||
|
||||
func testAddrPort() netip.AddrPort {
|
||||
return netip.MustParseAddrPort("192.0.2.1:51820")
|
||||
}
|
||||
|
||||
func newTestClientBind(reserved [3]uint8) *ClientBind {
|
||||
return &ClientBind{
|
||||
reservedForEndpoint: make(map[netip.AddrPort][3]uint8),
|
||||
reserved: reserved,
|
||||
}
|
||||
}
|
||||
|
||||
// testLoopbackDialer is the stand-in for a detour dialer: the only thing that
|
||||
// matters is that it is NOT a dialer.DefaultDialer, so Endpoint.Start would
|
||||
// pick ClientBind over conn.StdNetBind. It hands out a real loopback UDP
|
||||
// socket and remembers it so the test can address the bind.
|
||||
type testLoopbackDialer struct {
|
||||
packetConn net.PacketConn
|
||||
}
|
||||
|
||||
func (d *testLoopbackDialer) DialContext(ctx context.Context, network string, destination M.Socksaddr) (net.Conn, error) {
|
||||
return nil, os.ErrInvalid // the tests use the non-connect (ListenPacket) path
|
||||
}
|
||||
|
||||
func (d *testLoopbackDialer) ListenPacket(ctx context.Context, destination M.Socksaddr) (net.PacketConn, error) {
|
||||
var lc net.ListenConfig
|
||||
packetConn, err := lc.ListenPacket(ctx, "udp", "127.0.0.1:0")
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
d.packetConn = packetConn
|
||||
return packetConn, nil
|
||||
}
|
||||
|
||||
var _ N.Dialer = (*testLoopbackDialer)(nil)
|
||||
|
||||
// newLoopbackClientBind builds an Open()'d ClientBind over loopback UDP. The
|
||||
// struct is populated directly (rather than via NewClientBind) so the test does
|
||||
// not need a service registry for the pause manager; every field connect(),
|
||||
// Send() and receive() touch is set.
|
||||
func newLoopbackClientBind(t *testing.T, reserved [3]uint8) (*ClientBind, *testLoopbackDialer) {
|
||||
t.Helper()
|
||||
dialer := &testLoopbackDialer{}
|
||||
bind := &ClientBind{
|
||||
ctx: context.Background(),
|
||||
logger: logger.NOP(),
|
||||
dialer: dialer,
|
||||
reservedForEndpoint: make(map[netip.AddrPort][3]uint8),
|
||||
done: make(chan struct{}),
|
||||
reserved: reserved,
|
||||
}
|
||||
if _, _, err := bind.Open(0); err != nil {
|
||||
t.Fatalf("Open: %v", err)
|
||||
}
|
||||
t.Cleanup(func() { bind.Close() })
|
||||
if _, err := bind.connect(); err != nil {
|
||||
t.Fatalf("connect: %v", err)
|
||||
}
|
||||
return bind, dialer
|
||||
}
|
||||
|
||||
func awgTransportPacket() []byte {
|
||||
// 32 bytes: type(4) + receiver(4) + counter(8) + poly1305(16) — a keepalive,
|
||||
// the exact shape that failed on the router.
|
||||
packet := make([]byte, 32)
|
||||
binary.LittleEndian.PutUint32(packet[0:4], awgTransportMagic)
|
||||
return packet
|
||||
}
|
||||
|
||||
func TestClientBindHasReserved(t *testing.T) {
|
||||
if newTestClientBind([3]uint8{}).hasReserved() {
|
||||
t.Fatal("empty reserved must report false (AmneziaWG / plain WG)")
|
||||
}
|
||||
if !newTestClientBind([3]uint8{0, 0, 1}).hasReserved() {
|
||||
t.Fatal("non-zero global reserved must report true (WARP)")
|
||||
}
|
||||
bind := newTestClientBind([3]uint8{})
|
||||
bind.reservedForEndpoint[testAddrPort()] = [3]uint8{9, 9, 9}
|
||||
if !bind.hasReserved() {
|
||||
t.Fatal("non-zero per-endpoint reserved must report true (WARP)")
|
||||
}
|
||||
}
|
||||
|
||||
// TestClientBindSendPreservesAWGMagic is the send-side regression: with no
|
||||
// reserved configured, the 4-byte AmneziaWG magic header of a transport packet
|
||||
// must reach the wire byte-for-byte.
|
||||
func TestClientBindSendPreservesAWGMagic(t *testing.T) {
|
||||
peer, err := net.ListenPacket("udp", "127.0.0.1:0")
|
||||
if err != nil {
|
||||
t.Fatalf("listen peer: %v", err)
|
||||
}
|
||||
defer peer.Close()
|
||||
peerAddr, err := netip.ParseAddrPort(peer.LocalAddr().String())
|
||||
if err != nil {
|
||||
t.Fatalf("parse peer addr: %v", err)
|
||||
}
|
||||
|
||||
bind, _ := newLoopbackClientBind(t, [3]uint8{})
|
||||
if err = bind.Send([][]byte{awgTransportPacket()}, remoteEndpoint(peerAddr), 0); err != nil {
|
||||
t.Fatalf("Send: %v", err)
|
||||
}
|
||||
|
||||
buffer := make([]byte, 128)
|
||||
if err = peer.SetReadDeadline(time.Now().Add(10 * time.Second)); err != nil {
|
||||
t.Fatalf("set deadline: %v", err)
|
||||
}
|
||||
n, _, err := peer.ReadFrom(buffer)
|
||||
if err != nil {
|
||||
t.Fatalf("read from peer: %v", err)
|
||||
}
|
||||
if n != 32 {
|
||||
t.Fatalf("unexpected datagram size on the wire: got %d, want 32", n)
|
||||
}
|
||||
if got := binary.LittleEndian.Uint32(buffer[0:4]); got != awgTransportMagic {
|
||||
t.Fatalf("magic corrupted on the wire: got %d (0x%08x), want %d (0x%08x)",
|
||||
got, got, awgTransportMagic, awgTransportMagic)
|
||||
}
|
||||
}
|
||||
|
||||
// TestClientBindSendStampsReservedForWARP pins the unchanged WARP behaviour:
|
||||
// a non-zero reserved value is still written into bytes 1-3 on send.
|
||||
func TestClientBindSendStampsReservedForWARP(t *testing.T) {
|
||||
peer, err := net.ListenPacket("udp", "127.0.0.1:0")
|
||||
if err != nil {
|
||||
t.Fatalf("listen peer: %v", err)
|
||||
}
|
||||
defer peer.Close()
|
||||
peerAddr, err := netip.ParseAddrPort(peer.LocalAddr().String())
|
||||
if err != nil {
|
||||
t.Fatalf("parse peer addr: %v", err)
|
||||
}
|
||||
|
||||
bind, _ := newLoopbackClientBind(t, [3]uint8{1, 2, 3})
|
||||
if err = bind.Send([][]byte{awgTransportPacket()}, remoteEndpoint(peerAddr), 0); err != nil {
|
||||
t.Fatalf("Send: %v", err)
|
||||
}
|
||||
|
||||
buffer := make([]byte, 128)
|
||||
if err = peer.SetReadDeadline(time.Now().Add(10 * time.Second)); err != nil {
|
||||
t.Fatalf("set deadline: %v", err)
|
||||
}
|
||||
if _, _, err = peer.ReadFrom(buffer); err != nil {
|
||||
t.Fatalf("read from peer: %v", err)
|
||||
}
|
||||
want := [3]uint8{1, 2, 3}
|
||||
if got := [3]uint8{buffer[1], buffer[2], buffer[3]}; got != want {
|
||||
t.Fatalf("WARP reserved not stamped: got %v, want %v", got, want)
|
||||
}
|
||||
if wantByte0 := byte(awgTransportMagic & 0xFF); buffer[0] != wantByte0 {
|
||||
t.Fatalf("byte 0 must be untouched: got 0x%02x, want 0x%02x", buffer[0], wantByte0)
|
||||
}
|
||||
}
|
||||
|
||||
// TestClientBindReceivePreservesAWGMagic is the receive-side regression: an
|
||||
// inbound AmneziaWG transport packet must reach the device with its magic
|
||||
// intact when no reserved value is configured.
|
||||
func TestClientBindReceivePreservesAWGMagic(t *testing.T) {
|
||||
bind, dialer := newLoopbackClientBind(t, [3]uint8{})
|
||||
bindAddr, err := net.ResolveUDPAddr("udp", dialer.packetConn.LocalAddr().String())
|
||||
if err != nil {
|
||||
t.Fatalf("resolve bind addr: %v", err)
|
||||
}
|
||||
|
||||
sender, err := net.ListenPacket("udp", "127.0.0.1:0")
|
||||
if err != nil {
|
||||
t.Fatalf("listen sender: %v", err)
|
||||
}
|
||||
defer sender.Close()
|
||||
if _, err = sender.WriteTo(awgTransportPacket(), bindAddr); err != nil {
|
||||
t.Fatalf("write to bind: %v", err)
|
||||
}
|
||||
|
||||
// Bound the blocking ReadFrom inside receive() so a lost datagram fails the
|
||||
// test instead of hanging it.
|
||||
if err = dialer.packetConn.SetReadDeadline(time.Now().Add(10 * time.Second)); err != nil {
|
||||
t.Fatalf("set deadline: %v", err)
|
||||
}
|
||||
packets := [][]byte{make([]byte, 2048)}
|
||||
sizes := make([]int, 1)
|
||||
endpoints := make([]conn.Endpoint, 1)
|
||||
count, err := bind.receive(packets, sizes, endpoints)
|
||||
if err != nil {
|
||||
t.Fatalf("receive: %v", err)
|
||||
}
|
||||
if count != 1 || sizes[0] != 32 {
|
||||
t.Fatalf("unexpected receive result: count=%d size=%d", count, sizes[0])
|
||||
}
|
||||
if got := binary.LittleEndian.Uint32(packets[0][0:4]); got != awgTransportMagic {
|
||||
t.Fatalf("magic corrupted on receive: got %d (0x%08x), want %d (0x%08x)",
|
||||
got, got, awgTransportMagic, awgTransportMagic)
|
||||
}
|
||||
}
|
||||
|
||||
// TestClientBindReceiveStripsReservedForWARP pins the unchanged WARP behaviour
|
||||
// on the receive side: with a reserved value configured, bytes 1-3 are cleared.
|
||||
func TestClientBindReceiveStripsReservedForWARP(t *testing.T) {
|
||||
bind, dialer := newLoopbackClientBind(t, [3]uint8{0, 0, 1})
|
||||
bindAddr, err := net.ResolveUDPAddr("udp", dialer.packetConn.LocalAddr().String())
|
||||
if err != nil {
|
||||
t.Fatalf("resolve bind addr: %v", err)
|
||||
}
|
||||
|
||||
sender, err := net.ListenPacket("udp", "127.0.0.1:0")
|
||||
if err != nil {
|
||||
t.Fatalf("listen sender: %v", err)
|
||||
}
|
||||
defer sender.Close()
|
||||
if _, err = sender.WriteTo(awgTransportPacket(), bindAddr); err != nil {
|
||||
t.Fatalf("write to bind: %v", err)
|
||||
}
|
||||
|
||||
if err = dialer.packetConn.SetReadDeadline(time.Now().Add(10 * time.Second)); err != nil {
|
||||
t.Fatalf("set deadline: %v", err)
|
||||
}
|
||||
packets := [][]byte{make([]byte, 2048)}
|
||||
sizes := make([]int, 1)
|
||||
endpoints := make([]conn.Endpoint, 1)
|
||||
if _, err = bind.receive(packets, sizes, endpoints); err != nil {
|
||||
t.Fatalf("receive: %v", err)
|
||||
}
|
||||
if got := binary.LittleEndian.Uint32(packets[0][0:4]); got != awgTransportMagic&0xFF {
|
||||
t.Fatalf("WARP reserved not stripped: got %d, want %d", got, awgTransportMagic&0xFF)
|
||||
}
|
||||
}
|
||||
Reference in New Issue
Block a user