feat: real random strategy over the live pool; sweep becomes an honest knob; multi-WAN egress gateway

- random is a REAL urltest mode (lx SPEC 019 v2): uniform draw over LIVE
  slots only, pool sized to every member; dead slots keep their place
  (never-shrink) but are never picked, for random AND round_robin AND
  sticky (degrade-to-live). All-dead pools fall back to Select.
- Globals.SweepInterval + Globals.GroupHealth master switch, resolved by
  one pure function (model.SweepSchedule) shared by validator and apply;
  unparseable is warned-and-ON, never silently off. ConfigureSweep no
  longer resets the cursor on every cron reconcile (release blocker:
  a ~6-min cycle was restarted every 60s and never completed).
- multi-WAN egress gateway: ubus netifd status -> uci static -> main
  table; a gatewayless non-P2P egress warns CRITICAL instead of silently
  blackholing the second uplink.
- endpoint resolver (route.default_domain_resolver): bootstrap-direct
  clone of a named resolver, profile override beats globals.
- chains are composable: chain: hops flatten recursively, cycle-guarded,
  entry egress lifts only at position 0 (fail-closed mid-path).
- group test publishes its scope so "measuring" lights only the cards a
  run covers; health run is explicitly global (all_nodes).
- panel: biased-sample honesty (no ratio until a failure CAN be on
  record), profiles auto-pin plate, sweep/GroupHealth settings UI.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
This commit is contained in:
2026-07-22 18:20:26 +03:00
co-authored by Claude Fable 5
parent 08d5d6ccb8
commit 026bba904f
41 changed files with 3291 additions and 332 deletions
+1
View File
@@ -52,6 +52,7 @@ const (
const (
URLTestModeLeastTest = "least_test" // default — pick lowest-delay node (legacy urltest behaviour)
URLTestModeRoundRobin = "round_robin" // rotate over a fixed-size pool of live nodes
URLTestModeRandom = "random" // pick a uniformly-random LIVE node from the pool per connection
)
// lx: SPEC 019 — balancer.sticky_hash key components.
+66
View File
@@ -338,6 +338,17 @@ export interface GroupsHealth {
sweep?: HealthSweep
}
/**
* The health run's scope, as the daemon publishes it (panel/api.go NodeTestScope).
*
* It is a CONSTANT, not a list, and that is the whole point: a health run measures
* every node, every endpoint and every group's egress copies in one pass, so it can
* never be attributed to one card. Render it as a single global progress indicator.
* The scoped counterpart is {@link GroupTestStatus.scope}.
*/
export const NODE_TEST_SCOPE = 'all_nodes'
export type NodeTestScope = typeof NODE_TEST_SCOPE
/**
* GET /api/nodes/test — progress of a manual "Test all nodes" probe-all run.
* `running` is true while a run is in flight; `done`/`total` count finished vs
@@ -347,6 +358,8 @@ export interface NodeTestStatus {
running: boolean
done: number
total: number
/** Always {@link NODE_TEST_SCOPE}; absent on daemons older than the split. */
scope?: NodeTestScope
}
/** POST /api/nodes/test reply: `started` when a fresh run began, else `running`. */
@@ -410,8 +423,32 @@ export interface Globals {
ConfirmTimeout: number
ResolverDefault: string
ResolverFallback: string
/**
* `config resolver` used to resolve the SERVER DOMAINS of proxy configs
* (vless/awg endpoints) — sing-box route.default_domain_resolver, a bootstrap
* resolver that is always direct. "" ⇒ not set (model.Globals.EndpointResolver,
* UCI `endpoint_resolver`).
*/
EndpointResolver: string
ProbeURL: string
ProbeInterval: string
/**
* Tick period of the daemon's background health sweep — the walk that keeps
* every node's health fresh so the group cards fill in without anyone pressing
* a button (model.Globals.SweepInterval, UCI `sweep_interval`).
*
* `""` means ENABLED at the engine's default tick, NOT off. That default is
* load-bearing: urltest groups probe only while they are being used and selector
* groups never probe at all, so without the sweep a 376-node subscription reads
* almost entirely "not measured".
*
* Accepted: a duration ("30s", "5m", "1h") or a bare integer of seconds; any of
* `0` / `off` / `none` / `disabled` to switch the sweep off. Below 5 s the
* daemon raises it to that floor. An UNRECOGNISED value does not disable the
* sweep — the daemon warns and falls back to the default — so the panel refuses
* it at the input instead of saving something that will be silently ignored.
*/
SweepInterval: string
SchemaVersion: number
ActiveProfile: string
/**
@@ -439,6 +476,20 @@ export interface Globals {
DNSFilter?: boolean // master enable for the in-engine blocklist filter
DNSIntercept?: boolean // force ALL LAN plaintext DNS (:53) through the engine, incl. router-addressed queries
BlockDoH?: boolean // block known public DoH resolvers (by host + IP:443 + Firefox canary) so clients fall back to plaintext :53
/**
* Our background health sweep of group members — the walk that fills the alive /
* dead / tested numbers the Targets page shows for each group (model.Globals
* .GroupHealth, UCI `group_health`). Default ON.
*
* This governs OUR sweep and the health/testing UI ONLY. It does NOT touch
* sing-box's own internal urltest / least_test probes: a group always keeps
* picking a live member under the hood regardless of this switch. Turning it off
* just stops the extra sweep and hides the health statistics.
*
* ABSENT ⇒ enabled (an older config never wrote the key), so the invariant the
* panel reads by is `globals?.GroupHealth !== false` — never `=== true`.
*/
GroupHealth?: boolean
PanelPort?: number // admin-panel HTTP port (0 = default 8088)
// Statistics retention. For the three sizes: 0 = UNLIMITED (no trimming — that
// aggregate grows without bound, RAM-limited); a positive value is a fixed cap.
@@ -664,6 +715,8 @@ export interface Profile {
DisableRules?: string[] | null
DefaultTarget?: string
DefaultEgress?: string
/** Per-profile override of the endpoint resolver (keyed by active WAN: SIM→yandex, WiFi→DoH). "" ⇒ no override (model.Profile.EndpointResolver, UCI `endpoint_resolver`). */
EndpointResolver?: string
}
/**
@@ -1255,6 +1308,19 @@ export interface GroupTestStatus {
running: boolean
done: number
total: number
/**
* The group names THIS run covers. Always an array (never JSON null); absent
* only on daemons older than the split.
*
* It is what makes `running` usable. On its own that flag says only "a group
* test is happening somewhere", which is why pressing Test on one group used to
* put "measuring…" on every card. The rule: show the in-progress indicator on
* card g iff `running && scope.includes(g)`. A run started with a name carries
* exactly that name; a run started with no name carries every group, and then
* the indicator on every card is correct. The scope PERSISTS after the run
* ends, so displayed results stay attributable to the cards they came from.
*/
scope?: string[]
results: GroupTestResult[]
}
+129 -7
View File
@@ -23,15 +23,23 @@ const CONFIG: Model = {
ConfirmTimeout: 90,
ResolverDefault: 'cloudflare-doh',
ResolverFallback: 'router-local',
EndpointResolver: '',
ProbeURL: 'https://www.gstatic.com/generate_204',
ProbeInterval: '60s',
// '' = the engine's default tick (every 10 s), NOT off. Left blank on purpose
// so `?mock` shows the recommended state and the placeholder that says so.
SweepInterval: '',
SchemaVersion: 2,
ActiveProfile: '',
// Points at the iface-driven profile below, so `?mock` lands in the WAN-watcher
// AUTO-PIN state (not a manual override): the plate must say the router pins this
// itself by uplink, and `mobile-uplink` must be UNSELECTABLE in the override list.
ActiveProfile: 'mobile-uplink',
// Policy for traffic TPROXY physically can't carry (non-TCP/UDP). Override
// from the URL — ?mock&untun=icmp / &untun=direct — to see all three states.
Untunnelable: new URLSearchParams(typeof location === 'undefined' ? '' : location.search).get('untun') ?? 'block',
DNSIntercept: true, // force ALL LAN plaintext DNS (:53) through the engine
BlockDoH: false, // block known public DoH resolvers so clients fall back to plaintext :53
GroupHealth: true, // background group-member health sweep + Targets health stats (default on)
StatsRingSize: 500, // fixed cap — shows the "limit" rendering (500 rows)
StatsTimelineMinutes: 0, // 0 ⇒ "Unlimited" rendering + memory warning
StatsMaxDomains: 5000, // fixed cap — shows the "limit" rendering (5000)
@@ -213,6 +221,35 @@ const CONFIG: Model = {
Fallback: true,
},
],
// Two profiles exercising the honesty fixes. `mobile-uplink` is iface-driven, so
// it is the WAN-watcher's business: it is the ActiveProfile above (an auto-pin,
// never a manual override) and must NOT be a selectable override option. Its
// three DisableRules make the "+N more" truncation visible, and the rule names
// are the real ones in CONFIG.Rules so the chips read as genuine rules.
// `evening-direct` has no iface condition, so it IS a valid manual-override
// candidate and appears as selectable in the override dropdown.
Profiles: [
{
Name: 'mobile-uplink',
Enabled: true,
Priority: 20,
MatchIface: ['wan1'],
SchedDays: [],
EnableRules: ['default-tunnel'],
DisableRules: ['block-ads', 'ru-bypass', 'private-direct'],
DefaultTarget: 'group:auto',
},
{
Name: 'evening-direct',
Enabled: true,
Priority: 10,
MatchIface: [],
SchedDays: ['mon', 'tue', 'wed', 'thu', 'fri'],
SchedStart: '19:00',
SchedEnd: '23:00',
DefaultTarget: 'direct',
},
],
}
const wait = (ms = 220) => new Promise((r) => setTimeout(r, ms))
@@ -881,6 +918,31 @@ const HEALTH_SHAPE: Record<string, HealthShape> = {
fallback: { total: 24, alive: 0, dead: 0, bound: false, type: 'urltest', selectedIndex: null },
}
/**
* `?mock&bias=1|2|3` reproduces the optimistic-ratio defect the owner hit on the
* live router, on `auto` — 298 members, the same size he reported — and walks it
* through the three states the rendering has to tell apart.
*
* The mechanism: a group writes an entry when a member answers and DELETES it when
* one doesn't, so until the background sweep has been over the group its history
* holds nothing but successes. `11 / 11 alive` was that, not health.
*
* bias=1 → 11 alive, 0 dead, 287 unchecked, sweep mid-first-pass → NO RATIO
* bias=2 → 111 alive, 74 dead, 113 unchecked, sweep mid-first-pass → ratio (a
* failure is on record, so something is writing both outcomes)
* bias=3 → the same counts with a completed pass → ratio
*
* Read 1 → 2 → 3 in order: the reading must get MORE PRECISE, never "good, then
* suddenly bad".
*/
const BIAS_STAGE =
typeof location === 'undefined' ? null : new URLSearchParams(location.search).get('bias')
if (BIAS_STAGE === '1') {
HEALTH_SHAPE.auto = { ...HEALTH_SHAPE.auto, alive: 11, dead: 0 }
} else if (BIAS_STAGE === '2' || BIAS_STAGE === '3') {
HEALTH_SHAPE.auto = { ...HEALTH_SHAPE.auto, alive: 111, dead: 74 }
}
/** Build one group's member list from its shape. Deterministic, so screenshots
* of the same fixture are identical between runs. */
function buildMembers(group: string, shape: HealthShape): GroupMemberHealth[] {
@@ -992,11 +1054,36 @@ function advanceNodeTest(): void {
if (nodeTestQueue.length === 0) nodeTest = { ...nodeTest, running: false }
}
// ?mock&nodetest=1 lands straight in the running state (see the note above).
// ?mock&nodetest=1 lands straight in the running state (see the note above). This
// is the HEALTH run — global by nature, so the panel must render it as ONE
// indicator in the section header and never as a per-card badge.
if (typeof location !== 'undefined' && new URLSearchParams(location.search).has('nodetest')) {
startNodeTest()
}
/**
* The background sweep's progress, as GET /api/groups/health reports it.
*
* `cycles` is the field with teeth: 0 means the sweep has not been everywhere yet,
* so "not measured" is simply "not reached". Once it is ≥ 1 the sweep HAS been
* everywhere, and anything still unmeasured lost a reading it used to have.
*
* ?mock&sweep=first → mid first pass (cycles 0)
* ?mock&sweep=off → sweep disabled; nothing fills in on its own
*/
function mockSweep(): { enabled: boolean; cursor: number; total: number; cycles: number } {
const mode = typeof location === 'undefined' ? null : new URLSearchParams(location.search).get('sweep')
if (mode === 'off') return { enabled: false, cursor: 0, total: 0, cycles: 0 }
if (mode === 'first') return { enabled: true, cursor: 184, total: 707, cycles: 0 }
// The bias walkthrough drives the sweep too: stages 1 and 2 are mid-first-pass
// (the cursor advances between them, exactly as the live router's did), stage 3
// has completed one. See BIAS_STAGE.
if (BIAS_STAGE === '1') return { enabled: true, cursor: 312, total: 896, cycles: 0 }
if (BIAS_STAGE === '2') return { enabled: true, cursor: 696, total: 896, cycles: 0 }
if (BIAS_STAGE === '3') return { enabled: true, cursor: 148, total: 896, cycles: 1 }
return { enabled: true, cursor: 184, total: 707, cycles: 3 }
}
export async function getGroupsHealth(
opts: { group?: string; members?: boolean } = {},
): Promise<GroupsHealth> {
@@ -1008,7 +1095,7 @@ export async function getGroupsHealth(
node_test_total: nodeTest.total,
// The daemon's background sweep, on by default — it is what lets the panel
// promise that untested members resolve without anyone pressing anything.
sweep: { enabled: true, cursor: 184, total: 707, cycles: 3 },
sweep: mockSweep(),
}
if (opts.group) {
const members = GROUP_MEMBERS.get(opts.group)
@@ -1061,7 +1148,10 @@ export async function postNodesTest(): Promise<NodeTestStart> {
export async function getNodesTest(): Promise<NodeTestStatus> {
await wait(60)
advanceNodeTest()
return { ...nodeTest }
// The constant the daemon publishes (api.ts NODE_TEST_SCOPE, spelled inline
// because this module may only import TYPES from api.ts): this run is GLOBAL and
// can never be attributed to one group's card.
return { ...nodeTest, scope: 'all_nodes' }
}
// Mock group test. Deliberately covers every state the UI has to render, one per
@@ -1118,8 +1208,10 @@ function shapeFor(group: string, i: number): GroupTestResult {
return { ...base, group, tested_unix: Math.floor(Date.now() / 1000) }
}
let groupTest: GroupTestStatus = { running: false, done: 0, total: 0, results: [] }
let groupTest: GroupTestStatus = { running: false, done: 0, total: 0, scope: [], results: [] }
let groupTestQueue: string[] = []
/** A URL-seeded run stays running instead of draining (see the seed block below). */
let groupTestPinned = false
export async function postGroupsTest(name = ''): Promise<GroupTestStart> {
await wait(60)
@@ -1128,10 +1220,14 @@ export async function postGroupsTest(name = ''): Promise<GroupTestStart> {
const targets = name ? all.filter((g) => g === name) : all
if (targets.length === 0) return { started: false, reason: `no group named “${name}”` }
groupTestQueue = [...targets]
groupTestPinned = false
groupTest = {
running: true,
done: 0,
total: targets.length,
// The names this run covers — exactly what the panel tests card membership
// against. One name for a per-group run, every name for a run-all.
scope: [...targets],
// A re-test of ONE group replaces just that group's row and keeps the rest,
// exactly as a per-group daemon run would.
results: groupTest.results.filter((r) => !targets.includes(r.group)),
@@ -1141,7 +1237,7 @@ export async function postGroupsTest(name = ''): Promise<GroupTestStart> {
export async function getGroupsTest(): Promise<GroupTestStatus> {
await wait(60)
if (groupTest.running) {
if (groupTest.running && !groupTestPinned) {
const next = groupTestQueue.shift()
if (next !== undefined) {
groupTest = {
@@ -1152,7 +1248,33 @@ export async function getGroupsTest(): Promise<GroupTestStatus> {
}
if (groupTestQueue.length === 0) groupTest = { ...groupTest, running: false }
}
return { ...groupTest, results: groupTest.results.map((r) => ({ ...r })) }
return {
...groupTest,
scope: [...(groupTest.scope ?? [])],
results: groupTest.results.map((r) => ({ ...r })),
}
}
/**
* Land straight in a RUNNING group test, so the scoped in-progress indicator is
* screenshot-stable instead of draining a group per poll:
*
* ?mock&grouptest=stealth → one group in scope — the badge belongs to that card
* and to no other, which is the defect this fixture guards
* ?mock&grouptest=all → every group in scope — the badge on every card is then
* CORRECT, and the two cases must be told apart
*/
if (typeof location !== 'undefined') {
const want = new URLSearchParams(location.search).get('grouptest')
if (want) {
const all = (CONFIG.Groups ?? []).map((g) => g.Name)
const targets = want === 'all' || want === '1' ? all : all.filter((g) => g === want)
if (targets.length > 0) {
groupTestQueue = [...targets]
groupTestPinned = true
groupTest = { running: true, done: 0, total: targets.length, scope: [...targets], results: [] }
}
}
}
/** Exposed for potential UI hints; not part of the wire contract. */
+33
View File
@@ -538,6 +538,20 @@ export default function DNS() {
[config, save],
)
// The resolver used to look up the SERVER domains inside your proxy configs
// (vless/awg endpoints), separate from the client-facing DNS above. Empty ⇒
// the engine's built-in default.
const setEndpointResolver = useCallback(
(name: string) => {
if (!config) return
void save(
{ ...config, Globals: { ...config.Globals, EndpointResolver: name } },
name ? `Endpoint resolver → ${name}` : 'Endpoint resolver cleared',
)
},
[config, save],
)
// ---- DNS-rule mutations ---------------------------------------------------
// Rules are held sorted by ascending Order (first match wins). A DNS rule has no
// Name in the contract, so list POSITION is its only handle — every mutation
@@ -716,7 +730,26 @@ export default function DNS() {
/>
</dd>
</div>
<div className="dns-readout-item">
<dt>endpoint resolver</dt>
<dd>
<RoleSelect
value={globals?.EndpointResolver ?? ''}
names={resolverNames}
busy={busy}
disabled={!config}
ariaLabel="Endpoint resolver"
onChange={setEndpointResolver}
/>
</dd>
</div>
</dl>
<p className="dns-filter-sub dns-filter-note">
The endpoint resolver looks up the server domains in your proxy configs (the
vless / awg endpoints), separate from the client DNS above. Leave it on
<span className="mono"> none</span> to use the engine default; point it at a plain,
direct resolver so it can bootstrap before any tunnel is up.
</p>
</div>
</div>
+5
View File
@@ -435,6 +435,11 @@
.pf-erow-ctl {
min-width: 0;
}
/* Schedule sub-block dimmed for an iface-driven profile: the watcher ignores the
schedule then, so the fields stay editable but read as inert. */
.pf-erow--muted {
opacity: 0.55;
}
.pf-ehint {
margin: 6px 0 0;
font-family: var(--font-sans);
+142 -21
View File
@@ -22,6 +22,11 @@ const errText = (e: unknown): string =>
const asArray = <T,>(a: T[] | null | undefined): T[] => (a ? a : [])
// "block-ads, ru-bypass +1 more" — a couple of names for the summary chip, the
// rest counted. The full list rides in the chip's title for hover.
const ruleSummary = (names: string[], max = 2): string =>
names.length <= max ? names.join(', ') : `${names.slice(0, max).join(', ')} +${names.length - max} more`
// Weekday tokens as stored on the model (mon..sun), in display order.
const DAYS = ['mon', 'tue', 'wed', 'thu', 'fri', 'sat', 'sun'] as const
const cap = (s: string): string => (s ? s[0].toUpperCase() + s.slice(1) : s)
@@ -143,6 +148,7 @@ export default function Profiles() {
const presets = useMemo<Preset[]>(() => asArray(config?.Presets), [config])
const ruleNames = useMemo(() => namesOf(config?.Rules), [config])
const egresses = useMemo(() => asArray(config?.Egresses), [config])
const resolverNames = useMemo(() => namesOf(config?.Resolvers), [config])
const catalog = useMemo<TargetCatalog>(
() => ({
@@ -256,6 +262,31 @@ export default function Profiles() {
[renameProfile, profileNames],
)
// ---- live refresh of the active profile -----------------------------------
// The WAN-watcher rewrites Globals.ActiveProfile in the background (an uplink
// change re-pins the matching iface-profile), so a static page goes stale until
// a reload. Poll the whole config on a light interval — but ONLY while fully
// idle, since loadConfig() replaces `config` wholesale and would otherwise stomp
// an in-flight edit or flicker a toast. The guard is read from a ref so the
// interval stays armed across renders instead of tearing down every keystroke.
const idleRef = useRef(true)
idleRef.current = !dirty && !saving && !applying && openName === null
useEffect(() => {
const tick = () => {
if (document.hidden || !idleRef.current) return
void loadConfig()
}
const id = window.setInterval(tick, 4500)
const onVisible = () => {
if (!document.hidden) tick()
}
document.addEventListener('visibilitychange', onVisible)
return () => {
window.clearInterval(id)
document.removeEventListener('visibilitychange', onVisible)
}
}, [loadConfig])
const loading = config === null && loadError === null
return (
@@ -332,6 +363,7 @@ export default function Profiles() {
busy={busy}
ruleNames={ruleNames}
egresses={egresses}
resolverNames={resolverNames}
catalog={catalog}
valid={targetValid}
taken={profileNames}
@@ -402,8 +434,16 @@ function ActivePlate({
disabled: boolean
onChange: (name: string) => void
}) {
const manual = active !== ''
const missing = manual && !names.has(active)
// Four honest states. An ActiveProfile that points at an iface-driven profile is
// the WAN-watcher's auto-pin (it writes the matching profile's name on every
// uplink change), never a user's manual choice — so it must not read as one.
const matched = active !== '' ? profiles.find((p) => p.Name === active) : undefined
const missing = active !== '' && !names.has(active)
const matchIfaces = matched ? asArray(matched.MatchIface) : []
const ifaceDriven = matchIfaces.length > 0
const state: 'auto' | 'autopin' | 'manual' | 'missing' =
active === '' ? 'auto' : missing ? 'missing' : ifaceDriven ? 'autopin' : 'manual'
// Candidate order the router evaluates in auto mode: enabled, highest priority first.
const candidates = useMemo(
() =>
@@ -413,28 +453,25 @@ function ActivePlate({
.map((p) => p.Name),
[profiles],
)
// Only non-iface profiles are manual-override candidates — the watcher owns the
// iface ones and would stamp over any manual pin on its next tick.
const selectable = useMemo(
() => profiles.filter((p) => asArray(p.MatchIface).length === 0),
[profiles],
)
const ledLabel =
state === 'auto' ? 'Auto' : state === 'autopin' ? 'Auto · uplink' : 'Manual override'
return (
<div className="pf-active" aria-label="Active profile">
<div className="pf-active-lamp">
<Led variant={manual ? 'on' : 'amber'} label={manual ? 'Manual override' : 'Auto'} />
<Led variant={state === 'auto' ? 'amber' : 'on'} label={ledLabel} />
</div>
<div className="pf-active-copy">
<span className="pf-active-label mono">Active profile</span>
<p className="pf-active-sub">
{manual ? (
missing ? (
<>
Forced to <strong className="mono">{active}</strong>, which no longer exists — pick
another or switch back to auto.
</>
) : (
<>
Forced to <strong className="mono">{active}</strong>. A manual override wins over
every condition until you set it back to auto.
</>
)
) : (
{state === 'auto' && (
<>
Auto — the router picks the highest-priority profile whose conditions hold.
{candidates.length > 0 && (
@@ -445,6 +482,26 @@ function ActivePlate({
)}
</>
)}
{state === 'autopin' && (
<>
<strong className="mono">{active}</strong> is active because its uplink condition
matches the current WAN (<span className="mono">{matchIfaces.join(', ')}</span>). The
router switches this on its own when the active uplink changes — it isn’t a manual
override.
</>
)}
{state === 'manual' && (
<>
Forced to <strong className="mono">{active}</strong>. A manual override wins over every
condition until you set it back to auto.
</>
)}
{state === 'missing' && (
<>
Forced to <strong className="mono">{active}</strong>, which no longer exists — pick
another or switch back to auto.
</>
)}
</p>
</div>
<label className="pf-active-ctl">
@@ -457,11 +514,19 @@ function ActivePlate({
aria-label="Active profile override"
>
<option value="">Auto (by condition)</option>
{profiles.map((p) => (
{selectable.map((p) => (
<option key={p.Name} value={p.Name}>
{p.Name}
</option>
))}
{/* Auto-pin: show the watcher-managed profile as the current selection, but
disabled — the operator can leave it (Auto / a non-iface profile) yet
can't pick it, because the watcher would immediately overwrite it. */}
{state === 'autopin' && (
<option value={active} disabled>
Auto · managed by uplink ({active})
</option>
)}
{missing && <option value={active}>{active} (missing)</option>}
</select>
</label>
@@ -567,6 +632,7 @@ function ProfileRow({
busy,
ruleNames,
egresses,
resolverNames,
catalog,
valid,
taken,
@@ -582,6 +648,7 @@ function ProfileRow({
busy: boolean
ruleNames: string[]
egresses: { Name: string }[]
resolverNames: string[]
catalog: TargetCatalog
valid: Set<string>
taken: Set<string>
@@ -612,7 +679,9 @@ function ProfileRow({
<span className="pf-badge mono">prio {profile.Priority ?? 0}</span>
{isActive && (
<span className="pf-role">
<Led variant="on" label="Active now (manual)" />
{/* An iface-driven active profile is the watcher's auto-pin, not a
manual choice — only the non-iface case is a real manual override. */}
<Led variant="on" label={iface.length > 0 ? 'Active now · uplink' : 'Active now (manual)'} />
<span className="pf-badge pf-badge--accent">active</span>
</span>
)}
@@ -661,10 +730,20 @@ function ProfileRow({
</span>
)}
{enableRules.length > 0 && (
<span className="pf-chip pf-chip--on">+{enableRules.length} on</span>
<span
className="pf-chip pf-chip--on"
title={`Forced on while active: ${enableRules.join(', ')}`}
>
rules on: {ruleSummary(enableRules)}
</span>
)}
{disableRules.length > 0 && (
<span className="pf-chip pf-chip--offr">−{disableRules.length} off</span>
<span
className="pf-chip pf-chip--offr"
title={`Forced off while active: ${disableRules.join(', ')}`}
>
rules off: {ruleSummary(disableRules)}
</span>
)}
</div>
</div>
@@ -696,6 +775,7 @@ function ProfileRow({
busy={busy}
ruleNames={ruleNames}
egresses={egresses}
resolverNames={resolverNames}
catalog={catalog}
valid={valid}
taken={taken}
@@ -714,6 +794,7 @@ function ProfileEditor({
busy,
ruleNames,
egresses,
resolverNames,
catalog,
valid,
taken,
@@ -724,6 +805,7 @@ function ProfileEditor({
busy: boolean
ruleNames: string[]
egresses: { Name: string }[]
resolverNames: string[]
catalog: TargetCatalog
valid: Set<string>
taken: Set<string>
@@ -824,9 +906,14 @@ function ProfileEditor({
are gone from the daemon; the conditions that remain are the two below,
which are read for real. */}
<div className="pf-erow">
<div className={iface.length > 0 ? 'pf-erow pf-erow--muted' : 'pf-erow'}>
<span className="pf-elabel">Schedule</span>
<div className="pf-erow-ctl">
{iface.length > 0 && (
<p className="pf-ehint">
The router ignores the schedule while this profile is driven by its uplink interface.
</p>
)}
<div className="pf-days" role="group" aria-label="Active days (none = every day)">
{DAYS.map((d) => {
const on = days.includes(d)
@@ -934,6 +1021,40 @@ function ProfileEditor({
</div>
</div>
<div className="pf-erow">
<span className="pf-elabel">Endpoint resolver</span>
<div className="pf-erow-ctl">
<select
className="pf-select"
value={profile.EndpointResolver ?? ''}
onChange={(e) =>
onPatch(
{ EndpointResolver: e.target.value },
e.target.value
? `Endpoint resolver → ${e.target.value}`
: 'Endpoint resolver override cleared',
)
}
disabled={busy}
aria-label="Endpoint resolver"
>
<option value="">No override</option>
{resolverNames.map((rn) => (
<option key={rn} value={rn}>
{rn}
</option>
))}
{profile.EndpointResolver && !resolverNames.includes(profile.EndpointResolver) && (
<option value={profile.EndpointResolver}>{profile.EndpointResolver} (missing)</option>
)}
</select>
<p className="pf-ehint">
Overrides which resolver looks up your proxy server domains while this profile is active
— e.g. a plain resolver on SIM, DoH on Wi-Fi.
</p>
</div>
</div>
<div className="pf-erow">
<span className="pf-elabel">Rules</span>
<div className="pf-erow-ctl">
+101
View File
@@ -103,6 +103,64 @@ function parseDuration(raw: string): ParseResult<string> {
return { ok: true, value: s.toLowerCase() }
}
// ---- health sweep interval -------------------------------------------------
//
// The daemon's own vocabulary (model.SweepDisabledValues / SweepIntervalMin /
// SweepSchedule), mirrored here so the input accepts exactly what the router
// accepts and nothing else.
/** The four spellings that switch the background sweep OFF. */
const SWEEP_OFF = ['0', 'off', 'none', 'disabled']
/** Anything shorter is raised to this by the daemon, so the field raises it here. */
const SWEEP_MIN_S = 5
const UNIT_S: Record<string, number> = { ms: 0.001, s: 1, m: 60, h: 3600 }
/** Seconds in a duration this field already validated, or null. */
function durationSeconds(s: string): number | null {
const m = /^(\d+(?:\.\d+)?)(ms|s|m|h)$/.exec(s)
return m ? Number(m[1]) * UNIT_S[m[2]] : null
}
/**
* Parse the sweep tick.
*
* The rule worth stating: an unrecognised value does NOT mean "off". The daemon
* warns about it and quietly falls back to its default, so a value typed as a way
* to stop the sweep would leave the sweep running. The field refuses it here
* instead, where the person can still see what they typed.
*
* '' → the engine default (every 10 s)
* 0 / off / none / disabled → stopped
* 30, 30s, 5m, 1h → that tick, raised to the 5 s floor
*/
function parseSweepInterval(raw: string): ParseResult<string> {
const s = raw.trim().toLowerCase()
if (s === '') return { ok: true, value: '' }
if (SWEEP_OFF.includes(s)) return { ok: true, value: s }
// A bare integer is seconds, as everywhere else in the model — normalised so the
// stored value says which unit it meant.
const normalised = /^\d+$/.test(s) ? `${s}s` : s
const seconds = DURATION_RE.test(normalised) ? durationSeconds(normalised) : null
if (seconds === null) {
return {
ok: false,
error: 'Use a duration like 30s, 5m, or 1h — or “off” to stop the sweep.',
}
}
// The floor, applied where it is visible. Left to the daemon it would be a
// warning in a log nobody reads while the field went on showing 1s.
if (seconds < SWEEP_MIN_S) return { ok: true, value: `${SWEEP_MIN_S}s` }
return { ok: true, value: normalised }
}
/** What a saved sweep value means, as the toast that confirms it. */
function sweepSavedMsg(v: string): string {
if (v === '') return 'Health sweep → the default, every 10 s'
if (SWEEP_OFF.includes(v)) return 'Health sweep off — most nodes will read “not measured”'
if (v === `${SWEEP_MIN_S}s`) return `Health sweep → ${v} (the shortest allowed)`
return `Health sweep → every ${v}`
}
// `none` really does silence the engine log — it is emitted as the engine's own
// log-disable switch, not as a quieter level.
const LOG_LEVELS: ReadonlyArray<{ value: string; label: string }> = [
@@ -228,6 +286,11 @@ export default function Settings() {
// Retention controls are meaningless with logging off; disable them there.
const retentionDisabledCtl = busy || !ready || loggingOff
// Our group-member health sweep + all the health/testing stats on the Targets
// page. Absent ⇒ enabled (older config), so read it as `!== false`. When off, the
// sweep-interval field below is meaningless, so it is dimmed (value kept).
const groupHealthOn = globals?.GroupHealth !== false
const killSwitch = globals?.KillSwitch === 'open' ? 'open' : 'closed'
const killNote =
killSwitch === 'open'
@@ -373,6 +436,24 @@ export default function Settings() {
<p className="set-group-note">
Defaults for a group’s connectivity probe — a group without its own values falls back to these.
</p>
<Field
label="Group health checks"
note="Runs a background sweep that checks each group’s member nodes and powers the alive / tested health stats on the Targets page. Turn it off to stop that sweep and hide those stats — your groups still keep picking a live node on their own (sing-box probes them under the hood). The probe URL and interval below still feed those built-in checks."
>
<Toggle
pressed={groupHealthOn}
onChange={(on) =>
setGlobal(
'GroupHealth',
on,
on ? 'Group health checks on' : 'Group health checks off — sweep stopped',
)
}
label={groupHealthOn ? 'Turn off group health checks' : 'Turn on group health checks'}
size="md"
disabled={busy || !ready}
/>
</Field>
<Field label="Probe URL" note="Fetched to test whether a node is alive.">
<InlineEdit<string>
value={globals?.ProbeURL ?? ''}
@@ -400,6 +481,26 @@ export default function Settings() {
onCommit={(v) => setGlobal('ProbeInterval', v, `Probe interval → ${v}`)}
/>
</Field>
<Field
label="Health sweep interval"
note={
groupHealthOn
? 'How often the router probes nodes in the background so the health numbers on the Targets page stay fresh. Blank = every 10 s (recommended). Each tick measures only the nodes nobody else is checking, a few at a time, so a full pass takes roughly 5–10 minutes. Set “off” to stop it — at the cost of most nodes reading “not measured”.'
: 'Group health checks are off, so the sweep isn’t running and this interval has no effect. Turn group health checks back on to use it. (Your saved value is kept.)'
}
>
<InlineEdit<string>
value={globals?.SweepInterval ?? ''}
format={(s) => s}
parse={parseSweepInterval}
width="8rem"
placeholder="10s"
ariaLabel="Health sweep interval"
busy={busy}
disabled={!ready || !groupHealthOn}
onCommit={(v) => setGlobal('SweepInterval', v, sweepSavedMsg(v))}
/>
</Field>
</Group>
{/* ---- STATISTICS & LOGGING ---- */}
+75 -5
View File
@@ -617,15 +617,44 @@
padding: 6px 14px;
font-size: 11px;
}
.tg-testprog {
/* ---- run indicators (header) ----
* Two background runs report here, and they are different actions with different
* reach — a health refresh covers every group at once, an exit test covers only
* the groups it names. Each says WHAT it is and HOW FAR it has got, so neither can
* be mistaken for the other, and neither is ever repeated as a badge on the cards.
* The third is the sweep: not a run someone started, so no LED and no pulse — a
* quiet gauge that explains why untested members fill in by themselves. */
.tg-run {
display: inline-flex;
align-items: center;
gap: 6px;
padding: 3px 9px;
border: 1px solid var(--groove);
border-radius: 6px;
background: color-mix(in srgb, var(--sink) 55%, transparent);
font-size: 11px;
letter-spacing: 0.06em;
letter-spacing: 0.04em;
color: var(--dim);
white-space: nowrap;
}
.tg-run-what {
font-family: var(--font-sans);
letter-spacing: 0.02em;
}
.tg-run-n {
font-size: 10.5px;
color: var(--faint);
}
/* An in-flight run is amber-edged; the sweep is not a run and stays neutral. */
.tg-run--health,
.tg-run--exit {
border-color: color-mix(in srgb, var(--amber) 45%, var(--groove));
color: var(--ink);
}
.tg-run--sweep {
cursor: help;
color: var(--faint);
}
.tg-test-err {
margin: 10px 2px 0;
font-family: var(--font-sans);
@@ -715,6 +744,15 @@
background: var(--led-on);
box-shadow: 0 0 5px color-mix(in srgb, var(--led-on) 55%, transparent);
}
/* A one-sided sample (see isBiasedSample): every reading on record is a success,
because a group erases its own failures. The green is still TRUE — those members
really did answer — so it keeps its colour, but it loses the lit glow that reads
as a verdict. A confident bar under a sample nothing could have failed is the
same optimistic lie as the ratio, drawn instead of written. */
.gh--partial .gh-seg--alive {
box-shadow: none;
background: color-mix(in srgb, var(--led-on) 72%, var(--sink));
}
.gh-seg--dead {
background: var(--crit);
box-shadow: 0 0 5px color-mix(in srgb, var(--crit) 60%, transparent);
@@ -767,6 +805,33 @@
text-transform: uppercase;
color: var(--faint);
}
/* The open remainder of a one-sided sample. Not a failure and not a reassurance —
it is the size of what nobody has established yet, so it is neutral and plain. */
.gh-open {
padding: 1px 6px;
border: 1px dashed color-mix(in srgb, var(--dim) 40%, var(--groove));
border-radius: 4px;
font-size: 10.5px;
color: var(--dim);
white-space: nowrap;
cursor: help;
}
/* No verdict, so the headline count does not carry the weight of one. */
.gh--partial .gh-headline b {
color: var(--dim);
}
/* The failures. Loud enough to be found on a line of grey, quiet enough that it
never outranks the headline pair it qualifies — it answers "how bad", not
"what is this group". Crit carries it; the orange accent has no business here. */
.gh-dead {
padding: 1px 6px;
border-radius: 4px;
background: color-mix(in srgb, var(--crit) 12%, transparent);
font-size: 10.5px;
color: var(--crit);
white-space: nowrap;
cursor: help;
}
/* The quiet remainder. Deliberately the smallest, dimmest thing on the line: it
is context for the two numbers, never a third number competing with them. */
.gh-rest {
@@ -851,11 +916,16 @@
.gh-more[aria-expanded='true'] .gh-caret {
transform: rotate(90deg);
}
.gh-more--all {
color: var(--faint);
}
/* members */
/* The reading order, stated above the rows rather than inside the button that
opens them: the list is shown whole, and only the ORDER is opinionated. */
.gh-mems-order {
margin: 0;
font-family: var(--font-sans);
font-size: 11px;
color: var(--faint);
}
.gh-mems {
list-style: none;
margin: 0;
+333 -96
View File
@@ -21,6 +21,7 @@ import type {
GroupsHealth,
GroupTestResult,
GroupTestStatus,
HealthSweep,
Chain,
Egress,
Interface,
@@ -185,15 +186,16 @@ function refWarning(refs: RefSite[]): string {
}
/**
* The four ways a group can pick a member. Each label states the CONSEQUENCE for
* the operator's traffic, because the mechanical names lie by omission: "single"
* The ways a group can pick a member. Each label states the CONSEQUENCE for the
* operator's traffic, because the mechanical names lie by omission: "single"
* sounds like "the first one that's up" and is not, and "failover" sounds like it
* comes home again and does not.
*
* `Random` and `Least load` used to be here. Both quietly became "fastest node" —
* the engine has no randomised picker and no load metric — so they are gone rather
* than relabelled. A group still storing one of them shows it as unknown and keeps
* behaving as it always did until it is re-picked.
* `Least load` used to be here too. It quietly became "fastest node" — the engine
* has no load metric — so it is gone rather than relabelled. `Random` is a real
* picker again: the engine now spreads each connection across the live members. A
* group still storing a value this engine doesn't have shows it as unknown and
* keeps behaving as it always did until it is re-picked.
*/
const STRATEGIES: ReadonlyArray<{ id: string; label: string; blurb: string }> = [
{
@@ -206,6 +208,12 @@ const STRATEGIES: ReadonlyArray<{ id: string; label: string; blurb: string }> =
label: 'Round-robin — spread across every node',
blurb: 'Connections rotate through all members. Expect your exit IP to change as you browse.',
},
{
id: 'random',
label: 'Random — spread evenly across live nodes',
blurb:
'Each new connection lands on a random live member, so traffic ends up spread evenly across all of them — an approximation of round-robin. Expect your exit IP to vary as you browse.',
},
{
id: 'failover',
label: 'Failover — first working node, in order',
@@ -226,6 +234,7 @@ const STRATEGY_LABEL: Record<string, string> = Object.fromEntries(
const STRATEGY_SHORT: Record<string, string> = {
leastping: 'least ping',
roundrobin: 'round-robin',
random: 'random',
failover: 'failover',
single: 'single — no failover',
}
@@ -302,26 +311,78 @@ const DPI_TYPES = new Set(['interface', 'direct'])
// untested; rendering that as dead raises an alarm at the exact moment
// nothing is wrong, which teaches the operator to distrust every reading.
/** What a group's numbers add up to, in the five states they actually have. */
type Verdict = 'good' | 'degraded' | 'down' | 'unmeasured' | 'empty'
// ---- the sample is not always symmetric -------------------------------------
//
// A group's own history is written by TWO parties, and only one of them can
// record a failure:
//
// - the group itself probes its members, writes an entry on success, and on
// failure DELETES the entry (protocol/group/urltest.go:512). It is physically
// incapable of recording "dead" — its failures are indistinguishable from
// "never probed";
// - the background sweep (our overlay) is the only writer that puts a
// trustworthy "dead" on record.
//
// So until the sweep has been over a group, its history holds ONLY successes.
// "11 / 11 alive" then does not mean the group is healthy; it means the failures
// erased themselves. The real reading, minutes later, was 111 alive / 74 dead.
//
// That is an OPTIMISTIC lie, which is the worst kind here: it invites someone to
// route traffic through a group where 40% of the members are down. So while the
// sample is knowably one-sided the panel does not present it as a ratio at all —
// a ratio promises a denominator that was checked in both directions, and this
// one wasn't.
/** What a group's numbers add up to, in the six states they actually have. */
type Verdict = 'good' | 'partial' | 'degraded' | 'down' | 'unmeasured' | 'empty'
/**
* Is this group's sample knowably one-sided — only successes on record, with
* members still unaccounted for and nothing yet able to have recorded a failure?
*
* All three conditions matter:
* - `dead === 0` — the moment ONE failure is on record, something has been
* writing both outcomes and the ratio is honest;
* - `untested > 0` — a fully measured group has no room for hidden failures;
* - `!sweptOnce` — once the sweep has completed a pass it has had its say
* about every member, so silence now means something.
*
* It self-clears on the first failure or the first completed pass, whichever
* comes first — no timers, no flags, nothing to get stuck.
*/
function isBiasedSample(h: GroupHealth, sweptOnce: boolean): boolean {
return h.dead === 0 && h.untested > 0 && !sweptOnce
}
/**
* Read a group's counters as one verdict.
*
* `unmeasured` is deliberately NOT a failure and NOT a success: nothing has been
* measured, so there is nothing to claim. It is an invitation to measure.
* `unmeasured` and `partial` are deliberately NOT failures and NOT successes.
* `unmeasured` has nothing to claim at all; `partial` has confirmed answers but
* no basis for a proportion. Both decline to make a health claim rather than
* make a flattering one.
*/
function verdictOf(h: GroupHealth): Verdict {
function verdictOf(h: GroupHealth, sweptOnce: boolean): Verdict {
if (h.total === 0) return 'empty'
if (h.tested === 0) return 'unmeasured'
if (h.alive === 0) return 'down'
if (h.dead > 0) return 'degraded'
if (isBiasedSample(h, sweptOnce)) return 'partial'
return 'good'
}
/** Semantics carry the colour — good/warn/crit, never the orange accent. */
/**
* Semantics carry the colour — good/warn/crit, never the orange accent.
*
* `partial` is UNLIT, like `unmeasured`: an unlit lamp is this panel's way of
* saying "no verdict", and that is exactly the state. Lighting it green would be
* the lie; lighting it amber would claim a fault nobody has found. Unlit also
* makes the transition read correctly — going from no verdict to amber is the
* panel learning something, not the group getting worse.
*/
const VERDICT_LED: Record<Verdict, LedVariant> = {
good: 'on',
partial: 'off',
degraded: 'amber',
down: 'crit',
unmeasured: 'off',
@@ -345,25 +406,27 @@ function fmtAge(seconds: number): string {
}
/**
* Failures first, then everything that was actually measured (fastest first, so
* the member carrying traffic is near the top), and the untested tail last.
* Answering members first, fastest at the top; then the never-measured; then the
* ones that didn't answer, at the very bottom.
*
* Untested ranks LAST rather than second on purpose: on a 376-node subscription
* it is the overwhelming majority and it carries no information, so ordering it
* above the measurements would bury every real reading under hundreds of rows
* that all say the same nothing.
* This is ordered by USEFULNESS, because the question that makes someone open the
* list is "which nodes here are any good" — and the fastest are the most
* interesting answer to it. Failures are the least useful rows on the page, and
* their COUNT is already on the state line above, so nobody has to scroll to find
* out how many there are.
*
* Untested sits between the two: not a finding, but not a ruled-out node either —
* any of them may turn out to be the fastest member once something measures it.
* Ranking it below the confirmed failures would put the group's live candidates
* underneath its dead ends.
*/
const MEMBER_RANK: Record<GroupMemberHealth['state'], number> = { dead: 0, alive: 1, untested: 2 }
const MEMBER_RANK: Record<GroupMemberHealth['state'], number> = { alive: 0, untested: 1, dead: 2 }
function sortMembers(members: GroupMemberHealth[]): GroupMemberHealth[] {
return [...members].sort(
(a, b) => MEMBER_RANK[a.state] - MEMBER_RANK[b.state] || a.delay_ms - b.delay_ms,
)
}
/** A member list this long is capped until the operator asks for all of it — a
* 376-member group is otherwise a page of scrolling nobody reads. */
const MEMBER_PAGE = 60
/** Nothing read yet — an empty, honest starting state (groups is never null). */
const IDLE_HEALTH: GroupsHealth = {
groups: [],
@@ -373,20 +436,32 @@ const IDLE_HEALTH: GroupsHealth = {
}
/** No test has run (or the page hasn't read one yet). */
const IDLE_TEST: GroupTestStatus = { running: false, done: 0, total: 0, results: [] }
const IDLE_TEST: GroupTestStatus = { running: false, done: 0, total: 0, scope: [], results: [] }
/**
* Harden a test payload against a daemon that doesn't hold up its end. The
* contract says `results` is always an array, but the panel treats a null there
* as "no results" rather than crashing the whole page on someone else's bug.
* contract says `scope` and `results` are always arrays, but the panel treats a
* null there as "empty" rather than crashing the whole page on someone else's bug.
*/
const normalizeTest = (st: GroupTestStatus): GroupTestStatus => ({
running: !!st?.running,
done: st?.done ?? 0,
total: st?.total ?? 0,
scope: asArray(st?.scope),
results: asArray(st?.results),
})
/**
* How the header names the reach of a running group test. The name matters more
* than the number when there is only one: "auto" tells the operator which button
* they pressed; "1 group" tells them nothing they didn't already know.
*/
function scopeLabel(scope: string[], groupCount: number): string {
if (scope.length === 1) return scope[0]
if (scope.length === 0) return 'group exits' // pre-scope daemon — say nothing false
return scope.length >= groupCount ? 'every group' : `${scope.length} groups`
}
/** Which editor (add or edit-by-name) is open within a section. */
type Editor = { mode: 'add' } | { mode: 'edit'; name: string }
@@ -442,6 +517,12 @@ export default function Targets() {
toastTimer.current = window.setTimeout(() => setToast(null), 2600)
}, [])
// Our group-member health sweep + every health/testing control on this page. A
// group still picks a live node without it (sing-box probes internally); this
// switch (Settings → Health check) governs only OUR sweep and the stats it
// feeds. Absent ⇒ on. When off we stop polling and hide all of the health UI.
const groupHealthOn = config?.Globals?.GroupHealth !== false
// ---- per-group membership health -------------------------------------------
// The SUMMARY view only (counts, no member rows): a few hundred bytes even for
// a 376-member subscription group, and a pure read on the daemon side, so it is
@@ -456,6 +537,8 @@ export default function Targets() {
const [healthRead, setHealthRead] = useState(false)
useEffect(() => {
// Health checks disabled ⇒ nothing to read, and the whole health UI is hidden.
if (!groupHealthOn) return
let alive = true
const tick = async () => {
// Nothing to poll for while the tab is in the background.
@@ -481,7 +564,7 @@ export default function Targets() {
window.clearInterval(id)
document.removeEventListener('visibilitychange', onVisible)
}
}, [])
}, [groupHealthOn])
const healthByGroup = useMemo(
() => new Map(health.groups.map((h) => [h.group, h])),
@@ -539,6 +622,7 @@ export default function Targets() {
// One read on mount, so a run started elsewhere (another tab, the CLI) shows its
// progress instead of an idle page.
useEffect(() => {
if (!groupHealthOn) return
let alive = true
void readTest().then((st) => {
if (alive && st?.running) setPolling(true)
@@ -546,10 +630,10 @@ export default function Targets() {
return () => {
alive = false
}
}, [readTest])
}, [readTest, groupHealthOn])
useEffect(() => {
if (!polling) return
if (!polling || !groupHealthOn) return
let alive = true
const id = window.setInterval(() => {
void readTest().then((st) => {
@@ -562,7 +646,7 @@ export default function Targets() {
alive = false
window.clearInterval(id)
}
}, [polling, readTest, flash])
}, [polling, readTest, flash, groupHealthOn])
const runTest = useCallback(
async (name = '') => {
@@ -591,6 +675,17 @@ export default function Targets() {
[gtest],
)
/**
* The groups the current run covers. This — not `running` — is what puts the
* in-progress badge on a card: `running` alone only says a test is happening
* SOMEWHERE, which is why testing one group used to light up all four.
*
* A daemon too old to send a scope sends nothing, and an empty set means no card
* claims the run. That degrades to a quiet header-only indicator rather than to
* the wrong badge everywhere, which is the correct way round.
*/
const testScope = useMemo(() => new Set(asArray(gtest.scope)), [gtest])
const [dirty, setDirty] = useState(false) // saved-but-not-yet-applied
const [saving, setSaving] = useState(false)
const [applying, setApplying] = useState(false)
@@ -805,23 +900,63 @@ export default function Targets() {
<header className="tg-sec-hd">
<h2 className="tg-sec-title">Groups</h2>
<span className="tg-sec-count mono">{groups.length} configured</span>
{/* Two runs live here and they are NOT the same action. A health
refresh measures every node in every group at once — it is global by
construction, so it gets exactly one indicator, here, and never a
badge on a card. A group exit test is scoped to the groups it names,
so its progress says WHICH, and its badge lands only on those cards. */}
<div className="tg-sec-ctl">
{groupHealthOn && (
<>
{health.node_test_running && (
<span className="tg-testprog mono" role="status">
<span
className="tg-run tg-run--health"
role="status"
title="One health run measures every member of every group, including each group’s own egress copies. It covers all groups at once, so it is reported here and not on any single card."
>
<Led variant="amber" pulse />
measuring {health.node_test_done}/{health.node_test_total}
<span className="tg-run-what">health · all groups</span>
<span className="tg-run-n mono">
{health.node_test_done}/{health.node_test_total}
</span>
</span>
)}
{gtest.running && (
<span className="tg-testprog mono" role="status">
<span
className="tg-run tg-run--exit"
role="status"
title="An exit test sends one connection through each group it covers and reports the delay and the address the internet sees."
>
<Led variant="amber" pulse />
testing {gtest.done}/{gtest.total}
<span className="tg-run-what">
exit test · {scopeLabel(asArray(gtest.scope), groups.length)}
</span>
<span className="tg-run-n mono">
{gtest.done}/{gtest.total}
</span>
</span>
)}
{!health.node_test_running && health.sweep?.enabled && health.sweep.total > 0 && (
<span
className="tg-run tg-run--sweep"
title={
health.sweep.cycles === 0
? 'The router re-probes nodes in the background so these numbers stay fresh. It is still on its first pass, so members it hasn’t reached yet read “not measured”.'
: `The router re-probes nodes in the background so these numbers stay fresh. It has completed ${health.sweep.cycles} full pass${health.sweep.cycles === 1 ? '' : 'es'}, so anything still unmeasured lost a reading it used to have.`
}
>
<span className="tg-run-what">
sweep{health.sweep.cycles === 0 ? ' · first pass' : ''}
</span>
<span className="tg-run-n mono">
{health.sweep.cursor}/{health.sweep.total}
</span>
</span>
)}
<Button
onClick={() => void measureAll()}
disabled={measuring || health.node_test_running || groups.length === 0}
title="Probe every member of every group now, including each group’s own egress copies. Health also refreshes on its own in the background."
title="Probe every member of every group now, including each group’s own egress copies. One run, all groups — health also refreshes on its own in the background."
>
{health.node_test_running ? 'Measuring…' : 'Refresh health'}
</Button>
@@ -830,8 +965,10 @@ export default function Targets() {
disabled={busy || !config || groups.length === 0 || gtest.running}
title="Send one connection through each group and report the delay and the exit address the internet sees"
>
{gtest.running ? 'Testing…' : 'Test exits'}
{gtest.running ? 'Testing…' : 'Test every exit'}
</Button>
</>
)}
<Button
variant="primary"
onClick={() => setGroupEd({ mode: 'add' })}
@@ -849,14 +986,14 @@ export default function Targets() {
in one group and dead in another.
</p>
{healthErr && (
{groupHealthOn && healthErr && (
<p className="tg-test-err" role="alert">
Couldn’t read member health — {healthErr}. The numbers below are the last ones that
arrived.
</p>
)}
{gtestErr && (
{groupHealthOn && gtestErr && (
<p className="tg-test-err" role="alert">
Couldn’t read the test results — {gtestErr}.{' '}
<button className="linkish" onClick={() => void readTest()}>
@@ -919,13 +1056,17 @@ export default function Targets() {
key={g.Name}
group={g}
busy={busy}
showHealth={groupHealthOn}
health={healthByGroup.get(g.Name)}
healthKnown={healthRead}
measuring={health.node_test_running}
sweeping={health.sweep?.enabled ?? false}
sweep={health.sweep}
onMeasure={() => void measureAll()}
test={testByGroup.get(g.Name)}
testing={gtest.running}
// The badge is this card's business only when the run names it.
testing={gtest.running && testScope.has(g.Name)}
// …but the daemon runs one test at a time, so any run in flight
// is what disables the button.
testBusy={gtest.running}
onTest={() => void runTest(g.Name)}
onEdit={() => setGroupEd({ mode: 'edit', name: g.Name })}
onDelete={() => removeGroup(g.Name)}
@@ -1100,33 +1241,45 @@ export default function Targets() {
function GroupRow({
group,
busy,
showHealth,
health,
healthKnown,
measuring,
sweeping,
sweep,
onMeasure,
test,
testing,
testBusy,
onTest,
onEdit,
onDelete,
}: {
group: Group
busy: boolean
/** Group health checks are on (Settings). When false, the card drops its health
* readout, its exit-test readout and its Test button — it is config only. */
showHealth: boolean
/** This group's membership health, or undefined when the engine hasn't built
* it (not applied yet, or dropped for having no usable members). */
health?: GroupHealth
/** The health endpoint has answered at least once, so an absent entry really
* does mean "the engine doesn't have this group". */
healthKnown: boolean
/** A probe-all run is in flight — the numbers below are moving. */
measuring: boolean
/** The daemon's background sweep is on, so untested members resolve by
* themselves. Without it, "wait and it will fill in" would be a false promise. */
sweeping: boolean
/** The daemon's background sweep, so the card can say whether untested members
* resolve by themselves. Without it, "wait and it will fill in" is a false
* promise. Absent on daemons that don't report it. */
sweep?: HealthSweep
onMeasure: () => void
test?: GroupTestResult
/**
* A group exit test covering THIS group is in flight.
*
* Deliberately not "a test is running": the caller resolves it against the run's
* scope. There is no per-card equivalent for the health run — that one measures
* every group at once and is reported once, in the section header.
*/
testing: boolean
/** Any exit test is in flight; the daemon runs one at a time. */
testBusy: boolean
onTest: () => void
onEdit: () => void
onDelete: () => void
@@ -1177,16 +1330,19 @@ function GroupRow({
<div className="tg-row-l2 mono">
<span className="tg-row-detail">{detail}</span>
</div>
<GroupHealthReadout
name={group.Name}
egress={group.Egress ?? ''}
health={health}
healthKnown={healthKnown}
measuring={measuring}
sweeping={sweeping}
onMeasure={onMeasure}
/>
<GroupTestReadout test={test} pending={testing && !test} />
{showHealth && (
<>
<GroupHealthReadout
name={group.Name}
egress={group.Egress ?? ''}
health={health}
healthKnown={healthKnown}
sweep={sweep}
onMeasure={onMeasure}
/>
<GroupTestReadout test={test} pending={testing && !test} />
</>
)}
</div>
<RowActions
onEdit={onEdit}
@@ -1194,9 +1350,9 @@ function GroupRow({
busy={busy}
editLabel={`Edit group ${group.Name}`}
deleteLabel={`Delete group ${group.Name}`}
onTest={onTest}
testLabel={`Test group ${group.Name}`}
testDisabled={testing}
onTest={showHealth ? onTest : undefined}
testLabel={showHealth ? `Test the exit of group ${group.Name}` : undefined}
testDisabled={testBusy}
/>
</li>
)
@@ -1220,16 +1376,14 @@ function GroupHealthReadout({
egress,
health,
healthKnown,
measuring,
sweeping,
sweep,
onMeasure,
}: {
name: string
egress: string
health?: GroupHealth
healthKnown: boolean
measuring: boolean
sweeping: boolean
sweep?: HealthSweep
onMeasure: () => void
}) {
const [open, setOpen] = useState(false)
@@ -1256,7 +1410,10 @@ function GroupHealthReadout({
)
}
const v = verdictOf(health)
// A completed sweep pass is what makes silence meaningful; until then a group's
// own history can only have recorded successes. See isBiasedSample.
const sweptOnce = (sweep?.cycles ?? 0) >= 1
const v = verdictOf(health, sweptOnce)
const { total, tested, alive, dead, untested } = health
const boundTitle = `Every member of “${name}” is a private copy dialled through ${
egress ? `egress “${egress}”` : 'this group’s egress'
@@ -1271,9 +1428,11 @@ function GroupHealthReadout({
aria-label={
tested === 0
? `No members measured of ${total}`
: `${alive} of ${tested} measured members answering, ${dead} not answering${
untested > 0 ? `, ${untested} of ${total} not measured` : ''
}`
: v === 'partial'
? `${alive} of ${total} members confirmed answering, ${untested} not checked yet; no failures confirmed yet, so this is not a proportion`
: `${alive} of ${tested} measured members answering, ${dead} not answering${
untested > 0 ? `, ${untested} of ${total} not measured` : ''
}`
}
>
{alive > 0 && <span className="gh-seg gh-seg--alive" style={{ flexGrow: alive }} />}
@@ -1285,7 +1444,7 @@ function GroupHealthReadout({
)}
<div className="gh-line">
<Led variant={VERDICT_LED[v]} pulse={measuring} label={`${name} health`} />
<Led variant={VERDICT_LED[v]} label={`${name} health`} />
{v === 'empty' ? (
<span className="gh-quiet">no members</span>
@@ -1296,12 +1455,37 @@ function GroupHealthReadout({
{total} member{total === 1 ? '' : 's'}
</span>
</>
) : v === 'partial' ? (
/* No fraction here, on purpose. `alive / tested` would put a denominator
on a sample that nothing has yet been able to fail a member into, and
it always reads 100%. Two plain counts instead: what is confirmed, and
what is still open. When the sweep fills the gap this becomes a real
ratio — which reads as the panel getting more precise, not as the
group getting worse. */
<>
<span className="gh-headline mono">
<b>{alive}</b>
</span>
<span className="gh-word">confirmed answering</span>
<span className="gh-open mono" title={`${untested} of ${total} members have no measurement — some of them may be down`}>
{untested} not checked yet
</span>
</>
) : (
<>
<span className="gh-headline mono">
<b>{alive}</b> / {tested}
</span>
<span className="gh-word">alive</span>
{/* The failures, counted where the counts already are. This number used
to live only in the disclosure label — which then promised a list of
just those members and opened the whole thing instead. It belongs on
the state line: it is a state, not a filter. */}
{dead > 0 && (
<span className="gh-dead mono" title={`${dead} member${dead === 1 ? '' : 's'} were probed and did not answer`}>
{dead} not answering
</span>
)}
{/* The quiet remainder — present only when there IS one. */}
{untested > 0 && (
<span className="gh-rest mono">
@@ -1328,7 +1512,35 @@ function GroupHealthReadout({
)}
</div>
{/* The two states that need a sentence rather than a number. */}
{/* The three states that need a sentence rather than a number. */}
{/* Why there is no ratio. It names the mechanism, because "we're not sure"
without a reason reads as hedging — and because the mechanism is also the
answer to "when will I know": either the sweep gets there, or you ask. */}
{v === 'partial' && (
<p className="gh-say">
Not a proportion yet — a group records only the members that answer, so any that failed
are still counted as unchecked.{' '}
{sweep?.enabled ? (
<>
The background sweep is the only thing that confirms a member is down, and it hasn’t
finished its first pass over this group. It gets there on its own, or{' '}
<button type="button" className="linkish" onClick={onMeasure}>
measure now
</button>{' '}
to settle it.
</>
) : (
<>
The background sweep is what confirms a member is down, and it is off — so nothing
will settle this until you{' '}
<button type="button" className="linkish" onClick={onMeasure}>
measure now
</button>
.
</>
)}
</p>
)}
{v === 'down' && (
<p className="gh-say gh-say--bad">
None of the {tested} measured member{tested === 1 ? '' : 's'} answered
@@ -1336,17 +1548,35 @@ function GroupHealthReadout({
nowhere to go.
</p>
)}
{/* Why nothing has been measured, and what will change that. The sweep's
`cycles` is what separates the two honest readings of the same word:
cycles 0 means the sweep simply hasn't arrived; cycles ≥ 1 means it HAS
been everywhere, so these members lost the readings they once had. */}
{v === 'unmeasured' && (
<p className="gh-say">
{measuring ? (
'Measuring now — the numbers fill in as each probe lands.'
Nothing has been probed through this group yet, so there is nothing to report — not a
fault.{' '}
{!sweep?.enabled ? (
<>
The background sweep is off, so this fills in only when you ask:{' '}
<button type="button" className="linkish" onClick={onMeasure}>
measure now
</button>
.
</>
) : sweep.cycles === 0 ? (
<>
The background sweep is still on its first pass and gets to these within a few
minutes, or{' '}
<button type="button" className="linkish" onClick={onMeasure}>
measure now
</button>
.
</>
) : (
<>
Nothing has been probed through this group yet, so there is nothing to report — not a
fault.{' '}
{sweeping
? 'The background sweep gets to it within a few minutes, or '
: 'Measure it to find out: '}
The background sweep has already been everywhere, so these lost the readings they
had — probe them to find out where they stand:{' '}
<button type="button" className="linkish" onClick={onMeasure}>
measure now
</button>
@@ -1365,13 +1595,12 @@ function GroupHealthReadout({
onClick={() => setOpen((o) => !o)}
>
<span className="gh-caret" aria-hidden="true" />
{/* The label states the payoff when there is one: someone opens this to
find out WHICH members are not answering, not to read a list. */}
{open
? 'Hide members'
: dead > 0
? `Show the ${dead} not answering`
: `${total} member${total === 1 ? '' : 's'}`}
{/* A constant "N members" in both states — the caret (which rotates on
aria-expanded) and aria-expanded carry open/closed, so the words stay
a plain count of what the panel holds rather than a verb that flips.
Opening shows all `total` rows, nothing held back behind a second
button. */}
{total} member{total === 1 ? '' : 's'}
</button>
)}
@@ -1395,13 +1624,16 @@ function GroupHealthReadout({
* subscription group, so it is never pulled until someone opens the card, and it
* re-reads only when the summary poll says the counters actually changed.
*
* Sorted worst-first, because the question that makes anyone open this is "which
* ones are not answering" — not "list them in engine order".
* Once open it renders the WHOLE membership — no page, no second button. The list
* lives in its own scroll box, so a 298-row group costs one scrollable panel
* rather than a page of scrolling; see .gh-mems.
*
* Sorted fastest-first (see sortMembers): the question that makes anyone open this
* is "which nodes here are any good", not "list them in engine order".
*/
function GroupMembers({ id, group, version }: { id: string; group: string; version: string }) {
const [rows, setRows] = useState<GroupMemberHealth[] | null>(null)
const [err, setErr] = useState<string | null>(null)
const [showAll, setShowAll] = useState(false)
useEffect(() => {
let alive = true
@@ -1420,7 +1652,6 @@ function GroupMembers({ id, group, version }: { id: string; group: string; versi
}, [group, version])
const sorted = useMemo(() => (rows ? sortMembers(rows) : []), [rows])
const shown = showAll ? sorted : sorted.slice(0, MEMBER_PAGE)
if (err) {
return (
@@ -1448,16 +1679,19 @@ function GroupMembers({ id, group, version }: { id: string; group: string; versi
return (
<>
{/* The order, stated where it can be checked against the rows underneath —
never in the disclosure label, which must not promise a subset of a list
that is shown whole. */}
{sorted.length > 1 && (
<p className="gh-mems-order">
Fastest first, then the never-measured, then the ones that didn’t answer.
</p>
)}
<ul id={id} className="gh-mems">
{shown.map((m) => (
{sorted.map((m) => (
<MemberRow key={m.tag} member={m} />
))}
</ul>
{sorted.length > shown.length && (
<button type="button" className="gh-more gh-more--all" onClick={() => setShowAll(true)}>
Show all {sorted.length} members
</button>
)}
</>
)
}
@@ -1498,10 +1732,13 @@ function MemberRow({ member }: { member: GroupMemberHealth }) {
*/
function GroupTestReadout({ test, pending }: { test?: GroupTestResult; pending: boolean }) {
if (pending) {
// "testing", never "measuring": the health run owns that word and covers every
// group at once. Two runs that read the same on a card is how one group's test
// came to look like all four were busy.
return (
<div className="tg-test tg-test--wait" role="status">
<Led variant="amber" pulse />
<span className="tg-test-msg">measuring…</span>
<span className="tg-test-msg">testing this exit…</span>
</div>
)
}
+31 -5
View File
@@ -672,10 +672,22 @@ func (g *URLTestGroup) balancePoolFirstLive(ctx context.Context, size int) map[s
}
}
}
g.balancer.setSlots(next)
// result holds exactly the tags that tested live this round (pool re-test + hole fills);
// setSlots marks every other slot dead so pick() skips it. lx: SPEC 019 v2.
g.balancer.setSlots(next, liveSet(result))
return result
}
// liveSet turns a tag->delay result map (only live nodes are present) into the tag->live set
// setSlots consumes. lx: SPEC 019 v2 — pick() routes only through live slots.
func liveSet(result map[string]uint16) map[string]bool {
live := make(map[string]bool, len(result))
for tag := range result {
live[tag] = true
}
return live
}
// balancePoolTolerant (pool_tolerance > 0): test all nodes, then pick the top-`size` by delay,
// replacing a pool member only when an outside node beats it by more than the tolerance.
func (g *URLTestGroup) balancePoolTolerant(ctx context.Context, size int) map[string]uint16 {
@@ -690,7 +702,7 @@ func (g *URLTestGroup) balancePoolTolerant(ctx context.Context, size int) map[st
}
}
next := planTolerantPool(g.balancer.poolTags(), results, size, g.balancer.poolTolerance)
g.balancer.setSlots(next)
g.balancer.setSlots(next, liveSet(result))
return result
}
@@ -739,10 +751,16 @@ func (g *URLTestGroup) rebuildPool() {
for i, c := range fillCandidates {
fillOrder[i] = c.tag
}
g.balancer.setSlots(planFirstLivePool(current, live, fillOrder, size))
g.balancer.setSlots(planFirstLivePool(current, live, fillOrder, size), live)
return
}
g.balancer.setSlots(planTolerantPool(current, results, size, g.balancer.poolTolerance))
tolerantLive := make(map[string]bool, len(results))
for tag, c := range results {
if c.alive {
tolerantLive[tag] = true
}
}
g.balancer.setSlots(planTolerantPool(current, results, size, g.balancer.poolTolerance), tolerantLive)
}
// seedPool fills the pool before the first health-check: prefer nodes with live history (the
@@ -782,7 +800,15 @@ func (g *URLTestGroup) seedPool() {
seen[tag] = true
}
}
g.balancer.setSlots(next)
// Seed marks every seeded slot LIVE optimistically: no health-check has run yet, so we have
// no failure evidence, and marking them dead would blackhole all traffic to the fallback until
// the first check completes (killing cold-start spread). The first CheckOutbounds (kicked off
// right after seedPool in PostStart) corrects any that are actually down within one interval.
live := make(map[string]bool, len(next))
for _, tag := range next {
live[tag] = true
}
g.balancer.setSlots(next, live)
}
// outboundsByTags resolves slot tags back to live outbound objects (skipping empties/unknowns).
+78 -22
View File
@@ -3,6 +3,7 @@ package group
import (
"context"
"hash/fnv"
"math/rand/v2"
"strconv"
"sync"
"sync/atomic"
@@ -28,16 +29,21 @@ import (
// The pool is lazily health-checked: we test no more nodes than needed to keep it full of
// live nodes. Selection runs once per new connection (DialContext/ListenPacket).
// slot is one fixed position in the pool. tag is the node currently occupying it.
// slot is one fixed position in the pool. tag is the node currently occupying it; live is
// whether the last health-check found that node reachable. pick() routes ONLY through live
// slots (a dead occupant keeps its slot for the SPEC never-shrink invariant, but is skipped
// by selection until a health-check revives or replaces it).
type slot struct {
tag string
tag string
live bool
}
// balancer holds the per-group round_robin state. nil for least_test.
// balancer holds the per-group balancing state (round_robin and random). nil for least_test.
type balancer struct {
poolSize int // target slot count (config pool, before min(pool,nodes))
poolTolerance uint16 // ms; 0 = first-live-fill, > 0 = top-N-by-delay with eviction threshold
stickyHash []string // sticky key components; empty → no stickiness (counter rotation)
random bool // mode == random: pick a uniformly-random live slot per connection
access sync.Mutex // guards slots
slots []slot // the pool; len == min(poolSize, available nodes), index = fixed slot number
@@ -56,9 +62,16 @@ func newBalancer(options option.URLTestOutboundOptions) (*balancer, error) {
if mode == "" || mode == C.URLTestModeLeastTest {
return nil, nil
}
if mode != C.URLTestModeRoundRobin {
if mode != C.URLTestModeRoundRobin && mode != C.URLTestModeRandom {
return nil, E.New("unknown urltest mode: ", mode)
}
isRandom := mode == C.URLTestModeRandom
// stickyDefault: round_robin binds a flow to a slot by default (session stability); random is
// a per-connection independent draw, so stickiness is meaningless there — default it OFF.
var stickyDefault []string
if !isRandom {
stickyDefault = []string{C.URLTestStickyProcess, C.URLTestStickyDomain}
}
bo := options.Balancer
pool := C.DefaultURLTestPool
var stickyHash []string
@@ -71,19 +84,19 @@ func newBalancer(options option.URLTestOutboundOptions) (*balancer, error) {
if bo.Pool > 0 {
pool = bo.Pool
}
// Omitted (nil) or empty → default. To DISABLE stickiness use the explicit ["none"]
// Omitted (nil) or empty → mode default. To DISABLE stickiness use the explicit ["none"]
// sentinel: the config decoder collapses a bare [] to nil (see URLTestStickyNone), so an
// empty list cannot mean "off". ["none"] → stickiness off (no components).
if len(bo.StickyHash) == 0 {
stickyHash = []string{C.URLTestStickyProcess, C.URLTestStickyDomain}
stickyHash = stickyDefault
} else if len(bo.StickyHash) == 1 && bo.StickyHash[0] == C.URLTestStickyNone {
stickyHash = nil // sticky off → pure counter rotation
} else {
stickyHash = bo.StickyHash
}
} else {
// round_robin without balancer → defaults, stickiness on by default.
stickyHash = []string{C.URLTestStickyProcess, C.URLTestStickyDomain}
// no balancer block → mode defaults (round_robin: sticky on; random: sticky off).
stickyHash = stickyDefault
}
for _, component := range stickyHash {
switch component {
@@ -101,11 +114,13 @@ func newBalancer(options option.URLTestOutboundOptions) (*balancer, error) {
if bo != nil {
poolTolerance = bo.PoolTolerance
}
return &balancer{poolSize: pool, poolTolerance: poolTolerance, stickyHash: stickyHash}, nil
return &balancer{poolSize: pool, poolTolerance: poolTolerance, stickyHash: stickyHash, random: isRandom}, nil
}
// pick selects one node for this connection from the current pool. fallback is returned
// when the pool is empty (start, before the first health-check fills it).
// pick selects one node for this connection from the current pool. Selection runs ONLY over
// LIVE slots (health-checked reachable) — a dead slot is never returned while any live slot
// exists, for every balancing mode. fallback is returned when the pool is empty (start, before
// the first health-check fills it) or when every slot is currently dead.
func (b *balancer) pick(ctx context.Context, destination M.Socksaddr, fallback adapter.Outbound, resolve func(tag string) adapter.Outbound) adapter.Outbound {
b.access.Lock()
n := len(b.slots)
@@ -113,15 +128,39 @@ func (b *balancer) pick(ctx context.Context, destination M.Socksaddr, fallback a
b.access.Unlock()
return fallback
}
liveCount := 0
for i := range b.slots {
if b.slots[i].live {
liveCount++
}
}
var tag string
if len(b.stickyHash) > 0 {
// slot-hash: fixed slot index from the key; living node in its slot keeps its keys.
idx := int(hashKey(b.stickyKey(ctx, destination)) % uint64(n))
tag = b.slots[idx].tag
} else {
// plain round-robin over the fixed slots.
idx := int(b.counter.Add(1)-1) % n
tag = b.slots[idx].tag
switch {
case liveCount == 0:
// No live slot: nothing safe to route to. fallback (the group's Select) covers it —
// which itself only returns a node with a working history, else config order.
b.access.Unlock()
return fallback
case b.random:
// Uniform random draw over the LIVE slots. rand.IntN (math/rand/v2) is safe for
// concurrent use (per-P generator state, no shared global lock) and auto-seeded, so no
// mutex or manual seeding is needed — the lock here only guards the slots slice.
tag = b.nthLiveTag(rand.IntN(liveCount))
case len(b.stickyHash) > 0:
// slot-hash: fixed slot index from the key over ALL slots, so a living occupant keeps its
// keys (the strict-zero-reconnect invariant). If that slot is dead, degrade to a LIVE slot
// picked deterministically from the same key — the flow still lands somewhere working.
h := hashKey(b.stickyKey(ctx, destination))
idx := int(h % uint64(n))
if b.slots[idx].live {
tag = b.slots[idx].tag
} else {
tag = b.nthLiveTag(int(h % uint64(liveCount)))
}
default:
// plain round-robin over the LIVE slots only.
idx := int(b.counter.Add(1)-1) % liveCount
tag = b.nthLiveTag(idx)
}
b.access.Unlock()
if node := resolve(tag); node != nil {
@@ -130,6 +169,21 @@ func (b *balancer) pick(ctx context.Context, destination M.Socksaddr, fallback a
return fallback
}
// nthLiveTag returns the tag of the k-th live slot (0-based, in slot order). Caller must hold
// access and pass k in [0, liveCount). Walks without allocating (pools may be large for random).
func (b *balancer) nthLiveTag(k int) string {
for i := range b.slots {
if !b.slots[i].live {
continue
}
if k == 0 {
return b.slots[i].tag
}
k--
}
return ""
}
// poolTags returns the current slot tags (snapshot under lock). Used by Pool()/GetPool.
func (b *balancer) poolTags() []string {
b.access.Lock()
@@ -142,12 +196,14 @@ func (b *balancer) poolTags() []string {
}
// setSlots replaces the pool atomically. The health-check (urltest.go) computes the new
// occupancy — which tag sits in which slot — and hands the ordered tag list here.
func (b *balancer) setSlots(tags []string) {
// occupancy — which tag sits in which slot — and hands the ordered tag list here, plus the set
// of tags found LIVE this round. A slot whose tag is absent from live is retained (never-shrink)
// but marked dead, so pick() skips it until it is revived or replaced.
func (b *balancer) setSlots(tags []string, live map[string]bool) {
b.access.Lock()
b.slots = make([]slot, len(tags))
for i, tag := range tags {
b.slots[i] = slot{tag: tag}
b.slots[i] = slot{tag: tag, live: tag != "" && live[tag]}
}
b.access.Unlock()
// lx: SPEC 020 — the active pool changed; invalidate the reachable cache.
+152 -6
View File
@@ -48,6 +48,32 @@ func rrBalancer(t *testing.T, pool int, stickyHash []string) *balancer {
func destDomain(host string) M.Socksaddr { return M.Socksaddr{Fqdn: host} }
// allLive marks every given tag live — the common case for pick() tests that only exercise
// selection over a fully-healthy pool.
func allLive(tags ...string) map[string]bool {
m := make(map[string]bool, len(tags))
for _, t := range tags {
m[t] = true
}
return m
}
// randomBalancer builds a mode=random balancer with the given pool size (stickiness off).
func randomBalancer(t *testing.T, pool int) *balancer {
t.Helper()
b, err := newBalancer(option.URLTestOutboundOptions{
Mode: C.URLTestModeRandom,
Balancer: &option.URLTestBalancerOptions{Pool: pool},
})
if err != nil {
t.Fatalf("newBalancer(random): %v", err)
}
if b == nil {
t.Fatal("random balancer must not be nil")
}
return b
}
// --- newBalancer: modes, defaults, validation ---------------------------------------
func TestBalancerLeastTestIsNil(t *testing.T) {
@@ -162,7 +188,7 @@ func TestBalancerUnknownStickyComponentRejected(t *testing.T) {
func TestRoundRobinRotation(t *testing.T) {
b := rrBalancer(t, 3, []string{C.URLTestStickyNone}) // no stickiness → pure counter rotation
b.setSlots([]string{"a", "b", "c"})
b.setSlots([]string{"a", "b", "c"}, allLive("a", "b", "c"))
_, resolve := resolveFrom("a", "b", "c")
fb := &balNode{tag: "fb"}
const rounds = 3000
@@ -189,7 +215,7 @@ func TestPickEmptyPoolFallback(t *testing.T) {
func TestPickUnresolvableTagFallsBack(t *testing.T) {
b := rrBalancer(t, 1, []string{})
b.setSlots([]string{"gone"})
b.setSlots([]string{"gone"}, allLive("gone"))
_, resolve := resolveFrom() // resolves nothing
fb := &balNode{tag: "fb"}
if got := b.pick(context.Background(), M.Socksaddr{}, fb, resolve); got.Tag() != "fb" {
@@ -201,7 +227,7 @@ func TestPickUnresolvableTagFallsBack(t *testing.T) {
func TestStickySlotHashStable(t *testing.T) {
b := rrBalancer(t, 4, []string{C.URLTestStickyDomain})
b.setSlots([]string{"a", "b", "c", "d"})
b.setSlots([]string{"a", "b", "c", "d"}, allLive("a", "b", "c", "d"))
_, resolve := resolveFrom("a", "b", "c", "d")
first := b.pick(context.Background(), destDomain("example.com"), nil, resolve).Tag()
for i := 0; i < 100; i++ {
@@ -216,7 +242,7 @@ func TestStickySlotHashLivingNodeKeepsKeysAcrossOtherSlotChanges(t *testing.T) {
// The strict-zero-reconnect invariant: replacing the occupant of OTHER slots must not
// move a key whose slot occupant is unchanged.
b := rrBalancer(t, 4, []string{C.URLTestStickyDomain})
b.setSlots([]string{"a", "b", "c", "d"})
b.setSlots([]string{"a", "b", "c", "d"}, allLive("a", "b", "c", "d"))
_, resolve := resolveFrom("a", "b", "c", "d", "x", "y")
dst := destDomain("keep.me")
pinnedTag := b.pick(context.Background(), dst, nil, resolve).Tag()
@@ -229,7 +255,7 @@ func TestStickySlotHashLivingNodeKeepsKeysAcrossOtherSlotChanges(t *testing.T) {
newSlots[i] = []string{"x", "y", "x", "y"}[i]
}
}
b.setSlots(newSlots)
b.setSlots(newSlots, allLive(newSlots...))
_, resolve2 := resolveFrom(append(newSlots, pinnedTag)...)
if got := b.pick(context.Background(), dst, nil, resolve2).Tag(); got != pinnedTag {
t.Fatalf("living node in its slot must keep its key: %s != %s", got, pinnedTag)
@@ -239,7 +265,7 @@ func TestStickySlotHashLivingNodeKeepsKeysAcrossOtherSlotChanges(t *testing.T) {
func TestStickyEmptyKeyFixedSlot(t *testing.T) {
// All components empty (no domain) → key "" → one fixed slot, no rotation.
b := rrBalancer(t, 3, []string{C.URLTestStickyDomain})
b.setSlots([]string{"a", "b", "c"})
b.setSlots([]string{"a", "b", "c"}, allLive("a", "b", "c"))
_, resolve := resolveFrom("a", "b", "c")
first := b.pick(context.Background(), M.Socksaddr{}, nil, resolve).Tag()
for i := 0; i < 50; i++ {
@@ -476,3 +502,123 @@ func TestPlanFirstLivePoolGrowsKeepingExisting(t *testing.T) {
}
}
}
// --- random mode: config, uniform spread, live-only selection -----------------------
func TestRandomModeDefaultsStickyOff(t *testing.T) {
// mode=random with no balancer block → stickiness OFF (nil), random flag set.
b, err := newBalancer(option.URLTestOutboundOptions{Mode: C.URLTestModeRandom})
if err != nil {
t.Fatal(err)
}
if b == nil {
t.Fatal("random balancer must not be nil")
}
if !b.random {
t.Fatal("random flag must be set for mode=random")
}
if len(b.stickyHash) != 0 {
t.Fatalf("random mode must default stickiness off, got %v", b.stickyHash)
}
if b.poolSize != C.DefaultURLTestPool {
t.Fatalf("random default pool = %d, want %d", b.poolSize, C.DefaultURLTestPool)
}
}
// (а) random draws spread across MORE THAN ONE live node — the whole point of the mode.
func TestRandomSpreadsAcrossLiveNodes(t *testing.T) {
b := randomBalancer(t, 3)
b.setSlots([]string{"a", "b", "c"}, allLive("a", "b", "c"))
_, resolve := resolveFrom("a", "b", "c")
fb := &balNode{tag: "fb"}
const rounds = 3000
count := map[string]int{}
for i := 0; i < rounds; i++ {
count[b.pick(context.Background(), M.Socksaddr{}, fb, resolve).Tag()]++
}
for _, tag := range []string{"a", "b", "c"} {
// Uniform over 3 with 3000 draws → ~1000 each; a huge margin proves "more than one".
if count[tag] < rounds/10 {
t.Errorf("random must spread across live nodes; %s got only %d of %d", tag, count[tag], rounds)
}
}
if count["fb"] != 0 {
t.Errorf("fallback must never be used while live nodes exist, got %d", count["fb"])
}
}
// (б) pick must NEVER return a dead slot while a live one exists — for random AND round_robin.
func TestRandomNeverPicksDeadSlot(t *testing.T) {
b := randomBalancer(t, 3)
// b is dead this round; a and c are live.
b.setSlots([]string{"a", "b", "c"}, map[string]bool{"a": true, "c": true})
_, resolve := resolveFrom("a", "b", "c")
fb := &balNode{tag: "fb"}
for i := 0; i < 1000; i++ {
got := b.pick(context.Background(), M.Socksaddr{}, fb, resolve).Tag()
if got == "b" {
t.Fatal("random picked the dead slot b")
}
if got == "fb" {
t.Fatal("random fell back to fallback despite live slots")
}
}
}
func TestRoundRobinNeverPicksDeadSlot(t *testing.T) {
b := rrBalancer(t, 3, []string{C.URLTestStickyNone}) // counter rotation
b.setSlots([]string{"a", "b", "c"}, map[string]bool{"a": true, "c": true})
_, resolve := resolveFrom("a", "b", "c")
fb := &balNode{tag: "fb"}
count := map[string]int{}
for i := 0; i < 600; i++ {
got := b.pick(context.Background(), M.Socksaddr{}, fb, resolve).Tag()
if got == "b" {
t.Fatal("round_robin picked the dead slot b")
}
count[got]++
}
// Rotation over the two live slots must alternate evenly between a and c.
if count["a"] != 300 || count["c"] != 300 {
t.Fatalf("round_robin over live slots must split evenly a/c, got %v", count)
}
}
// (в) every slot dead → fallback (no live slot to route through).
func TestPickAllDeadFallsBack(t *testing.T) {
for _, name := range []string{"round_robin", "random"} {
var b *balancer
if name == "random" {
b = randomBalancer(t, 2)
} else {
b = rrBalancer(t, 2, []string{C.URLTestStickyNone})
}
b.setSlots([]string{"a", "b"}, map[string]bool{}) // nothing live
_, resolve := resolveFrom("a", "b")
fb := &balNode{tag: "fb"}
if got := b.pick(context.Background(), M.Socksaddr{}, fb, resolve).Tag(); got != "fb" {
t.Fatalf("%s: all-dead pool must return fallback, got %s", name, got)
}
}
}
// sticky flow whose hashed slot is dead must degrade to a LIVE slot, not to the fallback and not
// to the dead occupant.
func TestStickyDeadSlotDegradesToLive(t *testing.T) {
b := rrBalancer(t, 3, []string{C.URLTestStickyDomain})
tags := []string{"a", "b", "c"}
dst := destDomain("example.com")
deadSlot := int(hashKey("example.com") % 3)
live := allLive(tags...)
delete(live, tags[deadSlot]) // kill exactly the slot this flow hashes to
b.setSlots(tags, live)
_, resolve := resolveFrom(tags...)
fb := &balNode{tag: "fb"}
got := b.pick(context.Background(), dst, fb, resolve).Tag()
if got == tags[deadSlot] {
t.Fatalf("sticky flow must not land on its dead slot %s", tags[deadSlot])
}
if got == "fb" {
t.Fatal("sticky flow must degrade to a live slot, not the fallback")
}
}
+38 -14
View File
@@ -211,18 +211,30 @@ func groupProbeSpecs(m *model.Model) map[string]engine.GroupProbeSpec {
// configureSweep (re)installs the engine's scheduled health sweep from the model. It is
// called on every SUCCESSFUL apply, which is exactly when the facts it depends on — the
// group set, their egress bindings and their probe URLs — can have changed.
// group set, their egress bindings, their probe URLs and the schedule itself — can have
// changed.
//
// The sweep is ON by default and has no UCI knob yet: shipping it disabled would leave
// the problem it exists for (a 376-node subscription reading almost entirely "untested",
// because urltest groups probe only while in use and selector groups never probe at all)
// exactly as it was. See engine/sweep.go for the cost model.
// The schedule comes from Globals.SweepInterval via model.SweepSchedule, the same
// function ValidateGlobals warns from, so what the operator is told and what the engine
// does cannot diverge. An UNSET interval means ON at the engine's default tick: urltest
// groups probe only while in use and selector groups never probe at all, so with the
// sweep off a 376-node subscription reads almost entirely "untested". A zero Interval is
// passed through deliberately — it tells engine.SweepConfig.withDefaults to apply its
// own default, keeping that number in one place. See engine/sweep.go for the cost model.
func (a *Applier) configureSweep(m *model.Model) {
if a.eng == nil {
return
}
interval, enabled, _ := m.Globals.SweepSchedule()
// GroupHealth is the operator's master-switch for our sweep. When off, disable it
// exactly like SweepInterval="off" — no warning, this is an explicit choice — while
// PRESERVING the interval (so flipping the toggle back on restores the schedule).
if !m.Globals.GroupHealth {
enabled = false
}
a.eng.ConfigureSweep(engine.SweepConfig{
Enabled: true,
Enabled: enabled,
Interval: interval,
Specs: groupProbeSpecs(m),
FallbackURL: strings.TrimSpace(m.Globals.ProbeURL),
})
@@ -248,12 +260,12 @@ func (a *Applier) TestGroups(names []string, probeURL string) (started bool) {
return a.eng.TestGroups(names, probeURL)
}
// GroupTestStatus reports the engine's group-test progress and results. A nil engine
// reads as idle with an empty (never nil) result list, so the panel can render it
// unconditionally.
func (a *Applier) GroupTestStatus() (running bool, done, total int, results []engine.GroupTestResult) {
// GroupTestStatus reports the engine's group-test progress, the set of groups the run
// covers, and its results. A nil engine reads as idle with empty (never nil) slices, so
// the panel can render it unconditionally.
func (a *Applier) GroupTestStatus() (running bool, done, total int, scope []string, results []engine.GroupTestResult) {
if a.eng == nil {
return false, 0, 0, []engine.GroupTestResult{}
return false, 0, 0, []string{}, []engine.GroupTestResult{}
}
return a.eng.GroupTestStatus()
}
@@ -454,9 +466,17 @@ func (a *Applier) applyLocked(m *model.Model) (bool, error) {
// `local default dev lo` table. Harmless on a real change (the plane is being
// rebuilt anyway), but pointless churn on a no-op reconcile, so skip it when
// the ruleset is unchanged AND the rule is verifiably still installed.
// routeWarnings carries an egress that was BUILT but cannot route (no nexthop on a
// non-point-to-point device). It is deliberately part of the status warning set:
// such an egress looks applied everywhere in the UI while being unable to reach
// anything off its own subnet. On the fast path nothing was rebuilt, so there is
// nothing new to report and the previous set stands.
var routeWarnings []string
if !nftCurrent || !netplane.RoutingPresent(m.Globals) {
if err := netplane.ApplyRouting(m); err != nil {
return changed, err
var rerr error
routeWarnings, rerr = netplane.ApplyRoutingWithWarnings(m)
if rerr != nil {
return changed, rerr
}
}
if err := netplane.ApplySysctl(); err != nil {
@@ -486,7 +506,11 @@ func (a *Applier) applyLocked(m *model.Model) (bool, error) {
// Publish the warnings of THIS successful apply, in one normalised set, and log
// them in one consistent format. Status carries them to the panel so a
// fail-open degradation is visible in the UI instead of only in logread.
ws := collectWarnings(m.Globals, warnings, nftWarnings, configWarnings, untunPlan.Notes()...)
// routeWarnings joins the netplane channel, which is graded CRITICAL wholesale —
// correctly so here: an egress that cannot reach off its own subnet is a configured
// path that silently carries nothing, exactly the class of fault that channel exists
// for.
ws := collectWarnings(m.Globals, warnings, append(nftWarnings, routeWarnings...), configWarnings, untunPlan.Notes()...)
a.setWarnings(ws)
// The log only hears about a CHANGE. Status above always carries the full set;
// reprinting it on every no-op reconcile (cron, once a minute, plus every
+75
View File
@@ -2,7 +2,9 @@ package apply
import (
"testing"
"time"
"github.com/sagernet/sing-box/shater/engine"
"github.com/sagernet/sing-box/shater/model"
)
@@ -56,3 +58,76 @@ func TestGroupProbeSpecsNoOpinion(t *testing.T) {
t.Fatal("a model with no groups must yield nil specs")
}
}
// configureSweep must honour Globals.SweepInterval end to end — the knob is worthless
// if the value parses correctly in the model and is then ignored by the consumer.
func TestConfigureSweepHonoursInterval(t *testing.T) {
cases := []struct {
name string
interval string
wantEnabled bool
}{
{"unset is ON at the engine default", "", true},
{"an explicit tick is ON", "30s", true},
{"below the floor is still ON (clamped, not refused)", "1s", true},
{"garbage is still ON (a typo must not disable health data)", "soon", true},
{"0 is OFF", "0", false},
{"off is OFF", "off", false},
{"disabled is OFF", "disabled", false},
{"none is OFF", "none", false},
}
for _, c := range cases {
eng := engine.New()
a := New(eng, nil)
// GroupHealth: true is the default (DefaultGlobals seed); set it explicitly here
// because a bare model.Globals{} zeroes it, which would gate the sweep off and
// mask what this test is actually about (the SweepInterval grammar).
a.configureSweep(&model.Model{Globals: model.Globals{SweepInterval: c.interval, GroupHealth: true}})
enabled, _, _, _ := eng.SweepStatus()
if enabled != c.wantEnabled {
t.Errorf("%s: sweep_interval=%q -> enabled=%v, want %v", c.name, c.interval, enabled, c.wantEnabled)
}
eng.StopSweep()
}
}
// The GroupHealth master-switch gates the sweep independently of SweepInterval: when it
// is off the sweep is OFF even for a perfectly valid interval, and the interval value is
// still passed through untouched (flipping the toggle back on restores the schedule).
func TestConfigureSweepGroupHealthGate(t *testing.T) {
// GroupHealth off + a valid interval -> sweep disabled.
eng := engine.New()
a := New(eng, nil)
a.configureSweep(&model.Model{Globals: model.Globals{SweepInterval: "30s", GroupHealth: false}})
if enabled, _, _, _ := eng.SweepStatus(); enabled {
t.Fatalf("GroupHealth=false with sweep_interval=30s -> enabled=true, want the sweep gated off")
}
eng.StopSweep()
// GroupHealth on (the default) + the same interval -> sweep enabled, as before.
eng = engine.New()
a = New(eng, nil)
a.configureSweep(&model.Model{Globals: model.Globals{SweepInterval: "30s", GroupHealth: true}})
if enabled, _, _, _ := eng.SweepStatus(); !enabled {
t.Fatalf("GroupHealth=true with sweep_interval=30s -> enabled=false, want the sweep on")
}
eng.StopSweep()
}
// The interval actually reaches the engine (and the floor clamp with it), rather than
// every value collapsing onto the default.
func TestConfigureSweepPassesTickThrough(t *testing.T) {
eng := engine.New()
a := New(eng, nil)
defer eng.StopSweep()
a.configureSweep(&model.Model{Globals: model.Globals{SweepInterval: "45s", GroupHealth: true}})
if got := eng.SweepTickForTest(); got != 45*time.Second {
t.Errorf("tick = %v, want 45s", got)
}
a.configureSweep(&model.Model{Globals: model.Globals{SweepInterval: "1s", GroupHealth: true}})
if got := eng.SweepTickForTest(); got != model.SweepIntervalMin {
t.Errorf("tick = %v, want the %v floor", got, model.SweepIntervalMin)
}
}
+4
View File
@@ -95,6 +95,10 @@ type Engine struct {
groupTestTotal atomic.Int64
groupTestMu sync.Mutex
groupTestRes []GroupTestResult
// groupTestScope is the set of group names the current/last run covers. Published
// so the panel can put a "measuring" indicator on exactly the cards the run touches
// instead of on all of them; see GroupTestStatus.
groupTestScope []string
}
// New builds the engine context (with every protocol/dns/service registry
+53 -4
View File
@@ -6,6 +6,7 @@ import (
"io"
"net/http"
"net/netip"
"sort"
"strings"
"sync"
"time"
@@ -123,6 +124,10 @@ func (e *Engine) TestGroups(names []string, probeURL string) (started bool) {
e.groupTestTotal.Store(int64(len(targets) + len(missing)))
e.groupTestDone.Store(0)
e.setGroupTestResults(nil)
// Publish WHICH groups this run covers before returning started=true, so a poll
// issued immediately after the POST already knows where the run applies. See
// GroupTestStatus for why the scope is a set of names.
e.setGroupTestScope(scopeOf(targets, missing))
go func() {
defer e.groupTestRunning.Store(false)
@@ -347,16 +352,60 @@ func parsePlainIP(body string) (ip, country string) {
return v, ""
}
// GroupTestStatus reports the group-test progress and the results collected so far.
// results is ALWAYS non-nil so the panel can render it unconditionally. Lock-free
// on the counters; the results slice is copied under a short leaf lock.
func (e *Engine) GroupTestStatus() (running bool, done, total int, results []GroupTestResult) {
// GroupTestStatus reports the group-test progress, WHICH GROUPS the run covers, and the
// results collected so far. scope and results are ALWAYS non-nil so the panel can render
// them unconditionally. Lock-free on the counters; the slices are copied under a short
// leaf lock.
//
// # Why scope is a SET of names and not one name
//
// Without it this status is a bare global `running`, and the panel has no way to tell
// which card a run belongs to — so pressing "Test" on ONE group lit up "measuring" on
// every card, while the result correctly arrived for one. That was a defect in this
// contract, not in the panel: the information simply was not being published.
//
// A single "the group being tested" field would not have fixed it either, because the
// two launch modes have genuinely different reach: POST with a name tests one group,
// POST with an empty body tests EVERY group, and on the second one an indicator on every
// card is correct. A set expresses both uniformly — the panel renders the indicator on
// card g iff running && scope contains g — and it degenerates correctly at both ends
// (one name, or every name). A scalar would have forced a second flag beside it, and two
// fields describing one fact eventually disagree.
//
// The scope is resolved from the RUNNING BOX at launch and includes names the caller
// asked for that do not exist (those produce a not-found RESULT, so the operator sees an
// answer rather than silence — see TestGroups). It PERSISTS after the run finishes, so
// the panel can still tell which cards the displayed results belong to.
func (e *Engine) GroupTestStatus() (running bool, done, total int, scope []string, results []GroupTestResult) {
e.groupTestMu.Lock()
sc := append([]string{}, e.groupTestScope...)
e.groupTestMu.Unlock()
return e.groupTestRunning.Load(),
int(e.groupTestDone.Load()),
int(e.groupTestTotal.Load()),
sc,
e.groupTestResults()
}
// scopeOf is the set of group names a run covers: everything it will produce a result
// for, whether that result is a measurement or a "no such group". Sorted so the panel
// sees a stable order and two polls of the same run never differ.
func scopeOf(targets []adapter.Outbound, missing []string) []string {
out := make([]string, 0, len(targets)+len(missing))
for _, ob := range targets {
out = append(out, ob.Tag())
}
out = append(out, missing...)
sort.Strings(out)
return out
}
func (e *Engine) setGroupTestScope(names []string) {
e.groupTestMu.Lock()
e.groupTestScope = names
e.groupTestMu.Unlock()
}
func (e *Engine) setGroupTestResults(rs []GroupTestResult) {
e.groupTestMu.Lock()
e.groupTestRes = rs
+65 -5
View File
@@ -47,7 +47,7 @@ func TestGroupTestSingleton(t *testing.T) {
if e.TestGroups(nil, "") {
t.Fatalf("TestGroups started a SECOND run while one was in flight")
}
if running, _, _, _ := e.GroupTestStatus(); !running {
if running, _, _, _, _ := e.GroupTestStatus(); !running {
t.Fatalf("GroupTestStatus must report running while the guard is held")
}
e.groupTestRunning.Store(false)
@@ -69,7 +69,7 @@ func TestGroupTestStoppedEngineIsEmptyRun(t *testing.T) {
}
waitGroupTestIdle(t, e)
running, done, total, results := e.GroupTestStatus()
running, done, total, _, results := e.GroupTestStatus()
if running || done != 0 || total != 0 {
t.Fatalf("idle run: running=%v done=%d total=%d, want false/0/0", running, done, total)
}
@@ -91,7 +91,7 @@ func TestGroupTestNamedMissingGroupReportsIt(t *testing.T) {
}
waitGroupTestIdle(t, e)
_, done, total, results := e.GroupTestStatus()
_, done, total, _, results := e.GroupTestStatus()
if total != 1 || done != 1 {
t.Fatalf("done/total = %d/%d, want 1/1", done, total)
}
@@ -118,7 +118,7 @@ func TestGroupTestDedupesRequestedNames(t *testing.T) {
t.Fatalf("TestGroups must start")
}
waitGroupTestIdle(t, e)
if _, _, total, _ := e.GroupTestStatus(); total != 2 {
if _, _, total, _, _ := e.GroupTestStatus(); total != 2 {
t.Fatalf("total = %d, want 2 (deduped, blanks dropped)", total)
}
}
@@ -223,10 +223,70 @@ func waitGroupTestIdle(t *testing.T, e *Engine) {
t.Helper()
deadline := time.Now().Add(5 * time.Second)
for time.Now().Before(deadline) {
if running, _, _, _ := e.GroupTestStatus(); !running {
if running, _, _, _, _ := e.GroupTestStatus(); !running {
return
}
time.Sleep(5 * time.Millisecond)
}
t.Fatalf("group test did not finish within 5s")
}
// TestGroupTestScopeNamesItsTargets is the regression for the contract defect behind
// "press Test on one group, every card says measuring": the status used to publish a
// bare global `running` with no indication of which groups the run covered, so the
// panel had no way to attribute it and lit up all of them.
//
// A named run must scope to exactly that name; the scope must survive the run so the
// results can still be attributed; and a fresh engine must report an empty (never nil)
// scope so the panel can test membership unconditionally.
func TestGroupTestScopeNamesItsTargets(t *testing.T) {
e := New()
if _, _, _, scope, _ := e.GroupTestStatus(); scope == nil || len(scope) != 0 {
t.Fatalf("fresh engine scope = %v, want an empty non-nil slice", scope)
}
if !e.TestGroups([]string{"ghost"}, "") {
t.Fatal("TestGroups must start")
}
// The scope is published BEFORE started=true returns, so a poll issued immediately
// after the POST can already attribute the run.
_, _, _, scope, _ := e.GroupTestStatus()
if len(scope) != 1 || scope[0] != "ghost" {
t.Fatalf("scope = %v, want exactly [ghost] — a named run must not claim other groups", scope)
}
waitGroupTestIdle(t, e)
if _, _, _, scope, _ := e.GroupTestStatus(); len(scope) != 1 || scope[0] != "ghost" {
t.Fatalf("scope after the run = %v, want it to persist so results stay attributable", scope)
}
}
// A run started with no names covers every group in the running box — on a stopped
// engine that is none, and the scope is empty rather than "everything". An indicator
// driven off this scope is then correctly absent, instead of appearing on every card
// because `running` happened to be true.
func TestGroupTestScopeAllGroups(t *testing.T) {
e := New()
if !e.TestGroups(nil, "") {
t.Fatal("TestGroups must start")
}
waitGroupTestIdle(t, e)
if _, _, _, scope, _ := e.GroupTestStatus(); len(scope) != 0 {
t.Fatalf("scope = %v, want empty (a stopped engine has no groups to cover)", scope)
}
}
// The scope is deduped and sorted: a panel that double-sends a name must not see it
// twice, and two polls of one run must never disagree about the order.
func TestGroupTestScopeDedupedAndSorted(t *testing.T) {
e := New()
if !e.TestGroups([]string{"zeta", "alpha", "zeta", " "}, "") {
t.Fatal("TestGroups must start")
}
waitGroupTestIdle(t, e)
_, _, _, scope, _ := e.GroupTestStatus()
if len(scope) != 2 || scope[0] != "alpha" || scope[1] != "zeta" {
t.Fatalf("scope = %v, want [alpha zeta]", scope)
}
}
+102 -15
View File
@@ -125,24 +125,55 @@ type sweepState struct {
cycles uint64
// busy guards against a slow tick overlapping the next one.
busy bool
// planFn is the plan source. nil means Engine.BuildProbePlan, which is what
// production always uses; a test substitutes a stub so the cursor-preservation
// contract can be exercised without standing up a real box (the bug it guards
// against only shows up across repeated reconfigurations of a NON-empty plan).
planFn func(map[string]GroupProbeSpec, string) []ProbeJob
}
// buildPlan is the single plan-source indirection. Caller holds sweepMu.
func (e *Engine) buildPlan(cfg SweepConfig) []ProbeJob {
if e.sweep.planFn != nil {
return e.sweep.planFn(cfg.Specs, cfg.FallbackURL)
}
return e.BuildProbePlan(cfg.Specs, cfg.FallbackURL)
}
// ConfigureSweep installs (or updates, or stops) the scheduled sweep. It is idempotent
// and safe to call on every apply: the loop is started on the first enabled call,
// re-tuned in place afterwards, and stopped by an Enabled=false call.
//
// Changing Specs/FallbackURL invalidates the in-flight plan, because the probe URL a
// tag must be measured with may have changed with it.
// # It must NOT restart the cycle (this was a release-blocking bug)
//
// It is called from applyLocked on every SUCCESSFUL apply — and on this router cron
// reconciles EVERY MINUTE, so that is once a minute forever, almost always with an
// identical config. The first version rebuilt the plan and zeroed the cursor here. A
// full cycle is ~6 minutes (896 measurements at 24 per tick), so the cursor was reset
// five minutes before it could ever finish: observed live walking 96 -> 24 -> 72 with
// cycles stuck at 0. The sweep therefore never completed a pass, which in turn left the
// nodes past the cursor permanently "untested" — the exact condition the whole feature
// exists to remove, reintroduced by its own reconfiguration path.
//
// So progress is preserved across a reconfiguration that does not actually change the
// work: syncPlanLocked rebuilds the plan and keeps the cursor when the new plan is
// measurement-for-measurement identical to the old one. Only a genuinely different plan
// restarts the walk, which is correct — the old cursor indexes a list that no longer
// exists.
func (e *Engine) ConfigureSweep(cfg SweepConfig) {
cfg = cfg.withDefaults()
e.sweepMu.Lock()
defer e.sweepMu.Unlock()
e.sweep.cfg = cfg
// A config change always restarts the cycle: the plan encodes the old URLs.
e.sweep.plan = nil
e.sweep.cursor = 0
e.sweep.planHash = ""
if cfg.Enabled {
// Rebuild against the new specs/URLs, keeping our place if nothing moved.
e.syncPlanLocked(cfg)
} else {
e.sweep.plan = nil
e.sweep.cursor = 0
e.sweep.planHash = ""
}
switch {
case cfg.Enabled && e.sweep.stop == nil:
@@ -225,16 +256,19 @@ func (e *Engine) sweepTick() {
return
}
cfg := e.sweep.cfg
// Rebuild the plan at the start of a cycle, and whenever an Apply swap has changed
// the running config out from under it (tags may have appeared or vanished).
hash := e.Hash()
if e.sweep.plan == nil || e.sweep.cursor >= len(e.sweep.plan) || e.sweep.planHash != hash {
if e.sweep.plan != nil && e.sweep.cursor >= len(e.sweep.plan) {
e.sweep.cycles++
}
e.sweep.plan = e.BuildProbePlan(cfg.Specs, cfg.FallbackURL)
switch {
case e.sweep.plan != nil && e.sweep.cursor >= len(e.sweep.plan):
// Cycle complete: count it and start the next pass from the top. This is the
// ONE place the cursor is deliberately rewound.
e.sweep.cycles++
e.sweep.plan = e.buildPlan(cfg)
e.sweep.cursor = 0
e.sweep.planHash = hash
e.sweep.planHash = e.Hash()
case e.sweep.plan == nil || e.sweep.planHash != e.Hash():
// No plan yet, or an Apply swap changed the running config out from under it
// (tags may have appeared or vanished). Rebuild — but keep our place if the
// resulting work is the same, so a no-op apply cannot restart the walk.
e.syncPlanLocked(cfg)
}
// Take the next slice of the cycle.
start := e.sweep.cursor
@@ -274,6 +308,50 @@ func (e *Engine) sweepTick() {
wg.Wait()
}
// syncPlanLocked rebuilds the plan from the current box and specs, and RESETS the
// cursor only if the resulting work actually differs.
//
// "The same plan" is defined as measurement-for-measurement identity: same jobs, in the
// same order, each dialling the same tag with the same probe URL and recording into the
// same set of tags (plansEqual). That is the right comparison rather than, say, hashing
// the config, because the cursor's only meaning is a position in THIS list — if the list
// is identical the position is still valid, and if any measurement changed the position
// is meaningless whatever the config hash says.
//
// The plan is derived data (box outbounds + group membership + specs), so rebuilding and
// comparing is also self-correcting: there is no separate "is it stale" bookkeeping that
// could itself go wrong. The cost is one walk of the outbound list per call — the same
// walk a tick already does at cycle end.
//
// Caller holds sweepMu.
func (e *Engine) syncPlanLocked(cfg SweepConfig) {
next := e.buildPlan(cfg)
e.sweep.planHash = e.Hash()
if e.sweep.plan != nil && plansEqual(next, e.sweep.plan) {
return // identical work: keep walking where we were
}
e.sweep.plan = next
e.sweep.cursor = 0
}
// plansEqual reports whether two plans describe exactly the same measurements.
func plansEqual(a, b []ProbeJob) bool {
if len(a) != len(b) {
return false
}
for i := range a {
if a[i].Dial != b[i].Dial || a[i].URL != b[i].URL || len(a[i].Store) != len(b[i].Store) {
return false
}
for j := range a[i].Store {
if a[i].Store[j] != b[i].Store[j] {
return false
}
}
}
return true
}
// sweepShouldProbe reports whether a job is worth running now: yes when ANY tag it
// covers has no observation at all or one older than freshness.
//
@@ -290,3 +368,12 @@ func sweepShouldProbe(view HealthView, j ProbeJob, freshness time.Duration) bool
}
return false
}
// SweepTickForTest exposes the tick period the sweep is currently configured with, so a
// test in another package can assert that a configured interval actually reached the
// engine rather than collapsing onto the default. Not part of the operational API.
func (e *Engine) SweepTickForTest() time.Duration {
e.sweepMu.Lock()
defer e.sweepMu.Unlock()
return e.sweep.cfg.Interval
}
+145
View File
@@ -128,3 +128,148 @@ func TestSweepTickStoppedEngine(t *testing.T) {
t.Fatalf("stopped engine: cursor=%d total=%d, want 0/0", cursor, total)
}
}
// D1 REGRESSION (release blocker). ConfigureSweep is called from applyLocked on every
// successful apply, and cron reconciles this router every MINUTE. The first version
// rebuilt the plan and zeroed the cursor on each of those calls, so a ~6-minute cycle
// was restarted every 60s and never once completed: observed on the stand walking
// 96 -> 24 -> 72 with cycles stuck at 0, leaving everything past the cursor forever
// "untested" — the exact condition the sweep exists to remove.
//
// This replays that: walk part-way, then reconfigure repeatedly with an unchanged
// config, as cron does.
func TestConfigureSweepPreservesProgressWhenPlanUnchanged(t *testing.T) {
plan := []ProbeJob{
{Dial: "n1", URL: "u", Store: []string{"n1"}},
{Dial: "n2", URL: "u", Store: []string{"n2"}},
{Dial: "n3", URL: "u", Store: []string{"n3"}},
{Dial: "n4", URL: "u", Store: []string{"n4"}},
}
e := New()
e.sweepMu.Lock()
// A fresh slice every call, exactly as a real rebuild would produce.
e.sweep.planFn = func(map[string]GroupProbeSpec, string) []ProbeJob {
return append([]ProbeJob(nil), plan...)
}
e.sweepMu.Unlock()
cfg := SweepConfig{Enabled: true, FallbackURL: "u"}
e.ConfigureSweep(cfg)
defer e.StopSweep()
// Walk half the cycle.
e.sweepMu.Lock()
e.sweep.cursor = 2
e.sweepMu.Unlock()
// Five no-op reconciles, one "minute" apart.
for i := 0; i < 5; i++ {
e.ConfigureSweep(cfg)
if _, cursor, total, cycles := e.SweepStatus(); cursor != 2 || total != 4 || cycles != 0 {
t.Fatalf("reconcile #%d reset the walk: cursor=%d total=%d cycles=%d, want 2/4/0",
i+1, cursor, total, cycles)
}
}
// A GENUINE change must still restart the walk — the old cursor indexes a list
// that no longer exists.
plan = append(plan, ProbeJob{Dial: "n5", URL: "u", Store: []string{"n5"}})
e.ConfigureSweep(cfg)
if _, cursor, total, _ := e.SweepStatus(); cursor != 0 || total != 5 {
t.Fatalf("a changed plan must restart: cursor=%d total=%d, want 0/5", cursor, total)
}
}
// The cursor must also survive the tick path's own rebuild trigger (an Apply swap
// changed the engine hash) when the resulting work is identical.
func TestSweepTickRebuildPreservesProgress(t *testing.T) {
plan := []ProbeJob{
{Dial: "n1", URL: "u", Store: []string{"n1"}},
{Dial: "n2", URL: "u", Store: []string{"n2"}},
{Dial: "n3", URL: "u", Store: []string{"n3"}},
}
e := New()
e.sweepMu.Lock()
e.sweep.planFn = func(map[string]GroupProbeSpec, string) []ProbeJob {
return append([]ProbeJob(nil), plan...)
}
e.sweepMu.Unlock()
e.ConfigureSweep(SweepConfig{Enabled: true, Batch: 1, Concurrency: 1})
defer e.StopSweep()
e.sweepMu.Lock()
e.sweep.cursor = 1
e.sweep.planHash = "stale-hash" // force the tick's rebuild branch
e.sweepMu.Unlock()
e.sweepTick()
// The tick rebuilds (hash mismatch), finds identical work, keeps the cursor, and
// then consumes its batch of 1 — so 1 -> 2, never back to 0.
if _, cursor, _, cycles := e.SweepStatus(); cursor != 2 || cycles != 0 {
t.Fatalf("cursor=%d cycles=%d, want 2/0 (kept its place, then advanced one batch)", cursor, cycles)
}
}
// plansEqual is the definition of "the same work", so its edges are a contract: a
// different tag, a different probe URL, or a different set of recorded tags all mean a
// different plan — and a different plan legitimately restarts the walk.
func TestPlansEqual(t *testing.T) {
base := []ProbeJob{
{Dial: "n1", URL: "u", Store: []string{"n1", "group-a-m0-n1"}},
{Dial: "n2", URL: "u", Store: []string{"n2"}},
}
if !plansEqual(base, append([]ProbeJob(nil), base...)) {
t.Fatal("a copy must compare equal")
}
cases := map[string][]ProbeJob{
"shorter": base[:1],
"different dial tag": {
{Dial: "n9", URL: "u", Store: []string{"n1", "group-a-m0-n1"}},
{Dial: "n2", URL: "u", Store: []string{"n2"}},
},
"different probe URL": {
{Dial: "n1", URL: "other", Store: []string{"n1", "group-a-m0-n1"}},
{Dial: "n2", URL: "u", Store: []string{"n2"}},
},
"different store set": {
{Dial: "n1", URL: "u", Store: []string{"n1"}},
{Dial: "n2", URL: "u", Store: []string{"n2"}},
},
"different store member": {
{Dial: "n1", URL: "u", Store: []string{"n1", "group-b-m0-n1"}},
{Dial: "n2", URL: "u", Store: []string{"n2"}},
},
"empty": {},
"nil": nil,
}
for name, other := range cases {
if plansEqual(base, other) {
t.Errorf("%s: compared EQUAL to the base plan, want different", name)
}
}
}
// The one place the cursor is legitimately rewound is the completion of a cycle, and
// that must also be the one place `cycles` is bumped.
func TestSweepCycleCompletionRewinds(t *testing.T) {
e := New()
e.ConfigureSweep(SweepConfig{Enabled: true})
defer e.StopSweep()
e.sweepMu.Lock()
e.sweep.plan = []ProbeJob{{Dial: "n1", Store: []string{"n1"}}}
e.sweep.cursor = 1 // walked to the end
e.sweep.planHash = e.Hash()
e.sweepMu.Unlock()
e.sweepTick()
_, cursor, _, cycles := e.SweepStatus()
if cycles != 1 {
t.Errorf("cycles = %d, want 1 after finishing a pass", cycles)
}
if cursor != 0 {
t.Errorf("cursor = %d, want 0 at the start of the next pass", cursor)
}
}
+112 -58
View File
@@ -77,69 +77,31 @@ func (b *builder) resolveChain(name string) (string, bool) {
}
// buildChain materialises the hop wrappers for a chain and returns its entry tag
// (the last hop's wrapper). A 1-hop chain resolves straight to that hop (no
// wrapping/detour). On any unresolvable hop it rolls back every wrapper it
// appended so a failed chain leaves no orphan outbounds, then warns + skips.
// (the last hop's wrapper). The chain's hops are first FLATTENED: any `chain:<sub>`
// hop is spliced in place as the sub-chain's own hops (recursively, cycle-guarded),
// so the wrapper loop only ever sees a flat node:/group:/bare list. A 1-hop chain
// resolves straight to that hop (no wrapping/detour). On any unresolvable hop it
// rolls back every wrapper it appended so a failed chain leaves no orphan
// outbounds, then warns + skips.
func (b *builder) buildChain(name string) (string, bool) {
var ch *model.Chain
for i := range b.m.Chains {
if b.m.Chains[i].Name == name {
ch = &b.m.Chains[i]
break
}
}
if ch == nil {
if b.findChainDef(name) == nil {
b.warnf("chain %q: not defined, target skipped", name)
return "", false
}
// Gather the real hops, lifting a leading `egress:<name>` hop out as the chain's
// ENTRY interface. An egress hop is not a routable wrapper — it becomes the
// detour the FIRST real hop dials through, so the chain enters the path over
// that WAN/egress (multi-WAN). It is only meaningful at position 0 (the entry);
// anywhere else it is dropped with a warning (the last hop is the EXIT and must
// be a real node/group). A non-existent egress is likewise dropped. Fail-open:
// dropping an egress hop never aborts box.New — the chain builds without it.
// Flatten the chain into ONE ordered list of node:/group:/bare hops, splicing
// every `chain:<sub>` hop in place as the sub-chain's OWN hops. A leading
// `egress:` — written here, or contributed by a spliced sub-chain that itself
// leads with one — is lifted into baseDetour as the chain's ENTRY interface (the
// detour L1 dials through). expandHops fails the WHOLE chain (fail-closed) on a
// reference cycle, an undefined sub-chain, or a spliced sub-chain whose leading
// egress would fall anywhere but the entry: an egress is only meaningful as the
// path's single entry point, so reusing an entry-bound chain as a middle segment
// is a contradiction, not a droppable extra.
var hops []string
baseDetour := ""
pos := 0
for _, h := range ch.Hops {
h = strings.TrimSpace(h)
if h == "" {
continue
}
kind, egName := model.SplitTarget(h)
isEgress := strings.EqualFold(kind, "egress")
cur := pos
pos++
if isEgress {
if cur != 0 {
b.warnf("chain %q: egress hop only allowed first, dropped", name)
continue
}
tag := netplane.EgressOutboundTag(strings.TrimSpace(egName))
if !b.egressTags[tag] {
// FAIL-CLOSED, same contract as egressDetourOrBlock (outbound.go):
// the operator said this chain must ENTER the path over that WAN.
// Dropping the hop would have made L1 dial straight out over the
// default route instead — the chain would still work, which is why
// this was easy to miss, but its first hop would be visible to the
// wrong ISP from the router's real address. Block the entry instead.
b.warnf("chain %q: entry egress %q does not exist — either the name is wrong, or it was removed, or its type is one this build does not implement. The chain is FAIL-CLOSED: its first hop is BLOCKED rather than dialling out over the default WAN. Fix the egress name or type to restore the chain", name, egName)
baseDetour = tagBlock
continue
}
// applyDPI keys off the RULE's resolved target, which for a chain is the
// exit wrapper — never the egress. So a DPI preset carried by an egress
// used as a chain ENTRY hop cannot be stamped anywhere; say so rather
// than let the operator believe the desync is active.
if preset, dpi := b.egressDPI[tag]; dpi {
b.warnf("chain %q: entry egress %q carries dpi=%s, but a chain hop cannot apply a native tls_* preset; dpi ignored", name, egName, preset)
}
baseDetour = tag
continue
}
hops = append(hops, h)
if !b.expandHops(name, map[string]bool{}, &hops, &baseDetour, true) {
return "", false
}
if len(hops) == 0 {
// Empty, or ONLY a leading egress entry hop (no real exit to route to).
@@ -175,6 +137,100 @@ func (b *builder) buildChain(name string) (string, bool) {
return prev, true
}
// findChainDef returns a pointer to the model.Chain named name, or nil.
func (b *builder) findChainDef(name string) *model.Chain {
for i := range b.m.Chains {
if b.m.Chains[i].Name == name {
return &b.m.Chains[i]
}
}
return nil
}
// expandHops walks chain `name`'s hops in L1..Ln order, appending each real
// node:/group:/bare hop to *hops and flattening every `chain:<sub>` hop IN PLACE by
// recursing into the sub-chain (this is what makes chains reusable/composable).
// `seen` holds the chain names on the current expansion path — a name already in it
// is a reference cycle. A leading `egress:` is lifted into *baseDetour
// (liftEntryEgress) but ONLY at the entry position (nothing appended, no detour set
// yet). An egress anywhere else is dropped fail-open when `name` is the chain the
// rule directly targets (topLevel — the historical stray-egress-drop behaviour), but
// refused fail-closed when `name` was reached through a `chain:` hop: an entry-bound
// chain cannot legally become a middle segment. Returns false — failing the whole
// chain — on a cycle, an undefined sub-chain, or such a mis-positioned composed
// egress. Depth is bounded by the finite chain set minus the growing `seen` guard.
func (b *builder) expandHops(name string, seen map[string]bool, hops *[]string, baseDetour *string, topLevel bool) bool {
ch := b.findChainDef(name)
if ch == nil {
b.warnf("chain %q: hop references a chain that is not defined, target skipped", name)
return false
}
if seen[name] {
b.warnf("chain %q: hop references chain %q, forming a reference cycle; the whole chain is skipped", name, name)
return false
}
seen[name] = true
defer delete(seen, name)
for _, h := range ch.Hops {
h = strings.TrimSpace(h)
if h == "" {
continue
}
kind, arg := model.SplitTarget(h)
switch {
case strings.EqualFold(kind, "chain"):
// Splice the sub-chain's hops in place. topLevel=false: a leading egress the
// sub-chain carries can only survive if it lands at the overall entry
// (nothing appended yet); anywhere else it fails the whole chain closed.
if !b.expandHops(strings.TrimSpace(arg), seen, hops, baseDetour, false) {
return false
}
case strings.EqualFold(kind, "egress"):
if len(*hops) == 0 && *baseDetour == "" {
b.liftEntryEgress(name, arg, baseDetour)
continue
}
if topLevel {
// A stray egress in the chain the rule targets directly: historical
// fail-open drop — the surrounding real hops still chain.
b.warnf("chain %q: egress hop only allowed first, dropped", name)
continue
}
// This egress leads a chain spliced at a non-entry position: the operator
// tried to reuse a chain that ENTERS over an egress as a middle segment,
// which cannot mean anything. Fail the whole chain closed rather than
// silently drop the egress and let that segment dial out over the wrong WAN.
b.warnf("chain %q: an egress-entry chain cannot be spliced mid-path — its entry egress would fall at a non-entry position, where an egress means nothing; the whole chain is FAIL-CLOSED", name)
return false
default:
*hops = append(*hops, h)
}
}
return true
}
// liftEntryEgress resolves a chain's leading `egress:<eg>` hop into the detour its
// first real hop dials through — the WAN/egress the chain ENTERS over. A missing
// egress is FAIL-CLOSED to tagBlock rather than dropped: dropping it would make L1
// dial straight out over the default route from the router's real address (the
// tunnel would still come up, which is why it was easy to miss). A DPI preset the
// egress carries is reported as ignored — a chain hop cannot stamp a native tls_*
// preset (applyDPI keys off the rule's resolved target, the exit wrapper, never the
// egress). It always sets *baseDetour.
func (b *builder) liftEntryEgress(chainName, eg string, baseDetour *string) {
tag := netplane.EgressOutboundTag(strings.TrimSpace(eg))
if !b.egressTags[tag] {
b.warnf("chain %q: entry egress %q does not exist — either the name is wrong, or it was removed, or its type is one this build does not implement. The chain is FAIL-CLOSED: its first hop is BLOCKED rather than dialling out over the default WAN. Fix the egress name or type to restore the chain", chainName, eg)
*baseDetour = tagBlock
return
}
if preset, dpi := b.egressDPI[tag]; dpi {
b.warnf("chain %q: entry egress %q carries dpi=%s, but a chain hop cannot apply a native tls_* preset; dpi ignored", chainName, eg, preset)
}
*baseDetour = tag
}
// buildHopWrapper materialises one chain hop (a `node:`/`group:`/bare reference)
// as a per-chain copy tagged chain-<chainName>-h<idx>, detouring through the
// previous hop's tag (detour==""for L1). It returns the tag the NEXT hop should
@@ -275,8 +331,6 @@ func (b *builder) buildHopWrapper(chainName string, idx int, hop, detour string)
// which is a severe outcome for a typo — so name the actual cause here
// instead of leaving the operator to guess which of the two reasons applies.
switch kind {
case "chain":
b.warnf("chain %q: hop %d %q — a chain cannot be a hop of another chain (there is no outbound to copy and detour); inline the other chain's node/group hops here", chainName, idx, hop)
case "egress":
b.warnf("chain %q: hop %d %q — an egress is only valid as the FIRST hop, where it becomes the WAN the chain enters over; in any other position there is nothing to tunnel through", chainName, idx, hop)
case tagDirect, tagBlock:
+109
View File
@@ -105,6 +105,16 @@ func (b *builder) buildDNS() *option.DNSOptions {
}
}
// Endpoint resolver (route.default_domain_resolver): a dedicated bootstrap-direct
// server used ONLY to resolve proxy outbounds' server domains. Built by
// endpointResolver() (shared with buildRoute, which references its tag); appended
// here so the tag route names actually exists among the DNS servers. Not added to
// serverTags and not referenced by any DNS rule — it is a bootstrap server, not a
// client-facing one.
if b.endpointResolver() != "" && b.endpointResolverServer != nil {
servers = append(servers, *b.endpointResolverServer)
}
// DoH-block NXDOMAIN rules sit FIRST (BlockDoH): they answer known public DoH
// hostnames + the Firefox canary with NXDOMAIN so clients fall back to plaintext
// :53. Empty when BlockDoH is off. The synthetic local-dns rule prepended at the
@@ -179,6 +189,105 @@ func (b *builder) buildDNS() *option.DNSOptions {
}
}
// endpointResolverTagPrefix names the synthetic bootstrap-direct DNS server the
// endpoint resolver is cloned into, so its tag never collides with the user's own
// `config resolver` server tag it was cloned from.
const endpointResolverTagPrefix = "shater-endpoint-dns-"
// endpointResolver computes (once) the DNS server tag that route.default_domain_resolver
// must reference to resolve proxy outbounds' server DOMAINS (vless/awg endpoints).
// It returns "" when the feature is off or the named resolver cannot be used, in
// which case buildRoute leaves DefaultDomainResolver unset (never referencing a
// non-existent server, which would abort box.New).
//
// # Tag source priority
//
// The active WAN profile's EndpointResolver override wins over Globals.EndpointResolver.
// Both name an existing `config resolver` by name.
//
// # Bootstrap-direct
//
// An endpoint resolver MUST go DIRECT: resolving a proxy server's own domain THROUGH
// that proxy is an impossible bootstrap loop. The named `config resolver` may — and
// for anti-leak often SHOULD — carry Detour=proxy for the client DNS plane, so rather
// than mutate the shared server (which would break client anti-leak) or merely warn
// (which would leave the loop in place), this builds a DEDICATED server cloned from the
// named resolver with Detour FORCED to "direct", under its own tag. The client-facing
// copy keeps its own detour; the endpoint copy is always direct.
//
// # Zero client resolvers
//
// buildDNS early-returns nil when the model has no resolvers, but that case collapses
// here naturally: with no resolvers the named one does not exist, so this returns ""
// and DefaultDomainResolver is simply not set (the engine falls back to its built-in
// resolver for endpoint domains — the same as before this feature). No synthetic DNS
// plane is fabricated out of nothing.
func (b *builder) endpointResolver() string {
if b.endpointResolverComputed {
return b.endpointResolverTag
}
b.endpointResolverComputed = true
// Profile overrides live on the builder only after applyProfilesAndPresets; it is
// idempotent, so calling it here makes endpointResolver safe to invoke from either
// buildRoute or buildDNS regardless of order.
b.applyProfilesAndPresets()
name := b.profileEndpointResolver
src := "active profile"
if name == "" {
name = strings.TrimSpace(b.m.Globals.EndpointResolver)
src = "globals.endpoint_resolver"
}
if name == "" {
return "" // feature off — no default_domain_resolver, engine uses its built-in
}
var res *model.Resolver
for i := range b.m.Resolvers {
if b.m.Resolvers[i].Name == name {
res = &b.m.Resolvers[i]
break
}
}
if res == nil {
b.warnf("endpoint_resolver %q (%s): no such resolver — proxy server domains fall back to the engine's built-in resolver and route.default_domain_resolver is left unset", name, src)
return ""
}
// A fake-IP resolver mints synthetic addresses and never contacts an upstream, so
// using it to resolve a REAL proxy server's domain would hand the dialer a fake IP
// and the tunnel could never connect. Refuse it explicitly.
if strings.EqualFold(strings.TrimSpace(res.Type), "fakeip") {
b.warnf("endpoint_resolver %q (%s): a fakeip resolver cannot resolve a real proxy server domain (it invents addresses); route.default_domain_resolver is left unset", name, src)
return ""
}
// Clone under a dedicated tag with Detour forced direct (bootstrap-direct; see the
// doc comment). Guard the tag against colliding with any real resolver server tag.
clone := *res
clone.Detour = "direct"
tag := endpointResolverTagPrefix + res.Name
taken := make(map[string]bool, len(b.m.Resolvers))
for i := range b.m.Resolvers {
taken[b.m.Resolvers[i].Name] = true
}
for taken[tag] {
tag += "-x"
}
clone.Name = tag
srv, ok := b.dnsServer(clone)
if !ok {
// dnsServer already warned (bad address/type); leave the resolver unset.
b.warnf("endpoint_resolver %q (%s): could not be built as a bootstrap-direct server; route.default_domain_resolver is left unset", name, src)
return ""
}
b.endpointResolverServer = &srv
b.endpointResolverTag = tag
return tag
}
// resolverFallbackRules implements Globals.ResolverFallback: when the primary
// resolver produces NO response, the query is retried against the fallback.
//
+229
View File
@@ -0,0 +1,229 @@
package generate
import (
"strings"
"testing"
"github.com/sagernet/sing-box/option"
"github.com/sagernet/sing-box/shater/model"
)
// The endpoint resolver (route.default_domain_resolver) is the bootstrap-direct DNS
// server sing-box uses to resolve proxy outbounds' server DOMAINS (vless/awg
// endpoints). These tests verify the GENERATED config, not a model round-trip:
// - the referenced server tag is set on route.default_domain_resolver;
// - a per-profile override wins over Globals;
// - an unknown name leaves the field unset (so box.New can't abort on a dangling
// reference) and warns;
// - it works (or cleanly no-ops) when there are zero client resolvers;
// - the endpoint server is ALWAYS direct even when the named resolver detours
// through the proxy (bootstrap-direct: you can't resolve the proxy's own server
// through the proxy).
// endpointServerTag mirrors dns.go's endpointResolverTagPrefix + resolver name.
func endpointServerTag(resolver string) string {
return endpointResolverTagPrefix + resolver
}
// dnsServerByTag returns the generated DNS server with the given tag, or ok=false.
func dnsServerByTag(dns *option.DNSOptions, tag string) (option.DNSServerOptions, bool) {
if dns == nil {
return option.DNSServerOptions{}, false
}
for _, s := range dns.Servers {
if s.Tag == tag {
return s, true
}
}
return option.DNSServerOptions{}, false
}
// udpServerDetour extracts the dialer detour of a plain/udp DNS server.
func udpServerDetour(t *testing.T, s option.DNSServerOptions) string {
t.Helper()
opts, ok := s.Options.(*option.RemoteDNSServerOptions)
if !ok {
t.Fatalf("server %q: expected *RemoteDNSServerOptions, got %T", s.Tag, s.Options)
}
return opts.Detour
}
// TestEndpointResolverFromGlobals: Globals.EndpointResolver naming a real resolver
// makes route.default_domain_resolver point at a bootstrap-direct clone of it, and
// that clone actually exists among the DNS servers (else box.New would reject the
// reference).
func TestEndpointResolverFromGlobals(t *testing.T) {
m := &model.Model{
Globals: model.Globals{ResolverDefault: "cf", EndpointResolver: "cf"},
Resolvers: []model.Resolver{
{Name: "cf", Type: "udp", Address: "1.1.1.1", Detour: "direct"},
},
}
opts, warns, err := GenerateWithWarnings(m)
if err != nil {
t.Fatalf("GenerateWithWarnings: %v", err)
}
if opts.Route == nil || opts.Route.DefaultDomainResolver == nil {
t.Fatalf("route.default_domain_resolver must be set; route=%+v", opts.Route)
}
want := endpointServerTag("cf")
if got := opts.Route.DefaultDomainResolver.Server; got != want {
t.Fatalf("default_domain_resolver.server = %q, want %q", got, want)
}
if _, ok := dnsServerByTag(opts.DNS, want); !ok {
t.Fatalf("endpoint server %q must exist among DNS servers; servers=%+v", want, opts.DNS.Servers)
}
// The client-facing resolver "cf" must still be present in its own right.
if _, ok := dnsServerByTag(opts.DNS, "cf"); !ok {
t.Fatalf("client resolver %q must still be emitted", "cf")
}
for _, w := range warns {
if strings.Contains(w, "endpoint_resolver") {
t.Fatalf("no endpoint_resolver warning expected for a valid reference, got %q", w)
}
}
}
// TestEndpointResolverProfileOverridesGlobals: the active WAN profile's
// EndpointResolver wins over Globals.EndpointResolver.
func TestEndpointResolverProfileOverridesGlobals(t *testing.T) {
m := &model.Model{
Globals: model.Globals{ResolverDefault: "cf", EndpointResolver: "cf", ActiveProfile: "p"},
Resolvers: []model.Resolver{
{Name: "cf", Type: "udp", Address: "1.1.1.1", Detour: "direct"},
{Name: "q9", Type: "udp", Address: "9.9.9.9", Detour: "direct"},
},
Profiles: []model.Profile{
{Name: "p", Enabled: true, EndpointResolver: "q9"},
},
}
opts, _, err := GenerateWithWarnings(m)
if err != nil {
t.Fatalf("GenerateWithWarnings: %v", err)
}
if opts.Route == nil || opts.Route.DefaultDomainResolver == nil {
t.Fatalf("route.default_domain_resolver must be set; route=%+v", opts.Route)
}
want := endpointServerTag("q9") // the profile override, NOT globals' "cf"
if got := opts.Route.DefaultDomainResolver.Server; got != want {
t.Fatalf("profile override must win: default_domain_resolver.server = %q, want %q", got, want)
}
if _, ok := dnsServerByTag(opts.DNS, want); !ok {
t.Fatalf("endpoint server %q must exist among DNS servers; servers=%+v", want, opts.DNS.Servers)
}
}
// TestEndpointResolverUnknownNameUnset: an endpoint_resolver naming a resolver that
// does not exist must leave route.default_domain_resolver UNSET (no dangling
// reference for box.New to abort on) and warn. This is the box.New-safe outcome.
func TestEndpointResolverUnknownNameUnset(t *testing.T) {
m := &model.Model{
Globals: model.Globals{ResolverDefault: "cf", EndpointResolver: "ghost"},
Resolvers: []model.Resolver{
{Name: "cf", Type: "udp", Address: "1.1.1.1", Detour: "direct"},
},
}
opts, warns, err := GenerateWithWarnings(m)
if err != nil {
t.Fatalf("GenerateWithWarnings: %v", err)
}
if opts.Route != nil && opts.Route.DefaultDomainResolver != nil {
t.Fatalf("default_domain_resolver must be UNSET for an unknown resolver, got %+v", opts.Route.DefaultDomainResolver)
}
// And no phantom "shater-endpoint-dns-ghost" server was emitted.
if _, ok := dnsServerByTag(opts.DNS, endpointServerTag("ghost")); ok {
t.Fatalf("no endpoint server may be emitted for an unknown resolver")
}
if !warnsHaveSub(warns, "endpoint_resolver \"ghost\"") {
t.Fatalf("expected an unknown-resolver warning, got %v", warns)
}
}
// TestEndpointResolverZeroResolvers: with no client resolvers at all, the DNS plane
// is nil and the named endpoint resolver cannot exist, so the feature cleanly
// no-ops (no default_domain_resolver, a warning, no dangling reference).
func TestEndpointResolverZeroResolvers(t *testing.T) {
m := &model.Model{
Globals: model.Globals{EndpointResolver: "cf"},
// No Resolvers.
}
opts, warns, err := GenerateWithWarnings(m)
if err != nil {
t.Fatalf("GenerateWithWarnings: %v", err)
}
if opts.DNS != nil {
t.Fatalf("DNS plane must stay nil with zero resolvers, got %+v", opts.DNS)
}
if opts.Route != nil && opts.Route.DefaultDomainResolver != nil {
t.Fatalf("default_domain_resolver must be UNSET with zero resolvers, got %+v", opts.Route.DefaultDomainResolver)
}
if !warnsHaveSub(warns, "endpoint_resolver \"cf\"") {
t.Fatalf("expected a no-such-resolver warning, got %v", warns)
}
}
// TestEndpointResolverIsAlwaysDirect: the named resolver may detour through the
// proxy for the CLIENT DNS plane (anti-leak), but the endpoint copy must be forced
// DIRECT — resolving the proxy's own server domain through the proxy is an
// impossible bootstrap loop. The client server keeps its proxy detour; the endpoint
// clone is direct.
func TestEndpointResolverIsAlwaysDirect(t *testing.T) {
m := &model.Model{
Globals: model.Globals{ResolverDefault: "cf", EndpointResolver: "cf"},
Nodes: []model.Node{
{Name: "vps", Enabled: true, URI: "ss://aes-256-gcm:secret@203.0.113.1:8388#vps"},
},
Resolvers: []model.Resolver{
// Client DNS deliberately routed THROUGH the proxy node.
{Name: "cf", Type: "udp", Address: "1.1.1.1", Detour: "node:vps"},
},
}
opts, _, err := GenerateWithWarnings(m)
if err != nil {
t.Fatalf("GenerateWithWarnings: %v", err)
}
if opts.Route == nil || opts.Route.DefaultDomainResolver == nil {
t.Fatalf("route.default_domain_resolver must be set; route=%+v", opts.Route)
}
epTag := endpointServerTag("cf")
if got := opts.Route.DefaultDomainResolver.Server; got != epTag {
t.Fatalf("default_domain_resolver.server = %q, want %q", got, epTag)
}
// Client "cf" server must detour through the proxy node "vps".
client, ok := dnsServerByTag(opts.DNS, "cf")
if !ok {
t.Fatalf("client resolver %q must be emitted", "cf")
}
if d := udpServerDetour(t, client); d != "vps" {
t.Fatalf("client resolver detour = %q, want the proxy node %q", d, "vps")
}
// The endpoint clone MUST be direct, regardless of the client detour.
ep, ok := dnsServerByTag(opts.DNS, epTag)
if !ok {
t.Fatalf("endpoint server %q must be emitted; servers=%+v", epTag, opts.DNS.Servers)
}
if d := udpServerDetour(t, ep); d != "direct" {
t.Fatalf("endpoint resolver MUST be bootstrap-direct, detour = %q, want %q", d, "direct")
}
}
// TestEndpointResolverAbsentUnset: with no endpoint_resolver configured the field
// stays unset (prior behaviour preserved, engine uses its built-in resolver).
func TestEndpointResolverAbsentUnset(t *testing.T) {
m := &model.Model{
Globals: model.Globals{ResolverDefault: "cf"},
Resolvers: []model.Resolver{
{Name: "cf", Type: "udp", Address: "1.1.1.1", Detour: "direct"},
},
}
opts, _, err := GenerateWithWarnings(m)
if err != nil {
t.Fatalf("GenerateWithWarnings: %v", err)
}
if opts.Route != nil && opts.Route.DefaultDomainResolver != nil {
t.Fatalf("default_domain_resolver must be unset when the feature is off, got %+v", opts.Route.DefaultDomainResolver)
}
}
+7 -6
View File
@@ -83,9 +83,11 @@ func TestGroupStrategyBuildsExpectedOutbound(t *testing.T) {
{"roundrobin", C.TypeURLTest, C.URLTestModeRoundRobin, 0},
{"round_robin", C.TypeURLTest, C.URLTestModeRoundRobin, 0},
{"failover", C.TypeURLTest, C.URLTestModeRoundRobin, 1},
// `random` and `leastload` are DELETED, not special-cased: they are now
// ordinary unknown values and fall to least_test like any typo.
{"random", C.TypeURLTest, C.URLTestModeLeastTest, 0},
// `random` is the engine's real random mode over a pool spanning ALL members
// (2 here) — a uniform draw over the live set (see groupOutbound).
{"random", C.TypeURLTest, C.URLTestModeRandom, 2},
// `leastload` is DELETED, not special-cased: it is now an ordinary unknown
// value and falls to least_test like any typo.
{"leastload", C.TypeURLTest, C.URLTestModeLeastTest, 0},
{"whatever", C.TypeURLTest, C.URLTestModeLeastTest, 0},
// Case and stray whitespace resolve to what the operator typed.
@@ -137,7 +139,7 @@ func TestGroupStrategyBuildsExpectedOutbound(t *testing.T) {
// TestGroupStrategyUnknownWarns: a strategy this engine does not have is a naming
// mistake and must be reported as such, naming what the group actually does.
func TestGroupStrategyUnknownWarns(t *testing.T) {
for _, s := range []string{"random", "leastload", "nonsense"} {
for _, s := range []string{"leastload", "nonsense"} {
_, warns, err := GenerateWithWarnings(twoNodeGroupModel(s))
if err != nil {
t.Fatalf("strategy %q: Generate: %v", s, err)
@@ -153,7 +155,7 @@ func TestGroupStrategyUnknownWarns(t *testing.T) {
// silent. A warning on every apply for a correct config is noise, and noise is
// what makes the real warnings get ignored.
func TestGroupStrategyHonestValuesDoNotWarn(t *testing.T) {
for _, s := range []string{"", "single", "manual", "leastping", "roundrobin", "round_robin"} {
for _, s := range []string{"", "single", "manual", "leastping", "roundrobin", "round_robin", "random"} {
_, warns, err := GenerateWithWarnings(twoNodeGroupModel(s))
if err != nil {
t.Fatalf("strategy %q: Generate: %v", s, err)
@@ -637,7 +639,6 @@ func TestChainInvalidHopKindWarnsWithReason(t *testing.T) {
}{
{"direct", []string{"terminal route target"}},
{"block", []string{"terminal route target"}},
{"chain:other", []string{"cannot be a hop of another chain"}},
{"ghost", []string{"not an enabled/parseable node"}},
}
for _, tc := range cases {
+22 -13
View File
@@ -83,16 +83,14 @@ type badoptionDuration = badoption.Duration
// parseDuration parses a Go-style duration ("60s", "5m", "1h30m"). A bare
// integer is interpreted as seconds. Returns ok=false for empty/invalid input.
//
// It is a thin wrapper over model.ParseDuration and MUST stay one: this package
// and the model/validator layer both read the same UCI strings (probe_interval,
// update_interval, sweep_interval), so a second copy of the grammar is a second
// place for it to drift. The only thing added here is the option-layer type.
func parseDuration(s string) (badoption.Duration, bool) {
s = strings.TrimSpace(s)
if s == "" {
return 0, false
}
if v, err := strconv.Atoi(s); err == nil {
return badoption.Duration(time.Duration(v) * time.Second), true
}
d, err := time.ParseDuration(s)
if err != nil {
d, ok := model.ParseDuration(s)
if !ok {
return 0, false
}
return badoption.Duration(d), true
@@ -188,10 +186,21 @@ type builder struct {
// effectiveComputed; buildRoute consumes them. For a model with no
// profiles/presets effectiveRules is an exact copy of b.m.Rules (identical
// generate output) and the default overrides are empty.
effectiveRules []model.Rule // preset packs prepended + active-profile enable/disable applied
profileDefaultTarget string // active profile's DefaultTarget override ("" => none)
profileDefaultEgress string // active profile's DefaultEgress override ("" => none)
effectiveComputed bool // applyProfilesAndPresets has run
effectiveRules []model.Rule // preset packs prepended + active-profile enable/disable applied
profileDefaultTarget string // active profile's DefaultTarget override ("" => none)
profileDefaultEgress string // active profile's DefaultEgress override ("" => none)
profileEndpointResolver string // active profile's EndpointResolver override ("" => none)
effectiveComputed bool // applyProfilesAndPresets has run
// Endpoint resolver (route.default_domain_resolver): the bootstrap-direct DNS
// server used ONLY to resolve proxy outbounds' server DOMAINS. Computed once by
// endpointResolver() and shared between buildRoute (which references the tag) and
// buildDNS (which injects the server). endpointResolverServer is the synthetic
// direct-forced DNS server to append to the DNS plane, or nil when the feature is
// off / the named resolver does not exist. Guarded by endpointResolverComputed.
endpointResolverTag string
endpointResolverServer *option.DNSServerOptions
endpointResolverComputed bool
warnings []string
}
+57 -10
View File
@@ -19,22 +19,27 @@ import (
// # What the engine ACTUALLY has (and what it does not)
//
// There are exactly TWO group outbound types (constant/proxy.go): `selector` and
// `urltest`, and urltest has exactly TWO modes: `least_test` (upstream) and
// `round_robin` (lx SPEC 019). Everything else the model/panel offers has to map
// onto those four behaviours, so the mapping — and, where a value promises
// something the engine cannot do, the WARNING — is the whole contract:
// `urltest`, and urltest has THREE modes: `least_test` (upstream), `round_robin`
// and `random` (lx SPEC 019). Everything else the model/panel offers has to map
// onto those behaviours, so the mapping — and, where a value promises something the
// engine cannot do, the WARNING — is the whole contract:
//
// single / manual / "" -> selector (WORKS; manual is a synonym of single)
// leastping -> least_test (WORKS)
// roundrobin -> round_robin (WORKS)
// random -> random, pool = all members (WORKS — true random over live)
// failover -> round_robin, pool 1 (WORKS — see failoverBalancer)
// anything else -> least_test, warned as unknown
//
// That list is exhaustive on purpose. `random` and `leastload` used to have named
// branches here that mapped them to least_test with an apologetic warning; they
// were fictions — the engine has no randomised picker and no load metric of any
// kind — so they are GONE rather than dressed up. They now land in the unknown
// branch like any other typo. See normalizeGroupStrategy (the single mapping
// That list is exhaustive on purpose. `leastload` used to have a named branch here
// that mapped it to least_test with an apologetic warning; it was a fiction — the
// engine has no load metric of any kind — so it is GONE rather than dressed up and
// now lands in the unknown branch like any other typo.
//
// `random` is a REAL engine mode (protocol/group, lx SPEC 019 v2): a uniformly-random
// LIVE node is drawn per connection from a pool sized to hold every member. It is no
// longer an approximation of round-robin-over-all — groupOutbound documents the
// pool choice and its probing cost. See normalizeGroupStrategy (the single mapping
// point) and warnGroupStrategy (the single diagnostics point).
//
// # selector does NOT check liveness
@@ -209,6 +214,7 @@ const (
strategySelector groupStrategyMode = iota // selector: one fixed member, no probing
strategyLeastTest // urltest, mode least_test
strategyRoundRobin // urltest, mode round_robin (lx SPEC 019)
strategyRandom // urltest, mode random with a pool = all members
strategyFailover // urltest, round_robin with a 1-slot pool
)
@@ -310,6 +316,11 @@ func normalizeGroupStrategy(s string) groupStrategyMode {
return strategySelector
case "roundrobin", "round_robin":
return strategyRoundRobin
case "random":
// A real engine mode: uniform random over the LIVE members of a pool sized to
// hold every member (see groupOutbound). Honest whitelist in warnGroupStrategy
// (not warned as unknown).
return strategyRandom
case "failover":
return strategyFailover
default:
@@ -330,8 +341,11 @@ func normalizeGroupStrategy(s string) groupStrategyMode {
func (b *builder) warnGroupStrategy(g model.Group, members []string) {
switch strings.ToLower(strings.TrimSpace(g.Strategy)) {
case "", "single", "manual", "leastping", "least_ping", "least_test", "urltest",
"roundrobin", "round_robin":
"roundrobin", "round_robin", "random":
// Honest values (and their synonyms): what is emitted is what was asked for.
// "random" is a real engine mode (uniform random over the live members;
// groupOutbound documents the pool choice), so it is not warned as unknown;
// the operator asked to spread across members and gets exactly that.
return
case "failover":
b.warnFailoverGroup(g, members)
@@ -393,6 +407,39 @@ func (b *builder) groupOutbound(g model.Group, tag string, members []string) opt
Mode: C.URLTestModeRoundRobin,
},
}
case strategyRandom:
// A TRUE uniform random picker: mode=random (protocol/group, lx SPEC 019 v2) draws a
// uniformly-random LIVE node from the pool for every connection. Stickiness is off by
// default in random mode (a per-connection independent draw makes it meaningless), so no
// balancer.sticky_hash is emitted.
//
// Pool: len(members) — a slot for EVERY member, so the draw ranges over the whole live
// set instead of a default pool of 3. This is the "размазка по всем
// рабочим прокси" the model asks for: every reachable member is a
// candidate every connection.
//
// COST of the full-size pool: the health-check (balancePoolFirstLive, pool_tolerance 0)
// re-probes every pooled member each interval, so a full-size pool means one HEAD request
// per member per interval — a 376-node subscription group probes all 376 nodes per tick.
// That is the deliberate price of "spread across ALL working proxies"; the payoff is that
// pick() then draws only over the LIVE subset (a dead member keeps its slot for the
// never-shrink invariant but is SKIPPED by selection — no more connections rotating onto a
// dead exit, which is what the old round_robin approximation could not avoid). Operators
// who want the spread without probing an entire subscription can hand-pick a smaller
// member list or use roundrobin (default pool 3).
return option.Outbound{
Type: C.TypeURLTest,
Tag: tag,
Options: &option.URLTestOutboundOptions{
Outbounds: members,
URL: groupProbeURL(b.m.Globals, g),
Interval: groupInterval(b.m.Globals, g),
Mode: C.URLTestModeRandom,
Balancer: &option.URLTestBalancerOptions{
Pool: len(members),
},
},
}
case strategyFailover:
// A priority list: hold members[0] until it stops answering the probe, then
// move to the first member below it that does, and stay there. See
+67
View File
@@ -2,6 +2,7 @@ package generate
import (
"reflect"
"strings"
"testing"
C "github.com/sagernet/sing-box/constant"
@@ -10,6 +11,31 @@ import (
"github.com/sagernet/sing-box/shater/model"
)
// buildGroupOutbound wires a model through the real outbound builder and returns
// the built group outbound with the given tag, plus the builder's warnings.
func buildGroupOutbound(t *testing.T, m *model.Model, tag string) (option.Outbound, []string) {
t.Helper()
b := newBuilder(m)
b.buildOutboundsAndEndpoints()
groups := b.buildGroups()
for _, o := range groups {
if o.Tag == tag {
return o, b.warnings
}
}
t.Fatalf("built group %q not found (warnings: %v)", tag, b.warnings)
return option.Outbound{}, nil
}
func hasWarn(warns []string, sub string) bool {
for _, w := range warns {
if strings.Contains(w, sub) {
return true
}
}
return false
}
// buildMembers wires a model through the real outbound builder (which populates
// b.nodeTags for every enabled, parseable node) and returns groupMembers for the
// named group. This exercises the same nodeTags contract groupMembers depends on
@@ -170,3 +196,44 @@ func TestManualGroupUnchanged(t *testing.T) {
t.Fatalf("manual group: groupMembers = %v, want %v", got, want)
}
}
// TestRandomStrategyRoundRobinAllMembers proves strategy=random maps to the engine's
// real random mode whose pool spans EVERY member (stickiness is off by default in
// random mode, so none is emitted). It must NOT warn as unknown.
func TestRandomStrategyRoundRobinAllMembers(t *testing.T) {
m := &model.Model{
Nodes: []model.Node{
{Name: "n1", Enabled: true, URI: ss("203.0.113.1")},
{Name: "n2", Enabled: true, URI: ss("203.0.113.2")},
{Name: "n3", Enabled: true, URI: ss("203.0.113.3")},
},
Groups: []model.Group{
{Name: "spread", Source: "manual", Strategy: "random", Nodes: []string{"n1", "n2", "n3"}},
},
}
o, warns := buildGroupOutbound(t, m, "spread")
if o.Type != C.TypeURLTest {
t.Fatalf("random group: type = %q, want urltest", o.Type)
}
ut, ok := o.Options.(*option.URLTestOutboundOptions)
if !ok {
t.Fatalf("random group: options type = %T, want *URLTestOutboundOptions", o.Options)
}
if ut.Mode != C.URLTestModeRandom {
t.Fatalf("random group: mode = %q, want %q", ut.Mode, C.URLTestModeRandom)
}
if ut.Balancer == nil {
t.Fatalf("random group: balancer is nil, want pool=%d", len(ut.Outbounds))
}
if ut.Balancer.Pool != len(ut.Outbounds) {
t.Fatalf("random group: balancer.pool = %d, want len(members) = %d", ut.Balancer.Pool, len(ut.Outbounds))
}
// Random mode disables stickiness in the engine by default, so generate emits none.
if len(ut.Balancer.StickyHash) != 0 {
t.Fatalf("random group: sticky_hash = %v, want none (random defaults sticky off)", ut.Balancer.StickyHash)
}
if hasWarn(warns, "unknown strategy") {
t.Fatalf("random group: warned as unknown strategy; warnings: %v", warns)
}
}
+3
View File
@@ -54,6 +54,9 @@ func (b *builder) applyProfilesAndPresets() {
b.applyProfileRuleOverrides(eff, prof)
b.profileDefaultTarget = strings.TrimSpace(prof.DefaultTarget)
b.profileDefaultEgress = strings.TrimSpace(prof.DefaultEgress)
// Per-profile endpoint-resolver override (by the active WAN profile). Consumed
// by endpointResolver() with priority OVER Globals.EndpointResolver.
b.profileEndpointResolver = strings.TrimSpace(prof.EndpointResolver)
}
b.effectiveRules = eff
+14 -1
View File
@@ -176,11 +176,24 @@ func (b *builder) buildRoute() *option.RouteOptions {
rules = append(rules, b.dohBlockRouteRules()...)
rules = append(rules, general...)
return &option.RouteOptions{
route := &option.RouteOptions{
Rules: rules,
Final: final,
RuleSet: b.routeRuleSets,
}
// Endpoint resolver: point route.default_domain_resolver at the bootstrap-direct
// server that resolves proxy outbounds' server DOMAINS (endpointResolver builds it
// and buildDNS injects it). Set ONLY when a tag was actually produced — referencing
// a non-existent DNS server aborts box.New, exactly like a dangling Final. Leaving
// it unset when the feature is off preserves prior behaviour (engine built-in), but
// once set it also clears the engine's "missing route.default_domain_resolver"
// deprecation warning.
if tag := b.endpointResolver(); tag != "" {
route.DefaultDomainResolver = &option.DomainResolveOptions{Server: tag}
}
return route
}
// sniffRule is the leading protocol-sniff action rule (sniff all protocols).
+173 -7
View File
@@ -21,6 +21,7 @@ import (
"github.com/sagernet/sing-box/option"
"github.com/sagernet/sing-box/shater/model"
"github.com/sagernet/sing-box/shater/netplane"
)
// ruleModel wraps a set of rules in a minimal model with one usable node.
@@ -583,9 +584,14 @@ func TestCatchAllRuleEgressBecomesFinal(t *testing.T) {
// --- cycles ------------------------------------------------------------------
// TestChainCyclesTerminateFailClosed: a chain referencing itself (directly or
// through another chain) must be refused via the chainBuilt memo guard rather
// than recursing, and the referring rule must fail CLOSED.
// TestChainCyclesTerminateFailClosed: now that a `chain:<name>` hop is FLATTENED
// (the sub-chain's hops spliced in place), a chain that references itself — directly
// (self), through another chain (x -> y -> x), or as one of its own hops (mix) — is
// a reference cycle that expandHops's `seen` guard must catch and refuse, failing the
// whole chain CLOSED before any wrapper is materialised. No route rule is emitted, the
// referring traffic falls through to the fail-closed Final, and — because the flatten
// runs to completion before the wrapper loop — no orphan `chain-` outbound is ever
// created (there is nothing to roll back).
func TestChainCyclesTerminateFailClosed(t *testing.T) {
g := model.DefaultGlobals()
g.KillSwitch = "closed"
@@ -614,14 +620,174 @@ func TestChainCyclesTerminateFailClosed(t *testing.T) {
if opts.Route.Final != tagBlock {
t.Fatalf("Final = %q, want block", opts.Route.Final)
}
// The rolled-back "mix" chain must leave no orphan hop outbounds behind.
// Fail-closed BEFORE any wrapper is built: no orphan hop outbounds must remain.
for _, ob := range opts.Outbounds {
if strings.HasPrefix(ob.Tag, "chain-") {
t.Fatalf("orphan chain outbound after rollback: %q", ob.Tag)
t.Fatalf("orphan chain outbound after a refused cycle: %q", ob.Tag)
}
}
if !routeWarnsHave(warns, "unresolvable") {
t.Fatalf("expected an unresolvable-chain warning, got %v", warns)
if !routeWarnsHave(warns, "cycle") {
t.Fatalf("expected a reference-cycle warning, got %v", warns)
}
}
// --- chain-as-hop: reusable / composable chains ------------------------------
// chainFlattenModel is a model with three usable nodes and a closed kill-switch,
// ready for the chain-flatten cases below to attach chains + one targeting rule.
func chainFlattenModel() *model.Model {
g := model.DefaultGlobals()
g.KillSwitch = "closed"
return &model.Model{
Globals: g,
Nodes: []model.Node{
{Name: "a", Enabled: true, URI: ss("203.0.113.1")},
{Name: "b", Enabled: true, URI: ss("203.0.113.2")},
{Name: "c", Enabled: true, URI: ss("203.0.113.3")},
},
}
}
// TestChainFlattenInnerNoEgress: a `chain:inner` hop with NO egress is spliced in
// place, so outer=[node:a, chain:inner, node:c] with inner=[node:b] materialises the
// flat path a -> b -> c — three wrappers, Detour-linked, exit = c.
func TestChainFlattenInnerNoEgress(t *testing.T) {
m := chainFlattenModel()
m.Chains = []model.Chain{
{Name: "inner", Hops: []string{"node:b"}},
{Name: "outer", Hops: []string{"node:a", "chain:inner", "node:c"}},
}
m.Rules = []model.Rule{{Name: "via", Enabled: true, Order: 10, DstPort: "443", Target: "chain:outer"}}
opts, warns, err := GenerateWithWarnings(m)
if err != nil {
t.Fatalf("Generate: %v", err)
}
if len(warns) != 0 {
t.Fatalf("unexpected warnings: %v", warns)
}
got, ok := generalRouteOutbound(opts.Route)
if !ok || got != "chain-outer-h3" {
t.Fatalf("rule must route to the flattened exit chain-outer-h3, got %q (ok=%v)", got, ok)
}
// Detour direction: h3(c) -> h2(b) -> h1(a) -> "" (a dials directly).
for _, w := range []struct{ tag, detour string }{
{"chain-outer-h1", ""},
{"chain-outer-h2", "chain-outer-h1"},
{"chain-outer-h3", "chain-outer-h2"},
} {
ob := obByTag(opts, w.tag)
if ob == nil {
t.Fatalf("%s not emitted; outbounds=%v", w.tag, outboundTags(opts))
}
if d := outboundDetour(t, ob); d != w.detour {
t.Fatalf("%s.detour = %q, want %q", w.tag, d, w.detour)
}
}
}
// TestChainFlattenInnerEgressAtEntry: a `chain:inner` hop whose inner chain LEADS
// with an egress is legal only at the outer entry (position 0). There the inner's
// egress becomes the whole chain's entry interface: outer=[chain:inner, node:c] with
// inner=[egress:wan2, node:b] enters over wan2, so L1 (b) detours through the egress.
func TestChainFlattenInnerEgressAtEntry(t *testing.T) {
m := chainFlattenModel()
m.Egresses = []model.Egress{{Name: "wan2", Type: "direct"}}
m.Chains = []model.Chain{
{Name: "inner", Hops: []string{"egress:wan2", "node:b"}},
{Name: "outer", Hops: []string{"chain:inner", "node:c"}},
}
m.Rules = []model.Rule{{Name: "via", Enabled: true, Order: 10, DstPort: "443", Target: "chain:outer"}}
opts, warns, err := GenerateWithWarnings(m)
if err != nil {
t.Fatalf("Generate: %v", err)
}
if len(warns) != 0 {
t.Fatalf("unexpected warnings: %v", warns)
}
got, ok := generalRouteOutbound(opts.Route)
if !ok || got != "chain-outer-h2" {
t.Fatalf("rule must route to chain-outer-h2 (b->c), got %q (ok=%v)", got, ok)
}
h1 := obByTag(opts, "chain-outer-h1")
if h1 == nil {
t.Fatalf("chain-outer-h1 not emitted; outbounds=%v", outboundTags(opts))
}
want := netplane.EgressOutboundTag("wan2")
if d := outboundDetour(t, h1); d != want {
t.Fatalf("chain-outer-h1.detour = %q, want the entry egress %q", d, want)
}
}
// TestChainFlattenInnerEgressMidPathFailsClosed: the SAME egress-leading inner chain
// spliced at a NON-entry position (outer=[node:a, chain:inner]) is a contradiction —
// a chain that ENTERS over an egress cannot be a middle segment. It must be refused
// fail-closed: no route rule, Final stays block, no orphan wrappers, and a warning.
func TestChainFlattenInnerEgressMidPathFailsClosed(t *testing.T) {
m := chainFlattenModel()
m.Egresses = []model.Egress{{Name: "wan2", Type: "direct"}}
m.Chains = []model.Chain{
{Name: "inner", Hops: []string{"egress:wan2", "node:b"}},
{Name: "outer", Hops: []string{"node:a", "chain:inner"}},
}
m.Rules = []model.Rule{{Name: "via", Enabled: true, Order: 10, DstPort: "443", Target: "chain:outer"}}
opts, warns, err := GenerateWithWarnings(m)
if err != nil {
t.Fatalf("Generate: %v", err)
}
if n := len(generalRules(opts.Route)); n != 0 {
t.Fatalf("mid-path egress-entry chain must yield no route rule, got %d", n)
}
if opts.Route.Final != tagBlock {
t.Fatalf("Final = %q, want block", opts.Route.Final)
}
for _, ob := range opts.Outbounds {
if strings.HasPrefix(ob.Tag, "chain-") {
t.Fatalf("orphan chain outbound after fail-closed refusal: %q", ob.Tag)
}
}
if !routeWarnsHave(warns, "FAIL-CLOSED") {
t.Fatalf("expected a fail-closed warning, got %v", warns)
}
}
// TestChainFlattenDepth3: nesting deeper than two levels (A -> B -> C) must flatten
// through the recursion. chainA=[node:a, chain:chainB], chainB=[node:b, chain:chainC],
// chainC=[node:c] collapses to the flat path a -> b -> c.
func TestChainFlattenDepth3(t *testing.T) {
m := chainFlattenModel()
m.Chains = []model.Chain{
{Name: "chainC", Hops: []string{"node:c"}},
{Name: "chainB", Hops: []string{"node:b", "chain:chainC"}},
{Name: "chainA", Hops: []string{"node:a", "chain:chainB"}},
}
m.Rules = []model.Rule{{Name: "via", Enabled: true, Order: 10, DstPort: "443", Target: "chain:chainA"}}
opts, warns, err := GenerateWithWarnings(m)
if err != nil {
t.Fatalf("Generate: %v", err)
}
if len(warns) != 0 {
t.Fatalf("unexpected warnings: %v", warns)
}
got, ok := generalRouteOutbound(opts.Route)
if !ok || got != "chain-chainA-h3" {
t.Fatalf("rule must route to the flattened exit chain-chainA-h3, got %q (ok=%v)", got, ok)
}
for _, w := range []struct{ tag, detour string }{
{"chain-chainA-h1", ""},
{"chain-chainA-h2", "chain-chainA-h1"},
{"chain-chainA-h3", "chain-chainA-h2"},
} {
ob := obByTag(opts, w.tag)
if ob == nil {
t.Fatalf("%s not emitted; outbounds=%v", w.tag, outboundTags(opts))
}
if d := outboundDetour(t, ob); d != w.detour {
t.Fatalf("%s.detour = %q, want %q", w.tag, d, w.detour)
}
}
}
+122
View File
@@ -0,0 +1,122 @@
package model
import (
"fmt"
"strconv"
"strings"
"time"
)
// Duration parsing for the interval knobs, and the resolution of the one interval
// whose meaning is more than a number (Globals.SweepInterval).
// ParseDuration parses a Go-style duration ("60s", "5m", "1h30m"). A BARE INTEGER is
// accepted as SECONDS, which is what OpenWrt configs conventionally carry. Returns
// ok=false for an empty or unparseable value — the caller decides what that means, and
// the two callers deliberately decide differently (a missing probe_interval falls back
// to the engine default; a missing sweep_interval does too, but a MALFORMED one is
// warned about first).
//
// This is the one algorithm every interval in the project is read with. It is defined
// here, in the leaf package, because generate/, apply/ and engine/ all need it and all
// import model — the alternative, a private copy per consumer, is how "30s" ends up
// meaning one thing in one knob and something else in the next.
//
// NOTE: generate/generate.go still carries a byte-identical private parseDuration
// (returning badoption.Duration) that predates this one. It should be collapsed into a
// thin wrapper over this function; the behaviour is already identical and pinned by
// TestParseDuration.
func ParseDuration(s string) (time.Duration, bool) {
s = strings.TrimSpace(s)
if s == "" {
return 0, false
}
if v, err := strconv.Atoi(s); err == nil {
return time.Duration(v) * time.Second, true
}
d, err := time.ParseDuration(s)
if err != nil {
return 0, false
}
return d, true
}
// SweepDisabledValues turn the background health sweep OFF.
//
// The vocabulary deliberately mirrors SilentLogLevels ("none"/"off"/"disabled"), for
// the reason given there: an operator who has learned that "off" silences the log
// should not have to discover that a different word is required to stop the sweep.
// "silent" is dropped (it says nothing about a schedule) and "0" is added, because for
// a value that is otherwise a duration, zero is the obvious way to spell "never".
//
// It is a closed set on purpose. Anything outside it that also fails to parse as a
// duration is a MISTAKE, and SweepSchedule treats it as one — see there.
var SweepDisabledValues = []string{"0", "off", "none", "disabled"}
// SweepIntervalMin is the floor for Globals.SweepInterval.
//
// It equals the per-probe timeout in shater/engine (probeAllTimeout, 5s), and that is
// the principled reason for the number rather than a round guess: a tick shorter than
// the time a SINGLE probe may take cannot complete a batch, so the extra ticks are
// dropped by the sweeper's busy guard and buy nothing. What they do buy, whenever the
// probes ARE fast, is load: the sweep's cost is batch/interval probes per second, so
// honouring "1s" literally would run the default batch ten times faster than designed,
// forever, on a router whose CPU and uplink belong to the user's traffic.
//
// A value below the floor is RAISED to it with a warning rather than rejected. Refusing
// the config would be wildly disproportionate for a tuning knob — the kill-switch is
// closed while the engine is down, so "refuse to start" means "the LAN is offline" —
// and accepting it literally would be an invisible, permanent drain. Clamping is the
// only option that is both safe and visible.
const SweepIntervalMin = 5 * time.Second
// SweepSchedule resolves Globals.SweepInterval into what the engine should actually do.
//
// enabled false only when the value is explicitly one of SweepDisabledValues.
// interval the tick period; 0 means "use the engine's own default", which is what an
// EMPTY value (and any value we could not honour) resolves to. Returning 0
// rather than a number keeps the default in exactly one place — engine's
// sweep.go — instead of duplicating it here where it would drift.
// warn operator-facing text when the value could not be taken at face value; ""
// when it could.
//
// The three outcomes for a non-empty value are kept strictly apart:
//
// a disabling word -> off, no warning. The operator asked for this.
// a parseable duration -> on at that tick, clamped up to SweepIntervalMin with a
// warning if it is below the floor.
// anything else -> on at the DEFAULT tick, with a warning. Note what this is not:
// an unparseable value must never be read as "off". "I could not understand
// you" and "you asked me to stop" are different statements, and collapsing
// them would silently disable health data because of a typo — the operator
// would then see their own value echoed back in UCI and the panel and
// conclude the sweep was running.
//
// Pure, so both the validator (ValidateGlobals) and the consumer
// (apply.configureSweep) resolve the value through this one function and cannot drift.
func (g Globals) SweepSchedule() (interval time.Duration, enabled bool, warn string) {
raw := strings.TrimSpace(g.SweepInterval)
if raw == "" {
return 0, true, ""
}
if inSet(raw, SweepDisabledValues) {
return 0, false, ""
}
d, ok := ParseDuration(raw)
if !ok {
return 0, true, fmt.Sprintf(
"sweep interval %q is not a duration and is applied as the default tick "+
"(the background health sweep stays ON — an unreadable value is not "+
"read as \"off\"); use a duration like 30s/5m/1h, or %s to turn the "+
"sweep off.",
g.SweepInterval, strings.Join(SweepDisabledValues, "/"))
}
if d < SweepIntervalMin {
return SweepIntervalMin, true, fmt.Sprintf(
"sweep interval %q is below the %s floor and is applied as %s; a tick "+
"shorter than one probe's own timeout cannot finish a batch, and on a "+
"router the extra ticks are pure load.",
g.SweepInterval, SweepIntervalMin, SweepIntervalMin)
}
return d, true, ""
}
+129
View File
@@ -0,0 +1,129 @@
package model
import (
"strings"
"testing"
"time"
)
// ParseDuration is the ONE algorithm every interval in the project is read with, so its
// exact behaviour is a contract, not an implementation detail: a bare integer means
// SECONDS (the OpenWrt convention), Go duration syntax works, and anything else is a
// clean miss rather than a zero.
func TestParseDuration(t *testing.T) {
cases := []struct {
in string
want time.Duration
ok bool
}{
{"30s", 30 * time.Second, true},
{"5m", 5 * time.Minute, true},
{"1h", time.Hour, true},
{"1h30m", 90 * time.Minute, true},
{"60", 60 * time.Second, true}, // bare integer == seconds
{"0", 0, true},
{" 45s ", 45 * time.Second, true}, // trimmed
{"", 0, false},
{" ", 0, false},
{"soon", 0, false},
{"5 minutes", 0, false},
{"m5", 0, false},
}
for _, c := range cases {
got, ok := ParseDuration(c.in)
if got != c.want || ok != c.ok {
t.Errorf("ParseDuration(%q) = (%v,%v), want (%v,%v)", c.in, got, ok, c.want, c.ok)
}
}
}
// SweepSchedule keeps the three outcomes strictly apart. The one that matters most is
// the last: an unreadable value must NOT be read as "off".
func TestSweepSchedule(t *testing.T) {
cases := []struct {
name string
in string
wantInterval time.Duration
wantEnabled bool
wantWarn bool
}{
{
name: "unset means ON at the engine default (interval 0 = 'engine decides')",
in: "", wantInterval: 0, wantEnabled: true,
},
{name: "explicit duration", in: "30s", wantInterval: 30 * time.Second, wantEnabled: true},
{name: "minutes", in: "5m", wantInterval: 5 * time.Minute, wantEnabled: true},
{name: "bare seconds", in: "45", wantInterval: 45 * time.Second, wantEnabled: true},
{name: "exactly at the floor is honoured as-is", in: "5s", wantInterval: SweepIntervalMin, wantEnabled: true},
// Disabling: explicit, and silent (the operator asked for it).
{name: "off by zero", in: "0", wantEnabled: false},
{name: "off", in: "off", wantEnabled: false},
{name: "none", in: "none", wantEnabled: false},
{name: "disabled", in: "disabled", wantEnabled: false},
{name: "disabling words are case- and space-insensitive", in: " OFF ", wantEnabled: false},
// Below the floor: clamped UP, still on, and said out loud.
{
name: "1s is raised to the floor with a warning",
in: "1s", wantInterval: SweepIntervalMin, wantEnabled: true, wantWarn: true,
},
{
name: "a bare 2 (seconds) is raised too",
in: "2", wantInterval: SweepIntervalMin, wantEnabled: true, wantWarn: true,
},
// Unparseable: warned, defaulted, and CRITICALLY still enabled.
{
name: "garbage warns and falls back to the default tick, still ON",
in: "soon", wantInterval: 0, wantEnabled: true, wantWarn: true,
},
{
name: "a plausible-looking typo is still not 'off'",
in: "5 minutes", wantInterval: 0, wantEnabled: true, wantWarn: true,
},
}
for _, c := range cases {
interval, enabled, warn := Globals{SweepInterval: c.in}.SweepSchedule()
if interval != c.wantInterval || enabled != c.wantEnabled {
t.Errorf("%s: SweepSchedule(%q) = (%v, enabled=%v), want (%v, enabled=%v)",
c.name, c.in, interval, enabled, c.wantInterval, c.wantEnabled)
}
if (warn != "") != c.wantWarn {
t.Errorf("%s: SweepSchedule(%q) warn = %q, want warn=%v", c.name, c.in, warn, c.wantWarn)
}
}
}
// A disabling value must never produce a warning — the operator made a deliberate,
// supported choice, and nagging about it is how a warning list becomes noise nobody
// reads.
func TestSweepScheduleDisablingIsQuiet(t *testing.T) {
for _, v := range SweepDisabledValues {
if _, enabled, warn := (Globals{SweepInterval: v}).SweepSchedule(); enabled || warn != "" {
t.Errorf("%q: enabled=%v warn=%q, want disabled and quiet", v, enabled, warn)
}
}
}
// The warning surfaces through ValidateGlobals (which is what reaches the log and the
// panel), and it must name the offending value and the way to turn the sweep off.
func TestValidateGlobalsSweepInterval(t *testing.T) {
ws := ValidateGlobals(Globals{SweepInterval: "soon"})
if !hasWarning(ws, "globals", "soon") {
t.Fatalf("no warning naming the bad value: %+v", ws)
}
joined := ""
for _, w := range ws {
joined += w.Message
}
if !strings.Contains(joined, "off") {
t.Errorf("the warning must tell the operator how to disable the sweep: %q", joined)
}
// A good value, and a deliberate "off", are both silent.
for _, v := range []string{"", "30s", "off"} {
if ws := ValidateGlobals(Globals{SweepInterval: v}); len(ws) != 0 {
t.Errorf("sweep_interval=%q warned unnecessarily: %+v", v, ws)
}
}
}
+35
View File
@@ -109,12 +109,36 @@ type Globals struct {
ConfirmTimeout int
ResolverDefault string // resolver name consulted FIRST (servers list order)
ResolverFallback string // resolver name consulted LAST (fallback)
// EndpointResolver names the `config resolver` used to resolve the SERVER
// DOMAINS of proxy configs (vless/awg endpoints) — sing-box
// route.default_domain_resolver, a bootstrap resolver that is always direct.
// "" = not set. Consumed by the generator (generate/*, another agent); this is
// only the model contract.
EndpointResolver string
ProbeURL string // health-probe URL (default gstatic/generate_204)
ProbeInterval string // probe interval (e.g. 60s)
SchemaVersion int // UCI schema revision (0 = pre-versioned legacy)
ActiveProfile string // last profile switched to (display bookkeeping)
PanelPort int // admin-panel HTTP port; 0 = use the built-in default (8088)
// SweepInterval is the tick period of the background health sweep — the
// scheduled walk that keeps every node's health fresh so the panel and the
// group strategies have current data (see shater/engine/sweep.go).
//
// "" means ENABLED at the engine's default tick, not off. That default matters:
// urltest groups probe only while they are being used and selector groups never
// probe at all, so without the sweep a 376-node subscription reads almost
// entirely "untested" and the per-group health numbers stay empty until somebody
// presses "Test all nodes". Shipping the thing disabled would leave the problem
// it exists to solve exactly as it was.
//
// Values: a duration ("30s", "5m", "1h") or a bare integer of seconds, parsed by
// ParseDuration like every other interval in the model; any of
// SweepDisabledValues to turn the sweep off; anything else is warned about and
// falls back to the default (see SweepSchedule — an unparseable value is NOT
// silently treated as "off"). Below SweepIntervalMin it is raised to that floor.
SweepInterval string
// Geo-data source for `source=geosite`/`source=geoip` rule-sets. Resolved by
// generate.SetGeoProvider / GeoRuleSetURL; this package only carries the values
// (model must not import generate — generate imports model).
@@ -144,6 +168,12 @@ type Globals struct {
DNSFilter bool // master enable for the DNS bl/allow-list filter (D15); default false (opt-in)
DNSIntercept bool // force ALL LAN plaintext DNS (:53) through the engine, incl. router-addressed queries; default false (opt-in)
BlockDoH bool // block known public DoH resolvers (:443) so clients fall back to plaintext :53 (which the engine catches); default false (opt-in)
// GroupHealth is the master-switch of OUR background group health-sweep (the scheduled
// probe that keeps a group's member delays fresh, see apply.configureSweep); default TRUE
// (opt-out). Off disables the sweep exactly like SweepInterval="off" while KEEPING the
// interval value intact. It does NOT touch sing-box's own urltest probes inside a group —
// those live their own life in generate/*; only our sweep is gated here.
GroupHealth bool
// Untunnelable is the policy for LAN traffic that CANNOT be carried by the
// tunnel: everything that is not TCP or UDP. Kernel TPROXY only diverts those
@@ -207,6 +237,7 @@ func DefaultGlobals() Globals {
IPv6: true,
FwmarkBase: 0x2000,
TableBase: 0x2000,
GroupHealth: true,
StatsBackend: "memory",
StatsRingSize: 200,
StatsTimelineMinutes: 60,
@@ -502,6 +533,10 @@ type Profile struct {
DisableRules []string // rule names to disable
DefaultTarget string // override the default catch-all target (group:/node:/chain:/direct/block)
DefaultEgress string // override the default egress binding
// EndpointResolver is a per-profile override of the endpoint resolver (keyed by
// active WAN: SIM->yandex, WiFi->DoH). "" = no override. Consumed by the
// generator (another agent); model contract only.
EndpointResolver string
}
// Ruleset is a `config ruleset` (reusable domain/ip list).
+3
View File
@@ -13,6 +13,9 @@ func TestDefaultGlobals(t *testing.T) {
if g.FwmarkBase != 0x2000 || g.TableBase != 0x2000 {
t.Fatalf("default bases = %#x/%#x, want 0x2000/0x2000", g.FwmarkBase, g.TableBase)
}
if !g.GroupHealth {
t.Fatalf("default group_health = %v, want true (opt-out)", g.GroupHealth)
}
}
func TestEffectiveType(t *testing.T) {
+4
View File
@@ -52,8 +52,10 @@ func RenderUCIExport(m *Model) string {
w.intOpt("confirm_timeout", g.ConfirmTimeout)
w.strOpt("resolver_default", g.ResolverDefault)
w.strOpt("resolver_fallback", g.ResolverFallback)
w.strOpt("endpoint_resolver", g.EndpointResolver)
w.strOpt("probe_url", g.ProbeURL)
w.strOpt("probe_interval", g.ProbeInterval)
w.strOpt("sweep_interval", g.SweepInterval)
w.intOpt("schema_version", g.SchemaVersion)
w.strOpt("active_profile", g.ActiveProfile)
w.strOpt("geo_provider", g.GeoProvider)
@@ -65,6 +67,7 @@ func RenderUCIExport(m *Model) string {
w.boolOpt("dns_filter", g.DNSFilter)
w.boolOpt("dns_intercept", g.DNSIntercept)
w.boolOpt("block_doh", g.BlockDoH)
w.boolOpt("group_health", g.GroupHealth)
w.strOpt("untunnelable", g.Untunnelable)
w.strOpt("stats_backend", g.StatsBackend)
// The three stats-size knobs use intOptAlways (NOT the omit-zero intOpt) because 0
@@ -238,6 +241,7 @@ func RenderUCIExport(m *Model) string {
w.listOpt("disable_rule", p.DisableRules)
w.strOpt("default_target", p.DefaultTarget)
w.strOpt("default_egress", p.DefaultEgress)
w.strOpt("endpoint_resolver", p.EndpointResolver)
}
for _, res := range m.Resolvers {
+27
View File
@@ -291,6 +291,33 @@ func TestStatsBackendRoundTrip(t *testing.T) {
}
}
// TestGroupHealthRoundTrip pins the group_health master-switch: an EXPLICIT false
// (sweep off) survives WriteUCI->ReadUCI, and an ABSENT option falls back to the
// DefaultGlobals seed true (opt-out), never accidentally off.
func TestGroupHealthRoundTrip(t *testing.T) {
for _, v := range []bool{true, false} {
// group_health is only meaningful over an otherwise-default globals section, so
// start from DefaultGlobals and override just the field under test.
g := DefaultGlobals()
g.GroupHealth = v
got, err := ParseUCIExport(RenderUCIExport(&Model{Globals: g}))
if err != nil {
t.Fatalf("v=%v parse: %v", v, err)
}
if got.Globals.GroupHealth != v {
t.Fatalf("v=%v round-trip: group_health=%v, want %v", v, got.Globals.GroupHealth, v)
}
}
// Absent globals section => the DefaultGlobals seed true (opt-out).
got, err := ParseUCIExport("package shater\n")
if err != nil {
t.Fatal(err)
}
if !got.Globals.GroupHealth {
t.Fatalf("absent group_health = %v, want true (default, opt-out)", got.Globals.GroupHealth)
}
}
// TestNodeEgressRoundTrip pins the node-level egress binding (multi-WAN): a
// node's `egress` option survives WriteUCI->ReadUCI so the generator can bind the
// node's own upstream to that egress outbound. An absent option parses to "".
+18 -12
View File
@@ -187,18 +187,19 @@ func ParseUCIExport(text string) (*Model, error) {
})
case "profile":
m.Profiles = append(m.Profiles, Profile{
Name: s.optOr("name", s.Name),
Enabled: s.optBool("enabled", false),
Priority: parseInt(s.opt("priority"), 0),
MatchIface: nonEmpty(s.list("match_iface")),
SchedDays: s.list("sched_day"),
SchedStart: s.opt("sched_start"),
SchedEnd: s.opt("sched_end"),
SchedTZ: s.opt("sched_tz"),
EnableRules: nonEmpty(s.list("enable_rule")),
DisableRules: nonEmpty(s.list("disable_rule")),
DefaultTarget: s.opt("default_target"),
DefaultEgress: s.opt("default_egress"),
Name: s.optOr("name", s.Name),
Enabled: s.optBool("enabled", false),
Priority: parseInt(s.opt("priority"), 0),
MatchIface: nonEmpty(s.list("match_iface")),
SchedDays: s.list("sched_day"),
SchedStart: s.opt("sched_start"),
SchedEnd: s.opt("sched_end"),
SchedTZ: s.opt("sched_tz"),
EnableRules: nonEmpty(s.list("enable_rule")),
DisableRules: nonEmpty(s.list("disable_rule")),
DefaultTarget: s.opt("default_target"),
DefaultEgress: s.opt("default_egress"),
EndpointResolver: s.opt("endpoint_resolver"),
})
case "resolver":
m.Resolvers = append(m.Resolvers, Resolver{
@@ -281,8 +282,10 @@ func applyGlobals(g *Globals, s uciSection) {
g.ConfirmTimeout = parseInt(s.opt("confirm_timeout"), 0)
g.ResolverDefault = s.opt("resolver_default")
g.ResolverFallback = s.opt("resolver_fallback")
g.EndpointResolver = s.opt("endpoint_resolver")
g.ProbeURL = s.opt("probe_url")
g.ProbeInterval = s.opt("probe_interval")
g.SweepInterval = s.opt("sweep_interval")
g.SchemaVersion = parseInt(s.opt("schema_version"), g.SchemaVersion)
g.ActiveProfile = s.opt("active_profile")
// Geo-data source. No defaults here: "" IS the auto policy (country code ->
@@ -296,6 +299,9 @@ func applyGlobals(g *Globals, s uciSection) {
g.DNSFilter = s.optBool("dns_filter", g.DNSFilter)
g.DNSIntercept = s.optBool("dns_intercept", g.DNSIntercept)
g.BlockDoH = s.optBool("block_doh", g.BlockDoH)
// Default true comes from the DefaultGlobals seed (g.GroupHealth), so an ABSENT option
// stays ON (opt-out); an EXPLICIT "0" disables our sweep.
g.GroupHealth = s.optBool("group_health", g.GroupHealth)
g.Untunnelable = s.optOr("untunnelable", g.Untunnelable)
g.StatsBackend = s.optOr("stats_backend", g.StatsBackend)
// Stats-size knobs: 0 = UNLIMITED, N = limit. The default (g.*) is the DefaultGlobals
+6
View File
@@ -245,6 +245,12 @@ func ValidateGlobals(g Globals) []Warning {
g.LogLevel, strings.Join(EngineLogLevels, "/"), strings.Join(SilentLogLevels, "/")))
}
// The sweep schedule resolves through the SAME function apply uses, so the
// warning the operator reads and the behaviour they get cannot disagree.
if _, _, warn := g.SweepSchedule(); warn != "" {
add(warn)
}
switch strings.ToLower(strings.TrimSpace(g.Untunnelable)) {
case "", "block", "icmp", "direct":
default:
+183 -13
View File
@@ -9,6 +9,7 @@ import (
"bytes"
"encoding/json"
"fmt"
"net/netip"
"os/exec"
"strings"
"time"
@@ -86,12 +87,23 @@ func TeardownNft() error {
return nil
}
// ApplyRouting reconciles ip rule/route: fwmark(FwmarkBase) -> table(TableBase)
// with `local default dev lo`, so tproxy-marked packets are delivered locally,
// plus the per-egress mark->table->device bindings. Idempotent (del-then-add).
// ApplyRouting reconciles ip rule/route, discarding the non-fatal warnings. See
// ApplyRoutingWithWarnings for the full contract.
func ApplyRouting(m *model.Model) error {
_, err := ApplyRoutingWithWarnings(m)
return err
}
// ApplyRoutingWithWarnings reconciles ip rule/route: fwmark(FwmarkBase) ->
// table(TableBase) with `local default dev lo`, so tproxy-marked packets are delivered
// locally, plus the per-egress mark->table->device bindings. Idempotent (del-then-add).
//
// It mirrors RenderNftWithWarnings: the warnings are operator-facing statements about an
// egress that WAS built but cannot carry traffic, which apply folds into the status
// warning list so the panel shows them.
func ApplyRoutingWithWarnings(m *model.Model) ([]string, error) {
if err := addRouting(effFwmark(m.Globals), effTable(m.Globals), m.Globals.IPv6); err != nil {
return err
return nil, err
}
return addEgressRouting(m)
}
@@ -136,8 +148,9 @@ func removeRouting(mark, table uint32) {
// addEgressRouting realises the policy-routing half of an interface/tunnel egress
// (the generator emits the SO_BINDTODEVICE+SO_MARK outbound; this binds the mark
// to a table whose default route leaves via the egress device). Idempotent.
func addEgressRouting(m *model.Model) error {
func addEgressRouting(m *model.Model) ([]string, error) {
run := func(args ...string) { _ = execCommand("ip", args...).Run() }
var warnings []string
for i, eg := range m.Egresses {
t := strings.ToLower(eg.Type)
if (t != "interface" && t != "tunnel") || eg.Interface == "" {
@@ -156,14 +169,47 @@ func addEgressRouting(m *model.Model) error {
run(fam, "route", "flush", "table", fmt.Sprintf("%d", table))
// A gateway'd interface (WAN) needs `via <gw>`; a point-to-point tunnel
// (wg/awg) has no gateway and routes straight out the device.
if gw := defaultGateway(fam, dev); gw != "" {
if gw := egressGateway(fam, eg.Interface, dev); gw != "" {
run(fam, "route", "add", "default", "via", gw, "dev", dev, "table", fmt.Sprintf("%d", table))
} else {
run(fam, "route", "add", "default", "dev", dev, "table", fmt.Sprintf("%d", table))
continue
}
run(fam, "route", "add", "default", "dev", dev, "table", fmt.Sprintf("%d", table))
// No nexthop AND not a point-to-point device: the route just installed says
// "everything is on the local segment", which for any address outside this
// subnet is a guaranteed failure. The egress is configured, reports as
// applied, and cannot carry a single packet off-link — say so.
if !isPointToPoint(dev) {
// The `kind "name": message` prefix is the shape apply's warning
// normaliser parses (entityRe), so this lands as an egress-scoped,
// named warning in the panel rather than an anonymous line.
warnings = append(warnings, fmt.Sprintf(
"egress %q: no %s gateway could be found for interface %q (device %s) "+
"by any means, so its routing table sends traffic straight onto the "+
"local segment. The device is not point-to-point, so this egress "+
"CANNOT REACH ANYTHING outside its own subnet — every node, group and "+
"rule bound to it will fail to connect. Check that the interface is up "+
"and has a lease, or set a static nexthop (network.%s.%s).",
eg.Name, famLabel(fam), eg.Interface, dev, eg.Interface, uciGatewayOption(fam)))
}
}
}
return nil
return warnings, nil
}
// famLabel renders an `ip` family flag for an operator-facing message.
func famLabel(fam string) string {
if fam == "-6" {
return "IPv6"
}
return "IPv4"
}
// uciGatewayOption is the UCI option name carrying a static nexthop for a family.
func uciGatewayOption(fam string) string {
if fam == "-6" {
return "ip6gw"
}
return "gateway"
}
// removeEgressRouting tears down every possible egress table/rule in the reserved
@@ -180,9 +226,102 @@ func removeEgressRouting(m *model.Model) {
}
}
// defaultGateway returns the main-table default-route nexthop for a device
// (e.g. "10.0.2.2" for a WAN), or "" for a point-to-point device with no gateway.
func defaultGateway(fam, dev string) string {
// egressGateway returns the nexthop an egress interface's own default route should use,
// or "" when there is genuinely none to find.
//
// # Why this cannot read the main routing table alone
//
// It used to, and that broke MULTI-WAN — which is the entire reason egresses exist. With
// two uplinks the main table holds exactly ONE default route: the one that won on
// metric. The loser's gateway is perfectly well known to the system, it is simply not in
// that table, so `ip route show default dev eth1` printed nothing and we installed
// `default dev eth1 scope link` into the egress table — a route with no nexthop, which
// hands the packet to the local segment and fails for every off-link address. The second
// uplink was therefore silently dead on every dual-WAN router, with the panel showing a
// configured, applied egress. Observed on the stand: wan0 up, DHCP lease held, ubus
// reporting nexthop 10.0.2.2, and our table built without a via.
//
// # Order of sources, and why
//
// 1. ubus network.interface.<name>.status, the `route` entry with target 0.0.0.0 (or
// ::) and mask 0. This is netifd's view of what THIS interface's default route is,
// independent of which interface won the main table, so it is the only source that
// is correct on a multi-WAN box. It is also the same object IfaceDevice already
// reads for l3_device — one more read of a plane we are already talking to.
// 2. uci network.<name>.gateway / .ip6gw, a statically configured nexthop. Second
// because a running interface's ACTUAL route should beat the file that asked for it:
// a DHCP lease or a live operator change wins over stale config.
// 3. the main table, as before. Kept for a device set up outside netifd entirely, where
// neither of the above knows anything about it.
//
// It takes BOTH the interface name and the device because the two namespaces are
// different and each source is keyed by one of them: ubus and uci by the logical
// interface ("wan0"), the kernel by the device ("eth1"). Passing only the device is what
// made the authoritative source unreachable in the first place.
//
// An empty result is NOT automatically a defect — a point-to-point device (wireguard,
// amneziawg, ppp) legitimately has no nexthop and routes straight out the device. The
// caller separates the two with isPointToPoint and warns only about the other case.
func egressGateway(fam, iface, dev string) string {
if gw := ubusDefaultGateway(fam, iface); gw != "" {
return gw
}
if gw := uciGateway(fam, iface); gw != "" {
return gw
}
return mainTableGateway(fam, dev)
}
// ubusDefaultGateway reads the interface's OWN default-route nexthop from netifd.
func ubusDefaultGateway(fam, iface string) string {
if iface == "" {
return ""
}
out, err := execCommand("ubus", "call", "network.interface."+iface, "status").Output()
if err != nil {
return ""
}
var st struct {
Route []struct {
Target string `json:"target"`
Mask int `json:"mask"`
Nexthop string `json:"nexthop"`
} `json:"route"`
}
if json.Unmarshal(out, &st) != nil {
return ""
}
want := "0.0.0.0"
if fam == "-6" {
want = "::"
}
for _, r := range st.Route {
if r.Mask != 0 || r.Target != want {
continue
}
if gw := validGateway(fam, r.Nexthop); gw != "" {
return gw
}
}
return ""
}
// uciGateway reads a statically configured nexthop out of /etc/config/network.
func uciGateway(fam, iface string) string {
if iface == "" {
return ""
}
out, err := execCommand("uci", "-q", "get", "network."+iface+"."+uciGatewayOption(fam)).Output()
if err != nil {
return ""
}
return validGateway(fam, string(out))
}
// mainTableGateway is the original source: the main-table default route for a device.
// Correct when this interface is the one that won the main table, and useless otherwise
// — see egressGateway.
func mainTableGateway(fam, dev string) string {
out, err := execCommand("ip", fam, "route", "show", "default", "dev", dev).Output()
if err != nil {
return ""
@@ -190,12 +329,43 @@ func defaultGateway(fam, dev string) string {
f := strings.Fields(string(out))
for i := 0; i < len(f)-1; i++ {
if f[i] == "via" {
return f[i+1]
return validGateway(fam, f[i+1])
}
}
return ""
}
// validGateway accepts a nexthop only if it parses as an address of the RIGHT family and
// is not the unspecified address. Every source here is external text (ubus JSON, a UCI
// value, `ip` output) about to be pasted into an `ip route add`: a v6 address in a -4
// table, or a "0.0.0.0" placeholder, yields a route that either fails to install or
// silently blackholes the egress.
func validGateway(fam, raw string) string {
addr, err := netip.ParseAddr(strings.TrimSpace(raw))
if err != nil || !addr.IsValid() || addr.IsUnspecified() {
return ""
}
if (fam == "-6") != addr.Is6() {
return ""
}
return addr.String()
}
// isPointToPoint reports whether a device legitimately has no nexthop: wireguard,
// amneziawg, ppp and tun devices all carry IFF_POINTOPOINT, and for them a route straight
// out the device is correct rather than broken.
//
// Unknown or unreadable is reported as NOT point-to-point, so an egress we cannot
// classify is warned about rather than silently excused. A spurious warning costs a line
// of text; a suppressed one costs an uplink nobody knows is dead.
func isPointToPoint(dev string) bool {
out, err := execCommand("ip", "link", "show", "dev", dev).Output()
if err != nil {
return false
}
return strings.Contains(string(out), "POINTOPOINT")
}
// RoutingPresent reports whether our fwmark ip rule is currently installed, for
// EVERY family the model asks for. With Globals.IPv6 on, ApplyRouting installs
// both a -4 and a -6 rule, so checking only -4 was a half-truth: `ip -6 rule` is
+221
View File
@@ -0,0 +1,221 @@
package netplane
import (
"os"
"os/exec"
"strings"
"testing"
"github.com/sagernet/sing-box/shater/model"
)
// fakeNet intercepts the ubus/uci/ip reads the egress gateway resolution makes and
// records every `ip route add` it performs, so a test can assert the route that would
// actually be installed.
type fakeNet struct {
ubus map[string]string // interface name -> `ubus call ... status` JSON
uci map[string]string // "network.<iface>.<opt>" -> value
mainTab map[string]string // device -> `ip -N route show default dev <dev>` output
link map[string]string // device -> `ip link show dev <dev>` output
added []string // every `ip ... route add ...` invocation, joined
}
func (f *fakeNet) install(t *testing.T) {
t.Helper()
orig := execCommand
t.Cleanup(func() { execCommand = orig })
execCommand = func(name string, arg ...string) *exec.Cmd {
out := ""
switch {
case name == "ubus" && len(arg) >= 2 && arg[0] == "call":
out = f.ubus[strings.TrimPrefix(arg[1], "network.interface.")]
case name == "uci":
// uci -q get network.<iface>.<opt>
out = f.uci[arg[len(arg)-1]]
// `ip link show dev <dev>` has no family flag, so it starts at arg[0].
case name == "ip" && len(arg) >= 4 && arg[0] == "link" && arg[1] == "show":
out = f.link[arg[len(arg)-1]]
case name == "ip" && len(arg) >= 3 && arg[1] == "route" && arg[2] == "show":
out = f.mainTab[arg[len(arg)-1]]
case name == "ip" && len(arg) >= 3 && arg[1] == "route" && arg[2] == "add":
f.added = append(f.added, strings.Join(arg, " "))
}
cs := append([]string{"-test.run=TestEgressGatewayHelperProcess", "--", name}, arg...)
cmd := exec.Command(os.Args[0], cs...)
cmd.Env = append(os.Environ(), "GO_WANT_HELPER_PROCESS=1", "GO_HELPER_STDOUT="+out)
return cmd
}
}
// TestEgressGatewayHelperProcess is the canned-output child process.
func TestEgressGatewayHelperProcess(t *testing.T) {
if os.Getenv("GO_WANT_HELPER_PROCESS") != "1" {
return
}
os.Stdout.WriteString(os.Getenv("GO_HELPER_STDOUT"))
os.Exit(0)
}
// The stand's exact ubus payload for the SECOND uplink: up, DHCP lease held, its own
// default route known to netifd — and absent from the main table because the other WAN
// won on metric.
const wan0Status = `{
"up": true, "pending": false, "available": true, "l3_device": "eth1", "proto": "dhcp",
"route": [ { "target": "0.0.0.0", "mask": 0, "nexthop": "10.0.2.2", "source": "10.0.2.15/32" } ]
}`
// THE REGRESSION. A multi-WAN router holds one default route in the main table, so the
// losing uplink's gateway is invisible there — but netifd knows it perfectly well. The
// old code read only the main table, found nothing, and installed
// `default dev eth1 scope link`: a nexthop-less route that hands every packet to the
// local segment. The second uplink was dead on every dual-WAN box, with the egress
// showing as configured and applied.
//
// The egress table must carry `default via 10.0.2.2 dev eth1`, and nothing must warn.
func TestEgressGatewayFromUbusWhenAbsentFromMainTable(t *testing.T) {
f := &fakeNet{
ubus: map[string]string{"wan0": wan0Status},
mainTab: map[string]string{"eth1": ""}, // the other WAN won the main table
link: map[string]string{"eth1": "2: eth1: <BROADCAST,MULTICAST,UP,LOWER_UP> mtu 1500"},
}
f.install(t)
m := &model.Model{Egresses: []model.Egress{{Name: "g-wan0", Type: "interface", Interface: "wan0"}}}
warns, err := addEgressRouting(m)
if err != nil {
t.Fatalf("addEgressRouting: %v", err)
}
var got string
for _, add := range f.added {
if strings.Contains(add, "eth1") {
got = add
}
}
if !strings.Contains(got, "via 10.0.2.2") || !strings.Contains(got, "dev eth1") {
t.Fatalf("egress route = %q, want a nexthop from ubus (`via 10.0.2.2 dev eth1`).\n"+
"A route without `via` sends every off-subnet packet to the local segment.", got)
}
if len(warns) != 0 {
t.Fatalf("a working egress must not warn: %v", warns)
}
}
// A statically configured nexthop is honoured when netifd has nothing to say (interface
// down, or configured outside netifd).
func TestEgressGatewayFromUCI(t *testing.T) {
f := &fakeNet{
ubus: map[string]string{},
uci: map[string]string{"network.wan0.gateway": "192.168.9.1\n"},
mainTab: map[string]string{"eth1": ""},
link: map[string]string{"eth1": "<BROADCAST,MULTICAST>"},
}
f.install(t)
if got := egressGateway("-4", "wan0", "eth1"); got != "192.168.9.1" {
t.Fatalf("egressGateway = %q, want the static UCI nexthop", got)
}
}
// The main table remains the last resort, for a device netifd knows nothing about.
func TestEgressGatewayFallsBackToMainTable(t *testing.T) {
f := &fakeNet{
mainTab: map[string]string{"eth2": "default via 10.0.3.2 dev eth2 proto static"},
}
f.install(t)
if got := egressGateway("-4", "wan1", "eth2"); got != "10.0.3.2" {
t.Fatalf("egressGateway = %q, want the main-table nexthop 10.0.3.2", got)
}
}
// A point-to-point device legitimately has NO nexthop: routing straight out the device
// is correct there, so it must be installed without `via` and must NOT warn. This is the
// case the original comment was right about, and it must survive the fix.
func TestEgressPointToPointNeedsNoGatewayAndDoesNotWarn(t *testing.T) {
f := &fakeNet{
ubus: map[string]string{"wgvpn": `{"up":true,"l3_device":"wg0","route":[]}`},
link: map[string]string{"wg0": "5: wg0: <POINTOPOINT,NOARP,UP,LOWER_UP> mtu 1420"},
}
f.install(t)
m := &model.Model{Egresses: []model.Egress{{Name: "tun", Type: "tunnel", Interface: "wgvpn"}}}
warns, err := addEgressRouting(m)
if err != nil {
t.Fatalf("addEgressRouting: %v", err)
}
var got string
for _, add := range f.added {
if strings.Contains(add, "wg0") {
got = add
}
}
if got == "" || strings.Contains(got, "via") {
t.Fatalf("point-to-point route = %q, want a plain `dev wg0` route with no via", got)
}
if len(warns) != 0 {
t.Fatalf("a point-to-point egress must not warn: %v", warns)
}
}
// The remaining genuinely broken case: no nexthop from ANY source, on a device that is
// not point-to-point. The route is still installed (best effort) but the operator is
// told, in terms of consequence, that this egress cannot carry traffic off its subnet.
func TestEgressNoGatewayOnBroadcastDeviceWarns(t *testing.T) {
f := &fakeNet{
ubus: map[string]string{"wan0": `{"up":false,"l3_device":"eth1","route":[]}`},
link: map[string]string{"eth1": "2: eth1: <BROADCAST,MULTICAST> mtu 1500"},
}
f.install(t)
m := &model.Model{Egresses: []model.Egress{{Name: "g-wan0", Type: "interface", Interface: "wan0"}}}
warns, err := addEgressRouting(m)
if err != nil {
t.Fatalf("addEgressRouting: %v", err)
}
if len(warns) != 1 {
t.Fatalf("warnings = %v, want exactly one", warns)
}
w := warns[0]
for _, want := range []string{`egress "g-wan0":`, "IPv4", "wan0", "eth1", "CANNOT REACH", "network.wan0.gateway"} {
if !strings.Contains(w, want) {
t.Errorf("warning is missing %q:\n%s", want, w)
}
}
}
// validGateway is the last guard before the value is pasted into `ip route add`: wrong
// family, unspecified, or not an address at all must all be refused rather than
// installed as a route that fails or blackholes.
func TestValidGateway(t *testing.T) {
cases := []struct {
fam, in, want string
}{
{"-4", "10.0.2.2", "10.0.2.2"},
{"-4", " 10.0.2.2\n", "10.0.2.2"},
{"-6", "fe80::1", "fe80::1"},
{"-4", "fe80::1", ""}, // v6 nexthop in a v4 table
{"-6", "10.0.2.2", ""}, // v4 nexthop in a v6 table
{"-4", "0.0.0.0", ""}, // unspecified placeholder
{"-6", "::", ""}, // unspecified placeholder
{"-4", "", ""}, // nothing
{"-4", "not-an-ip", ""}, // garbage
{"-4", "10.0.2.2/24", ""}, // a prefix is not a nexthop
}
for _, c := range cases {
if got := validGateway(c.fam, c.in); got != c.want {
t.Errorf("validGateway(%q,%q) = %q, want %q", c.fam, c.in, got, c.want)
}
}
}
// isPointToPoint must fail SAFE: a device it cannot read is reported as not
// point-to-point, so an unclassifiable egress gets a warning rather than a silent pass.
func TestIsPointToPointFailsSafe(t *testing.T) {
f := &fakeNet{link: map[string]string{"wg0": "<POINTOPOINT,NOARP>"}}
f.install(t)
if !isPointToPoint("wg0") {
t.Error("wg0 must be recognised as point-to-point")
}
if isPointToPoint("unknown0") {
t.Error("an unreadable device must NOT be excused as point-to-point")
}
}
+43 -9
View File
@@ -1094,6 +1094,18 @@ func (s *Server) handleRuleSetCategories(w http.ResponseWriter, r *http.Request)
// --- node health probe-all (feedback #3 — "Test all nodes") -----------------
// NodeTestScope is the constant reported in nodeTestStatus.Scope. The health run has
// no per-group target and never will: it measures EVERY node, every endpoint and every
// group's egress copies in one pass, because that is what "refresh health" means.
//
// It is published as an explicit token rather than left implicit so the panel cannot
// confuse the two runs. They are different actions with different reach, and if both
// simply rendered as "measuring" on every card the operator would have no idea which
// one was happening. The intended split: a GROUP test is scoped (see
// groupTestStatus.Scope) and belongs on the cards it names; a HEALTH run is global and
// belongs in a global progress indicator driven by done/total, not on the cards.
const NodeTestScope = "all_nodes"
// nodeTestStatus is the GET /api/nodes/test body (also embedded in the POST reply's
// mirror): the live probe-all progress. running=false with done==total means the
// last run finished; a fresh daemon reports {false,0,0}.
@@ -1101,6 +1113,8 @@ type nodeTestStatus struct {
Running bool `json:"running"`
Done int `json:"done"`
Total int `json:"total"`
// Scope is always NodeTestScope — see there.
Scope string `json:"scope"`
}
// handleNodesTest → /api/nodes/test: force a health probe of EVERY node, or read
@@ -1114,12 +1128,17 @@ type nodeTestStatus struct {
// starts a second concurrent run). Each tag is probed with the URL of the group
// that owns it, falling back to Globals.ProbeURL and then to the engine's
// gstatic default.
// - GET → 200 {"running":bool,"done":N,"total":M} progress snapshot.
// - GET → 200 {"running":bool,"done":N,"total":M,"scope":"all_nodes"} progress
// snapshot. The scope is constant: this run is GLOBAL, so render it as one
// progress indicator, never as a per-group "measuring" badge — that badge belongs
// to /api/groups/test, which names the groups it covers.
func (s *Server) handleNodesTest(w http.ResponseWriter, r *http.Request) {
switch r.Method {
case http.MethodGet:
running, done, total := s.a.NodeTestStatus()
writeJSON(w, http.StatusOK, nodeTestStatus{Running: running, Done: done, Total: total})
writeJSON(w, http.StatusOK, nodeTestStatus{
Running: running, Done: done, Total: total, Scope: NodeTestScope,
})
case http.MethodPost:
// The Applier resolves the probe URLs itself (per-group, falling back to the
// global one) — a manual run must measure each tag with the instrument its
@@ -1137,12 +1156,22 @@ func (s *Server) handleNodesTest(w http.ResponseWriter, r *http.Request) {
// --- group test (F2 — "what am I going out through, and how fast") ----------
// groupTestStatus is the GET /api/groups/test body. Results is always a non-nil
// slice (never JSON null) so the panel can map over it without a guard.
// groupTestStatus is the GET /api/groups/test body. Scope and Results are always
// non-nil slices (never JSON null) so the panel can map over them without a guard.
//
// Scope is the set of group names THIS run covers, and it is what makes `running`
// usable: on its own that flag says only "a group test is happening somewhere", which
// is why pressing Test on one group used to show "measuring" on every card. The panel
// must render the in-progress indicator on card g only when running && scope contains g.
// A run started with a name carries exactly that name; a run started with an empty body
// carries every group in the running engine, and then the indicator on every card is
// correct. Scope persists after the run ends, so the displayed results can still be
// attributed to the cards they came from.
type groupTestStatus struct {
Running bool `json:"running"`
Done int `json:"done"`
Total int `json:"total"`
Scope []string `json:"scope"`
Results []engine.GroupTestResult `json:"results"`
}
@@ -1167,18 +1196,23 @@ type groupTestStartResponse struct {
// - POST {"name":"auto"} → test that group; an empty/absent name tests all.
// 200 {"started":true}, or 200 {"started":false,"reason":"already running"}
// when a run is in flight (never starts a second concurrent run).
// - GET → 200 {"running":bool,"done":N,"total":M,"results":[...]} — the same
// poll-for-progress shape as /api/nodes/test, with the results carried inline
// because nothing else in the API publishes an exit address.
// - GET → 200 {"running":bool,"done":N,"total":M,"scope":[...],"results":[...]} —
// the same poll-for-progress shape as /api/nodes/test, with the results carried
// inline because nothing else in the API publishes an exit address, and `scope`
// naming the groups this run covers so the panel can put the in-progress indicator
// on exactly those cards. See groupTestStatus.
func (s *Server) handleGroupsTest(w http.ResponseWriter, r *http.Request) {
switch r.Method {
case http.MethodGet:
running, done, total, results := s.a.GroupTestStatus()
running, done, total, scope, results := s.a.GroupTestStatus()
if results == nil {
results = []engine.GroupTestResult{}
}
if scope == nil {
scope = []string{}
}
writeJSON(w, http.StatusOK, groupTestStatus{
Running: running, Done: done, Total: total, Results: results,
Running: running, Done: done, Total: total, Scope: scope, Results: results,
})
case http.MethodPost:
// A missing or unparseable body means "all groups": the endpoint is also
+79 -3
View File
@@ -13,9 +13,10 @@ import (
// the test asserts the CONTRACT (field names + types) the panel codes against, not
// just whatever JSON happened to come out.
type groupTestBody struct {
Running bool `json:"running"`
Done int `json:"done"`
Total int `json:"total"`
Running bool `json:"running"`
Done int `json:"done"`
Total int `json:"total"`
Scope []string `json:"scope"`
Results []struct {
Group string `json:"group"`
Selected string `json:"selected"`
@@ -192,3 +193,78 @@ func TestGroupsTestMethodNotAllowed(t *testing.T) {
t.Fatalf("PUT /api/groups/test: got %d, want 405", code)
}
}
// The two runs must be distinguishable by the panel, because they are different
// actions with different reach. A GROUP test names the groups it covers, so the
// in-progress badge belongs on exactly those cards; a HEALTH run is global and reports
// a constant scope, so it belongs in a global progress indicator instead.
//
// This is the endpoint-level regression for "press Test on one group and every card
// says measuring": without `scope`, `running` alone cannot be attributed to a card.
func TestGroupAndNodeTestScopesAreDistinguishable(t *testing.T) {
s := newTestServer(t)
srv := httptest.NewServer(s.Handler())
defer srv.Close()
cookie := login(t, srv, s)
// Group test: scope is a LIST of group names, and must never be JSON null.
code, raw := groupsTestReq(t, srv, cookie, http.MethodGet, "")
if code != http.StatusOK {
t.Fatalf("GET /api/groups/test: got %d, want 200", code)
}
var g groupTestBody
if err := json.Unmarshal(raw, &g); err != nil {
t.Fatalf("decode group test: %v", err)
}
if g.Scope == nil {
t.Fatal("group test scope must encode as [] and never null")
}
// Structurally an ARRAY, not a scalar: the panel must be able to test membership.
var shape struct {
Scope json.RawMessage `json:"scope"`
}
if err := json.Unmarshal(raw, &shape); err != nil {
t.Fatalf("decode raw scope: %v", err)
}
if len(shape.Scope) == 0 || shape.Scope[0] != '[' {
t.Fatalf("group test scope must be a JSON array, got %s", shape.Scope)
}
// Health run: scope is the constant global token, not a list.
resp := do(t, srv, http.MethodGet, "/api/nodes/test", cookie, "")
defer resp.Body.Close()
var n struct {
Running bool `json:"running"`
Scope string `json:"scope"`
}
if err := json.NewDecoder(resp.Body).Decode(&n); err != nil {
t.Fatalf("decode node test: %v", err)
}
if n.Scope != NodeTestScope {
t.Fatalf("node test scope = %q, want the constant %q", n.Scope, NodeTestScope)
}
}
// A named group test scopes to that name alone — the endpoint-level version of the
// engine regression, so the contract the panel actually consumes is pinned too.
func TestGroupTestScopeFollowsRequestedName(t *testing.T) {
s := newTestServer(t)
srv := httptest.NewServer(s.Handler())
defer srv.Close()
cookie := login(t, srv, s)
if code, _ := groupsTestReq(t, srv, cookie, http.MethodPost, `{"name":"auto"}`); code != http.StatusOK {
t.Fatalf("POST /api/groups/test: got %d, want 200", code)
}
code, raw := groupsTestReq(t, srv, cookie, http.MethodGet, "")
if code != http.StatusOK {
t.Fatalf("GET /api/groups/test: got %d, want 200", code)
}
var g groupTestBody
if err := json.Unmarshal(raw, &g); err != nil {
t.Fatalf("decode: %v", err)
}
if len(g.Scope) != 1 || g.Scope[0] != "auto" {
t.Fatalf("scope = %v, want [auto] — a one-group run must not claim every card", g.Scope)
}
}