#!/bin/sh /etc/rc.common
# /etc/init.d/shater — procd supervisor for the Shater data plane (v0.2).
#
# v0.2 collapses the old xray + xrayctl + dnsmasq trio into ONE long-lived Go
# process: `shaterd run`. That process embeds the sing-box engine, the
# control-plane AND the in-process DNS server; it OWNS the nft table
# `inet shater`, the reserved policy-routing tables, and the :53 hijack. So this
# init no longer generates a run.json or drives an external engine — it just
# supervises `shaterd run` and lets the daemon apply/teardown its own data plane.
#
# DAEMON CONTRACT (see docs-shater/PORTING.md "Wave 3 daemon contract"):
#   * `shaterd run`       — the supervised daemon. Reads UCI on startup; if
#                           globals.enabled it applies engine+netplane, else it
#                           stays inert. Then blocks.
#   * SIGTERM (procd stop)-> honest teardown (engine.Close + netplane teardown),
#                           then exit. We do NOT tear nft/routing down from the
#                           shell on a normal stop — the daemon owns that.
#   * SIGHUP  (reconcile) -> re-read UCI + idempotent re-apply (config-hash gate
#                           means an unchanged apply does NOT churn the tunnel).
#
# RELIABILITY CONTRACT (the "железно" layer):
#   * The DATA PLANE is completely INERT unless globals.enabled=1. The daemon
#     itself is supervised either way (it needs to be up to serve the admin panel
#     and the control socket — that is the ONLY way to configure the box), but
#     with globals.enabled=0 `shaterd run` opens no engine, emits no nft rules and
#     touches no routing, so boot connectivity is NEVER affected.
#     (Bootstrap: the panel is served BY the daemon. Refusing to start it while
#     disabled made a fresh install unconfigurable — the LuCI "Open panel" button
#     needs a live daemon to mint a handoff token, and the daemon only becomes
#     enable-able THROUGH the panel. Hence: supervise always, intercept never
#     until enabled.)
#   * The interception data plane may only exist while this service is meant to
#     be running. `start` raises ACTIVE_FLAG, `stop` clears it; hotplug/cron
#     reconcile ONLY while the flag is up, so an admin `stop` STICKS — no
#     background actor may resurrect interception behind a stopped daemon.
#   * BEING REPLACED IS NOT BEING SWITCHED OFF. `restart` and `reload` (which is
#     stop+start, i.e. every LuCI Save & Apply) both run through `stop`, and the
#     daemon's SIGTERM teardown removes the fail-closed table unconditionally — it
#     does not consult kill_switch at all. Between that teardown and the
#     successor's first apply the init GUARANTEES a gap: it waits for the old
#     process to exit (shater_wait_stopped), then runs `shaterd migrate`, then
#     starts a daemon that still has to build an engine. So a restart is announced
#     with RESTART_FLAG, which tells the outgoing daemon to leave the fail-closed
#     holding plane STANDING — apply.TeardownExiting swaps it in with one nft
#     transaction and then skips the delete, so the table is never absent, not even
#     for the 80-90 ms the old arm-after-teardown order measured. A real `stop`
#     raises no flag and therefore still means what it says.
#     (A package UPGRADE does not come through here at all on apk v3: shater-core's
#     script table is post-install / pre-deinstall / post-upgrade, with no
#     pre-upgrade, so default_prerm — and its `stop` — runs only on REMOVAL.)
#   * The FAIL-CLOSED PLANE MUST ALSO EXIST BEFORE THIS SCRIPT DOES. START=99 is
#     after fw4 (19) and netifd (20), so at every boot the LAN forwards to the WAN
#     in the clear for as long as it takes procd to decompress the daemon off
#     flash and get an engine up. /etc/init.d/shater-armor (START=21) loads
#     BOOT_ARMOR — a copy of the holding plane the daemon persists on every apply
#     — to close that window. This script owns the DISARM half, and it owns it
#     with a CLOSED LIST: an operator's `stop`, or a removal, and nothing else.
#     Powering the box down must not — `shutdown` reaches stop_service too, and it
#     is not a person switching the product off (see shater_stop_disarms).
#   * The engine must never be permanently abandoned while interception stands:
#     respawn retries are infinite (procd never gives up); a sustained-dead
#     daemon is additionally escalated by the shater-cron watchdog.
#   * busybox ash only — no bashisms.

USE_PROCD=1
START=99          # after network + firewall
STOP=10

PROG=/usr/bin/shaterd
# Raised while the service is meant to be running; the ONLY token that lets
# hotplug/shater-cron touch the data plane. tmpfs => cleared by reboot, so
# nothing reconciles before this init has run at boot.
ACTIVE_FLAG=/var/run/shater.active
# Written by `shaterd run`; the single-owner token this init waits on so a
# restart never overlaps a new data plane with the previous one's teardown.
PIDFILE=/var/run/shaterd.pid
# Raised around a restart/reload, read by the OUTGOING `shaterd run` at SIGTERM:
# present => "you are being replaced, leave the fail-closed plane standing";
# absent  => "you are being switched off, take everything down". tmpfs, so a
# power cut can never make the next boot look like a restart.
RESTART_FLAG=/var/run/shater.restarting
# The persisted fail-closed holding plane. Written by the daemon on every apply,
# loaded by /etc/init.d/shater-armor at boot. Its PRESENCE is the arm token, so
# removing it here is how a deliberate stop stops the next boot from blocking.
BOOT_ARMOR=/etc/shater/boot.nft
# Seconds `start` will wait for a predecessor to finish its teardown. Must be
# >= term_timeout below (procd's hard cap on a predecessor's life after SIGTERM)
# so we never give up while procd is still letting it shut down cleanly.
STOP_WAIT_SECS=40

# WHICH ACTION rc.common was invoked with, frozen at source time.
#
# rc.common does, in this order:
#     initscript=$1; action=${2:-help}; shift 2; ...; . "$initscript"; $action "$@"
# so `action` is ALREADY assigned when this file is sourced, and every action then
# runs as a function in THAT SAME shell. MEASURED on the target (ImmortalWrt
# 25.12.1 r37978) with a throwaway probe init script, not read off documentation:
#
#     /etc/init.d/X restart  ->  stop_service action=[restart], start_service [restart]
#     /etc/init.d/X stop     ->  stop_service action=[stop]
#     /etc/init.d/X reload   ->  reload_service action=[reload]
#     `reboot`               ->  stop_service action=[SHUTDOWN]     <-- see below
#     the boot after it      ->  start_service action=[boot]
#
# A previous probe reported this variable EMPTY and the emptiness was written up as
# the defect. It was the probe: `sh -x /etc/init.d/shater restart` bypasses the
# `#!/bin/sh /etc/rc.common` shebang, so rc.common never runs, never assigns
# `action`, and the variable reads empty no matter what this file does.
#
# Frozen into our own variable because `action` is a short, generic name that other
# framework helpers also use as a local; a snapshot taken before any function runs
# cannot be shadowed later.
SHATER_RC_ACTION="$action"

# --- what an action MEANS --------------------------------------------------
#
# THE BUG THESE TWO PREDICATES REPLACE (v0.2.17, measured on the live router).
# The old stop_service was `case $action in restart|reload) keep;; *) DISARM;; esac`
# — an open default that swept up every action nobody had enumerated. `reboot` is
# one of them: procd runs the K-links with the action `shutdown`, so the shutdown
# path deleted the arm token on the way down and the next boot had nothing to load.
# The mechanism destroyed itself at exactly the moment it exists for. Instrument
# reading from the router, one minute apart across a reboot:
#
#     13:28  /etc/shater/boot.nft present
#     ----   reboot (stop_service action=[shutdown] -> old `*` branch -> rm)
#     18s    at_S22: NO_TABLE  armor_file=NO_FILE
#
# So both lists below are POSITIVE and CLOSED. An action nobody thought about —
# `shutdown` above all, but also whatever a future procd invents — falls through
# both and changes nothing. The default now fails in the recoverable direction: at
# worst a boot arms when it need not have, which costs the second before the daemon
# applies and is still gated by shater-armor's own four state refusals. The old
# default failed in the direction of the plaintext window the feature was built to
# close.
#
# They are predicates rather than an inline `case` so the test gate can execute the
# real thing: it sources THIS FILE in /bin/sh and calls them with every action procd
# actually uses (shater/cmd/shaterd/initscript_test.go). A comment claiming
# `shutdown` is handled is what shipped last time.

# True only for the ONE action that means "the operator switched the product off".
# Deliberately not `shutdown`: powering a router down is not turning a feature off.
#
# NOT sufficient on its own — see shater_stop_disarms. `stop` is also how the
# package manager's plumbing reaches us, and a package manager is not a person.
shater_action_disarms() {
	case "$1" in
		stop) return 0 ;;
		*)    return 1 ;;
	esac
}

# Is a package manager in the middle of a transaction RIGHT NOW?
#
# This is a state, read at the moment the decision is made, exactly like
# shater-armor's four refusals — not a record of an event. The same question is
# already asked (for the same reason: prerm/postinst plumbing is not a user
# action) by the detached bring-up in /etc/uci-defaults/30_shater-core.
shater_pkg_transaction() {
	pidof apk >/dev/null 2>&1 && return 0
	pidof opkg >/dev/null 2>&1 && return 0
	return 1
}

# Is the main service still enabled at boot? Same glob, and for the same reason,
# as shater-armor's own check: `/etc/init.d/shater enabled` would source procd.sh
# and take a blocking flock, which is not something to do from inside a package
# manager's transaction.
shater_rc_enabled() {
	local f
	for f in /etc/rc.d/S[0-9][0-9]shater; do
		[ -e "$f" ] && return 0
	done
	return 1
}

# THE ACTUAL DISARM DECISION.
#   $1 = action
#   $2 = 1 when a package transaction is in flight
#   $3 = 1 when the service is still enabled in rc.d
# All three are passed in rather than read inside, so the gate can drive every
# combination without a package manager or an /etc/rc.d.
#
# WHY IT IS NOT JUST THE ACTION. base-files' default_prerm runs, in this order:
#
#     if [ "$PKG_UPGRADE" != "1" ]; then "$i" disable; fi
#     "$i" stop
#
# so a package manager reaches stop_service wearing the operator's clothes. Two
# different intentions arrive as the same action, and the difference between them
# is readable at the moment of the decision:
#
#   REMOVAL  — prerm has ALREADY run `disable`, so S99shater is gone. The product
#              is going away; the armor goes with it. (It is belt-and-braces even
#              so: shater-armor refuses to arm without that symlink, and the whole
#              init script is about to be deleted anyway.)
#   REPLACED — the service is still enabled, so something intends to bring it
#              back. That is not an operator switching anything off, and deleting
#              the armor here would leave the next boot unprotected. "The next
#              apply will rewrite it" is not an answer: the armor exists precisely
#              to cover a reboot, and a reboot between an update and the first
#              apply is how this product is deployed.
#
# MEASURED, because the paragraph above is about a path I got wrong once already.
# On THIS target (apk-tools 3.0.5, ImmortalWrt 25.12.1) shater-core's script table
# is post-install / pre-deinstall / post-upgrade, with NO pre-upgrade — so an apk
# UPGRADE never executes default_prerm and never calls `stop` at all. Verified with
# a real `apk fix --reinstall shater-core` while sampling the armor file: 245 625
# samples, zero disappearances, even with this guard mutated off. The upgrade half
# of this predicate is therefore defence-in-depth for a shape that is one
# `pre-upgrade` script (or a returning opkg lane) away, NOT a fix for an observed
# failure. The removal half is live today.
shater_stop_disarms() {
	shater_action_disarms "$1" || return 1
	# No package manager involved => a person typed it. The escape hatch must work.
	[ "$2" = "1" ] || return 0
	# A package transaction that has NOT disabled the service is replacing it.
	[ "$3" = "1" ] && return 1
	return 0
}

# True when a successor is coming, so the outgoing daemon should leave the
# fail-closed holding plane standing instead of removing it.
#
# `shutdown` is deliberately NOT a handoff either: nothing is coming, and the
# kernel that would hold the plane is going away with it. Leaving the flag down
# there also keeps the marker's meaning exact — it says "you are being replaced",
# and at shutdown nothing is.
shater_action_handoff() {
	case "$1" in
		restart|reload) return 0 ;;
		*)              return 1 ;;
	esac
}

# --- helpers ---------------------------------------------------------------

# True only when the stack is explicitly enabled in UCI.
shater_enabled() {
	local en
	en=$(uci -q get shater.globals.enabled) || return 1
	[ "$en" = "1" ]
}

# Syslog line that honors globals.log_syslog — the SAME toggle that silences the
# daemon's own stderr->logread stream. With log_syslog=0 the operator asked for
# a silent syslog, and shell status lines emitted AROUND the binary must not
# leak past the binary's silence. Absent option (or unreadable UCI) = ON, which
# matches the daemon's default.
_slog() {
	[ "$(uci -q get shater.globals.log_syslog)" = "0" ] || logger -t shater "$@"
}

# A line the operator gets EVEN WITH globals.log_syslog=0, without going behind
# that setting's back.
#
# log_syslog is a statement about ONE destination: the syslog stream (see _slog
# above — the same toggle silences the daemon's own stderr->logread fan-out).
# Honouring it by staying silent everywhere turns "keep syslog quiet" into "never
# tell me the config could not be brought forward", which is not what it says and
# not what anybody means by it. So the refusal goes somewhere else instead:
#
#   * THIS SCRIPT'S OWN STDERR, unconditionally. That is not the syslog stream; it
#     is the reply to whoever invoked the script. Typed by hand it lands on the
#     operator's terminal at the moment they are looking at it; run from
#     30_shater-core inside `apk add` / `opkg install` it lands in the package
#     manager's output, which is the one screen an installing operator does read.
#     At boot it goes to procd's stderr (console) — not durable, hence the file.
#   * /etc/shater/migrate-failed, on flash, written on failure and REMOVED on the
#     first success. That is the durable half: it survives the reboot nobody
#     watched, `cat` reads it, and its absence is the honest all-clear. Written
#     AFTER the stderr line on purpose — a full /overlay is one of the named
#     causes of the failure it is reporting, so it must never be the only channel.
#   * syslog too, but only when log_syslog allows it — that channel keeps
#     obeying the operator exactly as before.
#
# One call, one text, three destinations, so the wording cannot drift between
# them. Failures only: _shout is not a status line.
SHATER_MIGRATE_BREADCRUMB=/etc/shater/migrate-failed
_shout() {
	echo "shater: $*" >&2
	_slog -p daemon.err "$*"
	mkdir -p "$(dirname "$SHATER_MIGRATE_BREADCRUMB")" 2>/dev/null
	echo "$(date -u '+%Y-%m-%dT%H:%M:%SZ') $*" \
		> "$SHATER_MIGRATE_BREADCRUMB" 2>/dev/null || :
}

# --- `shaterd migrate`: which of the four things happened --------------------
#
# The schema version on disk, as an integer. 0 for "absent" and 0 for anything
# non-numeric, DELIBERATELY the same two answers model.readSchemaVersion gives
# (`strconv.Atoi` of a garbage value is 0 with the error dropped) — this number
# is only ever used to name a version in a message, and a shell that disagreed
# with the binary about what v0 means would print a version the binary never saw.
shater_schema_version() {
	local v
	v=$(uci -q get shater.globals.schema_version) || v=""
	case "$v" in
		"")       echo 0 ;;
		*[!0-9]*) echo 0 ;;
		*)        echo "$v" ;;
	esac
}

# shater_migrate_class <rc> <output-of-shaterd-migrate> — prints EXACTLY one of:
#
#     ok          the binary reported success (it may or may not have had work)
#     downgrade   REFUSED: the config on disk is NEWER than this build
#     unreadable  /etc/config/shater could not be read at all
#     failed      it failed for a reason this script does not recognise
#
# A CLOSED POSITIVE LIST, and the last rung is the point of it. Until v0.2.19 all
# four of these were reported with ONE sentence, and that sentence described only
# the third one: "routing rules that still carry the removed dst_domain/dst_ip
# options stay DISABLED until this succeeds. Free space on /overlay and re-run".
# On a DOWNGRADE every clause of that is false — nothing is disabled, /overlay is
# not the problem, and re-running does not help, because the fix is to put the
# newer package back. A confident wrong diagnosis costs more than no diagnosis.
#
# `downgrade` is recognised from the binary's own words. That is a CONTRACT with
# shater/model: both refusals — model.migrateWith's "config schema v%d newer than
# this build (v%d); upgrade the package" and model.ErrSchemaTooNew's "config
# schema newer than this build" — contain the substring matched below, and
# TestMigrateDowngradeSignatureIsAContract (shater/cmd/shaterd) fails if either
# stops containing it. If the wording is ever changed anyway, this degrades to
# `failed`, which names itself as unrecognised and quotes the binary verbatim —
# the recoverable side. It cannot degrade into one of the confident branches.
shater_migrate_class() {
	local rc="$1" out="$2"

	[ "$rc" = "0" ] && { echo ok; return 0; }

	case "$out" in
		*"newer than this build"*) echo downgrade; return 0 ;;
	esac

	# Asked LAST, so a refusal we can name is never re-labelled as an I/O problem.
	# `uci export` fails both when the file is missing and when it does not parse,
	# which is the same thing from here: nothing can be said about a schema that
	# cannot be read.
	uci -q export shater >/dev/null 2>&1 || { echo unreadable; return 0; }

	echo failed
}

# Run the migration and report it. Called from start_service and mirrored by
# /etc/uci-defaults/30_shater-core; see the long note at the call site for why
# this never refuses to start.
shater_migrate() {
	local before after out rc class

	before=$(shater_schema_version)
	out=$("$PROG" migrate 2>&1)
	rc=$?
	class=$(shater_migrate_class "$rc" "$out")
	after=$(shater_schema_version)

	case "$class" in
		ok)
			# The all-clear is the ABSENCE of the breadcrumb, so a fixed router stops
			# claiming to be broken the moment it is fixed.
			rm -f "$SHATER_MIGRATE_BREADCRUMB"
			# Nothing to do is not news; obeys log_syslog like every other status line.
			[ "$before" = "$after" ] && return 0
			_slog -p daemon.info \
				"UCI schema migrated: v$before -> v$after. The config as it was at v$before was copied to /etc/shater/config.pre-v$before.bak before the first change."
			;;
		downgrade)
			_shout "UCI schema migration REFUSED — this is a DOWNGRADE, not a broken config. /etc/config/shater carries schema v$after, which is NEWER than this build understands, so nothing was migrated and nothing on disk was changed. Your settings are intact; they are also unchangeable, because the daemon and the panel refuse every config write for the same reason and a save from the panel will fail too. Nothing on this router fixes it: install a shater build that understands schema v$after — the one that ran here before the downgrade (docs-shater/INSTALL.md has the pinned per-version feed). Starting anyway, so the panel stays reachable. '$PROG migrate' said: ${out:-no output}"
			;;
		unreadable)
			_shout "UCI schema migration FAILED and /etc/config/shater CANNOT BE READ ('uci export shater' fails), so this script cannot even say which schema is on disk. A config that is missing or does not parse is neither migrated nor repaired here. Starting anyway — the daemon will come up on whatever it can parse, which may be nothing, leaving it inert with the panel still reachable. Check /etc/config/shater by hand; an /etc/shater/config.pre-v*.bak copy from an earlier migration may be next to it. '$PROG migrate' said: ${out:-no output}"
			;;
		failed)
			_shout "UCI schema migration FAILED for a reason this script does not recognise; the config on disk is still at schema v$after. Starting anyway: refusing to start would take the admin panel down with it, and the panel is the only way to fix the box. The mundane cause is a full /overlay, where 'uci commit' cannot write — check 'df /overlay' first, then re-run '$PROG migrate' or restart the service.$(
				[ "$after" = "1" ] && printf ' %s' "While the config stays at v1, routing rules that still carry the removed dst_domain/dst_ip options are held DISABLED by the daemon and reported as such — those rules are not in force."
			) '$PROG migrate' said: ${out:-no output}"
			;;
		*)
			# shater_migrate_class returns a closed set and every member of it is
			# handled above, so this is unreachable. It exists to say "this script
			# disagrees with itself" out loud instead of picking one of the confident
			# branches and being wrong quietly — which is the exact failure the closed
			# list replaced.
			_shout "INTERNAL: '$PROG migrate' produced a result /etc/init.d/shater cannot classify (class='$class', rc=$rc). That is a bug in this script, not a state of the router. Starting anyway. Output was: ${out:-no output}"
			;;
	esac
}

# Announce/withdraw "this daemon is being replaced, not switched off". Read by
# `shaterd run` when it receives SIGTERM.
shater_mark_restart() {
	mkdir -p "$(dirname "$RESTART_FLAG")" 2>/dev/null
	: > "$RESTART_FLAG"
}
shater_clear_restart() { rm -f "$RESTART_FLAG"; }

# Remove the persisted boot armor, so the LAN is NOT blocked at the next boot
# before the daemon starts. Called from exactly two places, both of which are a
# statement about the PRODUCT rather than about this process: an operator typing
# `stop`, and a daemon binary that is no longer on the box. In neither case is
# anything going to come along and replace the armor with a real data plane, and a
# kill switch with nothing behind it is just a brick.
#
# NOT called on the shutdown path. That is the whole fix — see
# shater_action_disarms.
shater_disarm_boot() { rm -f "$BOOT_ARMOR"; }

# Echo the pid of a LIVE `shaterd run`, or fail. The pidfile is written by the
# daemon itself and removed only by the daemon that owns it, AFTER its teardown
# has completed — so "pidfile names a live process" is precisely "the previous
# data plane has not been dismantled yet".
shater_daemon_pid() {
	local pid
	pid=$(cat "$PIDFILE" 2>/dev/null) || return 1
	[ -n "$pid" ] || return 1
	kill -0 "$pid" 2>/dev/null || return 1
	echo "$pid"
}

# Block until no predecessor daemon is left, bounded by STOP_WAIT_SECS.
#
# WHY THIS EXISTS. procd's `stop` is ASYNCHRONOUS: rc.common's `restart` is
# literally `stop; start`, and the `service delete` ubus call returns the moment
# procd has SENT SIGTERM — not when the instance is gone. `start` therefore
# re-adds the instance while the outgoing `shaterd run` is still executing its
# honest teardown (engine close, then `nft delete table`, `ip rule`/`ip route`
# removal and the per-iface sysctl restore). The result is that `restart` is NOT
# equivalent to `stop` + pause + `start`: the new plane is stood up on top of
# kernel state the old one has not finished removing, which is what B3 (DNS to
# the router's own LAN address dead after a restart, and never recovering) came
# out of. Waiting here restores the equivalence, and costs literally nothing when
# there is no predecessor — the check runs before the first sleep.
#
# Returning non-zero does NOT abort the start: the daemon carries its own
# single-owner guard and will refuse (or wait) on its side. Better to hand the
# decision to the process that can actually see the plane than to leave the box
# with no service at all.
shater_wait_stopped() {
	local i=0 pid
	pid=$(shater_daemon_pid) || return 0
	_slog -p daemon.info \
		"restart: waiting for the previous shaterd (pid $pid) to finish tearing the data plane down"
	while [ "$i" -lt "$STOP_WAIT_SECS" ]; do
		sleep 1
		i=$((i + 1))
		shater_daemon_pid >/dev/null || {
			_slog -p daemon.info "restart: previous shaterd exited after ${i}s; starting a fresh one"
			return 0
		}
	done
	_slog -p daemon.warn \
		"restart: previous shaterd (pid $pid) still alive after ${STOP_WAIT_SECS}s — starting anyway"
	return 1
}

# --- procd lifecycle -------------------------------------------------------

start_service() {
	# NOTE: deliberately NO `shater_enabled` guard here. The daemon is the box's
	# only configuration surface (it serves the admin panel + the control socket
	# that `shaterd apply`/`mint-token` talk to), so it must be reachable BEFORE
	# the stack is enabled — otherwise a fresh install can only be configured by
	# hand-editing UCI over SSH. With globals.enabled=0 the daemon logs
	# "staying inert" and applies NOTHING (no engine, no nft, no ip rules), so
	# supervising it here cannot affect connectivity. Interception is still gated
	# on globals.enabled — see ACTIVE_FLAG below.

	# Guard: never claim to run without the daemon binary. A half-removed/failed
	# shaterd upgrade must degrade to "plugin off", not to a box that thinks
	# interception is live with nothing behind it.
	#
	# "Plugin off" now has to include DISARMING. With the boot armor in play, a
	# missing binary is the one case where the fail-closed plane could stand
	# forever with nothing able to replace it: the armor loads at START=21, the
	# daemon never starts, and every later boot repeats it. The product being gone
	# is not a security event — it is an uninstall — so the plane comes down and
	# the LAN returns to plain routing, loudly.
	if [ ! -x "$PROG" ]; then
		shater_clear_restart
		shater_disarm_boot
		rm -f "$ACTIVE_FLAG"
		nft delete table inet shater 2>/dev/null
		_slog -p daemon.err \
			"shaterd binary missing/not executable at $PROG — refusing to start; the fail-closed plane and its boot armor have been REMOVED (LAN back to plain routing, unprotected). Reinstall shaterd."
		return 0
	fi

	# Do not stand a new data plane up on top of one that is still being taken
	# down. On `restart` procd has only just SIGTERMed the previous instance and
	# returned; this is the handshake that makes `restart` == `stop` + pause +
	# `start`. It also keeps `migrate` below from rewriting UCI underneath a
	# daemon that is still reading it. No-op (and no delay) when nothing is
	# running, which is the boot case.
	shater_wait_stopped

	# The predecessor is gone and has already consumed the flag (it reads it in its
	# SIGTERM handler). Withdraw it now, so a LATER `stop` is unambiguous even if
	# this start fails further down.
	shater_clear_restart

	# Bring the UCI schema forward before the daemon reads it (idempotent;
	# refuses a newer schema) so an upgraded package never applies a stale config.
	#
	# THE FAILURE IS REPORTED, NOT SWALLOWED, AND IT IS NAMED. This is the only
	# place the schema migration runs at boot (`shaterd run`, the SIGHUP reconcile
	# and the panel's config write all read UCI directly), so if it fails here it
	# does not get retried until the next start — which is also why this, and not
	# the uci-defaults call, is the report that matters: it comes back at every
	# boot and every restart for as long as the problem lasts.
	#
	# The three outcomes are three different problems with three different fixes
	# (free space / put the newer package back / the file is unreadable), and
	# shater_migrate says which one it was instead of asserting the middle one at
	# all of them. Start regardless in every case: refusing to start would take the
	# admin panel down with it, and the panel is the only way to fix the box.
	shater_migrate

	procd_open_instance shater
	# shaterd runs in the FOREGROUND under procd (must never daemonize). `run` is
	# the long-lived daemon that owns the in-process engine + netplane. It reads
	# UCI itself — there is NO -c config file to point at.
	procd_set_param command "$PROG" run
	# threshold(3600) timeout(5) retries(0=INFINITE): procd must NEVER give up on
	# the daemon while our interception rules stand — an abandoned engine with
	# live TPROXY+DNS-hijack rules is a permanent LAN blackout. A genuine crash
	# loop retries every 5s (cheap); sustained death is escalated by the
	# shater-cron watchdog (which fails open / alerts per kill_switch policy).
	procd_set_param respawn 3600 5 0
	# NOTE: deliberately NO `procd_set_param file` config watch. A file-watch
	# would bounce the tunnel (dropping every proxied connection) on EVERY UCI
	# commit, even when nothing relevant changed. Config changes reach the daemon
	# through reload_service (below) instead, and the daemon's config-hash gate
	# decides whether an apply actually rebuilds the engine.
	procd_set_param stdout 1                                # -> logread
	procd_set_param stderr 1
	# Give the daemon room to run its honest teardown (engine.Close + netplane
	# restore) before procd SIGKILLs it.
	#
	# 30s, not 10s: an engine holding a few hundred outbounds closes its
	# urltest/observatory goroutines and flushes experimental.cache_file to FLASH
	# before the netplane teardown even starts, and on eMMC/NAND that alone can
	# outlast 10s. A SIGKILL there aborts the teardown at an arbitrary point and
	# leaves the plane HALF removed — the nft table gone but the policy routing
	# still installed, or vice versa — which is precisely the class of leftover
	# state the successor's idempotent fast-path cannot see and never repairs.
	# Shutdown is bounded by procd either way; we are only choosing where.
	procd_set_param term_timeout 30
	procd_close_instance

	# Mark the stack live for hotplug/cron — but ONLY when interception is
	# actually meant to stand. While globals.enabled=0 the daemon applies nothing,
	# so raising the flag would licence hotplug/cron to poke a data plane that does
	# not exist. The daemon raises/clears the flag itself around apply/teardown;
	# this is just the boot-time seed for the enabled case.
	if shater_enabled; then
		mkdir -p "$(dirname "$ACTIVE_FLAG")"
		: > "$ACTIVE_FLAG"
	else
		rm -f "$ACTIVE_FLAG"
	fi
}

stop_service() {
	# Say WHY we are stopping before procd sends the signal, because the daemon
	# cannot tell from the signal alone and the answer changes what it leaves in
	# the kernel. Two INDEPENDENT questions, and the old code conflated them into
	# one two-armed `case` whose else-branch answered both wrongly for `shutdown`:
	#
	#   1. IS A SUCCESSOR COMING (this process only)?  restart / reload.
	#      Raise RESTART_FLAG so the outgoing daemon replaces its data plane with
	#      the fail-closed HOLDING plane instead of removing it. The gap until the
	#      successor applies is not a moment: this script waits out the old
	#      process, runs `shaterd migrate`, then starts a daemon that must build an
	#      engine — all of it, before this flag existed, with `lan -> wan ACCEPT`
	#      and nothing else.
	#
	#   2. IS THE PRODUCT BEING SWITCHED OFF (across boots)?  `stop` — and only
	#      `stop`, and only when a PERSON is behind it (shater_stop_disarms; the
	#      package manager reaches us through `stop` too). Then the boot armor goes
	#      with it, so the next boot does not quietly reinstate what the operator
	#      just switched off — the same rule ACTIVE_FLAG has always enforced for
	#      hotplug/cron.
	#
	# `shutdown` answers NO to both, which is the defect this replaced: a reboot is
	# not a successor and it is certainly not an operator switching the product off.
	# It is the boot the armor exists for. An upgrade answers NO to the second for
	# the same kind of reason.
	if shater_action_handoff "$SHATER_RC_ACTION"; then
		shater_mark_restart
	else
		shater_clear_restart
	fi
	local in_pkg=0 rc_en=0
	shater_pkg_transaction && in_pkg=1
	shater_rc_enabled && rc_en=1
	if shater_stop_disarms "$SHATER_RC_ACTION" "$in_pkg" "$rc_en"; then
		shater_disarm_boot
	elif [ "$in_pkg" = "1" ] && shater_action_disarms "$SHATER_RC_ACTION"; then
		_slog -p daemon.info \
			"stop came from a package transaction that left the service enabled — keeping the boot armor, so being replaced cannot leave the next boot unprotected"
	fi

	# Drop the live-flag FIRST so a concurrent hotplug/cron tick cannot rebuild
	# what we are about to tear down. procd then sends SIGTERM to `shaterd run`,
	# which runs its OWN honest teardown (engine.Close + netplane restore) — we
	# deliberately do NOT tear nft/routing down from the shell here, both to
	# avoid racing the daemon and because the daemon is the single owner of that
	# state. A crashed daemon that left rules standing is caught by the
	# shater-cron watchdog.
	rm -f "$ACTIVE_FLAG"
}

reload_service() {
	# Fired by the `shater` config.change reload-trigger (LuCI Save & Apply /
	# reload_config). Simplest correct behaviour: stop + start. `stop` clears the
	# flag and SIGTERMs the daemon (honest teardown); `start` WAITS for that
	# teardown to actually finish (shater_wait_stopped) and then launches a fresh
	# `shaterd run` that reads the new UCI and applies it. When the stack is
	# disabled, `start` is a no-op, so a disable+apply cleanly tears everything
	# down. Because the wait lives in start_service, this path gets the same
	# stop-then-start ordering guarantee as `restart`.
	#
	# Marked EXPLICITLY as well as via SHATER_RC_ACTION: this is the path a routine
	# Save & Apply takes, so it is the one that must not depend on reading an
	# rc.common variable correctly. Belt and braces, one line.
	shater_mark_restart
	stop
	start
}

service_triggers() {
	procd_add_reload_trigger "shater"
}
