#!/bin/sh /etc/rc.common
# /etc/init.d/shater-cron — periodic auto-updater for subscriptions & rulesets,
# plus the data-plane watchdog (v0.2).
#
# A tiny procd-supervised loop that, once per tick, checks every enabled
# subscription against its per-item `update_interval` and runs
#   shaterd sub update <name>       (subscriptions)
# when the item is due, then a single `shaterd reconcile` if anything changed
# (the daemon's config-hash gate rebuilds the engine only on a real change).
# `config ruleset` items are NOT touched here — see shater_run_due for who owns
# their refresh and where the remaining gap is.
#
# RELIABILITY CONTRACT (same "железно" posture as /etc/init.d/shater):
#   * The loop body is fully INERT unless globals.enabled=1 AND the main shater
#     service is live (ACTIVE_FLAG raised by its start, cleared by its stop).
#     An admin `stop` of the main service therefore STICKS — this loop idles.
#   * Only ever invokes shaterd verbs / the shater init — never touches the
#     data plane directly.
#   * Due-ness is tracked with epoch stamp FILES under a tmpfs run dir, so no
#     dependence on `date -r`. Stamps are lost on reboot => every item is due at
#     boot, giving a fetch-at-boot. Fetches are skipped (not stamped) while the
#     box has no default route, so the boot-time fetch actually happens once WAN
#     is up instead of silently burning the attempt.
#   * A stamp is written ONLY on success; a failed fetch is retried after a
#     short backoff (RETRY_SECS) instead of waiting out the full interval.
#   * WATCHDOG: if interception is meant to be live but the daemon has been dead
#     for WATCHDOG_TICKS consecutive ticks, we escalate: with kill_switch=open
#     the main service is STOPPED (tears interception down — fail-open, LAN
#     returns to plain routing); with kill_switch=closed the rules stay
#     (blocked-by-design) and we log loudly.
#   * CRASH-LOOP WATCHDOG: the dead-daemon counter above cannot see the failure
#     it matters most for. /etc/init.d/shater sets `respawn 3600 5 0`, so a
#     daemon that dies a few seconds into startup is back within 5s and a single
#     `pidof` per 60s tick nearly always finds a process — the counter resets and
#     never reaches WATCHDOG_TICKS, while the fail-closed plane keeps the LAN shut
#     and the panel (served BY the daemon) never comes up. So the tick's sleep is
#     spent SAMPLING the daemon's identity instead of sleeping blind, and a tick in
#     which several different daemons lived is counted as churn. See
#     shater_churn_scan / shater_churn_verdict / shater_churn_action.
#   * The loop never self-exits (procd would respawn-churn an exiting body);
#     it idles on its guards instead. busybox ash only — no bashisms.

USE_PROCD=1
START=96          # order vs the main init (99) is irrelevant; loop self-guards
STOP=11

# `loop` is an internal action procd re-execs to run the periodic body.
EXTRA_COMMANDS="loop"
EXTRA_HELP="	loop	internal: run the periodic update loop (invoked by procd)"

INIT_SCRIPT=/etc/init.d/shater-cron
SHATER_INIT=/etc/init.d/shater
SHATERD=/usr/bin/shaterd
ACTIVE_FLAG=/var/run/shater.active
# Raised by /etc/init.d/shater around a restart/reload and cleared by the
# successor's start_service. Read here ONLY as a "a person/package asked for this
# bounce" veto on the crash-loop verdict — never as a liveness signal.
RESTART_FLAG=/var/run/shater.restarting
# Written by `shaterd run` itself (main.go writePidfile) before it builds anything,
# and removed by that same process on a clean exit. It is the only handle that
# names THE daemon: `pidof shaterd` also matches the short-lived CLI verbs this
# very loop runs (`sub update`, `reconcile`, `schedule due`).
PIDFILE=/var/run/shaterd.pid
STAMP_DIR=/var/run/shater/cron
TICK=60                          # seconds between due-checks
RETRY_SECS=300                   # backoff before retrying a FAILED fetch
WATCHDOG_TICKS=5                 # consecutive dead-daemon ticks before escalating
DEFAULT_SUB_INTERVAL=6h
DEFAULT_BL_INTERVAL=24h          # url blocklist refresh interval (D16)

# --- crash-loop watchdog tuning --------------------------------------------
#
# Every number here is chosen against ONE question: what can a legitimate restart
# produce? A legitimate bounce (`restart`, LuCI Save & Apply -> reload, a package
# transaction) replaces the daemon EXACTLY ONCE, and it is announced twice over —
# /etc/init.d/shater raises RESTART_FLAG in stop_service and clears ACTIVE_FLAG for
# the duration. A crash loop is announced by nothing and repeats without bound.
LOOP_POLL=5                      # seconds between identity samples inside one tick
LOOP_MIN_GENS=3                  # distinct daemons in ONE tick that count as churn
LOOP_WINDOWS=2                   # consecutive churn ticks before we call it a loop
LOOP_REPORT_TICKS=30             # do not repeat the report more often than this

# --- helpers ---------------------------------------------------------------

shater_enabled() {
	local en
	en=$(uci -q get shater.globals.enabled) || return 1
	[ "$en" = "1" ]
}

# Syslog line that honors globals.log_syslog (same helper as /etc/init.d/shater,
# with this service's tag): with log_syslog=0 the operator asked for a silent
# syslog and the watchdog/updater status lines respect that too. Absent option
# (or unreadable UCI) = ON.
_slog() {
	[ "$(uci -q get shater.globals.log_syslog)" = "0" ] || logger -t shater-cron "$@"
}

# The main service raised its live-flag (start) and has not stopped since.
shater_active() {
	[ -f "$ACTIVE_FLAG" ]
}

# Best-effort "do we have an uplink" check; fetching before WAN is up at boot
# would waste each item's one boot attempt.
shater_has_uplink() {
	ip route show default 2>/dev/null | grep -q '^default' && return 0
	ip -6 route show default 2>/dev/null | grep -q '^default'
}

# Stamp names embed the UCI item name; keep them to one safe path component.
shater_safe_name() {
	echo "$1" | sed 's/[^A-Za-z0-9._-]/_/g'
}

# Convert 30m|6h|24h|2d|3600 -> seconds on stdout. $2 is a fallback token used
# when the input is empty or malformed (the FALLBACK is honored, not a
# hardcoded constant).
shater_ivl_secs() {
	local v="$1" def="$2" n u
	[ -n "$v" ] || v="$def"
	n=$(echo "$v" | sed 's/[^0-9].*$//')
	u=$(echo "$v" | sed 's/^[0-9]*//')
	if [ -z "$n" ]; then
		# Malformed (no leading digits): fall back to the caller's default; if
		# that is somehow malformed too, 6h.
		v="$def"
		n=$(echo "$v" | sed 's/[^0-9].*$//')
		u=$(echo "$v" | sed 's/^[0-9]*//')
		[ -n "$n" ] || { echo 21600; return; }
	fi
	case "$u" in
		s|"") echo "$n" ;;
		m|M)  echo $(( n * 60 )) ;;
		h|H)  echo $(( n * 3600 )) ;;
		d|D)  echo $(( n * 86400 )) ;;
		*)    echo $(( n )) ;;
	esac
}

# Due if now - stamp >= interval. $1 stamp file, $2 interval seconds.
shater_due() {
	local stamp="$1" ivl="$2" now last
	now=$(date +%s)
	last=$(cat "$stamp" 2>/dev/null)
	[ -n "$last" ] || last=0
	[ $(( now - last )) -ge "$ivl" ]
}

shater_stamp() { mkdir -p "$STAMP_DIR"; date +%s > "$1"; }

# Failed fetch: pretend the last success was (interval - RETRY_SECS) ago so the
# item comes due again after a short backoff instead of a full interval — but
# never sooner than RETRY_SECS (guards tiny intervals).
shater_stamp_retry() {
	local stamp="$1" ivl="$2" back
	back=$(( ivl - RETRY_SECS ))
	[ "$back" -gt 0 ] || back=0
	mkdir -p "$STAMP_DIR"
	echo $(( $(date +%s) - back )) > "$stamp"
}

# Walk anonymous `config subscription` / `config ruleset` sections by index and
# run any that are due. Echoes non-empty on stdout if at least one item updated.
#
# WHERE A SUBSCRIPTION'S BYTES TRAVEL, and what this loop sees when they cannot.
#
# `shaterd sub update <name>` honours the subscription's own `fetch_via`:
#   * fetch_via != proxy — fetched DIRECT by the short-lived CLI process, exactly
#     as it always was. Needs no daemon; works at cold start and first boot.
#   * fetch_via == proxy — DELEGATED to the running daemon over its control
#     socket, because only the daemon owns the engine the fetch has to travel
#     through. Until 2026-07-27 the CLI printed one `daemon.warn` line and
#     fetched DIRECT instead, so every tick of THIS loop and every fetch-at-boot
#     put the feed URL and the router's real address on the plain WAN — while the
#     panel's Refresh button (the same operation, through the daemon) worked, so
#     it looked configured. It now FAILS instead, exit 1.
#
# CONSEQUENCE HERE, stated rather than discovered later: with a proxy-fetched
# subscription and no live daemon, the `if` below takes the else arm every time —
# one syslog line and shater_stamp_retry, i.e. a fresh attempt every RETRY_SECS
# (300s) for as long as the daemon is down. That is ~288 lines a day per such
# subscription. It is NOT the `ruleset update` situation this file used to have:
# there the work had no owner and the retry could never succeed, whereas here the
# retry succeeds within RETRY_SECS of the daemon coming back, and the loop only
# runs at all while `shater_enabled && shater_active` — a dead daemon is already
# being escalated by shater_watchdog, and with kill_switch=open it stops the stack
# (clearing ACTIVE_FLAG), which makes this loop inert. Deliberately left noisy:
# a subscription that is silently going stale is worse than a repeated line.
shater_run_due() {
	local i name en ivl secs stamp changed=""

	# Subscriptions.
	i=0
	while uci -q get "shater.@subscription[$i]" >/dev/null 2>&1; do
		name=$(uci -q get "shater.@subscription[$i].name")
		en=$(uci -q get "shater.@subscription[$i].enabled")
		[ -z "$en" ] && en=1
		if [ -n "$name" ] && [ "$en" = "1" ]; then
			ivl=$(uci -q get "shater.@subscription[$i].update_interval")
			secs=$(shater_ivl_secs "$ivl" "$DEFAULT_SUB_INTERVAL")
			stamp="$STAMP_DIR/sub.$(shater_safe_name "$name")"
			if shater_due "$stamp" "$secs"; then
				if "$SHATERD" sub update "$name" >/dev/null 2>&1; then
					shater_stamp "$stamp"
					changed=1
				else
					_slog -p daemon.warn \
						"sub update '$name' failed; retrying in ${RETRY_SECS}s"
					shater_stamp_retry "$stamp" "$secs"
				fi
			fi
		fi
		i=$(( i + 1 ))
	done

	# RULE-SETS ARE NOT UPDATED FROM HERE, AND NEVER WERE.
	#
	# There used to be a second loop that ran `shaterd ruleset update <name>` for
	# every `config ruleset` with source=url. That verb has never existed: it
	# printed a note and exited 0, so this loop stamped the item as freshly updated
	# and raised `changed`, which cost a reconcile per item per interval and told
	# the operator the list was current when not one byte had been fetched. The verb
	# now exits non-zero (shater/cmd/shaterd/main.go, notImpl), which turns the same
	# loop into one failed attempt and one syslog line every RETRY_SECS — ~288 lines
	# a day, per rule-set, about work that has no owner here. Noise in the log hides
	# real problems as effectively as a lie about success does.
	#
	# WHO REFRESHES A url RULE-SET NOW, so the next reader does not think this was
	# forgotten. `source=url` splits into two shapes in shater/generate/ruleset.go:
	#
	#   * the URL serves an engine-native .srs/.json -> it stays a REMOTE rule-set
	#     and sing-box owns fetch/cache/refresh through RemoteRuleSet.UpdateInterval
	#     on the running box. This cron loop never had anything to contribute.
	#   * the URL serves a plain-text list -> it is compiled locally into
	#     /etc/shater/lists/<tag>.srs, and that artifact is refreshed by the
	#     GENERATOR, "when missing or older than update_interval" — i.e. only when
	#     something else already caused a generate. Nothing schedules one, so this
	#     shape has NO periodic refresh at all today. That is a real gap, and it is
	#     stated here rather than papered over with a call to a verb that does
	#     nothing: closing it needs a daemon-side timer (or a real `ruleset update`),
	#     not a shell loop, because only the daemon can force a rebuild past the
	#     config-hash gate.
	#
	# `config blocklist` url items are a different mechanism and DO refresh — see
	# shater_run_due_blocklists below.

	[ -n "$changed" ] && echo 1
}

# Walk anonymous `config blocklist` sections and, when a url blocklist is due,
# force a DNS-filter refresh via `shaterd blocklist update` (D16). Unlike subs /
# rulesets, that verb is a GLOBAL reconcile — it re-fetches EVERY remote rule-set
# on the fresh box — so it is fired at most ONCE per tick even when several url
# blocklists are due, but each due list is still stamped. It does its own
# reconcile, so this function does NOT feed the `changed` reconcile above (no
# double-signal). No-op unless at least one enabled url blocklist exists.
shater_run_due_blocklists() {
	local i name en src ivl secs stamp fired=""
	i=0
	while uci -q get "shater.@blocklist[$i]" >/dev/null 2>&1; do
		name=$(uci -q get "shater.@blocklist[$i].name")
		en=$(uci -q get "shater.@blocklist[$i].enabled")
		[ -z "$en" ] && en=1
		src=$(uci -q get "shater.@blocklist[$i].source")
		if [ -n "$name" ] && [ "$en" = "1" ] && [ "$src" = "url" ]; then
			ivl=$(uci -q get "shater.@blocklist[$i].update_interval")
			secs=$(shater_ivl_secs "$ivl" "$DEFAULT_BL_INTERVAL")
			stamp="$STAMP_DIR/bl.$(shater_safe_name "$name")"
			if shater_due "$stamp" "$secs"; then
				if [ -z "$fired" ]; then
					if "$SHATERD" blocklist update >/dev/null 2>&1; then
						fired=ok
					else
						fired=fail
					fi
				fi
				if [ "$fired" = "ok" ]; then
					shater_stamp "$stamp"
				else
					_slog -p daemon.warn \
						"blocklist update '$name' failed; retrying in ${RETRY_SECS}s"
					shater_stamp_retry "$stamp" "$secs"
				fi
			fi
		fi
		i=$(( i + 1 ))
	done
}

# Watchdog body for one tick: detect "interception live, daemon dead". $1 is the
# running dead-tick count; echoes the updated count. The loop only calls this
# while enabled+active, so a missing `shaterd run` process here means the engine
# that should be holding interception up is gone.
shater_watchdog() {
	local dead="$1" ks
	if pidof shaterd >/dev/null 2>&1; then
		echo 0
		return
	fi
	dead=$(( dead + 1 ))
	if [ "$dead" -eq "$WATCHDOG_TICKS" ]; then
		ks=$(uci -q get shater.globals.kill_switch)
		if [ "$ks" = "closed" ]; then
			# Fail-closed is a POLICY: dead engine + standing rules == traffic
			# blocked, which is what the admin asked for. Keep it, but say so.
			_slog -p daemon.crit \
				"shaterd dead for $(( dead * TICK ))s with interception live; kill_switch=closed keeps LAN blocked — fix the daemon or /etc/init.d/shater stop"
		else
			# Fail-open: durably stop the stack (clears the live-flag; the daemon
			# — or its next respawn — tears interception down) so the LAN returns
			# to plain routing.
			_slog -p daemon.crit \
				"shaterd dead for $(( dead * TICK ))s with interception live; kill_switch=open — stopping shater (fail-open, LAN back to plain routing)"
			"$SHATER_INIT" stop
			dead=0
		fi
	fi
	echo "$dead"
}

# --- crash-loop watchdog ----------------------------------------------------
#
# THE HOLE. shater_watchdog above answers "is a daemon there?" once every TICK
# seconds. /etc/init.d/shater sets `respawn 3600 5 0`, so a daemon that dies a few
# seconds into startup is back 5s later and that one sample nearly always finds a
# process: the dead-counter resets, never reaches WATCHDOG_TICKS, and the single
# failure the watchdog exists for — a new binary or a bad config that cannot get
# an engine up while the fail-closed plane holds the LAN shut — is the one it can
# never see. The daemon also serves the panel, so in that state the operator has
# neither internet nor a way to look at the box.
#
# THE SIGNAL. Not "is it there" but "is it the SAME one". The tick's sleep is
# spent taking an identity sample every LOOP_POLL seconds instead of sleeping
# blind, and a tick in which LOOP_MIN_GENS different daemons lived is a churn
# tick. LOOP_WINDOWS consecutive churn ticks is the verdict.
#
# WHY A LEGITIMATE RESTART CANNOT REACH IT. Four independent reasons, in order of
# how much they are relied on:
#
#   1. A bounce replaces the daemon ONCE. One restart scores 2 generations in the
#      tick it happens in and 1 in every tick after, so it cannot even produce a
#      single churn tick at LOOP_MIN_GENS=3, let alone LOOP_WINDOWS of them in a
#      row. Reaching the verdict takes >= 4 replacements inside 2 consecutive
#      minutes, >= 2 in each.
#   2. Bounces are ANNOUNCED. /etc/init.d/shater raises RESTART_FLAG in
#      stop_service and clears ACTIVE_FLAG for the whole stop->start, and either
#      one seen in any sample of a tick discards that tick outright.
#   3. The panel's Apply does not restart anything: it writes UCI and applies over
#      the daemon's control socket, in process. Only `restart`, a LuCI Save &
#      Apply (config.change -> reload) and a package transaction bounce the
#      daemon, and a human cannot produce those at four a minute.
#   4. The sample names THE daemon via its pidfile, not `pidof shaterd` — the
#      short-lived CLI verbs this very loop runs share that process name.
#
# WHAT IT DOES NOT COVER, stated rather than implied: a daemon that dies
# INSTANTLY (well under a second) is almost never caught alive by a 5s sample, so
# it scores few generations and this detector stays quiet. That case is exactly
# the one the existing dead-tick counter does see — its `pidof` misses too, tick
# after tick — so the two cover opposite ends and are deliberately left as two
# independent instruments rather than merged into one clever number.

# One identity sample: echoes the pid of the live `shaterd run`, or "-" for none.
#
# Through the PIDFILE, which `shaterd run` writes before it builds anything and
# removes on a clean exit, because that is the only handle that names THE daemon:
# `pidof shaterd` also matches `shaterd sub update` / `reconcile` / `schedule due`.
# /proc/<pid>/comm is checked so a stale pidfile whose pid has been reused by an
# unrelated process cannot read as a live daemon. No forks: `read` is a builtin.
shater_sample_pid() {
	local pid="" comm=""
	[ -r "$PIDFILE" ] && read -r pid 2>/dev/null < "$PIDFILE"
	case "$pid" in
		''|*[!0-9]*) echo -; return ;;
	esac
	[ -r "/proc/$pid/comm" ] && read -r comm 2>/dev/null < "/proc/$pid/comm"
	[ "$comm" = "shaterd" ] || { echo -; return; }
	echo "$pid"
}

# shater_churn_scan <sample>...  ->  "<generations> <absent-samples>"
#
# A GENERATION is one distinct daemon lifetime observed during the tick: a live
# pid that differs from the last live pid seen. A daemon that simply keeps running
# therefore scores exactly 1 generation and 0 absent samples for as long as it
# runs — the signal is flat unless something is actually being replaced.
#
# A GAP (samples with no daemon at all, e.g. procd's 5s respawn hole) is counted
# but does NOT by itself open a new generation: only a different pid does. An
# earlier draft reset the comparison across a gap so that "same pid seen again
# after a gap" would score two. That case cannot occur — a respawn always gets a
# fresh pid — so it was unfalsifiable code, and resetting also meant a momentarily
# unreadable pidfile could inflate the count. Not resetting is both simpler and
# the safer direction.
#
# Pure: no I/O, no globals, every input on the command line. That is what lets the
# gate drive it with synthetic sample streams instead of a live router.
shater_churn_scan() {
	local gens=0 absent=0 last="" s
	for s in "$@"; do
		if [ "$s" = "-" ]; then
			absent=$(( absent + 1 ))
			continue
		fi
		[ "$s" = "$last" ] || gens=$(( gens + 1 ))
		last="$s"
	done
	echo "$gens $absent"
}

# shater_churn_verdict <gens> <samples> <announced> <churn-so-far>
#   -> the new consecutive-churn-tick count
#
# Also pure. `announced`=1 means a sample during the tick saw RESTART_FLAG up or
# ACTIVE_FLAG down, i.e. /etc/init.d/shater said out loud that it was bouncing the
# daemon: that tick proves nothing and resets the run. A tick with no samples at
# all (the first pass through the loop) likewise scores 0 rather than guessing.
shater_churn_verdict() {
	local gens="$1" n="$2" announced="$3" churn="$4"
	[ "$announced" = "1" ] && { echo 0; return; }
	[ "$n" -gt 0 ] || { echo 0; return; }
	if [ "$gens" -ge "$LOOP_MIN_GENS" ]; then
		echo $(( churn + 1 ))
		return
	fi
	echo 0
}

# shater_churn_action <churn-ticks> <kill_switch>  ->  none | log | stop
#
# WHAT TO DO, and why it is not our call to make twice. A crash loop leaves the
# box in the same state a dead daemon does — no engine, fail-closed plane standing
# — so the answer is the one the operator already gave with kill_switch, not a new
# policy invented here:
#
#   open   The operator asked for connectivity over interception. Stop the stack,
#          exactly as shater_watchdog does for a sustained-dead daemon: the plane
#          comes down and the LAN returns to plain routing. It also disarms the
#          boot armor, so the NEXT boot is clean too instead of repeating the loop
#          behind a closed LAN. Nothing else can end the loop: procd's retries are
#          infinite by design.
#   closed The operator asked for blocked-over-leaking. Blocked is what they get,
#          and opening their LAN from a background loop would be the opposite of
#          what the knob says. Report it loudly and let the person decide; the
#          message names the one command that opens it.
#
# The list is POSITIVE and CLOSED, and the fall-through goes to `log`: an absent
# or unrecognised kill_switch is the model's documented default ("closed", see
# shater/model/model.go DefaultGlobals), and `log` is the recoverable side — it
# changes nothing and can be acted on, where a wrong `stop` silently drops a
# household onto the unproxied WAN.
#
# NOTE (not changed here, deliberately): shater_watchdog above answers the same
# question with `if closed ... else stop`, so for an ABSENT kill_switch it fails
# open — the opposite of the documented default. It is left alone because that
# behaviour predates this file's crash-loop work; it is reported upward instead.
shater_churn_action() {
	local churn="$1" ks="$2"
	[ "$churn" -ge "$LOOP_WINDOWS" ] || { echo none; return; }
	case "$ks" in
		open)   echo stop ;;
		closed) echo log ;;
		*)      echo log ;;
	esac
}

# Sleep out one tick in LOOP_POLL slices, sampling the daemon's identity as we go.
# Publishes CHURN_SAMPLES / CHURN_N / CHURN_ANNOUNCED for the next pass of loop().
# Deliberately NOT a subshell (globals must survive), and it always returns 0 so a
# false `[ -f ]` at the end cannot look like a failure.
shater_tick_sample() {
	local slept=0
	CHURN_SAMPLES=""
	CHURN_N=0
	CHURN_ANNOUNCED=0
	while [ "$slept" -lt "$TICK" ]; do
		sleep "$LOOP_POLL"
		slept=$(( slept + LOOP_POLL ))
		CHURN_SAMPLES="$CHURN_SAMPLES $(shater_sample_pid)"
		CHURN_N=$(( CHURN_N + 1 ))
		[ -f "$RESTART_FLAG" ] && CHURN_ANNOUNCED=1
		[ -f "$ACTIVE_FLAG" ] || CHURN_ANNOUNCED=1
	done
	return 0
}

# loop: the foreground body supervised by procd. Never exits on its own — it
# idles while disabled/inactive so procd is not respawn-churned by a
# self-exiting body when the stack is off.
loop() {
	# CRITICAL: drop the rc.common per-service flock before entering the eternal
	# body. For a USE_PROCD init script, rc.common sources /lib/functions/procd.sh
	# on EVERY action (enable/start/loop alike), and procd.sh's tail runs
	# `_procd_wrapper` -> `procd_lock`, which takes an EXCLUSIVE BLOCKING flock on
	# fd 1000 -> /var/lock/procd_shater-cron.lock for the lifetime of the process.
	# Every other action releases it by exiting — but `loop` never exits, so it
	# would hold that lock FOREVER. Every later `/etc/init.d/shater-cron <action>`
	# then blocks on the flock — including the `enable`/`start` that base-files'
	# default_postinst runs synchronously inside a live opkg/apk transaction,
	# which deadlocks the package manager (observed: `apk add` hung forever in
	# shater-core's post-install on BananaWRT 25.12). Closing fd 1000 releases
	# the flock immediately and keeps children (sleep/shaterd) from inheriting
	# it. A no-op where fd 1000 is not open (older procd.sh without procd_lock).
	exec 1000>&-
	local changed dead=0 churn=0 quiet=0 scan gens absent ks act
	# No tick has been sampled yet on the first pass; shater_churn_verdict scores
	# an empty tick as 0 rather than guessing.
	CHURN_SAMPLES=""
	CHURN_N=0
	CHURN_ANNOUNCED=0
	mkdir -p "$STAMP_DIR"
	while :; do
		if shater_enabled && shater_active; then
			if shater_has_uplink; then
				changed=$(shater_run_due)
				# url blocklists refresh independently (own reconcile, D16).
				shater_run_due_blocklists
			else
				changed=""
			fi
			if [ -n "$changed" ]; then
				# A sub/ruleset fetched new data: reconcile re-runs generate, which
				# ALSO re-evaluates time-scheduled rules for this tick — so no
				# separate `schedule due` is needed when we already reconcile.
				"$SHATERD" reconcile >/dev/null 2>&1
			else
				# Nothing else changed this tick: re-evaluate time-scheduled
				# (bedtime/school-hours) rules against the wall clock. `schedule
				# due` does its OWN reconcile in the daemon, hash-gated so it only
				# rebuilds the engine when a window boundary was actually crossed —
				# cheap to run every minute (a no-op between boundaries).
				"$SHATERD" schedule due >/dev/null 2>&1
			fi
			dead=$(shater_watchdog "$dead")

			# Crash-loop verdict on the tick that has just elapsed. Unquoted on
			# purpose: CHURN_SAMPLES is a whitespace-separated token list and word
			# splitting is how it becomes arguments.
			scan=$(shater_churn_scan $CHURN_SAMPLES)
			gens=${scan%% *}
			absent=${scan##* }
			churn=$(shater_churn_verdict "$gens" "$CHURN_N" "$CHURN_ANNOUNCED" "$churn")
			ks=$(uci -q get shater.globals.kill_switch)
			act=$(shater_churn_action "$churn" "$ks")
			case "$act" in
				stop)
					_slog -p daemon.crit \
						"shaterd is CRASH-LOOPING: $gens distinct daemons in the last ${TICK}s (absent in $absent of $CHURN_N samples), $churn such windows in a row — it is being respawned faster than it can bring an engine up. kill_switch=open, so shater is being STOPPED: interception comes down and the LAN returns to plain, UNPROXIED routing. Find the reason with 'logread -e shaterd', then '/etc/init.d/shater start'."
					"$SHATER_INIT" stop
					churn=0
					quiet="$LOOP_REPORT_TICKS"
					;;
				log)
					# Rate-limited: a standing condition, not an event. Never
					# silent for good, though — an operator who looks at the log an
					# hour later must still find it being said.
					if [ "$quiet" -le 0 ]; then
						_slog -p daemon.crit \
							"shaterd is CRASH-LOOPING: $gens distinct daemons in the last ${TICK}s (absent in $absent of $CHURN_N samples), $churn such windows in a row — it is being respawned faster than it can bring an engine up. kill_switch=${ks:-closed} keeps the fail-closed plane standing, so the LAN stays blocked and the admin panel is down with the daemon that serves it. Nothing is decided for you: find the reason with 'logread -e shaterd', or open the LAN with '/etc/init.d/shater stop'."
						quiet="$LOOP_REPORT_TICKS"
					fi
					churn=0
					;;
			esac
			[ "$quiet" -gt 0 ] && quiet=$(( quiet - 1 ))
		else
			dead=0
			churn=0
			quiet=0
		fi
		# Sleeps out the tick, sampling the daemon's identity while it does.
		shater_tick_sample
	done
}

# --- procd lifecycle -------------------------------------------------------

start_service() {
	[ -x "$SHATERD" ] || return 0
	# NOTE: deliberately NO `shater_enabled` guard here. The loop body re-checks
	# `shater_enabled && shater_active` on EVERY tick and idles when either is
	# false, so an outer guard buys nothing — but it costs correctness: the panel
	# enables the stack by writing UCI and applying over the daemon's control
	# socket, which emits no config.change event, so nothing would ever start this
	# loop afterwards. Subscriptions/rulesets/blocklists would then silently never
	# auto-update until the next reboot. Cost of always running: one `sleep 60`.

	procd_open_instance shater-cron
	# Re-exec ourselves as the loop body so procd supervises a single process.
	# Invoke through the rc.common shebang (NOT `/bin/sh $INIT_SCRIPT`) so the
	# `loop` action is dispatched by rc.common and procd tracks the resulting
	# foreground PID — running `/bin/sh <script> loop` detaches the loop from
	# procd (instance shows not-running, and procd respawn-churns it).
	procd_set_param command "$INIT_SCRIPT" loop
	procd_set_param respawn 3600 10 0     # threshold timeout always-retry
	procd_set_param file /etc/config/shater # restart the loop when UCI changes
	procd_set_param stdout 1
	procd_set_param stderr 1
	procd_close_instance
}

reload_service() {
	# UCI changed: bounce the loop so new intervals/items take effect. If the
	# stack was disabled, start re-guards to a no-op.
	stop
	start
}

service_triggers() {
	procd_add_reload_trigger "shater"
}
