#!/bin/sh
# /etc/uci-defaults/30_shater-core
#
# One-time, idempotent setup for shater-core. Runs on first boot (and via the
# default postinst on a live opkg/apk install). Must exit 0 so it is cleared and
# not retried. Everything here is safe to run more than once.

RT_TABLES=/etc/iproute2/rt_tables

# Append "<id>  <name>" to rt_tables only if neither the id nor the name is
# already present. Purely cosmetic (readable `ip rule`/`ip route` output);
# shaterd allocates the rest of the reserved block numerically.
seed_rt_table() {
	local id="$1" name="$2"
	[ -f "$RT_TABLES" ] || return 0
	grep -qE "^[[:space:]]*${id}[[:space:]]" "$RT_TABLES" && return 0
	grep -qE "[[:space:]]${name}[[:space:]]*\$" "$RT_TABLES" && return 0
	# A file without a trailing newline would otherwise get our entry glued onto
	# its last line, corrupting BOTH (and the glued line then defeats the grep
	# guards above, so every re-run would append again — breaking idempotency).
	[ -n "$(tail -c 1 "$RT_TABLES")" ] && printf '\n' >> "$RT_TABLES"
	printf '%s\t%s\n' "$id" "$name" >> "$RT_TABLES"
}

# Reserved routing-table base 0x2000 == 8192.
seed_rt_table 8192 shater

# Persistent home for sing-box's cache DB (experimental.cache_file, D16). The
# daemon also creates this on start; seeding it here means the very first apply
# lands the cache on /etc/shater/cache.db (survives reboot) instead of tmpfs.
mkdir -p /etc/shater

# NOTE: enable/restart of the init scripts is deliberately NOT done here inline
# — see the detached bring-up block at the end of this file. On a live
# opkg/apk install this script runs synchronously INSIDE the package manager's
# transaction, and any /etc/init.d/* invocation in that window risks blocking
# the transaction on rc.common's per-service flock (procd_lock).

# NO preset packs are seeded here, and the ones older releases seeded are removed.
#
# Until now this script created three `config preset` sections (block_ads,
# ru_bypass, private; all `enabled=0`) "so the LuCI Rules page renders their
# toggles". Both halves of that stopped being true in v0.2, and what was left was
# a knob wired to nothing:
#
#   * `preset` IS NOT A SECTION TYPE. The type switch in model.ParseUCIExport
#     (shater/model/uci.go) has no `preset` branch, and an unknown section type is
#     dropped on the floor rather than rejected — TestUnknownSectionAndOptionIgnored
#     pins that a config carrying one still parses, because the daemon has to come
#     up on whatever it finds. So `uci set shater.block_ads.enabled=1; uci commit`
#     edited the file and changed NOTHING about the running router, and there was
#     no error anywhere to say so. Nothing in shater/, panel/ or luci-app-shater
#     reads the type either.
#   * v0.2's luci-app-shater is a launcher for the daemon's own admin panel. There
#     is no LuCI Rules page for the toggles to appear on.
#   * The sections did not even persist. model.writeUCIWith replaces the WHOLE
#     package (`uci delete shater` + `uci import`), so the first save from the
#     panel deleted all three. A placeholder that erases itself reads as "this
#     broke", not as "this was never here" — which is worse than its absence.
#
# So the packs are not "missing": nothing in v0.2 lost a feature when the sections
# stopped being written, because nothing ever read them. And re-adding the seed is
# not how presets would come back. v0.1's packs were xray `geosite:`/`geoip:`
# matcher lists materialised into synthetic rules (`xrayctl/preset.go` on the v0.1
# branch); in v0.2 a rule's destination IS a `config ruleset` (schema v2), so the
# same pack is an ordinary ruleset + rule — which the panel's Routing page already
# builds, geosite/geoip sources included. Anything richer needs a section type the
# parser knows about, which has to land in shater/model FIRST; seeding UCI ahead
# of the parser only produces silence.
#
# The purge is narrow and safe by construction: it matches on the section TYPE
# being exactly `preset`, and that type is read by no consumer, so there is no
# setting to lose. Bounded and re-querying each round because a `config preset`
# may also be ANONYMOUS (`shater.@preset[0]`), where deleting from a list captured
# up front would shift the remaining indices out from under it.
purge_presets() {
	local s n=0 changed=""
	[ -f /etc/config/shater ] || return 0
	while [ "$n" -lt 32 ]; do
		s=$(uci -q show shater 2>/dev/null |
			sed -n 's/^shater\.\([^.=]*\)=preset$/\1/p' | head -n 1)
		[ -n "$s" ] || break
		uci -q delete "shater.$s" || break
		changed=1
		n=$((n + 1))
	done
	[ -n "$changed" ] && uci -q commit shater
	return 0
}
purge_presets

# Introduce the daemon-created `shater-l3*` TUN to fw4 (L3 ingress, D-L3). The
# daemon policy-routes LAN ICMP into that device from OUR nft table
# `inet shater`, but nftables runs EVERY table on every packet and a drop in
# any one of them wins — an accept in `inet shater` cannot override fw4. And
# fw4 WILL drop this forward: netifd knows nothing about a device the daemon
# creates at runtime, so it belongs to no zone and falls into fw4's zone-less
# defaults (REJECT). The device has to be declared to fw4 itself; it cannot be
# fixed from our own table.
#
# Seeded UNCONDITIONALLY (not gated on globals.l3_tunnel): uci-defaults run
# once, so gating on the option would require re-running this script when the
# option is flipped later — which never happens. An idle zone is harmless: its
# device match is a plain iifname/oifname STRING compare that simply never hits
# while the TUN does not exist.
#
# Idempotency: `config zone`/`config forwarding` are normally ANONYMOUS
# sections, and a naive `uci add firewall zone` would append a duplicate on
# every re-run (uci-defaults re-run on package upgrade/reinstall). The zone is
# NAMED instead, guarded by an existence check — a re-run re-finds the section
# and touches nothing. The forwardings are named too where the name is free, but
# their guard is a scan of the actual src/dest pairs, which is stronger; see
# seed_l3_forwarding below.
seed_l3_zone() {
	# No fw4 on this image (bare nftables build) => nothing drops the forward
	# on fw4's behalf and there is nothing to punch through.
	[ -f /etc/config/firewall ] || return 0
	if ! uci -q get firewall.shater_l3 >/dev/null; then
		uci set firewall.shater_l3=zone
		uci set firewall.shater_l3.name='shater_l3'
		uci set firewall.shater_l3.input='REJECT'
		uci set firewall.shater_l3.output='ACCEPT'
		uci set firewall.shater_l3.forward='REJECT'
		uci set firewall.shater_l3.masq='0'
		# INERT TODAY, kept for the day it is not. mtu_fix clamps forwarded TCP
		# MSS to the route MTU — but the L3 TUN is 65535 (deliberately: at any
		# smaller value the kernel fragments into the device, and the flow
		# dispatcher refuses to judge a fragment and lets the stack forge the
		# echo reply — see l3MTU in shater/generate/inbound.go), so the clamp has
		# nothing to clamp to. And only ICMP is ever marked into this device, so
		# no TCP rides here to be clamped in the first place. It earns its keep
		# the moment either of those changes; removing it would make that day
		# silent.
		uci set firewall.shater_l3.mtu_fix='1'
		# `list device`, deliberately NOT the usual `list network`: fw4
		# resolves a zone's networks through netifd, and netifd never learns
		# about a device the daemon creates at runtime — a stub interface
		# (proto none) would need to be brought UP to contribute an l3_device,
		# and nothing ever brings it up, so `list network` resolves to an
		# EMPTY device set and fw4 keeps dropping the forward. `list device`
		# instead compiles to an iifname/oifname STRING match, valid before
		# the TUN exists and matching from the moment shaterd creates it —
		# no netifd involvement and no firewall reload at enable time. Do not
		# "normalize" this to `list network` in a refactor; it breaks silently.
		#
		# The WILDCARD is load-bearing too. The daemon no longer opens one fixed
		# device: it alternates between `shater-l3a` and `shater-l3b` so that a
		# new engine generation never has to reopen the name the previous one is
		# still holding (that collision — TUNSETIFF: device or resource busy —
		# took the whole LAN down on the production router, because the recovery
		# path rebuilt the same config and hit the same busy name). fw4 compiles
		# `shater-l3*` to `iifname "shater-l3*"` / `oifname "shater-l3*"`,
		# verified on ImmortalWrt 25.12.1 with nftables 1.1.6, so ONE zone covers
		# every slot and no firewall reload is needed when the slot changes.
		uci add_list firewall.shater_l3.device='shater-l3*'
	fi
	seed_l3_forwardings
	uci -q commit firewall
}

# EVERY zone gets a forwarding into shater_l3, not just `lan`.
#
# The bug this closes is silent by construction. The daemon's divert set is built
# from every enabled `config inbound`'s network PLUS every device a rule names
# through an `iface:`/`zone:` source (shater/netplane/nft.go, nftDivertRefs) — so
# on a router with several LAN zones, ICMP from ALL of them is marked and routed
# into the TUN by our table. Our table then accepts it and fw4 drops it anyway,
# because the forward is judged in `forward_<source zone>` and only `lan` had a
# jump to `accept_to_shater_l3`. Result: ping through the tunnel works from one
# subnet and not from the next, with nothing in any log to say why — fw4's drop
# is the zone's policy verdict, not a rule with a name. The owner's production
# router has a single LAN zone, which is exactly why this went unnoticed; his
# second router has three.
#
# Every zone, including an uplink zone, and that is deliberate rather than lazy:
#
#   - The alternative is guessing which zones hold clients, and every available
#     signal is wrong somewhere. `masq='1'` marks the WAN on a stock config and
#     also marks a double-NAT LAN. The name `wan*` is a convention, not a rule.
#     A guess that is wrong reintroduces exactly the silent breakage above, while
#     a superfluous entry costs a line of ruleset.
#   - A forwarding into shater_l3 permits nothing on its own. It authorises the
#     forward of packets ROUTED INTO the TUN, and the only thing that routes a
#     packet there is our own fwmark rule, which matches solely on the divert
#     device set. A packet arriving on the WAN is not marked and never reaches
#     this decision; if an operator ever puts a WAN device in the divert set,
#     they meant to and this is the entry that makes it work.
#   - The reverse direction is NOT opened: no `src shater_l3` forwarding exists,
#     so nothing comes out of the TUN into a zone by way of these sections. The
#     engine's own replies return on the conntrack `established,related accept`
#     at the top of fw4's forward chain.
#
# LIMIT, stated because it is not obvious: this is a SNAPSHOT. uci-defaults run
# at first boot and on package install/upgrade, so a zone created AFTER the last
# shater-core install has no forwarding until the next one. Re-running this
# script (or reinstalling the package) re-seeds. The durable fix belongs in the
# daemon, which recomputes the divert set on every apply and already knows which
# zones are in it; it is deliberately not attempted from here.
seed_l3_forwardings() {
	uci -q show firewall 2>/dev/null |
		sed -n "s/^firewall\.\([^.=]*\)=zone\$/\1/p" |
		while read -r sid; do
			zone=$(uci -q get "firewall.$sid.name")
			# Unnamed zone: fw4 cannot reference it from a forwarding either.
			[ -n "$zone" ] || continue
			# Our own zone: a forwarding from shater_l3 to itself is meaningless.
			[ "$zone" = "shater_l3" ] && continue
			seed_l3_forwarding "$zone"
		done
}

# One `config forwarding` <zone> -> shater_l3, created only if no such forwarding
# exists yet.
#
# The guard scans the ACTUAL src/dest pairs rather than trusting a section id,
# which covers all three ways one can already be there: the legacy named section
# `shater_l3_fwd` seeded by earlier releases (src=lan), the per-zone names this
# function writes, and an anonymous one an operator added by hand. Without that,
# a re-run — uci-defaults re-run on every package upgrade — would append a
# duplicate for `lan` on every upgrade.
seed_l3_forwarding() {
	local zone="$1" sid found

	found=$(uci -q show firewall 2>/dev/null |
		sed -n "s/^firewall\.\([^.=]*\)=forwarding\$/\1/p" |
		while read -r f; do
			[ "$(uci -q get "firewall.$f.dest")" = "shater_l3" ] || continue
			[ "$(uci -q get "firewall.$f.src")" = "$zone" ] || continue
			echo yes
			break
		done)
	[ -n "$found" ] && return 0

	# Section ids are [a-zA-Z0-9_] only, while a zone name may legally carry a
	# hyphen — sanitise, and keep the legacy id for `lan` so an existing install
	# is recognised as already seeded rather than gaining a second section.
	if [ "$zone" = "lan" ]; then
		sid="shater_l3_fwd"
	else
		sid="shater_l3_fwd_$(printf '%s' "$zone" | sed 's/[^a-zA-Z0-9_]/_/g')"
	fi
	# The id may still be taken — by a section for a DIFFERENT zone whose name
	# sanitises to the same thing, or by something else entirely. Fall back to an
	# anonymous section rather than overwrite: the src/dest scan above is what
	# makes this idempotent, the name is only there to be readable.
	if uci -q get "firewall.$sid" >/dev/null; then
		sid=$(uci add firewall forwarding) || return 0
	else
		uci set "firewall.$sid=forwarding"
	fi
	uci set "firewall.$sid.src=$zone"
	uci set "firewall.$sid.dest=shater_l3"
}
seed_l3_zone

# Upgrade path for routers seeded by a pre-slot build.
#
# The block above only runs when the zone does NOT exist, which is exactly right
# for idempotency and exactly wrong here: an already-installed router has the
# zone with the OLD exact device `shater-l3`, that name matches no slot, and fw4
# would go back to dropping the forward — i.e. LAN ping through the tunnel dies
# silently on upgrade while everything reports healthy. Rewrite it in place.
#
# Narrow on purpose: only the literal legacy entry is replaced, and only when the
# wildcard is not already listed, so an operator who added devices of their own
# keeps them and a re-run changes nothing (uci-defaults re-run on every package
# upgrade). No `fw4 reload` here — uci-defaults run before the firewall starts on
# boot, and on a package upgrade the daemon's next apply is what needs the zone,
# not this script.
migrate_l3_zone_wildcard() {
	[ -f /etc/config/firewall ] || return 0
	uci -q get firewall.shater_l3 >/dev/null || return 0
	devs=$(uci -q get firewall.shater_l3.device) || return 0
	case " $devs " in
		*" shater-l3* "*) return 0 ;;   # already migrated
		*" shater-l3 "*) ;;             # legacy exact name present
		*) return 0 ;;
	esac
	uci -q del_list firewall.shater_l3.device='shater-l3'
	uci add_list firewall.shater_l3.device='shater-l3*'
	uci -q commit firewall
}
migrate_l3_zone_wildcard

# Bring the UCI schema forward on upgrade (idempotent; refuses a newer schema).
#
# THE RESULT IS NO LONGER THROWN AWAY. `>/dev/null 2>&1` discarded stdout, stderr
# AND the exit status, so a refusal here was indistinguishable from success — on
# the one screen the operator who caused it is actually looking at.
#
# WHY THIS STILL `exit 0`s (the script ends with one, and this block does not
# change that). A uci-defaults script that exits non-zero is NOT deleted and runs
# again at every boot. That is the wrong trade here, three times over:
#
#   * This file does far more than migrate — it seeds rt_tables, the shater_l3
#     fw4 zone and its per-zone forwardings, applies sysctl, and launches a
#     DETACHED BRING-UP that enables and RESTARTS shater/shater-cron and reloads
#     the firewall. Re-running all of that at every boot to carry one bit of "the
#     migration failed" would bounce the tunnel on every boot, after S99 had
#     already started it. One recoverable failure would become permanent churn.
#   * The exit status is not a reporting channel: nothing reads a uci-defaults
#     script's status, and neither procd nor the package manager surfaces it. It
#     buys no diagnosis, only the re-run.
#   * The retry it would buy already exists, and is better. /etc/init.d/shater
#     runs the same migration on EVERY start with the full classified report, so
#     a failure is retried at every boot and every restart regardless. And
#     re-running cannot fix either real cause anyway: a downgrade is fixed by
#     installing the right package, a full /overlay by freeing space.
#
# So: capture the status, name the cause, record it durably, and exit 0. The
# script has done everything it can do, and the failure is not lost.
#
# The classification is the same closed positive list /etc/init.d/shater uses
# (shater_migrate_class there, with the long argument for why the list is closed
# and why `downgrade` is matched on the binary's own words). It is repeated here
# rather than shared because the package installs no shell library the two could
# both source; TestMigrateClassifiersAgree (shater/cmd/shaterd) runs both over
# the same inputs and fails if they ever disagree.
SHATER_MIGRATE_BREADCRUMB=/etc/shater/migrate-failed

shater_migrate_class() {
	local rc="$1" out="$2"
	[ "$rc" = "0" ] && { echo ok; return 0; }
	case "$out" in
		*"newer than this build"*) echo downgrade; return 0 ;;
	esac
	uci -q export shater >/dev/null 2>&1 || { echo unreadable; return 0; }
	echo failed
}

# stderr, unconditionally: inside `apk add` / `opkg install` that is the package
# manager's own output, i.e. the installing operator's screen. Plus the durable
# breadcrumb on flash, which /etc/init.d/shater removes on the first successful
# migration. Deliberately NOT syslog — globals.log_syslog may be off by the
# operator's choice, and this path honours it by using other channels instead.
shater_migrate_shout() {
	echo "shater: $*" >&2
	mkdir -p "$(dirname "$SHATER_MIGRATE_BREADCRUMB")" 2>/dev/null
	echo "$(date -u '+%Y-%m-%dT%H:%M:%SZ') $*" \
		> "$SHATER_MIGRATE_BREADCRUMB" 2>/dev/null || :
}

run_migrate() {
	local out rc class schema
	[ -x /usr/bin/shaterd ] || return 0

	out=$(/usr/bin/shaterd migrate 2>&1)
	rc=$?
	class=$(shater_migrate_class "$rc" "$out")
	schema=$(uci -q get shater.globals.schema_version)
	case "$schema" in ""|*[!0-9]*) schema=0 ;; esac

	case "$class" in
		ok)
			rm -f "$SHATER_MIGRATE_BREADCRUMB"
			;;
		downgrade)
			shater_migrate_shout "install: UCI schema migration REFUSED — /etc/config/shater is schema v$schema and the shater build being installed understands an older one, so this is a DOWNGRADE. Nothing was migrated and nothing on disk was changed: your settings are intact, and also unchangeable, because the daemon and the panel refuse every config write for the same reason. Install a build that understands schema v$schema again (docs-shater/INSTALL.md has the pinned per-version feed). 'shaterd migrate' said: ${out:-no output}"
			;;
		unreadable)
			shater_migrate_shout "install: UCI schema migration FAILED and /etc/config/shater CANNOT BE READ ('uci export shater' fails), so the schema on disk cannot even be named. Check the file by hand before configuring anything. 'shaterd migrate' said: ${out:-no output}"
			;;
		failed)
			shater_migrate_shout "install: UCI schema migration FAILED for a reason this script does not recognise; the config is still at schema v$schema. The mundane cause is a full /overlay ('uci commit' cannot write) — check 'df /overlay'. /etc/init.d/shater retries this on every start and reports it there too. 'shaterd migrate' said: ${out:-no output}"
			;;
		*)
			# Unreachable: shater_migrate_class returns a closed set, all of it
			# handled above. Named rather than swept up, so a script that disagrees
			# with itself says so instead of picking a confident branch and being
			# wrong quietly.
			shater_migrate_shout "install: INTERNAL — 'shaterd migrate' produced a result this script cannot classify (class='$class', rc=$rc). That is a bug in /etc/uci-defaults/30_shater-core. Output was: ${out:-no output}"
			;;
	esac
	return 0
}
run_migrate

# Apply our sysctl knobs NOW (boot applies them via procd's sysctl service, but
# on a live opkg/apk install nothing else re-reads sysctl.d — without this, an
# install->enable->apply flow hits the rp_filter "rules match but nothing works"
# failure until the first reboot). Idempotent; unknown keys are ignored.
[ -f /etc/sysctl.d/99-shater.conf ] && sysctl -p /etc/sysctl.d/99-shater.conf >/dev/null 2>&1

# Bring the services up (enable + restart) — but DETACHED, never from this
# process. Why not inline: on a live `opkg install` / `apk add` this script is
# sourced synchronously by base-files' default_postinst, i.e. INSIDE the package
# manager's transaction. For a USE_PROCD init script EVERY rc.common action
# (even plain `enable`) first sources /lib/functions/procd.sh, whose
# `_procd_wrapper` -> `procd_lock` takes a BLOCKING exclusive flock on
# /var/lock/procd_<name>.lock. If any process spawned during the same
# transaction still holds that lock, the postinst blocks on the flock while the
# package manager waits on the postinst — a deadlock that hangs `apk add`
# forever (observed on BananaWRT 25.12; opkg holds its lock the same way).
# So the bring-up runs as a fully detached background job that first waits for
# the package manager to finish its transaction, then enables + restarts.
#
# Why still bring up at all (and not just `enable`): the admin panel is served
# BY the daemon, so on a live install nothing would run until the next reboot —
# the box would be uninstallable-then-unconfigurable in one step. `restart`
# (not `start`) also makes an upgraded init script / daemon binary take effect
# immediately. Safe in every context: with globals.enabled=0 the daemon is
# inert (no engine, no nft, no routing), and at FIRST boot (uci-defaults run by
# procd, no apk/opkg alive — the wait falls through instantly) this simply
# pre-starts what the S99 symlink would start moments later (procd dedupes the
# identical instance).
#
# Detach rules (both known package-manager gotchas):
#   * ALL fds go to /dev/null — apk waits for EOF on the postinst's stdout pipe,
#     so a background child keeping that pipe open would hang the transaction
#     just as surely as the flock does.
#   * The job stays short-lived (bounded wait + idempotent service calls): it
#     may inherit the package manager's own open lock fds, which must not be
#     kept alive long after the transaction.
#   * `setsid` (own session, survives any process-group signalling) when
#     available, plain background otherwise; busybox `timeout` as a last-resort
#     bound so a wedged service call can never leave an immortal orphan.
SHATER_BRINGUP='
	i=0
	while [ "$i" -lt 120 ]; do
		pidof apk >/dev/null 2>&1 || pidof opkg >/dev/null 2>&1 || break
		sleep 1
		i=$((i + 1))
	done
	[ -x /etc/init.d/shater ] && /etc/init.d/shater enable
	[ -x /etc/init.d/shater-cron ] && /etc/init.d/shater-cron enable
	# The boot-time fail-closed armor. `enable` only — it is a one-shot that loads
	# the persisted holding plane at START=21, and running it NOW would install a
	# block on a live box moments before the daemon replaces it anyway. It has to
	# be enabled here regardless of whether the stack is on: the file it loads only
	# exists while the daemon wants it to, so an enabled-but-unarmed service is a
	# no-op, and enabling it later would mean the first boot after an upgrade is
	# the one boot still exposed.
	[ -x /etc/init.d/shater-armor ] && /etc/init.d/shater-armor enable
	[ -x /etc/init.d/shater ] && /etc/init.d/shater restart
	[ -x /etc/init.d/shater-cron ] && /etc/init.d/shater-cron restart
	# Fold the seeded shater_l3 zone into the LIVE ruleset — matters on a live
	# opkg/apk install only, where firewall started long before our commit and
	# nothing else would re-read it until the next reboot. Gated on the fw4
	# table actually being loaded: at FIRST boot this job can run before the
	# S19 firewall start, and an early reload would install a ruleset built
	# from a half-initialized netifd AND make the later start a no-op (fw4
	# start skips when its table already exists). No table => the pending S19
	# start reads the committed config by itself, no reload needed.
	if nft list tables 2>/dev/null | grep -q "inet fw4"; then
		[ -x /etc/init.d/firewall ] && /etc/init.d/firewall reload
	fi
	exit 0
'
SHATER_TMO=""
command -v timeout >/dev/null 2>&1 && SHATER_TMO="timeout 300"
if command -v setsid >/dev/null 2>&1; then
	setsid $SHATER_TMO /bin/sh -c "$SHATER_BRINGUP" </dev/null >/dev/null 2>&1 &
else
	$SHATER_TMO /bin/sh -c "$SHATER_BRINGUP" </dev/null >/dev/null 2>&1 &
fi

exit 0
