Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
e38108a7c4 | ||
|
|
06c04c157d | ||
|
|
3654acf7fb | ||
|
|
3a9b3f523d | ||
|
|
201aa7c168 | ||
|
|
78bb6a1be8 | ||
|
|
ea3a4c518e | ||
|
|
0f69880150 | ||
|
|
164b703a7d | ||
|
|
de6fa8ebf4 | ||
|
|
b91fba1295 | ||
|
|
df078c3205 | ||
|
|
d0471b2418 | ||
|
|
3314927bef | ||
|
|
6801146240 | ||
|
|
2f8c692c39 | ||
|
|
c788425cad | ||
|
|
0144282f5e | ||
|
|
b642e5d8fe | ||
|
|
35e4900769 | ||
|
|
d3e33294c1 | ||
|
|
32aac89139 | ||
|
|
9e6dda22b3 | ||
|
|
fde4bed571 | ||
|
|
cb26936ebf | ||
|
|
8db29b6267 | ||
|
|
564033cd10 | ||
|
|
add90b5b2f | ||
|
|
a571bd0e1a | ||
|
|
1267d20fb8 | ||
|
|
35f697ed08 | ||
|
|
d0fb6befb1 | ||
|
|
81c96019b5 | ||
|
|
baed8ff8f2 | ||
|
|
f80fb4dd1b | ||
|
|
b71b793681 | ||
|
|
61c87ad1d9 | ||
|
|
4dee508e12 | ||
|
|
76da5134ef | ||
|
|
4c630c9a13 | ||
|
|
d8dbefcd07 | ||
|
|
974208fc05 | ||
|
|
2eb71e8244 | ||
|
|
668cccbf24 | ||
|
|
4ea4585402 | ||
|
|
dc6d102473 | ||
|
|
683afc0a47 | ||
|
|
2c3e20512e | ||
|
|
51b2f04672 | ||
|
|
f190c8251e |
@@ -35,11 +35,15 @@
|
||||
# invalidates every deployed router's trust.
|
||||
#
|
||||
# AUTO-RELEASE
|
||||
# push a tag `vX.Y.Z` -> versioned per-arch releases `apk-vX.Y.Z-<arch>`.
|
||||
# workflow_dispatch -> rolling per-arch `apk-latest-<arch>` (always-fresh
|
||||
# feed). Publish uses the Gitea API via curl (ci/gitea-release.sh) — no
|
||||
# external action needed. NOTE: the apk release tags deliberately do NOT start
|
||||
# with `v` so publishing them cannot re-trigger this workflow's `v*` filter.
|
||||
# The rolling per-arch `apk-latest-<arch>` is published on EVERY run — tag runs
|
||||
# included — and then read back over the API to assert it really serves the
|
||||
# version just built. A tag push `vX.Y.Z` publishes the pinnable per-arch
|
||||
# `apk-vX.Y.Z-<arch>` IN ADDITION. It is not an either/or: it used to be, and
|
||||
# the rolling pointer then froze at 0.2.0 while v0.2.9/v0.2.10 shipped (see the
|
||||
# long comment above the `release-apk` job). Publish uses the Gitea API via curl
|
||||
# (ci/gitea-release.sh) — no external action needed. NOTE: the apk release tags
|
||||
# deliberately do NOT start with `v` so publishing them cannot re-trigger this
|
||||
# workflow's `v*` filter.
|
||||
#
|
||||
# PACKAGE VERSIONING (bug B4)
|
||||
# PKG_VERSION/PKG_RELEASE are NOT hand-written in the Makefiles any more. They
|
||||
|
||||
@@ -6,61 +6,155 @@
|
||||
## Правила делегирования
|
||||
|
||||
1. ЛЮБАЯ реализация (код, тесты, конфиги, рефакторинг, отладка) выполняется
|
||||
субагентами через инструмент Agent с `model: "fable"`. Сам ты правишь файлы
|
||||
только в одном случае: тривиальная правка в 1–2 строки, где постановка
|
||||
задачи дороже самой правки.
|
||||
субагентами через инструмент Agent. Сам ты правишь файлы только в одном
|
||||
случае: тривиальная правка в 1–2 строки, где постановка задачи дороже самой
|
||||
правки.
|
||||
|
||||
2. Перед делегированием ты сам исследуешь код настолько, чтобы написать
|
||||
точное ТЗ. В каждом задании субагенту обязательно указывай:
|
||||
- контекст: что это за проект и над чем идёт работа;
|
||||
- конкретные файлы и функции, которые нужно менять (пути, а не «найди сам»);
|
||||
2. **Модель выбирает исполнитель задачи, а не привычка.** `fable` — быстрый и
|
||||
дешёвый, годится для механической работы с ясным контрактом. `opus` — для
|
||||
всего, где нужно рассуждение: поиск причины, аудит, дизайн, работа в чужом
|
||||
коде. Если у `fable` кончилась квота — молча переходи на `opus`, это не повод
|
||||
останавливать работу. Не спрашивай владельца, какую модель брать.
|
||||
|
||||
3. Перед делегированием ты сам исследуешь код настолько, чтобы написать точное
|
||||
ТЗ. В каждом задании субагенту обязательно указывай:
|
||||
- контекст: что за проект и над чем идёт работа;
|
||||
- конкретные файлы и функции (пути, а не «найди сам»);
|
||||
- контракт: сигнатуры, форматы данных, инварианты, что менять НЕЛЬЗЯ;
|
||||
- definition of done: как проверить, что задача выполнена
|
||||
(какие команды/тесты прогнать и какой ожидается результат);
|
||||
- что вернуть в финальном ответе: список изменённых файлов, результаты
|
||||
проверок, найденные проблемы и принятые решения.
|
||||
- definition of done: какие команды прогнать и какой ждать результат;
|
||||
- что вернуть: изменённые файлы, результаты проверок, найденные проблемы,
|
||||
принятые решения.
|
||||
|
||||
3. Скиллы: при постановке задачи посмотри список доступных скиллов и ЯВНО
|
||||
перечисли в ТЗ, какие скиллы субагент обязан вызвать через инструмент Skill
|
||||
до начала работы (например: «сначала вызови Skill "openwrt-procd-services"
|
||||
и следуй ему»). Субагент не видит наш диалог и сам не догадается — пиши
|
||||
названия скиллов прямо в текст задания.
|
||||
4. **Скиллы использовать по максимуму — и тебе, и агентам.** Это не
|
||||
формальность: в них лежит выстраданное знание по ровно тем предметным
|
||||
областям, в которых мы работаем, и игнорировать их — значит переоткрывать
|
||||
чужие грабли. См. раздел «Скиллы» ниже.
|
||||
|
||||
4. Независимые задачи запускай ПАРАЛЛЕЛЬНО — несколько вызовов Agent в одном
|
||||
сообщении, каждый с `model: "fable"`. Зависимые — последовательно, передавая
|
||||
в следующее ТЗ результаты предыдущего.
|
||||
5. Независимые задачи запускай ПАРАЛЛЕЛЬНО — несколько вызовов Agent в одном
|
||||
сообщении. Зависимые — последовательно, передавая результаты предыдущего.
|
||||
**Делишь файлы между параллельными агентами явно** и пишешь каждому, кто ещё
|
||||
работает в дереве и что трогать нельзя. Запрещай им `git stash`,
|
||||
`git checkout <файл>`, `git reset` — в этом проекте агент уже сносил правки
|
||||
соседа через `git stash push`.
|
||||
|
||||
5. Приёмка: результат каждого субагента ты проверяешь сам (читаешь diff
|
||||
ключевых мест, гоняешь проверки из definition of done). Если результат
|
||||
не принят — не переделывай сам, а верни задачу: доработку заказывай тому же
|
||||
агенту через SendMessage (у него сохранён контекст), а не новым спавном.
|
||||
6. Приёмка: результат каждого субагента ты проверяешь сам — читаешь diff
|
||||
ключевых мест, гоняешь проверки из definition of done. Не принимай отчёт на
|
||||
слово: сегодня отчёт «тесты зелёные» дважды сопровождался тестом, который
|
||||
ничего не прибивал. Если результат не принят — не переделывай сам, а верни
|
||||
задачу тому же агенту через SendMessage (у него сохранён контекст).
|
||||
|
||||
6. Финальный отчёт пользователю: что сделано, кем (сколько агентов),
|
||||
что проверено, что осталось.
|
||||
7. Финальный отчёт владельцу: что сделано, сколько агентов, что проверено,
|
||||
**что осталось непроверенным и почему** — последнее так же важно.
|
||||
|
||||
## Инженерные стандарты
|
||||
|
||||
Это не пожелания. Каждый пункт здесь появился после того, как его отсутствие
|
||||
стоило рабочего дня.
|
||||
|
||||
- **Тест обязан быть проверен мутацией.** Откатить фикс → показать, что тест
|
||||
падает, и с каким текстом → вернуть фикс. Тест, не падающий на сломанном коде,
|
||||
не тест, а украшение.
|
||||
|
||||
- **Прибор без контроля не доказывает ничего.** Отрицательный результат чего-то
|
||||
стоит, только если показано, что этот же прибор умеет дать положительный.
|
||||
«Утечки не нашли» прибором, который не мог её увидеть, — это не результат.
|
||||
|
||||
- **Опровержение ценнее согласия.** В каждом ТЗ прямо разрешай субагенту
|
||||
сказать «твоя версия неверна» и требуй доказательства, а не вежливости.
|
||||
Лучшие результаты этого проекта приходили именно так.
|
||||
|
||||
- **Не обещать непроверенного.** Комментарий, предупреждение и текст в панели —
|
||||
это утверждения о поведении. Если поведение не проверено, так и писать.
|
||||
Формально верная фраза, которая читается как «работает», — тоже ложь.
|
||||
|
||||
- **Умолчание падает в восстановимую сторону.** Открытый `default:` в разборе
|
||||
вариантов — источник целого класса дефектов: неучтённое значение уходит туда,
|
||||
где дороже всего ошибиться. Списки делать положительными и закрытыми.
|
||||
|
||||
- **Проверка присутствия обязана покрывать всё, что ставит её Apply-двойник.**
|
||||
Иначе идемпотентный быстрый путь становится ловушкой: «всё на месте» при
|
||||
отсутствующем маршруте.
|
||||
|
||||
- **Никакого молчаливого скипа.** Тест, который не выполнился, обязан быть
|
||||
назван поимённо в выводе гейта. Однажды CI гонял два теста из 116 файлов, и
|
||||
все считали, что покрыто.
|
||||
|
||||
## Скиллы
|
||||
|
||||
**Правило: если задача касается области, по которой есть скилл, — скилл
|
||||
вызывается ДО начала работы, а не после того, как что-то не заработало.**
|
||||
Это относится и к тебе, и к каждому субагенту.
|
||||
|
||||
Субагент не видит наш диалог и сам не догадается, что скиллы существуют.
|
||||
Поэтому **в каждом ТЗ перечисляй поимённо**, какие скиллы он обязан вызвать
|
||||
через инструмент Skill: «сначала вызови Skill "openwrt-nftables" и Skill
|
||||
"openwrt-networking", следуй им». Требуй в отчёте сказать, что именно из скилла
|
||||
он применил, — так видно, вызвал он его или упомянул.
|
||||
|
||||
Соответствие областей этого проекта и скиллов:
|
||||
|
||||
| Трогаешь | Обязательные скиллы |
|
||||
|---|---|
|
||||
| `/etc/config/*`, `uci`, uci-defaults, парсер модели | `openwrt-uci` |
|
||||
| nftables, fw4, зоны, метки, tproxy, kill-switch | `openwrt-nftables` |
|
||||
| интерфейсы, мосты, VLAN, policy routing, `ip rule`, sysctl, dnsmasq | `openwrt-networking` |
|
||||
| init-скрипты, procd, respawn, service triggers, boot armor | `openwrt-procd-services` |
|
||||
| перехват трафика целиком (tproxy + маршрутизация + DNS) | `openwrt-transparent-proxy` |
|
||||
| сборка пакетов, SDK, фид, CI, подпись, `apk`/`opkg` | `openwrt-package-build-ci`, `openwrt-native-packages` |
|
||||
| LuCI-приложение, ubus/rpcd, ucode | `openwrt-luci-plugin`, `openwrt-ubus-rpcd`, `openwrt-ucode` |
|
||||
| панель (React/TS) | `react-expert`, `frontend-design:frontend-design` |
|
||||
| Go: конкурентность, каналы, профилирование, идиоматика | `fullstack-dev-skills:golang-pro` |
|
||||
| TypeScript | `fullstack-dev-skills:typescript-pro` |
|
||||
| стратегия тестирования, покрытие, тестовые данные | `fullstack-dev-skills:test-master` |
|
||||
| поиск причины по логам и трассам | `fullstack-dev-skills:debugging-wizard` |
|
||||
| проверка в браузере, скриншоты | `fullstack-dev-skills:playwright-expert` |
|
||||
| ревью | `review`, `fullstack-dev-skills:code-reviewer` |
|
||||
| безопасность | `security-review`, `fullstack-dev-skills:security-reviewer` |
|
||||
| графики и визуализация данных | `dataviz` |
|
||||
|
||||
Список неполный — **смотри доступные скиллы под задачу**, а не только в эту
|
||||
таблицу. Если скилл выглядит смежным, дешевле вызвать его и не воспользоваться,
|
||||
чем не вызвать и потом отлаживать то, что там уже описано.
|
||||
|
||||
## Проверки
|
||||
|
||||
- **Гейт:** `bash scripts/run-tests.sh` — Linux в Docker, боевой набор тегов,
|
||||
`-race`, и шаг, требующий вердикта по имени для привилегированных тестов.
|
||||
Зелёный гейт — необходимое условие, но не достаточное: он не видит стыков с
|
||||
ядром, procd и nftables.
|
||||
- **Стенд:** сервер `local_openwrt` в ssh-manager — ImmortalWrt 25.12.1 той же
|
||||
ревизии, что боевой роутер. Сюда — всё, что касается init-скриптов, nft,
|
||||
policy routing, TUN.
|
||||
- **Боевой роутер:** `mini_router` (BPI-R3), через него идёт весь домашний
|
||||
трафик. Перед изменением конфигурации — резервная копия. Проверять приборно,
|
||||
а не по логу: лог может печатать одно и то же в честном и в ложном случае.
|
||||
|
||||
## Релиз и деплой
|
||||
|
||||
- Тег → CI (Gitea Actions) → apk-фид → установка на роутер.
|
||||
- **Обновлять только поимённо**, никогда не `apk upgrade` целиком:
|
||||
`apk upgrade shaterd shater-core luci-app-shater byedpi`.
|
||||
- **Не трогать кеш CI-раннера** — сборка растянется на часы.
|
||||
- Число тегов на порцию работы — на твоё усмотрение, если владелец не сказал
|
||||
иначе.
|
||||
|
||||
## Фронтенд (admin panel)
|
||||
|
||||
Дизайн-направление ЗАФИКСИРОВАНО: **Faceplate** (панель сетевого железа).
|
||||
Полная спека, токены, компоненты и ссылка на живой эталон — в
|
||||
[`docs-shater/DESIGN.md`](docs-shater/DESIGN.md). Эталон:
|
||||
https://claude.ai/code/artifact/9f7c07e8-d8ac-4ae1-b113-5b25d0ba5dd2
|
||||
Спека, токены и компоненты — в [`docs-shater/DESIGN.md`](docs-shater/DESIGN.md).
|
||||
Эталон: https://claude.ai/code/artifact/9f7c07e8-d8ac-4ae1-b113-5b25d0ba5dd2
|
||||
|
||||
- **Стек:** Vite + React + TypeScript, лёгкий (SPA встраивается в бинарь —
|
||||
без тяжёлых зависимостей). Расположение: папка `panel/` в корне.
|
||||
- **Порядок работ:**
|
||||
1. Сам (оркестратор) скаффолдишь `panel/`, переносишь токены из
|
||||
`docs-shater/DESIGN.md` в `panel/src/tokens.css` один-в-один и задаёшь каркас
|
||||
компонентов. Это фундамент — делай аккуратно сам или отдай ОДНОМУ агенту.
|
||||
2. Дизайн-систему в компоненты: `<Faceplate> <Module> <Toggle> <Led>
|
||||
<SegMeter> <QueryLog>` + кнопки — строго по эталону.
|
||||
3. Страницы раздаёшь ПАРАЛЛЕЛЬНО Opus-агентам (`model: "opus"`), по одной на
|
||||
агента: Overview, Nodes/Subscriptions, Routing rules, DNS/Blocklists,
|
||||
Devices, Apply/Rollback.
|
||||
- **В КАЖДОМ ТЗ агенту обязательно:** ссылка на `docs-shater/DESIGN.md` и на эталон;
|
||||
требование сначала вызвать Skill `react-expert` и Skill
|
||||
`frontend-design:frontend-design` и следовать им; список готовых компонентов,
|
||||
которые он ДОЛЖЕН переиспользовать (не изобретать заново); какие токены и
|
||||
семантические цвета применять; DoD — страница совпадает с языком эталона,
|
||||
адаптив + фокус + reduced-motion соблюдены.
|
||||
- **Не отходить от Faceplate.** Любой новый экран наследует ту же визуальную
|
||||
систему. Оранжевый — только акцент; семантика good/warn/crit — отдельно.
|
||||
- **Стек:** Vite + React + TypeScript в `panel/`. SPA встраивается в бинарь —
|
||||
тяжёлые зависимости недопустимы.
|
||||
- **Панель целиком на английском.** Ни одного символа кириллицы в `panel/src`.
|
||||
- **В КАЖДОМ ТЗ на панель:** ссылка на `DESIGN.md` и на эталон; требование
|
||||
сначала вызвать Skill `react-expert` и Skill
|
||||
`frontend-design:frontend-design`; список существующих компонентов, которые
|
||||
надо ПЕРЕИСПОЛЬЗОВАТЬ (`<Faceplate> <Module> <Toggle> <Led> <SegMeter>
|
||||
<QueryLog>` и кнопки), а не изобретать заново; какие токены и семантические
|
||||
цвета применять; DoD — совпадение с языком эталона, адаптив, фокус,
|
||||
`prefers-reduced-motion`.
|
||||
- Оранжевый — только акцент; семантика good/warn/crit — отдельно.
|
||||
- **Панель не должна врать про состояние.** Значение, которое движок примет,
|
||||
не может рисоваться как «never matches»; настройка, которой управляет другая
|
||||
подсистема, не может описываться так, будто управляет ею.
|
||||
|
||||
+30
-5
@@ -24,7 +24,8 @@ The engine is a **fork of [sing-box](https://github.com/SagerNet/sing-box) via
|
||||
[sing-box-lx](https://github.com/Leadaxe/sing-box-lx)**, compiled into a single Go
|
||||
binary `shaterd` together with the control plane, DNS filter, stats aggregator and
|
||||
the web panel itself. Broad protocol set: VLESS/VMess/Trojan/Shadowsocks,
|
||||
Reality/XTLS, WireGuard, **AmneziaWG 2.0**, Hysteria2, TUIC, XHTTP, MASQUE/CONNECT-IP.
|
||||
Reality/XTLS, WireGuard, **AmneziaWG 2.0**, Hysteria2, TUIC, XHTTP — exactly what
|
||||
`shater/parse` can read and `shater/registry` registers in the engine.
|
||||
|
||||
A thin **LuCI launcher** (mini-dashboard + "Open panel" button) hands the browser a
|
||||
single-use token into the standalone SPA the daemon serves on its own port
|
||||
@@ -39,8 +40,8 @@ single-use token into the standalone SPA the daemon serves on its own port
|
||||
selector / chain / direct / block; node groups with balancer/observatory;
|
||||
multi-hop chains; per-rule egress.
|
||||
- **Fail-closed kill-switch** (dead group → block, never a silent direct leak); own
|
||||
`inet shater` nft table; atomic apply with `nft -c` validation and commit-confirm
|
||||
auto-rollback.
|
||||
`inet shater` nft table; atomic apply with `nft -c` validation. Commit-confirm
|
||||
auto-rollback exists but **ships OFF** (`confirm_timeout=0`) — arm it yourself.
|
||||
- **DNS filtering & blocklists** with flexible sources (inline / file / url /
|
||||
geosite), compiled `.srs` matcher; Block-DoH/DoT to stop filter bypass.
|
||||
- Subscriptions (Clash / sing-box / Xray-JSON) and manual nodes; node health board.
|
||||
@@ -68,8 +69,23 @@ keeps the router current. Point the repo line at `apk-vX.Y.Z-<arch>` instead to
|
||||
pin a build; that file then has to be edited by hand for every upgrade.
|
||||
|
||||
shater ships **inert** (globals off) so install never breaks connectivity. After
|
||||
configuring nodes/rules: `uci set shater.globals.enabled=1 && uci commit shater`,
|
||||
then `shaterd apply` and `shaterd confirm`.
|
||||
configuring nodes/rules:
|
||||
|
||||
```sh
|
||||
uci set shater.globals.enabled=1
|
||||
uci set shater.globals.confirm_timeout=120 # commit-confirm ships OFF — arm it
|
||||
uci commit shater
|
||||
shaterd apply && shaterd confirm
|
||||
```
|
||||
|
||||
Without that middle line `shaterd apply` arms no auto-rollback (and says so), so an
|
||||
apply that costs you SSH/LuCI access has to be undone by hand.
|
||||
|
||||
Once an enabled, fail-closed config has been applied, `/etc/init.d/shater-armor`
|
||||
loads a saved fail-closed plane at **boot**, before the daemon exists: LAN→WAN
|
||||
forwarding is blocked until `shaterd` applies, while SSH/LuCI/the panel stay
|
||||
reachable on purpose (the chain hooks `forward` only). What arms it, what refuses
|
||||
to arm, and how to switch it off — `INSTALL.md` §4.
|
||||
|
||||
## Build from source
|
||||
|
||||
@@ -78,6 +94,15 @@ then `shaterd apply` and `shaterd confirm`.
|
||||
into `openwrt/shaterd/files/`. Details in
|
||||
[`docs-shater/INSTALL.md`](docs-shater/INSTALL.md).
|
||||
|
||||
`bash scripts/run-tests.sh` is the test gate: the whole suite under the **shipped**
|
||||
build tags (`scripts/router-tags.sh`), on linux (it re-execs in Docker from a
|
||||
non-linux host), with `-race`, plus three machine checks against a silent skip —
|
||||
the tag set may only add test files, every package with tests must report `ok` by
|
||||
name, and every `TestIntegration*` must produce a verdict by name.
|
||||
`scripts/check-router-tags.sh` separately proves no feature declared in
|
||||
`FEATURES.md` lost a build tag it needs. A green gate is necessary but not
|
||||
sufficient: it does not see the kernel, procd or nftables seams.
|
||||
|
||||
## Repository layout
|
||||
|
||||
| Path | What |
|
||||
|
||||
@@ -27,7 +27,8 @@ BananaWRT** (Banana Pi BPI-R3, BPI-R4 и совместимые). Он проз
|
||||
Go-бинарь `shaterd` вместе с control-plane, DNS-фильтром, агрегатором статистики и
|
||||
самой веб-панелью. За счёт sing-box поддерживается широкий и актуальный набор
|
||||
протоколов: VLESS/VMess/Trojan/Shadowsocks, Reality/XTLS, WireGuard,
|
||||
**AmneziaWG 2.0**, Hysteria2, TUIC, XHTTP, MASQUE/CONNECT-IP.
|
||||
**AmneziaWG 2.0**, Hysteria2, TUIC, XHTTP — ровно то, что умеет разобрать
|
||||
`shater/parse` и что регистрирует `shater/registry` в движке.
|
||||
|
||||
Интеграция в OpenWrt — тонкий **LuCI-лаунчер**: мини-дашборд и кнопка «Открыть
|
||||
панель», которая по одноразовому токену передаёт браузер в полноценную SPA-панель,
|
||||
@@ -50,8 +51,10 @@ Go-бинарь `shaterd` вместе с control-plane, DNS-фильтром,
|
||||
- **Fail-closed kill-switch**: мёртвая группа → block, а не тихая утечка мимо
|
||||
прокси; собственная nft-таблица `inet shater` и свои марки/таблицы, fw4 не
|
||||
трогаем.
|
||||
- Атомарный apply с валидацией движком и `nft -c`, **commit-confirm** с
|
||||
авто-откатом к последней рабочей конфигурации.
|
||||
- Атомарный apply с валидацией движком и `nft -c`. **Commit-confirm** с
|
||||
авто-откатом к последней рабочей конфигурации есть, но **на стоковой установке
|
||||
выключен**: `confirm_timeout` поставляется нулём, и apply не вооружает ничего,
|
||||
пока вы не зададите окно (см. «Включение»).
|
||||
- Идемпотентный reconcile из hotplug/boot под flock; management-bypass
|
||||
(SSH/LuCI/LAN) всегда в обход.
|
||||
|
||||
@@ -121,8 +124,11 @@ flowchart TB
|
||||
|
||||
Путь трафика: LAN-клиент → `nft tproxy` (mark → tproxy-порт) → tproxy-inbound
|
||||
sing-box (сниффинг SNI/Host/QUIC) → маршрут по правилу → outbound/selector/chain
|
||||
(проксировано) · direct (flow-offload) · block. Подробные диаграммы (auth-handoff,
|
||||
data-plane, DNS-flow, apply-flow) — в [`docs-shater/ARCHITECTURE.md`](docs-shater/ARCHITECTURE.md).
|
||||
(проксировано) · direct (обычный маршрут, без туннеля) · block. TPROXY несёт
|
||||
только TCP и UDP; ICMP и остальные протоколы — через отдельные опциональные
|
||||
механизмы (`l3_tunnel`, `untunnelable_egress`, ARCHITECTURE §3a). Подробные
|
||||
диаграммы (auth-handoff, data-plane, DNS-flow, apply-flow) — в
|
||||
[`docs-shater/ARCHITECTURE.md`](docs-shater/ARCHITECTURE.md).
|
||||
|
||||
---
|
||||
|
||||
@@ -195,15 +201,32 @@ shater ставится **инертным** (globals выключены), чт
|
||||
|
||||
```sh
|
||||
uci set shater.globals.enabled=1
|
||||
# Предохранитель: commit-confirm поставляется ВЫКЛЮЧЕННЫМ (confirm_timeout=0),
|
||||
# и без этой строки apply ничем не подстрахован. 120 с — окно на проверку связи.
|
||||
uci set shater.globals.confirm_timeout=120
|
||||
uci commit shater
|
||||
shaterd apply # apply + вооружить commit-confirm на живом демоне
|
||||
shaterd confirm # подтвердить (отменяет авто-откат)
|
||||
shaterd apply # применить и вооружить авто-откат на 120 с
|
||||
shaterd confirm # подтвердить в пределах окна (отменяет авто-откат)
|
||||
```
|
||||
|
||||
`shaterd apply` печатает, вооружил ли он что-нибудь, и почему нет: при
|
||||
`confirm_timeout=0` он прямо говорит, что автоматического отката НЕТ. Оставить
|
||||
ноль — сознательный выбор: тогда apply, отрезавший вам SSH/LuCI, придётся
|
||||
откатывать руками.
|
||||
|
||||
`/etc/init.d/shater enable && /etc/init.d/shater start` поднимает демона под procd.
|
||||
Кнопка «Открыть панель» в LuCI чеканит одноразовый токен и передаёт браузер в
|
||||
панель (`:8088` по умолчанию).
|
||||
|
||||
После первого же применённого включённого fail-closed конфига появляется
|
||||
**загрузочная защита**: `/etc/init.d/shater-armor` (START=21) грузит сохранённый
|
||||
fail-closed план ещё до старта демона, закрывая те секунды между поднятием LAN и
|
||||
первым apply, когда роутер форвардил трафик в WAN открытым. Форвардинг LAN→WAN
|
||||
заблокирован, пока `shaterd` не применит конфиг; SSH, LuCI и панель при этом
|
||||
доступны **намеренно** — цепочка вешается только на `forward`. Чем защита
|
||||
вооружается, когда отказывается вооружаться и как её снять —
|
||||
[`docs-shater/INSTALL.md`](docs-shater/INSTALL.md) §4.
|
||||
|
||||
---
|
||||
|
||||
## Сборка из исходников
|
||||
@@ -226,6 +249,28 @@ arm64}` с musl-static набором тегов (`CGO_ENABLED=0 GOOS=linux`), s
|
||||
(набор build-тегов, почему `shaterd` — prebuilt-пакет, порядок CI) — в
|
||||
[`docs-shater/INSTALL.md`](docs-shater/INSTALL.md).
|
||||
|
||||
### Проверка
|
||||
|
||||
```sh
|
||||
bash scripts/run-tests.sh # полный гейт
|
||||
bash scripts/run-tests.sh --no-race # без -race, для локального цикла
|
||||
```
|
||||
|
||||
Гейт гоняет весь набор **под теми же build-тегами, с которыми собирается
|
||||
роутерный бинарь** (`scripts/router-tags.sh`), на Linux (с не-Linux хоста — сам
|
||||
перезапускается в Docker), с `-race`, и содержит три машинные проверки против
|
||||
молчаливого скипа: набор тегов может только ДОБАВЛЯТЬ тест-файлы; каждый пакет с
|
||||
тестами обязан отчитаться `ok` поимённо; каждый `TestIntegration*` обязан выдать
|
||||
вердикт по имени. Причина такая: до 2026-07 релизный тракт не гонял почти ничего
|
||||
— 115 тест-файлов из 116 под `shater/**` в CI не исполнялись ни разу.
|
||||
|
||||
Отдельно `scripts/check-router-tags.sh` проверяет, что ни одна заявленная в
|
||||
`FEATURES.md` фича не потеряла нужный ей build-тег.
|
||||
|
||||
Зелёный гейт — необходимое, но не достаточное условие: он не видит стыков с
|
||||
ядром, procd и nftables. Это проверяется на стенде (см.
|
||||
[`docs-shater/CONTEXT.md`](docs-shater/CONTEXT.md)).
|
||||
|
||||
---
|
||||
|
||||
## Структура репозитория
|
||||
@@ -274,7 +319,10 @@ CI на **Gitea Actions** (`.gitea/workflows/release.yml`) собирает вс
|
||||
shater вкомпилирует **форк движка sing-box-lx** — тонкий downstream апстрима
|
||||
[SagerNet/sing-box](https://github.com/SagerNet/sing-box), добавляющий набор
|
||||
клиентских фич (XHTTP, AmneziaWG 2.0, MASQUE, расширения наблюдаемости) за
|
||||
build-тегами и живущий **ребейзом на каждый upstream-тег, а не merge**. Форк
|
||||
build-тегами и живущий **ребейзом на каждый upstream-тег, а не merge**. Это набор
|
||||
самого форка, а не shater: MASQUE/CONNECT-IP мы намеренно **не регистрируем** —
|
||||
`shater/generate` его не порождает, а отказ от него и остального незадействованного
|
||||
зоопарка экономит ~6 МБ бинаря и столько же RAM на роутере (`shater/registry`). Форк
|
||||
разрабатывается по Spec Kit; неизменяемые принципы — в
|
||||
[`SPECS/CONSTITUTION.md`](SPECS/CONSTITUTION.md), справочник фич движка — в
|
||||
[`docs-lx/lx-config.ru.md`](docs-lx/lx-config.ru.md).
|
||||
|
||||
@@ -0,0 +1,145 @@
|
||||
// lx:begin l3-honest-drop
|
||||
package adapter
|
||||
|
||||
import (
|
||||
"net/netip"
|
||||
"testing"
|
||||
|
||||
"github.com/sagernet/sing-tun"
|
||||
"github.com/sagernet/sing-tun/gtcpip/header"
|
||||
|
||||
"github.com/stretchr/testify/require"
|
||||
)
|
||||
|
||||
// judgeFlowRouter answers PreMatch with a canned verdict; JudgeFlow reads
|
||||
// nothing else off the Router.
|
||||
type judgeFlowRouter struct {
|
||||
Router
|
||||
result PreMatchResult
|
||||
}
|
||||
|
||||
func (r *judgeFlowRouter) PreMatch(InboundContext, []byte) PreMatchResult { return r.result }
|
||||
|
||||
// judgeFlowPort is the tun.Port half of a FlowOutbound. inet4 is what
|
||||
// PortAddresses reports for IPv4 — the one field the two ICMP consumers in
|
||||
// sing-tun disagree about (see the comment on
|
||||
// TestJudgeFlowICMPToBoundPortStaysAFlow).
|
||||
type judgeFlowPort struct {
|
||||
Outbound
|
||||
inet4 netip.Addr
|
||||
}
|
||||
|
||||
func (o *judgeFlowPort) Tag() string { return "wg-out" }
|
||||
func (o *judgeFlowPort) Type() string { return "wireguard" }
|
||||
func (o *judgeFlowPort) PortAddresses() (netip.Addr, netip.Addr) {
|
||||
return o.inet4, netip.Addr{}
|
||||
}
|
||||
func (o *judgeFlowPort) PortMTU() uint32 { return 1420 }
|
||||
func (o *judgeFlowPort) AttachReturn(tun.Return) error { return nil }
|
||||
func (o *judgeFlowPort) DetachReturn(tun.Return) error { return nil }
|
||||
func (o *judgeFlowPort) WritePackets(packets [][]byte) error { return nil }
|
||||
|
||||
// judgeFlowNonPort is a FlowOutbound-shaped result that is NOT a tun.Port — the
|
||||
// interface drift the second line of defense in JudgeFlow exists for.
|
||||
type judgeFlowNonPort struct {
|
||||
Outbound
|
||||
}
|
||||
|
||||
func (o *judgeFlowNonPort) Tag() string { return "drifted" }
|
||||
func (o *judgeFlowNonPort) Type() string { return "drifted" }
|
||||
|
||||
func judgeFlow(t *testing.T, protocol uint8, result PreMatchResult) tun.FlowVerdict {
|
||||
t.Helper()
|
||||
return JudgeFlow(
|
||||
&judgeFlowRouter{result: result},
|
||||
"l3-in", "tun", protocol,
|
||||
netip.MustParseAddrPort("192.168.1.2:1234"),
|
||||
netip.MustParseAddrPort("1.1.1.1:1234"),
|
||||
nil,
|
||||
)
|
||||
}
|
||||
|
||||
const (
|
||||
judgeFlowICMP = uint8(header.ICMPv4ProtocolNumber)
|
||||
judgeFlowTCP = uint8(header.TCPProtocolNumber)
|
||||
)
|
||||
|
||||
// TestJudgeFlowICMPToBoundPortStaysAFlow is the guard on the ONE fix that must
|
||||
// not be made here.
|
||||
//
|
||||
// sing-tun has two ICMP consumers with different requirements on the port:
|
||||
//
|
||||
// - ForwardDispatcher.createFlow (flow_dispatch.go) needs only a VALID port
|
||||
// address — it NATs the echo identifier and rewrites the source to that
|
||||
// address. This is the path every unfragmented LAN ping takes, and it is
|
||||
// what makes ping-through-WireGuard/AWG work at all.
|
||||
// - ICMPForwarder.installFlow (stack_gvisor_icmp.go) additionally requires the
|
||||
// address to be UNSPECIFIED, because it writes the packet to the port
|
||||
// unmodified. A WireGuard endpoint reports its concrete interface address
|
||||
// (transport/wireguard/port.go), so installFlow declines and HandlePacket
|
||||
// falls through to forging the echo reply.
|
||||
//
|
||||
// The tempting fix — "for ICMP, refuse ActionFlow when PortAddresses() is not
|
||||
// unspecified, so the verdict becomes a drop and the forgery is unreachable" —
|
||||
// is applied HERE, in the one function both consumers share, with byte-identical
|
||||
// arguments from either. It would therefore kill the working path too: every
|
||||
// ping through WireGuard/AWG, fragmented or not, would drop, and l3_tunnel would
|
||||
// carry nothing but `direct`. Keep this test failing loudly if anyone tries.
|
||||
func TestJudgeFlowICMPToBoundPortStaysAFlow(t *testing.T) {
|
||||
t.Parallel()
|
||||
port := &judgeFlowPort{inet4: netip.MustParseAddr("10.2.0.2")}
|
||||
verdict := judgeFlow(t, judgeFlowICMP, PreMatchResult{Action: PreMatchFlow, Outbound: port})
|
||||
require.Equal(t, tun.ActionFlow, verdict.Action,
|
||||
"ICMP to a WireGuard/AWG endpoint must stay a flow: the forward dispatcher NATs it by echo identifier and this is the whole point of l3_tunnel")
|
||||
require.Same(t, tun.Port(port), verdict.Port)
|
||||
}
|
||||
|
||||
// The `direct` shape: an unspecified port address. Both consumers accept it.
|
||||
func TestJudgeFlowICMPToUnspecifiedPortStaysAFlow(t *testing.T) {
|
||||
t.Parallel()
|
||||
port := &judgeFlowPort{inet4: netip.IPv4Unspecified()}
|
||||
verdict := judgeFlow(t, judgeFlowICMP, PreMatchResult{Action: PreMatchFlow, Outbound: port})
|
||||
require.Equal(t, tun.ActionFlow, verdict.Action)
|
||||
require.Same(t, tun.Port(port), verdict.Port)
|
||||
}
|
||||
|
||||
// PreMatchDrop is the honest verdict and must arrive as ActionDrop: it is the
|
||||
// only value (besides Reject) that stops ICMPForwarder.HandlePacket before the
|
||||
// Echo -> EchoReply rewrite.
|
||||
func TestJudgeFlowICMPDropReachesTheStackAsDrop(t *testing.T) {
|
||||
t.Parallel()
|
||||
verdict := judgeFlow(t, judgeFlowICMP, PreMatchResult{Action: PreMatchDrop})
|
||||
require.Equal(t, tun.ActionDrop, verdict.Action)
|
||||
}
|
||||
|
||||
// The second line of defense: a PreMatchFlow whose outbound is not a tun.Port
|
||||
// must not degrade ICMP to ActionAccept, because Accept is the forged reply.
|
||||
func TestJudgeFlowICMPNonPortOutboundDrops(t *testing.T) {
|
||||
t.Parallel()
|
||||
verdict := judgeFlow(t, judgeFlowICMP, PreMatchResult{Action: PreMatchFlow, Outbound: &judgeFlowNonPort{}})
|
||||
require.Equal(t, tun.ActionDrop, verdict.Action,
|
||||
"FlowOutbound and tun.Port are distinct interfaces; a drift between them must not silently re-enable the echo forger")
|
||||
}
|
||||
|
||||
func TestJudgeFlowTCPNonPortOutboundAccepts(t *testing.T) {
|
||||
t.Parallel()
|
||||
verdict := judgeFlow(t, judgeFlowTCP, PreMatchResult{Action: PreMatchFlow, Outbound: &judgeFlowNonPort{}})
|
||||
require.Equal(t, tun.ActionAccept, verdict.Action,
|
||||
"for TCP, falling back to Accept is upstream behaviour and must stay untouched")
|
||||
}
|
||||
|
||||
// TCP keeps every mapping it had, including the Continue -> Accept default that
|
||||
// is a forgery only for ICMP.
|
||||
func TestJudgeFlowTCPContinueStaysAccept(t *testing.T) {
|
||||
t.Parallel()
|
||||
verdict := judgeFlow(t, judgeFlowTCP, PreMatchResult{Action: PreMatchContinue})
|
||||
require.Equal(t, tun.ActionAccept, verdict.Action)
|
||||
}
|
||||
|
||||
func TestJudgeFlowTCPBypassStaysBypass(t *testing.T) {
|
||||
t.Parallel()
|
||||
verdict := judgeFlow(t, judgeFlowTCP, PreMatchResult{Action: PreMatchBypass})
|
||||
require.Equal(t, tun.ActionBypass, verdict.Action)
|
||||
}
|
||||
|
||||
// lx:end l3-honest-drop
|
||||
@@ -75,7 +75,18 @@ func JudgeFlow(router Router, inbound string, inboundType string, network uint8,
|
||||
case PreMatchFlow:
|
||||
port, isPort := result.Outbound.(tun.Port)
|
||||
if !isPort {
|
||||
// lx:begin l3-honest-drop
|
||||
// Second line of defense behind route.(*Router).preMatchFlow: a
|
||||
// PreMatchFlow result already implies the outbound is an
|
||||
// adapter.FlowOutbound, but FlowOutbound and tun.Port are distinct
|
||||
// interfaces, and a drift between them must not degrade ICMP to
|
||||
// ActionAccept — the TUN stack would then forge the echo reply
|
||||
// itself instead of admitting the tunnel cannot carry the packet.
|
||||
if networkName == N.NetworkICMP {
|
||||
return tun.FlowVerdict{Action: tun.ActionDrop}
|
||||
}
|
||||
return tun.FlowVerdict{Action: tun.ActionAccept}
|
||||
// lx:end l3-honest-drop
|
||||
}
|
||||
verdict := tun.FlowVerdict{Action: tun.ActionFlow, Port: port, UDPTimeout: result.UDPTimeout, NewTracker: result.NewTracker}
|
||||
if result.Destination.IsValid() {
|
||||
|
||||
@@ -0,0 +1,223 @@
|
||||
//go:build with_quic
|
||||
|
||||
package httpclient
|
||||
|
||||
import (
|
||||
"context"
|
||||
stdTLS "crypto/tls"
|
||||
"io"
|
||||
"net"
|
||||
"net/http"
|
||||
"net/http/httptest"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"github.com/sagernet/quic-go"
|
||||
"github.com/sagernet/quic-go/http3"
|
||||
sbTLS "github.com/sagernet/sing-box/common/tls"
|
||||
"github.com/sagernet/sing-box/option"
|
||||
"github.com/sagernet/sing/common/logger"
|
||||
M "github.com/sagernet/sing/common/metadata"
|
||||
N "github.com/sagernet/sing/common/network"
|
||||
)
|
||||
|
||||
// raceProbePayload is large enough that it cannot ride along in the response
|
||||
// headers: the caller has to read the body off the QUIC stream AFTER
|
||||
// roundTripHTTP3Race has returned. That is the whole point of the test.
|
||||
const raceProbePayload = 64 * 1024
|
||||
|
||||
var _ N.Dialer = (*plainDialer)(nil)
|
||||
|
||||
type plainDialer struct{}
|
||||
|
||||
func (d *plainDialer) DialContext(ctx context.Context, network string, destination M.Socksaddr) (net.Conn, error) {
|
||||
return (&net.Dialer{}).DialContext(ctx, network, destination.String())
|
||||
}
|
||||
|
||||
func (d *plainDialer) ListenPacket(ctx context.Context, destination M.Socksaddr) (net.PacketConn, error) {
|
||||
return net.ListenUDP("udp", nil)
|
||||
}
|
||||
|
||||
// splitDialer sends the HTTP/3 racer and the HTTP/2 racer to two different
|
||||
// listeners, so a test can decide which one of them wins without having to bind
|
||||
// a TCP and a UDP socket on the same port number.
|
||||
type splitDialer struct {
|
||||
udp M.Socksaddr
|
||||
tcp M.Socksaddr
|
||||
}
|
||||
|
||||
func (d *splitDialer) DialContext(ctx context.Context, network string, _ M.Socksaddr) (net.Conn, error) {
|
||||
destination := d.tcp
|
||||
if network == N.NetworkUDP {
|
||||
destination = d.udp
|
||||
}
|
||||
return (&net.Dialer{}).DialContext(ctx, network, destination.String())
|
||||
}
|
||||
|
||||
func (d *splitDialer) ListenPacket(ctx context.Context, destination M.Socksaddr) (net.PacketConn, error) {
|
||||
return net.ListenUDP("udp", nil)
|
||||
}
|
||||
|
||||
func startH3Server(t *testing.T, handler http.Handler) M.Socksaddr {
|
||||
t.Helper()
|
||||
certificate, err := sbTLS.GenerateKeyPair(nil, nil, nil, "localhost")
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
listener, err := quic.ListenAddrEarly("127.0.0.1:0", &stdTLS.Config{
|
||||
Certificates: []stdTLS.Certificate{*certificate},
|
||||
NextProtos: []string{http3.NextProtoH3},
|
||||
MinVersion: stdTLS.VersionTLS13,
|
||||
}, nil)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
server := &http3.Server{Handler: handler}
|
||||
go server.ServeListener(listener)
|
||||
t.Cleanup(func() {
|
||||
server.Close()
|
||||
listener.Close()
|
||||
})
|
||||
return M.ParseSocksaddr(listener.Addr().String())
|
||||
}
|
||||
|
||||
func newRaceProbeTransport(t *testing.T, serverAddr M.Socksaddr) (*http3FallbackTransport, string) {
|
||||
return newRaceProbeTransportWithDialer(t, &plainDialer{}, serverAddr)
|
||||
}
|
||||
|
||||
func newRaceProbeTransportWithDialer(t *testing.T, dialer N.Dialer, serverAddr M.Socksaddr) (*http3FallbackTransport, string) {
|
||||
t.Helper()
|
||||
baseTLSConfig, err := sbTLS.NewClient(context.Background(), logger.NOP(), "localhost", option.OutboundTLSOptions{
|
||||
Enabled: true,
|
||||
Insecure: true,
|
||||
ServerName: "localhost",
|
||||
})
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
h2Fallback, err := newHTTP2FallbackTransport(dialer, baseTLSConfig, option.HTTP2Options{})
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
inner, err := newHTTP3FallbackTransport(dialer, baseTLSConfig, h2Fallback, option.QUICOptions{}, 300*time.Millisecond)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
t.Cleanup(func() { inner.Close() })
|
||||
return inner.(*http3FallbackTransport), "https://" + serverAddr.String() + "/probe"
|
||||
}
|
||||
|
||||
// TestHTTP3RaceWinnerBodyStaysReadable pins that the response handed back by the
|
||||
// HTTP/3 race is a LIVE response: its body must still be readable after
|
||||
// roundTripHTTP3Race returns. Cancelling the context the winner was issued on
|
||||
// resets its QUIC stream, so a "successful" round trip would hand the caller a
|
||||
// response it can never read.
|
||||
func TestHTTP3RaceWinnerBodyStaysReadable(t *testing.T) {
|
||||
payload := make([]byte, raceProbePayload)
|
||||
for i := range payload {
|
||||
payload[i] = byte(i)
|
||||
}
|
||||
serverAddr := startH3Server(t, http.HandlerFunc(func(writer http.ResponseWriter, request *http.Request) {
|
||||
writer.Header().Set("Content-Type", "application/octet-stream")
|
||||
writer.Write(payload)
|
||||
}))
|
||||
transport, url := newRaceProbeTransport(t, serverAddr)
|
||||
|
||||
ctx, cancel := context.WithTimeout(context.Background(), 20*time.Second)
|
||||
defer cancel()
|
||||
request, err := http.NewRequestWithContext(ctx, http.MethodGet, url, nil)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
// No cached HTTP/3 connection yet and a bodyless GET is replayable, so this
|
||||
// takes the racing path.
|
||||
response, err := transport.RoundTrip(request)
|
||||
if err != nil {
|
||||
t.Fatal("round trip: ", err)
|
||||
}
|
||||
defer response.Body.Close()
|
||||
if response.ProtoMajor != 3 {
|
||||
t.Fatalf("expected the HTTP/3 racer to win, got HTTP/%d.%d", response.ProtoMajor, response.ProtoMinor)
|
||||
}
|
||||
body, err := io.ReadAll(response.Body)
|
||||
if err != nil {
|
||||
t.Fatalf("the race winner's body died with the race: %v (read %d of %d bytes)", err, len(body), len(payload))
|
||||
}
|
||||
if len(body) != len(payload) {
|
||||
t.Fatalf("short body: got %d bytes, want %d", len(body), len(payload))
|
||||
}
|
||||
}
|
||||
|
||||
// TestHTTP3RaceFallbackWinnerBodyStaysReadableAndH3LoserIsCancelled covers the
|
||||
// other half of the race: the HTTP/2 fallback wins, so its body must survive the
|
||||
// race, and the HTTP/3 racer that lost must be torn down instead of being left
|
||||
// to run to completion on the caller's behalf.
|
||||
func TestHTTP3RaceFallbackWinnerBodyStaysReadableAndH3LoserIsCancelled(t *testing.T) {
|
||||
payload := make([]byte, raceProbePayload)
|
||||
for i := range payload {
|
||||
payload[i] = byte(i)
|
||||
}
|
||||
|
||||
h3Started := make(chan struct{}, 1)
|
||||
h3Cancelled := make(chan struct{}, 1)
|
||||
// The HTTP/3 handler never answers, so the fallback wins on the timer.
|
||||
h3Addr := startH3Server(t, http.HandlerFunc(func(_ http.ResponseWriter, request *http.Request) {
|
||||
select {
|
||||
case h3Started <- struct{}{}:
|
||||
default:
|
||||
}
|
||||
<-request.Context().Done()
|
||||
select {
|
||||
case h3Cancelled <- struct{}{}:
|
||||
default:
|
||||
}
|
||||
}))
|
||||
|
||||
h2Server := httptest.NewUnstartedServer(http.HandlerFunc(func(writer http.ResponseWriter, _ *http.Request) {
|
||||
writer.Header().Set("Content-Type", "application/octet-stream")
|
||||
writer.Write(payload)
|
||||
}))
|
||||
h2Server.EnableHTTP2 = true
|
||||
h2Server.StartTLS()
|
||||
t.Cleanup(h2Server.Close)
|
||||
|
||||
transport, _ := newRaceProbeTransportWithDialer(t, &splitDialer{
|
||||
udp: h3Addr,
|
||||
tcp: M.ParseSocksaddr(h2Server.Listener.Addr().String()),
|
||||
}, h3Addr)
|
||||
|
||||
ctx, cancel := context.WithTimeout(context.Background(), 20*time.Second)
|
||||
defer cancel()
|
||||
request, err := http.NewRequestWithContext(ctx, http.MethodGet, "https://localhost:443/probe", nil)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
response, err := transport.RoundTrip(request)
|
||||
if err != nil {
|
||||
t.Fatal("round trip: ", err)
|
||||
}
|
||||
if response.ProtoMajor != 2 {
|
||||
t.Fatalf("expected the HTTP/2 fallback to win, got HTTP/%d.%d", response.ProtoMajor, response.ProtoMinor)
|
||||
}
|
||||
body, err := io.ReadAll(response.Body)
|
||||
if err != nil {
|
||||
t.Fatalf("the fallback winner's body died with the race: %v (read %d of %d bytes)", err, len(body), len(payload))
|
||||
}
|
||||
response.Body.Close()
|
||||
if len(body) != len(payload) {
|
||||
t.Fatalf("short body: got %d bytes, want %d", len(body), len(payload))
|
||||
}
|
||||
|
||||
select {
|
||||
case <-h3Started:
|
||||
case <-time.After(5 * time.Second):
|
||||
t.Fatal("the HTTP/3 racer never reached the server, the test proves nothing about cancelling it")
|
||||
}
|
||||
select {
|
||||
case <-h3Cancelled:
|
||||
case <-time.After(5 * time.Second):
|
||||
t.Fatal("the losing HTTP/3 request was left running after the fallback won")
|
||||
}
|
||||
}
|
||||
@@ -6,6 +6,7 @@ import (
|
||||
"context"
|
||||
stdTLS "crypto/tls"
|
||||
"errors"
|
||||
"io"
|
||||
"net/http"
|
||||
"sync"
|
||||
"time"
|
||||
@@ -168,32 +169,65 @@ func (t *http3FallbackTransport) roundTripHTTP3(request *http.Request) (*http.Re
|
||||
return t.roundTripHTTP3Race(request, authority)
|
||||
}
|
||||
|
||||
// cancelOnBodyClose releases a racer's context when the caller is done with the
|
||||
// response it won. The race cannot release it on the way out: the body is read
|
||||
// after RoundTrip returns, and the context the request was issued on is what
|
||||
// keeps its stream alive.
|
||||
type cancelOnBodyClose struct {
|
||||
io.ReadCloser
|
||||
cancel context.CancelFunc
|
||||
cancelOnce sync.Once
|
||||
}
|
||||
|
||||
func (b *cancelOnBodyClose) Close() error {
|
||||
err := b.ReadCloser.Close()
|
||||
b.cancelOnce.Do(b.cancel)
|
||||
return err
|
||||
}
|
||||
|
||||
func withCancelOnBodyClose(response *http.Response, cancel context.CancelFunc) *http.Response {
|
||||
if response == nil || response.Body == nil {
|
||||
cancel()
|
||||
return response
|
||||
}
|
||||
response.Body = &cancelOnBodyClose{ReadCloser: response.Body, cancel: cancel}
|
||||
return response
|
||||
}
|
||||
|
||||
func (t *http3FallbackTransport) roundTripHTTP3Race(request *http.Request, authority string) (*http.Response, error) {
|
||||
ctx, cancel := context.WithCancel(request.Context())
|
||||
defer cancel()
|
||||
type result struct {
|
||||
response *http.Response
|
||||
err error
|
||||
h3 bool
|
||||
}
|
||||
results := make(chan result, 2)
|
||||
startRoundTrip := func(request *http.Request, useH3 bool) {
|
||||
request = request.WithContext(ctx)
|
||||
var (
|
||||
response *http.Response
|
||||
err error
|
||||
)
|
||||
if useH3 {
|
||||
response, err = t.h3Transport.RoundTrip(request)
|
||||
} else {
|
||||
response, err = t.h2FallbackRoundTrip(request)
|
||||
}
|
||||
results <- result{response: response, err: err, h3: useH3}
|
||||
// Each racer runs on a context of its own. A context shared by both cannot be
|
||||
// cancelled when one of them wins: quic-go and net/http reset the winner's
|
||||
// stream on cancellation, so the caller would be handed a response whose body
|
||||
// stops mid-read with H3_REQUEST_CANCELLED. Only losers are cancelled here;
|
||||
// the winner's cancel travels with its body and fires on Close.
|
||||
startRoundTrip := func(useH3 bool) context.CancelFunc {
|
||||
ctx, cancel := context.WithCancel(request.Context())
|
||||
raceRequest := cloneRequestForRetry(request).WithContext(ctx)
|
||||
go func() {
|
||||
var (
|
||||
response *http.Response
|
||||
err error
|
||||
)
|
||||
if useH3 {
|
||||
response, err = t.h3Transport.RoundTrip(raceRequest)
|
||||
} else {
|
||||
response, err = t.h2FallbackRoundTrip(raceRequest)
|
||||
}
|
||||
results <- result{response: response, err: err, h3: useH3}
|
||||
}()
|
||||
return cancel
|
||||
}
|
||||
goroutines := 1
|
||||
received := 0
|
||||
var fallbackCancel context.CancelFunc
|
||||
h3Cancel := startRoundTrip(true)
|
||||
drainRemaining := func() {
|
||||
cancel()
|
||||
for range goroutines - received {
|
||||
go func() {
|
||||
loser := <-results
|
||||
@@ -203,7 +237,6 @@ func (t *http3FallbackTransport) roundTripHTTP3Race(request *http.Request, autho
|
||||
}()
|
||||
}
|
||||
}
|
||||
go startRoundTrip(cloneRequestForRetry(request), true)
|
||||
timer := time.NewTimer(t.fallbackDelay)
|
||||
defer timer.Stop()
|
||||
var (
|
||||
@@ -215,20 +248,28 @@ func (t *http3FallbackTransport) roundTripHTTP3Race(request *http.Request, autho
|
||||
case <-timer.C:
|
||||
if goroutines == 1 {
|
||||
goroutines++
|
||||
go startRoundTrip(cloneRequestForRetry(request), false)
|
||||
fallbackCancel = startRoundTrip(false)
|
||||
}
|
||||
case raceResult := <-results:
|
||||
received++
|
||||
if raceResult.err == nil {
|
||||
winnerCancel := fallbackCancel
|
||||
if raceResult.h3 {
|
||||
t.clearH3Broken(authority)
|
||||
winnerCancel = h3Cancel
|
||||
if fallbackCancel != nil {
|
||||
fallbackCancel()
|
||||
}
|
||||
} else {
|
||||
h3Cancel()
|
||||
}
|
||||
drainRemaining()
|
||||
return raceResult.response, nil
|
||||
return withCancelOnBodyClose(raceResult.response, winnerCancel), nil
|
||||
}
|
||||
if raceResult.h3 {
|
||||
t.markH3Broken(authority)
|
||||
h3Err = raceResult.err
|
||||
h3Cancel()
|
||||
if goroutines == 1 {
|
||||
goroutines++
|
||||
if !timer.Stop() {
|
||||
@@ -237,14 +278,21 @@ func (t *http3FallbackTransport) roundTripHTTP3Race(request *http.Request, autho
|
||||
default:
|
||||
}
|
||||
}
|
||||
go startRoundTrip(cloneRequestForRetry(request), false)
|
||||
fallbackCancel = startRoundTrip(false)
|
||||
}
|
||||
} else {
|
||||
fallbackErr = raceResult.err
|
||||
if fallbackCancel != nil {
|
||||
fallbackCancel()
|
||||
}
|
||||
}
|
||||
if received < goroutines {
|
||||
continue
|
||||
}
|
||||
h3Cancel()
|
||||
if fallbackCancel != nil {
|
||||
fallbackCancel()
|
||||
}
|
||||
drainRemaining()
|
||||
switch {
|
||||
case h3Err != nil && fallbackErr != nil:
|
||||
|
||||
+90
-3
@@ -10,6 +10,7 @@ import (
|
||||
"net/url"
|
||||
"strconv"
|
||||
"sync"
|
||||
"sync/atomic"
|
||||
"time"
|
||||
|
||||
"github.com/sagernet/sing-box/adapter"
|
||||
@@ -171,6 +172,73 @@ func (t *HTTPSTransport) Exchange(ctx context.Context, message *mDNS.Msg) (*mDNS
|
||||
return response, nil
|
||||
}
|
||||
|
||||
// requestBuffer owns the pooled buffer that backs one DoH query.
|
||||
//
|
||||
// Both transports behind HTTPSTransportWrapper write the request body on a
|
||||
// goroutine of their own and return from RoundTrip as soon as the response
|
||||
// HEADERS arrive: net/http's write loop is still copying out of the body a
|
||||
// bufferful at a time (4 KiB of write buffer, or io.Copy's 32 KiB once it hands
|
||||
// the body to the connection), and http2's writeRequestBody has read only the
|
||||
// first max-frame-size bytes of it. Returning the buffer to the pool at that
|
||||
// point handed live memory to the next caller while the query was still going
|
||||
// out — everything past that first copy left the router as whatever that caller
|
||||
// had written there. A data race, and a memory-disclosure primitive aimed at the
|
||||
// resolver. Measured, not reasoned: with the write parked mid-query the bytes on
|
||||
// the wire diverge from the bytes we packed at exactly one copy buffer in.
|
||||
//
|
||||
// Ownership is counted rather than handed over once, because a retry holds two
|
||||
// bodies at a time and the two transports order that differently:
|
||||
// http.Transport.rewindBody CLOSES the old body before asking GetBody for a
|
||||
// new one, while http2's shouldRetryRequest asks GetBody first and closes the
|
||||
// old body on a goroutine. exchange keeps a count of its own until RoundTrip
|
||||
// returns — the only window in which either can call GetBody — so neither
|
||||
// ordering can free the buffer under the other. If a transport ever fails to
|
||||
// close a body, the count never reaches zero and the buffer is simply not
|
||||
// reused: garbage, not corruption.
|
||||
type requestBuffer struct {
|
||||
buffer *buf.Buffer
|
||||
raw []byte
|
||||
refs atomic.Int32
|
||||
}
|
||||
|
||||
func newRequestBuffer(buffer *buf.Buffer, raw []byte) *requestBuffer {
|
||||
holder := &requestBuffer{buffer: buffer, raw: raw}
|
||||
holder.refs.Store(1)
|
||||
return holder
|
||||
}
|
||||
|
||||
// body hands out a reader over the packed query as one more owner. It refuses
|
||||
// once the buffer is back in the pool, so a late caller gets an error instead
|
||||
// of a reader over memory that now belongs to somebody else.
|
||||
func (b *requestBuffer) body() (*pooledRequestBody, bool) {
|
||||
for {
|
||||
refs := b.refs.Load()
|
||||
if refs < 1 {
|
||||
return nil, false
|
||||
}
|
||||
if b.refs.CompareAndSwap(refs, refs+1) {
|
||||
return &pooledRequestBody{Reader: bytes.NewReader(b.raw), owner: b}, true
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func (b *requestBuffer) release() {
|
||||
if b.refs.Add(-1) == 0 {
|
||||
b.buffer.Release()
|
||||
}
|
||||
}
|
||||
|
||||
type pooledRequestBody struct {
|
||||
*bytes.Reader
|
||||
owner *requestBuffer
|
||||
closeOne sync.Once
|
||||
}
|
||||
|
||||
func (b *pooledRequestBody) Close() error {
|
||||
b.closeOne.Do(b.owner.release)
|
||||
return nil
|
||||
}
|
||||
|
||||
func (t *HTTPSTransport) exchange(ctx context.Context, message *mDNS.Msg) (*mDNS.Msg, error) {
|
||||
exMessage := *message
|
||||
exMessage.Id = 0
|
||||
@@ -181,11 +249,31 @@ func (t *HTTPSTransport) exchange(ctx context.Context, message *mDNS.Msg) (*mDNS
|
||||
requestBuffer.Release()
|
||||
return nil, err
|
||||
}
|
||||
request, err := http.NewRequestWithContext(ctx, http.MethodPost, t.destination.String(), bytes.NewReader(rawMessage))
|
||||
queryBuffer := newRequestBuffer(requestBuffer, rawMessage)
|
||||
// Drops the count exchange holds once RoundTrip is done with the request;
|
||||
// the bodies handed to the transport keep their own until it closes them.
|
||||
defer queryBuffer.release()
|
||||
requestBody, _ := queryBuffer.body() // cannot fail: the count above is ours
|
||||
request, err := http.NewRequestWithContext(ctx, http.MethodPost, t.destination.String(), requestBody)
|
||||
if err != nil {
|
||||
requestBuffer.Release()
|
||||
requestBody.Close()
|
||||
return nil, err
|
||||
}
|
||||
// http.NewRequestWithContext infers both only for the body types it knows,
|
||||
// and pooledRequestBody is not one of them. Upstream got them for free from
|
||||
// *bytes.Reader; GetBody is what lets a POST be replayed when a pooled
|
||||
// connection turns out to have been closed under us. Being unknown to
|
||||
// net/http also costs one packet on the HTTP/1.1 leg: isKnownInMemoryReader
|
||||
// no longer recognises the body, so the request headers are flushed before
|
||||
// the query instead of travelling with it.
|
||||
request.ContentLength = int64(len(rawMessage))
|
||||
request.GetBody = func() (io.ReadCloser, error) {
|
||||
retryBody, ok := queryBuffer.body()
|
||||
if !ok {
|
||||
return nil, E.New("DoH request buffer already released")
|
||||
}
|
||||
return retryBody, nil
|
||||
}
|
||||
request.Header = t.headers.Clone()
|
||||
request.Header.Set("Content-Type", MimeType)
|
||||
request.Header.Set("Accept", MimeType)
|
||||
@@ -193,7 +281,6 @@ func (t *HTTPSTransport) exchange(ctx context.Context, message *mDNS.Msg) (*mDNS
|
||||
currentTransport := t.transport
|
||||
t.transportAccess.Unlock()
|
||||
response, err := currentTransport.RoundTrip(request)
|
||||
requestBuffer.Release()
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
|
||||
@@ -0,0 +1,541 @@
|
||||
package transport
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"context"
|
||||
"errors"
|
||||
"io"
|
||||
"net"
|
||||
"net/http"
|
||||
"net/http/httptest"
|
||||
"net/url"
|
||||
"os"
|
||||
"strconv"
|
||||
"sync"
|
||||
"sync/atomic"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
C "github.com/sagernet/sing-box/constant"
|
||||
"github.com/sagernet/sing-box/dns"
|
||||
"github.com/sagernet/sing/common/buf"
|
||||
"github.com/sagernet/sing/common/logger"
|
||||
M "github.com/sagernet/sing/common/metadata"
|
||||
|
||||
mDNS "github.com/miekg/dns"
|
||||
"golang.org/x/net/http2"
|
||||
)
|
||||
|
||||
// The request body of a DoH query is backed by a POOLED buffer. Neither
|
||||
// transport behind HTTPSTransportWrapper is done with that body when RoundTrip
|
||||
// returns: net/http hands the request to a write loop of its own and returns as
|
||||
// soon as the response HEADERS have been read, and golang.org/x/net/http2 writes
|
||||
// the body on the goroutine that runs writeRequest while roundTrip waits on
|
||||
// respHeaderRecv. Returning the buffer to the pool at that point hands live
|
||||
// memory to the next caller while the query is still being written to the wire,
|
||||
// and what goes out is whatever that next caller put there.
|
||||
//
|
||||
// Both tests below force a window that is normally microseconds wide to stay
|
||||
// open, and drain the pool while it is open:
|
||||
//
|
||||
// - HTTP/1.1: the client connection stops accepting writes past the request
|
||||
// headers, so net/http's write loop is parked having copied only the first
|
||||
// io.Copy buffer (32 KiB) of the query.
|
||||
// - HTTP/2: the server pins a 1 KiB stream receive window and does not read
|
||||
// the body, so writeRequestBody is parked in awaitFlowControl having copied
|
||||
// only the first max-frame-size bytes of the query.
|
||||
//
|
||||
// In both, the server sends the response HEADERS first and withholds the
|
||||
// response BODY until the pool has been drained, so Exchange has returned from
|
||||
// RoundTrip — and released the buffer, on the broken build — while the query is
|
||||
// still going out.
|
||||
//
|
||||
// Both queries are padded past the transport's copy buffer on purpose. Below it
|
||||
// the transport lifts the whole query out of the pooled buffer in a single Read
|
||||
// that RACES the release rather than provably following it, and a test built on
|
||||
// that race would be a coin toss. The ownership defect is the same at every
|
||||
// size; only its deterministic proof needs the padding.
|
||||
|
||||
const (
|
||||
// Past io.Copy's 32 KiB buffer, which is the granularity net/http moves a
|
||||
// request body at (persistConnWriter.ReadFrom -> io.Copy), and still inside
|
||||
// buf.MaxPooledBufferSize so the buffer really comes from the pool.
|
||||
httpsH1PaddedQuerySize = 40000
|
||||
// Past http2's max frame size, which is how much of the body
|
||||
// writeRequestBody lifts into its scratch buffer per round.
|
||||
httpsH2PaddedQuerySize = 20000
|
||||
// Pinned on the HTTP/2 server so the client cannot write the whole body
|
||||
// before the response headers come back.
|
||||
httpsPinnedStreamWindow = 1024
|
||||
// Pinned too: Go's HTTP/2 server advertises a 1 MiB max frame size by
|
||||
// default, and the client sizes its body-copy buffer from that — with the
|
||||
// default it would slurp a 20 KB query in one Read and the divergence would
|
||||
// be hidden by the copy size rather than absent. 16384 is the protocol
|
||||
// minimum and what real resolvers advertise.
|
||||
httpsPinnedMaxFrameSize = 16384
|
||||
// How many times the HTTP/2 scenario is repeated; see the test.
|
||||
httpsH2Rounds = 8
|
||||
// How long to wait after the response headers before draining the pool, so
|
||||
// that Exchange has certainly returned from RoundTrip.
|
||||
httpsReleaseSettleDelay = 200 * time.Millisecond
|
||||
)
|
||||
|
||||
// httpsPaddedQuery returns a query and the exact bytes HTTPSTransport.exchange
|
||||
// packs for it.
|
||||
func httpsPaddedQuery(t *testing.T, padding int) (*mDNS.Msg, []byte) {
|
||||
t.Helper()
|
||||
message := new(mDNS.Msg)
|
||||
message.SetQuestion("example.com.", mDNS.TypeA)
|
||||
opt := new(mDNS.OPT)
|
||||
opt.Hdr.Name = "."
|
||||
opt.Hdr.Rrtype = mDNS.TypeOPT
|
||||
opt.Option = append(opt.Option, &mDNS.EDNS0_PADDING{Padding: make([]byte, padding)})
|
||||
message.Extra = append(message.Extra, opt)
|
||||
|
||||
onWire := *message
|
||||
onWire.Id = 0
|
||||
onWire.Compress = true
|
||||
expected, err := onWire.Pack()
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
return message, expected
|
||||
}
|
||||
|
||||
func httpsTestReply(t *testing.T) []byte {
|
||||
t.Helper()
|
||||
query := new(mDNS.Msg)
|
||||
query.SetQuestion("example.com.", mDNS.TypeA)
|
||||
response := new(mDNS.Msg)
|
||||
response.SetReply(query)
|
||||
raw, err := response.Pack()
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
return raw
|
||||
}
|
||||
|
||||
// httpsPoisonPool takes buffers of one size class out of the pool and fills them
|
||||
// with a pattern no DNS message contains. They are returned, not released: the
|
||||
// caller holds them so nothing can hand them back while the check runs.
|
||||
func httpsPoisonPool(size int, count int) []*buf.Buffer {
|
||||
poison := make([]*buf.Buffer, 0, count)
|
||||
for range count {
|
||||
buffer := buf.NewSize(size)
|
||||
poison = append(poison, buffer)
|
||||
free := buffer.FreeBytes()
|
||||
for i := range free {
|
||||
free[i] = 0xEE
|
||||
}
|
||||
}
|
||||
return poison
|
||||
}
|
||||
|
||||
func httpsReleaseAll(buffers []*buf.Buffer) {
|
||||
for _, buffer := range buffers {
|
||||
buffer.Release()
|
||||
}
|
||||
}
|
||||
|
||||
// httpsRequirePoisonReachesReleasedBuffer is the CONTROL for the tests below. A
|
||||
// clean result there means nothing unless this instrument is shown to be able to
|
||||
// produce a dirty one: it must be true that a buffer released while its bytes
|
||||
// are still referenced comes back out of the pool and gets overwritten. If that
|
||||
// stops holding — a different allocator, a pool that zeroes, a size class that
|
||||
// is not pooled at all — the tests below would go green on broken code.
|
||||
//
|
||||
// Retried, because under -race sync.Pool.Put drops one object in four on
|
||||
// purpose. That same dice roll is why the checks below are 3-in-4 detectors
|
||||
// under -race and certainties without it; it can only make a broken build look
|
||||
// clean, never a clean build look broken.
|
||||
func httpsRequirePoisonReachesReleasedBuffer(t *testing.T, size int, pattern []byte) {
|
||||
t.Helper()
|
||||
for range 32 {
|
||||
control := buf.NewSize(size)
|
||||
free := control.FreeBytes()
|
||||
if len(free) < len(pattern) {
|
||||
t.Fatalf("control failed: a %d-byte buffer came back %d bytes long", size, len(free))
|
||||
}
|
||||
copy(free, pattern)
|
||||
alias := free[:len(pattern)]
|
||||
control.Release()
|
||||
|
||||
held := httpsPoisonPool(size, 8)
|
||||
poisoned := !bytes.Equal(alias, pattern)
|
||||
httpsReleaseAll(held)
|
||||
if poisoned {
|
||||
return
|
||||
}
|
||||
}
|
||||
t.Fatal("control failed: poisoning the pool never touched a released buffer, so a clean result below would prove nothing")
|
||||
}
|
||||
|
||||
// httpsTestDialer hands HTTPSTransportWrapper a connection to a local test
|
||||
// server, optionally wrapped.
|
||||
type httpsTestDialer struct {
|
||||
target string
|
||||
wrap func(net.Conn) net.Conn
|
||||
access sync.Mutex
|
||||
conns []net.Conn
|
||||
}
|
||||
|
||||
func (d *httpsTestDialer) DialContext(ctx context.Context, network string, destination M.Socksaddr) (net.Conn, error) {
|
||||
conn, err := (&net.Dialer{}).DialContext(ctx, "tcp", d.target)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
var wrapped net.Conn = conn
|
||||
if d.wrap != nil {
|
||||
wrapped = d.wrap(conn)
|
||||
}
|
||||
d.access.Lock()
|
||||
d.conns = append(d.conns, conn)
|
||||
d.access.Unlock()
|
||||
return wrapped, nil
|
||||
}
|
||||
|
||||
func (d *httpsTestDialer) ListenPacket(ctx context.Context, destination M.Socksaddr) (net.PacketConn, error) {
|
||||
return nil, os.ErrInvalid
|
||||
}
|
||||
|
||||
func (d *httpsTestDialer) closeAll() {
|
||||
d.access.Lock()
|
||||
defer d.access.Unlock()
|
||||
for _, conn := range d.conns {
|
||||
conn.Close()
|
||||
}
|
||||
}
|
||||
|
||||
// httpsGatedConn stops accepting writes once limit bytes have gone out, until
|
||||
// the gate is opened. HTTP/1.1 has no flow-control knob to park the writer with,
|
||||
// so the connection provides one.
|
||||
type httpsGatedConn struct {
|
||||
net.Conn
|
||||
limit int64
|
||||
written atomic.Int64
|
||||
gate chan struct{}
|
||||
}
|
||||
|
||||
func (c *httpsGatedConn) Write(p []byte) (int, error) {
|
||||
if c.written.Load()+int64(len(p)) > c.limit {
|
||||
select {
|
||||
case <-c.gate:
|
||||
case <-time.After(30 * time.Second):
|
||||
return 0, errors.New("gated conn: nobody opened the gate")
|
||||
}
|
||||
}
|
||||
n, err := c.Conn.Write(p)
|
||||
c.written.Add(int64(n))
|
||||
return n, err
|
||||
}
|
||||
|
||||
// httpsSlowServer is the handler both tests share: response HEADERS first, then
|
||||
// nothing until the pool has been drained, then the request body, then the
|
||||
// response body.
|
||||
type httpsSlowServer struct {
|
||||
reply []byte
|
||||
served atomic.Int32
|
||||
warmups int32
|
||||
headersSent chan struct{}
|
||||
bodyGate chan struct{}
|
||||
received chan []byte
|
||||
readErr chan error
|
||||
}
|
||||
|
||||
func newHTTPSSlowServer(reply []byte) *httpsSlowServer {
|
||||
return &httpsSlowServer{
|
||||
reply: reply,
|
||||
headersSent: make(chan struct{}, 1),
|
||||
bodyGate: make(chan struct{}),
|
||||
received: make(chan []byte, 1),
|
||||
readErr: make(chan error, 1),
|
||||
}
|
||||
}
|
||||
|
||||
func (s *httpsSlowServer) ServeHTTP(writer http.ResponseWriter, request *http.Request) {
|
||||
if s.served.Add(1) <= s.warmups {
|
||||
// Warm-up: answer normally, so the connection is established and the
|
||||
// client has applied the server's SETTINGS before the query that
|
||||
// matters goes out.
|
||||
io.Copy(io.Discard, request.Body)
|
||||
writer.Header().Set("Content-Type", MimeType)
|
||||
writer.Header().Set("Content-Length", strconv.Itoa(len(s.reply)))
|
||||
writer.Write(s.reply)
|
||||
return
|
||||
}
|
||||
// Without this, net/http's HTTP/1.1 server drains up to 256 KB of the
|
||||
// request body before it will write response headers, precisely so that a
|
||||
// half-duplex client cannot deadlock. That would consume the query before
|
||||
// the client is anywhere near done sending it, and there would be nothing
|
||||
// left in flight to catch. Full duplex is how a resolver that answers from
|
||||
// cache before reading the whole query behaves; HTTP/2 is full duplex
|
||||
// already and returns an error here, which is fine.
|
||||
http.NewResponseController(writer).EnableFullDuplex()
|
||||
writer.Header().Set("Content-Type", MimeType)
|
||||
// Content-Length matters: without it Exchange falls into io.ReadAll and
|
||||
// waits for the end of the response, which this handler is about to
|
||||
// withhold on purpose.
|
||||
writer.Header().Set("Content-Length", strconv.Itoa(len(s.reply)))
|
||||
writer.WriteHeader(http.StatusOK)
|
||||
writer.(http.Flusher).Flush()
|
||||
s.headersSent <- struct{}{}
|
||||
|
||||
// A real resolver would be reading the query by now. Withholding it is what
|
||||
// keeps the client parked mid-body while the pool is drained.
|
||||
<-s.bodyGate
|
||||
body, err := io.ReadAll(request.Body)
|
||||
s.readErr <- err
|
||||
s.received <- body
|
||||
|
||||
writer.Write(s.reply)
|
||||
}
|
||||
|
||||
// drainPoolOnceHeadersAreOut waits for the response headers, gives Exchange time
|
||||
// to return from RoundTrip, drains the size class the query buffer came from —
|
||||
// on this goroutine, so a buffer released on the way out lands in our hands and
|
||||
// not somewhere harmless — and only then lets the server read the query.
|
||||
func (s *httpsSlowServer) drainPoolOnceHeadersAreOut(bufferSize int) <-chan []*buf.Buffer {
|
||||
poisoned := make(chan []*buf.Buffer, 1)
|
||||
go func() {
|
||||
<-s.headersSent
|
||||
time.Sleep(httpsReleaseSettleDelay)
|
||||
poisoned <- httpsPoisonPool(bufferSize, 32)
|
||||
close(s.bodyGate)
|
||||
}()
|
||||
return poisoned
|
||||
}
|
||||
|
||||
func (s *httpsSlowServer) requireQueryOnWire(t *testing.T, expected []byte) {
|
||||
t.Helper()
|
||||
var sent []byte
|
||||
select {
|
||||
case sent = <-s.received:
|
||||
case <-time.After(30 * time.Second):
|
||||
t.Fatal("the server never received the request body")
|
||||
}
|
||||
if err := <-s.readErr; err != nil {
|
||||
t.Fatal("reading the request body: ", err)
|
||||
}
|
||||
if bytes.Equal(sent, expected) {
|
||||
return
|
||||
}
|
||||
firstDiff := -1
|
||||
for i := 0; i < len(sent) && i < len(expected); i++ {
|
||||
if sent[i] != expected[i] {
|
||||
firstDiff = i
|
||||
break
|
||||
}
|
||||
}
|
||||
t.Fatalf("the query on the wire is not the query we packed: %d of %d bytes received, first difference at offset %d — "+
|
||||
"the pooled request buffer was reused while the transport was still reading it", len(sent), len(expected), firstDiff)
|
||||
}
|
||||
|
||||
// TestHTTPSExchangeRequestBufferOutlivesRoundTripHTTP1 proves that the query an
|
||||
// HTTP/1.1 resolver receives is the query we asked to send, even when the pool
|
||||
// is drained the instant the response headers arrive.
|
||||
func TestHTTPSExchangeRequestBufferOutlivesRoundTripHTTP1(t *testing.T) {
|
||||
message, expected := httpsPaddedQuery(t, httpsH1PaddedQuerySize)
|
||||
bufferSize := 1 + message.Len()
|
||||
httpsRequirePoisonReachesReleasedBuffer(t, bufferSize, expected)
|
||||
|
||||
handler := newHTTPSSlowServer(httpsTestReply(t))
|
||||
server := httptest.NewServer(handler)
|
||||
t.Cleanup(server.Close)
|
||||
|
||||
dialer := &httpsTestDialer{
|
||||
target: server.Listener.Addr().String(),
|
||||
wrap: func(conn net.Conn) net.Conn {
|
||||
// One 4 KiB flush of net/http's write buffer gets through, which is
|
||||
// what carries the request headers to the server, and the write loop
|
||||
// parks on the next one — still holding the query.
|
||||
return &httpsGatedConn{Conn: conn, limit: 4096, gate: handler.bodyGate}
|
||||
},
|
||||
}
|
||||
t.Cleanup(dialer.closeAll)
|
||||
|
||||
// Scheme http puts HTTPSTransportWrapper on its HTTP/1.1 leg, the one it
|
||||
// also falls back to whenever a resolver does not negotiate h2.
|
||||
destination := &url.URL{Scheme: "http", Host: "doh.invalid", Path: "/dns-query"}
|
||||
dnsTransport := &HTTPSTransport{
|
||||
TransportAdapter: dns.NewTransportAdapter(C.DNSTypeHTTPS, "test-doh-h1", nil),
|
||||
logger: logger.NOP(),
|
||||
dialer: dialer,
|
||||
destination: destination,
|
||||
headers: http.Header{},
|
||||
transport: NewHTTPSTransportWrapper(dialer, M.ParseSocksaddr(server.Listener.Addr().String()), destination),
|
||||
}
|
||||
t.Cleanup(func() { dnsTransport.Close() })
|
||||
|
||||
poisoned := handler.drainPoolOnceHeadersAreOut(bufferSize)
|
||||
|
||||
ctx, cancel := context.WithTimeout(context.Background(), 60*time.Second)
|
||||
defer cancel()
|
||||
if _, err := dnsTransport.Exchange(ctx, message); err != nil {
|
||||
t.Fatal("exchange: ", err)
|
||||
}
|
||||
defer httpsReleaseAll(<-poisoned)
|
||||
|
||||
handler.requireQueryOnWire(t, expected)
|
||||
}
|
||||
|
||||
// TestHTTPSExchangeRequestBufferOutlivesRoundTripHTTP2 does the same over h2,
|
||||
// the leg every resolver that speaks HTTP/2 lands on.
|
||||
func TestHTTPSExchangeRequestBufferOutlivesRoundTripHTTP2(t *testing.T) {
|
||||
message, expected := httpsPaddedQuery(t, httpsH2PaddedQuerySize)
|
||||
bufferSize := 1 + message.Len()
|
||||
httpsRequirePoisonReachesReleasedBuffer(t, bufferSize, expected)
|
||||
// Repeated because a buffer released on the goroutine running Exchange
|
||||
// usually lands in that P's private sync.Pool slot, which the goroutine
|
||||
// draining the pool cannot steal: one round catches a broken build about
|
||||
// half the time, eight catch it better than 99 times in 100. Every round
|
||||
// must come back clean.
|
||||
for round := range httpsH2Rounds {
|
||||
if !t.Run(strconv.Itoa(round), func(t *testing.T) {
|
||||
httpsH2Round(t, message, expected, bufferSize)
|
||||
}) {
|
||||
return
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func httpsH2Round(t *testing.T, message *mDNS.Msg, expected []byte, bufferSize int) {
|
||||
handler := newHTTPSSlowServer(httpsTestReply(t))
|
||||
// x/net/http2 may put the first request on the wire before it has applied
|
||||
// the server's SETTINGS, and would then overrun the 1 KiB window this test
|
||||
// pins and be reset with FLOW_CONTROL_ERROR. One small query first settles
|
||||
// that: reading its response proves the SETTINGS frame ahead of it was
|
||||
// processed.
|
||||
handler.warmups = 1
|
||||
listener, err := net.Listen("tcp", "127.0.0.1:0")
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
t.Cleanup(func() { listener.Close() })
|
||||
h2server := &http2.Server{
|
||||
MaxUploadBufferPerStream: httpsPinnedStreamWindow,
|
||||
MaxReadFrameSize: httpsPinnedMaxFrameSize,
|
||||
}
|
||||
go func() {
|
||||
for {
|
||||
conn, acceptErr := listener.Accept()
|
||||
if acceptErr != nil {
|
||||
return
|
||||
}
|
||||
go h2server.ServeConn(conn, &http2.ServeConnOpts{Handler: handler})
|
||||
}
|
||||
}()
|
||||
|
||||
dialer := &httpsTestDialer{target: listener.Addr().String()}
|
||||
t.Cleanup(dialer.closeAll)
|
||||
|
||||
// Scheme https keeps HTTPSTransportWrapper on its h2 leg. The dialer hands
|
||||
// back a plain connection, which x/net/http2 speaks prior-knowledge h2 over;
|
||||
// TLS adds nothing this test is about.
|
||||
destination := &url.URL{Scheme: "https", Host: "doh.invalid", Path: "/dns-query"}
|
||||
dnsTransport := &HTTPSTransport{
|
||||
TransportAdapter: dns.NewTransportAdapter(C.DNSTypeHTTPS, "test-doh-h2", nil),
|
||||
logger: logger.NOP(),
|
||||
dialer: dialer,
|
||||
destination: destination,
|
||||
headers: http.Header{},
|
||||
transport: NewHTTPSTransportWrapper(dialer, M.ParseSocksaddr(listener.Addr().String()), destination),
|
||||
}
|
||||
t.Cleanup(func() { dnsTransport.Close() })
|
||||
|
||||
ctx, cancel := context.WithTimeout(context.Background(), 60*time.Second)
|
||||
defer cancel()
|
||||
warmup := new(mDNS.Msg)
|
||||
warmup.SetQuestion("warmup.invalid.", mDNS.TypeA)
|
||||
if _, err = dnsTransport.Exchange(ctx, warmup); err != nil {
|
||||
t.Fatal("warm-up exchange: ", err)
|
||||
}
|
||||
|
||||
poisoned := handler.drainPoolOnceHeadersAreOut(bufferSize)
|
||||
if _, err = dnsTransport.Exchange(ctx, message); err != nil {
|
||||
t.Fatal("exchange: ", err)
|
||||
}
|
||||
defer httpsReleaseAll(<-poisoned)
|
||||
|
||||
handler.requireQueryOnWire(t, expected)
|
||||
}
|
||||
|
||||
// TestHTTPSRequestBufferSurvivesRewind covers the second owner a retry creates.
|
||||
// net/http rewinds a dead connection's request by CLOSING the body it has and
|
||||
// then asking GetBody for another one (rewindBody), while x/net/http2 asks
|
||||
// GetBody first and closes the old body on a goroutine (shouldRetryRequest,
|
||||
// closeReqBodyLocked). Either ordering frees the buffer under the retry if the
|
||||
// first Close is what returns it to the pool, and the retry then sends whatever
|
||||
// the next pool user wrote — the same disclosure, one attempt later.
|
||||
func TestHTTPSRequestBufferSurvivesRewind(t *testing.T) {
|
||||
message, expected := httpsPaddedQuery(t, httpsH2PaddedQuerySize)
|
||||
bufferSize := 1 + message.Len()
|
||||
httpsRequirePoisonReachesReleasedBuffer(t, bufferSize, expected)
|
||||
|
||||
exMessage := *message
|
||||
exMessage.Id = 0
|
||||
exMessage.Compress = true
|
||||
requestBuffer := buf.NewSize(bufferSize)
|
||||
rawMessage, err := exMessage.PackBuffer(requestBuffer.FreeBytes())
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
queryBuffer := newRequestBuffer(requestBuffer, rawMessage)
|
||||
defer queryBuffer.release()
|
||||
|
||||
first, ok := queryBuffer.body()
|
||||
if !ok {
|
||||
t.Fatal("the first body was refused while exchange still holds the buffer")
|
||||
}
|
||||
// The transport got some of the query out before the connection turned out
|
||||
// to be dead, then closed the body.
|
||||
if _, err = io.CopyN(io.Discard, first, 128); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
first.Close()
|
||||
|
||||
// GetBody, as the retry would call it.
|
||||
second, ok := queryBuffer.body()
|
||||
if !ok {
|
||||
t.Fatal("GetBody was refused after the first body was closed: the retry has no query left to send")
|
||||
}
|
||||
poison := httpsPoisonPool(bufferSize, 32)
|
||||
defer httpsReleaseAll(poison)
|
||||
|
||||
retried, err := io.ReadAll(second)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if !bytes.Equal(retried, expected) {
|
||||
firstDiff := -1
|
||||
for i := 0; i < len(retried) && i < len(expected); i++ {
|
||||
if retried[i] != expected[i] {
|
||||
firstDiff = i
|
||||
break
|
||||
}
|
||||
}
|
||||
t.Fatalf("the retried query is not the query we packed: %d of %d bytes, first difference at offset %d — "+
|
||||
"closing the first body returned the buffer to the pool while the retry still needed it", len(retried), len(expected), firstDiff)
|
||||
}
|
||||
second.Close()
|
||||
}
|
||||
|
||||
// TestHTTPSRequestBufferRefusesBodyAfterRelease pins the recoverable end of the
|
||||
// contract: once the buffer really is back in the pool, GetBody must hand out an
|
||||
// error rather than a reader over memory that now belongs to somebody else.
|
||||
func TestHTTPSRequestBufferRefusesBodyAfterRelease(t *testing.T) {
|
||||
requestBuffer := buf.NewSize(64)
|
||||
rawMessage := requestBuffer.FreeBytes()[:8]
|
||||
queryBuffer := newRequestBuffer(requestBuffer, rawMessage)
|
||||
|
||||
body, ok := queryBuffer.body()
|
||||
if !ok {
|
||||
t.Fatal("the first body was refused while the caller still holds the buffer")
|
||||
}
|
||||
body.Close()
|
||||
body.Close() // http3 and net/http both manage to close a body twice
|
||||
queryBuffer.release()
|
||||
|
||||
if _, ok = queryBuffer.body(); ok {
|
||||
t.Fatal("a body was handed out over a buffer that is already back in the pool")
|
||||
}
|
||||
}
|
||||
@@ -162,15 +162,34 @@ func (t *HTTP3Transport) Exchange(ctx context.Context, message *mDNS.Msg) (*mDNS
|
||||
exMessage := *message
|
||||
exMessage.Id = 0
|
||||
exMessage.Compress = true
|
||||
requestBuffer := buf.NewSize(1 + message.Len())
|
||||
rawMessage, err := exMessage.PackBuffer(requestBuffer.FreeBytes())
|
||||
// NOT a pooled buffer, deliberately — the request body must own memory this
|
||||
// transport can never hand back.
|
||||
//
|
||||
// quic-go writes the request body on a goroutine of its own (http3's
|
||||
// doRequest spawns it and goes on to block in ReadResponse), and NOTHING ever
|
||||
// joins that goroutine. On the success path sendRequestBody closes the body
|
||||
// when it is finished, but on every error path RoundTripOpt closes it as soon
|
||||
// as doRequest returns — and doRequest waits only on the request-cancellation
|
||||
// watchdog, not on the writer. So there is no moment at which this code can
|
||||
// know the body is no longer being read, and therefore no moment at which it
|
||||
// may return a pooled buffer. Releasing on Close looks like an ownership
|
||||
// handoff and is not one.
|
||||
//
|
||||
// Owning it costs nothing here, measured rather than assumed: for a typical
|
||||
// query (a 36-byte name, A record) Pack is 87 ns/op at 64 B and 1 alloc,
|
||||
// against 108 ns/op at 64 B and 1 alloc for packing into a pooled buffer. The
|
||||
// pool never avoided an allocation on this path — buf.NewSize allocates the
|
||||
// Buffer struct itself, the same 64 bytes the message needs — it only added
|
||||
// Get/Put on top. This path is hot in queries, not in bytes.
|
||||
//
|
||||
// The response buffer below stays pooled: it is read to completion and
|
||||
// unpacked before Exchange returns, and nothing outlives it.
|
||||
rawMessage, err := exMessage.Pack()
|
||||
if err != nil {
|
||||
requestBuffer.Release()
|
||||
return nil, err
|
||||
}
|
||||
request, err := http.NewRequestWithContext(ctx, http.MethodPost, t.destination.String(), bytes.NewReader(rawMessage))
|
||||
if err != nil {
|
||||
requestBuffer.Release()
|
||||
return nil, err
|
||||
}
|
||||
request.Header = t.headers.Clone()
|
||||
@@ -180,7 +199,6 @@ func (t *HTTP3Transport) Exchange(ctx context.Context, message *mDNS.Msg) (*mDNS
|
||||
currentTransport := t.transport
|
||||
t.transportAccess.Unlock()
|
||||
response, err := currentTransport.RoundTrip(request)
|
||||
requestBuffer.Release()
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
|
||||
@@ -0,0 +1,426 @@
|
||||
package quic
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"context"
|
||||
"crypto/rand"
|
||||
"crypto/tls"
|
||||
"io"
|
||||
"net"
|
||||
"net/http"
|
||||
"net/url"
|
||||
"strconv"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"github.com/sagernet/quic-go"
|
||||
"github.com/sagernet/quic-go/http3"
|
||||
C "github.com/sagernet/sing-box/constant"
|
||||
"github.com/sagernet/sing-box/dns"
|
||||
"github.com/sagernet/sing-box/dns/transport"
|
||||
"github.com/sagernet/sing/common/buf"
|
||||
"github.com/sagernet/sing/common/logger"
|
||||
M "github.com/sagernet/sing/common/metadata"
|
||||
|
||||
mDNS "github.com/miekg/dns"
|
||||
)
|
||||
|
||||
// The request body of a DoH3 query used to be backed by a POOLED buffer. quic-go
|
||||
// sends that body on a goroutine of its own which outlives RoundTrip (http3's
|
||||
// doRequest spawns it and returns as soon as the response HEADERS arrive), and
|
||||
// NOTHING joins that goroutine, so there is no moment at which the transport may
|
||||
// hand the buffer back.
|
||||
//
|
||||
// Two tests, because the two paths are observable in different ways.
|
||||
//
|
||||
// - On the SUCCESS path the body keeps flowing, so the damage is visible on the
|
||||
// wire: TestHTTP3ExchangeRequestBufferOutlivesRoundTrip pins a 2 KB server
|
||||
// stream window and answers before reading the body, so the client is still
|
||||
// writing when Exchange returns, and compares what the server received.
|
||||
//
|
||||
// - On the FAILURE and CANCELLATION paths the damage is not visible on the wire
|
||||
// at all: every ReadResponse error in quic-go calls str.CancelWrite BEFORE
|
||||
// RoundTripOpt closes the body, so whatever the writer reads afterwards is
|
||||
// thrown at a dead stream. What is left is a read of memory that belongs to
|
||||
// somebody else. TestHTTP3ExchangeNeverPacksQueriesIntoPooledMemory therefore
|
||||
// pins the CAUSE instead of the symptom: the bytes of a query must never end
|
||||
// up in a buffer this transport can return to the pool.
|
||||
|
||||
const (
|
||||
// Big enough to need more than one 8 KiB read out of the request body
|
||||
// (http3's bodyCopyBufferSize), small enough to still come from the pool
|
||||
// (buf.MaxPooledBufferSize).
|
||||
paddedQuerySize = 20000
|
||||
// Pinned on the server so the client cannot write the whole body before the
|
||||
// response comes back.
|
||||
pinnedStreamWindow = 2048
|
||||
// Padding for the marked query of the ownership test. Only has to be
|
||||
// distinctive and pooled, not large.
|
||||
markedQueryPadding = 4096
|
||||
markedQueryNeedle = 64
|
||||
// How deep to drain a size class when looking for the needle.
|
||||
poolScanDepth = 64
|
||||
// How many times a CONTROL may repeat before it gives up.
|
||||
//
|
||||
// Both controls in this file assert the same thing — a buffer released while
|
||||
// its bytes are still referenced comes back out of the pool — and under
|
||||
// `-race` that is a DICE ROLL, not a certainty: sync.Pool.Put drops one
|
||||
// object in four on purpose (runtime_randn(4) == 0, sync/pool.go). Measured
|
||||
// in golang:1.26 with `go test -race -count=60`: the single-attempt control
|
||||
// failed 18 times out of 60, i.e. the gate's -race pass had a ~30% chance of
|
||||
// going red on a tree with nothing wrong with it.
|
||||
//
|
||||
// A retry is the honest repair rather than a papering-over, because the
|
||||
// control's claim is EXISTENTIAL — "this instrument is able to find a
|
||||
// released, still-referenced buffer" — and one success proves it. It is not
|
||||
// an average over attempts, so nothing is diluted by taking more than one.
|
||||
// 32 attempts leave a (1/4)^32 chance of a false alarm.
|
||||
//
|
||||
// What this does NOT do, said plainly: it does not make the VERDICT below
|
||||
// certain under -race. The same 1-in-4 drop means a scan that comes back
|
||||
// clean has a 1-in-4 chance of being clean because the pool threw the
|
||||
// evidence away. That direction is the safe one — it can only let a broken
|
||||
// build look clean, never make a clean build look broken — and the -race
|
||||
// pass is not the only one that runs this test: [2/7] of scripts/run-tests.sh
|
||||
// runs the same file WITHOUT -race, where both the control and the verdict
|
||||
// are certainties.
|
||||
controlAttempts = 32
|
||||
)
|
||||
|
||||
func paddedQuery(t *testing.T) (*mDNS.Msg, []byte) {
|
||||
t.Helper()
|
||||
message := new(mDNS.Msg)
|
||||
message.SetQuestion("example.com.", mDNS.TypeA)
|
||||
opt := new(mDNS.OPT)
|
||||
opt.Hdr.Name = "."
|
||||
opt.Hdr.Rrtype = mDNS.TypeOPT
|
||||
opt.Option = append(opt.Option, &mDNS.EDNS0_PADDING{Padding: make([]byte, paddedQuerySize)})
|
||||
message.Extra = append(message.Extra, opt)
|
||||
|
||||
// Exactly what HTTP3Transport.Exchange puts on the wire.
|
||||
onWire := *message
|
||||
onWire.Id = 0
|
||||
onWire.Compress = true
|
||||
expected, err := onWire.Pack()
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
return message, expected
|
||||
}
|
||||
|
||||
// poisonPool takes buffers of one size class out of the pool and fills them with
|
||||
// a pattern no DNS message contains. The buffers are returned, not released: the
|
||||
// caller holds them so nothing can hand them back while the check runs.
|
||||
func poisonPool(size int, count int) []*buf.Buffer {
|
||||
poison := make([]*buf.Buffer, 0, count)
|
||||
for range count {
|
||||
buffer := buf.NewSize(size)
|
||||
poison = append(poison, buffer)
|
||||
free := buffer.FreeBytes()
|
||||
for i := range free {
|
||||
free[i] = 0xEE
|
||||
}
|
||||
}
|
||||
return poison
|
||||
}
|
||||
|
||||
func releaseAll(buffers []*buf.Buffer) {
|
||||
for _, buffer := range buffers {
|
||||
buffer.Release()
|
||||
}
|
||||
}
|
||||
|
||||
// requirePoisonReachesReleasedBuffer is the CONTROL for the test below. A clean
|
||||
// result there means nothing unless this instrument is shown to be able to
|
||||
// produce a dirty one: it must be true that a buffer released while its bytes
|
||||
// are still referenced comes back out of the pool and gets overwritten. If this
|
||||
// stops holding — a different allocator, a pool that zeroes, a size class that
|
||||
// is not pooled at all — the test below would go green on broken code.
|
||||
//
|
||||
// Retried, because under -race sync.Pool.Put drops one object in four on
|
||||
// purpose. That same dice roll is why the check below is a 3-in-4 detector under
|
||||
// -race and a certainty without it; it can only make a broken build look clean,
|
||||
// never a clean build look broken. See controlAttempts.
|
||||
func requirePoisonReachesReleasedBuffer(t *testing.T, size int, pattern []byte) {
|
||||
t.Helper()
|
||||
for range controlAttempts {
|
||||
control := buf.NewSize(size)
|
||||
free := control.FreeBytes()
|
||||
if len(free) < len(pattern) {
|
||||
t.Fatalf("control failed: a %d-byte buffer came back %d bytes long", size, len(free))
|
||||
}
|
||||
copy(free, pattern)
|
||||
alias := free[:len(pattern)]
|
||||
control.Release()
|
||||
|
||||
held := poisonPool(size, 8)
|
||||
poisoned := !bytes.Equal(alias, pattern)
|
||||
releaseAll(held)
|
||||
if poisoned {
|
||||
return
|
||||
}
|
||||
}
|
||||
t.Fatal("control failed: poisoning the pool never touched a released buffer, so a clean result below would prove nothing")
|
||||
}
|
||||
|
||||
// TestHTTP3ExchangeRequestBufferOutlivesRoundTrip proves that the query the
|
||||
// server receives is the query we asked to send, even when the pool is drained
|
||||
// the instant Exchange returns.
|
||||
func TestHTTP3ExchangeRequestBufferOutlivesRoundTrip(t *testing.T) {
|
||||
message, expected := paddedQuery(t)
|
||||
bufferSize := 1 + message.Len()
|
||||
requirePoisonReachesReleasedBuffer(t, bufferSize, expected)
|
||||
|
||||
drainGate := make(chan struct{})
|
||||
received := make(chan []byte, 1)
|
||||
mux := http.NewServeMux()
|
||||
mux.HandleFunc("/dns-query", func(writer http.ResponseWriter, request *http.Request) {
|
||||
// Answer BEFORE reading the request body. A real resolver would not, but
|
||||
// any peer, middlebox or loss pattern that delays the body has the same
|
||||
// effect, and this makes the window deterministic.
|
||||
response := new(mDNS.Msg)
|
||||
response.SetReply(testQuery())
|
||||
rawResponse, err := response.Pack()
|
||||
if err != nil {
|
||||
writer.WriteHeader(http.StatusInternalServerError)
|
||||
return
|
||||
}
|
||||
writer.Header().Set("Content-Type", transport.MimeType)
|
||||
// Content-Length matters here: without it Exchange falls into io.ReadAll
|
||||
// and waits for the stream FIN, which this handler is about to withhold.
|
||||
writer.Header().Set("Content-Length", strconv.Itoa(len(rawResponse)))
|
||||
writer.Write(rawResponse)
|
||||
writer.(http.Flusher).Flush()
|
||||
|
||||
<-drainGate
|
||||
body, _ := io.ReadAll(request.Body)
|
||||
received <- body
|
||||
})
|
||||
listener, err := quic.ListenAddrEarly("127.0.0.1:0", testServerTLSConfig(t, []string{http3.NextProtoH3}), &quic.Config{
|
||||
InitialStreamReceiveWindow: pinnedStreamWindow,
|
||||
MaxStreamReceiveWindow: pinnedStreamWindow,
|
||||
InitialConnectionReceiveWindow: 1 << 16,
|
||||
MaxConnectionReceiveWindow: 1 << 16,
|
||||
})
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
server := &http3.Server{Handler: mux}
|
||||
go server.ServeListener(listener)
|
||||
t.Cleanup(func() {
|
||||
server.Close()
|
||||
listener.Close()
|
||||
})
|
||||
|
||||
dialer := &trackingDialer{}
|
||||
t.Cleanup(dialer.closeAll)
|
||||
dnsTransport := &HTTP3Transport{
|
||||
TransportAdapter: dns.NewTransportAdapter(C.DNSTypeHTTP3, "test-doh3-buffer", nil),
|
||||
logger: logger.NOP(),
|
||||
dialer: dialer,
|
||||
destination: &url.URL{Scheme: "https", Host: "localhost", Path: "/dns-query"},
|
||||
headers: http.Header{},
|
||||
serverAddr: M.ParseSocksaddr(listener.Addr().String()),
|
||||
tlsConfig: &tls.Config{
|
||||
InsecureSkipVerify: true,
|
||||
ServerName: "localhost",
|
||||
NextProtos: []string{http3.NextProtoH3},
|
||||
MinVersion: tls.VersionTLS13,
|
||||
},
|
||||
}
|
||||
dnsTransport.transport = dnsTransport.newTransport()
|
||||
t.Cleanup(func() { dnsTransport.Close() })
|
||||
|
||||
ctx, cancel := context.WithTimeout(context.Background(), 30*time.Second)
|
||||
defer cancel()
|
||||
if _, err = dnsTransport.Exchange(ctx, message); err != nil {
|
||||
t.Fatal("exchange: ", err)
|
||||
}
|
||||
|
||||
// Exchange has returned, the body is still in flight. Drain the size class it
|
||||
// came from, on this very goroutine, so a buffer released on the way out lands
|
||||
// in our hands and not somewhere harmless. The buffers are held until after
|
||||
// the comparison below.
|
||||
poison := poisonPool(bufferSize, 32)
|
||||
defer releaseAll(poison)
|
||||
|
||||
close(drainGate)
|
||||
var sent []byte
|
||||
select {
|
||||
case sent = <-received:
|
||||
case <-time.After(20 * time.Second):
|
||||
t.Fatal("the server never received the request body")
|
||||
}
|
||||
if !bytes.Equal(sent, expected) {
|
||||
firstDiff := -1
|
||||
for i := 0; i < len(sent) && i < len(expected); i++ {
|
||||
if sent[i] != expected[i] {
|
||||
firstDiff = i
|
||||
break
|
||||
}
|
||||
}
|
||||
t.Fatalf("the query on the wire is not the query we packed: %d of %d bytes received, first difference at offset %d — "+
|
||||
"the pooled request buffer was reused while quic-go was still reading it", len(sent), len(expected), firstDiff)
|
||||
}
|
||||
}
|
||||
|
||||
// markedQuery builds a query whose EDNS0 padding carries a random tag, so the
|
||||
// packed bytes contain a needle that can be searched for in pool memory and
|
||||
// cannot collide with anything else.
|
||||
func markedQuery(t *testing.T) (*mDNS.Msg, []byte) {
|
||||
t.Helper()
|
||||
padding := make([]byte, markedQueryPadding)
|
||||
if _, err := rand.Read(padding); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
message := new(mDNS.Msg)
|
||||
message.SetQuestion("example.com.", mDNS.TypeA)
|
||||
opt := new(mDNS.OPT)
|
||||
opt.Hdr.Name = "."
|
||||
opt.Hdr.Rrtype = mDNS.TypeOPT
|
||||
opt.Option = append(opt.Option, &mDNS.EDNS0_PADDING{Padding: padding})
|
||||
message.Extra = append(message.Extra, opt)
|
||||
return message, padding[:markedQueryNeedle]
|
||||
}
|
||||
|
||||
// poolHoldsNeedle drains one size class of the buffer pool and reports whether
|
||||
// any buffer in it still carries the needle. It must run on the goroutine that
|
||||
// released the buffer: sync.Pool keeps a per-P private slot that no other P can
|
||||
// steal from, and on the paths this test covers the release happens inline in
|
||||
// RoundTripOpt, on the caller's own goroutine.
|
||||
func poolHoldsNeedle(size int, needle []byte, count int) bool {
|
||||
held := make([]*buf.Buffer, 0, count)
|
||||
defer func() { releaseAll(held) }()
|
||||
var found bool
|
||||
for range count {
|
||||
buffer := buf.NewSize(size)
|
||||
held = append(held, buffer)
|
||||
if bytes.Contains(buffer.FreeBytes(), needle) {
|
||||
found = true
|
||||
}
|
||||
}
|
||||
return found
|
||||
}
|
||||
|
||||
// requireInstrumentFindsPackedQuery is the CONTROL. It does exactly what the old
|
||||
// Exchange did — pack a query into a pooled buffer and release it — and demands
|
||||
// that the scan below FINDS the needle. Without it, "the pool does not hold the
|
||||
// query" would also be the verdict for a scan that can never find anything.
|
||||
//
|
||||
// Retried for the same reason its sibling control above is, and it was NOT
|
||||
// before: under -race sync.Pool.Put drops one object in four, so a single
|
||||
// attempt made this control — and with it the whole -race pass of the gate —
|
||||
// fail on 18 of 60 measured runs with nothing wrong in the tree. A fresh
|
||||
// needle is packed on each attempt, so a later one cannot be answered by an
|
||||
// earlier one's bytes. See controlAttempts for what the retry does and does not
|
||||
// buy.
|
||||
func requireInstrumentFindsPackedQuery(t *testing.T) {
|
||||
t.Helper()
|
||||
for range controlAttempts {
|
||||
message, needle := markedQuery(t)
|
||||
size := 1 + message.Len()
|
||||
exMessage := *message
|
||||
exMessage.Id = 0
|
||||
exMessage.Compress = true
|
||||
|
||||
buffer := buf.NewSize(size)
|
||||
if _, err := exMessage.PackBuffer(buffer.FreeBytes()); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
buffer.Release()
|
||||
|
||||
if poolHoldsNeedle(size, needle, poolScanDepth) {
|
||||
return
|
||||
}
|
||||
}
|
||||
t.Fatalf("control failed: %d times in a row, a query packed into a pooled buffer and released was NOT "+
|
||||
"found by the scan, so a clean verdict below would prove nothing. Under -race sync.Pool.Put drops "+
|
||||
"one object in four, which is what the retries absorb; this many consecutive misses is something "+
|
||||
"else — a pool that zeroes on Put, a size class that stopped being pooled, or buf.Buffer no longer "+
|
||||
"handing its array back at all", controlAttempts)
|
||||
}
|
||||
|
||||
// TestHTTP3ExchangeNeverPacksQueriesIntoPooledMemory pins the ownership rule the
|
||||
// failure paths depend on.
|
||||
//
|
||||
// quic-go's http3.Transport closes the request body on every error path
|
||||
// (transport.go RoundTripOpt) the moment doRequest returns, and doRequest waits
|
||||
// only on the request-cancellation watchdog — never on the goroutine writing the
|
||||
// body. So releasing the buffer when the body is closed is not an ownership
|
||||
// handoff, and the only safe arrangement is for the query never to live in pool
|
||||
// memory at all.
|
||||
//
|
||||
// This test encodes THAT design. A future guarded-pool design (a lock around
|
||||
// Read and Close, refusing reads after release) would also be correct and would
|
||||
// fail this test on purpose — it would have to replace it, and say so.
|
||||
func TestHTTP3ExchangeNeverPacksQueriesIntoPooledMemory(t *testing.T) {
|
||||
requireInstrumentFindsPackedQuery(t)
|
||||
|
||||
// A UDP socket nobody answers on: the handshake runs to the context deadline
|
||||
// instead of being refused, which is the shape a router sees when the tunnel
|
||||
// carrying its resolver drops.
|
||||
blackhole, err := net.ListenUDP("udp", &net.UDPAddr{IP: net.IPv4(127, 0, 0, 1)})
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
t.Cleanup(func() { blackhole.Close() })
|
||||
|
||||
for _, testCase := range []struct {
|
||||
name string
|
||||
ctx func(t *testing.T) (context.Context, context.CancelFunc)
|
||||
}{
|
||||
{
|
||||
// RoundTripOpt closes the body after the handshake gives up.
|
||||
name: "server never answers",
|
||||
ctx: func(t *testing.T) (context.Context, context.CancelFunc) {
|
||||
return context.WithTimeout(context.Background(), 500*time.Millisecond)
|
||||
},
|
||||
},
|
||||
{
|
||||
// The cancellation watchdog fires, then RoundTripOpt closes the body.
|
||||
name: "context already cancelled",
|
||||
ctx: func(t *testing.T) (context.Context, context.CancelFunc) {
|
||||
ctx, cancel := context.WithCancel(context.Background())
|
||||
cancel()
|
||||
return ctx, func() {}
|
||||
},
|
||||
},
|
||||
} {
|
||||
t.Run(testCase.name, func(t *testing.T) {
|
||||
message, needle := markedQuery(t)
|
||||
size := 1 + message.Len()
|
||||
|
||||
dialer := &trackingDialer{}
|
||||
t.Cleanup(dialer.closeAll)
|
||||
dnsTransport := &HTTP3Transport{
|
||||
TransportAdapter: dns.NewTransportAdapter(C.DNSTypeHTTP3, "test-doh3-ownership", nil),
|
||||
logger: logger.NOP(),
|
||||
dialer: dialer,
|
||||
destination: &url.URL{Scheme: "https", Host: "localhost", Path: "/dns-query"},
|
||||
headers: http.Header{},
|
||||
serverAddr: M.ParseSocksaddr(blackhole.LocalAddr().String()),
|
||||
tlsConfig: &tls.Config{
|
||||
InsecureSkipVerify: true,
|
||||
ServerName: "localhost",
|
||||
NextProtos: []string{http3.NextProtoH3},
|
||||
MinVersion: tls.VersionTLS13,
|
||||
},
|
||||
}
|
||||
dnsTransport.transport = dnsTransport.newTransport()
|
||||
t.Cleanup(func() { dnsTransport.Close() })
|
||||
|
||||
ctx, cancel := testCase.ctx(t)
|
||||
defer cancel()
|
||||
if _, err := dnsTransport.Exchange(ctx, message); err == nil {
|
||||
t.Fatal("expected the exchange to fail; this test is about the failure path")
|
||||
}
|
||||
|
||||
// Same goroutine that ran RoundTripOpt, so the per-P private slot a
|
||||
// release would have landed in is the one being drained.
|
||||
if poolHoldsNeedle(size, needle, poolScanDepth) {
|
||||
t.Fatal("the bytes of the query came back out of the buffer pool: the request body was packed into pooled " +
|
||||
"memory and released while quic-go's body writer could still be reading it")
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
@@ -12,6 +12,50 @@ as GitHub **pre-releases** and never become "Latest".
|
||||
|
||||
#### Unreleased (shater)
|
||||
|
||||
**`l3-honest-drop` — ICMP routed to an L4-only outbound is dropped, not
|
||||
forged** — ships with `shaterd` (part of the shater L3 ingress,
|
||||
`docs-shater/DECISIONS.md` D25), not as an lx release tag; recorded here because
|
||||
it edits two upstream files. Without it the TUN stack answers an unroutable echo
|
||||
ITSELF — sing-tun's `ICMPForwarder.HandlePacket` rewrites Echo→EchoReply
|
||||
whenever the flow judgment comes back Accept (`stack_gvisor_icmp.go`) — so a
|
||||
ping routed to vless/vmess/… would read as a working tunnel while the packet
|
||||
never left the router.
|
||||
|
||||
* **`route/route.go` (`PreMatch`)** — the pre-match walk was renamed to
|
||||
`preMatch` and the exported `PreMatch` became a thin FUNNEL that rewrites
|
||||
`PreMatchContinue` and `PreMatchBypass` to `PreMatchDrop` for
|
||||
`N.NetworkICMP`. An earlier version overrode `continueResult` inside
|
||||
`preMatchFlow` instead; that covered only the exits reaching that function and
|
||||
left three of the walk's own exits forging — the `prepareMatchMetadata` error
|
||||
return, the sniff bail-outs, and the `default:` arm of the rule-action switch
|
||||
(every action pre-match has no arm for: `hijack-dns`, `direct`, …). A guard on
|
||||
the single return value cannot be outgrown by a new exit. `PreMatchBypass` is
|
||||
folded in because sing-tun implements `ActionBypass` on the nfqueue plane only
|
||||
— on the TUN path it lands in the same `default:` arm as Accept, i.e. forges.
|
||||
* **`adapter/router.go` (`JudgeFlow`, the `!isPort` branch)** — ICMP returns
|
||||
`ActionDrop` where it fell through to `ActionAccept`. Second line of defense:
|
||||
`adapter.FlowOutbound` and `tun.Port` are distinct interfaces, and a drift
|
||||
between them must not quietly re-enable the forged reply.
|
||||
* **TCP/UDP behaviour is unchanged** — `PreMatchContinue` still means "take the
|
||||
ordinary connection route" for both, `PreMatchBypass` still means bypass, and
|
||||
the `!isPort` fallthrough still returns `ActionAccept` for them; pinned by
|
||||
`route/prematch_icmp_lx_test.go` and `adapter/judgeflow_icmp_lx_test.go`
|
||||
(both inside the marker), each ICMP case having an explicit TCP/UDP twin.
|
||||
* **NOT covered: a FRAGMENTED echo to a WireGuard/AWG outbound is still
|
||||
forged** — sing-tun's `ForwardDispatcher.Dispatch` returns before asking for a
|
||||
verdict at all when `parsed.fragment`, and the reassembled packet reaches
|
||||
`ICMPForwarder.HandlePacket`, whose `installFlow` demands an UNSPECIFIED port
|
||||
address that a WireGuard endpoint never has. Fixing it inside `JudgeFlow`
|
||||
is NOT possible — both consumers call it with identical arguments and the
|
||||
working path needs the concrete address. Full chain, the two viable fixes and
|
||||
the trap are in `docs-shater/DECISIONS.md` D25, under "What is still NOT
|
||||
covered, said plainly", item 2.
|
||||
* **Rebase cost: two small marked blocks** (`lx:begin/end l3-honest-drop`, a
|
||||
wrapper function in `route/route.go` and one branch body in
|
||||
`adapter/router.go`) plus the two self-contained test files — carried across
|
||||
an upstream rebase by eye. Note that `PreMatch`'s own body now lives in
|
||||
`preMatch`, so an upstream change to the walk applies to that function.
|
||||
|
||||
**Fork-layer + control-plane rework of proxy health** — ships with `shaterd`
|
||||
(the shater router daemon), not as an lx release tag; recorded here because the
|
||||
load-bearing half lives in fork zones (`common/urltest`, `protocol/group`).
|
||||
|
||||
@@ -62,16 +62,61 @@ flowchart LR
|
||||
C["LAN client"] -->|"nft tproxy, mark → tproxy port"| IN["sing-box tproxy inbound (sniff SNI/Host/QUIC)"]
|
||||
IN --> R{"route: rule match — src / dst / list / geo / client"}
|
||||
R -->|"proxied"| OUT["outbound / selector (balancer, chain)"]
|
||||
R -->|"direct"| DIR["direct (flow-offload on)"]
|
||||
R -->|"direct"| DIR["direct (out the normal route, untunnelled)"]
|
||||
R -->|"blocked"| BLK["block"]
|
||||
OUT --> NET["exit — VLESS/Reality/AmneziaWG2/Hysteria2/…"]
|
||||
```
|
||||
|
||||
Reliability (ported from v0.1): own nft table `inet shater` + own marks/tables
|
||||
(never touch fw4); atomic validate→stage→swap; commit-confirm rollback;
|
||||
idempotent reconcile under flock; management-bypass always; fail-closed
|
||||
(never touch fw4); atomic validate→stage→swap; commit-confirm rollback (opt-in —
|
||||
see §5); idempotent reconcile under flock; management-bypass always; fail-closed
|
||||
kill-switch (dead group → block, not a silent direct leak).
|
||||
|
||||
Only TCP and UDP reach that path — TPROXY carries nothing else. What happens to
|
||||
the rest is §3a.
|
||||
|
||||
### 3a. L3 ingress and kernel egress — what TPROXY cannot carry
|
||||
|
||||
Two opt-in globals cover the protocols the tproxy plane leaves on the floor.
|
||||
Both are off in a stock config, and both are configured through UCI only (the
|
||||
panel does not expose them).
|
||||
|
||||
**`globals.l3_tunnel` — LAN ICMP through the tunnel.** The generator adds a
|
||||
synthetic `tun` inbound tagged `l3-in` (gVisor stack, `auto_route` **off**, MTU
|
||||
65535, `shater/generate/inbound.go`), so ICMP is routed by the engine's own rules
|
||||
instead of being dropped or answered by a forged local reply. The device is not
|
||||
one fixed name: the generator emits a stable placeholder (so a no-op reconcile
|
||||
still hashes identical and does not rebuild the engine once a minute), and
|
||||
`shater/engine/l3slot.go` substitutes one of the two slots `shater-l3a` /
|
||||
`shater-l3b` (`netplane/l3.go`) just before `box.New` — a new generation must
|
||||
never reopen the name the outgoing one still holds
|
||||
(`TUNSETIFF: device or resource busy` took the whole LAN down once). The routing half is scoped and lives entirely outside
|
||||
the main table: our nft prerouting chain stamps LAN `icmp`/`ipv6-icmp` with
|
||||
`L3Mark` (`fwmark_base + 0x80`), and `netplane.addL3Routing` binds that mark to
|
||||
`L3Table` (`table_base + 8`), whose only content is a default route out the live
|
||||
slot. Because the daemon creates the device at runtime, netifd never learns about
|
||||
it and fw4 would reject the forward on its own account — so `30_shater-core`
|
||||
seeds a **`shater_l3` zone in the user's `/etc/config/firewall`**, matching
|
||||
`list device 'shater-l3*'` (a string match that is valid before the TUN exists
|
||||
and covers both slots). Ceiling: ICMP echo only, and only for L3-capable
|
||||
egresses; see `DECISIONS.md` D25 for what is still not covered.
|
||||
|
||||
**`globals.untunnelable_egress` — everything else, carried by the kernel.** It
|
||||
names an existing interface/tunnel egress. Whatever the L3 block above did not
|
||||
claim — ESP/AH, GRE, IGMP, SCTP, and ICMP too when `l3_tunnel` is off — is
|
||||
stamped in prerouting with **that egress's own mark** (`netplane/nft.go`,
|
||||
`UntunnelableEgressBinding`) and accepted; the `fwmark → table` pair
|
||||
`addEgressRouting` already installed for the egress then routes it out the
|
||||
egress's device. No new mark, no new table, and the engine never sees a byte —
|
||||
which is why any IP protocol works here while the L3 TUN is narrow. Order is
|
||||
load-bearing: this sweep runs **after** the L3 marking (first match wins) and
|
||||
**after** the local-plane accepts, so LAN-to-LAN, router-addressed traffic and
|
||||
IPv6 neighbour discovery never leave through an uplink. With `ipv6=0` the mark
|
||||
is scoped to `nfproto ipv4`, because `addEgressRouting` installs the `-6`
|
||||
rule/table pair only when IPv6 is on and marked v6 without it would fall through
|
||||
to the main table past the kill-switch. `globals.untunnelable` (block | icmp |
|
||||
direct) stays in charge of whatever neither mechanism carries.
|
||||
|
||||
## 4. DNS + filtering + stats
|
||||
|
||||
```mermaid
|
||||
@@ -79,7 +124,7 @@ flowchart LR
|
||||
C["client :53"] -->|"hijack"| DNS["sing-box DNS (in-process)"]
|
||||
DNS --> FILT{"shater filter: blocklists + allowlist + per-device policy"}
|
||||
FILT -->|"blocked"| NX["NXDOMAIN / 0.0.0.0"]
|
||||
FILT -->|"allowed"| RES["resolvers (DoH/DoT/plain/FakeIP) + nftset for routing"]
|
||||
FILT -->|"allowed"| RES["resolvers (DoH/DoT/plain/local/FakeIP), per-rule detour"]
|
||||
DNS -->|"query events (engine observability)"| AGG["shater stats aggregator"]
|
||||
AGG --> PANEL["panel: top domains · per-device · allowed/blocked · timeline"]
|
||||
```
|
||||
@@ -87,9 +132,10 @@ flowchart LR
|
||||
Because the engine's DNS runs **in our process**, every query (domain, client,
|
||||
verdict, latency) is available to the stats aggregator without log-scraping —
|
||||
this is the payoff of embedding. Blocklist matching uses an efficient compiled
|
||||
matcher, not dnsmasq megalists (see `DECISIONS.md` D5). Per-device blocking =
|
||||
engine route/DNS rule keyed by client, or nftset(device) × nftset(blocked-domain)
|
||||
→ drop.
|
||||
matcher, not dnsmasq megalists (see `DECISIONS.md` D5). Per-device blocking is an
|
||||
engine route/DNS rule keyed by client. Routing decisions come from in-engine
|
||||
rule-sets: the v0.1 mechanism where dnsmasq populated nft sets does not exist in
|
||||
v0.2 (`generate/dns.go`).
|
||||
|
||||
## 5. Config & apply flow
|
||||
|
||||
@@ -100,11 +146,20 @@ stateDiagram-v2
|
||||
Render --> Validate: engine config check + nft -c
|
||||
Validate --> KeepOld: fail
|
||||
Validate --> Apply: ok (atomic swap: engine reload + nft/route reconcile)
|
||||
Apply --> ConfirmWindow
|
||||
Apply --> Committed: confirm_timeout = 0 (SHIPPED DEFAULT — nothing armed)
|
||||
Apply --> ConfirmWindow: confirm_timeout > 0
|
||||
ConfirmWindow --> Committed: confirmed
|
||||
ConfirmWindow --> Rollback: timeout
|
||||
Rollback --> LastGood
|
||||
```
|
||||
|
||||
**The confirm window is opt-in and ships closed.** `model.DefaultGlobals()` leaves
|
||||
`ConfirmTimeout` at zero, the shipped `/etc/config/shater` says
|
||||
`option confirm_timeout '0'`, and `apply.ArmRollback` returns immediately on a
|
||||
non-positive timeout — so on a stock install every apply takes the left edge above
|
||||
and there is no net under it. `shaterd apply` reports which edge it took
|
||||
(`reason: commit-confirm-off` vs an armed window). Set
|
||||
`globals.confirm_timeout` to arm it.
|
||||
|
||||
## 6. Roadmap tiers
|
||||
See `ROADMAP.md` for the phased plan and `FEATURES.md` for the full feature list.
|
||||
|
||||
+29
-16
@@ -51,6 +51,9 @@ We are rebasing onto a new engine and a new UI architecture. Full rationale in
|
||||
MASQUE/WARP, and gRPC observability (DNS queries / rules / outbounds). Upstream
|
||||
sing-box brings VLESS/VMess/Trojan/Shadowsocks/WireGuard/Reality + Hysteria2/
|
||||
TUIC. It is library-first (`libbox`) and **GPL-3.0** (compatible with us).
|
||||
That list is what the FORK can build, not what shater ships: `shater/registry`
|
||||
registers only what `shater/generate` can emit, and MASQUE is one of the types
|
||||
deliberately left out (~6 MB of binary and resident RAM). See `FEATURES.md`.
|
||||
- We **fork it** (not just depend on it) so we can embed literally everything —
|
||||
control-plane, admin panel, DNS filter — and integrate tightly with the
|
||||
engine internals (DNS, routing, stats). This is a deliberate, decided
|
||||
@@ -84,13 +87,13 @@ We are rebasing onto a new engine and a new UI architecture. Full rationale in
|
||||
|
||||
## Repository model
|
||||
|
||||
- **`shater` `main` = our fork of sing-box-lx.** After Phase 1 it contains the
|
||||
full sing-box-lx tree PLUS our additive overlay (`shater/`, `panel/`,
|
||||
`openwrt/`, `docs-shater/`). Upstream is tracked via a git remote and merged by tag.
|
||||
- **`shater` `main` = our fork of sing-box-lx.** It contains the full sing-box-lx
|
||||
tree PLUS our additive overlay (`shater/`, `panel/`, `openwrt/`, `docs-shater/`,
|
||||
`scripts/`, `ci/`). Upstream is tracked via a git remote and merged by tag.
|
||||
Phase 1 merged the engine in on 2026-07-14 (`v1.14.0-lx.3`); `main` has not been
|
||||
a docs-only seed since.
|
||||
- **`shater` branch `v0.1`** = the standalone xray-based version (frozen, ported
|
||||
from).
|
||||
- Until Phase 1 merges the engine in, `main` is the docs-first overlay seed you
|
||||
are reading now (LICENSE, README, `docs-shater/`, the feed signing key).
|
||||
|
||||
## What to port from v0.1 (don't rewrite these ideas)
|
||||
|
||||
@@ -119,20 +122,30 @@ filter/stats engine wired into sing-box's DNS.
|
||||
v0.2 fork; branch `v0.1` = the working xray-based version.
|
||||
- **Upstream to track:** `https://github.com/Leadaxe/sing-box-lx` (which tracks
|
||||
`https://github.com/SagerNet/sing-box`).
|
||||
- **CI:** Gitea Actions (act_runner + Docker). v0.1's workflow was removed from
|
||||
`main`; new CI is added when the v0.2 build exists.
|
||||
- **CI:** Gitea Actions (act_runner + Docker), `.gitea/workflows/release.yml` —
|
||||
builds the four packages through the ImmortalWrt 25.12.1 SDK and publishes the
|
||||
signed per-arch apk repo. The opkg/`.ipk` lane was deleted, not disabled (D22).
|
||||
- **Test gate:** `bash scripts/run-tests.sh` — the whole suite under the SHIPPED
|
||||
build tags, on linux (in Docker from a non-linux host), with `-race`, and with
|
||||
three anti-silent-skip checks. Not optional reading before touching `shater/`.
|
||||
- **Feed signing:** EC (prime256v1) key for the apk index; secret in the repo
|
||||
secret `KEY_APK`; public key `dist/shater-apk.pem`, installed on routers as
|
||||
`/etc/apk/keys/shater-apk.pem`. Never regenerate it (D22).
|
||||
- **Test VM:** OpenWrt 24.10.3 x86_64 in Docker (`docker ps --filter
|
||||
name=openwrt-vm`). SSH via the ssh-manager MCP server `local_openwrt`
|
||||
(localhost:2222, root/openwrt). LuCI at `http://127.0.0.1:8080` (root/openwrt),
|
||||
drivable with the Playwright MCP.
|
||||
- **Test VM:** **ImmortalWrt 25.12.1** (`r37978-cd0a06bfd3fd`) x86_64 in Docker
|
||||
(`docker ps --filter name=openwrt-vm`), apk-tools 3.0.5 — deliberately the same
|
||||
revision as `mini_router`, and required: the only package format we publish is
|
||||
`.apk`, which does not install on 24.10 at all. SSH via the ssh-manager MCP
|
||||
server `local_openwrt` (localhost:2222, root/openwrt). LuCI at
|
||||
`http://127.0.0.1:8080` (root/openwrt), drivable with the Playwright MCP.
|
||||
- **Routers:** `mini_router` (BPi-R3 Mini, ImmortalWrt 25.12.1) carries the real
|
||||
home traffic; `main_router` (BPi-R4, OpenWrt 25.12.0). Both `aarch64_cortex-a53`,
|
||||
both apk-tools 3.0.5 — see the table in D22.
|
||||
|
||||
## Current status
|
||||
|
||||
Repo reset done: v0.1 preserved on its branch; `main` cleaned to this docs-first
|
||||
scaffold. Next is Phase 1 in `ROADMAP.md` — fork sing-box-lx into `main`
|
||||
(add upstream remote, merge a pinned tag), stand up the embedding prototype
|
||||
(prove AmneziaWG 2.0, measure binary size with feature-trim + `-s -w` + UPX)
|
||||
before building the control plane and panel.
|
||||
**v0.2 is feature-complete and running on real hardware.** ROADMAP Phases 0–8 are
|
||||
done and VM-verified; the product ships as a signed apk feed and is installed on
|
||||
`mini_router`. Read `ROADMAP.md` for what each phase delivered, `FEATURES.md` for
|
||||
the honest MVP/T1/T2 state of each feature (including what is declared but not
|
||||
shipped), and `DECISIONS.md` for why. Work since Phase 8 has been correctness and
|
||||
honesty passes rather than new phases.
|
||||
|
||||
@@ -319,6 +319,10 @@ Three values, not two, because the leaks differ in *kind*: an ICMP echo is ephem
|
||||
user-initiated and reveals the address only to a host the user deliberately contacted,
|
||||
whereas ESP/GRE is a standing second tunnel carrying arbitrary traffic beside ours. A
|
||||
single toggle would make "I want ping to work" mean "I allow a parallel VPN bypass".
|
||||
*(Refined 2026-07-26 by D25: still true of TPROXY — but ICMP echo now has an
|
||||
opt-in data plane of its own, the dedicated L3 TUN, so the policy no longer
|
||||
speaks alone for ping; it keeps sole charge of ESP/GRE/IGMP and of the degraded
|
||||
paths.)*
|
||||
|
||||
**Fail-open degradations must be visible in the panel, not only in `logread`.** The
|
||||
audit deliberately converted many aborts into warn-and-continue (an unfetchable list,
|
||||
@@ -772,3 +776,645 @@ server, which restores exactly the pre-D24 behaviour and clears the notice. Unti
|
||||
that lands, an operator can get the same result by setting `endpoint_resolver` to a
|
||||
direct resolver. Note the hazard is **not** created by D24 — any config with two
|
||||
resolvers has it today; the default merely makes it universal.
|
||||
|
||||
## D25 — L3 ingress: LAN ICMP rides a dedicated TUN through the tunnel, not a policy verdict
|
||||
Decided 2026-07-26. D17 made everything TPROXY cannot divert an explicit policy
|
||||
(`Globals.Untunnelable` = block | icmp | direct) — and its premise still holds:
|
||||
kernel TPROXY delivers a packet by handing it to a listening SOCKET, and sockets
|
||||
exist for TCP and UDP only, so an ICMP echo has nothing to be handed to. But a
|
||||
policy can only choose between losing the packet and leaking it with the
|
||||
client's real source address; neither ever puts a ping THROUGH the tunnel. This
|
||||
decision adds the data plane D17 could not have: **`globals.l3_tunnel` (opt-in,
|
||||
default off; `model.Globals.L3Tunnel`) opens a second, dedicated ingress — a TUN
|
||||
device — and LAN ICMP enters the engine as raw IP packets**, where the ordinary
|
||||
route rules pick an outbound exactly as for any flow. The policy is refined, not
|
||||
repealed: it keeps sole charge of the protocols the engine cannot ingest at all,
|
||||
and of the degraded paths (both below).
|
||||
|
||||
**The whole mechanism is one mark, one rule, one device — the TPROXY plane is
|
||||
untouched.** The nft prerouting chain stamps `L3Mark` (= fwmark_base + 0x80,
|
||||
`netplane/nft.go` `l3MarkOffset`) on LAN `ip protocol icmp` / `meta l4proto
|
||||
ipv6-icmp` ONLY, and only after every local plane was already accepted
|
||||
(fib-local, RFC1918/link-local/multicast daddr sets) and — for v6 — after a
|
||||
unicast ND/NA carve-out, because one tunnelled neighbour probe is enough to take
|
||||
the LAN's v6 plane down (`renderNft`, the L3 block). `addL3Routing`
|
||||
(`netplane/apply.go`) binds that mark to a table (= table_base + 0x08) whose
|
||||
only content is `default dev shater-l3`; del-then-add idempotent, and a failed
|
||||
rule or route is a NAMED operator warning, never an apply abort. `generate`
|
||||
emits the synthetic `l3-in` TUN inbound bound to exactly `netplane.L3Device`,
|
||||
MTU 65535 (the largest IP datagram there can be, so the KERNEL can never
|
||||
fragment on the way in — see "the device MTU is not a tunnel budget" below),
|
||||
point-to-point /30 + /126 addresses from private space,
|
||||
the v6 one only when `globals.ipv6` is on — and only next to a tproxy inbound:
|
||||
the ingress rides the same LAN divert plane, and without one the TUN would sit
|
||||
dark while the config claims ICMP is tunnelled, so it is skipped with a warning
|
||||
(`generate/inbound.go`, `appendL3TunInbound`). `shater/registry` registers the
|
||||
`tun` inbound type; that costs no new build tag and no meaningful size because
|
||||
`with_wireguard` already requires `with_gvisor` (D23, `scripts/router-tags.sh`).
|
||||
|
||||
**`auto_route: false` is load-bearing, not a default we happened to keep.**
|
||||
sing-box's auto_route rewrites the router's MAIN routing table — it would drag
|
||||
everything the router itself sends (WAN traffic, DNS, the tunnel's own underlay)
|
||||
into this TUN. The fwmark rule + dedicated table above is deliberately the ONLY
|
||||
entrance, and disabling the feature can never strand a stale default route in
|
||||
main (`generate/inbound.go`; pinned by `TestL3TunnelEmitsTunInbound`).
|
||||
|
||||
**`stack: "gvisor"` is a deliberate choice, and the tempting reason for it is
|
||||
wrong.** It is TRUE that sing-tun's system stack answers an ICMP echo LOCALLY —
|
||||
`processIPv4ICMP` rewrites Echo→EchoReply in place and swaps the addresses
|
||||
(sing-tun `stack_system.go:648`; the v6 twin sits right under it). It is FALSE
|
||||
that this makes the system stack unusable here: `dispatchIPv4`
|
||||
(`stack_system.go:355-372`) hands the packet to the SAME `ForwardDispatcher`
|
||||
first and only falls through to that forger for packets addressed to the TUN
|
||||
itself, exactly as the gVisor filter does (`stack_gvisor_filter.go:52-113`).
|
||||
Both stacks would forward. gvisor is chosen because it is already linked —
|
||||
`with_wireguard` requires `with_gvisor` (D23), so it costs no tag and no new
|
||||
code path — and because it is the combination the integration test actually
|
||||
exercises. Do not re-derive this as "the system stack fakes ping": it fakes ping
|
||||
only where the dispatcher declined the packet.
|
||||
|
||||
**The ceiling is ICMP echo, and it is upstream's dispatcher — NOT the netstack.**
|
||||
This distinction matters because the netstack answer is the intuitive one and it
|
||||
is wrong. On the forward path a WireGuard/AWG endpoint never consults gVisor at
|
||||
all: `Endpoint.WritePackets` (`transport/wireguard/port.go:21-58`) reads the IP
|
||||
version and the destination address and hands the raw bytes to
|
||||
`wgDevice.InputPackets` — the protocol byte is never examined — and
|
||||
`returnDeviceWrapper.Write` (`:127-157`) offers every decrypted packet to
|
||||
`returnPath.ReturnPackets` before the stack sees it. WireGuard would carry ESP
|
||||
today if anything handed it one. What refuses is `ForwardDispatcher`: its parser
|
||||
sets `hasFlow` for TCP, UDP and ICMP echo alone (`flow_parse.go`,
|
||||
`parseTransport`, the echo identifier serving as the pseudo-port), and
|
||||
`createFlow` NATs through a port-shaped selector (`flow_dispatch.go:325`,
|
||||
`allocateSelector`) that ESP, AH and GRE do not have. So ESP/AH/GRE/IGMP/SCTP
|
||||
cannot enter the engine in ANY configuration and REMAIN on the D17 policy —
|
||||
or on the kernel egress of D26, which sidesteps the dispatcher entirely. The nft
|
||||
plane encodes the same boundary on purpose: it marks `icmp`/`ipv6-icmp` only,
|
||||
never `l4proto != { tcp, udp }`, because a marked ESP packet would enter the
|
||||
device and vanish — a black hole wearing a tunnel's name — instead of receiving
|
||||
the policy's honest verdict (`netplane/nft.go`, the prerouting L3 comment).
|
||||
|
||||
**What works and what does not, read off the upstream source.** ping v4/v6 —
|
||||
yes. Windows `tracert` — yes: the gVisor return path recognises
|
||||
`ICMPv4TimeExceeded` and `ICMPv4DstUnreachable` alongside EchoReply and NATs
|
||||
them back to the LAN client (`stack_gvisor_icmp.go:341+`, `returnPacket`). IPv6
|
||||
traceroute — intermediate hops stay invisible: the v6 branch of the same
|
||||
function accepts EchoReply only, so just the final destination answers. Several
|
||||
LAN clients behind the one tunnel address are already solved upstream:
|
||||
`ForwardDispatcher` NATs by echo identifier and rewrites the source to the
|
||||
outbound's port address (`flow_dispatch.go:325+`, `createFlow`; `icmpFlowKey`) —
|
||||
we wrote no NAT of our own.
|
||||
|
||||
**Which outbounds can carry it.** The contract is `adapter.FlowOutbound`
|
||||
(= `Outbound` + `tun.Port` + `PreMatchFlow`, `adapter/outbound.go`). In-tree
|
||||
implementors: the WireGuard/AWG endpoint (`protocol/wireguard`), `direct`
|
||||
(`protocol/direct`), `bridge` (`protocol/bridge`), `tailscale`
|
||||
(`protocol/tailscale`). Of those, the shaterd registry can construct only
|
||||
WireGuard/AWG and direct (`shater/registry/registry.go` — bridge and tailscale
|
||||
are not registered). Every proxy protocol — vless/vmess/trojan/shadowsocks/
|
||||
hysteria2/tuic/socks/http/shadowtls — is L4-only and cannot. Recorded as a known
|
||||
gap: `masque` is L3 by nature (CONNECT-IP; it builds a userspace gVisor stack
|
||||
per tunnel, `protocol/masque/outbound.go`) but implements no `tun.Port` and is
|
||||
not in the shater registry, so today it cannot carry the ingress. Wiring it up
|
||||
is possible future work, not a promise.
|
||||
|
||||
**ICMP to an L4-only outbound is DROPPED, and that took patching upstream files
|
||||
(the `lx:l3-honest-drop` delta — see `docs-lx/lx-changelog.md`).** In the gVisor
|
||||
stack the fallthrough verdict is a forgery: `ICMPForwarder.HandlePacket` answers
|
||||
the echo ITSELF (Echo→EchoReply + address swap) whenever the flow judgment comes
|
||||
back Accept (`stack_gvisor_icmp.go:120`), and upstream maps "no flow route" to
|
||||
exactly that Accept — so a ping routed to vless would read as tunnelled while
|
||||
the packet died on the router. Two small marked hunks make the truth observable:
|
||||
`route/route.go` wraps the whole pre-match walk — the walk itself became
|
||||
`preMatch`, and the exported `PreMatch` is now a FUNNEL that rewrites
|
||||
`PreMatchContinue` and `PreMatchBypass` to `PreMatchDrop` for `N.NetworkICMP` —
|
||||
and `adapter/router.go` (`JudgeFlow`, the `!isPort` branch) returns `ActionDrop`
|
||||
for ICMP where it fell through to `ActionAccept` — the second line of defense,
|
||||
because `FlowOutbound` and `tun.Port` are distinct interfaces and a drift
|
||||
between them must not quietly re-enable the forger. TCP/UDP verdicts are
|
||||
byte-identical; `route/prematch_icmp_lx_test.go` and
|
||||
`adapter/judgeflow_icmp_lx_test.go` pin both directions. The operator-facing
|
||||
text says the same out loud (`shater/apply/warnings.go`): proxy-routed addresses
|
||||
"cannot be pinged at all — deliberately".
|
||||
|
||||
> **Why a funnel and not an override inside the walk.** The first version of
|
||||
> this delta overrode the pre-declared `continueResult` inside `preMatchFlow`
|
||||
> and claimed to cover "every exit point of the function at once". It covered
|
||||
> every exit of THAT function; the walk above it has exits of its own that never
|
||||
> reach it — the `prepareMatchMetadata` error return (which arrived later, with
|
||||
> the shared-metadata refactor, upstream `b911fb078`), the sniff bail-outs, and
|
||||
> the `default:` arm of the rule-action switch, which catches every action
|
||||
> pre-match has no arm for (`hijack-dns`, `direct`, and whatever upstream adds
|
||||
> next). Each of those returned `PreMatchContinue`, i.e. `tun.ActionAccept`,
|
||||
> i.e. the forged reply. A guard on the single return value cannot be outgrown
|
||||
> by a new exit. `PreMatchBypass` joined the drop for the same reason: sing-tun
|
||||
> implements `ActionBypass` on the nfqueue plane only — the name appears nowhere
|
||||
> in `flow_dispatch.go` or `stack_gvisor_icmp.go` — so on the TUN path it lands
|
||||
> in the same `default:` arm as Accept and forges too. There is no honest bypass
|
||||
> for a packet that is already inside the engine's TUN.
|
||||
|
||||
**The device MTU is NOT a tunnel budget, and pretending it was manufactured
|
||||
forged replies.** `l3-in` is created with MTU **65535**, not the tunnel's 1420,
|
||||
and the maximum is the whole argument. This MTU governs exactly one thing:
|
||||
whether the KERNEL splits a packet on its way INTO the device. What the engine
|
||||
then puts into the tunnel is sized separately and correctly, against the
|
||||
OUTBOUND's MTU — `ForwardDispatcher.forwardToPort` (`flow_dispatch.go:445-481`)
|
||||
measures every forwarded packet against `Port.PortMTU()` and either fragments to
|
||||
it (no DF, `fragmentIPv4Packet`) or answers a well-formed `fragmentation needed`
|
||||
quoting it (DF, `buildFragmentationNeeded`, source = the far host, so PMTU
|
||||
discovery works end to end). That machinery was always there; it was simply
|
||||
never handed a whole packet.
|
||||
|
||||
At 1420 it wasn't. Anything above 1392 bytes of payload was fragmented by the
|
||||
kernel at this device, and a fragment is the one thing sing-tun will not judge:
|
||||
`Dispatch` (`flow_dispatch.go:176-177`) returns on `parsed.fragment` BEFORE
|
||||
calling `JudgeFlow` at all. The fragments fell through to the gVisor stack —
|
||||
promiscuous and spoofing (`stack_gvisor.go:219-223`) — which reassembled them
|
||||
and handed the echo to `ICMPForwarder.HandlePacket` (`stack_gvisor_icmp.go:105+`),
|
||||
whose `installFlow` (`:233-244`) writes to the port UNMODIFIED and therefore
|
||||
demands a port address that is valid **and UNSPECIFIED**. `direct` qualifies
|
||||
(`IPv4Unspecified()`); a WireGuard/AWG endpoint reports its concrete interface
|
||||
address (`transport/wireguard/port.go:13`) and does not. So it declined, and
|
||||
`HandlePacket` fell past the switch and FORGED the reply: `SetType(EchoReply)` +
|
||||
address swap. Net effect on the operator's bench: `ping -s 1392` honest,
|
||||
`ping -s 1393` a lie told by the router — and the lie was, of course, only for
|
||||
the outbounds this feature exists for. (Upstream applies the very same
|
||||
unspecified test and answers it honestly in the cloudflared ICMP handler,
|
||||
`protocol/cloudflare/inbound.go:163-167`: it drops. Only the TUN path forges.)
|
||||
|
||||
65535 rather than "big enough": no IP datagram can exceed it, so the kernel
|
||||
CANNOT fragment at this device, for any packet, ever. Any smaller value leaves
|
||||
a band of sizes open and re-opens the class. It is also sing-box's own default
|
||||
TUN MTU on Linux. Pinned by `TestL3TunnelMTULeavesNothingForTheKernelToFragment`
|
||||
and `TestL3TunnelMTUIsNotATunnelBudget` (`generate/l3mtu_test.go`), and — the
|
||||
assertion that matters — by the integration test reading the MTU back off the
|
||||
real kernel device, since a kernel that clamped it would restore the forgery
|
||||
without changing a generated byte.
|
||||
|
||||
Memory was MEASURED, not reasoned about: three paired runs of
|
||||
`TestIntegrationL3TunInboundStarts` under `-test.memprofilerate=1` (exact
|
||||
accounting, not sampled) allocate 5.41 / 5.48 / 5.47 MB at 65535 against
|
||||
5.76 / 5.46 / 5.70 MB at 1420, and a `-diff_base` profile attributes every
|
||||
difference to netlink interface enumeration. Nothing in the read path scales
|
||||
with the MTU: gVisor reads through `fdbased.BufConfig`, which sing-tun's `init`
|
||||
pins to a single 65535-byte view regardless of MTU, and `fdbased` keeps `mtu`
|
||||
only to return it from `MTU()`. Two adjacent facts, recorded because both are
|
||||
easy to derive wrongly: (a) `protocol/tun` computes
|
||||
`enableGSO = stack == gvisor && mtu < 49152`, so this MTU turns GSO off there —
|
||||
and then `StartStateStart` turns it back ON unconditionally because an
|
||||
`adapter.FlowOutbound` exists in the config, so the ~1.98 MB of TCP/UDP GRO
|
||||
scaffolding is present at BOTH MTUs and is priced by the flow-capable outbound,
|
||||
not by this number; (b) the `mtu_fix` on the `shater_l3` fw4 zone is now inert —
|
||||
only ICMP is ever marked into the device — and its uci-defaults comment still
|
||||
says "the tunnel MTU is 1420".
|
||||
|
||||
**What is still NOT covered, said plainly.**
|
||||
|
||||
1. **A big ping does not start WORKING — it starts FAILING HONESTLY.** Upstream's
|
||||
ICMP NAT is unfragmented-only in BOTH directions: `classifyReturn`
|
||||
(`flow_dispatch.go:703-710`) returns `returnPass` on `parsed.fragment` exactly
|
||||
as the forward path does. So a non-DF `ping -s 2000` now genuinely leaves the
|
||||
router (fragmented to the tunnel MTU by `forwardToPort`), the far host really
|
||||
answers, and the reply — fragmented by the peer to fit the tunnel — is not
|
||||
NAT'd back to the LAN client. The operator sees a timeout. That is the
|
||||
feature's promise ("travels or fails honestly"), not a capability claim.
|
||||
Carrying oversized ICMP end to end would need reassembly upstream does not
|
||||
have; it is not planned.
|
||||
2. **A client that puts fragments on the wire ITSELF.** The device MTU cannot
|
||||
un-fragment what already arrived fragmented, so such packets still reach the
|
||||
gVisor stack, still get reassembled there, and still receive a forged reply
|
||||
when the outbound is WireGuard/AWG. This is the residue the planned
|
||||
`ip frag-off & 0x3fff != 0` prerouting carve-out (`netplane/nft.go`) is for.
|
||||
**Whoever writes that rule must first check whether it can ever match:** fw4's
|
||||
ruleset uses conntrack, conntrack pulls in `nf_defrag_ipv4`/`nf_defrag_ipv6`,
|
||||
and defrag REASSEMBLES in PREROUTING before our marking rules run. Where
|
||||
defrag is active the case does not arise (the MTU covers it) and the rule is
|
||||
dead; where it is not, the rule is the only cover. Verify on the bench with
|
||||
`nft list ruleset | grep -c ct` and a fragment counter, do not assume.
|
||||
3. **The DF path changed hands and is untested on hardware.** It used to be the
|
||||
kernel that answered `fragmentation needed` (from the router's LAN address,
|
||||
MTU 1420); it is now the engine (from the far host's address, quoting
|
||||
`Port.PortMTU()`). Both are correct PMTUD; only the first has ever run on a
|
||||
real router.
|
||||
**fw4 has to be told about the device, and `list device` is the only spelling
|
||||
that works.** nftables runs EVERY table on every packet and a drop in any one of
|
||||
them wins — an accept in `inet shater` cannot override fw4, and fw4 WILL reject
|
||||
this forward: netifd never learns about a device the daemon creates at runtime,
|
||||
so `shater-l3` belongs to no zone and falls into fw4's zone-less defaults. Hence
|
||||
a real fw4 zone `shater_l3` + a lan→shater_l3 forwarding, seeded idempotently
|
||||
(NAMED sections) and unconditionally in uci-defaults
|
||||
(`openwrt/shater-core/files/etc/uci-defaults/30_shater-core`, `seed_l3_zone`),
|
||||
with `mtu_fix` set. That `mtu_fix` is now inert and should be read as such: it
|
||||
clamps forwarded TCP MSS to the route MTU, the device MTU is 65535, and nothing
|
||||
but ICMP is ever marked into this device — the uci-defaults comment still says
|
||||
"the tunnel MTU is 1420" and is stale. The device is attached
|
||||
via `list device`, deliberately NOT `list network`: fw4 resolves a zone's
|
||||
networks through netifd, which yields an EMPTY device set for a runtime-created
|
||||
TUN (a proto-none stub would have to be brought UP to contribute an l3_device,
|
||||
and nothing ever brings it up), while `list device` compiles to a plain
|
||||
iifname/oifname string match — valid before the TUN exists, matching from the
|
||||
moment shaterd creates it. `kmod-tun` joined DEPENDS so a slimmed image cannot
|
||||
lose `/dev/net/tun` (`openwrt/shater-core/Makefile`). Our own forward chain
|
||||
accepts both TUN legs ahead of the fail-closed drops — accepts that speak for
|
||||
OUR table only (`netplane/nft.go`, forward chain step 4).
|
||||
|
||||
**What the policy still owns, and the one combination that now warns.** With the
|
||||
ingress on, the mark is stamped in prerouting and the ROUTING decision carries
|
||||
echo into the TUN before the forward chain — where the policy's verdicts live —
|
||||
is ever consulted; that holds under every `untunnelable` value. The policy
|
||||
therefore governs exactly two things: the never-markable protocols above, and
|
||||
the fallback when the L3 rule/route did not come up (engine down, partial apply)
|
||||
— `block` turns that failure into an honest loss, `direct` into a silent leak
|
||||
with the real address. That is why `l3_tunnel` + `untunnelable=direct` draws a
|
||||
validation warning naming the safe choice (`model/validate.go`), why every
|
||||
rule/route failure surfaces as a named panel warning rather than an abort
|
||||
(`addL3Routing`), and why the D17 HOLDING plane never marks: the TUN is created
|
||||
BY the engine, and the holding plane exists precisely because the engine is not
|
||||
running — marking would dead-end ping in a device that does not exist
|
||||
(`netplane/nft.go`, hold comment).
|
||||
|
||||
- **Rejected: `auto_route` / letting the engine own the routing.** It rewrites
|
||||
the main table and intercepts the router's own WAN/DNS/underlay traffic; the
|
||||
blast radius of a toggle meant for LAN ping would be the whole router.
|
||||
- **Rejected: marking all `l4proto != { tcp, udp }` into the TUN.** ESP/AH/GRE/
|
||||
IGMP/SCTP cannot be parsed into flows upstream; they would vanish inside the
|
||||
device. A drop with a name (the policy's) beats a silent black hole.
|
||||
- **Rejected: keeping upstream's accept-and-forge for unroutable ICMP.** A ping
|
||||
that "works" without leaving the router is the inverted lie this project keeps
|
||||
deleting (D17's fiction purge, D23's dead WireGuard, D24's obedient-client
|
||||
leak).
|
||||
|
||||
**Proven, and not proven, said plainly.** The cold start is PROVEN, not assumed:
|
||||
`TestIntegrationL3TunInboundStarts`
|
||||
(`shater/generate/l3_integration_linux_test.go`, run as root with NET_ADMIN and
|
||||
`/dev/net/tun`, PASS) drives an `l3_tunnel=1` config through the SLIM registry
|
||||
(`registry.Context`, not upstream's `include.Context`) under the shipped router
|
||||
tag set: `box.New` + `Start` accept it, the kernel really ends up with the
|
||||
`shater-l3` device at the contract MTU 65535 — the assertion the value exists
|
||||
for, since a kernel that clamped it would silently restore the forged-reply
|
||||
band — and Close removes it; precisely
|
||||
the "built with X, verified with Y" gap class D23 exists for (a lost
|
||||
`tun.RegisterInbound` or a trimmed `with_gvisor` changes no generated byte and
|
||||
would otherwise surface only on the operator's router). Each layer contract is
|
||||
pinned besides (`generate` `TestL3Tunnel*`, `netplane` `TestL3Ingress*`,
|
||||
`route/prematch_icmp_lx_test.go`). Exactly two things remain UNVERIFIED:
|
||||
(a) the end-to-end path on live hardware — LAN client → prerouting mark →
|
||||
ip rule → TUN → WireGuard peer → reply back to the client — has not been
|
||||
exercised on a real router; (b) the steady-state memory cost of the second
|
||||
gVisor netstack (the `l3-in` TUN beside the WireGuard endpoint's own) is
|
||||
unmeasured on the target hardware. An indicative figure exists and is only
|
||||
that: on x86_64 in a container, idle and carrying no flows, peak RSS of a
|
||||
process that brought the same engine up went from ~26.0-26.8 MB without
|
||||
`l3_tunnel` to ~28.3-28.7 MB with it over three paired runs — about +2.2 MB.
|
||||
That was measured on a throwaway harness, not on aarch64, not under load, and
|
||||
with an empty ICMP NAT table, so it bounds nothing on the router. Neither
|
||||
item is folded into any claim above.
|
||||
|
||||
Consequence: a ping from the LAN either genuinely travels through the tunnel
|
||||
(WireGuard/AWG, direct) or fails honestly, at every size the router itself can
|
||||
put into the device — and a router that never opts in renders the pre-feature
|
||||
plane byte-for-byte (`TestL3IngressOptIn` pins the off-state render). Read
|
||||
"fails honestly" strictly: above the tunnel MTU a non-DF ping now leaves the
|
||||
router for real and then times out, because upstream's ICMP NAT does not carry
|
||||
fragments back either. The one qualifier left is item 2 above — a client that
|
||||
puts fragments on the wire ITSELF, on a router where conntrack defrag is not
|
||||
reassembling them first. This paragraph has been overclaimed twice already;
|
||||
extend it only against a bench result, never against a reading.
|
||||
|
||||
## D26 — What the engine cannot carry, the kernel carries: `untunnelable_egress`
|
||||
Decided 2026-07-26. D25 ended with ESP/AH/GRE/IGMP/SCTP still owned by the D17
|
||||
policy — that is, with a choice between dropping them and leaking them out the
|
||||
WAN, never a data plane. This decision gives them one, and deliberately NOT
|
||||
ours: **`globals.untunnelable_egress` (default empty;
|
||||
`model.Globals.UntunnelableEgress`) names an existing egress of type
|
||||
interface/tunnel, and LAN traffic that is neither TCP nor UDP is stamped in
|
||||
prerouting with that egress's own mark, so the KERNEL routes it out that
|
||||
egress's device with the kernel's own NAT.** No proxy, no engine, no userspace
|
||||
stack ever touches the packet — which is exactly why every protocol works.
|
||||
|
||||
**Where the engine's boundary actually is — recorded so nobody digs for it
|
||||
twice.** It is NOT the gVisor stack, and it is not WireGuard: on the forward
|
||||
path the WG/AWG endpoint never consults gVisor at all. `Endpoint.WritePackets`
|
||||
(`transport/wireguard/port.go:21-58`) takes the raw IP packet bytes, reads
|
||||
exactly the IP version and the destination address, and hands
|
||||
`device.InputPacketRef`s to `wgDevice.InputPackets` — the protocol byte is
|
||||
never read; on the way back (`port.go:127-157`) `returnDeviceWrapper.Write`
|
||||
offers every decrypted packet to `returnPath.ReturnPackets` first and only the
|
||||
unconsumed remainder falls through to the gVisor device. gVisor serves
|
||||
`DialContext`/`ListenPacket` — traffic the ENGINE originates — while forwarded
|
||||
traffic bypasses the stack in both directions, indifferent to protocol. The
|
||||
real ceiling sits one step earlier, in sing-tun's `ForwardDispatcher`:
|
||||
`parseTransport` (`flow_parse.go:106-153`) sets `hasFlow` for exactly TCP, UDP,
|
||||
ICMPv4 Echo/EchoReply and ICMPv6 EchoRequest/EchoReply — a packet of any other
|
||||
protocol is never dispatched as a flow — and `createFlow`
|
||||
(`flow_dispatch.go:325`) builds its NAT through
|
||||
`allocateSelector(packet.protocol, …, packet.source.Port())` (line 355), which
|
||||
needs a port-like selector that ESP/AH/GRE simply do not have (SCTP has ports,
|
||||
but the parser above never grants it a flow either). Tailscale documents the
|
||||
same frontier for its own userspace mode — "Any IP protocol other than TCP or
|
||||
UDP (such as SCTP) is not supported in userspace mode… All IP protocols are
|
||||
supported" in kernel mode
|
||||
(https://tailscale.com/docs/reference/kernel-vs-userspace-routers) — useful as
|
||||
external corroboration of where userspace data planes generally end, though OUR
|
||||
boundary is the dispatcher, not the stack. The kernel egress was therefore
|
||||
chosen not because userspace "cannot" in principle, but because the kernel
|
||||
delivers all protocols with zero new code on the hot path.
|
||||
|
||||
**The mechanism already existed; the feature is one binding and one marking
|
||||
step.** `addEgressRouting` (`netplane/apply.go`) has always installed, for
|
||||
every interface/tunnel egress, an `ip rule fwmark <EgressMark> lookup
|
||||
<EgressTable>` plus a `default dev <device>` route in that table — per-rule
|
||||
egress selection rides on it. The only missing piece was that nothing ever
|
||||
marked non-TCP/UDP traffic: `untunnelable=direct` merely ACCEPTED it in the
|
||||
forward chain, so it left over the main table, i.e. the WAN.
|
||||
`UntunnelableEgressBinding` (`netplane/nft.go`) resolves the option to the
|
||||
egress's index, its OWN mark and its OWN device — deliberately no third
|
||||
mark/table pair to keep coherent — and the prerouting chain stamps that mark on
|
||||
the untunnelable protocols. A name that does not resolve to an interface/tunnel
|
||||
egress with a device renders nothing and is reported: the D17 policy stays in
|
||||
sole charge, which is the fail-closed reading of a typo.
|
||||
|
||||
**Why `l4proto != { tcp, udp }` is safe here when D25 banned it.** D25 rejected
|
||||
the broad filter because the receiving side was the `ForwardDispatcher`, which
|
||||
classifies nothing beyond TCP/UDP/ICMP echo — a marked ESP packet would enter
|
||||
the TUN and vanish, a black hole wearing a tunnel's name. Here the receiving
|
||||
side is the kernel, which forwards ANY IP protocol and NATs what it has
|
||||
machinery for: SCTP carries ports and NATs like TCP/UDP; GRE is NATed only
|
||||
through the PPTP helper keyed on the call-id — the kernel's own comment calls
|
||||
GRE "generally not very suited for NAT, as it has no protocol-specific part as
|
||||
port numbers" (`net/netfilter/nf_conntrack_proto_gre.c`); ESP/AH pass as plain
|
||||
routed IP. Nothing on this path can silently swallow a protocol it does not
|
||||
understand, which was the entire objection.
|
||||
|
||||
**Order against D25: the L3 ingress claims ICMP first.** With `l3_tunnel` on,
|
||||
LAN ICMP is marked into the engine's TUN before the egress carrier is consulted
|
||||
— the engine path routes ping by the operator's rules, which a kernel egress
|
||||
cannot do — and only the remaining protocols go to the egress. With `l3_tunnel`
|
||||
off, ICMP goes to the egress with everything else. In both shapes marked
|
||||
traffic is settled by ROUTING before the forward chain speaks, so the D17
|
||||
policy now governs exactly the failure case — the rule or route that did not
|
||||
come up — the same division D25 already established for the L3 mark.
|
||||
|
||||
**What the feature refuses to promise — and the operator text refuses with it
|
||||
(`shater/apply/warnings.go`, the egress-carrier note).** (a) It is not a tunnel
|
||||
per se: the option accepts any interface/tunnel egress, and on the target
|
||||
routers a WireGuard device is the exception (`kmod-wireguard` is usually
|
||||
absent) while a second WAN is routine. Through a WireGuard egress this
|
||||
genuinely is a tunnel; through a second WAN it is simply another uplink, and
|
||||
the destination sees that uplink's real address. No text, comment or doc line
|
||||
may call it a tunnel unconditionally. (b) It does not revive IPTV: IGMP is
|
||||
LAN-side multicast group management, WireGuard is L3 point-to-point and carries
|
||||
no multicast, and multicast never crossed this router under any setting —
|
||||
routing IGMP out an egress restores nothing, and no wording may hint otherwise.
|
||||
(c) IPsec through NAT-T never needed it: RFC 3948 encapsulates ESP in UDP/4500,
|
||||
so a modern IPsec client behind NAT is ordinary UDP that already follows the
|
||||
routing rules; the raw-ESP case this feature carries is the no-NAT-T remainder.
|
||||
|
||||
- **Rejected: teaching the engine these protocols.** Extending `parseTransport`
|
||||
and the selector NAT upstream would be new hot-path code in an
|
||||
actively-maintained adversarial area, for protocols the kernel already
|
||||
forwards for free — and for ESP/AH/GRE there is no port-like selector to NAT
|
||||
by in the first place.
|
||||
- **Rejected 2026-07-26: carrying them through the userspace AWG endpoint
|
||||
site-to-site, with no NAT at all.** This is the alternative the "no port-like
|
||||
selector" line above does NOT dispose of, and it is written down because the
|
||||
obvious reading of that line — "impossible" — is wrong and would be
|
||||
re-derived. The endpoint is already protocol-blind in both directions
|
||||
(`transport/wireguard/port.go:21-58`, `:127-157`), so an ESP packet could be
|
||||
forwarded UNTOUCHED, keeping the LAN client's own source address, and the
|
||||
reply would come back addressed to that client and need only be written to the
|
||||
TUN. No selector, no NAT, every protocol. It needs two things we declined to
|
||||
take on: lx-owned code in the forward hot path, bypassing `ForwardDispatcher`
|
||||
on both legs — precisely the surface CONSTITUTION §2 exists to keep small on
|
||||
an actively-maintained upstream — and a SERVER-side prerequisite (our LAN
|
||||
prefix in the peer's `AllowedIPs`, plus a route back), which turns a router
|
||||
option into a deployment contract. The kernel egress above buys the same
|
||||
protocols with zero hot-path code, so this stays a design on file, not a gap.
|
||||
- **Rejected: a dedicated mark/table pair for the carrier.** `addEgressRouting`
|
||||
already binds `EgressMark`/`EgressTable` to the device; a third pair would be
|
||||
a second copy of the same route that could drift from the first.
|
||||
|
||||
**Not verified, said plainly.** The end-to-end path — LAN client → prerouting
|
||||
mark → ip rule → egress device → far end and back — has not been exercised with
|
||||
real ESP or GRE on live hardware. Nothing above claims it has.
|
||||
|
||||
Consequence: raw IPsec, PPTP/GRE, SCTP — and ICMP when the L3 ingress is off —
|
||||
leave through an egress the operator explicitly named, under kernel routing and
|
||||
kernel NAT, instead of being dropped or silently leaking out the WAN; and with
|
||||
the option empty (the default) the plane renders byte-for-byte as before, with
|
||||
the D17 policy in sole charge.
|
||||
|
||||
## D27 — The `sing-quic` pin moves forward; two use-after-release defects in the QUIC/HTTP-3 client path
|
||||
Decided 2026-07-26, after an audit of `common/httpclient`, `dns/transport/quic`
|
||||
and `transport/v2rayquic`. Two suspicions were put to a test rather than to a
|
||||
reading. Both turned out to be real, and neither was the resource leak the
|
||||
suspicion named — both are objects released while still in use.
|
||||
|
||||
**1. The HTTP/3 race handed back a response nobody could read
|
||||
(`common/httpclient/http3_transport.go`).** `roundTripHTTP3Race` ran both racers
|
||||
on one `context.WithCancel` child and its `drainRemaining()` called `cancel()`
|
||||
before returning the WINNER. quic-go and net/http both reset a request's stream
|
||||
when its context is cancelled, so the caller received a `*http.Response` whose
|
||||
body died mid-read. Measured, not inferred:
|
||||
`H3_REQUEST_CANCELLED (local) (read 2687 of 65536 bytes)`. The path is taken
|
||||
whenever there is no cached HTTP/3 connection and the request is replayable —
|
||||
that is, the FIRST request to every host, plus every request after an idle
|
||||
close. Anything configured with `"version": 3` and no
|
||||
`disable_version_fallback` was affected: subscription fetches, remote rule-set
|
||||
downloads, URLTest probes.
|
||||
Fixed by giving each racer a context of its own. Losers are cancelled where the
|
||||
old code cancelled everything; the winner's `cancel` travels with its body and
|
||||
fires on `Close`. Pinned by
|
||||
`common/httpclient/http3_race_lx_test.go` — one test per winner, and the
|
||||
loser-is-torn-down assertion so the fix cannot be "stop cancelling" either.
|
||||
|
||||
**How defect 2 is pinned, and why it takes two tests.** The success path is
|
||||
visible on the wire, so `TestHTTP3ExchangeRequestBufferOutlivesRoundTrip`
|
||||
compares the query the server received with the query we packed. The failure
|
||||
path is NOT visible on the wire — see the `CancelWrite` note below — so
|
||||
`TestHTTP3ExchangeNeverPacksQueriesIntoPooledMemory` pins the CAUSE instead of
|
||||
the symptom: it tags a query with a random needle, runs an exchange that fails
|
||||
(a server that never answers; a context already cancelled) and then drains the
|
||||
buffer pool on the same goroutine `RoundTripOpt` ran on, demanding the needle is
|
||||
not in it. Both carry a control that must produce a DIRTY result first — the
|
||||
poison must be shown to reach a released buffer, the scan must be shown to find a
|
||||
query that really was packed into pool memory — because a clean verdict from an
|
||||
instrument that cannot produce a dirty one is not evidence. The second test
|
||||
encodes the chosen design; a future guarded-pool implementation would fail it on
|
||||
purpose and would have to replace it, saying so.
|
||||
|
||||
**2. DoH3 sent the DNS server whatever the next caller put in a recycled buffer
|
||||
(`dns/transport/quic/http3.go`).** `Exchange` packed the query into a POOLED
|
||||
`buf.Buffer`, handed `bytes.NewReader` over it to the request, and called
|
||||
`requestBuffer.Release()` the moment `RoundTrip` returned. But http3's
|
||||
`doRequest` writes the request body on a goroutine of its own and returns as
|
||||
soon as the response HEADERS arrive — the body is still being read. The test
|
||||
holds that window open (a 2 KB server stream window, an answer written before
|
||||
the body is read) and shows the query on the wire diverging from the query we
|
||||
packed **at exactly offset 8192** — `bodyCopyBufferSize`, the amount quic-go had
|
||||
already copied out before the buffer went back to the pool. Everything past that
|
||||
was the next pool user's memory, sent to the resolver. That is a data race and a
|
||||
small memory-disclosure primitive, not a slowdown.
|
||||
**The first fix for this was wrong, and the way it was wrong is the point.** It
|
||||
transferred ownership to the transport: the body became a `pooledRequestBody`
|
||||
whose `Close` released the buffer, justified as "`http3.Transport` closes the
|
||||
request body on every path, hence the `sync.Once`". That sentence is true about
|
||||
HOW MANY TIMES the body is closed and says nothing about WHEN — the exact shape
|
||||
of dishonesty this document keeps having to name. Review caught it before
|
||||
release. On the failure path `RoundTripOpt` (`http3/transport.go:167-173`) closes
|
||||
the body the moment `doRequest` returns, and `doRequest`
|
||||
(`http3/client.go:338-341`) waits only on the request-CANCELLATION watchdog —
|
||||
`close(reqDone); <-done` — never on the goroutine writing the body. Nothing in
|
||||
quic-go ever joins that goroutine. So `Close` is not a handoff point, and the
|
||||
`sync.Once` prevented a double `Release` while doing nothing about a read after
|
||||
one.
|
||||
**One correction to the review's severity, for the record:** on the failure path
|
||||
the damage stops at the data race. Every `ReadResponse` error branch
|
||||
(`http3/stream.go:325`, `:336`, `:343`, `:363`) calls `str.CancelWrite` BEFORE
|
||||
`RoundTripOpt` closes the body, so the bytes the writer reads out of the recycled
|
||||
buffer are thrown at an already-cancelled stream and never reach the resolver.
|
||||
The memory-disclosure primitive is the SUCCESS path only. The failure path is
|
||||
"merely" a read of memory owned by somebody else — still undefined behaviour,
|
||||
still a `-race` finding, still not shippable.
|
||||
**Fixed instead by not sharing at all:** the query is packed with `Pack()` into
|
||||
memory the request body owns outright, and no pooled buffer is involved. The
|
||||
alternative on the table — a lock around `Read` and `Close` so reads after
|
||||
release return an error — would also be correct, and was rejected because it
|
||||
keeps a released-but-referenced object alive and leaves a live invariant for the
|
||||
next person to break, which is now twice in one day that an assumption about
|
||||
quic-go's internal lifetimes has been wrong.
|
||||
The cost turned out to be negative, measured rather than assumed: `Pack` runs
|
||||
87 ns/op at 64 B and 1 alloc against 108 ns/op at 64 B and 1 alloc for the pooled
|
||||
version, because `buf.NewSize` allocates the `Buffer` struct itself — the same 64
|
||||
bytes — and then adds `Get`/`Put` on top. **The pool was never saving an
|
||||
allocation on this path.** The response buffer stays pooled: it is read and
|
||||
unpacked before `Exchange` returns and nothing outlives it.
|
||||
Both files thereby DIVERGE from upstream again, six hours after `0a6689b29`
|
||||
made them byte-identical on purpose. That was the right call then and this is
|
||||
the right call now; upstream carries defect 2 in `dns/transport/https.go` as
|
||||
well (same shape, HTTP/1.1 and HTTP/2 write bodies asynchronously too) and that
|
||||
file was left alone — it is outside the audit's scope, and it is written down
|
||||
here so the next person finds it instead of rediscovering it.
|
||||
|
||||
**3. The pin moved: `sing-quic` v0.6.2-0.20260525051024 -> v0.6.4-0.20260709034545.**
|
||||
`quic.go` — `Dial`/`DialEarly`/`CreateTransport` — is byte-identical across the
|
||||
two, so the packet-conn ownership fix in `0a6689b29` is NOT duplicated by the
|
||||
bump and is not made redundant by it: quic-go still does not own the socket, and
|
||||
we still close it. What the newer module does carry is the OTHER half of the
|
||||
same family, and we had taken only our half:
|
||||
`clientConn.Close()` in `tuic/`, `hysteria/` and `hysteria2/` now sets a past
|
||||
write deadline after `Stream.Close()`, word for word the fix
|
||||
`transport/v2rayquic/stream.go` already had — quic-go's `Stream.Close` does not
|
||||
release a write blocked on flow control. We ship tuic and hysteria2, so on the
|
||||
old pin every such close could park a goroutine for the life of the process.
|
||||
It also brings a QUIC-connection-death watchdog and a handshake deadline to the
|
||||
hysteria clients.
|
||||
**The cost, measured, not estimated:** six new indirect modules (`libp2p/go-nat`
|
||||
and its UPnP/NAT-PMP/gopacket tail) for hysteria2's realm port mapping, and
|
||||
**+256 KiB exactly** on the stripped aarch64 `shaterd` (26 542 242 ->
|
||||
26 804 386 bytes, +0.99%), before UPX. The port-mapping path is unreachable from
|
||||
anything `shater/generate` emits — `realm` is only built when the JSON names it,
|
||||
and it never does — so the growth is dead weight, but it is small dead weight
|
||||
next to a goroutine leak on the two QUIC protocols we actually ship. `upstream/lx`
|
||||
is already on this pin, so keeping the old one would mean fighting every rebase.
|
||||
**Not verified here:** the tuic/hysteria2 close fix is read from the module diff,
|
||||
not exercised — those packages are outside this audit's file set and testing them
|
||||
needs a live tuic/hysteria2 server. `go test ./common/... ./dns/... ./transport/...
|
||||
./protocol/...` is green under the shipped tag set apart from
|
||||
`common/windivert`'s `TestIntegration*`, which want Windows SCM access and fail
|
||||
on any developer machine, bump or no bump.
|
||||
|
||||
## D28 — The L3 TUN is a reclaimable SLOT, not a name: a fixed device made every apply fatal
|
||||
Decided 2026-07-26. D25 gave the L3 ingress one fixed device, `shater-l3`. On the
|
||||
production router that single name made **every** configuration change with
|
||||
`l3_tunnel=1` an outage, three times in a row:
|
||||
|
||||
```
|
||||
19:10:33 reconcile failed: start inbound/tun[l3-in]: open tun: TUNSETIFF: device or resource busy
|
||||
19:14:02 start instance failed and could not restore previous config; engine stopped
|
||||
19:14:38 reconcile failed: TUNSETIFF: device or resource busy
|
||||
```
|
||||
|
||||
followed by `plane: hold` — the fail-closed ruleset — i.e. the whole house
|
||||
offline until somebody ran `/etc/init.d/shater restart` by hand.
|
||||
|
||||
**The mechanism is a collision between two GENERATIONS of the engine, and the
|
||||
fatal part is where the collision lands.** An apply builds a fresh box and starts
|
||||
it; with one name that box must open the device the outgoing box still holds.
|
||||
That alone is a failed apply. What turned it into an outage is the recovery path:
|
||||
`closeOldThenStart` answers a failed start by rebuilding the PREVIOUS config —
|
||||
and that config names the same device, so **the rescue failed for exactly the
|
||||
reason the rescue was needed**. A recovery path must never depend on the resource
|
||||
whose contention it is recovering from; that sentence, not the device name, is
|
||||
the decision here.
|
||||
|
||||
**Two slots (`shater-l3a` / `shater-l3b`), chosen by the ENGINE at box-build
|
||||
time.** Not by `generate`, and that is load-bearing rather than incidental:
|
||||
`generate` runs on every reconcile including the once-a-minute no-ops, and its
|
||||
output is what `Apply` hashes to decide whether anything changed. A device name
|
||||
that alternated there would change the hash every minute and rebuild the whole
|
||||
engine forever; a name that tracked "whichever slot exists" would hand the BUSY
|
||||
one to every real change. Only the engine knows it is building a new generation.
|
||||
So `generate` emits `netplane.L3DeviceBase` as a **placeholder that is never
|
||||
created**, and `engine.newBox` substitutes a slot on a COPY of the options —
|
||||
after the hash, so the stored config stays canonical (`engine/l3slot.go`).
|
||||
|
||||
**Rotation alone is NOT the fix, and that was measured, not reasoned.** The
|
||||
two-slot build survived five applies of five different kinds and then failed on
|
||||
4 of 10 back-to-back changes with the original outage in full. A retired
|
||||
generation does not hand its device back when its replacement is adopted:
|
||||
`Box.Close` walks its subsystems under a budget and the TUN dies with the last
|
||||
fd. One apply gives it time; two inside that window do not. Rotation widens the
|
||||
race by one generation — a better outage, not the absence of one.
|
||||
|
||||
**So an occupied non-current slot is DELETED, not waited for** (`L3SlotFor`).
|
||||
The running generation's slot is excluded first and never touched; every other
|
||||
slot belongs to a retired generation that is not in the L3 routing table and is
|
||||
carrying nothing, so taking its device away is safe and, if anything, helps the
|
||||
close already in flight. There is deliberately **no bounded wait**: waiting on an
|
||||
asynchronous kernel teardown is the race this design removes, and adding it back
|
||||
as a "safety net" would only make the failure intermittent.
|
||||
|
||||
**The firewall never learns which slot is live.** Our forward accepts and the fw4
|
||||
zone match by PREFIX — `iifname "shater-l3*"` / `oifname "shater-l3*"`, and
|
||||
`list device 'shater-l3*'` in uci-defaults. Verified on ImmortalWrt 25.12.1
|
||||
(kernel 6.12.94, nftables 1.1.6) that both forms validate AND load, and that fw4
|
||||
compiles the wildcard into exactly those matches. So the rendered ruleset is
|
||||
byte-identical across a swap: no nft reload, and no window in which the accept
|
||||
names a device that is already gone. Routers seeded by a pre-slot build are
|
||||
migrated in place (`migrate_l3_zone_wildcard`); without it an upgrade would
|
||||
silently go back to fw4 dropping the forward.
|
||||
|
||||
**A2 — turning the feature off used to leave the plane installed.** `addL3Routing`
|
||||
returned early when `l3_tunnel=0`, so after switching it off the router still had
|
||||
the device, `ip rule fwmark 0x2080 lookup 8200` and table 8200. Nobody else
|
||||
removes them: the engine's new config simply has no TUN inbound. The surviving
|
||||
device is also the commonest way back into the EBUSY above. The disabled branch
|
||||
now removes rule, table and device, symmetrically with `removeEgressRouting`, and
|
||||
`TeardownRouting` sweeps every name including the legacy `shater-l3`.
|
||||
|
||||
**Two smaller lies found while proving the above, both measured on the stand and
|
||||
both fixed here.** `ip -6 route flush table N` does not remove a non-unicast
|
||||
route while the v4 flush does, so (a) the floor survived the flush and the next
|
||||
`add` answered `File exists`, which was reported as a CRITICAL "this table has NO
|
||||
fail-closed floor, traffic can leave over the plain WAN" — on every single apply,
|
||||
about a table whose floor was sitting right there; and (b) teardown left that
|
||||
floor behind. An already-present floor is now success, and teardown deletes it
|
||||
explicitly. (b) cannot misroute anything — no rule points at the table — but a
|
||||
teardown whose result is not "the table prints nothing" is one nobody can verify.
|
||||
|
||||
**Verified on the stand** (`local_openwrt`, ImmortalWrt 25.12.1, kernel 6.12.94 —
|
||||
the router's revision), before-and-after with binaries built from the same tree:
|
||||
the pre-fix binary reproduces the production failure on the "edit a node URI and
|
||||
apply" round (engine stopped, `plane: hold`) and leaves device + rule behind on
|
||||
`l3_tunnel=0`; the fixed binary survives all five apply kinds, 12 back-to-back
|
||||
changes, and leaves nothing at all — no device, no rule on either family, both
|
||||
tables printing empty.
|
||||
|
||||
**Known fragility, deliberately NOT addressed here.** Our `ip rule` priorities
|
||||
are whatever the kernel hands out (the L3 rule lands at 32764, counting down from
|
||||
32765), so the ORDER of our rules between reboots depends on insertion sequence.
|
||||
It is not a live defect — the marks are disjoint, each rule catches its own, and
|
||||
all of them end up above `main` — but it is luck, not design. Moving to explicit
|
||||
`pref` values touches every existing rule and needs a migration for rules already
|
||||
installed on implicit numbers; that is its own piece of work, not a rider on this
|
||||
one.
|
||||
|
||||
+39
-3
@@ -6,12 +6,44 @@ usable release, **[T1]** next, **[T2]** later. Phases refer to `ROADMAP.md`.
|
||||
## Proxy engine & protocols (from the sing-box fork)
|
||||
- **[MVP]** VLESS, VMess, Trojan, Shadowsocks, WireGuard, Reality/XTLS.
|
||||
- **[MVP]** **AmneziaWG 2.0** (I1–I5 CPS decoy packets) — a driving requirement.
|
||||
- **[T1]** Hysteria2, TUIC, ShadowTLS, XHTTP, MASQUE/CONNECT-IP (Cloudflare WARP).
|
||||
- **[MVP]** Hysteria2, TUIC (`hysteria2://`/`hy2://`/`tuic://`, `shater/parse`),
|
||||
XHTTP transport — all shipped: the router tag set carries `with_quic` and
|
||||
`with_xhttp` and `shater/registry` registers them (`scripts/router-tags.sh`,
|
||||
`buildtags.Features`).
|
||||
- **[T1]** ShadowTLS — half-built: `shater/generate` emits it and `shater/registry`
|
||||
registers it, but no parser produces one (there is no `shadowtls://` share link
|
||||
and no subscription path), so a config cannot reach it today.
|
||||
- **NOT SHIPPED** MASQUE/CONNECT-IP (Cloudflare WARP). `masque` appears nowhere in
|
||||
`shater/parse`, `shater/generate` or `shater/model`, and `shater/registry` names
|
||||
it among the upstream types it deliberately does not register (~6 MB of binary
|
||||
and resident RAM). The engine fork can build it; this product does not.
|
||||
- **[MVP]** Transports: TCP/WS/gRPC/HTTPUpgrade/H2/QUIC as upstream provides.
|
||||
|
||||
## Transparent proxying & routing
|
||||
- **[MVP]** TPROXY transparent proxy for multiple LAN interfaces (TCP + UDP), SNI/
|
||||
Host/QUIC sniffing.
|
||||
- **[MVP]** **L3 ingress for ICMP** (`globals.l3_tunnel`, opt-in, default off):
|
||||
LAN ping travels THROUGH the tunnel instead of being dropped or answered by a
|
||||
forged local reply. The engine opens a dedicated TUN (`shater-l3`, gVisor
|
||||
stack, `auto_route` off); nft marks LAN icmp/icmpv6 only and a scoped
|
||||
`ip rule` routes it in — the TPROXY plane and the main routing table stay
|
||||
untouched (D25). Carried only by L3-capable egresses (WireGuard/AmneziaWG,
|
||||
direct); ICMP routed to vless/vmess/… is honestly dropped, never faked.
|
||||
Ceiling is upstream sing-tun's: ICMP echo only — Windows tracert works, IPv6
|
||||
traceroute shows just the destination; ESP/AH/GRE/IGMP stay with the
|
||||
`untunnelable` policy (D17) unless `untunnelable_egress` carries them (D26).
|
||||
- **[MVP]** **Kernel egress for untunnelable protocols**
|
||||
(`globals.untunnelable_egress`, opt-in, default empty): names an existing
|
||||
interface/tunnel egress, and IPsec (ESP/AH), PPTP/GRE, SCTP — everything that
|
||||
is neither TCP nor UDP, plus ICMP when the L3 ingress is off — is routed out
|
||||
that egress's device by the KERNEL with kernel NAT, reusing the egress's own
|
||||
fwmark/table from `addEgressRouting`; the proxy never sees a byte, which is
|
||||
why every protocol works (D26). What that buys depends on the device: a
|
||||
WireGuard interface really is a tunnel, a second WAN is just another uplink
|
||||
whose real address the destination sees. It does not revive multicast IPTV,
|
||||
and UDP-based VPNs (WireGuard, OpenVPN-UDP, IPsec NAT-T) never needed it —
|
||||
they follow the routing rules as before. The `untunnelable` policy (D17)
|
||||
keeps only the failure case: a route that did not come up.
|
||||
- **[MVP]** First-match routing rules by source (IP/CIDR/MAC/interface/zone),
|
||||
destination, port, proto → target (outbound/selector/chain/direct/block) + egress.
|
||||
A rule names its **destination through a rule-set only** — a reusable named list
|
||||
@@ -87,8 +119,12 @@ usable release, **[T1]** next, **[T2]** later. Phases refer to `ROADMAP.md`.
|
||||
## Reliability ("железно")
|
||||
- **[MVP]** Fail-closed kill-switch (dead group → block, never silent direct leak);
|
||||
IPv6 dropped when disabled.
|
||||
- **[MVP]** Atomic apply with engine + `nft -c` validation; commit-confirm
|
||||
auto-rollback to last-good.
|
||||
- **[MVP]** Atomic apply with engine + `nft -c` validation. Commit-confirm
|
||||
auto-rollback to last-good is built and works, but it is **opt-in and ships
|
||||
OFF**: `DefaultGlobals()` leaves `ConfirmTimeout` at 0, the shipped
|
||||
`/etc/config/shater` says `confirm_timeout '0'`, and `apply.ArmRollback` returns
|
||||
at once on a non-positive timeout. Until an operator sets a window, an apply on
|
||||
a stock box has no net under it — and `shaterd apply` says so.
|
||||
- **[MVP]** Idempotent reconcile from hotplug/boot under flock; restart engine only
|
||||
on real config change; management-bypass (SSH/LuCI/LAN) always exempt.
|
||||
- **[MVP]** Own nft table `inet shater` + own marks/tables; never touch fw4.
|
||||
|
||||
+79
-8
@@ -88,7 +88,7 @@ Four OpenWrt packages live under `openwrt/`:
|
||||
| Package | Arch | What it ships |
|
||||
|--------------------|-----------|---------------|
|
||||
| `shaterd` | per-arch | **Prebuilt** static `shaterd` binary → `/usr/bin/shaterd` (this is the ship artifact from step 1). |
|
||||
| `shater-core` | all | procd init (supervises `shaterd run`), cron, hotplug, sysctl, inert default UCI. `DEPENDS:=+shaterd +kmod-nft-tproxy +kmod-nft-socket +ip-full`. |
|
||||
| `shater-core` | all | procd init (supervises `shaterd run`), the boot armor (§4), cron, hotplug, sysctl, inert default UCI. `DEPENDS:=+shaterd +kmod-nft-tproxy +kmod-nft-socket +kmod-tun +ip-full +nftables-json +ca-bundle`. |
|
||||
| `luci-app-shater` | all | Thin LuCI launcher: mini dashboard + token-handoff "Open panel" button. `DEPENDS:=+shater-core +rpcd`. |
|
||||
| `byedpi` | per-arch | *Optional* ByeDPI (`ciadpi`) local desync SOCKS proxy for a `type='byedpi'` egress. |
|
||||
|
||||
@@ -176,11 +176,27 @@ install. Configure nodes/rules (via the LuCI panel or `uci`), then enable and ap
|
||||
|
||||
```sh
|
||||
uci set shater.globals.enabled=1
|
||||
# The safety net is NOT on by default — see below. 120 s is a window wide enough
|
||||
# to re-open SSH/LuCI and decide whether the new config is any good.
|
||||
uci set shater.globals.confirm_timeout=120
|
||||
uci commit shater
|
||||
shaterd apply # apply + arm commit-confirm on the running daemon
|
||||
shaterd confirm # confirm (cancels the auto-rollback)
|
||||
shaterd apply # apply + arm the auto-rollback for 120 s
|
||||
shaterd confirm # confirm inside that window (cancels the auto-rollback)
|
||||
```
|
||||
|
||||
> **Commit-confirm ships OFF.** `model.DefaultGlobals()` does not seed
|
||||
> `ConfirmTimeout`, the shipped `/etc/config/shater` carries
|
||||
> `option confirm_timeout '0'`, and `apply.ArmRollback` returns immediately on a
|
||||
> non-positive timeout — so on a stock box `shaterd apply` arms **nothing** and an
|
||||
> apply that costs you SSH/LuCI access simply stays. The daemon says so rather
|
||||
> than implying otherwise: the `commit-confirm-off` outcome of `shaterd apply`
|
||||
> prints *"globals.confirm_timeout is 0, so commit-confirm is switched OFF: this
|
||||
> apply armed NO automatic rollback"*, and the panel's Overview reads
|
||||
> `confirm: no auto-rollback`. Set a window (UCI as above, or Settings in the
|
||||
> panel) if you want the net. Non-obvious detail: the option is written back only
|
||||
> when non-zero, so an explicit `0` disappears from `/etc/config/shater` on the
|
||||
> first write — absent and `0` mean the same thing.
|
||||
|
||||
`/etc/init.d/shater enable && /etc/init.d/shater start` brings up the procd-supervised
|
||||
daemon (`shaterd run`), which owns the engine, the `inet shater` data plane, policy
|
||||
routing, in-process DNS, and the admin panel (default `:8088`). The LuCI app's
|
||||
@@ -218,6 +234,55 @@ Your `0` is kept: `/etc/config/shater` is a conffile (upgrades never replace it)
|
||||
the daemon always writes the option back explicitly, so it is never re-enabled by a
|
||||
default.
|
||||
|
||||
### The boot-time fail-closed armor
|
||||
|
||||
`shater-core` installs a **third** init script, `/etc/init.d/shater-armor`, and
|
||||
`30_shater-core` enables it at install time. It exists because `/etc/init.d/shater`
|
||||
is `START=99`: by then fw4 (19) has loaded `lan -> wan ACCEPT` and netifd (20) has
|
||||
brought the LAN bridge up, so between link-up and the daemon's first apply the
|
||||
router forwards LAN traffic to the WAN in the clear — on router hardware with a
|
||||
UPX-packed binary that is the seconds in which Wi-Fi associates and every client
|
||||
reconnects. `kill_switch=closed` covered none of it, because the protection lived
|
||||
inside a process that had not started.
|
||||
|
||||
**How it works.** On every apply the daemon persists a copy of its fail-closed
|
||||
*holding plane* — the same ruleset it installs when the engine is down — to
|
||||
`/etc/shater/boot.nft`. `shater-armor` runs at `START=21` (after fw4 and netifd),
|
||||
validates that file with `nft -c` and loads it. When the daemon comes up it
|
||||
replaces the table atomically, so there is never a moment with no table. Its
|
||||
`stop()` is deliberately a no-op.
|
||||
|
||||
**LAN forwarding is blocked until the daemon applies — management access is not.**
|
||||
The chain hooks `forward` only, so SSH, LuCI and the admin panel (all `input` hook,
|
||||
to the router's own addresses) stay reachable **on purpose**: a kill switch you
|
||||
cannot switch off is a brick. If you see the syslog line
|
||||
|
||||
```
|
||||
fail-closed plane armed from /etc/shater/boot.nft: LAN->WAN forwarding is BLOCKED
|
||||
until shaterd applies. SSH, LuCI and the admin panel stay reachable.
|
||||
```
|
||||
|
||||
that is the mechanism working, not a fault.
|
||||
|
||||
**When it refuses to arm** — each is a state check made at boot, never a record of
|
||||
something that happened on the way down:
|
||||
|
||||
| Condition | Behaviour |
|
||||
|---|---|
|
||||
| `/etc/shater/boot.nft` absent | Nothing to do, silent. The file exists only while the last applied config was **both** `enabled=1` **and** `kill_switch=closed`; either being off removes it at the next apply, and an operator-typed `/etc/init.d/shater stop` removes it there and then. Powering off does **not** — and neither does the `stop` a package upgrade issues while the service stays enabled, so being replaced cannot leave the next boot unprotected. |
|
||||
| the file is empty, or fails `nft -c` | Refuses, logs an error — the LAN is unprotected until `shaterd` starts. |
|
||||
| `/usr/bin/shaterd` missing, or no `S??shater` symlink in `/etc/rc.d` | Refuses: nothing would ever come along to replace the block with a working data plane. This is what makes an uninstalled or disabled product safe regardless of what the file says. |
|
||||
| UCI is readable **and** says `globals.enabled` is not `1` | Removes `boot.nft` and does not arm. An **unreadable** UCI is not a refusal — that case is exactly why the armor is a file rather than a query. |
|
||||
| `nft` not installed | Refuses, logs an error. |
|
||||
|
||||
**Turning it off.** The durable off-states are the two the script itself asks
|
||||
about — `uci set shater.globals.enabled=0 && uci commit shater && shaterd apply`
|
||||
(the next apply removes `boot.nft`), or `/etc/init.d/shater disable`. A bare
|
||||
`/etc/init.d/shater stop` typed at the shell also removes the file, but it is not
|
||||
durable: `S99shater` is still linked, so procd starts the daemon again on the next
|
||||
boot. To remove just the armor and keep the stack: `/etc/init.d/shater-armor
|
||||
disable`.
|
||||
|
||||
## 5. The signed apk repo (the normal install path)
|
||||
|
||||
OpenWrt/ImmortalWrt **25.12** packages with Alpine's **apk**: `.apk` files, a
|
||||
@@ -321,8 +386,14 @@ The mtk-vendor channel (base: `SuperKali/immortalwrt-mt798x-rebase`, branch
|
||||
`downloads.immortalwrt.org/releases/25.12-SNAPSHOT` — so packages built with the
|
||||
vanilla ImmortalWrt 25.12 filogic SDK install cleanly; no SuperKali-special SDK
|
||||
is needed. We ship **no kmods** (shaterd is a static Go binary, byedpi plain C),
|
||||
so the vendor 6.6 kernel is irrelevant to our packages; the kmod *dependencies*
|
||||
of shater-core (`kmod-nft-tproxy`, `kmod-nft-socket`, plus `ip-full`) are
|
||||
already **baked into the BananaWRT mtk-vendor image** (verified in its
|
||||
`config.buildinfo`). On a self-built 25.12 image, make sure those kmods come
|
||||
from the image's own kernel build.
|
||||
so the vendor 6.6 kernel is irrelevant to our packages.
|
||||
|
||||
What was actually checked in the BananaWRT mtk-vendor `config.buildinfo` is
|
||||
`kmod-nft-tproxy`, `kmod-nft-socket` and `ip-full` — those three are baked into
|
||||
the image. `shater-core` also depends on `kmod-tun`, `nftables-json` and
|
||||
`ca-bundle` (added later; see the annotated `DEPENDS` in
|
||||
`openwrt/shater-core/Makefile`), and **those were not part of that check**. They
|
||||
are ordinarily present on a stock image — apk will pull whatever is missing from
|
||||
the distfeeds — but if you install offline or from a slimmed image, verify them
|
||||
yourself. On a self-built 25.12 image, make sure the kmods come from the image's
|
||||
own kernel build.
|
||||
|
||||
+24
-5
@@ -273,10 +273,12 @@ Apply/rollback: `apSnapshot` (run→last-good, nft→last-good.nft, route marks)
|
||||
| `block_doh` | `0` | NXDOMAIN the known public DoH hostnames + the Firefox canary and reject `:443` to their IPs, so clients fall back to `:53` (which the engine catches) |
|
||||
| `group_health` | `1` | OUR background group probing (the observatory). Does not touch sing-box's own urltest inside a group |
|
||||
| `untunnelable` | `block` | policy for what TPROXY cannot carry (ICMP/IGMP/ESP/AH/GRE/SCTP): `block` \| `icmp` (echo out, rest dropped) \| `direct` (all out, bypassing the tunnel) |
|
||||
| `l3_tunnel` | `0` | **opt-in**, UCI-only (the panel does not expose it). Opens the synthetic `l3-in` TUN so LAN ICMP is routed by the engine instead of dropped/forged; nft marks LAN `icmp`/`ipv6-icmp` with `fwmark_base+0x80` and a scoped `ip rule` sends it to table `table_base+8`. Absent option ⇒ OFF; only an explicit `1` opens it. See D25 and `ARCHITECTURE.md` §3a |
|
||||
| `untunnelable_egress` | unset | **opt-in**, UCI-only. Names a `config egress`; everything the L3 block did not claim (ESP/AH, GRE, IGMP, SCTP, and ICMP when `l3_tunnel=0`) is stamped with that egress's OWN mark and routed out its device by the kernel — no new mark, no new table, engine not in the path. Empty ⇒ `untunnelable` above stays in sole charge (D26) |
|
||||
| `geo_provider` | unset = auto | `sagernet` \| `loyalsoldier` \| `metacubex` \| `custom`; auto = country codes from SagerNet, everything else from Loyalsoldier |
|
||||
| `geosite_url` / `geoip_url` | unset | `{category}` templates, honoured only when `geo_provider=custom` |
|
||||
| `geosite_index_url` / `geoip_index_url` | unset | git-trees URLs used to SUGGEST categories in the panel; empty = no suggestions |
|
||||
| `stats_backend` | `memory` | `off` (no aggregation at all) \| `memory` (RAM, lost on restart) \| `sqlite` (aggregates in RAM + query/connection log on disk) |
|
||||
| `stats_backend` | `memory` | `off` (no aggregation at all) \| `memory` (RAM, lost on restart) \| `sqlite` (aggregates in RAM + query/connection log on disk). The value NAME is historical: the on-disk store is **bbolt**, not SQLite, since the migration — a leftover sqlite-era `stats.db` is detected by its file magic and replaced (`shater/stats/boltring.go`) |
|
||||
| `stats_ring_size` / `stats_timeline_minutes` / `stats_max_domains` | `200` / `60` / `5000` | live-log length, sparkline minutes, domain-map cap. **`0` = UNLIMITED** (grows with traffic), which is why these three are always emitted |
|
||||
| `stats_disk_limit_mb` | `64` | on-disk cap of `stats.db`; only meaningful for `stats_backend=sqlite`; `0` = unlimited |
|
||||
| `stats_retention_disabled` | `0` | master switch that turns OFF all trimming/pruning — every aggregate then grows unbounded |
|
||||
@@ -285,7 +287,12 @@ Apply/rollback: `apSnapshot` (run→last-good, nft→last-good.nft, route marks)
|
||||
|
||||
Deleted options still parse (unknown keys are ignored) and drain out on the next
|
||||
render: `dns_mode` (D17 — fake-IP is a resolver TYPE), `sweep_interval` (D19).
|
||||
- `config inbound`: name, enabled, type, network, tproxy_port(12345), listen, port, auth, user, pass, target_addr, target_port, target_network, tcp, udp, sniff.
|
||||
- `config inbound`: name, enabled, type, network, tproxy_port(12345), listen, port, auth, user, pass, target_addr, target_port, target_network, tcp, udp.
|
||||
**No `sniff`.** Since sing-box 1.11 sniffing is a leading route ACTION rule with no
|
||||
inbound matcher, so every inbound is sniffed always; the flag was read by nothing but
|
||||
its own UCI round-trip. Re-adding it would be a regression, not a restored feature —
|
||||
the hijack-dns rule matches the SNIFFED `dns` protocol, so a per-inbound toggle is a
|
||||
DNS-leak switch wearing a performance label (`model.go`, `Inbound`).
|
||||
- `config subscription`: name, enabled, url, update_interval, fetch_via(direct|proxy), ua, hwid, device_os, ver_os, device_model, list header, format, list include/exclude/filter_proto/filter_country, dedup, expire_alert_days.
|
||||
- `config node`: name, enabled, uri, mux, mux_concurrency, xudp_concurrency, xudp_udp443, sockopt_mark, tcp_fast_open, tcp_keepalive_idle.
|
||||
- `config group`: name, source, subscription, list node, strategy, include/exclude/filter_proto/filter_country, dedup, probe_url, probe_interval.
|
||||
@@ -296,8 +303,19 @@ Apply/rollback: `apSnapshot` (run→last-good, nft→last-good.nft, route marks)
|
||||
destination is a `config ruleset` and nothing else. `shaterd migrate` folds each legacy
|
||||
list into a generated `rule-<name>` (and `rule-<name>-ip`) inline ruleset; see
|
||||
`DECISIONS.md` D21 for the entry-by-entry conversion table.
|
||||
- `config preset`: name, enabled, order, target. `config profile`: name, enabled, priority, list match_iface, probe_url, probe_mode, sched_*, list enable_rule/disable_rule, default_target, default_egress.
|
||||
- `config profile`: name, enabled, priority, list match_iface, probe_url, probe_mode, sched_*, list enable_rule/disable_rule, default_target, default_egress.
|
||||
- `config resolver`: name, type, address, detour, pool. `config dns_rule`: order, list match_domain/match_src, resolver.
|
||||
- `config blocklist`: name, enabled, source(inline|file|url|geosite), url, path, list category, list entry, response(nxdomain), update_interval.
|
||||
- `config allowlist`: the same minus `response` (an allowlist has no verdict to render); it overrides every blocklist.
|
||||
- `config device`: name, mac, ip, enabled, list block, list allow.
|
||||
- `config alert`: name, enabled, type(telegram), token, chat_id, url, list event, via, fallback.
|
||||
- **`config preset` is NOT a section type.** `ReadUCI`'s type switch has no `preset`
|
||||
branch, so such a section is parsed by nothing and reaches no part of the model.
|
||||
It survives only because `30_shater-core` still seeds three of them
|
||||
(`block_ads`, `ru_bypass`, `private`) with the comment "so the LuCI Rules page
|
||||
renders their toggles" — and v0.2's LuCI app is a thin launcher with no Rules
|
||||
page. Those `uci set` calls should be dropped from the uci-defaults script; until
|
||||
they are, three inert sections appear in every fresh `/etc/config/shater`.
|
||||
|
||||
### Subscriptions & HAPP fetch
|
||||
Schemes: `vless:// vmess:// trojan:// ss:// wireguard:// wg://`. Body formats (`DetectSubFormat`): clash-YAML, xray-JSON, singbox-JSON, base64/plain link list. All converge to URIs re-parsed by `ParseShareLink`. HAPP fetch: UA default `Happ/3.13.0`; headers `x-hwid` (auto UUIDv4/sub), `x-device-os`, `x-ver-os`, `x-device-model`, custom. `fetch_via=proxy` dials local socks. Quota/expiry from `Subscription-Userinfo` (`upload;download;total;expire`). Reconcile by `Fingerprint` (sha256 of proto|addr|port|id|net|sec|sni|path) → new/keep/stale (drop after 3 stale refreshes).
|
||||
@@ -305,10 +323,11 @@ Schemes: `vless:// vmess:// trojan:// ss:// wireguard:// wg://`. Body formats (`
|
||||
---
|
||||
|
||||
## PART B — v0.1 packaging (`shater-core/`, branch `v0.1`)
|
||||
Pure scripts+config, `PKGARCH:=all`. v0.1 DEPENDS: `+xrayctl +xray-core +dnsmasq-full +kmod-nft-tproxy +kmod-nft-socket +ip-full`. → **v0.2 deps: `+shaterd +kmod-nft-tproxy +kmod-nft-socket +ip-full`** (engine does DNS in-process, so dnsmasq-full may be droppable — confirm the :53 listener is our engine). `/etc/config/shater` is a conffile.
|
||||
Pure scripts+config, `PKGARCH:=all`. v0.1 DEPENDS: `+xrayctl +xray-core +dnsmasq-full +kmod-nft-tproxy +kmod-nft-socket +ip-full`. → **v0.2 deps (authoritative: `openwrt/shater-core/Makefile`, which annotates each one): `+shaterd +kmod-nft-tproxy +kmod-nft-socket +kmod-tun +ip-full +nftables-json +ca-bundle`** — `dnsmasq-full` is gone (the engine owns the `:53` hijack listener); `kmod-tun` is `/dev/net/tun` for the L3 ingress, `nftables-json` is the `nft -j` output `netplane/stats.go` parses, `ca-bundle` is the cert store a `CGO_ENABLED=0` binary has no host fallback for. `/etc/config/shater` is a conffile.
|
||||
- **init.d/shater** (procd, START=99/STOP=10): v0.1 supervised `xray run -c /etc/xray/run.json`; → v0.2 supervises `shaterd`. `respawn 3600 5 0` (infinite). **No `procd_set_param file` watch** (would bounce tunnel on commit). Inert unless `globals.enabled=1`. `ACTIVE_FLAG=/var/run/shater.active` gates hotplug/cron. `stop` clears flag + tears down nft table + reserved routing tables. `reload_service`→start/stop. trigger `procd_add_reload_trigger "shater"`.
|
||||
- **init.d/shater-cron** (START=96): supervised `loop`; per-item due-check, runs sub/ruleset update + reconcile + schedule due; watchdog: engine dead 5 ticks ⇒ kill_switch=open stops stack (fail-open), closed logs crit.
|
||||
- **uci-defaults/30_shater-core**: seed `rt_tables` (8192 shater), enable both inits, seed preset packs (disabled), run migrate, apply sysctl.
|
||||
- **init.d/shater-armor** (START=21/STOP=89, v0.2-only — no v0.1 counterpart): the fail-closed plane BEFORE the daemon exists. `/etc/init.d/shater` is START=99, so from netifd's `ifup` until the daemon's first apply the router forwarded LAN→WAN in the clear. The daemon persists its holding plane to `/etc/shater/boot.nft` on every apply; this loads it after fw4 (19) and netifd (20), `nft -c`-validated. Four state checks refuse to arm (no/empty/invalid file, missing `shaterd`, no `S??shater` rc-link, readable UCI saying `enabled≠1`) — asked ON THE WAY UP, deliberately not recorded on the way down. Hooks `forward` only, so SSH/LuCI/panel stay reachable. `stop()` is a NO-OP. Operator-facing writeup: `INSTALL.md` §4.
|
||||
- **uci-defaults/30_shater-core**: seed `rt_tables` (8192 shater), `mkdir /etc/shater`, seed the `shater_l3` fw4 zone + `lan→shater_l3` forwarding (named sections, `list device 'shater-l3*'`) and migrate a legacy exact-name entry to the wildcard, run `shaterd migrate`, apply sysctl, then a DETACHED bring-up (enable+restart `shater`/`shater-cron`, enable `shater-armor`, conditional `firewall reload`) — detached because an inline init call inside an apk/opkg transaction deadlocks on procd's flock. Also still seeds three `config preset` sections, which nothing parses (see the schema note above); those calls should go.
|
||||
- **hotplug.d/iface/99-shater**: ifup/ifdown → debounced (2s) `reconcile` (netifd wipes ip rules on reload). Guarded by enabled + ACTIVE_FLAG.
|
||||
- **sysctl.d/99-shater.conf**: `ip_forward=1`, `rp_filter=0` (all+default), `lo.route_localnet=1`, `lo.accept_local=1`, `all.src_valid_mark=1`, `ipv6.all.forwarding=1`.
|
||||
|
||||
|
||||
@@ -49,7 +49,7 @@ build new logic in the `shater/`, `panel/`, `openwrt/` overlay.
|
||||
the gate: fail-closed forward drop (4f618140), engine apply-swap close-first
|
||||
fallback (9b6b9406), DNS hijack-dns per D14 (86194ce6).
|
||||
|
||||
## Phase 2b — DPI-bypass egress = ByeDPI (D13)
|
||||
## Phase 2b — DPI-bypass egress = ByeDPI (D13) ✅ DONE
|
||||
- The one external desync tool is **ByeDPI (ciadpi)** — chosen over zapret because
|
||||
it *is* a SOCKS egress (fits shater's "routing picks the egress" model with zero
|
||||
packet-plane conflict); zapret is explicitly rejected (see D13).
|
||||
@@ -86,7 +86,7 @@ build new logic in the `shater/`, `panel/`, `openwrt/` overlay.
|
||||
well-known lists (StevenBlack/OISD/AdGuard).
|
||||
- **Gate:** ad/tracker domains blocked network-wide; big list loads fast; RAM sane.
|
||||
|
||||
## Phase 5 — Statistics (per-domain / client / device)
|
||||
## Phase 5 — Statistics (per-domain / client / device) ✅ DONE
|
||||
- Stats aggregator consuming the engine's DNS/routing/stats observability + nft
|
||||
counters: top domains, allowed vs blocked, per-device breakdown, timelines,
|
||||
per-node/per-rule traffic, live query log with one-click block.
|
||||
|
||||
@@ -46,7 +46,7 @@ require (
|
||||
github.com/sagernet/sing v0.8.12-0.20260702081104-2ded2af32d3d
|
||||
github.com/sagernet/sing-cloudflared v0.1.3-0.20260706062323-d9787e794aa3
|
||||
github.com/sagernet/sing-mux v0.3.5
|
||||
github.com/sagernet/sing-quic v0.6.2-0.20260525051024-9467ede27fb7
|
||||
github.com/sagernet/sing-quic v0.6.4-0.20260709034545-e23afe1172dc
|
||||
github.com/sagernet/sing-shadowsocks v0.2.8
|
||||
github.com/sagernet/sing-shadowsocks2 v0.2.1
|
||||
github.com/sagernet/sing-shadowtls v0.2.1
|
||||
@@ -104,14 +104,20 @@ require (
|
||||
github.com/google/btree v1.1.3 // indirect
|
||||
github.com/google/go-cmp v0.7.0 // indirect
|
||||
github.com/google/go-querystring v1.1.0 // indirect
|
||||
github.com/google/gopacket v1.1.19 // indirect
|
||||
github.com/google/nftables v0.2.1-0.20240414091927-5e242ec57806 // indirect
|
||||
github.com/google/uuid v1.6.0 // indirect
|
||||
github.com/hashicorp/yamux v0.1.2 // indirect
|
||||
github.com/hdevalence/ed25519consensus v0.2.0 // indirect
|
||||
github.com/huin/goupnp v1.2.0 // indirect
|
||||
github.com/inconshreveable/mousetrap v1.1.0 // indirect
|
||||
github.com/jackpal/go-nat-pmp v1.0.2 // indirect
|
||||
github.com/klauspost/compress v1.18.0 // indirect
|
||||
github.com/klauspost/cpuid/v2 v2.3.0 // indirect
|
||||
github.com/koron/go-ssdp v0.0.4 // indirect
|
||||
github.com/kr/fs v0.1.0 // indirect
|
||||
github.com/libp2p/go-nat v1.0.1-0.20250821073202-01afc089f138 // indirect
|
||||
github.com/libp2p/go-netroute v0.2.1 // indirect
|
||||
github.com/mdlayher/socket v0.5.1 // indirect
|
||||
github.com/mitchellh/go-ps v1.0.0 // indirect
|
||||
github.com/philhofer/fwd v1.2.0 // indirect
|
||||
|
||||
@@ -101,6 +101,8 @@ github.com/google/go-cmp v0.7.0 h1:wk8382ETsv4JYUZwIsn6YpYiWiBsYLSJiTsyBybVuN8=
|
||||
github.com/google/go-cmp v0.7.0/go.mod h1:pXiqmnSA92OHEEa9HXL2W4E7lf9JzCmGVUdgjX3N/iU=
|
||||
github.com/google/go-querystring v1.1.0 h1:AnCroh3fv4ZBgVIf1Iwtovgjaw/GiKJo8M8yD/fhyJ8=
|
||||
github.com/google/go-querystring v1.1.0/go.mod h1:Kcdr2DB4koayq7X8pmAG4sNG59So17icRSOU623lUBU=
|
||||
github.com/google/gopacket v1.1.19 h1:ves8RnFZPGiFnTS0uPQStjwru6uO6h+nlr9j6fL7kF8=
|
||||
github.com/google/gopacket v1.1.19/go.mod h1:iJ8V8n6KS+z2U1A8pUwu8bW5SyEMkXJB8Yo/Vo+TKTo=
|
||||
github.com/google/nftables v0.2.1-0.20240414091927-5e242ec57806 h1:wG8RYIyctLhdFk6Vl1yPGtSRtwGpVkWyZww1OCil2MI=
|
||||
github.com/google/nftables v0.2.1-0.20240414091927-5e242ec57806/go.mod h1:Beg6V6zZ3oEn0JuiUQ4wqwuyqqzasOltcoXPtgLbFp4=
|
||||
github.com/google/uuid v1.6.0 h1:NIvaJDMOsjHA8n1jAhLSgzrAzy1Hgr+hNrb57e+94F0=
|
||||
@@ -109,10 +111,14 @@ github.com/hashicorp/yamux v0.1.2 h1:XtB8kyFOyHXYVFnwT5C3+Bdo8gArse7j2AQ0DA0Uey8
|
||||
github.com/hashicorp/yamux v0.1.2/go.mod h1:C+zze2n6e/7wshOZep2A70/aQU6QBRWJO/G6FT1wIns=
|
||||
github.com/hdevalence/ed25519consensus v0.2.0 h1:37ICyZqdyj0lAZ8P4D1d1id3HqbbG1N3iBb1Tb4rdcU=
|
||||
github.com/hdevalence/ed25519consensus v0.2.0/go.mod h1:w3BHWjwJbFU29IRHL1Iqkw3sus+7FctEyM4RqDxYNzo=
|
||||
github.com/huin/goupnp v1.2.0 h1:uOKW26NG1hsSSbXIZ1IR7XP9Gjd1U8pnLaCMgntmkmY=
|
||||
github.com/huin/goupnp v1.2.0/go.mod h1:gnGPsThkYa7bFi/KWmEysQRf48l2dvR5bxr2OFckNX8=
|
||||
github.com/inconshreveable/mousetrap v1.1.0 h1:wN+x4NVGpMsO7ErUn/mUI3vEoE6Jt13X2s0bqwp9tc8=
|
||||
github.com/inconshreveable/mousetrap v1.1.0/go.mod h1:vpF70FUmC8bwa3OWnCshd2FqLfsEA9PFc4w1p2J65bw=
|
||||
github.com/insomniacslk/dhcp v0.0.0-20260220084031-5adc3eb26f91 h1:u9i04mGE3iliBh0EFuWaKsmcwrLacqGmq1G3XoaM7gY=
|
||||
github.com/insomniacslk/dhcp v0.0.0-20260220084031-5adc3eb26f91/go.mod h1:qfvBmyDNp+/liLEYWRvqny/PEz9hGe2Dz833eXILSmo=
|
||||
github.com/jackpal/go-nat-pmp v1.0.2 h1:KzKSgb7qkJvOUTqYl9/Hg/me3pWgBmERKrTGD7BdWus=
|
||||
github.com/jackpal/go-nat-pmp v1.0.2/go.mod h1:QPH045xvCAeXUZOxsnwmrtiCoxIr9eob+4orBN1SBKc=
|
||||
github.com/jessevdk/go-flags v1.4.0/go.mod h1:4FA24M0QyGHXBuZZK/XkWh8h0e1EYbRYJSGM75WSRxI=
|
||||
github.com/jsimonetti/rtnetlink v1.4.0 h1:Z1BF0fRgcETPEa0Kt0MRk3yV5+kF1FWTni6KUFKrq2I=
|
||||
github.com/jsimonetti/rtnetlink v1.4.0/go.mod h1:5W1jDvWdnthFJ7fxYX1GMK07BUpI4oskfOqvPteYS6E=
|
||||
@@ -122,6 +128,8 @@ github.com/klauspost/compress v1.18.0 h1:c/Cqfb0r+Yi+JtIEq73FWXVkRonBlf0CRNYc8Zt
|
||||
github.com/klauspost/compress v1.18.0/go.mod h1:2Pp+KzxcywXVXMr50+X0Q/Lsb43OQHYWRCY2AiWywWQ=
|
||||
github.com/klauspost/cpuid/v2 v2.3.0 h1:S4CRMLnYUhGeDFDqkGriYKdfoFlDnMtqTiI/sFzhA9Y=
|
||||
github.com/klauspost/cpuid/v2 v2.3.0/go.mod h1:hqwkgyIinND0mEev00jJYCxPNVRVXFQeu1XKlok6oO0=
|
||||
github.com/koron/go-ssdp v0.0.4 h1:1IDwrghSKYM7yLf7XCzbByg2sJ/JcNOZRXS2jczTwz0=
|
||||
github.com/koron/go-ssdp v0.0.4/go.mod h1:oDXq+E5IL5q0U8uSBcoAXzTzInwy5lEgC91HoKtbmZk=
|
||||
github.com/kr/fs v0.1.0 h1:Jskdu9ieNAYnjxsi0LbQp1ulIKZV1LAFgK1tWhpZgl8=
|
||||
github.com/kr/fs v0.1.0/go.mod h1:FFnZGqtBN9Gxj7eW1uZ42v5BccTP0vu6NEaFoC2HwRg=
|
||||
github.com/kylelemons/godebug v1.1.0 h1:RPNrshWIDI6G2gRW9EHilWtl7Z6Sb1BR0xunSBf0SNc=
|
||||
@@ -138,6 +146,10 @@ github.com/libdns/cloudflare v0.2.2 h1:XWHv+C1dDcApqazlh08Q6pjytYLgR2a+Y3xrXFu0v
|
||||
github.com/libdns/cloudflare v0.2.2/go.mod h1:w9uTmRCDlAoafAsTPnn2nJ0XHK/eaUMh86DUk8BWi60=
|
||||
github.com/libdns/libdns v1.1.1 h1:wPrHrXILoSHKWJKGd0EiAVmiJbFShguILTg9leS/P/U=
|
||||
github.com/libdns/libdns v1.1.1/go.mod h1:4Bj9+5CQiNMVGf87wjX4CY3HQJypUHRuLvlsfsZqLWQ=
|
||||
github.com/libp2p/go-nat v1.0.1-0.20250821073202-01afc089f138 h1:YohuNPT/1k3VcThCQlBZ43PCPWPfMRS1zcxWBF2SLK8=
|
||||
github.com/libp2p/go-nat v1.0.1-0.20250821073202-01afc089f138/go.mod h1:TXQg5tfSy+bUjnhT5728j5j/MBj7keIYqqZ1+8k/ui8=
|
||||
github.com/libp2p/go-netroute v0.2.1 h1:V8kVrpD8GK0Riv15/7VN6RbUQ3URNZVosw7H2v9tksU=
|
||||
github.com/libp2p/go-netroute v0.2.1/go.mod h1:hraioZr0fhBjG0ZRXJJ6Zj2IVEVNx6tDTFQfSmcq7mQ=
|
||||
github.com/logrusorgru/aurora v2.0.3+incompatible h1:tOpm7WcpBTn4fjmVfgpQq0EfczGlG91VSDkswnjF5A8=
|
||||
github.com/logrusorgru/aurora v2.0.3+incompatible/go.mod h1:7rIyQOR62GCctdiQpZ/zOJlFyk6y+94wXzv6RNZgaR4=
|
||||
github.com/mdlayher/netlink v1.9.0 h1:G8+GLq2x3v4D4MVIqDdNUhTUC7TKiCy/6MDkmItfKco=
|
||||
@@ -264,8 +276,8 @@ github.com/sagernet/sing-cloudflared v0.1.3-0.20260706062323-d9787e794aa3 h1:3y6
|
||||
github.com/sagernet/sing-cloudflared v0.1.3-0.20260706062323-d9787e794aa3/go.mod h1:XEqEDYRCAYLaoPjZ1ifVWJg5iWAJHL2gOAXe/PM28Cg=
|
||||
github.com/sagernet/sing-mux v0.3.5 h1:RHnhVEc+SFqkrK4xMygYjDwwLhzp2Bj3lztSukONfhI=
|
||||
github.com/sagernet/sing-mux v0.3.5/go.mod h1:QvlKMyNBNrQoyX4x+gq028uPbLM2XeRpWtDsWBJbFSk=
|
||||
github.com/sagernet/sing-quic v0.6.2-0.20260525051024-9467ede27fb7 h1:hFLPJ21uNZSbRnzhOKz4Zv0b4F93mpDorWyN93BeRcM=
|
||||
github.com/sagernet/sing-quic v0.6.2-0.20260525051024-9467ede27fb7/go.mod h1:+oqD54aHel4ALKkp1hVXWCgLU/EjLojvm6AUzDfvj0I=
|
||||
github.com/sagernet/sing-quic v0.6.4-0.20260709034545-e23afe1172dc h1:zdc0fj4JdAdgAmQIoh7ZF+B/wPTEF2X75lYDqTmvlaw=
|
||||
github.com/sagernet/sing-quic v0.6.4-0.20260709034545-e23afe1172dc/go.mod h1:9k+dzGsWMttUGldBzq3dU792YHXzW6NgfbOGltnXq+0=
|
||||
github.com/sagernet/sing-shadowsocks v0.2.8 h1:PURj5PRoAkqeHh2ZW205RWzN9E9RtKCVCzByXruQWfE=
|
||||
github.com/sagernet/sing-shadowsocks v0.2.8/go.mod h1:lo7TWEMDcN5/h5B8S0ew+r78ZODn6SwVaFhvB6H+PTI=
|
||||
github.com/sagernet/sing-shadowsocks2 v0.2.1 h1:dWV9OXCeFPuYGHb6IRqlSptVnSzOelnqqs2gQ2/Qioo=
|
||||
@@ -360,6 +372,8 @@ go4.org/mem v0.0.0-20240501181205-ae6ca9944745 h1:Tl++JLUCe4sxGu8cTpDzRLd3tN7US4
|
||||
go4.org/mem v0.0.0-20240501181205-ae6ca9944745/go.mod h1:reUoABIJ9ikfM5sgtSF3Wushcza7+WeD01VB9Lirh3g=
|
||||
go4.org/netipx v0.0.0-20231129151722-fdeea329fbba h1:0b9z3AuHCjxk0x/opv64kcgZLBseWJUpBw5I82+2U4M=
|
||||
go4.org/netipx v0.0.0-20231129151722-fdeea329fbba/go.mod h1:PLyyIXexvUFg3Owu6p/WfdlivPbZJsZdgWZlrGope/Y=
|
||||
golang.org/x/crypto v0.0.0-20190308221718-c2843e01d9a2/go.mod h1:djNgcEr1/C05ACkg1iLfiJU5Ep61QUkGW8qpdssI0+w=
|
||||
golang.org/x/crypto v0.0.0-20191011191535-87dc89f01550/go.mod h1:yigFU9vqHzYiE8UmvKecakEJjdnWj3jj499lnFckfCI=
|
||||
golang.org/x/crypto v0.0.0-20210513164829-c07d793c2f9a/go.mod h1:P+XmwS30IXTQdn5tA2iutPOUgjI07+tq3H3K9MVA1s8=
|
||||
golang.org/x/crypto v0.48.0 h1:/VRzVqiRSggnhY7gNRxPauEQ5Drw9haKdM0jqfcCFts=
|
||||
golang.org/x/crypto v0.48.0/go.mod h1:r0kV5h3qnFPlQnBSrULhlsRfryS2pmewsg+XfMgkVos=
|
||||
@@ -367,17 +381,24 @@ golang.org/x/exp v0.0.0-20251219203646-944ab1f22d93 h1:fQsdNF2N+/YewlRZiricy4P1i
|
||||
golang.org/x/exp v0.0.0-20251219203646-944ab1f22d93/go.mod h1:EPRbTFwzwjXj9NpYyyrvenVh9Y+GFeEvMNh7Xuz7xgU=
|
||||
golang.org/x/image v0.27.0 h1:C8gA4oWU/tKkdCfYT6T2u4faJu3MeNS5O8UPWlPF61w=
|
||||
golang.org/x/image v0.27.0/go.mod h1:xbdrClrAUway1MUTEZDq9mz/UpRwYAkFFNUslZtcB+g=
|
||||
golang.org/x/lint v0.0.0-20200302205851-738671d3881b/go.mod h1:3xt1FjdF8hUf6vQPIChWIBhFzV8gjjsPE/fR3IyQdNY=
|
||||
golang.org/x/mod v0.1.1-0.20191105210325-c90efee705ee/go.mod h1:QqPTAvyqsEbceGzBzNggFXnrqF1CaUcvgkdR5Ot7KZg=
|
||||
golang.org/x/mod v0.33.0 h1:tHFzIWbBifEmbwtGz65eaWyGiGZatSrT9prnU8DbVL8=
|
||||
golang.org/x/mod v0.33.0/go.mod h1:swjeQEj+6r7fODbD2cqrnje9PnziFuw4bmLbBZFrQ5w=
|
||||
golang.org/x/net v0.0.0-20190404232315-eb5bcb51f2a3/go.mod h1:t9HGtf8HONx5eT2rtn7q6eTqICYqUVnKs3thJo3Qplg=
|
||||
golang.org/x/net v0.0.0-20190620200207-3b0461eec859/go.mod h1:z5CRVTTTmAJ677TzLLGU+0bjPO0LkuOLi4/5GtJWs/s=
|
||||
golang.org/x/net v0.0.0-20210226172049-e18ecbb05110/go.mod h1:m0MpNAwzfU5UDzcl9v0D8zg8gWTRqZa9RBIspLL5mdg=
|
||||
golang.org/x/net v0.0.0-20210525063256-abc453219eb5/go.mod h1:9nx3DQGgdP8bBQD5qxJ1jj9UTztislL4KSBs9R2vV5Y=
|
||||
golang.org/x/net v0.50.0 h1:ucWh9eiCGyDR3vtzso0WMQinm2Dnt8cFMuQa9K33J60=
|
||||
golang.org/x/net v0.50.0/go.mod h1:UgoSli3F/pBgdJBHCTc+tp3gmrU4XswgGRgtnwWTfyM=
|
||||
golang.org/x/oauth2 v0.34.0 h1:hqK/t4AKgbqWkdkcAeI8XLmbK+4m4G5YeQRrmiotGlw=
|
||||
golang.org/x/oauth2 v0.34.0/go.mod h1:lzm5WQJQwKZ3nwavOZ3IS5Aulzxi68dUSgRHujetwEA=
|
||||
golang.org/x/sync v0.0.0-20190423024810-112230192c58/go.mod h1:RxMgew5VJxzue5/jJTE5uejpjVlOe/izrB70Jof72aM=
|
||||
golang.org/x/sync v0.0.0-20210220032951-036812b2e83c/go.mod h1:RxMgew5VJxzue5/jJTE5uejpjVlOe/izrB70Jof72aM=
|
||||
golang.org/x/sync v0.19.0 h1:vV+1eWNmZ5geRlYjzm2adRgW2/mcpevXNg50YZtPCE4=
|
||||
golang.org/x/sync v0.19.0/go.mod h1:9KTHXmSnoGruLpwFjVSX0lNNA75CykiMECbovNTZqGI=
|
||||
golang.org/x/sys v0.0.0-20190215142949-d0b11bdaac8a/go.mod h1:STP8DvDyc/dI5b8T5hshtkjS+E42TnysNCUPdjciGhY=
|
||||
golang.org/x/sys v0.0.0-20190412213103-97732733099d/go.mod h1:h1NjWce9XRLGQEsW7wpKNCjG9DtNlClVuFLEZdDNbEs=
|
||||
golang.org/x/sys v0.0.0-20200217220822-9197077df867/go.mod h1:h1NjWce9XRLGQEsW7wpKNCjG9DtNlClVuFLEZdDNbEs=
|
||||
golang.org/x/sys v0.0.0-20200728102440-3e129f6d46b1/go.mod h1:h1NjWce9XRLGQEsW7wpKNCjG9DtNlClVuFLEZdDNbEs=
|
||||
golang.org/x/sys v0.0.0-20201119102817-f84b799fce68/go.mod h1:h1NjWce9XRLGQEsW7wpKNCjG9DtNlClVuFLEZdDNbEs=
|
||||
@@ -389,6 +410,7 @@ golang.org/x/sys v0.41.0/go.mod h1:OgkHotnGiDImocRcuBABYBEXf8A9a87e/uXjp9XT3ks=
|
||||
golang.org/x/term v0.0.0-20201126162022-7de9c90e9dd1/go.mod h1:bj7SfCRtBDWHUb9snDiAeCFNEtKQo2Wmx5Cou7ajbmo=
|
||||
golang.org/x/term v0.40.0 h1:36e4zGLqU4yhjlmxEaagx2KuYbJq3EwY8K943ZsHcvg=
|
||||
golang.org/x/term v0.40.0/go.mod h1:w2P8uVp06p2iyKKuvXIm7N/y0UCRt3UfJTfZ7oOpglM=
|
||||
golang.org/x/text v0.3.0/go.mod h1:NqM8EUOU14njkJ3fqMW+pc6Ldnwhi/IjpwHt7yyuwOQ=
|
||||
golang.org/x/text v0.3.3/go.mod h1:5Zoc/QRtKVWzQhOtBMvqHzDpF6irO9z98xDceosuGiQ=
|
||||
golang.org/x/text v0.3.6/go.mod h1:5Zoc/QRtKVWzQhOtBMvqHzDpF6irO9z98xDceosuGiQ=
|
||||
golang.org/x/text v0.34.0 h1:oL/Qq0Kdaqxa1KbNeMKwQq0reLCCaFtqu2eNuSeNHbk=
|
||||
@@ -396,8 +418,10 @@ golang.org/x/text v0.34.0/go.mod h1:homfLqTYRFyVYemLBFl5GgL/DWEiH5wcsQ5gSh1yziA=
|
||||
golang.org/x/time v0.11.0 h1:/bpjEDfN9tkoN/ryeYHnv5hcMlc8ncjMcM4XBk5NWV0=
|
||||
golang.org/x/time v0.11.0/go.mod h1:CDIdPxbZBQxdj6cxyCIdrNogrJKMJ7pr37NYpMcMDSg=
|
||||
golang.org/x/tools v0.0.0-20180917221912-90fa682c2a6e/go.mod h1:n7NCudcB/nEzxVGmLbDWY5pfWTLqBcC2KZ6jyYvM4mQ=
|
||||
golang.org/x/tools v0.0.0-20200130002326-2f3ba24bd6e7/go.mod h1:TB2adYChydJhpapKDTa4BR/hXlZSLoq2Wpct/0txZ28=
|
||||
golang.org/x/tools v0.42.0 h1:uNgphsn75Tdz5Ji2q36v/nsFSfR/9BRFvqhGBaJGd5k=
|
||||
golang.org/x/tools v0.42.0/go.mod h1:Ma6lCIwGZvHK6XtgbswSoWroEkhugApmsXyrUmBhfr0=
|
||||
golang.org/x/xerrors v0.0.0-20191011141410-1b5146add898/go.mod h1:I/5z698sn9Ka8TeJc9MKroUUfqBBauWjQqLJ2OPfmY0=
|
||||
golang.org/x/xerrors v0.0.0-20191204190536-9bdfabe68543/go.mod h1:I/5z698sn9Ka8TeJc9MKroUUfqBBauWjQqLJ2OPfmY0=
|
||||
golang.org/x/xerrors v0.0.0-20200804184101-5ec99f83aff1 h1:go1bK/D/BFZV2I8cIQd1NKEZ+0owSTG1fDTci4IqFcE=
|
||||
golang.org/x/xerrors v0.0.0-20200804184101-5ec99f83aff1/go.mod h1:I/5z698sn9Ka8TeJc9MKroUUfqBBauWjQqLJ2OPfmY0=
|
||||
|
||||
@@ -26,15 +26,32 @@ var callMintToken = rpc.declare({
|
||||
expect: { '': {} }
|
||||
});
|
||||
|
||||
// led renders a small status dot: state is 'good' | 'warn' | 'bad'.
|
||||
// LED palette. 'unknown' is an UNLIT socket — never amber and never green.
|
||||
// Amber is this page's "degraded", and there is nothing to be degraded about
|
||||
// when no reading has arrived; green on a missing reading is how the panel used
|
||||
// to claim health it had not measured (see panel/src/planeState.ts, which says
|
||||
// the same thing and is the wording this page is kept in step with).
|
||||
var LED_COLORS = {
|
||||
good: '#37b24d',
|
||||
warn: '#f59f00',
|
||||
bad: '#e03131',
|
||||
unknown: '#6b6b6b'
|
||||
};
|
||||
|
||||
// led renders a small status dot: state is 'good' | 'warn' | 'bad' | 'unknown'.
|
||||
// The dot is decorative — every row states its condition in words beside it — so
|
||||
// it is hidden from assistive tech rather than being the only carrier of meaning.
|
||||
function led(state) {
|
||||
var color = state === 'good' ? '#37b24d'
|
||||
: state === 'warn' ? '#f59f00'
|
||||
: '#e03131';
|
||||
// Closed positive list. An unrecognised state resolves to UNKNOWN, never to
|
||||
// green: an open default here is exactly how a state nobody thought about
|
||||
// ends up painted healthy.
|
||||
var color = Object.prototype.hasOwnProperty.call(LED_COLORS, state)
|
||||
? LED_COLORS[state] : LED_COLORS.unknown;
|
||||
var glow = (color === LED_COLORS.unknown) ? '' : ';box-shadow:0 0 5px ' + color;
|
||||
return E('span', {
|
||||
'aria-hidden': 'true',
|
||||
'style': 'display:inline-block;width:.72em;height:.72em;border-radius:50%;' +
|
||||
'margin-right:.6em;vertical-align:-.05em;background:' + color +
|
||||
';box-shadow:0 0 5px ' + color
|
||||
'margin-right:.6em;vertical-align:-.05em;background:' + color + glow
|
||||
});
|
||||
}
|
||||
|
||||
@@ -49,63 +66,251 @@ function row(state, label, value) {
|
||||
]);
|
||||
}
|
||||
|
||||
// statusRows maps the shaterd status object to LED rows. An empty object (the
|
||||
// ubus call failed / daemon down) degrades every row to a "down" reading.
|
||||
function statusRows(st) {
|
||||
st = st || {};
|
||||
var down = (st.running !== true);
|
||||
// ---------------------------------------------------------------------------
|
||||
// Pure state derivation — no DOM below this line until statusRows().
|
||||
//
|
||||
// statusReadout() maps a shaterd status object to a list of
|
||||
// { state, label, value } descriptors. It is deliberately free of E()/DOM so it
|
||||
// can be run against recorded fixtures offline; tests/status-readout.test.js
|
||||
// does exactly that for the three cases this page has to tell apart.
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
// PLANES is the closed set of values a LIVE Applier.Status() can put in `plane`
|
||||
// (shater/apply/apply.go: "full" | "hold" | "none"). It is also how this page
|
||||
// tells a live daemon from a dead one — see daemonState().
|
||||
var PLANES = { full: true, hold: true, none: true };
|
||||
|
||||
// daemonState — is the shaterd PROCESS answering?
|
||||
//
|
||||
// 'up' — a live Applier produced this status.
|
||||
// 'down' — proven not: `shaterd status` printed its OFFLINE STUB.
|
||||
// 'unknown' — no usable answer, or an answer from a daemon older than `plane`.
|
||||
//
|
||||
// `running` MUST NOT be used for this. It changed meaning on 2026-07-26
|
||||
// (a8970b8ac): it used to be a hardcoded true, and is now the ENGINE's liveness
|
||||
// (apply.go `Running: engineUp`). A daemon that is perfectly alive with a dead
|
||||
// engine reports running=false — and this page used to answer that with a red
|
||||
// "Daemon: not running", the advice "start the Shater service first", and a
|
||||
// DISABLED button to the one place the config can be fixed. The holding plane
|
||||
// keeps management reachable on purpose (shater/netplane/nft.go); LuCI was the
|
||||
// only thing taking that guarantee away.
|
||||
//
|
||||
// Nor is "the ubus call returned" sufficient, which is the trap here: the rpcd
|
||||
// plugin shells out to `shaterd status`, and that command EXITS 0 WITH A
|
||||
// FABRICATED STATUS when the daemon is unreachable (cmd/shaterd/main.go,
|
||||
// cmdStatus offline stub). The stub is the apply.Status zero value plus a UCI
|
||||
// read, so it carries enabled/table/kill_switch/panel_port but leaves `plane` at
|
||||
// "" — a value no live daemon ever emits, because Status() always assigns one of
|
||||
// the three words. So a known plane word is the one positive proof on the wire
|
||||
// that a daemon answered, and an explicit empty one is positive proof that none
|
||||
// did.
|
||||
//
|
||||
// Everything else is unknown and is painted as unknown: {} from a failed ubus
|
||||
// call, {"error":...} from the plugin (which is ALSO what a live-but-wedged
|
||||
// daemon produces — cmdStatus prints nothing and exits 1 on a control-socket
|
||||
// timeout, so "wedged" must not be reported as "dead"), and a status from a
|
||||
// daemon predating the `plane` field.
|
||||
function daemonState(st) {
|
||||
if (!st || typeof st !== 'object')
|
||||
return 'unknown';
|
||||
if (typeof st.plane === 'string' && PLANES[st.plane] === true)
|
||||
return 'up';
|
||||
if (st.plane === '')
|
||||
return 'down';
|
||||
return 'unknown';
|
||||
}
|
||||
|
||||
// engineState — is a sing-box instance actually started?
|
||||
//
|
||||
// The engine lives INSIDE the shaterd process, so a dead daemon is a dead engine
|
||||
// and this page may say so without guessing. With the daemon up, `engine_running`
|
||||
// is the self-documenting field and `running` carries the same fact by
|
||||
// construction; either may prove a NEGATIVE, and a negative always wins. Neither
|
||||
// asserting anything leaves 'unknown'.
|
||||
function engineState(st) {
|
||||
var d = daemonState(st);
|
||||
if (d === 'down')
|
||||
return 'down';
|
||||
if (d === 'unknown')
|
||||
return 'unknown';
|
||||
if (st.running === false || st.engine_running === false)
|
||||
return 'down';
|
||||
if (st.running === true || st.engine_running === true)
|
||||
return 'up';
|
||||
return 'unknown';
|
||||
}
|
||||
|
||||
function mk(state, label, value) {
|
||||
return { state: state, label: label, value: value };
|
||||
}
|
||||
|
||||
function statusReadout(st) {
|
||||
st = (st && typeof st === 'object') ? st : {};
|
||||
|
||||
var dstate = daemonState(st);
|
||||
var estate = engineState(st);
|
||||
var traffic = (st.traffic && typeof st.traffic === 'object') ? st.traffic : {};
|
||||
var rows = [];
|
||||
|
||||
// Daemon process itself.
|
||||
rows.push(row(
|
||||
down ? 'bad' : 'good',
|
||||
_('Daemon (shaterd)'),
|
||||
down ? _('not running') : _('running')
|
||||
));
|
||||
// --- The shaterd process itself. ------------------------------------------
|
||||
// Its own liveness is not a field; it is whether a live daemon answered.
|
||||
if (dstate === 'up')
|
||||
rows.push(mk('good', _('Daemon (shaterd)'), _('responding')));
|
||||
else if (dstate === 'down')
|
||||
rows.push(mk('bad', _('Daemon (shaterd)'),
|
||||
_('not responding — start the Shater service')));
|
||||
else
|
||||
rows.push(mk('unknown', _('Daemon (shaterd)'),
|
||||
_('no usable answer — state unknown')));
|
||||
|
||||
// Desired state: globals.enabled in UCI.
|
||||
rows.push(row(
|
||||
st.enabled ? 'good' : 'warn',
|
||||
_('Service enabled'),
|
||||
st.enabled ? _('enabled') : _('inert (disabled)')
|
||||
));
|
||||
// --- The engine (sing-box) inside it. -------------------------------------
|
||||
if (estate === 'up')
|
||||
rows.push(mk('good', _('Engine (sing-box)'), _('running')));
|
||||
else if (estate === 'down' && dstate === 'down')
|
||||
rows.push(mk('bad', _('Engine (sing-box)'),
|
||||
_('stopped — it runs inside shaterd, which is not answering')));
|
||||
else if (estate === 'down')
|
||||
rows.push(mk('bad', _('Engine (sing-box)'),
|
||||
_('stopped — the daemon is up but no instance is running')));
|
||||
else
|
||||
rows.push(mk('unknown', _('Engine (sing-box)'), _('not reported')));
|
||||
|
||||
// Interception raised (ACTIVE_FLAG present after a successful enabled apply).
|
||||
rows.push(row(
|
||||
st.active ? 'good' : (st.enabled ? 'warn' : 'bad'),
|
||||
_('Interception'),
|
||||
st.active ? _('active') : _('inactive')
|
||||
));
|
||||
// --- Desired state: globals.enabled in UCI. -------------------------------
|
||||
if (st.enabled === true)
|
||||
rows.push(mk('good', _('Service enabled'), _('enabled')));
|
||||
else if (st.enabled === false)
|
||||
rows.push(mk('warn', _('Service enabled'), _('inert (disabled)')));
|
||||
else
|
||||
rows.push(mk('unknown', _('Service enabled'), _('not reported')));
|
||||
|
||||
// Data plane: the `inet shater` nft table is loaded.
|
||||
rows.push(row(
|
||||
st.table ? 'good' : (st.enabled ? 'warn' : 'bad'),
|
||||
_('Data plane'),
|
||||
st.table ? _('nft table inet shater loaded') : _('not loaded')
|
||||
));
|
||||
// --- The ACTIVE_FLAG latch. -----------------------------------------------
|
||||
// NOT a health signal, and this row must never read as one. apply.go states
|
||||
// the contract: it is the "the service is meant to be running" latch that
|
||||
// gates hotplug and cron; it is raised by a successful enabled apply and
|
||||
// cleared only by teardown, so it STAYS UP while the engine is down and the
|
||||
// fail-closed holding plane is blocking the LAN — deliberately, because
|
||||
// clearing it would switch off the very cron reconcile that brings the engine
|
||||
// back. This page used to render it as "Interception: active", in green, over
|
||||
// a dead engine and a blocked LAN. The lamp now reports only whether the
|
||||
// latch AGREES with globals.enabled.
|
||||
if (typeof st.active !== 'boolean')
|
||||
rows.push(mk('unknown', _('Service latch'), _('not reported')));
|
||||
else if (st.active)
|
||||
rows.push(mk(st.enabled === true ? 'good' : 'warn', _('Service latch'),
|
||||
_('raised — the service is meant to be running')));
|
||||
else
|
||||
rows.push(mk(st.enabled === false ? 'good' : 'warn', _('Service latch'),
|
||||
_('cleared — the service is torn down')));
|
||||
|
||||
// Kill-switch: fail-closed ("closed") is the safe posture; "open" leaks
|
||||
// LAN→WAN if the engine goes down. Unknown (older daemon) degrades to warn.
|
||||
var ks = st.kill_switch;
|
||||
rows.push(row(
|
||||
ks === 'closed' ? 'good' : 'warn',
|
||||
_('Kill-switch'),
|
||||
ks === 'closed' ? _('closed (fail-closed)')
|
||||
: ks === 'open' ? _('open (leaky)')
|
||||
: _('unknown')
|
||||
));
|
||||
// --- What is loaded in the kernel right now. ------------------------------
|
||||
// The row this page was missing. `plane` distinguishes the working ruleset
|
||||
// from the FAIL-CLOSED HOLDING PLANE, which `table` cannot: `table` is a bare
|
||||
// existence check, so a held LAN and a working one look identical through it.
|
||||
switch (st.plane) {
|
||||
case 'full':
|
||||
// Deliberately mechanical wording. "full" means the table, the policy
|
||||
// routing and the engine are all in place — it does NOT mean traffic is
|
||||
// tunnelled. That claim belongs to the traffic verdict below.
|
||||
rows.push(mk('good', _('Traffic plane'),
|
||||
_('full — ruleset, routing and engine are all installed')));
|
||||
break;
|
||||
case 'hold':
|
||||
rows.push(mk('bad', _('Traffic plane'),
|
||||
_('hold — the engine is down and LAN→WAN forwarding is BLOCKED')));
|
||||
break;
|
||||
case 'none':
|
||||
rows.push(mk(st.kill_switch === 'closed' ? 'bad' : 'warn', _('Traffic plane'),
|
||||
st.kill_switch === 'closed'
|
||||
? _('none — nothing is installed; traffic reaches the WAN unprotected')
|
||||
: _('none — no data plane is installed')));
|
||||
break;
|
||||
default:
|
||||
rows.push(mk('unknown', _('Traffic plane'),
|
||||
dstate === 'down'
|
||||
? _('not reported — no daemon answered')
|
||||
: _('not reported by this daemon')));
|
||||
break;
|
||||
}
|
||||
|
||||
// Running engine config hash ("" when the engine is not started).
|
||||
rows.push(row(
|
||||
st.hash ? 'good' : 'warn',
|
||||
_('Config hash'),
|
||||
st.hash ? st.hash : '—'
|
||||
));
|
||||
// --- Where the traffic goes under the running config. ---------------------
|
||||
// Separate from the plane on purpose: a router with one `default -> direct`
|
||||
// rule has a fully installed plane and sends every packet out the plain WAN
|
||||
// with its real address.
|
||||
switch (traffic.verdict) {
|
||||
case 'tunnel':
|
||||
rows.push(mk('good', _('Traffic verdict'), _('tunnel — unmatched traffic is proxied')));
|
||||
break;
|
||||
case 'split':
|
||||
rows.push(mk('warn', _('Traffic verdict'),
|
||||
_('split — the default leaves directly; only matched rules are tunnelled')));
|
||||
break;
|
||||
case 'direct':
|
||||
rows.push(mk('warn', _('Traffic verdict'),
|
||||
_('direct — nothing is tunnelled; traffic leaves over the plain WAN')));
|
||||
break;
|
||||
case 'blocked':
|
||||
rows.push(mk('warn', _('Traffic verdict'), _('blocked — unmatched traffic is dropped')));
|
||||
break;
|
||||
default:
|
||||
rows.push(mk('unknown', _('Traffic verdict'), _('not reported')));
|
||||
break;
|
||||
}
|
||||
|
||||
// --- The nft table, as a bare presence check. -----------------------------
|
||||
// Kept because the offline stub still reads it straight from the kernel, so
|
||||
// it is the one plane fact available when no daemon answers. It says nothing
|
||||
// about WHICH ruleset is loaded — that is the Traffic plane row.
|
||||
if (st.table === true)
|
||||
rows.push(mk('good', _('nft table'), _('inet shater is loaded')));
|
||||
else if (st.table === false)
|
||||
rows.push(mk(st.enabled === false ? 'warn' : 'bad', _('nft table'), _('not loaded')));
|
||||
else
|
||||
rows.push(mk('unknown', _('nft table'), _('not reported')));
|
||||
|
||||
// --- Kill-switch: the configured policy, and whether it is in force. ------
|
||||
// "closed" with no plane installed is a setting that is not in effect, which
|
||||
// is worse news than "open" and must not share its amber lamp.
|
||||
if (st.kill_switch === 'closed' && st.plane === 'none')
|
||||
rows.push(mk('bad', _('Kill-switch'),
|
||||
_('closed, but NOT in effect — no data plane is installed')));
|
||||
else if (st.kill_switch === 'closed' && dstate === 'up')
|
||||
rows.push(mk('good', _('Kill-switch'), _('closed (fail-closed)')));
|
||||
else if (st.kill_switch === 'closed')
|
||||
rows.push(mk('warn', _('Kill-switch'),
|
||||
_('configured closed; whether it is installed is not known')));
|
||||
else if (st.kill_switch === 'open')
|
||||
rows.push(mk('warn', _('Kill-switch'), _('open (leaky)')));
|
||||
else
|
||||
rows.push(mk('unknown', _('Kill-switch'), _('not reported')));
|
||||
|
||||
// --- Running engine config hash ("" when the engine is not started). ------
|
||||
if (typeof st.hash === 'string' && st.hash !== '')
|
||||
rows.push(mk('good', _('Config hash'), st.hash));
|
||||
else if (estate === 'down')
|
||||
rows.push(mk('unknown', _('Config hash'), _('none — the engine is not started')));
|
||||
else
|
||||
rows.push(mk('unknown', _('Config hash'), _('not reported')));
|
||||
|
||||
return rows;
|
||||
}
|
||||
|
||||
// statusRows turns the readout into LED table rows.
|
||||
function statusRows(st) {
|
||||
return statusReadout(st).map(function(r) {
|
||||
return row(r.state, r.label, r.value);
|
||||
});
|
||||
}
|
||||
|
||||
// panelHint describes the button's target and what is known about it. It never
|
||||
// promises the panel is up — only where the launcher will point.
|
||||
function panelTitle(dstate) {
|
||||
if (dstate === 'up')
|
||||
return _('Mint a session token and open the admin panel');
|
||||
if (dstate === 'down')
|
||||
return _('shaterd is not answering, so this will probably fail — but the panel is served by the daemon, not by the engine, so it is worth trying: any failure is reported here.');
|
||||
return _('The daemon state is not known. Try it — a failure is reported here rather than hidden.');
|
||||
}
|
||||
|
||||
// handleOpenPanel mints a single-use token and hands it to the panel via the
|
||||
// ARCHITECTURE §2 browser bridge: GET http://<router>:<port>/?t=<token>. The panel
|
||||
// validates+consumes the token and drops a session cookie.
|
||||
@@ -151,13 +356,17 @@ return view.extend({
|
||||
handleSave: null,
|
||||
handleReset: null,
|
||||
|
||||
// Exposed so the offline fixture harness (tests/status-readout.test.js) can
|
||||
// exercise the state derivation without a browser, a router, or a DOM.
|
||||
statusReadout: statusReadout,
|
||||
daemonState: daemonState,
|
||||
engineState: engineState,
|
||||
|
||||
load: function() {
|
||||
return L.resolveDefault(callStatus(), {});
|
||||
},
|
||||
|
||||
render: function(st) {
|
||||
var self = this;
|
||||
|
||||
var table = E('table', { 'class': 'table' }, statusRows(st));
|
||||
|
||||
var openBtn = E('button', {
|
||||
@@ -174,25 +383,29 @@ return view.extend({
|
||||
'style': 'margin-left:1em;color:#888;font-size:90%'
|
||||
}, hintText());
|
||||
|
||||
// Reflect daemon reachability on the button up front, then keep the whole
|
||||
// dashboard live. Also track the panel port reported in status so the
|
||||
// launcher redirect and hint follow globals.panel_port.
|
||||
// Track the panel port reported in status so the launcher redirect and the
|
||||
// hint follow globals.panel_port.
|
||||
//
|
||||
// THE BUTTON IS NEVER DISABLED. It used to be locked whenever
|
||||
// `running !== true`, which after a8970b8ac means "the engine is down" —
|
||||
// precisely the situation the panel exists to get you out of, and one in
|
||||
// which the daemon and its web server are still up and still minting
|
||||
// tokens (cmd/shaterd/main.go starts the panel server independently of the
|
||||
// engine). Locking it on a guess is the failure; a mint that fails already
|
||||
// reports itself through ui.addNotification, which is the recoverable
|
||||
// direction for an unknown state.
|
||||
function reflect(state) {
|
||||
state = state || {};
|
||||
state = (state && typeof state === 'object') ? state : {};
|
||||
panelPort = state.panel_port || DEFAULT_PANEL_PORT;
|
||||
hint.textContent = hintText();
|
||||
var down = (state.running !== true);
|
||||
openBtn.disabled = down;
|
||||
openBtn.title = down
|
||||
? _('shaterd is not running — start the Shater service first')
|
||||
: _('Mint a session token and open the admin panel');
|
||||
openBtn.title = panelTitle(daemonState(state));
|
||||
}
|
||||
reflect(st || {});
|
||||
reflect(st);
|
||||
|
||||
poll.add(function() {
|
||||
return L.resolveDefault(callStatus(), {}).then(function(s) {
|
||||
dom.content(table, statusRows(s));
|
||||
reflect(s || {});
|
||||
reflect(s);
|
||||
});
|
||||
}, 5);
|
||||
|
||||
@@ -209,7 +422,7 @@ return view.extend({
|
||||
E('div', { 'class': 'cbi-section' }, [
|
||||
E('h3', {}, _('Admin panel')),
|
||||
E('p', { 'class': 'cbi-value-description' },
|
||||
_('The rich admin panel is served by shaterd on its own port. LuCI mints a short-lived, single-use token for your browser — the panel has no separate login.')),
|
||||
_('The rich admin panel is served by shaterd on its own port — by the daemon, not by the engine, so it stays reachable while the engine is down. LuCI mints a short-lived, single-use token for your browser; the panel has no separate login.')),
|
||||
E('div', {}, [ openBtn, hint ])
|
||||
])
|
||||
]);
|
||||
|
||||
@@ -4,9 +4,27 @@
|
||||
# Registers the ubus object "shater" (object name == this file's name) with two
|
||||
# read-side methods the thin LuCI launcher calls over ubus:
|
||||
#
|
||||
# status -> passthrough of `shaterd status` ({running,enabled,active,table,hash})
|
||||
# status -> passthrough of `shaterd status`
|
||||
# mint_token -> passthrough of `shaterd mint-token` ({"token":"..."} | {"error":"..."})
|
||||
#
|
||||
# The status object is whatever apply.Status marshals (shater/apply/apply.go is the
|
||||
# only definition; this script never parses or reshapes it). As of 2026-07-26 that is:
|
||||
#
|
||||
# running, engine_running, enabled, active, table, plane, traffic, hash,
|
||||
# kill_switch, panel_port, can_rollback, warnings, started_unix, uptime_seconds
|
||||
#
|
||||
# Two of those are load-bearing for the caller and easy to misread:
|
||||
#
|
||||
# running / engine_running — the ENGINE's liveness, not this daemon's. `running`
|
||||
# was a hardcoded true until a8970b8ac (2026-07-26) and is now `engineUp`, so
|
||||
# a healthy daemon with a dead engine reports running=false. The daemon's own
|
||||
# liveness is not a field at all.
|
||||
# plane — "full" | "hold" | "none" from a LIVE daemon. `shaterd status` also has an
|
||||
# OFFLINE STUB path: when the daemon is unreachable it still exits 0 and prints
|
||||
# a status built from the apply.Status zero value plus a UCI read, which leaves
|
||||
# plane at "". So this method returning an object is NOT evidence that a daemon
|
||||
# answered; a known plane word is. dashboard.js relies on exactly that.
|
||||
#
|
||||
# Why shell out to shaterd instead of talking to /var/run/shaterd.ctl directly:
|
||||
# a reliable AF_UNIX client is NOT guaranteed on stock OpenWrt (busybox `nc` is
|
||||
# usually built without `-U`; socat/ucode-socket aren't in the base image). shaterd
|
||||
|
||||
@@ -0,0 +1,236 @@
|
||||
#!/usr/bin/env node
|
||||
/*
|
||||
* Offline harness for the dashboard's state derivation.
|
||||
*
|
||||
* Run: node openwrt/luci-app-shater/tests/status-readout.test.js
|
||||
*
|
||||
* Why this exists: the LuCI page is the ONE screen an operator reaches when the
|
||||
* engine is down and the fail-closed holding plane is blocking the LAN. What it
|
||||
* says there is a claim about the router's behaviour, and until now nothing
|
||||
* checked those claims. There is no browser and no router in this loop — the view
|
||||
* exposes statusReadout/daemonState/engineState as plain functions, and this file
|
||||
* feeds them recorded status objects.
|
||||
*
|
||||
* The fixtures are not invented. Each is what the wire actually carries:
|
||||
*
|
||||
* ENGINE_UP — apply.Status() from a live daemon with a started engine.
|
||||
* ENGINE_DOWN — apply.Status() from a live daemon whose engine died; the
|
||||
* holding plane is installed and the LAN is blocked.
|
||||
* DAEMON_DOWN — the OFFLINE STUB `shaterd status` prints when the daemon is
|
||||
* unreachable (cmd/shaterd/main.go cmdStatus): apply.Status zero
|
||||
* value + a UCI read, marshalled by Status.JSON(), so every field
|
||||
* is present and `plane` is "".
|
||||
* NO_ANSWER — {} , what L.resolveDefault hands render() when the ubus call
|
||||
* fails outright.
|
||||
* PLUGIN_ERROR — {"error":...} from the rpcd plugin, which is ALSO what a
|
||||
* live-but-wedged daemon produces.
|
||||
* LEGACY — a daemon predating plane/engine_running (packages do not update
|
||||
* atomically).
|
||||
*
|
||||
* Mutation check: revert dashboard.js to reading `st.running` for daemon
|
||||
* liveness and this file fails on ENGINE_DOWN with the exact text the operator
|
||||
* would have been shown.
|
||||
*/
|
||||
|
||||
'use strict';
|
||||
|
||||
var fs = require('fs');
|
||||
var path = require('path');
|
||||
|
||||
// --- Load the view module with LuCI's globals stubbed. ----------------------
|
||||
// The view file is a module body LuCI wraps in a function, so it ends in a
|
||||
// top-level `return` and cannot be require()d. Wrapping it in new Function is the
|
||||
// same thing LuCI's loader does. The 'require x' lines are bare string literals
|
||||
// and evaluate to nothing.
|
||||
// DASHBOARD_JS points the harness at a copy of the view. It exists so the
|
||||
// mutation check is repeatable: copy dashboard.js, reintroduce the defect in the
|
||||
// copy, run this file against it, and watch the named assertions fail. A test
|
||||
// that cannot be shown to fail on the broken code is decoration.
|
||||
var SRC = process.env.DASHBOARD_JS || path.join(__dirname, '..', 'htdocs',
|
||||
'luci-static', 'resources', 'view', 'shater', 'dashboard.js');
|
||||
|
||||
function loadView() {
|
||||
var src = fs.readFileSync(SRC, 'utf8');
|
||||
var factory = new Function('view', 'dom', 'poll', 'rpc', 'ui', 'E', '_', 'L',
|
||||
'window', src);
|
||||
return factory(
|
||||
{ extend: function(o) { return o; } }, // view
|
||||
{ content: function() {} }, // dom
|
||||
{ add: function() {} }, // poll
|
||||
{ declare: function() { return function() {}; } }, // rpc
|
||||
{ createHandlerFn: function() { return function() {}; }, addNotification: function() {} },
|
||||
function() { return {}; }, // E
|
||||
function(s) { return s; }, // _ (identity)
|
||||
{ resolveDefault: function(p, d) { return Promise.resolve(d); } },
|
||||
{ location: { hostname: 'router' }, open: function() { return null; } }
|
||||
);
|
||||
}
|
||||
|
||||
var page = loadView();
|
||||
|
||||
// --- Fixtures ---------------------------------------------------------------
|
||||
|
||||
var ENGINE_UP = {
|
||||
running: true, engine_running: true, enabled: true, active: true, table: true,
|
||||
plane: 'full', traffic: { verdict: 'tunnel', default: 'proxy', tunnel_rules: 3 },
|
||||
hash: 'a1b2c3d4', kill_switch: 'closed', panel_port: 8088, can_rollback: true,
|
||||
warnings: [], started_unix: 1753500000, uptime_seconds: 3600
|
||||
};
|
||||
|
||||
var ENGINE_DOWN = {
|
||||
running: false, engine_running: false, enabled: true, active: true, table: true,
|
||||
plane: 'hold', traffic: { verdict: '', default: '', tunnel_rules: 0 },
|
||||
hash: '', kill_switch: 'closed', panel_port: 8088, can_rollback: true,
|
||||
warnings: [], started_unix: 1753500000, uptime_seconds: 3600
|
||||
};
|
||||
|
||||
// Exactly what Status.JSON() emits for the cmdStatus offline stub.
|
||||
var DAEMON_DOWN = {
|
||||
running: false, engine_running: false, enabled: true, active: true, table: true,
|
||||
plane: '', traffic: { verdict: '', default: '', tunnel_rules: 0 },
|
||||
hash: '', kill_switch: 'closed', panel_port: 8088, can_rollback: false,
|
||||
warnings: null, started_unix: 0, uptime_seconds: 0
|
||||
};
|
||||
|
||||
var NO_ANSWER = {};
|
||||
var PLUGIN_ERROR = { error: 'shaterd unavailable' };
|
||||
var LEGACY = {
|
||||
running: true, enabled: true, active: true, table: true, hash: 'deadbeef',
|
||||
kill_switch: 'closed', panel_port: 8088
|
||||
};
|
||||
|
||||
// --- Assertions -------------------------------------------------------------
|
||||
|
||||
var failures = [];
|
||||
|
||||
function check(name, cond, detail) {
|
||||
if (cond) return;
|
||||
failures.push(name + (detail ? ': ' + detail : ''));
|
||||
}
|
||||
|
||||
// readout indexes the rows by label. A row that is NOT emitted must fail by name
|
||||
// rather than by throwing on `undefined.state`: a harness that dies mid-run stops
|
||||
// reporting the assertions after it, which is the silent-skip failure this
|
||||
// project has been bitten by. Missing rows come back as a loud sentinel instead.
|
||||
var MISSING = { state: '<row absent>', value: '<row absent>', missing: true };
|
||||
|
||||
function readout(st) {
|
||||
var out = {};
|
||||
page.statusReadout(st).forEach(function(r) { out[r.label] = r; });
|
||||
return new Proxy(out, {
|
||||
get: function(t, k) {
|
||||
if (typeof k !== 'string' || k in t) return t[k];
|
||||
return MISSING;
|
||||
},
|
||||
has: function(t, k) { return k in t; }
|
||||
});
|
||||
}
|
||||
|
||||
function show(title, st) {
|
||||
process.stdout.write('\n=== ' + title + ' ===\n');
|
||||
process.stdout.write(' daemon=' + page.daemonState(st) +
|
||||
' engine=' + page.engineState(st) + '\n');
|
||||
page.statusReadout(st).forEach(function(r) {
|
||||
process.stdout.write(' [' + r.state.padEnd(7) + '] ' +
|
||||
r.label.padEnd(18) + ' ' + r.value + '\n');
|
||||
});
|
||||
}
|
||||
|
||||
function lamps(st) {
|
||||
return page.statusReadout(st).map(function(r) { return r.state; });
|
||||
}
|
||||
|
||||
// 1. Live daemon, engine up.
|
||||
show('A. daemon alive, engine running', ENGINE_UP);
|
||||
check('A/daemon', page.daemonState(ENGINE_UP) === 'up');
|
||||
check('A/engine', page.engineState(ENGINE_UP) === 'up');
|
||||
check('A/no-red', lamps(ENGINE_UP).indexOf('bad') === -1,
|
||||
'a fully healthy router must show no red lamp');
|
||||
check('A/no-unknown', lamps(ENGINE_UP).indexOf('unknown') === -1,
|
||||
'every field is present, so nothing may read as unknown');
|
||||
|
||||
// 2. THE DEFECT. Live daemon, dead engine, LAN held.
|
||||
show('B. daemon alive, engine DOWN, holding plane', ENGINE_DOWN);
|
||||
check('B/daemon-up', page.daemonState(ENGINE_DOWN) === 'up',
|
||||
'the daemon is answering; calling it dead is the bug being fixed');
|
||||
check('B/engine-down', page.engineState(ENGINE_DOWN) === 'down');
|
||||
var b = readout(ENGINE_DOWN);
|
||||
check('B/daemon-row-green', b['Daemon (shaterd)'].state === 'good',
|
||||
'got ' + b['Daemon (shaterd)'].state + ' / ' + b['Daemon (shaterd)'].value);
|
||||
check('B/daemon-row-no-start-advice',
|
||||
b['Daemon (shaterd)'].value.indexOf('start') === -1,
|
||||
'must not tell the operator to start a service that is already running');
|
||||
check('B/plane-red', b['Traffic plane'].state === 'bad');
|
||||
check('B/plane-says-blocked', /BLOCKED/.test(b['Traffic plane'].value));
|
||||
check('B/latch-not-called-interception',
|
||||
!b['Service latch'].missing && b['Interception'].missing === true,
|
||||
'`active` is the run latch, not a "we are proxying" signal — apply.go: ' +
|
||||
'"Never render it as \'we are proxying\'"');
|
||||
check('B/latch-value-is-a-latch', /meant to be running/.test(b['Service latch'].value),
|
||||
'the latch row must state the latch, not claim traffic is being proxied');
|
||||
check('B/latch-not-active-word', !/^active$/.test(b['Service latch'].value));
|
||||
check('B/verdict-unknown', b['Traffic verdict'].state === 'unknown',
|
||||
'no verdict was published; it must not be painted as tunnel');
|
||||
check('B/some-red', lamps(ENGINE_DOWN).indexOf('bad') !== -1,
|
||||
'a blocked LAN must not be an all-green screen');
|
||||
|
||||
// The button is a property of render(), not of the pure readout, so it is
|
||||
// guarded at the source level: nothing may ever set `disabled` on the launcher.
|
||||
// Locking the way into the panel while the engine is down is the defect this
|
||||
// whole file exists for, and it must not come back by a different route.
|
||||
check('B/button-never-disabled',
|
||||
!/openBtn\s*\.\s*disabled/.test(fs.readFileSync(SRC, 'utf8')),
|
||||
'dashboard.js assigns openBtn.disabled — the launcher must never be locked');
|
||||
|
||||
// 3. Daemon not answering at all — the offline stub.
|
||||
show('C. daemon NOT running (offline stub)', DAEMON_DOWN);
|
||||
check('C/daemon-down', page.daemonState(DAEMON_DOWN) === 'down',
|
||||
'plane:"" is the stub signature; got ' + page.daemonState(DAEMON_DOWN));
|
||||
check('C/engine-down', page.engineState(DAEMON_DOWN) === 'down');
|
||||
var c = readout(DAEMON_DOWN);
|
||||
check('C/daemon-row-red', c['Daemon (shaterd)'].state === 'bad');
|
||||
check('C/plane-unknown', c['Traffic plane'].state === 'unknown',
|
||||
'the stub reports no plane; got ' + c['Traffic plane'].value);
|
||||
check('C/kill-switch-not-green', c['Kill-switch'].state !== 'good',
|
||||
'"closed" from a dead daemon proves nothing is installed to enforce it');
|
||||
check('C/distinct-from-B',
|
||||
c['Daemon (shaterd)'].value !== b['Daemon (shaterd)'].value,
|
||||
'engine-down and daemon-down must not render identically');
|
||||
|
||||
// 4/5/6. Degenerate answers must degrade to unknown, never to healthy.
|
||||
[['D. ubus call failed ({})', NO_ANSWER],
|
||||
['E. rpcd plugin error / wedged daemon', PLUGIN_ERROR],
|
||||
['F. legacy daemon (no plane, no engine_running)', LEGACY]].forEach(function(p) {
|
||||
show(p[0], p[1]);
|
||||
var st = p[1];
|
||||
check(p[0] + '/daemon-unknown', page.daemonState(st) === 'unknown');
|
||||
check(p[0] + '/engine-unknown', page.engineState(st) === 'unknown');
|
||||
var r = readout(st);
|
||||
check(p[0] + '/plane-unknown', r['Traffic plane'].state === 'unknown');
|
||||
check(p[0] + '/daemon-row-unknown', r['Daemon (shaterd)'].state === 'unknown');
|
||||
check(p[0] + '/no-false-green-plane', r['Traffic plane'].state !== 'good');
|
||||
});
|
||||
|
||||
// The legacy fixture additionally must not crash and must not lose the fields it
|
||||
// DOES carry — a non-atomic package update must degrade, not black out.
|
||||
var f = readout(LEGACY);
|
||||
check('F/enabled-still-read', f['Service enabled'].state === 'good');
|
||||
check('F/hash-still-read', f['Config hash'].value === 'deadbeef');
|
||||
check('F/table-still-read', f['nft table'].state === 'good');
|
||||
|
||||
// A plane word this build does not know about must land in unknown, not in the
|
||||
// last-listed branch. (Closed positive list, recoverable default.)
|
||||
var FUTURE = Object.assign({}, ENGINE_UP, { plane: 'partial' });
|
||||
check('G/unknown-plane-word', page.daemonState(FUTURE) === 'unknown',
|
||||
'an unrecognised plane value must not be read as a live daemon');
|
||||
check('G/unknown-plane-row', readout(FUTURE)['Traffic plane'].state === 'unknown');
|
||||
|
||||
// --- Report -----------------------------------------------------------------
|
||||
|
||||
process.stdout.write('\n');
|
||||
if (failures.length) {
|
||||
process.stdout.write('FAIL (' + failures.length + ')\n');
|
||||
failures.forEach(function(f) { process.stdout.write(' - ' + f + '\n'); });
|
||||
process.exit(1);
|
||||
}
|
||||
process.stdout.write('OK — all cases distinguished\n');
|
||||
@@ -40,6 +40,10 @@ define Package/shater-core
|
||||
# shaterd : the daemon our init supervises (`shaterd run`)
|
||||
# kmod-nft-tproxy : kernel TPROXY (shaterd emits the `inet shater` rules)
|
||||
# kmod-nft-socket : socket match used by the tproxy divert chain
|
||||
# kmod-tun : /dev/net/tun — the daemon opens the `shater-l3` TUN
|
||||
# for L3 ingress (globals.l3_tunnel); usually built-in
|
||||
# on stock images, but a slimmed image without it would
|
||||
# make the option fail with a cryptic open() error.
|
||||
# ip-full : `ip rule`/`ip route`/rt_tables for policy routing
|
||||
# nftables-json : shaterd shells out to `nft`, and netplane/stats.go
|
||||
# parses `nft -j list ...` — the JSON output only exists
|
||||
@@ -49,7 +53,7 @@ define Package/shater-core
|
||||
# ca-bundle : the daemon is CGO_ENABLED=0, so crypto/x509 has no
|
||||
# host cert fallback — without /etc/ssl/certs every
|
||||
# HTTPS subscription / .srs ruleset fetch fails.
|
||||
DEPENDS:=+shaterd +kmod-nft-tproxy +kmod-nft-socket +ip-full +nftables-json +ca-bundle
|
||||
DEPENDS:=+shaterd +kmod-nft-tproxy +kmod-nft-socket +kmod-tun +ip-full +nftables-json +ca-bundle
|
||||
PKGARCH:=all
|
||||
endef
|
||||
|
||||
@@ -96,6 +100,16 @@ define Package/shater-core/install
|
||||
|
||||
$(INSTALL_DIR) $(1)/etc/uci-defaults
|
||||
$(INSTALL_BIN) ./files/etc/uci-defaults/30_shater-core $(1)/etc/uci-defaults/30_shater-core
|
||||
|
||||
# sysupgrade's "keep settings" walks /lib/upgrade/keep.d/*, and without this the
|
||||
# node inventory in /etc/shater/subs does NOT survive a flash: the restored box
|
||||
# has its rules and its groups and no nodes for them to point at, and the only
|
||||
# repair is `sub update`, which needs the internet the tunnel was going to
|
||||
# provide. Package metadata, not user config, so INSTALL_DATA and not
|
||||
# INSTALL_CONF. (/etc/config/shater needs no entry — it is a conffile and
|
||||
# sysupgrade already keeps it that way.)
|
||||
$(INSTALL_DIR) $(1)/lib/upgrade/keep.d
|
||||
$(INSTALL_DATA) ./files/lib/upgrade/keep.d/shater-core $(1)/lib/upgrade/keep.d/shater-core
|
||||
endef
|
||||
|
||||
$(eval $(call BuildPackage,shater-core))
|
||||
|
||||
@@ -49,17 +49,58 @@ config globals 'globals'
|
||||
# Queries aimed at an EXTERNAL resolver are dropped with the rest of the LAN's
|
||||
# forwarded traffic.
|
||||
option dns_intercept '1'
|
||||
# Carry LAN ping through the tunnel. ON by default, and the alternative is
|
||||
# why: without it a ping is decided by `untunnelable` below, whose rungs are
|
||||
# "drop it" (block, the default) or "let it out of the WAN interface with the
|
||||
# client's real IP on it" (icmp/direct). There was no setting in which ping
|
||||
# both worked and stayed inside the tunnel. With this on, the engine opens a
|
||||
# TUN, LAN ICMP is routed into it, and an outbound that speaks layer 3
|
||||
# (WireGuard/AmneziaWG, or a direct route) carries the echo for real. An
|
||||
# outbound that does not (vless/trojan/shadowsocks) makes the ping DROP —
|
||||
# honestly: no reply is forged, ping reports loss. So a ping that used to
|
||||
# "work" through such a node was a ping that was leaking.
|
||||
#
|
||||
# It costs a permanent TUN device plus the gVisor netstack behind it, about
|
||||
# 2 MB of RSS for as long as the daemon runs.
|
||||
#
|
||||
# Set to '0' to opt out — worth it on a 32/64 MB router, or to bisect whether
|
||||
# the L3 ingress is what broke something. `shaterd apply` will tell you what
|
||||
# the off state costs. Your explicit value is never overwritten: this file is
|
||||
# a conffile and the daemon always writes the option back as '1'/'0'.
|
||||
#
|
||||
# NOTE the interaction: with this ON, `untunnelable` no longer governs ping at
|
||||
# all (the L3 route decision happens before the firewall chain its verdicts
|
||||
# live in). It still governs ESP/AH/GRE/IGMP/SCTP, which no tunnel of ours can
|
||||
# carry. `untunnelable 'icmp'` in particular stops meaning "block plus working
|
||||
# ping" and is reported as such.
|
||||
option l3_tunnel '1'
|
||||
option ipv6 '1'
|
||||
# Reserved fwmark base and routing-table base (do not overlap fw4/other apps).
|
||||
option fwmark_base '0x2000'
|
||||
option table_base '0x2000'
|
||||
# Seconds to auto-rollback an unconfirmed apply (0 = commit-confirm off).
|
||||
# Seconds to auto-rollback an unconfirmed apply. SHIPPED AS 0, i.e.
|
||||
# commit-confirm is OFF: `shaterd apply` arms nothing, and an apply that costs
|
||||
# you SSH/LuCI access stays until you undo it by hand. Set a window (e.g.
|
||||
# '120') to arm it, and run `shaterd confirm` inside that window to keep the
|
||||
# new config. Note the option is written back only when NON-zero, so an
|
||||
# explicit '0' disappears from this file on the first write by the daemon or
|
||||
# the panel — absent and 0 are the same thing.
|
||||
option confirm_timeout '0'
|
||||
option schema_version '1'
|
||||
# Master enable of the DNS blocklist/allowlist filter (D15). OFF by default;
|
||||
# it needs at least one `config resolver` to have a DNS plane to filter with.
|
||||
# See the "DNS filter" section at the end of this file.
|
||||
option dns_filter '0'
|
||||
option schema_version '2'
|
||||
|
||||
# LAN interception inbound. `network` is a UCI interface name; shaterd resolves
|
||||
# it to its device (e.g. 'lan' -> br-lan) for the nft TPROXY plane. Enable
|
||||
# globals above and adjust `network` to the interface(s) you want proxied.
|
||||
#
|
||||
# There is no per-inbound `sniff` option: since sing-box 1.11 sniffing is a
|
||||
# leading route ACTION rule with no inbound matcher, so EVERY inbound is sniffed,
|
||||
# always. Do not add one back — the hijack-dns rule matches the SNIFFED `dns`
|
||||
# protocol, so a per-inbound sniff toggle would be a DNS-leak switch (D14, and
|
||||
# the long argument at shater/model/model.go Inbound).
|
||||
config inbound
|
||||
option name 'lan'
|
||||
option enabled '1'
|
||||
@@ -68,7 +109,6 @@ config inbound
|
||||
option tproxy_port '12345'
|
||||
option tcp '1'
|
||||
option udp '1'
|
||||
option sniff '1'
|
||||
|
||||
# --- Commented examples (copy, uncomment, adjust, then enable globals) -------
|
||||
#
|
||||
@@ -153,8 +193,8 @@ config inbound
|
||||
#
|
||||
# --- DNS filter (D15) -------------------------------------------------------
|
||||
# Network-wide domain blocking, built on sing-box rule-sets + reject DNS rules.
|
||||
# Turn it ON by setting `option dns_filter '1'` in `config globals` above (it is
|
||||
# OFF by default). Filtering needs at least one `config resolver` (the in-engine
|
||||
# Turn it ON by flipping `option dns_filter` to '1' in `config globals` above (it
|
||||
# is shipped '0'). Filtering needs at least one `config resolver` (the in-engine
|
||||
# DNS plane). A blocklist answers matched domains with NXDOMAIN; an allowlist
|
||||
# always OVERRIDES the blocklists (allowlisted domains resolve normally).
|
||||
#
|
||||
|
||||
@@ -3,11 +3,12 @@
|
||||
# plus the data-plane watchdog (v0.2).
|
||||
#
|
||||
# A tiny procd-supervised loop that, once per tick, checks every enabled
|
||||
# subscription and url-ruleset against its per-item `update_interval` and runs
|
||||
# subscription against its per-item `update_interval` and runs
|
||||
# shaterd sub update <name> (subscriptions)
|
||||
# shaterd ruleset update <name> (url rulesets)
|
||||
# when the item is due, then a single `shaterd reconcile` if anything changed
|
||||
# (the daemon's config-hash gate rebuilds the engine only on a real change).
|
||||
# `config ruleset` items are NOT touched here — see shater_run_due for who owns
|
||||
# their refresh and where the remaining gap is.
|
||||
#
|
||||
# RELIABILITY CONTRACT (same "железно" posture as /etc/init.d/shater):
|
||||
# * The loop body is fully INERT unless globals.enabled=1 AND the main shater
|
||||
@@ -27,6 +28,15 @@
|
||||
# the main service is STOPPED (tears interception down — fail-open, LAN
|
||||
# returns to plain routing); with kill_switch=closed the rules stay
|
||||
# (blocked-by-design) and we log loudly.
|
||||
# * CRASH-LOOP WATCHDOG: the dead-daemon counter above cannot see the failure
|
||||
# it matters most for. /etc/init.d/shater sets `respawn 3600 5 0`, so a
|
||||
# daemon that dies a few seconds into startup is back within 5s and a single
|
||||
# `pidof` per 60s tick nearly always finds a process — the counter resets and
|
||||
# never reaches WATCHDOG_TICKS, while the fail-closed plane keeps the LAN shut
|
||||
# and the panel (served BY the daemon) never comes up. So the tick's sleep is
|
||||
# spent SAMPLING the daemon's identity instead of sleeping blind, and a tick in
|
||||
# which several different daemons lived is counted as churn. See
|
||||
# shater_churn_scan / shater_churn_verdict / shater_churn_action.
|
||||
# * The loop never self-exits (procd would respawn-churn an exiting body);
|
||||
# it idles on its guards instead. busybox ash only — no bashisms.
|
||||
|
||||
@@ -42,14 +52,34 @@ INIT_SCRIPT=/etc/init.d/shater-cron
|
||||
SHATER_INIT=/etc/init.d/shater
|
||||
SHATERD=/usr/bin/shaterd
|
||||
ACTIVE_FLAG=/var/run/shater.active
|
||||
# Raised by /etc/init.d/shater around a restart/reload and cleared by the
|
||||
# successor's start_service. Read here ONLY as a "a person/package asked for this
|
||||
# bounce" veto on the crash-loop verdict — never as a liveness signal.
|
||||
RESTART_FLAG=/var/run/shater.restarting
|
||||
# Written by `shaterd run` itself (main.go writePidfile) before it builds anything,
|
||||
# and removed by that same process on a clean exit. It is the only handle that
|
||||
# names THE daemon: `pidof shaterd` also matches the short-lived CLI verbs this
|
||||
# very loop runs (`sub update`, `reconcile`, `schedule due`).
|
||||
PIDFILE=/var/run/shaterd.pid
|
||||
STAMP_DIR=/var/run/shater/cron
|
||||
TICK=60 # seconds between due-checks
|
||||
RETRY_SECS=300 # backoff before retrying a FAILED fetch
|
||||
WATCHDOG_TICKS=5 # consecutive dead-daemon ticks before escalating
|
||||
DEFAULT_SUB_INTERVAL=6h
|
||||
DEFAULT_RS_INTERVAL=24h
|
||||
DEFAULT_BL_INTERVAL=24h # url blocklist refresh interval (D16)
|
||||
|
||||
# --- crash-loop watchdog tuning --------------------------------------------
|
||||
#
|
||||
# Every number here is chosen against ONE question: what can a legitimate restart
|
||||
# produce? A legitimate bounce (`restart`, LuCI Save & Apply -> reload, a package
|
||||
# transaction) replaces the daemon EXACTLY ONCE, and it is announced twice over —
|
||||
# /etc/init.d/shater raises RESTART_FLAG in stop_service and clears ACTIVE_FLAG for
|
||||
# the duration. A crash loop is announced by nothing and repeats without bound.
|
||||
LOOP_POLL=5 # seconds between identity samples inside one tick
|
||||
LOOP_MIN_GENS=3 # distinct daemons in ONE tick that count as churn
|
||||
LOOP_WINDOWS=2 # consecutive churn ticks before we call it a loop
|
||||
LOOP_REPORT_TICKS=30 # do not repeat the report more often than this
|
||||
|
||||
# --- helpers ---------------------------------------------------------------
|
||||
|
||||
shater_enabled() {
|
||||
@@ -133,7 +163,7 @@ shater_stamp_retry() {
|
||||
# Walk anonymous `config subscription` / `config ruleset` sections by index and
|
||||
# run any that are due. Echoes non-empty on stdout if at least one item updated.
|
||||
shater_run_due() {
|
||||
local i name en ivl secs stamp src changed=""
|
||||
local i name en ivl secs stamp changed=""
|
||||
|
||||
# Subscriptions.
|
||||
i=0
|
||||
@@ -159,28 +189,36 @@ shater_run_due() {
|
||||
i=$(( i + 1 ))
|
||||
done
|
||||
|
||||
# Rulesets (only url sources auto-update; others have nothing to fetch).
|
||||
i=0
|
||||
while uci -q get "shater.@ruleset[$i]" >/dev/null 2>&1; do
|
||||
name=$(uci -q get "shater.@ruleset[$i].name")
|
||||
src=$(uci -q get "shater.@ruleset[$i].source")
|
||||
if [ -n "$name" ] && [ "$src" = "url" ]; then
|
||||
ivl=$(uci -q get "shater.@ruleset[$i].update_interval")
|
||||
secs=$(shater_ivl_secs "$ivl" "$DEFAULT_RS_INTERVAL")
|
||||
stamp="$STAMP_DIR/rs.$(shater_safe_name "$name")"
|
||||
if shater_due "$stamp" "$secs"; then
|
||||
if "$SHATERD" ruleset update "$name" >/dev/null 2>&1; then
|
||||
shater_stamp "$stamp"
|
||||
changed=1
|
||||
else
|
||||
_slog -p daemon.warn \
|
||||
"ruleset update '$name' failed; retrying in ${RETRY_SECS}s"
|
||||
shater_stamp_retry "$stamp" "$secs"
|
||||
fi
|
||||
fi
|
||||
fi
|
||||
i=$(( i + 1 ))
|
||||
done
|
||||
# RULE-SETS ARE NOT UPDATED FROM HERE, AND NEVER WERE.
|
||||
#
|
||||
# There used to be a second loop that ran `shaterd ruleset update <name>` for
|
||||
# every `config ruleset` with source=url. That verb has never existed: it
|
||||
# printed a note and exited 0, so this loop stamped the item as freshly updated
|
||||
# and raised `changed`, which cost a reconcile per item per interval and told
|
||||
# the operator the list was current when not one byte had been fetched. The verb
|
||||
# now exits non-zero (shater/cmd/shaterd/main.go, notImpl), which turns the same
|
||||
# loop into one failed attempt and one syslog line every RETRY_SECS — ~288 lines
|
||||
# a day, per rule-set, about work that has no owner here. Noise in the log hides
|
||||
# real problems as effectively as a lie about success does.
|
||||
#
|
||||
# WHO REFRESHES A url RULE-SET NOW, so the next reader does not think this was
|
||||
# forgotten. `source=url` splits into two shapes in shater/generate/ruleset.go:
|
||||
#
|
||||
# * the URL serves an engine-native .srs/.json -> it stays a REMOTE rule-set
|
||||
# and sing-box owns fetch/cache/refresh through RemoteRuleSet.UpdateInterval
|
||||
# on the running box. This cron loop never had anything to contribute.
|
||||
# * the URL serves a plain-text list -> it is compiled locally into
|
||||
# /etc/shater/lists/<tag>.srs, and that artifact is refreshed by the
|
||||
# GENERATOR, "when missing or older than update_interval" — i.e. only when
|
||||
# something else already caused a generate. Nothing schedules one, so this
|
||||
# shape has NO periodic refresh at all today. That is a real gap, and it is
|
||||
# stated here rather than papered over with a call to a verb that does
|
||||
# nothing: closing it needs a daemon-side timer (or a real `ruleset update`),
|
||||
# not a shell loop, because only the daemon can force a rebuild past the
|
||||
# config-hash gate.
|
||||
#
|
||||
# `config blocklist` url items are a different mechanism and DO refresh — see
|
||||
# shater_run_due_blocklists below.
|
||||
|
||||
[ -n "$changed" ] && echo 1
|
||||
}
|
||||
@@ -256,6 +294,171 @@ shater_watchdog() {
|
||||
echo "$dead"
|
||||
}
|
||||
|
||||
# --- crash-loop watchdog ----------------------------------------------------
|
||||
#
|
||||
# THE HOLE. shater_watchdog above answers "is a daemon there?" once every TICK
|
||||
# seconds. /etc/init.d/shater sets `respawn 3600 5 0`, so a daemon that dies a few
|
||||
# seconds into startup is back 5s later and that one sample nearly always finds a
|
||||
# process: the dead-counter resets, never reaches WATCHDOG_TICKS, and the single
|
||||
# failure the watchdog exists for — a new binary or a bad config that cannot get
|
||||
# an engine up while the fail-closed plane holds the LAN shut — is the one it can
|
||||
# never see. The daemon also serves the panel, so in that state the operator has
|
||||
# neither internet nor a way to look at the box.
|
||||
#
|
||||
# THE SIGNAL. Not "is it there" but "is it the SAME one". The tick's sleep is
|
||||
# spent taking an identity sample every LOOP_POLL seconds instead of sleeping
|
||||
# blind, and a tick in which LOOP_MIN_GENS different daemons lived is a churn
|
||||
# tick. LOOP_WINDOWS consecutive churn ticks is the verdict.
|
||||
#
|
||||
# WHY A LEGITIMATE RESTART CANNOT REACH IT. Four independent reasons, in order of
|
||||
# how much they are relied on:
|
||||
#
|
||||
# 1. A bounce replaces the daemon ONCE. One restart scores 2 generations in the
|
||||
# tick it happens in and 1 in every tick after, so it cannot even produce a
|
||||
# single churn tick at LOOP_MIN_GENS=3, let alone LOOP_WINDOWS of them in a
|
||||
# row. Reaching the verdict takes >= 4 replacements inside 2 consecutive
|
||||
# minutes, >= 2 in each.
|
||||
# 2. Bounces are ANNOUNCED. /etc/init.d/shater raises RESTART_FLAG in
|
||||
# stop_service and clears ACTIVE_FLAG for the whole stop->start, and either
|
||||
# one seen in any sample of a tick discards that tick outright.
|
||||
# 3. The panel's Apply does not restart anything: it writes UCI and applies over
|
||||
# the daemon's control socket, in process. Only `restart`, a LuCI Save &
|
||||
# Apply (config.change -> reload) and a package transaction bounce the
|
||||
# daemon, and a human cannot produce those at four a minute.
|
||||
# 4. The sample names THE daemon via its pidfile, not `pidof shaterd` — the
|
||||
# short-lived CLI verbs this very loop runs share that process name.
|
||||
#
|
||||
# WHAT IT DOES NOT COVER, stated rather than implied: a daemon that dies
|
||||
# INSTANTLY (well under a second) is almost never caught alive by a 5s sample, so
|
||||
# it scores few generations and this detector stays quiet. That case is exactly
|
||||
# the one the existing dead-tick counter does see — its `pidof` misses too, tick
|
||||
# after tick — so the two cover opposite ends and are deliberately left as two
|
||||
# independent instruments rather than merged into one clever number.
|
||||
|
||||
# One identity sample: echoes the pid of the live `shaterd run`, or "-" for none.
|
||||
#
|
||||
# Through the PIDFILE, which `shaterd run` writes before it builds anything and
|
||||
# removes on a clean exit, because that is the only handle that names THE daemon:
|
||||
# `pidof shaterd` also matches `shaterd sub update` / `reconcile` / `schedule due`.
|
||||
# /proc/<pid>/comm is checked so a stale pidfile whose pid has been reused by an
|
||||
# unrelated process cannot read as a live daemon. No forks: `read` is a builtin.
|
||||
shater_sample_pid() {
|
||||
local pid="" comm=""
|
||||
[ -r "$PIDFILE" ] && read -r pid 2>/dev/null < "$PIDFILE"
|
||||
case "$pid" in
|
||||
''|*[!0-9]*) echo -; return ;;
|
||||
esac
|
||||
[ -r "/proc/$pid/comm" ] && read -r comm 2>/dev/null < "/proc/$pid/comm"
|
||||
[ "$comm" = "shaterd" ] || { echo -; return; }
|
||||
echo "$pid"
|
||||
}
|
||||
|
||||
# shater_churn_scan <sample>... -> "<generations> <absent-samples>"
|
||||
#
|
||||
# A GENERATION is one distinct daemon lifetime observed during the tick: a live
|
||||
# pid that differs from the last live pid seen. A daemon that simply keeps running
|
||||
# therefore scores exactly 1 generation and 0 absent samples for as long as it
|
||||
# runs — the signal is flat unless something is actually being replaced.
|
||||
#
|
||||
# A GAP (samples with no daemon at all, e.g. procd's 5s respawn hole) is counted
|
||||
# but does NOT by itself open a new generation: only a different pid does. An
|
||||
# earlier draft reset the comparison across a gap so that "same pid seen again
|
||||
# after a gap" would score two. That case cannot occur — a respawn always gets a
|
||||
# fresh pid — so it was unfalsifiable code, and resetting also meant a momentarily
|
||||
# unreadable pidfile could inflate the count. Not resetting is both simpler and
|
||||
# the safer direction.
|
||||
#
|
||||
# Pure: no I/O, no globals, every input on the command line. That is what lets the
|
||||
# gate drive it with synthetic sample streams instead of a live router.
|
||||
shater_churn_scan() {
|
||||
local gens=0 absent=0 last="" s
|
||||
for s in "$@"; do
|
||||
if [ "$s" = "-" ]; then
|
||||
absent=$(( absent + 1 ))
|
||||
continue
|
||||
fi
|
||||
[ "$s" = "$last" ] || gens=$(( gens + 1 ))
|
||||
last="$s"
|
||||
done
|
||||
echo "$gens $absent"
|
||||
}
|
||||
|
||||
# shater_churn_verdict <gens> <samples> <announced> <churn-so-far>
|
||||
# -> the new consecutive-churn-tick count
|
||||
#
|
||||
# Also pure. `announced`=1 means a sample during the tick saw RESTART_FLAG up or
|
||||
# ACTIVE_FLAG down, i.e. /etc/init.d/shater said out loud that it was bouncing the
|
||||
# daemon: that tick proves nothing and resets the run. A tick with no samples at
|
||||
# all (the first pass through the loop) likewise scores 0 rather than guessing.
|
||||
shater_churn_verdict() {
|
||||
local gens="$1" n="$2" announced="$3" churn="$4"
|
||||
[ "$announced" = "1" ] && { echo 0; return; }
|
||||
[ "$n" -gt 0 ] || { echo 0; return; }
|
||||
if [ "$gens" -ge "$LOOP_MIN_GENS" ]; then
|
||||
echo $(( churn + 1 ))
|
||||
return
|
||||
fi
|
||||
echo 0
|
||||
}
|
||||
|
||||
# shater_churn_action <churn-ticks> <kill_switch> -> none | log | stop
|
||||
#
|
||||
# WHAT TO DO, and why it is not our call to make twice. A crash loop leaves the
|
||||
# box in the same state a dead daemon does — no engine, fail-closed plane standing
|
||||
# — so the answer is the one the operator already gave with kill_switch, not a new
|
||||
# policy invented here:
|
||||
#
|
||||
# open The operator asked for connectivity over interception. Stop the stack,
|
||||
# exactly as shater_watchdog does for a sustained-dead daemon: the plane
|
||||
# comes down and the LAN returns to plain routing. It also disarms the
|
||||
# boot armor, so the NEXT boot is clean too instead of repeating the loop
|
||||
# behind a closed LAN. Nothing else can end the loop: procd's retries are
|
||||
# infinite by design.
|
||||
# closed The operator asked for blocked-over-leaking. Blocked is what they get,
|
||||
# and opening their LAN from a background loop would be the opposite of
|
||||
# what the knob says. Report it loudly and let the person decide; the
|
||||
# message names the one command that opens it.
|
||||
#
|
||||
# The list is POSITIVE and CLOSED, and the fall-through goes to `log`: an absent
|
||||
# or unrecognised kill_switch is the model's documented default ("closed", see
|
||||
# shater/model/model.go DefaultGlobals), and `log` is the recoverable side — it
|
||||
# changes nothing and can be acted on, where a wrong `stop` silently drops a
|
||||
# household onto the unproxied WAN.
|
||||
#
|
||||
# NOTE (not changed here, deliberately): shater_watchdog above answers the same
|
||||
# question with `if closed ... else stop`, so for an ABSENT kill_switch it fails
|
||||
# open — the opposite of the documented default. It is left alone because that
|
||||
# behaviour predates this file's crash-loop work; it is reported upward instead.
|
||||
shater_churn_action() {
|
||||
local churn="$1" ks="$2"
|
||||
[ "$churn" -ge "$LOOP_WINDOWS" ] || { echo none; return; }
|
||||
case "$ks" in
|
||||
open) echo stop ;;
|
||||
closed) echo log ;;
|
||||
*) echo log ;;
|
||||
esac
|
||||
}
|
||||
|
||||
# Sleep out one tick in LOOP_POLL slices, sampling the daemon's identity as we go.
|
||||
# Publishes CHURN_SAMPLES / CHURN_N / CHURN_ANNOUNCED for the next pass of loop().
|
||||
# Deliberately NOT a subshell (globals must survive), and it always returns 0 so a
|
||||
# false `[ -f ]` at the end cannot look like a failure.
|
||||
shater_tick_sample() {
|
||||
local slept=0
|
||||
CHURN_SAMPLES=""
|
||||
CHURN_N=0
|
||||
CHURN_ANNOUNCED=0
|
||||
while [ "$slept" -lt "$TICK" ]; do
|
||||
sleep "$LOOP_POLL"
|
||||
slept=$(( slept + LOOP_POLL ))
|
||||
CHURN_SAMPLES="$CHURN_SAMPLES $(shater_sample_pid)"
|
||||
CHURN_N=$(( CHURN_N + 1 ))
|
||||
[ -f "$RESTART_FLAG" ] && CHURN_ANNOUNCED=1
|
||||
[ -f "$ACTIVE_FLAG" ] || CHURN_ANNOUNCED=1
|
||||
done
|
||||
return 0
|
||||
}
|
||||
|
||||
# loop: the foreground body supervised by procd. Never exits on its own — it
|
||||
# idles while disabled/inactive so procd is not respawn-churned by a
|
||||
# self-exiting body when the stack is off.
|
||||
@@ -274,7 +477,12 @@ loop() {
|
||||
# the flock immediately and keeps children (sleep/shaterd) from inheriting
|
||||
# it. A no-op where fd 1000 is not open (older procd.sh without procd_lock).
|
||||
exec 1000>&-
|
||||
local changed dead=0
|
||||
local changed dead=0 churn=0 quiet=0 scan gens absent ks act
|
||||
# No tick has been sampled yet on the first pass; shater_churn_verdict scores
|
||||
# an empty tick as 0 rather than guessing.
|
||||
CHURN_SAMPLES=""
|
||||
CHURN_N=0
|
||||
CHURN_ANNOUNCED=0
|
||||
mkdir -p "$STAMP_DIR"
|
||||
while :; do
|
||||
if shater_enabled && shater_active; then
|
||||
@@ -299,10 +507,44 @@ loop() {
|
||||
"$SHATERD" schedule due >/dev/null 2>&1
|
||||
fi
|
||||
dead=$(shater_watchdog "$dead")
|
||||
|
||||
# Crash-loop verdict on the tick that has just elapsed. Unquoted on
|
||||
# purpose: CHURN_SAMPLES is a whitespace-separated token list and word
|
||||
# splitting is how it becomes arguments.
|
||||
scan=$(shater_churn_scan $CHURN_SAMPLES)
|
||||
gens=${scan%% *}
|
||||
absent=${scan##* }
|
||||
churn=$(shater_churn_verdict "$gens" "$CHURN_N" "$CHURN_ANNOUNCED" "$churn")
|
||||
ks=$(uci -q get shater.globals.kill_switch)
|
||||
act=$(shater_churn_action "$churn" "$ks")
|
||||
case "$act" in
|
||||
stop)
|
||||
_slog -p daemon.crit \
|
||||
"shaterd is CRASH-LOOPING: $gens distinct daemons in the last ${TICK}s (absent in $absent of $CHURN_N samples), $churn such windows in a row — it is being respawned faster than it can bring an engine up. kill_switch=open, so shater is being STOPPED: interception comes down and the LAN returns to plain, UNPROXIED routing. Find the reason with 'logread -e shaterd', then '/etc/init.d/shater start'."
|
||||
"$SHATER_INIT" stop
|
||||
churn=0
|
||||
quiet="$LOOP_REPORT_TICKS"
|
||||
;;
|
||||
log)
|
||||
# Rate-limited: a standing condition, not an event. Never
|
||||
# silent for good, though — an operator who looks at the log an
|
||||
# hour later must still find it being said.
|
||||
if [ "$quiet" -le 0 ]; then
|
||||
_slog -p daemon.crit \
|
||||
"shaterd is CRASH-LOOPING: $gens distinct daemons in the last ${TICK}s (absent in $absent of $CHURN_N samples), $churn such windows in a row — it is being respawned faster than it can bring an engine up. kill_switch=${ks:-closed} keeps the fail-closed plane standing, so the LAN stays blocked and the admin panel is down with the daemon that serves it. Nothing is decided for you: find the reason with 'logread -e shaterd', or open the LAN with '/etc/init.d/shater stop'."
|
||||
quiet="$LOOP_REPORT_TICKS"
|
||||
fi
|
||||
churn=0
|
||||
;;
|
||||
esac
|
||||
[ "$quiet" -gt 0 ] && quiet=$(( quiet - 1 ))
|
||||
else
|
||||
dead=0
|
||||
churn=0
|
||||
quiet=0
|
||||
fi
|
||||
sleep "$TICK"
|
||||
# Sleeps out the tick, sampling the daemon's identity while it does.
|
||||
shater_tick_sample
|
||||
done
|
||||
}
|
||||
|
||||
|
||||
@@ -59,6 +59,199 @@ if uci -q get shater.globals >/dev/null 2>&1 || [ -f /etc/config/shater ]; then
|
||||
uci -q commit shater
|
||||
fi
|
||||
|
||||
# Introduce the daemon-created `shater-l3*` TUN to fw4 (L3 ingress, D-L3). The
|
||||
# daemon policy-routes LAN ICMP into that device from OUR nft table
|
||||
# `inet shater`, but nftables runs EVERY table on every packet and a drop in
|
||||
# any one of them wins — an accept in `inet shater` cannot override fw4. And
|
||||
# fw4 WILL drop this forward: netifd knows nothing about a device the daemon
|
||||
# creates at runtime, so it belongs to no zone and falls into fw4's zone-less
|
||||
# defaults (REJECT). The device has to be declared to fw4 itself; it cannot be
|
||||
# fixed from our own table.
|
||||
#
|
||||
# Seeded UNCONDITIONALLY (not gated on globals.l3_tunnel): uci-defaults run
|
||||
# once, so gating on the option would require re-running this script when the
|
||||
# option is flipped later — which never happens. An idle zone is harmless: its
|
||||
# device match is a plain iifname/oifname STRING compare that simply never hits
|
||||
# while the TUN does not exist.
|
||||
#
|
||||
# Idempotency: `config zone`/`config forwarding` are normally ANONYMOUS
|
||||
# sections, and a naive `uci add firewall zone` would append a duplicate on
|
||||
# every re-run (uci-defaults re-run on package upgrade/reinstall). The zone is
|
||||
# NAMED instead, guarded by an existence check — a re-run re-finds the section
|
||||
# and touches nothing. The forwardings are named too where the name is free, but
|
||||
# their guard is a scan of the actual src/dest pairs, which is stronger; see
|
||||
# seed_l3_forwarding below.
|
||||
seed_l3_zone() {
|
||||
# No fw4 on this image (bare nftables build) => nothing drops the forward
|
||||
# on fw4's behalf and there is nothing to punch through.
|
||||
[ -f /etc/config/firewall ] || return 0
|
||||
if ! uci -q get firewall.shater_l3 >/dev/null; then
|
||||
uci set firewall.shater_l3=zone
|
||||
uci set firewall.shater_l3.name='shater_l3'
|
||||
uci set firewall.shater_l3.input='REJECT'
|
||||
uci set firewall.shater_l3.output='ACCEPT'
|
||||
uci set firewall.shater_l3.forward='REJECT'
|
||||
uci set firewall.shater_l3.masq='0'
|
||||
# INERT TODAY, kept for the day it is not. mtu_fix clamps forwarded TCP
|
||||
# MSS to the route MTU — but the L3 TUN is 65535 (deliberately: at any
|
||||
# smaller value the kernel fragments into the device, and the flow
|
||||
# dispatcher refuses to judge a fragment and lets the stack forge the
|
||||
# echo reply — see l3MTU in shater/generate/inbound.go), so the clamp has
|
||||
# nothing to clamp to. And only ICMP is ever marked into this device, so
|
||||
# no TCP rides here to be clamped in the first place. It earns its keep
|
||||
# the moment either of those changes; removing it would make that day
|
||||
# silent.
|
||||
uci set firewall.shater_l3.mtu_fix='1'
|
||||
# `list device`, deliberately NOT the usual `list network`: fw4
|
||||
# resolves a zone's networks through netifd, and netifd never learns
|
||||
# about a device the daemon creates at runtime — a stub interface
|
||||
# (proto none) would need to be brought UP to contribute an l3_device,
|
||||
# and nothing ever brings it up, so `list network` resolves to an
|
||||
# EMPTY device set and fw4 keeps dropping the forward. `list device`
|
||||
# instead compiles to an iifname/oifname STRING match, valid before
|
||||
# the TUN exists and matching from the moment shaterd creates it —
|
||||
# no netifd involvement and no firewall reload at enable time. Do not
|
||||
# "normalize" this to `list network` in a refactor; it breaks silently.
|
||||
#
|
||||
# The WILDCARD is load-bearing too. The daemon no longer opens one fixed
|
||||
# device: it alternates between `shater-l3a` and `shater-l3b` so that a
|
||||
# new engine generation never has to reopen the name the previous one is
|
||||
# still holding (that collision — TUNSETIFF: device or resource busy —
|
||||
# took the whole LAN down on the production router, because the recovery
|
||||
# path rebuilt the same config and hit the same busy name). fw4 compiles
|
||||
# `shater-l3*` to `iifname "shater-l3*"` / `oifname "shater-l3*"`,
|
||||
# verified on ImmortalWrt 25.12.1 with nftables 1.1.6, so ONE zone covers
|
||||
# every slot and no firewall reload is needed when the slot changes.
|
||||
uci add_list firewall.shater_l3.device='shater-l3*'
|
||||
fi
|
||||
seed_l3_forwardings
|
||||
uci -q commit firewall
|
||||
}
|
||||
|
||||
# EVERY zone gets a forwarding into shater_l3, not just `lan`.
|
||||
#
|
||||
# The bug this closes is silent by construction. The daemon's divert set is built
|
||||
# from every enabled `config inbound`'s network PLUS every device a rule names
|
||||
# through an `iface:`/`zone:` source (shater/netplane/nft.go, nftDivertRefs) — so
|
||||
# on a router with several LAN zones, ICMP from ALL of them is marked and routed
|
||||
# into the TUN by our table. Our table then accepts it and fw4 drops it anyway,
|
||||
# because the forward is judged in `forward_<source zone>` and only `lan` had a
|
||||
# jump to `accept_to_shater_l3`. Result: ping through the tunnel works from one
|
||||
# subnet and not from the next, with nothing in any log to say why — fw4's drop
|
||||
# is the zone's policy verdict, not a rule with a name. The owner's production
|
||||
# router has a single LAN zone, which is exactly why this went unnoticed; his
|
||||
# second router has three.
|
||||
#
|
||||
# Every zone, including an uplink zone, and that is deliberate rather than lazy:
|
||||
#
|
||||
# - The alternative is guessing which zones hold clients, and every available
|
||||
# signal is wrong somewhere. `masq='1'` marks the WAN on a stock config and
|
||||
# also marks a double-NAT LAN. The name `wan*` is a convention, not a rule.
|
||||
# A guess that is wrong reintroduces exactly the silent breakage above, while
|
||||
# a superfluous entry costs a line of ruleset.
|
||||
# - A forwarding into shater_l3 permits nothing on its own. It authorises the
|
||||
# forward of packets ROUTED INTO the TUN, and the only thing that routes a
|
||||
# packet there is our own fwmark rule, which matches solely on the divert
|
||||
# device set. A packet arriving on the WAN is not marked and never reaches
|
||||
# this decision; if an operator ever puts a WAN device in the divert set,
|
||||
# they meant to and this is the entry that makes it work.
|
||||
# - The reverse direction is NOT opened: no `src shater_l3` forwarding exists,
|
||||
# so nothing comes out of the TUN into a zone by way of these sections. The
|
||||
# engine's own replies return on the conntrack `established,related accept`
|
||||
# at the top of fw4's forward chain.
|
||||
#
|
||||
# LIMIT, stated because it is not obvious: this is a SNAPSHOT. uci-defaults run
|
||||
# at first boot and on package install/upgrade, so a zone created AFTER the last
|
||||
# shater-core install has no forwarding until the next one. Re-running this
|
||||
# script (or reinstalling the package) re-seeds. The durable fix belongs in the
|
||||
# daemon, which recomputes the divert set on every apply and already knows which
|
||||
# zones are in it; it is deliberately not attempted from here.
|
||||
seed_l3_forwardings() {
|
||||
uci -q show firewall 2>/dev/null |
|
||||
sed -n "s/^firewall\.\([^.=]*\)=zone\$/\1/p" |
|
||||
while read -r sid; do
|
||||
zone=$(uci -q get "firewall.$sid.name")
|
||||
# Unnamed zone: fw4 cannot reference it from a forwarding either.
|
||||
[ -n "$zone" ] || continue
|
||||
# Our own zone: a forwarding from shater_l3 to itself is meaningless.
|
||||
[ "$zone" = "shater_l3" ] && continue
|
||||
seed_l3_forwarding "$zone"
|
||||
done
|
||||
}
|
||||
|
||||
# One `config forwarding` <zone> -> shater_l3, created only if no such forwarding
|
||||
# exists yet.
|
||||
#
|
||||
# The guard scans the ACTUAL src/dest pairs rather than trusting a section id,
|
||||
# which covers all three ways one can already be there: the legacy named section
|
||||
# `shater_l3_fwd` seeded by earlier releases (src=lan), the per-zone names this
|
||||
# function writes, and an anonymous one an operator added by hand. Without that,
|
||||
# a re-run — uci-defaults re-run on every package upgrade — would append a
|
||||
# duplicate for `lan` on every upgrade.
|
||||
seed_l3_forwarding() {
|
||||
local zone="$1" sid found
|
||||
|
||||
found=$(uci -q show firewall 2>/dev/null |
|
||||
sed -n "s/^firewall\.\([^.=]*\)=forwarding\$/\1/p" |
|
||||
while read -r f; do
|
||||
[ "$(uci -q get "firewall.$f.dest")" = "shater_l3" ] || continue
|
||||
[ "$(uci -q get "firewall.$f.src")" = "$zone" ] || continue
|
||||
echo yes
|
||||
break
|
||||
done)
|
||||
[ -n "$found" ] && return 0
|
||||
|
||||
# Section ids are [a-zA-Z0-9_] only, while a zone name may legally carry a
|
||||
# hyphen — sanitise, and keep the legacy id for `lan` so an existing install
|
||||
# is recognised as already seeded rather than gaining a second section.
|
||||
if [ "$zone" = "lan" ]; then
|
||||
sid="shater_l3_fwd"
|
||||
else
|
||||
sid="shater_l3_fwd_$(printf '%s' "$zone" | sed 's/[^a-zA-Z0-9_]/_/g')"
|
||||
fi
|
||||
# The id may still be taken — by a section for a DIFFERENT zone whose name
|
||||
# sanitises to the same thing, or by something else entirely. Fall back to an
|
||||
# anonymous section rather than overwrite: the src/dest scan above is what
|
||||
# makes this idempotent, the name is only there to be readable.
|
||||
if uci -q get "firewall.$sid" >/dev/null; then
|
||||
sid=$(uci add firewall forwarding) || return 0
|
||||
else
|
||||
uci set "firewall.$sid=forwarding"
|
||||
fi
|
||||
uci set "firewall.$sid.src=$zone"
|
||||
uci set "firewall.$sid.dest=shater_l3"
|
||||
}
|
||||
seed_l3_zone
|
||||
|
||||
# Upgrade path for routers seeded by a pre-slot build.
|
||||
#
|
||||
# The block above only runs when the zone does NOT exist, which is exactly right
|
||||
# for idempotency and exactly wrong here: an already-installed router has the
|
||||
# zone with the OLD exact device `shater-l3`, that name matches no slot, and fw4
|
||||
# would go back to dropping the forward — i.e. LAN ping through the tunnel dies
|
||||
# silently on upgrade while everything reports healthy. Rewrite it in place.
|
||||
#
|
||||
# Narrow on purpose: only the literal legacy entry is replaced, and only when the
|
||||
# wildcard is not already listed, so an operator who added devices of their own
|
||||
# keeps them and a re-run changes nothing (uci-defaults re-run on every package
|
||||
# upgrade). No `fw4 reload` here — uci-defaults run before the firewall starts on
|
||||
# boot, and on a package upgrade the daemon's next apply is what needs the zone,
|
||||
# not this script.
|
||||
migrate_l3_zone_wildcard() {
|
||||
[ -f /etc/config/firewall ] || return 0
|
||||
uci -q get firewall.shater_l3 >/dev/null || return 0
|
||||
devs=$(uci -q get firewall.shater_l3.device) || return 0
|
||||
case " $devs " in
|
||||
*" shater-l3* "*) return 0 ;; # already migrated
|
||||
*" shater-l3 "*) ;; # legacy exact name present
|
||||
*) return 0 ;;
|
||||
esac
|
||||
uci -q del_list firewall.shater_l3.device='shater-l3'
|
||||
uci add_list firewall.shater_l3.device='shater-l3*'
|
||||
uci -q commit firewall
|
||||
}
|
||||
migrate_l3_zone_wildcard
|
||||
|
||||
# Bring the UCI schema forward on upgrade (idempotent; refuses a newer schema).
|
||||
[ -x /usr/bin/shaterd ] && /usr/bin/shaterd migrate >/dev/null 2>&1
|
||||
|
||||
@@ -120,6 +313,17 @@ SHATER_BRINGUP='
|
||||
[ -x /etc/init.d/shater-armor ] && /etc/init.d/shater-armor enable
|
||||
[ -x /etc/init.d/shater ] && /etc/init.d/shater restart
|
||||
[ -x /etc/init.d/shater-cron ] && /etc/init.d/shater-cron restart
|
||||
# Fold the seeded shater_l3 zone into the LIVE ruleset — matters on a live
|
||||
# opkg/apk install only, where firewall started long before our commit and
|
||||
# nothing else would re-read it until the next reboot. Gated on the fw4
|
||||
# table actually being loaded: at FIRST boot this job can run before the
|
||||
# S19 firewall start, and an early reload would install a ruleset built
|
||||
# from a half-initialized netifd AND make the later start a no-op (fw4
|
||||
# start skips when its table already exists). No table => the pending S19
|
||||
# start reads the committed config by itself, no reload needed.
|
||||
if nft list tables 2>/dev/null | grep -q "inet fw4"; then
|
||||
[ -x /etc/init.d/firewall ] && /etc/init.d/firewall reload
|
||||
fi
|
||||
exit 0
|
||||
'
|
||||
SHATER_TMO=""
|
||||
|
||||
@@ -0,0 +1,80 @@
|
||||
# /lib/upgrade/keep.d/shater-core — what sysupgrade and LuCI "Backup" must carry
|
||||
# out of /etc/shater.
|
||||
#
|
||||
# HOW THIS FILE IS READ. /sbin/sysupgrade (base-files, list_static_conffiles):
|
||||
#
|
||||
# find $(sed -ne '/^[[:space:]]*$/d; /^#/d; p' \
|
||||
# /etc/sysupgrade.conf /lib/upgrade/keep.d/* 2>/dev/null) \
|
||||
# \( -type f -o -type l \) $filter 2>/dev/null
|
||||
#
|
||||
# so blank lines and lines starting with '#' are stripped, and every other line is
|
||||
# a path handed to `find`: a directory is recursed, a path that does not exist is
|
||||
# silently skipped (hence a trailing '/' for the two directories, and no need to
|
||||
# guard for a fresh install that has neither). The result is tarred and, on a real
|
||||
# sysupgrade, HELD IN RAM across the flash — which is why this is a per-file
|
||||
# decision and not simply "/etc/shater/".
|
||||
#
|
||||
# WHY IT EXISTS. Everything the product knows besides /etc/config/shater lives in
|
||||
# /etc/shater, and nothing shipped a keep.d entry for it. A "keep settings"
|
||||
# sysupgrade, or a LuCI backup restored onto a new router, therefore produced a
|
||||
# box whose config looked complete and whose node inventory was EMPTY — silently.
|
||||
#
|
||||
# /etc/config/shater is NOT listed here: it is declared in
|
||||
# Package/shater-core/conffiles, and sysupgrade backs CHANGED conffiles up on its
|
||||
# own (list_changed_conffiles). Listing it again would work, but it would claim
|
||||
# ownership of a mechanism that already covers it.
|
||||
|
||||
# THE NODE INVENTORY. Subscription-fetched nodes deliberately live OUTSIDE UCI
|
||||
# (shater/model/subcache.go) — one JSON file per subscription. Without them the
|
||||
# restored box has groups and rules that reference nodes which do not exist, so no
|
||||
# tunnel comes up, and the only repair is `sub update`, which needs the internet
|
||||
# the tunnel was supposed to be providing. Indented JSON: a few hundred KiB even
|
||||
# for a several-hundred-node subscription.
|
||||
/etc/shater/subs/
|
||||
|
||||
# THE BOOT-ARMOR ARM TOKEN. Its PRESENCE is what lets /etc/init.d/shater-armor
|
||||
# (START=21) load the fail-closed plane before fw4's `lan -> wan ACCEPT` is the
|
||||
# only rule on the box. Without it the first boot after a restore forwards LAN to
|
||||
# WAN in the clear until the daemon has built an engine. One small nft script.
|
||||
/etc/shater/boot.nft
|
||||
|
||||
# COMPILED LIST ARTIFACTS (.srs). Losing these fails SILENTLY in the worst
|
||||
# direction: a missing LOCAL rule-set is left out of the generated config and the
|
||||
# engine starts perfectly happily with the filtering simply gone
|
||||
# (shater/generate/ruleset.go, compiledListRuleSet). "It will re-download itself"
|
||||
# is NOT true for them either — a compiled url list is rebuilt only by the next
|
||||
# generate, and nothing schedules one (see the note in /etc/init.d/shater-cron
|
||||
# about `ruleset update`). Cheap to keep: compiled .srs is 3-6% of the source
|
||||
# text (~80 KiB for a 150k-domain list), under a 4 MiB soft cap.
|
||||
/etc/shater/lists/
|
||||
|
||||
# ALERT DE-DUPLICATION STATE. A few hundred bytes mapping subscription -> when its
|
||||
# expiry warning last fired. Without it every subscription already announced
|
||||
# announces itself again on the restored box — the exact re-alert storm the file
|
||||
# was created to prevent (shater/alert/expiry.go).
|
||||
/etc/shater/alert-state.json
|
||||
|
||||
# DELIBERATELY NOT KEPT. Each of these is history or cache, and the backup is
|
||||
# built in RAM:
|
||||
#
|
||||
# /etc/shater/stats.db Traffic/query HISTORY, not configuration. Bounded
|
||||
# only by globals.stats_disk_limit_mb, whose default is
|
||||
# 64 MB and whose 0 means UNLIMITED — one file able to
|
||||
# outweigh everything else here by two orders of
|
||||
# magnitude, and the only entry whose loss costs the
|
||||
# operator nothing but a chart.
|
||||
# /etc/shater/cache.db sing-box's own cache (8 MiB cap, deleted above it).
|
||||
# Rebuilt on demand by design, and a stale rule-set
|
||||
# cache carried onto a different box is worse than no
|
||||
# cache at all.
|
||||
# /etc/shater/shaterd.log A log (capped by globals.log_max_kb). A restored box
|
||||
# wants its own log, and this one carries the DNS query
|
||||
# history of the box it came from — which is not
|
||||
# something to move into an archive a person then puts
|
||||
# somewhere else.
|
||||
#
|
||||
# ON SECRECY, since this archive routinely ends up in cloud storage: subs/*.json
|
||||
# carries every node's credentials (UUID/password/keys). That is not a NEW
|
||||
# exposure — /etc/config/shater already carries the subscription URLs and every
|
||||
# manual node's credentials, and it is already in the backup as a conffile — but a
|
||||
# shater backup is a secret-bearing file and should be treated as one.
|
||||
+8
-1
@@ -351,8 +351,15 @@ function UnauthPlate() {
|
||||
<Faceplate ariaLabel="shater — not authenticated" header={<FaceplateHeader wordmark="SHATER" subline="v0.2 · openwrt appliance" />}>
|
||||
<div className="plate-msg">
|
||||
<Module name="Session" value="LOCKED" led={{ variant: 'amber' }}>
|
||||
{/* THE ONLY RECOVERY INSTRUCTION THE PRODUCT GIVES, so it has to point at
|
||||
the real menu entry. It said "System → shater"; the page is registered
|
||||
at `admin/services/shater` (luci-app-shater/root/usr/share/luci/menu.d/
|
||||
luci-app-shater.json, title "Shater"), which LuCI renders under
|
||||
SERVICES. Anyone reading this line has just lost access to the panel
|
||||
and is looking for the one door back — sending them to the wrong menu
|
||||
costs far more than its size. */}
|
||||
<p className="placeholder-note">
|
||||
No active session. Open the panel from the LuCI menu (System → shater →{' '}
|
||||
No active session. Open the panel from the LuCI menu (Services → Shater →{' '}
|
||||
<strong>Open panel</strong>) to hand off a fresh access token.
|
||||
</p>
|
||||
</Module>
|
||||
|
||||
+150
-15
@@ -112,13 +112,55 @@ export interface StatusWarning {
|
||||
/** GET /api/status — live daemon + data-plane state. */
|
||||
export interface Status {
|
||||
running: boolean
|
||||
/**
|
||||
* `globals.enabled` in UCI — MEANINGLESS unless `config_readable` is true. Test
|
||||
* that first; see it for why "off" and "cannot tell" must never share a branch.
|
||||
*/
|
||||
enabled: boolean
|
||||
active: boolean
|
||||
table: boolean
|
||||
hash: string
|
||||
version: string
|
||||
// The RAW `option kill_switch` string, echoed straight off model.Globals
|
||||
// (apply.go: `s.KillSwitch = m.Globals.KillSwitch`) — NOT normalised. So it can
|
||||
// be "Closed", " closed ", or "" as well as the two documented spellings, and
|
||||
// the daemon reads it as closed unless it case-insensitively equals "open"
|
||||
// (apply.killSwitchClosed). Never compare it with `===`; use
|
||||
// planeState.killSwitchClosed, which is that same rule.
|
||||
kill_switch?: string // "closed" (fail-closed) | "open"
|
||||
panel_port?: number // configured admin-panel port (default 8088)
|
||||
// The CONFIGURED admin-panel port (default 8088) — NOT a port anything has
|
||||
// confirmed is being listened on. The daemon echoes the config value, and a
|
||||
// failed listen is only `logger.Warn("panel server unavailable (daemon
|
||||
// continues)")`, so this field reads exactly the same whether the panel is up or
|
||||
// was never bound. Do not render it as "the panel is at :N": say configured.
|
||||
panel_port?: number
|
||||
/**
|
||||
* COULD THE CONFIGURATION BE READ AT ALL when this status was taken?
|
||||
*
|
||||
* It QUALIFIES the only three fields sourced from the config — `enabled`,
|
||||
* `kill_switch`, `panel_port`. When it is false those three are zero values and
|
||||
* mean NOTHING: not "switched off", not "kill switch unset", not "port 0". Test
|
||||
* it before reading any of them.
|
||||
*
|
||||
* The phrasing is positive on purpose, and the panel must keep it that way: a
|
||||
* client that predates the field sees it missing, reads `false`, and lands on the
|
||||
* ALARMING side. Reading it as `!== false` would invert that and hand the
|
||||
* reassuring branch to every daemon too old to answer.
|
||||
*
|
||||
* Why it matters more than it looks: the failure is a full /overlay, or a
|
||||
* `uci commit` caught half-written — precisely when the fail-closed plane has the
|
||||
* whole LAN cut off on purpose. The daemon then publishes `plane:"hold"` WITH
|
||||
* `enabled:false`, and a panel that checks `!enabled` first renders "Turned off",
|
||||
* amber, no alarm, and points at a Settings page backed by the same unreadable
|
||||
* file. It tells the owner they did this to themselves while the house has no
|
||||
* internet. See planeState.protectionState, where the order of those two checks
|
||||
* is the whole fix.
|
||||
*/
|
||||
config_readable?: boolean
|
||||
/** Why the config read failed, verbatim, or absent/"" when it did not. A
|
||||
* diagnostic, not the operator-facing sentence — that one is published as a
|
||||
* critical warning in section "config", name "unreadable". */
|
||||
config_error?: string
|
||||
can_rollback?: boolean // a rollback would revert something (armed snapshot or engine last-good)
|
||||
// Is the sing-box engine process actually up? Absent on older daemons.
|
||||
engine_running?: boolean
|
||||
@@ -583,12 +625,46 @@ export interface Stats {
|
||||
|
||||
// --- Model shapes (PascalCase keys; slices may be null) ---------------------
|
||||
|
||||
/**
|
||||
* `T` with every key REQUIRED to be written down — `undefined` still allowed as a
|
||||
* VALUE, so nothing changes on the wire (`JSON.stringify` omits undefined, and the
|
||||
* result stays assignable to `T`).
|
||||
*
|
||||
* WHAT IT IS FOR. Several editors REBUILD a model object from their form controls
|
||||
* instead of extending the one they were given, because rebuilding is what stops a
|
||||
* stale field surviving a change of shape (an inbound switched from `socks` to
|
||||
* `tproxy` must not keep its old `Listen`). The cost is that the rebuild is only
|
||||
* correct for as long as somebody remembers to touch it: add a sixteenth field to
|
||||
* `Inbound` and every edit silently drops it, with no error anywhere. That already
|
||||
* happened once, to `Ruleset.Format` (see ruleset.ts), and the value could only be
|
||||
* put back over SSH.
|
||||
*
|
||||
* Annotating the rebuilt literal `Complete<T>` turns the next occurrence into a
|
||||
* BUILD failure, in the function that has to decide, naming the field it forgot.
|
||||
* Writing `undefined` for a field this shape has no use for is then a statement
|
||||
* rather than an omission.
|
||||
*
|
||||
* Only the OPTIONAL keys get `| undefined`. A bare `{[K in keyof T]-?: T[K] |
|
||||
* undefined}` would widen the required ones too — `Name: string | undefined` —
|
||||
* which both weakens them and stops the result being assignable back to `T`; an
|
||||
* intersection with `T` does not fix it either, since intersecting an optional
|
||||
* `string` with `string | undefined` collapses back to `string`. So the two halves
|
||||
* are split explicitly.
|
||||
*/
|
||||
type OptionalKeys<T> = { [K in keyof T]-?: object extends Pick<T, K> ? K : never }[keyof T]
|
||||
export type Complete<T> = Pick<T, Exclude<keyof T, OptionalKeys<T>>> & {
|
||||
[K in OptionalKeys<T>]-?: T[K] | undefined
|
||||
}
|
||||
|
||||
export interface Globals {
|
||||
Enabled: boolean
|
||||
// debug|info|warning|error|none. `none` really is silent — it is emitted as the
|
||||
// engine's own log-disable switch, not as a quieter level. Anything unrecognised
|
||||
// falls back to warning (an unknown level fails engine start outright).
|
||||
LogLevel: string
|
||||
// The saved policy, verbatim. Same caveat as Status.kill_switch: the daemon
|
||||
// normalises with EqualFold+TrimSpace and defaults to CLOSED, so read it through
|
||||
// planeState.killSwitchClosed rather than comparing the string.
|
||||
KillSwitch: string // "closed" | "open"
|
||||
// There is deliberately no DNSMode. It was removed from the Go model (see
|
||||
// model.go's package comment): "nftset" named the v0.1 dnsmasq architecture that
|
||||
@@ -621,19 +697,60 @@ export interface Globals {
|
||||
* no representation in them. So this traffic can only be dropped or let out
|
||||
* directly; there is no third physical option, and the setting picks WHICH.
|
||||
*
|
||||
* block — (default) drop it all. No ping/traceroute out, no multicast IPTV,
|
||||
* no client IPsec/PPTP passthrough. Nothing leaks.
|
||||
* block — (default) drop it all. No ping/traceroute out, no client
|
||||
* IPsec/PPTP passthrough. Nothing leaks.
|
||||
* icmp — let ICMP/ICMPv6 echo out directly. Ping and traceroute work; the
|
||||
* host being pinged sees the real WAN IP. IPTV/VPN passthrough stay
|
||||
* blocked.
|
||||
* direct — let all of it out directly. Ping, IPTV and IPsec/PPTP passthrough
|
||||
* work, and all of it bypasses the tunnel with the real IP.
|
||||
* host being pinged sees the real WAN IP. The rest gets out only
|
||||
* toward addresses the routing rules already send direct.
|
||||
* direct — let all of it out directly. Ping and IPsec/PPTP passthrough work,
|
||||
* and all of it bypasses the tunnel with the real IP.
|
||||
*
|
||||
* MULTICAST IPTV IS NOT ONE OF THE THINGS THIS DECIDES, on any of the three.
|
||||
* The stream is UDP, every line this policy emits carries `meta l4proto !=
|
||||
* { tcp, udp }` (netplane/untunnelable.go), and the fail-closed forward chain's
|
||||
* surviving accepts cover the RFC1918/link-local daddr sets only — 224.0.0.0/4
|
||||
* is not among them (netplane/nft.go). The daemon says so itself in the notes it
|
||||
* publishes for this section. Listing IPTV as something `direct` restores is the
|
||||
* one lie this comment previously told.
|
||||
*
|
||||
* Absent, empty, or unrecognised ⇒ `block` (the daemon normalises to the safe
|
||||
* side). Note the policy is inert while KillSwitch is "open", because then
|
||||
* nothing is being blocked in the first place.
|
||||
* side). Three other settings override it, and the panel must read them before
|
||||
* describing it: an OPEN KillSwitch (the forward chain then has no drops at all,
|
||||
* so nothing is blocked whatever this says), L3Tunnel (takes ICMP echo into the
|
||||
* tunnel in prerouting, before the forward chain is consulted), and
|
||||
* UntunnelableEgress (routes ESP/AH/GRE/IGMP/SCTP — and ICMP too, when L3Tunnel
|
||||
* is off — out a named interface). See netplane/nft.go's prerouting chain.
|
||||
*/
|
||||
Untunnelable?: string // block|icmp|direct
|
||||
/**
|
||||
* The L3 ingress (model.Globals.L3Tunnel, UCI `l3_tunnel`). The engine opens a
|
||||
* TUN device and prerouting policy-routes ICMP ECHO AND NOTHING ELSE into it,
|
||||
* where the engine's own route rules pick the outbound — so ping and Windows
|
||||
* tracert travel THROUGH the tunnel toward every address the rules send to an
|
||||
* outbound that can carry plain IP (WireGuard/AmneziaWG), and are not answered
|
||||
* at all for addresses routed to a stream-only outbound. Raw ESP/AH/GRE cannot
|
||||
* enter it: sing-tun's dispatcher NATs through a port-shaped selector they do
|
||||
* not have.
|
||||
*
|
||||
* Load-bearing for the panel because it happens BEFORE the forward chain, so it
|
||||
* silently rewrites the ICMP half of every Untunnelable promise. Absent ⇒ false
|
||||
* (opt-in), so read it as `=== true`.
|
||||
*/
|
||||
L3Tunnel?: boolean
|
||||
/**
|
||||
* The name of an interface/tunnel egress that carries what the engine will not
|
||||
* dispatch — ESP, AH, GRE, IGMP, SCTP, plus ICMP when L3Tunnel is off
|
||||
* (model.Globals.UntunnelableEgress, UCI `untunnelable_egress`). The kernel
|
||||
* routes those packets out that device with its own NAT; the forward chain,
|
||||
* where Untunnelable's verdicts live, never decides them.
|
||||
*
|
||||
* It is NOT necessarily a tunnel — the option takes any interface/tunnel egress,
|
||||
* and a second WAN is just another uplink whose real address the far end sees.
|
||||
* The daemon's own note for this section says which, from whether the device is
|
||||
* point-to-point, so the panel does not guess. "" (the default) ⇒ Untunnelable
|
||||
* is in sole charge.
|
||||
*/
|
||||
UntunnelableEgress?: string
|
||||
DNSFilter?: boolean // master enable for the in-engine blocklist filter
|
||||
DNSIntercept?: boolean // force ALL LAN plaintext DNS (:53) through the engine, incl. router-addressed queries
|
||||
BlockDoH?: boolean // block known public DoH resolvers (by host + IP:443 + Firefox canary) so clients fall back to plaintext :53
|
||||
@@ -663,13 +780,20 @@ export interface Globals {
|
||||
StatsTimelineMinutes?: number // trailing per-minute sparkline buckets; 0 = unlimited
|
||||
StatsMaxDomains?: number // network-wide domain-map size before prune; 0 = unlimited
|
||||
StatsRetentionDisabled?: boolean // master switch: disable trimming for all aggregates
|
||||
// SQLite-only: hard cap on the on-disk stats.db size, in MB. 0 = unlimited (bounded
|
||||
// only by the device); a positive N caps the DB at N MB (oldest rows pruned + VACUUM).
|
||||
// Applies only when StatsBackend === "sqlite"; ignored for off/memory.
|
||||
// Disk-backend only: hard cap on the on-disk stats.db size, in MB. 0 = unlimited
|
||||
// (bounded only by the device); a positive N aims the DB at N MB — oldest rows are
|
||||
// deleted, then the file is REBUILT to give the pages back (stats/boltring.go:
|
||||
// bbolt.Compact into a temp file + atomic swap). Not sqlite and not VACUUM: the
|
||||
// store is bbolt, and the rebuild is SKIPPED when the filesystem cannot fit the
|
||||
// transient second copy, which leaves the DB over its cap until space frees up.
|
||||
// Applies only when StatsBackend === "sqlite" (a historical value name — see
|
||||
// StatsBackend below); ignored for off/memory.
|
||||
StatsDiskLimitMB?: number // stats.db disk cap in MB; 0 = unlimited
|
||||
// Logging/stats backend selector. "off" collects nothing (all stats lists empty);
|
||||
// "memory" keeps aggregates in RAM (lost on restart); "sqlite" persists the query
|
||||
// and connection logs to an on-disk, disk-bounded store that survives a restart
|
||||
// "memory" keeps aggregates in RAM (lost on restart); "sqlite" — a HISTORICAL value
|
||||
// name, kept because it is on disk in every shipped config; the store behind it is
|
||||
// bbolt (stats/boltring.go), pure Go and already linked into the binary — persists
|
||||
// the query and connection logs to an on-disk, disk-bounded store that survives a restart
|
||||
// (the DNS/nft aggregates stay in RAM; if the DB can't be opened it falls back to
|
||||
// memory and the snapshot honestly reports "memory"). Default "memory". The
|
||||
// EFFECTIVE running backend is echoed on the /api/stats snapshot.
|
||||
@@ -959,11 +1083,22 @@ export interface Inbound {
|
||||
* address. Route to a group/node/chain for the former, and to the `block` TARGET
|
||||
* for the latter.
|
||||
*/
|
||||
/*
|
||||
* THERE IS DELIBERATELY NO `Target`. It was declared here as "legacy field of the
|
||||
* removed `proxy` type; read by nothing", and it is not in the Go model at all —
|
||||
* so it could only ever be `undefined`, which made every panel branch asking
|
||||
* "which egress points at this node/group?" (Nodes.tsx, Targets.tsx) permanently
|
||||
* unreachable: dead code that read as coverage. The other end was worse than
|
||||
* inert. PUT /api/config decodes with DisallowUnknownFields, so the day anything
|
||||
* had put a string on it — a rename pass, a migration, a hand-written fixture —
|
||||
* `JSON.stringify` would have started emitting it and the daemon would have
|
||||
* rejected the WHOLE write with `json: unknown field "Target"` (verified against
|
||||
* the running daemon), losing an unrelated edit somewhere else on the page.
|
||||
*/
|
||||
export interface Egress {
|
||||
Name: string
|
||||
Type: string // interface|direct|byedpi
|
||||
Interface?: string // type=interface: the UCI interface name
|
||||
Target?: string // legacy field of the removed `proxy` type; read by nothing
|
||||
Port?: number // type=byedpi ONLY: the local ciadpi listen port (default 1080)
|
||||
DPI?: string // type=interface|direct: off|fragment|record|spoof (byedpi desyncs itself)
|
||||
}
|
||||
|
||||
@@ -0,0 +1,116 @@
|
||||
import { test } from 'node:test'
|
||||
import assert from 'node:assert/strict'
|
||||
import type { Egress } from './api.ts'
|
||||
import {
|
||||
DPI_TYPES,
|
||||
EGRESS_TYPES,
|
||||
UNKNOWN_EGRESS_TYPE_HINT,
|
||||
isKnownEgressType,
|
||||
nextEgress,
|
||||
} from './egressEdit.ts'
|
||||
|
||||
// The egress editor's save merge. The defect these pin: the submit handler
|
||||
// cleared Interface, Port and DPI for every type it did not have a branch for —
|
||||
// including types it renders no field for at all — so opening an egress the
|
||||
// panel calls "(unknown)" and changing only its NAME deleted its interface. On a
|
||||
// `tunnel` egress that was live damage: the data plane routed it for real, and
|
||||
// `untunnelable_egress` resolves through exactly that field, so the ESP/AH/GRE/
|
||||
// IGMP/SCTP carrier silently stopped existing and those protocols fell back to
|
||||
// the untunnelable policy.
|
||||
|
||||
test('an unknown type keeps the fields the editor never showed', () => {
|
||||
const initial: Egress = { Name: 'vpn', Type: 'wireguard', Interface: 'wg0', DPI: 'fragment' }
|
||||
// The form as the editor would hold it for an unknown type: no Interface
|
||||
// input is rendered, no port input, no DPI select. Only the name was touched.
|
||||
const out = nextEgress(initial, {
|
||||
name: 'vpn-renamed',
|
||||
type: 'wireguard',
|
||||
iface: 'wg0',
|
||||
port: '',
|
||||
dpi: 'off',
|
||||
})
|
||||
assert.equal(out.Name, 'vpn-renamed')
|
||||
assert.equal(out.Type, 'wireguard')
|
||||
assert.equal(
|
||||
out.Interface,
|
||||
'wg0',
|
||||
'renaming an egress whose type this panel does not know must not delete its interface',
|
||||
)
|
||||
assert.equal(out.DPI, 'fragment', 'nor any other field the form declined to display')
|
||||
})
|
||||
|
||||
test('an unknown type with a blank form state still keeps what was stored', () => {
|
||||
// The stricter version: the editor's `iface` state is seeded from `initial`,
|
||||
// so a test that passes the same value back could pass on a broken merge too.
|
||||
// Blank the form and the stored value must still survive.
|
||||
const initial: Egress = { Name: 'vpn', Type: 'wireguard', Interface: 'wg0', Port: 9050 }
|
||||
const out = nextEgress(initial, { name: 'vpn', type: 'wireguard', iface: '', port: '', dpi: '' })
|
||||
assert.equal(out.Interface, 'wg0')
|
||||
assert.equal(out.Port, 9050)
|
||||
})
|
||||
|
||||
test('the inputs are not mutated — the caller keeps a usable `initial`', () => {
|
||||
const initial: Egress = { Name: 'vpn', Type: 'wireguard', Interface: 'wg0' }
|
||||
nextEgress(initial, { name: 'other', type: 'interface', iface: 'wan2', port: '', dpi: 'off' })
|
||||
assert.deepEqual(initial, { Name: 'vpn', Type: 'wireguard', Interface: 'wg0' })
|
||||
})
|
||||
|
||||
test('a known type still clears the fields it does not use', () => {
|
||||
// The other half of the contract: for a type the editor DOES render, stale
|
||||
// settings from the previous type must go, or the config keeps a value the new
|
||||
// type ignores and the panel shows a setting that does nothing.
|
||||
const initial: Egress = { Name: 'e', Type: 'interface', Interface: 'wan2', DPI: 'fragment' }
|
||||
const out = nextEgress(initial, { name: 'e', type: 'byedpi', iface: 'wan2', port: '1081', dpi: 'fragment' })
|
||||
assert.equal(out.Type, 'byedpi')
|
||||
assert.equal(out.Interface, undefined, 'a byedpi egress has no interface')
|
||||
assert.equal(out.Port, 1081)
|
||||
assert.equal(out.DPI, undefined, 'byedpi desyncs itself; the native preset is not applied')
|
||||
})
|
||||
|
||||
test('an interface egress carries its interface and DPI, and no port', () => {
|
||||
const out = nextEgress(undefined, {
|
||||
name: ' wan-direct ',
|
||||
type: 'interface',
|
||||
iface: ' wan2 ',
|
||||
port: '1080',
|
||||
dpi: 'record',
|
||||
})
|
||||
assert.equal(out.Name, 'wan-direct', 'the name is trimmed')
|
||||
assert.equal(out.Interface, 'wan2', 'the interface is trimmed')
|
||||
assert.equal(out.Port, undefined, 'only a byedpi egress dials a port')
|
||||
assert.equal(out.DPI, 'record')
|
||||
})
|
||||
|
||||
test('a byedpi egress with no port falls back to the ciadpi default', () => {
|
||||
const out = nextEgress(undefined, { name: 'b', type: 'byedpi', iface: '', port: ' ', dpi: 'off' })
|
||||
assert.equal(out.Port, 1080)
|
||||
})
|
||||
|
||||
test('the type list is the closed set the daemon builds outbounds for', () => {
|
||||
// model.KnownEgressTypes. `tunnel` must NOT be here: the daemon folds it to
|
||||
// `interface` on read (model.NormalizeEgressTypes), so the panel receives the
|
||||
// canonical spelling and a second entry would put the split back into the UI.
|
||||
assert.deepEqual(
|
||||
EGRESS_TYPES.map((t) => t.id),
|
||||
['interface', 'direct', 'byedpi'],
|
||||
)
|
||||
assert.equal(isKnownEgressType('tunnel'), false)
|
||||
assert.equal(isKnownEgressType('interface'), true)
|
||||
assert.deepEqual([...DPI_TYPES].sort(), ['direct', 'interface'])
|
||||
})
|
||||
|
||||
test('the unknown-type hint describes what actually happens, both halves of it', () => {
|
||||
// It has to name the ROUTING as well as the outbound — the old text said only
|
||||
// "this engine builds no outbound", which was false for the one unknown type
|
||||
// anybody had, because the data plane was building that egress a real routing
|
||||
// table at the same time. And it has to promise what nextEgress now keeps.
|
||||
assert.match(UNKNOWN_EGRESS_TYPE_HINT, /routing rule or table/)
|
||||
assert.match(UNKNOWN_EGRESS_TYPE_HINT, /no outbound/)
|
||||
assert.match(UNKNOWN_EGRESS_TYPE_HINT, /blocked/)
|
||||
assert.match(UNKNOWN_EGRESS_TYPE_HINT, /leaves this egress’s other settings/)
|
||||
assert.doesNotMatch(
|
||||
UNKNOWN_EGRESS_TYPE_HINT,
|
||||
/This engine builds no outbound for that type/,
|
||||
'the superseded sentence claimed the engine was the only half involved',
|
||||
)
|
||||
})
|
||||
@@ -0,0 +1,142 @@
|
||||
import type { Egress } from './api'
|
||||
|
||||
/**
|
||||
* The egress editor's data half — the closed type list and the one function that
|
||||
* decides which fields a save writes.
|
||||
*
|
||||
* It lives outside `pages/Targets.tsx` because it is the part that must be
|
||||
* TESTED, and the panel's test runner is `node --test src/*.test.ts`: plain
|
||||
* modules only, no JSX, no DOM. Extracting it is not tidiness — the bug below
|
||||
* shipped precisely because "which fields does Save write?" was three lines
|
||||
* buried in a submit handler that nothing could call.
|
||||
*/
|
||||
|
||||
/**
|
||||
* The three egress kinds that produce a real way out, in the order the editor
|
||||
* offers them. This is the panel's copy of `model.KnownEgressTypes` and must
|
||||
* stay equal to it: the daemon builds no outbound for anything else, and the
|
||||
* router installs no mark, no `ip rule` and no routing table for it either, so
|
||||
* every node, group and rule bound to such an egress is blocked.
|
||||
*
|
||||
* `tunnel` is deliberately NOT here. It is an accepted spelling in
|
||||
* `/etc/config/shater`, but the daemon folds it to `interface` on read
|
||||
* (model.NormalizeEgressTypes), so an egress written that way arrives at this
|
||||
* panel already saying `interface` — with its Interface field rendered, its
|
||||
* blurb correct and no "(unknown)" label. Adding a fourth entry here would put
|
||||
* the second spelling back into a UI that has to agree with two backend halves.
|
||||
*
|
||||
* `proxy` and `block` were removed: neither ever created an outbound, so
|
||||
* everything bound to them fell through to the plain WAN with the real address.
|
||||
* Send traffic through a proxy by routing it at a group/node/chain, and drop it
|
||||
* with the `block` target on a rule.
|
||||
*/
|
||||
export const EGRESS_TYPES: ReadonlyArray<{ id: string; label: string; blurb: string }> = [
|
||||
{
|
||||
id: 'interface',
|
||||
label: 'Interface — out a specific WAN or tunnel',
|
||||
blurb: 'Binds to one device (wan, wg0, …) so this traffic leaves over that uplink.',
|
||||
},
|
||||
{
|
||||
id: 'direct',
|
||||
label: 'Direct — straight out, with an optional DPI preset',
|
||||
blurb: 'Uses the normal route. Its point is the DPI preset below, applied to what you route here.',
|
||||
},
|
||||
{
|
||||
id: 'byedpi',
|
||||
label: 'ByeDPI — through the local ciadpi desync proxy',
|
||||
blurb: 'Hands traffic to ciadpi on 127.0.0.1, which desyncs it and goes out direct.',
|
||||
},
|
||||
]
|
||||
|
||||
/** Lookup by id, or undefined when the stored type is not one this panel knows. */
|
||||
export function egressTypeInfo(type: string): (typeof EGRESS_TYPES)[number] | undefined {
|
||||
return EGRESS_TYPES.find((t) => t.id === type)
|
||||
}
|
||||
|
||||
/** Whether this panel has a definition — and therefore its own fields — for the type. */
|
||||
export function isKnownEgressType(type: string): boolean {
|
||||
return egressTypeInfo(type) !== undefined
|
||||
}
|
||||
|
||||
/**
|
||||
* Types whose native DPI-bypass preset applies. NOT byedpi: the desync happens
|
||||
* inside the ciadpi process, and the engine's tls_* flags are never stamped on
|
||||
* top of it — so the control is hidden there rather than accepted and dropped.
|
||||
*/
|
||||
export const DPI_TYPES: ReadonlySet<string> = new Set(['interface', 'direct'])
|
||||
|
||||
/**
|
||||
* What the editor shows under the Type select when the stored type is not one of
|
||||
* the three. It has to describe what the router actually does with such an
|
||||
* egress, and what THIS FORM does to it on save — both halves were wrong before.
|
||||
*
|
||||
* It used to read: "This engine builds no outbound for that type, so everything
|
||||
* routed here is blocked. Pick one above." Two problems. It said "this engine",
|
||||
* as if only the engine were involved, at a time when the data plane happily
|
||||
* built an `ip rule`, a routing table and a mark bypass for a `tunnel` egress and
|
||||
* `untunnelable_egress` carried live ESP/GRE out of it — so the sentence was
|
||||
* flatly false for the one unknown type anybody had. And it stayed silent about
|
||||
* the thing this form was doing to the egress: saving it wiped `Interface`,
|
||||
* because the field is only rendered for `type === 'interface'` and the submit
|
||||
* handler cleared every field it did not render. That is fixed in nextEgress
|
||||
* below, and the text now says so, because a promise about saving is only worth
|
||||
* making next to the code that keeps it.
|
||||
*/
|
||||
export const UNKNOWN_EGRESS_TYPE_HINT =
|
||||
'The router does not recognise this type: it builds no outbound for it and installs no ' +
|
||||
'routing rule or table, so everything routed here is blocked — never sent out over the ' +
|
||||
'plain WAN. Pick a type above to fix it; saving leaves this egress’s other settings ' +
|
||||
'exactly as they are until you do.'
|
||||
|
||||
/** The editor's form state, as strings straight out of the inputs. */
|
||||
export interface EgressForm {
|
||||
name: string
|
||||
type: string
|
||||
iface: string
|
||||
port: string
|
||||
dpi: string
|
||||
}
|
||||
|
||||
/**
|
||||
* Merge the form back onto the egress being edited.
|
||||
*
|
||||
* # The rule, and why it is the rule
|
||||
*
|
||||
* A save may only CLEAR a field the editor was in a position to show. For the
|
||||
* three known types the editor renders every field that type uses, so clearing
|
||||
* the others is right: switching `interface` → `byedpi` must drop the stale
|
||||
* interface name, or the config keeps a setting the new type ignores.
|
||||
*
|
||||
* For a type this panel has no definition for, the editor renders NONE of those
|
||||
* fields — and used to clear all three anyway:
|
||||
*
|
||||
* base.Interface = type === 'interface' ? iface.trim() : undefined
|
||||
*
|
||||
* So opening an egress the panel calls "(unknown)", changing nothing but its
|
||||
* name, and pressing Save silently deleted its `interface`. That was not
|
||||
* hypothetical damage. `tunnel` was such a type, the data plane routed it for
|
||||
* real, and `untunnelable_egress` pointing at it is resolved by
|
||||
* netplane.UntunnelableEgressBinding through exactly that field: with the
|
||||
* interface gone the binding fails, the ESP/AH/GRE/IGMP/SCTP protection quietly
|
||||
* stops existing, and those protocols fall back to the untunnelable policy —
|
||||
* from one rename, with no message anywhere.
|
||||
*
|
||||
* The daemon no longer hands this panel a `tunnel` (it is folded to `interface`
|
||||
* on read), so that particular type is gone. The rule stays, because the next
|
||||
* type the backend gains before the panel learns it would repeat the whole
|
||||
* thing: an editor must not delete what it declines to display.
|
||||
*
|
||||
* `initial` is never mutated — the caller keeps a usable object if the save
|
||||
* fails.
|
||||
*/
|
||||
export function nextEgress(initial: Egress | undefined, form: EgressForm): Egress {
|
||||
const type = form.type
|
||||
const base: Egress = { ...(initial ?? ({} as Egress)), Name: form.name.trim(), Type: type }
|
||||
// There is no `Target` to clear — the field is not in the Go model, so GET
|
||||
// never delivers one and the spread above cannot produce one.
|
||||
if (!isKnownEgressType(type)) return base
|
||||
base.Interface = type === 'interface' ? form.iface.trim() : undefined
|
||||
base.Port = type === 'byedpi' ? Number(form.port.trim()) || 1080 : undefined
|
||||
base.DPI = DPI_TYPES.has(type) ? form.dpi : undefined
|
||||
return base
|
||||
}
|
||||
+97
-4
@@ -6,8 +6,15 @@
|
||||
// state mutates in-memory so the Apply / Confirm / Rollback flow is exercisable.
|
||||
//
|
||||
// Type-only imports from api.ts (erased at build) keep this free of a runtime cycle.
|
||||
import { killSwitchClosed } from './planeState'
|
||||
import type { ApplyResult, ChainHealth, ChainHopHealth, ConnLogEntry, DiscoveredDevice, GroupHealth, GroupMemberHealth, GroupsHealth, GroupTestResult, GroupTestStart, GroupTestStatus, Interface, Model, Profile, QueryLogEntry, RuleReach, RulesReachability, RulesetCategories, RulesetCheck, RulesetStatus, Stats, StatsLogPage, StatsLogQuery, Status, StatusWarning, Traffic } from './api'
|
||||
|
||||
/** One URL knob, safe to read before `location` exists (SSR-less builds/tests). */
|
||||
function mockParam(name: string): string | null {
|
||||
if (typeof location === 'undefined') return null
|
||||
return new URLSearchParams(location.search).get(name)
|
||||
}
|
||||
|
||||
let armed = false // a pending commit-confirm auto-rollback
|
||||
let hasLastGood = false // a predecessor config exists to roll back to (post-apply)
|
||||
let hash = 'sha256:9f7c07e8d8ac4ae1'
|
||||
@@ -33,7 +40,13 @@ const CONFIG: Model = {
|
||||
ActiveProfile: 'mobile-uplink',
|
||||
// Policy for traffic TPROXY physically can't carry (non-TCP/UDP). Override
|
||||
// from the URL — ?mock&untun=icmp / &untun=direct — to see all three states.
|
||||
Untunnelable: new URLSearchParams(typeof location === 'undefined' ? '' : location.search).get('untun') ?? 'block',
|
||||
Untunnelable: mockParam('untun') ?? 'block',
|
||||
// The two settings that decide part of the untunnelable traffic BEFORE the
|
||||
// policy above is consulted, so the Networks copy has to change shape for
|
||||
// them: ?mock&l3=1 (ping rides the tunnel) and ?mock&uegress=wg0 (the kernel
|
||||
// routes ESP/GRE/SCTP out that interface). Both off by default, as shipped.
|
||||
L3Tunnel: mockParam('l3') === '1',
|
||||
UntunnelableEgress: mockParam('uegress') ?? '',
|
||||
DNSIntercept: true, // force ALL LAN plaintext DNS (:53) through the engine
|
||||
BlockDoH: false, // block known public DoH resolvers so clients fall back to plaintext :53
|
||||
GroupHealth: true, // observatory: background probing of used groups/chains + Targets health stats (default on)
|
||||
@@ -42,7 +55,7 @@ const CONFIG: Model = {
|
||||
StatsMaxDomains: 5000, // fixed cap — shows the "limit" rendering (5000)
|
||||
StatsRetentionDisabled: false,
|
||||
StatsBackend: 'memory', // logging backend: off | memory | sqlite
|
||||
StatsDiskLimitMB: 64, // SQLite-only disk cap (MB); shows once backend=sqlite (0 ⇒ Unlimited)
|
||||
StatsDiskLimitMB: 64, // disk-backend-only cap (MB); shows once backend=sqlite (0 ⇒ Unlimited)
|
||||
// Daemon operational log (shaterd's own log): both destinations on, file in
|
||||
// tmpfs (the default), 2 MB cap — the defaults a fresh install ships with.
|
||||
LogToSyslog: true,
|
||||
@@ -511,7 +524,12 @@ function mockPlane(): {
|
||||
} {
|
||||
const q = typeof location === 'undefined' ? '' : location.search
|
||||
const params = new URLSearchParams(q)
|
||||
const killSwitch = params.get('ks') === 'open' ? 'open' : 'closed'
|
||||
// Passed through VERBATIM, because that is what the daemon does: apply.go sets
|
||||
// `s.KillSwitch = m.Globals.KillSwitch` with no normalisation, so `?ks=Closed`,
|
||||
// `?ks=%20closed%20` and `?ks=` are all reachable readings of a router that
|
||||
// BLOCKS. The mock used to fold everything that wasn't "open" to "closed",
|
||||
// which made the panel's own `=== 'closed'` bug unreproducible here.
|
||||
const killSwitch = params.get('ks') ?? 'closed'
|
||||
const p = params.get('plane')
|
||||
if (p === 'hold') return { plane: 'hold', engine: false, killSwitch: 'closed' }
|
||||
if (p === 'none') return { plane: 'none', engine: false, killSwitch }
|
||||
@@ -606,6 +624,21 @@ const MOCK_WARNINGS: StatusWarning[] = [
|
||||
* "this list is incomplete" rendering is exercisable — it used to be dropped
|
||||
* wholesale by the panel's `info` filter and reached no screen at all.
|
||||
*/
|
||||
/**
|
||||
* The daemon's own critical finding when it cannot read the configuration
|
||||
* (apply.go, section "config" / name "unreadable"). Copied close to verbatim: the
|
||||
* sentence about NOT switching anything off is the load-bearing one — the instinct
|
||||
* in front of a dead LAN is to turn things off, and that is the single action that
|
||||
* makes this worse.
|
||||
*/
|
||||
const CONFIG_UNREADABLE_WARNING: StatusWarning = {
|
||||
severity: 'critical',
|
||||
section: 'config',
|
||||
name: 'unreadable',
|
||||
message:
|
||||
"the router's configuration could NOT be read (uci show shater: exit status 1), so this status cannot say whether shater is switched on, whether the kill switch is closed, or which port this panel is served on — enabled, kill_switch and panel_port are placeholders here, not readings. If traffic is being blocked, that is the fail-closed plane doing its job and NOT the service being switched off: do not turn anything off to fix it. The usual causes are a full /overlay and a `uci commit` interrupted part-way; free space, check /etc/config/shater, then restart shaterd.",
|
||||
}
|
||||
|
||||
const MOCK_TRUNCATION: StatusWarning = {
|
||||
severity: 'info',
|
||||
section: 'generate',
|
||||
@@ -620,7 +653,36 @@ const MOCK_TRUNCATION: StatusWarning = {
|
||||
* is inert entirely while the kill-switch is open.
|
||||
*/
|
||||
function untunnelableNote(mode: string, killSwitch: string): StatusWarning[] {
|
||||
if (killSwitch === 'open') {
|
||||
const g = CONFIG.Globals as { L3Tunnel?: boolean; UntunnelableEgress?: string }
|
||||
const egress = (g.UntunnelableEgress ?? '').trim()
|
||||
// The daemon's own precedence: the egress carrier owns the whole story, then
|
||||
// the L3 ingress, then the kill-switch, then the policy (apply/warnings.go).
|
||||
if (egress) {
|
||||
return [
|
||||
{
|
||||
severity: 'info',
|
||||
section: 'untunnelable',
|
||||
name: egress,
|
||||
message:
|
||||
(g.L3Tunnel
|
||||
? 'ping and Windows tracert travel THROUGH the tunnel; everything else the tunnel cannot carry — IPsec (ESP/AH), PPTP/GRE, SCTP — now leaves'
|
||||
: 'ping, Windows tracert, IPsec (ESP/AH), PPTP/GRE, SCTP and every other protocol that is neither TCP nor UDP now leave') +
|
||||
` through egress "${egress}": the kernel routes them out that interface with that interface's own NAT, and none of it follows your routing rules. Multicast IPTV does not pass this router under any setting, and carrying IGMP out an egress cannot change that.`,
|
||||
},
|
||||
]
|
||||
}
|
||||
if (g.L3Tunnel) {
|
||||
return [
|
||||
{
|
||||
severity: 'info',
|
||||
section: 'untunnelable',
|
||||
name: '',
|
||||
message:
|
||||
'ping and Windows tracert work and travel THROUGH the tunnel, toward every address your rules send to an outbound that can carry plain IP (WireGuard/AmneziaWG); addresses your rules send anywhere else cannot be pinged at all, deliberately. Raw VPN passthrough (IPsec ESP/AH, PPTP/GRE) cannot enter the tunnel and stays with the untunnelable policy. Multicast IPTV does not pass this router on any setting; the L3 ingress does not change that.',
|
||||
},
|
||||
]
|
||||
}
|
||||
if (!killSwitchClosed(killSwitch)) {
|
||||
return [
|
||||
{
|
||||
severity: 'info',
|
||||
@@ -673,6 +735,34 @@ export async function getStatus(): Promise<Status> {
|
||||
await wait(120)
|
||||
const enabled = (CONFIG.Globals as { Enabled: boolean }).Enabled
|
||||
const { plane, engine, killSwitch } = mockPlane()
|
||||
// ?mock&cfg=unreadable — the daemon could not READ the configuration (full
|
||||
// /overlay, or a `uci commit` caught half-written). It is not a hypothetical: it
|
||||
// is the situation the fail-closed plane exists for, so it ships with the plane
|
||||
// HOLDING and the whole LAN cut off deliberately — while `enabled`, `kill_switch`
|
||||
// and `panel_port` are placeholders that mean nothing. Reproducing it here is how
|
||||
// the "Turned off" misreading stays fixed: the panel must alarm, not reassure.
|
||||
if (mockParam('cfg') === 'unreadable') {
|
||||
return {
|
||||
running: true,
|
||||
enabled: false, // a placeholder, NOT "the owner switched it off"
|
||||
active: false,
|
||||
table: true,
|
||||
hash,
|
||||
version: '1.11.0-shater',
|
||||
kill_switch: '', // placeholder likewise
|
||||
panel_port: 0, // placeholder likewise
|
||||
config_readable: false,
|
||||
config_error: 'uci show shater: exit status 1',
|
||||
can_rollback: armed || hasLastGood,
|
||||
engine_running: false,
|
||||
plane: 'hold',
|
||||
traffic: mockTraffic('hold'),
|
||||
warnings: [CONFIG_UNREADABLE_WARNING, ...mockWarnings(killSwitch)],
|
||||
started_unix: MOCK_STARTED_UNIX,
|
||||
uptime_seconds: Math.floor(Date.now() / 1000) - MOCK_STARTED_UNIX,
|
||||
byedpi_installed: true,
|
||||
}
|
||||
}
|
||||
return {
|
||||
running: true,
|
||||
enabled,
|
||||
@@ -682,6 +772,9 @@ export async function getStatus(): Promise<Status> {
|
||||
version: '1.11.0-shater',
|
||||
kill_switch: killSwitch,
|
||||
panel_port: 8088,
|
||||
// The daemon read the config fine in every other mock state. Sent explicitly
|
||||
// rather than left off: absent means "no reading", which is a different claim.
|
||||
config_readable: true,
|
||||
can_rollback: armed || hasLastGood,
|
||||
engine_running: engine,
|
||||
plane,
|
||||
|
||||
@@ -274,9 +274,11 @@ export default function Apply() {
|
||||
// The LIVE kill-switch wins over the saved one, exactly as on Overview: this row
|
||||
// is a status readout, and the config on disk can already differ from what is
|
||||
// installed. Falls back to the config only while /api/status is unread.
|
||||
const killArmed = (status?.kill_switch ?? globals?.KillSwitch ?? 'closed') === 'closed'
|
||||
// Whether that setting is actually installed — same three-plus-unknown reading
|
||||
// as Overview, so the two pages cannot disagree about the same router.
|
||||
// as Overview, so the two pages cannot disagree about the same router. There is
|
||||
// no separate `killArmed` here any more: it compared the raw string (so "Closed"
|
||||
// read as fail-OPEN) and, being a boolean, could not express "the configuration
|
||||
// could not be read". Both facts come off this one readout now.
|
||||
const kill = killSwitchReadout(status, globals?.KillSwitch)
|
||||
const killWord =
|
||||
kill.state === 'open'
|
||||
@@ -285,7 +287,12 @@ export default function Apply() {
|
||||
? 'fail-closed'
|
||||
: kill.state === 'inert'
|
||||
? 'closed · not in effect'
|
||||
: 'closed · not reported'
|
||||
: // The unknown branch splits: "closed · not reported" asserts the policy
|
||||
// and doubts only the install, which is wrong when the policy itself is
|
||||
// a placeholder from a configuration nothing could read.
|
||||
kill.setting === 'not known'
|
||||
? 'not known'
|
||||
: 'closed · not reported'
|
||||
// Every engine mark on this page comes from ONE reading, and that reading is
|
||||
// able to say "stopped" — see planeState.engineState for why `status.running`
|
||||
// could not. This page is where someone lands when the network is down; three
|
||||
@@ -296,8 +303,14 @@ export default function Apply() {
|
||||
// that is a leak (crit); under an open one it is the documented choice (amber).
|
||||
// It used to go amber whenever `running` was true — i.e. always — and unlit
|
||||
// otherwise, so the one state worth shouting about had no colour of its own.
|
||||
const dataVariant: LedVariant = status?.table ? 'on' : !status ? 'off' : killArmed ? 'crit' : 'amber'
|
||||
const configVariant: LedVariant = status?.enabled ? 'on' : 'amber'
|
||||
const dataVariant: LedVariant =
|
||||
status?.table ? 'on' : !status ? 'off' : kill.state !== 'open' ? 'crit' : 'amber'
|
||||
// `enabled` is sourced from the configuration, so it means nothing when that
|
||||
// could not be read (Status.config_readable): unlit, not amber, and the pip
|
||||
// beside it says so rather than printing "disabled".
|
||||
const configUnreadable = status?.config_readable === false
|
||||
const configVariant: LedVariant = configUnreadable ? 'off' : status?.enabled ? 'on' : 'amber'
|
||||
const configWord = configUnreadable ? 'unreadable' : status?.enabled ? 'enabled' : 'disabled'
|
||||
|
||||
const liveHash = short(status?.hash ?? '')
|
||||
const pct = armed ? Math.max(0, Math.round((armed.remaining / armed.pending.total) * 100)) : 0
|
||||
@@ -320,7 +333,7 @@ export default function Apply() {
|
||||
<StatusPip
|
||||
label="Config"
|
||||
variant={configVariant}
|
||||
value={status?.enabled ? 'enabled' : 'disabled'}
|
||||
value={configWord}
|
||||
/>
|
||||
<StatusPip
|
||||
label="Data plane"
|
||||
@@ -361,7 +374,7 @@ export default function Apply() {
|
||||
v: status?.table ? 'nft installed' : 'no table',
|
||||
hot: dataVariant === 'crit',
|
||||
},
|
||||
{ k: 'kill-switch', v: killWord, hot: kill.variant === 'crit' || !killArmed },
|
||||
{ k: 'kill-switch', v: killWord, hot: kill.variant === 'crit' || kill.settingHot },
|
||||
]}
|
||||
/>
|
||||
<Module
|
||||
@@ -375,7 +388,10 @@ export default function Apply() {
|
||||
led={{ variant: engineVariant }}
|
||||
rows={[
|
||||
{ k: 'state', v: engine.word, hot: engineVariant === 'crit' },
|
||||
{ k: 'config', v: status?.enabled ? 'enabled' : 'disabled' },
|
||||
// Same reading as the pip above — `status.enabled` is a placeholder
|
||||
// when the configuration could not be read, and "disabled" is the one
|
||||
// word that must not be printed for it.
|
||||
{ k: 'config', v: configWord, hot: configUnreadable },
|
||||
{ k: 'schema', v: globals ? `v${globals.SchemaVersion}` : '—' },
|
||||
]}
|
||||
/>
|
||||
|
||||
@@ -104,6 +104,16 @@
|
||||
font-size: 11.5px;
|
||||
color: var(--faint);
|
||||
}
|
||||
/* Nested inside .dns-filter-copy the note is an ordinary paragraph, but the
|
||||
endpoint-resolver footnote sits as a DIRECT child of the card — which makes it
|
||||
a grid item. Without a span it auto-placed into the toggle's `auto` column and
|
||||
sized that column to its own max-content (322px on desktop, 237px at 390px),
|
||||
which starved the `1fr` copy column down to 0px: the heading then laid out one
|
||||
word per line and spilled 2px past the viewport, scrolling the whole page
|
||||
sideways. It is a full-width footnote under the readout — say so. */
|
||||
.dns-filter-card > .dns-filter-note {
|
||||
grid-column: 1 / -1;
|
||||
}
|
||||
.dns-readout {
|
||||
display: flex;
|
||||
flex-direction: column;
|
||||
|
||||
+10
-3
@@ -9,7 +9,7 @@ import {
|
||||
updateRuleset as apiUpdateRuleset,
|
||||
ApiError,
|
||||
} from '../api'
|
||||
import type { DNSRule, Model, Resolver, RulesetStatus } from '../api'
|
||||
import type { Complete, DNSRule, Model, Resolver, RulesetStatus } from '../api'
|
||||
import { everyLabel, relFetch } from '../format'
|
||||
|
||||
// The DNS / Blocklists page is a thin editor over the desired-state Model —
|
||||
@@ -1944,8 +1944,15 @@ function ruleToDraft(r: DNSRule): DNSRuleDraft {
|
||||
}
|
||||
}
|
||||
|
||||
/** Build the DNSRule to persist. Empty matcher lists are omitted, not sent as []. */
|
||||
function draftToRule(d: DNSRuleDraft): DNSRule {
|
||||
/**
|
||||
* Build the DNSRule to persist. Empty matcher lists are omitted, not sent as [].
|
||||
*
|
||||
* The return type is `Complete<DNSRule>` for the same reason as Networks.fromDraft:
|
||||
* this REBUILDS the rule from the draft rather than extending the one it was given,
|
||||
* so a field added to `DNSRule` would otherwise be dropped on every edit with
|
||||
* nothing to notice it. Completing the type makes that a build failure here.
|
||||
*/
|
||||
function draftToRule(d: DNSRuleDraft): Complete<DNSRule> {
|
||||
const domains = parseRuleDomains(d.Domains)
|
||||
const order = Number.parseInt(d.Order, 10)
|
||||
return {
|
||||
|
||||
@@ -495,6 +495,22 @@
|
||||
border-left: 0;
|
||||
border-top: 1px solid var(--groove);
|
||||
}
|
||||
/* Collapsing to one column was not enough on a phone. A grid column is sized by
|
||||
its widest item's MIN-CONTENT, and a <select> reports the width of its longest
|
||||
option ("Allow everything — most compatible", in the mono face) — so the
|
||||
column stayed ~20px wider than the plate and the COPY beside it was clipped
|
||||
mid-word at the right edge, which is how a sentence about what leaks loses its
|
||||
second half. The select is allowed to shrink and ellipsise its own label
|
||||
instead; the chosen option is still fully readable once opened, and no text
|
||||
that states a consequence is cut. */
|
||||
.nw-policy-ctl {
|
||||
min-width: 0;
|
||||
}
|
||||
.nw-policy-ctl .fp-select {
|
||||
max-width: 100%;
|
||||
min-width: 0;
|
||||
text-overflow: ellipsis;
|
||||
}
|
||||
}
|
||||
|
||||
/* A daemon info note about the current policy — neutral by design: it states a
|
||||
|
||||
+296
-34
@@ -2,9 +2,10 @@ import './Networks.css'
|
||||
import { useCallback, useEffect, useMemo, useRef, useState } from 'react'
|
||||
import { Button, Led, Select, Toggle, useConfirm } from '../components'
|
||||
import { apply as apiApply, getConfig, putConfig, ApiError } from '../api'
|
||||
import type { Inbound, Interface, Model, Status } from '../api'
|
||||
import type { Complete, Inbound, Interface, Model, Status } from '../api'
|
||||
import { isLanNetwork, isWanNetwork, useInterfaces } from '../srcOptions'
|
||||
import { sectionNotes } from '../findings'
|
||||
import { killSwitchClosed } from '../planeState'
|
||||
|
||||
// The Networks page is the INGRESS editor — a thin editor over Model.Inbounds,
|
||||
// following the same save-then-Apply contract as Nodes/DNS/Routing: every edit
|
||||
@@ -98,6 +99,36 @@ const DEFAULT_TPROXY_PORT = 12345
|
||||
*
|
||||
* `icmp` still carries that destination-dependence for its NON-ping half, so its
|
||||
* cost line says so rather than claiming "nothing else gets out".
|
||||
*
|
||||
* WHY THIS COPY IS NOW A FUNCTION AND NOT A TABLE. Two audit findings, one cause:
|
||||
* a constant string cannot be true about a router whose behaviour three OTHER
|
||||
* settings can override.
|
||||
*
|
||||
* 1. MULTICAST IPTV WAS PROMISED, AND NEVER WORKS. `direct` said "Ping, multicast
|
||||
* IPTV, and connecting to a VPN ... all work" — an INSTRUCTION, and the worst
|
||||
* kind of wrong: someone who wants IPTV reads it, moves to the most open rung
|
||||
* on the ladder (which also permits a client's ESP/GRE straight past the
|
||||
* proxy), and still has no IPTV. The stream is UDP; every rule this policy
|
||||
* emits carries `meta l4proto != { tcp, udp }` so UDP never reaches one, and
|
||||
* the fail-closed forward chain accepts only the RFC1918/link-local daddr
|
||||
* sets — 224.0.0.0/4 is not there, and unconditional drops follow. The daemon
|
||||
* says exactly this in the note rendered a few pixels below on this same page.
|
||||
* IPTV is now stated ONCE, as its own line, and it says it does not work.
|
||||
* 2. THE `block` COST LINE WAS UNCONDITIONAL. Three settings contradict it:
|
||||
* - an OPEN kill-switch — the forward chain emits no drops at all;
|
||||
* - Globals.L3Tunnel — prerouting marks ICMP echo into the engine's TUN
|
||||
* BEFORE the forward chain, so ping keeps working, through the tunnel;
|
||||
* - Globals.UntunnelableEgress — ESP/AH/GRE/IGMP/SCTP are marked and routed
|
||||
* out a named interface, so the forward chain never rules on them.
|
||||
* Neither of the last two existed in the panel's `Globals` type, so the page
|
||||
* could not have told the truth about them even in principle; they were added
|
||||
* (api.ts) rather than papered over with a vaguer sentence.
|
||||
*
|
||||
* The copy therefore describes only what the POLICY still decides, and a separate
|
||||
* line names whatever another setting has taken off it. Detail beyond that belongs
|
||||
* to the daemon's own note for this section (`policyNotes`), which is computed
|
||||
* from the running plane and rendered right underneath — this copy's job is to not
|
||||
* contradict it.
|
||||
*/
|
||||
type Untunnelable = 'block' | 'icmp' | 'direct'
|
||||
|
||||
@@ -122,31 +153,197 @@ interface PolicyCopy {
|
||||
works: string
|
||||
cost: string | null
|
||||
tone: 'good' | 'warn'
|
||||
/** What some OTHER setting decides instead of this one. `null` ⇒ nothing; this policy owns it all. */
|
||||
claimed: string | null
|
||||
}
|
||||
|
||||
const UNTUNNELABLE_COPY: Record<Untunnelable, PolicyCopy> = {
|
||||
block: {
|
||||
// Scoped to "this traffic" on purpose. The old line — "Nothing leaves except
|
||||
// through the tunnel" — was doubly loose: it was false (see the note above),
|
||||
// and even read charitably it collides with directly-routed TCP, which does
|
||||
// leave outside the tunnel by design.
|
||||
works:
|
||||
'None of this traffic leaves the router — it’s dropped, whatever your routing rules say. It’s the only setting whose promise doesn’t depend on how the rules are written.',
|
||||
cost: 'Ping and traceroute stop working from your devices. So do IPsec and PPTP VPN connections made from a device on your network, multicast IPTV, and SCTP. VPNs that run over UDP — WireGuard, OpenVPN-UDP, and IPsec through NAT (IKEv2/NAT-T) — are unaffected: they go through the tunnel like everything else.',
|
||||
tone: 'good',
|
||||
},
|
||||
icmp: {
|
||||
works: 'Ping and traceroute work everywhere, so you can check whether something is reachable.',
|
||||
cost: 'Whatever you ping sees your real IP address instead of the tunnel’s. IPsec, PPTP and IPTV also get out — but only toward addresses your routing rules already send direct, so a VPN app on a device can still open its own connection beside this one if its server is one of those.',
|
||||
tone: 'warn',
|
||||
},
|
||||
direct: {
|
||||
works: 'Ping, multicast IPTV, and connecting to a VPN from a device on your network all work.',
|
||||
cost: 'All of it goes out with your real IP, around the tunnel. A VPN app left running on a device keeps its own connection open beside this one — traffic through it isn’t proxied or filtered.',
|
||||
tone: 'warn',
|
||||
},
|
||||
/**
|
||||
* How much of this traffic the policy still decides.
|
||||
*
|
||||
* Only three combinations are reachable, which is why this is an enum and not two
|
||||
* booleans: UntunnelableEgress claims EVERY untunnelable protocol (the kernel
|
||||
* routes them out its device before the forward chain runs), so once it is set
|
||||
* there is nothing left for L3Tunnel to change about the policy's scope.
|
||||
*
|
||||
* all — neither override is on. The policy decides everything.
|
||||
* exceptPing — L3Tunnel only. Ping rides the tunnel; ESP/AH/GRE/SCTP are the
|
||||
* policy's.
|
||||
* none — UntunnelableEgress is set. Routing settles all of it first; the
|
||||
* policy answers only for the case where that route fails to come up.
|
||||
*/
|
||||
type PolicyScope = 'all' | 'exceptPing' | 'none'
|
||||
|
||||
interface PolicyContext {
|
||||
/** Globals.L3Tunnel. */
|
||||
l3: boolean
|
||||
/** Globals.UntunnelableEgress, trimmed. */
|
||||
egress: string
|
||||
/** The LIVE kill-switch, normalised the daemon's way. Open ⇒ the chain has no drops. */
|
||||
killSwitchOpen: boolean
|
||||
}
|
||||
|
||||
function policyScope(ctx: PolicyContext): PolicyScope {
|
||||
if (ctx.egress) return 'none'
|
||||
return ctx.l3 ? 'exceptPing' : 'all'
|
||||
}
|
||||
|
||||
/** The sentence that stops an operator "fixing" a UDP VPN that was never broken. */
|
||||
const UDP_VPNS_FINE =
|
||||
'VPNs that run over UDP — WireGuard, OpenVPN-UDP, and IPsec through NAT (IKEv2/NAT-T) — are unaffected either way: they go through the tunnel like everything else.'
|
||||
|
||||
/** Which setting took this traffic off the policy, and what it does with it. */
|
||||
function claimedCopy(ctx: PolicyContext): string | null {
|
||||
// Each of these says only WHY the copy above has the shape it has — which other
|
||||
// setting took the traffic, and therefore why the familiar promise is missing.
|
||||
// What that setting then DOES with it is the daemon's note, published for this
|
||||
// same section and rendered immediately below from the RUNNING plane. Saying it
|
||||
// twice would make the shorter, staler one look like a second opinion.
|
||||
if (ctx.egress && ctx.l3) {
|
||||
return `Two other settings decide this before the one above is asked: ping goes through the tunnel (l3_tunnel), and everything else the tunnel can’t carry is routed out “${ctx.egress}” (untunnelable_egress).`
|
||||
}
|
||||
if (ctx.egress) {
|
||||
return `Another setting decides this before the one above is asked: everything the tunnel can’t carry is routed out “${ctx.egress}” (untunnelable_egress).`
|
||||
}
|
||||
if (ctx.l3) {
|
||||
// Deliberately shorter than the two above: when only the L3 ingress is on, the
|
||||
// daemon publishes its own note for this section directly underneath and says
|
||||
// the rest (which addresses can be pinged, and why the others cannot). This
|
||||
// line exists to explain the SHAPE of the copy above it — why ping is missing
|
||||
// from a policy that used to decide it — not to restate the daemon.
|
||||
return 'Ping and Windows tracert are taken through the tunnel before the setting above is asked (l3_tunnel), so it no longer decides them.'
|
||||
}
|
||||
return null
|
||||
}
|
||||
|
||||
/**
|
||||
* The copy for the policy as it is actually behaving right now.
|
||||
*
|
||||
* Read it as: an open kill-switch beats everything (no drops are emitted at all,
|
||||
* so no rung promises anything), then the scope decides how much of the ladder's
|
||||
* usual story is still this setting's to tell.
|
||||
*/
|
||||
function policyCopy(policy: Untunnelable, ctx: PolicyContext): PolicyCopy {
|
||||
const scope = policyScope(ctx)
|
||||
const claimed = claimedCopy(ctx)
|
||||
|
||||
// Fail-open: the forward chain emits no drops, so every rung is inert. Saying
|
||||
// what IS happening beats repeating a promise nothing is keeping.
|
||||
if (ctx.killSwitchOpen) {
|
||||
if (scope === 'none') {
|
||||
return {
|
||||
works:
|
||||
'Nothing is being dropped, and nothing is left for this setting to decide: the kill-switch is open, and another setting has already taken this traffic.',
|
||||
cost: 'If that route ever fails to come up, the traffic leaves through your normal connection with your real IP address, quietly, instead of failing.',
|
||||
tone: 'warn',
|
||||
claimed,
|
||||
}
|
||||
}
|
||||
return {
|
||||
works:
|
||||
scope === 'exceptPing'
|
||||
? 'Nothing is being dropped: with the kill-switch open the forward chain has no drops at all, so a device’s own IPsec or PPTP connection works too.'
|
||||
: 'Nothing is being dropped: with the kill-switch open the forward chain has no drops at all, so ping, traceroute and a device’s own IPsec or PPTP connection all work.',
|
||||
// No "set the kill-switch to fail-closed" here: the moot note below owns
|
||||
// that instruction, and printing it twice in one section is how the second
|
||||
// copy stops being read.
|
||||
cost: `It reaches the internet with your real IP address, around the tunnel. ${UDP_VPNS_FINE}`,
|
||||
tone: 'warn',
|
||||
claimed,
|
||||
}
|
||||
}
|
||||
|
||||
switch (policy) {
|
||||
case 'block':
|
||||
if (scope === 'none') {
|
||||
return {
|
||||
works:
|
||||
'Where this setting still applies, the packet is dropped rather than let out — so a route that fails to come up fails honestly instead of leaking.',
|
||||
cost: null,
|
||||
tone: 'good',
|
||||
claimed,
|
||||
}
|
||||
}
|
||||
return {
|
||||
// Scoped to "this traffic" on purpose. The old line — "Nothing leaves
|
||||
// except through the tunnel" — was doubly loose: it was false (see the
|
||||
// note above), and even read charitably it collides with directly-routed
|
||||
// TCP, which does leave outside the tunnel by design.
|
||||
works:
|
||||
scope === 'exceptPing'
|
||||
? 'Everything this setting still decides is dropped, whatever your routing rules say. If the ping route ever fails to come up, ping fails outright rather than leaking.'
|
||||
: 'None of this traffic leaves the router — it’s dropped, whatever your routing rules say. It’s the only setting whose promise doesn’t depend on how the rules are written.',
|
||||
cost:
|
||||
scope === 'exceptPing'
|
||||
? `IPsec and PPTP VPN connections made from a device on your network stop working, and so does SCTP. ${UDP_VPNS_FINE}`
|
||||
: `Ping and traceroute stop working from your devices. So do IPsec and PPTP VPN connections made from a device on your network, and SCTP. ${UDP_VPNS_FINE}`,
|
||||
tone: 'good',
|
||||
claimed,
|
||||
}
|
||||
|
||||
case 'icmp':
|
||||
if (scope === 'none') {
|
||||
return {
|
||||
works:
|
||||
'Where this setting still applies, it lets ping out directly and drops the rest.',
|
||||
cost: 'So if a route ever fails to come up, ping quietly leaves with your real IP address instead of failing.',
|
||||
tone: 'warn',
|
||||
claimed,
|
||||
}
|
||||
}
|
||||
if (scope === 'exceptPing') {
|
||||
return {
|
||||
works:
|
||||
'Ping already travels through the tunnel, so this rung’s exception for it only matters if that route fails to come up.',
|
||||
cost: 'IPsec, PPTP and SCTP get out toward addresses your routing rules already send direct, with your real IP address — so a VPN app on a device can still open its own connection beside this one if its server is one of those. And if the ping route fails, ping leaves with your real address rather than failing.',
|
||||
tone: 'warn',
|
||||
claimed,
|
||||
}
|
||||
}
|
||||
return {
|
||||
works:
|
||||
'Ping and traceroute work everywhere, so you can check whether something is reachable.',
|
||||
cost: 'Whatever you ping sees your real IP address instead of the tunnel’s. IPsec and PPTP also get out — but only toward addresses your routing rules already send direct, so a VPN app on a device can still open its own connection beside this one if its server is one of those.',
|
||||
tone: 'warn',
|
||||
claimed,
|
||||
}
|
||||
|
||||
case 'direct':
|
||||
if (scope === 'none') {
|
||||
return {
|
||||
works:
|
||||
'Nothing is left for this setting to decide: another setting has already taken this traffic.',
|
||||
cost: 'If that route ever fails to come up, this setting lets the traffic leave through your normal connection with your real IP address, quietly, instead of failing.',
|
||||
tone: 'warn',
|
||||
claimed,
|
||||
}
|
||||
}
|
||||
return {
|
||||
works:
|
||||
scope === 'exceptPing'
|
||||
? 'A device on your network can make its own IPsec or PPTP VPN connection. Ping already travels through the tunnel.'
|
||||
: 'Ping and traceroute work, and a device on your network can make its own IPsec or PPTP VPN connection.',
|
||||
cost:
|
||||
scope === 'exceptPing'
|
||||
? 'That traffic goes out with your real IP, around the tunnel. A VPN app left running on a device keeps its own connection open beside this one — traffic through it isn’t proxied or filtered. If the ping route ever fails to come up, ping does the same instead of failing.'
|
||||
: 'All of it goes out with your real IP, around the tunnel. A VPN app left running on a device keeps its own connection open beside this one — traffic through it isn’t proxied or filtered.',
|
||||
tone: 'warn',
|
||||
claimed,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Multicast IPTV, said once and said straight.
|
||||
*
|
||||
* It is stated unconditionally because it is unconditionally true — the stream is
|
||||
* UDP and no rule this policy emits can match UDP, on any of the three rungs — and
|
||||
* it is stated at all because the page used to promise the opposite under `direct`
|
||||
* and under `icmp`. Someone whose IPTV is broken arrives here looking for the
|
||||
* setting that fixes it; the useful thing to tell them is that there isn't one.
|
||||
*/
|
||||
const IPTV_LINE =
|
||||
'Multicast IPTV is not one of these things: it doesn’t pass this router on any of the three settings, and “Allow everything” won’t bring it back.'
|
||||
|
||||
/** The addr:port an inbound binds — the generator's clash key (listenKey). */
|
||||
function listenKey(in_: Inbound): string {
|
||||
if (effectiveType(in_) === 'tproxy') {
|
||||
@@ -348,11 +545,24 @@ export default function Networks({ status }: { status?: Status | null }) {
|
||||
// Untunnelable-traffic policy. Normalised the same way the daemon does, so an
|
||||
// absent/unknown UCI value reads as `block` here too rather than as blank.
|
||||
const untunnelable = normUntunnelable(config?.Globals?.Untunnelable)
|
||||
const untunnelableCopy = UNTUNNELABLE_COPY[untunnelable]
|
||||
// Prefer the LIVE kill-switch off /api/status; fall back to the saved config
|
||||
// when the shell hasn't got a status yet.
|
||||
const killSwitchOpen =
|
||||
(status?.kill_switch ?? config?.Globals?.KillSwitch ?? 'closed').toLowerCase() === 'open'
|
||||
// when the shell hasn't got a status yet. The comparison is the daemon's own
|
||||
// (planeState.killSwitchClosed) — this page normalised and planeState.ts did
|
||||
// not, so the same router read differently on two pages.
|
||||
const killSwitchOpen = !killSwitchClosed(status?.kill_switch ?? config?.Globals?.KillSwitch)
|
||||
// The two settings that decide part of this traffic BEFORE the policy is
|
||||
// consulted. Both are read from the saved config, like `untunnelable` itself:
|
||||
// /api/status reports neither, and this section describes the setting the
|
||||
// operator is editing. The daemon's own note below is the live counterpart.
|
||||
const untunnelableCopy = useMemo(
|
||||
() =>
|
||||
policyCopy(untunnelable, {
|
||||
l3: config?.Globals?.L3Tunnel === true,
|
||||
egress: (config?.Globals?.UntunnelableEgress ?? '').trim(),
|
||||
killSwitchOpen,
|
||||
}),
|
||||
[untunnelable, config?.Globals?.L3Tunnel, config?.Globals?.UntunnelableEgress, killSwitchOpen],
|
||||
)
|
||||
// The daemon's info notes about this policy — shown beside the control they
|
||||
// describe. The fail-open case has its own dedicated line below, so drop that
|
||||
// one here to avoid saying the same thing twice.
|
||||
@@ -519,8 +729,8 @@ export default function Networks({ status }: { status?: Status | null }) {
|
||||
</header>
|
||||
<p className="nw-sec-note">
|
||||
The tunnel carries the traffic almost everything uses — web, video, games, email. A few
|
||||
things can’t go through it no matter what: ping, and the protocols that carry IPTV or a VPN
|
||||
connection. Choose what happens to those.
|
||||
things can’t go through it no matter what: ping, and the protocols a device uses to make
|
||||
its own VPN connection. Choose what happens to those.
|
||||
</p>
|
||||
|
||||
<div className="nw-policy">
|
||||
@@ -549,13 +759,28 @@ export default function Networks({ status }: { status?: Status | null }) {
|
||||
</div>
|
||||
</div>
|
||||
|
||||
{/* Fail-open makes the whole policy moot — say so instead of letting the
|
||||
page imply something is being blocked when nothing is. */}
|
||||
{/* What another setting decides instead of this one. Unlit lamp, like the
|
||||
daemon's notes below: it reports a configuration, not a fault. */}
|
||||
{untunnelableCopy.claimed && (
|
||||
<p className="nw-sec-note nw-policy-note" role="status">
|
||||
<Led variant="off" />
|
||||
<span>{untunnelableCopy.claimed}</span>
|
||||
</p>
|
||||
)}
|
||||
|
||||
{/* Said once, on every setting, because it is true on every setting. */}
|
||||
<p className="nw-sec-note nw-policy-note">
|
||||
<Led variant="off" />
|
||||
<span>{IPTV_LINE}</span>
|
||||
</p>
|
||||
|
||||
{/* Fail-open makes the whole policy moot. The copy above now says what IS
|
||||
happening; this line stays because it is the one that says what to DO. */}
|
||||
{killSwitchOpen && (
|
||||
<p className="nw-sec-note nw-policy-moot" role="status">
|
||||
<Led variant="amber" /> This setting isn’t doing anything right now: the kill-switch is
|
||||
set to fail-open, so traffic keeps flowing directly whenever the tunnel is down. Set it
|
||||
to fail-closed in Settings for this choice to take effect.
|
||||
<Led variant="amber" /> The kill-switch is set to fail-open, so traffic keeps flowing
|
||||
directly whenever the tunnel is down. Set it to fail-closed in Settings for this choice
|
||||
to take effect.
|
||||
</p>
|
||||
)}
|
||||
|
||||
@@ -758,7 +983,22 @@ function toDraft(in_: Inbound): Draft {
|
||||
* and a dokodemo listener binds whatever `TargetNetwork` says. Writing anything
|
||||
* else would make the daemon warn about a flag no one can see.
|
||||
*/
|
||||
function fromDraft(d: Draft, base: Partial<Inbound>, enabled: boolean): Inbound {
|
||||
/**
|
||||
* Build the Inbound this draft describes — REBUILT per type, never extended.
|
||||
*
|
||||
* Rebuilding is the point: an inbound switched from `socks` to `tproxy` binds
|
||||
* 0.0.0.0:TproxyPort, so a surviving `Listen`/`Auth` from its previous life would
|
||||
* be a setting the panel shows nobody and the generator ignores. `base` is
|
||||
* accepted and deliberately unused for that reason.
|
||||
*
|
||||
* The return type is `Complete<Inbound>` so the rebuild cannot go stale in
|
||||
* silence: adding a field to `Inbound` fails the BUILD in all three branches
|
||||
* below until each says what it wants done with it. That is the only mechanism
|
||||
* here that makes the omission loud — `Ruleset.Format` is what a missing one
|
||||
* costs (see ruleset.ts). `undefined` is a positive statement of "this shape has
|
||||
* no use for it", and JSON.stringify drops it, so the wire form is unchanged.
|
||||
*/
|
||||
function fromDraft(d: Draft, base: Partial<Inbound>, enabled: boolean): Complete<Inbound> {
|
||||
void base
|
||||
const common = { Name: d.Name.trim(), Enabled: enabled }
|
||||
const port = Number.parseInt(d.Port, 10)
|
||||
@@ -771,6 +1011,16 @@ function fromDraft(d: Draft, base: Partial<Inbound>, enabled: boolean): Inbound
|
||||
TproxyPort: Number.parseInt(d.TproxyPort, 10) || DEFAULT_TPROXY_PORT,
|
||||
TCP: d.TCP,
|
||||
UDP: d.UDP,
|
||||
// Binds 0.0.0.0:TproxyPort and cannot authenticate or rewrite a
|
||||
// destination, so none of the listener fields mean anything here.
|
||||
Listen: undefined,
|
||||
Port: undefined,
|
||||
Auth: undefined,
|
||||
User: undefined,
|
||||
Pass: undefined,
|
||||
TargetAddr: undefined,
|
||||
TargetPort: undefined,
|
||||
TargetNetwork: undefined,
|
||||
}
|
||||
case 'socks':
|
||||
case 'http':
|
||||
@@ -784,6 +1034,12 @@ function fromDraft(d: Draft, base: Partial<Inbound>, enabled: boolean): Inbound
|
||||
Pass: d.Auth === 'password' ? d.Pass : '',
|
||||
TCP: true,
|
||||
UDP: true,
|
||||
// A local listener diverts no network and has no fixed target.
|
||||
Network: undefined,
|
||||
TproxyPort: undefined,
|
||||
TargetAddr: undefined,
|
||||
TargetPort: undefined,
|
||||
TargetNetwork: undefined,
|
||||
}
|
||||
case 'dokodemo':
|
||||
return {
|
||||
@@ -796,6 +1052,12 @@ function fromDraft(d: Draft, base: Partial<Inbound>, enabled: boolean): Inbound
|
||||
TargetNetwork: d.TargetNetwork,
|
||||
TCP: true,
|
||||
UDP: true,
|
||||
// Diverts no network, and forwards everything on without authenticating.
|
||||
Network: undefined,
|
||||
TproxyPort: undefined,
|
||||
Auth: undefined,
|
||||
User: undefined,
|
||||
Pass: undefined,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -271,9 +271,10 @@ function findNodeReferences(m: Model, name: string): NodeRefSite[] {
|
||||
if (isNodeRef(s.FetchDetour, name))
|
||||
out.push({ kind: 'subscription', label: `subscription “${s.Name}” fetch` })
|
||||
}
|
||||
for (const e of asArray(m.Egresses)) {
|
||||
if (isNodeRef(e.Target, name)) out.push({ kind: 'egress', label: `egress “${e.Name}” target` })
|
||||
}
|
||||
// An egress does not reference a node. The loop that used to sit here read
|
||||
// `e.Target`, a field the Go model has never had — so it was `undefined` on
|
||||
// every egress and the branch could not fire. It read as coverage for a
|
||||
// reference site that does not exist, which is worse than the gap it hid.
|
||||
return out
|
||||
}
|
||||
|
||||
@@ -336,7 +337,9 @@ function renameNodeReferences(m: Model, from: string, to: string): Model {
|
||||
if (m.Alerts) next.Alerts = m.Alerts.map((a) => ({ ...a, Via: pfx(a.Via) }))
|
||||
if (m.Subscriptions)
|
||||
next.Subscriptions = m.Subscriptions.map((s) => ({ ...s, FetchDetour: pfx(s.FetchDetour) }))
|
||||
if (m.Egresses) next.Egresses = m.Egresses.map((e) => ({ ...e, Target: pfx(e.Target) }))
|
||||
// No Egresses pass: an egress holds no node reference to rewrite (it has no
|
||||
// Target field), and rewriting one would have WRITTEN the key onto every egress
|
||||
// — which PUT rejects wholesale under DisallowUnknownFields.
|
||||
return next
|
||||
}
|
||||
|
||||
|
||||
@@ -226,11 +226,11 @@ export function Overview({
|
||||
|
||||
// ---- derived display state ----
|
||||
const g = config?.Globals
|
||||
// The LIVE kill-switch from /api/status wins over the saved config: this is a
|
||||
// status readout, so it must describe what is actually installed. Reading the
|
||||
// config here let the strip claim "fail-closed" while an apply-time finding
|
||||
// said the running plane was fail-open — two truths on one screen.
|
||||
const killArmed = (status?.kill_switch ?? g?.KillSwitch ?? 'closed') === 'closed'
|
||||
// There is deliberately no local `killArmed` any more. The page asked the same
|
||||
// question twice — once here and once inside killSwitchReadout — and the local
|
||||
// copy was the poorer of the two: it compared the raw string (so "Closed" read
|
||||
// as fail-OPEN) and it was a boolean, which cannot say "the configuration could
|
||||
// not be read". Both answers now come from the readout below.
|
||||
// Offer rollback only when the daemon has something to revert to (armed
|
||||
// commit-confirm snapshot or an engine last-good); otherwise hide the control.
|
||||
const canRollback = status?.can_rollback ?? false
|
||||
@@ -459,7 +459,11 @@ export function Overview({
|
||||
value={kill.value}
|
||||
led={{ variant: kill.variant }}
|
||||
rows={[
|
||||
{ k: 'setting', v: killArmed ? 'fail-closed' : 'fail-open', hot: !killArmed },
|
||||
// From the readout, not from a second local comparison: the old
|
||||
// `killArmed ? 'fail-closed' : 'fail-open'` had no third answer, so an
|
||||
// unreadable configuration printed a confident "fail-closed" beneath a
|
||||
// lamp that said NOT REPORTED. See planeState.killSwitchReadout.
|
||||
{ k: 'setting', v: kill.setting, hot: kill.settingHot },
|
||||
...(kill.blockingNow
|
||||
? [{ k: 'blocking now', v: kill.blockingNow, hot: kill.hot }]
|
||||
: [{ k: 'ipv6', v: g?.IPv6 ? 'covered' : 'off' }]),
|
||||
|
||||
+67
-15
@@ -13,6 +13,8 @@ import {
|
||||
} from '../api'
|
||||
import type { Model, Rule, RuleReach, Ruleset, RulesetStatus } from '../api'
|
||||
import { everyLabel, relFetch } from '../format'
|
||||
import { killSwitchClosed } from '../planeState'
|
||||
import { carryRulesetFormat } from '../ruleset'
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// The api.ts `Rule` is a deliberately thin subset (Name/Enabled/Order/Target/
|
||||
@@ -60,16 +62,31 @@ type RuleForce = {
|
||||
}
|
||||
|
||||
/**
|
||||
* Everything `Proto` can match, and nothing else. The engine understands two
|
||||
* transports and exactly ten application protocols its sniffers can name
|
||||
* (generate/route.go sniffedProtocols); a value outside this set builds a rule
|
||||
* that is perfectly valid and can never fire — so its traffic quietly falls
|
||||
* Everything `Proto` can match, and nothing else. A value outside this set builds
|
||||
* a rule that is perfectly valid and can never fire — so its traffic quietly falls
|
||||
* through to whatever rule sits below it. That is why this is a closed list and
|
||||
* not a text box.
|
||||
*
|
||||
* Split into two groups because they answer different questions: the transport is
|
||||
* known the moment a packet arrives, while an app protocol is only known once the
|
||||
* first bytes have been read and labelled.
|
||||
* Three groups, because the engine reads them through three different matchers
|
||||
* (generate/route.go, ruleMatchers) and they answer different questions:
|
||||
*
|
||||
* - Transport — the L4 network. Known the moment a packet arrives.
|
||||
* - Detected protocol — the L7 label a sniffer puts on a connection once its
|
||||
* first bytes have been read. This group, and ONLY this group, is the engine's
|
||||
* `sniffedProtocols` set; anything else routed into that matcher is inert.
|
||||
* - Layer 3 — ICMP. Not a sniffed label: it lands in the emitted rule's
|
||||
* `network`, never in `protocol` (the sniffers are skipped outright for an
|
||||
* ICMP flow, so they never report "icmp"). All three spellings are the SAME
|
||||
* one network; `icmpv4`/`icmpv6` additionally pin `ip_version`, which the
|
||||
* engine derives from the destination address.
|
||||
*
|
||||
* ICMP carries caveats the picker deliberately does not try to enforce, because
|
||||
* the daemon reports each one against the whole config on apply: it reaches the
|
||||
* engine only while globals l3_tunnel is on, it has no ports (a port matcher
|
||||
* beside it can never be satisfied), `icmpv6` also needs globals ipv6 on, and it
|
||||
* is DROPPED rather than falling through when routed at a target that cannot
|
||||
* carry layer 3 — i.e. every proxy protocol. Only wireguard/AmneziaWG nodes and
|
||||
* direct/interface egresses can carry a ping.
|
||||
*/
|
||||
const PROTO_TRANSPORT: ReadonlyArray<{ id: string; label: string }> = [
|
||||
{ id: 'tcp', label: 'TCP' },
|
||||
@@ -87,10 +104,27 @@ const PROTO_APP: ReadonlyArray<{ id: string; label: string }> = [
|
||||
{ id: 'rdp', label: 'RDP' },
|
||||
{ id: 'ntp', label: 'NTP' },
|
||||
]
|
||||
const PROTO_VALUES = new Set([...PROTO_TRANSPORT, ...PROTO_APP].map((p) => p.id))
|
||||
/**
|
||||
* The family-qualified spellings are offered next to plain `icmp` rather than
|
||||
* hidden behind it: the engine treats them as first-class and the difference is
|
||||
* observable (an `ip_version` item on the same rule), so hiding them would leave a
|
||||
* capability reachable only by hand-editing /etc/config/shater — and would mean
|
||||
* that anyone who edited such a rule here lost the narrowing on the next save.
|
||||
*/
|
||||
const PROTO_L3: ReadonlyArray<{ id: string; label: string }> = [
|
||||
{ id: 'icmp', label: 'ICMP (ping)' },
|
||||
{ id: 'icmpv4', label: 'ICMP — IPv4 only' },
|
||||
{ id: 'icmpv6', label: 'ICMP — IPv6 only' },
|
||||
]
|
||||
const PROTO_VALUES = new Set(
|
||||
[...PROTO_TRANSPORT, ...PROTO_APP, ...PROTO_L3].map((p) => p.id),
|
||||
)
|
||||
|
||||
/** The Proto picker's option list — shared by the inline add row and the editor. */
|
||||
function ProtoOptions({ value }: { value: string }) {
|
||||
// The engine lower-cases `Proto` before matching it, so a hand-written `ICMP`
|
||||
// is a working rule; judge it the same way and flag only what really is inert.
|
||||
const matches = PROTO_VALUES.has(value.trim().toLowerCase())
|
||||
return (
|
||||
<>
|
||||
<option value="">any</option>
|
||||
@@ -108,10 +142,18 @@ function ProtoOptions({ value }: { value: string }) {
|
||||
</option>
|
||||
))}
|
||||
</optgroup>
|
||||
{/* A stored value the engine can't detect is kept and flagged, never
|
||||
silently rewritten — the rule it belongs to is live right now. */}
|
||||
<optgroup label="Layer 3">
|
||||
{PROTO_L3.map((p) => (
|
||||
<option key={p.id} value={p.id}>
|
||||
{p.label}
|
||||
</option>
|
||||
))}
|
||||
</optgroup>
|
||||
{/* A stored value none of the groups spells verbatim is kept and offered as
|
||||
written, never silently rewritten — the rule it belongs to is live right
|
||||
now. It is flagged only when the engine cannot match it either. */}
|
||||
{value !== '' && !PROTO_VALUES.has(value) && (
|
||||
<option value={value}>{value} — never matches</option>
|
||||
<option value={value}>{matches ? value : `${value} — never matches`}</option>
|
||||
)}
|
||||
</>
|
||||
)
|
||||
@@ -732,7 +774,10 @@ export default function Routing() {
|
||||
[rules],
|
||||
)
|
||||
|
||||
const killSwitch = (config?.Globals?.KillSwitch ?? 'closed') === 'open' ? 'open' : 'closed'
|
||||
// Normalised by the daemon's own rule — `=== 'open'` read "OPEN" as fail-CLOSED,
|
||||
// so this dialog would have described a blocking router that isn't blocking.
|
||||
// See planeState.killSwitchClosed.
|
||||
const killSwitch = killSwitchClosed(config?.Globals?.KillSwitch) ? 'closed' : 'open'
|
||||
|
||||
const onToggle = useCallback(
|
||||
async (name: string) => {
|
||||
@@ -2030,9 +2075,16 @@ function RulesetForm({
|
||||
return
|
||||
}
|
||||
// Type is meaningless for geosite/geoip (the remote .srs is self-describing).
|
||||
const rs: Ruleset = isGeoSource(source)
|
||||
? { Name: nm, Source: source }
|
||||
: { Name: nm, Type: type, Source: source }
|
||||
//
|
||||
// `Format` is CARRIED, not rebuilt: this form has no control for it, so every
|
||||
// save that dropped it destroyed a value only SSH could put back — and
|
||||
// renaming a list came through here. See ruleset.ts for what it costs.
|
||||
const rs: Ruleset = carryRulesetFormat(
|
||||
isGeoSource(source)
|
||||
? { Name: nm, Source: source }
|
||||
: { Name: nm, Type: type, Source: source },
|
||||
initial,
|
||||
)
|
||||
if (source === 'inline') {
|
||||
const entries = parseLines(text)
|
||||
if (entries.length === 0) {
|
||||
|
||||
@@ -5,6 +5,7 @@ import { Button, Led, Select, Toggle, useConfirm } from '../components'
|
||||
import { AlertsSection } from './Alerts'
|
||||
import { apply as apiApply, downloadLog, getConfig, putConfig, ApiError } from '../api'
|
||||
import type { Globals, LogRange, Model } from '../api'
|
||||
import { killSwitchClosed } from '../planeState'
|
||||
|
||||
// The Settings page is a thin editor over the desired-state Model's Globals —
|
||||
// same save→apply split as DNS.tsx: every edit rewrites model.Globals in-place,
|
||||
@@ -117,10 +118,18 @@ const LOG_LEVELS: ReadonlyArray<{ value: string; label: string }> = [
|
||||
// Logging/stats backend. "off" collects nothing; "memory" keeps aggregates in RAM
|
||||
// (lost on restart); "sqlite" persists logs to /etc/shater/stats.db so they survive
|
||||
// a restart, bounded by the retention row caps + the disk-limit knob below.
|
||||
//
|
||||
// THE VALUE `sqlite` IS A HISTORICAL NAME AND THE LABELS NO LONGER REPEAT IT. There
|
||||
// is no SQLite in the daemon: the store is bbolt (stats/boltring.go) — pure Go, no
|
||||
// CGO, already linked into the binary via experimental/cachefile — and it was chosen
|
||||
// precisely to be rid of "the stop-the-world window the sqlite VACUUM used to
|
||||
// impose", in that file's own words. The wire value has to stay (it is in every
|
||||
// shipped config, and the daemon still matches on it); what the operator READS
|
||||
// should describe where the logs go, which is the disk.
|
||||
const STATS_BACKENDS: ReadonlyArray<{ value: string; label: string }> = [
|
||||
{ value: 'off', label: 'Off — no logging' },
|
||||
{ value: 'memory', label: 'Memory (RAM)' },
|
||||
{ value: 'sqlite', label: 'SQLite · persistent' },
|
||||
{ value: 'sqlite', label: 'Disk · survives a restart' },
|
||||
]
|
||||
|
||||
// ---- page ------------------------------------------------------------------
|
||||
@@ -219,14 +228,16 @@ export default function Settings() {
|
||||
const ringUnlimited = (globals?.StatsRingSize ?? 0) === 0
|
||||
const timelineUnlimited = (globals?.StatsTimelineMinutes ?? 0) === 0
|
||||
const domainsUnlimited = (globals?.StatsMaxDomains ?? 0) === 0
|
||||
// SQLite disk cap: 0 = unlimited (stats.db grows with the disk).
|
||||
// Disk cap: 0 = unlimited (stats.db grows with the disk).
|
||||
const diskUnlimited = (globals?.StatsDiskLimitMB ?? 0) === 0
|
||||
|
||||
// Logging backend. Default "memory" when the field is absent (older config). When
|
||||
// "off", nothing is collected, so the retention sizes below don't apply — dim them.
|
||||
const statsBackend = globals?.StatsBackend || 'memory'
|
||||
const loggingOff = statsBackend === 'off'
|
||||
const loggingSqlite = statsBackend === 'sqlite'
|
||||
// The wire value is still `sqlite` (historical — see STATS_BACKENDS); the store
|
||||
// is bbolt on disk, so everything the operator reads calls it the disk backend.
|
||||
const loggingDisk = statsBackend === 'sqlite'
|
||||
// Retention controls are meaningless with logging off; disable them there.
|
||||
const retentionDisabledCtl = busy || !ready || loggingOff
|
||||
|
||||
@@ -258,7 +269,11 @@ export default function Settings() {
|
||||
// the Targets page. Absent ⇒ enabled (older config), so read it as `!== false`.
|
||||
const groupHealthOn = globals?.GroupHealth !== false
|
||||
|
||||
const killSwitch = globals?.KillSwitch === 'open' ? 'open' : 'closed'
|
||||
// Normalised the daemon's way (planeState.killSwitchClosed), not by string
|
||||
// equality: `kill_switch 'OPEN'` is fail-OPEN on the router, and `=== 'open'`
|
||||
// read it as closed — the panel would have drawn the protective setting over a
|
||||
// router that has none.
|
||||
const killSwitch = killSwitchClosed(globals?.KillSwitch) ? 'closed' : 'open'
|
||||
|
||||
/**
|
||||
* The master switch, which is the most destructive control in the panel and was
|
||||
@@ -301,10 +316,16 @@ export default function Settings() {
|
||||
},
|
||||
[confirm, killSwitch, setGlobal],
|
||||
)
|
||||
// "so nothing leaks unproxied" claimed more than the holding plane promises.
|
||||
// netplane/nft.go states its own contract as "No client TRAFFIC reaches the WAN"
|
||||
// and names the exception in the same paragraph: clients still reach the router's
|
||||
// resolver and dnsmasq forwards those lookups to the ISP in the clear. It cannot
|
||||
// be closed — blocking it would also cut the daemon's own name resolution, and
|
||||
// with it any chance of recovering unattended.
|
||||
const killNote =
|
||||
killSwitch === 'open'
|
||||
? 'Fail-open — if the engine stops, traffic falls back to the direct WAN. Stays online, but unprotected.'
|
||||
: 'Fail-closed — if the engine stops, LAN→WAN is blocked so nothing leaks unproxied.'
|
||||
: 'Fail-closed — if the engine stops, LAN→WAN is blocked so no traffic from your devices reaches the internet. DNS is the exception: lookups sent to the router still go out to your provider in the clear, which is what lets the router recover on its own.'
|
||||
|
||||
const loading = config === null && loadError === null
|
||||
|
||||
@@ -375,7 +396,16 @@ export default function Settings() {
|
||||
|
||||
<Field
|
||||
label="Panel port"
|
||||
note="Admin-panel port. 0 uses the default 8088. A change needs a restart to rebind."
|
||||
// "needs a restart to rebind" read as a promise that the rebind
|
||||
// happens. It is not one the panel can make: cmd/shaterd/main.go
|
||||
// treats a failed panel listen as `logger.Warn("panel server
|
||||
// unavailable (daemon continues)")` and carries on — the daemon keeps
|
||||
// routing traffic and the panel simply is not there. Nothing reports
|
||||
// it in the UI either, because the UI is what went missing, and
|
||||
// `status.panel_port` keeps naming the CONFIGURED port regardless
|
||||
// (which is also what LuCI builds its "Open panel" button from). So
|
||||
// the note names the failure and where the answer actually is.
|
||||
note="Admin-panel port. 0 uses the default 8088. A change takes effect on restart — and if the new port is already taken the panel does not come back at all: the daemon keeps running and only says so in its log."
|
||||
>
|
||||
<InlineEdit<number>
|
||||
value={globals?.PanelPort ?? 0}
|
||||
@@ -635,7 +665,7 @@ export default function Settings() {
|
||||
>
|
||||
<Field
|
||||
label="Logging backend"
|
||||
note="Off: collect nothing. Memory: fast, lost on restart, RAM-bounded. SQLite: survives restart, disk-bounded."
|
||||
note="Off: collect nothing. Memory: fast, lost on restart, RAM-bounded. Disk: survives a restart, disk-bounded."
|
||||
>
|
||||
<Select
|
||||
value={statsBackend}
|
||||
@@ -650,7 +680,7 @@ export default function Settings() {
|
||||
v === 'off'
|
||||
? 'Logging off — collecting nothing'
|
||||
: v === 'sqlite'
|
||||
? 'Logging backend → SQLite (persistent)'
|
||||
? 'Logging backend → disk (survives a restart)'
|
||||
: 'Logging backend → memory',
|
||||
)
|
||||
}
|
||||
@@ -663,13 +693,13 @@ export default function Settings() {
|
||||
Insights page shows an off state. The retention limits below apply once logging
|
||||
is turned back on.
|
||||
</p>
|
||||
) : loggingSqlite ? (
|
||||
) : loggingDisk ? (
|
||||
<p className="set-group-note">
|
||||
Logs persist to <span className="mono">/etc/shater/stats.db</span> and survive a
|
||||
restart. The <strong>entries</strong> limit below caps rows kept per log table; the{' '}
|
||||
<strong>disk limit</strong> caps the whole <span className="mono">stats.db</span>{' '}
|
||||
file (oldest rows are pruned to stay under it). Set any size to <strong>0</strong>{' '}
|
||||
for <strong>Unlimited</strong>.
|
||||
<strong>disk limit</strong> aims the whole <span className="mono">stats.db</span>{' '}
|
||||
file at a size (oldest rows are deleted and the file rebuilt to stay near it). Set
|
||||
any size to <strong>0</strong> for <strong>Unlimited</strong>.
|
||||
</p>
|
||||
) : (
|
||||
<p className="set-group-note">
|
||||
@@ -706,10 +736,16 @@ export default function Settings() {
|
||||
</p>
|
||||
)}
|
||||
|
||||
{loggingSqlite && (
|
||||
{loggingDisk && (
|
||||
<Field
|
||||
label="SQLite disk limit (MB) (0 = unlimited)"
|
||||
note="Hard cap on the on-disk stats.db file. A positive number is the ceiling — oldest rows are pruned and the DB vacuumed to stay under it; 0 lets it grow with the disk."
|
||||
label="Disk limit (MB) (0 = unlimited)"
|
||||
// Was: "oldest rows are pruned and the DB vacuumed". There is no
|
||||
// SQLite and no VACUUM here — the store is bbolt, and reclaiming
|
||||
// space means rebuilding the file (bbolt.Compact + atomic swap).
|
||||
// The rebuild is SKIPPED when the filesystem cannot fit the
|
||||
// transient second copy, so "ceiling" was a promise too: the DB
|
||||
// then sits over the cap until space frees up. Both are said.
|
||||
note="Target size for the on-disk stats.db file. Above it, the oldest rows are deleted and the file is rebuilt to give the space back — the rebuild needs room for a temporary second copy, so on a full disk the file stays over the limit until space frees up. 0 lets it grow with the disk."
|
||||
>
|
||||
<InlineEdit<number>
|
||||
value={globals?.StatsDiskLimitMB ?? 0}
|
||||
@@ -719,7 +755,7 @@ export default function Settings() {
|
||||
inputMode="numeric"
|
||||
width="9rem"
|
||||
placeholder="Unlimited"
|
||||
ariaLabel="SQLite disk limit in MB (0 = unlimited)"
|
||||
ariaLabel="Stats database disk limit in MB (0 = unlimited)"
|
||||
busy={busy}
|
||||
disabled={retentionDisabledCtl}
|
||||
onCommit={(v) =>
|
||||
@@ -728,7 +764,7 @@ export default function Settings() {
|
||||
/>
|
||||
</Field>
|
||||
)}
|
||||
{loggingSqlite && diskUnlimited && (
|
||||
{loggingDisk && diskUnlimited && (
|
||||
<p className="set-warn" role="status">
|
||||
Unlimited — stats.db grows with disk; set a cap (MB) to bound it.
|
||||
</p>
|
||||
|
||||
+19
-46
@@ -29,6 +29,13 @@ import type {
|
||||
Node,
|
||||
} from '../api'
|
||||
import { fmtClock, fmtDuration } from '../format'
|
||||
import {
|
||||
DPI_TYPES,
|
||||
EGRESS_TYPES,
|
||||
UNKNOWN_EGRESS_TYPE_HINT,
|
||||
egressTypeInfo,
|
||||
nextEgress,
|
||||
} from '../egressEdit'
|
||||
|
||||
// The Targets page is a thin editor over the desired-state Model — the same
|
||||
// shape as DNS.tsx and Nodes.tsx. It manages the three things a routing rule
|
||||
@@ -112,9 +119,9 @@ function findReferences(m: Model, kind: RefKind, name: string): RefSite[] {
|
||||
if (asArray(c.Hops).some((h) => isPrefixed(h, kind, name)))
|
||||
out.push({ label: `chain “${c.Name}” hop` })
|
||||
}
|
||||
for (const e of asArray(m.Egresses)) {
|
||||
if (isPrefixed(e.Target, kind, name)) out.push({ label: `egress “${e.Name}” target` })
|
||||
}
|
||||
// No Egresses loop: an egress carries no target. The one that stood here read
|
||||
// `e.Target`, absent from the Go model, so it was `undefined` on every egress
|
||||
// and never once matched — a dead branch shaped like a covered case.
|
||||
for (const r of asArray(m.Resolvers)) {
|
||||
if (isPrefixed(r.Detour, kind, name)) out.push({ label: `resolver “${r.Name}” DNS path` })
|
||||
}
|
||||
@@ -152,7 +159,8 @@ function renameReferences(m: Model, kind: RefKind, from: string, to: string): Mo
|
||||
if (m.Rules) next.Rules = m.Rules.map((r) => ({ ...r, Target: pfx(r.Target), Egress: br(r.Egress) }))
|
||||
if (m.Chains)
|
||||
next.Chains = m.Chains.map((c) => ({ ...c, Hops: c.Hops ? c.Hops.map((h) => pfx(h) ?? h) : c.Hops }))
|
||||
if (m.Egresses) next.Egresses = m.Egresses.map((e) => ({ ...e, Target: pfx(e.Target) }))
|
||||
// No Egresses pass — see targetRefs above: an egress holds no target to rewrite,
|
||||
// and writing the key on would make PUT reject the whole rename with a 400.
|
||||
if (m.Resolvers) next.Resolvers = m.Resolvers.map((r) => ({ ...r, Detour: pfx(r.Detour) }))
|
||||
if (m.Alerts) next.Alerts = m.Alerts.map((a) => ({ ...a, Via: pfx(a.Via) }))
|
||||
if (m.Subscriptions)
|
||||
@@ -244,29 +252,6 @@ function normStrategy(raw: string | undefined): string {
|
||||
|
||||
const PROTOS = ['vless', 'vmess', 'trojan', 'ss'] as const
|
||||
|
||||
/**
|
||||
* The three egress kinds that produce a real way out. `Proxy` and `Block` were
|
||||
* removed: neither ever created an outbound, so everything bound to them fell
|
||||
* through to the plain WAN with the real address. Send traffic through a proxy by
|
||||
* routing it at a group/node/chain, and drop it with the `block` target on a rule.
|
||||
*/
|
||||
const EGRESS_TYPES: ReadonlyArray<{ id: string; label: string; blurb: string }> = [
|
||||
{
|
||||
id: 'interface',
|
||||
label: 'Interface — out a specific WAN or tunnel',
|
||||
blurb: 'Binds to one device (wan, wg0, …) so this traffic leaves over that uplink.',
|
||||
},
|
||||
{
|
||||
id: 'direct',
|
||||
label: 'Direct — straight out, with an optional DPI preset',
|
||||
blurb: 'Uses the normal route. Its point is the DPI preset below, applied to what you route here.',
|
||||
},
|
||||
{
|
||||
id: 'byedpi',
|
||||
label: 'ByeDPI — through the local ciadpi desync proxy',
|
||||
blurb: 'Hands traffic to ciadpi on 127.0.0.1, which desyncs it and goes out direct.',
|
||||
},
|
||||
]
|
||||
const EGRESS_TYPE_LABEL: Record<string, string> = Object.fromEntries(
|
||||
EGRESS_TYPES.map((t) => [t.id, t.label.split(' — ')[0]]),
|
||||
)
|
||||
@@ -278,12 +263,6 @@ const DPI_PRESETS: ReadonlyArray<{ id: string; label: string }> = [
|
||||
{ id: 'spoof', label: 'Spoof' },
|
||||
]
|
||||
|
||||
/**
|
||||
* Types whose native DPI-bypass preset applies. NOT byedpi: the desync happens
|
||||
* inside the ciadpi process, and the engine's tls_* flags are never stamped on top
|
||||
* of it — so the control is hidden there rather than accepted and dropped.
|
||||
*/
|
||||
const DPI_TYPES = new Set(['interface', 'direct'])
|
||||
|
||||
// ---- group membership health ------------------------------------------------
|
||||
//
|
||||
@@ -2839,7 +2818,7 @@ function EgressEditor({
|
||||
const [port, setPort] = useState(initial?.Port != null ? String(initial.Port) : '')
|
||||
const [dpi, setDpi] = useState(initial?.DPI || 'off')
|
||||
const [err, setErr] = useState<string | null>(null)
|
||||
const typeInfo = EGRESS_TYPES.find((t) => t.id === type)
|
||||
const typeInfo = egressTypeInfo(type)
|
||||
// This egress was byedpi when the editor opened — its own type stays legal
|
||||
// even with the package gone, so saved config can always round-trip.
|
||||
const wasByedpi = initial?.Type === 'byedpi'
|
||||
@@ -2856,14 +2835,11 @@ function EgressEditor({
|
||||
if (type === 'byedpi' && byedpiLocked)
|
||||
return setErr('Install the byedpi package to add a ByeDPI egress.')
|
||||
setErr(null)
|
||||
const base: Egress = { ...(initial ?? ({} as Egress)), Name: nm, Type: type }
|
||||
// Only carry the fields the chosen type uses; clear the rest. `Target` belonged
|
||||
// to the removed `proxy` type and is cleared unconditionally.
|
||||
base.Interface = type === 'interface' ? iface.trim() : undefined
|
||||
base.Target = undefined
|
||||
base.Port = type === 'byedpi' ? Number(port.trim()) || 1080 : undefined
|
||||
base.DPI = DPI_TYPES.has(type) ? dpi : undefined
|
||||
await onSave(base)
|
||||
// Carry the fields the chosen type uses and clear the rest — but ONLY for a
|
||||
// type this editor renders those fields for. An unknown type keeps every
|
||||
// stored setting untouched, because this form showed the operator none of
|
||||
// them and must not delete what it declined to display. See nextEgress.
|
||||
await onSave(nextEgress(initial, { name: nm, type, iface, port, dpi }))
|
||||
}
|
||||
|
||||
return (
|
||||
@@ -2912,10 +2888,7 @@ function EgressEditor({
|
||||
})}
|
||||
{!typeInfo && <option value={type}>{type || '—'} (unknown)</option>}
|
||||
</select>
|
||||
<p className="tg-fhint">
|
||||
{typeInfo?.blurb ??
|
||||
'This engine builds no outbound for that type, so everything routed here is blocked. Pick one above.'}
|
||||
</p>
|
||||
<p className="tg-fhint">{typeInfo?.blurb ?? UNKNOWN_EGRESS_TYPE_HINT}</p>
|
||||
{byedpiLocked && (
|
||||
<p className="tg-fhint">
|
||||
Install the <code>byedpi</code> package to enable the ByeDPI egress.
|
||||
|
||||
@@ -15,7 +15,13 @@
|
||||
import { test } from 'node:test'
|
||||
import assert from 'node:assert/strict'
|
||||
|
||||
import { engineReadout, engineState, killSwitchReadout, protectionState } from './planeState.ts'
|
||||
import {
|
||||
engineReadout,
|
||||
engineState,
|
||||
killSwitchClosed,
|
||||
killSwitchReadout,
|
||||
protectionState,
|
||||
} from './planeState.ts'
|
||||
import type { Status, Traffic } from './api.ts'
|
||||
|
||||
/** A healthy, fully-installed router; `traffic` is what each case varies. */
|
||||
@@ -251,3 +257,175 @@ test('the live kill_switch wins over the saved one; the saved one only fills a g
|
||||
assert.equal(killSwitchReadout(noKill as Status, 'open').state, 'open')
|
||||
assert.equal(killSwitchReadout(noKill as Status, 'closed').state, 'armed')
|
||||
})
|
||||
|
||||
// --- killSwitchClosed: the daemon's spelling, not the panel's ---------------
|
||||
//
|
||||
// `Status.kill_switch` is `option kill_switch` echoed verbatim (apply.go:
|
||||
// `s.KillSwitch = m.Globals.KillSwitch`), and the daemon decides with
|
||||
//
|
||||
// !strings.EqualFold(strings.TrimSpace(g.KillSwitch), "open")
|
||||
//
|
||||
// so case, surrounding whitespace and an empty value all read as CLOSED on the
|
||||
// router. The panel compared `=== 'closed'`, which read every one of them as
|
||||
// OPEN: an amber lamp over the word OPEN and "Nothing is meant to be blocked" on
|
||||
// a router that blocks — and, through protectionState's `failClosed`, a
|
||||
// plane-less router downgraded from crit to amber on the same misreading.
|
||||
//
|
||||
// These are the inputs a hand-edited /etc/config/shater actually produces.
|
||||
|
||||
test('closed is closed however it is spelled — case, padding, and empty', () => {
|
||||
for (const raw of ['closed', 'Closed', 'CLOSED', ' closed ', '\tclosed\n', '', ' ']) {
|
||||
assert.equal(killSwitchClosed(raw), true, `killSwitchClosed(${JSON.stringify(raw)})`)
|
||||
}
|
||||
// Absent is the same question with no answer, and it falls closed too.
|
||||
assert.equal(killSwitchClosed(undefined), true)
|
||||
assert.equal(killSwitchClosed(null), true)
|
||||
// Anything unrecognised is NOT taken as permission to stop blocking.
|
||||
assert.equal(killSwitchClosed('nonsense'), true)
|
||||
})
|
||||
|
||||
test('only "open" is open — but every spelling of it is', () => {
|
||||
for (const raw of ['open', 'Open', 'OPEN', ' open ', 'oPeN']) {
|
||||
assert.equal(killSwitchClosed(raw), false, `killSwitchClosed(${JSON.stringify(raw)})`)
|
||||
}
|
||||
})
|
||||
|
||||
test('a router blocking under "Closed" is never drawn as OPEN', () => {
|
||||
for (const raw of ['Closed', ' closed ', '']) {
|
||||
const k = killSwitchReadout(status({ kill_switch: raw }))
|
||||
assert.equal(k.state, 'armed', `state for ${JSON.stringify(raw)}`)
|
||||
assert.equal(k.value, 'ARMED')
|
||||
assert.equal(k.variant, 'on')
|
||||
assert.notEqual(k.value, 'OPEN')
|
||||
}
|
||||
})
|
||||
|
||||
test('the saved policy is normalised too, not just the live one', () => {
|
||||
const { kill_switch, ...noKill } = status()
|
||||
void kill_switch
|
||||
assert.equal(killSwitchReadout(noKill as Status, ' Closed ').state, 'armed')
|
||||
assert.equal(killSwitchReadout(noKill as Status, 'OPEN').state, 'open')
|
||||
})
|
||||
|
||||
test('"Closed" with no plane is the CRIT state, not the amber fail-open one', () => {
|
||||
// The inverted-lamp case with teeth: the daemon is blocking-by-policy and has
|
||||
// no plane installed, which is "not protected AND nothing is stopping it".
|
||||
// Reading "Closed" as fail-open downgraded that from crit to amber and told
|
||||
// the operator it was their own configured choice.
|
||||
for (const raw of ['Closed', ' closed ', '']) {
|
||||
const p = protectionState(status({ kill_switch: raw, plane: 'none' }))
|
||||
assert.equal(p.variant, 'crit', `variant for ${JSON.stringify(raw)}`)
|
||||
assert.equal(p.headline, 'Not protected — traffic is going out directly')
|
||||
assert.equal(p.alarm, true)
|
||||
}
|
||||
// The genuinely fail-open router still gets the amber, deliberate-choice text.
|
||||
const open = protectionState(status({ kill_switch: 'OPEN', plane: 'none' }))
|
||||
assert.equal(open.variant, 'amber')
|
||||
assert.equal(open.headline, 'Not protected — running direct')
|
||||
})
|
||||
|
||||
// --- the holding plane does not promise "nothing" ---------------------------
|
||||
//
|
||||
// netplane/nft.go's RenderHoldNft states its own contract as "No client TRAFFIC
|
||||
// reaches the WAN" and names the exception in the same paragraph: clients still
|
||||
// reach the router's resolver and dnsmasq forwards those lookups to the ISP in
|
||||
// the clear, deliberately, because blocking it would also cut the daemon's own
|
||||
// name resolution. The banner used to round that up to "Nothing is being let
|
||||
// out".
|
||||
|
||||
// --- an unreadable config is not "switched off" -----------------------------
|
||||
//
|
||||
// `enabled`, `kill_switch` and `panel_port` all come from the configuration, so
|
||||
// when the daemon could not READ it they are zero values that mean nothing
|
||||
// (Status.config_readable). The failure happens on a full /overlay or a `uci
|
||||
// commit` caught half-written — which is precisely when the fail-closed plane has
|
||||
// the LAN cut off on purpose — and the daemon then publishes `plane:"hold"` WITH
|
||||
// `enabled:false`. Checking `!enabled` first turned that into "Turned off", amber,
|
||||
// no alarm, "turn the service on in Settings": the owner of a house with no
|
||||
// internet told they did it themselves, and pointed at a page reading the same
|
||||
// unreadable file. The ORDER of the two checks is the fix, so these pin the order.
|
||||
|
||||
/** The proven field state: hold plane, placeholders, and no config reading. */
|
||||
function unreadable(over: Partial<Status> = {}): Status {
|
||||
return status({
|
||||
config_readable: false,
|
||||
config_error: 'uci show shater: exit status 1',
|
||||
enabled: false,
|
||||
kill_switch: '',
|
||||
plane: 'hold',
|
||||
engine_running: false,
|
||||
active: false,
|
||||
...over,
|
||||
})
|
||||
}
|
||||
|
||||
test('an unreadable config is a crit alarm, never "Turned off"', () => {
|
||||
const p = protectionState(unreadable())
|
||||
assert.equal(p.variant, 'crit')
|
||||
assert.equal(p.alarm, true)
|
||||
assert.notEqual(p.headline, 'Turned off')
|
||||
assert.match(p.headline, /configuration/i)
|
||||
})
|
||||
|
||||
test('it says not to switch anything off — the instinct that makes it worse', () => {
|
||||
const d = protectionState(unreadable()).detail
|
||||
assert.match(d, /don’t turn anything off|do not turn anything off/i)
|
||||
assert.match(d, /fail-closed plane doing its job/i)
|
||||
assert.doesNotMatch(d, /Turn the service on in Settings/)
|
||||
})
|
||||
|
||||
test('it wins over `enabled` whatever that placeholder happens to say', () => {
|
||||
// `enabled` is a zero value here; neither of its readings may reach a branch.
|
||||
for (const enabled of [false, true]) {
|
||||
const p = protectionState(unreadable({ enabled }))
|
||||
assert.equal(p.variant, 'crit', `enabled=${enabled}`)
|
||||
assert.notEqual(p.headline, 'Turned off')
|
||||
}
|
||||
})
|
||||
|
||||
test('an OLDER daemon that never sends the field is not alarmed at forever', () => {
|
||||
// Absent ⇒ "no reading", which is NOT "the read failed". Such a daemon's
|
||||
// `enabled` means what it says, so the ordinary branches must still run —
|
||||
// reading the field as `!== true` would have alarmed on every one of them.
|
||||
const { config_readable, ...noField } = unreadable({ enabled: false })
|
||||
void config_readable
|
||||
assert.equal(protectionState(noField as Status).headline, 'Turned off')
|
||||
// And a healthy older daemon still reaches its ordinary green readout.
|
||||
const running = withTraffic({ verdict: 'tunnel', tunnel_rules: 2 })
|
||||
assert.equal(protectionState(running).headline, 'Protected')
|
||||
assert.equal(protectionState(running).variant, 'on')
|
||||
})
|
||||
|
||||
test('the kill-switch readout refuses to name a policy it could not read', () => {
|
||||
// kill_switch is "" here — which normalises to CLOSED, and that is exactly the
|
||||
// trap: an unreadable config printed ARMED, green, "setting: fail-closed" from a
|
||||
// placeholder. The `setting` word carries the third answer so no caller has to
|
||||
// re-derive a boolean that cannot hold it.
|
||||
const k = killSwitchReadout(unreadable())
|
||||
assert.equal(k.state, 'unknown')
|
||||
assert.notEqual(k.value, 'ARMED')
|
||||
assert.equal(k.variant, 'off')
|
||||
assert.equal(k.setting, 'not known')
|
||||
assert.equal(k.settingHot, true)
|
||||
})
|
||||
|
||||
test('a readable config still names its policy, both ways', () => {
|
||||
const closed = killSwitchReadout(status({ config_readable: true }))
|
||||
assert.equal(closed.setting, 'fail-closed')
|
||||
assert.equal(closed.settingHot, false)
|
||||
const open = killSwitchReadout(status({ config_readable: true, kill_switch: 'open' }))
|
||||
assert.equal(open.setting, 'fail-open')
|
||||
assert.equal(open.settingHot, true)
|
||||
})
|
||||
|
||||
test('a readable config still reaches the calm "Turned off"', () => {
|
||||
const p = protectionState(status({ config_readable: true, enabled: false }))
|
||||
assert.equal(p.headline, 'Turned off')
|
||||
assert.equal(p.alarm, false)
|
||||
})
|
||||
|
||||
test('the hold banner does not claim nothing leaves, and names DNS', () => {
|
||||
const d = protectionState(status({ plane: 'hold', engine_running: false })).detail
|
||||
assert.match(d, /DNS/)
|
||||
assert.doesNotMatch(d, /^Nothing is being let out/)
|
||||
})
|
||||
|
||||
+123
-5
@@ -86,6 +86,33 @@ export function engineReadout(status: Status | null): { variant: LedVariant; wor
|
||||
*/
|
||||
export type KillSwitchState = 'armed' | 'inert' | 'unknown' | 'open'
|
||||
|
||||
/**
|
||||
* IS THE KILL-SWITCH CLOSED? The daemon's rule, exactly:
|
||||
*
|
||||
* killSwitchClosed = !strings.EqualFold(strings.TrimSpace(g.KillSwitch), "open")
|
||||
* (apply/apply.go)
|
||||
*
|
||||
* Three things it does that `=== 'closed'` did not, each of which had inverted a
|
||||
* lamp. `Status.kill_switch` is the RAW UCI string, echoed with no normalisation
|
||||
* (`s.KillSwitch = m.Globals.KillSwitch`), so all three inputs are reachable from
|
||||
* a hand-edited /etc/config/shater:
|
||||
*
|
||||
* - CASE. `kill_switch 'Closed'` blocks on the router. `'Closed' === 'closed'`
|
||||
* is false, so the panel drew OPEN, amber, "Nothing is meant to be blocked" —
|
||||
* and `plane: 'none'` was downgraded from crit to amber on the same reading.
|
||||
* - WHITESPACE. `' closed '` likewise.
|
||||
* - THE DEFAULT SIDE. An empty value blocks on the router. `??` only falls
|
||||
* through null/undefined, so `''` reached the comparison, failed it, and read
|
||||
* as OPEN — the default landing on the side that cannot be recovered from by
|
||||
* looking at the page.
|
||||
*
|
||||
* Everything in the panel that asks this question must ask it here. Two pages
|
||||
* having their own spelling of the same comparison is how they came to disagree.
|
||||
*/
|
||||
export function killSwitchClosed(raw: string | null | undefined): boolean {
|
||||
return (raw ?? '').trim().toLowerCase() !== 'open'
|
||||
}
|
||||
|
||||
export interface KillSwitchReadout {
|
||||
state: KillSwitchState
|
||||
/** The word the module puts in its readout. */
|
||||
@@ -95,6 +122,19 @@ export interface KillSwitchReadout {
|
||||
blockingNow: string | null
|
||||
/** True when `blockingNow` is bad news and should be drawn hot. */
|
||||
hot: boolean
|
||||
/**
|
||||
* The CONFIGURED policy, as a word for the "setting" row: 'fail-closed',
|
||||
* 'fail-open', or 'not known'.
|
||||
*
|
||||
* It lives here rather than being re-derived at each call site because it was
|
||||
* re-derived at each call site: both pages kept their own `killArmed ?
|
||||
* 'fail-closed' : 'fail-open'` beside this readout, which cannot express the third
|
||||
* answer — so a router whose configuration could not be READ printed a confident
|
||||
* "fail-closed" under a lamp that already said NOT REPORTED.
|
||||
*/
|
||||
setting: string
|
||||
/** True when `setting` should be drawn hot (fail-open, or not known). */
|
||||
settingHot: boolean
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -113,22 +153,60 @@ export interface KillSwitchReadout {
|
||||
* /api/status has not reported one. The live value wins wherever it exists, as
|
||||
* everywhere else in the panel: this is a status readout, and the config on disk
|
||||
* can already differ from what is installed.
|
||||
*
|
||||
* AN UNREADABLE CONFIGURATION IS ITS OWN ANSWER, and it comes first. `kill_switch`
|
||||
* is one of the three fields sourced from the config, so when `config_readable` is
|
||||
* false it is a zero value — and the empty string normalises to "closed", which is
|
||||
* how a router nobody could read printed ARMED, green, "setting: fail-closed". The
|
||||
* lamp said NOT REPORTED two lines further down in an earlier draft of this fix and
|
||||
* the row still said fail-closed, which is the same defect twice.
|
||||
*/
|
||||
export function killSwitchReadout(
|
||||
status: Status | null,
|
||||
configured?: string,
|
||||
): KillSwitchReadout {
|
||||
const closed = (status?.kill_switch ?? configured ?? 'closed') === 'closed'
|
||||
if (status?.config_readable === false) {
|
||||
return {
|
||||
state: 'unknown',
|
||||
value: 'NOT REPORTED',
|
||||
variant: 'off',
|
||||
// The plane may well be blocking (a hold plane is installed in exactly this
|
||||
// situation) — what is unknown is the SETTING, so that is what this says.
|
||||
blockingNow: 'setting can’t be read',
|
||||
hot: false,
|
||||
setting: 'not known',
|
||||
settingHot: true,
|
||||
}
|
||||
}
|
||||
|
||||
const closed = killSwitchClosed(status?.kill_switch ?? configured)
|
||||
const setting = closed
|
||||
? { setting: 'fail-closed', settingHot: false }
|
||||
: { setting: 'fail-open', settingHot: true }
|
||||
|
||||
if (!closed) {
|
||||
return { state: 'open', value: 'OPEN', variant: 'amber', blockingNow: null, hot: false }
|
||||
return {
|
||||
state: 'open',
|
||||
value: 'OPEN',
|
||||
variant: 'amber',
|
||||
blockingNow: null,
|
||||
hot: false,
|
||||
...setting,
|
||||
}
|
||||
}
|
||||
|
||||
switch (status?.plane) {
|
||||
case 'full':
|
||||
case 'hold':
|
||||
// Something is installed, so the fail-closed guard is really in the path.
|
||||
return { state: 'armed', value: 'ARMED', variant: 'on', blockingNow: null, hot: false }
|
||||
return {
|
||||
state: 'armed',
|
||||
value: 'ARMED',
|
||||
variant: 'on',
|
||||
blockingNow: null,
|
||||
hot: false,
|
||||
...setting,
|
||||
}
|
||||
case 'none':
|
||||
return {
|
||||
state: 'inert',
|
||||
@@ -136,6 +214,7 @@ export function killSwitchReadout(
|
||||
variant: 'crit',
|
||||
blockingNow: 'no — nothing installed',
|
||||
hot: true,
|
||||
...setting,
|
||||
}
|
||||
default:
|
||||
return {
|
||||
@@ -146,6 +225,7 @@ export function killSwitchReadout(
|
||||
// wrapped onto three lines beside a one-word key.
|
||||
blockingNow: status ? 'not known' : 'no reading yet',
|
||||
hot: false,
|
||||
...setting,
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -189,7 +269,36 @@ export function protectionState(status: Status | null): ProtectionState {
|
||||
}
|
||||
}
|
||||
|
||||
const failClosed = (status.kill_switch ?? 'closed') === 'closed'
|
||||
// COULD THE CONFIG BE READ? This has to come FIRST, before `enabled` and before
|
||||
// `kill_switch`, because both of those are sourced from the configuration and are
|
||||
// zero values when it could not be read (Status.config_readable).
|
||||
//
|
||||
// The order was the bug, and it is not a cosmetic one. The daemon cannot read the
|
||||
// config when /overlay is full or a `uci commit` was interrupted part-way — which
|
||||
// is exactly when the fail-closed plane has the LAN cut off on purpose. It then
|
||||
// publishes `plane:"hold"` together with `enabled:false`, and checking `!enabled`
|
||||
// first turned that into "Turned off", amber, alarm:false, "turn the service on in
|
||||
// Settings" — telling the owner of a house with no internet that they switched it
|
||||
// off themselves, and sending them to a page backed by the same unreadable file.
|
||||
//
|
||||
// Read as `=== false`, never `!== true`: the field is positively phrased so that
|
||||
// its ABSENCE (an older daemon) reads as "no reading" — but "no reading" is not
|
||||
// the same as "the read failed", and only the latter earns this crit. An older
|
||||
// daemon has an `enabled` that means what it says, so it must fall through to the
|
||||
// branches below rather than be alarmed at forever.
|
||||
if (status.config_readable === false) {
|
||||
return {
|
||||
variant: 'crit',
|
||||
headline: 'The router’s configuration can’t be read',
|
||||
// Carries the daemon's own load-bearing sentence: the instinct here is to
|
||||
// switch things off, and that is the one action that makes it worse.
|
||||
detail:
|
||||
'This page can’t say whether shater is on, or whether the kill-switch is closed. If traffic is being blocked, that is the fail-closed plane doing its job — not the service being off, so don’t turn anything off to fix it. The usual causes are a full disk or an interrupted save; see the findings below.',
|
||||
alarm: true,
|
||||
}
|
||||
}
|
||||
|
||||
const failClosed = killSwitchClosed(status.kill_switch)
|
||||
|
||||
// The service being switched off is a deliberate state, not a fault.
|
||||
if (!status.enabled) {
|
||||
@@ -205,11 +314,20 @@ export function protectionState(status: Status | null): ProtectionState {
|
||||
case 'full':
|
||||
return fullPlaneState(status.traffic)
|
||||
case 'hold':
|
||||
// "Nothing is being let out" was one word too wide, and the word was load-
|
||||
// bearing. The holding plane's own contract (netplane/nft.go RenderHoldNft)
|
||||
// says "No client TRAFFIC reaches the WAN" and then names the exception in
|
||||
// the same breath: clients can still reach the router's resolver, and
|
||||
// dnsmasq forwards those queries to the ISP in the clear. That is deliberate
|
||||
// and cannot be closed — blocking it would also cut the daemon's own name
|
||||
// resolution, and with it any chance of fetching what it choked on and
|
||||
// recovering. A panel that rounds "no traffic" up to "nothing" is claiming a
|
||||
// guarantee the plane below it never made.
|
||||
return {
|
||||
variant: 'amber',
|
||||
headline: 'Traffic blocked — the tunnel is down',
|
||||
detail:
|
||||
'Nothing is being let out rather than let out unprotected. Devices can still reach each other and the router, so you can fix it from here.',
|
||||
'No traffic from your devices is reaching the internet — it’s blocked rather than let out unprotected. Devices can still reach each other and the router, so you can fix it from here. DNS is the exception: lookups sent to the router still go out to your provider in the clear.',
|
||||
alarm: true,
|
||||
}
|
||||
case 'none':
|
||||
|
||||
@@ -0,0 +1,72 @@
|
||||
// carryRulesetFormat — the field the rule-set edit form cannot show and therefore
|
||||
// must not drop.
|
||||
//
|
||||
// Run with `npm test` (node's built-in test runner + native type stripping).
|
||||
// ruleset.ts has no runtime imports, so this runs against the real module.
|
||||
//
|
||||
// THE CASE THIS FILE WAS WRITTEN FOR is the boring one: open a rule-set that was
|
||||
// given an explicit `Format` over SSH, change nothing but its NAME, press Save.
|
||||
// The form rebuilt the object from its own controls, `Format` was on none of them,
|
||||
// and `updateRuleset` swapped the element whole — so the value was gone, with no
|
||||
// control anywhere in the panel able to put it back. What follows is silent: per
|
||||
// generate/ruleset.go the list then matches nothing, the rule using it stops
|
||||
// firing, and its traffic falls through to the next rule with nothing logged.
|
||||
|
||||
import { test } from 'node:test'
|
||||
import assert from 'node:assert/strict'
|
||||
|
||||
import { carryRulesetFormat } from './ruleset.ts'
|
||||
import type { Ruleset } from './api.ts'
|
||||
|
||||
/** A saved rule-set carrying a format the panel has no control for. */
|
||||
function saved(over: Partial<Ruleset> = {}): Ruleset {
|
||||
return { Name: 'blocked-ru', Type: 'domain', Source: 'file', Path: '/etc/shater/ru.lst', Format: 'binary', ...over }
|
||||
}
|
||||
|
||||
/** What the edit form rebuilds from its own controls — no Format among them. */
|
||||
function rebuilt(over: Partial<Ruleset> = {}): Ruleset {
|
||||
return { Name: 'blocked-ru', Type: 'domain', Source: 'file', Path: '/etc/shater/ru.lst', ...over }
|
||||
}
|
||||
|
||||
test('renaming a file rule-set keeps its explicit format', () => {
|
||||
const out = carryRulesetFormat(rebuilt({ Name: 'ru-blocked' }), saved())
|
||||
assert.equal(out.Format, 'binary')
|
||||
assert.equal(out.Name, 'ru-blocked')
|
||||
// Everything else the form rebuilt is untouched.
|
||||
assert.equal(out.Path, '/etc/shater/ru.lst')
|
||||
assert.equal(out.Source, 'file')
|
||||
})
|
||||
|
||||
test('a url rule-set keeps it too — that is what stops a .srs being read as text', () => {
|
||||
const out = carryRulesetFormat(
|
||||
rebuilt({ Source: 'url', Path: undefined, URL: 'https://example.invalid/list' }),
|
||||
saved({ Source: 'url', Format: 'binary' }),
|
||||
)
|
||||
assert.equal(out.Format, 'binary')
|
||||
})
|
||||
|
||||
test('the sources the generator ignores it on do not get it, so it cannot raise a warning', () => {
|
||||
for (const Source of ['inline', 'geosite', 'geoip']) {
|
||||
const out = carryRulesetFormat(rebuilt({ Source, Path: undefined }), saved({ Source }))
|
||||
assert.equal(out.Format, undefined, `Source=${Source}`)
|
||||
}
|
||||
})
|
||||
|
||||
test('nothing to carry is not something to invent', () => {
|
||||
const next = rebuilt()
|
||||
// No prior rule-set at all (the ADD path).
|
||||
assert.equal(carryRulesetFormat(next, null).Format, undefined)
|
||||
assert.equal(carryRulesetFormat(next, undefined).Format, undefined)
|
||||
// A prior rule-set with an empty or blank format.
|
||||
assert.equal(carryRulesetFormat(next, saved({ Format: '' })).Format, undefined)
|
||||
assert.equal(carryRulesetFormat(next, saved({ Format: ' ' })).Format, undefined)
|
||||
})
|
||||
|
||||
test('the inputs are not mutated — the caller keeps a usable `initial`', () => {
|
||||
const next = rebuilt()
|
||||
const prev = saved()
|
||||
const out = carryRulesetFormat(next, prev)
|
||||
assert.notEqual(out, next)
|
||||
assert.equal(next.Format, undefined, 'the rebuilt literal must not be written through')
|
||||
assert.equal(prev.Format, 'binary')
|
||||
})
|
||||
@@ -0,0 +1,52 @@
|
||||
// Rule-set model rules the panel must not get wrong — the ones about fields the
|
||||
// panel does not show.
|
||||
//
|
||||
// This module exists so they can be tested. The forms that apply them live in
|
||||
// page components (Routing.tsx), which import CSS and React and therefore cannot
|
||||
// be loaded by `npm test`; the imports here are type-only, exactly like
|
||||
// planeState.ts, so the test runs against the real code with nothing stubbed.
|
||||
|
||||
import type { Ruleset } from './api'
|
||||
|
||||
/**
|
||||
* The sources for which the GENERATOR reads `Ruleset.Format`.
|
||||
*
|
||||
* Positive and closed, not "everything except inline". generate/ruleset.go
|
||||
* consults the field in exactly two branches and WARNS about it in the others
|
||||
* ("format set on a source that has no use for it"), so a source that grows into
|
||||
* the model later must be added here deliberately rather than inheriting a
|
||||
* meaning nobody chose for it.
|
||||
*/
|
||||
const FORMAT_BEARING_SOURCES: ReadonlySet<string> = new Set(['url', 'file'])
|
||||
|
||||
/**
|
||||
* Carry `Format` from the ruleset being edited onto the one being saved.
|
||||
*
|
||||
* WHY THIS IS NOT JUST A SPREAD. The edit form rebuilds a `Ruleset` literal per
|
||||
* source branch, and rebuilding is right: switching a list from `url` to `file`
|
||||
* must not drag the old `URL` and `UpdateInterval` along. But `Format` is not like
|
||||
* those — the panel has NO CONTROL for it. It cannot be set here and cannot be
|
||||
* restored here, so a save that dropped it destroyed a value only an SSH session
|
||||
* could put back, and the trigger was as domestic as RENAMING the list (the rename
|
||||
* path runs through the same rebuild, and the update replaces the element whole).
|
||||
*
|
||||
* WHAT IT COSTS WHEN IT GOES, per generate/ruleset.go:
|
||||
*
|
||||
* - `file` — an explicit format OVERRIDES the filename extension, and is the
|
||||
* only way to load a rule-set whose name says nothing about its contents.
|
||||
* - `url` — it is what stops a compiled binary `.srs`, served without a
|
||||
* recognisable extension, from being parsed as a plain-text domain list.
|
||||
*
|
||||
* Either way the rule-set silently matches nothing afterwards. Nothing errors: the
|
||||
* rule that uses it simply stops firing and its traffic falls through to whatever
|
||||
* rule comes next, which is the failure mode that costs a day to find.
|
||||
*
|
||||
* Returns a NEW ruleset; the inputs are not modified. An empty/absent Format, or a
|
||||
* source the generator ignores it on, yields `next` unchanged.
|
||||
*/
|
||||
export function carryRulesetFormat(next: Ruleset, initial: Ruleset | null | undefined): Ruleset {
|
||||
const format = (initial?.Format ?? '').trim()
|
||||
if (!format) return next
|
||||
if (!FORMAT_BEARING_SOURCES.has((next.Source ?? '').trim().toLowerCase())) return next
|
||||
return { ...next, Format: format }
|
||||
}
|
||||
@@ -0,0 +1,268 @@
|
||||
// lx:begin l3-honest-drop
|
||||
package route
|
||||
|
||||
import (
|
||||
"context"
|
||||
"net/netip"
|
||||
"testing"
|
||||
|
||||
"github.com/sagernet/sing-box/adapter"
|
||||
C "github.com/sagernet/sing-box/constant"
|
||||
"github.com/sagernet/sing-box/log"
|
||||
"github.com/sagernet/sing-box/option"
|
||||
R "github.com/sagernet/sing-box/route/rule"
|
||||
"github.com/sagernet/sing/common/json/badoption"
|
||||
M "github.com/sagernet/sing/common/metadata"
|
||||
N "github.com/sagernet/sing/common/network"
|
||||
|
||||
"github.com/stretchr/testify/require"
|
||||
)
|
||||
|
||||
// The contract under test: PreMatch never answers "continue" (nor "bypass") for
|
||||
// an ICMP flow. adapter.JudgeFlow maps both to tun.ActionAccept, and the TUN
|
||||
// stack answers Accept by FORGING the echo reply itself
|
||||
// (sing-tun stack_gvisor_icmp.go ICMPForwarder.HandlePacket, the fallthrough
|
||||
// under the Flow/Reject/Drop switch). A verdict of "continue" therefore reads to
|
||||
// the operator as a working ping off a tunnel that never carried the packet.
|
||||
//
|
||||
// Every test below has a TCP/UDP twin: the honest drop must not leak into the
|
||||
// protocols where "continue" really does mean "take the ordinary connection
|
||||
// route".
|
||||
|
||||
// icmpL4Outbound is a minimal L4-only outbound (the vless/vmess/... shape): it
|
||||
// does NOT implement adapter.FlowOutbound, and Network() lists only TCP/UDP.
|
||||
// Unused Outbound methods come from the embedded nil interface and are never
|
||||
// called on the pre-match paths under test.
|
||||
type icmpL4Outbound struct {
|
||||
adapter.Outbound
|
||||
tag string
|
||||
}
|
||||
|
||||
func (o *icmpL4Outbound) Tag() string { return o.tag }
|
||||
func (o *icmpL4Outbound) Type() string { return "vless" }
|
||||
func (o *icmpL4Outbound) Network() []string { return []string{N.NetworkTCP, N.NetworkUDP} }
|
||||
|
||||
// icmpOutboundManager resolves tags from a fixed map and hands the same L4-only
|
||||
// outbound out as the default; the rest of the OutboundManager surface is never
|
||||
// touched by the pre-match walk.
|
||||
type icmpOutboundManager struct {
|
||||
adapter.OutboundManager
|
||||
defaultOutbound adapter.Outbound
|
||||
outbounds map[string]adapter.Outbound
|
||||
}
|
||||
|
||||
func (m *icmpOutboundManager) Default() adapter.Outbound { return m.defaultOutbound }
|
||||
|
||||
func (m *icmpOutboundManager) Outbound(tag string) (adapter.Outbound, bool) {
|
||||
outbound, loaded := m.outbounds[tag]
|
||||
return outbound, loaded
|
||||
}
|
||||
|
||||
// icmpDNSRouter / icmpDNSTransportManager implement only what
|
||||
// prepareMatchMetadata reaches. FakeIP returns nil unless a transport is
|
||||
// installed, which is how the "fakeip lookup failed" exit is driven below.
|
||||
type icmpDNSRouter struct {
|
||||
adapter.DNSRouter
|
||||
}
|
||||
|
||||
func (s *icmpDNSRouter) LookupReverseMapping(netip.Addr) (string, bool) { return "", false }
|
||||
|
||||
type icmpDNSTransportManager struct {
|
||||
adapter.DNSTransportManager
|
||||
fakeIP adapter.FakeIPTransport
|
||||
}
|
||||
|
||||
func (s *icmpDNSTransportManager) FakeIP() adapter.FakeIPTransport {
|
||||
if s.fakeIP == nil {
|
||||
return nil
|
||||
}
|
||||
return s.fakeIP
|
||||
}
|
||||
|
||||
// icmpMissingFakeIPTransport claims every address and then fails to look any of
|
||||
// them up — exactly the "missing fakeip record, try enable
|
||||
// `experimental.cache_file`" error prepareMatchMetadata returns.
|
||||
type icmpMissingFakeIPTransport struct {
|
||||
adapter.FakeIPTransport
|
||||
}
|
||||
|
||||
func (t *icmpMissingFakeIPTransport) Store() adapter.FakeIPStore {
|
||||
return &icmpMissingFakeIPStore{}
|
||||
}
|
||||
|
||||
type icmpMissingFakeIPStore struct {
|
||||
adapter.FakeIPStore
|
||||
}
|
||||
|
||||
func (s *icmpMissingFakeIPStore) Contains(netip.Addr) bool { return true }
|
||||
func (s *icmpMissingFakeIPStore) Lookup(netip.Addr) (string, bool) { return "", false }
|
||||
|
||||
type icmpRouterOptions struct {
|
||||
fakeIP adapter.FakeIPTransport
|
||||
rules []option.Rule
|
||||
}
|
||||
|
||||
func icmpTestRouter(t *testing.T, options icmpRouterOptions) *Router {
|
||||
t.Helper()
|
||||
logger := log.NewNOPFactory().NewLogger("test")
|
||||
defaultOutbound := &icmpL4Outbound{tag: "proxy-out"}
|
||||
router := &Router{
|
||||
ctx: context.Background(),
|
||||
logger: logger,
|
||||
dns: &icmpDNSRouter{},
|
||||
dnsTransport: &icmpDNSTransportManager{fakeIP: options.fakeIP},
|
||||
outbound: &icmpOutboundManager{
|
||||
defaultOutbound: defaultOutbound,
|
||||
outbounds: map[string]adapter.Outbound{defaultOutbound.Tag(): defaultOutbound},
|
||||
},
|
||||
}
|
||||
for i, ruleOptions := range options.rules {
|
||||
rule, err := R.NewRule(router.ctx, logger, ruleOptions, false)
|
||||
require.NoError(t, err, "build rule[%d]", i)
|
||||
router.rules = append(router.rules, rule)
|
||||
}
|
||||
return router
|
||||
}
|
||||
|
||||
func icmpTestMetadata(network string) adapter.InboundContext {
|
||||
return adapter.InboundContext{
|
||||
Inbound: "l3-in",
|
||||
InboundType: C.TypeTun,
|
||||
Network: network,
|
||||
Source: M.SocksaddrFrom(netip.MustParseAddr("192.168.1.2"), 0),
|
||||
Destination: M.SocksaddrFrom(netip.MustParseAddr("1.1.1.1"), 0),
|
||||
}
|
||||
}
|
||||
|
||||
// lanRuleWithAction matches every packet from the test source, so the action is
|
||||
// what the test is actually about.
|
||||
func lanRuleWithAction(action option.RuleAction) option.Rule {
|
||||
return option.Rule{
|
||||
Type: C.RuleTypeDefault,
|
||||
DefaultOptions: option.DefaultRule{
|
||||
RawDefaultRule: option.RawDefaultRule{
|
||||
SourceIPCIDR: badoption.Listable[string]{"192.168.1.0/24"},
|
||||
},
|
||||
RuleAction: action,
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
// --- exit 1: an outbound that cannot carry layer 3 --------------------------
|
||||
|
||||
func TestPreMatchICMPToL4OutboundDrops(t *testing.T) {
|
||||
t.Parallel()
|
||||
router := icmpTestRouter(t, icmpRouterOptions{})
|
||||
result := router.PreMatch(icmpTestMetadata(N.NetworkICMP), nil)
|
||||
require.Equal(t, adapter.PreMatchDrop, result.Action,
|
||||
"ICMP to an L4-only outbound fell through to the ordinary pre-match path: the TUN stack will forge the echo reply and ping will lie about a tunnel that never saw the packet")
|
||||
}
|
||||
|
||||
func TestPreMatchTCPToL4OutboundContinues(t *testing.T) {
|
||||
t.Parallel()
|
||||
router := icmpTestRouter(t, icmpRouterOptions{})
|
||||
result := router.PreMatch(icmpTestMetadata(N.NetworkTCP), nil)
|
||||
require.Equal(t, adapter.PreMatchContinue, result.Action,
|
||||
"TCP to an L4-only outbound must keep taking the ordinary connection route; the ICMP honest-drop must not leak into TCP/UDP pre-match")
|
||||
}
|
||||
|
||||
func TestPreMatchUDPToL4OutboundContinues(t *testing.T) {
|
||||
t.Parallel()
|
||||
router := icmpTestRouter(t, icmpRouterOptions{})
|
||||
result := router.PreMatch(icmpTestMetadata(N.NetworkUDP), nil)
|
||||
require.Equal(t, adapter.PreMatchContinue, result.Action,
|
||||
"UDP to an L4-only outbound must keep taking the ordinary connection route")
|
||||
}
|
||||
|
||||
// --- exit 2: prepareMatchMetadata failed before any rule was walked ---------
|
||||
|
||||
// This exit arrived with the shared prepareMatchMetadata refactor (upstream
|
||||
// b911fb078): it returns before the rule walk, so it never reaches preMatchFlow
|
||||
// where the ICMP override used to live.
|
||||
func TestPreMatchICMPMetadataErrorDrops(t *testing.T) {
|
||||
t.Parallel()
|
||||
router := icmpTestRouter(t, icmpRouterOptions{fakeIP: &icmpMissingFakeIPTransport{}})
|
||||
result := router.PreMatch(icmpTestMetadata(N.NetworkICMP), nil)
|
||||
require.Equal(t, adapter.PreMatchDrop, result.Action,
|
||||
"a fakeip record that cannot be resolved must not degrade ICMP to continue: continue is tun.ActionAccept, and Accept is a forged echo reply")
|
||||
}
|
||||
|
||||
func TestPreMatchTCPMetadataErrorContinues(t *testing.T) {
|
||||
t.Parallel()
|
||||
router := icmpTestRouter(t, icmpRouterOptions{fakeIP: &icmpMissingFakeIPTransport{}})
|
||||
result := router.PreMatch(icmpTestMetadata(N.NetworkTCP), nil)
|
||||
require.Equal(t, adapter.PreMatchContinue, result.Action,
|
||||
"for TCP the metadata-error exit must keep meaning `take the ordinary connection route`")
|
||||
}
|
||||
|
||||
// --- exit 3: a rule action the pre-match walk does not handle ---------------
|
||||
|
||||
// hijack-dns is one of the actions PreMatch's switch has no arm for, so it lands
|
||||
// in the default arm. Any future unhandled action lands there too — that is why
|
||||
// the guard is a funnel on the return value and not a per-arm override.
|
||||
func TestPreMatchICMPUnhandledRuleActionDrops(t *testing.T) {
|
||||
t.Parallel()
|
||||
router := icmpTestRouter(t, icmpRouterOptions{
|
||||
rules: []option.Rule{lanRuleWithAction(option.RuleAction{Action: C.RuleActionTypeHijackDNS})},
|
||||
})
|
||||
result := router.PreMatch(icmpTestMetadata(N.NetworkICMP), nil)
|
||||
require.Equal(t, adapter.PreMatchDrop, result.Action,
|
||||
"an unhandled rule action must not degrade ICMP to continue: continue is tun.ActionAccept, and Accept is a forged echo reply")
|
||||
}
|
||||
|
||||
func TestPreMatchTCPUnhandledRuleActionContinues(t *testing.T) {
|
||||
t.Parallel()
|
||||
router := icmpTestRouter(t, icmpRouterOptions{
|
||||
rules: []option.Rule{lanRuleWithAction(option.RuleAction{Action: C.RuleActionTypeHijackDNS})},
|
||||
})
|
||||
result := router.PreMatch(icmpTestMetadata(N.NetworkTCP), nil)
|
||||
require.Equal(t, adapter.PreMatchContinue, result.Action,
|
||||
"the unhandled-action exit must stay a continue for TCP")
|
||||
}
|
||||
|
||||
// --- exit 4: an explicit bypass ---------------------------------------------
|
||||
|
||||
// sing-tun implements ActionBypass on the nfqueue plane only; on the TUN path it
|
||||
// falls into the same default arm as Accept (flow_dispatch.go judgeAndInstall,
|
||||
// and the ICMP forwarder's switch has no Bypass case either), i.e. into the same
|
||||
// forgery. There is no honest bypass for a packet already inside the engine's
|
||||
// TUN.
|
||||
func TestPreMatchICMPBypassDrops(t *testing.T) {
|
||||
t.Parallel()
|
||||
router := icmpTestRouter(t, icmpRouterOptions{
|
||||
rules: []option.Rule{lanRuleWithAction(option.RuleAction{Action: C.RuleActionTypeBypass})},
|
||||
})
|
||||
result := router.PreMatch(icmpTestMetadata(N.NetworkICMP), nil)
|
||||
require.Equal(t, adapter.PreMatchDrop, result.Action,
|
||||
"bypass degrades to tun.ActionAccept on the TUN path, which is the forged echo reply again")
|
||||
}
|
||||
|
||||
func TestPreMatchTCPBypassIsStillBypass(t *testing.T) {
|
||||
t.Parallel()
|
||||
router := icmpTestRouter(t, icmpRouterOptions{
|
||||
rules: []option.Rule{lanRuleWithAction(option.RuleAction{Action: C.RuleActionTypeBypass})},
|
||||
})
|
||||
result := router.PreMatch(icmpTestMetadata(N.NetworkTCP), nil)
|
||||
require.Equal(t, adapter.PreMatchBypass, result.Action,
|
||||
"the ICMP honest-drop must not turn a TCP bypass rule into a drop")
|
||||
}
|
||||
|
||||
// --- the verdicts that must pass through untouched ---------------------------
|
||||
|
||||
// A reject rule already carries its own honest verdict; the funnel must not
|
||||
// rewrite it (a Reject sends an ICMP unreachable, which is information, not a
|
||||
// forged liveness signal).
|
||||
func TestPreMatchICMPRejectIsNotRewritten(t *testing.T) {
|
||||
t.Parallel()
|
||||
router := icmpTestRouter(t, icmpRouterOptions{
|
||||
rules: []option.Rule{lanRuleWithAction(option.RuleAction{
|
||||
Action: C.RuleActionTypeReject,
|
||||
RejectOptions: option.RejectActionOptions{Method: C.RuleActionRejectMethodDefault},
|
||||
})},
|
||||
})
|
||||
result := router.PreMatch(icmpTestMetadata(N.NetworkICMP), nil)
|
||||
require.Equal(t, adapter.PreMatchReject, result.Action,
|
||||
"the ICMP funnel must only rewrite continue/bypass, never an explicit reject")
|
||||
}
|
||||
|
||||
// lx:end l3-honest-drop
|
||||
@@ -314,7 +314,48 @@ func (r *Router) routePacketConnection(ctx context.Context, conn N.PacketConn, m
|
||||
return nil
|
||||
}
|
||||
|
||||
// lx:begin l3-honest-drop
|
||||
// PreMatch funnels every verdict of the pre-match walk through one ICMP check.
|
||||
//
|
||||
// An ICMP flow has no fallback path, so PreMatchContinue is not "try the
|
||||
// ordinary connection route" the way it is for TCP and UDP: the TUN stack takes
|
||||
// the packet back and answers the echo ITSELF (sing-tun stack_gvisor_icmp.go —
|
||||
// adapter.JudgeFlow maps Continue to tun.ActionAccept, and the ICMP forwarder
|
||||
// answers Accept by rewriting Echo into EchoReply and swapping the addresses).
|
||||
// A ping routed to an outbound that cannot carry layer 3 — every proxy
|
||||
// protocol; only adapter.FlowOutbound can — would therefore return a FORGED
|
||||
// reply, and the operator would read a working ping off a tunnel that never saw
|
||||
// the packet. Dropping instead reports the truth.
|
||||
//
|
||||
// PreMatchBypass is folded into the same drop because sing-tun implements
|
||||
// bypass for the nfqueue plane only (`ActionBypass` appears nowhere in
|
||||
// flow_dispatch.go / stack_gvisor_icmp.go): on the TUN path it degrades to the
|
||||
// same Accept, i.e. to the same forgery. There is no honest bypass for an ICMP
|
||||
// packet that is already inside the engine's TUN.
|
||||
//
|
||||
// This is a funnel and not an override inside the walk on purpose: the walk has
|
||||
// several independent exits that say "continue" (the prepareMatchMetadata error
|
||||
// return, the sniff bail-outs, the un-routable `bypass`, and the default arm of
|
||||
// the rule-action switch), and an earlier version of this delta guarded only
|
||||
// the ones that pass through preMatchFlow — leaving the others as narrow paths
|
||||
// to the forged reply. Guarding the single return value cannot be outgrown by a
|
||||
// new exit.
|
||||
func (r *Router) PreMatch(metadata adapter.InboundContext, firstPacket []byte) adapter.PreMatchResult {
|
||||
result := r.preMatch(metadata, firstPacket)
|
||||
if metadata.Network == N.NetworkICMP {
|
||||
switch result.Action {
|
||||
case adapter.PreMatchContinue, adapter.PreMatchBypass:
|
||||
return adapter.PreMatchResult{Action: adapter.PreMatchDrop}
|
||||
}
|
||||
}
|
||||
return result
|
||||
}
|
||||
|
||||
// preMatch is upstream's PreMatch body, unchanged; only the name moved, so that
|
||||
// the funnel above owns the exported entry point. An upstream change to the
|
||||
// pre-match walk applies to THIS function.
|
||||
func (r *Router) preMatch(metadata adapter.InboundContext, firstPacket []byte) adapter.PreMatchResult {
|
||||
// lx:end l3-honest-drop
|
||||
ctx := log.ContextWithNewID(r.ctx)
|
||||
metadata.PreMatch = true
|
||||
continueResult := adapter.PreMatchResult{Action: adapter.PreMatchContinue}
|
||||
@@ -440,6 +481,11 @@ func applyRouteOptionsOverride(metadata *adapter.InboundContext, routeOptions *R
|
||||
|
||||
func (r *Router) preMatchFlow(ctx context.Context, metadata *adapter.InboundContext, packetDestination M.Socksaddr, matchedRule adapter.Rule, outboundTag string) adapter.PreMatchResult {
|
||||
continueResult := adapter.PreMatchResult{Action: adapter.PreMatchContinue}
|
||||
// lx: ICMP does NOT get a local override here any more — the honest drop is
|
||||
// applied once, to the single return value of PreMatch (see the funnel
|
||||
// there, marker l3-honest-drop). Overriding continueResult in this function
|
||||
// covered only the exits that reach it and left the walk's own exits
|
||||
// forging.
|
||||
var outbound adapter.Outbound
|
||||
if outboundTag == "" {
|
||||
outbound = r.outbound.Default()
|
||||
|
||||
+539
-20
@@ -22,11 +22,34 @@
|
||||
# 2. It runs on linux. shater/generate has 44 test files on linux against 32 on
|
||||
# windows/darwin; the linux-only half is where the routing, ruleset, DNS and
|
||||
# health tests live.
|
||||
# 3. Nothing is skipped SILENTLY. Two machine checks:
|
||||
# 3. Nothing is skipped SILENTLY. Six machine checks:
|
||||
# - the tag set may only ADD test files, never hide them (a test behind
|
||||
# `//go:build !with_awg` would vanish from the gate — this fails first);
|
||||
# - every package that has tests must report `ok` by name; a suite that
|
||||
# compiles down to "no test files" fails the gate instead of passing it.
|
||||
# compiles down to "no test files" fails the gate instead of passing it;
|
||||
# - every ORDINARY test that calls t.Skip is named in the output and must
|
||||
# be DECLARED in SKIP_DECLARED below with the reason it cannot run here;
|
||||
# an undeclared skip fails the gate. This is why the suites run with -v:
|
||||
# without it a skipped test prints nothing whatsoever and the package
|
||||
# still reports `ok`. It was not a hypothetical — shater/apply's
|
||||
# TestApplyInstallsHoldWhenEngineFailsToStart, the W5 regression for
|
||||
# "the engine died, the LAN must not be left open", guarded itself with
|
||||
# a t.Skip whose condition had become permanently true, so it asserted
|
||||
# nothing at all while the gate reported `ok shater/apply`;
|
||||
# - every ^TestIntegration under the fork's trees must produce a verdict
|
||||
# BY NAME ([5/7]). `ok <pkg>` is printed whether the privileged tests in
|
||||
# that package ran or called t.Skip, so the second check cannot see them
|
||||
# — and the gate would keep saying "passes every test we own" while the
|
||||
# tests that need a real kernel never executed;
|
||||
# - every NON-GO test file in the tree must be claimed by a named runner
|
||||
# ([6/7]). The four checks above are all built on `go list`/`go test`, so
|
||||
# a test in another language is invisible to them BY CONSTRUCTION — and
|
||||
# that is not hypothetical either: openwrt/luci-app-shater/tests/
|
||||
# status-readout.test.js, 24 assertions over the one screen an operator
|
||||
# reaches while the LAN is cut off, was run by nothing at all;
|
||||
# - the non-Go suites this gate owns produce a verdict BY NAME ([7/7]),
|
||||
# including "did not run: no node here", which then replaces the closing
|
||||
# banner.
|
||||
# A guard that silently runs nothing is worse than no guard (same rule as
|
||||
# scripts/check-router-tags.sh).
|
||||
#
|
||||
@@ -38,6 +61,10 @@
|
||||
# SHATER_GO_IMAGE docker image used to reach linux from a non-linux host
|
||||
# (default golang:1.26 — keep it >= go.mod's toolchain).
|
||||
# SHATER_NO_DOCKER=1 fail instead of falling back to docker.
|
||||
# SHATER_REQUIRE_PRIVILEGED=1
|
||||
# turn [5/7]'s "did not run here" report into a hard
|
||||
# failure. Use it on the OpenWrt VM or in any pre-release
|
||||
# run that must actually have exercised the kernel paths.
|
||||
set -euo pipefail
|
||||
|
||||
SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
|
||||
@@ -48,7 +75,7 @@ RACE=1
|
||||
for a in "$@"; do
|
||||
case "$a" in
|
||||
--no-race) RACE=0 ;;
|
||||
-h|--help) sed -n '2,41p' "$0" | sed 's/^# \{0,1\}//'; exit 0 ;;
|
||||
-h|--help) sed -n '2,67p' "$0" | sed 's/^# \{0,1\}//'; exit 0 ;;
|
||||
*) echo "run-tests: unknown flag: $a" >&2; exit 2 ;;
|
||||
esac
|
||||
done
|
||||
@@ -72,6 +99,12 @@ ROOTS_COMMON=(./common/...)
|
||||
# rest of common/ can be a real gate instead of a permanently red one. (On
|
||||
# linux every tlsspoof test is a TestIntegration*, so that package is
|
||||
# effectively uncovered here; it is covered by the VM runs.)
|
||||
# The ^TestIntegration prefix is the fork-wide marker for "needs capabilities
|
||||
# the ordinary gate lacks", and [5/7] below leans on the same convention to
|
||||
# catch privileged tests inside ROOTS, which are NOT name-filtered and would
|
||||
# otherwise skip behind a green `ok <pkg>`. ROOTS_COMMON stays out of [5/7]:
|
||||
# these fail rather than skip without the capability, and that is a decision
|
||||
# about upstream code, not about the fork's own coverage.
|
||||
SKIP_COMMON='^TestIntegration'
|
||||
|
||||
# SKIP, WITH REASON: the first -race run over this tree (2026-07-26 — nobody
|
||||
@@ -90,6 +123,124 @@ SKIP_COMMON='^TestIntegration'
|
||||
# the race is fixed.
|
||||
RACE_SKIP='^$'
|
||||
|
||||
# --- the ONLY skips this gate accepts ----------------------------------------
|
||||
# A t.Skip is invisible to every other check here: the test binary exits 0, the
|
||||
# package prints `ok <pkg>`, and the name of the test that did not run appears
|
||||
# NOWHERE unless -v is on. That is how shater/apply's W5 regression —
|
||||
# TestApplyInstallsHoldWhenEngineFailsToStart, the only END-TO-END test between
|
||||
# "the engine died" and "the LAN forwards to the WAN in the clear" — came to
|
||||
# assert nothing at all: it broke the engine by pointing a rule-set at
|
||||
# /nonexistent/nope.srs and stood itself down with t.Skip when that failed to
|
||||
# break anything, and it stopped breaking anything once LocalRuleSet.reloadFile
|
||||
# began treating an unreadable file as an empty one. Measured 2026-07-26 in
|
||||
# golang:1.26: the skip fired unconditionally, and the package still printed
|
||||
# `ok shater/apply`.
|
||||
#
|
||||
# So the suites below run with -v and every `--- SKIP` is matched against this
|
||||
# list. A skip that is not here fails the gate BY NAME. Skips that genuinely
|
||||
# cannot run in some environment are not forbidden — they are DECLARED, with the
|
||||
# reason, and printed on every run so nobody mistakes the gate's silence for
|
||||
# coverage.
|
||||
#
|
||||
# Format: '<extended regexp matched against the full test name>|<reason>'.
|
||||
# The reason is shown verbatim next to the test on every run; write it for
|
||||
# someone deciding whether the gate proved what they think it proved.
|
||||
SKIP_DECLARED=(
|
||||
'^TestIntegration|privileged: needs root + CAP_NET_ADMIN + /dev/net/tun. NOT waved through — [5/7] below gives every one of these a verdict by name, and an unrunnable one REPLACES the closing banner so this run cannot claim it covered them.'
|
||||
'^TestCompiledTagsMatchTheShippedSet$|compares the tags COMPILED INTO a binary with router-tags.sh, and needs the harness that builds that binary and sets SHATER_ROUTER_TAG_CHECK=1. The harness is scripts/check-router-tags.sh, which the release tract runs separately; here there is no such binary to read.'
|
||||
)
|
||||
|
||||
# --- every NON-GO test file must be claimed by a runner ----------------------
|
||||
# The Go half of this gate cannot see a test written in another language, and the
|
||||
# review of 2026-07-26 found what that costs: openwrt/luci-app-shater/tests/
|
||||
# status-readout.test.js — 236 lines, six recorded fixtures, 24 assertions, the
|
||||
# only thing checking what the LuCI dashboard tells an operator while the engine
|
||||
# is down — was executed by NOTHING. Not by panel/package.json's `test` script
|
||||
# (`node --test src/*.test.ts`, panel/src only), not by scripts/run-panel-tests.sh
|
||||
# (same glob), not by any step here. And [1/7] could not report it, because
|
||||
# `go list` is the instrument and a .js file is invisible to it BY CONSTRUCTION.
|
||||
#
|
||||
# So [6/7] enumerates the tree's non-Go test files and requires each to be claimed
|
||||
# by a runner named HERE. A new .test.js/.test.ts/.spec.*/test_*.py that no runner
|
||||
# picks up fails the gate by name on the day it is committed, instead of sitting
|
||||
# there looking like coverage.
|
||||
#
|
||||
# Format: '<extended regexp matched against the repo-relative path>|<runner>'.
|
||||
# Positive and CLOSED on purpose: a file that matches nothing is a failure, not a
|
||||
# default. The Go files are deliberately NOT in scope — `_test.go` under the
|
||||
# declared ROOTS is what [1/7]+[2/7] already prove ran, and pulling the rest of
|
||||
# the upstream tree in here would be a different decision.
|
||||
NONGO_TEST_RUNNERS=(
|
||||
'^panel/src/[^/]+\.test\.ts$|scripts/run-panel-tests.sh (node --test via panel/package.json); CI runs it as its own step on node 24, before this script'
|
||||
'^openwrt/luci-app-shater/tests/[^/]+\.test\.js$|[7/7] of this script'
|
||||
)
|
||||
|
||||
# The non-Go suites [7/7] RUNS, as `node <file>`. panel/src is not here: it has its
|
||||
# own script with its own npm install, and duplicating it would mean two places to
|
||||
# keep right. Each entry is a glob; a glob that matches nothing is a failure (a
|
||||
# renamed file must be reported as that, not as a fast green step).
|
||||
JS_SUITES=('openwrt/luci-app-shater/tests/*.test.js')
|
||||
|
||||
# JS_UNVERIFIED collects the suites that did NOT run, by path, for the closing
|
||||
# banner — the same treatment PRIV_UNVERIFIED gets, and for the same reason.
|
||||
JS_UNVERIFIED=""
|
||||
|
||||
# js_step runs the declared non-Go suites and prints a verdict for each BY NAME.
|
||||
# Defined up here because it is called from TWO places: the non-linux re-exec
|
||||
# below runs it on the HOST, where node usually exists, rather than let the
|
||||
# golang image (which has none) report "did not run" on every single local run —
|
||||
# a banner that always fires is a banner nobody reads.
|
||||
#
|
||||
# `node <file>`: these are standalone harnesses that exit non-zero on a failed
|
||||
# assertion, not `node --test` modules.
|
||||
#
|
||||
# KNOWN LIMIT, stated rather than papered over: the verdict is the process exit
|
||||
# code. A harness gutted of its assertions that still exits 0 reads as a pass —
|
||||
# the same limit scripts/run-panel-tests.sh already names for `node --test`.
|
||||
# Deletion, rename, a throw and a failed assertion are all caught.
|
||||
#
|
||||
# Returns 1 if a suite failed or the globs matched nothing. Sets JS_UNVERIFIED
|
||||
# when there is no node to run them with.
|
||||
js_step() { # $1 = where we are, in words, for the "no node" line
|
||||
local where="$1" f g rc bad=0 tmp
|
||||
local files=()
|
||||
shopt -s nullglob
|
||||
for g in "${JS_SUITES[@]}"; do
|
||||
files+=($g)
|
||||
done
|
||||
shopt -u nullglob
|
||||
if [ "${#files[@]}" -eq 0 ]; then
|
||||
echo " FAILED [js]: JS_SUITES matched no file at all. Either the glob is wrong or" >&2
|
||||
echo " the suite was renamed/deleted — both must be said, not passed over." >&2
|
||||
return 1
|
||||
fi
|
||||
if ! command -v node >/dev/null 2>&1; then
|
||||
echo " node: NOT AVAILABLE $where — these suites did NOT run:"
|
||||
for f in "${files[@]}"; do
|
||||
echo " DID NOT RUN $f"
|
||||
JS_UNVERIFIED="$JS_UNVERIFIED $f"
|
||||
done
|
||||
return 0
|
||||
fi
|
||||
echo " node: $(node --version) $where, ${#files[@]} suite(s)"
|
||||
tmp="$(mktemp)"
|
||||
for f in "${files[@]}"; do
|
||||
set +e
|
||||
node "$f" >"$tmp" 2>&1
|
||||
rc=$?
|
||||
set -e
|
||||
if [ "$rc" -eq 0 ]; then
|
||||
echo " RAN $f"
|
||||
else
|
||||
echo " FAILED $f (exit $rc)" >&2
|
||||
sed 's/^/ | /' "$tmp" >&2
|
||||
bad=1
|
||||
fi
|
||||
done
|
||||
rm -f "$tmp"
|
||||
return "$bad"
|
||||
}
|
||||
|
||||
echo "== shater test gate =="
|
||||
echo " tags : $SHATER_ROUTER_TAGS"
|
||||
echo " ldflags: $SHATER_ROUTER_LDFLAGS"
|
||||
@@ -111,14 +262,67 @@ if [ "$(go env GOOS)" != "linux" ] && [ "${SHATER_TESTS_IN_DOCKER:-0}" != "1" ];
|
||||
echo "== re-exec on linux via docker ($image) =="
|
||||
host_repo="$REPO"
|
||||
command -v cygpath >/dev/null 2>&1 && host_repo="$(cygpath -w "$REPO")"
|
||||
# Hand the container CAP_NET_ADMIN and /dev/net/tun when this host's docker
|
||||
# can. shater/generate's ^TestIntegration tests open a real TUN and stand a
|
||||
# real engine on it; without the device they skip, and a dev running the gate
|
||||
# by hand would get a green result that never touched the kernel path the
|
||||
# branch is about. The dev host CAN give them (Docker Desktop's VM has the tun
|
||||
# module) — the CI runner cannot, which is what [5/7] exists to say out loud.
|
||||
# PROBED, never assumed: a docker whose kernel lacks tun refuses --device and
|
||||
# would take the whole gate down with it.
|
||||
priv_flags=()
|
||||
if MSYS2_ARG_CONV_EXCL='*' MSYS_NO_PATHCONV=1 docker run --rm \
|
||||
--cap-add NET_ADMIN --device /dev/net/tun "$image" true >/dev/null 2>&1; then
|
||||
priv_flags=(--cap-add NET_ADMIN --device /dev/net/tun)
|
||||
echo " CAP_NET_ADMIN + /dev/net/tun: available — the privileged tests will really run"
|
||||
else
|
||||
echo " CAP_NET_ADMIN + /dev/net/tun: NOT available from this docker — [5/7] will report the gap"
|
||||
fi
|
||||
# The non-Go suites do not need linux, and this host very likely has node while
|
||||
# the golang image certainly does not. Run them HERE, so the local loop really
|
||||
# executes them instead of being told every single time that it did not: a
|
||||
# banner that always fires is a banner nobody reads, and that is how a report
|
||||
# stops being a report. Their verdict is folded into this script's exit status
|
||||
# below, and the container is told not to repeat them.
|
||||
host_js_rc=0
|
||||
js_flags=()
|
||||
if command -v node >/dev/null 2>&1; then
|
||||
echo "== [7/7] the non-Go suites (node), run on this host before the re-exec =="
|
||||
js_step "on this host ($image has none)" || host_js_rc=1
|
||||
js_flags=(-e SHATER_JS_ALREADY_RAN=1)
|
||||
echo
|
||||
fi
|
||||
MSYS2_ARG_CONV_EXCL='*' MSYS_NO_PATHCONV=1 docker run --rm \
|
||||
"${priv_flags[@]+"${priv_flags[@]}"}" \
|
||||
"${js_flags[@]+"${js_flags[@]}"}" \
|
||||
-v "$host_repo":/src \
|
||||
-v shater-tagcheck-gomod:/go/pkg/mod \
|
||||
-v shater-tagcheck-gocache:/root/.cache/go-build \
|
||||
-w /src \
|
||||
-e SHATER_TESTS_IN_DOCKER=1 \
|
||||
"$image" bash scripts/run-tests.sh "$@"
|
||||
exit $?
|
||||
-e SHATER_REQUIRE_PRIVILEGED="${SHATER_REQUIRE_PRIVILEGED:-0}" \
|
||||
"$image" bash -c '
|
||||
# netplane.L3SlotFor asks the kernel through `ip link show` and reclaims a
|
||||
# stale slot through `ip link del`. Without iproute2 EVERY slot reads as
|
||||
# free, so TestIntegrationL3StaleSlotIsReclaimed refuses to run rather than
|
||||
# pass while proving the opposite of what it claims — and [5/7] then fails
|
||||
# the whole gate, correctly. golang:1.26 ships no iproute2, so install it
|
||||
# here rather than let the image quietly narrow what this gate can verify.
|
||||
# On a Linux host the script never re-execs, and the router has ip-full as
|
||||
# a hard dependency, so this is the docker path only.
|
||||
if ! command -v ip >/dev/null 2>&1; then
|
||||
echo " iproute2: absent from '"$image"' — installing (the slot-reclaim test needs it)"
|
||||
apt-get update -qq >/dev/null 2>&1 && apt-get install -y -qq iproute2 >/dev/null 2>&1 \
|
||||
|| echo " iproute2: INSTALL FAILED — [5/7] will report the gap by name"
|
||||
fi
|
||||
exec bash scripts/run-tests.sh "$@"
|
||||
' _ "$@"
|
||||
docker_rc=$?
|
||||
if [ "$host_js_rc" -ne 0 ]; then
|
||||
echo "== TEST GATE FAILED — a non-Go suite failed on the host (see [7/7] above). ==" >&2
|
||||
exit 1
|
||||
fi
|
||||
exit "$docker_rc"
|
||||
fi
|
||||
|
||||
ALL_ROOTS=("${ROOTS[@]}" "${ROOTS_COMMON[@]}")
|
||||
@@ -128,7 +332,7 @@ ALL_ROOTS=("${ROOTS[@]}" "${ROOTS_COMMON[@]}")
|
||||
# shipped tags REMOVES a test file from any package, that test exists but the
|
||||
# gate would never see it — which is the failure mode this whole script is about,
|
||||
# just pointed the other way.
|
||||
echo "== [1/4] no test file is hidden by the shipped tag set =="
|
||||
echo "== [1/7] no test file is hidden by the shipped tag set =="
|
||||
LISTFMT='{{.ImportPath}} {{len .TestGoFiles}} {{len .XTestGoFiles}}'
|
||||
plain="$(go list -f "$LISTFMT" "${ALL_ROOTS[@]}")"
|
||||
tagged="$(go list -tags "$SHATER_ROUTER_TAGS" -f "$LISTFMT" "${ALL_ROOTS[@]}")"
|
||||
@@ -162,14 +366,67 @@ fi
|
||||
echo
|
||||
|
||||
# --- the runner --------------------------------------------------------------
|
||||
# Runs one suite and then PROVES it ran: every package `go list` says has tests
|
||||
# must appear as `ok <pkg>` in the output. `go test` over a package whose tests
|
||||
# all vanished behind a build constraint prints "[no test files]" and exits 0 —
|
||||
# a green run that verified nothing.
|
||||
# Runs one suite and then PROVES it ran, at BOTH granularities:
|
||||
# - package: every package `go list` says has tests must appear as `ok <pkg>`.
|
||||
# `go test` over a package whose tests all vanished behind a build constraint
|
||||
# prints "[no test files]" and exits 0 — a green run that verified nothing.
|
||||
# - test: every `--- SKIP` must be declared in SKIP_DECLARED (check_skips).
|
||||
# `ok <pkg>` is printed whether the tests inside ran or stood themselves down.
|
||||
LOG="$(mktemp)"
|
||||
trap 'rm -f "$LOG"' EXIT
|
||||
FAILED=0
|
||||
|
||||
# check_skips reads the per-test verdicts of the suite in $LOG and refuses to let
|
||||
# a t.Skip through unnamed. Returns non-zero on an undeclared skip.
|
||||
check_skips() { # $1=label ; reads $LOG
|
||||
local label="$1" name reason matched entry re bad=0
|
||||
|
||||
# THE CONTROL, and it comes first on purpose. Everything below reads `--- SKIP`
|
||||
# lines, which exist only under `go test -v`. Drop the -v and this function
|
||||
# reports a clean bill of health over a suite that skipped every test it had —
|
||||
# a check against silent skipping that is itself silently skipping, which is
|
||||
# exactly how the first cut of [5/7] shipped (`go test -list` failing to link,
|
||||
# swallowed by `|| true`, reporting "none declared"). `=== RUN` is printed for
|
||||
# every test the binary starts, so its absence means the verdicts are not being
|
||||
# produced at all and this instrument is reading a blank page.
|
||||
if ! grep -q '^=== RUN ' "$LOG"; then
|
||||
echo " FAILED [$label]: not one '=== RUN' line in the output — per-test verdicts are" >&2
|
||||
echo " not being produced (is -v still on?), so the skip check was reading a" >&2
|
||||
echo " blank page and its silence means nothing." >&2
|
||||
return 1
|
||||
fi
|
||||
|
||||
while read -r name; do
|
||||
[ -n "$name" ] || continue
|
||||
matched=""
|
||||
for entry in "${SKIP_DECLARED[@]}"; do
|
||||
re="${entry%%|*}"
|
||||
reason="${entry#*|}"
|
||||
if grep -qE "$re" <<<"$name"; then
|
||||
matched="$reason"
|
||||
break
|
||||
fi
|
||||
done
|
||||
if [ -n "$matched" ]; then
|
||||
echo " DECLARED SKIP $name"
|
||||
echo " -> $matched"
|
||||
else
|
||||
echo " UNDECLARED SKIP $name" >&2
|
||||
bad=1
|
||||
fi
|
||||
done < <(sed -n 's/^[[:space:]]*--- SKIP: \([^[:space:]]*\).*/\1/p' "$LOG" | sort -u)
|
||||
|
||||
if [ "$bad" -ne 0 ]; then
|
||||
echo " FAILED [$label]: the test(s) above called t.Skip and are not declared in" >&2
|
||||
echo " SKIP_DECLARED at the top of this script. A skipped test is a test that" >&2
|
||||
echo " DID NOT RUN, and the package's 'ok' line says nothing about it. Either" >&2
|
||||
echo " make it run here, or declare it by name with the reason it cannot —" >&2
|
||||
echo " the reason is printed on every run, so it has to hold up." >&2
|
||||
return 1
|
||||
fi
|
||||
return 0
|
||||
}
|
||||
|
||||
run_suite() { # $1=label $2=extra go-test flags (may be empty) $3..=packages
|
||||
local label="$1" extra="$2"
|
||||
shift 2
|
||||
@@ -187,17 +444,41 @@ run_suite() { # $1=label $2=extra go-test flags (may be empty) $3..=packages
|
||||
echo " packages with tests: $(wc -l <<<"$expect" | tr -d ' ')"
|
||||
|
||||
set +e
|
||||
# -v is NOT optional: it is the only way a t.Skip becomes visible at all (see
|
||||
# check_skips). It costs no test TIME — measured 2026-07-26 over the fork's
|
||||
# trees, warm cache, three alternating runs each: 38/25/24 s plain against
|
||||
# 38/24/24 s with -v. What it costs is OUTPUT: 5 KB -> 257 KB, which is why the
|
||||
# printing below is filtered rather than the flag dropped.
|
||||
# shellcheck disable=SC2086 # $extra is a deliberate word-split flag list
|
||||
go test -count=1 $extra \
|
||||
go test -count=1 -v $extra \
|
||||
-tags "$SHATER_ROUTER_TAGS" -ldflags "$SHATER_ROUTER_LDFLAGS" \
|
||||
"${pkgs[@]}" >"$LOG" 2>&1
|
||||
rc=$?
|
||||
set -e
|
||||
sed 's/^/ /' "$LOG"
|
||||
if [ "$rc" -ne 0 ]; then
|
||||
# A failure needs the whole story, t.Logf output and all.
|
||||
sed 's/^/ /' "$LOG"
|
||||
else
|
||||
# A green run gets what the non-verbose gate always printed — one line per
|
||||
# package — plus every skip verdict. The rest of -v's output is a
|
||||
# `=== RUN`/`--- PASS` pair per test (250 KB a suite); printing it would bury
|
||||
# the handful of lines anyone reads.
|
||||
grep -E '^(ok|FAIL|\?)[[:space:]]|^[[:space:]]*--- SKIP: ' "$LOG" | sed 's/^/ /' || true
|
||||
fi
|
||||
|
||||
if [ "$rc" -ne 0 ]; then
|
||||
echo " FAILED [$label]: go test exited $rc" >&2
|
||||
FAILED=1
|
||||
# Name the skips anyway. A suite that failed somewhere else must not become
|
||||
# a hiding place for a test that did not run — that is the same sin one
|
||||
# level down, and while a red tree is being fixed is exactly when a skip
|
||||
# gets added "temporarily". The verdict is already FAILED, so this only
|
||||
# reports. Guarded on tests having actually run: a BUILD failure produces no
|
||||
# verdicts to read, and check_skips' own control would then fire and bury
|
||||
# the compiler error under a complaint about -v.
|
||||
if grep -q '^=== RUN ' "$LOG"; then
|
||||
check_skips "$label" || true
|
||||
fi
|
||||
return
|
||||
fi
|
||||
|
||||
@@ -214,31 +495,233 @@ run_suite() { # $1=label $2=extra go-test flags (may be empty) $3..=packages
|
||||
FAILED=1
|
||||
return
|
||||
fi
|
||||
if ! check_skips "$label"; then
|
||||
FAILED=1
|
||||
return
|
||||
fi
|
||||
echo " OK [$label]"
|
||||
}
|
||||
|
||||
# --- [2/4] the fork's trees, shipped tags, linux -----------------------------
|
||||
echo "== [2/4] go test — the fork's trees (shipped tags, linux) =="
|
||||
# --- [2/7] the fork's trees, shipped tags, linux -----------------------------
|
||||
echo "== [2/7] go test — the fork's trees (shipped tags, linux) =="
|
||||
run_suite main "" "${ROOTS[@]}"
|
||||
echo
|
||||
|
||||
# --- [3/4] common/, minus the tests that need CAP_NET_ADMIN ------------------
|
||||
echo "== [3/4] go test — common/ (minus the CAP_NET_ADMIN integration tests) =="
|
||||
# --- [3/7] common/, minus the tests that need CAP_NET_ADMIN ------------------
|
||||
echo "== [3/7] go test — common/ (minus the CAP_NET_ADMIN integration tests) =="
|
||||
run_suite common "-skip $SKIP_COMMON" "${ROOTS_COMMON[@]}"
|
||||
echo
|
||||
|
||||
# --- [4/4] -race over the same trees -----------------------------------------
|
||||
# --- [4/7] -race over the same trees -----------------------------------------
|
||||
# Everything, not a subset: shater/netplane alone is ~110 s under -race and it is
|
||||
# the single most concurrency-critical package we own (the nft data plane), so
|
||||
# once it is in, adding the rest costs ~40 s more. common/ is left out — it is
|
||||
# upstream code exercised by upstream CI.
|
||||
if [ "$RACE" -eq 1 ]; then
|
||||
echo "== [4/4] go test -race — the fork's trees =="
|
||||
echo "== [4/7] go test -race — the fork's trees =="
|
||||
echo " nothing is skipped under -race"
|
||||
|
||||
|
||||
run_suite race "-race -skip $RACE_SKIP" "${ROOTS[@]}"
|
||||
else
|
||||
echo "== [4/4] -race pass skipped (--no-race) =="
|
||||
echo "== [4/7] -race pass skipped (--no-race) =="
|
||||
fi
|
||||
echo
|
||||
|
||||
# --- [5/7] the privileged tests may not skip in silence ----------------------
|
||||
# THE HOLE THIS CLOSES. Some tests can only prove what they claim against a real
|
||||
# kernel: shater/generate's TestIntegrationL3TunInboundStarts opens /dev/net/tun
|
||||
# and stands a real engine on it, TestIntegrationL3EgressICMPIsAFlow binds a real
|
||||
# socket to a real device. Both guard themselves with t.Skip when root or the
|
||||
# device is missing — the honest thing for a test to do, and completely INVISIBLE
|
||||
# above: `go test` prints `ok <pkg>` whether they ran or skipped, so [2/7]'s
|
||||
# per-package `ok` check is satisfied either way and the gate closes by claiming
|
||||
# it "passes every test we own". That is precisely the failure this whole script
|
||||
# was written for (115 of 116 test files never running while CI stayed green),
|
||||
# one level down and harder to see.
|
||||
#
|
||||
# The list is DISCOVERED, not hand-kept — `go test -list` over the same ROOTS —
|
||||
# so a privileged test written next month joins this check on the day it is
|
||||
# named, with no edit here. It keys on the ^TestIntegration prefix, already this
|
||||
# fork's marker for "needs capabilities the ordinary gate lacks" (SKIP_COMMON
|
||||
# above excludes common/tlsspoof's TestIntegration* for exactly that reason).
|
||||
# Name a privileged test anything else and it is invisible again — so don't.
|
||||
#
|
||||
# Verdicts, per test, by name:
|
||||
# RAN — it executed here; printed so that is visible rather than assumed.
|
||||
# FAILED — fatal, like any other failure.
|
||||
# MISSING — `go test -list` named it and the run produced no verdict for it:
|
||||
# fatal. A test that vanished between listing and running is the
|
||||
# same class of hole as one hidden by a build tag.
|
||||
# SKIPPED while this environment HAS root and /dev/net/tun — fatal. The
|
||||
# capability guard cannot be what skipped it, so something else did
|
||||
# and only the test knows what.
|
||||
# SKIPPED because the environment genuinely cannot run it — reported loudly,
|
||||
# by name, and it REPLACES the closing banner, so the last line of
|
||||
# the gate can never claim coverage it does not have. Deliberately
|
||||
# not fatal by default: the act_runner is an LXC guest whose kernel
|
||||
# has no tun module at all (checked 2026-07-26 on 10.10.10.211 —
|
||||
# `modprobe tun` answers "Module tun not found", /dev/net does not
|
||||
# exist, and act_runner runs job containers with privileged:false
|
||||
# and no container.options), so the device cannot be handed down
|
||||
# without reconfiguring the Proxmox host. Making it fatal would
|
||||
# paint CI permanently red and teach everyone to ignore the gate.
|
||||
# SHATER_REQUIRE_PRIVILEGED=1 makes it fatal for the runs that can.
|
||||
echo "== [5/7] the privileged tests (^TestIntegration) produced a verdict by name =="
|
||||
PRIV_RE='^TestIntegration'
|
||||
PRIV_UNVERIFIED=""
|
||||
# -ldflags is NOT optional on the discovery call either: `go test -list` LINKS
|
||||
# each test binary before it can enumerate its tests, and without
|
||||
# -checklinkname=0 every package that pulls common/badtls fails to link. The
|
||||
# first cut of this step omitted it, swallowed the error with `2>/dev/null ||
|
||||
# true`, and reported "none declared" — a check against silent skipping that was
|
||||
# itself silently skipping. Hence also: the exit status is inspected, and an
|
||||
# empty list is only ever reported after a SUCCESSFUL enumeration.
|
||||
set +e
|
||||
priv_expect_raw="$(go test -list "$PRIV_RE" \
|
||||
-tags "$SHATER_ROUTER_TAGS" -ldflags "$SHATER_ROUTER_LDFLAGS" "${ROOTS[@]}" 2>&1)"
|
||||
priv_list_rc=$?
|
||||
set -e
|
||||
priv_expect="$(grep -E "$PRIV_RE" <<<"$priv_expect_raw" | sort -u || true)"
|
||||
if [ "$priv_list_rc" -ne 0 ]; then
|
||||
echo " FAILED [privileged]: could not enumerate the privileged tests (go test -list exited $priv_list_rc)." >&2
|
||||
echo " An unreadable list is NOT an empty list — this check refuses to" >&2
|
||||
echo " report 'nothing to verify' on the strength of a failed command." >&2
|
||||
sed 's/^/ /' <<<"$priv_expect_raw" | grep -vE '^\s+(ok|\?)\s' >&2 || true
|
||||
FAILED=1
|
||||
elif [ -z "$priv_expect" ]; then
|
||||
echo " none declared under the fork's trees — nothing to verify"
|
||||
else
|
||||
priv_capable=0
|
||||
if [ "$(id -u)" = "0" ] && [ -e /dev/net/tun ]; then
|
||||
priv_capable=1
|
||||
fi
|
||||
echo " declared: $(wc -l <<<"$priv_expect" | tr -d ' ')"
|
||||
echo " this environment: uid=$(id -u), /dev/net/tun $([ -e /dev/net/tun ] && echo present || echo MISSING) => can run them: $([ "$priv_capable" -eq 1 ] && echo yes || echo NO)"
|
||||
set +e
|
||||
go test -count=1 -v -run "$PRIV_RE" \
|
||||
-tags "$SHATER_ROUTER_TAGS" -ldflags "$SHATER_ROUTER_LDFLAGS" \
|
||||
"${ROOTS[@]}" >"$LOG" 2>&1
|
||||
priv_rc=$?
|
||||
set -e
|
||||
# The verdict lines plus whatever reason the test printed just before them,
|
||||
# so a skip is readable here and not just counted.
|
||||
grep -E '^(--- (PASS|SKIP|FAIL): |[[:space:]]+[^[:space:]]+\.go:[0-9]+: )' "$LOG" \
|
||||
| sed 's/^/ | /' || true
|
||||
priv_bad=0
|
||||
while read -r name; do
|
||||
[ -n "$name" ] || continue
|
||||
if grep -qE "^--- PASS: ${name}([[:space:]]|\$)" "$LOG"; then
|
||||
echo " RAN $name"
|
||||
elif grep -qE "^--- FAIL: ${name}([[:space:]]|\$)" "$LOG"; then
|
||||
echo " FAILED $name" >&2
|
||||
priv_bad=1
|
||||
elif grep -qE "^--- SKIP: ${name}([[:space:]]|\$)" "$LOG"; then
|
||||
if [ "$priv_capable" -eq 1 ]; then
|
||||
echo " SKIPPED $name — but this environment HAS root and /dev/net/tun, so the capability guard is NOT what skipped it" >&2
|
||||
priv_bad=1
|
||||
else
|
||||
echo " DID NOT RUN $name — skipped: no root and/or no /dev/net/tun here"
|
||||
PRIV_UNVERIFIED="$PRIV_UNVERIFIED $name"
|
||||
fi
|
||||
else
|
||||
echo " MISSING $name — go test -list named it, the run produced no verdict for it" >&2
|
||||
priv_bad=1
|
||||
fi
|
||||
done <<<"$priv_expect"
|
||||
if [ "$priv_rc" -ne 0 ] && [ "$priv_bad" -eq 0 ]; then
|
||||
echo " FAILED [privileged]: go test exited $priv_rc with every named test accounted for —" >&2
|
||||
echo " a build or package-level failure, see the log above." >&2
|
||||
priv_bad=1
|
||||
fi
|
||||
if [ "$priv_bad" -ne 0 ]; then
|
||||
FAILED=1
|
||||
fi
|
||||
fi
|
||||
echo
|
||||
|
||||
# --- [6/7] every non-Go test file is claimed by a runner ---------------------
|
||||
# See NONGO_TEST_RUNNERS above for why this exists. The instrument is `git
|
||||
# ls-files`, not a filesystem walk: a test file that is not committed is not
|
||||
# anybody's coverage, and a walk would also drag node_modules in.
|
||||
#
|
||||
# The candidate pattern is a CLOSED positive list of the shapes a test file takes
|
||||
# in this repo and the ones it plausibly will (js/ts/jsx/tsx, python, bats, shell,
|
||||
# ucode). It is deliberately wider than what exists today: the whole point is to
|
||||
# catch the file somebody adds next month in a language no step here knows about.
|
||||
echo "== [6/7] every non-Go test file is claimed by a runner =="
|
||||
NONGO_TEST_RE='(\.(test|spec)\.(js|mjs|cjs|jsx|ts|tsx)|(^|/)test_[^/]*\.py|_test\.py|\.bats|_test\.sh|_test\.uc)$'
|
||||
set +e
|
||||
tracked="$(git ls-files 2>&1)"
|
||||
tracked_rc=$?
|
||||
set -e
|
||||
if [ "$tracked_rc" -ne 0 ]; then
|
||||
echo " FAILED [claimed]: could not enumerate the tree (git ls-files exited $tracked_rc)." >&2
|
||||
echo " An unreadable list is NOT an empty list — same rule as [5/7]." >&2
|
||||
sed 's/^/ /' <<<"$tracked" >&2
|
||||
FAILED=1
|
||||
else
|
||||
nongo="$(grep -E "$NONGO_TEST_RE" <<<"$tracked" | sort || true)"
|
||||
if [ -z "$nongo" ]; then
|
||||
echo " FAILED [claimed]: not one non-Go test file found in a tree that has several." >&2
|
||||
echo " The pattern stopped matching; this check would pass having looked" >&2
|
||||
echo " at nothing." >&2
|
||||
FAILED=1
|
||||
else
|
||||
echo " non-Go test files: $(wc -l <<<"$nongo" | tr -d ' ')"
|
||||
unclaimed=0
|
||||
while read -r f; do
|
||||
[ -n "$f" ] || continue
|
||||
owner=""
|
||||
for entry in "${NONGO_TEST_RUNNERS[@]}"; do
|
||||
if grep -qE "${entry%%|*}" <<<"$f"; then
|
||||
owner="${entry#*|}"
|
||||
break
|
||||
fi
|
||||
done
|
||||
if [ -n "$owner" ]; then
|
||||
echo " claimed $f"
|
||||
echo " -> $owner"
|
||||
else
|
||||
echo " UNCLAIMED $f — no runner in this gate executes it" >&2
|
||||
unclaimed=1
|
||||
fi
|
||||
done <<<"$nongo"
|
||||
if [ "$unclaimed" -ne 0 ]; then
|
||||
echo " FAILED [claimed]: the file(s) above are test files that NOTHING runs." >&2
|
||||
echo " That is a test which cannot fail — the most expensive kind, because" >&2
|
||||
echo " it reads as coverage. Either wire a runner (JS_SUITES below, or" >&2
|
||||
echo " scripts/run-panel-tests.sh) and declare it in NONGO_TEST_RUNNERS, or" >&2
|
||||
echo " delete the file. Declaring it without wiring one is not an option:" >&2
|
||||
echo " the runner named there is the one [7/7] reports a verdict for." >&2
|
||||
FAILED=1
|
||||
fi
|
||||
fi
|
||||
fi
|
||||
echo
|
||||
|
||||
# --- [7/7] the non-Go suites this gate owns, with a verdict by name ----------
|
||||
# The work is in js_step() at the top of this file; see there for what `node
|
||||
# <file>` proves and what it cannot.
|
||||
#
|
||||
# NODE MAY BE ABSENT, and that is handled the way [5/7] handles a missing
|
||||
# /dev/net/tun: the files are named, the run says out loud that they DID NOT RUN,
|
||||
# and that notice REPLACES the closing banner so this script can never end by
|
||||
# claiming coverage it does not have. Not fatal by default, because the
|
||||
# golang:1.26 image the non-linux re-exec uses has no node. CI does: both
|
||||
# .gitea/workflows/test.yml and release.yml run actions/setup-node@v4 (node 24)
|
||||
# and scripts/run-panel-tests.sh BEFORE this script, in the same job, so on the
|
||||
# release path node is on PATH here and these really execute.
|
||||
#
|
||||
# SHATER_JS_ALREADY_RAN=1 means the re-exec that started this container ran them
|
||||
# on the host first and will fold their verdict into its own exit status — so
|
||||
# re-running them here would only be slower and, without node, would print a
|
||||
# "did not run" that is not true of this gate as a whole.
|
||||
echo "== [7/7] the non-Go suites (node) produced a verdict by name =="
|
||||
if [ "${SHATER_JS_ALREADY_RAN:-0}" = "1" ]; then
|
||||
echo " already run on the host before the re-exec into this container (see above);"
|
||||
echo " that run's verdict is folded into the exit status of the script that started it."
|
||||
else
|
||||
js_step "here" || FAILED=1
|
||||
fi
|
||||
echo
|
||||
|
||||
@@ -246,4 +729,40 @@ if [ "$FAILED" -ne 0 ]; then
|
||||
echo "== TEST GATE FAILED — nothing may be published from this run. ==" >&2
|
||||
exit 1
|
||||
fi
|
||||
if [ -n "$PRIV_UNVERIFIED" ] || [ -n "$JS_UNVERIFIED" ]; then
|
||||
echo "== !! PASSED, BUT NOT FULLY VERIFIED !! =================================="
|
||||
echo " Every test that COULD run here passed. These did not run at all:"
|
||||
for t in $PRIV_UNVERIFIED $JS_UNVERIFIED; do
|
||||
echo " - $t"
|
||||
done
|
||||
echo
|
||||
fi
|
||||
if [ -n "$JS_UNVERIFIED" ]; then
|
||||
echo " The file(s) above with a path are non-Go suites and this environment has"
|
||||
echo " no \`node\`. The golang image the non-linux re-exec uses does not ship one;"
|
||||
echo " CI does (actions/setup-node@v4, node 24, in the same job before this"
|
||||
echo " script), so on the release path they DO run. To run them here:"
|
||||
echo " node openwrt/luci-app-shater/tests/status-readout.test.js"
|
||||
echo
|
||||
fi
|
||||
if [ -n "$PRIV_UNVERIFIED" ]; then
|
||||
echo " The named tests above need root + CAP_NET_ADMIN + /dev/net/tun, which"
|
||||
echo " this environment does not have. Nothing about the kernel paths they"
|
||||
echo " cover was verified by this run. To actually run them, from a host whose"
|
||||
echo " docker can:"
|
||||
echo " scripts/run-tests.sh # the re-exec hands the container both"
|
||||
echo " or directly:"
|
||||
echo " docker run --rm --cap-add NET_ADMIN --device /dev/net/tun \\"
|
||||
echo " -v \"\$PWD\":/src -w /src golang:1.26 bash scripts/run-tests.sh"
|
||||
echo " or on the OpenWrt VM. SHATER_REQUIRE_PRIVILEGED=1 makes this a hard"
|
||||
echo " failure instead of this notice."
|
||||
fi
|
||||
if [ -n "$PRIV_UNVERIFIED" ] || [ -n "$JS_UNVERIFIED" ]; then
|
||||
echo "=========================================================================="
|
||||
if [ -n "$PRIV_UNVERIFIED" ] && [ "${SHATER_REQUIRE_PRIVILEGED:-0}" = "1" ]; then
|
||||
echo "== TEST GATE FAILED: SHATER_REQUIRE_PRIVILEGED=1 and the tests above did not run. ==" >&2
|
||||
exit 1
|
||||
fi
|
||||
exit 0
|
||||
fi
|
||||
echo "== OK: the shipped tag set, on linux, passes every test we own. =="
|
||||
|
||||
@@ -0,0 +1,123 @@
|
||||
#!/bin/sh
|
||||
# scripts/testbed-lao.sh — add a SECOND LAN network ("lao", 10.67.1.0/24) in its
|
||||
# own fw4 zone, on a testbed router.
|
||||
#
|
||||
# WHY IT EXISTS
|
||||
#
|
||||
# Almost every zone-related defect in this project is invisible on a router with
|
||||
# one LAN zone, because "the zone" and "the LAN" are the same thing there. The
|
||||
# divert set the daemon builds spans every LAN inbound and every `iface:`/`zone:`
|
||||
# rule source, so on a multi-zone router traffic from the other zones is marked,
|
||||
# routed, accepted by `inet shater` — and then dropped by fw4's zone policy,
|
||||
# silently. Reproducing that needs a second zone and nothing else: no second
|
||||
# physical port, no client, no traffic. This script makes one.
|
||||
#
|
||||
# WHY IT IS NOT SHIPPED
|
||||
#
|
||||
# It lives in scripts/ and is NOT installed by openwrt/shater-core/Makefile,
|
||||
# which lists every file it installs by name. That is the whole opt-in mechanism,
|
||||
# and it was chosen over the alternatives on purpose:
|
||||
#
|
||||
# - an extra /etc/uci-defaults/ file would run on EVERY install, handing a
|
||||
# second network and a second firewall zone to every ordinary user — the one
|
||||
# thing this must not do;
|
||||
# - an environment variable read inside 30_shater-core is unreachable in
|
||||
# practice: that script is deleted after its first successful run, so there
|
||||
# is no later moment at which an operator could set the variable and re-run it;
|
||||
# - a separate package would need a feed entry, a build, a release and a
|
||||
# version, for a file that exists to be scp'd onto one VM.
|
||||
#
|
||||
# USAGE
|
||||
#
|
||||
# scp scripts/testbed-lao.sh root@testbed:/tmp/ && ssh root@testbed sh /tmp/testbed-lao.sh
|
||||
# ssh root@testbed sh /tmp/testbed-lao.sh --remove
|
||||
#
|
||||
# It is idempotent (every section is NAMED and guarded), purely additive, and
|
||||
# touches no existing section. Running it three times in a row leaves exactly one
|
||||
# of everything.
|
||||
|
||||
set -e
|
||||
|
||||
REMOVE=0
|
||||
[ "$1" = "--remove" ] && REMOVE=1
|
||||
|
||||
if [ "$REMOVE" = 1 ]; then
|
||||
uci -q delete firewall.lao_fwd
|
||||
uci -q delete firewall.lao
|
||||
uci -q delete dhcp.lao
|
||||
uci -q delete network.lao
|
||||
uci -q delete network.br_lao
|
||||
uci -q commit firewall
|
||||
uci -q commit dhcp
|
||||
uci -q commit network
|
||||
/etc/init.d/network reload
|
||||
/etc/init.d/firewall reload
|
||||
echo "lao removed"
|
||||
exit 0
|
||||
fi
|
||||
|
||||
# --- L2: an empty bridge -----------------------------------------------------
|
||||
# No ports on purpose: the point is a second ROUTED network with its own firewall
|
||||
# zone, and giving it a switch port would mean re-cabling a testbed for nothing.
|
||||
#
|
||||
# bridge_empty is what makes a portless bridge usable. Without it netifd leaves a
|
||||
# member-less bridge down (no carrier), the `lao` interface never comes up, fw4
|
||||
# resolves `list network 'lao'` to an EMPTY device set, and the zone silently
|
||||
# matches nothing — which would make this script a worse instrument than no
|
||||
# instrument, since it would look set up and prove nothing.
|
||||
if ! uci -q get network.br_lao >/dev/null; then
|
||||
uci set network.br_lao=device
|
||||
uci set network.br_lao.name='br-lao'
|
||||
uci set network.br_lao.type='bridge'
|
||||
uci set network.br_lao.bridge_empty='1'
|
||||
fi
|
||||
|
||||
# --- L3: the interface -------------------------------------------------------
|
||||
# 10.67.1.0/24 is deliberately far from anything a home LAN or a proxy node uses.
|
||||
# ipaddr+netmask rather than CIDR: CIDR in `ipaddr` is a 25.12 convenience and
|
||||
# this script should also run on an older testbed image.
|
||||
if ! uci -q get network.lao >/dev/null; then
|
||||
uci set network.lao=interface
|
||||
uci set network.lao.proto='static'
|
||||
uci set network.lao.device='br-lao'
|
||||
uci set network.lao.ipaddr='10.67.1.1'
|
||||
uci set network.lao.netmask='255.255.255.0'
|
||||
fi
|
||||
|
||||
# --- DHCP: same shape as lan -------------------------------------------------
|
||||
if ! uci -q get dhcp.lao >/dev/null; then
|
||||
uci set dhcp.lao=dhcp
|
||||
uci set dhcp.lao.interface='lao'
|
||||
uci set dhcp.lao.start='100'
|
||||
uci set dhcp.lao.limit='150'
|
||||
uci set dhcp.lao.leasetime='12h'
|
||||
fi
|
||||
|
||||
# --- Firewall: its OWN zone, which is the entire point -----------------------
|
||||
# Same policies as the stock lan zone and its own forwarding to wan, so the
|
||||
# network behaves like a second LAN. What it does NOT get here is a forwarding
|
||||
# into shater_l3: seeding that for every zone is the job under test, done by
|
||||
# /etc/uci-defaults/30_shater-core. If this script seeded it, the test would be
|
||||
# testing itself.
|
||||
if ! uci -q get firewall.lao >/dev/null; then
|
||||
uci set firewall.lao=zone
|
||||
uci set firewall.lao.name='lao'
|
||||
uci set firewall.lao.input='ACCEPT'
|
||||
uci set firewall.lao.output='ACCEPT'
|
||||
uci set firewall.lao.forward='ACCEPT'
|
||||
uci add_list firewall.lao.network='lao'
|
||||
fi
|
||||
if ! uci -q get firewall.lao_fwd >/dev/null; then
|
||||
uci set firewall.lao_fwd=forwarding
|
||||
uci set firewall.lao_fwd.src='lao'
|
||||
uci set firewall.lao_fwd.dest='wan'
|
||||
fi
|
||||
|
||||
uci commit network
|
||||
uci commit dhcp
|
||||
uci commit firewall
|
||||
|
||||
/etc/init.d/network reload
|
||||
/etc/init.d/firewall reload
|
||||
|
||||
echo "lao seeded: br-lao 10.67.1.1/24, fw4 zone lao -> wan"
|
||||
+569
-33
@@ -61,6 +61,30 @@ type Applier struct {
|
||||
// silently skipped. "" until the first successful nft load.
|
||||
lastNft string
|
||||
|
||||
// lastRouteWarnings is the LAST MEASUREMENT of the policy-routing state — what
|
||||
// ApplyRoutingWithWarnings reported the last time it actually ran. Guarded by
|
||||
// a.mu, like lastNft, and written only from applyLocked.
|
||||
//
|
||||
// It exists because the routing step is skipped on the fast path while the
|
||||
// conditions it reports on go on standing: an egress with no gateway, a routing
|
||||
// table that could not be given its fail-closed floor. Neither makes
|
||||
// RoutingPresent false, so the fast path is taken forever and the finding was
|
||||
// re-published as "nothing found" once a minute.
|
||||
//
|
||||
// It is a MEASUREMENT, not a verified fact, and the difference is the residual
|
||||
// risk of this design, named rather than hidden: if the operator fixes the cause
|
||||
// out from under us — plugs the second WAN in, gives it a gateway — nothing in
|
||||
// this process re-measures until something forces a full rebuild (any config
|
||||
// edit, or any event that empties the table and makes RoutingPresent false), so
|
||||
// the finding outlives its cause until then. That direction is the deliberate
|
||||
// one: a warning that outlives its cause sends the operator to look and find it
|
||||
// fixed, while a warning that vanishes under a standing cause is the leak this
|
||||
// field exists to stop. Re-measuring on every pass is not the alternative it
|
||||
// looks like — ApplyRoutingWithWarnings is idempotent BY del-then-add, so it
|
||||
// would churn every egress binding once a minute, which is the exact cost the
|
||||
// fast path was introduced to avoid.
|
||||
lastRouteWarnings []string
|
||||
|
||||
// holding is the LATCH half of the hold state: true while a fail-closed HOLDING
|
||||
// PLANE that THIS Applier installed (holdLocked) is in the kernel — the engine
|
||||
// is down and LAN->WAN forwarding is blocked. Surfaced in Status so the panel
|
||||
@@ -80,6 +104,27 @@ type Applier struct {
|
||||
stateMu sync.RWMutex
|
||||
holding bool
|
||||
|
||||
// engineDownCause is WHY the engine is not running, as last recorded by
|
||||
// holdLocked. Guarded by stateMu.
|
||||
//
|
||||
// It exists because `plane: "hold"` with an EMPTY findings list was a normal,
|
||||
// reachable state of this product: holdLocked wrote the cause to the log and
|
||||
// nowhere else, and Status.Warnings carries the findings of the last SUCCESSFUL
|
||||
// apply — which, when the engine never started, is no apply at all. The panel
|
||||
// then said "traffic is blocked, the tunnel is down, fix it from here" and
|
||||
// nothing anywhere said WHAT to fix. The two workarounds were both obscure:
|
||||
// press Apply again and read the POST's error body, or read the log — which
|
||||
// defaults to tmpfs, so an engine that failed to start at boot leaves nothing
|
||||
// behind for an owner who sits down to investigate after a reboot.
|
||||
//
|
||||
// It is published at READ time (see Warnings) and only while the engine is
|
||||
// really down, so it is a statement about the router's condition NOW and not a
|
||||
// replay of the log: one cause, the current one, gone the instant the engine
|
||||
// starts. Same reasoning as the abandoned-generation warnings beside it and as
|
||||
// foreignHold — a latch set from a condition would have to be remembered to be
|
||||
// cleared, and this one clears itself.
|
||||
engineDownCause error
|
||||
|
||||
// traffic is WHERE THE TRAFFIC GOES under the config that is currently running:
|
||||
// tunnelled, split, straight out, or blocked (see generate.TrafficOf). It is
|
||||
// computed from the generated option.Options at the moment they are handed to
|
||||
@@ -297,20 +342,269 @@ func chainNamesFromModel() []string {
|
||||
// HTTPClient returns an http.Client that dials THROUGH the running engine's
|
||||
// outbound named by `via` (feedback #1/#8). It delegates to the engine; a stopped
|
||||
// engine yields engine.ErrEngineStopped and an unknown tag engine.ErrOutboundUnknown.
|
||||
//
|
||||
// `chain:<name>` is resolved HERE, before the engine sees it, because a chain has
|
||||
// no outbound named after itself — see resolveVia.
|
||||
func (a *Applier) HTTPClient(via string) (*http.Client, error) {
|
||||
if a.eng == nil {
|
||||
return nil, engine.ErrEngineStopped
|
||||
}
|
||||
via, err := a.resolveVia(via)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
return a.eng.HTTPClient(via)
|
||||
}
|
||||
|
||||
// chainViaPrefix is the `via`/detour selector that names a multi-hop chain.
|
||||
const chainViaPrefix = "chain:"
|
||||
|
||||
// resolveVia rewrites a `chain:<name>` selector into the outbound tag the RUNNING
|
||||
// box actually carries for that chain; every other form is passed through
|
||||
// untouched for engine.ViaToTag to handle.
|
||||
//
|
||||
// # Why this exists
|
||||
//
|
||||
// engine.ViaToTag maps "chain:X" to the bare tag "X", and there is no outbound
|
||||
// called "X": the generator materialises a chain as one wrapper per hop, tagged
|
||||
// chain-<X>-h1..chain-<X>-hN (see shater/generate/chain.go), and routes into the
|
||||
// LAST one. So `fetch_detour=chain:X` looked "X" up in the OutboundManager, missed,
|
||||
// and failed with ErrOutboundUnknown — the feature never worked. Worse than a plain
|
||||
// miss: when a NODE OR GROUP happens to share the chain's name, the lookup HITS it
|
||||
// and the subscription is fetched through a completely different outbound, silently.
|
||||
//
|
||||
// The resolution mirrors the generator, including the case the generator cannot
|
||||
// serve at all: chains are built LAZILY, only for a chain some enabled rule/egress/
|
||||
// DNS detour targets, and a fetch detour is not one of those references. A chain
|
||||
// nothing else points at therefore has no outbounds in the box, and the honest
|
||||
// answer is a named refusal — never a quiet fall back to direct, which would pull
|
||||
// the feed over the plain WAN with the owner's real address, the exact thing
|
||||
// fetch_via=proxy exists to prevent.
|
||||
//
|
||||
// A stopped engine is reported as engine.ErrEngineStopped; every resolution failure
|
||||
// wraps engine.ErrOutboundUnknown, which the panel already maps to 400.
|
||||
func (a *Applier) resolveVia(via string) (string, error) {
|
||||
if _, ok := cutPrefixFold(strings.TrimSpace(via), chainViaPrefix); !ok {
|
||||
return via, nil
|
||||
}
|
||||
tags := a.runningTags()
|
||||
if tags == nil {
|
||||
return "", engine.ErrEngineStopped
|
||||
}
|
||||
// The model is only needed for the single-hop case below, so it is read lazily
|
||||
// and a failed read (off-router, no `uci`) is not fatal: chainOutboundTag then
|
||||
// sees no chain definitions and says so, which is the truthful answer.
|
||||
var chains []model.Chain
|
||||
if m, err := model.ReadUCI(); err == nil && m != nil {
|
||||
chains = m.Chains
|
||||
}
|
||||
return resolveViaTag(via, tags, chains)
|
||||
}
|
||||
|
||||
// resolveViaTag is resolveVia's pure core: no engine, no UCI, just the running
|
||||
// box's tag set and the model's chain definitions. Split out so every branch is
|
||||
// reachable from a unit test without a box.
|
||||
func resolveViaTag(via string, tags map[string]bool, chains []model.Chain) (string, error) {
|
||||
name, ok := cutPrefixFold(strings.TrimSpace(via), chainViaPrefix)
|
||||
if !ok {
|
||||
return via, nil
|
||||
}
|
||||
return chainOutboundTag(name, tags, chains, map[string]bool{})
|
||||
}
|
||||
|
||||
// chainOutboundTag resolves a chain NAME to the tag the running box carries for it.
|
||||
//
|
||||
// Two shapes exist, both produced by shater/generate/chain.go:
|
||||
//
|
||||
// 1. Every chain with more than one hop (or with an `egress:` entry hop) is built
|
||||
// as wrappers chain-<name>-h1..chain-<name>-hN; the LAST one is the entry the
|
||||
// traffic is routed into. Member copies of a group hop are tagged
|
||||
// chain-<name>-h<i>-<member> and are NOT hops — the digits check rejects them.
|
||||
// 2. A chain that flattens to a single real hop is not wrapped at all: it IS that
|
||||
// hop, and resolves to the hop's own tag.
|
||||
//
|
||||
// Anything else is refused by name. The kind switch is a closed, positive list on
|
||||
// purpose: an unlisted spelling must land in a refusal the operator can read, not
|
||||
// in a lookup that might accidentally hit an unrelated outbound.
|
||||
func chainOutboundTag(name string, tags map[string]bool, chains []model.Chain, seen map[string]bool) (string, error) {
|
||||
name = strings.TrimSpace(name)
|
||||
if name == "" {
|
||||
return "", fmt.Errorf("%w: detour %q names no chain", engine.ErrOutboundUnknown, chainViaPrefix)
|
||||
}
|
||||
if seen[name] {
|
||||
return "", fmt.Errorf("%w: chain %q is part of a reference cycle", engine.ErrOutboundUnknown, name)
|
||||
}
|
||||
seen[name] = true
|
||||
|
||||
// (1) Hop wrappers in the running box: the highest hop index is the entry.
|
||||
if tag, ok := chainEntryTag(name, tags); ok {
|
||||
return tag, nil
|
||||
}
|
||||
|
||||
// (2) No wrappers. Either the chain collapses to one hop, or it was never built.
|
||||
def := findChainDef(chains, name)
|
||||
if def == nil {
|
||||
return "", fmt.Errorf("%w: no chain named %q is defined", engine.ErrOutboundUnknown, name)
|
||||
}
|
||||
hops := make([]string, 0, len(def.Hops))
|
||||
for _, h := range def.Hops {
|
||||
if h = strings.TrimSpace(h); h != "" {
|
||||
hops = append(hops, h)
|
||||
}
|
||||
}
|
||||
if len(hops) == 0 {
|
||||
return "", fmt.Errorf("%w: chain %q has no hops", engine.ErrOutboundUnknown, name)
|
||||
}
|
||||
if len(hops) > 1 {
|
||||
return "", errChainNotBuilt(name)
|
||||
}
|
||||
|
||||
hop := hops[0]
|
||||
kind, base := model.SplitTarget(hop)
|
||||
switch strings.ToLower(kind) {
|
||||
case "chain":
|
||||
// A chain whose only hop is another chain IS that chain (the generator
|
||||
// splices sub-chains in place before wrapping anything).
|
||||
return chainOutboundTag(base, tags, chains, seen)
|
||||
case "node", "group":
|
||||
if tags[base] {
|
||||
return base, nil
|
||||
}
|
||||
return "", errChainHopMissing(name, hop)
|
||||
case "egress":
|
||||
// An egress hop is a chain's ENTRY interface, never its exit. Alone it leaves
|
||||
// nothing to route into, which is what the generator reports as "no hops".
|
||||
return "", fmt.Errorf("%w: chain %q has only the egress entry hop %q and no exit to fetch through",
|
||||
engine.ErrOutboundUnknown, name, hop)
|
||||
case "direct", "block":
|
||||
// The generator's single-hop shortcut would alias these to the terminal
|
||||
// route target. We refuse instead, and the divergence is deliberate: this is
|
||||
// a FETCH detour, and "fetch the subscription through direct" is the
|
||||
// un-tunnelled fetch that fetch_via=proxy was set to avoid. A named refusal
|
||||
// is the recoverable side of the mistake; a silent direct fetch is not.
|
||||
// (The generator itself rejects direct/block in every multi-hop chain.)
|
||||
return "", fmt.Errorf("%w: chain %q's only hop is %q, a terminal route target and not a tunnel — set fetch_via=direct if an un-tunnelled fetch is what you want",
|
||||
engine.ErrOutboundUnknown, name, hop)
|
||||
default:
|
||||
// Bare name: model.SplitTarget puts a colon-less hop in KIND, so the hop
|
||||
// text itself is the node/group name (same rule buildHopWrapper follows).
|
||||
if tags[hop] {
|
||||
return hop, nil
|
||||
}
|
||||
return "", errChainHopMissing(name, hop)
|
||||
}
|
||||
}
|
||||
|
||||
// errChainNotBuilt names the case that has no fix at this layer: the chain is
|
||||
// configured, but nothing the generator walks refers to it, so the box was built
|
||||
// without it. Said in full because the operator otherwise sees a correct-looking
|
||||
// config and an "unknown outbound" with no way to connect the two.
|
||||
func errChainNotBuilt(name string) error {
|
||||
return fmt.Errorf("%w: chain %q is defined but the running engine holds no outbound for it — a chain is only built when an enabled rule, egress or DNS detour targets it, and a subscription fetch detour is not one of those; point a rule at %q, or fetch through node:/group:/egress: instead",
|
||||
engine.ErrOutboundUnknown, name, name)
|
||||
}
|
||||
|
||||
// errChainHopMissing reports a single-hop chain whose hop is not in the box (a
|
||||
// disabled/unparseable node, a group with no usable members, a typo).
|
||||
func errChainHopMissing(name, hop string) error {
|
||||
return fmt.Errorf("%w: chain %q is the single hop %q, which the running engine holds no outbound for",
|
||||
engine.ErrOutboundUnknown, name, hop)
|
||||
}
|
||||
|
||||
// chainEntryTag finds the chain's ENTRY wrapper in a tag set: the tag
|
||||
// "chain-<name>-h<i>" with the largest all-digit i. This is the same tag grammar
|
||||
// engine/grouphealth.go and engine/grouptest.go project chains out of the running
|
||||
// pool with; a member copy ("chain-<name>-h<i>-<member>") fails the digits test and
|
||||
// is skipped, as is another chain that merely shares the prefix.
|
||||
func chainEntryTag(name string, tags map[string]bool) (string, bool) {
|
||||
prefix := "chain-" + name + "-h"
|
||||
best, bestTag := -1, ""
|
||||
for tag := range tags {
|
||||
rest, ok := strings.CutPrefix(tag, prefix)
|
||||
if !ok {
|
||||
continue
|
||||
}
|
||||
idx, ok := allDigits(rest)
|
||||
if !ok {
|
||||
continue
|
||||
}
|
||||
if idx > best {
|
||||
best, bestTag = idx, tag
|
||||
}
|
||||
}
|
||||
return bestTag, best >= 0
|
||||
}
|
||||
|
||||
// allDigits parses a non-empty all-digit string.
|
||||
func allDigits(s string) (int, bool) {
|
||||
if s == "" {
|
||||
return 0, false
|
||||
}
|
||||
n := 0
|
||||
for _, c := range s {
|
||||
if c < '0' || c > '9' {
|
||||
return 0, false
|
||||
}
|
||||
n = n*10 + int(c-'0')
|
||||
}
|
||||
return n, true
|
||||
}
|
||||
|
||||
// findChainDef returns the model.Chain named name, or nil.
|
||||
func findChainDef(chains []model.Chain, name string) *model.Chain {
|
||||
for i := range chains {
|
||||
if strings.TrimSpace(chains[i].Name) == name {
|
||||
return &chains[i]
|
||||
}
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// cutPrefixFold is strings.CutPrefix with a case-insensitive prefix match, so
|
||||
// "CHAIN:x" is the selector "chain:x" — engine.ViaToTag accepts the same spellings.
|
||||
func cutPrefixFold(s, prefix string) (string, bool) {
|
||||
if len(s) >= len(prefix) && strings.EqualFold(s[:len(prefix)], prefix) {
|
||||
return s[len(prefix):], true
|
||||
}
|
||||
return "", false
|
||||
}
|
||||
|
||||
// runningTags is the set of outbound tags the RUNNING box can dial. Outbounds and
|
||||
// endpoints are UNIONED: a WG/AWG hop is an endpoint, and OutboundManager.Outbounds()
|
||||
// does not list those even though its Outbound(tag) lookup falls through to the
|
||||
// endpoint manager — so both are dialable and both must be visible here. nil (not
|
||||
// empty) when no box is running, which the caller reports as ErrEngineStopped.
|
||||
func (a *Applier) runningTags() map[string]bool {
|
||||
inst := a.eng.Instance()
|
||||
if inst == nil {
|
||||
return nil
|
||||
}
|
||||
om := inst.Outbound()
|
||||
if om == nil {
|
||||
return nil
|
||||
}
|
||||
tags := make(map[string]bool)
|
||||
for _, ob := range om.Outbounds() {
|
||||
tags[ob.Tag()] = true
|
||||
}
|
||||
if em := inst.Endpoint(); em != nil {
|
||||
for _, ep := range em.Endpoints() {
|
||||
tags[ep.Tag()] = true
|
||||
}
|
||||
}
|
||||
return tags
|
||||
}
|
||||
|
||||
// UpdateSubscription fetches the named subscription, folds the parsed nodes into
|
||||
// the model (replacing exactly that sub's cache), persists them to the sub's
|
||||
// JSON cache file (model.SaveSubCache) + the userinfo counters to UCI, and
|
||||
// reconciles so the running engine picks them up (feedback #8). When the
|
||||
// subscription's FetchVia=="proxy" the fetch is routed THROUGH the engine
|
||||
// outbound named by its FetchDetour (group:/node:/egress:/direct); otherwise the
|
||||
// fetch is DIRECT. Zero-node safety is inherited from subscribe.UpdateSubscription
|
||||
// outbound named by its FetchDetour (group:/node:/egress:/chain:/direct); a detour
|
||||
// that cannot be resolved FAILS the update rather than falling back — a silent
|
||||
// direct fetch would put the feed, and the owner's real address, on the plain WAN.
|
||||
// With FetchVia!="proxy" the fetch is DIRECT, as configured.
|
||||
// Zero-node safety is inherited from subscribe.UpdateSubscription
|
||||
// (a bad body leaves the cache untouched); nothing is written on a failed parse.
|
||||
// Returns the node count now cached.
|
||||
//
|
||||
@@ -454,6 +748,10 @@ func (a *Applier) applyLocked(m *model.Model) (bool, error) {
|
||||
// (3) netplane, fail-closed: on any failure return the error WITHOUT tearing
|
||||
// the engine/table down (kill-switch/table stay up; the watchdog decides).
|
||||
plane, perr := applyDataPlane(a, m, opts, now)
|
||||
// Resolve "nothing was found" against "nothing was checked" BEFORE anything is
|
||||
// published, on both the success and the failure path — abortAfterSwap republishes
|
||||
// the same plane outcome and must not drop the findings either.
|
||||
a.standingRouteWarnings(&plane)
|
||||
if perr != nil {
|
||||
// The engine is ALREADY running the new config at this point, and the data
|
||||
// plane is not — so nothing the previous apply published is true any more.
|
||||
@@ -469,6 +767,11 @@ func (a *Applier) applyLocked(m *model.Model) (bool, error) {
|
||||
a.stateGen.Add(1)
|
||||
}
|
||||
a.setHolding(false)
|
||||
// The engine carried this apply, so the last start failure is history. The
|
||||
// read-time gate would hide it anyway while the engine runs; dropping it here as
|
||||
// well means a LATER engine death cannot resurrect an unrelated old reason and
|
||||
// present it as the current one.
|
||||
a.setEngineDownCause(nil)
|
||||
a.lastGood = m
|
||||
// Publish where this config actually sends traffic, read off the very options
|
||||
// the engine was just handed (a.eng.Apply above). The engine's hash fast-path
|
||||
@@ -511,10 +814,65 @@ type planeOutcome struct {
|
||||
changed bool
|
||||
// stage names the netplane step that failed, in operator words, or "" on
|
||||
// success. It is what the failure warning is addressed to.
|
||||
stage string
|
||||
planNotes []string
|
||||
nftWarnings []string
|
||||
stage string
|
||||
planNotes []string
|
||||
nftWarnings []string
|
||||
|
||||
// routeWarnings is what the policy-routing step found, and routeMeasured is
|
||||
// whether the step RAN AT ALL this pass.
|
||||
//
|
||||
// The pair exists because nil is two different facts here and conflating them
|
||||
// published a lie. The routing step is skipped whenever the ruleset is
|
||||
// unchanged and the rules are verifiably still installed (the fast path), so
|
||||
// on a no-op reconcile — cron, once a minute — routeWarnings was nil not
|
||||
// because the router is healthy but because nobody looked. applyLocked then
|
||||
// REPLACED the published set with one that no longer contained the finding,
|
||||
// and a critical warning about a standing fault ("this egress CANNOT REACH
|
||||
// ANYTHING outside its own subnet", "this table could not be given a
|
||||
// fail-closed floor") erased itself sixty seconds after it appeared. Neither
|
||||
// of those faults makes RoutingPresent false, so nothing brought it back.
|
||||
//
|
||||
// They are written together by measuredRouting and never separately, so the
|
||||
// flag cannot drift away from the finding it qualifies.
|
||||
routeWarnings []string
|
||||
routeMeasured bool
|
||||
}
|
||||
|
||||
// measuredRouting records the result of an ApplyRoutingWithWarnings call. It is
|
||||
// the ONLY way routeWarnings is set, which is what keeps routeMeasured honest:
|
||||
// "there is nothing to report" and "nothing was checked" cannot be spelled the
|
||||
// same way by accident.
|
||||
func (o *planeOutcome) measuredRouting(ws []string) {
|
||||
o.routeWarnings, o.routeMeasured = ws, true
|
||||
}
|
||||
|
||||
// standingRouteWarnings makes out.routeWarnings the CURRENT set of standing
|
||||
// policy-routing findings rather than the findings of this pass, so the caller can
|
||||
// keep publishing it unconditionally and still not erase a fault that is still
|
||||
// there. Caller holds a.mu.
|
||||
//
|
||||
// Two cases, and the asymmetry is the whole point:
|
||||
//
|
||||
// - the routing WAS measured: this pass looked, so what it saw replaces the
|
||||
// memory outright — including when it saw nothing, which is how a finding is
|
||||
// retired once its cause is gone. Without that half this would be a latch, and
|
||||
// a latch is the same defect pointed the other way.
|
||||
// - the routing was NOT measured (the fast path): nothing looked, so the last
|
||||
// measurement is still the best statement anyone here can make about the
|
||||
// router, and it stands. The alternative — publishing the empty set — is a
|
||||
// positive claim of health that no observation supports.
|
||||
//
|
||||
// The memory is valid across the fast path because the fast path is only taken
|
||||
// when the rendered ruleset is byte-identical to the loaded one; any ruleset change
|
||||
// forces !nftCurrent, which forces the routing step to run and re-measure. So what
|
||||
// is carried is always a measurement of the plane that is loaded right now, never
|
||||
// one belonging to a config that has since been replaced.
|
||||
func (a *Applier) standingRouteWarnings(out *planeOutcome) {
|
||||
if out.routeMeasured {
|
||||
a.lastRouteWarnings = out.routeWarnings
|
||||
return
|
||||
}
|
||||
out.routeWarnings = a.lastRouteWarnings
|
||||
}
|
||||
|
||||
// engineApply and applyDataPlane are the two heavy halves of applyLocked, behind
|
||||
@@ -529,6 +887,29 @@ var (
|
||||
applyDataPlane = (*Applier).applyDataPlaneLocked
|
||||
)
|
||||
|
||||
// The policy-routing/sysctl half of applyDataPlaneLocked, behind seams for the
|
||||
// same reason as the two above and as tableExists/teardownNft: whether a given
|
||||
// pass MEASURED the routing or fast-pathed past it is the fact the published
|
||||
// warning set now depends on, and it is decided by `ip rule show` against a
|
||||
// kernel no test has. Stubbing them is also what keeps a unit test from writing
|
||||
// real sysctls into the host it runs on. Production uses netplane.
|
||||
// readConfig is model.ReadUCI behind a seam, used by Status ALONE.
|
||||
//
|
||||
// Deliberately not threaded through the other readers in this file: what needs
|
||||
// exercising is not "the model is read" but the FAILURE of that read on the status
|
||||
// path, because the failure is itself part of the published contract now
|
||||
// (config_readable). model.ReadUCI shells out to `uci`, so the unreadable-config
|
||||
// state — a full /overlay, a `uci commit` caught half-written, the state the boot
|
||||
// armor exists for — can otherwise only be produced on a router by breaking it.
|
||||
var readConfig = model.ReadUCI
|
||||
|
||||
var (
|
||||
routingPresent = netplane.RoutingPresent
|
||||
applyRoutingWithWarnings = netplane.ApplyRoutingWithWarnings
|
||||
applySysctl = netplane.ApplySysctl
|
||||
applyIfaceSysctlsAt = netplane.ApplyIfaceSysctlsAt
|
||||
)
|
||||
|
||||
// applyDataPlaneLocked is step (3) of the pipeline: nft ruleset, policy routing
|
||||
// and sysctls, in that order, fail-closed. Caller holds a.mu and has ALREADY
|
||||
// swapped the engine, so every failure here leaves the router in a mixed state —
|
||||
@@ -585,17 +966,24 @@ func (a *Applier) applyDataPlaneLocked(m *model.Model, opts option.Options, now
|
||||
// routeWarnings carries an egress that was BUILT but cannot route (no nexthop on a
|
||||
// non-point-to-point device). It is deliberately part of the status warning set:
|
||||
// such an egress looks applied everywhere in the UI while being unable to reach
|
||||
// anything off its own subnet. On the fast path nothing was rebuilt, so there is
|
||||
// nothing new to report and the previous set stands.
|
||||
if !nftCurrent || !netplane.RoutingPresent(m.Globals) {
|
||||
routeWarnings, rerr := netplane.ApplyRoutingWithWarnings(m)
|
||||
out.routeWarnings = routeWarnings
|
||||
// anything off its own subnet.
|
||||
//
|
||||
// The fast path takes NO measurement, and that is reported as such (routeMeasured
|
||||
// stays false) rather than as an empty result. The sentence that used to stand
|
||||
// here — "nothing was rebuilt, so there is nothing new to report and the previous
|
||||
// set stands" — described an intention the code did not implement: applyLocked
|
||||
// publishes unconditionally, so the previous set was REPLACED by one without these
|
||||
// findings. Keeping the previous set standing is now applyLocked's job
|
||||
// (standingRouteWarnings), and this flag is what tells it which case it is in.
|
||||
if !nftCurrent || !routingPresent(m.Globals) {
|
||||
routeWarnings, rerr := applyRoutingWithWarnings(m)
|
||||
out.measuredRouting(routeWarnings)
|
||||
if rerr != nil {
|
||||
out.stage = "installing the policy routing"
|
||||
return out, rerr
|
||||
}
|
||||
}
|
||||
if err := netplane.ApplySysctl(); err != nil {
|
||||
if err := applySysctl(); err != nil {
|
||||
out.stage = "setting the kernel sysctls"
|
||||
return out, err
|
||||
}
|
||||
@@ -607,7 +995,7 @@ func (a *Applier) applyDataPlaneLocked(m *model.Model, opts option.Options, now
|
||||
// bridge (`ifup lan`, a VLAN change) hands back a device with the kernel
|
||||
// defaults, silently un-setting accept_local behind our back. Fail-closed like
|
||||
// the other netplane steps: return without teardown.
|
||||
if err := netplane.ApplyIfaceSysctlsAt(m, now); err != nil {
|
||||
if err := applyIfaceSysctlsAt(m, now); err != nil {
|
||||
out.stage = "setting the per-interface sysctls"
|
||||
return out, err
|
||||
}
|
||||
@@ -706,6 +1094,11 @@ func (a *Applier) holdLocked(m *model.Model, cause error) {
|
||||
// "tunnel" verdict left behind by a config that is no longer running is the same
|
||||
// reassuring lie in a different place.
|
||||
a.setTraffic(generate.Traffic{})
|
||||
// Record WHY, before either branch below returns. It belongs on the fail-open
|
||||
// branch too: there the plane is "none" and the panel says "not protected,
|
||||
// traffic is going out directly", which is an even worse place to be told
|
||||
// nothing about the cause.
|
||||
a.setEngineDownCause(cause)
|
||||
if !killSwitchClosed(m.Globals) {
|
||||
a.log.Warn("engine is down and kill_switch=open: LAN traffic is NOT protected (documented fail-open): ", cause)
|
||||
return
|
||||
@@ -905,6 +1298,45 @@ func (a *Applier) setHolding(v bool) {
|
||||
a.stateMu.Unlock()
|
||||
}
|
||||
|
||||
func (a *Applier) setEngineDownCause(err error) {
|
||||
a.stateMu.Lock()
|
||||
a.engineDownCause = err
|
||||
a.stateMu.Unlock()
|
||||
}
|
||||
|
||||
// engineDownWarning turns a recorded start failure into the operator-facing
|
||||
// finding. It leads with the CONSEQUENCE the owner is already looking at ("this is
|
||||
// why your traffic is blocked") and then the cause, because the panel has already
|
||||
// told them the tunnel is down — what it could not tell them is what to fix.
|
||||
//
|
||||
// The two spellings are a closed choice, not a default: the boot-time arm
|
||||
// (errEngineNotStartedYet) is a normal few seconds of a healthy start-up and must
|
||||
// not spend the panel's alarm banner, which lights on critical and on nothing else.
|
||||
// A real start failure is critical, because a router that cannot proxy anything is
|
||||
// exactly what that banner is for.
|
||||
func engineDownWarning(cause error) Warning {
|
||||
if errors.Is(cause, errEngineNotStartedYet) {
|
||||
return Warning{
|
||||
Severity: SeverityWarning,
|
||||
Section: "engine",
|
||||
Name: "starting",
|
||||
Message: "the tunnel engine has not started yet, so traffic from your devices is being blocked " +
|
||||
"rather than let out unprotected. That is the normal state for the first seconds after a boot " +
|
||||
"or a restart. If it stays this way, the reason will appear here as soon as a start is attempted.",
|
||||
}
|
||||
}
|
||||
return Warning{
|
||||
Severity: SeverityCritical,
|
||||
Section: "engine",
|
||||
Name: "start",
|
||||
Message: fmt.Sprintf("the tunnel engine could NOT be started, and that is why nothing is being "+
|
||||
"proxied or filtered: %v. The daemon retries every minute, so a cause that clears itself (a "+
|
||||
"rule-set that could not be downloaded at boot, a port still held by the previous instance) will "+
|
||||
"recover on its own; anything else needs the configuration fixed. Check the node or subscription "+
|
||||
"named above if one is.", cause),
|
||||
}
|
||||
}
|
||||
|
||||
// Traffic returns where the traffic of the CURRENTLY RUNNING config goes. The
|
||||
// zero value means no config of this process's is running (nothing applied yet,
|
||||
// or the plane was torn down / put on hold), and callers must render that as
|
||||
@@ -932,25 +1364,49 @@ func (a *Applier) setWarnings(ws []Warning) {
|
||||
// empty slice means "the last apply was clean", which the panel must render
|
||||
// differently from "no apply has run yet" (Active/Plane cover that).
|
||||
//
|
||||
// The live half is currently the engine's abandoned generations
|
||||
// (engineTeardownWarnings). It is computed at READ time rather than folded into
|
||||
// lastWarnings on purpose: a superseded box that will not shut down is a
|
||||
// condition of the process, not a property of a config. Folding it in would make
|
||||
// it appear only after the NEXT successful apply and then stay published long
|
||||
// after the shutdown finally completed — reporting a leak that is over, and
|
||||
// staying silent about one that is not. Read-time means it shows up the instant
|
||||
// it happens and clears itself the instant it resolves.
|
||||
// The live half is TWO things, and both are computed at READ time rather than
|
||||
// folded into lastWarnings on purpose.
|
||||
//
|
||||
// - The engine's abandoned generations (engineTeardownWarnings). A superseded box
|
||||
// that will not shut down is a condition of the process, not a property of a
|
||||
// config. Folding it in would make it appear only after the NEXT successful
|
||||
// apply and then stay published long after the shutdown finally completed —
|
||||
// reporting a leak that is over, and staying silent about one that is not.
|
||||
// - WHY THE ENGINE IS DOWN (engineDownCause). A failed start produces no
|
||||
// successful apply, so it can leave nothing in lastWarnings by construction;
|
||||
// that is how `plane: "hold"` with zero findings became a normal state of this
|
||||
// product, and the panel ended up saying "the tunnel is down, fix it from here"
|
||||
// with the cause recorded only in a log that defaults to tmpfs.
|
||||
//
|
||||
// Read-time means both show up the instant they happen and clear themselves the
|
||||
// instant they resolve. The engine-down entry in particular is gated on the engine
|
||||
// being down RIGHT NOW, so a start that eventually succeeds retires it without
|
||||
// anything having to remember to.
|
||||
func (a *Applier) Warnings() []Warning {
|
||||
var out []Warning
|
||||
if a.eng != nil {
|
||||
out = engineTeardownWarnings(a.eng.PendingCloses())
|
||||
}
|
||||
return a.warningsWith(a.eng != nil && a.eng.Running())
|
||||
}
|
||||
|
||||
// warningsWith is Warnings against an engine-liveness fact the caller has already
|
||||
// established. Same shape and same two reasons as holdingWith: Status must not ask
|
||||
// the engine twice for one poll, and — the one that matters — two independent reads
|
||||
// can straddle an engine swap and publish "the engine could NOT be started" beside
|
||||
// running=true, which is a contradiction the panel has no way to resolve.
|
||||
func (a *Applier) warningsWith(engineUp bool) []Warning {
|
||||
a.stateMu.RLock()
|
||||
defer a.stateMu.RUnlock()
|
||||
if out == nil && a.lastWarnings == nil {
|
||||
cause, last := a.engineDownCause, a.lastWarnings
|
||||
a.stateMu.RUnlock()
|
||||
|
||||
var out []Warning
|
||||
if cause != nil && !engineUp {
|
||||
out = append(out, engineDownWarning(cause))
|
||||
}
|
||||
if a.eng != nil {
|
||||
out = append(out, engineTeardownWarnings(a.eng.PendingCloses())...)
|
||||
}
|
||||
if out == nil && last == nil {
|
||||
return []Warning{}
|
||||
}
|
||||
return append(out, a.lastWarnings...)
|
||||
return append(out, last...)
|
||||
}
|
||||
|
||||
// Reconcile re-reads UCI and either tears down (disabled) or re-applies (enabled).
|
||||
@@ -1065,6 +1521,12 @@ func (a *Applier) teardown(arm func() bool) error {
|
||||
clearActiveFlag(a.log)
|
||||
a.lastGood = nil
|
||||
a.lastNft = ""
|
||||
// TeardownRouting just removed the rules and tables those findings were about,
|
||||
// so the last measurement no longer describes anything. Dropped here and NOT in
|
||||
// holdLocked, deliberately: the holding plane replaces the nft table and leaves
|
||||
// the policy routing exactly where it was, so there the measurement is still
|
||||
// true and carrying it forward is correct.
|
||||
a.lastRouteWarnings = nil
|
||||
// Not a blanket false any more: with a holding plane standing, forwarded LAN
|
||||
// traffic really IS being blocked, and saying otherwise here is the inverted lie
|
||||
// Holding()'s doc comment is about — the process is exiting, but Status can
|
||||
@@ -1072,6 +1534,10 @@ func (a *Applier) teardown(arm func() bool) error {
|
||||
a.setHolding(kept)
|
||||
a.setTraffic(generate.Traffic{})
|
||||
a.setWarnings(nil)
|
||||
// The engine is down because it was TOLD to be. Keeping a start failure here
|
||||
// would make an operator's own `stop` look like a malfunction for as long as the
|
||||
// process lives to answer.
|
||||
a.setEngineDownCause(nil)
|
||||
// The plane is gone, so the logged set no longer describes anything. Forget it,
|
||||
// and the next apply re-announces its warnings in full rather than staying
|
||||
// silent because they happen to match what a now-dismantled plane had.
|
||||
@@ -1393,7 +1859,9 @@ func ActiveFlagPresent() bool {
|
||||
// "Is the daemon process alive?" is a different question and is answered
|
||||
// by whether the status call returned at all (plus uptime_seconds, which
|
||||
// only a live daemon can produce).
|
||||
// enabled globals.enabled in UCI.
|
||||
// enabled globals.enabled in UCI. ONLY MEANINGFUL WITH config_readable=true —
|
||||
// see that field; false with config_readable=false means "unknown", and
|
||||
// rendering it as "switched off" is a documented defect.
|
||||
// active ACTIVE_FLAG present. This is the "the service is meant to be running"
|
||||
// latch that gates hotplug and cron, NOT a health signal: it is raised by
|
||||
// a successful enabled apply and cleared only by teardown, so it stays up
|
||||
@@ -1404,23 +1872,51 @@ func ActiveFlagPresent() bool {
|
||||
// table the `inet shater` nft table is loaded.
|
||||
// hash the running engine's config hash ("" when the engine is not started).
|
||||
// kill_switch globals.kill_switch in UCI (closed = fail-closed, open = leaky).
|
||||
// Only meaningful with config_readable=true.
|
||||
// panel_port the CONFIGURED admin-panel port (globals.panel_port, or 8088 when
|
||||
// unset). This is the configured port, not necessarily the live bound
|
||||
// one — an env override (SHATER_PANEL_ADDR) is not reflected here.
|
||||
// Only meaningful with config_readable=true; 0 means "not read".
|
||||
// config_readable whether the three fields above could be read at all, and
|
||||
// config_error why not. See the fields.
|
||||
// can_rollback whether a rollback would actually revert something: true when a
|
||||
// commit-confirm snapshot is armed, OR the engine holds a last-good
|
||||
// predecessor config. When false the panel hides the rollback control
|
||||
// (there is nothing to roll back to).
|
||||
type Status struct {
|
||||
Running bool `json:"running"`
|
||||
Enabled bool `json:"enabled"`
|
||||
Active bool `json:"active"`
|
||||
Running bool `json:"running"`
|
||||
Enabled bool `json:"enabled"`
|
||||
Active bool `json:"active"`
|
||||
Table bool `json:"table"`
|
||||
Hash string `json:"hash"`
|
||||
KillSwitch string `json:"kill_switch"`
|
||||
PanelPort int `json:"panel_port"`
|
||||
CanRollback bool `json:"can_rollback"`
|
||||
|
||||
// ConfigReadable is whether the configuration could be READ when this status was
|
||||
// taken. It qualifies Enabled, KillSwitch and PanelPort, which are the only
|
||||
// fields sourced from it: when this is false those three are ZERO VALUES AND
|
||||
// MEAN NOTHING — not "switched off", not "kill switch unset", not "port 0".
|
||||
//
|
||||
// A consumer MUST test it before reading Enabled. "Not enabled" and "we cannot
|
||||
// tell" are opposite situations: the first is the owner's own choice and calls
|
||||
// for a calm note, the second happens when the flash is full or a commit was
|
||||
// interrupted, and on this router that is exactly when the fail-closed plane has
|
||||
// the LAN cut off. Presenting the second as the first tells the owner they
|
||||
// switched something off and sends them to a settings page backed by the same
|
||||
// unreadable file.
|
||||
//
|
||||
// Positive phrasing is deliberate: a client that does not know this field yet
|
||||
// sees it missing, reads false, and lands on the alarming side rather than the
|
||||
// reassuring one.
|
||||
ConfigReadable bool `json:"config_readable"`
|
||||
|
||||
// ConfigError is why the read failed, verbatim, or "" when it did not. It is a
|
||||
// diagnostic for the panel and the log — the operator-facing sentence is carried
|
||||
// in Warnings, as a critical entry in section "config", so that consumers which
|
||||
// only render the warning list still show it.
|
||||
ConfigError string `json:"config_error"`
|
||||
|
||||
// EngineRunning is whether a sing-box instance is actually started.
|
||||
//
|
||||
// It was added as the honest field to stand beside a `running` that was hard-wired
|
||||
@@ -1545,7 +2041,7 @@ func (a *Applier) Status() Status {
|
||||
CanRollback: a.canRollback(),
|
||||
EngineRunning: engineUp,
|
||||
Traffic: a.Traffic(),
|
||||
Warnings: a.Warnings(),
|
||||
Warnings: a.warningsWith(engineUp),
|
||||
}
|
||||
s.StartedUnix, s.UptimeSeconds = processUptime(time.Now())
|
||||
switch {
|
||||
@@ -1560,10 +2056,50 @@ func (a *Applier) Status() Status {
|
||||
default:
|
||||
s.Plane = "full"
|
||||
}
|
||||
if m, err := model.ReadUCI(); err == nil {
|
||||
// The three config-derived fields, and the fact that they ARE derived from a
|
||||
// read that can fail.
|
||||
//
|
||||
// The failure used to be silent: `if err == nil { ... }` and nothing else, so an
|
||||
// unreadable configuration published enabled=false, kill_switch="" and
|
||||
// panel_port=0 with no indication that they were placeholders. The panel checks
|
||||
// `!enabled` before it looks at `plane` and renders "Turned off" — amber, no
|
||||
// alarm, "turn the service on in Settings" — which is a lie in the one situation
|
||||
// this daemon has a fail-closed plane FOR: a full /overlay, or a `uci commit`
|
||||
// caught half-written, in which the boot armor has cut the LAN off and the panel
|
||||
// says the owner did it to themselves. Settings, of course, reads the same
|
||||
// unreadable configuration.
|
||||
//
|
||||
// So the read result is published as a fact of its own. config_readable is
|
||||
// phrased positively — absent or false means "no reading", the alarming side —
|
||||
// so a consumer that has not been taught about it cannot be reassured by its
|
||||
// absence.
|
||||
if m, err := readConfig(); err == nil {
|
||||
s.ConfigReadable = true
|
||||
s.Enabled = m.Globals.Enabled
|
||||
s.KillSwitch = m.Globals.KillSwitch
|
||||
s.PanelPort = effectivePanelPort(m.Globals.PanelPort)
|
||||
} else {
|
||||
s.ConfigError = err.Error()
|
||||
// Say it in the warning list too, and not only in a new field. The warning
|
||||
// list is already rendered by every consumer, so this reaches the owner on the
|
||||
// day the field ships rather than on the day someone reads it; and being
|
||||
// computed at read time it clears itself the moment the configuration becomes
|
||||
// readable again — the same reasoning as Warnings' live half and foreignHold.
|
||||
//
|
||||
// Prepended without re-running finalizeWarnings: critical is where the sort
|
||||
// would put it anyway, and re-finalizing an already-capped set would rewrite
|
||||
// its "N further warning(s) suppressed" disclosure with a wrong count.
|
||||
s.Warnings = append([]Warning{{
|
||||
Severity: SeverityCritical,
|
||||
Section: "config",
|
||||
Name: "unreadable",
|
||||
Message: fmt.Sprintf("the router's configuration could NOT be read (%v), so this status cannot say "+
|
||||
"whether shater is switched on, whether the kill switch is closed, or which port this panel is "+
|
||||
"served on — enabled, kill_switch and panel_port are placeholders here, not readings. If traffic "+
|
||||
"is being blocked, that is the fail-closed plane doing its job and NOT the service being switched "+
|
||||
"off: do not turn anything off to fix it. The usual causes are a full /overlay and a `uci commit` "+
|
||||
"interrupted part-way; free space, check /etc/config/shater, then restart shaterd.", err),
|
||||
}}, s.Warnings...)
|
||||
}
|
||||
return s
|
||||
}
|
||||
|
||||
+156
-19
@@ -1,15 +1,48 @@
|
||||
package apply
|
||||
|
||||
import (
|
||||
"errors"
|
||||
"os"
|
||||
"strings"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"github.com/sagernet/sing-box/option"
|
||||
"github.com/sagernet/sing-box/shater/engine"
|
||||
"github.com/sagernet/sing-box/shater/model"
|
||||
"github.com/sagernet/sing-box/shater/netplane"
|
||||
)
|
||||
|
||||
// TestMain takes THIS TEST BINARY off the real network namespace, for the one
|
||||
// operation of netplane's that destroys something another test binary can be
|
||||
// using: the L3-ingress TUN devices.
|
||||
//
|
||||
// Applier.Teardown calls netplane.TeardownRouting for real here, and its
|
||||
// device sweep deletes BOTH slots unconditionally. `go test` runs package
|
||||
// binaries concurrently (-p defaults to GOMAXPROCS) and they all share one
|
||||
// network namespace, so on a runner with iproute2 and /dev/net/tun this binary
|
||||
// was issuing 9 `ip link del shater-l3a` and 9 `ip link del shater-l3b` per
|
||||
// gate run — measured with an `ip` shim on PATH — into the namespace where
|
||||
// shater/generate's privileged tests hold a live TUN. What that looks like from
|
||||
// the other side is a red TestIntegrationL3* in a package that did nothing
|
||||
// wrong:
|
||||
//
|
||||
// no [shater-l3a shater-l3b] device exists after a successful Start
|
||||
//
|
||||
// netplane.L3StubKernelForTest carries the full measurement and the reasoning.
|
||||
//
|
||||
// Scope, stated rather than implied: this diverts ONLY the L3 device deletes.
|
||||
// The `ip rule del` / `ip route flush` on the reserved tables that teardown also
|
||||
// performs still run for real from here. They are left alone because nothing
|
||||
// else in the gate reads those tables, so unlike the devices they have no
|
||||
// observed victim — not because they are harmless on a machine that matters.
|
||||
func TestMain(m *testing.M) {
|
||||
restore := netplane.L3StubKernelForTest()
|
||||
code := m.Run()
|
||||
restore()
|
||||
os.Exit(code)
|
||||
}
|
||||
|
||||
// TestCanRollback pins the signal the panel gates its rollback control on:
|
||||
// canRollback is false on a fresh applier (no armed commit-confirm snapshot AND
|
||||
// the engine holds no last-good predecessor), and flips to true once a snapshot
|
||||
@@ -372,36 +405,140 @@ func TestHoldLockedOpenInstallsNothing(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
// TestApplyInstallsHoldWhenEngineFailsToStart is the end-to-end W5 wiring: a
|
||||
// model the engine cannot come up on must leave the data plane PROTECTING, not
|
||||
// absent. On the VM the trigger was an unreachable remote rule-set at boot; any
|
||||
// construction failure takes the same path.
|
||||
// errEngineStartFailedInTest stands in for what really happens on the router:
|
||||
// engine.Apply -> newBox -> box.New refuses the config (an unreachable remote
|
||||
// rule-set at boot, a node the registry cannot build, a busy port), so the engine
|
||||
// never starts. TestEngineApplyReallyFailsWithoutStarting below is the CONTROL
|
||||
// that this is a faithful model and not a convenient fiction.
|
||||
var errEngineStartFailedInTest = errors.New("create instance: initialize outbound[0]: outbound type not found")
|
||||
|
||||
// stubEngineApply replaces the engineApply seam — applyLocked's engine-swap step,
|
||||
// documented in apply.go as existing precisely so a failing stage can be tested
|
||||
// without a router — and COUNTS the calls.
|
||||
//
|
||||
// The counter is not decoration. This is the test that has to survive the way its
|
||||
// predecessor did not: it used to break the engine by pointing a rule-set at
|
||||
// /nonexistent/nope.srs and then guard itself with `if err == nil ||
|
||||
// a.eng.Running() { t.Skip(...) }`. Once LocalRuleSet.reloadFile started treating
|
||||
// an unreadable file as an EMPTY one, the engine came up fine, the guard fired on
|
||||
// every platform, and the package still reported `ok` — the single most important
|
||||
// test of this package asserted nothing at all for months. A seam the test drives
|
||||
// itself cannot rot that way; the counter closes the one remaining hole, which is
|
||||
// applyLocked ceasing to go through the seam at all.
|
||||
func stubEngineApply(t *testing.T, err error) *int {
|
||||
t.Helper()
|
||||
var calls int
|
||||
orig := engineApply
|
||||
engineApply = func(*Applier, option.Options) (bool, error) {
|
||||
calls++
|
||||
return false, err
|
||||
}
|
||||
t.Cleanup(func() { engineApply = orig })
|
||||
return &calls
|
||||
}
|
||||
|
||||
// stubTableExists pins the "is our nft table in the kernel" fact, which Status
|
||||
// reads to name the plane. A unit test has no kernel table, so without this the
|
||||
// only reachable verdict is "none" and the "hold" branch — the one the panel shows
|
||||
// the operator — is never exercised.
|
||||
func stubTableExists(t *testing.T, loaded bool) {
|
||||
t.Helper()
|
||||
orig := tableExists
|
||||
tableExists = func() bool { return loaded }
|
||||
t.Cleanup(func() { tableExists = orig })
|
||||
}
|
||||
|
||||
// TestApplyInstallsHoldWhenEngineFailsToStart is the end-to-end W5 wiring: a model
|
||||
// the engine cannot come up on must leave the data plane PROTECTING, not absent.
|
||||
// On the VM the trigger was an unreachable remote rule-set at boot; any
|
||||
// construction failure takes the same path, which is why the failure is injected
|
||||
// at the seam rather than reproduced through one particular cause — the branch
|
||||
// under test is applyLocked's, and a cause that stops causing (see stubEngineApply)
|
||||
// silently retires the test.
|
||||
//
|
||||
// Everything except the engine swap is REAL here: the model, generate, the
|
||||
// kill-switch decision, netplane.RenderHoldNft, the latch and Status. Only the
|
||||
// load into the kernel is intercepted (withHoldProbe), so what the assertions read
|
||||
// is the ruleset that would have gone to `nft -f`.
|
||||
func TestApplyInstallsHoldWhenEngineFailsToStart(t *testing.T) {
|
||||
loaded := withHoldProbe(t)
|
||||
calls := stubEngineApply(t, errEngineStartFailedInTest)
|
||||
stubTableExists(t, true) // the holding plane we install below IS a loaded table
|
||||
a := New(engine.New(), nil)
|
||||
m := holdModel("closed")
|
||||
// A rule-set file that cannot be read: generate warns, and the engine build
|
||||
// fails for one reason or another on every platform we run on.
|
||||
m.Rulesets = []model.Ruleset{{
|
||||
Name: "badfile", Type: "domain", Source: "file",
|
||||
Path: "/nonexistent/nope.srs", Format: "binary",
|
||||
}}
|
||||
|
||||
_, err := a.Apply(m)
|
||||
if err == nil || a.eng.Running() {
|
||||
t.Skip("this platform started the engine anyway; the hold path is covered by TestHoldLockedInstallsBlockingPlane")
|
||||
|
||||
// The instrument first: an assertion suite that never reached the branch is
|
||||
// worth nothing, and saying so by name is the whole lesson of this test.
|
||||
if *calls != 1 {
|
||||
t.Fatalf("the engine-swap seam ran %d times, want exactly 1 — applyLocked no longer goes "+
|
||||
"through engineApply, so this test is NOT exercising the engine-failure branch", *calls)
|
||||
}
|
||||
if len(*loaded) == 0 {
|
||||
t.Fatalf("engine failed to start (%v) but NO holding plane was installed — "+
|
||||
"the router would forward LAN traffic to the WAN in the clear with kill_switch=closed", err)
|
||||
if !errors.Is(err, errEngineStartFailedInTest) {
|
||||
t.Fatalf("Apply returned %v, want the engine failure unwrapped — the caller (cmd/shaterd, "+
|
||||
"the panel) decides what to tell the operator from this error", err)
|
||||
}
|
||||
if !strings.Contains((*loaded)[0], "drop") {
|
||||
t.Errorf("the installed plane does not drop:\n%s", (*loaded)[0])
|
||||
if a.eng.Running() {
|
||||
t.Fatalf("precondition: the engine must not be running after a failed swap")
|
||||
}
|
||||
|
||||
if len(*loaded) != 1 {
|
||||
t.Fatalf("engine failed to start (%v) but %d holding planes were installed, want 1 — "+
|
||||
"with none, the router forwards LAN traffic to the WAN in the clear while "+
|
||||
"kill_switch=closed", err, len(*loaded))
|
||||
}
|
||||
rs := (*loaded)[0]
|
||||
if !strings.Contains(rs, "meta nfproto ipv4 drop") || !strings.Contains(rs, "meta nfproto ipv6 drop") {
|
||||
t.Errorf("the installed plane does not block forwarded traffic on both families:\n%s", rs)
|
||||
}
|
||||
if strings.Contains(rs, "hook input") || strings.Contains(rs, "hook output") {
|
||||
t.Errorf("the installed plane must only hook forward, or the operator loses management access:\n%s", rs)
|
||||
}
|
||||
if strings.Contains(rs, "tproxy ") {
|
||||
t.Errorf("the installed plane must not divert to a dead engine socket:\n%s", rs)
|
||||
}
|
||||
|
||||
if !a.Holding() {
|
||||
t.Errorf("Holding() must report true so the panel can show 'protected, not proxying'")
|
||||
}
|
||||
if got := a.Status().Plane; got != "hold" && got != "none" {
|
||||
t.Errorf("Status().Plane = %q, want hold (or none where nft is unavailable)", got)
|
||||
s := a.Status()
|
||||
if s.Plane != "hold" {
|
||||
t.Errorf("Status().Plane = %q, want \"hold\" — with a table loaded and the engine down, "+
|
||||
"\"full\" would tell the operator traffic is being proxied when nothing is", s.Plane)
|
||||
}
|
||||
if s.EngineRunning {
|
||||
t.Errorf("Status().EngineRunning must be false after a failed engine swap")
|
||||
}
|
||||
if s.Traffic.Verdict != "" {
|
||||
t.Errorf("Traffic.Verdict = %q, want \"\" (unknown): no config of ours is running, and a "+
|
||||
"leftover verdict is a reassuring lie", s.Traffic.Verdict)
|
||||
}
|
||||
}
|
||||
|
||||
// TestEngineApplyReallyFailsWithoutStarting is the CONTROL for the stub above: it
|
||||
// proves that the seam's PRODUCTION twin, engine.Apply, really can return an error
|
||||
// with the engine left stopped — i.e. that the state
|
||||
// TestApplyInstallsHoldWhenEngineFailsToStart simulates is a state this fork can
|
||||
// actually be in. Without it, that test would be an instrument with no proof it
|
||||
// measures anything real.
|
||||
//
|
||||
// The trigger is chosen to be independent of every moving part around it: an
|
||||
// outbound type no registry can ever hold fails in adapter/outbound.Registry.
|
||||
// CreateOutbound ("outbound type not found"), before any listener is opened, so it
|
||||
// needs no root, no network, no TUN and no nft — and it cannot quietly start
|
||||
// succeeding the way an unreadable rule-set file did.
|
||||
func TestEngineApplyReallyFailsWithoutStarting(t *testing.T) {
|
||||
e := engine.New()
|
||||
_, err := e.Apply(option.Options{
|
||||
Outbounds: []option.Outbound{{Type: "shater-no-such-outbound-type", Tag: "probe"}},
|
||||
})
|
||||
if err == nil {
|
||||
t.Fatalf("engine.Apply accepted an outbound type that cannot exist; the failure this " +
|
||||
"package's hold path is built for would never occur and the W5 test above simulates nothing")
|
||||
}
|
||||
if e.Running() {
|
||||
t.Errorf("engine.Apply failed with %v but left the engine RUNNING — the hold path is "+
|
||||
"gated on !Running(), so it would never engage", err)
|
||||
}
|
||||
}
|
||||
|
||||
@@ -0,0 +1,366 @@
|
||||
package apply
|
||||
|
||||
import (
|
||||
"errors"
|
||||
"strings"
|
||||
"testing"
|
||||
|
||||
"github.com/sagernet/sing-box/shater/engine"
|
||||
"github.com/sagernet/sing-box/shater/generate"
|
||||
"github.com/sagernet/sing-box/shater/model"
|
||||
)
|
||||
|
||||
// A subscription's fetch_detour is resolved by Applier.HTTPClient, and `chain:<X>`
|
||||
// is the one form the engine cannot resolve on its own: a chain has no outbound
|
||||
// named after itself. These tests pin the resolution against the tag grammar the
|
||||
// GENERATOR actually emits, and pin what happens on every miss — the answer must
|
||||
// never be "fetch it direct", because that puts the feed and the router's real
|
||||
// address on the plain WAN, which is what fetch_via=proxy exists to prevent.
|
||||
|
||||
// tagSet builds a running-box tag set for the pure resolver.
|
||||
func tagSet(tags ...string) map[string]bool {
|
||||
out := make(map[string]bool, len(tags))
|
||||
for _, t := range tags {
|
||||
out[t] = true
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
// ssURI is a parseable share link so generate can build a real outbound from it.
|
||||
func ssURI(host, name string) string {
|
||||
return "ss://YWVzLTI1Ni1nY206cGFzcw@" + host + ":8388#" + name
|
||||
}
|
||||
|
||||
// TestChainDetourTagMatchesGenerator is the tripwire: it runs the REAL generator
|
||||
// over a model whose rule targets a chain, and requires that the tag resolveViaTag
|
||||
// hands the engine is one the generator actually emitted — and specifically the
|
||||
// chain's entry (highest hop index), not the chain's own name.
|
||||
//
|
||||
// This is the test that fails on the unfixed code: engine.ViaToTag("chain:work")
|
||||
// yields "work", which appears nowhere in the emitted config.
|
||||
func TestChainDetourTagMatchesGenerator(t *testing.T) {
|
||||
m := &model.Model{
|
||||
Globals: model.Globals{Enabled: true},
|
||||
Nodes: []model.Node{
|
||||
{Name: "n1", Enabled: true, URI: ssURI("1.2.3.4", "n1")},
|
||||
{Name: "n2", Enabled: true, URI: ssURI("5.6.7.8", "n2")},
|
||||
},
|
||||
Chains: []model.Chain{{Name: "work", Hops: []string{"node:n1", "node:n2"}}},
|
||||
Rules: []model.Rule{
|
||||
{Name: "r", Enabled: true, Target: "chain:work", DstPort: "443"},
|
||||
},
|
||||
}
|
||||
opts, _, err := generate.GenerateWithWarnings(m)
|
||||
if err != nil {
|
||||
t.Fatalf("generate: %v", err)
|
||||
}
|
||||
emitted := map[string]bool{}
|
||||
for _, ob := range opts.Outbounds {
|
||||
emitted[ob.Tag] = true
|
||||
}
|
||||
for _, ep := range opts.Endpoints {
|
||||
emitted[ep.Tag] = true
|
||||
}
|
||||
if !emitted["chain-work-h2"] {
|
||||
t.Fatalf("fixture broken: generator emitted no chain-work-h2; tags=%v", emitted)
|
||||
}
|
||||
|
||||
got, err := resolveViaTag("chain:work", emitted, m.Chains)
|
||||
if err != nil {
|
||||
t.Fatalf("resolveViaTag(chain:work): %v", err)
|
||||
}
|
||||
if !emitted[got] {
|
||||
t.Fatalf("resolved %q, which the generator never emitted (tags=%v)", got, emitted)
|
||||
}
|
||||
if got != "chain-work-h2" {
|
||||
t.Fatalf("resolved %q, want the chain ENTRY chain-work-h2", got)
|
||||
}
|
||||
}
|
||||
|
||||
// TestChainDetourEntryIsTheLastHop pins entry = highest hop index, and that a group
|
||||
// hop's member copies (chain-<n>-h<i>-<member>) are not mistaken for hops. Dialling
|
||||
// a member copy instead of its wrapper is not a harmless near-miss: the wrapper IS
|
||||
// the group's selector, so the copy pins one member and throws the balancing away.
|
||||
func TestChainDetourEntryIsTheLastHop(t *testing.T) {
|
||||
cases := []struct {
|
||||
name string
|
||||
tags map[string]bool
|
||||
want string
|
||||
}{
|
||||
{
|
||||
name: "highest index wins, 10 beats 2",
|
||||
tags: tagSet(
|
||||
"direct", "block", "n1", "n2", "n3",
|
||||
"chain-work-h1", "chain-work-h1-n1",
|
||||
"chain-work-h2", "chain-work-h2-n2", "chain-work-h2-n3",
|
||||
"chain-work-h10", "chain-work-h10-n3",
|
||||
"chain-workshop-h1", // another chain sharing the prefix
|
||||
),
|
||||
want: "chain-work-h10",
|
||||
},
|
||||
{
|
||||
// The digits check is what makes this case come out right: every tag here
|
||||
// starts with "chain-work-h", and only one of them is a hop.
|
||||
name: "a lone group hop resolves to the WRAPPER, never to a member copy",
|
||||
tags: tagSet(
|
||||
"direct", "n1", "n2",
|
||||
"chain-work-h1",
|
||||
"chain-work-h1-n1", "chain-work-h1-n2", "chain-work-h1-zzz",
|
||||
),
|
||||
want: "chain-work-h1",
|
||||
},
|
||||
}
|
||||
for _, tc := range cases {
|
||||
t.Run(tc.name, func(t *testing.T) {
|
||||
got, err := resolveViaTag("chain:work", tc.tags, nil)
|
||||
if err != nil {
|
||||
t.Fatalf("resolveViaTag: %v", err)
|
||||
}
|
||||
if got != tc.want {
|
||||
t.Fatalf("entry = %q, want %q", got, tc.want)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
// TestChainDetourNotBuiltRefusesInsteadOfDirect is the leak guard. A chain nothing
|
||||
// else references is never materialised, so no wrapper exists. The resolver must
|
||||
// say so by name — and must NOT resolve to "direct" or to a node/group that merely
|
||||
// shares the chain's name.
|
||||
func TestChainDetourNotBuiltRefusesInsteadOfDirect(t *testing.T) {
|
||||
// "work" is ALSO a group tag here: the trap the unfixed code fell into, where
|
||||
// engine.ViaToTag("chain:work") -> "work" hits an unrelated outbound.
|
||||
tags := tagSet("direct", "block", "n1", "n2", "work")
|
||||
chains := []model.Chain{{Name: "work", Hops: []string{"node:n1", "node:n2"}}}
|
||||
|
||||
got, err := resolveViaTag("chain:work", tags, chains)
|
||||
if err == nil {
|
||||
t.Fatalf("unbuilt chain resolved to %q, want a refusal", got)
|
||||
}
|
||||
if got != "" {
|
||||
t.Fatalf("refusal also returned a tag %q; a caller could dial it", got)
|
||||
}
|
||||
if !errors.Is(err, engine.ErrOutboundUnknown) {
|
||||
t.Fatalf("err = %v, want it to wrap engine.ErrOutboundUnknown (panel maps that to 400)", err)
|
||||
}
|
||||
for _, want := range []string{"chain \"work\"", "only built when an enabled rule"} {
|
||||
if !strings.Contains(err.Error(), want) {
|
||||
t.Fatalf("err %q does not explain the cause (missing %q)", err, want)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestChainDetourUndefinedIsNamed: a fetch_detour pointing at a chain that does not
|
||||
// exist at all must say exactly that.
|
||||
func TestChainDetourUndefinedIsNamed(t *testing.T) {
|
||||
got, err := resolveViaTag("chain:ghost", tagSet("direct", "ghost"), nil)
|
||||
if err == nil {
|
||||
t.Fatalf("undefined chain resolved to %q, want a refusal", got)
|
||||
}
|
||||
if !errors.Is(err, engine.ErrOutboundUnknown) {
|
||||
t.Fatalf("err = %v, want engine.ErrOutboundUnknown", err)
|
||||
}
|
||||
if !strings.Contains(err.Error(), "no chain named \"ghost\"") {
|
||||
t.Fatalf("err %q does not name the missing chain", err)
|
||||
}
|
||||
}
|
||||
|
||||
// TestChainDetourSingleHop covers the shape the generator does NOT wrap: a chain
|
||||
// that flattens to one hop IS that hop.
|
||||
func TestChainDetourSingleHop(t *testing.T) {
|
||||
tags := tagSet("direct", "block", "n1", "grp")
|
||||
cases := []struct {
|
||||
name string
|
||||
chains []model.Chain
|
||||
via string
|
||||
want string
|
||||
}{
|
||||
{
|
||||
name: "node hop",
|
||||
chains: []model.Chain{{Name: "solo", Hops: []string{"node:n1"}}},
|
||||
via: "chain:solo",
|
||||
want: "n1",
|
||||
},
|
||||
{
|
||||
name: "group hop",
|
||||
chains: []model.Chain{{Name: "solo", Hops: []string{"group:grp"}}},
|
||||
via: "chain:solo",
|
||||
want: "grp",
|
||||
},
|
||||
{
|
||||
name: "bare hop",
|
||||
chains: []model.Chain{{Name: "solo", Hops: []string{"n1"}}},
|
||||
via: "chain:solo",
|
||||
want: "n1",
|
||||
},
|
||||
{
|
||||
name: "sub-chain hop is spliced",
|
||||
chains: []model.Chain{
|
||||
{Name: "outer", Hops: []string{"chain:inner"}},
|
||||
{Name: "inner", Hops: []string{"node:n1"}},
|
||||
},
|
||||
via: "chain:outer",
|
||||
want: "n1",
|
||||
},
|
||||
{
|
||||
name: "blank hops are ignored",
|
||||
chains: []model.Chain{{Name: "solo", Hops: []string{"", "node:n1", " "}}},
|
||||
via: "chain:solo",
|
||||
want: "n1",
|
||||
},
|
||||
}
|
||||
for _, tc := range cases {
|
||||
t.Run(tc.name, func(t *testing.T) {
|
||||
got, err := resolveViaTag(tc.via, tags, tc.chains)
|
||||
if err != nil {
|
||||
t.Fatalf("resolveViaTag(%q): %v", tc.via, err)
|
||||
}
|
||||
if got != tc.want {
|
||||
t.Fatalf("resolveViaTag(%q) = %q, want %q", tc.via, got, tc.want)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
// TestChainDetourRefusals pins every remaining miss as a NAMED refusal with no tag,
|
||||
// so none of them can degrade into a fetch through the wrong outbound.
|
||||
func TestChainDetourRefusals(t *testing.T) {
|
||||
tags := tagSet("direct", "block", "n1", "egress-wg0")
|
||||
cases := []struct {
|
||||
name string
|
||||
chains []model.Chain
|
||||
via string
|
||||
wantMsg string
|
||||
}{
|
||||
{
|
||||
name: "hop not in the running box",
|
||||
chains: []model.Chain{{Name: "solo", Hops: []string{"node:gone"}}},
|
||||
via: "chain:solo",
|
||||
wantMsg: "holds no outbound for",
|
||||
},
|
||||
{
|
||||
name: "egress entry hop alone has no exit",
|
||||
chains: []model.Chain{{Name: "solo", Hops: []string{"egress:wg0"}}},
|
||||
via: "chain:solo",
|
||||
wantMsg: "no exit to fetch through",
|
||||
},
|
||||
{
|
||||
name: "direct is not a tunnel hop",
|
||||
chains: []model.Chain{{Name: "solo", Hops: []string{"direct"}}},
|
||||
via: "chain:solo",
|
||||
wantMsg: "terminal route target",
|
||||
},
|
||||
{
|
||||
name: "block is not a tunnel hop",
|
||||
chains: []model.Chain{{Name: "solo", Hops: []string{"block"}}},
|
||||
via: "chain:solo",
|
||||
wantMsg: "terminal route target",
|
||||
},
|
||||
{
|
||||
name: "chain with no hops",
|
||||
chains: []model.Chain{{Name: "solo", Hops: []string{" "}}},
|
||||
via: "chain:solo",
|
||||
wantMsg: "has no hops",
|
||||
},
|
||||
{
|
||||
name: "chain: with no name",
|
||||
chains: nil,
|
||||
via: "chain:",
|
||||
wantMsg: "names no chain",
|
||||
},
|
||||
{
|
||||
name: "self-referencing chain terminates",
|
||||
chains: []model.Chain{{Name: "loop", Hops: []string{"chain:loop"}}},
|
||||
via: "chain:loop",
|
||||
wantMsg: "reference cycle",
|
||||
},
|
||||
{
|
||||
name: "mutual cycle terminates",
|
||||
chains: []model.Chain{
|
||||
{Name: "a", Hops: []string{"chain:b"}},
|
||||
{Name: "b", Hops: []string{"chain:a"}},
|
||||
},
|
||||
via: "chain:a",
|
||||
wantMsg: "reference cycle",
|
||||
},
|
||||
}
|
||||
for _, tc := range cases {
|
||||
t.Run(tc.name, func(t *testing.T) {
|
||||
got, err := resolveViaTag(tc.via, tags, tc.chains)
|
||||
if err == nil {
|
||||
t.Fatalf("resolveViaTag(%q) = %q, want a refusal", tc.via, got)
|
||||
}
|
||||
if got != "" {
|
||||
t.Fatalf("refusal also returned tag %q", got)
|
||||
}
|
||||
if got == "direct" {
|
||||
t.Fatalf("refusal degraded to direct — that is a WAN leak")
|
||||
}
|
||||
if !errors.Is(err, engine.ErrOutboundUnknown) {
|
||||
t.Fatalf("err = %v, want engine.ErrOutboundUnknown", err)
|
||||
}
|
||||
if !strings.Contains(err.Error(), tc.wantMsg) {
|
||||
t.Fatalf("err %q does not contain %q", err, tc.wantMsg)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
// TestChainDetourCaseInsensitive: engine.ViaToTag accepts "CHAIN:x"; so must this,
|
||||
// or a spelling the engine would have honoured falls through as a bare tag.
|
||||
func TestChainDetourCaseInsensitive(t *testing.T) {
|
||||
tags := tagSet("chain-work-h1", "chain-work-h2")
|
||||
for _, via := range []string{"chain:work", "Chain:work", "CHAIN:work", " chain:work "} {
|
||||
got, err := resolveViaTag(via, tags, nil)
|
||||
if err != nil {
|
||||
t.Fatalf("resolveViaTag(%q): %v", via, err)
|
||||
}
|
||||
if got != "chain-work-h2" {
|
||||
t.Fatalf("resolveViaTag(%q) = %q, want chain-work-h2", via, got)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestNonChainDetoursUnchanged is the regression guard: every other fetch_detour
|
||||
// form must reach engine.ViaToTag byte-for-byte as configured, so group:/node:/
|
||||
// egress:/direct/bare keep behaving exactly as before this resolver existed.
|
||||
func TestNonChainDetoursUnchanged(t *testing.T) {
|
||||
// A tag set that WOULD satisfy a chain lookup, to catch a resolver that starts
|
||||
// treating everything as a chain name.
|
||||
tags := tagSet("direct", "auto", "n1", "egress-wg0", "chain-auto-h1", "chain-n1-h1")
|
||||
chains := []model.Chain{{Name: "auto", Hops: []string{"node:n1"}}}
|
||||
|
||||
for _, via := range []string{"", "direct", "DIRECT", "group:auto", "node:n1", "egress:wg0", "n1", " group:auto "} {
|
||||
got, err := resolveViaTag(via, tags, chains)
|
||||
if err != nil {
|
||||
t.Fatalf("resolveViaTag(%q): %v", via, err)
|
||||
}
|
||||
if got != via {
|
||||
t.Fatalf("resolveViaTag(%q) = %q, want it passed through unchanged for engine.ViaToTag", via, got)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestHTTPClientChainOnStoppedEngine: with no box running there are no tags to
|
||||
// resolve against, and the caller must hear "engine stopped" — not a confusing
|
||||
// claim about the chain, and not a client dialling anything.
|
||||
func TestHTTPClientChainOnStoppedEngine(t *testing.T) {
|
||||
a := New(engine.New(), nil)
|
||||
for _, via := range []string{"chain:work", "group:auto", "direct", ""} {
|
||||
c, err := a.HTTPClient(via)
|
||||
if !errors.Is(err, engine.ErrEngineStopped) {
|
||||
t.Fatalf("HTTPClient(%q): err = %v, want engine.ErrEngineStopped", via, err)
|
||||
}
|
||||
if c != nil {
|
||||
t.Fatalf("HTTPClient(%q) returned a client %v on a stopped engine", via, c)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestHTTPClientNilEngine keeps the pre-existing nil-engine contract.
|
||||
func TestHTTPClientNilEngine(t *testing.T) {
|
||||
a := &Applier{}
|
||||
if _, err := a.HTTPClient("chain:work"); !errors.Is(err, engine.ErrEngineStopped) {
|
||||
t.Fatalf("err = %v, want engine.ErrEngineStopped", err)
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,576 @@
|
||||
package apply
|
||||
|
||||
// Regression tests for two more ways this package reported calm over a router that
|
||||
// was not doing what its config said. Both are the INVERTED failure — not an error
|
||||
// raised when things are fine, but silence (or an amber "you switched it off")
|
||||
// while something is really wrong — which is the only kind that gets someone hurt.
|
||||
//
|
||||
// 1. a CRITICAL policy-routing finding published by one apply and erased by the
|
||||
// next no-op reconcile, sixty seconds later, while its cause stood;
|
||||
// 2. a configuration that cannot be READ published as enabled=false, i.e. as the
|
||||
// owner's own choice, while the fail-closed plane had the LAN cut off.
|
||||
|
||||
import (
|
||||
"errors"
|
||||
"strings"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"github.com/sagernet/sing-box/option"
|
||||
"github.com/sagernet/sing-box/shater/engine"
|
||||
"github.com/sagernet/sing-box/shater/model"
|
||||
"github.com/sagernet/sing-box/shater/netplane"
|
||||
)
|
||||
|
||||
// The package's tests are written against a ROUTER-LIKE BASELINE, in which the
|
||||
// configuration can be read. That baseline used to be supplied by accident: on a
|
||||
// build host there is no `uci`, model.ReadUCI failed on every Status() call, and
|
||||
// the failure was silent — which is defect 2 itself. Now that the failure is
|
||||
// published, leaving it in place would make every test in this package assert
|
||||
// against a router whose configuration is unreadable, which is nobody's intended
|
||||
// fixture and would have masked the real subject of several of them.
|
||||
//
|
||||
// So the baseline is made explicit here rather than left to the absence of a
|
||||
// binary. A test that wants the UNREADABLE case sets readConfig itself and restores
|
||||
// it (see TestStatusSaysWhenTheConfigCannotBeRead) — which is also the only way to
|
||||
// exercise that branch deterministically on a router, where `uci` does exist.
|
||||
//
|
||||
// An init() rather than a TestMain on purpose: init functions compose, so another
|
||||
// test file in this package can add its own without a conflict.
|
||||
func init() {
|
||||
readConfig = func() (*model.Model, error) { return &model.Model{Globals: model.DefaultGlobals()}, nil }
|
||||
}
|
||||
|
||||
// noGatewayFinding is netplane/apply.go's real text for the fault this is all
|
||||
// about, shortened to its load-bearing clause: an egress that was BUILT, reports as
|
||||
// applied everywhere in the UI, and cannot carry a packet off its own subnet.
|
||||
// Quoted rather than invented so the test breaks if that warning is ever reworded
|
||||
// out of the critical class.
|
||||
const noGatewayFinding = `egress "wan2": no IPv4 gateway could be found for interface "wan2" (device eth1) ` +
|
||||
`by any means, so its routing table sends traffic straight onto the local segment. The device is not ` +
|
||||
`point-to-point, so this egress CANNOT REACH ANYTHING outside its own subnet — every node, group and ` +
|
||||
`rule bound to it will fail to connect.`
|
||||
|
||||
// findFinding reports whether the published warning set still carries the finding,
|
||||
// and at what severity.
|
||||
func findFinding(ws []Warning, substr string) (Warning, bool) {
|
||||
for _, w := range ws {
|
||||
if strings.Contains(w.Message, substr) {
|
||||
return w, true
|
||||
}
|
||||
}
|
||||
return Warning{}, false
|
||||
}
|
||||
|
||||
// TestStandingRouteWarningSurvivesTheFastPath is the defect-1 regression, stated in
|
||||
// the terms the owner experiences it.
|
||||
//
|
||||
// A second WAN comes up without a gateway. The apply that notices publishes a
|
||||
// critical finding. A minute later cron reconciles; nothing has changed, so the
|
||||
// data plane takes its fast path and measures no routing — and applyLocked then
|
||||
// published `collectWarnings(..., plane.routeWarnings, ...)` UNCONDITIONALLY, with
|
||||
// routeWarnings nil, replacing the set with one that no longer contained the
|
||||
// finding. Neither of the two findings that survive to the fast path (no gateway,
|
||||
// no fail-closed floor) makes RoutingPresent false, so nothing ever brought it
|
||||
// back: the panel showed zero findings, plane full, green, over an egress carrying
|
||||
// nothing — or over a routing table whose next ifdown is a leak onto the plain WAN.
|
||||
//
|
||||
// THE CONTROL IS THE POINT. Pass 1 proves this instrument can see a live finding at
|
||||
// all, and pass 3 proves it can see one DISAPPEAR. Without both, "the warning is
|
||||
// still there after the fast path" would be satisfied by an instrument that always
|
||||
// says yes, and would equally be satisfied by turning the fix into a latch — which
|
||||
// is the same defect pointed the other way.
|
||||
func TestStandingRouteWarningSurvivesTheFastPath(t *testing.T) {
|
||||
a := New(engine.New(), nil)
|
||||
m := holdModel("closed")
|
||||
|
||||
// The plane stage is driven directly: what is under test is what applyLocked
|
||||
// PUBLISHES for a given plane outcome, and the outcome itself is produced by
|
||||
// TestApplyDataPlaneFastPathTakesNoRoutingMeasurement below.
|
||||
var outcome planeOutcome
|
||||
restore := stubApplyStages(t,
|
||||
func(*Applier, option.Options) (bool, error) { return false, nil },
|
||||
func(*Applier, *model.Model, option.Options, time.Time) (planeOutcome, error) {
|
||||
return outcome, nil
|
||||
})
|
||||
defer restore()
|
||||
|
||||
apply := func(t *testing.T, o planeOutcome) []Warning {
|
||||
t.Helper()
|
||||
outcome = o
|
||||
a.mu.Lock()
|
||||
_, err := a.applyLocked(m)
|
||||
a.mu.Unlock()
|
||||
if err != nil {
|
||||
t.Fatalf("applyLocked: %v", err)
|
||||
}
|
||||
return a.Warnings()
|
||||
}
|
||||
|
||||
// (1) CONTROL — the instrument can see a live finding. The routing was measured
|
||||
// and it found the egress unable to route.
|
||||
var measured planeOutcome
|
||||
measured.measuredRouting([]string{noGatewayFinding})
|
||||
ws := apply(t, measured)
|
||||
w, ok := findFinding(ws, "CANNOT REACH ANYTHING outside its own subnet")
|
||||
if !ok {
|
||||
t.Fatalf("control failed: a MEASURED route finding is not published at all, so this test "+
|
||||
"could not detect its loss either; warnings = %+v", ws)
|
||||
}
|
||||
if w.Severity != SeverityCritical {
|
||||
t.Errorf("an egress that cannot reach off its own subnet is critical, got %q", w.Severity)
|
||||
}
|
||||
|
||||
// (2) THE DEFECT — the cron reconcile a minute later. Nothing changed, the fast
|
||||
// path took no measurement, and the fault is still standing.
|
||||
ws = apply(t, planeOutcome{}) // routeMeasured false: nothing was checked
|
||||
if _, ok := findFinding(ws, "CANNOT REACH ANYTHING outside its own subnet"); !ok {
|
||||
t.Errorf("the fast path erased a CRITICAL finding whose cause is still standing: "+
|
||||
"the panel now shows a clean, green router over an egress that carries nothing. "+
|
||||
"warnings = %+v", ws)
|
||||
}
|
||||
|
||||
// (3) CONTROL — the instrument can see the finding go away. The cause was fixed
|
||||
// (a gateway appeared), something forced a real rebuild, and the routing was
|
||||
// measured again with nothing to report. A fix that merely latched the warning
|
||||
// would fail here.
|
||||
var clean planeOutcome
|
||||
clean.measuredRouting(nil)
|
||||
ws = apply(t, clean)
|
||||
if w, ok := findFinding(ws, "CANNOT REACH ANYTHING outside its own subnet"); ok {
|
||||
t.Errorf("a re-MEASURED clean routing state must retire the finding, still published: %+v", w)
|
||||
}
|
||||
|
||||
// And the erasure is not permanent either: a later measurement that finds it
|
||||
// again publishes it again.
|
||||
ws = apply(t, measured)
|
||||
if _, ok := findFinding(ws, "CANNOT REACH ANYTHING outside its own subnet"); !ok {
|
||||
t.Errorf("a finding that recurs must be published again; warnings = %+v", ws)
|
||||
}
|
||||
}
|
||||
|
||||
// TestStandingRouteWarningSurvivesAPostSwapAbort pins the same property on the
|
||||
// other publishing path. abortAfterSwap republishes the plane outcome with its own
|
||||
// critical entry in front; if it were handed the fast path's empty routeWarnings it
|
||||
// would drop the standing finding exactly like the success path did.
|
||||
func TestStandingRouteWarningSurvivesAPostSwapAbort(t *testing.T) {
|
||||
a := New(engine.New(), nil)
|
||||
m := holdModel("closed")
|
||||
|
||||
var outcome planeOutcome
|
||||
var stageErr error
|
||||
restore := stubApplyStages(t,
|
||||
func(*Applier, option.Options) (bool, error) { return true, nil },
|
||||
func(*Applier, *model.Model, option.Options, time.Time) (planeOutcome, error) {
|
||||
return outcome, stageErr
|
||||
})
|
||||
defer restore()
|
||||
|
||||
// A successful apply records the standing finding.
|
||||
outcome.measuredRouting([]string{noGatewayFinding})
|
||||
a.mu.Lock()
|
||||
_, err := a.applyLocked(m)
|
||||
a.mu.Unlock()
|
||||
if err != nil {
|
||||
t.Fatalf("applyLocked (seed): %v", err)
|
||||
}
|
||||
|
||||
// Now the sysctl stage fails on a pass that fast-pathed the routing.
|
||||
outcome = planeOutcome{stage: "setting the kernel sysctls"}
|
||||
stageErr = errors.New("sysctl: read-only file system")
|
||||
a.mu.Lock()
|
||||
_, err = a.applyLocked(m)
|
||||
a.mu.Unlock()
|
||||
if err == nil {
|
||||
t.Fatalf("applyLocked must surface the netplane failure")
|
||||
}
|
||||
|
||||
ws := a.Warnings()
|
||||
if _, ok := findFinding(ws, "could NOT be completed"); !ok {
|
||||
t.Errorf("the abort must still say the data plane is incomplete; warnings = %+v", ws)
|
||||
}
|
||||
if _, ok := findFinding(ws, "CANNOT REACH ANYTHING outside its own subnet"); !ok {
|
||||
t.Errorf("a half-installed plane must not also erase the standing routing finding; "+
|
||||
"warnings = %+v", ws)
|
||||
}
|
||||
}
|
||||
|
||||
// TestStandingRouteWarningsClearedOnTeardown: teardown removes the ip rules and
|
||||
// tables those findings are ABOUT, so the last measurement stops describing
|
||||
// anything and must not be carried into the next apply.
|
||||
func TestStandingRouteWarningsClearedOnTeardown(t *testing.T) {
|
||||
a := New(engine.New(), nil)
|
||||
m := holdModel("closed")
|
||||
|
||||
var outcome planeOutcome
|
||||
outcome.measuredRouting([]string{noGatewayFinding})
|
||||
restore := stubApplyStages(t,
|
||||
func(*Applier, option.Options) (bool, error) { return false, nil },
|
||||
func(*Applier, *model.Model, option.Options, time.Time) (planeOutcome, error) {
|
||||
return outcome, nil
|
||||
})
|
||||
defer restore()
|
||||
|
||||
a.mu.Lock()
|
||||
if _, err := a.applyLocked(m); err != nil {
|
||||
a.mu.Unlock()
|
||||
t.Fatalf("applyLocked: %v", err)
|
||||
}
|
||||
a.mu.Unlock()
|
||||
if len(a.lastRouteWarnings) == 0 {
|
||||
t.Fatalf("control failed: the measurement was not remembered, so this test cannot show it cleared")
|
||||
}
|
||||
|
||||
// Teardown with everything stubbed out: nothing here may touch a real kernel.
|
||||
origTeardownNft, origTable := teardownNft, tableExists
|
||||
teardownNft = func() error { return nil }
|
||||
tableExists = func() bool { return false }
|
||||
defer func() { teardownNft, tableExists = origTeardownNft, origTable }()
|
||||
|
||||
if err := a.Teardown(); err != nil {
|
||||
t.Fatalf("Teardown: %v", err)
|
||||
}
|
||||
if len(a.lastRouteWarnings) != 0 {
|
||||
t.Errorf("teardown removed the rules and tables those findings describe, but kept them: %+v",
|
||||
a.lastRouteWarnings)
|
||||
}
|
||||
|
||||
// The next apply's fast path must therefore publish nothing, not the findings of
|
||||
// a plane that no longer exists.
|
||||
outcome = planeOutcome{}
|
||||
a.mu.Lock()
|
||||
if _, err := a.applyLocked(m); err != nil {
|
||||
a.mu.Unlock()
|
||||
t.Fatalf("applyLocked after teardown: %v", err)
|
||||
}
|
||||
a.mu.Unlock()
|
||||
if w, ok := findFinding(a.Warnings(), "CANNOT REACH ANYTHING outside its own subnet"); ok {
|
||||
t.Errorf("a finding about a torn-down plane is still published: %+v", w)
|
||||
}
|
||||
}
|
||||
|
||||
// TestApplyDataPlaneFastPathTakesNoRoutingMeasurement is the other half of the
|
||||
// instrument, and without it the tests above prove nothing about production: they
|
||||
// drive applyDataPlane through its seam, so they would pass just as happily if the
|
||||
// REAL data-plane stage marked its fast path as a measurement.
|
||||
//
|
||||
// It exercises applyDataPlaneLocked itself, with the ruleset pre-loaded so the nft
|
||||
// fast path is taken and only the routing decision varies.
|
||||
func TestApplyDataPlaneFastPathTakesNoRoutingMeasurement(t *testing.T) {
|
||||
m := holdModel("closed")
|
||||
now := time.Now()
|
||||
opts := option.Options{}
|
||||
|
||||
a := New(engine.New(), nil)
|
||||
// Reproduce exactly what the stage will render, so `ruleset == a.lastNft` holds
|
||||
// and ApplyNft is never reached (there is no nft binary here, and a test must not
|
||||
// load a ruleset into the host it runs on).
|
||||
ruleset, _, err := netplane.RenderNftPlanAt(m, a.untunnelablePlanFor(m, opts), now)
|
||||
if err != nil {
|
||||
t.Fatalf("RenderNftPlanAt: %v", err)
|
||||
}
|
||||
|
||||
origTable := tableExists
|
||||
origPresent, origApplyRouting := routingPresent, applyRoutingWithWarnings
|
||||
origSysctl, origIface := applySysctl, applyIfaceSysctlsAt
|
||||
tableExists = func() bool { return true }
|
||||
applySysctl = func() error { return nil }
|
||||
applyIfaceSysctlsAt = func(*model.Model, time.Time) error { return nil }
|
||||
defer func() {
|
||||
tableExists = origTable
|
||||
routingPresent, applyRoutingWithWarnings = origPresent, origApplyRouting
|
||||
applySysctl, applyIfaceSysctlsAt = origSysctl, origIface
|
||||
}()
|
||||
|
||||
tests := []struct {
|
||||
name string
|
||||
present bool
|
||||
wantCalled bool
|
||||
wantMeasured bool
|
||||
}{
|
||||
{
|
||||
name: "routing intact: the fast path measures nothing",
|
||||
present: true,
|
||||
wantCalled: false,
|
||||
wantMeasured: false,
|
||||
},
|
||||
{
|
||||
// The CONTROL for the case above: the same stage, one input flipped, does
|
||||
// call the routing and does report a measurement. A stage that never
|
||||
// measured anything would satisfy the first row on its own.
|
||||
name: "routing missing: the full path measures",
|
||||
present: false,
|
||||
wantCalled: true,
|
||||
wantMeasured: true,
|
||||
},
|
||||
}
|
||||
for _, tc := range tests {
|
||||
t.Run(tc.name, func(t *testing.T) {
|
||||
called := false
|
||||
routingPresent = func(model.Globals) bool { return tc.present }
|
||||
applyRoutingWithWarnings = func(*model.Model) ([]string, error) {
|
||||
called = true
|
||||
return []string{noGatewayFinding}, nil
|
||||
}
|
||||
|
||||
a.lastNft = ruleset
|
||||
a.mu.Lock()
|
||||
out, err := a.applyDataPlaneLocked(m, opts, now)
|
||||
a.mu.Unlock()
|
||||
if err != nil {
|
||||
t.Fatalf("applyDataPlaneLocked: %v (stage %q)", err, out.stage)
|
||||
}
|
||||
// Guards on the setup itself: if the nft fast path was NOT taken, the run
|
||||
// below says nothing about the routing decision.
|
||||
if out.stage != "" || out.changed {
|
||||
t.Fatalf("setup broken: the nft fast path was not taken (stage %q, changed %v)",
|
||||
out.stage, out.changed)
|
||||
}
|
||||
if called != tc.wantCalled {
|
||||
t.Errorf("ApplyRoutingWithWarnings called = %v, want %v", called, tc.wantCalled)
|
||||
}
|
||||
if out.routeMeasured != tc.wantMeasured {
|
||||
t.Errorf("routeMeasured = %v, want %v — a pass that took no measurement must not "+
|
||||
"report one, or applyLocked will publish 'nothing found' as if something had looked",
|
||||
out.routeMeasured, tc.wantMeasured)
|
||||
}
|
||||
if tc.wantMeasured && len(out.routeWarnings) != 1 {
|
||||
t.Errorf("a measured pass must carry what it found, got %+v", out.routeWarnings)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
// --- defect 3: the engine is down and nothing says why -----------------------
|
||||
|
||||
// TestEngineStartFailureIsPublished is the defect-3 regression.
|
||||
//
|
||||
// holdLocked wrote the cause of a failed engine start to the LOG and nowhere else:
|
||||
// it sets traffic to unknown, installs the fail-closed plane, and never calls
|
||||
// setWarnings. Status.Warnings carries the findings of the last SUCCESSFUL apply,
|
||||
// and an engine that never started produced none — so `plane: "hold"` with an empty
|
||||
// findings list was a normal, reachable state of the product. The panel correctly
|
||||
// said "traffic is blocked, the tunnel is down, fix it from here", and nothing
|
||||
// anywhere said WHAT to fix. The two workarounds were pressing Apply again to read
|
||||
// the POST's error body, and reading a log that lives in tmpfs by default — gone
|
||||
// after the reboot the owner sat down to investigate.
|
||||
func TestEngineStartFailureIsPublished(t *testing.T) {
|
||||
a := New(engine.New(), nil)
|
||||
m := holdModel("closed")
|
||||
|
||||
// CONTROL — nothing has failed yet, so nothing is claimed. Without this, an
|
||||
// implementation that warns unconditionally would satisfy every check below.
|
||||
if w, ok := findFinding(a.Warnings(), "could NOT be started"); ok {
|
||||
t.Fatalf("control failed: an engine that has not failed must not be reported as failed: %+v", w)
|
||||
}
|
||||
|
||||
// The holding plane is installed without touching a real kernel.
|
||||
origHold := applyHoldNft
|
||||
applyHoldNft = func(string) error { return nil }
|
||||
defer func() { applyHoldNft = origHold }()
|
||||
|
||||
cause := errors.New("start outbound/vless[node-tokyo]: parse server address: invalid IP")
|
||||
a.mu.Lock()
|
||||
a.holdLocked(m, cause)
|
||||
a.mu.Unlock()
|
||||
|
||||
if !a.holding {
|
||||
t.Fatalf("precondition: holdLocked must have installed the holding plane")
|
||||
}
|
||||
w, ok := findFinding(a.Warnings(), "could NOT be started")
|
||||
if !ok {
|
||||
t.Fatalf("the engine is down and the LAN is held, and NOTHING says why; warnings = %+v",
|
||||
a.Warnings())
|
||||
}
|
||||
if w.Severity != SeverityCritical {
|
||||
t.Errorf("a router that cannot proxy anything is critical, got %q", w.Severity)
|
||||
}
|
||||
if w.Section != "engine" {
|
||||
t.Errorf("the finding must be attributable to the engine, got section %q", w.Section)
|
||||
}
|
||||
if !strings.Contains(w.Message, "parse server address: invalid IP") {
|
||||
t.Errorf("the finding must carry the CAUSE verbatim — that is the whole point: %q", w.Message)
|
||||
}
|
||||
// It must reach Status, which is what the panel actually reads.
|
||||
origTable := tableExists
|
||||
tableExists = func() bool { return true }
|
||||
defer func() { tableExists = origTable }()
|
||||
s := a.Status()
|
||||
if s.Plane != "hold" {
|
||||
t.Fatalf("precondition: plane = %q, want \"hold\" — the state this defect is about", s.Plane)
|
||||
}
|
||||
if _, ok := findFinding(s.Warnings, "could NOT be started"); !ok {
|
||||
t.Errorf("plane=hold with an EMPTY findings list is the defect; warnings = %+v", s.Warnings)
|
||||
}
|
||||
|
||||
// CONTROL — it is a CONDITION, not a log feed. With the same cause still on
|
||||
// record, an engine that is up must publish nothing: the entry is gated on the
|
||||
// engine being down right now, which is what makes it self-clearing rather than a
|
||||
// latch somebody has to remember to reset. (Driven through warningsWith because
|
||||
// engine liveness is unexported state; Status feeds it the same fact it puts in
|
||||
// `running`, so the two can never contradict each other.)
|
||||
if w, ok := findFinding(a.warningsWith(true), "could NOT be started"); ok {
|
||||
t.Errorf("with the engine UP, a past start failure is not a current condition: %+v", w)
|
||||
}
|
||||
if _, ok := findFinding(a.warningsWith(false), "could NOT be started"); !ok {
|
||||
t.Errorf("control failed: warningsWith(false) must still show it, or the check above proves nothing")
|
||||
}
|
||||
|
||||
// And the record itself is dropped by a successful apply, so a LATER engine death
|
||||
// cannot resurrect this reason and present it as the current one.
|
||||
restore := stubApplyStages(t,
|
||||
func(*Applier, option.Options) (bool, error) { return true, nil },
|
||||
func(*Applier, *model.Model, option.Options, time.Time) (planeOutcome, error) {
|
||||
var out planeOutcome
|
||||
out.measuredRouting(nil)
|
||||
return out, nil
|
||||
})
|
||||
defer restore()
|
||||
a.mu.Lock()
|
||||
_, err := a.applyLocked(m)
|
||||
a.mu.Unlock()
|
||||
if err != nil {
|
||||
t.Fatalf("applyLocked (recovery): %v", err)
|
||||
}
|
||||
if w, ok := findFinding(a.Warnings(), "could NOT be started"); ok {
|
||||
t.Errorf("a start failure must retire itself once the engine runs, still published: %+v", w)
|
||||
}
|
||||
}
|
||||
|
||||
// TestEngineNotStartedYetIsNotAnAlarm: the boot-time arm (ArmHold) goes through the
|
||||
// same holdLocked, and the few seconds between the daemon starting and the engine
|
||||
// coming up are a healthy start-up, not a fault. The panel's alarm banner lights on
|
||||
// critical and on nothing else, so spending it here would teach the owner that the
|
||||
// banner means nothing — and the next one, about a node that really cannot be
|
||||
// parsed, is the one they would not read.
|
||||
func TestEngineNotStartedYetIsNotAnAlarm(t *testing.T) {
|
||||
a := New(engine.New(), nil)
|
||||
m := holdModel("closed")
|
||||
|
||||
origHold := applyHoldNft
|
||||
applyHoldNft = func(string) error { return nil }
|
||||
defer func() { applyHoldNft = origHold }()
|
||||
|
||||
a.ArmHold(m)
|
||||
|
||||
w, ok := findFinding(a.Warnings(), "has not started yet")
|
||||
if !ok {
|
||||
t.Fatalf("the boot arm blocks the LAN and must still say why; warnings = %+v", a.Warnings())
|
||||
}
|
||||
if w.Severity != SeverityWarning {
|
||||
t.Errorf("a normal start-up must not spend the critical banner, got %q", w.Severity)
|
||||
}
|
||||
}
|
||||
|
||||
// TestEngineDownCauseClearedOnTeardown: a `stop` is the operator's own decision, and
|
||||
// a start failure left standing would present it as a malfunction for as long as the
|
||||
// process lives to answer status calls.
|
||||
func TestEngineDownCauseClearedOnTeardown(t *testing.T) {
|
||||
a := New(engine.New(), nil)
|
||||
|
||||
origHold, origTeardown, origTable := applyHoldNft, teardownNft, tableExists
|
||||
applyHoldNft = func(string) error { return nil }
|
||||
teardownNft = func() error { return nil }
|
||||
tableExists = func() bool { return false }
|
||||
defer func() { applyHoldNft, teardownNft, tableExists = origHold, origTeardown, origTable }()
|
||||
|
||||
a.mu.Lock()
|
||||
a.holdLocked(holdModel("closed"), errors.New("port 12345 already in use"))
|
||||
a.mu.Unlock()
|
||||
if _, ok := findFinding(a.Warnings(), "could NOT be started"); !ok {
|
||||
t.Fatalf("control failed: the cause was not published, so this test cannot show it cleared")
|
||||
}
|
||||
|
||||
if err := a.Teardown(); err != nil {
|
||||
t.Fatalf("Teardown: %v", err)
|
||||
}
|
||||
if w, ok := findFinding(a.Warnings(), "could NOT be started"); ok {
|
||||
t.Errorf("after a deliberate teardown the engine is down BY REQUEST, not by failure: %+v", w)
|
||||
}
|
||||
}
|
||||
|
||||
// --- defect 2: an unreadable configuration -----------------------------------
|
||||
|
||||
// TestStatusSaysWhenTheConfigCannotBeRead is the defect-2 regression.
|
||||
//
|
||||
// `if m, err := model.ReadUCI(); err == nil { ... }` left Enabled/KillSwitch/
|
||||
// PanelPort at their zero values and recorded NOWHERE that the read had failed. The
|
||||
// panel checks `!status.enabled` before it looks at `plane` and renders "Turned
|
||||
// off" — amber, alarm:false, "turn the service on in Settings" — so the exact
|
||||
// situation the boot armor exists for (a full /overlay, a `uci commit` caught
|
||||
// half-written) came out as the owner's own choice, with the whole LAN cut off and
|
||||
// the suggested remedy pointing at a settings page backed by the same unreadable
|
||||
// file.
|
||||
func TestStatusSaysWhenTheConfigCannotBeRead(t *testing.T) {
|
||||
a := New(engine.New(), nil)
|
||||
|
||||
origRead, origTable := readConfig, tableExists
|
||||
tableExists = func() bool { return true } // a plane IS loaded: the LAN is being held
|
||||
defer func() { readConfig, tableExists = origRead, origTable }()
|
||||
|
||||
// CONTROL — the readable case. Without it, every assertion below is satisfied by
|
||||
// a status that reports "unreadable" unconditionally.
|
||||
m := holdModel("closed")
|
||||
m.Globals.PanelPort = 8443
|
||||
readConfig = func() (*model.Model, error) { return m, nil }
|
||||
s := a.Status()
|
||||
if !s.ConfigReadable {
|
||||
t.Fatalf("control failed: a successful read must report config_readable=true")
|
||||
}
|
||||
if !s.Enabled || s.KillSwitch != "closed" || s.PanelPort != 8443 {
|
||||
t.Fatalf("control failed: a successful read must publish the config: %+v", s)
|
||||
}
|
||||
if s.ConfigError != "" {
|
||||
t.Errorf("config_error must be empty on a successful read, got %q", s.ConfigError)
|
||||
}
|
||||
if w, ok := findFinding(s.Warnings, "configuration could NOT be read"); ok {
|
||||
t.Errorf("a readable config must not warn about being unreadable: %+v", w)
|
||||
}
|
||||
|
||||
// THE DEFECT — the read fails.
|
||||
readConfig = func() (*model.Model, error) { return nil, errors.New("uci export shater: exit status 1") }
|
||||
s = a.Status()
|
||||
|
||||
if s.ConfigReadable {
|
||||
t.Errorf("config_readable = true after a failed read")
|
||||
}
|
||||
if !strings.Contains(s.ConfigError, "exit status 1") {
|
||||
t.Errorf("config_error must carry the reason, got %q", s.ConfigError)
|
||||
}
|
||||
// enabled is still false — it has no honest value to take — so the ONLY thing
|
||||
// standing between the owner and "Turned off" is that this is distinguishable.
|
||||
if s.Enabled {
|
||||
t.Errorf("enabled must not be invented on a failed read")
|
||||
}
|
||||
if s.ConfigReadable == !s.Enabled {
|
||||
// i.e. false == true; guards against a future refactor that makes
|
||||
// config_readable track enabled and stops distinguishing anything.
|
||||
t.Errorf("config_readable must be an independent fact from enabled")
|
||||
}
|
||||
w, ok := findFinding(s.Warnings, "configuration could NOT be read")
|
||||
if !ok {
|
||||
t.Fatalf("an unreadable configuration must be in the warning list — that is the only "+
|
||||
"channel every consumer already renders; warnings = %+v", s.Warnings)
|
||||
}
|
||||
if w.Severity != SeverityCritical {
|
||||
t.Errorf("an unreadable configuration is critical, got %q", w.Severity)
|
||||
}
|
||||
if w.Section != "config" || w.Name != "unreadable" {
|
||||
t.Errorf("the warning must be attributable (section/name), got %q/%q", w.Section, w.Name)
|
||||
}
|
||||
// The one sentence that keeps the owner from "fixing" the fail-closed plane by
|
||||
// switching the protection off.
|
||||
if !strings.Contains(w.Message, "NOT the service being switched off") {
|
||||
t.Errorf("the warning must say the block is not the service being off: %q", w.Message)
|
||||
}
|
||||
|
||||
// And it self-clears: computed at read time, so the next successful read drops it
|
||||
// without anything having to remember to.
|
||||
readConfig = func() (*model.Model, error) { return m, nil }
|
||||
if s = a.Status(); !s.ConfigReadable {
|
||||
t.Errorf("config_readable must go true again the moment the config is readable")
|
||||
}
|
||||
if w, ok := findFinding(s.Warnings, "configuration could NOT be read"); ok {
|
||||
t.Errorf("the unreadable warning must clear itself, still published: %+v", w)
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,197 @@
|
||||
package apply
|
||||
|
||||
// The panel may not tell an operator that traceroute works.
|
||||
//
|
||||
// It did, in four different ways, for as long as the untunnelable notes have
|
||||
// existed: "on Linux and macOS traceroute sends UDP probes instead, which still
|
||||
// follow your rules", "which ARE tunnelled — the hops it prints are the tunnel's
|
||||
// path", and twice "Ping and traceroute work everywhere". Measured on the
|
||||
// production router, a plain `traceroute` from a LAN device prints `* * *` and
|
||||
// nothing else — under every rung of the ladder, `direct` included, with the L3
|
||||
// ingress on or off, with the kill switch open or closed. There is no mechanism
|
||||
// that could print a hop: the UDP probe is delivered LOCALLY to the engine by
|
||||
// tproxy, local delivery is not forwarding, so the TTL is never decremented and
|
||||
// no router on the path is provoked into a `time-exceeded`.
|
||||
//
|
||||
// That makes the old texts the most expensive kind of wrong: an operator who
|
||||
// reads "traceroute works" over a screen of stars goes looking for a fault in
|
||||
// their own network, and there is none to find.
|
||||
//
|
||||
// This file is a RATCHET, not a prose test. It walks every branch of
|
||||
// untunnelablePolicyWarnings and asserts two things about each note:
|
||||
//
|
||||
// 1. the retired sentences never come back, in any branch;
|
||||
// 2. a note that mentions traceroute/tracert at all carries the shared
|
||||
// udpTracerouteFacts verbatim — so a future edit cannot keep the claim and
|
||||
// drop the correction, and cannot fork the wording into a second version.
|
||||
//
|
||||
// The matrix is exhaustive over the four inputs that select a branch, so a new
|
||||
// branch added without the facts fails here rather than shipping.
|
||||
|
||||
import (
|
||||
"strings"
|
||||
"testing"
|
||||
|
||||
"github.com/sagernet/sing-box/shater/model"
|
||||
"github.com/sagernet/sing-box/shater/netplane"
|
||||
)
|
||||
|
||||
// retiredTracerouteClaims are the exact phrases that were false. Each is quoted
|
||||
// from the text it replaced, with the branch it lived in, so a reviewer can see
|
||||
// this list is a record of what was actually said rather than a guess at what
|
||||
// someone might say.
|
||||
var retiredTracerouteClaims = []string{
|
||||
// the untunnelable_egress branch, L3 off
|
||||
"which still follow your rules",
|
||||
// the `block` branch
|
||||
"which ARE tunnelled — the hops it prints are the tunnel's path",
|
||||
// `direct`, `icmp`, and the kill-switch-open note
|
||||
"Ping and traceroute work",
|
||||
"ping, traceroute and raw VPN passthrough",
|
||||
// the shape of the claim, not one phrasing of it: any promise that the
|
||||
// UDP-probe default follows the routing rules or shows a path.
|
||||
"traceroute sends UDP probes instead",
|
||||
}
|
||||
|
||||
// tracerouteMentions is what makes rule 2 above bite. A note that talks about
|
||||
// tracing at all has taken on the duty to say what a plain `traceroute` does.
|
||||
func tracerouteMentions(msg string) bool {
|
||||
return strings.Contains(msg, "traceroute") || strings.Contains(msg, "tracert")
|
||||
}
|
||||
|
||||
// untunnelableCases is the exhaustive cross-product of the inputs that pick a
|
||||
// branch in untunnelablePolicyWarnings: the three policy rungs (plus one
|
||||
// unrecognised value, which EffectiveUntunnelable folds into `block`), the kill
|
||||
// switch, the L3 ingress, and untunnelable_egress.
|
||||
func untunnelableCases() []model.Globals {
|
||||
var out []model.Globals
|
||||
for _, policy := range []string{
|
||||
netplane.UntunnelableBlock,
|
||||
netplane.UntunnelableICMP,
|
||||
netplane.UntunnelableDirect,
|
||||
"", // a config written before the option existed
|
||||
} {
|
||||
for _, kill := range []string{"closed", "open"} {
|
||||
for _, l3 := range []bool{true, false} {
|
||||
for _, egress := range []string{"", "wan2"} {
|
||||
g := model.DefaultGlobals()
|
||||
g.Untunnelable = policy
|
||||
g.KillSwitch = kill
|
||||
g.L3Tunnel = l3
|
||||
g.UntunnelableEgress = egress
|
||||
out = append(out, g)
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
func caseLabel(g model.Globals) string {
|
||||
return "policy=" + netplane.EffectiveUntunnelable(g) +
|
||||
" kill=" + g.KillSwitch +
|
||||
" l3=" + map[bool]string{true: "on", false: "off"}[g.L3Tunnel] +
|
||||
" egress=" + map[bool]string{true: g.UntunnelableEgress, false: "-"}[g.UntunnelableEgress != ""]
|
||||
}
|
||||
|
||||
// TestNoNoteClaimsPlainTracerouteWorks is the ratchet described at the top.
|
||||
func TestNoNoteClaimsPlainTracerouteWorks(t *testing.T) {
|
||||
seenMentions := 0
|
||||
for _, g := range untunnelableCases() {
|
||||
label := caseLabel(g)
|
||||
for _, w := range untunnelablePolicyWarnings(g, nil) {
|
||||
for _, claim := range retiredTracerouteClaims {
|
||||
if strings.Contains(w.Message, claim) {
|
||||
t.Errorf("%s: the untunnelable note says %q again.\n"+
|
||||
"That claim was measured false on the production router: a plain "+
|
||||
"`traceroute` prints `* * *` and no hops under every setting here. "+
|
||||
"Say what udpTracerouteFacts says, or say nothing about traceroute.\n"+
|
||||
"note: %s", label, claim, w.Message)
|
||||
}
|
||||
}
|
||||
if !tracerouteMentions(w.Message) {
|
||||
continue
|
||||
}
|
||||
seenMentions++
|
||||
if !strings.Contains(w.Message, udpTracerouteFacts) {
|
||||
t.Errorf("%s: the untunnelable note talks about tracing but does not carry "+
|
||||
"udpTracerouteFacts.\n"+
|
||||
"A note that mentions traceroute/tracert has taken on the duty to say that "+
|
||||
"the Linux/macOS default prints no hops at all, why (local delivery is not "+
|
||||
"forwarding), and what to use instead (`traceroute -I`). Append the shared "+
|
||||
"constant rather than re-writing it — netplane/untunnelable.go states the "+
|
||||
"same fact and the two must not fork.\nnote: %s", label, w.Message)
|
||||
}
|
||||
}
|
||||
}
|
||||
// Control. Without this, deleting every mention of tracing from every branch
|
||||
// would leave a silent green test that proves nothing — the same instrument
|
||||
// that returns "no lies found" when it cannot see a lie.
|
||||
if seenMentions == 0 {
|
||||
t.Fatal("no untunnelable note mentioned traceroute or tracert in the whole matrix — " +
|
||||
"this test then asserts nothing at all. Either the notes stopped talking about " +
|
||||
"ping diagnostics entirely, or untunnelablePolicyWarnings is no longer being " +
|
||||
"reached from here.")
|
||||
}
|
||||
t.Logf("checked %d notes that mention tracing", seenMentions)
|
||||
}
|
||||
|
||||
// TestUDPTracerouteFactsSayTheThreeThings pins the CONTENT of the shared text,
|
||||
// not just its presence. Without this the constant could be emptied to "" and
|
||||
// every assertion above would still pass — strings.Contains(x, "") is true.
|
||||
func TestUDPTracerouteFactsSayTheThreeThings(t *testing.T) {
|
||||
for _, want := range []string{
|
||||
// it prints nothing — the symptom the operator is staring at
|
||||
"no hops at all",
|
||||
"* * *",
|
||||
// under every setting on this page, so nobody goes hunting the knob
|
||||
"`direct` included",
|
||||
// the cause, short enough to be read
|
||||
"local delivery is not forwarding",
|
||||
// the way out
|
||||
"traceroute -I",
|
||||
} {
|
||||
if !strings.Contains(udpTracerouteFacts, want) {
|
||||
t.Errorf("udpTracerouteFacts no longer contains %q.\n"+
|
||||
"The text has to carry all of: the symptom (no hops, only stars), that it is "+
|
||||
"the same under every rung, the one-clause cause, and the working alternative. "+
|
||||
"Drop any of them and the note stops being an answer.\ntext: %s", want, udpTracerouteFacts)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestL3NoteNamesTheDirectCarriers guards the second correction in this pass.
|
||||
//
|
||||
// The L3 notes used to say ping travels the tunnel toward "an outbound that can
|
||||
// carry plain IP (WireGuard/AmneziaWG)" and that everything else "cannot be
|
||||
// pinged at all". generate/route.go's l3Target is the authority, and its list is
|
||||
// exhaustive by adapter registration: a wireguard/AWG node AND the direct
|
||||
// outbound behind `direct` or an interface/direct egress. In the commonest
|
||||
// configuration here — tunnel the blocked list, send the rest direct — the
|
||||
// second half is most of the address space, and those pings answer out of the
|
||||
// ordinary uplink with its real address. Reading the old text, an operator
|
||||
// concluded either "tunnelled" or "dropped"; it was neither.
|
||||
func TestL3NoteNamesTheDirectCarriers(t *testing.T) {
|
||||
for _, egress := range []string{"", "wan2"} {
|
||||
g := model.DefaultGlobals()
|
||||
g.L3Tunnel = true
|
||||
g.UntunnelableEgress = egress
|
||||
ws := untunnelablePolicyWarnings(g, nil)
|
||||
if len(ws) != 1 {
|
||||
t.Fatalf("egress=%q: expected exactly one untunnelable note, got %d", egress, len(ws))
|
||||
}
|
||||
msg := ws[0].Message
|
||||
if !strings.Contains(msg, "`direct`") || !strings.Contains(msg, "interface egress") {
|
||||
t.Errorf("egress=%q: the L3 note does not name the other outbounds that carry an "+
|
||||
"echo.\nPing also answers for addresses routed `direct` or out an interface "+
|
||||
"egress (generate/route.go l3Target) — and it answers with the real address of "+
|
||||
"that uplink, which is a disclosure the operator is entitled to read here.\n"+
|
||||
"note: %s", egress, msg)
|
||||
}
|
||||
if !strings.Contains(msg, "real address of that uplink") {
|
||||
t.Errorf("egress=%q: the L3 note names the direct carriers but not what they cost: "+
|
||||
"such a ping leaves with the uplink's real address, not the tunnel's.\nnote: %s",
|
||||
egress, msg)
|
||||
}
|
||||
}
|
||||
}
|
||||
+184
-13
@@ -328,6 +328,37 @@ func finalizeWarnings(out []Warning) []Warning {
|
||||
return out
|
||||
}
|
||||
|
||||
// udpTracerouteFacts is the one thing this file is allowed to say about plain
|
||||
// `traceroute`, shared by every branch below so the panel cannot carry two
|
||||
// versions of it.
|
||||
//
|
||||
// It replaces four claims that were false, and false in the most expensive
|
||||
// direction: they told an operator whose trace printed nothing that traceroute
|
||||
// "works", or "still follows your rules", or that the hops it prints "are the
|
||||
// tunnel's path". Measured on the production router, an ordinary `traceroute`
|
||||
// from a LAN device prints `* * *` and nothing else — under every rung of the
|
||||
// untunnelable ladder, `direct` included, with the L3 ingress on or off.
|
||||
//
|
||||
// There is no mechanism by which it could print a hop, which is why the text
|
||||
// says so unconditionally rather than hedging: the UDP probe is diverted by
|
||||
// tproxy and delivered LOCALLY to the engine's socket, and local delivery is not
|
||||
// forwarding — the TTL is never decremented, so no router on the path is ever
|
||||
// provoked into a `time-exceeded`. The engine then opens its OWN connection with
|
||||
// a fresh TTL, and an ICMP error raised against that has no way back to the
|
||||
// client's original datagram; the final `port-unreachable` is absorbed in the
|
||||
// same place. `traceroute -I` and Windows `tracert` are unaffected because they
|
||||
// are ICMP echo, which the L3 ingress (or the `icmp` rung) carries for real.
|
||||
//
|
||||
// netplane/untunnelable.go states the same fact in the same terms where it
|
||||
// prices `block` (the ECHO bullet). One product, one version of this: change
|
||||
// one, change both.
|
||||
const udpTracerouteFacts = "Plain `traceroute` on Linux and macOS is a separate matter, and it reads " +
|
||||
"the same under every setting here, `direct` included: its UDP probes are tunnelled and DO reach " +
|
||||
"the target, but it prints no hops at all, only `* * *`. tproxy delivers each probe locally to the " +
|
||||
"engine, and local delivery is not forwarding, so nothing on the path is ever asked for a " +
|
||||
"`time-exceeded` — there is no fault at your end to go looking for. Use `traceroute -I` (ICMP " +
|
||||
"probes, which is what Windows `tracert` already sends) for a trace that prints hops."
|
||||
|
||||
// untunnelablePolicyWarnings explains, in terms of what the user will actually
|
||||
// experience, what the untunnelable-protocol policy costs them.
|
||||
//
|
||||
@@ -360,6 +391,142 @@ func untunnelablePolicyWarnings(g model.Globals, planNotes []string) []Warning {
|
||||
return append(out, Warning{Severity: SeverityInfo, Section: section, Name: name, Message: msg})
|
||||
}
|
||||
|
||||
// The egress carrier owns the whole story the moment untunnelable_egress
|
||||
// names an egress: the carried protocols get that egress's own mark in
|
||||
// prerouting and the ROUTING decision sends them out its device with the
|
||||
// kernel's NAT — the forward chain, where the policy's verdicts live, no
|
||||
// longer decides their fate. So this branch sits above every other and
|
||||
// returns its own text; with the option empty, the notes below are
|
||||
// byte-for-byte what they were. Two phrasing rules here are load-bearing.
|
||||
// First, the egress is NEVER called a tunnel unconditionally: the option
|
||||
// accepts any interface/tunnel egress, and on the routers this ships to
|
||||
// that is at least as often a second WAN — another uplink, whose real
|
||||
// address the far end sees — as a WireGuard device. Second, the note must
|
||||
// say out loud that UDP-based VPNs are none of this option's business: an
|
||||
// operator who reads "VPN passthrough" and enables it for a WireGuard
|
||||
// client that already worked through the ordinary tunnel has been misled,
|
||||
// not helped. What the policy still owns is exactly the failure path — a
|
||||
// name that resolves to no interface/tunnel egress, a rule or route that
|
||||
// did not come up — and each tail below says what that failure looks like,
|
||||
// because under `direct` (or an open kill switch) it is a silent leak
|
||||
// through the normal uplink with the real address, and nothing anywhere
|
||||
// else would say so.
|
||||
if egressName := strings.TrimSpace(g.UntunnelableEgress); egressName != "" {
|
||||
var msg, failure string
|
||||
if netplane.L3Enabled(g) {
|
||||
msg = "Ping and Windows tracert keep travelling THROUGH the tunnel, toward every " +
|
||||
"address your rules send to a WireGuard/AmneziaWG node — the L3 ingress " +
|
||||
"claims ICMP before this option is consulted. Addresses your rules send " +
|
||||
"`direct`, or out an interface egress, answer as well, but those pings " +
|
||||
"leave the way that traffic does, carrying the real address of that uplink " +
|
||||
"rather than the tunnel's; addresses your rules send anywhere else " +
|
||||
"(vless/vmess/trojan/shadowsocks and the like) still cannot be pinged at " +
|
||||
"all, deliberately. " + udpTracerouteFacts + " Everything else the proxy cannot carry — IPsec " +
|
||||
"(ESP/AH), PPTP/GRE, SCTP and every other protocol that is neither TCP " +
|
||||
"nor UDP — now leaves through egress \"" + egressName + "\": the kernel " +
|
||||
"routes it out that interface with that interface's own NAT, and none of " +
|
||||
"it goes through the proxy or follows your routing rules. "
|
||||
failure = "when one of the routes is not in place — the L3 route for ping, the " +
|
||||
"egress route for the rest (a name that matches no interface/tunnel " +
|
||||
"egress, or a rule or route that failed to come up): "
|
||||
} else {
|
||||
msg = "Ping, Windows tracert, IPsec (ESP/AH), PPTP/GRE, SCTP and every other " +
|
||||
"protocol that is neither TCP nor UDP now leave through egress \"" +
|
||||
egressName + "\": the kernel routes them out that interface with that " +
|
||||
"interface's own NAT, and none of it goes through the proxy or follows " +
|
||||
"your routing rules — the hops Windows tracert and `traceroute -I` print " +
|
||||
"are that interface's path. " + udpTracerouteFacts + " "
|
||||
failure = "when the egress route is not in place (a name that matches no " +
|
||||
"interface/tunnel egress, or a rule or route that failed to come up): "
|
||||
}
|
||||
msg += "What that buys depends entirely on what the interface IS: a WireGuard " +
|
||||
"interface really is a tunnel, but a second WAN is not — it is just another " +
|
||||
"uplink, and the host on the far end sees that uplink's real address. Two " +
|
||||
"things this option does NOT do: multicast IPTV does not pass this router " +
|
||||
"under any setting, and carrying IGMP out an egress cannot change that; and " +
|
||||
"VPNs that run over UDP (WireGuard, OpenVPN-UDP, IPsec through NAT — IKE on " +
|
||||
"UDP 500, NAT-T on UDP 4500) never needed it: they are ordinary tunnelled " +
|
||||
"traffic, keep following your routing rules exactly as before, and gain " +
|
||||
"nothing from this option. The `untunnelable` policy no longer decides this " +
|
||||
"traffic's fate — routing settles it before the forward chain gets a say — " +
|
||||
"and answers only for failure, " + failure
|
||||
switch {
|
||||
case !killSwitchClosed(g):
|
||||
msg += "with the kill switch open nothing is dropped, so whatever loses its " +
|
||||
"route quietly leaves through your normal uplink with your real IP address."
|
||||
case policy == netplane.UntunnelableDirect:
|
||||
msg += "\"direct\" quietly lets it leave through your normal uplink with " +
|
||||
"your real IP address."
|
||||
case policy == netplane.UntunnelableICMP:
|
||||
msg += "\"icmp\" drops it, excepting only ping — which then quietly leaves " +
|
||||
"with your real IP address instead of failing."
|
||||
default:
|
||||
msg += "\"block\" drops it — an honest loss rather than a silent leak."
|
||||
}
|
||||
return note(egressName, msg)
|
||||
}
|
||||
|
||||
// The L3 ingress rewrites the ICMP half of every note below, so it gets one
|
||||
// text of its own rather than four patched variants: echo is marked in
|
||||
// prerouting and the ROUTING decision carries it into the engine's TUN before
|
||||
// the forward chain — where the policy accepts and the kill-switch drops
|
||||
// live — is ever consulted. That holds under all three policy values and
|
||||
// with the kill switch open alike, which is why this branch sits above the
|
||||
// kill-switch note: "reaches the internet with your real IP address" stops
|
||||
// being true for ping the moment the divert exists. What the policy still
|
||||
// owns is exactly two things, and both are said: the protocols the engine
|
||||
// cannot ingest at all (raw IPsec, PPTP/GRE), and the fallback path a marked
|
||||
// packet takes when the L3 route failed to install — under `direct` (or an
|
||||
// open kill switch) that failure is a SILENT leak with the real address,
|
||||
// under `block` an honest packet loss. The unpingable-through-proxy sentence
|
||||
// is deliberate too: those pings used to be answered by the router itself,
|
||||
// and a fake "alive" is worse than a truthful timeout.
|
||||
//
|
||||
// The set of outbounds that carry an echo is NOT "WireGuard/AmneziaWG", and
|
||||
// saying so was an understatement that hid a disclosure. generate/route.go's
|
||||
// l3Target is the authority and its list is exhaustive by adapter
|
||||
// registration: a wireguard/AWG node, AND the direct outbound behind `direct`
|
||||
// or an interface/direct egress. In the commonest configuration on this router
|
||||
// — tunnel the blocked list, send the rest direct — that second half is most
|
||||
// of the address space, and those pings do answer, out of the ordinary uplink
|
||||
// with its real address. An operator who read the old text concluded either
|
||||
// "tunnelled" or "dropped", and neither was what their ping was doing.
|
||||
if netplane.L3Enabled(g) {
|
||||
msg := "Ping and Windows tracert work and travel THROUGH the tunnel, toward every " +
|
||||
"address your rules send to a WireGuard/AmneziaWG node. Addresses your rules " +
|
||||
"send `direct`, or out an interface egress, answer as well — but those pings " +
|
||||
"leave the way that traffic does, carrying the real address of that uplink " +
|
||||
"rather than the tunnel's. Addresses your rules send anywhere else " +
|
||||
"(vless/vmess/trojan/shadowsocks and the like) cannot be pinged at all — " +
|
||||
"deliberately: those pings used to be answered by the router itself, reporting " +
|
||||
"hosts alive it had never reached. The hops Windows tracert and `traceroute -I` " +
|
||||
"print are the tunnel's path, not your own; over IPv6 that same trace shows only " +
|
||||
"the destination and none of the hops on the way. " + udpTracerouteFacts +
|
||||
" Raw VPN passthrough (IPsec ESP/AH, PPTP/GRE) cannot " +
|
||||
"enter the tunnel at all and stays with the untunnelable policy: "
|
||||
switch {
|
||||
case !killSwitchClosed(g):
|
||||
msg += "with the kill switch open none of it is dropped, so it leaves with your " +
|
||||
"real IP address — and if the L3 route ever fails to come up, ping quietly " +
|
||||
"does the same instead of failing."
|
||||
case policy == netplane.UntunnelableDirect:
|
||||
msg += "\"direct\" lets it out with your real IP address — and if the L3 route " +
|
||||
"ever fails to come up, ping quietly does the same instead of failing."
|
||||
case policy == netplane.UntunnelableICMP:
|
||||
msg += "\"icmp\" drops it, excepting only echo — which now rides the tunnel " +
|
||||
"anyway, so the exception matters just once: if the L3 route ever fails to " +
|
||||
"come up, it lets ping quietly leave with your real IP address instead of " +
|
||||
"failing."
|
||||
default:
|
||||
msg += "\"block\" drops it — and if the L3 route ever fails to come up, ping " +
|
||||
"fails outright rather than leaking."
|
||||
}
|
||||
msg += " VPNs that run over UDP (WireGuard, OpenVPN-UDP, IPsec through NAT) are " +
|
||||
"ordinary tunnelled traffic and are unaffected either way. Multicast IPTV does " +
|
||||
"not pass this router on any setting; the L3 ingress does not change that."
|
||||
return note(policy, msg)
|
||||
}
|
||||
|
||||
// With the kill switch open the forward chain has no drops at all, so nothing
|
||||
// is restricted whatever the policy says. Saying that is more useful than
|
||||
// repeating a promise which is not being kept.
|
||||
@@ -377,26 +544,30 @@ func untunnelablePolicyWarnings(g model.Globals, planNotes []string) []Warning {
|
||||
}
|
||||
return note(policy,
|
||||
"This setting has no effect while the kill switch is open: with the kill switch open the "+
|
||||
"forward chain has no drops at all, so ping, traceroute and raw VPN passthrough "+
|
||||
"(IPsec ESP/AH, PPTP/GRE) all work — and every one of them reaches the internet with "+
|
||||
"your real IP address. IPTV is not part of that: multicast does not pass this router "+
|
||||
"forward chain has no drops at all, so ping, an ICMP trace (`traceroute -I`, Windows "+
|
||||
"`tracert`) and raw VPN passthrough (IPsec ESP/AH, PPTP/GRE) all work — and every one "+
|
||||
"of them reaches the internet with your real IP address. "+udpTracerouteFacts+
|
||||
" IPTV is not part of that: multicast does not pass this router "+
|
||||
"on any setting, which is a separate matter from this one.")
|
||||
}
|
||||
|
||||
switch policy {
|
||||
case netplane.UntunnelableDirect:
|
||||
return note(policy,
|
||||
"Ping and traceroute work everywhere, and so does raw VPN passthrough (IPsec ESP/AH, "+
|
||||
"Ping works everywhere, and so do an ICMP trace (`traceroute -I`, Windows `tracert`) and "+
|
||||
"raw VPN passthrough (IPsec ESP/AH, "+
|
||||
"PPTP/GRE) — but all of it goes straight out with your real IP address instead of "+
|
||||
"through the tunnel, because a tunnel cannot carry this kind of traffic. VPNs that "+
|
||||
"through the tunnel, because a tunnel cannot carry this kind of traffic. "+
|
||||
udpTracerouteFacts+" VPNs that "+
|
||||
"run over UDP (WireGuard, OpenVPN-UDP, IPsec through NAT) are ordinary tunnelled "+
|
||||
"traffic and are unaffected either way. IPTV is not covered by this setting at all: "+
|
||||
"multicast does not pass this router on any of the three, so switching to `direct` "+
|
||||
"will not bring it back.")
|
||||
case netplane.UntunnelableICMP:
|
||||
return note(policy,
|
||||
"Ping and traceroute work everywhere, including addresses you send through the tunnel; "+
|
||||
"the host you ping sees your real IP address. IPTV and VPN passthrough (IPsec/PPTP) "+
|
||||
"Ping works everywhere, including addresses you send through the tunnel, and so does an "+
|
||||
"ICMP trace (`traceroute -I`, Windows `tracert`); the host you ping sees your real IP "+
|
||||
"address. "+udpTracerouteFacts+" IPTV and VPN passthrough (IPsec/PPTP) "+
|
||||
"work only toward addresses your rules route directly.")
|
||||
default:
|
||||
// This text used to say these things "work only toward addresses your rules
|
||||
@@ -409,13 +580,13 @@ func untunnelablePolicyWarnings(g model.Globals, planNotes []string) []Warning {
|
||||
// through NAT is the difference between "my VPN broke" and "my VPN is fine".
|
||||
return note(netplane.UntunnelableBlock,
|
||||
"Ping, traceroute, IPsec/PPTP VPN passthrough and IPTV do not work from your devices at "+
|
||||
"all — not even toward addresses your rules route directly. None of this traffic can "+
|
||||
"travel through a tunnel, so rather than let it out with your real IP address it is "+
|
||||
"dropped. Concretely: ping and Windows tracert fail (on Linux and macOS traceroute "+
|
||||
"sends UDP probes instead, which ARE tunnelled — the hops it prints are the tunnel's "+
|
||||
"path, not your own), and so do raw IPsec (ESP/AH) and PPTP/GRE — a PPTP session will "+
|
||||
"all — not even toward addresses your rules route directly. None of the traffic this "+
|
||||
"policy decides can travel through a tunnel, so rather than let it out with your real IP address it is "+
|
||||
"dropped. Concretely: ping and every ICMP trace (`traceroute -I`, Windows `tracert`) "+
|
||||
"fail, and so do raw IPsec (ESP/AH) and PPTP/GRE — a PPTP session will "+
|
||||
"even look connected, because its control channel is TCP and only the payload is "+
|
||||
"dropped. VPNs that run over UDP are NOT affected: WireGuard, OpenVPN-UDP and IPsec "+
|
||||
"dropped. "+udpTracerouteFacts+
|
||||
" VPNs that run over UDP are NOT affected: WireGuard, OpenVPN-UDP and IPsec "+
|
||||
"through NAT (IKE on UDP 500, NAT-T on UDP 4500) keep working normally. Multicast "+
|
||||
"IPTV does not cross this router under any setting; that one is not this policy.")
|
||||
}
|
||||
|
||||
@@ -139,9 +139,35 @@ func TestCollectWarningsCapKeepsCriticals(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
// withReadableConfig pins the one outside fact this file's warning-COUNT
|
||||
// assertions depend on: whether Status could read the router's configuration.
|
||||
//
|
||||
// It is here because of a defect this test used to have, of the same family as the
|
||||
// silent skip. The assertion "a fresh Status carries no warnings" was true only by
|
||||
// accident of environment: a build host has no `uci`, model.ReadUCI failed on every
|
||||
// Status() call, and that failure was SILENT, so it contributed nothing to count.
|
||||
// On a router, where `uci` exists and the config reads fine, it was also zero — for
|
||||
// the opposite reason. The test therefore never stated which world it was in, and
|
||||
// the moment the unreadable-config failure started publishing a critical warning
|
||||
// (apply.go, "a finding that is still true may not erase itself") the same source
|
||||
// line meant two different things in the two environments.
|
||||
//
|
||||
// A package-level init() in standing_state_test.go now supplies a router-like
|
||||
// baseline for the whole package, which is the right home for a package-wide
|
||||
// fixture. This helper is NOT a duplicate of it: a test that counts warnings must
|
||||
// not depend on any ambient baseline at all, whoever set it up and whether or not
|
||||
// it is still there tomorrow. It says what it needs, in its own body.
|
||||
func withReadableConfig(t *testing.T) {
|
||||
t.Helper()
|
||||
orig := readConfig
|
||||
readConfig = func() (*model.Model, error) { return &model.Model{Globals: model.DefaultGlobals()}, nil }
|
||||
t.Cleanup(func() { readConfig = orig })
|
||||
}
|
||||
|
||||
// TestStatusWarningsAlwaysNonNil: the panel maps over this array unconditionally,
|
||||
// so it must serialise as [] and never null.
|
||||
func TestStatusWarningsAlwaysNonNil(t *testing.T) {
|
||||
withReadableConfig(t) // counting warnings requires knowing which world we are in
|
||||
a := New(engine.New(), nil)
|
||||
s := a.Status()
|
||||
if s.Warnings == nil {
|
||||
@@ -216,8 +242,19 @@ func TestWarningsAgainstRealGenerateOutput(t *testing.T) {
|
||||
if err != nil {
|
||||
t.Fatalf("GenerateWithWarnings: %v", err)
|
||||
}
|
||||
// NOT a skip. This model is built to be unloadable — a binary rule-set at a
|
||||
// path that does not exist, and a url rule-set with no url — and both are
|
||||
// protection the operator configured that is not in force. If generate stops
|
||||
// saying so, the panel goes back to being indistinguishable from "not
|
||||
// configured" and the user believes ad blocking is on when it is not: that is
|
||||
// the W7 regression itself, not a reason to stand this test down. (The
|
||||
// predecessor of this line was a t.Skip, in the same family as the one that
|
||||
// retired TestApplyInstallsHoldWhenEngineFailsToStart.)
|
||||
if len(genWarnings) == 0 {
|
||||
t.Skip("generate produced no warnings for this model; nothing to classify")
|
||||
t.Fatal("generate produced NO warning for a model whose blocklists cannot load " +
|
||||
"(/nonexistent/ads.srs does not exist, and the second list has an empty url) — an " +
|
||||
"unloadable blocklist must always be reported, or nothing in the UI distinguishes " +
|
||||
"it from one that is working")
|
||||
}
|
||||
t.Logf("real generate warnings: %q", genWarnings)
|
||||
|
||||
|
||||
@@ -0,0 +1,294 @@
|
||||
// The verdict half of `shaterd apply`.
|
||||
//
|
||||
// `apply` is not "push the config" — the daemon reconciles anyway, from SIGHUP,
|
||||
// from the cron/hotplug `shaterd reconcile`, from the panel, and from its own
|
||||
// startup. What `apply` ADDS, and the only reason to type it, is the safety net:
|
||||
// snapshot the current last-good, apply, and arm an automatic rollback so a change
|
||||
// that costs you access to the router undoes itself.
|
||||
//
|
||||
// THE DEFECT THIS FILE EXISTS FOR (hit on the live router, 2026-07-26). The verb
|
||||
// answered `{"changed":false}` and nothing else. That reads as "all good, nothing
|
||||
// to do". It was not: the operator had edited UCI and run `uci commit`, the
|
||||
// `config.change` reload trigger had already restarted the daemon, and the fresh
|
||||
// daemon had applied the new config on startup. By the time `apply` ran there was
|
||||
// nothing left to apply — and, worse, the last-good it snapshotted as the ROLLBACK
|
||||
// TARGET was the newly applied config itself. So the auto-rollback was armed onto
|
||||
// the very configuration it was supposed to protect against: firing it would have
|
||||
// restored exactly what was already loaded. The safety net was absent, the output
|
||||
// said nothing about it, and the house went down.
|
||||
//
|
||||
// So this file answers ONE question in words, on every apply: IS THERE A SAFETY
|
||||
// NET, AND IF NOT, WHY NOT. It does not build a new one — that is separate work.
|
||||
//
|
||||
// The classification is a pure function of facts the daemon already has, so the
|
||||
// whole vocabulary is testable on a dev host with no router, no root and no nft.
|
||||
package main
|
||||
|
||||
import (
|
||||
"crypto/sha256"
|
||||
"encoding/hex"
|
||||
"encoding/json"
|
||||
"fmt"
|
||||
"os"
|
||||
"time"
|
||||
|
||||
"github.com/sagernet/sing-box/shater/model"
|
||||
)
|
||||
|
||||
// uciConfigPath is the file `uci commit shater` rewrites. A var so tests can point
|
||||
// the mtime probe at a fixture.
|
||||
var uciConfigPath = "/etc/config/shater"
|
||||
|
||||
// daemonStarted is when THIS process began.
|
||||
//
|
||||
// Round(0) strips the monotonic reading on purpose: the only comparison made with
|
||||
// it is against a FILE mtime, which is wall-clock-only, so keeping a monotonic
|
||||
// reading here would just make the comparison silently fall back to the wall clock
|
||||
// anyway. Being explicit is worth a word.
|
||||
//
|
||||
// The router has no RTC, so this instant is captured before NTP steps the clock
|
||||
// (typically forward, by years). That bias is in the recoverable direction for the
|
||||
// one use below: a start instant that reads OLDER than it was can only make an
|
||||
// edit look like it happened after the daemon came up — which, for any edit made
|
||||
// after boot, it did.
|
||||
var daemonStarted = time.Now().Round(0)
|
||||
|
||||
// applyVerdict is the `apply` verb's answer on the control socket.
|
||||
//
|
||||
// `changed` and `error` are unchanged from the old ctlResult, so every existing
|
||||
// consumer keeps working; the rest is the part that was missing. RollbackArmed is
|
||||
// the load-bearing field: it is the answer to "if this just broke my network, will
|
||||
// anything undo it?".
|
||||
type applyVerdict struct {
|
||||
Changed bool `json:"changed"`
|
||||
Error string `json:"error,omitempty"`
|
||||
|
||||
// RollbackArmed is true ONLY when an automatic rollback was armed AND it would
|
||||
// take the router somewhere other than where it already is. An armed watcher
|
||||
// whose target is the running configuration is not a safety net, and is not
|
||||
// reported as one.
|
||||
RollbackArmed bool `json:"rollback_armed"`
|
||||
|
||||
// Reason is a closed vocabulary (the reason* constants). Closed and positive on
|
||||
// purpose: an unenumerated outcome must not fall into a bucket that reads
|
||||
// reassuring.
|
||||
Reason string `json:"reason"`
|
||||
|
||||
// Message says the same thing in the operator's words. Never empty.
|
||||
Message string `json:"message"`
|
||||
|
||||
// ConfirmTimeout is the configured commit-confirm window in seconds, echoed so
|
||||
// the answer carries the setting it depends on (0 = the feature is off).
|
||||
ConfirmTimeout int `json:"confirm_timeout"`
|
||||
}
|
||||
|
||||
// The closed reason vocabulary.
|
||||
const (
|
||||
// reasonApplyFailed — the apply returned an error; nothing was armed.
|
||||
reasonApplyFailed = "apply-failed"
|
||||
// reasonConfigUnreadable — the apply worked but /etc/config/shater could not be
|
||||
// re-read to learn the confirm timeout, so ArmRollback was never called.
|
||||
reasonConfigUnreadable = "config-unreadable"
|
||||
// reasonCommitConfirmOff — globals.confirm_timeout is 0. This is the SHIPPED
|
||||
// DEFAULT, so on a stock box it is the usual answer.
|
||||
reasonCommitConfirmOff = "commit-confirm-off"
|
||||
// reasonDisabled — globals.enabled=0, so the reconcile tore the plane down.
|
||||
reasonDisabled = "disabled"
|
||||
// reasonApplied — the running configuration moved and a real rollback target
|
||||
// was recorded. The only outcome where the net exists.
|
||||
reasonApplied = "applied"
|
||||
// reasonAlreadyApplied — nothing moved, and the config file was edited AFTER
|
||||
// this daemon started: something else applied it before this command ran.
|
||||
reasonAlreadyApplied = "already-applied"
|
||||
// reasonNothingToApply — nothing moved, and there is no evidence either way
|
||||
// about whether the config was edited (a daemon restart erases it).
|
||||
reasonNothingToApply = "nothing-to-apply"
|
||||
)
|
||||
|
||||
// applyFacts is everything one `apply` learned, as plain values.
|
||||
type applyFacts struct {
|
||||
// engineChanged is the Applier's own `changed` — TRUE only when the ENGINE
|
||||
// config moved. It is deliberately not used to decide whether a rollback target
|
||||
// is meaningful: a netplane-only change (the kill-switch flipping, a rule
|
||||
// gaining a schedule window) hashes identical for the engine and still moves
|
||||
// the router. before/after do that job.
|
||||
engineChanged bool
|
||||
err error
|
||||
|
||||
// uciRead is whether /etc/config/shater could be re-read after the apply — the
|
||||
// read that supplies confirmTimeout and gates ArmRollback.
|
||||
uciRead bool
|
||||
enabled bool
|
||||
confirmTimeout int
|
||||
|
||||
// before is the last-good model at the instant of Snapshot(), i.e. the rollback
|
||||
// target this apply recorded. after is the last-good once the apply finished,
|
||||
// i.e. what is running now. nil means "no successful apply is on record".
|
||||
before *model.Model
|
||||
after *model.Model
|
||||
|
||||
// configMTime is the mtime of /etc/config/shater, daemonStart is when this
|
||||
// process began. Zero means unknown, and unknown must never be read as "no".
|
||||
configMTime time.Time
|
||||
daemonStart time.Time
|
||||
}
|
||||
|
||||
// classifyApply turns the facts into the verdict. Every branch sets Reason AND
|
||||
// Message, and every branch that leaves RollbackArmed false says so in words: a
|
||||
// silent `{"changed":false}` is the exact failure this replaces.
|
||||
func classifyApply(f applyFacts) applyVerdict {
|
||||
v := applyVerdict{Changed: f.engineChanged, ConfirmTimeout: f.confirmTimeout}
|
||||
|
||||
switch {
|
||||
case f.err != nil:
|
||||
v.Error = f.err.Error()
|
||||
v.Reason = reasonApplyFailed
|
||||
v.Message = fmt.Sprintf("apply FAILED (%v) and armed NO automatic rollback. "+
|
||||
"Run `shaterd status`: a failed apply can leave the engine down with the "+
|
||||
"fail-closed plane holding the LAN.", f.err)
|
||||
|
||||
case !f.uciRead:
|
||||
v.Reason = reasonConfigUnreadable
|
||||
v.Message = "the apply itself succeeded, but /etc/config/shater could not be re-read " +
|
||||
"afterwards, so the commit-confirm window was never armed. There is NO automatic " +
|
||||
"rollback for what was just applied."
|
||||
|
||||
case f.confirmTimeout <= 0:
|
||||
v.Reason = reasonCommitConfirmOff
|
||||
v.Message = "globals.confirm_timeout is 0, so commit-confirm is switched OFF: this apply " +
|
||||
"armed NO automatic rollback and nothing will undo it if it cost you access to the " +
|
||||
"router. Arm it with `uci set shater.@globals[0].confirm_timeout=<seconds>`."
|
||||
if sameConfig(f.before, f.after) {
|
||||
v.Message += " Nothing was applied either — the running configuration already " +
|
||||
"equals /etc/config/shater."
|
||||
}
|
||||
|
||||
case !f.enabled:
|
||||
// The reconcile tore the plane down. The rollback target is the configuration
|
||||
// that was running BEFORE, so firing it would switch shater back ON — worth
|
||||
// saying out loud, because "apply" after a disable looks like a no-op.
|
||||
if f.before == nil {
|
||||
v.Reason = reasonDisabled
|
||||
v.Message = "shater is disabled (globals.enabled=0) and the data plane was torn down. " +
|
||||
"No earlier configuration is on record, so NO automatic rollback is armed."
|
||||
break
|
||||
}
|
||||
v.RollbackArmed = true
|
||||
v.Reason = reasonDisabled
|
||||
v.Message = fmt.Sprintf("shater is disabled (globals.enabled=0) and the data plane was torn "+
|
||||
"down. An automatic rollback is armed: in %d s the PREVIOUS configuration is re-applied "+
|
||||
"— i.e. shater switches back on — unless you run `shaterd confirm`.", f.confirmTimeout)
|
||||
|
||||
case !sameConfig(f.before, f.after):
|
||||
v.RollbackArmed = true
|
||||
v.Reason = reasonApplied
|
||||
if f.before == nil {
|
||||
v.Message = fmt.Sprintf("applied. No earlier configuration is on record (this is the "+
|
||||
"first apply this daemon completed), so the automatic rollback in %d s REMOVES the "+
|
||||
"shater data plane entirely (safe teardown) unless you run `shaterd confirm`.",
|
||||
f.confirmTimeout)
|
||||
break
|
||||
}
|
||||
v.Message = fmt.Sprintf("applied. An automatic rollback to the previous configuration is "+
|
||||
"armed for %d s — run `shaterd confirm` to keep this one.", f.confirmTimeout)
|
||||
|
||||
case configEditedSinceStart(f):
|
||||
v.Reason = reasonAlreadyApplied
|
||||
v.Message = fmt.Sprintf("nothing was applied: /etc/config/shater was last modified %s, AFTER "+
|
||||
"this daemon started %s, and the running configuration ALREADY matches it — so something "+
|
||||
"other than this command applied it (the `config.change` reload trigger, `shaterd "+
|
||||
"reconcile` from cron/hotplug, the panel, or SIGHUP). NO automatic rollback is armed: the "+
|
||||
"rollback target recorded here is the configuration that is already running, so nothing "+
|
||||
"can undo that change.",
|
||||
stampUTC(f.configMTime), stampUTC(f.daemonStart))
|
||||
|
||||
default:
|
||||
v.Reason = reasonNothingToApply
|
||||
v.Message = "nothing was applied: the running configuration already equals " +
|
||||
"/etc/config/shater. NO automatic rollback is armed — the rollback target recorded here " +
|
||||
"IS the running configuration, so firing it would restore exactly what is loaded now. " +
|
||||
"That is harmless if you changed nothing. If you DID edit the config, it was applied " +
|
||||
"before this command ran (a `uci commit` fires the reload trigger, which restarts the " +
|
||||
"daemon, and the fresh daemon applies on startup) — and then that change is running with " +
|
||||
"no safety net. shaterd cannot tell those two cases apart across a daemon restart."
|
||||
}
|
||||
return v
|
||||
}
|
||||
|
||||
// configEditedSinceStart reports whether /etc/config/shater was written after this
|
||||
// daemon process began. It is the ONE reliable discriminator between "you changed
|
||||
// nothing" and "something else applied your change": an edit landing during the
|
||||
// daemon's lifetime, with the running config already matching it, can only mean a
|
||||
// reconcile beat this command to it.
|
||||
//
|
||||
// It is deliberately one-directional. False does NOT mean "nothing was edited" —
|
||||
// the `uci commit` reload trigger RESTARTS the daemon, which moves daemonStart
|
||||
// past the edit — which is why the false branch says so instead of claiming
|
||||
// everything is fine.
|
||||
func configEditedSinceStart(f applyFacts) bool {
|
||||
if f.configMTime.IsZero() || f.daemonStart.IsZero() {
|
||||
return false
|
||||
}
|
||||
return f.configMTime.After(f.daemonStart)
|
||||
}
|
||||
|
||||
// sameConfig reports whether two models are the same desired state — i.e. whether
|
||||
// a rollback to `a` would leave the router where `b` already has it.
|
||||
//
|
||||
// Compared over the WHOLE model, not the engine's option hash: a change the engine
|
||||
// hashes identical (kill-switch, divert set, DNS intercept) still moves the router
|
||||
// and still deserves a real rollback target.
|
||||
func sameConfig(a, b *model.Model) bool {
|
||||
if a == nil || b == nil {
|
||||
return a == nil && b == nil
|
||||
}
|
||||
return configHash(a) == configHash(b)
|
||||
}
|
||||
|
||||
// configHash is a content hash of a model. encoding/json sorts map keys, so it is
|
||||
// stable across runs. A marshal failure yields a UNIQUE value rather than a shared
|
||||
// sentinel: two configs that could not be hashed must not be reported as equal,
|
||||
// because "equal" is the answer that says the safety net is missing.
|
||||
func configHash(m *model.Model) string {
|
||||
b, err := json.Marshal(m)
|
||||
if err != nil {
|
||||
return fmt.Sprintf("unhashable-%p", m)
|
||||
}
|
||||
sum := sha256.Sum256(b)
|
||||
return hex.EncodeToString(sum[:])
|
||||
}
|
||||
|
||||
// stampUTC formats an instant for the operator. UTC, like every other timestamp
|
||||
// this daemon prints (the box has no tzdata).
|
||||
func stampUTC(t time.Time) string { return t.UTC().Format("2006-01-02 15:04:05 UTC") }
|
||||
|
||||
// applyNoticeLine renders the operator-facing stderr line for a control-socket
|
||||
// reply, or "" when there is nothing to add. Only the `apply` verb has anything to
|
||||
// say here; every other verb keeps its old, quiet output.
|
||||
//
|
||||
// It is the CLI half of the same honesty rule: when no rollback was armed the line
|
||||
// LEADS with that fact, because the reader of a terminal scans the first words.
|
||||
func applyNoticeLine(cmd, resp string) string {
|
||||
if cmd != "apply" {
|
||||
return ""
|
||||
}
|
||||
var v applyVerdict
|
||||
if json.Unmarshal([]byte(resp), &v) != nil || v.Message == "" {
|
||||
return ""
|
||||
}
|
||||
if v.RollbackArmed {
|
||||
return "shaterd apply: " + v.Message
|
||||
}
|
||||
return "shaterd apply: NO AUTOMATIC ROLLBACK — " + v.Message
|
||||
}
|
||||
|
||||
// uciConfigMTime returns the mtime of /etc/config/shater, or the zero time when it
|
||||
// cannot be stat'ed (which classifyApply treats as "unknown", never as "not
|
||||
// edited").
|
||||
func uciConfigMTime() time.Time {
|
||||
st, err := os.Stat(uciConfigPath)
|
||||
if err != nil {
|
||||
return time.Time{}
|
||||
}
|
||||
return st.ModTime()
|
||||
}
|
||||
@@ -0,0 +1,466 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"encoding/json"
|
||||
"errors"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"github.com/sagernet/sing-box/shater/model"
|
||||
)
|
||||
|
||||
// --- fixtures ---------------------------------------------------------------
|
||||
|
||||
var (
|
||||
fixtureStart = time.Date(2026, 7, 26, 12, 0, 0, 0, time.UTC)
|
||||
beforeStart = fixtureStart.Add(-1 * time.Hour)
|
||||
afterStart = fixtureStart.Add(1 * time.Minute)
|
||||
)
|
||||
|
||||
// cfg builds a distinguishable model. killSwitch is varied rather than an engine
|
||||
// field on purpose: it is a change the ENGINE hashes identical, so a test that
|
||||
// separates cfg("open") from cfg("closed") also proves the comparison does not
|
||||
// lean on the engine's own `changed` flag.
|
||||
func cfg(killSwitch string) *model.Model {
|
||||
m := &model.Model{Globals: model.DefaultGlobals()}
|
||||
m.Globals.KillSwitch = killSwitch
|
||||
m.Nodes = []model.Node{{Name: "n1", Enabled: true, URI: "vless://example"}}
|
||||
return m
|
||||
}
|
||||
|
||||
// armedFacts is the healthy baseline: enabled, a 30 s confirm window, config read.
|
||||
func armedFacts() applyFacts {
|
||||
return applyFacts{
|
||||
uciRead: true,
|
||||
enabled: true,
|
||||
confirmTimeout: 30,
|
||||
before: cfg("closed"),
|
||||
after: cfg("closed"),
|
||||
configMTime: beforeStart,
|
||||
daemonStart: fixtureStart,
|
||||
}
|
||||
}
|
||||
|
||||
// --- the two cases the verb must tell apart ---------------------------------
|
||||
|
||||
// TestApplyReportsSomethingElseAlreadyApplied is the live-router regression.
|
||||
//
|
||||
// Sequence on mini_router (2026-07-26): `uci set` + `uci commit`, which fires the
|
||||
// `config.change` reload trigger, which restarts the daemon / a cron reconcile
|
||||
// picks it up — either way the new config is already running by the time `shaterd
|
||||
// apply` is typed. The verb answered `{"changed":false}`, which reads as "all
|
||||
// good", while the rollback target it had just snapshotted WAS the newly applied
|
||||
// config. There was no safety net and nothing said so.
|
||||
//
|
||||
// Here the discriminator is available: the file was written after the daemon
|
||||
// started, so the reconcile that applied it can only have been someone else's.
|
||||
func TestApplyReportsSomethingElseAlreadyApplied(t *testing.T) {
|
||||
f := armedFacts()
|
||||
f.engineChanged = false
|
||||
f.configMTime = afterStart // edited while this daemon was already up
|
||||
|
||||
v := classifyApply(f)
|
||||
|
||||
if v.RollbackArmed {
|
||||
t.Fatalf("rollback_armed = true, but the rollback target is the config that is "+
|
||||
"already running — nothing could be undone. verdict: %+v", v)
|
||||
}
|
||||
if v.Reason != reasonAlreadyApplied {
|
||||
t.Errorf("reason = %q, want %q", v.Reason, reasonAlreadyApplied)
|
||||
}
|
||||
mustSay(t, v.Message,
|
||||
"NO automatic rollback", // the missing safety net, in words
|
||||
"ALREADY matches", // WHY nothing was applied
|
||||
"other than this command",
|
||||
)
|
||||
// The evidence the operator needs to believe it must be in the sentence.
|
||||
mustSay(t, v.Message, stampUTC(afterStart), stampUTC(fixtureStart))
|
||||
}
|
||||
|
||||
// TestApplyUnchangedIsHonestAboutTheAmbiguity covers the other case: nothing moved
|
||||
// and there is NO evidence either way, because the `uci commit` reload trigger is
|
||||
// stop+start — it moves the daemon's start past the edit, erasing the mtime
|
||||
// discriminator.
|
||||
//
|
||||
// The verb must not read as "all good". It must say the net is absent, that this
|
||||
// is harmless if nothing was changed, AND that a change applied before the command
|
||||
// ran is running unprotected. A formally-true sentence that reads as success is
|
||||
// the same lie in a nicer suit.
|
||||
func TestApplyUnchangedIsHonestAboutTheAmbiguity(t *testing.T) {
|
||||
f := armedFacts()
|
||||
f.engineChanged = false
|
||||
f.configMTime = beforeStart // the file predates this daemon: no evidence
|
||||
|
||||
v := classifyApply(f)
|
||||
|
||||
if v.RollbackArmed {
|
||||
t.Fatalf("rollback_armed = true with target == running config: %+v", v)
|
||||
}
|
||||
if v.Reason != reasonNothingToApply {
|
||||
t.Errorf("reason = %q, want %q", v.Reason, reasonNothingToApply)
|
||||
}
|
||||
mustSay(t, v.Message,
|
||||
"NO automatic rollback",
|
||||
"harmless if you changed nothing", // the benign half, named as conditional
|
||||
"no safety net", // the dangerous half, named
|
||||
"cannot tell", // the ambiguity, admitted
|
||||
)
|
||||
}
|
||||
|
||||
// TestApplyArmsWhenTheConfigActuallyMoved is the CONTROL for both tests above: the
|
||||
// same instrument must be able to report a real safety net, or "no net" proves
|
||||
// nothing. (Engineering standard: a negative result needs a positive control.)
|
||||
func TestApplyArmsWhenTheConfigActuallyMoved(t *testing.T) {
|
||||
f := armedFacts()
|
||||
f.engineChanged = true
|
||||
f.before = cfg("open")
|
||||
f.after = cfg("closed")
|
||||
|
||||
v := classifyApply(f)
|
||||
|
||||
if !v.RollbackArmed {
|
||||
t.Fatalf("rollback_armed = false, but the rollback target differs from the "+
|
||||
"running config — the net is real here: %+v", v)
|
||||
}
|
||||
if v.Reason != reasonApplied {
|
||||
t.Errorf("reason = %q, want %q", v.Reason, reasonApplied)
|
||||
}
|
||||
mustSay(t, v.Message, "30 s", "shaterd confirm")
|
||||
}
|
||||
|
||||
// TestApplyArmsOnNetplaneOnlyChange pins that the verdict does not trust the
|
||||
// engine's `changed`. A kill-switch flip hashes identical for the engine
|
||||
// (option.Options are unchanged) yet reloads the nft ruleset and really can take
|
||||
// the LAN off the air — exactly the apply that most needs a rollback.
|
||||
func TestApplyArmsOnNetplaneOnlyChange(t *testing.T) {
|
||||
f := armedFacts()
|
||||
f.engineChanged = false // the engine hash did not move
|
||||
f.before = cfg("open")
|
||||
f.after = cfg("closed")
|
||||
|
||||
v := classifyApply(f)
|
||||
|
||||
if !v.RollbackArmed || v.Reason != reasonApplied {
|
||||
t.Fatalf("a netplane-only change must still arm a rollback; got %+v", v)
|
||||
}
|
||||
if v.Changed {
|
||||
t.Errorf("changed must keep reporting the ENGINE flag verbatim (false here), got true")
|
||||
}
|
||||
}
|
||||
|
||||
// --- the other ways the net is silently absent ------------------------------
|
||||
|
||||
// TestApplyConfirmTimeoutZeroIsNotSilent covers the SHIPPED DEFAULT:
|
||||
// openwrt/shater-core/files/etc/config/shater sets `confirm_timeout '0'`, and
|
||||
// apply.ArmRollback returns immediately for a non-positive timeout. So on a stock
|
||||
// box every `apply` arms nothing at all, and used to say `{"changed":true}`.
|
||||
func TestApplyConfirmTimeoutZeroIsNotSilent(t *testing.T) {
|
||||
f := armedFacts()
|
||||
f.engineChanged = true
|
||||
f.confirmTimeout = 0
|
||||
f.before = cfg("open")
|
||||
f.after = cfg("closed")
|
||||
|
||||
v := classifyApply(f)
|
||||
|
||||
if v.RollbackArmed {
|
||||
t.Fatalf("confirm_timeout=0 arms nothing (ArmRollback returns early); got %+v", v)
|
||||
}
|
||||
if v.Reason != reasonCommitConfirmOff {
|
||||
t.Errorf("reason = %q, want %q", v.Reason, reasonCommitConfirmOff)
|
||||
}
|
||||
mustSay(t, v.Message, "confirm_timeout", "NO automatic rollback")
|
||||
if v.ConfirmTimeout != 0 {
|
||||
t.Errorf("confirm_timeout echoed as %d, want 0", v.ConfirmTimeout)
|
||||
}
|
||||
}
|
||||
|
||||
// TestApplyUnreadableConfigIsNotSilent: the daemon arms the window from a SECOND
|
||||
// model.ReadUCI after the reconcile. When that read fails ArmRollback is never
|
||||
// called — a hole that produced a plain success answer.
|
||||
func TestApplyUnreadableConfigIsNotSilent(t *testing.T) {
|
||||
f := armedFacts()
|
||||
f.engineChanged = true
|
||||
f.uciRead = false
|
||||
f.before = cfg("open")
|
||||
f.after = cfg("closed")
|
||||
|
||||
v := classifyApply(f)
|
||||
|
||||
if v.RollbackArmed || v.Reason != reasonConfigUnreadable {
|
||||
t.Fatalf("an unreadable config after a successful apply arms nothing; got %+v", v)
|
||||
}
|
||||
mustSay(t, v.Message, "NO automatic", "could not be re-read")
|
||||
}
|
||||
|
||||
// TestApplyFailureSaysNothingWasArmed: on an error the daemon skips ArmRollback
|
||||
// entirely, and a failed apply is precisely when an operator assumes a net exists.
|
||||
func TestApplyFailureSaysNothingWasArmed(t *testing.T) {
|
||||
f := armedFacts()
|
||||
f.err = errors.New("engine: address already in use")
|
||||
|
||||
v := classifyApply(f)
|
||||
|
||||
if v.RollbackArmed || v.Reason != reasonApplyFailed {
|
||||
t.Fatalf("a failed apply arms nothing; got %+v", v)
|
||||
}
|
||||
if v.Error != "engine: address already in use" {
|
||||
t.Errorf("error = %q, want the applier's error verbatim", v.Error)
|
||||
}
|
||||
mustSay(t, v.Message, "NO automatic rollback")
|
||||
}
|
||||
|
||||
// TestApplyFirstEverArmsATeardown: with no last-good on record the rollback target
|
||||
// is nil, which apply.rollbackTo turns into a SAFE TEARDOWN. That is a real net —
|
||||
// but a surprising one, so the words must name it.
|
||||
func TestApplyFirstEverArmsATeardown(t *testing.T) {
|
||||
f := armedFacts()
|
||||
f.engineChanged = true
|
||||
f.before = nil
|
||||
f.after = cfg("closed")
|
||||
|
||||
v := classifyApply(f)
|
||||
|
||||
if !v.RollbackArmed || v.Reason != reasonApplied {
|
||||
t.Fatalf("a first-ever apply still arms a rollback (safe teardown); got %+v", v)
|
||||
}
|
||||
mustSay(t, v.Message, "REMOVES the shater data plane")
|
||||
}
|
||||
|
||||
// TestApplyDisabledSaysTheRollbackTurnsItBackOn: `apply` after globals.enabled=0
|
||||
// tears the plane down and returns changed=false, yet the armed rollback re-applies
|
||||
// the previous (enabled) config. Reported as a no-op, that is a trap.
|
||||
func TestApplyDisabledSaysTheRollbackTurnsItBackOn(t *testing.T) {
|
||||
f := armedFacts()
|
||||
f.enabled = false
|
||||
f.engineChanged = false
|
||||
|
||||
v := classifyApply(f)
|
||||
|
||||
if !v.RollbackArmed || v.Reason != reasonDisabled {
|
||||
t.Fatalf("a disable leaves a real rollback target armed; got %+v", v)
|
||||
}
|
||||
mustSay(t, v.Message, "switches back on", "shaterd confirm")
|
||||
}
|
||||
|
||||
// TestApplyDisabledWithNoHistoryArmsNothing: same path, but nothing was ever
|
||||
// applied, so the "rollback" is a teardown of a plane that is already down.
|
||||
func TestApplyDisabledWithNoHistoryArmsNothing(t *testing.T) {
|
||||
f := armedFacts()
|
||||
f.enabled = false
|
||||
f.before = nil
|
||||
f.after = nil
|
||||
|
||||
v := classifyApply(f)
|
||||
|
||||
if v.RollbackArmed || v.Reason != reasonDisabled {
|
||||
t.Fatalf("no history means nothing to roll back to; got %+v", v)
|
||||
}
|
||||
mustSay(t, v.Message, "NO automatic rollback")
|
||||
}
|
||||
|
||||
// --- invariants across the whole vocabulary ---------------------------------
|
||||
|
||||
// TestEveryVerdictSpeaks is the anti-silence gate: whatever the facts, the answer
|
||||
// carries a reason from the closed vocabulary and a non-empty message, and any
|
||||
// answer WITHOUT a rollback says so unmistakably. `{"changed":false}` with nothing
|
||||
// else must be unreachable.
|
||||
func TestEveryVerdictSpeaks(t *testing.T) {
|
||||
known := map[string]bool{
|
||||
reasonApplyFailed: true, reasonConfigUnreadable: true, reasonCommitConfirmOff: true,
|
||||
reasonDisabled: true, reasonApplied: true, reasonAlreadyApplied: true,
|
||||
reasonNothingToApply: true,
|
||||
}
|
||||
cases := map[string]applyFacts{}
|
||||
for _, uciRead := range []bool{true, false} {
|
||||
for _, enabled := range []bool{true, false} {
|
||||
for _, timeout := range []int{0, 30} {
|
||||
for _, mtime := range []time.Time{{}, beforeStart, afterStart} {
|
||||
for i, pair := range [][2]*model.Model{
|
||||
{nil, nil}, {nil, cfg("closed")},
|
||||
{cfg("closed"), cfg("closed")}, {cfg("open"), cfg("closed")},
|
||||
} {
|
||||
for _, e := range []error{nil, errors.New("boom")} {
|
||||
f := applyFacts{
|
||||
engineChanged: i == 3, err: e, uciRead: uciRead, enabled: enabled,
|
||||
confirmTimeout: timeout, before: pair[0], after: pair[1],
|
||||
configMTime: mtime, daemonStart: fixtureStart,
|
||||
}
|
||||
name := strings.Join([]string{
|
||||
boolName(uciRead), boolName(enabled), stampUTC(mtime),
|
||||
}, "/") + "/" + string(rune('a'+i))
|
||||
if e != nil {
|
||||
name += "/err"
|
||||
}
|
||||
if timeout == 0 {
|
||||
name += "/t0"
|
||||
}
|
||||
cases[name] = f
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
for name, f := range cases {
|
||||
v := classifyApply(f)
|
||||
if !known[v.Reason] {
|
||||
t.Errorf("%s: reason %q is outside the closed vocabulary", name, v.Reason)
|
||||
}
|
||||
if strings.TrimSpace(v.Message) == "" {
|
||||
t.Errorf("%s: empty message — this is the silent {\"changed\":false} defect", name)
|
||||
}
|
||||
if !v.RollbackArmed && !strings.Contains(strings.ToLower(v.Message), "no automatic rollback") {
|
||||
t.Errorf("%s: rollback_armed=false but the message never says so: %q", name, v.Message)
|
||||
}
|
||||
if v.RollbackArmed && strings.Contains(strings.ToLower(v.Message), "no automatic rollback") {
|
||||
t.Errorf("%s: rollback_armed=true but the message denies it: %q", name, v.Message)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestVerdictNeverClaimsANetOverAnIdenticalTarget is the single load-bearing
|
||||
// invariant, asserted independently of the branch order: a rollback target that
|
||||
// equals the running config can NEVER be reported as armed, because firing it
|
||||
// would restore what is already loaded.
|
||||
func TestVerdictNeverClaimsANetOverAnIdenticalTarget(t *testing.T) {
|
||||
for _, timeout := range []int{0, 1, 30, 3600} {
|
||||
for _, mtime := range []time.Time{{}, beforeStart, afterStart} {
|
||||
f := armedFacts()
|
||||
f.confirmTimeout = timeout
|
||||
f.configMTime = mtime
|
||||
f.before, f.after = cfg("closed"), cfg("closed")
|
||||
if v := classifyApply(f); v.RollbackArmed {
|
||||
t.Errorf("timeout=%d mtime=%v: armed over an identical target: %+v", timeout, mtime, v)
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// --- the comparison instrument itself ---------------------------------------
|
||||
|
||||
// TestSameConfigDetectsBothWays: a comparison that always says "equal" would make
|
||||
// every apply look unprotected, and one that always says "different" would make
|
||||
// every apply look safe. Both directions are pinned.
|
||||
func TestSameConfigDetectsBothWays(t *testing.T) {
|
||||
if !sameConfig(cfg("closed"), cfg("closed")) {
|
||||
t.Errorf("identical models must compare equal")
|
||||
}
|
||||
if sameConfig(cfg("open"), cfg("closed")) {
|
||||
t.Errorf("a kill-switch difference must compare different (the engine hashes it identical)")
|
||||
}
|
||||
a := cfg("closed")
|
||||
b := cfg("closed")
|
||||
b.Nodes[0].URI = "vless://other"
|
||||
if sameConfig(a, b) {
|
||||
t.Errorf("a node URI difference must compare different")
|
||||
}
|
||||
if !sameConfig(nil, nil) {
|
||||
t.Errorf("nil/nil is the same state (nothing applied)")
|
||||
}
|
||||
if sameConfig(nil, cfg("closed")) || sameConfig(cfg("closed"), nil) {
|
||||
t.Errorf("nil vs a model must compare different")
|
||||
}
|
||||
}
|
||||
|
||||
// --- wire contract ----------------------------------------------------------
|
||||
|
||||
// TestApplyVerdictWire pins the JSON the control socket emits: `changed` survives
|
||||
// for existing consumers, and the new fields are always present (rollback_armed is
|
||||
// NOT omitempty — a missing field would read as "unknown", and this answer is the
|
||||
// one place that must not be ambiguous).
|
||||
func TestApplyVerdictWire(t *testing.T) {
|
||||
f := armedFacts()
|
||||
f.configMTime = afterStart
|
||||
b, err := json.Marshal(classifyApply(f))
|
||||
if err != nil {
|
||||
t.Fatalf("marshal: %v", err)
|
||||
}
|
||||
var got map[string]any
|
||||
if err := json.Unmarshal(b, &got); err != nil {
|
||||
t.Fatalf("unmarshal: %v", err)
|
||||
}
|
||||
for _, k := range []string{"changed", "rollback_armed", "reason", "message", "confirm_timeout"} {
|
||||
if _, ok := got[k]; !ok {
|
||||
t.Errorf("key %q missing from the apply reply: %s", k, b)
|
||||
}
|
||||
}
|
||||
if got["rollback_armed"] != false {
|
||||
t.Errorf("rollback_armed = %v, want false", got["rollback_armed"])
|
||||
}
|
||||
}
|
||||
|
||||
// TestApplyNoticeLine pins the terminal line. It must LEAD with the missing net —
|
||||
// the operator scans the first words — and must stay silent for other verbs.
|
||||
func TestApplyNoticeLine(t *testing.T) {
|
||||
unarmed, _ := json.Marshal(applyVerdict{Reason: reasonAlreadyApplied, Message: "already applied."})
|
||||
line := applyNoticeLine("apply", string(unarmed))
|
||||
if !strings.HasPrefix(line, "shaterd apply: NO AUTOMATIC ROLLBACK") {
|
||||
t.Errorf("unarmed notice = %q, want it to lead with the missing rollback", line)
|
||||
}
|
||||
armed, _ := json.Marshal(applyVerdict{RollbackArmed: true, Reason: reasonApplied, Message: "applied."})
|
||||
line = applyNoticeLine("apply", string(armed))
|
||||
if strings.Contains(line, "NO AUTOMATIC") || !strings.Contains(line, "applied.") {
|
||||
t.Errorf("armed notice = %q, want the plain message", line)
|
||||
}
|
||||
if got := applyNoticeLine("confirm", string(unarmed)); got != "" {
|
||||
t.Errorf("applyNoticeLine(confirm) = %q, want silence for other verbs", got)
|
||||
}
|
||||
if got := applyNoticeLine("apply", "not json"); got != "" {
|
||||
t.Errorf("applyNoticeLine on garbage = %q, want silence", got)
|
||||
}
|
||||
}
|
||||
|
||||
// --- the mtime probe --------------------------------------------------------
|
||||
|
||||
// TestUCIConfigMTime: a real stat, and an absent file reported as UNKNOWN (zero),
|
||||
// which configEditedSinceStart must never read as "not edited".
|
||||
func TestUCIConfigMTime(t *testing.T) {
|
||||
dir := t.TempDir()
|
||||
path := filepath.Join(dir, "shater")
|
||||
if err := os.WriteFile(path, []byte("config globals\n"), 0o644); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
orig := uciConfigPath
|
||||
uciConfigPath = path
|
||||
defer func() { uciConfigPath = orig }()
|
||||
|
||||
if got := uciConfigMTime(); got.IsZero() {
|
||||
t.Errorf("mtime of an existing file must not be zero")
|
||||
}
|
||||
uciConfigPath = filepath.Join(dir, "absent")
|
||||
if got := uciConfigMTime(); !got.IsZero() {
|
||||
t.Errorf("mtime of an absent file = %v, want the zero time (unknown)", got)
|
||||
}
|
||||
// Unknown must fall to the CAUTIOUS branch, not to "already applied".
|
||||
f := armedFacts()
|
||||
f.configMTime = time.Time{}
|
||||
if configEditedSinceStart(f) {
|
||||
t.Errorf("an unknown mtime must not be reported as an edit")
|
||||
}
|
||||
if v := classifyApply(f); v.Reason != reasonNothingToApply {
|
||||
t.Errorf("unknown mtime: reason = %q, want the honest-ambiguity branch %q",
|
||||
v.Reason, reasonNothingToApply)
|
||||
}
|
||||
}
|
||||
|
||||
// --- helpers ----------------------------------------------------------------
|
||||
|
||||
func mustSay(t *testing.T, msg string, phrases ...string) {
|
||||
t.Helper()
|
||||
for _, p := range phrases {
|
||||
if !strings.Contains(msg, p) {
|
||||
t.Errorf("message does not contain %q:\n %s", p, msg)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func boolName(b bool) string {
|
||||
if b {
|
||||
return "y"
|
||||
}
|
||||
return "n"
|
||||
}
|
||||
+172
-18
@@ -16,7 +16,8 @@
|
||||
// status | nodes | stats read-side JSON for LuCI/panel
|
||||
// blocklist update force a DNS-filter refresh (reconcile; no-op if down)
|
||||
// schedule due re-evaluate time-scheduled rules now (reconcile; no-op if down)
|
||||
// sub update | ruleset update Phase-2b no-op stubs (cron-safe)
|
||||
// sub update [<name>] fetch + re-cache subscription nodes (real; exits 1 on failure)
|
||||
// ruleset update NOT IMPLEMENTED — exits non-zero, does nothing
|
||||
//
|
||||
// See docs-shater/PORTING.md "Wave 3 daemon contract" and DECISIONS D11/D12.
|
||||
package main
|
||||
@@ -51,7 +52,11 @@ import (
|
||||
"github.com/sagernet/sing-box/shater/subscribe"
|
||||
)
|
||||
|
||||
const pidfilePath = "/var/run/shaterd.pid"
|
||||
// pidfilePath is where the daemon publishes its pid and where every other verb
|
||||
// looks for it. A var, not a const, for the same reason ctlPath is one: the
|
||||
// daemon-reachable / daemon-absent split is a contract worth testing, and a test
|
||||
// must be able to stage both without writing to the real /var/run.
|
||||
var pidfilePath = "/var/run/shaterd.pid"
|
||||
|
||||
func main() { os.Exit(dispatch(os.Args[1:])) }
|
||||
|
||||
@@ -130,13 +135,30 @@ usage: shaterd <verb>
|
||||
schedule due re-evaluate time-scheduled rules now (reconcile; no-op if down)
|
||||
alert test send a test notification to every enabled alert (no daemon needed)
|
||||
sub update [<name>] fetch + re-cache subscription nodes (all enabled, or one named)
|
||||
ruleset update (Phase 2b) not yet implemented
|
||||
ruleset update NOT IMPLEMENTED — exits non-zero without updating anything
|
||||
`)
|
||||
}
|
||||
|
||||
// notImplExitCode is what an unimplemented verb exits with.
|
||||
//
|
||||
// It is NOT 0, and that is the entire point. `ruleset update` used to print a
|
||||
// note on stderr and exit 0, which made every caller that checks the exit status
|
||||
// believe the work was done: /etc/init.d/shater-cron runs it as
|
||||
// `"$SHATERD" ruleset update "$name" >/dev/null 2>&1` and, on success, STAMPS the
|
||||
// item as freshly updated and sets changed=1 — so every `config ruleset` with
|
||||
// source=url was permanently reported up to date by a verb that never fetched a
|
||||
// byte, and each stamp also triggered a reconcile with an unchanged config.
|
||||
//
|
||||
// It is also NOT 2: `dispatch` reserves 2 for "I do not know this verb" (usage),
|
||||
// and a caller must be able to tell a typo from a verb that exists but does not
|
||||
// work yet.
|
||||
const notImplExitCode = 1
|
||||
|
||||
func notImpl(verb string) int {
|
||||
fmt.Fprintf(os.Stderr, "shaterd: %s not yet implemented (Phase 2b)\n", verb)
|
||||
return 0
|
||||
fmt.Fprintf(os.Stderr, "shaterd: %s is NOT implemented — nothing was updated. "+
|
||||
"Remote rule-sets are refreshed by the engine itself "+
|
||||
"(ruleset.update_interval on the running box), not by this verb.\n", verb)
|
||||
return notImplExitCode
|
||||
}
|
||||
|
||||
// --- run: THE daemon --------------------------------------------------------
|
||||
@@ -868,11 +890,84 @@ func cmdScheduleDue() int {
|
||||
return 0
|
||||
}
|
||||
|
||||
// statusDaemonAnsweredKey is the ONE field of `shaterd status` that says, BY
|
||||
// CONTRACT, where the object under it came from:
|
||||
//
|
||||
// true — a running daemon answered over the control socket; every other field
|
||||
// is that daemon's own Applier.Status().
|
||||
// false — no daemon answered. What follows is the OFFLINE STUB: the apply.Status
|
||||
// zero value plus what could be read from UCI and from the kernel. It is
|
||||
// NOT a status report, and the fields that only a live daemon can know
|
||||
// (running/engine_running/plane/traffic/hash/warnings/uptime) are
|
||||
// placeholders, not measurements.
|
||||
//
|
||||
// It exists because the stub used to be indistinguishable from an answer: it is
|
||||
// the same struct, printed by the same marshaller, and `shaterd status` exited 0
|
||||
// either way. The one thing that happened to differ was `plane` being "" — a
|
||||
// value a live Applier.Status() cannot produce because it always assigns one of
|
||||
// three words — and luci-app-shater's dashboard.js was forced to key its
|
||||
// "daemon: down" verdict off exactly that. Detection by a side effect is not a
|
||||
// contract: filling `plane` in the stub, for any reason, would silently turn
|
||||
// "the daemon is dead" into "the daemon is fine" on the LuCI page.
|
||||
//
|
||||
// The field is ADDITIVE. Every pre-existing key keeps its name, its value and
|
||||
// its position, so no current consumer breaks; `plane: ""` in particular is
|
||||
// still emitted, deliberately, so dashboard.js keeps working until it is moved
|
||||
// onto this field.
|
||||
const statusDaemonAnsweredKey = "daemon_answered"
|
||||
|
||||
// markStatusOrigin prefixes a status JSON object with the daemon_answered
|
||||
// verdict, leaving every other key untouched and in place.
|
||||
//
|
||||
// It splices rather than re-marshals on purpose: the live branch relays the
|
||||
// DAEMON's own words, and a decode/encode round-trip would silently reorder
|
||||
// them and drop any field this build does not know about (a newer daemon's, an
|
||||
// older CLI's). The input is validated as a JSON object first, so the splice is
|
||||
// well defined; a reply that is not an object is an error, never something we
|
||||
// print unmarked.
|
||||
func markStatusOrigin(obj string, answered bool) (string, error) {
|
||||
t := strings.TrimSpace(obj)
|
||||
var probe map[string]json.RawMessage
|
||||
if err := json.Unmarshal([]byte(t), &probe); err != nil {
|
||||
return "", fmt.Errorf("status reply is not a JSON object: %w", err)
|
||||
}
|
||||
if _, dup := probe[statusDaemonAnsweredKey]; dup {
|
||||
return "", fmt.Errorf("status reply already carries %q", statusDaemonAnsweredKey)
|
||||
}
|
||||
field := `"` + statusDaemonAnsweredKey + `": ` + strconv.FormatBool(answered)
|
||||
rest := t[1:] // Unmarshal succeeded and t is trimmed, so t[0] is '{'
|
||||
switch {
|
||||
case strings.HasPrefix(rest, "\n"):
|
||||
return "{\n " + field + "," + rest, nil // apply.Status.JSON() is indented
|
||||
case strings.TrimSpace(rest) == "}":
|
||||
return "{" + field + "}", nil // the empty object
|
||||
default:
|
||||
return "{" + field + "," + rest, nil
|
||||
}
|
||||
}
|
||||
|
||||
// cmdStatus prints the daemon's status as JSON.
|
||||
//
|
||||
// Three outcomes, and a caller can tell them apart without guessing:
|
||||
//
|
||||
// exit 0 — a daemon answered; stdout carries its status with
|
||||
// "daemon_answered": true.
|
||||
// exit 1 + stdout — no daemon answered; stdout carries the OFFLINE STUB with
|
||||
// "daemon_answered": false (see statusDaemonAnsweredKey).
|
||||
// exit 1 + NO stdout — the process is alive but wedged. Printing the stub here
|
||||
// would assert running=false about a daemon that is running, and would
|
||||
// hide a data plane that is very probably still installed and still
|
||||
// enforcing.
|
||||
func cmdStatus() int {
|
||||
if _, ok := daemonAlive(); ok {
|
||||
resp, err := ctlRequest("status")
|
||||
if err == nil {
|
||||
fmt.Println(strings.TrimSpace(resp))
|
||||
out, merr := markStatusOrigin(resp, true)
|
||||
if merr != nil {
|
||||
fmt.Fprintf(os.Stderr, "shaterd status: %v\n", merr)
|
||||
return 1
|
||||
}
|
||||
fmt.Println(out)
|
||||
return 0
|
||||
}
|
||||
// The daemon PROCESS exists but did not answer in time. Falling through to
|
||||
@@ -886,7 +981,13 @@ func cmdStatus() int {
|
||||
return 1
|
||||
}
|
||||
}
|
||||
// Offline stub: the daemon isn't reachable, so running=false / hash="".
|
||||
// OFFLINE STUB. No daemon answered, so the only honest fields are the ones
|
||||
// read from the kernel and from UCI right here; everything else is the
|
||||
// apply.Status zero value and is marked as such by daemon_answered=false.
|
||||
//
|
||||
// `plane` stays "" on purpose: it is the signal luci-app-shater currently
|
||||
// keys "daemon: down" off, and dropping it would break that page silently.
|
||||
// It is now a COMPATIBILITY carry-over, not the contract.
|
||||
s := apply.Status{
|
||||
Running: false,
|
||||
Active: apply.ActiveFlagPresent(),
|
||||
@@ -902,8 +1003,18 @@ func cmdStatus() int {
|
||||
}
|
||||
}
|
||||
b, _ := s.JSON()
|
||||
fmt.Println(string(b))
|
||||
return 0
|
||||
out, merr := markStatusOrigin(string(b), false)
|
||||
if merr != nil {
|
||||
fmt.Fprintf(os.Stderr, "shaterd status: %v\n", merr)
|
||||
return 1
|
||||
}
|
||||
// stdout keeps carrying a parseable object (the rpcd plugin reads stdout and
|
||||
// ignores the exit status), but the exit code no longer reports success for a
|
||||
// status nobody produced.
|
||||
fmt.Println(out)
|
||||
fmt.Fprintln(os.Stderr, "shaterd status: the daemon did not answer — the object on "+
|
||||
"stdout is the offline stub (daemon_answered=false), not a status report.")
|
||||
return 1
|
||||
}
|
||||
|
||||
// cmdNodes prints the node inventory as a JSON array (see nodes.go for the
|
||||
@@ -1052,15 +1163,8 @@ func handleCtl(conn net.Conn, a *apply.Applier, ps *panel.Server, sa stats.Stats
|
||||
}
|
||||
writeResult(conn, ch, err)
|
||||
case "apply":
|
||||
a.Snapshot()
|
||||
changed, err := a.Reconcile()
|
||||
if m, e := model.ReadUCI(); e == nil {
|
||||
if err == nil {
|
||||
a.ArmRollback(m.Globals.ConfirmTimeout)
|
||||
}
|
||||
refreshBootArmor(m, l)
|
||||
}
|
||||
writeResult(conn, changed, err)
|
||||
b, _ := json.Marshal(runApplyVerb(a, l))
|
||||
writeLine(conn, string(b))
|
||||
case "confirm":
|
||||
writeResult(conn, false, a.Confirm())
|
||||
case "rollback":
|
||||
@@ -1105,6 +1209,49 @@ func handleCtl(conn net.Conn, a *apply.Applier, ps *panel.Server, sa stats.Stats
|
||||
}
|
||||
}
|
||||
|
||||
// runApplyVerb is the daemon side of `shaterd apply`: snapshot the rollback
|
||||
// target, reconcile, arm the commit-confirm window — and then REPORT whether a
|
||||
// safety net actually exists. The arming behaviour is byte-for-byte what it was
|
||||
// (Snapshot, Reconcile, ArmRollback only on success and only with a readable UCI,
|
||||
// refreshBootArmor either way); what is new is that the answer says what happened.
|
||||
// See applyverb.go for why silence here cost a house's connectivity.
|
||||
func runApplyVerb(a *apply.Applier, l log.ContextLogger) applyVerdict {
|
||||
// The rollback target this apply is about to record. Snapshot() reads the same
|
||||
// LastGood() under the apply mutex; reading it here first is the only way to see
|
||||
// it, and the two reads cannot disagree in practice — an Apply holds that mutex
|
||||
// for its whole (multi-second) duration, so both calls either precede it or
|
||||
// follow it.
|
||||
before := a.LastGood()
|
||||
a.Snapshot()
|
||||
changed, err := a.Reconcile()
|
||||
f := applyFacts{
|
||||
engineChanged: changed,
|
||||
err: err,
|
||||
before: before,
|
||||
after: a.LastGood(),
|
||||
configMTime: uciConfigMTime(),
|
||||
daemonStart: daemonStarted,
|
||||
}
|
||||
if m, e := model.ReadUCI(); e == nil {
|
||||
f.uciRead = true
|
||||
f.enabled = m.Globals.Enabled
|
||||
f.confirmTimeout = m.Globals.ConfirmTimeout
|
||||
if err == nil {
|
||||
a.ArmRollback(m.Globals.ConfirmTimeout)
|
||||
}
|
||||
refreshBootArmor(m, l)
|
||||
}
|
||||
v := classifyApply(f)
|
||||
// The CLI answer is read once; logread is what an incident is reconstructed
|
||||
// from. A missing safety net is a WARNING there, not an info line.
|
||||
if v.RollbackArmed {
|
||||
l.Info("apply (", v.Reason, "): ", v.Message)
|
||||
} else {
|
||||
l.Warn("apply (", v.Reason, "): NO automatic rollback armed: ", v.Message)
|
||||
}
|
||||
return v
|
||||
}
|
||||
|
||||
func writeResult(conn net.Conn, changed bool, err error) {
|
||||
r := ctlResult{Changed: changed}
|
||||
if err != nil {
|
||||
@@ -1141,6 +1288,13 @@ func ctlMutate(cmd string, requireDaemon bool) int {
|
||||
return 1
|
||||
}
|
||||
fmt.Println(strings.TrimSpace(resp))
|
||||
// `apply` answers with more than {changed,error}: it says whether a
|
||||
// commit-confirm window was actually armed. That is the entire reason the verb
|
||||
// exists, so it is also said in plain words on stderr — a one-line JSON object
|
||||
// on stdout is exactly what let a missing safety net pass for a healthy apply.
|
||||
if line := applyNoticeLine(cmd, resp); line != "" {
|
||||
fmt.Fprintln(os.Stderr, line)
|
||||
}
|
||||
return 0
|
||||
}
|
||||
|
||||
|
||||
@@ -0,0 +1,284 @@
|
||||
//go:build linux
|
||||
|
||||
package main
|
||||
|
||||
import (
|
||||
"bufio"
|
||||
"encoding/json"
|
||||
"io"
|
||||
"net"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"strconv"
|
||||
"strings"
|
||||
"testing"
|
||||
)
|
||||
|
||||
// What this file pins
|
||||
//
|
||||
// 1. `shaterd status` must let its caller tell "a daemon answered" from "no
|
||||
// daemon answered" BY CONTRACT. Until now the offline stub was the same
|
||||
// struct printed by the same marshaller with exit 0, and the only thing that
|
||||
// happened to differ was `plane` being "" — a side effect luci-app-shater was
|
||||
// forced to key its "daemon down" verdict off.
|
||||
// 2. `shaterd ruleset update` must not report success for work it does not do.
|
||||
// shater-cron stamps the item as freshly updated on a zero exit.
|
||||
|
||||
// --- helpers ---------------------------------------------------------------
|
||||
|
||||
// captureStdout runs fn with os.Stdout redirected and returns what it printed.
|
||||
func captureStdout(t *testing.T, fn func() int) (string, int) {
|
||||
t.Helper()
|
||||
r, w, err := os.Pipe()
|
||||
if err != nil {
|
||||
t.Fatalf("pipe: %v", err)
|
||||
}
|
||||
orig := os.Stdout
|
||||
os.Stdout = w
|
||||
done := make(chan string, 1)
|
||||
go func() {
|
||||
b, _ := io.ReadAll(r)
|
||||
done <- string(b)
|
||||
}()
|
||||
code := fn()
|
||||
_ = w.Close()
|
||||
os.Stdout = orig
|
||||
out := <-done
|
||||
_ = r.Close()
|
||||
return out, code
|
||||
}
|
||||
|
||||
// noDaemon points the pid lookup at a path that cannot name a live process, so
|
||||
// cmdStatus takes the offline branch. Restored by t.Cleanup.
|
||||
func noDaemon(t *testing.T) {
|
||||
t.Helper()
|
||||
orig := pidfilePath
|
||||
pidfilePath = filepath.Join(t.TempDir(), "absent.pid")
|
||||
t.Cleanup(func() { pidfilePath = orig })
|
||||
}
|
||||
|
||||
// liveDaemon stages a reachable daemon: a pidfile naming THIS process (so
|
||||
// daemonAlive's kill(pid,0) probe succeeds) and a control socket that answers a
|
||||
// single `status` request with reply. Restored by t.Cleanup.
|
||||
func liveDaemon(t *testing.T, reply string) {
|
||||
t.Helper()
|
||||
dir := t.TempDir()
|
||||
|
||||
origPid := pidfilePath
|
||||
pidfilePath = filepath.Join(dir, "shaterd.pid")
|
||||
if err := os.WriteFile(pidfilePath, []byte(strconv.Itoa(os.Getpid())+"\n"), 0o644); err != nil {
|
||||
t.Fatalf("write pidfile: %v", err)
|
||||
}
|
||||
t.Cleanup(func() { pidfilePath = origPid })
|
||||
|
||||
origCtl := ctlPath
|
||||
ctlPath = filepath.Join(dir, "shaterd.ctl")
|
||||
ln, err := net.Listen("unix", ctlPath)
|
||||
if err != nil {
|
||||
t.Fatalf("listen %s: %v", ctlPath, err)
|
||||
}
|
||||
t.Cleanup(func() { _ = ln.Close(); ctlPath = origCtl })
|
||||
|
||||
go func() {
|
||||
for {
|
||||
conn, err := ln.Accept()
|
||||
if err != nil {
|
||||
return
|
||||
}
|
||||
go func(c net.Conn) {
|
||||
defer c.Close()
|
||||
line, _ := bufio.NewReader(c).ReadString('\n')
|
||||
if strings.TrimSpace(line) != "status" {
|
||||
_, _ = c.Write([]byte(`{"error":"unknown command"}` + "\n"))
|
||||
return
|
||||
}
|
||||
_, _ = c.Write([]byte(reply + "\n"))
|
||||
}(conn)
|
||||
}
|
||||
}()
|
||||
}
|
||||
|
||||
// liveStatusReply is a realistic Applier.Status() answer: indented (Status.JSON
|
||||
// uses MarshalIndent) and carrying a plane word only a live daemon can emit.
|
||||
const liveStatusReply = `{
|
||||
"running": true,
|
||||
"enabled": true,
|
||||
"active": true,
|
||||
"table": true,
|
||||
"hash": "abc123",
|
||||
"kill_switch": "closed",
|
||||
"panel_port": 8088,
|
||||
"can_rollback": false,
|
||||
"engine_running": true,
|
||||
"plane": "full",
|
||||
"traffic": {"verdict": "tunnelled"},
|
||||
"warnings": [],
|
||||
"started_unix": 1753500000,
|
||||
"uptime_seconds": 42
|
||||
}`
|
||||
|
||||
// --- defect 1: the offline stub must not pass for a status ------------------
|
||||
|
||||
// TestStatusOfflineStubIsMarkedAndFails is the whole contract of the daemon-down
|
||||
// case: stdout still carries a parseable object (the rpcd plugin reads stdout and
|
||||
// ignores the exit status, so removing it would blank the LuCI page), but that
|
||||
// object SAYS it is not a daemon's answer, and the exit code says so too.
|
||||
func TestStatusOfflineStubIsMarkedAndFails(t *testing.T) {
|
||||
noDaemon(t)
|
||||
|
||||
out, code := captureStdout(t, cmdStatus)
|
||||
|
||||
if code == 0 {
|
||||
t.Fatal("`shaterd status` exited 0 with no daemon to ask — a caller that " +
|
||||
"checks the exit status is told the status is real")
|
||||
}
|
||||
var got map[string]any
|
||||
if err := json.Unmarshal([]byte(out), &got); err != nil {
|
||||
t.Fatalf("stdout is not a JSON object (%v):\n%s", err, out)
|
||||
}
|
||||
v, present := got[statusDaemonAnsweredKey]
|
||||
if !present {
|
||||
t.Fatalf("the offline stub carries no %q field, so it is indistinguishable "+
|
||||
"from a daemon's answer except by guesswork:\n%s", statusDaemonAnsweredKey, out)
|
||||
}
|
||||
if v != false {
|
||||
t.Fatalf("%s = %v, want false for the offline stub", statusDaemonAnsweredKey, v)
|
||||
}
|
||||
// The compatibility carry-over luci-app-shater currently keys off. It is
|
||||
// deliberately still emitted; if it ever goes away, dashboard.js has to be
|
||||
// moved onto daemon_answered FIRST.
|
||||
if plane, ok := got["plane"]; !ok || plane != "" {
|
||||
t.Fatalf(`the stub no longer emits plane:"" (got %v, present=%v) — `+
|
||||
"luci-app-shater's dashboard.js derives \"daemon: down\" from exactly "+
|
||||
"that and would go silent; move it onto %q before removing it",
|
||||
got["plane"], ok, statusDaemonAnsweredKey)
|
||||
}
|
||||
}
|
||||
|
||||
// TestStatusLiveAnswerIsMarkedAndRelayedVerbatim is the CONTROL for the test
|
||||
// above: the same command, the same field, the opposite verdict. Without it,
|
||||
// "daemon_answered is false" proves nothing — a build that hardcoded false would
|
||||
// pass. It also pins that the daemon's own words are relayed unchanged: the CLI
|
||||
// adds one key and reorders/drops nothing, including fields this build has never
|
||||
// heard of.
|
||||
func TestStatusLiveAnswerIsMarkedAndRelayedVerbatim(t *testing.T) {
|
||||
liveDaemon(t, liveStatusReply)
|
||||
|
||||
out, code := captureStdout(t, cmdStatus)
|
||||
|
||||
if code != 0 {
|
||||
t.Fatalf("`shaterd status` exited %d while a daemon answered", code)
|
||||
}
|
||||
var got map[string]any
|
||||
if err := json.Unmarshal([]byte(out), &got); err != nil {
|
||||
t.Fatalf("stdout is not a JSON object (%v):\n%s", err, out)
|
||||
}
|
||||
if v, ok := got[statusDaemonAnsweredKey]; !ok || v != true {
|
||||
t.Fatalf("%s = %v (present=%v), want true for a daemon's answer:\n%s",
|
||||
statusDaemonAnsweredKey, got[statusDaemonAnsweredKey], ok, out)
|
||||
}
|
||||
var want map[string]any
|
||||
if err := json.Unmarshal([]byte(liveStatusReply), &want); err != nil {
|
||||
t.Fatalf("fixture: %v", err)
|
||||
}
|
||||
for k, wv := range want {
|
||||
gv, ok := got[k]
|
||||
if !ok {
|
||||
t.Errorf("the CLI dropped the daemon's %q field", k)
|
||||
continue
|
||||
}
|
||||
wb, _ := json.Marshal(wv)
|
||||
gb, _ := json.Marshal(gv)
|
||||
if string(wb) != string(gb) {
|
||||
t.Errorf("the CLI changed the daemon's %q: got %s, daemon said %s", k, gb, wb)
|
||||
}
|
||||
}
|
||||
if len(got) != len(want)+1 {
|
||||
t.Errorf("the CLI added more than the one verdict field: %d keys vs the daemon's %d",
|
||||
len(got), len(want))
|
||||
}
|
||||
// Field order is not JSON semantics, but it IS what an operator reads first.
|
||||
if !strings.HasPrefix(strings.TrimSpace(out), `{`+"\n \""+statusDaemonAnsweredKey+`"`) {
|
||||
t.Errorf("the verdict is not the first field of the object:\n%s", out)
|
||||
}
|
||||
}
|
||||
|
||||
// TestMarkStatusOriginRefusesWhatItCannotMark: a reply that is not a JSON object
|
||||
// must NOT be printed unmarked. Printing it would put an unlabelled blob on
|
||||
// stdout, which is the exact ambiguity this field exists to remove.
|
||||
func TestMarkStatusOriginRefusesWhatItCannotMark(t *testing.T) {
|
||||
for _, bad := range []string{"", "not json", "[1,2,3]", `"a string"`, "{", `{"a":}`} {
|
||||
if out, err := markStatusOrigin(bad, true); err == nil {
|
||||
t.Errorf("markStatusOrigin(%q) accepted it and produced %q", bad, out)
|
||||
}
|
||||
}
|
||||
// A reply that already carries the key is refused too: silently keeping the
|
||||
// daemon's value would let a future daemon assert its own reachability.
|
||||
if _, err := markStatusOrigin(`{"`+statusDaemonAnsweredKey+`":false}`, true); err == nil {
|
||||
t.Error("a reply that already carries the verdict field was accepted")
|
||||
}
|
||||
}
|
||||
|
||||
// TestMarkStatusOriginShapes covers the object shapes the two producers emit.
|
||||
func TestMarkStatusOriginShapes(t *testing.T) {
|
||||
for _, tc := range []struct {
|
||||
name, in string
|
||||
answered bool
|
||||
}{
|
||||
{"empty object", `{}`, false},
|
||||
{"compact", `{"a":1,"b":"x"}`, true},
|
||||
{"indented", "{\n \"a\": 1\n}", false},
|
||||
} {
|
||||
t.Run(tc.name, func(t *testing.T) {
|
||||
out, err := markStatusOrigin(tc.in, tc.answered)
|
||||
if err != nil {
|
||||
t.Fatalf("markStatusOrigin: %v", err)
|
||||
}
|
||||
var got map[string]any
|
||||
if err := json.Unmarshal([]byte(out), &got); err != nil {
|
||||
t.Fatalf("result is not valid JSON (%v): %s", err, out)
|
||||
}
|
||||
if got[statusDaemonAnsweredKey] != tc.answered {
|
||||
t.Fatalf("%s = %v, want %v", statusDaemonAnsweredKey,
|
||||
got[statusDaemonAnsweredKey], tc.answered)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
// --- defect 3: an unimplemented verb must not report success ----------------
|
||||
|
||||
// TestRulesetUpdateExitsNonZero. /etc/init.d/shater-cron runs
|
||||
// `"$SHATERD" ruleset update "$name" >/dev/null 2>&1` and, on a ZERO exit,
|
||||
// stamps the ruleset as freshly updated and sets changed=1. With the old exit 0
|
||||
// every url ruleset was permanently "just updated" by a verb that fetched
|
||||
// nothing, and every stamp triggered a reconcile with an identical config.
|
||||
func TestRulesetUpdateExitsNonZero(t *testing.T) {
|
||||
if code := dispatch([]string{"ruleset", "update"}); code == 0 {
|
||||
t.Fatal("`shaterd ruleset update` exited 0 without updating anything — " +
|
||||
"shater-cron reads that as a successful fetch and stamps the item")
|
||||
}
|
||||
if code := dispatch([]string{"ruleset", "update", "geosite"}); code == 0 {
|
||||
t.Fatal("`shaterd ruleset update <name>` exited 0 without updating anything")
|
||||
}
|
||||
}
|
||||
|
||||
// TestNotImplNeverReportsSuccess guards the helper itself, so a verb added to it
|
||||
// later cannot inherit a zero exit. It also pins that "not implemented" is
|
||||
// distinguishable from "no such verb" (usage, exit 2) — a caller must be able to
|
||||
// tell a typo from a stub.
|
||||
func TestNotImplNeverReportsSuccess(t *testing.T) {
|
||||
if code := notImpl("some future verb"); code == 0 {
|
||||
t.Fatal("notImpl reports success")
|
||||
}
|
||||
if notImplExitCode == 2 {
|
||||
t.Fatal("the not-implemented exit code collides with the usage exit code (2), " +
|
||||
"so a caller cannot tell an unknown verb from an unimplemented one")
|
||||
}
|
||||
if code := dispatch([]string{"ruleset", "frobnicate"}); code != 2 {
|
||||
t.Fatalf("an unknown ruleset sub-verb exited %d, want the usage code 2", code)
|
||||
}
|
||||
if code := dispatch([]string{"nosuchverb"}); code != 2 {
|
||||
t.Fatalf("an unknown verb exited %d, want the usage code 2", code)
|
||||
}
|
||||
}
|
||||
+67
-10
@@ -40,6 +40,7 @@ import (
|
||||
"github.com/sagernet/sing-box/option"
|
||||
"github.com/sagernet/sing-box/route"
|
||||
"github.com/sagernet/sing-box/route/rule"
|
||||
"github.com/sagernet/sing-box/shater/netplane"
|
||||
"github.com/sagernet/sing-box/shater/registry"
|
||||
E "github.com/sagernet/sing/common/exceptions"
|
||||
"github.com/sagernet/sing/common/json"
|
||||
@@ -73,6 +74,13 @@ type Engine struct {
|
||||
// PendingCloses when a shutdown overruns its budget.
|
||||
instanceGen uint64
|
||||
|
||||
// l3Device is the L3-ingress TUN slot the RUNNING instance opened, or "" when
|
||||
// nothing is running (or the running config has no L3 ingress). It is the
|
||||
// input to the slot choice for the next generation — the one name that
|
||||
// generation may not take — and it is what the netplane is told to point the
|
||||
// L3 routing table at. See l3slot.go and netplane/l3.go. Guarded by mu.
|
||||
l3Device string
|
||||
|
||||
// pending holds the retirements in flight (see teardown.go). Its OWN leaf
|
||||
// lock, deliberately not mu: PendingCloses is read by the status path, and
|
||||
// the moment that read matters most is while an apply is holding mu waiting
|
||||
@@ -246,7 +254,7 @@ func (e *Engine) applyLocked(opts option.Options) (bool, error) {
|
||||
|
||||
// (2) build + validate on its OWN cancellable context. box.New constructs and
|
||||
// validates every adapter.
|
||||
nb, nbCancel, err := e.newBox(opts)
|
||||
nb, nbCancel, nbDev, err := e.newBox(opts)
|
||||
if err != nil {
|
||||
// Validation failed: keep the running instance, do not swap.
|
||||
return false, E.Cause(err, "create instance")
|
||||
@@ -307,7 +315,7 @@ func (e *Engine) applyLocked(opts option.Options) (bool, error) {
|
||||
// an ERROR line naming it — visible, but not mistaken for a failed apply.
|
||||
_ = e.retireLocked(e.instance, e.instanceCancel, e.instanceGen, e.hash)
|
||||
}
|
||||
e.adoptLocked(nb, nbCancel, opts, newHash)
|
||||
e.adoptLocked(nb, nbCancel, opts, newHash, nbDev)
|
||||
return true, nil
|
||||
}
|
||||
|
||||
@@ -315,7 +323,15 @@ func (e *Engine) applyLocked(opts option.Options) (bool, error) {
|
||||
// returns the cancel alongside it. Every goroutine the box starts inherits that
|
||||
// context, so cancelling it is what unwinds the ones Close does not reach; see
|
||||
// teardown.go for why the shared, never-cancelled context was the defect.
|
||||
func (e *Engine) newBox(opts option.Options) (*box.Box, context.CancelFunc, error) {
|
||||
//
|
||||
// It also picks the L3-ingress TUN slot this box will open and returns it, so
|
||||
// the caller can publish it on adoption. The generated config carries only a
|
||||
// placeholder name; substituting it HERE — after the caller hashed the canonical
|
||||
// options, and once per box actually built — is what stops two generations from
|
||||
// contending for one TUN device. See l3slot.go. The returned device is "" for
|
||||
// every config without an L3 ingress, which is all of them by default.
|
||||
func (e *Engine) newBox(opts option.Options) (*box.Box, context.CancelFunc, string, error) {
|
||||
opts, dev := l3RetargetForNext(opts, e.l3Device)
|
||||
ctx, cancel := context.WithCancel(e.ctx)
|
||||
b, err := box.New(box.Options{
|
||||
Context: ctx,
|
||||
@@ -324,20 +340,39 @@ func (e *Engine) newBox(opts option.Options) (*box.Box, context.CancelFunc, erro
|
||||
})
|
||||
if err != nil {
|
||||
cancel()
|
||||
return nil, nil, err
|
||||
return nil, nil, "", err
|
||||
}
|
||||
return b, cancel, nil
|
||||
return b, cancel, dev, nil
|
||||
}
|
||||
|
||||
// adoptLocked installs a started box as THE running instance and gives it the
|
||||
// next generation number. Caller holds e.mu and has already retired whatever was
|
||||
// running before.
|
||||
func (e *Engine) adoptLocked(b *box.Box, cancel context.CancelFunc, opts option.Options, hash string) {
|
||||
//
|
||||
// opts is the CANONICAL (placeholder-named) config — the one the hash describes
|
||||
// and the one a later reconcile is compared against. dev is the L3 slot the box
|
||||
// really opened, kept apart from opts for exactly that reason, and published to
|
||||
// the netplane so ApplyRouting points the L3 table at the device that exists
|
||||
// rather than at a name it guessed.
|
||||
func (e *Engine) adoptLocked(b *box.Box, cancel context.CancelFunc, opts option.Options, hash string, dev string) {
|
||||
e.instanceGen++
|
||||
e.instance = b
|
||||
e.instanceCancel = cancel
|
||||
e.current = opts
|
||||
e.hash = hash
|
||||
e.setL3DeviceLocked(dev)
|
||||
}
|
||||
|
||||
// setL3DeviceLocked records the L3 slot the running instance holds and publishes
|
||||
// it to the netplane. Caller holds e.mu.
|
||||
//
|
||||
// The two live together on purpose: they answered different questions once (the
|
||||
// engine's "which name may the next generation not take" and the netplane's
|
||||
// "which device does the route point at") and any state where they disagree is a
|
||||
// state where one of them is lying about the same device.
|
||||
func (e *Engine) setL3DeviceLocked(dev string) {
|
||||
e.l3Device = dev
|
||||
netplane.RememberL3Device(dev)
|
||||
}
|
||||
|
||||
// closeOldThenStart is the close-old-then-start-new swap. It is taken both
|
||||
@@ -377,9 +412,17 @@ func (e *Engine) closeOldThenStart(discard *box.Box, discardCancel context.Cance
|
||||
prevHasLastGood := e.hasLastGood
|
||||
_ = e.retireLocked(e.instance, e.instanceCancel, e.instanceGen, prevHash)
|
||||
e.instance, e.instanceCancel = nil, nil
|
||||
// Nothing is running any more, so no L3 slot is spoken for. Clearing this
|
||||
// BEFORE the rebuilds below is what gives them a real choice: with the old
|
||||
// generation's slot still recorded, the rebuild and the restore would each
|
||||
// have exactly one candidate left, and the restore's would be the slot the
|
||||
// failed rebuild just released. Cleared, the slot picker takes whichever the
|
||||
// kernel says is actually free — which after a failed Start is the one that
|
||||
// has been free the longest. See netplane.L3SlotFor.
|
||||
e.setL3DeviceLocked("")
|
||||
|
||||
// (c) Build a FRESH box for opts (the discarded one cannot be reused).
|
||||
nb2, cancel2, err := e.newBox(opts)
|
||||
nb2, cancel2, dev2, err := e.newBox(opts)
|
||||
if err == nil {
|
||||
err = nb2.Start()
|
||||
if err != nil {
|
||||
@@ -391,14 +434,23 @@ func (e *Engine) closeOldThenStart(discard *box.Box, discardCancel context.Cance
|
||||
// (d) Success: the old config we just closed becomes last-good.
|
||||
e.lastGood = prevOpts
|
||||
e.hasLastGood = true
|
||||
e.adoptLocked(nb2, cancel2, opts, newHash)
|
||||
e.adoptLocked(nb2, cancel2, opts, newHash, dev2)
|
||||
return true, nil
|
||||
}
|
||||
|
||||
// (e) The fresh box could not come up and the old one is already closed —
|
||||
// interception is currently down. Try to RESTORE the previous config so we
|
||||
// do not leave the tunnel dead.
|
||||
rb, rcancel, rerr := e.newBox(prevOpts)
|
||||
//
|
||||
// This rebuild goes through newBox like every other, which means it gets its
|
||||
// own L3 slot rather than the name prevOpts was originally started under.
|
||||
// That is the whole point: this path used to fail for the SAME reason it was
|
||||
// entered — a fixed TUN name that the generation we just closed had not
|
||||
// finished handing back — so the rescue was wired to the resource whose
|
||||
// contention it was rescuing from, and the measured outcome on the router was
|
||||
// "start instance failed and could not restore previous config; engine
|
||||
// stopped" with the kill-switch closed and the LAN dark.
|
||||
rb, rcancel, rdev, rerr := e.newBox(prevOpts)
|
||||
if rerr == nil {
|
||||
rerr = rb.Start()
|
||||
if rerr != nil {
|
||||
@@ -409,7 +461,7 @@ func (e *Engine) closeOldThenStart(discard *box.Box, discardCancel context.Cance
|
||||
if rerr == nil {
|
||||
// Old config restored: keep current/hash/last-good exactly as they were
|
||||
// (do NOT advance them). Report that opts was not applied.
|
||||
e.adoptLocked(rb, rcancel, prevOpts, prevHash)
|
||||
e.adoptLocked(rb, rcancel, prevOpts, prevHash, rdev)
|
||||
e.lastGood = prevLastGood
|
||||
e.hasLastGood = prevHasLastGood
|
||||
return false, E.Cause(err, "start instance (config not applied; previous config restored)")
|
||||
@@ -420,6 +472,7 @@ func (e *Engine) closeOldThenStart(discard *box.Box, discardCancel context.Cance
|
||||
// safe (no unproxied leak) even though interception is down.
|
||||
e.instance, e.instanceCancel = nil, nil
|
||||
e.hash = ""
|
||||
e.setL3DeviceLocked("")
|
||||
return false, E.Cause(E.Errors(err, rerr), "start instance failed and could not restore previous config; engine stopped")
|
||||
}
|
||||
|
||||
@@ -535,6 +588,10 @@ func (e *Engine) Close() error {
|
||||
err := e.retireLocked(e.instance, e.instanceCancel, e.instanceGen, e.hash)
|
||||
e.instance, e.instanceCancel = nil, nil
|
||||
e.hash = ""
|
||||
// No instance, no slot. Leaving the old name published would make the next
|
||||
// generation avoid a device nobody holds, and would make the netplane point
|
||||
// the L3 route at a device that is on its way out.
|
||||
e.setL3DeviceLocked("")
|
||||
return err
|
||||
}
|
||||
|
||||
|
||||
@@ -0,0 +1,86 @@
|
||||
package engine
|
||||
|
||||
// Which L3-ingress TUN device a NEW generation opens.
|
||||
//
|
||||
// generate emits the L3 inbound with netplane.L3DeviceBase as a placeholder;
|
||||
// this file substitutes a real slot just before box.New. netplane/l3.go carries
|
||||
// the full argument for the two-slot design and for why the choice belongs
|
||||
// here rather than in generate. The short version, because it is the part that
|
||||
// is easy to "simplify" back into a bug:
|
||||
//
|
||||
// - generate runs on every reconcile, including the once-a-minute no-ops, and
|
||||
// Apply's fast path is a hash of what generate produced. A device name that
|
||||
// alternated in generate would change that hash every minute and rebuild the
|
||||
// whole engine — so the name in the CONFIG has to be stable.
|
||||
// - the name the KERNEL gets must not be stable, because the fresh box and the
|
||||
// box it replaces are alive at the same time (or the old one is still being
|
||||
// unregistered), and one name for both is TUNSETIFF EBUSY.
|
||||
//
|
||||
// Those two requirements are only compatible if the substitution happens after
|
||||
// the hash and before box.New. That is exactly here.
|
||||
//
|
||||
// The retarget deliberately does NOT mutate the caller's options: the canonical
|
||||
// (placeholder) form is what the engine stores as e.current and what every later
|
||||
// hash is compared against, so a mutation would make the next identical
|
||||
// reconcile look like a change and swap the engine for nothing.
|
||||
|
||||
import (
|
||||
C "github.com/sagernet/sing-box/constant"
|
||||
"github.com/sagernet/sing-box/option"
|
||||
|
||||
"github.com/sagernet/sing-box/shater/netplane"
|
||||
)
|
||||
|
||||
// l3InboundIndex returns the index of the L3-ingress TUN inbound in opts, or -1
|
||||
// when this config has none (l3_tunnel off, or the inbound was skipped).
|
||||
//
|
||||
// It matches on the TUN type AND on the device name being one this project
|
||||
// owns (netplane.L3IsManagedName) rather than on the tag: the tag is generate's
|
||||
// private constant, and importing it here would be an import cycle — the
|
||||
// generate package's own tests import this package.
|
||||
func l3InboundIndex(opts option.Options) int {
|
||||
for i, in := range opts.Inbounds {
|
||||
if in.Type != C.TypeTun {
|
||||
continue
|
||||
}
|
||||
to, ok := in.Options.(*option.TunInboundOptions)
|
||||
if !ok || !netplane.L3IsManagedName(to.InterfaceName) {
|
||||
continue
|
||||
}
|
||||
return i
|
||||
}
|
||||
return -1
|
||||
}
|
||||
|
||||
// l3Retarget returns a copy of opts whose L3-ingress TUN inbound opens dev, and
|
||||
// reports the device that copy will actually create ("" when opts has no L3
|
||||
// inbound, in which case opts is returned untouched).
|
||||
//
|
||||
// The copy is shallow except for the two things that must not be shared: the
|
||||
// Inbounds slice (a slice header copy still aliases the backing array) and the
|
||||
// TunInboundOptions struct the entry points at (it is a pointer behind an
|
||||
// `any`, so writing through it would reach every holder of the original —
|
||||
// including e.current, e.lastGood and the applier's own copy).
|
||||
func l3Retarget(opts option.Options, dev string) (option.Options, string) {
|
||||
idx := l3InboundIndex(opts)
|
||||
if idx < 0 {
|
||||
return opts, ""
|
||||
}
|
||||
inbounds := make([]option.Inbound, len(opts.Inbounds))
|
||||
copy(inbounds, opts.Inbounds)
|
||||
tun := *(inbounds[idx].Options.(*option.TunInboundOptions))
|
||||
tun.InterfaceName = dev
|
||||
inbounds[idx].Options = &tun
|
||||
opts.Inbounds = inbounds
|
||||
return opts, dev
|
||||
}
|
||||
|
||||
// l3RetargetForNext rewrites opts to open the slot the NEXT generation may use,
|
||||
// given the device the currently running generation holds (current, "" when
|
||||
// nothing is running). It returns the rewritten options and the device chosen.
|
||||
func l3RetargetForNext(opts option.Options, current string) (option.Options, string) {
|
||||
if l3InboundIndex(opts) < 0 {
|
||||
return opts, ""
|
||||
}
|
||||
return l3Retarget(opts, netplane.L3SlotFor(current))
|
||||
}
|
||||
@@ -0,0 +1,223 @@
|
||||
package engine
|
||||
|
||||
// The slot choice is what stops two generations of the engine from wanting one
|
||||
// TUN device. Everything here is about the two ways that guarantee can be lost:
|
||||
// picking the name the running generation holds, and letting the retarget leak
|
||||
// back into the options the hash gate compares.
|
||||
|
||||
import (
|
||||
"os"
|
||||
"testing"
|
||||
|
||||
C "github.com/sagernet/sing-box/constant"
|
||||
"github.com/sagernet/sing-box/option"
|
||||
|
||||
"github.com/sagernet/sing-box/shater/netplane"
|
||||
)
|
||||
|
||||
// TestMain takes THIS TEST BINARY off the real network namespace.
|
||||
//
|
||||
// The tests below call l3RetargetForNext for its return value, and that reaches
|
||||
// netplane.L3SlotFor — which does not merely ask the kernel about a device, it
|
||||
// DELETES one it finds occupying a candidate slot. `go test` runs package
|
||||
// binaries concurrently (-p defaults to GOMAXPROCS) and every one of them shares
|
||||
// the host's network namespace, so on a runner that has iproute2 and a real
|
||||
// /dev/net/tun these unit tests were deleting the TUN device shater/generate's
|
||||
// privileged tests had just opened:
|
||||
//
|
||||
// post-start inbound/tun[l3-in]: starting TUN interface: find tun interface: Link not found
|
||||
//
|
||||
// which is a red gate in a package that did nothing wrong. netplane.L3StubKernelForTest
|
||||
// carries the measurement and the reasoning; netplane's own
|
||||
// TestL3StubKernelTakesTheSlotChoiceOffTheKernel is the control that the hook
|
||||
// still diverts.
|
||||
//
|
||||
// The fake kernel starts EMPTY, so every slot reads as free and no reclaim is
|
||||
// ever attempted from here. That costs this file nothing: the reclaim is
|
||||
// netplane's subject (TestL3SlotForReclaimsARetiredSlot, against netplane's own
|
||||
// exec fake), the live-kernel proof is shater/generate's
|
||||
// TestIntegrationL3StaleSlotIsReclaimed, and what the tests below are about —
|
||||
// that the slot handed to the next generation is never the running one, and
|
||||
// that the substitution does not leak into the canonical options — is answered
|
||||
// by the real L3SlotFor either way.
|
||||
//
|
||||
// It is a TestMain rather than a per-test helper deliberately: a helper is
|
||||
// something the next test added here can forget, and the failure that causes
|
||||
// lands in a DIFFERENT package, on some runs only.
|
||||
func TestMain(m *testing.M) {
|
||||
restore := netplane.L3StubKernelForTest()
|
||||
code := m.Run()
|
||||
restore()
|
||||
os.Exit(code)
|
||||
}
|
||||
|
||||
// l3Opts is a config carrying one L3-ingress TUN inbound exactly as generate
|
||||
// emits it: the canonical placeholder name, which is the only name the retarget
|
||||
// is allowed to recognise.
|
||||
func l3Opts() option.Options {
|
||||
return option.Options{
|
||||
Inbounds: []option.Inbound{
|
||||
{
|
||||
Type: C.TypeTun,
|
||||
Tag: "l3-in",
|
||||
Options: &option.TunInboundOptions{InterfaceName: netplane.L3DeviceBase, MTU: 65535},
|
||||
},
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
func tunName(t *testing.T, opts option.Options, idx int) string {
|
||||
t.Helper()
|
||||
to, ok := opts.Inbounds[idx].Options.(*option.TunInboundOptions)
|
||||
if !ok {
|
||||
t.Fatalf("inbound %d options are %T, want *option.TunInboundOptions", idx, opts.Inbounds[idx].Options)
|
||||
}
|
||||
return to.InterfaceName
|
||||
}
|
||||
|
||||
// TestL3SlotNeverCollidesWithTheRunningGeneration is the A1 invariant in one
|
||||
// assertion: whatever the running generation holds, the next one is given
|
||||
// something else.
|
||||
//
|
||||
// This is the whole fix. With a single fixed name, an apply built a box that had
|
||||
// to open the device the running box still held (or that the kernel had not
|
||||
// finished unregistering), TUNSETIFF answered EBUSY, and — because the recovery
|
||||
// path rebuilds the PREVIOUS config, which named the same device — the rescue
|
||||
// failed for the very reason it was needed. Measured on the production router:
|
||||
// "start instance failed and could not restore previous config; engine stopped",
|
||||
// then `plane: hold`, i.e. the whole LAN offline until a manual restart.
|
||||
func TestL3SlotNeverCollidesWithTheRunningGeneration(t *testing.T) {
|
||||
for _, current := range append([]string{"", netplane.L3DeviceBase}, netplane.L3Slots[:]...) {
|
||||
got, dev := l3RetargetForNext(l3Opts(), current)
|
||||
if dev == "" {
|
||||
t.Fatalf("current=%q: no device was chosen for a config that HAS an L3 inbound — the engine would then start it under the placeholder name, which is the single-name collision this design removes", current)
|
||||
}
|
||||
if dev == current {
|
||||
t.Errorf("current=%q: the next generation was handed the SAME device %q the running one holds. Start would answer `TUNSETIFF: device or resource busy`, and on the router that is a LAN-wide outage, not a failed apply", current, dev)
|
||||
}
|
||||
if name := tunName(t, got, 0); name != dev {
|
||||
t.Errorf("current=%q: chose %q but the returned options still say %q", current, dev, name)
|
||||
}
|
||||
if !netplane.L3IsManagedName(dev) || dev == netplane.L3DeviceBase {
|
||||
t.Errorf("current=%q: chose %q, which is not one of the slots %v — the fw4 zone and our nft accepts match the slot prefix, and a name outside it is a device the firewall drops", current, dev, netplane.L3Slots)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestL3RetargetLeavesTheCanonicalOptionsAlone guards the OTHER half of the
|
||||
// design, and it is the half that is easy to lose in a "simplification": the
|
||||
// substitution must not reach the options the caller keeps.
|
||||
//
|
||||
// applyLocked hashes opts and stores it as e.current; every later reconcile
|
||||
// compares a freshly generated (canonical) config against that stored one. If
|
||||
// the retarget wrote through the shared *TunInboundOptions pointer, the stored
|
||||
// config would carry a per-generation device name, the next identical reconcile
|
||||
// would hash differently, and the engine would rebuild itself — dropping every
|
||||
// connection through the tunnel — once a minute, forever.
|
||||
func TestL3RetargetLeavesTheCanonicalOptionsAlone(t *testing.T) {
|
||||
canonical := l3Opts()
|
||||
origOptions := canonical.Inbounds[0].Options
|
||||
|
||||
got, dev := l3RetargetForNext(canonical, "")
|
||||
if dev == netplane.L3DeviceBase || dev == "" {
|
||||
t.Fatalf("retarget chose %q — nothing below would prove anything", dev)
|
||||
}
|
||||
|
||||
if name := tunName(t, canonical, 0); name != netplane.L3DeviceBase {
|
||||
t.Errorf("the caller's options were MUTATED to %q. That name then becomes e.current, every later reconcile hashes differently against it, and the engine rebuilds on every one-minute no-op reconcile.", name)
|
||||
}
|
||||
if canonical.Inbounds[0].Options == got.Inbounds[0].Options {
|
||||
t.Error("the returned options share the TunInboundOptions pointer with the caller's — they are one struct behind two `any`s, so the next retarget writes through both")
|
||||
}
|
||||
if origOptions != canonical.Inbounds[0].Options {
|
||||
t.Error("the caller's inbound now points at a different options struct")
|
||||
}
|
||||
// The backing array too: copying only the slice header still aliases it.
|
||||
if len(canonical.Inbounds) > 0 && len(got.Inbounds) > 0 &&
|
||||
&canonical.Inbounds[0] == &got.Inbounds[0] {
|
||||
t.Error("the Inbounds slices share a backing array — writing the retargeted inbound reached the caller's slice")
|
||||
}
|
||||
}
|
||||
|
||||
// TestL3RetargetIgnoresInboundsThatAreNotOurs: the retarget matches on the TUN
|
||||
// type AND on a name this project owns. A tun inbound that is not the L3
|
||||
// ingress must be left exactly as configured — renaming a device out from under
|
||||
// its owner is a bigger failure than not renaming ours.
|
||||
func TestL3RetargetIgnoresInboundsThatAreNotOurs(t *testing.T) {
|
||||
cases := []struct {
|
||||
name string
|
||||
in option.Inbound
|
||||
}{
|
||||
{"foreign tun device", option.Inbound{Type: C.TypeTun, Tag: "vpn", Options: &option.TunInboundOptions{InterfaceName: "tun0"}}},
|
||||
{"not a tun at all", option.Inbound{Type: C.TypeTProxy, Tag: "lan", Options: &option.TProxyInboundOptions{}}},
|
||||
}
|
||||
for _, tc := range cases {
|
||||
t.Run(tc.name, func(t *testing.T) {
|
||||
opts := option.Options{Inbounds: []option.Inbound{tc.in}}
|
||||
got, dev := l3RetargetForNext(opts, "")
|
||||
if dev != "" {
|
||||
t.Errorf("chose device %q for a config with no L3 ingress", dev)
|
||||
}
|
||||
if to, ok := got.Inbounds[0].Options.(*option.TunInboundOptions); ok && to.InterfaceName != "tun0" {
|
||||
t.Errorf("renamed a foreign tun device to %q", to.InterfaceName)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
// TestL3RetargetNoInboundIsNotADevice: with l3_tunnel off (the default) there is
|
||||
// no TUN inbound, so there is no slot to claim and nothing to publish to the
|
||||
// netplane. Reporting a device here would make addL3Routing point the L3 table
|
||||
// at an interface nothing ever creates.
|
||||
func TestL3RetargetNoInboundIsNotADevice(t *testing.T) {
|
||||
opts := option.Options{Inbounds: []option.Inbound{
|
||||
{Type: C.TypeTProxy, Tag: "lan", Options: &option.TProxyInboundOptions{}},
|
||||
}}
|
||||
if _, dev := l3RetargetForNext(opts, netplane.L3Slots[0]); dev != "" {
|
||||
t.Errorf("device = %q for a config without an L3 ingress, want \"\"", dev)
|
||||
}
|
||||
}
|
||||
|
||||
// TestEngineSlotHandoffSurvivesAFailedSwap walks the exact sequence
|
||||
// closeOldThenStart performs, through the ENGINE's own state, and asserts the
|
||||
// property the production outage violated: no step is ever handed the device
|
||||
// the step before it was using.
|
||||
//
|
||||
// The sequence, with the router log it reproduces:
|
||||
//
|
||||
// gen1 running on some slot (19:10 healthy)
|
||||
// an apply builds gen2 while gen1 still holds its ← must differ, or TUNSETIFF
|
||||
// gen1 is closed, gen2's Start fails (19:14:02)
|
||||
// the RESTORE rebuilds the previous config ← must not be handed gen2's
|
||||
//
|
||||
// The last step is the one that turned a failed apply into an outage: it used to
|
||||
// rebuild a config naming the one fixed device, so it collided with the teardown
|
||||
// still in flight and the engine stopped with the kill-switch closed.
|
||||
func TestEngineSlotHandoffSurvivesAFailedSwap(t *testing.T) {
|
||||
e := &Engine{}
|
||||
|
||||
_, gen1 := l3RetargetForNext(l3Opts(), e.l3Device)
|
||||
e.l3Device = gen1 // gen1 adopted and running
|
||||
|
||||
_, gen2 := l3RetargetForNext(l3Opts(), e.l3Device)
|
||||
if gen2 == gen1 {
|
||||
t.Fatalf("gen2 was handed gen1's device %q while gen1 is still running — this is the TUNSETIFF EBUSY at 19:10:33", gen1)
|
||||
}
|
||||
|
||||
// closeOldThenStart: gen1 retired, nothing running. The engine clears the
|
||||
// slot here precisely so the two rebuilds below get a real choice instead of
|
||||
// inheriting a single forced candidate.
|
||||
e.l3Device = ""
|
||||
|
||||
_, restore := l3RetargetForNext(l3Opts(), e.l3Device)
|
||||
if restore == "" {
|
||||
t.Fatal("the restore path was given no device at all")
|
||||
}
|
||||
// The restore must not be forced onto the device the failed generation was
|
||||
// using. Without the clear above, gen1's name would still be recorded, the
|
||||
// only remaining candidate would be gen2's — the one that just failed — and
|
||||
// the rescue would again depend on the resource it is rescuing from.
|
||||
if e.l3Device != "" {
|
||||
t.Fatalf("the engine still claims device %q with nothing running", e.l3Device)
|
||||
}
|
||||
}
|
||||
@@ -27,7 +27,7 @@ func outboundByTag(opts option.Options, tag string) *option.Outbound {
|
||||
|
||||
func byedpiModel(port int) *model.Model {
|
||||
return &model.Model{
|
||||
Globals: nonDNSGlobals(),
|
||||
Globals: plainGlobals(),
|
||||
Egresses: []model.Egress{{Name: "bd", Type: "byedpi", Port: port}},
|
||||
Rules: []model.Rule{
|
||||
{Name: "desync", Enabled: true, Order: 10, Src: []string{"192.168.1.0/24"}, Target: "egress:bd"},
|
||||
|
||||
@@ -117,7 +117,7 @@ func TestChainDeadExitMarkedByObservatory(t *testing.T) {
|
||||
eng.StopObservatory()
|
||||
_ = eng.Close()
|
||||
})
|
||||
if _, err := eng.Apply(opts); err != nil {
|
||||
if _, err := eng.Apply(withoutL3Ingress(t, opts)); err != nil {
|
||||
t.Fatalf("engine.Apply (box.New + start): %v", err)
|
||||
}
|
||||
|
||||
@@ -174,7 +174,7 @@ func TestChainDeadExitDialFailsClosed(t *testing.T) {
|
||||
}
|
||||
eng := engine.New()
|
||||
t.Cleanup(func() { _ = eng.Close() })
|
||||
if _, err := eng.Apply(opts); err != nil {
|
||||
if _, err := eng.Apply(withoutL3Ingress(t, opts)); err != nil {
|
||||
t.Fatalf("engine.Apply: %v", err)
|
||||
}
|
||||
inst := eng.Instance()
|
||||
|
||||
@@ -47,7 +47,7 @@ func generalRouteOutbound(rt *option.RouteOptions) (string, bool) {
|
||||
|
||||
func threeNodeChainModel() *model.Model {
|
||||
return &model.Model{
|
||||
Globals: nonDNSGlobals(),
|
||||
Globals: plainGlobals(),
|
||||
Nodes: []model.Node{
|
||||
{Name: "a", Enabled: true, URI: "ss://aes-256-gcm:secret@203.0.113.1:8388#a"},
|
||||
{Name: "b", Enabled: true, URI: "ss://aes-256-gcm:secret@203.0.113.2:8388#b"},
|
||||
@@ -122,7 +122,7 @@ func TestChainMultiHopDetourWiring(t *testing.T) {
|
||||
// (no wrapper, no detour outbound).
|
||||
func TestChainSingleHopResolvesToHop(t *testing.T) {
|
||||
m := &model.Model{
|
||||
Globals: nonDNSGlobals(),
|
||||
Globals: plainGlobals(),
|
||||
Nodes: []model.Node{
|
||||
{Name: "a", Enabled: true, URI: "ss://aes-256-gcm:secret@203.0.113.1:8388#a"},
|
||||
},
|
||||
@@ -210,7 +210,7 @@ func TestChainEmptyHopsWarnsBlocks(t *testing.T) {
|
||||
// through the previous hop; the next hop detours into the group wrapper.
|
||||
func TestChainGroupHop(t *testing.T) {
|
||||
m := &model.Model{
|
||||
Globals: nonDNSGlobals(),
|
||||
Globals: plainGlobals(),
|
||||
Nodes: []model.Node{
|
||||
{Name: "a", Enabled: true, URI: "ss://aes-256-gcm:secret@203.0.113.1:8388#a"},
|
||||
{Name: "b", Enabled: true, URI: "ss://aes-256-gcm:secret@203.0.113.2:8388#b"},
|
||||
|
||||
@@ -275,7 +275,14 @@ func TestDNSFilterLiveAnswers(t *testing.T) {
|
||||
var mgr deprecated.Manager = &recordingDeprecated{}
|
||||
ctx = service.ContextWith(ctx, mgr)
|
||||
ctx = registry.Context(ctx)
|
||||
b, err := box.New(box.Options{Context: ctx, Options: opts})
|
||||
// withoutL3Ingress for the reasons given at its definition, plus one that is
|
||||
// sharper here than anywhere else in the suite: this test builds the box
|
||||
// DIRECTLY, so it never reaches engine.newBox and never gets a slot — it
|
||||
// would open the device under the PLACEHOLDER name, which is the one name
|
||||
// every generation wants and therefore the one name that must never exist.
|
||||
// That is not a hypothesis: it is where the `shater-l3` device in
|
||||
// TestIntegrationL3TunInboundStarts' failure came from.
|
||||
b, err := box.New(box.Options{Context: ctx, Options: withoutL3Ingress(t, opts)})
|
||||
if err != nil {
|
||||
t.Fatalf("box.New: %v", err)
|
||||
}
|
||||
@@ -393,7 +400,10 @@ func applyWithDeprecations(t *testing.T, m *model.Model) (option.Options, []depr
|
||||
var mgr deprecated.Manager = rec
|
||||
ctx = service.ContextWith(ctx, mgr)
|
||||
ctx = registry.Context(ctx)
|
||||
b, err := box.New(box.Options{Context: ctx, Options: opts})
|
||||
// Direct box.New, so the L3 ingress must come out first — see the note at
|
||||
// the identical call in TestDNSFilterLiveAnswers, and withoutL3Ingress for
|
||||
// the whole argument.
|
||||
b, err := box.New(box.Options{Context: ctx, Options: withoutL3Ingress(t, opts)})
|
||||
if err != nil {
|
||||
t.Fatalf("box.New failed: %v\nwarnings: %v", err, warns)
|
||||
}
|
||||
|
||||
@@ -35,7 +35,7 @@ func routeActionFor(opts option.Options, tag string) *option.RouteActionOptions
|
||||
|
||||
func dpiModel(dpi string) *model.Model {
|
||||
return &model.Model{
|
||||
Globals: nonDNSGlobals(),
|
||||
Globals: plainGlobals(),
|
||||
Egresses: []model.Egress{{Name: "frag", Type: "direct", DPI: dpi}},
|
||||
Rules: []model.Rule{
|
||||
{Name: "desync", Enabled: true, Order: 10, Src: []string{"192.168.1.0/24"}, Target: "egress:frag"},
|
||||
|
||||
@@ -0,0 +1,148 @@
|
||||
// Defect 1: an `interface` egress with no interface used to be bound to the LAN
|
||||
// bridge.
|
||||
//
|
||||
// The chain, because none of it is visible in the generated JSON: this generator
|
||||
// resolved the bind device with netplane.IfaceDevice(eg.Interface), and
|
||||
// IfaceDevice("") falls back to "br-lan" (the right default for an INBOUND with
|
||||
// no network). netplane.EgressDevice — the resolution the DATA plane uses —
|
||||
// returns "" for the same egress on purpose, and calls br-lan "catastrophic
|
||||
// here", so addEgressRouting installed no `ip rule` and no routing table for that
|
||||
// egress's mark, and the prerouting marking and the forward-chain accept skipped
|
||||
// it too.
|
||||
//
|
||||
// The result was an outbound with SO_BINDTODEVICE=br-lan and a routing mark
|
||||
// nothing routed: every node, group and rule bound to that egress dialled public
|
||||
// addresses out of the LAN bridge. Not a leak — the bind pins the socket to the
|
||||
// LAN — but a total, silent black hole, with the panel showing a configured,
|
||||
// applied egress and no findings at all.
|
||||
package generate
|
||||
|
||||
import (
|
||||
"strings"
|
||||
"testing"
|
||||
|
||||
"github.com/sagernet/sing-box/option"
|
||||
|
||||
"github.com/sagernet/sing-box/shater/model"
|
||||
"github.com/sagernet/sing-box/shater/netplane"
|
||||
)
|
||||
|
||||
// deviceLessEgressModel is the config at issue: one interface egress with no
|
||||
// interface, and a rule that sends a LAN subnet through it.
|
||||
func deviceLessEgressModel(iface string) *model.Model {
|
||||
return &model.Model{
|
||||
Globals: model.DefaultGlobals(),
|
||||
Inbounds: []model.Inbound{
|
||||
{Name: "lan", Enabled: true, Type: "tproxy", TproxyPort: 12345, TCP: true, UDP: true},
|
||||
},
|
||||
Egresses: []model.Egress{{Name: "hole", Type: "interface", Interface: iface}},
|
||||
Rules: []model.Rule{
|
||||
{Name: "via-hole", Enabled: true, Order: 10, Src: []string{"192.168.9.0/24"}, Target: "egress:hole"},
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
// bindInterfaces returns every BindInterface the generated outbounds carry,
|
||||
// keyed by outbound tag. Only direct outbounds can carry one.
|
||||
func bindInterfaces(opts option.Options) map[string]string {
|
||||
out := map[string]string{}
|
||||
for i := range opts.Outbounds {
|
||||
if do, ok := opts.Outbounds[i].Options.(*option.DirectOutboundOptions); ok {
|
||||
out[opts.Outbounds[i].Tag] = do.BindInterface
|
||||
}
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
// TestInterfaceEgressWithNoInterfaceEmitsNoOutbound is the defect proper.
|
||||
func TestInterfaceEgressWithNoInterfaceEmitsNoOutbound(t *testing.T) {
|
||||
opts, warns, err := GenerateWithWarnings(deviceLessEgressModel(""))
|
||||
if err != nil {
|
||||
t.Fatalf("Generate: unexpected error: %v", err)
|
||||
}
|
||||
tag := netplane.EgressOutboundTag("hole")
|
||||
binds := bindInterfaces(opts)
|
||||
|
||||
if dev, ok := binds[tag]; ok {
|
||||
t.Errorf("outbound %q was emitted, bound to device %q. netplane.EgressDevice resolves this egress to "+
|
||||
"NO device, so addEgressRouting installs neither its `ip rule` nor its routing table and the "+
|
||||
"prerouting mark and forward accept skip it: this outbound is bound to a device the router does not "+
|
||||
"route for, under a mark that leads nowhere. Every node, group and rule bound to this egress "+
|
||||
"disappears into it. Emit nothing instead, so the binding fails closed and is visible.", tag, dev)
|
||||
}
|
||||
for otag, dev := range binds {
|
||||
if dev == "br-lan" {
|
||||
t.Errorf("outbound %q is bound to %q — the LAN BRIDGE. That is IfaceDevice's empty-name fallback "+
|
||||
"leaking into an egress bind: the socket is pinned to the LAN and dials public addresses out of "+
|
||||
"it. Nothing leaves the house, and nothing works, and nothing says why.", otag, dev)
|
||||
}
|
||||
}
|
||||
|
||||
// The skip must be LOUD. A silent skip only moves the black hole from the data
|
||||
// plane into the panel.
|
||||
if !hasEgressWarning(warns, "hole") {
|
||||
t.Errorf("no warning names egress %q. The egress is configured, the panel lists it, rules are bound to "+
|
||||
"it, and it carries nothing — the operator has to discover that by noticing their traffic stopped. "+
|
||||
"Warnings were:\n %s", "hole", strings.Join(warns, "\n "))
|
||||
}
|
||||
}
|
||||
|
||||
// TestInterfaceEgressWithBlankInterfaceEmitsNoOutbound is the same defect through
|
||||
// the whitespace door — the exact shape that had already produced one
|
||||
// validator/data-plane divergence in this codebase (see EgressDevice's comment).
|
||||
// IfaceDevice(" ") hands back " ", which is neither empty nor a device, so the
|
||||
// old code bound the socket to a device name made of spaces.
|
||||
func TestInterfaceEgressWithBlankInterfaceEmitsNoOutbound(t *testing.T) {
|
||||
opts, warns, err := GenerateWithWarnings(deviceLessEgressModel(" "))
|
||||
if err != nil {
|
||||
t.Fatalf("Generate: unexpected error: %v", err)
|
||||
}
|
||||
tag := netplane.EgressOutboundTag("hole")
|
||||
if dev, ok := bindInterfaces(opts)[tag]; ok {
|
||||
t.Errorf("outbound %q was emitted, bound to %q. netplane.EgressDevice trims and resolves this to no "+
|
||||
"device, so the router routes nothing for this egress's mark.", tag, dev)
|
||||
}
|
||||
if !hasEgressWarning(warns, "hole") {
|
||||
t.Errorf("blank interface skipped silently; warnings were:\n %s", strings.Join(warns, "\n "))
|
||||
}
|
||||
}
|
||||
|
||||
// TestInterfaceEgressWithADeviceStillEmitsItsOutbound is the CONTROL. Without it
|
||||
// the two tests above are equally satisfied by a generator that emits no egress
|
||||
// outbound ever — an instrument that cannot produce a positive proves nothing by
|
||||
// producing a negative.
|
||||
func TestInterfaceEgressWithADeviceStillEmitsItsOutbound(t *testing.T) {
|
||||
m := deviceLessEgressModel("wan2")
|
||||
opts, warns, err := GenerateWithWarnings(m)
|
||||
if err != nil {
|
||||
t.Fatalf("Generate: unexpected error: %v", err)
|
||||
}
|
||||
tag := netplane.EgressOutboundTag("hole")
|
||||
dev, ok := bindInterfaces(opts)[tag]
|
||||
if !ok {
|
||||
t.Fatalf("no outbound %q for an egress that DOES resolve to a device — the fail-closed skip has swallowed "+
|
||||
"a healthy egress, and every rule bound to it is now blocked. Warnings: %v", tag, warns)
|
||||
}
|
||||
if want := netplane.EgressDevice(m.Egresses[0]); dev != want {
|
||||
t.Errorf("BindInterface = %q, want %q — the bind device must be the one netplane.EgressDevice resolves, "+
|
||||
"because that is the device addEgressRouting builds the mark's routing table around. Any other "+
|
||||
"string binds the socket to one device while the kernel routes its mark out another.", dev, want)
|
||||
}
|
||||
if hasEgressWarning(warns, "hole") {
|
||||
t.Errorf("a healthy interface egress must not be reported as device-less; warnings were:\n %s",
|
||||
strings.Join(warns, "\n "))
|
||||
}
|
||||
}
|
||||
|
||||
// hasEgressWarning reports whether any warning is about this egress AND about it
|
||||
// having no device — matching on the entity prefix apply's normaliser parses
|
||||
// (`egress "name": ...`) plus the substance, so an unrelated egress warning (a
|
||||
// stray port, an unknown type) cannot pass for this one.
|
||||
func hasEgressWarning(warns []string, name string) bool {
|
||||
for _, w := range warns {
|
||||
if strings.HasPrefix(w, `egress "`+name+`": `) && strings.Contains(w, "no device") {
|
||||
return true
|
||||
}
|
||||
}
|
||||
return false
|
||||
}
|
||||
@@ -13,7 +13,7 @@ import (
|
||||
// the egress tag (multi-WAN). No copies: the binding lands on the node itself.
|
||||
func TestNodeEgressBindsOutbound(t *testing.T) {
|
||||
m := &model.Model{
|
||||
Globals: nonDNSGlobals(),
|
||||
Globals: plainGlobals(),
|
||||
Egresses: []model.Egress{{Name: "wan2", Type: "direct"}},
|
||||
Nodes: []model.Node{
|
||||
{Name: "a", Enabled: true, URI: "ss://aes-256-gcm:secret@203.0.113.1:8388#a", Egress: "wan2"},
|
||||
@@ -80,7 +80,7 @@ func TestNodeEgressMissingIsFailClosed(t *testing.T) {
|
||||
// outbound (instead of dialing directly from the router).
|
||||
func TestChainEgressEntryHop(t *testing.T) {
|
||||
m := &model.Model{
|
||||
Globals: nonDNSGlobals(),
|
||||
Globals: plainGlobals(),
|
||||
Egresses: []model.Egress{{Name: "wan2", Type: "direct"}},
|
||||
Nodes: []model.Node{
|
||||
{Name: "x", Enabled: true, URI: "ss://aes-256-gcm:secret@203.0.113.1:8388#x"},
|
||||
|
||||
@@ -364,10 +364,18 @@ func TestBootstrapResolverSetWhenPlaneHasTwoTransports(t *testing.T) {
|
||||
}
|
||||
|
||||
// The shipped default: one DoH resolver + intercept. The clone must still be DoH.
|
||||
//
|
||||
// The tproxy inbound is part of "the shipped default shape" and is here for
|
||||
// that reason, not to dodge a warning: /etc/config/shater ships one, and the
|
||||
// seeded-ON l3_tunnel needs it — the L3 ingress is fed only by the tproxy
|
||||
// divert plane, so a model without one is a shape nobody actually installs
|
||||
// and it earns an honest `icmp "tunnel"` diagnostic. Asserting "no warnings"
|
||||
// over a fixture that omits it was asserting it about the wrong config.
|
||||
g := model.DefaultGlobals()
|
||||
g.ResolverDefault = "cf"
|
||||
m2 := &model.Model{
|
||||
Globals: g,
|
||||
Globals: g,
|
||||
Inbounds: []model.Inbound{lanTproxy()},
|
||||
Resolvers: []model.Resolver{
|
||||
{Name: "cf", Type: "doh", Address: "https://1.1.1.1/dns-query", Detour: "direct"},
|
||||
},
|
||||
|
||||
@@ -43,7 +43,7 @@ func warnMatching(warns []string, subs ...string) []string {
|
||||
// carrying the given strategy.
|
||||
func twoNodeGroupModel(strategy string) *model.Model {
|
||||
return &model.Model{
|
||||
Globals: nonDNSGlobals(),
|
||||
Globals: plainGlobals(),
|
||||
Nodes: []model.Node{
|
||||
{Name: "n1", Enabled: true, URI: ss("203.0.113.1")},
|
||||
{Name: "n2", Enabled: true, URI: ss("203.0.113.2")},
|
||||
@@ -313,7 +313,7 @@ func TestFailoverWarnsOnceAcrossChainCopy(t *testing.T) {
|
||||
|
||||
func egressModel(egType string) *model.Model {
|
||||
return &model.Model{
|
||||
Globals: nonDNSGlobals(),
|
||||
Globals: plainGlobals(),
|
||||
Egresses: []model.Egress{{Name: "e1", Type: egType, Interface: "eth1"}},
|
||||
}
|
||||
}
|
||||
@@ -375,7 +375,7 @@ func TestEgressTypeUnknownWarns(t *testing.T) {
|
||||
// egress, but the native tls_* flags are never applied to one.
|
||||
func TestByedpiEgressDPIWarnsIgnored(t *testing.T) {
|
||||
m := &model.Model{
|
||||
Globals: nonDNSGlobals(),
|
||||
Globals: plainGlobals(),
|
||||
Egresses: []model.Egress{{Name: "bd", Type: "byedpi", Port: 1080, DPI: "fragment"}},
|
||||
}
|
||||
_, warns, err := GenerateWithWarnings(m)
|
||||
@@ -387,7 +387,7 @@ func TestByedpiEgressDPIWarnsIgnored(t *testing.T) {
|
||||
}
|
||||
for _, d := range []string{"", "off"} {
|
||||
m2 := &model.Model{
|
||||
Globals: nonDNSGlobals(),
|
||||
Globals: plainGlobals(),
|
||||
Egresses: []model.Egress{{Name: "bd", Type: "byedpi", DPI: d}},
|
||||
}
|
||||
if _, w, _ := GenerateWithWarnings(m2); len(w) != 0 {
|
||||
@@ -413,7 +413,7 @@ func TestEgressPortIgnoredOnNonByedpi(t *testing.T) {
|
||||
}
|
||||
// byedpi uses it, so no warning.
|
||||
m := &model.Model{
|
||||
Globals: nonDNSGlobals(),
|
||||
Globals: plainGlobals(),
|
||||
Egresses: []model.Egress{{Name: "e1", Type: "byedpi", Port: 9050}},
|
||||
}
|
||||
if _, warns, _ := GenerateWithWarnings(m); len(warns) != 0 {
|
||||
@@ -564,7 +564,7 @@ func findChainOutbound(b *builder, tag string) *option.Outbound {
|
||||
// resolves must still produce a plain detour to the egress, with no warning.
|
||||
func TestEgressBindingValidIsUnaffected(t *testing.T) {
|
||||
m := &model.Model{
|
||||
Globals: nonDNSGlobals(),
|
||||
Globals: plainGlobals(),
|
||||
Egresses: []model.Egress{{Name: "wan2", Type: "direct"}},
|
||||
Nodes: []model.Node{
|
||||
{Name: "n1", Enabled: true, URI: ss("203.0.113.1"), Egress: "wan2"},
|
||||
|
||||
@@ -99,7 +99,12 @@ func TestFailoverBalancerIsEngineValidated(t *testing.T) {
|
||||
}
|
||||
e := engine.New()
|
||||
t.Cleanup(func() { _ = e.Close() })
|
||||
if _, err := e.Apply(opts); err == nil {
|
||||
// The L3 ingress comes out here for the same reason as everywhere else in
|
||||
// this suite (see withoutL3Ingress), and for one more that is specific to a
|
||||
// NEGATIVE test: with the TUN left in, an Apply on a host without
|
||||
// /dev/net/tun fails whatever the balancer says, and this control would
|
||||
// report "validated" while validating nothing.
|
||||
if _, err := e.Apply(withoutL3Ingress(t, opts)); err == nil {
|
||||
t.Fatalf("engine accepted a bogus sticky_hash — the balancer is NOT validated, so TestFailoverGroupApplies proves nothing")
|
||||
}
|
||||
}
|
||||
|
||||
@@ -9,30 +9,53 @@ import (
|
||||
"github.com/sagernet/sing-box/shater/model"
|
||||
)
|
||||
|
||||
// nonDNSGlobals is model.DefaultGlobals for a fixture whose subject is NOT the
|
||||
// DNS plane: egress binding, chain flattening, rule-set compilation, byedpi
|
||||
// wiring.
|
||||
// plainGlobals is model.DefaultGlobals for a fixture whose subject is neither the
|
||||
// DNS plane nor the L3 ingress: egress binding, chain flattening, rule-set
|
||||
// compilation, byedpi wiring. It turns OFF the two router-wide planes that are
|
||||
// seeded ON, so such a fixture can keep asserting "this config produces no
|
||||
// diagnostics at all" and have that assertion still mean what it says.
|
||||
//
|
||||
// D24 made `dns_intercept` ON by default, so a model built straight from
|
||||
// DefaultGlobals now declares "force every LAN :53 into the engine, including the
|
||||
// queries addressed to the router". buildDNS answers that declaration honestly on
|
||||
// both sides: a model that intercepts DNS while configuring no `config resolver`
|
||||
// earns a warning (there is no resolver plane to intercept INTO — the queries end
|
||||
// up at the system resolver), and a model that does have one gets the .lan / PTR
|
||||
// preservation rule prepended ahead of everything else.
|
||||
// # dns_intercept (D24)
|
||||
//
|
||||
// Both are correct, and both are noise in a test about SOCKS outbounds. So the
|
||||
// fixtures that use this helper state plainly that they do not intercept DNS,
|
||||
// rather than keeping an "unexpected warnings" assertion that would silently
|
||||
// become a DNS assertion. The tests that ARE about the DNS plane —
|
||||
// dns_intercept_test.go, dnsfilter_test.go, doh_test.go, devices_test.go — keep
|
||||
// the real default: that is where the intercept contract belongs and is pinned.
|
||||
func nonDNSGlobals() model.Globals {
|
||||
g := model.DefaultGlobals()
|
||||
// A model built straight from DefaultGlobals declares "force every LAN :53 into
|
||||
// the engine, including the queries addressed to the router". buildDNS answers
|
||||
// that declaration honestly on both sides: a model that intercepts DNS while
|
||||
// configuring no `config resolver` earns a warning (there is no resolver plane to
|
||||
// intercept INTO — the queries end up at the system resolver), and a model that
|
||||
// does have one gets the .lan / PTR preservation rule prepended ahead of
|
||||
// everything else.
|
||||
//
|
||||
// # l3_tunnel (the default flip, 164b703a7)
|
||||
//
|
||||
// Same shape, one layer down. l3_tunnel is now seeded ON, and a model with no
|
||||
// tproxy inbound therefore earns the `icmp "tunnel"` warning from
|
||||
// appendL3TunInbound: the L3 ingress is only fed by the tproxy divert plane, and
|
||||
// a fixture that declares no inbounds at all raises no such plane. The warning is
|
||||
// TRUE of these fixtures — they are engine-topology models, not routers — and it
|
||||
// is noise in a test about SOCKS outbounds.
|
||||
//
|
||||
// Both defaults are correct. So the fixtures that use this helper state plainly
|
||||
// that they run neither plane, rather than keeping an "unexpected warnings"
|
||||
// assertion that would silently decay into a DNS or an ICMP assertion. The tests
|
||||
// that ARE about those planes keep the real default and pin it there:
|
||||
// dns_intercept_test.go, dnsfilter_test.go, doh_test.go, devices_test.go for DNS;
|
||||
// inbound_test.go (TestL3TunnelOnByDefaultEmitsTunInbound and its neighbours) and
|
||||
// l3mtu_test.go / l3_egress_test.go for the L3 ingress.
|
||||
func plainGlobals() model.Globals {
|
||||
g := withoutL3Tunnel(model.DefaultGlobals())
|
||||
g.DNSIntercept = false
|
||||
return g
|
||||
}
|
||||
|
||||
// withoutL3Tunnel turns the L3 ingress off on g, for the fixtures that must keep
|
||||
// the real dns_intercept default (their subject IS the DNS plane) but still are
|
||||
// not routers and have no tproxy inbound to feed the ingress with. See
|
||||
// plainGlobals for the whole argument.
|
||||
func withoutL3Tunnel(g model.Globals) model.Globals {
|
||||
g.L3Tunnel = false
|
||||
return g
|
||||
}
|
||||
|
||||
// isLocalZoneRule reports whether a DNS rule is the .lan / private-PTR
|
||||
// preservation rule that dns_intercept prepends (buildDNS): a domain_suffix rule
|
||||
// routing those zones to the synthetic `shater-local-dns` server, which points at
|
||||
|
||||
@@ -17,6 +17,8 @@
|
||||
// # What this package covers (Phase-2 MVP gate)
|
||||
//
|
||||
// - tproxy inbound (+ mixed/socks/dokodemo local listeners)
|
||||
// - the synthetic "l3-in" TUN inbound (globals l3_tunnel): the L3 ingress that
|
||||
// lets the engine carry ICMP, which kernel TPROXY cannot divert at all
|
||||
// - outbounds: vless, vmess, trojan, shadowsocks, hysteria2, tuic, shadowtls
|
||||
// with the shared TLS/Reality/uTLS container and ws/grpc/httpupgrade/http/
|
||||
// quic/xhttp transports, plus per-node multiplex
|
||||
|
||||
@@ -18,12 +18,17 @@ import (
|
||||
|
||||
"github.com/sagernet/sing-box/shater/engine"
|
||||
"github.com/sagernet/sing-box/shater/model"
|
||||
"github.com/sagernet/sing-box/shater/netplane"
|
||||
"github.com/sagernet/sing-box/shater/parse"
|
||||
)
|
||||
|
||||
// applyAndClose generates from m, feeds the result through
|
||||
// engine.New().Apply (which runs box.New for validation AND Start), and returns
|
||||
// the built options + the changed flag. A failed box.New/Start fails the test.
|
||||
//
|
||||
// The options RETURNED are the ones generate really emitted; what is handed to
|
||||
// the engine is that config minus the L3-ingress TUN inbound. See
|
||||
// withoutL3Ingress for why, and for what covers the ingress instead.
|
||||
func applyAndClose(t *testing.T, m *model.Model) (option.Options, []string, bool) {
|
||||
t.Helper()
|
||||
opts, warns, err := GenerateWithWarnings(m)
|
||||
@@ -32,13 +37,67 @@ func applyAndClose(t *testing.T, m *model.Model) (option.Options, []string, bool
|
||||
}
|
||||
e := engine.New()
|
||||
t.Cleanup(func() { _ = e.Close() })
|
||||
changed, err := e.Apply(opts)
|
||||
changed, err := e.Apply(withoutL3Ingress(t, opts))
|
||||
if err != nil {
|
||||
t.Fatalf("engine.Apply (box.New validate + start) failed: %v\nwarnings: %v", err, warns)
|
||||
}
|
||||
return opts, warns, changed
|
||||
}
|
||||
|
||||
// withoutL3Ingress returns opts with the synthetic L3-ingress TUN inbound
|
||||
// removed. The caller's copy is untouched (a fresh Inbounds slice is built), so
|
||||
// a test can still assert on the config generate really emitted.
|
||||
//
|
||||
// # Why the engine instrument in this suite does not start the L3 ingress
|
||||
//
|
||||
// Since 164b703a7 l3_tunnel is ON by default, so every fixture here that enables
|
||||
// a tproxy inbound now also carries `l3-in`. It is the only inbound this project
|
||||
// emits that is a KERNEL DEVICE rather than a socket, and a test about DNS
|
||||
// filtering, chains, rule-sets or egress presets must not open one — for two
|
||||
// reasons, and the second is the one that matters:
|
||||
//
|
||||
// - it cannot run on the gate at all. The act_runner is an LXC guest with no
|
||||
// /dev/net (let alone /dev/net/tun), so Start fails with `open /dev/net/tun:
|
||||
// no such file or directory`. Measured: 32 ORDINARY tests in this package
|
||||
// red, which is the whole CI, and the privileged/ordinary split
|
||||
// scripts/run-tests.sh is built on dissolved with them;
|
||||
// - and on a host that DOES have the device it is worse, because it is SILENT.
|
||||
// There are exactly TWO L3 slots (netplane/l3.go) and they are global to the
|
||||
// process, shared with the tests that really are about the ingress. Measured
|
||||
// under `docker run --cap-add NET_ADMIN --device /dev/net/tun`: run alone,
|
||||
// TestIntegrationL3TunInboundStarts, TestIntegrationL3EgressICMPIsAFlow and
|
||||
// TestDNSFilterLiveAnswers all PASS; run as a package they all FAIL — and
|
||||
// one of them fails by finding a `shater-l3` device that a DNS-filter test
|
||||
// created. A test that does not own a resource must not take it.
|
||||
//
|
||||
// Nothing about the ingress goes unproven because of this. Its SHAPE is pinned
|
||||
// portably (inbound_test.go TestL3Tunnel*), the fact that box.New accepts it
|
||||
// under the slim registry is pinned by an ordinary test
|
||||
// (TestL3TunInboundIsAcceptedByBoxNew), that it is the ONLY difference between a
|
||||
// default config and an l3_tunnel=0 one is pinned by
|
||||
// TestL3TunnelChangesNothingButTheTunInbound — which is what makes the Starts
|
||||
// below speak for the default config too — and Start against a real kernel is
|
||||
// TestIntegrationL3TunInboundStarts, which scripts/run-tests.sh [5/7] gives a
|
||||
// verdict by name.
|
||||
func withoutL3Ingress(t *testing.T, opts option.Options) option.Options {
|
||||
t.Helper()
|
||||
kept := make([]option.Inbound, 0, len(opts.Inbounds))
|
||||
dropped := 0
|
||||
for _, in := range opts.Inbounds {
|
||||
if to, ok := in.Options.(*option.TunInboundOptions); ok &&
|
||||
in.Type == C.TypeTun && netplane.L3IsManagedName(to.InterfaceName) {
|
||||
dropped++
|
||||
continue
|
||||
}
|
||||
kept = append(kept, in)
|
||||
}
|
||||
if dropped > 1 {
|
||||
t.Fatalf("withoutL3Ingress dropped %d TUN inbounds, want at most 1 — generate emits exactly one L3 ingress, so more than that means this helper is eating something it was never meant to touch and the engine below is validating a config nobody wrote", dropped)
|
||||
}
|
||||
opts.Inbounds = kept
|
||||
return opts
|
||||
}
|
||||
|
||||
// validKey returns a valid 32-byte base64 WireGuard key seeded by fill.
|
||||
func validKey(fill byte) string {
|
||||
b := make([]byte, 32)
|
||||
@@ -222,12 +281,13 @@ func TestAllReachableProtocols(t *testing.T) {
|
||||
const vmessLink = "vmess://eyJ2IjoiMiIsInBzIjoidm1lc3MtdyIsImFkZCI6ImV4YW1wbGUubmV0IiwicG9ydCI6IjQ0MyIsImlkIjoiMzMzMzMzMzMtMzMzMy0zMzMzLTMzMzMtMzMzMzMzMzMzMzMzIiwiYWlkIjoiMCIsInNjeSI6ImF1dG8iLCJuZXQiOiJ3cyIsImhvc3QiOiJleGFtcGxlLm5ldCIsInBhdGgiOiIvd3MiLCJ0bHMiOiJ0bHMifQ=="
|
||||
|
||||
m := &model.Model{
|
||||
// nonDNSGlobals, not DefaultGlobals: the subject here is "every share-link
|
||||
// protocol reaches box.New", and this model configures no `config resolver`,
|
||||
// so the D24 default (dns_intercept ON) would earn its own honest warning
|
||||
// about an intercept with nothing to intercept into — turning the
|
||||
// no-warnings assertion below into a DNS assertion by accident.
|
||||
Globals: nonDNSGlobals(),
|
||||
// plainGlobals, not DefaultGlobals: the subject here is "every share-link
|
||||
// protocol reaches box.New", and this model configures no `config resolver`
|
||||
// and no inbound, so the two seeded-ON planes would each earn their own
|
||||
// honest warning — dns_intercept with nothing to intercept into (D24), and
|
||||
// l3_tunnel with no tproxy divert plane to feed the L3 ingress — turning the
|
||||
// no-warnings assertion below into a DNS/ICMP assertion by accident.
|
||||
Globals: plainGlobals(),
|
||||
Inbounds: []model.Inbound{{Name: "lan", Enabled: true, Type: "tproxy", TproxyPort: 12363}},
|
||||
Nodes: []model.Node{
|
||||
{Name: "vless-ws", Enabled: true, URI: "vless://11111111-1111-1111-1111-111111111111@example.com:443?type=ws&security=tls&path=/vl&host=cdn.example.com&sni=cdn.example.com#vless-ws"},
|
||||
|
||||
@@ -56,7 +56,7 @@ func anyDetour(t *testing.T, opts option.Options, tag string) string {
|
||||
// and one deliberately not.
|
||||
func twoGroupsOneSubModel() *model.Model {
|
||||
return &model.Model{
|
||||
Globals: nonDNSGlobals(),
|
||||
Globals: plainGlobals(),
|
||||
Egresses: []model.Egress{{Name: "awg", Type: "direct"}},
|
||||
Nodes: []model.Node{
|
||||
{Name: "n1", Enabled: true, URI: ss("203.0.113.1"), FromSub: "qomar"},
|
||||
|
||||
+196
-2
@@ -9,11 +9,15 @@ import (
|
||||
C "github.com/sagernet/sing-box/constant"
|
||||
"github.com/sagernet/sing-box/option"
|
||||
"github.com/sagernet/sing/common/auth"
|
||||
"github.com/sagernet/sing/common/json/badoption"
|
||||
|
||||
"github.com/sagernet/sing-box/shater/model"
|
||||
"github.com/sagernet/sing-box/shater/netplane"
|
||||
)
|
||||
|
||||
// buildInbounds maps every enabled model.Inbound to a typed sing-box inbound.
|
||||
// buildInbounds maps every enabled model.Inbound to a typed sing-box inbound,
|
||||
// then appends the synthetic L3-ingress TUN inbound when globals l3_tunnel
|
||||
// calls for one (see appendL3TunInbound).
|
||||
//
|
||||
// # What each Type really produces
|
||||
//
|
||||
@@ -87,7 +91,197 @@ func (b *builder) buildInbounds() []option.Inbound {
|
||||
seenTag[built.Tag] = true
|
||||
inbounds = append(inbounds, built)
|
||||
}
|
||||
return inbounds
|
||||
return b.appendL3TunInbound(inbounds, seenTag)
|
||||
}
|
||||
|
||||
// The L3-ingress TUN parameters below are a fixed contract with the netplane:
|
||||
// the nft prerouting chain stamps netplane.L3Mark on LAN traffic the tunnel can
|
||||
// only carry at layer 3, ApplyRouting points the L3 table's default route at
|
||||
// one of netplane.L3Slots, and this package opens the engine's end of it.
|
||||
// None of it is operator-tunable, which is why the inbound is synthesised from
|
||||
// globals instead of being a model.Inbound.
|
||||
const (
|
||||
// l3InboundTag is how route rules and diagnostics address the L3 ingress.
|
||||
l3InboundTag = "l3-in"
|
||||
// l3PlaceholderDevice is the interface_name this package puts into the L3
|
||||
// TUN inbound. It is a PLACEHOLDER: engine.newBox rewrites it to one of
|
||||
// netplane.L3Slots before box.New (engine/l3slot.go carries the argument for
|
||||
// why the choice belongs to the engine and not here).
|
||||
//
|
||||
// It is deliberately LONGER than the kernel will take. IFNAMSIZ is 16
|
||||
// including the NUL, so 15 characters is the maximum a device name can have,
|
||||
// and this one has 21. The consequence is the whole point:
|
||||
//
|
||||
// box.New ACCEPTS it — measured — so the config generate emits is still a
|
||||
// config the engine can VALIDATE, which is what every
|
||||
// box.New-based check in this repo relies on;
|
||||
// Start REFUSES it, with `start inbound/tun[l3-in]: configure tun
|
||||
// interface: open tun: create ifreq: invalid argument`, and
|
||||
// creates NO device.
|
||||
//
|
||||
// So a config that reaches a running box WITHOUT passing through
|
||||
// engine.Apply fails immediately and visibly instead of quietly creating a
|
||||
// device under the placeholder name. That silent creation was not
|
||||
// hypothetical: two tests in this package called box.New directly, the
|
||||
// `shater-l3` device they left behind outlived them by ~4s (the kernel
|
||||
// defers unregister_netdevice long after the fd is closed), and it was found
|
||||
// by TestIntegrationL3TunInboundStarts complaining about a device it had not
|
||||
// created. On a router the same bypass is worse in kind, because the
|
||||
// placeholder is a name EVERY generation would want: that is the TUNSETIFF
|
||||
// EBUSY outage netplane/l3.go exists to describe, and it arrives
|
||||
// intermittently, which is the one failure mode that design explicitly
|
||||
// refuses to trade for.
|
||||
//
|
||||
// If you are reading this because Start failed with `create ifreq: invalid
|
||||
// argument`: nothing is wrong with the kernel or the tag set. Some caller
|
||||
// built a box from generate's output directly. Route it through
|
||||
// engine.Apply, or drop the L3 inbound if the caller is only validating.
|
||||
l3PlaceholderDevice = netplane.L3DeviceBase + "-placeholder"
|
||||
// l3MTU is 65535 — the largest total length an IPv4 datagram can carry —
|
||||
// and that maximum IS the reason, not a round number.
|
||||
//
|
||||
// This MTU is NOT a tunnel budget. It decides exactly one thing: whether
|
||||
// the KERNEL fragments a packet on its way INTO the device. What the
|
||||
// engine then puts into the tunnel is sized against the OUTBOUND's own
|
||||
// MTU: sing-tun's ForwardDispatcher.forwardToPort measures every forwarded
|
||||
// packet against Port.PortMTU() — the WireGuard/AWG endpoint's MTU — and
|
||||
// either fragments to it (no DF) or answers a proper `fragmentation
|
||||
// needed` quoting it (DF). It already does that work correctly; it only
|
||||
// has to be handed a WHOLE packet to do it.
|
||||
//
|
||||
// The previous value, 1420, was the WireGuard payload budget copied one
|
||||
// layer too far out, and it was not merely useless — it was a forgery
|
||||
// generator. Anything bigger was fragmented by the kernel at this device,
|
||||
// and sing-tun's dispatcher returns on `parsed.fragment` BEFORE asking for
|
||||
// a routing verdict at all (flow_dispatch.go). The fragments then reached
|
||||
// the gVisor stack, which reassembled them and handed the echo to
|
||||
// ICMPForwarder.HandlePacket, whose installFlow demands an UNSPECIFIED
|
||||
// port address that a WireGuard endpoint never has — so it fell through
|
||||
// and ANSWERED THE ECHO ITSELF. Net effect: `ping -s 1392` was honest and
|
||||
// `ping -s 1393` was a lie told by the router. See D25.
|
||||
//
|
||||
// 65535 is chosen over any other large value because no IP datagram can
|
||||
// exceed it: the kernel therefore CANNOT fragment at this device, for any
|
||||
// packet, ever. Any smaller value leaves a band open and re-opens the bug.
|
||||
// It is also sing-box's own default TUN MTU on Linux
|
||||
// (protocol/tun/inbound.go), so it is a well-trodden value.
|
||||
//
|
||||
// It costs NO memory, and that was measured rather than assumed: three
|
||||
// paired runs of TestIntegrationL3TunInboundStarts under
|
||||
// -test.memprofilerate=1 (exact accounting, not sampled) allocate 5.41 /
|
||||
// 5.48 / 5.47 MB at 65535 against 5.76 / 5.46 / 5.70 MB at 1420, and a
|
||||
// -diff_base profile attributes every difference to netlink interface
|
||||
// enumeration, not to the MTU. Nothing in the read path scales with it:
|
||||
// gVisor reads through fdbased.BufConfig, which sing-tun's init pins to a
|
||||
// single 65535-byte view regardless of MTU, and fdbased keeps `mtu` only
|
||||
// to return it from MTU().
|
||||
//
|
||||
// Do NOT expect a saving from GSO either, tempting as the arithmetic is.
|
||||
// protocol/tun computes `enableGSO = stack == gvisor && mtu < 49152`, so
|
||||
// this MTU turns it off there — and then StartStateStart turns it back ON
|
||||
// unconditionally because an adapter.FlowOutbound exists in the config
|
||||
// (protocol/tun/inbound.go, the outbound/endpoint scan). The ~1.98 MB of
|
||||
// TCP/UDP GRO scaffolding is therefore present at BOTH MTUs; it is priced
|
||||
// by the presence of a flow-capable outbound, not by this number.
|
||||
l3MTU = 65535
|
||||
)
|
||||
|
||||
// l3Addr4/l3Addr6 are the device's point-to-point addresses — the kernel only
|
||||
// routes over an interface that has one. The prefixes are deliberately tiny
|
||||
// (/30, /126) and from private space no sane LAN uses, so they cannot shadow a
|
||||
// real subnet.
|
||||
var (
|
||||
l3Addr4 = netip.MustParsePrefix("172.19.242.1/30")
|
||||
l3Addr6 = netip.MustParsePrefix("fdfe:d3ad:b33f::1/126")
|
||||
)
|
||||
|
||||
// appendL3TunInbound appends the synthetic L3-ingress TUN inbound when globals
|
||||
// l3_tunnel opted in. Kernel TPROXY diverts nothing but TCP/UDP, so the rest of
|
||||
// the LAN's traffic (in practice: ICMP echo) can reach the engine only as raw
|
||||
// IP packets through a TUN device; the netplane policy-routes such packets into
|
||||
// the live netplane L3 slot and the engine picks them up here.
|
||||
//
|
||||
// There is deliberately no loop-guard RoutingMark: a TUN inbound is not a
|
||||
// socket (option.TunInboundOptions carries no ListenOptions), and the loop
|
||||
// guard already lives on every outbound dialer (DialerOptions.RoutingMark =
|
||||
// LoopMark), which is what keeps the engine's own egress out of the divert.
|
||||
//
|
||||
// Warnings carry the `icmp "tunnel"` entity prefix — the same entity the
|
||||
// netplane's L3 warnings use — so the panel groups everything about the L3
|
||||
// ingress under one section instead of scattering it.
|
||||
func (b *builder) appendL3TunInbound(inbounds []option.Inbound, seenTag map[string]bool) []option.Inbound {
|
||||
if !b.m.Globals.L3Tunnel {
|
||||
return inbounds
|
||||
}
|
||||
hasTproxy := false
|
||||
for _, in := range inbounds {
|
||||
if in.Type == C.TypeTProxy {
|
||||
hasTproxy = true
|
||||
break
|
||||
}
|
||||
}
|
||||
if !hasTproxy {
|
||||
// The L3 ingress rides the same LAN divert plane as tproxy: with no
|
||||
// tproxy inbound the netplane raises no divert chain and marks nothing
|
||||
// (nft.go gates the L3 marking on `L3Enabled(globals) && dnsIif != ""`,
|
||||
// and that interface name comes from the primary tproxy inbound), so the
|
||||
// TUN would sit dark while the config claims ICMP is tunnelled.
|
||||
//
|
||||
// The wording deliberately does NOT open with "l3_tunnel is on". Since
|
||||
// the default flip that is no longer a decision the reader made, and a
|
||||
// message that reads as "you turned this on" sends them hunting for a
|
||||
// switch they never touched. What they need instead is: what is not
|
||||
// happening, why, and which of the two exits applies to their router.
|
||||
b.warnf("icmp \"tunnel\": the %q L3 ingress is not started — LAN ping is not carried by the tunnel on this router. l3_tunnel is ON BY DEFAULT, so this is not a switch anyone flipped here; the ingress is simply not a standalone device. It carries only the LAN traffic the tproxy divert plane marks into it, and this config enables no tproxy inbound, so no divert plane is raised and the TUN would sit dark. Enable a tproxy inbound if this router is meant to divert LAN traffic; on a proxy-only box (socks/http listeners only) nothing is missing here — set `option l3_tunnel '0'` to say so and this notice goes away", l3InboundTag)
|
||||
return inbounds
|
||||
}
|
||||
if seenTag[l3InboundTag] {
|
||||
// Unreachable today — inboundTag prefixes every model-derived tag with
|
||||
// "in-" — but kept on the same fail-degraded principle as the guards in
|
||||
// buildInbounds: losing the L3 ingress is strictly smaller damage than
|
||||
// two inbounds fighting over one tag.
|
||||
b.warnf("icmp \"tunnel\": inbound tag %q is already taken — the L3 ingress inbound is skipped so the existing listener keeps working (rename the clashing inbound to restore it)", l3InboundTag)
|
||||
return inbounds
|
||||
}
|
||||
addrs := badoption.Listable[netip.Prefix]{l3Addr4}
|
||||
if b.m.Globals.IPv6 {
|
||||
addrs = append(addrs, l3Addr6)
|
||||
}
|
||||
return append(inbounds, option.Inbound{
|
||||
Type: C.TypeTun,
|
||||
Tag: l3InboundTag,
|
||||
Options: &option.TunInboundOptions{
|
||||
// A PLACEHOLDER, not the device that gets created — and one the
|
||||
// kernel cannot take, so it cannot BECOME the device that gets
|
||||
// created either. The engine rewrites this to one of
|
||||
// netplane.L3Slots in newBox, because two generations must never
|
||||
// contend for one device name (l3.go has the whole argument,
|
||||
// including why the choice cannot be made here: this function runs
|
||||
// on every no-op reconcile, and its output is what the engine
|
||||
// hashes, so alternating the name here would rebuild the engine once
|
||||
// a minute). See l3PlaceholderDevice for why it is over-long.
|
||||
InterfaceName: l3PlaceholderDevice,
|
||||
MTU: l3MTU,
|
||||
Address: addrs,
|
||||
// AutoRoute MUST stay false. auto_route rewrites the router's MAIN
|
||||
// routing table and would drag everything the router itself sends —
|
||||
// WAN traffic, DNS, the tunnel's own underlay — into this TUN. The
|
||||
// netplane installs the scoped rule/route (fwmark L3Mark -> L3Table
|
||||
// -> the live slot) itself; that is the whole routing story.
|
||||
AutoRoute: false,
|
||||
// gvisor is a CHOICE, not a necessity, and the tempting reason for
|
||||
// it is wrong: BOTH sing-tun stacks forward ICMP through the same
|
||||
// ForwardDispatcher first, and both forge an echo reply only for
|
||||
// what that dispatcher declined (system: stack_system.go
|
||||
// dispatchIPv4 -> processIPv4ICMP; gvisor: stack_gvisor_filter.go
|
||||
// -> ICMPForwarder.HandlePacket). Do not re-derive this as "the
|
||||
// system stack fakes ping" — D25 says so explicitly. gvisor is
|
||||
// picked because it is already linked (with_wireguard requires
|
||||
// with_gvisor, D23), so it costs no build tag and no new code path,
|
||||
// and because it is the combination the integration test exercises.
|
||||
Stack: "gvisor",
|
||||
},
|
||||
})
|
||||
}
|
||||
|
||||
// listenKey returns the "addr:port" an inbound binds, for the clash guard. ok is
|
||||
|
||||
+357
-10
@@ -10,19 +10,40 @@
|
||||
package generate
|
||||
|
||||
import (
|
||||
"context"
|
||||
"strings"
|
||||
"testing"
|
||||
|
||||
C "github.com/sagernet/sing-box/constant"
|
||||
"github.com/sagernet/sing-box/option"
|
||||
"github.com/sagernet/sing/common/json"
|
||||
|
||||
"github.com/sagernet/sing-box/shater/model"
|
||||
"github.com/sagernet/sing-box/shater/netplane"
|
||||
"github.com/sagernet/sing-box/shater/registry"
|
||||
)
|
||||
|
||||
// buildInboundsOf runs buildInbounds over a single inbound and returns the
|
||||
// emitted inbounds plus the warnings.
|
||||
//
|
||||
// l3_tunnel is turned OFF here, and only here, because every test that uses this
|
||||
// helper asks a question about ONE model.Inbound: which listener type it becomes,
|
||||
// which of its fields reach the engine, which of them are reported as ignored.
|
||||
// The synthetic L3 ingress is not derived from a model.Inbound at all — it is
|
||||
// synthesised from globals — so with the seeded-ON default it would add a second
|
||||
// emitted inbound to every count below and a second warning to every
|
||||
// "no warnings" assertion, neither of which is about the inbound under test.
|
||||
// The L3 ingress has its own tests further down this file, and they use globals
|
||||
// that say what they mean.
|
||||
func buildInboundsOf(in model.Inbound) ([]option.Inbound, []string) {
|
||||
b := newBuilder(&model.Model{Globals: model.DefaultGlobals(), Inbounds: []model.Inbound{in}})
|
||||
return buildInboundsWith(withoutL3Tunnel(model.DefaultGlobals()), in)
|
||||
}
|
||||
|
||||
// buildInboundsWith is buildInboundsOf with explicit globals and any number of
|
||||
// inbounds — for the knobs (l3_tunnel, ipv6) that decide WHICH inbounds exist
|
||||
// rather than how a single one is shaped.
|
||||
func buildInboundsWith(g model.Globals, ins ...model.Inbound) ([]option.Inbound, []string) {
|
||||
b := newBuilder(&model.Model{Globals: g, Inbounds: ins})
|
||||
return b.buildInbounds(), b.warnings
|
||||
}
|
||||
|
||||
@@ -288,20 +309,346 @@ func TestInboundBadListenAddrWarns(t *testing.T) {
|
||||
// sniff-related may appear on the emitted inbound options. option.InboundOptions
|
||||
// is the legacy pre-1.11 home of sniff/domain_strategy and sing-box 1.13+ REJECTS
|
||||
// it, so a regression here fails box.New and takes the tunnel down.
|
||||
// It checks EVERY emitted listener rather than ins[0], and counts how many it
|
||||
// checked. The old shape asserted "exactly 1 inbound" purely so it could index
|
||||
// ins[0] — a count that says nothing about sniffing and that any future
|
||||
// synthetic inbound (the L3 ingress was the first) breaks for no reason. The
|
||||
// checked counter replaces the guarantee that assertion was really providing:
|
||||
// that the loop ran at all.
|
||||
func TestSniffIsNotAnInboundField(t *testing.T) {
|
||||
ins, _ := buildInboundsOf(model.Inbound{
|
||||
Name: "lan", Enabled: true, Type: "tproxy", TproxyPort: 12345,
|
||||
TCP: true, UDP: true,
|
||||
})
|
||||
if len(ins) != 1 {
|
||||
t.Fatalf("want 1 inbound, got %d", len(ins))
|
||||
checked := 0
|
||||
for _, in := range ins {
|
||||
lw, ok := in.Options.(option.ListenOptionsWrapper)
|
||||
if !ok {
|
||||
continue // e.g. a TUN inbound: no socket, no ListenOptions to carry it
|
||||
}
|
||||
checked++
|
||||
//nolint:staticcheck // asserting the legacy field stays untouched
|
||||
if lw.TakeListenOptions().InboundOptions != (option.InboundOptions{}) {
|
||||
t.Fatalf("inbound %q: legacy inbound sniff fields must stay empty (rejected by sing-box 1.13+)", in.Tag)
|
||||
}
|
||||
}
|
||||
lw, ok := ins[0].Options.(option.ListenOptionsWrapper)
|
||||
if !ok {
|
||||
t.Fatal("inbound carries no listen options")
|
||||
}
|
||||
//nolint:staticcheck // asserting the legacy field stays untouched
|
||||
if lw.TakeListenOptions().InboundOptions != (option.InboundOptions{}) {
|
||||
t.Fatal("legacy inbound sniff fields must stay empty (rejected by sing-box 1.13+)")
|
||||
if checked == 0 {
|
||||
t.Fatal("no listener was emitted, so nothing was actually checked")
|
||||
}
|
||||
}
|
||||
|
||||
// tunInbounds filters the emitted inbounds down to the TUN ones. The synthetic
|
||||
// L3 ingress is their only source — TestInboundTypeMapping pins that a user
|
||||
// inbound of type "tun" stays rejected.
|
||||
func tunInbounds(ins []option.Inbound) []option.Inbound {
|
||||
var tuns []option.Inbound
|
||||
for _, in := range ins {
|
||||
if in.Type == C.TypeTun {
|
||||
tuns = append(tuns, in)
|
||||
}
|
||||
}
|
||||
return tuns
|
||||
}
|
||||
|
||||
// lanTproxy is the minimal enabled tproxy inbound the L3 tests pair with: the
|
||||
// L3 ingress only exists alongside the LAN divert plane.
|
||||
func lanTproxy() model.Inbound {
|
||||
return model.Inbound{Name: "lan", Enabled: true, Type: "tproxy", TproxyPort: 12345, TCP: true, UDP: true}
|
||||
}
|
||||
|
||||
// TestL3TunnelOffEmitsNoTunInbound: the opt-OUT is a real answer and it must be
|
||||
// honoured. l3_tunnel is now seeded ON (164b703a7), so "off" is no longer the
|
||||
// absence of a decision — it is an explicit `option l3_tunnel '0'` from an
|
||||
// operator with a reason: the synthetic TUN inbound brings a device, a gVisor
|
||||
// netstack and ~2 MB of standing RSS with it, which is real money on a 32/64 MB
|
||||
// router, and turning it off is also how you bisect whether the L3 ingress is
|
||||
// what broke a box. A generator that emitted the TUN anyway would make that
|
||||
// answer unavailable, and the operator would have no other way to reach it.
|
||||
//
|
||||
// This is the SAME property the test pinned before the flip; only its premise
|
||||
// changed, from "the default is off" to "off is what was asked for". Which is
|
||||
// why the globals now say so explicitly instead of relying on the default —
|
||||
// leaning on the default is exactly what made this test start passing vacuously
|
||||
// (it asserted nothing about l3_tunnel at all, only about DefaultGlobals).
|
||||
func TestL3TunnelOffEmitsNoTunInbound(t *testing.T) {
|
||||
g := model.DefaultGlobals()
|
||||
g.L3Tunnel = false
|
||||
ins, _ := buildInboundsWith(g, lanTproxy())
|
||||
if got := tunInbounds(ins); len(got) != 0 {
|
||||
t.Fatalf("l3_tunnel is explicitly off, yet a TUN inbound was emitted — the opt-out is the only way off a standing TUN + gVisor netstack, and it was ignored")
|
||||
}
|
||||
}
|
||||
|
||||
// TestL3TunnelOnByDefaultEmitsTunInbound is the other half, and the one that
|
||||
// actually pins the FLIP: a model that says nothing about l3_tunnel — the shape
|
||||
// every install predating the option comes back as, and the shape DefaultGlobals
|
||||
// hands out — must get the L3 ingress. Without this the whole feature could be
|
||||
// switched back to opt-in and the only test that noticed would be in
|
||||
// shater/model (which pins the seed value, not what the generator does with it).
|
||||
//
|
||||
// It deliberately reads the default rather than setting L3Tunnel itself: the
|
||||
// claim under test is "the default emits it", and a fixture that assigns the
|
||||
// field can only ever prove "true emits it", which TestL3TunnelEmitsTunInbound
|
||||
// already covers in full detail.
|
||||
func TestL3TunnelOnByDefaultEmitsTunInbound(t *testing.T) {
|
||||
g := model.DefaultGlobals()
|
||||
if !g.L3Tunnel {
|
||||
t.Fatal("model.DefaultGlobals().L3Tunnel = false — the L3 ingress is meant to be seeded ON; the rest of this test would pass vacuously, so it stops here")
|
||||
}
|
||||
ins, warns := buildInboundsWith(g, lanTproxy())
|
||||
tuns := tunInbounds(ins)
|
||||
if len(tuns) != 1 {
|
||||
t.Fatalf("a model that never mentions l3_tunnel emitted %d TUN inbounds, want exactly 1 — on the seeded-ON default every router carries the L3 ingress, and without it a LAN ping is either dropped or sent out of the WAN with the client's real address (warns=%v)", len(tuns), warns)
|
||||
}
|
||||
if tuns[0].Tag != l3InboundTag {
|
||||
t.Fatalf("tag = %q, want %q", tuns[0].Tag, l3InboundTag)
|
||||
}
|
||||
if len(warns) != 0 {
|
||||
t.Fatalf("the default posture (l3_tunnel on + a tproxy inbound) must be silent, got %v", warns)
|
||||
}
|
||||
}
|
||||
|
||||
// TestL3TunnelEmitsTunInbound pins the data-plane contract of the L3 ingress.
|
||||
// Every field is load-bearing: auto_route off keeps sing-box away from the
|
||||
// router's MAIN routing table, and only the gVisor stack actually forwards ICMP
|
||||
// — the system stack forges echo replies locally.
|
||||
//
|
||||
// interface_name must be the canonical PLACEHOLDER netplane.L3DeviceBase and
|
||||
// nothing else. The engine substitutes a real slot for it at box-build time
|
||||
// (engine/l3slot.go), and it can only do that if the name it is handed is one it
|
||||
// recognises — but the deeper reason this is pinned is the other direction: what
|
||||
// generate emits is what the engine HASHES, so a per-apply name here would make
|
||||
// every once-a-minute no-op reconcile look like a config change and rebuild the
|
||||
// whole engine, dropping every connection through the tunnel.
|
||||
func TestL3TunnelEmitsTunInbound(t *testing.T) {
|
||||
g := model.DefaultGlobals()
|
||||
g.L3Tunnel = true
|
||||
ins, warns := buildInboundsWith(g, lanTproxy())
|
||||
tuns := tunInbounds(ins)
|
||||
if len(tuns) != 1 {
|
||||
t.Fatalf("want exactly one L3 TUN inbound, got %d (warns=%v)", len(tuns), warns)
|
||||
}
|
||||
if tuns[0].Tag != "l3-in" {
|
||||
t.Fatalf("tag = %q, want %q — route rules address the L3 ingress by exactly this tag, so any other spelling detaches it from its routing", tuns[0].Tag, "l3-in")
|
||||
}
|
||||
to, ok := tuns[0].Options.(*option.TunInboundOptions)
|
||||
if !ok {
|
||||
t.Fatalf("options are %T, want *option.TunInboundOptions", tuns[0].Options)
|
||||
}
|
||||
if to.InterfaceName != l3PlaceholderDevice {
|
||||
t.Fatalf("interface_name = %q, want the canonical placeholder %q.\n"+
|
||||
"Three things break if this moves: the engine's slot substitution only recognises a name "+
|
||||
"netplane.L3IsManagedName accepts; this string is part of what the engine "+
|
||||
"hashes to decide whether a config CHANGED, so anything that varies per apply rebuilds the "+
|
||||
"engine on every one-minute reconcile; and the name must stay one the KERNEL refuses, "+
|
||||
"see TestL3PlaceholderCannotBecomeAKernelDevice.", to.InterfaceName, l3PlaceholderDevice)
|
||||
}
|
||||
if !netplane.L3IsManagedName(to.InterfaceName) {
|
||||
t.Fatalf("netplane.L3IsManagedName(%q) = false — engine/l3slot.go finds the L3 inbound by exactly this predicate, so the substitution would not happen and the config would go to box.New with the placeholder still in it", to.InterfaceName)
|
||||
}
|
||||
if to.AutoRoute {
|
||||
t.Fatal("auto_route is on — sing-box would rewrite the router's MAIN routing table and drag the router's own WAN/DNS/underlay traffic into the tunnel; the netplane owns the scoped L3 rules")
|
||||
}
|
||||
if to.Stack != "gvisor" {
|
||||
t.Fatalf("stack = %q, want gvisor — the system stack answers ICMP echo locally instead of forwarding it, which is the exact forged reply l3_tunnel exists to remove", to.Stack)
|
||||
}
|
||||
if to.MTU != 65535 {
|
||||
t.Fatalf("mtu = %d, want 65535 — see TestL3TunnelMTULeavesNothingForTheKernelToFragment for why the number is the maximum and not a tunnel budget", to.MTU)
|
||||
}
|
||||
if len(to.Address) != 2 || to.Address[0].String() != "172.19.242.1/30" || to.Address[1].String() != "fdfe:d3ad:b33f::1/126" {
|
||||
t.Fatalf("address = %v, want [172.19.242.1/30 fdfe:d3ad:b33f::1/126] — the netplane's routes are built against exactly these prefixes", to.Address)
|
||||
}
|
||||
}
|
||||
|
||||
// TestL3TunnelIPv6OffKeepsV4Only: with globals ipv6 off the netplane installs
|
||||
// no v6 rules, so a v6 prefix on the TUN would advertise an ICMPv6 path that
|
||||
// dead-ends inside the device.
|
||||
func TestL3TunnelIPv6OffKeepsV4Only(t *testing.T) {
|
||||
g := model.DefaultGlobals()
|
||||
g.L3Tunnel = true
|
||||
g.IPv6 = false
|
||||
ins, _ := buildInboundsWith(g, lanTproxy())
|
||||
tuns := tunInbounds(ins)
|
||||
if len(tuns) != 1 {
|
||||
t.Fatalf("want the L3 TUN inbound, got %d", len(tuns))
|
||||
}
|
||||
to, ok := tuns[0].Options.(*option.TunInboundOptions)
|
||||
if !ok {
|
||||
t.Fatalf("options are %T, want *option.TunInboundOptions", tuns[0].Options)
|
||||
}
|
||||
if len(to.Address) != 1 || !to.Address[0].Addr().Is4() {
|
||||
t.Fatalf("address = %v — on an ipv6=off router only the v4 prefix may remain; a v6 address advertises an ICMPv6 path the netplane never routes", to.Address)
|
||||
}
|
||||
}
|
||||
|
||||
// TestL3TunnelWithoutTproxySkipped: the L3 ingress rides the LAN divert plane
|
||||
// that only exists alongside a tproxy inbound. Without one the TUN would sit
|
||||
// dark while the config claims ICMP is tunnelled — so it is skipped, and the
|
||||
// skip is said out loud.
|
||||
func TestL3TunnelWithoutTproxySkipped(t *testing.T) {
|
||||
g := model.DefaultGlobals()
|
||||
g.L3Tunnel = true
|
||||
ins, warns := buildInboundsWith(g, model.Inbound{
|
||||
Name: "sock", Enabled: true, Type: "socks", Listen: "127.0.0.1", Port: 1080, TCP: true, UDP: true,
|
||||
})
|
||||
if got := tunInbounds(ins); len(got) != 0 {
|
||||
t.Fatalf("no tproxy inbound is enabled, yet the L3 TUN inbound was emitted — it would carry nothing while claiming ICMP coverage")
|
||||
}
|
||||
if !warnsHave(warns, "no tproxy inbound") {
|
||||
t.Fatalf("expected a warning explaining the skipped L3 ingress, got %v", warns)
|
||||
}
|
||||
if !warnsHave(warns, `icmp "tunnel": `) {
|
||||
t.Fatalf("the L3 warning must carry the `icmp \"tunnel\"` entity prefix — without it the panel cannot group it with the data plane's L3 warnings and it lands nameless under a bare \"generate\" section; got %v", warns)
|
||||
}
|
||||
// Both exits must be named. Since the default flip the reader did not choose
|
||||
// this state, so a message that only says what is broken sends them looking
|
||||
// for a switch they never touched: one exit restores the ingress (a tproxy
|
||||
// inbound), the other says the router does not want it (`l3_tunnel '0'`), and
|
||||
// which one is right is a fact about their router that the generator cannot
|
||||
// know. Naming only one of them would push every proxy-only box toward
|
||||
// enabling a tproxy listener it has no use for.
|
||||
if !warnsHave(warns, "ON BY DEFAULT") {
|
||||
t.Fatalf("the warning must say the ingress is on by DEFAULT — otherwise it reads as an accusation about a switch the operator never flipped; got %v", warns)
|
||||
}
|
||||
if !warnsHave(warns, "l3_tunnel '0'") {
|
||||
t.Fatalf("the warning must name the opt-out as the other legitimate exit, or a proxy-only box has no way out of it but to enable a tproxy listener it does not want; got %v", warns)
|
||||
}
|
||||
|
||||
// The same holds for a config with no inbounds at all (every one disabled).
|
||||
ins, warns = buildInboundsWith(g)
|
||||
if got := tunInbounds(ins); len(got) != 0 {
|
||||
t.Fatalf("an inbound-less config emitted the L3 TUN inbound — it would carry nothing while claiming ICMP coverage")
|
||||
}
|
||||
if !warnsHave(warns, "no tproxy inbound") {
|
||||
t.Fatalf("expected a warning explaining the skipped L3 ingress, got %v", warns)
|
||||
}
|
||||
}
|
||||
|
||||
// TestL3PlaceholderCannotBecomeAKernelDevice pins the property that keeps the
|
||||
// generated config from creating a TUN behind the engine's back: the
|
||||
// interface_name generate emits is LONGER than any Linux device name may be.
|
||||
//
|
||||
// IFNAMSIZ is 16 including the terminating NUL, so 15 characters is the maximum
|
||||
// and anything at or above 16 is refused. Measured in `docker run --cap-add
|
||||
// NET_ADMIN --device /dev/net/tun` (golang:1.26, kernel 6.x): box.New ACCEPTS
|
||||
// the over-long name — so the config generate emits is still one the engine can
|
||||
// validate — while Start refuses it with `configure tun interface: open tun:
|
||||
// create ifreq: invalid argument` and creates NO device. With the old 9-character
|
||||
// placeholder the same Start succeeded and left a `shater-l3` device behind.
|
||||
//
|
||||
// That device is not a cosmetic problem. `shater-l3` is the ONE name no
|
||||
// generation may take, because every generation would want it — that is the
|
||||
// TUNSETIFF EBUSY outage netplane/l3.go was written for. The two slots exist so
|
||||
// the name is never contended; a bypass that opens the placeholder puts the
|
||||
// contention back, and (measured) does it silently.
|
||||
//
|
||||
// The assertion is on the LENGTH rather than on the literal so it keeps its
|
||||
// meaning if the spelling changes: what must survive is "the kernel will not
|
||||
// take this", not "it is spelled -placeholder".
|
||||
func TestL3PlaceholderCannotBecomeAKernelDevice(t *testing.T) {
|
||||
const ifnamsiz = 16 // linux/if.h: IFNAMSIZ, including the NUL
|
||||
if len(l3PlaceholderDevice) < ifnamsiz {
|
||||
t.Fatalf("l3PlaceholderDevice = %q is %d bytes, which the kernel ACCEPTS (max %d + NUL). "+
|
||||
"Then a box built from generate's output without going through engine.Apply creates a real "+
|
||||
"device under the placeholder name — the one name every generation wants, i.e. the TUNSETIFF "+
|
||||
"EBUSY outage netplane/l3.go exists to prevent, and it arrives intermittently. Keep it at %d "+
|
||||
"bytes or more.", l3PlaceholderDevice, len(l3PlaceholderDevice), ifnamsiz-1, ifnamsiz)
|
||||
}
|
||||
if !netplane.L3IsManagedName(l3PlaceholderDevice) {
|
||||
t.Fatalf("netplane.L3IsManagedName(%q) = false — the engine finds the L3 inbound by that predicate "+
|
||||
"(engine/l3slot.go), so an unrecognised placeholder is never substituted and EVERY apply then "+
|
||||
"fails at Start with the over-long name instead of opening a slot", l3PlaceholderDevice)
|
||||
}
|
||||
}
|
||||
|
||||
// TestL3TunnelChangesNothingButTheTunInbound is what lets the engine suite in
|
||||
// this package start its configs with the L3 ingress taken out (see
|
||||
// withoutL3Ingress) and still speak about the DEFAULT config.
|
||||
//
|
||||
// Without it the argument would be an assumption: "removing the TUN inbound
|
||||
// probably changes nothing else". This measures it instead — the canonical JSON
|
||||
// of a default (l3_tunnel on) config, minus the l3-in inbound, must be
|
||||
// byte-identical to the canonical JSON of the same model with l3_tunnel off. Same
|
||||
// marshaller the engine hashes with (engine.hashOptions), so "identical" here is
|
||||
// the same word the apply fast path uses.
|
||||
//
|
||||
// SCOPE, stated rather than implied: l3_tunnel has exactly one other reader in
|
||||
// this package, warnICMPRule (route.go), which emits WARNINGS for `proto icmp`
|
||||
// rules and never changes a config byte. The model below deliberately has no
|
||||
// ICMP rule, so the equality is about the config; the warning half is pinned by
|
||||
// the route tests.
|
||||
func TestL3TunnelChangesNothingButTheTunInbound(t *testing.T) {
|
||||
newModel := func(l3 bool) *model.Model {
|
||||
g := model.DefaultGlobals()
|
||||
g.L3Tunnel = l3
|
||||
g.ResolverDefault = "cf"
|
||||
return &model.Model{
|
||||
Globals: g,
|
||||
Inbounds: []model.Inbound{{Name: "lan", Enabled: true, Type: "tproxy", TproxyPort: 12421, TCP: true, UDP: true}},
|
||||
Nodes: []model.Node{{Name: "exit", Enabled: true, URI: "ss://aes-256-gcm:secret@203.0.113.1:8388#exit"}},
|
||||
Resolvers: []model.Resolver{
|
||||
{Name: "cf", Type: "doh", Address: "https://1.1.1.1/dns-query", Detour: "direct"},
|
||||
},
|
||||
Rules: []model.Rule{{Name: "via-exit", Enabled: true, Order: 10, DstPort: "443", Target: "node:exit"}},
|
||||
}
|
||||
}
|
||||
|
||||
on, warnsOn, err := GenerateWithWarnings(newModel(true))
|
||||
if err != nil {
|
||||
t.Fatalf("Generate(l3_tunnel on): %v", err)
|
||||
}
|
||||
off, warnsOff, err := GenerateWithWarnings(newModel(false))
|
||||
if err != nil {
|
||||
t.Fatalf("Generate(l3_tunnel off): %v", err)
|
||||
}
|
||||
if len(warnsOn) != 0 || len(warnsOff) != 0 {
|
||||
t.Fatalf("this fixture must be silent on both sides, got on=%v off=%v — a warning means one of the two configs was degraded and the comparison below would be between something other than what it claims", warnsOn, warnsOff)
|
||||
}
|
||||
|
||||
// Control first: the two configs must actually DIFFER, or the equality below
|
||||
// would be satisfied by a generator that ignores l3_tunnel entirely.
|
||||
if len(on.Inbounds) != len(off.Inbounds)+1 {
|
||||
t.Fatalf("l3_tunnel on emitted %d inbounds and off emitted %d, want exactly one more — the whole comparison assumes the ingress is one inbound", len(on.Inbounds), len(off.Inbounds))
|
||||
}
|
||||
if len(tunInbounds(off.Inbounds)) != 0 {
|
||||
t.Fatalf("l3_tunnel off still emitted a TUN inbound")
|
||||
}
|
||||
|
||||
stripped := withoutL3IngressOpts(on)
|
||||
if a, b := canonicalJSON(t, stripped), canonicalJSON(t, off); a != b {
|
||||
t.Fatalf("a default config with the l3-in inbound removed is NOT the l3_tunnel=0 config.\n"+
|
||||
"l3_tunnel therefore changes something besides that one inbound, and every engine.Apply in this "+
|
||||
"package — which starts the stripped form (withoutL3Ingress) — stops speaking for the default "+
|
||||
"config.\n on(stripped) = %s\n off = %s", a, b)
|
||||
}
|
||||
}
|
||||
|
||||
// withoutL3IngressOpts is the non-testing half of withoutL3Ingress: it drops the
|
||||
// L3-ingress TUN inbound from a copy of opts. It lives here, without a *testing.T,
|
||||
// so the portable comparison above can use it on every platform (withoutL3Ingress
|
||||
// is in the linux-only engine suite).
|
||||
func withoutL3IngressOpts(opts option.Options) option.Options {
|
||||
kept := make([]option.Inbound, 0, len(opts.Inbounds))
|
||||
for _, in := range opts.Inbounds {
|
||||
if to, ok := in.Options.(*option.TunInboundOptions); ok &&
|
||||
in.Type == C.TypeTun && netplane.L3IsManagedName(to.InterfaceName) {
|
||||
continue
|
||||
}
|
||||
kept = append(kept, in)
|
||||
}
|
||||
opts.Inbounds = kept
|
||||
return opts
|
||||
}
|
||||
|
||||
// canonicalJSON renders opts through the SAME marshaller engine.hashOptions
|
||||
// hashes with (the context-aware one, so the typed inbound/outbound options can
|
||||
// be encoded at all). Comparing these strings is comparing what the engine's
|
||||
// apply fast path would compare.
|
||||
func canonicalJSON(t *testing.T, opts option.Options) string {
|
||||
t.Helper()
|
||||
data, err := json.MarshalContext(registry.Context(context.Background()), opts)
|
||||
if err != nil {
|
||||
t.Fatalf("marshal options: %v", err)
|
||||
}
|
||||
return string(data)
|
||||
}
|
||||
|
||||
@@ -0,0 +1,161 @@
|
||||
//go:build linux
|
||||
|
||||
// The egress half of the L3 ICMP story, live-engine part: not the SHAPE of the
|
||||
// config (l3_egress_test.go pins that portably) but what the running engine
|
||||
// BELIEVES about the two egress kinds. The belief is the whole feature:
|
||||
//
|
||||
// - An interface egress must come up as an adapter.FlowOutbound whose
|
||||
// PreMatchFlow answers Flow for ICMP. That answer exists only if
|
||||
// protocol/direct's constructor actually built its ping.Port, which it does
|
||||
// only when dialer.NewWithOptions hands back a *dialer.DefaultDialer for a
|
||||
// dialer that carries BindInterface + RoutingMark. Nothing in the portable
|
||||
// suite can see this: if that cast ever stops holding (an upstream bump
|
||||
// wrapping the bound dialer, say), the generated JSON stays byte-identical,
|
||||
// every codegen test stays green, and ping through every interface egress
|
||||
// silently degrades from "leaves via the second WAN" to "dropped".
|
||||
// - A byedpi egress must NOT look ICMP-capable: route.preMatchFlow
|
||||
// (l3-honest-drop) drops an ICMP flow whose outbound either lacks icmp in
|
||||
// Network() or is not a FlowOutbound, and SOCKS satisfies both refusals. If
|
||||
// it ever stops refusing, the drop stops happening — and the TUN stack's
|
||||
// alternative is forging the echo reply itself.
|
||||
//
|
||||
// Gating mirrors l3_integration_linux_test.go, whose comment carries the full
|
||||
// argument: the model opts into l3_tunnel, so Start opens /dev/net/tun and
|
||||
// needs root + CAP_NET_ADMIN, which the ordinary gate's containers do not
|
||||
// expose; the TestIntegration prefix and the honest skips below keep the gate
|
||||
// green while telling a human exactly how to run this for real.
|
||||
package generate
|
||||
|
||||
import (
|
||||
"net/netip"
|
||||
"os"
|
||||
"slices"
|
||||
"testing"
|
||||
|
||||
"github.com/sagernet/sing-box/adapter"
|
||||
N "github.com/sagernet/sing/common/network"
|
||||
|
||||
"github.com/sagernet/sing-box/shater/engine"
|
||||
"github.com/sagernet/sing-box/shater/model"
|
||||
"github.com/sagernet/sing-box/shater/netplane"
|
||||
)
|
||||
|
||||
// l3EgressICMPModel is the smallest l3_tunnel=1 config that carries both
|
||||
// egress kinds: the mandatory tproxy divert plane, a resolver, one interface
|
||||
// egress and one byedpi egress, each with a rule routing into it.
|
||||
//
|
||||
// The interface egress binds to `lo` — deliberately: BindInterface must name a
|
||||
// device that EXISTS on the runner (netplane.IfaceDevice passes an
|
||||
// unresolvable name through unchanged, and every kernel has lo), and this test
|
||||
// never dials, so nothing actually leaves through it. IPv6 is off for the same
|
||||
// reason as the sibling model: a disable_ipv6=1 host must not masquerade as
|
||||
// the regression this test hunts.
|
||||
func l3EgressICMPModel() *model.Model {
|
||||
g := model.DefaultGlobals()
|
||||
g.KillSwitch = "closed"
|
||||
g.ResolverDefault = "cf"
|
||||
g.L3Tunnel = true
|
||||
g.IPv6 = false
|
||||
return &model.Model{
|
||||
Globals: g,
|
||||
Inbounds: []model.Inbound{
|
||||
// 12404: next free port above the package's hand-allocated tproxy
|
||||
// band (12403 belongs to l3_integration_linux_test.go). Start
|
||||
// binds for real, so a clash with a sibling would fail this test
|
||||
// for reasons that have nothing to do with the egresses.
|
||||
{Name: "lan", Enabled: true, Type: "tproxy", TproxyPort: 12404, TCP: true, UDP: true},
|
||||
},
|
||||
Resolvers: []model.Resolver{
|
||||
{Name: "cf", Type: "doh", Address: "https://1.1.1.1/dns-query", Detour: "direct"},
|
||||
},
|
||||
Egresses: []model.Egress{
|
||||
{Name: "lo", Type: "interface", Interface: "lo"},
|
||||
{Name: "bd", Type: "byedpi", Port: 1080},
|
||||
},
|
||||
Rules: []model.Rule{
|
||||
{Name: "ping-lo", Enabled: true, Order: 10, Src: []string{"192.168.88.0/24"}, Target: "egress:lo"},
|
||||
{Name: "desync", Enabled: true, Order: 20, Src: []string{"192.168.89.0/24"}, Target: "egress:bd"},
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
// TestIntegrationL3EgressICMPIsAFlow proves the live engine's verdict on ICMP
|
||||
// through each egress kind, in failure-mode order:
|
||||
//
|
||||
// 1. the interface egress outbound is an adapter.FlowOutbound advertising
|
||||
// icmp — the two static gates route.preMatchFlow checks before it even
|
||||
// asks the outbound;
|
||||
// 2. its PreMatchFlow(icmp) answers Flow — the dynamic gate, true only when
|
||||
// the ping.Port was really constructed despite BindInterface+RoutingMark
|
||||
// on the dialer. THE assertion of this file: its failure mode is a ping
|
||||
// that silently turns into a drop with not one generated byte changed;
|
||||
// 3. the byedpi egress outbound fails at least one of the same static gates,
|
||||
// which is precisely what makes l3-honest-drop DROP a ping routed at it
|
||||
// instead of the TUN stack forging the echo reply.
|
||||
func TestIntegrationL3EgressICMPIsAFlow(t *testing.T) {
|
||||
if os.Geteuid() != 0 {
|
||||
t.Skipf("needs root to open and configure a TUN device (euid=%d) — run as root with CAP_NET_ADMIN and /dev/net/tun, e.g. on the OpenWrt VM or via `docker run --cap-add NET_ADMIN --device /dev/net/tun`", os.Geteuid())
|
||||
}
|
||||
if _, err := os.Stat("/dev/net/tun"); err != nil {
|
||||
t.Skipf("/dev/net/tun is not available (%v) — expose it (modprobe tun; in docker: --device /dev/net/tun --cap-add NET_ADMIN) and run as root", err)
|
||||
}
|
||||
|
||||
opts, warns, err := GenerateWithWarnings(l3EgressICMPModel())
|
||||
if err != nil {
|
||||
t.Fatalf("Generate: unexpected error: %v", err)
|
||||
}
|
||||
if len(warns) != 0 {
|
||||
t.Fatalf("generate degraded the model (warnings: %v) — the egress and L3 skip paths warn instead of failing, so a warning here usually means an egress outbound or the TUN inbound was silently dropped and the engine verdicts below would prove nothing", warns)
|
||||
}
|
||||
|
||||
e := engine.New()
|
||||
// Idempotent; covers every Fatalf below. The wait is no longer needed to
|
||||
// keep the NEXT test in this package from meeting `TUNSETIFF: device or
|
||||
// resource busy` — with two slots (netplane/l3.go) it simply takes the other
|
||||
// one. It stays as a LEAK check: a slot this test does not hand back is a
|
||||
// slot gone for the life of the process, and the collision only comes back
|
||||
// once both are.
|
||||
t.Cleanup(func() {
|
||||
_ = e.Close()
|
||||
l3WaitDevicesGone(t)
|
||||
})
|
||||
if _, err := e.Apply(opts); err != nil {
|
||||
t.Fatalf("engine.Apply (box.New validate + Start) rejected the config: %v\n%s", err, l3StartFailureHint(err))
|
||||
}
|
||||
om := e.Instance().Outbound()
|
||||
if om == nil {
|
||||
t.Fatalf("running box has no OutboundManager — nothing below could prove anything")
|
||||
}
|
||||
|
||||
// 1+2. The interface egress: the engine must consider it ICMP-capable, and
|
||||
// capable FOR REAL (the ping.Port exists), not just by interface shape.
|
||||
loTag := netplane.EgressOutboundTag("lo")
|
||||
loOb, ok := om.Outbound(loTag)
|
||||
if !ok {
|
||||
t.Fatalf("running box has no outbound %q — every rule bound to this egress resolved into nothing, so its traffic is fail-closed blocked and the second-WAN path this feature sells does not exist", loTag)
|
||||
}
|
||||
if !slices.Contains(loOb.Network(), N.NetworkICMP) {
|
||||
t.Fatalf("outbound %q Network() = %v, without %q — route.preMatchFlow refuses the flow at its first static gate, so every ping routed through an interface egress is dropped while TCP/UDP keep flowing", loTag, loOb.Network(), N.NetworkICMP)
|
||||
}
|
||||
flow, isFlow := loOb.(adapter.FlowOutbound)
|
||||
if !isFlow {
|
||||
t.Fatalf("outbound %q (%T) is not an adapter.FlowOutbound — route.preMatchFlow can then never answer Flow for it, so every ping routed through an interface egress is dropped while the config still claims the egress carries L3", loTag, loOb)
|
||||
}
|
||||
if got := flow.PreMatchFlow(N.NetworkICMP, netip.MustParseAddr("203.0.113.9")); got != adapter.PreMatchFlow {
|
||||
t.Fatalf("PreMatchFlow(icmp) = %v, want adapter.PreMatchFlow — the direct outbound started WITHOUT its ping.Port, i.e. dialer.NewWithOptions no longer yields a *dialer.DefaultDialer once BindInterface+RoutingMark are set (protocol/direct only builds icmpPort behind that cast); ping through every interface egress then silently turns into a drop with not one generated byte changed, so only this live check can catch it", got)
|
||||
}
|
||||
|
||||
// 3. The byedpi egress: at least one static gate must refuse it. Both
|
||||
// refusing is today's reality (SOCKS advertises no icmp and is no
|
||||
// FlowOutbound); the regression is BOTH passing, because then
|
||||
// l3-honest-drop stops dropping and the TUN stack answers the echo itself
|
||||
// — a forged reply from a desync hop that never saw the packet.
|
||||
bdTag := netplane.EgressOutboundTag("bd")
|
||||
bdOb, ok := om.Outbound(bdTag)
|
||||
if !ok {
|
||||
t.Fatalf("running box has no outbound %q — every rule bound to this egress resolved into nothing, so its domains lost the desync entirely", bdTag)
|
||||
}
|
||||
if _, isFlow := bdOb.(adapter.FlowOutbound); isFlow && slices.Contains(bdOb.Network(), N.NetworkICMP) {
|
||||
t.Fatalf("outbound %q (%T, networks %v) passes both of route.preMatchFlow's static gates for ICMP — l3-honest-drop then no longer drops a ping routed at the byedpi egress, and the TUN stack forges the echo reply locally: the operator reads a working ping off a SOCKS hop that cannot carry the packet", bdTag, bdOb, bdOb.Network())
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,123 @@
|
||||
// The egress half of the L3 ICMP story, portable codegen part: WHAT the
|
||||
// generator must emit so a ping routed to `egress:<name>` behaves honestly.
|
||||
//
|
||||
// The chain these tests pin (all upstream sing-box internals, none of them
|
||||
// visible in the generated JSON): protocol/direct is the ONLY proxy outbound
|
||||
// in the shipped registry that implements adapter.FlowOutbound — at
|
||||
// construction it wraps a ping.Port around the dialer's Control chain, and
|
||||
// common/dialer/default.go appends BindInterface + RoutingMark to exactly
|
||||
// that chain (dialer.Control -> DefaultDialer.dialer4 ->
|
||||
// DialerForICMPDestination -> the raw ICMP socket). So the SHAPE asserted
|
||||
// here — a direct outbound carrying the egress device and the deterministic
|
||||
// egress mark — is precisely what makes a tunnelled ping leave through the
|
||||
// right interface under the right policy table. A SOCKS outbound (byedpi)
|
||||
// sits on the other side of the same line: it cannot implement tun.Port, so
|
||||
// route.preMatchFlow's l3-honest-drop block DROPS ICMP routed at it instead
|
||||
// of letting the TUN stack forge an echo reply locally.
|
||||
//
|
||||
// Portable (no box.New): these assert on the generated option.Options only.
|
||||
// The live-engine proof that the direct outbound REALLY constructs its
|
||||
// icmpPort despite bind+mark lives in l3_egress_linux_test.go.
|
||||
package generate
|
||||
|
||||
import (
|
||||
"testing"
|
||||
|
||||
C "github.com/sagernet/sing-box/constant"
|
||||
"github.com/sagernet/sing-box/option"
|
||||
|
||||
"github.com/sagernet/sing-box/shater/model"
|
||||
"github.com/sagernet/sing-box/shater/netplane"
|
||||
)
|
||||
|
||||
// l3EgressOutbound finds the outbound emitted under tag. A local twin of
|
||||
// generate_test.go's findOutbound, which lives in the linux-gated engine
|
||||
// suite and does not exist on other platforms.
|
||||
func l3EgressOutbound(opts option.Options, tag string) *option.Outbound {
|
||||
for i := range opts.Outbounds {
|
||||
if opts.Outbounds[i].Tag == tag {
|
||||
return &opts.Outbounds[i]
|
||||
}
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// TestEgressInterfaceOutboundCarriesICMP pins the three fields that decide
|
||||
// whether a ping routed to an interface egress ACTUALLY leaves through that
|
||||
// interface:
|
||||
//
|
||||
// - Type direct — the one registered outbound type whose constructor builds
|
||||
// a ping.Port (adapter.FlowOutbound); any other type demotes ICMP through
|
||||
// this egress to the honest drop.
|
||||
// - BindInterface = the egress device — appended to dialer.Control, which
|
||||
// DefaultDialer.dialer4 carries and DialerForICMPDestination hands to the
|
||||
// raw ICMP socket.
|
||||
// - RoutingMark = the deterministic egress mark — same Control chain; it is
|
||||
// what the netplane's policy rule matches to steer the packet into the
|
||||
// egress table and past the tproxy divert.
|
||||
func TestEgressInterfaceOutboundCarriesICMP(t *testing.T) {
|
||||
m := &model.Model{
|
||||
Globals: model.DefaultGlobals(),
|
||||
Inbounds: []model.Inbound{
|
||||
{Name: "lan", Enabled: true, Type: "tproxy", TproxyPort: 12345, TCP: true, UDP: true},
|
||||
},
|
||||
Egresses: []model.Egress{{Name: "wan2", Type: "interface", Interface: "wan2"}},
|
||||
Rules: []model.Rule{
|
||||
{Name: "via-wan2", Enabled: true, Order: 10, Src: []string{"192.168.2.0/24"}, Target: "egress:wan2"},
|
||||
},
|
||||
}
|
||||
opts, warns, err := GenerateWithWarnings(m)
|
||||
if err != nil {
|
||||
t.Fatalf("Generate: unexpected error: %v", err)
|
||||
}
|
||||
tag := netplane.EgressOutboundTag("wan2")
|
||||
ob := l3EgressOutbound(opts, tag)
|
||||
if ob == nil {
|
||||
t.Fatalf("no outbound %q was emitted (warnings: %v) — every rule bound to this egress then resolves through the fail-closed path (egressDetourOrBlock) and the operator's second WAN silently carries nothing", tag, warns)
|
||||
}
|
||||
if ob.Type != C.TypeDirect {
|
||||
t.Fatalf("egress outbound type = %q, want %q — direct is the only proxy outbound in the shipped registry that implements adapter.FlowOutbound (its constructor builds the ping.Port), so any other type turns every ping routed through this egress into a drop while TCP/UDP keep flowing, and nobody can tell the second WAN's L3 path is dead", ob.Type, C.TypeDirect)
|
||||
}
|
||||
do, ok := ob.Options.(*option.DirectOutboundOptions)
|
||||
if !ok {
|
||||
t.Fatalf("egress outbound options are %T, want *option.DirectOutboundOptions — without the typed dialer options there is no BindInterface/RoutingMark to reach the ICMP socket's Control chain at all", ob.Options)
|
||||
}
|
||||
if want := netplane.IfaceDevice("wan2"); do.BindInterface != want {
|
||||
t.Fatalf("BindInterface = %q, want %q — this field is appended to dialer.Control (common/dialer/default.go), which lands in DefaultDialer.dialer4, whose Control DialerForICMPDestination hands to the ICMP socket; losing it sends the echo over the MAIN routing table, i.e. out the plain default WAN instead of this egress — a silent leak, not a visible failure", do.BindInterface, want)
|
||||
}
|
||||
if want := option.FwMark(netplane.EgressMark(m.Globals, 0)); do.RoutingMark != want {
|
||||
t.Fatalf("RoutingMark = %#x, want %#x — the mark rides the same Control chain into the ICMP socket, and it is what the netplane's per-egress policy rule matches; without it the egress's own packets are routed by the main table (leaking past the egress) or re-caught by the tproxy divert (a routing loop)", uint32(do.RoutingMark), uint32(want))
|
||||
}
|
||||
}
|
||||
|
||||
// TestEgressByeDPIOutboundCannotCarryICMP pins the OTHER side of the line: a
|
||||
// byedpi egress is a SOCKS5 hop into the local ciadpi desync proxy, and SOCKS
|
||||
// does not (and cannot) implement tun.Port, so route.preMatchFlow's
|
||||
// l3-honest-drop block DROPS ICMP routed at it. That drop is the feature: the
|
||||
// only alternative the TUN stack offers is answering the echo ITSELF
|
||||
// (stack_gvisor_icmp.go), i.e. a forged reply from a path that never saw the
|
||||
// packet. Pinning the SOCKS type here pins the reason the drop happens.
|
||||
func TestEgressByeDPIOutboundCannotCarryICMP(t *testing.T) {
|
||||
m := &model.Model{
|
||||
Globals: model.DefaultGlobals(),
|
||||
Inbounds: []model.Inbound{
|
||||
{Name: "lan", Enabled: true, Type: "tproxy", TproxyPort: 12345, TCP: true, UDP: true},
|
||||
},
|
||||
Egresses: []model.Egress{{Name: "bd", Type: "byedpi", Port: 1080}},
|
||||
Rules: []model.Rule{
|
||||
{Name: "desync", Enabled: true, Order: 10, Src: []string{"192.168.3.0/24"}, Target: "egress:bd"},
|
||||
},
|
||||
}
|
||||
opts, warns, err := GenerateWithWarnings(m)
|
||||
if err != nil {
|
||||
t.Fatalf("Generate: unexpected error: %v", err)
|
||||
}
|
||||
tag := netplane.EgressOutboundTag("bd")
|
||||
ob := l3EgressOutbound(opts, tag)
|
||||
if ob == nil {
|
||||
t.Fatalf("no outbound %q was emitted (warnings: %v) — every rule bound to this egress then resolves through the fail-closed path and the desync stops covering its domains", tag, warns)
|
||||
}
|
||||
if ob.Type != C.TypeSOCKS {
|
||||
t.Fatalf("byedpi egress outbound type = %q, want %q — the type is load-bearing twice over: only a SOCKS hop actually reaches the local ciadpi process (anything else skips the desync entirely), and its inability to implement tun.Port is exactly what makes the l3-honest-drop block DROP a ping routed here instead of the TUN stack forging an echo reply from a path that never carried the packet", ob.Type, C.TypeSOCKS)
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,467 @@
|
||||
//go:build linux
|
||||
|
||||
// The engine half of the L3-ingress proof — the one that needs a real kernel.
|
||||
//
|
||||
// Everything else about l3_tunnel is already pinned by unprivileged tests: the
|
||||
// portable codegen suite (inbound_test.go TestL3Tunnel*) fixes the SHAPE of the
|
||||
// synthetic "l3-in" TUN inbound (tag, netplane.L3Device name, auto_route off,
|
||||
// gvisor stack, MTU, addresses), and the netplane tests fix the nft marks and
|
||||
// the ip rule/route plan around it. What NONE of them prove is that the engine
|
||||
// actually accepts that config on the router build: shaterd swaps upstream's
|
||||
// include.Context for the slim shater/registry (engine.New wires
|
||||
// registry.Context), where tun.RegisterInbound is a deliberate hand-kept
|
||||
// entry, and the gVisor netstack the inbound demands (stack: gvisor) is only
|
||||
// compiled in because with_wireguard drags with_gvisor along
|
||||
// (scripts/router-tags.sh). Losing either — the registry entry deleted as
|
||||
// "unused", the tag trimmed for size — changes no generated byte, so every
|
||||
// codegen assertion stays green, and the failure surfaces as a dead engine on
|
||||
// the operator's router at the first l3_tunnel=1 apply (the 2026-07-25
|
||||
// WireGuard outage was exactly this "built with X, verified with Y" class).
|
||||
// This file closes that gap, in two layers rather than one:
|
||||
//
|
||||
// TestL3TunInboundIsAcceptedByBoxNew ORDINARY. box.New alone — no root, no
|
||||
// /dev/net/tun — so the registry half runs
|
||||
// on EVERY gate, act_runner included.
|
||||
// TestIntegrationL3TunInboundStarts PRIVILEGED. Start against a real kernel:
|
||||
// the device exists, carries the contract
|
||||
// MTU, and is returned on Close.
|
||||
// TestIntegrationL3StaleSlotIsReclaimed
|
||||
// PRIVILEGED. What a RESTART finds: an
|
||||
// engine with no memory of slots, next to
|
||||
// a device it did not open.
|
||||
//
|
||||
// The split is the point. The registry regression is the one that costs a LAN
|
||||
// outage and it is now caught by a test CI can actually run; only the parts that
|
||||
// genuinely need a kernel are left to the privileged one.
|
||||
//
|
||||
// Separate file, TestIntegration name, honest skips: opening /dev/net/tun and
|
||||
// configuring the device needs root + CAP_NET_ADMIN, which the ordinary gate
|
||||
// does not have — scripts/run-tests.sh reaches linux through containers (the
|
||||
// act_runner job container, or the docker re-exec from a dev host) that expose
|
||||
// no /dev/net/tun, and its SKIP_COMMON already excludes common/tlsspoof's
|
||||
// TestIntegration* for the same capability reason; the TestIntegration prefix
|
||||
// keeps this test inside that naming convention. The shater/... roots are not
|
||||
// name-filtered, so what keeps the ordinary gate green here are the guards
|
||||
// below: no root or no /dev/net/tun means a loud skip that says how to run it
|
||||
// for real (the OpenWrt VM, or
|
||||
// `docker run --cap-add NET_ADMIN --device /dev/net/tun`).
|
||||
package generate
|
||||
|
||||
import (
|
||||
"context"
|
||||
"fmt"
|
||||
"net"
|
||||
"os"
|
||||
"os/exec"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
box "github.com/sagernet/sing-box"
|
||||
C "github.com/sagernet/sing-box/constant"
|
||||
"github.com/sagernet/sing-box/option"
|
||||
|
||||
"github.com/sagernet/sing-box/shater/engine"
|
||||
"github.com/sagernet/sing-box/shater/model"
|
||||
"github.com/sagernet/sing-box/shater/netplane"
|
||||
"github.com/sagernet/sing-box/shater/registry"
|
||||
)
|
||||
|
||||
// l3GoneTimeout bounds how long assertion 4 waits for the kernel to drop the
|
||||
// device after Close.
|
||||
//
|
||||
// It is 20s, not the 5s it used to be, and the number comes from a measurement
|
||||
// rather than from taste. Instrumented run in a CONTAINER (`docker run
|
||||
// --cap-add NET_ADMIN --device /dev/net/tun`, golang:1.26, kernel 6.x), three
|
||||
// repeats:
|
||||
//
|
||||
// Apply 15-36 ms
|
||||
// engine.Close returned 288-372 µs
|
||||
// /dev/net/tun fds open 0, already at the instant Close returned
|
||||
// (control: 1 immediately before Close, so the
|
||||
// instrument can see an open fd when there is one)
|
||||
// device actually gone 3.82 s / 4.29 s / 4.59 s LATER
|
||||
//
|
||||
// The engine therefore hands the fd back at once; the wait is the KERNEL's
|
||||
// deferred unregister_netdevice, and nothing in this process can shorten it.
|
||||
//
|
||||
// Where the delay comes from matters, so it was measured on the other side too:
|
||||
// on the stand (local_openwrt, ImmortalWrt 25.12.1 r37978, kernel 6.12.94 — the
|
||||
// production router's revision) this whole test takes 0.10 s, i.e. the device is
|
||||
// gone essentially immediately. The multi-second lag is a nested-netns container
|
||||
// artefact, not what a router does.
|
||||
//
|
||||
// So 5s was a flake generator on exactly the environment the gate runs in, for
|
||||
// nothing: a LEAK is unbounded, so a longer budget costs one slow failure and
|
||||
// gives up no sensitivity whatsoever.
|
||||
//
|
||||
// This is also why netplane.L3SlotFor DELETES a lingering slot rather than
|
||||
// waiting for it, and why there are two slots: a second apply inside that window
|
||||
// is normal, not exceptional.
|
||||
const l3GoneTimeout = 20 * time.Second
|
||||
|
||||
// l3IntegrationModel is the smallest realistic l3_tunnel=1 config: one enabled
|
||||
// tproxy inbound (the L3 ingress refuses to exist without the divert plane it
|
||||
// rides), one resolver, one shadowsocks node on TEST-NET and a rule into it —
|
||||
// the same skeleton as the other *_linux_test.go models, plus the L3 opt-in.
|
||||
//
|
||||
// IPv6 is off deliberately: the v6 address half of the TUN contract is pinned
|
||||
// by the portable codegen tests, and carrying it here would couple THIS proof
|
||||
// (the tun inbound starts under the slim registry) to the runner kernel's
|
||||
// ipv6 sysctls — a disable_ipv6=1 host would fail address configuration and
|
||||
// masquerade as the registry/tag regression this test hunts.
|
||||
func l3IntegrationModel() *model.Model {
|
||||
g := model.DefaultGlobals()
|
||||
g.KillSwitch = "closed"
|
||||
g.ResolverDefault = "cf"
|
||||
g.L3Tunnel = true
|
||||
g.IPv6 = false
|
||||
return &model.Model{
|
||||
Globals: g,
|
||||
Inbounds: []model.Inbound{
|
||||
// 12403: first port above the package's hand-allocated tproxy band
|
||||
// (currently topping out at 12402). Start binds for real, so a
|
||||
// clash with a sibling's port would fail this test for reasons
|
||||
// that have nothing to do with the TUN.
|
||||
{Name: "lan", Enabled: true, Type: "tproxy", TproxyPort: 12403, TCP: true, UDP: true},
|
||||
},
|
||||
Nodes: []model.Node{
|
||||
{Name: "exit", Enabled: true, URI: "ss://aes-256-gcm:secret@203.0.113.1:8388#exit"},
|
||||
},
|
||||
Resolvers: []model.Resolver{
|
||||
{Name: "cf", Type: "doh", Address: "https://1.1.1.1/dns-query", Detour: "direct"},
|
||||
},
|
||||
Rules: []model.Rule{
|
||||
{Name: "via-exit", Enabled: true, Order: 10, DstPort: "443", Target: "node:exit"},
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
// TestIntegrationL3TunInboundStarts proves the generated l3_tunnel config is
|
||||
// not merely well-formed but ALIVE: the engine (slim registry, compiled tag
|
||||
// set) accepts it, the kernel ends up with the device the netplane routes
|
||||
// into, and Close returns the name. Four assertions, in failure-mode order:
|
||||
//
|
||||
// 1. engine.Apply (box.New + Start) succeeds — a rejection here is the slim
|
||||
// registry losing tun.RegisterInbound or the build losing the gVisor
|
||||
// stack, see l3StartFailureHint;
|
||||
// 2. net.InterfaceByName(netplane.L3Device) finds the device — the L3 table's
|
||||
// default route points at exactly that name, so no device means marked
|
||||
// LAN ICMP blackholes while the config claims it is tunnelled;
|
||||
// 3. the kernel ACCEPTS the contract MTU 65535 on a TUN — the whole point of
|
||||
// the value is that no IP datagram can exceed it, so the kernel can never
|
||||
// fragment on the way in (D25); a kernel that clamped it would silently
|
||||
// restore the forged-reply band;
|
||||
// 4. after Close the device is GONE — a leak would jam every later apply
|
||||
// (each swap reopens the same name) until shaterd itself is restarted.
|
||||
func TestIntegrationL3TunInboundStarts(t *testing.T) {
|
||||
if os.Geteuid() != 0 {
|
||||
t.Skipf("needs root to open and configure a TUN device (euid=%d) — run as root with CAP_NET_ADMIN and /dev/net/tun, e.g. on the OpenWrt VM or via `docker run --cap-add NET_ADMIN --device /dev/net/tun`", os.Geteuid())
|
||||
}
|
||||
if _, err := os.Stat("/dev/net/tun"); err != nil {
|
||||
t.Skipf("/dev/net/tun is not available (%v) — expose it (modprobe tun; in docker: --device /dev/net/tun --cap-add NET_ADMIN) and run as root", err)
|
||||
}
|
||||
|
||||
opts, warns, err := GenerateWithWarnings(l3IntegrationModel())
|
||||
if err != nil {
|
||||
t.Fatalf("Generate: unexpected error: %v", err)
|
||||
}
|
||||
if len(warns) != 0 {
|
||||
t.Fatalf("generate degraded the model (warnings: %v) — the L3 skip paths warn instead of failing, so a warning here usually means the TUN inbound was silently dropped and the engine run below would prove nothing", warns)
|
||||
}
|
||||
// Precondition, not the point: the SHAPE of the tun inbound is
|
||||
// inbound_test.go's job. If it is missing here, fail with the right
|
||||
// address instead of a misleading "no such interface" three steps later.
|
||||
hasTun := false
|
||||
for _, in := range opts.Inbounds {
|
||||
if in.Type == C.TypeTun {
|
||||
hasTun = true
|
||||
break
|
||||
}
|
||||
}
|
||||
if !hasTun {
|
||||
t.Fatalf("generated options carry no tun inbound — a codegen regression (inbound_test.go TestL3TunnelEmitsTunInbound should be red too), not an engine one; nothing below could prove anything")
|
||||
}
|
||||
|
||||
// 1. The engine — slim registry, the tag set this binary was built with —
|
||||
// takes the config and starts it.
|
||||
e := engine.New()
|
||||
t.Cleanup(func() { _ = e.Close() }) // idempotent; covers every Fatalf below
|
||||
changed, err := e.Apply(opts)
|
||||
if err != nil {
|
||||
t.Fatalf("engine.Apply (box.New validate + Start) rejected the l3_tunnel config: %v\n%s", err, l3StartFailureHint(err))
|
||||
}
|
||||
if !changed {
|
||||
t.Fatalf("expected Apply changed==true on a fresh engine")
|
||||
}
|
||||
|
||||
// 2. The device is REAL, and it is one of the SLOTS — not the placeholder
|
||||
// name generate emitted. Start constructs the TUN synchronously (protocol/tun
|
||||
// StartStateStart -> tun.New), so no settling loop is needed on this side.
|
||||
//
|
||||
// Asserting the slot rather than a fixed name is the point of the check now:
|
||||
// if the engine ever stopped substituting (engine/l3slot.go), the device
|
||||
// would come up as the placeholder `shater-l3`, every generation would want
|
||||
// that one name again, and the TUNSETIFF EBUSY that took the LAN down on the
|
||||
// production router would be back — while every codegen test stayed green,
|
||||
// because generate's output would not have changed by one byte.
|
||||
dev := l3LiveSlot(t)
|
||||
iface, err := net.InterfaceByName(dev)
|
||||
if err != nil {
|
||||
t.Fatalf("Start reported success but %q does not exist (%v) — the netplane's L3 table points its default route at the live slot, so no device means marked LAN ICMP blackholes while the config claims it is tunnelled", dev, err)
|
||||
}
|
||||
|
||||
// 3. The kernel device carries the contract MTU. This is the assertion the
|
||||
// value was chosen for: at 65535 no IP datagram can exceed the device MTU,
|
||||
// so the kernel cannot fragment on the way in and sing-tun'''s dispatcher
|
||||
// always gets a whole packet to judge. A kernel that silently clamped this
|
||||
// (or a driver with a lower max_mtu) would put the forged-reply band back
|
||||
// without changing a single generated byte.
|
||||
if iface.MTU != 65535 {
|
||||
t.Fatalf("%s MTU = %d, want 65535 — the kernel did not take the contract MTU; anything smaller means the kernel fragments packets above it INTO this device, sing-tun'''s ForwardDispatcher declines fragments before asking for a verdict, and the gVisor ICMP forwarder forges the echo reply for WireGuard/AWG (D25)", dev, iface.MTU)
|
||||
}
|
||||
|
||||
// 4. Close returns the device name to the kernel.
|
||||
if err := e.Close(); err != nil {
|
||||
t.Fatalf("engine Close failed: %v — a box that does not stop keeps %s open", err, dev)
|
||||
}
|
||||
l3WaitDevicesGone(t)
|
||||
}
|
||||
|
||||
// l3LiveSlot returns the ONE netplane.L3Slots device the kernel currently has,
|
||||
// failing if there is not exactly one.
|
||||
//
|
||||
// "Exactly one" is an assertion, not a convenience. Zero means the engine did
|
||||
// not create the device at all (or created it under some third name), and two
|
||||
// means a generation leaked one — which is the state that eventually exhausts
|
||||
// the slots and brings TUNSETIFF EBUSY back. Neither can be allowed to pass as
|
||||
// "found a device, carry on".
|
||||
func l3LiveSlot(t *testing.T) string {
|
||||
t.Helper()
|
||||
var live []string
|
||||
for _, s := range netplane.L3Slots {
|
||||
if _, err := net.InterfaceByName(s); err == nil {
|
||||
live = append(live, s)
|
||||
}
|
||||
}
|
||||
// The legacy single name. It is no longer reachable from a current build —
|
||||
// generate's placeholder is now too long for the kernel to accept
|
||||
// (l3PlaceholderDevice), so a config that skipped the substitution fails at
|
||||
// Start instead of creating this device — but an installation upgraded from
|
||||
// the single-name build can still have one lying around, and it would be
|
||||
// indistinguishable from a live slot to everything that matches by prefix.
|
||||
if _, err := net.InterfaceByName(netplane.L3DeviceBase); err == nil {
|
||||
t.Fatalf("the kernel has a device named %q — the legacy single name. A current build cannot create it (the placeholder generate emits is longer than IFNAMSIZ allows), so this is either a leftover from an upgraded install that teardown did not sweep, or somebody made the placeholder short again. Every generation would want this one name, which is exactly the TUNSETIFF EBUSY that takes the LAN down on a live router", netplane.L3DeviceBase)
|
||||
}
|
||||
switch len(live) {
|
||||
case 1:
|
||||
return live[0]
|
||||
case 0:
|
||||
t.Fatalf("no %v device exists after a successful Start — the netplane's L3 table has nothing to route marked LAN ICMP into, so it blackholes while the config claims it is tunnelled", netplane.L3Slots)
|
||||
default:
|
||||
t.Fatalf("both slots exist at once (%v) — a generation leaked its TUN; two slots tolerate one late unregister, not an accumulating leak, so this is how the EBUSY outage comes back", live)
|
||||
}
|
||||
return ""
|
||||
}
|
||||
|
||||
// l3WaitDevicesGone blocks until NO L3 slot device is left in the kernel.
|
||||
//
|
||||
// The two-slot design (netplane/l3.go) means this is no longer needed to keep
|
||||
// one test from colliding with the next — a second engine simply takes the
|
||||
// other slot. It is kept because it is still the only place that would catch a
|
||||
// LEAK: a box whose Close does not hand the device back consumes a slot
|
||||
// permanently, and once both are consumed the collision the slots exist to
|
||||
// prevent is back, on a live router, with the fail-closed plane holding the LAN
|
||||
// down. Removal is normally immediate with the last fd but unregister_netdevice
|
||||
// may defer briefly under load, so this polls rather than sleeping a fixed
|
||||
// amount.
|
||||
func l3WaitDevicesGone(t *testing.T) {
|
||||
t.Helper()
|
||||
deadline := time.Now().Add(l3GoneTimeout)
|
||||
names := append([]string{netplane.L3DeviceBase}, netplane.L3Slots[:]...)
|
||||
for {
|
||||
var left []string
|
||||
for _, n := range names {
|
||||
if _, err := net.InterfaceByName(n); err == nil {
|
||||
left = append(left, n)
|
||||
}
|
||||
}
|
||||
if len(left) == 0 {
|
||||
return // gone: the kernel dropped the devices with the engine's fds
|
||||
}
|
||||
if time.Now().After(deadline) {
|
||||
t.Fatalf("%v still exist %v after Close — the engine leaked the TUN. There are only two slots; a leak consumes one for the life of the process, and when both are gone every apply is back to `TUNSETIFF: device or resource busy` with the LAN held down by the fail-closed plane", left, l3GoneTimeout)
|
||||
}
|
||||
time.Sleep(50 * time.Millisecond)
|
||||
}
|
||||
}
|
||||
|
||||
// l3StartFailureHint names the specific regression (or environment defect) an
|
||||
// Apply failure most likely is, so the gate output points at the fix instead
|
||||
// of a bare engine error. Substring matching is the same pragmatism engine.go
|
||||
// itself applies to swap conflicts (isAddrInUse/isCacheLockConflict): the
|
||||
// upstream errors carry no exported sentinels.
|
||||
func l3StartFailureHint(err error) string {
|
||||
msg := err.Error()
|
||||
switch {
|
||||
case strings.Contains(msg, "type not found: tun"):
|
||||
return "hint: the slim registry no longer registers the tun inbound (shater/registry InboundRegistry must call tun.RegisterInbound) — on the router, l3_tunnel=1 then leaves the engine DOWN, and with the fail-closed nft plane that blackholes the whole LAN, not just ICMP"
|
||||
case strings.Contains(msg, "not included in this build"):
|
||||
return "hint: the gVisor netstack is compiled out — the tag set lost with_gvisor (scripts/router-tags.sh keeps it via with_wireguard); this is the 2026-07-25 outage class: green codegen, dead engine at the first Start on the operator's router"
|
||||
case strings.Contains(msg, "operation not permitted"):
|
||||
return "hint: environment, not code — this runner has root and /dev/net/tun but the kernel refused the device (missing CAP_NET_ADMIN? seccomp?); rerun with --cap-add NET_ADMIN or on the OpenWrt VM"
|
||||
default:
|
||||
return "hint: whatever the cause, on the router this means applying l3_tunnel=1 leaves the engine down, and with the fail-closed nft plane that is a LAN-wide outage"
|
||||
}
|
||||
}
|
||||
|
||||
// TestL3TunInboundIsAcceptedByBoxNew is the ORDINARY (unprivileged) half of the
|
||||
// proof above, and it is deliberately not named TestIntegration*: it needs no
|
||||
// root, no CAP_NET_ADMIN and no /dev/net/tun, so it runs on every gate — the
|
||||
// act_runner LXC guest included — where the privileged test can only report
|
||||
// "did not run here".
|
||||
//
|
||||
// What it covers is the failure that costs the most and needs the least: box.New
|
||||
// constructs every adapter from the registry, so a slim registry that has lost
|
||||
// tun.RegisterInbound (shater/registry) fails HERE, with `type not found: tun`,
|
||||
// on every CI run. That regression changes no generated byte — every codegen
|
||||
// assertion stays green — and on the router it surfaces as l3_tunnel=1 leaving
|
||||
// the engine DOWN, which under the fail-closed nft plane is a LAN-wide outage.
|
||||
// Until now the only thing that could see it was a test the gate cannot run.
|
||||
//
|
||||
// What it does NOT cover, said plainly so nobody reads more into a green run:
|
||||
// box.New builds no device and touches no kernel, so the gVisor stack, the
|
||||
// contract MTU and Close returning the device are all beyond it. Those are
|
||||
// TestIntegrationL3TunInboundStarts' job, and scripts/run-tests.sh [5/7] reports
|
||||
// by name whether that one really ran.
|
||||
func TestL3TunInboundIsAcceptedByBoxNew(t *testing.T) {
|
||||
opts, warns, err := GenerateWithWarnings(l3IntegrationModel())
|
||||
if err != nil {
|
||||
t.Fatalf("Generate: unexpected error: %v", err)
|
||||
}
|
||||
if len(warns) != 0 {
|
||||
t.Fatalf("generate degraded the model (warnings: %v) — the L3 skip paths warn instead of failing, so a warning here usually means the TUN inbound was never emitted and box.New below would prove nothing", warns)
|
||||
}
|
||||
// Precondition, not the point: without the inbound this test is vacuous.
|
||||
if len(tunInbounds(opts.Inbounds)) != 1 {
|
||||
t.Fatalf("generated options carry %d TUN inbounds, want exactly 1 — a codegen regression (inbound_test.go should be red too), not a registry one", len(tunInbounds(opts.Inbounds)))
|
||||
}
|
||||
|
||||
b, err := box.New(box.Options{Context: registry.Context(context.Background()), Options: opts})
|
||||
if err != nil {
|
||||
t.Fatalf("box.New rejected the default l3_tunnel config: %v\n%s", err, l3StartFailureHint(err))
|
||||
}
|
||||
// Never Started, so there is nothing to shut down; Close anyway, because a
|
||||
// constructed-but-unstarted box still owns whatever its constructors opened.
|
||||
_ = b.Close()
|
||||
}
|
||||
|
||||
// TestIntegrationL3StaleSlotIsReclaimed answers the two questions a RESTART
|
||||
// raises, neither of which any test asked before: a fresh shaterd starts with
|
||||
// engine.l3Device == "" — no memory of which slot the process before it held —
|
||||
// and a device from an earlier generation can still be in the kernel, because
|
||||
// (measured, see l3GoneTimeout) unregister_netdevice lags the fd close by
|
||||
// several seconds.
|
||||
//
|
||||
// The occupied slot here is held by a SECOND LIVE ENGINE in this process, not
|
||||
// planted with `ip tuntap add`. That is not fussiness, it is the difference
|
||||
// between testing something and testing nothing: a device created by
|
||||
// `ip tuntap add` is PERSISTENT and therefore ATTACHABLE — TUNSETIFF on it
|
||||
// simply succeeds — so a planted device is reused whether or not anything
|
||||
// reclaims it. Measured: with the reclaim loop deleted from
|
||||
// netplane.L3SlotFor, the planted version of this test still passed. A slot held
|
||||
// by a live box is what actually answers EBUSY, and it is also what a router
|
||||
// really has (a generation whose Close has not finished).
|
||||
//
|
||||
// Two steps, in the order they happen on a router:
|
||||
//
|
||||
// 1. a slot is occupied and NOTHING in the new engine knows why. It must come
|
||||
// up on the OTHER slot and leave the occupied one alone. This is the
|
||||
// restart-inside-the-unregister-window case, and getting it wrong is
|
||||
// `TUNSETIFF: device or resource busy` at the first apply after a restart;
|
||||
// 2. the next apply of that engine has only the occupied slot left — its own
|
||||
// live one is excluded. It must RECLAIM it (L3SlotFor deletes rather than
|
||||
// waits) instead of failing. A wait would be a race by construction: the
|
||||
// kernel's unregister is asynchronous, so "it will be free in a moment" is
|
||||
// exactly the intermittent failure the two-slot design refuses to trade for.
|
||||
func TestIntegrationL3StaleSlotIsReclaimed(t *testing.T) {
|
||||
if os.Geteuid() != 0 {
|
||||
t.Skipf("needs root to open and configure TUN devices (euid=%d) — run as root with CAP_NET_ADMIN and /dev/net/tun, e.g. on the OpenWrt VM or via `docker run --cap-add NET_ADMIN --device /dev/net/tun`", os.Geteuid())
|
||||
}
|
||||
if _, err := os.Stat("/dev/net/tun"); err != nil {
|
||||
t.Skipf("/dev/net/tun is not available (%v) — expose it (modprobe tun; in docker: --device /dev/net/tun --cap-add NET_ADMIN) and run as root", err)
|
||||
}
|
||||
if _, err := exec.LookPath("ip"); err != nil {
|
||||
t.Skipf("iproute2 is not installed (%v) — netplane.L3SlotFor asks the kernel through `ip link show` and reclaims through `ip link del`; without the binary every slot looks free and this test would prove the opposite of what it says", err)
|
||||
}
|
||||
|
||||
// staleModel/freshModel differ only in the tproxy port: both listeners bind
|
||||
// for real, so two engines in one process need two ports. 12405-12407 are the
|
||||
// next free ones above the band this package hand-allocates (12403, 12404).
|
||||
cacheDir := t.TempDir()
|
||||
modelOnPort := func(port int) *model.Model {
|
||||
m := l3IntegrationModel()
|
||||
m.Inbounds[0].TproxyPort = port
|
||||
return m
|
||||
}
|
||||
//
|
||||
// The cache_file is repointed per engine for the same reason: bbolt holds it
|
||||
// under an EXCLUSIVE lock, so two live engines in one process would fail on
|
||||
// `initialize cache-file: timeout` long before reaching the slot question.
|
||||
// One process running two engines is an artefact of this test, not of the
|
||||
// router, so the contention it creates is removed rather than tested around.
|
||||
generate := func(port int) option.Options {
|
||||
t.Helper()
|
||||
opts, warns, err := GenerateWithWarnings(modelOnPort(port))
|
||||
if err != nil {
|
||||
t.Fatalf("Generate(port %d): unexpected error: %v", port, err)
|
||||
}
|
||||
if len(warns) != 0 {
|
||||
t.Fatalf("Generate(port %d) degraded the model (warnings: %v)", port, warns)
|
||||
}
|
||||
if opts.Experimental == nil || opts.Experimental.CacheFile == nil {
|
||||
t.Fatalf("Generate(port %d) emitted no cache_file — this helper repoints it, and a silent no-op here brings back the bbolt lock contention it exists to remove", port)
|
||||
}
|
||||
opts.Experimental.CacheFile.Path = filepath.Join(cacheDir, fmt.Sprintf("cache-%d.db", port))
|
||||
return opts
|
||||
}
|
||||
|
||||
// The generation that is "still around": a live box holding one slot.
|
||||
occupier := engine.New()
|
||||
t.Cleanup(func() { _ = occupier.Close() })
|
||||
if _, err := occupier.Apply(generate(12405)); err != nil {
|
||||
t.Fatalf("could not start the occupying engine: %v\n%s", err, l3StartFailureHint(err))
|
||||
}
|
||||
occupied := l3LiveSlot(t)
|
||||
|
||||
// 1. A brand-new Engine: l3Device == "", exactly like a restarted daemon.
|
||||
fresh := engine.New()
|
||||
t.Cleanup(func() { _ = fresh.Close() })
|
||||
if _, err := fresh.Apply(generate(12406)); err != nil {
|
||||
t.Fatalf("a fresh engine could not start next to the occupied slot %s: %v\n%s\nThis is the first apply after a restart that lands inside the kernel's unregister window; failing it leaves the LAN down behind the fail-closed plane until somebody intervenes", occupied, err, l3StartFailureHint(err))
|
||||
}
|
||||
if _, err := net.InterfaceByName(occupied); err != nil {
|
||||
t.Fatalf("%s is gone (%v) — the fresh engine took the slot it must never touch. On a router that device can belong to a generation that is still carrying connections", occupied, err)
|
||||
}
|
||||
free := netplane.L3Slots[0]
|
||||
if free == occupied {
|
||||
free = netplane.L3Slots[1]
|
||||
}
|
||||
if _, err := net.InterfaceByName(free); err != nil {
|
||||
t.Fatalf("the fresh engine did not come up on %s (%v) — with %s already in the kernel there was exactly one slot left to take", free, err, occupied)
|
||||
}
|
||||
|
||||
// 2. The next apply has only the OCCUPIED slot to choose from, and it is held
|
||||
// by a live box that will answer EBUSY. Reclaim, not wait.
|
||||
changed, err := fresh.Apply(generate(12407))
|
||||
if err != nil {
|
||||
t.Fatalf("the second apply could not reclaim %s: %v\n%s\nBoth other candidates are unavailable — %s is carrying this engine's own traffic — so a slot held by another generation MUST be deleted rather than waited for. This is the three-in-a-row TUNSETIFF EBUSY that took the production LAN down (netplane/l3.go)", occupied, err, l3StartFailureHint(err), free)
|
||||
}
|
||||
if !changed {
|
||||
t.Fatalf("the second config hashed equal to the first — it was meant to differ (different tproxy port), so the swap this step is about never happened")
|
||||
}
|
||||
if _, err := net.InterfaceByName(occupied); err != nil {
|
||||
t.Fatalf("after the swap %s does not exist (%v) — it was the only slot the new generation could take, so it should have been reclaimed and reopened", occupied, err)
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,88 @@
|
||||
package generate
|
||||
|
||||
import (
|
||||
"testing"
|
||||
|
||||
"github.com/sagernet/sing-box/option"
|
||||
"github.com/sagernet/sing-box/shater/model"
|
||||
)
|
||||
|
||||
// maxIPv4Datagram is the largest value the 16-bit IPv4 Total Length field can
|
||||
// hold, i.e. the largest IP datagram that can exist on the wire at all. It is
|
||||
// spelled out here rather than written as a literal because it IS the argument:
|
||||
// a device whose MTU is this large cannot be fragmented into.
|
||||
const maxIPv4Datagram = 65535
|
||||
|
||||
// TestL3TunnelMTULeavesNothingForTheKernelToFragment is the reason the number
|
||||
// is the number. Read this before changing l3MTU.
|
||||
//
|
||||
// The MTU of `shater-l3` is NOT a tunnel budget. It governs exactly one thing:
|
||||
// whether the KERNEL splits a packet on its way INTO the device. What the
|
||||
// engine then puts into the tunnel is sized separately and correctly, against
|
||||
// the OUTBOUND's own MTU — sing-tun's ForwardDispatcher.forwardToPort measures
|
||||
// each forwarded packet against Port.PortMTU() and either fragments to it (no
|
||||
// DF) or answers a `fragmentation needed` quoting it (DF).
|
||||
//
|
||||
// The value used to be 1420, the WireGuard payload budget, copied one layer too
|
||||
// far out. That did not make pings fit the tunnel; it made the kernel fragment
|
||||
// everything above 1392 bytes of payload right here, and a fragment is the one
|
||||
// thing sing-tun's dispatcher will not judge: Dispatch returns on
|
||||
// `parsed.fragment` BEFORE calling JudgeFlow, the fragments fall through to the
|
||||
// gVisor stack, which reassembles them and hands the echo to
|
||||
// ICMPForwarder.HandlePacket — whose installFlow requires an UNSPECIFIED port
|
||||
// address that a WireGuard/AWG endpoint never has. It declines, and HandlePacket
|
||||
// FORGES the echo reply. So `ping -s 1392` was honest and `ping -s 1393` was a
|
||||
// lie told by the router (D25).
|
||||
//
|
||||
// Hence the invariant, not merely the constant: the MTU must be at least the
|
||||
// largest datagram that can exist, so that NO packet can ever be fragmented
|
||||
// into this device. Anything smaller re-opens a band of sizes where a ping
|
||||
// reads as tunnelled without leaving the router.
|
||||
func TestL3TunnelMTULeavesNothingForTheKernelToFragment(t *testing.T) {
|
||||
to := l3TunOptions(t)
|
||||
if to.MTU < maxIPv4Datagram {
|
||||
t.Fatalf("l3-in MTU = %d, must be >= %d (the largest IP datagram there can be).\n"+
|
||||
"Below that the kernel fragments packets between the MTU and the datagram size on their way INTO shater-l3, sing-tun's ForwardDispatcher declines fragments without ever asking for a routing verdict, and the gVisor ICMP forwarder answers the echo ITSELF for any outbound whose port address is not unspecified — every WireGuard/AWG endpoint.\n"+
|
||||
"The result is not a dropped ping, it is a FORGED reply: the operator reads a working tunnel off a packet that died on the router. Do not size this to the tunnel MTU — forwardToPort already sizes against Port.PortMTU().",
|
||||
to.MTU, maxIPv4Datagram)
|
||||
}
|
||||
// The concrete value, so that a change is a decision and not a drift. 65535
|
||||
// is also sing-box's own default TUN MTU on Linux (protocol/tun/inbound.go).
|
||||
if to.MTU != maxIPv4Datagram {
|
||||
t.Fatalf("l3-in MTU = %d, want exactly %d — larger is impossible on the wire and buys nothing; if you have a reason, put it in the l3MTU comment and change this test deliberately", to.MTU, maxIPv4Datagram)
|
||||
}
|
||||
}
|
||||
|
||||
// TestL3TunnelMTUIsNotATunnelBudget guards the specific regression: someone
|
||||
// reading "the tunnel is 1420" and "re-aligning" the device to it. The two
|
||||
// numbers are unrelated, and making them equal is what produced the forged
|
||||
// replies in the first place.
|
||||
func TestL3TunnelMTUIsNotATunnelBudget(t *testing.T) {
|
||||
to := l3TunOptions(t)
|
||||
// 1500 is the largest MTU a plain Ethernet LAN hands the router in one
|
||||
// piece; every plausible tunnel budget (1420 for WireGuard, 1280 for a
|
||||
// conservative v6 path, 1412 for AWG with headers) sits below it. An l3-in
|
||||
// MTU anywhere in that range means the kernel is fragmenting into the
|
||||
// device again.
|
||||
if to.MTU <= 1500 {
|
||||
t.Fatalf("l3-in MTU = %d — that is tunnel/LAN-sized, so the kernel will fragment into shater-l3 and the gVisor ICMP forwarder will forge echo replies for WireGuard/AWG. The device MTU and the tunnel MTU are NOT the same number; see l3MTU's comment and D25.", to.MTU)
|
||||
}
|
||||
}
|
||||
|
||||
// l3TunOptions builds the l3_tunnel-on config and returns the TUN inbound's
|
||||
// options, failing with a useful address if the inbound is missing entirely.
|
||||
func l3TunOptions(t *testing.T) *option.TunInboundOptions {
|
||||
t.Helper()
|
||||
g := model.DefaultGlobals()
|
||||
g.L3Tunnel = true
|
||||
ins, warns := buildInboundsWith(g, lanTproxy())
|
||||
tuns := tunInbounds(ins)
|
||||
if len(tuns) != 1 {
|
||||
t.Fatalf("want exactly one L3 TUN inbound, got %d (warns=%v)", len(tuns), warns)
|
||||
}
|
||||
to, ok := tuns[0].Options.(*option.TunInboundOptions)
|
||||
if !ok {
|
||||
t.Fatalf("options are %T, want *option.TunInboundOptions", tuns[0].Options)
|
||||
}
|
||||
return to
|
||||
}
|
||||
@@ -21,7 +21,12 @@ import (
|
||||
// tag schema (per-group copy names, egress detours, the rule->group routing) is the
|
||||
// one the running router actually produces.
|
||||
func egressReachModel() *model.Model {
|
||||
g := model.DefaultGlobals()
|
||||
// l3_tunnel off: this fixture is an OUTBOUND topology (which nodes/groups/
|
||||
// chains a probe plan can reach), and it declares no inbounds at all. On the
|
||||
// seeded-ON default that earns the honest `icmp "tunnel"` warning — the L3
|
||||
// ingress has no tproxy divert plane to be fed by — which would turn the
|
||||
// zero-warnings assertion below into an assertion about ICMP.
|
||||
g := withoutL3Tunnel(model.DefaultGlobals())
|
||||
g.KillSwitch = "closed"
|
||||
g.ResolverDefault = "cf"
|
||||
g.ProbeURL = "https://probe.example/204"
|
||||
|
||||
@@ -43,6 +43,9 @@ func (b *builder) buildOutboundsAndEndpoints() ([]option.Outbound, []option.Endp
|
||||
// - interface: a direct outbound BOUND to the egress device and stamped
|
||||
// with the deterministic egress mark so engine-originated traffic through
|
||||
// it is routed by the matching policy table and bypasses the tproxy divert.
|
||||
// The device comes from netplane.EgressDevice — the SAME resolution the
|
||||
// routing uses — and an egress whose `interface` is empty resolves to no
|
||||
// device at all: it emits no outbound and is reported (see the case body).
|
||||
// - direct (and ""): a plain direct outbound (loop-guard mark only). Its
|
||||
// purpose is to carry a per-ruleset native DPI-bypass preset (D13): a rule
|
||||
// routed to it goes DIRECT but with tls_fragment/record/spoof applied on
|
||||
@@ -73,9 +76,37 @@ func (b *builder) buildOutboundsAndEndpoints() ([]option.Outbound, []option.Endp
|
||||
}
|
||||
switch egType {
|
||||
case "interface":
|
||||
dev := netplane.IfaceDevice(eg.Interface)
|
||||
// ONE resolution with the data plane, by calling the data plane's own.
|
||||
// netplane.EgressDevice is what addEgressRouting (the `ip rule` + table
|
||||
// for this egress's mark), the prerouting marking and the forward-chain
|
||||
// accept all consult; calling it here is not tidiness, it is the only
|
||||
// way this outbound cannot be bound to a device whose routing was never
|
||||
// installed.
|
||||
//
|
||||
// It had drifted, and the drift was a black hole. This code used to call
|
||||
// IfaceDevice(eg.Interface) directly, and IfaceDevice("") falls back to
|
||||
// "br-lan" — the right default for an INBOUND with no network, ruinous
|
||||
// here. An egress with no `interface` therefore produced an outbound
|
||||
// BOUND TO THE LAN BRIDGE and stamped with an egress mark that
|
||||
// EgressDevice had made addEgressRouting skip, so every node, group and
|
||||
// rule bound to that egress dialled public addresses out of br-lan under
|
||||
// a mark with no routing table, and neither the panel nor the log said
|
||||
// anything. (The `if dev == "" { dev = eg.Interface }` line that stood
|
||||
// here read as a guard against exactly this and could never execute:
|
||||
// IfaceDevice never returns "".) EgressDevice also trims, so
|
||||
// `option interface ' eth1 '` binds the same device the netplane routes.
|
||||
dev := netplane.EgressDevice(eg)
|
||||
if dev == "" {
|
||||
dev = eg.Interface
|
||||
// Fail-closed and SAY SO. Emitting nothing makes every reference to
|
||||
// this egress resolve through egressDetourOrBlock to tagBlock, which
|
||||
// stops that traffic rather than letting it out over the plain WAN.
|
||||
b.warnf("egress %q: type=interface but no interface name is set, so this egress resolves to no device — "+
|
||||
"the router installs no routing rule and no routing table for its mark. No outbound is built for it, so every "+
|
||||
"node, group and rule bound to this egress is FAIL-CLOSED: that traffic is blocked, not sent out over the "+
|
||||
"default WAN. The alternative was worse and used to be what happened — the outbound was bound to the LAN "+
|
||||
"bridge and its traffic disappeared into it with no diagnostic at all. Set `option interface` on this egress "+
|
||||
"to the uplink or tunnel it should leave through", eg.Name)
|
||||
continue
|
||||
}
|
||||
outbounds = append(outbounds, option.Outbound{
|
||||
Type: C.TypeDirect,
|
||||
|
||||
+301
-2
@@ -7,9 +7,11 @@ import (
|
||||
C "github.com/sagernet/sing-box/constant"
|
||||
"github.com/sagernet/sing-box/option"
|
||||
"github.com/sagernet/sing/common/json/badoption"
|
||||
N "github.com/sagernet/sing/common/network"
|
||||
|
||||
"github.com/sagernet/sing-box/shater/model"
|
||||
"github.com/sagernet/sing-box/shater/netplane"
|
||||
"github.com/sagernet/sing-box/shater/parse"
|
||||
)
|
||||
|
||||
// buildRoute assembles option.RouteOptions:
|
||||
@@ -117,6 +119,15 @@ func (b *builder) buildRoute() *option.RouteOptions {
|
||||
continue
|
||||
}
|
||||
|
||||
// An ICMP rule is the one shape whose traffic reaches the engine at layer 3,
|
||||
// and layer 3 has prerequisites the rule itself cannot state. Diagnose it
|
||||
// only when its target RESOLVED: an unresolved one already earned the much
|
||||
// louder ruleKillFallback warning above, and the kill fallback is a policy
|
||||
// decision about a broken reference, not about ICMP.
|
||||
if isICMPProto(r.Proto) && ok {
|
||||
b.warnICMPRule(r, want)
|
||||
}
|
||||
|
||||
route := option.RouteActionOptions{Outbound: target}
|
||||
b.applyDPI(&route, target, r.Name)
|
||||
general = append(general, option.Rule{
|
||||
@@ -378,6 +389,35 @@ func (b *builder) ruleMatchers(r model.Rule) (raw option.RawDefaultRule, matched
|
||||
switch p {
|
||||
case "tcp", "udp":
|
||||
raw.Network = badoption.Listable[string]{p}
|
||||
case protoICMP, protoICMPv4, protoICMPv6:
|
||||
// ICMP is a NETWORK, not a sniffed L7 label. Routed through the default
|
||||
// branch below it landed in RawDefaultRule.Protocol, where it is compared
|
||||
// against what the sniffers reported — and the sniffers never report
|
||||
// "icmp" (route.go's pre-match skips the sniff action for an ICMP flow
|
||||
// outright), so the rule was valid, warned about, and dead. The L3 ingress
|
||||
// therefore had no way to express its target at all: every ping fell to
|
||||
// whatever the catch-all resolved to, which on a real config is a group of
|
||||
// proxy nodes that cannot carry layer 3.
|
||||
//
|
||||
// NetworkItem.Match is a plain map lookup over metadata.Network, which
|
||||
// adapter.JudgeFlow sets to N.NetworkICMP for BOTH ICMPv4 and ICMPv6
|
||||
// (header.ICMPv4ProtocolNumber and header.ICMPv6ProtocolNumber share the
|
||||
// one case there). So there is exactly ONE network value here and `icmp`
|
||||
// covers both families — a separate `icmpv6` network would match nothing,
|
||||
// forever.
|
||||
//
|
||||
// The family is still expressible, and precisely: metadata.IPVersion is
|
||||
// derived from the destination address (route.go prepareMatchMetadata,
|
||||
// which PreMatch runs before the rules), and an ICMPv6 packet always
|
||||
// carries an IPv6 destination. `icmpv4`/`icmpv6` therefore narrow the SAME
|
||||
// network with an ip_version item rather than inventing a second network —
|
||||
// no false positives and no false negatives, unlike an inert Protocol
|
||||
// matcher (which would also silently widen to both families if we mapped
|
||||
// it onto plain `icmp`).
|
||||
raw.Network = badoption.Listable[string]{N.NetworkICMP}
|
||||
if v := icmpProtoIPVersion(p); v != 0 {
|
||||
raw.IPVersion = v
|
||||
}
|
||||
default:
|
||||
// Sniffed L7 protocol. route/rule.NewProtocolItem does NO validation —
|
||||
// it just compares the string against what the sniffers reported — so an
|
||||
@@ -386,7 +426,7 @@ func (b *builder) ruleMatchers(r model.Rule) (raw option.RawDefaultRule, matched
|
||||
// the rule to ALL traffic, which for a `direct` target is a leak), but
|
||||
// the operator is told it is inert.
|
||||
if !sniffedProtocols[p] {
|
||||
b.warnf("rule %q: proto %q is not something this engine can detect — the rule is kept but can NEVER match, so its traffic silently follows the rules below it. Use tcp, udp or one of: %s", r.Name, r.Proto, sniffedProtocolList())
|
||||
b.warnf("rule %q: proto %q is not something this engine can detect — the rule is kept but can NEVER match, so its traffic silently follows the rules below it. Use tcp, udp, icmp (icmpv4/icmpv6 narrow it to one family) or one of: %s", r.Name, r.Proto, sniffedProtocolList())
|
||||
}
|
||||
raw.Protocol = badoption.Listable[string]{p}
|
||||
}
|
||||
@@ -411,10 +451,269 @@ func (b *builder) ruleMatchers(r model.Rule) (raw option.RawDefaultRule, matched
|
||||
return raw, matched
|
||||
}
|
||||
|
||||
// The ICMP spellings `Rule.Proto` accepts. All three become the SAME engine
|
||||
// network (N.NetworkICMP); the two family-qualified ones additionally pin
|
||||
// RawDefaultRule.IPVersion. See the switch in ruleMatchers for why there is only
|
||||
// one network and why the family is an ip_version item rather than a second one.
|
||||
const (
|
||||
protoICMP = "icmp"
|
||||
protoICMPv4 = "icmpv4"
|
||||
protoICMPv6 = "icmpv6"
|
||||
)
|
||||
|
||||
// isICMPProto reports whether a Rule.Proto value asks for the L3 ingress.
|
||||
func isICMPProto(proto string) bool {
|
||||
switch strings.TrimSpace(strings.ToLower(proto)) {
|
||||
case protoICMP, protoICMPv4, protoICMPv6:
|
||||
return true
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
// icmpProtoIPVersion is the ip_version an ICMP proto narrows to; 0 = both
|
||||
// families (plain `icmp`) or not an ICMP proto at all.
|
||||
func icmpProtoIPVersion(proto string) int {
|
||||
switch strings.TrimSpace(strings.ToLower(proto)) {
|
||||
case protoICMPv4:
|
||||
return 4
|
||||
case protoICMPv6:
|
||||
return 6
|
||||
}
|
||||
return 0
|
||||
}
|
||||
|
||||
// warnICMPRule reports the ways an emitted `proto icmp` rule can still be a
|
||||
// no-op, none of which is visible anywhere else.
|
||||
//
|
||||
// A rule that matches nothing is normally cheap: its traffic falls through to
|
||||
// the rules below it. Not here. ICMP has no fall-through — the packet either
|
||||
// reaches an L3-capable outbound or is DROPPED (route.preMatchFlow's
|
||||
// l3-honest-drop block, which exists so the TUN stack cannot forge an echo reply
|
||||
// for a path that never carried the packet). So an ICMP rule that cannot fire is
|
||||
// not a dead setting, it is ping that stops working, with a rule in the UI that
|
||||
// says it should.
|
||||
//
|
||||
// Deliberately worded clear of shater/apply's criticalMarkers: a dropped ping is
|
||||
// not a protection gap (nothing leaks — the failure is fail-CLOSED), so these are
|
||||
// warnings, and phrases like "never applies" / "NOT emitted" would light the
|
||||
// panel's alarm banner for something that costs the operator ping and nothing else.
|
||||
func (b *builder) warnICMPRule(r model.Rule, want string) {
|
||||
if !b.m.Globals.L3Tunnel {
|
||||
// The prerequisite, and the only one whose absence makes the other checks
|
||||
// moot: with no L3 ingress the packet never enters the engine, so the target
|
||||
// is not consulted at all.
|
||||
b.warnf("rule %q: proto %s matches only traffic that reaches the engine at layer 3, and nothing does while globals l3_tunnel is off — kernel TPROXY diverts TCP and UDP and nothing else, so no ping ever enters the engine and this rule cannot fire. ICMP is governed by globals untunnelable (%s) instead, whatever this rule's target %q says. Turn l3_tunnel on to make the rule live", r.Name, strings.ToLower(strings.TrimSpace(r.Proto)), untunnelablePolicyName(b.m.Globals.Untunnelable), want)
|
||||
return
|
||||
}
|
||||
if icmpProtoIPVersion(r.Proto) == 6 && !b.m.Globals.IPv6 {
|
||||
// netplane/nft.go marks ipv6-icmp into the L3 TUN only under Globals.IPv6,
|
||||
// and generate/inbound.go gives that TUN an IPv6 address on the same
|
||||
// condition. With IPv6 off the v6 half of the ingress simply does not exist.
|
||||
b.warnf("rule %q: proto icmpv6 needs globals ipv6 on — the L3 ingress marks ipv6-icmp into the engine's TUN only then, and the TUN is given no IPv6 address either, so with IPv6 off this rule cannot fire. Use proto icmp (one value, both families) or turn ipv6 on", r.Name)
|
||||
return
|
||||
}
|
||||
// A port matcher and ICMP are mutually exclusive by construction:
|
||||
// adapter.JudgeFlow zeroes source and destination ports for an ICMP flow
|
||||
// before PreMatch runs, so a PortItem next to the network item can never be
|
||||
// satisfied. The rule is emitted (dropping the port would WIDEN it) but it is
|
||||
// dead.
|
||||
if single, ranges, _ := splitPorts(r.DstPort); len(single)+len(ranges) > 0 {
|
||||
b.warnf("rule %q: proto %s together with a port matcher can never match — an ICMP packet has no port, and the engine zeroes both ports on an ICMP flow before matching it (adapter.JudgeFlow). Drop the port from this rule, or move the port part into a rule of its own", r.Name, strings.ToLower(strings.TrimSpace(r.Proto)))
|
||||
}
|
||||
switch b.l3Target(want, 0) {
|
||||
case l3Drops:
|
||||
b.warnf("rule %q: proto %s is routed to %q, which cannot carry a layer-3 packet — only wireguard/AmneziaWG nodes and direct/interface egresses reach the engine's flow path; every proxy protocol (vless, vmess, trojan, shadowsocks, hysteria2, tuic, shadowtls) and the byedpi SOCKS egress cannot, because none of them can implement it. The engine DROPS a ping routed at such a target rather than let the TUN stack answer the echo itself from a path that never carried the packet, so this rule makes those pings fail instead of tunnelling them. Point it at a wireguard/AmneziaWG node, at an interface egress, or at a chain whose LAST hop is one of those", r.Name, strings.ToLower(strings.TrimSpace(r.Proto)), want)
|
||||
case l3Partial:
|
||||
b.warnf("rule %q: proto %s is routed to %q, whose members disagree about layer 3 — the ping is carried while the group's current member is a wireguard/AmneziaWG node and dropped while it is a proxy-protocol one, and nothing in the UI says which is in force right now. Point the rule at the L3-capable node itself (or at a group holding only those) if ping must behave the same from one minute to the next", r.Name, strings.ToLower(strings.TrimSpace(r.Proto)), want)
|
||||
}
|
||||
}
|
||||
|
||||
// untunnelablePolicyName spells the Untunnelable policy for a diagnostic, naming
|
||||
// the empty value as the "block" it means (model.Globals.Untunnelable).
|
||||
func untunnelablePolicyName(policy string) string {
|
||||
if p := strings.ToLower(strings.TrimSpace(policy)); p != "" {
|
||||
return p
|
||||
}
|
||||
return "block"
|
||||
}
|
||||
|
||||
// l3Verdict is what a route target does with a layer-3 packet.
|
||||
type l3Verdict int
|
||||
|
||||
const (
|
||||
// l3Unknown: undecidable from the model alone — say nothing rather than guess.
|
||||
l3Unknown l3Verdict = iota
|
||||
// l3Carries: the emitted outbound implements adapter.FlowOutbound.
|
||||
l3Carries
|
||||
// l3Drops: it does not, so route.preMatchFlow drops the packet.
|
||||
l3Drops
|
||||
// l3Blocks: `block`. It drops the packet too, but that IS the stated policy,
|
||||
// so it is never reported.
|
||||
l3Blocks
|
||||
// l3Partial: a group whose members disagree; the answer changes with the pick.
|
||||
l3Partial
|
||||
)
|
||||
|
||||
// l3Target answers, from the MODEL, whether a rule target ends up at an outbound
|
||||
// that can carry a layer-3 packet.
|
||||
//
|
||||
// # Why this is decidable here at all
|
||||
//
|
||||
// The capability is not a runtime property to be discovered: it is fixed by the
|
||||
// outbound TYPE this package is about to emit, and adapter registration makes the
|
||||
// list exhaustive. Only `direct` (protocol/direct/outbound.go:66), `wireguard`
|
||||
// (protocol/wireguard/endpoint.go:73), `tailscale` and `bridge` declare
|
||||
// N.NetworkICMP among their networks, and of those exactly two are reachable from
|
||||
// a shater model: a wireguard/AmneziaWG node, and the direct outbound behind
|
||||
// `direct` or an interface/direct egress. Everything else this generator emits —
|
||||
// vless, vmess, trojan, shadowsocks, hysteria2, tuic, shadowtls and the byedpi
|
||||
// SOCKS egress — cannot.
|
||||
//
|
||||
// # The predicate that MUST change in lockstep
|
||||
//
|
||||
// "Is this node an endpoint?" is decided in outbound.go by
|
||||
// `p.WG != nil || p.Protocol == "wireguard"`. l3Node repeats that test. If one
|
||||
// side learns a new L3-capable node kind and the other does not, this diagnosis
|
||||
// starts lying in whichever direction the drift went — a false alarm on a working
|
||||
// ping, or silence on a broken one.
|
||||
//
|
||||
// depth bounds the chain->chain recursion; expandHops already refuses cycles, so
|
||||
// it is a belt on top of a brace.
|
||||
func (b *builder) l3Target(want string, depth int) l3Verdict {
|
||||
if depth > 8 {
|
||||
return l3Unknown
|
||||
}
|
||||
// The kind switch mirrors resolveTarget exactly — the two must agree about what
|
||||
// a target string means, or this diagnoses a different outbound than the one the
|
||||
// rule is routed to.
|
||||
kind, name := model.SplitTarget(want)
|
||||
switch strings.ToLower(kind) {
|
||||
case tagDirect, "":
|
||||
if strings.EqualFold(want, tagBlock) {
|
||||
return l3Blocks
|
||||
}
|
||||
return l3Carries
|
||||
case tagBlock:
|
||||
return l3Blocks
|
||||
case "node":
|
||||
return b.l3Node(name)
|
||||
case "group":
|
||||
return b.l3Group(name)
|
||||
case "egress":
|
||||
return b.l3Egress(name)
|
||||
case "chain":
|
||||
return b.l3Chain(name, depth)
|
||||
default:
|
||||
// Bare name: node then group, same order resolveTarget uses.
|
||||
if b.nodeTags[want] {
|
||||
return b.l3Node(want)
|
||||
}
|
||||
if b.groupTags[want] {
|
||||
return b.l3Group(want)
|
||||
}
|
||||
return l3Unknown
|
||||
}
|
||||
}
|
||||
|
||||
// l3Node: a node carries layer 3 exactly when it is emitted as a WireGuard/
|
||||
// AmneziaWG ENDPOINT rather than a proxy outbound.
|
||||
func (b *builder) l3Node(name string) l3Verdict {
|
||||
for i := range b.m.Nodes {
|
||||
n := b.m.Nodes[i]
|
||||
if n.Name != name || !n.Enabled {
|
||||
continue
|
||||
}
|
||||
p, err := parse.ParseShareLink(n.URI)
|
||||
if err != nil {
|
||||
// Unreachable from warnICMPRule (an unparseable node has no tag, so the
|
||||
// target would not have resolved), and not ours to report twice anyway.
|
||||
return l3Unknown
|
||||
}
|
||||
if p.WG != nil || p.Protocol == "wireguard" {
|
||||
return l3Carries
|
||||
}
|
||||
return l3Drops
|
||||
}
|
||||
return l3Unknown
|
||||
}
|
||||
|
||||
// l3Group folds its members' verdicts. route.preMatchFlow unwraps a group through
|
||||
// group.Now() before testing the outbound, so the group's answer IS its current
|
||||
// member's — which is why a mixed group is reported as its own case rather than
|
||||
// rounded to either side.
|
||||
func (b *builder) l3Group(name string) l3Verdict {
|
||||
g, found := b.findGroup(name)
|
||||
if !found {
|
||||
return l3Unknown
|
||||
}
|
||||
var members []string
|
||||
// groupMembers re-emits the member diagnostics buildGroups already surfaced.
|
||||
b.withoutNewWarnings(func() { members = b.groupMembers(g) })
|
||||
carries, drops := 0, 0
|
||||
for _, member := range members {
|
||||
switch b.l3Node(member) {
|
||||
case l3Carries:
|
||||
carries++
|
||||
case l3Drops:
|
||||
drops++
|
||||
}
|
||||
}
|
||||
switch {
|
||||
case carries == 0 && drops == 0:
|
||||
return l3Unknown
|
||||
case carries == 0:
|
||||
return l3Drops
|
||||
case drops == 0:
|
||||
return l3Carries
|
||||
default:
|
||||
return l3Partial
|
||||
}
|
||||
}
|
||||
|
||||
// l3Egress: an interface/direct egress is a DIRECT outbound (the one proxy
|
||||
// outbound in the shipped registry that builds a ping.Port), byedpi is SOCKS, and
|
||||
// any other type emits no outbound at all — so it never reaches this diagnosis.
|
||||
func (b *builder) l3Egress(name string) l3Verdict {
|
||||
for _, eg := range b.m.Egresses {
|
||||
if eg.Name != name {
|
||||
continue
|
||||
}
|
||||
switch strings.ToLower(strings.TrimSpace(eg.Type)) {
|
||||
case "interface", "direct", "":
|
||||
return l3Carries
|
||||
default:
|
||||
return l3Drops
|
||||
}
|
||||
}
|
||||
return l3Unknown
|
||||
}
|
||||
|
||||
// l3Chain: a rule routed at a chain enters at the LAST hop's wrapper (chain.go —
|
||||
// Detour points backwards, so Ln is the outbound the rule is handed to and the
|
||||
// exit on the wire). That wrapper is a copy of Ln's own outbound, so Ln's
|
||||
// capability is the chain's.
|
||||
func (b *builder) l3Chain(name string, depth int) l3Verdict {
|
||||
var (
|
||||
hops []string
|
||||
baseDetour string
|
||||
expanded bool
|
||||
)
|
||||
// expandHops flattens sub-chains and reports cycles/undefined references;
|
||||
// resolveChain already surfaced all of that for this same chain.
|
||||
b.withoutNewWarnings(func() {
|
||||
expanded = b.expandHops(name, map[string]bool{}, &hops, &baseDetour, true)
|
||||
})
|
||||
if !expanded || len(hops) == 0 {
|
||||
return l3Unknown
|
||||
}
|
||||
return b.l3Target(hops[len(hops)-1], depth+1)
|
||||
}
|
||||
|
||||
// sniffedProtocols is exactly the set of L7 labels this engine's sniffers can
|
||||
// ever put on a connection (route/rule.RuleActionSniff.build + the default
|
||||
// stream/packet sniffer sets in route/route.go). A `proto` outside this set —
|
||||
// and outside tcp/udp — matches nothing, forever.
|
||||
// and outside the tcp/udp/icmp NETWORKS handled ahead of it in ruleMatchers —
|
||||
// matches nothing, forever.
|
||||
var sniffedProtocols = map[string]bool{
|
||||
C.ProtocolTLS: true,
|
||||
C.ProtocolHTTP: true,
|
||||
|
||||
@@ -124,8 +124,16 @@ func TestDuplicateTproxyPortStillStarts(t *testing.T) {
|
||||
if !changed {
|
||||
t.Fatalf("expected Apply changed==true (warnings: %v)", warns)
|
||||
}
|
||||
if len(opts.Inbounds) != 1 {
|
||||
t.Fatalf("want 1 inbound after the clash guard, got %d", len(opts.Inbounds))
|
||||
// Count the TPROXY listeners, not len(opts.Inbounds): the subject is that the
|
||||
// second CLASHING listener is gone (that is what used to fail Start), and a
|
||||
// total also counts the synthetic L3 TUN, whose presence is decided by an
|
||||
// unrelated global. See tproxyListeners in route_matchers_test.go.
|
||||
got := tproxyListeners(opts)
|
||||
if len(got) != 1 {
|
||||
t.Fatalf("want 1 tproxy listener after the clash guard, got %d", len(got))
|
||||
}
|
||||
if got[0].Tag != "in-lan" {
|
||||
t.Fatalf("the surviving listener is %q, want the first-declared in-lan", got[0].Tag)
|
||||
}
|
||||
if !routeWarnsHave(warns, "already used by inbound") {
|
||||
t.Fatalf("expected a listen-clash warning, got %v", warns)
|
||||
|
||||
@@ -478,6 +478,21 @@ func findOutboundByTag(opts option.Options, tag string) *option.Outbound {
|
||||
|
||||
// --- inbounds ----------------------------------------------------------------
|
||||
|
||||
// tproxyListeners returns the emitted TPROXY inbounds. The clash-guard tests below
|
||||
// count THESE and not len(opts.Inbounds): the subject is how many tproxy listeners
|
||||
// survive the guard, and a total that also counts the synthetic L3 TUN (or any
|
||||
// future synthetic inbound) answers a different question — one whose "right"
|
||||
// number changes whenever an unrelated global flips.
|
||||
func tproxyListeners(opts option.Options) []option.Inbound {
|
||||
var out []option.Inbound
|
||||
for _, in := range opts.Inbounds {
|
||||
if in.Type == C.TypeTProxy {
|
||||
out = append(out, in)
|
||||
}
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
// TestDuplicateTproxyPortSkipped: two tproxy inbounds on the same port fail to
|
||||
// bind at Start, which takes the WHOLE engine down — and with the engine down the
|
||||
// fail-closed plane blackholes the LAN. The second must be dropped with a warning.
|
||||
@@ -493,8 +508,15 @@ func TestDuplicateTproxyPortSkipped(t *testing.T) {
|
||||
if err != nil {
|
||||
t.Fatalf("Generate: %v", err)
|
||||
}
|
||||
if len(opts.Inbounds) != 1 {
|
||||
t.Fatalf("want 1 inbound after the clash guard, got %d", len(opts.Inbounds))
|
||||
got := tproxyListeners(opts)
|
||||
if len(got) != 1 {
|
||||
t.Fatalf("want 1 tproxy listener after the clash guard, got %d", len(got))
|
||||
}
|
||||
// The SURVIVOR must be the first one declared, not the second: the guard drops
|
||||
// the inbound that would have failed to bind, and a last-wins guard would
|
||||
// silently hand the LAN's divert port to a different section.
|
||||
if got[0].Tag != "in-lan" {
|
||||
t.Fatalf("the surviving listener is %q, want the first-declared in-lan", got[0].Tag)
|
||||
}
|
||||
if !routeWarnsHave(warns, "already used by inbound") {
|
||||
t.Fatalf("expected a listen-clash warning, got %v", warns)
|
||||
@@ -502,7 +524,8 @@ func TestDuplicateTproxyPortSkipped(t *testing.T) {
|
||||
}
|
||||
|
||||
// TestMultiLanDistinctTproxyPortsBothKept: the legitimate multi-LAN shape (two
|
||||
// tproxy listeners on DIFFERENT ports) must still emit both.
|
||||
// tproxy listeners on DIFFERENT ports) must still emit both — the clash guard has
|
||||
// to be about the ADDRESS, not about the type.
|
||||
func TestMultiLanDistinctTproxyPortsBothKept(t *testing.T) {
|
||||
m := &model.Model{
|
||||
Globals: model.DefaultGlobals(),
|
||||
@@ -515,8 +538,21 @@ func TestMultiLanDistinctTproxyPortsBothKept(t *testing.T) {
|
||||
if err != nil {
|
||||
t.Fatalf("Generate: %v", err)
|
||||
}
|
||||
if len(opts.Inbounds) != 2 {
|
||||
t.Fatalf("want both tproxy inbounds, got %d (warnings %v)", len(opts.Inbounds), warns)
|
||||
got := tproxyListeners(opts)
|
||||
if len(got) != 2 {
|
||||
t.Fatalf("want both tproxy listeners, got %d (warnings %v)", len(got), warns)
|
||||
}
|
||||
for i, wantPort := range []uint16{12345, 12346} {
|
||||
to, ok := got[i].Options.(*option.TProxyInboundOptions)
|
||||
if !ok {
|
||||
t.Fatalf("listener %d options are %T, want *option.TProxyInboundOptions", i, got[i].Options)
|
||||
}
|
||||
if to.ListenPort != wantPort {
|
||||
t.Fatalf("listener %d binds port %d, want %d — both LANs must keep the port their nft divert aims at", i, to.ListenPort, wantPort)
|
||||
}
|
||||
}
|
||||
if routeWarnsHave(warns, "already used by inbound") {
|
||||
t.Fatalf("distinct ports must not trip the clash guard, got %v", warns)
|
||||
}
|
||||
}
|
||||
|
||||
@@ -737,7 +773,7 @@ func TestChainCyclesTerminateFailClosed(t *testing.T) {
|
||||
// chainFlattenModel is a model with three usable nodes and a closed kill-switch,
|
||||
// ready for the chain-flatten cases below to attach chains + one targeting rule.
|
||||
func chainFlattenModel() *model.Model {
|
||||
g := nonDNSGlobals()
|
||||
g := plainGlobals()
|
||||
g.KillSwitch = "closed"
|
||||
return &model.Model{
|
||||
Globals: g,
|
||||
|
||||
@@ -0,0 +1,377 @@
|
||||
// The rule side of the L3 ingress: whether `proto icmp` can be SAID at all, and
|
||||
// whether saying it is honest.
|
||||
//
|
||||
// Before this, `icmp` fell through ruleMatchers' proto switch into
|
||||
// RawDefaultRule.Protocol — the sniffed-L7 field — where it was compared against
|
||||
// labels the sniffers report. They never report "icmp" (route.go's pre-match
|
||||
// skips the sniff action for an ICMP flow outright), so the rule was structurally
|
||||
// valid and permanently dead. The consequence was not a dead setting but a dead
|
||||
// FEATURE: with no way to write "ICMP goes here", every ping fell to whatever the
|
||||
// catch-all resolved to, and on the configuration this was built for that is a
|
||||
// group of VLESS nodes, which cannot carry layer 3 at all.
|
||||
//
|
||||
// Portable (no box.New): these assert on the generated option.Options and on the
|
||||
// warning texts only. The live-engine proofs that a direct/wireguard outbound
|
||||
// really carries the flow live in l3_egress_linux_test.go and
|
||||
// l3_integration_linux_test.go.
|
||||
package generate
|
||||
|
||||
import (
|
||||
"strings"
|
||||
"testing"
|
||||
|
||||
"github.com/sagernet/sing-box/option"
|
||||
|
||||
"github.com/sagernet/sing-box/shater/model"
|
||||
)
|
||||
|
||||
// --- fixture ----------------------------------------------------------------
|
||||
|
||||
// icmpModel builds a model carrying one of everything the L3 verdict has to tell
|
||||
// apart: a WireGuard node (an ENDPOINT — carries layer 3), a Shadowsocks node (a
|
||||
// proxy outbound — cannot), an interface egress (a direct outbound — carries), a
|
||||
// byedpi egress (SOCKS — cannot), three groups spanning all-proxy/all-L3/mixed,
|
||||
// and two chains that differ only in which kind of node they EXIT through.
|
||||
func icmpModel(l3Tunnel bool, rules ...model.Rule) *model.Model {
|
||||
g := model.DefaultGlobals()
|
||||
g.DNSIntercept = false // not the subject; keeps the DNS diagnostics out
|
||||
g.KillSwitch = "closed"
|
||||
g.L3Tunnel = l3Tunnel
|
||||
return &model.Model{
|
||||
Globals: g,
|
||||
Inbounds: []model.Inbound{
|
||||
{Name: "lan", Enabled: true, Type: "tproxy", TproxyPort: 12345, TCP: true, UDP: true},
|
||||
},
|
||||
Nodes: []model.Node{
|
||||
{Name: "wg1", Enabled: true, URI: wgDedupURI(wgDedupKey(1), wgDedupKey(2), "203.0.113.10", 51820)},
|
||||
{Name: "wg2", Enabled: true, URI: wgDedupURI(wgDedupKey(3), wgDedupKey(4), "203.0.113.11", 51820)},
|
||||
{Name: "ss1", Enabled: true, URI: "ss://aes-256-gcm:secret@203.0.113.1:8388#ss1"},
|
||||
},
|
||||
Egresses: []model.Egress{
|
||||
{Name: "wan2", Type: "interface", Interface: "wan2"},
|
||||
{Name: "bd", Type: "byedpi", Port: 1080},
|
||||
},
|
||||
Groups: []model.Group{
|
||||
{Name: "proxies", Source: "manual", Nodes: []string{"ss1"}},
|
||||
{Name: "l3only", Source: "manual", Nodes: []string{"wg1", "wg2"}},
|
||||
{Name: "mixed", Source: "manual", Nodes: []string{"ss1", "wg1"}},
|
||||
},
|
||||
Chains: []model.Chain{
|
||||
{Name: "exit-proxy", Hops: []string{"node:wg1", "node:ss1"}},
|
||||
{Name: "exit-wg", Hops: []string{"node:ss1", "node:wg1"}},
|
||||
},
|
||||
Rules: rules,
|
||||
}
|
||||
}
|
||||
|
||||
// genICMP generates the fixture and returns the model-derived route rules plus
|
||||
// the warnings.
|
||||
func genICMP(t *testing.T, l3Tunnel bool, rules ...model.Rule) ([]option.Rule, []string) {
|
||||
t.Helper()
|
||||
opts, warns, err := GenerateWithWarnings(icmpModel(l3Tunnel, rules...))
|
||||
if err != nil {
|
||||
t.Fatalf("Generate: %v", err)
|
||||
}
|
||||
return generalRules(opts.Route), warns
|
||||
}
|
||||
|
||||
// oneICMPRule generates a single-rule model and returns that rule's matchers.
|
||||
func oneICMPRule(t *testing.T, l3Tunnel bool, r model.Rule) (option.RawDefaultRule, []string) {
|
||||
t.Helper()
|
||||
rules, warns := genICMP(t, l3Tunnel, r)
|
||||
if len(rules) != 1 {
|
||||
t.Fatalf("want exactly 1 emitted route rule, got %d (warns=%v) — a rule the generator drops carries no target at all, and for ICMP that is not a fall-through but a drop", len(rules), warns)
|
||||
}
|
||||
return rules[0].DefaultOptions.RawDefaultRule, warns
|
||||
}
|
||||
|
||||
// icmpRule is a `proto <p>` rule scoped to one LAN source, pointed at target.
|
||||
func icmpRule(name, proto, target string) model.Rule {
|
||||
return model.Rule{
|
||||
Name: name, Enabled: true, Order: 10,
|
||||
Src: []string{"192.168.1.0/24"},
|
||||
Proto: proto,
|
||||
Target: target,
|
||||
}
|
||||
}
|
||||
|
||||
// --- what the matcher must become -------------------------------------------
|
||||
|
||||
// TestProtoICMPBecomesTheNetworkNotTheSniffedProtocol is the fix itself.
|
||||
//
|
||||
// route/rule.NewNetworkItem matches metadata.Network with a plain map lookup, and
|
||||
// adapter.JudgeFlow sets that to N.NetworkICMP ("icmp") for an ICMP flow — so the
|
||||
// value belongs in Network. RawDefaultRule.Protocol is the sniffed-L7 field
|
||||
// (NewProtocolItem compares against what the sniffers labelled the connection),
|
||||
// and nothing ever labels a flow "icmp": route.go's PreMatch skips the sniff
|
||||
// action for ICMP before it starts. A value there is a rule that cannot fire.
|
||||
func TestProtoICMPBecomesTheNetworkNotTheSniffedProtocol(t *testing.T) {
|
||||
raw, warns := oneICMPRule(t, true, icmpRule("ping", "icmp", "node:wg1"))
|
||||
if len(raw.Protocol) != 0 {
|
||||
t.Fatalf("proto icmp landed in RawDefaultRule.Protocol = %v — that field is matched against the SNIFFED protocol label, and no sniffer ever produces \"icmp\" (PreMatch skips sniffing for an ICMP flow), so the rule is valid, silent and permanently dead: every ping keeps falling to the catch-all instead of the target the operator wrote", raw.Protocol)
|
||||
}
|
||||
if len(raw.Network) != 1 || raw.Network[0] != "icmp" {
|
||||
t.Fatalf("RawDefaultRule.Network = %v, want [icmp] — NetworkItem.Match is a map lookup over metadata.Network, which adapter.JudgeFlow sets to exactly this string for an ICMP flow; anything else and the L3 ingress has no expressible target at all", raw.Network)
|
||||
}
|
||||
if raw.IPVersion != 0 {
|
||||
t.Fatalf("RawDefaultRule.IPVersion = %d, want 0 — plain `icmp` must cover BOTH families (JudgeFlow maps ICMPv4 and ICMPv6 to the one network), so narrowing it here would silently drop half the pings the operator asked for", raw.IPVersion)
|
||||
}
|
||||
if routeWarnsHave(warns, "is not something this engine can detect") {
|
||||
t.Fatalf("icmp was reported as an undetectable sniffed protocol: %v — it is a NETWORK, and the warning tells the operator to stop using the only spelling that works", warns)
|
||||
}
|
||||
}
|
||||
|
||||
// TestProtoICMPFamilySpellingsNarrowByIPVersion pins the answer to "is ICMPv6 a
|
||||
// second network?": it is not. adapter.JudgeFlow folds
|
||||
// header.ICMPv4ProtocolNumber and header.ICMPv6ProtocolNumber into ONE case and
|
||||
// sets N.NetworkICMP for both, so an `icmpv6` network value would match nothing,
|
||||
// forever. The family is still expressible, and exactly: metadata.IPVersion comes
|
||||
// from the destination address (prepareMatchMetadata, which PreMatch runs), and
|
||||
// an ICMPv6 packet always has an IPv6 destination. So the family spellings narrow
|
||||
// the same network with ip_version instead of inventing a second one.
|
||||
func TestProtoICMPFamilySpellingsNarrowByIPVersion(t *testing.T) {
|
||||
for _, tc := range []struct {
|
||||
proto string
|
||||
want int
|
||||
}{{"icmpv4", 4}, {"icmpv6", 6}} {
|
||||
raw, warns := oneICMPRule(t, true, icmpRule("ping", tc.proto, "node:wg1"))
|
||||
if len(raw.Network) != 1 || raw.Network[0] != "icmp" {
|
||||
t.Fatalf("proto %s: Network = %v, want [icmp] — there is no separate icmpv6 network in this engine; emitting one produces a rule that can never match", tc.proto, raw.Network)
|
||||
}
|
||||
if raw.IPVersion != tc.want {
|
||||
t.Fatalf("proto %s: IPVersion = %d, want %d — without it the rule silently covers the OTHER family too, which is a different rule than the one written", tc.proto, raw.IPVersion, tc.want)
|
||||
}
|
||||
if len(raw.Protocol) != 0 {
|
||||
t.Fatalf("proto %s: landed in Protocol = %v (warns=%v) — the sniffed-L7 field, where it can never match", tc.proto, raw.Protocol, warns)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestProtoTCPUDPAndSniffedAreUnchanged: the new branch must not move any value
|
||||
// that already worked. tcp/udp stay networks, a sniffed L7 label stays in
|
||||
// Protocol, and neither picks up an ip_version it never had.
|
||||
func TestProtoTCPUDPAndSniffedAreUnchanged(t *testing.T) {
|
||||
for _, network := range []string{"tcp", "udp"} {
|
||||
raw, _ := oneICMPRule(t, true, icmpRule("t", network, "node:ss1"))
|
||||
if len(raw.Network) != 1 || raw.Network[0] != network {
|
||||
t.Fatalf("proto %s: Network = %v, want [%s]", network, raw.Network, network)
|
||||
}
|
||||
if len(raw.Protocol) != 0 || raw.IPVersion != 0 {
|
||||
t.Fatalf("proto %s: Protocol = %v / IPVersion = %d, want empty/0 — the ICMP branch leaked into the transport branch", network, raw.Protocol, raw.IPVersion)
|
||||
}
|
||||
}
|
||||
raw, warns := oneICMPRule(t, true, icmpRule("t", "tls", "node:ss1"))
|
||||
if len(raw.Protocol) != 1 || raw.Protocol[0] != "tls" {
|
||||
t.Fatalf("proto tls: Protocol = %v, want [tls] — a sniffed L7 label belongs in the sniffed field", raw.Protocol)
|
||||
}
|
||||
if len(raw.Network) != 0 || raw.IPVersion != 0 {
|
||||
t.Fatalf("proto tls: Network = %v / IPVersion = %d, want empty/0", raw.Network, raw.IPVersion)
|
||||
}
|
||||
if routeWarnsHave(warns, "is not something this engine can detect") {
|
||||
t.Fatalf("tls is sniffable and was reported as not: %v", warns)
|
||||
}
|
||||
// The vocabulary the undetectable-proto warning offers must now name icmp —
|
||||
// it is the list an operator reads when their spelling was rejected, and
|
||||
// leaving icmp out of it points them away from the only working value.
|
||||
_, warns = oneICMPRule(t, true, icmpRule("t", "ping", "node:ss1"))
|
||||
if !routeWarnsHave(warns, "is not something this engine can detect") {
|
||||
t.Fatalf("proto \"ping\" is neither a network nor a sniffed label, yet nothing was reported: %v", warns)
|
||||
}
|
||||
if !routeWarnsHave(warns, "Use tcp, udp, icmp") {
|
||||
t.Fatalf("the undetectable-proto warning still offers only tcp/udp + sniffed labels: %v — an operator who wrote \"ping\" is told every value EXCEPT the one that would work", warns)
|
||||
}
|
||||
}
|
||||
|
||||
// --- the two prerequisites the rule cannot state itself ----------------------
|
||||
|
||||
// TestProtoICMPWithoutL3TunnelIsReported: with globals l3_tunnel off, kernel
|
||||
// TPROXY diverts TCP and UDP and nothing else, so no ICMP packet ever enters the
|
||||
// engine and the rule — perfectly well-formed — matches nothing at all. Silence
|
||||
// here is the worst kind: the panel shows a rule that says ping is tunnelled.
|
||||
func TestProtoICMPWithoutL3TunnelIsReported(t *testing.T) {
|
||||
_, warns := oneICMPRule(t, false, icmpRule("ping", "icmp", "node:wg1"))
|
||||
if !routeWarnsHave(warns, "globals l3_tunnel is off") {
|
||||
t.Fatalf("a proto icmp rule under l3_tunnel=0 was accepted in silence: %v — the L3 ingress is the ONLY path an ICMP packet has into the engine, so without it the rule is decoration", warns)
|
||||
}
|
||||
// And the opposite: turning the ingress on must silence it, or the warning is
|
||||
// noise that trains the operator to ignore the section.
|
||||
_, warns = oneICMPRule(t, true, icmpRule("ping", "icmp", "node:wg1"))
|
||||
if routeWarnsHave(warns, "globals l3_tunnel is off") {
|
||||
t.Fatalf("l3_tunnel is ON and the rule was still reported as dead: %v", warns)
|
||||
}
|
||||
}
|
||||
|
||||
// TestProtoICMPv6WithoutIPv6IsReported: the v6 half of the ingress is gated twice
|
||||
// on globals ipv6 — netplane/nft.go marks ipv6-icmp into the TUN only then, and
|
||||
// generate/inbound.go gives the TUN an IPv6 address on the same condition. An
|
||||
// icmpv6 rule with IPv6 off is therefore inert for a reason nothing else states.
|
||||
func TestProtoICMPv6WithoutIPv6IsReported(t *testing.T) {
|
||||
m := icmpModel(true, icmpRule("ping6", "icmpv6", "node:wg1"))
|
||||
m.Globals.IPv6 = false
|
||||
_, warns, err := GenerateWithWarnings(m)
|
||||
if err != nil {
|
||||
t.Fatalf("Generate: %v", err)
|
||||
}
|
||||
if !routeWarnsHave(warns, "proto icmpv6 needs globals ipv6 on") {
|
||||
t.Fatalf("an icmpv6 rule under ipv6=0 was accepted in silence: %v — neither the nft marking nor the TUN address exists in that configuration, so the rule cannot fire", warns)
|
||||
}
|
||||
_, warns = oneICMPRule(t, true, icmpRule("ping6", "icmpv6", "node:wg1"))
|
||||
if routeWarnsHave(warns, "proto icmpv6 needs globals ipv6 on") {
|
||||
t.Fatalf("ipv6 is on and the icmpv6 rule was still reported: %v", warns)
|
||||
}
|
||||
}
|
||||
|
||||
// TestProtoICMPWithPortMatcherIsReported: adapter.JudgeFlow zeroes both ports on
|
||||
// an ICMP flow before PreMatch runs, so a port item sitting next to the network
|
||||
// item can never be satisfied. The rule is still EMITTED — dropping the port
|
||||
// matcher would widen it — but it is dead, and only this says so.
|
||||
func TestProtoICMPWithPortMatcherIsReported(t *testing.T) {
|
||||
r := icmpRule("ping", "icmp", "node:wg1")
|
||||
r.DstPort = "443"
|
||||
_, warns := oneICMPRule(t, true, r)
|
||||
if !routeWarnsHave(warns, "an ICMP packet has no port") {
|
||||
t.Fatalf("proto icmp + dst_port was accepted in silence: %v — the two matchers are mutually exclusive by construction, so the rule can never fire", warns)
|
||||
}
|
||||
_, warns = oneICMPRule(t, true, icmpRule("ping", "icmp", "node:wg1"))
|
||||
if routeWarnsHave(warns, "an ICMP packet has no port") {
|
||||
t.Fatalf("a portless icmp rule was reported as having a port: %v", warns)
|
||||
}
|
||||
}
|
||||
|
||||
// --- does the TARGET carry layer 3? -----------------------------------------
|
||||
|
||||
// TestProtoICMPToAnL4TargetIsReported is the honesty half. An ICMP flow has no
|
||||
// fall-through: route.preMatchFlow's l3-honest-drop block DROPS a ping routed at
|
||||
// an outbound that cannot carry layer 3, precisely so the TUN stack cannot forge
|
||||
// an echo reply for a path that never saw the packet. So "ICMP -> group of VLESS
|
||||
// nodes" is not a dead setting, it is ping that stops working — and the operator
|
||||
// wrote the rule believing the opposite.
|
||||
//
|
||||
// Every target below is decidable from the model, because the capability is fixed
|
||||
// by the outbound TYPE this package emits: only direct (protocol/direct) and the
|
||||
// wireguard/AmneziaWG endpoints declare N.NetworkICMP.
|
||||
func TestProtoICMPToAnL4TargetIsReported(t *testing.T) {
|
||||
for _, target := range []string{
|
||||
"node:ss1", // a proxy outbound
|
||||
"group:proxies", // a group of nothing but proxy outbounds
|
||||
"egress:bd", // byedpi: a SOCKS hop, which cannot implement tun.Port
|
||||
"chain:exit-proxy", // the rule enters at the LAST hop, and that one is ss1
|
||||
} {
|
||||
_, warns := oneICMPRule(t, true, icmpRule("ping", "icmp", target))
|
||||
if !routeWarnsHave(warns, "cannot carry a layer-3 packet") {
|
||||
t.Fatalf("proto icmp -> %s was accepted in silence: %v — the engine DROPS that ping (route.preMatchFlow, l3-honest-drop) and nothing else in the UI says so; the operator reads a rule claiming ping is tunnelled and a ping that fails", target, warns)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestProtoICMPToAnL3TargetIsSilent: the same check must stay quiet for every
|
||||
// target that really does carry the packet, including `block` — a blocked ping is
|
||||
// a stated policy, not an accident — and a chain that merely PASSES THROUGH a
|
||||
// proxy hop on its way to a WireGuard exit.
|
||||
func TestProtoICMPToAnL3TargetIsSilent(t *testing.T) {
|
||||
for _, target := range []string{
|
||||
"node:wg1", // a wireguard endpoint
|
||||
"group:l3only", // a group of nothing else
|
||||
"egress:wan2", // an interface egress = a direct outbound
|
||||
"direct", // the baseline direct outbound
|
||||
"block", // dropping the ping IS the policy here
|
||||
"chain:exit-wg", // enters at the LAST hop, which is wg1
|
||||
} {
|
||||
_, warns := oneICMPRule(t, true, icmpRule("ping", "icmp", target))
|
||||
if routeWarnsHave(warns, "cannot carry a layer-3 packet") {
|
||||
t.Fatalf("proto icmp -> %s was reported as unable to carry layer 3: %v — a false alarm on a working path teaches the operator to ignore the one that is real", target, warns)
|
||||
}
|
||||
if routeWarnsHave(warns, "disagree about layer 3") {
|
||||
t.Fatalf("proto icmp -> %s was reported as a mixed group: %v", target, warns)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestProtoICMPToAMixedGroupIsReported: route.preMatchFlow unwraps a group
|
||||
// through group.Now() before testing the outbound, so a group holding both kinds
|
||||
// answers differently from one pick to the next — ping works, then does not, with
|
||||
// nothing in the UI to say which member is in force. That is its own report, not
|
||||
// a rounding of either side.
|
||||
func TestProtoICMPToAMixedGroupIsReported(t *testing.T) {
|
||||
_, warns := oneICMPRule(t, true, icmpRule("ping", "icmp", "group:mixed"))
|
||||
if !routeWarnsHave(warns, "disagree about layer 3") {
|
||||
t.Fatalf("proto icmp -> a group of one wireguard and one proxy node was accepted in silence: %v — the ping's fate follows the group's current pick", warns)
|
||||
}
|
||||
if routeWarnsHave(warns, "cannot carry a layer-3 packet") {
|
||||
t.Fatalf("a MIXED group was reported as unable to carry layer 3 at all: %v — it can, half the time, and telling the operator otherwise sends them to change a target that is only unstable", warns)
|
||||
}
|
||||
}
|
||||
|
||||
// TestProtoICMPDiagnosticsAreQuietForNonICMPRules: none of the above may fire on
|
||||
// the tcp/udp/sniffed rules that make up every existing config. A byedpi egress
|
||||
// carrying TCP is exactly right, and saying otherwise would flood the panel.
|
||||
func TestProtoICMPDiagnosticsAreQuietForNonICMPRules(t *testing.T) {
|
||||
_, warns := genICMP(t, true,
|
||||
icmpRule("a", "tcp", "egress:bd"),
|
||||
icmpRule("b", "udp", "group:proxies"),
|
||||
icmpRule("c", "tls", "node:ss1"),
|
||||
)
|
||||
for _, marker := range []string{
|
||||
"cannot carry a layer-3 packet",
|
||||
"disagree about layer 3",
|
||||
"globals l3_tunnel is off",
|
||||
"an ICMP packet has no port",
|
||||
} {
|
||||
if routeWarnsHave(warns, marker) {
|
||||
t.Fatalf("a non-ICMP rule drew the ICMP diagnostic %q: %v", marker, warns)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestProtoICMPUnresolvedTargetIsNotDoubleReported: a rule whose target does not
|
||||
// resolve already earns ruleKillFallback's much louder warning and is routed to
|
||||
// block. A capability verdict on the target it never reaches would compete with
|
||||
// it — the operator's problem is the broken reference, and it is the one sentence
|
||||
// that must be read first.
|
||||
//
|
||||
// The shape below is the one where the two actually collide: a group named after
|
||||
// a node is SKIPPED by buildGroups (the outbound manager last-wins on a duplicate
|
||||
// tag), so `group:ss1` does not resolve — while the group DEFINITION is still in
|
||||
// the model, holding a proxy member, so a verdict is perfectly computable for it.
|
||||
func TestProtoICMPUnresolvedTargetIsNotDoubleReported(t *testing.T) {
|
||||
m := icmpModel(true, icmpRule("ping", "icmp", "group:ss1"))
|
||||
m.Groups = append(m.Groups, model.Group{Name: "ss1", Source: "manual", Nodes: []string{"ss1"}})
|
||||
_, warns, err := GenerateWithWarnings(m)
|
||||
if err != nil {
|
||||
t.Fatalf("Generate: %v", err)
|
||||
}
|
||||
if !routeWarnsHave(warns, "unresolved target") {
|
||||
t.Fatalf("a rule pointing at a non-existent node lost its unresolved-target warning: %v", warns)
|
||||
}
|
||||
if routeWarnsHave(warns, "cannot carry a layer-3 packet") {
|
||||
t.Fatalf("the unresolved target was ALSO reported as unable to carry layer 3: %v — the operator's problem is the missing node, and the second sentence competes with the first", warns)
|
||||
}
|
||||
}
|
||||
|
||||
// TestProtoICMPWarningsStayBelowTheAlarmThreshold: shater/apply grades generate's
|
||||
// free text, and a handful of substrings ("never applies", "NOT emitted", …) light
|
||||
// the panel's alarm banner. A ping that fails is fail-CLOSED — nothing leaks — so
|
||||
// these belong under `warning`, and the wording must not drift into the markers.
|
||||
// Checked here rather than in shater/apply because that package imports this one.
|
||||
func TestProtoICMPWarningsStayBelowTheAlarmThreshold(t *testing.T) {
|
||||
var texts []string
|
||||
_, w := oneICMPRule(t, false, icmpRule("a", "icmp", "node:wg1"))
|
||||
texts = append(texts, w...)
|
||||
_, w = oneICMPRule(t, true, icmpRule("b", "icmp", "node:ss1"))
|
||||
texts = append(texts, w...)
|
||||
_, w = oneICMPRule(t, true, icmpRule("c", "icmp", "group:mixed"))
|
||||
texts = append(texts, w...)
|
||||
|
||||
// The exact substrings shater/apply/warnings.go greps for.
|
||||
markers := []string{"NOT emitted", "inert", "has NO effect", "never applies", "in the clear"}
|
||||
for _, text := range texts {
|
||||
if !strings.Contains(text, "layer 3") && !strings.Contains(text, "layer-3") && !strings.Contains(text, "l3_tunnel") {
|
||||
continue // somebody else's warning
|
||||
}
|
||||
for _, m := range markers {
|
||||
if strings.Contains(text, m) {
|
||||
t.Fatalf("ICMP warning %q contains the critical marker %q — it would light the panel's alarm banner for a fail-CLOSED ping failure, and every cosmetic critical teaches the operator to skip the real one", text, m)
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -85,7 +85,7 @@ func contains(list []string, want string) bool {
|
||||
// so it runs on every platform.
|
||||
func TestRoutingRuleSetInlineDomain(t *testing.T) {
|
||||
m := &model.Model{
|
||||
Globals: nonDNSGlobals(),
|
||||
Globals: plainGlobals(),
|
||||
Rulesets: []model.Ruleset{
|
||||
{Name: "ads", Type: "domain", Source: "inline", Entries: []string{"ads.example", "doubleclick.net"}},
|
||||
},
|
||||
@@ -134,7 +134,7 @@ func TestRoutingRuleSetInlineDomain(t *testing.T) {
|
||||
// looks configured and matches nothing.
|
||||
func TestRoutingRuleSetInlineDomainRegex(t *testing.T) {
|
||||
m := &model.Model{
|
||||
Globals: nonDNSGlobals(),
|
||||
Globals: plainGlobals(),
|
||||
Rulesets: []model.Ruleset{
|
||||
{Name: "ads", Type: "domain", Source: "inline", Entries: []string{`regexp:^ads\.`}},
|
||||
},
|
||||
@@ -173,7 +173,7 @@ func TestRoutingRuleSetInlineDomainRegex(t *testing.T) {
|
||||
// and the route rule references it.
|
||||
func TestRoutingRuleSetInlineIPCIDR(t *testing.T) {
|
||||
m := &model.Model{
|
||||
Globals: nonDNSGlobals(),
|
||||
Globals: plainGlobals(),
|
||||
Rulesets: []model.Ruleset{
|
||||
{Name: "cn", Type: "ipcidr", Source: "inline", Entries: []string{"10.0.0.0/8", "192.168.0.0/16"}},
|
||||
},
|
||||
@@ -259,7 +259,7 @@ func TestRoutingRuleSetUnusedNotEmitted(t *testing.T) {
|
||||
// The single-category tag now carries the category suffix (rs-<name>-<category>).
|
||||
func TestRoutingRuleSetGeositeCategory(t *testing.T) {
|
||||
m := &model.Model{
|
||||
Globals: nonDNSGlobals(),
|
||||
Globals: plainGlobals(),
|
||||
Rulesets: []model.Ruleset{
|
||||
{Name: "yt", Source: "geosite", Categories: []string{"youtube"}},
|
||||
},
|
||||
@@ -302,7 +302,7 @@ func TestRoutingRuleSetGeositeCategory(t *testing.T) {
|
||||
// references the ruleset matches BOTH tags.
|
||||
func TestRoutingRuleSetGeositeMultiCategory(t *testing.T) {
|
||||
m := &model.Model{
|
||||
Globals: nonDNSGlobals(),
|
||||
Globals: plainGlobals(),
|
||||
Rulesets: []model.Ruleset{
|
||||
{Name: "social", Source: "geosite", Categories: []string{"youtube", "telegram"}},
|
||||
},
|
||||
@@ -346,7 +346,7 @@ func TestRoutingRuleSetGeositeMultiCategory(t *testing.T) {
|
||||
// lower-cased (rs-ru-ru).
|
||||
func TestRoutingRuleSetGeoipCountryLowercased(t *testing.T) {
|
||||
m := &model.Model{
|
||||
Globals: nonDNSGlobals(),
|
||||
Globals: plainGlobals(),
|
||||
Rulesets: []model.Ruleset{
|
||||
{Name: "ru", Source: "geoip", Categories: []string{"RU"}},
|
||||
},
|
||||
@@ -415,7 +415,13 @@ func TestRoutingRuleSetGeositeNoCategoryWarnsSkipped(t *testing.T) {
|
||||
// materialises a REMOTE sing-geosite .srs rule-set (bl-<name>) — the DNS-filter
|
||||
// counterpart of TestRoutingRuleSetGeositeCategory. Pure codegen (no box.New).
|
||||
func TestFilterGeositeBlocklistRemote(t *testing.T) {
|
||||
g := model.DefaultGlobals()
|
||||
// withoutL3Tunnel: the subject here is rule-set materialisation, so
|
||||
// dns_intercept keeps its real default — but this model declares no inbounds,
|
||||
// and on the seeded-ON l3_tunnel that earns the honest `icmp "tunnel"` warning
|
||||
// (there is no tproxy divert plane to feed the L3 ingress). The zero-warnings
|
||||
// assertion below would otherwise swallow it as if it said something about
|
||||
// rule-sets.
|
||||
g := withoutL3Tunnel(model.DefaultGlobals())
|
||||
g.DNSFilter = true
|
||||
g.ResolverDefault = "cf"
|
||||
m := &model.Model{
|
||||
@@ -454,7 +460,13 @@ func TestFilterGeositeBlocklistRemote(t *testing.T) {
|
||||
// materialises TWO remote sing-geosite rule-sets (tags bl-<name>-<cat>), and BOTH
|
||||
// are referenced by the block DNS rule so either category is filtered.
|
||||
func TestFilterGeositeBlocklistMultiCategory(t *testing.T) {
|
||||
g := model.DefaultGlobals()
|
||||
// withoutL3Tunnel: the subject here is rule-set materialisation, so
|
||||
// dns_intercept keeps its real default — but this model declares no inbounds,
|
||||
// and on the seeded-ON l3_tunnel that earns the honest `icmp "tunnel"` warning
|
||||
// (there is no tproxy divert plane to feed the L3 ingress). The zero-warnings
|
||||
// assertion below would otherwise swallow it as if it said something about
|
||||
// rule-sets.
|
||||
g := withoutL3Tunnel(model.DefaultGlobals())
|
||||
g.DNSFilter = true
|
||||
g.ResolverDefault = "cf"
|
||||
m := &model.Model{
|
||||
@@ -528,7 +540,13 @@ func TestFilterGeositeNoCategoryWarnsSkipped(t *testing.T) {
|
||||
// TestRoutingAndFilterRuleSetsCoexist: a DNS-filter blocklist (bl-*) and a
|
||||
// routing ruleset (rs-*) live in one config with distinct tags, no collision.
|
||||
func TestRoutingAndFilterRuleSetsCoexist(t *testing.T) {
|
||||
g := model.DefaultGlobals()
|
||||
// withoutL3Tunnel: the subject here is rule-set materialisation, so
|
||||
// dns_intercept keeps its real default — but this model declares no inbounds,
|
||||
// and on the seeded-ON l3_tunnel that earns the honest `icmp "tunnel"` warning
|
||||
// (there is no tproxy divert plane to feed the L3 ingress). The zero-warnings
|
||||
// assertion below would otherwise swallow it as if it said something about
|
||||
// rule-sets.
|
||||
g := withoutL3Tunnel(model.DefaultGlobals())
|
||||
g.DNSFilter = true
|
||||
g.ResolverDefault = "cf"
|
||||
m := &model.Model{
|
||||
@@ -1185,7 +1203,7 @@ func TestFileSourceExistingPathEmitted(t *testing.T) {
|
||||
t.Fatalf("write: %v", err)
|
||||
}
|
||||
m := &model.Model{
|
||||
Globals: nonDNSGlobals(),
|
||||
Globals: plainGlobals(),
|
||||
Rulesets: []model.Ruleset{
|
||||
{Name: "j", Source: "file", Path: jsonPath},
|
||||
{Name: "s", Source: "file", Path: srsPath},
|
||||
|
||||
@@ -165,11 +165,13 @@ func TestShippedTagSetConstructsDeclaredProtocols(t *testing.T) {
|
||||
// No inbound on purpose: this test is about protocol construction, and a
|
||||
// tproxy listener would demand CAP_NET_ADMIN from every runner. The tproxy
|
||||
// path is covered by the rest of the suite.
|
||||
// nonDNSGlobals, not DefaultGlobals: this test is about PROTOCOL construction
|
||||
// plainGlobals, not DefaultGlobals: this test is about PROTOCOL construction
|
||||
// under the shipped tag set, and it asserts zero generator warnings. With the
|
||||
// default dns_intercept (D24) a model with no `config resolver` earns a DNS
|
||||
// warning that has nothing to say about whether vless or hysteria2 compiled in.
|
||||
m := &model.Model{Globals: nonDNSGlobals(), Nodes: nodes}
|
||||
// warning, and with the default l3_tunnel a model with no tproxy inbound earns
|
||||
// the `icmp "tunnel"` one; neither has anything to say about whether vless or
|
||||
// hysteria2 compiled in.
|
||||
m := &model.Model{Globals: plainGlobals(), Nodes: nodes}
|
||||
|
||||
opts, warns, changed := applyAndClose(t, m)
|
||||
if !changed {
|
||||
|
||||
+513
-120
@@ -1,6 +1,7 @@
|
||||
package generate
|
||||
|
||||
import (
|
||||
"encoding/json"
|
||||
"fmt"
|
||||
"sort"
|
||||
"strings"
|
||||
@@ -50,6 +51,33 @@ import (
|
||||
// whatever put a second device in the config, it is gone before the config
|
||||
// reaches box.New.
|
||||
//
|
||||
// # Two copies is not automatically a conflict — merge before you block
|
||||
//
|
||||
// The physical limit is about DEVICES, not about how many times a device is
|
||||
// written down. If two copies would build the SAME device — same key, same peers,
|
||||
// same address/MTU/AWG parameters and, crucially, the same dialer (so the same
|
||||
// uplink under the same mark) — then there is only ever one device on the wire and
|
||||
// nothing to fight over. Those copies are MERGED: one survives and every reference
|
||||
// to the others is rewritten to point at it. Nothing is blocked and nothing is
|
||||
// reported, because nothing is wrong.
|
||||
//
|
||||
// This is what makes the shape an operator actually wants expressible:
|
||||
//
|
||||
// chain:ewan-wg-subs = egress:ewan -> node:awgout -> group:sub0 -> ... (default route)
|
||||
// chain:ewan-wg = egress:ewan -> node:awgout (ICMP terminal)
|
||||
//
|
||||
// Both chains enter over the same egress, so `awgout` materialises identically in
|
||||
// both; one copy is the whole house's default route and the other is a one-rule
|
||||
// terminal. Merging keeps BOTH. Blocking one of them, as this pass used to do,
|
||||
// meant picking one to break — and it picked by tag order, which on the live router
|
||||
// took out the default route for the entire house.
|
||||
//
|
||||
// Only a REAL incompatibility — a different detour, a different peer set, different
|
||||
// device parameters — is two devices, and then one genuinely has to go. There the
|
||||
// survivor is chosen by WEIGHT (see tagWeight): the copy the default route depends
|
||||
// on outranks a copy one narrow rule depends on, and the warning says in words what
|
||||
// stops working.
|
||||
//
|
||||
// # What is left alone
|
||||
//
|
||||
// Non-wireguard outbounds are untouched (their copies are legitimate), groups
|
||||
@@ -96,22 +124,68 @@ func wgPrivateKeyOf(deviceKey string) string {
|
||||
return priv
|
||||
}
|
||||
|
||||
// dedupWireGuardEndpoints enforces "one private key = at most one live device" on
|
||||
// wgMaterialisation is the FULL identity of the device an endpoint will build:
|
||||
// everything about it except the tag it is filed under. Two endpoints with equal
|
||||
// materialisations are one device written down twice, and folding them together
|
||||
// changes nothing on the wire.
|
||||
//
|
||||
// It is the whole marshalled options blob rather than a hand-picked field list on
|
||||
// purpose. A hand-picked list has to be maintained: the day someone adds a field to
|
||||
// WireGuardEndpointOptions (or to the embedded DialerOptions — bind_interface,
|
||||
// netns, routing_mark, udp_fragment are all in there and all change what device you
|
||||
// get) a list that was not updated starts merging two devices that differ, which is
|
||||
// a silent routing change. The blob is closed by construction: an unknown
|
||||
// difference reads as "different", which costs at worst a duplicate reported
|
||||
// instead of merged — the recoverable direction.
|
||||
//
|
||||
// ok=false when the options cannot be marshalled at all; the caller then treats the
|
||||
// endpoint as its own class, i.e. refuses to merge what it cannot prove identical.
|
||||
func wgMaterialisation(ep option.Endpoint) (string, bool) {
|
||||
raw, err := json.Marshal(ep.Options)
|
||||
if err != nil {
|
||||
return "", false
|
||||
}
|
||||
return ep.Type + "|" + string(raw), true
|
||||
}
|
||||
|
||||
// wgClass is one set of endpoint tags that all build the SAME device. Merging
|
||||
// happens inside a class; blocking happens only between classes.
|
||||
type wgClass struct {
|
||||
tags []string // every tag in the class, in tag order
|
||||
rep string // the one that survives if this class is kept
|
||||
weight tagWeight // how much traffic depends on the class as a whole
|
||||
}
|
||||
|
||||
// wgConflict is one reported "two devices, one key" decision, held until the
|
||||
// rewrite has been applied. It is not warned about on the spot because the honest
|
||||
// version of the warning has to describe the config that RESULTS — see
|
||||
// wgLoseConsequence, which must not tell the operator his default route is dead
|
||||
// when the group it points at still has other live members.
|
||||
type wgConflict struct {
|
||||
node string
|
||||
keep wgClass
|
||||
lose wgClass
|
||||
}
|
||||
|
||||
// dedupWireGuardEndpoints enforces "one private key = at most one live DEVICE" on
|
||||
// the fully assembled config. Called from GenerateWithWarningsAt once opts is
|
||||
// complete (chain outbounds/endpoints folded in, route built), because only then
|
||||
// is every producer's output visible and the reachability walk meaningful.
|
||||
//
|
||||
// Per group of endpoints sharing a device identity:
|
||||
// Per group of endpoints sharing a device identity (wgDeviceKey):
|
||||
//
|
||||
// - endpoints nothing can route to are DELETED, silently. This is the normal
|
||||
// case (the always-emitted base endpoint of a node used only in a chain) and
|
||||
// warning about it would train the operator to ignore the warning list.
|
||||
// - if more than one REACHABLE copy remains, the config asks for something a
|
||||
// single key cannot express — e.g. the same WG node entered over two different
|
||||
// WANs in two chains, which is two devices by definition. The first copy in
|
||||
// tag order is kept (deterministic across runs), the rest are deleted and every
|
||||
// reference to them is rewritten to `block`, and each is reported as critical.
|
||||
// - exactly one left: nothing to say.
|
||||
// - the reachable ones are partitioned by wgMaterialisation. Copies inside one
|
||||
// partition are the same device: one survives, the others are deleted and every
|
||||
// reference to them is REWRITTEN TO THE SURVIVOR. Silent — nothing was
|
||||
// misconfigured and nothing stopped working.
|
||||
// - only if more than one partition remains does the config genuinely ask for two
|
||||
// devices from one key, which the peer cannot give. The HEAVIEST partition wins
|
||||
// (tagWeight: the default route outranks a narrow rule); the losers are deleted
|
||||
// and their references fail-closed to `block`, with a warning that names the
|
||||
// consequence in words.
|
||||
//
|
||||
// Deleting rather than merely re-pointing is essential: box.New starts EVERY
|
||||
// endpoint in the config regardless of whether anything routes to it, so a
|
||||
@@ -123,6 +197,7 @@ func (b *builder) dedupWireGuardEndpoints(opts *option.Options) {
|
||||
}
|
||||
|
||||
tagsByKey := map[string][]string{}
|
||||
epByTag := map[string]option.Endpoint{}
|
||||
var keyOrder []string
|
||||
for i := range opts.Endpoints {
|
||||
key, ok := wgDeviceKey(opts.Endpoints[i])
|
||||
@@ -133,6 +208,7 @@ func (b *builder) dedupWireGuardEndpoints(opts *option.Options) {
|
||||
keyOrder = append(keyOrder, key)
|
||||
}
|
||||
tagsByKey[key] = append(tagsByKey[key], opts.Endpoints[i].Tag)
|
||||
epByTag[opts.Endpoints[i].Tag] = opts.Endpoints[i]
|
||||
}
|
||||
var dupKeys []string
|
||||
for _, key := range keyOrder {
|
||||
@@ -144,47 +220,249 @@ func (b *builder) dedupWireGuardEndpoints(opts *option.Options) {
|
||||
return // the common case: no key is materialised twice, nothing to do
|
||||
}
|
||||
|
||||
reachable := reachableOptionTags(opts, b.subscriptionDetourSeeds()...)
|
||||
// The subscription fetch detours are seeds AND pins: see wgPinnedTags.
|
||||
fetchSeeds := b.subscriptionDetourSeeds()
|
||||
weights := optionTagWeights(opts, fetchSeeds...)
|
||||
pinned := wgPinnedTags(fetchSeeds)
|
||||
var names map[string]string // private key -> node name, built only if we must name one
|
||||
drop := map[string]bool{}
|
||||
var conflicts []wgConflict
|
||||
replace := map[string]string{}
|
||||
for _, key := range dupKeys {
|
||||
tags := append([]string(nil), tagsByKey[key]...)
|
||||
sort.Strings(tags) // deterministic survivor, independent of emission order
|
||||
sort.Strings(tags) // deterministic, independent of emission order
|
||||
var live []string
|
||||
for _, tag := range tags {
|
||||
if reachable[tag] {
|
||||
if weights[tag].reachable() {
|
||||
live = append(live, tag)
|
||||
continue
|
||||
}
|
||||
drop[tag] = true // dead weight: a device nothing routes to
|
||||
replace[tag] = tagBlock // dead weight: a device nothing routes to
|
||||
}
|
||||
if len(live) < 2 {
|
||||
continue
|
||||
}
|
||||
|
||||
classes := wgClassesOf(live, epByTag, weights, pinned)
|
||||
|
||||
// One class = one device described several times. Fold them together and say
|
||||
// nothing: the config is not wrong, it is redundant, and after this pass it is
|
||||
// not even that.
|
||||
winner := classes[0]
|
||||
for _, other := range classes[1:] {
|
||||
if other.weight.heavierThan(winner.weight) {
|
||||
winner = other
|
||||
}
|
||||
}
|
||||
for _, tag := range winner.tags {
|
||||
if tag != winner.rep {
|
||||
replace[tag] = winner.rep
|
||||
}
|
||||
}
|
||||
if len(classes) == 1 {
|
||||
continue
|
||||
}
|
||||
|
||||
// More than one class: two devices are actually being asked for, and one key
|
||||
// cannot be two devices. The losers go, fail-closed.
|
||||
if names == nil {
|
||||
names = b.wgNodeNamesByKey()
|
||||
}
|
||||
name := names[wgPrivateKeyOf(key)]
|
||||
if name == "" {
|
||||
name = live[0] // unknown to the model (defensive): name it by its tag
|
||||
name = winner.rep // unknown to the model (defensive): name it by its tag
|
||||
}
|
||||
keep := live[0]
|
||||
for _, tag := range live[1:] {
|
||||
drop[tag] = true
|
||||
b.warnf("node %q: this WireGuard node is materialised twice in the engine config — as %q and as %q — and traffic can reach both. A WireGuard peer keeps ONE session per public key, so two devices built from one private key evict each other continuously and NEITHER tunnel passes traffic. Only %q is kept; everything that routed through %q is fail-closed (blocked) instead of leaving over the plain WAN. Give the second path its own WireGuard node (its own key), or route both paths through the same one",
|
||||
name, keep, tag, keep, tag)
|
||||
for _, lose := range classes {
|
||||
if lose.rep == winner.rep {
|
||||
continue
|
||||
}
|
||||
for _, tag := range lose.tags {
|
||||
replace[tag] = tagBlock
|
||||
}
|
||||
conflicts = append(conflicts, wgConflict{node: name, keep: winner, lose: lose})
|
||||
}
|
||||
}
|
||||
if len(drop) == 0 {
|
||||
if len(replace) == 0 {
|
||||
return
|
||||
}
|
||||
|
||||
// Rewrite first, delete second: the rewrite must still see the tags it is
|
||||
// replacing. Every dangling reference is pointed at `block`, never at `direct`
|
||||
// — a consumer whose tunnel just disappeared must stop, not fall out onto the
|
||||
// plain WAN with the router's real address.
|
||||
retargetDroppedTags(opts, drop, tagBlock)
|
||||
removeDroppedEndpoints(opts, drop)
|
||||
// replacing. A merged tag is repointed at its survivor — the same device, so
|
||||
// nothing changes on the wire. A blocked tag is pointed at `block`, never at
|
||||
// `direct`: a consumer whose tunnel just disappeared must stop, not fall out onto
|
||||
// the plain WAN with the router's real address.
|
||||
remapTags(opts, replace)
|
||||
removeRemappedEndpoints(opts, replace)
|
||||
|
||||
// Warn LAST, against the config that actually resulted. Whether losing a copy
|
||||
// takes the default route down is not decidable from the copy alone — the default
|
||||
// route may point at a GROUP that still has other live members — and a warning
|
||||
// that guesses is a warning that lies. defaultRouteBlocked reads the answer off
|
||||
// the finished config instead.
|
||||
if len(conflicts) > 0 {
|
||||
dead := defaultRouteBlocked(opts)
|
||||
for _, c := range conflicts {
|
||||
b.warnWGDeviceConflict(c, dead)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// defaultRouteBlocked reports whether the default route — route.Final, where every
|
||||
// packet that matched no rule goes — now ends at `block`, i.e. whether the LAN has
|
||||
// lost its way out. A group counts as usable while ANY member is usable, which is
|
||||
// the whole reason this is measured rather than inferred: dropping one member of a
|
||||
// balancing pool is not the same event as dropping the pool.
|
||||
func defaultRouteBlocked(opts *option.Options) bool {
|
||||
if opts.Route == nil {
|
||||
return false
|
||||
}
|
||||
return !usableTerminal(opts.Route.Final, optionGraph(opts), map[string]bool{})
|
||||
}
|
||||
|
||||
// usableTerminal answers "can traffic sent to this tag still leave the router".
|
||||
//
|
||||
// It follows DETOURS as well as group membership, because an outbound is only as
|
||||
// alive as the thing it dials through: a chain exit whose previous hop was just
|
||||
// deleted has Detour=block and carries exactly nothing, even though the exit
|
||||
// outbound itself is still sitting in the config looking healthy. Stopping at the
|
||||
// first non-group tag would have reported such a route as working.
|
||||
//
|
||||
// The visiting set is the cycle guard: a tag that can only reach itself carries
|
||||
// nothing, so returning false on re-entry is both safe and correct.
|
||||
func usableTerminal(tag string, graph map[string]optionNode, visiting map[string]bool) bool {
|
||||
if tag == "" || tag == tagBlock || visiting[tag] {
|
||||
return false
|
||||
}
|
||||
node, ok := graph[tag]
|
||||
if !ok {
|
||||
return false // dangling: box.New will refuse it, and it certainly carries nothing
|
||||
}
|
||||
visiting[tag] = true
|
||||
defer delete(visiting, tag)
|
||||
if node.group {
|
||||
for _, member := range node.members {
|
||||
if usableTerminal(member, graph, visiting) {
|
||||
return true
|
||||
}
|
||||
}
|
||||
return false
|
||||
}
|
||||
if node.detour == "" {
|
||||
return true // a real outbound/endpoint dialling straight out of the router
|
||||
}
|
||||
return usableTerminal(node.detour, graph, visiting)
|
||||
}
|
||||
|
||||
// wgPinnedTags marks the tags a merge may NOT rename away from.
|
||||
//
|
||||
// remapTags can rewrite every reference that lives inside option.Options, and that
|
||||
// is what makes merging safe there. The subscription fetch detour does not live
|
||||
// there: it is a model string that apply.UpdateSubscription hands to
|
||||
// engine.HTTPClient, which resolves it against the RUNNING box (see
|
||||
// subscriptionDetourSeeds). Nothing in this pass can rewrite it, so if the tag it
|
||||
// names is the copy that got merged away, updating the subscription fails with
|
||||
// "unknown outbound tag" — the same regression the seed itself was added to
|
||||
// prevent, re-entering through the merge.
|
||||
//
|
||||
// So such a tag wins the survivor slot outright. It costs nothing: every copy in a
|
||||
// class is the same device, so which one keeps its name is free to choose, and this
|
||||
// is the one choice that has an external constraint on it.
|
||||
func wgPinnedTags(fetchSeeds []string) map[string]bool {
|
||||
if len(fetchSeeds) == 0 {
|
||||
return nil
|
||||
}
|
||||
pinned := make(map[string]bool, len(fetchSeeds))
|
||||
for _, tag := range fetchSeeds {
|
||||
if tag != "" {
|
||||
pinned[tag] = true
|
||||
}
|
||||
}
|
||||
return pinned
|
||||
}
|
||||
|
||||
// wgClassesOf partitions the reachable copies of one device key into "same device"
|
||||
// classes, in tag order so the result is stable across runs. Each class's rep is the
|
||||
// tag that survives if the class is kept: a pinned tag first (an outside reference
|
||||
// that cannot be rewritten — see wgPinnedTags), then the heaviest, then tag order.
|
||||
// So the surviving name is the one the most important path already refers to.
|
||||
//
|
||||
// An endpoint whose materialisation cannot be computed gets a class of its own:
|
||||
// unable to prove it identical, this refuses to merge it. That is the fail-closed
|
||||
// direction — it can cost a warning, never a silent routing change.
|
||||
func wgClassesOf(live []string, epByTag map[string]option.Endpoint, weights map[string]tagWeight, pinned map[string]bool) []wgClass {
|
||||
betterRep := func(candidate, current string) bool {
|
||||
if pinned[candidate] != pinned[current] {
|
||||
return pinned[candidate]
|
||||
}
|
||||
return weights[candidate].heavierThan(weights[current])
|
||||
}
|
||||
var classes []wgClass
|
||||
byMat := map[string]int{}
|
||||
for _, tag := range live {
|
||||
mat, ok := wgMaterialisation(epByTag[tag])
|
||||
if i, seen := byMat[mat]; ok && seen {
|
||||
classes[i].tags = append(classes[i].tags, tag)
|
||||
if betterRep(tag, classes[i].rep) {
|
||||
classes[i].rep = tag
|
||||
}
|
||||
classes[i].weight = classes[i].weight.merge(weights[tag])
|
||||
continue
|
||||
}
|
||||
if ok {
|
||||
byMat[mat] = len(classes)
|
||||
}
|
||||
classes = append(classes, wgClass{tags: []string{tag}, rep: tag, weight: weights[tag]})
|
||||
}
|
||||
return classes
|
||||
}
|
||||
|
||||
// warnWGDeviceConflict reports the one case this pass cannot fix for the operator:
|
||||
// two copies of a key that are NOT the same device, so one of them has to be turned
|
||||
// off. The old wording stated the mechanism correctly and the consequence not at
|
||||
// all — "everything that routed through X is fail-closed" is true of a stray test
|
||||
// rule and of the default route for the entire house, and the operator who hit this
|
||||
// on the live router read it as the former. So the consequence is spelled out.
|
||||
func (b *builder) warnWGDeviceConflict(c wgConflict, defaultRouteDead bool) {
|
||||
b.warnf("node %q: this WireGuard node is materialised twice in the engine config — as %q and as %q — and the two are NOT the same device: they differ in how the device itself is built (its detour/uplink, its peers, or its device parameters), so they cannot be folded into one. A WireGuard peer keeps ONE session per public key, so two devices built from one private key evict each other continuously and NEITHER tunnel passes traffic. %q is kept because %s. %q is deleted and everything that routed through it is fail-closed — BLOCKED, not sent out the plain WAN with the router's real address. CONSEQUENCE: %s. Fix it by giving the second path its own WireGuard node with its own private key, or by making both paths build the SAME device (same entry egress, same peers) — identical copies are merged automatically and cost no warning",
|
||||
c.node, c.keep.rep, c.lose.rep, c.keep.rep, wgKeepReason(c.keep, c.lose), c.lose.rep, wgLoseConsequence(c.lose, defaultRouteDead))
|
||||
}
|
||||
|
||||
// wgKeepReason says why this copy outranked the other, in terms of what depends on
|
||||
// it rather than in terms of the algorithm.
|
||||
func wgKeepReason(keep, lose wgClass) string {
|
||||
switch {
|
||||
case keep.weight.defaultRoute:
|
||||
return "it carries the DEFAULT ROUTE — the traffic of every client that has no more specific rule"
|
||||
case keep.weight.entryPoints > lose.weight.entryPoints:
|
||||
return fmt.Sprintf("more of the routing depends on it: %d entry points against %d", keep.weight.entryPoints, lose.weight.entryPoints)
|
||||
default:
|
||||
return "nothing in the config makes either copy more important than the other, so the first tag in alphabetical order was kept — that choice is arbitrary and you should make it deliberately"
|
||||
}
|
||||
}
|
||||
|
||||
// wgLoseConsequence names what actually stops working, in the words the operator
|
||||
// would use to describe the outage. Three cases, and they are three because
|
||||
// conflating them is how the previous text managed to be true and useless:
|
||||
//
|
||||
// - the default route went through the lost copy and has no other way out. This
|
||||
// is the live-router failure: say so in the loudest terms available.
|
||||
// - the default route went through it but survives (its group still has live
|
||||
// members). Saying "your default route is blocked" here would be a lie, and one
|
||||
// the operator would act on.
|
||||
// - the default route never touched it. Then it is worth saying so explicitly:
|
||||
// the operator's first question on reading this warning is whether the house is
|
||||
// about to go offline.
|
||||
//
|
||||
// defaultRouteDead is measured on the finished config (defaultRouteBlocked), not
|
||||
// guessed from the copy.
|
||||
func wgLoseConsequence(lose wgClass, defaultRouteDead bool) string {
|
||||
switch {
|
||||
case lose.weight.defaultRoute && defaultRouteDead:
|
||||
return "this BLOCKS YOUR DEFAULT ROUTE — every packet with no more specific rule is now dropped, which means the whole LAN loses the internet until you fix the config"
|
||||
case lose.weight.defaultRoute:
|
||||
return fmt.Sprintf("your default route ALSO went through this copy, but it still has another way out, so it keeps working; what stops is the %d routing entry point(s) that dialled through this copy specifically", lose.weight.entryPoints)
|
||||
default:
|
||||
return fmt.Sprintf("your default route is NOT affected; what stops is the %d routing entry point(s) that dialled through this copy", lose.weight.entryPoints)
|
||||
}
|
||||
}
|
||||
|
||||
// wgNodeNamesByKey maps a WireGuard private key back to the model node name that
|
||||
@@ -228,31 +506,108 @@ type optionNode struct {
|
||||
detour string // ordinary outbound/endpoint: its DialerOptions.Detour
|
||||
}
|
||||
|
||||
// reachableOptionTags is the option-layer twin of route.walkReachable
|
||||
// (route/reachability_lx.go): the set of outbound/endpoint tags traffic can
|
||||
// currently reach. That one walks live adapters inside a running box, which does
|
||||
// not exist yet at generate time, so the same question is answered over the
|
||||
// option structures.
|
||||
// tagWeight is "how much traffic depends on this tag", measured in the only two
|
||||
// terms the option layer can actually observe — and it exists because the previous
|
||||
// tie-break, "first in alphabetical order", once chose to block the default route
|
||||
// of a whole household in favour of a single ICMP rule.
|
||||
//
|
||||
// - defaultRoute: the tag is reachable from route.Final. Final is not merely one
|
||||
// more rule; it is where every packet that matched nothing else goes (buildRoute
|
||||
// points it at the catch-all model rule's target, or at block/direct per the
|
||||
// kill-switch). Losing it is losing the internet for everything unrouted, so it
|
||||
// dominates the comparison absolutely — no number of narrow rules outweighs it.
|
||||
// - entryPoints: how many distinct entry points into the outbound graph reach the
|
||||
// tag (route rules, DNS detours, rule-set download detours, the subscription
|
||||
// fetch detours). A proxy for breadth of use, used only to break a tie between
|
||||
// two copies that are both off the default route.
|
||||
//
|
||||
// A zero weight means unreachable: nothing routes to the tag at all.
|
||||
type tagWeight struct {
|
||||
defaultRoute bool
|
||||
entryPoints int
|
||||
}
|
||||
|
||||
// reachable reports whether anything can route to the tag at all.
|
||||
func (w tagWeight) reachable() bool { return w.entryPoints > 0 }
|
||||
|
||||
// heavierThan is the survivor comparison: default route first, breadth of use
|
||||
// second, and equal weights deliberately report false so the caller's stable tag
|
||||
// order decides — an arbitrary choice must at least be the SAME arbitrary choice
|
||||
// on every run.
|
||||
func (w tagWeight) heavierThan(o tagWeight) bool {
|
||||
if w.defaultRoute != o.defaultRoute {
|
||||
return w.defaultRoute
|
||||
}
|
||||
return w.entryPoints > o.entryPoints
|
||||
}
|
||||
|
||||
// merge combines the weights of two copies that turned out to be one device. The
|
||||
// entry-point counts are MAXed rather than summed: the two copies are usually
|
||||
// reached from overlapping sets of seeds, so summing would count the same rule
|
||||
// twice and inflate a class purely for having been written down more often.
|
||||
func (w tagWeight) merge(o tagWeight) tagWeight {
|
||||
return tagWeight{
|
||||
defaultRoute: w.defaultRoute || o.defaultRoute,
|
||||
entryPoints: max(w.entryPoints, o.entryPoints),
|
||||
}
|
||||
}
|
||||
|
||||
// optionTagWeights is the option-layer twin of route.walkReachable
|
||||
// (route/reachability_lx.go): which outbound/endpoint tags traffic can currently
|
||||
// reach, and — unlike that one — how much traffic. The runtime version walks live
|
||||
// adapters inside a running box, which does not exist yet at generate time, so the
|
||||
// same question is answered over the option structures.
|
||||
//
|
||||
// It walks once PER SEED rather than once in total, because the per-seed identity
|
||||
// is the whole point: "reachable from route.Final" is a different fact from
|
||||
// "reachable from some rule", and the pass has to be able to tell them apart.
|
||||
//
|
||||
// One deliberate difference from the runtime walk: a group contributes ALL of its
|
||||
// members, not just the one it would select right now. The runtime walk can ask a
|
||||
// selector what it has chosen; here nothing has been chosen yet, and treating the
|
||||
// unselected members as unreachable would delete an endpoint the group is free to
|
||||
// switch to a second later. Over-approximating is the safe direction — the worst it
|
||||
// costs is a duplicate reported as critical instead of being removed silently.
|
||||
//
|
||||
// One deliberate difference: a group contributes ALL of its members, not just the
|
||||
// one it would select right now. The runtime walk can ask a selector what it has
|
||||
// chosen; here nothing has been chosen yet, and treating the unselected members as
|
||||
// unreachable would delete an endpoint the group is free to switch to a second
|
||||
// later. Over-approximating is the safe direction — the worst it costs is a
|
||||
// duplicate reported as critical instead of being removed silently.
|
||||
// extraSeeds are references that exist OUTSIDE option.Options — today the
|
||||
// subscription fetch detours (see subscriptionDetourSeeds). They are entry points
|
||||
// exactly like a route rule, so they are walked identically.
|
||||
func reachableOptionTags(opts *option.Options, extraSeeds ...string) map[string]bool {
|
||||
// exactly like a route rule, so they are walked identically, and they are never the
|
||||
// default route.
|
||||
func optionTagWeights(opts *option.Options, extraSeeds ...string) map[string]tagWeight {
|
||||
graph := optionGraph(opts)
|
||||
reachable := map[string]bool{}
|
||||
for _, seed := range optionSeedTags(opts) {
|
||||
walkOptionReachable(seed, graph, reachable)
|
||||
seeds := optionSeeds(opts)
|
||||
for _, tag := range extraSeeds {
|
||||
seeds = append(seeds, optionSeed{tag: tag})
|
||||
}
|
||||
for _, seed := range extraSeeds {
|
||||
walkOptionReachable(seed, graph, reachable)
|
||||
|
||||
// Collapse repeated seed tags first, so the metric counts DISTINCT entry tags.
|
||||
// Five rules naming one chain are one way in, not five, and route.Final very
|
||||
// often equals some rule's target — without this collapse a copy would gain
|
||||
// "weight" merely by being written into more rules that all take the same path.
|
||||
isDefault := map[string]bool{}
|
||||
var order []string
|
||||
for _, seed := range seeds {
|
||||
if seed.tag == "" {
|
||||
continue
|
||||
}
|
||||
if _, seen := isDefault[seed.tag]; !seen {
|
||||
order = append(order, seed.tag)
|
||||
}
|
||||
isDefault[seed.tag] = isDefault[seed.tag] || seed.defaultRoute
|
||||
}
|
||||
return reachable
|
||||
|
||||
weights := map[string]tagWeight{}
|
||||
for _, seed := range order {
|
||||
reached := map[string]bool{}
|
||||
walkOptionReachable(seed, graph, reached)
|
||||
for tag := range reached {
|
||||
w := weights[tag]
|
||||
w.entryPoints++
|
||||
w.defaultRoute = w.defaultRoute || isDefault[seed]
|
||||
weights[tag] = w
|
||||
}
|
||||
}
|
||||
return weights
|
||||
}
|
||||
|
||||
// subscriptionDetourSeeds returns the outbound tags the SUBSCRIPTION fetcher dials
|
||||
@@ -353,37 +708,48 @@ func optionGraph(opts *option.Options) map[string]optionNode {
|
||||
return graph
|
||||
}
|
||||
|
||||
// optionSeedTags collects every tag traffic can ENTER the outbound graph at: the
|
||||
// route's Final (the kill-switch backstop), each route rule's routed/bypassed
|
||||
// outbound, each DNS server's detour, and the download detours of the remote
|
||||
// rule-sets / geo sources / clash external UI (those dial through an outbound too,
|
||||
// so a tag one of them names is in use even if no user traffic reaches it).
|
||||
func optionSeedTags(opts *option.Options) []string {
|
||||
var seeds []string
|
||||
// optionSeed is one entry point into the outbound graph, carrying whether it is
|
||||
// THE default route. Only route.Final is: buildRoute points it at the catch-all
|
||||
// model rule's target (or at block/direct per the kill-switch), so it is where every
|
||||
// packet that matched no rule ends up.
|
||||
type optionSeed struct {
|
||||
tag string
|
||||
defaultRoute bool
|
||||
}
|
||||
|
||||
// optionSeeds collects every tag traffic can ENTER the outbound graph at: the
|
||||
// route's Final (the kill-switch backstop / default route), each route rule's
|
||||
// routed/bypassed outbound, each DNS server's detour, and the download detours of
|
||||
// the remote rule-sets / geo sources / clash external UI (those dial through an
|
||||
// outbound too, so a tag one of them names is in use even if no user traffic
|
||||
// reaches it).
|
||||
func optionSeeds(opts *option.Options) []optionSeed {
|
||||
var seeds []optionSeed
|
||||
add := func(tag string) { seeds = append(seeds, optionSeed{tag: tag}) }
|
||||
if rt := opts.Route; rt != nil {
|
||||
seeds = append(seeds, rt.Final)
|
||||
seeds = append(seeds, optionSeed{tag: rt.Final, defaultRoute: true})
|
||||
for i := range rt.Rules {
|
||||
seeds = append(seeds, ruleActionOutbound(rt.Rules[i]))
|
||||
add(ruleActionOutbound(rt.Rules[i]))
|
||||
}
|
||||
for i := range rt.RuleSet {
|
||||
if rt.RuleSet[i].Type == C.RuleSetTypeRemote {
|
||||
seeds = append(seeds, rt.RuleSet[i].RemoteOptions.DownloadDetour)
|
||||
add(rt.RuleSet[i].RemoteOptions.DownloadDetour)
|
||||
}
|
||||
}
|
||||
if rt.GeoIP != nil {
|
||||
seeds = append(seeds, rt.GeoIP.DownloadDetour)
|
||||
add(rt.GeoIP.DownloadDetour)
|
||||
}
|
||||
if rt.Geosite != nil {
|
||||
seeds = append(seeds, rt.Geosite.DownloadDetour)
|
||||
add(rt.Geosite.DownloadDetour)
|
||||
}
|
||||
}
|
||||
if opts.DNS != nil {
|
||||
for i := range opts.DNS.Servers {
|
||||
seeds = append(seeds, dialerDetour(opts.DNS.Servers[i].Options))
|
||||
add(dialerDetour(opts.DNS.Servers[i].Options))
|
||||
}
|
||||
}
|
||||
if opts.Experimental != nil && opts.Experimental.ClashAPI != nil {
|
||||
seeds = append(seeds, opts.Experimental.ClashAPI.ExternalUIDownloadDetour)
|
||||
add(opts.Experimental.ClashAPI.ExternalUIDownloadDetour)
|
||||
}
|
||||
return seeds
|
||||
}
|
||||
@@ -451,117 +817,135 @@ func groupMembersOf(o any) (members []string, def string, ok bool) {
|
||||
return nil, "", false
|
||||
}
|
||||
|
||||
// retargetDroppedTags repoints every reference to a dropped endpoint tag, so the
|
||||
// config stays internally consistent once the endpoint is gone. A dangling tag is
|
||||
// not merely untidy: box.New resolves group members and route targets eagerly and
|
||||
// refuses the whole config over one of them, which with the kill-switch closed is
|
||||
// the entire LAN offline.
|
||||
// remapTags repoints every reference to a removed endpoint tag, so the config stays
|
||||
// internally consistent once the endpoint is gone. A dangling tag is not merely
|
||||
// untidy: box.New resolves group members and route targets eagerly and refuses the
|
||||
// whole config over one of them, which with the kill-switch closed is the entire LAN
|
||||
// offline.
|
||||
//
|
||||
// Detours, route targets and the route Final go to `to` (block) — the fail-closed
|
||||
// direction. Group MEMBER lists instead drop the tag and only fall back to a lone
|
||||
// block member when that empties the group: a group that still has other tunnels
|
||||
// to balance over should use them, and a block sitting in a urltest pool would
|
||||
// otherwise be probed and reported dead forever on the group health card.
|
||||
func retargetDroppedTags(opts *option.Options, drop map[string]bool, to string) {
|
||||
// `replace` maps a removed tag to what its references become. Two kinds of value,
|
||||
// and the difference is the whole point of this pass:
|
||||
//
|
||||
// - a SURVIVOR tag — the removed copy and the survivor build the same device, so
|
||||
// the reference keeps working exactly as written;
|
||||
// - tagBlock — the reference has genuinely lost its tunnel and must fail closed,
|
||||
// never fall back to `direct` and out the plain WAN with the router's real
|
||||
// address.
|
||||
//
|
||||
// Group MEMBER lists are the one place the two are not interchangeable: a merged
|
||||
// member is REPLACED by the survivor (still a working tunnel to balance over, and
|
||||
// de-duplicated so the survivor cannot appear twice in one pool), while a blocked
|
||||
// member is REMOVED, falling back to a lone block member only if that empties the
|
||||
// group. A block sitting in a urltest pool would otherwise be probed and reported
|
||||
// dead forever on the group health card.
|
||||
func remapTags(opts *option.Options, replace map[string]string) {
|
||||
for i := range opts.Outbounds {
|
||||
if !retargetGroupMembers(opts.Outbounds[i].Options, drop, to) {
|
||||
retargetDetour(opts.Outbounds[i].Options, drop, to)
|
||||
if !remapGroupMembers(opts.Outbounds[i].Options, replace) {
|
||||
remapDetour(opts.Outbounds[i].Options, replace)
|
||||
}
|
||||
}
|
||||
for i := range opts.Endpoints {
|
||||
retargetDetour(opts.Endpoints[i].Options, drop, to)
|
||||
remapDetour(opts.Endpoints[i].Options, replace)
|
||||
}
|
||||
if opts.DNS != nil {
|
||||
for i := range opts.DNS.Servers {
|
||||
retargetDetour(opts.DNS.Servers[i].Options, drop, to)
|
||||
remapDetour(opts.DNS.Servers[i].Options, replace)
|
||||
}
|
||||
}
|
||||
if rt := opts.Route; rt != nil {
|
||||
if drop[rt.Final] {
|
||||
rt.Final = to
|
||||
}
|
||||
remapField(&rt.Final, replace)
|
||||
for i := range rt.Rules {
|
||||
retargetRuleAction(&rt.Rules[i], drop, to)
|
||||
remapRuleAction(&rt.Rules[i], replace)
|
||||
}
|
||||
for i := range rt.RuleSet {
|
||||
if rt.RuleSet[i].Type == C.RuleSetTypeRemote && drop[rt.RuleSet[i].RemoteOptions.DownloadDetour] {
|
||||
rt.RuleSet[i].RemoteOptions.DownloadDetour = to
|
||||
if rt.RuleSet[i].Type == C.RuleSetTypeRemote {
|
||||
remapField(&rt.RuleSet[i].RemoteOptions.DownloadDetour, replace)
|
||||
}
|
||||
}
|
||||
if rt.GeoIP != nil && drop[rt.GeoIP.DownloadDetour] {
|
||||
rt.GeoIP.DownloadDetour = to
|
||||
if rt.GeoIP != nil {
|
||||
remapField(&rt.GeoIP.DownloadDetour, replace)
|
||||
}
|
||||
if rt.Geosite != nil && drop[rt.Geosite.DownloadDetour] {
|
||||
rt.Geosite.DownloadDetour = to
|
||||
if rt.Geosite != nil {
|
||||
remapField(&rt.Geosite.DownloadDetour, replace)
|
||||
}
|
||||
}
|
||||
if opts.Experimental != nil && opts.Experimental.ClashAPI != nil &&
|
||||
drop[opts.Experimental.ClashAPI.ExternalUIDownloadDetour] {
|
||||
opts.Experimental.ClashAPI.ExternalUIDownloadDetour = to
|
||||
if opts.Experimental != nil && opts.Experimental.ClashAPI != nil {
|
||||
remapField(&opts.Experimental.ClashAPI.ExternalUIDownloadDetour, replace)
|
||||
}
|
||||
}
|
||||
|
||||
// retargetRuleAction points a route rule whose target was dropped at `to`. It is
|
||||
// the write half of ruleActionOutbound and must stay in step with it: a rule the
|
||||
// seed walk counted as a reference is a rule this has to be able to repoint.
|
||||
func retargetRuleAction(rule *option.Rule, drop map[string]bool, to string) {
|
||||
// remapField rewrites one tag-valued field in place, leaving anything not in the
|
||||
// map alone.
|
||||
func remapField(field *string, replace map[string]string) {
|
||||
if to, ok := replace[*field]; ok {
|
||||
*field = to
|
||||
}
|
||||
}
|
||||
|
||||
// remapRuleAction points a route rule whose target was removed at its replacement.
|
||||
// It is the write half of ruleActionOutbound and must stay in step with it: a rule
|
||||
// the seed walk counted as a reference is a rule this has to be able to repoint.
|
||||
func remapRuleAction(rule *option.Rule, replace map[string]string) {
|
||||
action := &rule.DefaultOptions.RuleAction
|
||||
if rule.Type == C.RuleTypeLogical {
|
||||
action = &rule.LogicalOptions.RuleAction
|
||||
}
|
||||
switch action.Action {
|
||||
case C.RuleActionTypeRoute, "":
|
||||
if drop[action.RouteOptions.Outbound] {
|
||||
action.RouteOptions.Outbound = to
|
||||
}
|
||||
remapField(&action.RouteOptions.Outbound, replace)
|
||||
case C.RuleActionTypeBypass:
|
||||
if drop[action.BypassOptions.Outbound] {
|
||||
action.BypassOptions.Outbound = to
|
||||
}
|
||||
remapField(&action.BypassOptions.Outbound, replace)
|
||||
}
|
||||
}
|
||||
|
||||
// retargetDetour points a dropped Detour at `to`, leaving any other value alone.
|
||||
func retargetDetour(o any, drop map[string]bool, to string) {
|
||||
// remapDetour rewrites a removed Detour, leaving any other value alone.
|
||||
func remapDetour(o any, replace map[string]string) {
|
||||
w, ok := o.(option.DialerOptionsWrapper)
|
||||
if !ok {
|
||||
return
|
||||
}
|
||||
d := w.TakeDialerOptions()
|
||||
if !drop[d.Detour] {
|
||||
to, hit := replace[d.Detour]
|
||||
if !hit {
|
||||
return
|
||||
}
|
||||
d.Detour = to
|
||||
w.ReplaceDialerOptions(d)
|
||||
}
|
||||
|
||||
// retargetGroupMembers prunes dropped members from a group, reporting whether the
|
||||
// options were a group at all (so the caller knows not to look for a detour).
|
||||
// A selector whose Default was dropped falls back to an empty Default, which the
|
||||
// engine reads as "the first member" — always a member that still exists.
|
||||
func retargetGroupMembers(o any, drop map[string]bool, to string) bool {
|
||||
// remapGroupMembers rewrites a group's member list, reporting whether the options
|
||||
// were a group at all (so the caller knows not to look for a detour). A selector
|
||||
// whose Default was BLOCKED falls back to an empty Default, which the engine reads
|
||||
// as "the first member" — always a member that still exists; a Default that was
|
||||
// merely merged follows its survivor and keeps the operator's choice.
|
||||
func remapGroupMembers(o any, replace map[string]string) bool {
|
||||
switch g := o.(type) {
|
||||
case *option.SelectorOutboundOptions:
|
||||
g.Outbounds = pruneMembers(g.Outbounds, drop, to)
|
||||
if drop[g.Default] {
|
||||
g.Default = ""
|
||||
g.Outbounds = remapMembers(g.Outbounds, replace)
|
||||
if to, ok := replace[g.Default]; ok {
|
||||
if to == tagBlock {
|
||||
g.Default = ""
|
||||
} else {
|
||||
g.Default = to
|
||||
}
|
||||
}
|
||||
return true
|
||||
case *option.URLTestOutboundOptions:
|
||||
g.Outbounds = pruneMembers(g.Outbounds, drop, to)
|
||||
g.Outbounds = remapMembers(g.Outbounds, replace)
|
||||
return true
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
// pruneMembers removes dropped tags from a member list, substituting a single
|
||||
// fallback member when that would leave the group empty (an empty selector/urltest
|
||||
// is refused by box.New, and the fallback is `block` so the emptied group is
|
||||
// fail-closed rather than a hole).
|
||||
func pruneMembers(in []string, drop map[string]bool, fallback string) []string {
|
||||
// remapMembers rewrites a member list: a merged member becomes its survivor (kept
|
||||
// once, however many members map onto it), a blocked member is dropped, and a list
|
||||
// left empty falls back to a single `block` member — an empty selector/urltest is
|
||||
// refused by box.New, and block makes the emptied group fail-closed rather than a
|
||||
// hole.
|
||||
func remapMembers(in []string, replace map[string]string) []string {
|
||||
hit := false
|
||||
for _, tag := range in {
|
||||
if drop[tag] {
|
||||
if _, ok := replace[tag]; ok {
|
||||
hit = true
|
||||
break
|
||||
}
|
||||
@@ -570,24 +954,33 @@ func pruneMembers(in []string, drop map[string]bool, fallback string) []string {
|
||||
return in
|
||||
}
|
||||
out := make([]string, 0, len(in))
|
||||
seen := make(map[string]bool, len(in))
|
||||
for _, tag := range in {
|
||||
if drop[tag] {
|
||||
if to, ok := replace[tag]; ok {
|
||||
if to == tagBlock {
|
||||
continue
|
||||
}
|
||||
tag = to
|
||||
}
|
||||
if seen[tag] {
|
||||
continue
|
||||
}
|
||||
seen[tag] = true
|
||||
out = append(out, tag)
|
||||
}
|
||||
if len(out) == 0 {
|
||||
return []string{fallback}
|
||||
return []string{tagBlock}
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
// removeDroppedEndpoints filters the dropped endpoints out of the config. This is
|
||||
// the step that actually stops the duplicate device from being created.
|
||||
func removeDroppedEndpoints(opts *option.Options, drop map[string]bool) {
|
||||
// removeRemappedEndpoints filters the remapped endpoints out of the config. This is
|
||||
// the step that actually stops the duplicate device from being created: box.New
|
||||
// starts every endpoint present, referenced or not.
|
||||
func removeRemappedEndpoints(opts *option.Options, replace map[string]string) {
|
||||
kept := opts.Endpoints[:0]
|
||||
for _, ep := range opts.Endpoints {
|
||||
if drop[ep.Tag] {
|
||||
if _, ok := replace[ep.Tag]; ok {
|
||||
continue
|
||||
}
|
||||
kept = append(kept, ep)
|
||||
|
||||
@@ -311,27 +311,33 @@ func TestWGDedupFetchDetourAbsentBaseStillDropped(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
// Case 2 — node in the chain AND named as a fetch detour: the config asks for the
|
||||
// node on two different dial paths, which one private key cannot provide. Both are
|
||||
// reachable, so one survives and the other is fail-closed with the critical
|
||||
// warning. Choosing silently for the operator is what we must NOT do.
|
||||
// Case 2 — node in the chain AND named as a fetch detour. Both copies are
|
||||
// reachable, but the chain has no entry egress, so the chain hop and the base
|
||||
// endpoint dial IDENTICALLY: one device, written twice. They are merged, and the
|
||||
// survivor must be the BASE tag `wg1`.
|
||||
//
|
||||
// That last part is not cosmetic. The fetch detour is the one reference this pass
|
||||
// cannot rewrite: `fetch_detour=node:wg1` lives in the model, and
|
||||
// apply.UpdateSubscription resolves it against the RUNNING box (see wgPinnedTags).
|
||||
// Merging into the chain tag would have deleted `wg1` and broken the subscription
|
||||
// update with "unknown outbound tag" — the very regression subscriptionDetourSeeds
|
||||
// was added to prevent, re-entering through the merge.
|
||||
//
|
||||
// This case USED to be reported as a conflict and fail-closed. It never was one.
|
||||
func TestWGDedupFetchDetourAndChainBothReachable(t *testing.T) {
|
||||
opts, warns, err := GenerateWithWarnings(fetchDetourModel(true, "node:wg1"))
|
||||
if err != nil {
|
||||
t.Fatalf("Generate: %v", err)
|
||||
}
|
||||
// Tag order decides: "chain-c-h1" < "wg1", so the chain copy is the survivor.
|
||||
if got := endpointTags(opts); len(got) != 1 || got[0] != "chain-c-h1" {
|
||||
t.Fatalf("endpoints = %v, want exactly [chain-c-h1] (one device per key)", got)
|
||||
if got := endpointTags(opts); len(got) != 1 || got[0] != "wg1" {
|
||||
t.Fatalf("endpoints = %v, want exactly [wg1]: one device, and it must keep the tag the subscription fetch detour names", got)
|
||||
}
|
||||
found := warningsContaining(warns, "materialised twice")
|
||||
if len(found) != 1 {
|
||||
t.Fatalf("want exactly one duplicate warning, got %v (all: %v)", found, warns)
|
||||
// The chain still works — its L2 now detours through the survivor.
|
||||
if d := anyDetour(t, opts, "chain-c-h2"); d != "wg1" {
|
||||
t.Fatalf("chain-c-h2 detour = %q, want wg1 (the merged survivor)", d)
|
||||
}
|
||||
for _, want := range []string{`node "wg1"`, "chain-c-h1", `"wg1"`, "fail-closed"} {
|
||||
if !strings.Contains(found[0], want) {
|
||||
t.Fatalf("warning must contain %q, got: %s", want, found[0])
|
||||
}
|
||||
if got := warningsContaining(warns, "materialised twice"); len(got) != 0 {
|
||||
t.Fatalf("the chain hop and the base endpoint dial identically — merging is silent, got %v", got)
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -0,0 +1,470 @@
|
||||
package generate
|
||||
|
||||
import (
|
||||
"strings"
|
||||
"testing"
|
||||
|
||||
C "github.com/sagernet/sing-box/constant"
|
||||
"github.com/sagernet/sing-box/option"
|
||||
|
||||
"github.com/sagernet/sing-box/shater/model"
|
||||
)
|
||||
|
||||
// These tests pin the SECOND half of the WireGuard de-duplication contract: two
|
||||
// copies of a key are only a conflict when they would build two DIFFERENT devices.
|
||||
// When they would build the same one, they are merged and both paths keep working.
|
||||
//
|
||||
// The first half (one key = one live device) is pinned in wgdedup_test.go. This file
|
||||
// exists because that half, on its own, once took the whole house off the internet:
|
||||
// the operator added a narrow rule pointing at a WireGuard node that was already the
|
||||
// first hop of the default-route chain, and the pass resolved the "conflict" by
|
||||
// blocking the default route. There was no conflict — both copies dialled out over
|
||||
// the same egress and were the same device.
|
||||
//
|
||||
// Not Linux-gated, for the same reason as wgdedup_test.go: nothing here reaches
|
||||
// box.New, so the routing_mark platform restriction does not apply.
|
||||
|
||||
// wgMergeEndpointCount counts the wireguard endpoints in a config — the number of
|
||||
// DEVICES the engine will bring up, which is the quantity the whole pass is about.
|
||||
func wgMergeEndpointCount(opts option.Options) int {
|
||||
n := 0
|
||||
for i := range opts.Endpoints {
|
||||
if opts.Endpoints[i].Type == C.TypeWireGuard {
|
||||
n++
|
||||
}
|
||||
}
|
||||
return n
|
||||
}
|
||||
|
||||
// wgMergeRuleTarget returns the outbound the single non-catch-all rule routes to.
|
||||
// It fails when there is not exactly one, so a test can never silently assert about
|
||||
// the wrong rule.
|
||||
func wgMergeRuleTarget(t *testing.T, opts option.Options) string {
|
||||
t.Helper()
|
||||
got := routeRuleOutbounds(opts.Route)
|
||||
if len(got) != 1 {
|
||||
t.Fatalf("expected exactly one routing rule, got %v", got)
|
||||
}
|
||||
return got[0]
|
||||
}
|
||||
|
||||
// groupMembersByTag returns a group outbound's member list.
|
||||
func groupMembersByTag(t *testing.T, opts option.Options, tag string) []string {
|
||||
t.Helper()
|
||||
ob := obByTag(opts, tag)
|
||||
if ob == nil {
|
||||
t.Fatalf("no outbound tagged %q; outbounds=%v", tag, outboundTags(opts))
|
||||
}
|
||||
members, _, ok := groupMembersOf(ob.Options)
|
||||
if !ok {
|
||||
t.Fatalf("outbound %q is not a group (%T)", tag, ob.Options)
|
||||
}
|
||||
return members
|
||||
}
|
||||
|
||||
// --- the case that broke the live router ------------------------------------
|
||||
|
||||
// The exact shape the owner wanted and could not have: one WireGuard node used
|
||||
// BOTH as the first hop of the default-route chain and as the whole of a second,
|
||||
// one-rule chain, both entering over the same egress.
|
||||
//
|
||||
// chain:ewan-wg-subs = egress:ewan -> node:awgout -> node:sub1 (default route)
|
||||
// chain:ewan-wg = egress:ewan -> node:awgout (narrow terminal)
|
||||
//
|
||||
// Both copies of awgout carry the same key, the same peer and the same detour, so
|
||||
// they are ONE device written twice. The pass must merge them: one endpoint, both
|
||||
// chains alive, every reference pointing at the survivor, and not a word of warning
|
||||
// — there is nothing wrong with this config.
|
||||
//
|
||||
// Before the merge existed this produced "materialised twice" and fail-closed the
|
||||
// default route: the whole LAN lost the internet because of one narrow rule.
|
||||
func TestWGMergeIdenticalCopiesShareOneDevice(t *testing.T) {
|
||||
priv, pub := wgDedupKey(41), wgDedupKey(51)
|
||||
m := &model.Model{
|
||||
Globals: model.DefaultGlobals(),
|
||||
Nodes: []model.Node{
|
||||
{Name: "awgout", Enabled: true, URI: wgDedupURI(priv, pub, "203.0.113.10", 51820)},
|
||||
{Name: "sub1", Enabled: true, URI: "ss://aes-256-gcm:secret@203.0.113.2:8388#sub1"},
|
||||
},
|
||||
Egresses: []model.Egress{{Name: "ewan", Type: "interface", Interface: "eth0"}},
|
||||
Chains: []model.Chain{
|
||||
{Name: "ewan-wg-subs", Hops: []string{"egress:ewan", "node:awgout", "node:sub1"}},
|
||||
{Name: "ewan-wg", Hops: []string{"egress:ewan", "node:awgout"}},
|
||||
},
|
||||
Rules: []model.Rule{
|
||||
// The narrow rule. On the live router this was an ICMP terminal; the port
|
||||
// matcher stands in for it so the test is about the de-duplication and not
|
||||
// about what the engine can match at layer 3.
|
||||
{Name: "terminal", Enabled: true, Order: 10, DstPort: "443", Target: "chain:ewan-wg"},
|
||||
// Matcher-less => the default route for everything else.
|
||||
{Name: "default", Enabled: true, Order: 100, Target: "chain:ewan-wg-subs"},
|
||||
},
|
||||
}
|
||||
|
||||
opts, warns, err := GenerateWithWarnings(m)
|
||||
if err != nil {
|
||||
t.Fatalf("Generate: %v", err)
|
||||
}
|
||||
|
||||
// ONE device, and it is the copy the default route depends on.
|
||||
if n := wgMergeEndpointCount(opts); n != 1 {
|
||||
t.Fatalf("wireguard endpoints = %d (%v), want exactly 1 device", n, endpointTags(opts))
|
||||
}
|
||||
const survivor = "chain-ewan-wg-subs-h1"
|
||||
if got := endpointTags(opts); len(got) != 1 || got[0] != survivor {
|
||||
t.Fatalf("endpoints = %v, want [%s]", got, survivor)
|
||||
}
|
||||
// It is the RIGHT device: still entering over the chain's egress.
|
||||
if d := anyDetour(t, opts, survivor); d != "egress-ewan" {
|
||||
t.Fatalf("%s detour = %q, want egress-ewan", survivor, d)
|
||||
}
|
||||
|
||||
// Both paths survive, and both now lead to the one device.
|
||||
if got := wgMergeRuleTarget(t, opts); got != survivor {
|
||||
t.Fatalf("the narrow rule routes to %q, want the survivor %q — a merged copy must be repointed, not blocked", got, survivor)
|
||||
}
|
||||
if opts.Route.Final != "chain-ewan-wg-subs-h2" {
|
||||
t.Fatalf("route.Final = %q, want chain-ewan-wg-subs-h2 (the default-route chain's exit)", opts.Route.Final)
|
||||
}
|
||||
if d := anyDetour(t, opts, "chain-ewan-wg-subs-h2"); d != survivor {
|
||||
t.Fatalf("chain-ewan-wg-subs-h2 detour = %q, want %q", d, survivor)
|
||||
}
|
||||
|
||||
// Nothing was fail-closed, and nothing was reported: the config is fine.
|
||||
if opts.Route.Final == tagBlock {
|
||||
t.Fatalf("the default route was blocked; route=%+v", opts.Route.Final)
|
||||
}
|
||||
if got := warningsContaining(warns, "materialised twice"); len(got) != 0 {
|
||||
t.Fatalf("identical copies are one device and must merge silently, got %v", got)
|
||||
}
|
||||
}
|
||||
|
||||
// --- a REAL incompatibility still fails closed, and says what it costs -------
|
||||
|
||||
// Two chains entering over two DIFFERENT egresses bind the same key to two
|
||||
// different uplinks. That IS two devices, and one has to go.
|
||||
//
|
||||
// The tags are named so that ALPHABETICAL order and IMPORTANCE disagree:
|
||||
// "chain-aaa-narrow-h1" sorts first but carries one port rule, while
|
||||
// "chain-zzz-default-h1" sorts last and carries the default route. The old pass
|
||||
// kept the first tag and blocked the default route — the failure this test exists
|
||||
// to make impossible.
|
||||
func TestWGMergeIncompatibleKeepsTheDefaultRoute(t *testing.T) {
|
||||
priv, pub := wgDedupKey(61), wgDedupKey(71)
|
||||
m := &model.Model{
|
||||
Globals: model.DefaultGlobals(),
|
||||
Nodes: []model.Node{{Name: "awgout", Enabled: true, URI: wgDedupURI(priv, pub, "203.0.113.10", 51820)}},
|
||||
Egresses: []model.Egress{
|
||||
{Name: "ewan", Type: "interface", Interface: "eth0"},
|
||||
{Name: "lte", Type: "interface", Interface: "eth1"},
|
||||
},
|
||||
Chains: []model.Chain{
|
||||
{Name: "aaa-narrow", Hops: []string{"egress:lte", "node:awgout"}},
|
||||
{Name: "zzz-default", Hops: []string{"egress:ewan", "node:awgout"}},
|
||||
},
|
||||
Rules: []model.Rule{
|
||||
{Name: "narrow", Enabled: true, Order: 10, DstPort: "443", Target: "chain:aaa-narrow"},
|
||||
{Name: "default", Enabled: true, Order: 100, Target: "chain:zzz-default"},
|
||||
},
|
||||
}
|
||||
|
||||
opts, warns, err := GenerateWithWarnings(m)
|
||||
if err != nil {
|
||||
t.Fatalf("Generate: %v", err)
|
||||
}
|
||||
|
||||
const survivor = "chain-zzz-default-h1"
|
||||
if got := endpointTags(opts); len(got) != 1 || got[0] != survivor {
|
||||
t.Fatalf("endpoints = %v, want [%s]: the copy the DEFAULT ROUTE depends on must win, not the first tag alphabetically", got, survivor)
|
||||
}
|
||||
// Identity check, not just a name check: the survivor must be the copy bound to
|
||||
// the default route's egress.
|
||||
if d := anyDetour(t, opts, survivor); d != "egress-ewan" {
|
||||
t.Fatalf("%s detour = %q, want egress-ewan — the wrong copy survived under the right tag", survivor, d)
|
||||
}
|
||||
if opts.Route.Final != survivor {
|
||||
t.Fatalf("route.Final = %q, want %q — the default route must still reach its tunnel", opts.Route.Final, survivor)
|
||||
}
|
||||
// The loser is fail-closed, never dropped to direct.
|
||||
if got := wgMergeRuleTarget(t, opts); got != tagBlock {
|
||||
t.Fatalf("the losing rule routes to %q, want %q (fail-closed, not out the plain WAN)", got, tagBlock)
|
||||
}
|
||||
|
||||
found := warningsContaining(warns, "materialised twice")
|
||||
if len(found) != 1 {
|
||||
t.Fatalf("want exactly one conflict warning, got %v (all: %v)", found, warns)
|
||||
}
|
||||
w := found[0]
|
||||
for _, want := range []string{`node "awgout"`, survivor, "chain-aaa-narrow-h1", "fail-closed"} {
|
||||
if !strings.Contains(w, want) {
|
||||
t.Fatalf("warning must contain %q, got: %s", want, w)
|
||||
}
|
||||
}
|
||||
// The point of the rewrite: the operator must be able to read the CONSEQUENCE,
|
||||
// not just the mechanism. The old text was true and useless.
|
||||
lower := strings.ToLower(w)
|
||||
if !strings.Contains(lower, "default route") {
|
||||
t.Fatalf("warning never mentions the default route, which is exactly what the operator needs to weigh: %s", w)
|
||||
}
|
||||
if !strings.Contains(lower, "it carries the default route") {
|
||||
t.Fatalf("warning must say WHY this copy was kept (that it carries the default route), got: %s", w)
|
||||
}
|
||||
if !strings.Contains(lower, "your default route is not affected") {
|
||||
t.Fatalf("warning must say what the loss costs — here, that the default route survives: %s", w)
|
||||
}
|
||||
}
|
||||
|
||||
// --- regression: no duplication at all, nothing touched ---------------------
|
||||
|
||||
// Two DIFFERENT WireGuard nodes, each materialised exactly once, alongside a group
|
||||
// that is the default route. No key is duplicated, so the pass must be a no-op: no
|
||||
// endpoint removed, no member list rewritten, no rule repointed, no warning. This
|
||||
// guards the fast path — an over-eager merge that keyed off something weaker than
|
||||
// the private key would collapse these two into one device and silently reroute
|
||||
// half the config.
|
||||
func TestWGMergeNoDuplicateLeavesConfigAlone(t *testing.T) {
|
||||
m := &model.Model{
|
||||
Globals: model.DefaultGlobals(),
|
||||
Nodes: []model.Node{
|
||||
{Name: "wgpool", Enabled: true, URI: wgDedupURI(wgDedupKey(81), wgDedupKey(91), "203.0.113.20", 51820)},
|
||||
{Name: "wgdirect", Enabled: true, URI: wgDedupURI(wgDedupKey(82), wgDedupKey(92), "203.0.113.21", 51820)},
|
||||
{Name: "ss1", Enabled: true, URI: "ss://aes-256-gcm:secret@203.0.113.2:8388#ss1"},
|
||||
},
|
||||
Groups: []model.Group{
|
||||
{Name: "pool", Source: "manual", Nodes: []string{"wgpool", "ss1"}},
|
||||
},
|
||||
Rules: []model.Rule{
|
||||
{Name: "to-node", Enabled: true, Order: 10, DstPort: "443", Target: "node:wgdirect"},
|
||||
{Name: "default", Enabled: true, Order: 100, Target: "group:pool"},
|
||||
},
|
||||
}
|
||||
|
||||
opts, warns, err := GenerateWithWarnings(m)
|
||||
if err != nil {
|
||||
t.Fatalf("Generate: %v", err)
|
||||
}
|
||||
|
||||
if got := endpointTags(opts); len(got) != 2 || got[0] != "wgdirect" || got[1] != "wgpool" {
|
||||
t.Fatalf("endpoints = %v, want [wgdirect wgpool] — nothing to de-duplicate, nothing to remove", got)
|
||||
}
|
||||
if got := groupMembersByTag(t, opts, "pool"); len(got) != 2 || got[0] != "wgpool" || got[1] != "ss1" {
|
||||
t.Fatalf("group members = %v, want [wgpool ss1] untouched", got)
|
||||
}
|
||||
if got := wgMergeRuleTarget(t, opts); got != "wgdirect" {
|
||||
t.Fatalf("rule target = %q, want wgdirect untouched", got)
|
||||
}
|
||||
if opts.Route.Final != "pool" {
|
||||
t.Fatalf("route.Final = %q, want pool untouched", opts.Route.Final)
|
||||
}
|
||||
if got := warningsContaining(warns, "materialised twice"); len(got) != 0 {
|
||||
t.Fatalf("no key is duplicated, so there must be no warning; got %v", got)
|
||||
}
|
||||
}
|
||||
|
||||
// --- a merge inside a group must not put the survivor in the pool twice ------
|
||||
|
||||
// Two model nodes carrying the SAME WireGuard config — the shape a subscription
|
||||
// that lists one server twice produces — are one device, and both are members of
|
||||
// the same balancing pool. After the merge the pool must name the survivor ONCE:
|
||||
// a tag repeated in a selector/urltest list is a member the engine probes and
|
||||
// balances over twice, i.e. a silent weighting of one server against the rest.
|
||||
func TestWGMergeGroupMemberCollapsesWithoutDuplicate(t *testing.T) {
|
||||
uri := wgDedupURI(wgDedupKey(101), wgDedupKey(111), "203.0.113.30", 51820)
|
||||
m := &model.Model{
|
||||
Globals: model.DefaultGlobals(),
|
||||
Nodes: []model.Node{
|
||||
{Name: "wgA", Enabled: true, URI: uri},
|
||||
{Name: "wgB", Enabled: true, URI: uri},
|
||||
{Name: "ss1", Enabled: true, URI: "ss://aes-256-gcm:secret@203.0.113.2:8388#ss1"},
|
||||
},
|
||||
Groups: []model.Group{
|
||||
{Name: "pool", Source: "manual", Nodes: []string{"wgA", "wgB", "ss1"}},
|
||||
},
|
||||
Rules: []model.Rule{
|
||||
{Name: "default", Enabled: true, Order: 100, Target: "group:pool"},
|
||||
},
|
||||
}
|
||||
|
||||
opts, warns, err := GenerateWithWarnings(m)
|
||||
if err != nil {
|
||||
t.Fatalf("Generate: %v", err)
|
||||
}
|
||||
|
||||
if got := endpointTags(opts); len(got) != 1 || got[0] != "wgA" {
|
||||
t.Fatalf("endpoints = %v, want [wgA]: one config written twice is one device", got)
|
||||
}
|
||||
got := groupMembersByTag(t, opts, "pool")
|
||||
if len(got) != 2 || got[0] != "wgA" || got[1] != "ss1" {
|
||||
t.Fatalf("group members = %v, want [wgA ss1] — the merged member must be folded into the survivor exactly once", got)
|
||||
}
|
||||
for _, member := range got {
|
||||
if member == tagBlock {
|
||||
t.Fatalf("group members = %v: a MERGED member must be repointed, never blocked", got)
|
||||
}
|
||||
}
|
||||
if got := warningsContaining(warns, "materialised twice"); len(got) != 0 {
|
||||
t.Fatalf("identical members merge silently, got %v", got)
|
||||
}
|
||||
}
|
||||
|
||||
// --- the warning must not overstate the damage either ------------------------
|
||||
|
||||
// Two INCOMPATIBLE copies of one key (different egress bindings, so different
|
||||
// devices) that are both members of the group serving as the default route. One is
|
||||
// blocked — but the pool keeps its other members, so the default route does NOT go
|
||||
// down.
|
||||
//
|
||||
// The warning has to say that. "This blocks your default route" is exactly as
|
||||
// harmful when it is a false alarm as when it is missing: an operator who is told
|
||||
// the house is offline will go and change something that was working.
|
||||
func TestWGMergeConflictInsideDefaultGroupDoesNotClaimOutage(t *testing.T) {
|
||||
priv, pub := wgDedupKey(121), wgDedupKey(131)
|
||||
uri := wgDedupURI(priv, pub, "203.0.113.40", 51820)
|
||||
m := &model.Model{
|
||||
Globals: model.DefaultGlobals(),
|
||||
Nodes: []model.Node{
|
||||
// Same key and same peer — one device identity — but bound to two
|
||||
// different uplinks, which is genuinely two devices.
|
||||
{Name: "wgA", Enabled: true, URI: uri, Egress: "e1"},
|
||||
{Name: "wgB", Enabled: true, URI: uri, Egress: "e2"},
|
||||
{Name: "ss1", Enabled: true, URI: "ss://aes-256-gcm:secret@203.0.113.2:8388#ss1"},
|
||||
},
|
||||
Egresses: []model.Egress{
|
||||
{Name: "e1", Type: "interface", Interface: "eth0"},
|
||||
{Name: "e2", Type: "interface", Interface: "eth1"},
|
||||
},
|
||||
Groups: []model.Group{
|
||||
{Name: "pool", Source: "manual", Nodes: []string{"wgA", "wgB", "ss1"}},
|
||||
},
|
||||
Rules: []model.Rule{
|
||||
{Name: "default", Enabled: true, Order: 100, Target: "group:pool"},
|
||||
},
|
||||
}
|
||||
|
||||
opts, warns, err := GenerateWithWarnings(m)
|
||||
if err != nil {
|
||||
t.Fatalf("Generate: %v", err)
|
||||
}
|
||||
|
||||
if got := endpointTags(opts); len(got) != 1 || got[0] != "wgA" {
|
||||
t.Fatalf("endpoints = %v, want [wgA]: two uplinks from one key is two devices, one has to go", got)
|
||||
}
|
||||
// The default route is demonstrably still up: the pool kept live members.
|
||||
members := groupMembersByTag(t, opts, "pool")
|
||||
if len(members) != 2 || members[0] != "wgA" || members[1] != "ss1" {
|
||||
t.Fatalf("group members = %v, want [wgA ss1]", members)
|
||||
}
|
||||
if opts.Route.Final != "pool" {
|
||||
t.Fatalf("route.Final = %q, want pool", opts.Route.Final)
|
||||
}
|
||||
|
||||
found := warningsContaining(warns, "materialised twice")
|
||||
if len(found) != 1 {
|
||||
t.Fatalf("want exactly one conflict warning, got %v (all: %v)", found, warns)
|
||||
}
|
||||
lower := strings.ToLower(found[0])
|
||||
if strings.Contains(lower, "blocks your default route") ||
|
||||
strings.Contains(lower, "whole lan loses the internet") {
|
||||
t.Fatalf("the default route is still up (pool=%v) — the warning must NOT announce an outage: %s", members, found[0])
|
||||
}
|
||||
if !strings.Contains(lower, "it still has another way out") {
|
||||
t.Fatalf("the warning must say the default route survives and why, got: %s", found[0])
|
||||
}
|
||||
}
|
||||
|
||||
// ...and when the default route really IS taken down, it must say THAT, in the
|
||||
// words the operator would use.
|
||||
//
|
||||
// A chain that puts one WireGuard node behind itself asks for two devices from one
|
||||
// key by construction (L1 dials directly, L2 dials through L1 — different dialers,
|
||||
// different devices). One of them goes, which severs the chain, and the chain is the
|
||||
// default route: the whole LAN is off the internet. This is the sentence the live
|
||||
// outage needed and did not get.
|
||||
func TestWGMergeConflictKillingDefaultRouteSaysSo(t *testing.T) {
|
||||
priv, pub := wgDedupKey(141), wgDedupKey(151)
|
||||
m := &model.Model{
|
||||
Globals: model.DefaultGlobals(),
|
||||
Nodes: []model.Node{{Name: "wg1", Enabled: true, URI: wgDedupURI(priv, pub, "203.0.113.50", 51820)}},
|
||||
Chains: []model.Chain{{Name: "def", Hops: []string{"node:wg1", "node:wg1"}}},
|
||||
Rules: []model.Rule{
|
||||
{Name: "default", Enabled: true, Order: 100, Target: "chain:def"},
|
||||
},
|
||||
}
|
||||
|
||||
opts, warns, err := GenerateWithWarnings(m)
|
||||
if err != nil {
|
||||
t.Fatalf("Generate: %v", err)
|
||||
}
|
||||
|
||||
if opts.Route.Final != tagBlock {
|
||||
t.Fatalf("route.Final = %q, want %q — the chain lost a hop, so the default route must fail CLOSED", opts.Route.Final, tagBlock)
|
||||
}
|
||||
found := warningsContaining(warns, "materialised twice")
|
||||
if len(found) != 1 {
|
||||
t.Fatalf("want exactly one conflict warning, got %v (all: %v)", found, warns)
|
||||
}
|
||||
lower := strings.ToLower(found[0])
|
||||
if !strings.Contains(lower, "blocks your default route") {
|
||||
t.Fatalf("the default route is dead and the warning must say so outright, got: %s", found[0])
|
||||
}
|
||||
if !strings.Contains(lower, "whole lan loses the internet") {
|
||||
t.Fatalf("the warning must name the outage in the operator's terms, got: %s", found[0])
|
||||
}
|
||||
if strings.Contains(lower, "not affected") || strings.Contains(lower, "another way out") {
|
||||
t.Fatalf("the warning claims the default route is fine while it is blocked: %s", found[0])
|
||||
}
|
||||
}
|
||||
|
||||
// The same outage, but hidden one hop deeper — and this is the shape that decides
|
||||
// whether "is the default route dead?" is measured or merely assumed.
|
||||
//
|
||||
// chain:def = node:wg1 -> node:wg1 -> node:ss1 (the default route)
|
||||
//
|
||||
// The conflict is between the two wg1 hops; the loser is L2, so route.Final still
|
||||
// points at L3, which is a perfectly healthy shadowsocks outbound — except that its
|
||||
// Detour is now `block`, so it carries nothing. An outbound is only as alive as what
|
||||
// it dials through, and a check that stops at the first non-group tag reports this
|
||||
// dead route as working and tells the operator his default route is fine.
|
||||
func TestWGMergeDefaultRouteDeadBehindASurvivingExit(t *testing.T) {
|
||||
priv, pub := wgDedupKey(161), wgDedupKey(171)
|
||||
m := &model.Model{
|
||||
Globals: model.DefaultGlobals(),
|
||||
Nodes: []model.Node{
|
||||
{Name: "wg1", Enabled: true, URI: wgDedupURI(priv, pub, "203.0.113.60", 51820)},
|
||||
{Name: "ss1", Enabled: true, URI: "ss://aes-256-gcm:secret@203.0.113.2:8388#ss1"},
|
||||
},
|
||||
Chains: []model.Chain{{Name: "def", Hops: []string{"node:wg1", "node:wg1", "node:ss1"}}},
|
||||
Rules: []model.Rule{
|
||||
{Name: "default", Enabled: true, Order: 100, Target: "chain:def"},
|
||||
},
|
||||
}
|
||||
|
||||
opts, warns, err := GenerateWithWarnings(m)
|
||||
if err != nil {
|
||||
t.Fatalf("Generate: %v", err)
|
||||
}
|
||||
|
||||
// The premise: Final still names a live outbound, which is what makes a naive
|
||||
// "does Final exist?" check answer the wrong question.
|
||||
if opts.Route.Final != "chain-def-h3" {
|
||||
t.Fatalf("route.Final = %q, want chain-def-h3 — this test is only meaningful while the exit survives", opts.Route.Final)
|
||||
}
|
||||
if obByTag(opts, "chain-def-h3") == nil {
|
||||
t.Fatalf("chain-def-h3 was removed; outbounds=%v", outboundTags(opts))
|
||||
}
|
||||
if d := anyDetour(t, opts, "chain-def-h3"); d != tagBlock {
|
||||
t.Fatalf("chain-def-h3 detour = %q, want %q — the exit must have lost the hop it dialled through", d, tagBlock)
|
||||
}
|
||||
|
||||
found := warningsContaining(warns, "materialised twice")
|
||||
if len(found) != 1 {
|
||||
t.Fatalf("want exactly one conflict warning, got %v (all: %v)", found, warns)
|
||||
}
|
||||
lower := strings.ToLower(found[0])
|
||||
if !strings.Contains(lower, "blocks your default route") {
|
||||
t.Fatalf("the default route dead-ends at block one hop behind Final, and the warning must say so: %s", found[0])
|
||||
}
|
||||
if strings.Contains(lower, "another way out") || strings.Contains(lower, "not affected") {
|
||||
t.Fatalf("the warning reassures the operator about a default route that carries nothing: %s", found[0])
|
||||
}
|
||||
}
|
||||
@@ -86,11 +86,30 @@ func TestShippedConfigEnablesDNSIntercept(t *testing.T) {
|
||||
path := filepath.Join("..", "..", "openwrt", "shater-core", "files", "etc", "config", "shater")
|
||||
raw, err := os.ReadFile(path)
|
||||
if err != nil {
|
||||
t.Skipf("shipped config not readable from this checkout (%v)", err)
|
||||
// NOT a Skip. This is the only check that the installed file and
|
||||
// DefaultGlobals agree, so "could not read it" is a failure of the check,
|
||||
// not an excuse from it — and a silent skip is precisely how a guard ends
|
||||
// up reporting `ok` while guarding nothing.
|
||||
t.Fatalf("shipped config %s is unreadable (%v)", path, err)
|
||||
}
|
||||
text := string(raw)
|
||||
if !strings.Contains(text, "option dns_intercept '1'") {
|
||||
t.Error("the shipped /etc/config/shater must set dns_intercept '1' explicitly: a fresh install reads this file, and an operator who later opts out must see the option they are flipping")
|
||||
// Line-wise and comment-aware on purpose. That file is more than half
|
||||
// commented-out examples, so a plain strings.Contains is satisfied by
|
||||
// `#option dns_intercept '1'` — and the parse below cannot tell the difference
|
||||
// either, because a commented option falls back to the seed, which since D24 is
|
||||
// also true. So a config that ships the option COMMENTED OUT would pass both
|
||||
// halves of this test while giving a fresh install no visible option to flip,
|
||||
// which is the one thing this test exists to prevent. Nothing else here can
|
||||
// catch that.
|
||||
live := false
|
||||
for _, line := range strings.Split(text, "\n") {
|
||||
if strings.TrimSpace(line) == "option dns_intercept '1'" {
|
||||
live = true
|
||||
break
|
||||
}
|
||||
}
|
||||
if !live {
|
||||
t.Error("the shipped /etc/config/shater must carry an UNCOMMENTED `option dns_intercept '1'`: a fresh install reads this file, and an operator who later opts out must see the option they are flipping")
|
||||
}
|
||||
m, err := ParseUCIExport(text)
|
||||
if err != nil {
|
||||
|
||||
@@ -0,0 +1,187 @@
|
||||
package model
|
||||
|
||||
// Tests for the `l3_tunnel` default flip: carrying LAN ping through the tunnel
|
||||
// is now the DEFAULT, not an opt-in.
|
||||
//
|
||||
// Why it flipped, so a future reader does not "restore" the old default as a
|
||||
// safety measure: with l3_tunnel off, a LAN ping is decided by `untunnelable`
|
||||
// alone, and every rung of that ladder is either a drop (block) or a disclosure
|
||||
// (icmp/direct let the echo out of the WAN interface with the client's real
|
||||
// address). There was no configuration in which ping both worked and stayed
|
||||
// inside the tunnel. With l3_tunnel on, an L3-capable outbound carries the echo
|
||||
// and one that is not drops it honestly — adapter.JudgeFlow returns
|
||||
// tun.ActionDrop for an ICMP flow whose outbound is not a tun.Port
|
||||
// (adapter/judgeflow_icmp_lx_test.go), so no reply is ever forged. Turning it on
|
||||
// therefore breaks nothing that was working truthfully.
|
||||
//
|
||||
// Four properties are pinned, and the middle two pull against each other:
|
||||
//
|
||||
// 1. the seed and an option-less config both come back ON (every install
|
||||
// written before the option existed gains the ingress on upgrade);
|
||||
// 2. an EXPLICIT `option l3_tunnel '0'` stays OFF across a WriteUCI->ReadUCI
|
||||
// round-trip — a bool omitted at false would be re-read as the seed and
|
||||
// silently flip the operator's decision back on;
|
||||
// 3. the SHIPPED /etc/config/shater agrees with the seed (it is a conffile: a
|
||||
// fresh install's posture is that file's literal text, not DefaultGlobals);
|
||||
// 4. neither the off state nor the combinations the flip made misleading are
|
||||
// reached in silence — ValidateGlobals reports all three.
|
||||
|
||||
import (
|
||||
"os"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
"testing"
|
||||
)
|
||||
|
||||
// TestL3TunnelDefaultOn: the seed and an option-less config both say ON.
|
||||
func TestL3TunnelDefaultOn(t *testing.T) {
|
||||
if !DefaultGlobals().L3Tunnel {
|
||||
t.Fatal("DefaultGlobals().L3Tunnel = false, want true — with the L3 ingress off a LAN ping is either dropped or sent out of the WAN with the client's real address, and neither is what a tunnel is for")
|
||||
}
|
||||
m, err := ParseUCIExport("package shater\n\nconfig globals 'globals'\n\toption enabled '1'\n")
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if !m.Globals.L3Tunnel {
|
||||
t.Fatal("a config with no l3_tunnel option parsed as OFF; an absent option must fall back to the ON seed, or every install predating the option keeps leaking its pings")
|
||||
}
|
||||
}
|
||||
|
||||
// TestL3TunnelExplicitOffPreserved: the operator's opt-out survives both the
|
||||
// parse (over an ON seed) and the render->parse round-trip. Off is a supported
|
||||
// answer — a 32/64 MB router where the standing TUN + gVisor netstack costs real
|
||||
// money, or a bisection — so it has to be a decision the config keeps.
|
||||
func TestL3TunnelExplicitOffPreserved(t *testing.T) {
|
||||
m, err := ParseUCIExport("package shater\n\nconfig globals 'globals'\n" +
|
||||
"\toption enabled '1'\n\toption l3_tunnel '0'\n")
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if m.Globals.L3Tunnel {
|
||||
t.Fatal("explicit l3_tunnel '0' was overwritten by the ON seed")
|
||||
}
|
||||
|
||||
rendered := RenderUCIExport(m)
|
||||
if !strings.Contains(rendered, "option l3_tunnel '0'") {
|
||||
t.Fatalf("render must emit the explicit '0' (an omitted bool would be re-read as ON):\n%s", rendered)
|
||||
}
|
||||
back, err := ParseUCIExport(rendered)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if back.Globals.L3Tunnel {
|
||||
t.Fatal("l3_tunnel '0' did not survive the WriteUCI->ReadUCI round-trip — the opt-out would be silently re-enabled on the next save")
|
||||
}
|
||||
}
|
||||
|
||||
// TestShippedConfigEnablesL3Tunnel reads the file the package actually installs.
|
||||
// /etc/config/shater is a CONFFILE: written once on first install and never
|
||||
// replaced on upgrade, so a fresh install's ping posture is decided by this
|
||||
// file's literal text and not by DefaultGlobals. The two must agree, and only a
|
||||
// test that reads the shipped bytes can say that they do.
|
||||
func TestShippedConfigEnablesL3Tunnel(t *testing.T) {
|
||||
path := filepath.Join("..", "..", "openwrt", "shater-core", "files", "etc", "config", "shater")
|
||||
raw, err := os.ReadFile(path)
|
||||
if err != nil {
|
||||
t.Fatalf("shipped config %s is unreadable (%v) — this test must not be skipped: it is the only check that the installed file and DefaultGlobals agree", path, err)
|
||||
}
|
||||
text := string(raw)
|
||||
// Line-wise and comment-aware on purpose. That file is more than half
|
||||
// commented-out examples, so a plain strings.Contains would be satisfied by
|
||||
// `#option l3_tunnel '1'` — and the parse below cannot tell the difference
|
||||
// either, because a commented option falls back to the seed, which is now
|
||||
// also true. Nothing else in this test can catch that, so this check has to.
|
||||
live := false
|
||||
for _, line := range strings.Split(text, "\n") {
|
||||
if strings.TrimSpace(line) == "option l3_tunnel '1'" {
|
||||
live = true
|
||||
break
|
||||
}
|
||||
}
|
||||
if !live {
|
||||
t.Error("the shipped /etc/config/shater must carry an UNCOMMENTED `option l3_tunnel '1'`: a fresh install reads this file, and an operator who later opts out must be able to see the option they are flipping")
|
||||
}
|
||||
m, err := ParseUCIExport(text)
|
||||
if err != nil {
|
||||
t.Fatalf("shipped config does not parse: %v", err)
|
||||
}
|
||||
if !m.Globals.L3Tunnel {
|
||||
t.Error("shipped config parses with L3Tunnel off")
|
||||
}
|
||||
if m.Globals.Enabled {
|
||||
t.Error("shipped config must stay inert (enabled '0'): a fresh install may not touch connectivity")
|
||||
}
|
||||
}
|
||||
|
||||
// TestL3TunnelOffIsWarned: nobody reaches the off state by accident any more —
|
||||
// absent means ON — so a false is an explicit opt-out and must be told what it
|
||||
// costs, rather than be discovered later as "ping stopped working" or, worse and
|
||||
// silently, as "ping works, from the WAN address".
|
||||
func TestL3TunnelOffIsWarned(t *testing.T) {
|
||||
g := DefaultGlobals()
|
||||
g.L3Tunnel = false
|
||||
ws := ValidateGlobals(g)
|
||||
if !warnsMention(ws, "l3_tunnel is off") {
|
||||
t.Fatalf("turning the L3 ingress off produced no warning: %v", warningTexts(ws))
|
||||
}
|
||||
// The default policy has to be NAMED, not implied: "block" and "direct" have
|
||||
// opposite consequences for the packet and the operator cannot tell which
|
||||
// they are in from the option's absence.
|
||||
if !warnsMention(ws, "untunnelable (block)") {
|
||||
t.Fatalf("the off warning does not name the policy that takes over instead: %v", warningTexts(ws))
|
||||
}
|
||||
|
||||
// The control: with the ingress ON the same prober must stay quiet, or a
|
||||
// green "no warning" above would prove nothing about the check firing.
|
||||
if ws := ValidateGlobals(DefaultGlobals()); warnsMention(ws, "l3_tunnel is off") {
|
||||
t.Fatalf("the default (L3 ingress ON) was reported as off: %v", warningTexts(ws))
|
||||
}
|
||||
}
|
||||
|
||||
// TestL3TunnelWithUntunnelableICMPIsReported is the combination the flip made
|
||||
// misleading. `icmp` used to mean "block, plus working ping". With l3_tunnel on
|
||||
// the prerouting L3 mark claims every ICMP packet from a diverted device before
|
||||
// the forward chain where that echo accept lives, so the ping half is dead and
|
||||
// only the half the operator may not have wanted survives: ESP/AH/GRE/IGMP/SCTP
|
||||
// out with the real address toward directly-routed destinations. Under the old
|
||||
// opt-in default the pair was exotic; under the new one it is ordinary, and
|
||||
// silence about it would be the panel lying about state.
|
||||
func TestL3TunnelWithUntunnelableICMPIsReported(t *testing.T) {
|
||||
g := DefaultGlobals()
|
||||
g.Untunnelable = "icmp"
|
||||
ws := ValidateGlobals(g)
|
||||
if !warnsMention(ws, "ping no longer reaches this policy") {
|
||||
t.Fatalf("l3_tunnel=1 + untunnelable=icmp was accepted in silence: %v", warningTexts(ws))
|
||||
}
|
||||
|
||||
// Control 1: the same policy WITHOUT the L3 ingress is the state the rung was
|
||||
// designed for and must not be reported.
|
||||
g.L3Tunnel = false
|
||||
if ws := ValidateGlobals(g); warnsMention(ws, "ping no longer reaches this policy") {
|
||||
t.Fatalf("untunnelable=icmp under l3_tunnel=0 was reported as dead: %v", warningTexts(ws))
|
||||
}
|
||||
// Control 2: the L3 ingress with the DEFAULT policy is the ordinary posture
|
||||
// and must not be reported either.
|
||||
if ws := ValidateGlobals(DefaultGlobals()); warnsMention(ws, "ping no longer reaches this policy") {
|
||||
t.Fatalf("the default posture (l3_tunnel=1 + untunnelable=block) was reported: %v", warningTexts(ws))
|
||||
}
|
||||
}
|
||||
|
||||
// warnsMention reports whether any warning text contains sub.
|
||||
func warnsMention(ws []Warning, sub string) bool {
|
||||
for _, w := range ws {
|
||||
if strings.Contains(w.Message, sub) {
|
||||
return true
|
||||
}
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
// warningTexts flattens warnings for a failure message.
|
||||
func warningTexts(ws []Warning) []string {
|
||||
out := make([]string, 0, len(ws))
|
||||
for _, w := range ws {
|
||||
out = append(out, w.Message)
|
||||
}
|
||||
return out
|
||||
}
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user