#!/bin/sh

# Copyright (C) 2025 asvow
# SPDX-License-Identifier: GPL-3.0-only
#
# tailscale_helper: one-shot post-start configurator for tailscaled.
# Invoked by /etc/init.d/tailscale start_instance. Reads the same CLI flags
# that tailscale's own `up`/`set` consumes, then:
#   1. waits for tailscaled to accept IPC;
#   2. runs `tailscale up` (first connect) or `tailscale set` (thereafter);
#   3. provisions the OpenWrt firewall zone + forwardings;
#   4. applies the kill-switch when an exit node is configured.

# ---------- Constants ------------------------------------------------------

LOCK_FILE=/var/lock/tailscale.lock
STATE_DIR=/var/lib/tailscale
KILLSWITCH_STATE=$STATE_DIR/killswitch_lan_wan
TAILSCALE=/usr/sbin/tailscale

# Name of the firewall zone we own (firewall.tszone.name).
TS_ZONE_NAME=tailscale

# Path of the transient 0600 file holding the authkey, when one is in use.
AUTHKEY_FILE=""

# ---------- Logging --------------------------------------------------------

log_info() { logger -p daemon.info -t tailscale_helper "$*"; }
log_warn() { logger -p daemon.warn -t tailscale_helper "$*"; }
log_err()  { logger -p daemon.err  -t tailscale_helper "$*"; }

# ---------- Exit helpers ---------------------------------------------------

# Exit cleanly on SIGTERM from init.d stop. We used to `uci -q revert firewall`
# here, but that's "friendly fire": it also drops pending changes that other
# writers (user in a LuCI tab editing port forwards, fw4-gui, etc.) hadn't
# committed yet. All of helper's mutations are idempotent, so leaving them in
# the pending state is safe:
#   * if stop_service commits next, its own deletes of tszone/ts_ac_*
#     supersede our partial sets, yielding clean state;
#   * if helper is restarted, it re-runs the same steps and converges.
graceful_exit() {
	log_warn "received signal, aborting"
	discard_authkey_file
	exit 0
}

# Fatal error path. Does NOT recurse into init.d stop — that would killall us
# mid-exit.
fatal() {
	log_err "$1"
	discard_authkey_file
	exit 1
}

# ---------- Authkey handling ----------------------------------------------
# The key arrives in the environment (TS_AUTHKEY), never in argv — helper's
# own command line is readable by any local user via `ps`. Handing it to
# `tailscale up` has the same problem, so we spill it into a 0600 file on
# tmpfs and use the documented `--auth-key=file:` form instead.

write_authkey_file() {
	[ -n "$AUTHKEY" ] || return 1
	AUTHKEY_FILE=$(mktemp /var/run/ts_authkey.XXXXXX 2>/dev/null) || {
		AUTHKEY_FILE=""
		return 1
	}
	chmod 600 "$AUTHKEY_FILE" 2>/dev/null
	printf '%s' "$AUTHKEY" > "$AUTHKEY_FILE" || {
		discard_authkey_file
		return 1
	}
	return 0
}

discard_authkey_file() {
	[ -n "$AUTHKEY_FILE" ] && rm -f "$AUTHKEY_FILE"
	AUTHKEY_FILE=""
}

# ---------- Step: wait for tailscaled -------------------------------------
# tailscaled comes up async (procd starts it in parallel). Helper must wait
# for the local IPC socket to answer before issuing commands.

wait_for_tailscaled() {
	# `tailscale version` returns 0 even when the daemon is down (it's a pure
	# local binary print). `tailscale status --peers=false` returns exit 1 in
	# NeedsLogin state ("Logged out."). Only `--json` both pokes the IPC
	# socket and returns 0 regardless of login/auth state.
	local n=0
	while ! $TAILSCALE status --json >/dev/null 2>&1; do
		sleep 1
		n=$((n + 1))
		[ $n -ge 30 ] && fatal "tailscaled not ready after 30s"
	done
}

# ---------- Step: lock -----------------------------------------------------
# Flock guards against concurrent helper runs (LuCI can queue multiple
# save/applies in quick succession; the init.d reload also spawns a helper).
# We wait up to 30s for a queued run — silent exit 0 would drop the config
# change the user just made.

acquire_lock() {
	mkdir -p "$(dirname "$LOCK_FILE")" 2>/dev/null
	exec 9> "$LOCK_FILE" || fatal "cannot open $LOCK_FILE"
	if ! flock -xn 9; then
		log_warn "another helper is running; waiting up to 30s"
		flock -w 30 9 || fatal "could not acquire lock within 30s"
	fi
}

# ---------- Step: parse CLI args ------------------------------------------
# We reset each variable first — otherwise $HOSTNAME (and potentially others)
# leak in from the shell environment.

parse_args() {
	ACCEPT_ROUTES=""; ACCEPT_DNS=""; TS_HOSTNAME=""
	ADVERTISE_EXIT_NODE=""; EXIT_NODE=""; ADVERTISE_ROUTES=""
	SNAT_SUBNET_ROUTES=""; LOGIN_SERVER=""; DNS_MODE=""
	EXTRA_FLAGS=""

	# init.d hands the key over in the environment. `--authkey=` stays
	# supported so the helper is still usable by hand, but nothing in the
	# package puts the key on a command line any more.
	AUTHKEY="${TS_AUTHKEY:-}"
	unset TS_AUTHKEY

	local arg
	for arg in "$@"; do
		case "$arg" in
			--accept-routes=*)       ACCEPT_ROUTES="${arg#*=}" ;;
			--accept-dns=*)          ACCEPT_DNS="${arg#*=}" ;;
			--hostname=*)            TS_HOSTNAME="${arg#*=}" ;;
			--advertise-exit-node=*) ADVERTISE_EXIT_NODE="${arg#*=}" ;;
			--exit-node=*)           EXIT_NODE="${arg#*=}" ;;
			--advertise-routes=*)    ADVERTISE_ROUTES="${arg#*=}" ;;
			--snat-subnet-routes=*)  SNAT_SUBNET_ROUTES="${arg#*=}" ;;
			--login-server=*)        LOGIN_SERVER="${arg#*=}" ;;
			--dns-mode=*)            DNS_MODE="${arg#*=}" ;;
			--authkey=*)             AUTHKEY="${arg#*=}" ;;
			--*)                     EXTRA_FLAGS="${EXTRA_FLAGS} $arg" ;;
		esac
	done
}

# ---------- Step: tailscale up / set --------------------------------------
# `up` vs `set`:
#   - `up` is the full reconfigure path — required for the first login and
#     any time we want to hand tailscaled a fresh authkey/login-server.
#   - `set` is incremental preference updates on an already-authenticated
#     node. Much cheaper; doesn't churn routes/firewall rules.
#
# We pick `up` only when there is no tailscale IPv4 address yet. On every
# subsequent LuCI save/apply we use `set`.

# Both builders assemble their arguments in the positional parameters and run
# tailscale from the same shell. They used to `echo "$@"` and let the caller
# re-split the string, which silently dropped quoting — a Device Name with a
# space in it arrived as two arguments and the call failed.

ts_up() {
	set --
	set -- "$@" --timeout=30s
	[ "$ACCEPT_ROUTES" = "1" ] \
		&& set -- "$@" --accept-routes \
		|| set -- "$@" --accept-routes=false
	# Default DNS off: tailscale MagicDNS conflicts with OpenWrt dnsmasq.
	[ "$ACCEPT_DNS" = "1" ] \
		&& set -- "$@" --accept-dns \
		|| set -- "$@" --accept-dns=false
	[ "$ADVERTISE_EXIT_NODE" = "1" ] \
		&& set -- "$@" --advertise-exit-node \
		|| set -- "$@" --advertise-exit-node=false
	[ "$SNAT_SUBNET_ROUTES" = "1" ] \
		&& set -- "$@" --snat-subnet-routes \
		|| set -- "$@" --snat-subnet-routes=false
	[ -n "$TS_HOSTNAME" ] && set -- "$@" "--hostname=$TS_HOSTNAME"
	if [ -n "$EXIT_NODE" ]; then
		set -- "$@" "--exit-node=$EXIT_NODE" --exit-node-allow-lan-access
	else
		set -- "$@" "--exit-node="
	fi
	set -- "$@" "--advertise-routes=$ADVERTISE_ROUTES"
	# Force netfilter off: tailscaled's connmark-restore rule assigns instead
	# of merging and clobbers external fwmarks (sing-box tproxy, mwan3). Still
	# true as of 1.102.2 — see cleanup_tailscale_mangle below.
	set -- "$@" --netfilter-mode=off

	[ -n "$LOGIN_SERVER" ] && set -- "$@" "--login-server=$LOGIN_SERVER"
	if write_authkey_file; then
		set -- "$@" "--auth-key=file:$AUTHKEY_FILE"
	elif [ -n "$AUTHKEY" ]; then
		log_warn "could not stage authkey file; continuing without a key"
	fi
	# EXTRA_FLAGS is a space-separated list of whole flags by construction, so
	# splitting it here is deliberate.
	local f
	for f in $EXTRA_FLAGS; do set -- "$@" "$f"; done

	run_ts "$TAILSCALE" up "$@"
}

# Minimal args for `set`. Omissions (intentional — each has a past incident):
#   --authkey            one-shot; re-passing spams "authkey not valid"
#   --login-server       requires --force-reauth; spams "can't change..."
#   --netfilter-mode     only relevant on daemon init; re-pass spams warnings
#   --snat-subnet-routes re-asserts SNAT rules even on no-op runs
#   --exit-node-allow-lan-access  rejected by `set` without --exit-node
ts_set() {
	set --
	[ "$ACCEPT_ROUTES" = "1" ] \
		&& set -- "$@" --accept-routes \
		|| set -- "$@" --accept-routes=false
	[ "$ACCEPT_DNS" = "1" ] \
		&& set -- "$@" --accept-dns \
		|| set -- "$@" --accept-dns=false
	[ "$ADVERTISE_EXIT_NODE" = "1" ] \
		&& set -- "$@" --advertise-exit-node \
		|| set -- "$@" --advertise-exit-node=false
	[ -n "$TS_HOSTNAME" ] && set -- "$@" "--hostname=$TS_HOSTNAME"
	if [ -n "$EXIT_NODE" ]; then
		set -- "$@" "--exit-node=$EXIT_NODE"
	else
		set -- "$@" "--exit-node="
	fi
	set -- "$@" "--advertise-routes=$ADVERTISE_ROUTES"

	run_ts "$TAILSCALE" set "$@"
}

# Run a tailscale command, logging its output while preserving exit status.
run_ts() {
	local out status
	out=$("$@" 2>&1)
	status=$?
	[ -n "$out" ] && echo "$out" | logger -p daemon.warn -t tailscale_helper
	return $status
}

apply_tailscale_config() {
	# Pick command based on BackendState — authoritative source.
	# Using `tailscale ip -4` emptiness was unreliable: it returns empty
	# in NeedsLogin AND NeedsMachineAuth, which caused repeated `up` spam.
	local state st
	state=$($TAILSCALE status --json 2>/dev/null | jsonfilter -e '@.BackendState' 2>/dev/null)

	case "$state" in
		Running|Starting)
			# Node is up (or nearly so) — incremental update only.
			ts_set
			st=$?
			[ $st -ne 0 ] && { log_warn "tailscale set failed (exit $st)"; ts_config_failed=1; }
			;;
		NeedsMachineAuth)
			# Device is registered and waiting for admin approval. Re-running
			# `up` would just re-authenticate with the same (possibly spent)
			# key. Leave daemon alone — user must approve in admin console.
			log_info "BackendState=NeedsMachineAuth — awaiting admin approval, skipping up/set"
			;;
		*)
			# Stopped, NoState, NeedsLogin — daemon hasn't authenticated yet.
			ts_up
			st=$?
			discard_authkey_file
			[ $st -ne 0 ] && { log_err "tailscale up failed (exit $st)"; ts_config_failed=1; }
			;;
	esac
}

# ---------- Step: DNS mode -------------------------------------------------
# Three ways to reach MagicDNS names, selected by the `dns_mode` UCI option:
#
#   disabled  leave DNS alone (default)
#   magicdns  --accept-dns=true; tailscaled rewrites resolv.conf, which fights
#             dnsmasq on OpenWrt — offered, but not the recommended route
#   dnsmasq   --accept-dns=false and dnsmasq forwards just the tailnet zone to
#             the MagicDNS resolver, so dnsmasq stays in charge of everything
#
# The quad-100 resolver answers even when --accept-dns is false (verified
# against 1.94.2), which is what makes the third mode work at all.
#
# The zone is read from MagicDNSSuffix rather than hardcoded to "ts.net":
# self-hosted control planes (headscale, ionscale) issue their own suffix.

MAGICDNS_ADDR=100.100.100.100

# The suffix only exists once the daemon has a netmap, which can lag a few
# seconds behind `tailscale up` on a cold boot. Give it a short window rather
# than giving up until the next reload — that window was long enough for the
# forward to simply never appear after a reboot.
#
# Only worth waiting while the node is actually coming up: when it is logged
# out there will never be a suffix, and blocking 10s on every helper run would
# be pure cost.
magicdns_suffix() {
	local n=0 s state

	while [ $n -lt 10 ]; do
		s=$($TAILSCALE status --json 2>/dev/null \
			| jsonfilter -e '@.MagicDNSSuffix' 2>/dev/null)
		[ -n "$s" ] && { echo "$s"; return 0; }

		state=$($TAILSCALE status --json 2>/dev/null \
			| jsonfilter -e '@.BackendState' 2>/dev/null)
		case "$state" in
			Running|Starting) ;;
			*) return 1 ;;
		esac

		n=$((n + 1))
		sleep 1
	done

	return 1
}

# Every dnsmasq server entry that points at the MagicDNS resolver, whatever
# zone it names — catches leftovers after a tailnet rename.
list_magicdns_forwards() {
	uci -q get dhcp.@dnsmasq[0].server 2>/dev/null \
		| tr ' ' '\n' \
		| grep "/${MAGICDNS_ADDR}\$"
}

drop_magicdns_forwards() {
	local e
	for e in $(list_magicdns_forwards); do
		uci -q del_list "dhcp.@dnsmasq[0].server=$e" && dns_mutated=1
	done
}

apply_dns_mode() {
	dns_mutated=0

	if [ "$DNS_MODE" != "dnsmasq" ]; then
		drop_magicdns_forwards
	else
		local suffix entry have
		suffix=$(magicdns_suffix)
		if [ -z "$suffix" ]; then
			# Before first login there is no suffix yet; leave whatever is
			# configured alone and retry on the next helper run rather than
			# tearing down a working forward.
			log_warn "dns_mode=dnsmasq but MagicDNSSuffix is not known yet; skipping"
			return
		fi
		entry="/$suffix/$MAGICDNS_ADDR"

		# `uci del_list` removes *every* copy of a value, so entries cannot be
		# pruned one at a time. Compare against the desired end state instead —
		# exactly one forward, naming the current suffix — and rebuild if it
		# differs. That collapses duplicates and drops stale suffixes left over
		# from a tailnet rename in the same pass.
		local total match
		total=$(list_magicdns_forwards | wc -l)
		match=$(list_magicdns_forwards | grep -Fxc "$entry")

		if [ "$total" != "1" ] || [ "$match" != "1" ]; then
			drop_magicdns_forwards
			uci -q add_list "dhcp.@dnsmasq[0].server=$entry" && dns_mutated=1
		fi
	fi

	# Commit only our own change, and only when there is one: /etc/config/dhcp
	# belongs to other writers too.
	if [ "$dns_mutated" = "1" ] && [ -n "$(uci changes dhcp)" ]; then
		uci commit dhcp && {
			/etc/init.d/dnsmasq reload >/dev/null 2>&1 \
				|| /etc/init.d/dnsmasq restart >/dev/null 2>&1
			log_info "dns_mode=$DNS_MODE: updated dnsmasq forwards"
		}
	fi
}

# ---------- Step: firewall zone -------------------------------------------
# We flip fw_mutated=1 on every uci mutation, so the final commit runs only
# when something actually changed. Without that flag, idempotent helper runs
# would trigger useless `/etc/init.d/firewall reload` cycles.

ensure_firewall_zone() {
	[ -n "$(uci -q get firewall.tszone)" ] && return
	uci set firewall.tszone='zone'
	uci set firewall.tszone.name='tailscale'
	uci set firewall.tszone.input='ACCEPT'
	uci set firewall.tszone.output='ACCEPT'
	uci set firewall.tszone.forward='ACCEPT'
	uci set firewall.tszone.masq='1'
	uci set firewall.tszone.mtu_fix='1'
	uci add_list firewall.tszone.device='tailscale+'
	fw_mutated=1
}

# ---------- Step: forwarding rules ----------------------------------------
# Reconcile against $ACCESS: create rules the user enabled, remove ones they
# disabled. Without this, stale forwardings accumulate over time.

want_rule() {
	local name="$1" src="$2" dest="$3"
	case " $ACCESS " in
		*" $name "*)
			[ -n "$(uci -q get firewall.$name)" ] && return
			uci set firewall.$name='forwarding'
			uci set firewall.$name.src="$src"
			uci set firewall.$name.dest="$dest"
			fw_mutated=1
			;;
		*)
			[ -z "$(uci -q get firewall.$name)" ] && return
			uci -q delete firewall.$name
			fw_mutated=1
			;;
	esac
}

reconcile_forwardings() {
	want_rule ts_ac_lan tailscale lan
	want_rule ts_ac_wan tailscale wan
	want_rule lan_ac_ts lan tailscale
	want_rule wan_ac_ts wan tailscale
}

# ---------- Step: kill-switch --------------------------------------------
# With an exit node set, break the direct LAN → WAN path so traffic can't
# leak around the tunnel. We record every rule we disable so we can restore
# the prior state — which may differ per rule — when exit_node is cleared.

# A destination zone is an escape hatch if it NATs traffic out (masq=1) —
# that covers `wan` but also VPN-uplink zones like wg/awg, which the old
# hardcoded dest=wan test walked straight past, leaving the LAN a way around
# the exit node. Our own tailscale zone sets masq too, so exclude it by name.
#
# The section-type check is not decoration: rules, redirects AND forwardings
# may all carry a `.name`, so matching on name alone picks up non-zones.
zone_is_escape() {
	local want="$1" sect
	[ -n "$want" ] || return 1
	[ "$want" = "$TS_ZONE_NAME" ] && return 1
	for sect in $(uci show firewall 2>/dev/null \
		| grep "\.name='$want'\$" \
		| cut -d'.' -f2 | cut -d'=' -f1); do
		[ "$(uci -q get "firewall.$sect")" = "zone" ] || continue
		[ "$(uci -q get "firewall.$sect.masq")" = "1" ] && return 0
	done
	return 1
}

list_lan_escape_fwd() {
	local s dest
	for s in $(uci show firewall 2>/dev/null | grep '=forwarding$' \
		| cut -d'.' -f2 | cut -d'=' -f1); do
		[ "$(uci -q get firewall.$s.src)" = "lan" ] || continue
		dest=$(uci -q get firewall.$s.dest)
		zone_is_escape "$dest" || continue
		echo "$s"
	done
}

apply_killswitch() {
	if [ -n "$EXIT_NODE" ]; then
		# Deliberately fail closed. If `tailscale set` did not take, the exit
		# node is not actually carrying traffic — but opening the direct path
		# instead would leak exactly what the kill-switch exists to prevent.
		# Say so loudly; the Settings page shows the same mismatch as a banner
		# by comparing the saved exit node against `tailscale debug prefs`.
		if [ "$ts_config_failed" = "1" ]; then
			log_err "exit node configured but tailscale reconfiguration failed:" \
			        "keeping the kill-switch engaged, LAN has no direct egress"
		fi
		killswitch_on
	else
		killswitch_off
	fi
}

# Original-state is stored as a sibling UCI key `firewall.$s.ts_saved_enabled`
# in the same /etc/config/firewall file that holds the `enabled` we flip.
# This survives reboot (tmpfs /var/lib was lost on every boot, which permanently
# stuck lan->wan in the disabled state if the kill-switch had been active at
# the time of reboot).
killswitch_on() {
	local s prev
	for s in $(list_lan_escape_fwd); do
		# Record original state only if we haven't already — covers repeated
		# helper runs AND new escape routes added while kill-switch is on.
		if [ -z "$(uci -q get firewall.$s.ts_saved_enabled)" ]; then
			prev=$(uci -q get "firewall.$s.enabled")
			uci -q set "firewall.$s.ts_saved_enabled=${prev:-<unset>}"
			fw_mutated=1
		fi
		if [ "$(uci -q get firewall.$s.enabled)" != "0" ]; then
			uci -q set firewall.$s.enabled='0'
			fw_mutated=1
		fi
	done
}

killswitch_off() {
	# Find every forwarding section we previously marked and restore its
	# original `enabled` value.
	local s saved
	for s in $(uci show firewall 2>/dev/null \
		| grep '\.ts_saved_enabled=' \
		| cut -d'.' -f2 | cut -d'=' -f1); do
		saved=$(uci -q get "firewall.$s.ts_saved_enabled")
		if [ "$saved" = "<unset>" ]; then
			uci -q delete "firewall.$s.enabled"
		else
			uci -q set "firewall.$s.enabled=$saved"
		fi
		uci -q delete "firewall.$s.ts_saved_enabled"
		fw_mutated=1
	done

	# Migration: clean up old tmpfs state file from pre-1.2.6-40 versions.
	rm -f "$KILLSWITCH_STATE"
}

# ---------- Step: commit + reload -----------------------------------------
# Commit only when WE actually mutated something. fw_mutated keeps us from
# triggering expensive fw4 reloads on no-op helper runs.

commit_firewall() {
	[ "$fw_mutated" = "1" ] || return
	[ -n "$(uci changes firewall)" ] || return
	uci commit firewall && /etc/init.d/firewall reload >/dev/null 2>&1
}

# ---------- Step: remove tailscaled's clobbering connmark rules -----------
# tailscaled's connmark-restore rule assigns rather than merges:
# `meta mark set ct mark & MASK` wipes every mark bit outside MASK. Upstream
# documents this as a standing limitation of the nftables expression VM (see
# makeConnmarkRestoreExprs in util/linuxfw/nftables_runner.go) — as of 1.102.2
# it is still present, guarded only against the total-wipe case. That is what
# breaks co-tenants such as sing-box tproxy and mwan3.
#
# tsconst.LinuxFwmarkMask is 0xff0000 on every release we have shipped
# (1.94.2, 1.96.4, 1.102.2). Rules have also been seen rendered with the mask
# byte-swapped to 0x0000ff00, so the pattern below targets only that swapped
# form (canonical "0x0000ff00" and compact "0xff00") and deliberately leaves
# correctly-masked 0xff0000 rules alone.
#
# Belt and braces: we run tailscaled with --netfilter-mode=off, and with that
# setting no ip/ip6 mangle table is created at all (verified on 1.94.2). This
# purge only matters if netfilter mode is turned back on, or if a saved
# profile briefly installs rules before our prefs land.

cleanup_tailscale_mangle() {
	local rule_file fam chain h
	command -v nft >/dev/null 2>&1 || return 0
	rule_file=$(mktemp /tmp/ts_rules.XXXXXX 2>/dev/null) || return 0
	for fam in ip ip6; do
		for chain in PREROUTING OUTPUT; do
			nft -a list chain $fam mangle $chain >"$rule_file" 2>/dev/null || continue
			# The trailing guard matters: without it `0x0*ff00` also matches
			# inside 0xff0000 — the mask tailscaled actually uses — so the
			# purge would delete correctly-masked rules too.
			awk '/(meta mark set ct mark|ct mark set meta mark) & 0x0*ff00([^0-9a-fA-F]|$)/ {
				match($0, /handle [0-9]+$/);
				if (RSTART) print substr($0, RSTART+7)
			}' "$rule_file" | while read -r h; do
				[ -n "$h" ] && nft delete rule $fam mangle $chain handle "$h" 2>/dev/null
			done
		done
	done
	rm -f "$rule_file"
}

# ---------- Main -----------------------------------------------------------

main() {
	log_info "Starting tailscale_helper"
	trap graceful_exit TERM INT

	wait_for_tailscaled
	acquire_lock
	parse_args "$@"

	fw_mutated=0
	ts_config_failed=0

	apply_tailscale_config
	apply_dns_mode
	ensure_firewall_zone
	reconcile_forwardings
	apply_killswitch
	commit_firewall
	cleanup_tailscale_mangle

	if [ "$ts_config_failed" = "1" ]; then
		log_warn "Completed with errors (see above)"
	else
		log_info "Completed successfully"
	fi
}

main "$@"
