#!/bin/sh
# shellcheck shell=dash
# oaf-update.sh — Nightly updater for OAF and family-safe packages
#
# Deployed to the router at /usr/sbin/oaf-update.sh by:
#   - 'make router-install' (from dev machine)
#   - Baked into custom firmware via build-image.sh
#
# Invoked by:
#   - Nightly cron: random 02:xx–03:xx UTC (set per-device at first boot by 99-family-safe)
#   - Manually:     /usr/sbin/oaf-update.sh [--force]
#
# POSIX sh — compatible with OpenWrt busybox ash (no bashisms).
# Uses only: wget, awk, grep, sed, uname, logger, opkg (apk fallback)
#
# Log format: logfmt — every line is space-separated key=value pairs.
#   Mandatory fields on every line: device=  level=  event=
#   Values containing spaces are double-quoted.
#   Parse examples:
#     grep 'event=kmod_skip'              # all kmod ABI mismatches across fleet
#     grep 'device=e4671eaf156c'          # everything from one router
#     awk -F'pkg=' '{print $2}' | awk '{print $1}'   # package names touched
#     grep 'level=warn\|level=error'      # problems only
#
# S3 contract:
#   manifest-oaf.env      (written by OpenAppFilter repo CI)
#   manifest-familysafe.env (written by this repo's CI)
#   packages/             versioned .ipk files
#
# Exit codes:
#   0 — success (or graceful skip due to network/S3 failure)
#   1 — hard error (no package manager found)

# Update channel: the literal path segment under the single router bucket
# (e.g. "dev", "main", "router-prod-tenbay-wr3000k") -- not a dev/prod enum.
# Set at build time to match wherever this build's own package was published,
# so a device keeps tracking its own channel (a WR3000K test unit tracks
# router-prod-tenbay-wr3000k indefinitely, not just "main").
# Override by setting UPDATE_CHANNEL=<path> in /etc/oaf-update.conf.
#
# OAF_CHANNEL is separate from UPDATE_CHANNEL because the two repos release on
# independent schedules -- a device can track a firmware channel and an OAF
# channel that move at different times. Where app-filter publishes a matching
# channel (router-prod-tenbay-wr3000k), the image bakes it in via
# /etc/oaf-update.conf.
#
# These are defaults for a device with no conf file. Do not remove them, and
# keep them pointing at a channel that both EXISTS and holds THIS product.
# They used to be `main`, which is wrong in two different ways — measured
# 2026-08-17, do not collapse them into one claim:
#
#   router/main/oaf-update.sh      -> 404   (this is the fatal one: the
#                                            bootstrap retries a nonexistent
#                                            URL forever behind its backoff
#                                            and family-safe is never
#                                            installed)
#   router/main/manifest-oaf.env   -> 200   (exists, but it is the generic
#                                            product's OAF manifest, not the
#                                            WR3000K one — wrong packages,
#                                            not a missing URL)
#
# On this branch (prod-tenbay-wr3000k) the WR3000K prod channel is the only
# product built, so it is the correct fallback for both. CI-built images
# override these via /etc/oaf-update.conf below.
# See MISTAKES.md 2026-08-17.
UPDATE_CHANNEL="router-prod-tenbay-wr3000k"
OAF_CHANNEL="router-prod-tenbay-wr3000k"
if [ -f /etc/oaf-update.conf ]; then
    # shellcheck disable=SC1091
    . /etc/oaf-update.conf
fi
# Package/firmware CDN host. Matches KAHF_CDN_DOMAIN in config/lab.env.
KAHF_HOST="${KAHF_HOST:-router.kahf.co}"
S3_BASE="https://${KAHF_HOST}/router/${UPDATE_CHANNEL}"
MANIFEST_OAF_URL="https://${KAHF_HOST}/router/${OAF_CHANNEL}/manifest-oaf.env"
# Allow CI / staging override: if MANIFEST_FS_URL is already set in the
# environment (e.g. by verify-image pointing at a staging prefix), honour it.
# Otherwise fall back to the canonical channel URL.
MANIFEST_FS_URL="${MANIFEST_FS_URL:-${S3_BASE}/manifest-familysafe.env}"
TMP_DIR="/tmp/oaf-update"
LOG_TAG="oaf-update"
# Unique device identifier — MAC of br-lan, colons stripped (e.g. "a4b1c2d3e4f5").
# Prefixed on every log line so fleet-wide aggregated syslog can be filtered per router.
DEVICE_ID="$(cat /sys/class/net/br-lan/address 2>/dev/null | tr -d ':')"
[ -z "${DEVICE_ID}" ] && DEVICE_ID="unknown"
WGET_TIMEOUT=30
WGET_DOWNLOAD_TIMEOUT=120

FORCE=0
for arg in "$@"; do
    case "$arg" in
        --force|-f) FORCE=1 ;;
    esac
done

# Emit a logfmt line to both syslog and stdout.
# Callers pass bare key=value pairs; device= and level= are prepended automatically.
# Values with spaces must be quoted by the caller: key="some value"
log()   { local _e="device=${DEVICE_ID} level=info $*";  logger -p daemon.info    -t "${LOG_TAG}" "${_e}" 2>/dev/null; echo "${_e}"; }
warn()  { local _e="device=${DEVICE_ID} level=warn $*";  logger -p daemon.warning -t "${LOG_TAG}" "${_e}" 2>/dev/null; echo "${_e}"; }
error() { local _e="device=${DEVICE_ID} level=error $*"; logger -p daemon.err     -t "${LOG_TAG}" "${_e}" 2>/dev/null; echo "${_e}"; }

log "event=start ts=$(date -u +%Y-%m-%dT%H:%M:%SZ) channel=${UPDATE_CHANNEL} s3=${S3_BASE}"

# ------------------------------------------------------------------
# Diagnostics: free space + inode availability for the filesystem(s) a
# local write here can actually fail against. OpenWrt symlinks /var to
# /tmp, so apk's own package-index cache (populated by pkglist_refresh
# below) and this script's own TMP_DIR scratch share ONE ram-backed
# tmpfs -- and a local-write failure is otherwise unexplained after the
# fact: wget/uclient-fetch's exit=3 means "could not open the destination
# file" (not DNS/connect/TLS -- those are 4 and 5), which is exactly what
# tmpfs exhaustion looks like, but nothing in the existing log said
# whether the filesystem was actually full. Verify-image's WR1205K
# failure (pkglist_refresh ok, then all three of this run's local writes
# failing in the same short window) is the concrete case this exists to
# stop being a guess. One compact line per call site, only at the few
# fixed points below -- never in a loop, so a healthy run stays exactly as
# quiet as before.
# ------------------------------------------------------------------
_log_space() {
    local _point="$1" _fs _blk _ino _root_dev _tmp_dev
    for _fs in /tmp /; do
        # Skip / when it's the SAME filesystem as /tmp (the common case: a
        # router's whole overlay is one squashfs+overlay mount and /tmp is
        # a separate tmpfs) -- otherwise this would log every point twice
        # for no new information, which is exactly the chattiness this
        # instrumentation must not add.
        if [ "${_fs}" = "/" ]; then
            _tmp_dev="$(df -P /tmp 2>/dev/null | awk 'NR==2{print $1}')"
            _root_dev="$(df -P / 2>/dev/null | awk 'NR==2{print $1}')"
            [ -n "${_tmp_dev}" ] && [ "${_tmp_dev}" = "${_root_dev}" ] && continue
        fi
        _blk="$(df -Pk "${_fs}" 2>/dev/null | awk 'NR==2{printf "avail_kb=%s use_pct=%s", $4, $5}')"
        _ino="$(df -Pi "${_fs}" 2>/dev/null | awk 'NR==2{printf "iavail=%s iuse_pct=%s", $4, $5}')"
        log "event=diskspace point=${_point} fs=${_fs} ${_blk:-avail_kb=? use_pct=?} ${_ino:-iavail=? iuse_pct=?}"
    done
}

# ------------------------------------------------------------------
# Front-panel LED: blink GREEN for the whole run (2026-08-28).
#
# On a FIRST BOOT this window is minutes long -- OAF, the family-safe
# package, and the blocklists (~22MB once the gambling category lands) all
# download and install here -- and the router is deliberately not in its
# steady state while it happens: services restart, the WAN comes and goes,
# and led-status.sh's normal probe would flap the light RED mid-install and
# read as a fault to anyone watching the unit. Blinking green says "busy
# setting up", which is what is actually true.
#
# The marker is a one-line epoch DEADLINE, the same shape (and the same
# reader, _led_marker_unexpired) as the mesh join/adopt markers -- so if this
# process is killed hard enough to skip the trap below, the blink still ends
# on its own instead of leaving a permanently busy-looking router. 1800s is
# far longer than any real run and far shorter than "forever".
#
# The trap clears it on EVERY exit path -- success, failure, or signal --
# because the requirement is that the light SETTLES either way. led-status.sh
# then re-probes on its next pass and lands on online/offline/mesh_agent on
# the merits; a failed update deliberately does not get an error colour of
# its own (the hardware has three LEDs and red already means "no internet").
# This script has no other EXIT trap -- if one is ever added, COMBINE them in
# a single `trap` call rather than adding a second, which silently REPLACES
# the first.
FS_INSTALLING_FILE="${FS_INSTALLING_FILE:-/tmp/family-safe/installing}"
_fs_led_installing_begin() {
    mkdir -p "$(dirname "${FS_INSTALLING_FILE}")" 2>/dev/null || return 0
    _fli_now="$(date +%s 2>/dev/null)" || return 0
    case "${_fli_now}" in ''|*[!0-9]*) return 0 ;; esac
    printf '%s\n' "$(( _fli_now + 1800 ))" > "${FS_INSTALLING_FILE}" 2>/dev/null || true
}
_fs_led_installing_end() { rm -f "${FS_INSTALLING_FILE}" 2>/dev/null || true; }
trap '_fs_led_installing_end' EXIT INT TERM
_fs_led_installing_begin

# ------------------------------------------------------------------
# Debug logging window: turn ON kernel logging for family-safe's firewall
# reject/drop rules for the duration of this run, so a failure anywhere
# below (package install, blocklist apply, dnsmasq/firewall restart) leaves
# kern.warn evidence behind. Turned back OFF near Step 6, but only if the
# post-update health check actually passes — see that block for why a
# failed run deliberately leaves this on. No-op (and safe) on a router with
# no family-safe package installed yet, or before its first successful run.
# fs_set_debug_logging() is itself flash-wear-safe: this is a real nightly
# firewall reload when logging was off (the common steady-state case, since
# a healthy prior run turned it off), not a blind daily write.
# Gate on a PACKAGE-ONLY file: lib-family-safe.sh is baked into the image, so
# its presence never meant "family-safe is installed" — and running this before
# the first install created /etc/family-safe.env, which apk then kept as the
# conffile (see fs_set_debug_logging).
if [ -x /usr/lib/family-safe/install-family-safe.sh ] && [ -f /etc/family-safe.env ]; then
    (SCRIPT_DIR=/usr/lib/family-safe \
     && . /usr/lib/family-safe/lib-family-safe.sh \
     && load_config \
     && fs_set_debug_logging 1) \
        && log "event=debug_logging status=enabled" \
        || warn "event=debug_logging status=enable_failed"
fi

# ------------------------------------------------------------------
# Rollout ring gate (T2.1): derive a stable 0–99 cohort bucket from
# the device's install ID (br-lan MAC).  The manifest carries
# ROLLOUT_PERCENT; if our bucket >= ROLLOUT_PERCENT we defer this run.
# Default 100 = ship to all devices (safe backward-compat fallback).
# ------------------------------------------------------------------
_rollout_bucket() {
    local _mac="${1:-unknown}"
    local _hex
    _hex="$(printf '%s' "${_mac}" | sha256sum 2>/dev/null | cut -c1-4)"
    [ -z "${_hex}" ] && { echo 50; return; }
    printf '%d\n' "0x${_hex}" | awk '{print $1 % 100}'
}

# T2.2: version comparison helper — returns 0 if $1 >= $2 (semver).
# Uses sort -V (available in busybox ≥1.30 and GNU coreutils).
_ver_gte() {
    [ "$(printf '%s\n%s\n' "$1" "$2" | sort -V | head -1)" = "$2" ]
}

ROLLOUT_PERCENT=100
DEVICE_BUCKET="$(_rollout_bucket "${DEVICE_ID}")"
log "event=rollout_init bucket=${DEVICE_BUCKET} threshold=${ROLLOUT_PERCENT}"

mkdir -p "${TMP_DIR}"

# ------------------------------------------------------------------
# Detect package manager — opkg preferred (23.05.5/ipk fleet)
# ------------------------------------------------------------------
PKG_MGR=""
if command -v opkg >/dev/null 2>&1; then PKG_MGR="opkg"; fi
if [ -z "${PKG_MGR}" ] && command -v apk >/dev/null 2>&1; then PKG_MGR="apk"; fi

if [ -z "${PKG_MGR}" ]; then
    error "event=abort reason=no_pkgmgr"
    rm -rf "${TMP_DIR}"
    exit 1
fi

log "event=pkgmgr name=${PKG_MGR}"

# ------------------------------------------------------------------
# Cron migration + deduplication: ensure exactly one nightly cron entry
# exists, and that it invokes the runner (/usr/lib/family-safe/
# oaf-update-runner.sh) rather than this script directly.
#
# Historical reason to dedup at all: config-preserving sysupgrades re-run
# 99-family-safe, which appends a new timed entry without removing the old
# one, accumulating duplicates across flashes.
#
# Runner migration: nightly delivery of this script now goes through
# oaf-update-runner.sh (family-safe-openwrt/package/family-safe/files/
# oaf-update-runner.sh — installed by the family-safe package, so it ships
# on the SAME nightly cadence and reaches routers without a firmware flash;
# see that file's header for the full fetch/verify/fallback design). A
# router whose crontab still has the old direct entry — because it hasn't
# picked up this migration yet — must be MOVED onto the runner here, not
# merely deduplicated in place, or it never gets the nightly-delivery
# benefit at all.
#
# THE TRAP: the runner's filename is "oaf-update-runner.sh", which does NOT
# contain the literal substring "oaf-update.sh" (there is a "-runner"
# between "oaf-update" and the extension) — so reusing the old single
# `grep -c 'oaf-update\.sh'` pattern against a crontab that already has only
# the correct, lone runner entry would count ZERO matches. That would make
# this block think no nightly entry exists at all and append a SECOND cron
# line for the direct /usr/sbin/oaf-update.sh — giving two nightly runs and
# silently reverting the migration, every single night. Both entry types are
# therefore counted explicitly and independently below; neither pattern can
# match the other's line, so the two counts never overlap.
# ------------------------------------------------------------------
_CRON_FILE="/etc/crontabs/root"
_RUNNER_BIN="/usr/lib/family-safe/oaf-update-runner.sh"
# Plain `grep -c` (no `|| echo 0` fallback): when the file exists but has
# zero matching lines, `grep -c` itself already prints "0" and exits 1 — an
# `|| echo 0` on that exit status would append a SECOND "0" line, corrupting
# the value used below in arithmetic. Only a missing/unreadable file yields
# truly empty output, which the blank-check defaults to 0.
_direct_cron_count="$(grep -c 'oaf-update\.sh' "${_CRON_FILE}" 2>/dev/null)"
_runner_cron_count="$(grep -c 'oaf-update-runner\.sh' "${_CRON_FILE}" 2>/dev/null)"
case "${_direct_cron_count}" in ''|*[!0-9]*) _direct_cron_count=0 ;; esac
case "${_runner_cron_count}" in ''|*[!0-9]*) _runner_cron_count=0 ;; esac
_total_cron_count=$(( _direct_cron_count + _runner_cron_count ))

# Correct, expected, idempotent end-state: exactly one nightly entry total,
# and it is the runner entry. Anything else — zero entries, more than one of
# either kind, or a lingering direct entry alongside/instead of the runner —
# gets collapsed to a single fresh runner entry with a newly randomized
# time. When already in the correct state this whole block is a no-op, so
# repeated nightly runs never reshuffle a correctly-migrated cron time.
if [ "${_total_cron_count}" -ne 1 ] || [ "${_runner_cron_count}" -ne 1 ]; then
    log "event=cron_dedup direct_found=${_direct_cron_count} runner_found=${_runner_cron_count} action=replacing_with_single_runner_entry"
    sed -i '/oaf-update\.sh/d' "${_CRON_FILE}"
    sed -i '/oaf-update-runner\.sh/d' "${_CRON_FILE}"
    _cron_hex="$(head -c 4 /dev/urandom 2>/dev/null | md5sum | cut -c1-4)"
    _cron_r="$(printf '%d' "0x${_cron_hex}" 2>/dev/null)"
    [ -z "${_cron_r}" ] && _cron_r=7531
    _cron_offset=$(( _cron_r % 120 ))
    if [ "${_cron_offset}" -lt 30 ]; then
        _cron_hour=23; _cron_min=$(( _cron_offset + 30 ))
    elif [ "${_cron_offset}" -lt 90 ]; then
        _cron_hour=0;  _cron_min=$(( _cron_offset - 30 ))
    else
        _cron_hour=1;  _cron_min=$(( _cron_offset - 90 ))
    fi
    printf '%s\n' "${_cron_min} ${_cron_hour} * * * ${_RUNNER_BIN} >> /var/log/oaf-update.log 2>&1" >> "${_CRON_FILE}"
    /etc/init.d/cron reload 2>/dev/null || true
    log "event=cron_dedup_done new_time=${_cron_hour}:$(printf '%02d' ${_cron_min}) target=runner"
fi

# ------------------------------------------------------------------
# Step 1: Refresh package lists (do NOT full-upgrade — risky on routers)
# ------------------------------------------------------------------
_log_space "pre_pkglist_refresh"
if [ "${PKG_MGR}" = "opkg" ]; then
    _out="$(opkg update 2>&1)"; _rc=$?
    printf '%s\n' "${_out}"
    if [ "${_rc}" -eq 0 ]; then
        log "event=pkglist_refresh status=ok"
    else
        warn "event=pkglist_refresh status=fail exit=${_rc} detail=\"$(printf '%s' "${_out}" | awk 'END{print}')\""
    fi
elif [ "${PKG_MGR}" = "apk" ]; then
    _out="$(apk update 2>&1)"; _rc=$?
    printf '%s\n' "${_out}"
    if [ "${_rc}" -eq 0 ]; then
        log "event=pkglist_refresh status=ok"
    else
        warn "event=pkglist_refresh status=fail exit=${_rc} detail=\"$(printf '%s' "${_out}" | awk 'END{print}')\""
    fi
fi
_log_space "post_pkglist_refresh"

# ------------------------------------------------------------------
# Step 1b: ensure zram-swap is installed (2.4.22 — dnsmasq OOM mitigation)
#
# dnsmasq --test forks a second full parse of the whole blocklist config
# (20MB+ once gambling etc. are enabled) alongside the live daemon it is
# validating for. On a ~234MB-RAM unit that transient double footprint can
# exceed available memory, and the kernel OOM-kills the LIVE dnsmasq (not
# the disposable test fork) rather than failing the test cleanly —
# reproduced on hardware 2026-09-09 (see LOG.md), fixed by giving the
# kernel swap headroom for the spike. kmod-zram/zram-swap are stock
# OpenWrt feed packages, not part of this repo's own signed manifest set,
# so they can't ride the update_pkg() path below — install them directly
# from the feed like any other stock package, right after the pkglist
# refresh above. Idempotent (skips instantly once already active) and
# fail-open (an unreachable feed tonight just retries tomorrow rather than
# aborting the rest of the update).
# ------------------------------------------------------------------
if grep -q '^/dev/zram' /proc/swaps 2>/dev/null; then
    log "event=zram_swap_install status=already_active"
elif [ "${PKG_MGR}" = "opkg" ]; then
    _out="$(opkg install kmod-zram zram-swap 2>&1)"; _rc=$?
    if [ "${_rc}" -eq 0 ] && grep -q '^/dev/zram' /proc/swaps 2>/dev/null; then
        log "event=zram_swap_install status=ok"
    else
        warn "event=zram_swap_install status=fail exit=${_rc} detail=\"$(printf '%s' "${_out}" | awk 'END{print}')\""
    fi
elif [ "${PKG_MGR}" = "apk" ]; then
    _out="$(apk add kmod-zram zram-swap 2>&1)"; _rc=$?
    if [ "${_rc}" -eq 0 ] && grep -q '^/dev/zram' /proc/swaps 2>/dev/null; then
        log "event=zram_swap_install status=ok"
    else
        warn "event=zram_swap_install status=fail exit=${_rc} detail=\"$(printf '%s' "${_out}" | awk 'END{print}')\""
    fi
fi

# ------------------------------------------------------------------
# Step 2: Backup /etc/config/appfilter (best-effort, before any changes)
# ------------------------------------------------------------------
if [ -f /etc/config/appfilter ]; then
    cp /etc/config/appfilter "${TMP_DIR}/appfilter.bak" 2>/dev/null \
        && log "event=backup path=/etc/config/appfilter status=ok" \
        || warn "event=backup path=/etc/config/appfilter status=fail"
fi

# ------------------------------------------------------------------
# Post-update health check + auto-rollback (T2.3)
# Runs after package installs to verify the router is still healthy.
# On failure: restores appfilter config, restarts services, and alerts.
# Returns 0 (healthy) or 1 (unhealthy — rollback triggered).
# ------------------------------------------------------------------
_health_check() {
    local _attempt=0 _live=0 _dns=0
    # POLL, don't single-shot. This runs seconds after Step 5c restarts dnsmasq
    # (and, on an upgrade night, after the package install restarted it too). A
    # restart empties the cache, so the first query after it goes all the way
    # upstream through stubby/https-dns-proxy and can take a moment to settle. A
    # one-shot check here would read a perfectly healthy router as broken and
    # trigger a rollback that restores config nothing was wrong with.
    while [ "${_attempt}" -lt 6 ]; do
        _attempt=$(( _attempt + 1 ))
        _live=1
        pgrep -f rpcd >/dev/null 2>&1 || _live=0
        pgrep -x dnsmasq >/dev/null 2>&1 || _live=0
        if [ "${_live}" = "1" ] && nslookup kahf.co 127.0.0.1 >/dev/null 2>&1; then
            _dns=1
            break
        fi
        sleep 5
    done

    if [ "${_live}" = "1" ] && [ "${_dns}" = "1" ]; then
        return 0
    fi

    if [ "${_live}" = "0" ]; then
        pgrep -f rpcd  >/dev/null 2>&1 || warn "event=health_fail reason=rpcd_not_running"
        pgrep -x dnsmasq >/dev/null 2>&1 || warn "event=health_fail reason=dnsmasq_not_running"
        return 1
    fi

    # Daemons are up but nothing resolves. Distinguish "we broke DNS" from "the
    # ISP is down" before rolling back — an outage at 03:00 is not a reason to
    # restore config and declare the update bad. Probe an upstream IP directly
    # (no DNS involved) two ways: ICMP ping AND a TCP/HTTP fetch. ICMP alone is
    # not enough — plenty of WAN paths (some ISPs, mobile hotspots, hotel/
    # corporate uplinks) filter ping but pass TCP, and a router behind one of
    # those would otherwise read "ping failed" as "ISP is down" and skip a
    # rollback that a genuinely broken DNS config needs. Bias toward rollback
    # when the signal is ambiguous: only skip it when BOTH probes fail.
    _internet_up=0
    if ping -c1 -W2 1.1.1.1 >/dev/null 2>&1 || ping -c1 -W2 9.9.9.9 >/dev/null 2>&1; then
        _internet_up=1
    fi
    if [ "${_internet_up}" = "0" ]; then
        # Best-effort: never let the probe itself abort health_check.
        if wget -q --timeout=3 -O /dev/null http://1.1.1.1/ 2>/dev/null \
            || wget -q --timeout=3 -O /dev/null http://9.9.9.9/ 2>/dev/null; then
            _internet_up=1
        fi
    fi
    if [ "${_internet_up}" = "1" ]; then
        warn "event=health_fail reason=dns_not_resolving internet=up"
        return 1
    fi
    warn "event=health_degraded reason=dns_not_resolving internet=down action=none hint=isp_outage_not_a_rollback_trigger"
    return 0
}

_rollback() {
    warn "event=rollback_start reason=health_check_failed"
    # Rollback scope: only /etc/config/appfilter is restored here.
    # dhcp, firewall, and family-safe UCI are NOT rolled back — a failed
    # family-safe package upgrade leaves those configs in their new state.
    # DNS filtering is restored by reloading dnsmasq + firewall from current UCI.
    if [ -f "${TMP_DIR}/appfilter.bak" ]; then
        cp "${TMP_DIR}/appfilter.bak" /etc/config/appfilter 2>/dev/null \
            && log "event=rollback config=appfilter status=ok" \
            || warn "event=rollback config=appfilter status=fail"
    fi
    # restart, not reload: reload never re-reads /etc/dnsmasq.d/, and a rollback
    # is exactly the case where the drop-ins on disk may differ from what the
    # running daemon loaded. Prefer fs_dnsmasq_restart — if a family-safe drop-in
    # is what is keeping dnsmasq down, it quarantines them and gets DNS back
    # (filtering off is recoverable on the next download-blocklists.sh run; a
    # router with no resolver is not).
    if [ -f /usr/lib/family-safe/lib-family-safe.sh ]; then
        (SCRIPT_DIR=/usr/lib/family-safe \
         && . /usr/lib/family-safe/lib-family-safe.sh \
         && load_config \
         && fs_dnsmasq_restart "rollback") \
            || warn "event=rollback config=dnsmasq status=degraded"
    else
        /etc/init.d/dnsmasq restart 2>/dev/null || true
    fi
    /etc/init.d/firewall restart 2>/dev/null || true
    warn "event=rollback_done hint=router_should_be_filtering"
}

# ------------------------------------------------------------------
# Helper: wget with 3 attempts, 5-second delay between retries.
# Prints stderr of the last failed attempt on stdout; returns its exit code.
# ------------------------------------------------------------------
_wget_retry() {
    local _url="$1" _dest="$2" _timeout="$3"
    local _attempt=0 _out _rc=1
    while [ "${_attempt}" -lt 3 ]; do
        _attempt=$(( _attempt + 1 ))
        # No -q. uclient-fetch (the wget this ships as) gates several of its
        # own perror() diagnostics -- including "Cannot open output file",
        # the exact message that would have named a local-write failure
        # instantly instead of leaving only a bare exit=3 -- on `!quiet`.
        # -q suppressed exactly that message during the WR1205K verify-image
        # investigation this instrumentation exists for. Dropping it costs
        # nothing on the happy path: ${_out} is only ever read below, on the
        # loop's LAST iteration, and only when every attempt already failed
        # (a successful attempt returns immediately, before ${_out} is used
        # for anything). `tr` collapses any \r/\n a longer-running fetch's
        # progress output might contain into spaces, so a caller embedding
        # this in one logfmt event= line can't have it split across lines --
        # the local-open failure this targets fails before any progress
        # output exists at all, so in practice this is just belt-and-braces.
        #
        # No pipe here on purpose: `$?` after `cmd | tr ...` is tr's exit
        # status, not wget's (no `set -o pipefail` in this script, and
        # adding one here would be a bigger behavioural change than this
        # instrumentation should make) -- capture raw output first, THEN
        # sanitize it as a separate step so ${_rc} is always wget's own.
        _out="$(wget --timeout="${_timeout}" -O "${_dest}" "${_url}" 2>&1)"
        _rc=$?
        [ "${_rc}" -eq 0 ] && return 0
        _out="$(printf '%s' "${_out}" | tr '\r\n' '  ')"
        [ "${_attempt}" -lt 3 ] && sleep 5
    done
    printf '%s' "${_out}"
    return "${_rc}"
}

# ------------------------------------------------------------------
# Helper: fetch a manifest from S3.  Returns 1 if unreachable/missing.
# ------------------------------------------------------------------
fetch_manifest() {
    local url="$1"
    local dest="$2"
    local _fm_out _fm_rc
    log "event=manifest_fetch url=${url}"
    _log_space "pre_manifest_fetch"
    _fm_out="$(_wget_retry "${url}" "${dest}" "${WGET_TIMEOUT}")"
    _fm_rc=$?
    if [ "${_fm_rc}" -eq 0 ]; then
        return 0
    fi
    warn "event=manifest_fetch_fail url=${url} exit=${_fm_rc}${_fm_out:+ detail=\"${_fm_out}\"}"
    return 1
}

# ------------------------------------------------------------------
# Strict manifest parser — replaces unsafe `. manifest` sourcing.
# Only reads a known allowlist of keys; rejects any value containing
# shell metacharacters or path traversal sequences.
# Usage: parse_manifest <file> <KEY1> [KEY2 ...]
#   Sets each KEY in the current shell environment.
#   Returns 1 if any required key is absent or its value is invalid.
# ------------------------------------------------------------------
# Allowed value pattern: printable ASCII, no shell metacharacters,
# no newlines, no control chars. Max 512 chars per value.
# Note: busybox grep -E does not support {n,m} interval quantifiers.
# Use + (one-or-more) and enforce max length separately with ${#var}.
# Hyphen must be first in the class to avoid range mis-interpretation.
_MANIFEST_VALUE_RE='^[-A-Za-z0-9_./:@=+%,]+$'

parse_manifest() {
    local _pm_file="$1"; shift
    local _pm_key _pm_val _pm_raw _pm_ok
    for _pm_key in "$@"; do
        # Extract raw value: match KEY=VALUE or export KEY=VALUE lines only
        _pm_raw="$(grep -m1 "^export ${_pm_key}=\|^${_pm_key}=" "${_pm_file}" 2>/dev/null \
            | sed "s/^export ${_pm_key}=//;s/^${_pm_key}=//" \
            | tr -d "'\"")"
        if [ -z "${_pm_raw}" ]; then
            warn "event=manifest_parse_fail key=${_pm_key} reason=missing_key"
            return 1
        fi
        # Enforce max length (busybox ${#var} is portable; avoids grep {n,m}).
        if [ "${#_pm_raw}" -gt 512 ]; then
            warn "event=manifest_parse_fail key=${_pm_key} reason=value_too_long"
            return 1
        fi
        # Validate against allowlist pattern (no eval/glob/subshell chars).
        _pm_ok="$(printf '%s' "${_pm_raw}" | grep -cE "${_MANIFEST_VALUE_RE}" 2>/dev/null || echo 0)"
        if [ "${_pm_ok}" != "1" ]; then
            warn "event=manifest_parse_fail key=${_pm_key} reason=invalid_value_chars"
            return 1
        fi
        # Export safely — no eval, no subshell expansion
        export "${_pm_key}=${_pm_raw}"
    done
    return 0
}

# ------------------------------------------------------------------
# Helper: get installed version of a package
# ------------------------------------------------------------------
get_installed_version() {
    local pkg="$1"
    if [ "${PKG_MGR}" = "opkg" ]; then
        opkg list-installed 2>/dev/null | awk -v p="${pkg}" '$1==p{print $3}'
    elif [ "${PKG_MGR}" = "apk" ]; then
        apk info "${pkg}" 2>/dev/null | awk 'NR==1{print $1}' | sed "s/^${pkg}-//"
    fi
}

# ------------------------------------------------------------------
# Helper: download and install a package
#   update_pkg <pkgname> <manifest_ver> <url> [extra_opkg_flags]
# ------------------------------------------------------------------
update_pkg() {
    local pkg="$1"
    local manifest_ver="$2"
    local url="$3"
    local extra_flags="${4:-}"
    local pkg_file
    pkg_file="${TMP_DIR}/$(basename "${url}")"

    installed_ver="$(get_installed_version "${pkg}")"

    if [ "${FORCE}" = "0" ] && [ -n "${installed_ver}" ] && [ "${installed_ver}" = "${manifest_ver}" ]; then
        log "event=pkg_check pkg=${pkg} installed=${installed_ver} latest=${manifest_ver} status=current"
        return 0
    fi

    # T9d: monotonic version floor — refuse to install an older version (downgrade
    # protection). _ver_gte returns 0 when installed >= manifest, i.e. a rollback/
    # poisoned manifest would push to a lower version. --force bypasses for lab use.
    if [ "${FORCE}" = "0" ] && [ -n "${installed_ver}" ] \
       && [ "${installed_ver}" != "${manifest_ver}" ] \
       && _ver_gte "${installed_ver}" "${manifest_ver}"; then
        warn "event=downgrade_refused pkg=${pkg} installed=${installed_ver} manifest=${manifest_ver}"
        return 0
    fi

    if [ -n "${installed_ver}" ]; then
        log "event=pkg_update pkg=${pkg} from=${installed_ver} to=${manifest_ver}"
    else
        log "event=pkg_install pkg=${pkg} version=${manifest_ver}"
    fi

    local dl_out dl_rc inst_out inst_rc
    dl_out="$(_wget_retry "${url}" "${pkg_file}" "${WGET_DOWNLOAD_TIMEOUT}")"
    dl_rc=$?
    if [ "${dl_rc}" -ne 0 ]; then
        warn "event=pkg_dl_fail pkg=${pkg} url=${url} exit=${dl_rc}${dl_out:+ detail=\"${dl_out}\"}"
        rm -f "${pkg_file}"
        return 1
    fi

    # T1.5: Verify package signature before install.
    # Pub key baked into firmware at /etc/oaf-pkg-signing.pub (prod key only on prod builds).
    local pkg_pubkey="/etc/oaf-pkg-signing.pub"
    local pkg_sig="${pkg_file}.sig"
    if [ -f "${pkg_pubkey}" ] && command -v usign >/dev/null 2>&1; then
        local sig_out sig_rc
        sig_out="$(_wget_retry "${url}.sig" "${pkg_sig}" 15)"
        sig_rc=$?
        if [ "${sig_rc}" -ne 0 ] || [ ! -s "${pkg_sig}" ]; then
            warn "event=pkg_sig_fetch_fail pkg=${pkg} reason=no_signature_at_${url}.sig"
            rm -f "${pkg_file}" "${pkg_sig}"
            return 1
        fi
        if ! usign -V -m "${pkg_file}" -p "${pkg_pubkey}" -x "${pkg_sig}" >/dev/null 2>&1; then
            warn "event=pkg_sig_invalid pkg=${pkg} url=${url} reason=signature_verification_failed"
            rm -f "${pkg_file}" "${pkg_sig}"
            return 1
        fi
        log "event=pkg_sig_ok pkg=${pkg}"
        rm -f "${pkg_sig}"
    else
        log "event=pkg_sig_skip pkg=${pkg} reason=no_pubkey_or_usign_missing pubkey=${pkg_pubkey}"
    fi

    if [ "${PKG_MGR}" = "opkg" ]; then
        # shellcheck disable=SC2086
        inst_out="$(opkg install --force-reinstall ${extra_flags} "${pkg_file}" 2>&1)"
        inst_rc=$?
        printf '%s\n' "${inst_out}"
        if [ "${inst_rc}" -eq 0 ]; then
            log "event=pkg_ok pkg=${pkg} version=${manifest_ver}"
        else
            warn "event=pkg_install_fail pkg=${pkg} version=${manifest_ver} exit=${inst_rc} detail=\"$(printf '%s' "${inst_out}" | awk 'END{print}')\""
        fi
    elif [ "${PKG_MGR}" = "apk" ]; then
        # --allow-untrusted is ALWAYS required for local .apk installs.
        # Our packages are authenticated by the usign .sig layer (verified above)
        # which is entirely separate from apk's native per-package keystore model
        # (/etc/apk/keys/).  We do NOT sign with apk's native signing format, so
        # apk would reject the package without --allow-untrusted regardless of
        # whether oaf-pkg-signing.pub is present.  The usign verification above
        # is the authenticity gate; --allow-untrusted here only bypasses the
        # apk-native trust check, not our usign layer.
        inst_out="$(apk add --allow-untrusted "${pkg_file}" 2>&1)"
        inst_rc=$?
        printf '%s\n' "${inst_out}"
        if [ "${inst_rc}" -eq 0 ]; then
            log "event=pkg_ok pkg=${pkg} version=${manifest_ver}"
        else
            warn "event=pkg_install_fail pkg=${pkg} version=${manifest_ver} exit=${inst_rc} detail=\"$(printf '%s' "${inst_out}" | awk 'END{print}')\""
        fi
    fi

    rm -f "${pkg_file}"
}

# ------------------------------------------------------------------
# Step 3: OAF packages (manifest-oaf.env)
# ------------------------------------------------------------------
# NOTE on from-source builds (WR3000K since lab/build-openwrt-source.sh):
# kmod-oaf is compiled IN-TREE against that build's own kernel and baked into
# the image, because this manifest's kmod-oaf is built by OpenAppFilter's CI
# against the OFFICIAL stock kernel and can never apk-install onto a
# from-source kernel (apk hashes the whole kernel source tree — see
# docs/openwrt-source-build.md's "OAF kernel-ABI" section). The kmod guard
# below already handles that correctly and permanently: it logs kmod_skip and
# moves on, leaving the baked-in kmod untouched.
#
# The appfilter/luci-app-oaf USERSPACE packages below are deliberately NOT
# gated on that — they carry no kernel ABI, so a newer build of them installs
# and runs fine alongside an older baked-in kmod-oaf, and this is the ONLY
# path by which an OpenAppFilter userspace change reaches a deployed router
# without reflashing it. Do not "fix" the apparent asymmetry by skipping them
# too; that would silently strand every future OAF userspace release.
# A change to oaf.ko ITSELF cannot ship this way on a from-source device —
# that needs a new full-source build + sysupgrade, by construction.
MANIFEST_OAF="${TMP_DIR}/manifest-oaf.env"
if fetch_manifest "${MANIFEST_OAF_URL}" "${MANIFEST_OAF}"; then
    if ! parse_manifest "${MANIFEST_OAF}" \
            APPFILTER_VERSION APPFILTER_URL \
            LUCI_VERSION LUCI_URL \
            KMOD_VERSION KMOD_URL KMOD_KERNEL_ABI \
            OPENWRT_ARCH BUILD_COMMIT BUILD_DATE; then
        warn "event=manifest_invalid url=${MANIFEST_OAF_URL} reason=parse_failed"
    else
        # Rollout gate: parse optional ROLLOUT_PERCENT from manifest.
        _rp_raw="$(grep -m1 '^ROLLOUT_PERCENT=' "${MANIFEST_OAF}" 2>/dev/null | sed 's/^ROLLOUT_PERCENT=//')"
        if [ -n "${_rp_raw}" ] && [ "${_rp_raw}" -ge 0 ] 2>/dev/null && [ "${_rp_raw}" -le 100 ] 2>/dev/null; then
            ROLLOUT_PERCENT="${_rp_raw}"
        fi
        if [ "${FORCE}" = "0" ] && [ "${DEVICE_BUCKET}" -ge "${ROLLOUT_PERCENT}" ]; then
            log "event=rollout_defer bucket=${DEVICE_BUCKET} threshold=${ROLLOUT_PERCENT} hint=will_retry_next_run"
            rm -rf "${TMP_DIR}"
            exit 0
        fi
        log "event=oaf_manifest commit=${BUILD_COMMIT:-unknown} date=${BUILD_DATE:-unknown} kmod_ver=${KMOD_VERSION:-?} kmod_abi=${KMOD_KERNEL_ABI:-?} appfilter_ver=${APPFILTER_VERSION:-?} luci_ver=${LUCI_VERSION:-?} rollout_pct=${ROLLOUT_PERCENT}"

        # T2.2: Version stepping gate — honour MIN_FROM_VERSION so a long-offline
        # router doesn't blind-jump over a known-bad intermediate version.
        SKIP_OAF=0
        _min_from="$(grep -m1 '^MIN_FROM_VERSION=' "${MANIFEST_OAF}" 2>/dev/null | sed 's/^MIN_FROM_VERSION=//')"
        if [ -n "${_min_from}" ]; then
            _cur_oaf="$(opkg list-installed 2>/dev/null | awk '$1=="appfilter"{print $3}')"
            if [ -z "${_cur_oaf}" ] && [ "${PKG_MGR}" = "apk" ]; then
                _cur_oaf="$(apk info 2>/dev/null | awk '/^appfilter-/{gsub(/^appfilter-/,""); print; exit}')"
            fi
            if [ -n "${_cur_oaf}" ] && ! _ver_gte "${_cur_oaf}" "${_min_from}"; then
                warn "event=version_pin_defer installed=${_cur_oaf} min_from_version=${_min_from} hint=need_intermediate_update"
                SKIP_OAF=1
            fi
        fi

        # kmod guard: compare ABI string from manifest against installed kernel
        # KMOD_KERNEL_ABI from manifest is the exact "kernel (= X)" dep value,
        # e.g. "5.15.167-1-03ba5b5fee..."; the router reports the same via opkg.
        ROUTER_KERNEL_VER="$(opkg list-installed 2>/dev/null | awk '$1=="kernel"{print $3}')"
        if [ -z "${ROUTER_KERNEL_VER}" ] && [ "${PKG_MGR}" = "apk" ]; then
            ROUTER_KERNEL_VER="$(uname -r)"
        fi

        # Architecture guard: KMOD_KERNEL_ABI on apk systems is only the bare
        # kernel version (e.g. "6.12.87"), stripped of the target-specific
        # build hash -- the same version number is shared across different
        # architectures/targets on the same OpenWrt release, so it alone
        # cannot distinguish e.g. aarch64 (mediatek/filogic, real WR3000K)
        # from x86_64 (QEMU verify-image test target). Cross-check uname -m
        # against the manifest's OPENWRT_ARCH so a same-version, wrong-arch
        # kmod-oaf is skipped instead of failing apk's dependency resolver.
        ROUTER_ARCH="$(uname -m)"
        case "${OPENWRT_ARCH:-}" in
            "${ROUTER_ARCH}"*) ARCH_OK=1 ;;
            *) ARCH_OK=0 ;;
        esac

        if [ "${SKIP_OAF}" = "0" ]; then
            if [ "${ARCH_OK}" = "1" ] && [ -n "${KMOD_KERNEL_ABI:-}" ] && [ -n "${ROUTER_KERNEL_VER}" ] \
               && [ "${ROUTER_KERNEL_VER}" = "${KMOD_KERNEL_ABI}" ]; then
                update_pkg "kmod-oaf" "${KMOD_VERSION}" "${KMOD_URL}"
            else
                warn "event=kmod_skip reason=abi_mismatch router_kernel=${ROUTER_KERNEL_VER:-unknown} manifest_abi=${KMOD_KERNEL_ABI:-unknown} router_arch=${ROUTER_ARCH} manifest_arch=${OPENWRT_ARCH:-unknown} hint=firmware_upgrade_required"
            fi

            # appfilter and luci-app-oaf are also arch-specific compiled
            # packages (not arch=all) -- same guard applies to both, since
            # they'd fail apk dependency resolution the same way kmod-oaf did.
            if [ "${ARCH_OK}" = "1" ]; then
                update_pkg "appfilter"    "${APPFILTER_VERSION}" "${APPFILTER_URL}"
                update_pkg "luci-app-oaf" "${LUCI_VERSION}"      "${LUCI_URL}"
            else
                warn "event=oaf_userspace_skip reason=arch_mismatch router_arch=${ROUTER_ARCH} manifest_arch=${OPENWRT_ARCH:-unknown} hint=qemu_test_target_or_firmware_upgrade_required"
            fi
        fi

        # Fix: appfilter package ships feature.cfg with CN app IDs (3001=TikTok).
        # Overwrite with the EN version where 3001=YouTube, matching the app catalog.
        if [ -f /etc/appfilter/feature_en.cfg ]; then
            cp /etc/appfilter/feature_en.cfg /etc/appfilter/feature.cfg \
                && log "event=feature_cfg_fix source=feature_en.cfg" \
                || warn "event=feature_cfg_fix status=fail"
        fi

        # Fix: user_mode is not in the default UCI config; writing an empty string
        # to /proc/sys/oaf/user_mode causes EINVAL in oaf_rule reload.
        uci -q get appfilter.global.user_mode >/dev/null 2>&1 \
            || { uci set appfilter.global.user_mode=0; uci commit appfilter; }

        # Restart the appfilter daemon so the new binary takes effect --
        # only if it's actually installed (skipped above on arch mismatch,
        # e.g. the QEMU x86_64 verify-image target).
        if [ -x /etc/init.d/appfilter ]; then
            _restart_out="$(/etc/init.d/appfilter restart 2>&1)"
            _restart_rc=$?
            if [ "${_restart_rc}" -eq 0 ]; then
                log "event=service_restart name=appfilter status=ok"
            else
                warn "event=service_restart name=appfilter status=fail exit=${_restart_rc}${_restart_out:+ detail=\"${_restart_out}\"}"
            fi
        else
            log "event=service_restart name=appfilter status=skip reason=not_installed"
        fi
    fi
else
    warn "event=oaf_skip reason=manifest_unavailable"
fi

# ------------------------------------------------------------------
# Step 4: family-safe package (manifest-familysafe.env)
# ------------------------------------------------------------------
# Tell the package's postinst (install-family-safe.sh) not to fire its own
# background blocklist download: Step 5c below refreshes the lists synchronously
# and then restarts dnsmasq exactly once. Without this the postinst's background
# run would duplicate the fetch, collide on the blocklist lock, and restart
# dnsmasq a second time at an unpredictable moment — right when Step 6's health
# check is looking. Exported so it reaches the postinst child process.
FS_BLOCKLIST_REFRESH_DEFERRED=1
export FS_BLOCKLIST_REFRESH_DEFERRED

MANIFEST_FS="${TMP_DIR}/manifest-familysafe.env"
if fetch_manifest "${MANIFEST_FS_URL}" "${MANIFEST_FS}"; then
    if ! parse_manifest "${MANIFEST_FS}" \
            FAMILYSAFE_VERSION FAMILYSAFE_URL \
            BUILD_COMMIT BUILD_DATE; then
        warn "event=manifest_invalid url=${MANIFEST_FS_URL} reason=parse_failed"
    else
        log "event=fs_manifest commit=${BUILD_COMMIT:-unknown} date=${BUILD_DATE:-unknown} version=${FAMILYSAFE_VERSION:-?}"

        # --force-depends: package may depend on dnsmasq-full; avoid blocking
        # if the user has a custom dnsmasq variant that satisfies the function.
        update_pkg "family-safe" "${FAMILYSAFE_VERSION}" "${FAMILYSAFE_URL}"
    fi
else
    warn "event=fs_skip reason=manifest_unavailable"
fi

# ------------------------------------------------------------------
# Step 5: Refresh block page IP in dnsmasq (blockpage.conf)
# ------------------------------------------------------------------
if [ -f /usr/lib/family-safe/lib-family-safe.sh ]; then
    (SCRIPT_DIR=/usr/lib/family-safe \
     && . /usr/lib/family-safe/lib-family-safe.sh \
     && load_config \
     && update_block_page) \
        && log "event=blockpage_refresh status=ok" \
        || warn "event=blockpage_refresh status=fail"
fi

# ------------------------------------------------------------------
# Step 5b: Ensure default root password (MAC-derived) — migration path
# for routers already in the field that are still sitting on a blank
# root password (flashed before this default existed, or the app hasn't
# connected yet). No-ops instantly once any password exists.
# ------------------------------------------------------------------
if [ -f /usr/lib/family-safe/lib-family-safe.sh ]; then
    (SCRIPT_DIR=/usr/lib/family-safe \
     && . /usr/lib/family-safe/lib-family-safe.sh \
     && ensure_default_root_password) \
        && log "event=default_root_password_check status=ok" \
        || warn "event=default_root_password_check status=fail"
fi

# ------------------------------------------------------------------
# Step 5b2: Apply locale (country/timezone) build defaults — migration path
# for routers already in the field. Findings #7/#8, PRODUCTION_FIX_PLANS.md
# "7 + 8. Country and timezone".
#
# NIGHTLY-ONLY, deliberately: apply_locale_build_defaults() can write
# wireless.*.country for the FIRST time on a router that has never had one
# (every unit flashed before this feature existed), which restarts the
# radio and briefly drops every client. This is the ONE call site for that
# function in the whole codebase — never call it from
# install-family-safe.sh or a WAN-ifup hotplug, both of which can fire
# outside this randomized low-traffic window (23:30-01:30 UTC at each
# router's own local time once this same migration has set its timezone —
# see 99-family-safe/CLAUDE.md for the cron-window consequence). On a
# router that already has a country set (a fresh unit, or a genuine
# customer/operator value) this is a same-value, no-restart no-op — see the
# function's own doc comment in lib-family-safe.sh.
#
# PRODUCTION_READINESS_REVIEW.md FINDING #3: apply_locale_build_defaults
# takes the same GEO_LOCK_FILE geo-locale.sh (WAN-ifup path) uses, and is
# non-blocking (tries once, never waits) — on contention it logs and
# `return 0` rather than making this nightly run wait on a lock some other
# script is holding right now. "status=ok" below therefore covers BOTH "ran
# and (maybe) applied a default" and "skipped this cycle, lock was held" —
# both are a normal, expected outcome and either way there is always a
# tomorrow-night retry; see `logread | grep 'locale:'` for which one
# actually happened.
# ------------------------------------------------------------------
if [ -f /usr/lib/family-safe/lib-family-safe.sh ]; then
    (SCRIPT_DIR=/usr/lib/family-safe \
     && . /usr/lib/family-safe/lib-family-safe.sh \
     && load_config \
     && apply_locale_build_defaults) \
        && log "event=locale_build_defaults status=ok" \
        || warn "event=locale_build_defaults status=fail"
fi

# ------------------------------------------------------------------
# Step 5c: Refresh the DNS blocklists, then restart dnsmasq — but only if
# something actually needs it.
#
# The refresh runs here rather than from its own cron entry so it always lands
# immediately after the package update that may have changed the category
# manifest, the category list, or download-blocklists.sh itself — and so there is
# at most one point in the night where dnsmasq is bounced. Lists are therefore at
# most a day old; the old schedule was weekly (Sunday ~03:xx), which meant a
# newly published category or an overlay wipe could sit unfixed for up to 7 days.
# install-family-safe.sh removes that legacy cron line on update.
#
# Synchronous, not --bg: the restart below must happen after the confs are
# written. --no-restart because that restart is ours — otherwise the nightly path
# would bounce dnsmasq twice within seconds, and every extra bounce is another
# window for a health check to misfire.
#
# A failed refresh is not fatal: every list is validated and only swapped in
# atomically, so a fetch failure keeps the previous good copy.
#
# T4-4b: download-blocklists.sh now compares a tiny detached .sig before ever
# downloading a full list, so a steady-state night where CDN content hasn't
# changed does real work but writes nothing under /etc/dnsmasq.d. It signals
# that via a status marker file (FS_BLOCKLIST_STATUS_FILE, default
# /tmp/family-safe-blocklist-refresh.status) — NOT via its exit code, which
# stays exactly 0/non-zero as before so this stays byte-for-byte compatible
# with any caller (including an older oaf-update.sh) that only checks
# `if download-blocklists.sh ...; then`.
#
# Skipping the restart below on a confirmed-unchanged night is safe because
# the other two things that used to rely on this being unconditional already
# restart dnsmasq themselves, independently, only when THEY actually change
# something:
#   - a family-safe package install/upgrade (Step 4) restarts from its own
#     postinst (install-family-safe.sh -> fs_dnsmasq_restart "install")
#   - the block-page refresh (Step 5, update_block_page()) restarts itself,
#     and only when its own conf actually changed (_install_if_changed)
# So the only thing this restart still needs to cover is the blocklist
# refresh directly above — and it now tells us whether that happened.
#
# Safe-default rule: missing marker file, unreadable, or any content other
# than the exact string "unchanged" -> treat as changed and RESTART. Only an
# exact "unchanged" read, AND a status=ok refresh, skips it. A missed restart
# means a router silently keeps serving stale blocklists, which is far worse
# than one unnecessary restart — so every ambiguous case below falls to
# restarting.
# ------------------------------------------------------------------
_BL_STATUS_FILE="${FS_BLOCKLIST_STATUS_FILE:-/tmp/family-safe-blocklist-refresh.status}"
_bl_skip_restart=0
if [ -x /usr/lib/family-safe/download-blocklists.sh ]; then
    # Belt-and-braces against a script downgrade: download-blocklists.sh
    # itself resets this marker at the very start of its own run, but that
    # only helps if the copy that actually executes is new enough to contain
    # that reset. oaf-update.sh is fetched from the CDN independently of the
    # family-safe package (see CLAUDE.md "Backward Compatibility"), so an
    # older download-blocklists.sh could in theory run here without ever
    # having heard of the marker at all. Removing it HERE, immediately before
    # invoking download-blocklists.sh, means the marker is provably absent
    # unless THIS run's download-blocklists.sh writes "unchanged" — so an old
    # downloader that never touches the marker correctly falls through to the
    # safe default (missing file -> restart) instead of ever reading a stale
    # "unchanged" left over from some earlier new-downloader run.
    rm -f "${_BL_STATUS_FILE}" 2>/dev/null || true
    if /usr/lib/family-safe/download-blocklists.sh --no-restart >/dev/null 2>&1; then
        log "event=blocklist_refresh status=ok"
        # A failed `cat` (missing/unreadable file) leaves _bl_status empty,
        # which never matches "unchanged" below -> falls through to restart.
        _bl_status="$(cat "${_BL_STATUS_FILE}" 2>/dev/null || true)"
        if [ "${_bl_status}" = "unchanged" ]; then
            _bl_skip_restart=1
        fi
    else
        warn "event=blocklist_refresh status=fail hint=previous_lists_kept"
    fi
fi

# `/etc/init.d/dnsmasq reload` never re-reads /etc/dnsmasq.d/ (OpenWrt's
# reload_service() is rc_procd + SIGHUP; procd does not track confdir contents
# and dnsmasq's SIGHUP handler does not re-read configuration). So any conf
# written tonight by the refresh above needs a restart to take effect — unless
# $_bl_skip_restart says nothing was actually written.
#
# This is the ONLY conditional dnsmasq restart in the nightly path — the
# refresh above ran with --no-restart, and install-family-safe.sh skips its
# bootstrap download when FS_BLOCKLIST_REFRESH_DEFERRED=1. Keeping bounces to
# at most one matters: every bounce is a few seconds where a health check
# (ours below, or a lab/CI smoke check) can misread a settling resolver as a
# broken one. The on-device DNS detectors are not at risk — dns-upstream-
# guard.sh and dns-health-monitor.sh both probe the stubby/https-dns-proxy
# loopback ports directly, never dnsmasq's :53, so a dnsmasq restart can
# never cause a transport downgrade.
#
# fs_dnsmasq_restart is fail-safe on its own (it will not restart into a config
# that fails `dnsmasq --test`, and it quarantines family-safe's drop-ins and
# recovers if a config that passed --test still kills the daemon). This runs
# BEFORE Step 6 so that anything it could not fix trips the health check and
# the existing rollback, rather than leaving the router without a resolver.
if [ "${_bl_skip_restart}" = "1" ]; then
    log "event=dnsmasq_nightly_restart status=skipped reason=blocklists_unchanged"
elif [ -f /usr/lib/family-safe/lib-family-safe.sh ]; then
    (SCRIPT_DIR=/usr/lib/family-safe \
     && . /usr/lib/family-safe/lib-family-safe.sh \
     && load_config \
     && fs_dnsmasq_restart "nightly oaf-update") \
        && log "event=dnsmasq_nightly_restart status=ok" \
        || warn "event=dnsmasq_nightly_restart status=fail"
else
    # No family-safe package installed (OAF-only router): plain restart.
    # _bl_skip_restart is always 0 here — download-blocklists.sh doesn't even
    # exist on an OAF-only router, so the block above never sets it to 1.
    /etc/init.d/dnsmasq restart 2>/dev/null || true
fi

# ------------------------------------------------------------------
# Step 5d: IP (CIDR) blocklist refresh — Tor / VPN / open-proxy / "bad"
# ranges, the firewall-layer sibling of Step 5c above. download-ip-
# blocklists.sh only downloads and validates (see its own "job boundary"
# header comment — it never touches nftables itself), so a changed list
# still needs an explicit apply to take effect. fs_ip_blocklist_apply()
# rebuilds the guard's nft chain directly from whatever is now on disk plus
# the block_*_ip UCI flags — deliberately NOT `/etc/init.d/firewall
# restart`/`reload`, so calling it here can never become a second "firewall
# reload" for the nightly run to worry about.
#
# Independent of Step 5c on purpose, same as the two downloaders' separate
# locks: a failure here must neither block nor be blocked by the DNS
# blocklist step above, and vice versa.
#
# Same safe-default contract as Step 5c's _bl_skip_restart: only a
# confirmed successful run AND an exact "unchanged" marker skips the apply;
# a failed download, a missing marker, or any other content all fall
# through to applying (the same "ambiguous case -> act" reasoning as Step
# 5c — a missed apply means a router silently keeps enforcing a stale IP
# blocklist, worse than one unnecessary rebuild). Removing the marker HERE
# first is the same belt-and-braces as Step 5c's rm, for the same
# independently-fetched-oaf-update.sh reason given there.
_IPBL_STATUS_FILE="${FS_IP_BLOCKLIST_STATUS_FILE:-/tmp/family-safe-ip-blocklist-refresh.status}"
_ipbl_skip_apply=0
if [ -x /usr/lib/family-safe/download-ip-blocklists.sh ]; then
    rm -f "${_IPBL_STATUS_FILE}" 2>/dev/null || true
    if /usr/lib/family-safe/download-ip-blocklists.sh --no-restart >/dev/null 2>&1; then
        log "event=ip_blocklist_refresh status=ok"
        _ipbl_status="$(cat "${_IPBL_STATUS_FILE}" 2>/dev/null || true)"
        [ "${_ipbl_status}" = "unchanged" ] && _ipbl_skip_apply=1
    else
        warn "event=ip_blocklist_refresh status=fail hint=previous_lists_kept"
    fi

    if [ "${_ipbl_skip_apply}" = "1" ]; then
        log "event=ip_blocklist_apply status=skipped reason=lists_unchanged"
    elif [ -f /usr/lib/family-safe/lib-family-safe.sh ]; then
        (SCRIPT_DIR=/usr/lib/family-safe \
         && . /usr/lib/family-safe/lib-family-safe.sh \
         && load_config \
         && fs_ip_blocklist_apply) \
            && log "event=ip_blocklist_apply status=ok" \
            || warn "event=ip_blocklist_apply status=fail"
    else
        # Unreachable in practice today (both scripts ship in the same
        # family-safe package) but kept for the same defense-in-depth
        # reason as Step 5c's own library-presence check.
        log "event=ip_blocklist_apply status=skipped reason=no_family_safe_package"
    fi
fi

# ------------------------------------------------------------------
# Step 6: Post-update health check
# ------------------------------------------------------------------
if _health_check; then
    # Update completed and the router is healthy: turn debug logging back
    # off. Deliberately skipped on the failure branch below — an admin
    # investigating a rollback needs tonight's kern.warn lines, and the
    # NEXT run's enable step (top of this script) re-arms logging anyway
    # before it touches anything again.
    if [ -f /usr/lib/family-safe/lib-family-safe.sh ]; then
        (SCRIPT_DIR=/usr/lib/family-safe \
         && . /usr/lib/family-safe/lib-family-safe.sh \
         && load_config \
         && fs_set_debug_logging 0) \
            && log "event=debug_logging status=disabled" \
            || warn "event=debug_logging status=disable_failed"
    fi
else
    warn "event=debug_logging status=left_enabled reason=health_check_failed"
    _rollback
fi

# ------------------------------------------------------------------
# Cleanup
# ------------------------------------------------------------------
rm -rf "${TMP_DIR}"
log "event=finish ts=$(date -u +%Y-%m-%dT%H:%M:%SZ)"
exit 0
