#!/bin/sh
#
# S99upgrade - Commit DCENT_OS first-boot recovery flag (am3-aml)
#
# `.78` stock-return is recovery-flag 0x02 â†’ recover_to_stock.
# firstboot WAL is a DCENT companion write, not a second revert arm.
#
#   (a) Recovery flag byte at mtd5 LOCAL offset from system.sh
#       (.78 dmesg physical mtd5 0x06700000 â†’ local 0x04D00000).
#       Never default to 0x05300000 (that assumed size-sum 0x06100000).
#       Values:
#         0x1  RECOVERY_FLAG_INSTALLED   (host installer)
#         0x2  RECOVERY_FLAG_FIRST_BOOT  (U-Boot, before kernel handoff)
#         0x3  RECOVERY_FLAG_SUCCESSFUL  (this script, on healthy boot)
#       If a reboot happens while flag still = 0x2, U-Boot does NOT
#       bootm mtd2 stock_system. `.78` bootcmd falls through to
#       recover_to_stock (recover_env + nand erase.part nvdata + reset).
#       We promote 0x2 -> 0x3 only after health checks pass.
#
#   (b) U-Boot env `firstboot` is a DCENT S99 WAL companion only.
#       `.78` bootcmd never reads firstboot. Stock-return is flag 0x02
#       -> recover_to_stock. fw_setenv firstboot=0 is not the revert disarm.
#
# CRITICAL fail-closed: if libubootenv-tools (fw_setenv/fw_printenv) is
# missing, log a clear ERROR and exit non-zero. Silent ignore = brick risk
# (see feedback_am2_dcentos_fw_setenv_missing.md â€” this exact bug bricked
# am2 and is being prevented here per O.3 spec).
#
# fw_env.config layout used here lives at /etc/fw_env.config and points at
# /dev/nand_env (Amlogic NAND env chardev, single 64KB instance). Live .78
# validation on 2026-04-29 confirmed standard U-Boot env format:
# CRC32 + null-terminated key=value entries, fw_printenv/fw_setenv compatible.
#
# Wave 10 B2: write-ahead log eliminates the split-state window between
# recovery-flag commit (mtd5 byte 0x02 -> 0x03) and U-Boot env commit
# (firstboot=0). If commit_recovery_flag succeeds but clear_uboot_env crashes
# / loses power / EINTRs, /data/.firstboot-pending remains as a replay
# intent. On every subsequent boot, replay_pending_env_clear() runs before
# any recovery-flag check and idempotently re-attempts fw_setenv firstboot=0
# until it succeeds. See docs/dev/2026-04-30-wave10-protected/B2-s99upgrade-wal.md.
#

HEALTH_OK=true

[ -f /lib/functions/system.sh ] && . /lib/functions/system.sh

# Local flag comes from system.sh mtd_device_offset (adds the 6 MiB
# mtd0â†’mtd1 hole). Refuse the naive 0x05300000 size-sum pairing.
RECOVERY_FLAG_OFFSET=${DCENTOS_LOCAL_RECOVERY_FLAGS_OFFSET:-}
if [ -z "$RECOVERY_FLAG_OFFSET" ]; then
    echo "ERROR: S99upgrade: DCENTOS_LOCAL_RECOVERY_FLAGS_OFFSET unset; refuse naive 0x05300000" >&2
    HEALTH_OK=false
fi
RECOVERY_MTD=${DCENTOS_SYSTEM_MTD:-/dev/mtd5}
RECOVERY_ERASESIZE_EXPECTED=${DCENTOS_RECOVERY_ERASESIZE_EXPECTED:-131072}
RECOVERY_STAGING_ROOT=${DCENTOS_RECOVERY_STAGING_ROOT:-/data}
DCENT_RELEASE_IMAGE_MARKER=/etc/dcentos/release-image
MUTATION_POLICY_FILE=${DCENTOS_MUTATION_POLICY_FILE:-/etc/dcentos/mutation_policy}
MUTATION_POLICY_HELPER=${DCENTOS_MUTATION_POLICY_HELPER:-/usr/libexec/dcentos/mutation-policy.sh}

# Wave 10 B2: write-ahead log file. Existence == "I owe a fw_setenv firstboot=0".
# Lives in /data/ (persistent overlay) so it survives reboots. Removed only after
# fw_printenv confirms firstboot=0 was actually written. /data/ is mounted by
# S40 (well before S99), so this is always reachable when S99upgrade runs.
FIRSTBOOT_PENDING_FILE=${DCENTOS_FIRSTBOOT_PENDING_FILE:-/data/.firstboot-pending}
# Live SKU marker. Override is host-test only (WAL durability harness).
BOARD_TARGET_FILE=${DCENTOS_BOARD_TARGET_FILE:-/etc/dcentos/board_target}
# Independent live platform marker. Neither package metadata nor board_target
# alone is authority for raw NAND mutation.
PLATFORM_FILE=${DCENTOS_PLATFORM_FILE:-/etc/dcentos-platform}

wait_for_http() {
    LABEL="$1"
    URL="$2"
    LIMIT="$3"
    [ -n "$LIMIT" ] || LIMIT=30

    echo "  Waiting for $LABEL HTTP health (up to ${LIMIT}s)..."
    for i in $(seq 1 "$LIMIT"); do
        if command -v wget > /dev/null 2>&1; then
            if wget -q -T 2 -O /dev/null "$URL" 2>/dev/null; then
                echo "  [OK] $LABEL HTTP health"
                return 0
            fi
        fi
        sleep 1
    done

    echo "  [FAIL] $LABEL HTTP health did not respond at $URL"
    return 1
}

# W8 parity (ported from board/zynq/.../S99upgrade): the length of the
# boot-success window the daemon must survive as the SAME PID before its slot is
# committed (catches a bind-then-exit failure).
MIN_HEALTHY_UPTIME_S=${DCENTOS_BOOT_SUCCESS_WINDOW_S:-25}

# Real (not socket-bind) health verdict from /api/system/health. Returns
# "healthy" (live daemon, uptime>0), "unhealthy" (reachable but zero/dead
# uptime -> positive failure), or "unknown" (no decisive signal -> soft-pass).
daemon_real_health_verdict() {
    URL="http://127.0.0.1:8080/api/system/health"
    if ! command -v wget > /dev/null 2>&1; then
        echo "unknown"
        return 0
    fi
    BODY=$(wget -q -T 4 -O - "$URL" 2>/dev/null)
    if [ -z "$BODY" ]; then
        echo "unknown"
        return 0
    fi
    UPTIME=$(printf '%s' "$BODY" \
        | tr ',' '\n' \
        | sed -n 's/.*"uptime_s"[[:space:]]*:[[:space:]]*\([0-9][0-9]*\).*/\1/p' \
        | head -1)
    if [ -z "$UPTIME" ]; then
        echo "unknown"
        return 0
    fi
    if [ "$UPTIME" -gt 0 ] 2>/dev/null; then
        echo "healthy"
    else
        echo "unhealthy"
    fi
    return 0
}

check_health() {
    # S40network runs udhcpc in background, so DHCP may not complete before S99.
    echo "  Waiting for network (up to 60s)..."
    for i in $(seq 1 60); do
        if ip addr show eth0 2>/dev/null | grep -q 'inet '; then
            break
        fi
        sleep 1
    done

    if ! ip addr show eth0 2>/dev/null | grep -q 'inet '; then
        echo "  [FAIL] eth0 has no IP address after 60s"
        HEALTH_OK=false
        return
    fi
    echo "  [OK] Network interface has IP"

    SSH_LISTENING=false
    if ! netstat -tln 2>/dev/null | grep -q ':22 '; then
        if [ -f "$DCENT_RELEASE_IMAGE_MARKER" ]; then
            echo "  [OK] SSH intentionally locked on release image until owner credentials or authorized_keys are provisioned"
        else
            echo "  [FAIL] SSH server not listening on port 22"
            HEALTH_OK=false
            return
        fi
    else
        SSH_LISTENING=true
        echo "  [OK] SSH server listening"
    fi

    if grep -q '\-s' /etc/default/dropbear 2>/dev/null; then
        if [ ! -s /root/.ssh/authorized_keys ] && [ ! -s /data/keys/dropbear/authorized_keys ]; then
            if [ -f "$DCENT_RELEASE_IMAGE_MARKER" ] && [ "$SSH_LISTENING" = "false" ]; then
                echo "  [OK] SSH key-only mode has no keys yet because release-image SSH is locked before owner setup"
            else
                echo "  [FAIL] SSH key-only mode (-s) but no authorized_keys - locked out!"
                HEALTH_OK=false
                return
            fi
        fi
    fi
    echo "  [OK] SSH access gate verified"

    if ! pidof dcentrald > /dev/null 2>&1; then
        if [ ! -x /usr/local/bin/dcentrald ]; then
            echo "  [FAIL] dcentrald binary not found at /usr/local/bin/dcentrald"
            HEALTH_OK=false
            return
        fi
        echo "  [WAIT] dcentrald not running yet - waiting up to 30s"
        for i in $(seq 1 30); do
            if pidof dcentrald > /dev/null 2>&1; then
                echo "  [OK] dcentrald started after ${i}s"
                break
            fi
            sleep 1
        done
    fi

    if ! pidof dcentrald > /dev/null 2>&1; then
        echo "  [FAIL] dcentrald did not start within 30s"
        HEALTH_OK=false
        return
    fi
    echo "  [OK] dcentrald is running"

    if ! wait_for_http "dashboard" "http://127.0.0.1/api/dashboard/health" 30; then
        HEALTH_OK=false
        return
    fi

    if ! wait_for_http "dcentrald API" "http://127.0.0.1:8080/api/status" 45; then
        HEALTH_OK=false
        return
    fi

    # Check 6 (W8 parity â€” ported from zynq S99upgrade): REAL health verdict.
    # The socket-bind probes above answer the instant dcentrald binds a port â€”
    # they do NOT prove the daemon is actually alive+steady. On NoPic Amlogic a
    # bind-then-crash daemon would otherwise be committed permanently, defeating
    # U-Boot first-boot auto-revert (the "defeated by S99" hole).
    HEALTH_VERDICT=$(daemon_real_health_verdict)
    case "$HEALTH_VERDICT" in
        healthy)
            echo "  [OK] /api/system/health reports a live daemon with positive uptime"
            ;;
        unhealthy)
            echo "  [FAIL] /api/system/health reachable but reports a dead/zero-uptime daemon"
            HEALTH_OK=false
            return
            ;;
        *)
            echo "  [SOFT-PASS] /api/system/health gave no decisive signal; not blocking commit on its absence"
            ;;
    esac

    # Check 7 (W8 parity): BOOT-SUCCESS WINDOW. Require dcentrald to stay up for
    # MIN_HEALTHY_UPTIME_S as the SAME PID. Catches a bind-then-exit failure:
    # /api/status answers the instant the socket binds, but if the daemon then
    # dies S82dcentrald leaves it stopped pending hardware-state resolution.
    # The same-PID proof also catches explicit replacement. Probe in 1s steps
    # so an early death is caught fast (and surfaced as a positive failure).
    START_PID=$(pidof dcentrald 2>/dev/null | awk '{print $1}')
    if [ -z "$START_PID" ]; then
        echo "  [FAIL] dcentrald PID disappeared before the boot-success window"
        HEALTH_OK=false
        return
    fi
    echo "  Boot-success window: confirming dcentrald PID $START_PID stays up ${MIN_HEALTHY_UPTIME_S}s..."
    WINDOW_LEFT=$MIN_HEALTHY_UPTIME_S
    while [ "$WINDOW_LEFT" -gt 0 ]; do
        sleep 1
        WINDOW_LEFT=$((WINDOW_LEFT - 1))
        if ! kill -0 "$START_PID" 2>/dev/null; then
            echo "  [FAIL] dcentrald PID $START_PID died inside the boot-success window"
            HEALTH_OK=false
            return
        fi
    done
    NOW_PID=$(pidof dcentrald 2>/dev/null | awk '{print $1}')
    if [ "$NOW_PID" != "$START_PID" ]; then
        echo "  [FAIL] dcentrald PID changed during the window ($START_PID -> ${NOW_PID:-none}); hardware-owner continuity was lost"
        HEALTH_OK=false
        return
    fi
    echo "  [OK] dcentrald survived the ${MIN_HEALTHY_UPTIME_S}s boot-success window as PID $START_PID"
}

# Wave 272: OTA-08 is Amlogic-overlay only. Same sealed identities as S37.
# Missing, unknown, alias, or mixed platform:target identity fails closed
# before WAL replay or any recovery-flag NAND read/write/erase.
# Wave 398: `.78` bootcmd ignores firstboot. Blocking 0x03 on WAL failure
# leaves 0x02 â†’ recover_to_stock after a healthy S19k boot. Other Amlogic
# identities keep the WAL fail-closed gate.
s19k_firstboot_is_wal_companion_only() {
    require_amlogic_ota08_identity || return 1
    [ "$AML_OTA08_IDENTITY" = "am3-aml-s19k:am3-s19k" ]
}

require_amlogic_ota08_identity() {
    if [ ! -r "$PLATFORM_FILE" ]; then
        echo "  *** ERROR: missing live $PLATFORM_FILE; refuse OTA-08 NAND ***"
        return 1
    fi
    if [ ! -r "$BOARD_TARGET_FILE" ]; then
        echo "  *** ERROR: missing live $BOARD_TARGET_FILE; refuse OTA-08 NAND ***"
        return 1
    fi
    PLATFORM=$(tr -d ' \t\r\n' < "$PLATFORM_FILE")
    BT=$(tr -d ' \t\r\n' < "$BOARD_TARGET_FILE")
    AML_OTA08_IDENTITY="$PLATFORM:$BT"
    case "$AML_OTA08_IDENTITY" in
        am3-aml-s19k:am3-s19k) ;;
        am3-aml-s19jpro:am3-s19jpro-aml) ;;
        am3-aml-s21:am3-s21) ;;
        am3-aml-s21pro:am3-s21pro) ;;
        *)
            echo "  *** ERROR: live platform:target='$AML_OTA08_IDENTITY' is not an exact admitted Amlogic identity; refuse OTA-08 NAND ***"
            return 1
            ;;
    esac
    return 0
}

require_ota_mutation_policy() {
    POLICY_TARGET=$(tr -d ' \t\r\n' < "$BOARD_TARGET_FILE" 2>/dev/null || true)
    POLICY_PLATFORM=$(tr -d ' \t\r\n' < "$PLATFORM_FILE" 2>/dev/null || true)
    case "$POLICY_PLATFORM:$POLICY_TARGET" in
        am3-aml-s19jproa:*|*:am3-s19jproa|am3-aml-s19jproplus:*|*:am3-s19jproplus)
            echo "  *** ERROR: A126/Plus120 nvdata image does not admit legacy OTA-08 environment/NAND commits ***"
            return 1
            ;;
    esac
    case "$POLICY_TARGET" in
        am3-s19jpro-aml)
            # This candidate mounts the exact seven-partition nvdata layout.
            # The legacy system-partition flag/WAL is a different contract;
            # neither an inherited pending file nor missing policy admits it.
            echo "  *** ERROR: Classic AML nvdata image does not admit legacy OTA-08 environment/NAND commits ***"
            return 1
            ;;
    esac
    if [ ! -r "$MUTATION_POLICY_HELPER" ]; then
        echo "  *** ERROR: mutation-policy helper is missing; refuse OTA/recovery mutation ***"
        return 1
    fi
    . "$MUTATION_POLICY_HELPER"
    if ! dcent_mutation_policy_has "$MUTATION_POLICY_FILE" ota-storage; then
        echo "  *** ERROR: ota-storage capability is missing or insecure; refuse OTA/recovery mutation ***"
        return 1
    fi
}

read_recovery_flag() {
    # Returns the single byte at mtd5+RECOVERY_FLAG_OFFSET as 0x?? on stdout.
    if [ -z "$RECOVERY_FLAG_OFFSET" ] || [ "$RECOVERY_FLAG_OFFSET" = "0x05300000" ]; then
        echo "ERR_FLAG_OFFSET"
        return 1
    fi
    if ! command -v nanddump > /dev/null 2>&1; then
        echo "ERR_NO_NANDDUMP"
        return 1
    fi
    if [ ! -e "$RECOVERY_MTD" ]; then
        echo "ERR_NO_MTD5"
        return 1
    fi
    # padbad keeps the requested physical offset meaningful if this block ever
    # becomes bad. A compacting read could silently return a byte from the next
    # good eraseblock and admit the wrong recovery state.
    BYTE=$(nanddump --bb=padbad --omitoob -s "$RECOVERY_FLAG_OFFSET" -l 1 \
        "$RECOVERY_MTD" 2>/dev/null | od -An -tx1 | tr -d ' \n')
    if [ -z "$BYTE" ]; then
        echo "ERR_DUMP_FAILED"
        return 1
    fi
    echo "0x$BYTE"
}

commit_recovery_flag() {
    # OTA-08: AMLOGIC_RAW_NAND_RECOVERY_FLAG_EXCEPTION.
    #
    # This is the only target-side raw NAND write this init script is allowed to
    # perform: a read/modify/write of the complete eraseblock containing the
    # Amlogic mtd5 recovery-state byte, after boot health has passed. Erasing a
    # block and programming only one byte destroys every adjacent byte, even
    # when the flag happens to be block-aligned. It must not be copied to
    # Zynq/CVitek/BB, and it does not update U-Boot env; firstboot remains
    # fw_setenv-only below.
    #
    # Two identical padbad/omitoob snapshots are required before erase. The
    # expected 128 KiB image changes exactly the flag byte and the complete
    # eraseblock must compare equal after write. The single-byte readback below
    # is retained as a second semantic check.
    # Wave 272: also require a sealed Amlogic board_target and a pre-write 0x02.
    if [ -z "$RECOVERY_FLAG_OFFSET" ] || [ "$RECOVERY_FLAG_OFFSET" = "0x05300000" ]; then
        echo "  *** ERROR: recovery flag offset unset or naive 0x05300000; refuse NAND write ***"
        return 1
    fi
    case "$RECOVERY_FLAG_OFFSET" in
        0x*|0X*) FLAG_HEX_DIGITS=${RECOVERY_FLAG_OFFSET#??} ;;
        *) FLAG_HEX_DIGITS= ;;
    esac
    case "$FLAG_HEX_DIGITS" in
        ''|*[!0-9A-Fa-f]*)
            echo "  *** ERROR: recovery flag offset is not canonical hexadecimal: '$RECOVERY_FLAG_OFFSET' ***"
            return 1
            ;;
    esac
    if ! require_amlogic_ota08_identity; then
        return 1
    fi
    OLD=$(read_recovery_flag)
    if [ "$OLD" != "0x02" ]; then
        echo "  *** ERROR: recovery flag pre-write = $OLD (expected 0x02); refuse OTA-08 NAND ***"
        return 1
    fi
    for TOOL in nanddump flash_erase nandwrite cp dd od cmp mktemp wc; do
        if ! command -v "$TOOL" > /dev/null 2>&1; then
            echo "  *** ERROR: $TOOL missing - cannot preserve/rewrite recovery eraseblock ***"
            return 1
        fi
    done
    case "$RECOVERY_ERASESIZE_EXPECTED" in
        ''|*[!0-9]*)
            echo "  *** ERROR: invalid recovery erasesize '$RECOVERY_ERASESIZE_EXPECTED' ***"
            return 1
            ;;
    esac
    if [ "$RECOVERY_ERASESIZE_EXPECTED" -ne 131072 ]; then
        echo "  *** ERROR: recovery erasesize $RECOVERY_ERASESIZE_EXPECTED != sealed 131072 ***"
        return 1
    fi
    MTD_BASENAME=${RECOVERY_MTD##*/}
    MTD_ERASESIZE_FILE="/sys/class/mtd/$MTD_BASENAME/erasesize"
    if [ -r "$MTD_ERASESIZE_FILE" ]; then
        LIVE_ERASESIZE=$(tr -d ' \t\r\n' < "$MTD_ERASESIZE_FILE")
        case "$LIVE_ERASESIZE" in
            ''|*[!0-9]*)
                echo "  *** ERROR: live recovery erasesize is invalid: '$LIVE_ERASESIZE' ***"
                return 1
                ;;
        esac
        if [ "$LIVE_ERASESIZE" -ne "$RECOVERY_ERASESIZE_EXPECTED" ]; then
            echo "  *** ERROR: live recovery erasesize $LIVE_ERASESIZE != expected $RECOVERY_ERASESIZE_EXPECTED ***"
            return 1
        fi
    fi
    FLAG_DEC=$((RECOVERY_FLAG_OFFSET))
    ERASEBLOCK_START_DEC=$((FLAG_DEC / RECOVERY_ERASESIZE_EXPECTED * RECOVERY_ERASESIZE_EXPECTED))
    FLAG_IN_BLOCK=$((FLAG_DEC - ERASEBLOCK_START_DEC))
    ERASEBLOCK_START=$(printf '0x%08X' "$ERASEBLOCK_START_DEC")

    if [ ! -d "$RECOVERY_STAGING_ROOT" ] || [ -L "$RECOVERY_STAGING_ROOT" ]; then
        echo "  *** ERROR: persistent recovery staging root is unavailable or a symlink: $RECOVERY_STAGING_ROOT ***"
        return 1
    fi
    FLAG_TXN=$(mktemp -d "$RECOVERY_STAGING_ROOT/.s99-recovery-flag.XXXXXX") || {
        echo "  *** ERROR: cannot create persistent recovery-flag transaction directory ***"
        return 1
    }
    chmod 0700 "$FLAG_TXN" 2>/dev/null || {
        rmdir "$FLAG_TXN" 2>/dev/null || true
        echo "  *** ERROR: cannot protect recovery-flag transaction directory ***"
        return 1
    }
    FLAG_BEFORE="$FLAG_TXN/eraseblock.before.bin"
    FLAG_BEFORE_2="$FLAG_TXN/eraseblock.before-duplicate.bin"
    FLAG_EXPECTED="$FLAG_TXN/eraseblock.expected.bin"
    FLAG_READBACK="$FLAG_TXN/eraseblock.readback.bin"

    cleanup_flag_transaction() {
        rm -f "$FLAG_BEFORE" "$FLAG_BEFORE_2" "$FLAG_EXPECTED" "$FLAG_READBACK" 2>/dev/null || true
        rmdir "$FLAG_TXN" 2>/dev/null || true
    }
    retain_flag_transaction() {
        echo "  *** RECOVERY: preserved pre-write eraseblock transaction at $FLAG_TXN ***"
    }

    if ! nanddump --bb=padbad --omitoob -s "$ERASEBLOCK_START" \
        -l "$RECOVERY_ERASESIZE_EXPECTED" "$RECOVERY_MTD" > "$FLAG_BEFORE"; then
        echo "  *** ERROR: first recovery eraseblock snapshot failed ***"
        cleanup_flag_transaction
        return 1
    fi
    if ! nanddump --bb=padbad --omitoob -s "$ERASEBLOCK_START" \
        -l "$RECOVERY_ERASESIZE_EXPECTED" "$RECOVERY_MTD" > "$FLAG_BEFORE_2"; then
        echo "  *** ERROR: duplicate recovery eraseblock snapshot failed ***"
        cleanup_flag_transaction
        return 1
    fi
    BEFORE_SIZE=$(wc -c < "$FLAG_BEFORE" | tr -d ' \t\r\n')
    BEFORE_2_SIZE=$(wc -c < "$FLAG_BEFORE_2" | tr -d ' \t\r\n')
    if [ "$BEFORE_SIZE" != "$RECOVERY_ERASESIZE_EXPECTED" ] || \
       [ "$BEFORE_2_SIZE" != "$RECOVERY_ERASESIZE_EXPECTED" ] || \
       ! cmp -s "$FLAG_BEFORE" "$FLAG_BEFORE_2"; then
        echo "  *** ERROR: recovery eraseblock snapshots are short or unstable ***"
        cleanup_flag_transaction
        return 1
    fi
    SNAPSHOT_OLD=$(dd if="$FLAG_BEFORE" bs=1 skip="$FLAG_IN_BLOCK" count=1 2>/dev/null \
        | od -An -tx1 | tr -d ' \n')
    if [ "$SNAPSHOT_OLD" != "02" ]; then
        echo "  *** ERROR: snapshotted recovery flag = 0x$SNAPSHOT_OLD (expected 0x02) ***"
        cleanup_flag_transaction
        return 1
    fi
    if ! cp "$FLAG_BEFORE" "$FLAG_EXPECTED" || \
       ! printf '\003' | dd of="$FLAG_EXPECTED" bs=1 seek="$FLAG_IN_BLOCK" conv=notrunc 2>/dev/null; then
        echo "  *** ERROR: cannot construct preserved recovery eraseblock image ***"
        cleanup_flag_transaction
        return 1
    fi
    EXPECTED_SIZE=$(wc -c < "$FLAG_EXPECTED" | tr -d ' \t\r\n')
    if [ "$EXPECTED_SIZE" != "$RECOVERY_ERASESIZE_EXPECTED" ]; then
        echo "  *** ERROR: preserved recovery eraseblock image changed size ***"
        cleanup_flag_transaction
        return 1
    fi

    # Durably retain the old and intended eraseblocks before the destructive
    # boundary. On any later failure the directory is intentionally retained.
    if ! sync; then
        echo "  *** ERROR: recovery eraseblock snapshots are not durable; refuse erase ***"
        cleanup_flag_transaction
        return 1
    fi
    if ! flash_erase "$RECOVERY_MTD" "$ERASEBLOCK_START" 1 2>&1; then
        echo "  *** ERROR: flash_erase $RECOVERY_MTD $ERASEBLOCK_START failed ***"
        retain_flag_transaction
        return 1
    fi
    if ! nandwrite -p -s "$ERASEBLOCK_START" "$RECOVERY_MTD" "$FLAG_EXPECTED" 2>&1; then
        echo "  *** ERROR: full recovery eraseblock nandwrite failed ***"
        retain_flag_transaction
        return 1
    fi
    if ! nanddump --bb=padbad --omitoob -s "$ERASEBLOCK_START" \
        -l "$RECOVERY_ERASESIZE_EXPECTED" "$RECOVERY_MTD" > "$FLAG_READBACK"; then
        echo "  *** ERROR: full recovery eraseblock readback failed ***"
        retain_flag_transaction
        return 1
    fi
    READBACK_SIZE=$(wc -c < "$FLAG_READBACK" | tr -d ' \t\r\n')
    if [ "$READBACK_SIZE" != "$RECOVERY_ERASESIZE_EXPECTED" ] || \
       ! cmp -s "$FLAG_EXPECTED" "$FLAG_READBACK"; then
        echo "  *** ERROR: full recovery eraseblock readback mismatch ***"
        retain_flag_transaction
        return 1
    fi
    NEW=$(read_recovery_flag)
    if [ "$NEW" != "0x03" ]; then
        echo "  *** ERROR: recovery flag readback = $NEW (expected 0x03) ***"
        retain_flag_transaction
        return 1
    fi
    cleanup_flag_transaction
    echo "  [OK] recovery flag promoted 0x02 -> 0x03 by verified full-eraseblock rewrite (firmware committed)"
    return 0
}

clear_uboot_env() {
    # Best-effort: clear the AML first-boot guard after recovery-flag commit.
    # Live .78 U-Boot env uses `firstboot` (not `first_boot`).
    #
    # Wave 10 B2: idempotent. Safe to call multiple times. If firstboot is
    # already 0, returns success without re-writing. The caller manages the
    # /data/.firstboot-pending intent file (write before, remove after).
    if ! command -v fw_setenv > /dev/null 2>&1; then
        echo "  *** ERROR: fw_setenv missing - libubootenv-tools not installed ***"
        echo "  *** Recovery-flag byte was committed, but U-Boot env vars NOT ***"
        echo "  *** cleared. Add BR2_PACKAGE_LIBUBOOTENV{,_TOOLS}=y to defconfig ***"
        return 1
    fi

    # Short-circuit if already cleared (idempotent replay).
    if fw_printenv firstboot 2>/dev/null | grep -q '^firstboot=0$'; then
        echo "  [OK] U-Boot env firstboot already 0 (no write needed)"
        return 0
    fi

    if ! fw_setenv firstboot 0 2>/dev/null; then
        echo "  *** WARN: failed to set firstboot=0 in /dev/nand_env ***"
        return 1
    fi
    if fw_printenv firstboot 2>/dev/null | grep -q '^firstboot=0$'; then
        echo "  [OK] U-Boot env firstboot=0 committed"
    else
        echo "  *** WARN: firstboot did not read back as 0 after fw_setenv ***"
        return 1
    fi
    return 0
}

# Wave 10 B2: WAL helpers --------------------------------------------------

mark_firstboot_pending() {
    # Record intent to write firstboot=0 BEFORE we commit the recovery flag.
    # If the system reboots between recovery-flag commit and env-clear, this
    # file remains; replay_pending_env_clear() picks it up on next boot.
    WAL_DIR=${FIRSTBOOT_PENDING_FILE%/*}
    [ "$WAL_DIR" != "$FIRSTBOOT_PENDING_FILE" ] || WAL_DIR=.
    if [ ! -d "$WAL_DIR" ]; then
        echo "  *** WARN: WAL directory $WAL_DIR is not mounted; cannot persist firstboot WAL ***"
        return 1
    fi
    # Publish through a same-directory temp + rename. A direct `: > marker`
    # can leave a newly-created directory entry non-durable at the exact crash
    # point this WAL exists to cover. The first sync makes the staged inode
    # durable; the second makes the rename durable. Do not proceed to the raw
    # recovery-flag commit unless both boundaries complete.
    WAL_TMP="${FIRSTBOOT_PENDING_FILE}.tmp.$$"
    rm -f "$WAL_TMP" 2>/dev/null || true
    if ! (umask 077; : > "$WAL_TMP") 2>/dev/null; then
        echo "  *** WARN: failed to stage WAL marker $WAL_TMP ***"
        rm -f "$WAL_TMP" 2>/dev/null || true
        return 1
    fi
    if ! sync; then
        echo "  *** WARN: failed to sync staged WAL marker $WAL_TMP ***"
        rm -f "$WAL_TMP" 2>/dev/null || true
        return 1
    fi
    if ! mv -f "$WAL_TMP" "$FIRSTBOOT_PENDING_FILE" 2>/dev/null; then
        echo "  *** WARN: failed to publish WAL marker $FIRSTBOOT_PENDING_FILE ***"
        rm -f "$WAL_TMP" 2>/dev/null || true
        return 1
    fi
    if ! sync; then
        echo "  *** WARN: WAL marker is visible but directory durability is unproven: $FIRSTBOOT_PENDING_FILE ***"
        return 1
    fi
    return 0
}

clear_firstboot_pending() {
    if [ -f "$FIRSTBOOT_PENDING_FILE" ]; then
        rm -f "$FIRSTBOOT_PENDING_FILE" 2>/dev/null
        sync
    fi
    return 0
}

replay_pending_env_clear() {
    # Called FIRST in the start case. If a previous boot crashed between
    # commit_recovery_flag and clear_uboot_env, $FIRSTBOOT_PENDING_FILE
    # exists and we owe a fw_setenv firstboot=0. Idempotent â€” safe to
    # invoke whether or not the file exists.
    [ -f "$FIRSTBOOT_PENDING_FILE" ] || return 0
    echo "[WAL] $FIRSTBOOT_PENDING_FILE present - replaying firstboot=0 commit"
    if clear_uboot_env; then
        clear_firstboot_pending
        echo "  [OK] WAL replay succeeded; intent file cleared"
        return 0
    fi
    echo "  *** WAL replay failed; will retry on next boot ***"
    return 1
}
# --------------------------------------------------------------------------

case "$1" in
    start)
        # This gate precedes WAL replay because replay can call fw_setenv.
        # Target policy therefore blocks every env/NAND mutation, not only the
        # later recovery-flag commit.
        require_ota_mutation_policy || exit 1
        if [ ! -e "$RECOVERY_MTD" ]; then
            # No mtd5 â€” not on Amlogic NAND. Nothing to do.
            exit 0
        fi
        require_amlogic_ota08_identity || exit 1

        # Wave 10 B2: WAL replay runs BEFORE the recovery-flag check. If a
        # previous boot succeeded at commit_recovery_flag but failed at
        # clear_uboot_env, the recovery flag is now 0x03 and the 0x02 case
        # below would skip env clearing entirely. The replay closes that gap.
        replay_pending_env_clear

        FLAG=$(read_recovery_flag)
        case "$FLAG" in
            0x02)
                echo "Recovery flag = 0x02 (FIRST_BOOT) - running health checks before committing..."
                check_health
                if [ "$HEALTH_OK" != "true" ]; then
                    echo "  *** HEALTH CHECK FAILED - leaving recovery flag at 0x02 ***"
                    echo "  *** slot remains uncommitted; use the platform recovery runbook ***"
                    echo "  *** (serial/SD boundary unless rollback is proven for this lane) ***"
                    exit 0
                fi
                echo "Health checks passed - committing recovery flag..."

                # Wave 10 B2: write WAL intent BEFORE commit_recovery_flag so
                # a crash during the mtd5 write or anywhere between here and
                # successful clear_uboot_env leaves a replay marker. The
                # marker is cleared only after fw_printenv confirms firstboot=0.
                if ! mark_firstboot_pending; then
                    if s19k_firstboot_is_wal_companion_only; then
                        echo "  *** WARN: firstboot WAL not proven; S19k bootcmd ignores firstboot; proceeding to 0x02->0x03 ***"
                    else
                        echo "  *** ERROR: durable WAL intent was not proven - refusing recovery-flag commit ***"
                        echo "  *** Slot remains uncommitted at recovery flag 0x02; next reboot is recover_to_stock ***"
                        exit 1
                    fi
                fi

                if ! commit_recovery_flag; then
                    echo "  *** ERROR: recovery flag commit failed - leftover 0x02 is recover_to_stock on next reboot ***"
                    # Recovery flag is still 0x02; clear WAL marker since no env
                    # commit will be attempted this boot.
                    clear_firstboot_pending
                    exit 1
                fi
                if ! clear_uboot_env; then
                    if s19k_firstboot_is_wal_companion_only; then
                        echo "  *** WARN: firstboot=0 failed after 0x03; S19k recover_to_stock is already disarmed by flag 0x03 ***"
                    else
                        echo "  *** ERROR: U-Boot env commit failed after recovery flag commit ***"
                        echo "  *** WAL marker $FIRSTBOOT_PENDING_FILE retained for next-boot replay ***"
                        exit 1
                    fi
                fi
                # Both halves succeeded; clear the WAL marker.
                clear_firstboot_pending
                ;;
            0x03)
                # Already committed â€” normal boot, nothing to do.
                exit 0
                ;;
            0x01)
                # InstallArm leftover. U-Boot try_to_boot_bos_after_install
                # must write 0x02 before kernel handoff so first-boot revert
                # is armed. Do not promote 0x01 â†’ 0x03 (skips health + WAL).
                echo "  *** ERROR: recovery flag = 0x01 (INSTALLED) leftover in userspace; U-Boot did not write 0x02; refuse OTA-08 ***"
                exit 1
                ;;
            ERR_*)
                # mtd5 exists (start already exited 0 if it does not).
                # Unreadable flag can hide leftover 0x01. Do not promote.
                echo "  *** ERROR: could not read recovery flag ($FLAG); refuse OTA-08 ***"
                exit 1
                ;;
            *)
                # U-Boot treats unknown bytes as recover_to_stock. Userspace
                # must not treat them as a healthy committed boot.
                echo "  *** ERROR: unexpected recovery flag value: $FLAG; refuse OTA-08 ***"
                exit 1
                ;;
        esac
        ;;

    stop)
        ;;

    *)
        echo "Usage: $0 {start|stop}"
        exit 1
        ;;
esac

exit 0
