diff --git a/.gitignore b/.gitignore index b200a9f0f..1818b8711 100644 --- a/.gitignore +++ b/.gitignore @@ -2,6 +2,14 @@ # Build output of the standalone BDK canary crate (scripts/canary/bdk-canary). # Its Cargo.lock IS committed (reproducible downstream pin); its target/ is not. /scripts/canary/bdk-canary/target + +# Same for the reference push relay (contrib/push-relay): a standalone crate +# excluded from the workspace, so the root /target rule does not cover it. The +# rule lives this early in the stack, ahead of the crate it protects, because +# 831 MiB of its build output was once committed by a stray `git add -A` four +# PRs before the crate itself existed — and .gitignore does not untrack what is +# already tracked. +/contrib/push-relay/target *.swp .idea/ .vscode/ diff --git a/CHANGELOG.md b/CHANGELOG.md index fff3a0a37..f9c842913 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -130,6 +130,29 @@ layout) per [`STABILITY_POLICY.md`](STABILITY_POLICY.md). capability gates on the RPC surface. Not replayable (no cursor — detectors re-raise standing conditions after a restart). Wire schema only in this change; the detectors that emit them land next. +- Alerting: node-health detectors. satd now watches six conditions about itself + — stalled tip, low disk, congested mempool, peer starvation, IBD completion, + deep reorg — and reports each through three surfaces at once: a `status` + streaming event, an entry in `getwarnings` (which fires the Core-compatible + `alertnotify` hook), and a `satd_alert_active{kind}` gauge. Standing + conditions raise once and clear once, with hysteresis so a value sitting on + the threshold does not flap. The one-shot events (`ibd_complete`, + `deep_reorg`) fire `alertnotify` and the streaming event but deliberately do + **not** enter `getwarnings` — nothing would ever clear them, and on chains + where multi-block reorgs are routine that would wedge + `getblockchaininfo.warnings` and the TUI modal permanently. Five new hot-reloadable thresholds + (`alerttipstallseconds=3600` — `0` on regtest, `alertdiskfreemb=10240`, + `alertmempoolfullpct=90`, `alertpeerfloor=3` — `0` on regtest, and capped by + the number of `-connect=` peers when that is set — + `alertreorgdepth=3` — `10` on test networks, `0` on regtest); `0` disables a + detector. `deep_reorg` depth, fork + height and new tip are read from the durable reorg log, so they are exact + regardless of chain-event lag. Two new gauges close longstanding observability gaps + independently of alerting: `satd_tip_last_connect_age_seconds` and + `satd_disk_free_bytes`, the latter sampled even when `alertdiskfreemb=0`. The + free-space probe is bounded and carried across polls, so a `blocksdir` on an + unresponsive network mount stalls only the disk alert — never the other + detectors — and strands at most one blocking thread rather than one per poll. ## Releases diff --git a/docs/api/streaming.md b/docs/api/streaming.md index c726474d7..78bbbc3fb 100644 --- a/docs/api/streaming.md +++ b/docs/api/streaming.md @@ -973,6 +973,33 @@ re-raise the ones still standing, which is what makes health alerting at-least-once across a restart. A condition that both raised and fully cleared while a consumer was away is stale by definition and is not reconstructed. +**`details` keys by kind.** Values are strings. Most are decimal numbers, but +not all — parse per key rather than assuming the whole map is numeric. + +| Kind | Keys | +|---|---| +| `ibd_complete` | `height` | +| `tip_stall` | raise: `seconds_since_block`, `threshold_seconds`, `tip_height`. Cleared by a block: `height`. Cleared by the poll: `seconds_since_block`, `threshold_seconds` | +| `disk_low` | `free_bytes`, `threshold_bytes`; `clear_threshold_bytes` when cleared by recovered space | +| `mempool_congested` | `bytes_used`, `bytes_cap`, `threshold_pct`, `mempoolminfee_sat_per_kvb` (raise only) | +| `peer_floor` | `peers`, `peers_outbound`, `peers_inbound`, `threshold` (raise only) | +| `deep_reorg` | `depth`, `from_height`, `to_height`, `fork_height` | + +`deep_reorg` figures are exact. Depth, fork height and the reconnected chain +are read from the reorg log record that `perform_reorg` writes and fsyncs, not +reconstructed by counting disconnect events off the event bus — so a reorg deep +enough to overrun the bus ring is reported with the same precision as a shallow +one. `to_height` is the new tip, not the first reconnected block. + +Any kind may additionally carry `reason` — a non-numeric token on a `cleared` +event emitted because the detector can no longer evaluate the condition, rather +than because the condition recovered: + +| Token | Meaning | +|---|---| +| `detector_disabled` | The operator set this detector's threshold to `0`. | +| `mempool_cap_zero` | `maxmempool` is `0`, so there is no occupancy ratio to evaluate. `alertmempoolfullpct` is still armed — this is not the operator turning the detector off, and a consumer that suppresses `detector_disabled` should not suppress this. | + **Additive by construction.** `details` is a string map and `StatusKind` is an open enum: new kinds and new detail keys ship without a schema bump (§4). A consumer must tolerate an unrecognized `kind` — `message` and `severity` remain diff --git a/docs/manual/src/config-reference.md b/docs/manual/src/config-reference.md index bd317a052..edb8b57b9 100644 --- a/docs/manual/src/config-reference.md +++ b/docs/manual/src/config-reference.md @@ -394,6 +394,26 @@ Core ZMQ wire-format compatible.) | `reorgwebhook` | none | hot | satd | HTTP(S) endpoint receiving a POST on reorg detection. | | `reorgwebhooksecret` | none | hot | satd | HMAC-SHA256 secret signing webhook bodies via `X-Satd-Signature`. | +## Health alerts + +Thresholds for the node-health detectors. Each raises a `status` event on the +[Streaming Consumption API](streaming.md) and an entry in `getwarnings` (which +also fires `alertnotify`) when its condition is entered, and retracts both when +it recovers. Every one is hot-reloadable — retuning an alert should not need a +restart, since you are usually retuning it *because* it is firing. + +Set a threshold to `0` to disable that detector. See +[Observability → Node-health alerts](observability.md#node-health-alerts) for +the taxonomy and the details each event carries. + +| Key | Default | Reload | Compat | Description | +|---|---|---|---|---| +| `alerttipstallseconds` | `3600` (`0` on regtest) | hot | satd | Raise `tip_stall` after this many seconds with no connected block. Defaults to disabled on regtest only, where blocks exist just when a test mines them and an idle chain is normal; every other network — test networks included — keeps the hour, since going an hour without a block is not an ordinary property of thin hashrate the way a shallow reorg is. *Not* suppressed during initial block download — `is_initial_block_download()` compares the tip header's timestamp against the wall clock rather than tracking sync progress, so a node that was caught up and then wedged re-enters it precisely when you need paging. A node that is genuinely syncing connects blocks continuously and so never crosses the threshold. Cleared the moment a block connects — or, if you raise this value past the current tip age, on the next detector poll. | +| `alertdiskfreemb` | `10240` | hot | satd | Raise `disk_low` below this many MiB free on the blocks directory (or the data directory when `blocksdir` is not split out). Clears at 1.5× the floor, or as soon as you lower the floor below the current reading. | +| `alertmempoolfullpct` | `90` | hot | satd | Raise `mempool_congested` at this percentage of `maxmempool`. Clears below 75 % of the raise line, or as soon as you raise the threshold above the current occupancy. Values above 100 are clamped. | +| `alertpeerfloor` | `3` (`0` on regtest; capped by the `-connect=` count) | hot | satd | Raise `peer_floor` below this many connected peers (inbound + outbound). The count must hold for 60 s in either direction, so ordinary peer churn does not page you, and it does not raise until 90 s after startup or the first peer, whichever is sooner. Defaults to disabled on regtest only, where a node with no peers is normal; **signet keeps the floor** — set `alertpeerfloor=0` explicitly on a deliberately isolated signet node. When `connect=` is set the default drops to that many peers (never above `3`), since `connect=` suppresses DNS and fixed seeds and the node can never exceed the addresses you named — an explicit value here still overrides it. | +| `alertreorgdepth` | `3` (`10` on test networks, `0` on regtest) | hot | satd | Emit the one-shot `deep_reorg` event for a reorg that rolls back at least this many blocks. Depth 3 is an incident on mainnet and ordinary on a chain with thin, volatile hashrate, so signet/testnet/testnet4 default to `10` — above the 6-confirmation convention, so a reorg that invalidated something a wallet called final still reports. Regtest defaults off: its test suites reorg deliberately. | + > **Note.** The `*notify` shell hooks (`blocknotify`, `alertnotify`, > `startupnotify`, `shutdownnotify`) exist for drop-in Bitcoin Core > compatibility and quick scripts. They are best-effort shell execs with no diff --git a/docs/manual/src/observability.md b/docs/manual/src/observability.md index f81d97008..c2b0cfe09 100644 --- a/docs/manual/src/observability.md +++ b/docs/manual/src/observability.md @@ -45,6 +45,106 @@ labels) and does not consume an RPC worker on every scrape. > one-time handshake bytes are not included, so absolute socket totals read > marginally lower than the kernel's. +## Node-health alerts + +Metrics tell you what the node is doing; health alerts tell you when it has +stopped doing it. satd watches six conditions about *itself* and reports each +one through three surfaces at once, so they can never disagree: + +* a `status` event on the [Streaming Consumption API](streaming.md) (category + bit 16 — see §7.8 of the wire spec), +* an entry in `getwarnings` (and therefore in `getblockchaininfo.warnings` + and the TUI), which also fires the Core-compatible `alertnotify` hook, +* a `satd_alert_active{kind="..."}` gauge on `/metrics`. + +**One-shot events are the exception to the middle surface.** `ibd_complete` and +`deep_reorg` describe something that *happened*; there is no state for anything +to later clear. They fire `alertnotify` and emit their `status` event, but they +do **not** create a `getwarnings` entry. An entry nothing clears would sit in +`getblockchaininfo.warnings` for the life of the process and hold the TUI's +warning modal open — which on signet and testnet4, where reorgs several blocks +deep are ordinary, would happen on the first one and never stop. The durable +record of a reorg is the reorg log (`getreorghistory`), not the warnings set. + +| Condition | Severity | Raises when | Clears when | +|---|---|---|---| +| `ibd_complete` | info | initial block download finishes | one-shot | +| `tip_stall` | critical | no block connected for `alerttipstallseconds`, outside IBD | the next block connects, or the threshold no longer considers the tip stalled | +| `disk_low` | critical | free space below `alertdiskfreemb` | free space reaches 1.5× the floor, or the floor is lowered below the current reading | +| `mempool_congested` | warning | mempool at `alertmempoolfullpct` of its cap | occupancy drops below 75 % of the raise line, or the threshold is raised above the current occupancy | +| `peer_floor` | warning | fewer than `alertpeerfloor` peers for 60 s (after a 90 s startup grace) | at or above the floor for 60 s | +| `deep_reorg` | critical | a reorg rolled back ≥ `alertreorgdepth` blocks (default `3` on mainnet, `10` on test networks, off on regtest) | one-shot | + +Every standing condition raises **once** on entry and clears **once** on +recovery — you get a pair of events, not a stream of repeats — and the gap +between the raise and clear lines (a ratio, a hold time, or both) means a value +sitting on the threshold does not flap your pager. `ibd_complete` and +`deep_reorg` describe things that happened rather than states that persist, so +they are one-shot: they never clear, and for the same reason they never enter +`getwarnings` at all. + +Thresholds are configured with the `alert*` keys in the +[Configuration Reference](config-reference.md#health-alerts); all of them are +hot-reloadable, and setting one to `0` disables that detector. Each event +carries a `details` map with the numbers behind it (free bytes and the floor, +seconds since the last block and the tip height, the reorg's true depth and +fork height, the mempool's current `mempoolminfee`), so an alert is actionable +without a follow-up query. The watched *path* is deliberately not in the event — +it goes to every `status` subscriber and into push-notification bodies, and an +absolute datadir path usually names the account it runs under. The node logs it +instead. + +**Retuning a threshold always clears its own alert.** The gap between each raise +and clear line stops a value hovering at the threshold from flapping, but it +would otherwise trap the operator who raises a threshold *because* the alert is +firing: the unchanged reading lands between the new raise line and the new clear +line, where neither fires. So a standing condition also clears when the +threshold moves such that it would no longer raise. Without this, +`mempool_congested` in particular was inescapable — `alertmempoolfullpct` clamps +at 100 and the clear line is 75 % of the raise line, so past 75 % occupancy no +setting could clear it. + +`alertreorgdepth` defaults to `3` on mainnet, where a reorg that deep costs real +hashrate and invalidates transactions merchants have started treating as +settled. Signet, testnet and testnet4 default to `10`: those chains are not +economically secured, and reorgs a few blocks deep are an ordinary consequence +of thin, volatile hashrate rather than an incident. Defaulting them to mainnet's +sensitivity would run `-alertnotify` for the network working as designed, and an +alert that fires during normal operation is one you learn to ignore — which +costs you the mainnet alert too. It is raised rather than switched off because +past the 6-confirmation convention a wallet has been told something false, and +that is worth reporting on any chain. Regtest is off entirely; its test suites +reorg deliberately. Set `alertreorgdepth=3` explicitly if you want mainnet +sensitivity on a test network. + +`alertpeerfloor` defaults to `3` everywhere except regtest, where it is `0` +(disabled) because running with no peers at all is a regtest node's normal +operating state rather than a fault. Signet keeps the floor: it is a public +network with real peers, and a detector defaulted off is indistinguishable from +a healthy one — `satd_alert_active{kind="peer_floor"}` reads `0` either way. Set +`alertpeerfloor=0` explicitly on a deliberately isolated signet node. + +Where the floor is active, a node that has never seen a peer gets a 90 s startup +grace, and the ordinary hold begins when that grace expires or when the first +peer arrives, whichever comes first. The grace defers the start of the hold +rather than shortening it, so a node still dialing out does not page anyone on +the way up. + +**Durability.** Health events are not replayable: they carry no resume cursor, +and a `from_cursor` reconnect never yields one. Instead the detectors +re-evaluate from scratch on startup and re-raise anything still standing, so a +consumer that was disconnected across a restart still learns about a live +problem. A condition that both raised and cleared while nothing was listening is +stale by definition and is not reconstructed. For the same reason, a subscriber +that attaches *after* a condition raised will not see it until the condition +changes — check `getwarnings` for current state on connect. + +Two of the gauges are useful independently of alerting: +`satd_tip_last_connect_age_seconds` (seconds since the last connected block) +and `satd_disk_free_bytes` (free space on the watched directory). The latter is +omitted rather than reported as zero when the filesystem cannot be +interrogated. + ## Structured JSON Logging `satd` logs to stdout. Use `--log-format=json` to switch from the text format diff --git a/docs/release-notes/0.5.0-pre.md b/docs/release-notes/0.5.0-pre.md index 437993716..a6ac85419 100644 --- a/docs/release-notes/0.5.0-pre.md +++ b/docs/release-notes/0.5.0-pre.md @@ -255,6 +255,134 @@ daemon had stalled, filled its disk, or lost its peers had to poll enum, and `details` is a string map, so future conditions and fields ship without breaking a deployed consumer. +### One detection, three surfaces + +An operator should not have to pick a monitoring style to find out their node is +in trouble. Each detection is reported simultaneously as: + +- a `status` event on the streaming API, +- an entry in `getwarnings` (and so in `getblockchaininfo.warnings` and the + TUI), which also fires the Core-compatible `alertnotify` shell hook, +- a `satd_alert_active{kind="..."}` gauge on `/metrics`. + +They are driven from the same state machine, so they cannot disagree: if the +gauge reads 1, the warning is present and the raise event was published. +`satd_alert_active` is pre-registered at 0 for every standing condition from the +first scrape, so a Prometheus rule can reference a series before it ever fires. + +Each event carries the numbers behind it in `details` — free bytes and the +watched path, seconds since the last block and the tip height, a reorg's true +depth and fork height, the mempool's current `mempoolminfee` — so an alert is +actionable without a follow-up query. (`ChainEvent::Reorg` does not carry a +fork point, so the detector derives the true depth by counting the reorg's +disconnects rather than reporting the threshold it crossed.) + +### Configuration + +Five new thresholds, all hot-reloadable — retuning an alert should not require a +restart, since you are usually retuning it *because* it is firing. Setting one +to `0` disables that detector: + +| Key | Default | +|---|---| +| `alerttipstallseconds` | `3600` (`0` on regtest) | +| `alertdiskfreemb` | `10240` | +| `alertmempoolfullpct` | `90` | +| `alertpeerfloor` | `3` (`0` on regtest; capped by the `-connect=` count) | +| `alertreorgdepth` | `3` (`10` on test networks, `0` on regtest) | + +Three defaults are network-dependent, all for the same reason: an alert that +fires while the network behaves exactly as designed is one operators learn to +ignore, and that costs them the mainnet alert too. + +`alerttipstallseconds` is `0` on regtest. Regtest blocks exist only when +something calls `generatetoaddress`, so an idle chain is its resting state +rather than a stall — and the alert is *critical*, so a node left running for an +hour between test runs would pin `getwarnings`, hold the error flag, and raise +the TUI's blocking modal on a chain doing exactly what it should. Every other +network keeps the hour, test networks included: unlike a shallow reorg, going an +hour without a block is not an ordinary property of thin hashrate. + +`alertreorgdepth` is `3` on mainnet, where a reorg that deep costs real hashrate +and invalidates transactions merchants have begun treating as settled. On +signet, testnet and testnet4 it is `10`: those chains are not economically +secured, and reorgs a few blocks deep are an ordinary consequence of thin, +volatile hashrate rather than an incident. It is raised rather than disabled — +past the 6-confirmation convention a wallet has been told something false, and +that is worth reporting on any chain. Regtest is off entirely, since its test +suites reorg on purpose and constantly. + +`alertpeerfloor` gets `0` on regtest, because a regtest node normally runs +entirely alone and a permanent critical warning is a poor greeting for a first +run. Every other network — signet included — keeps the floor, since a detector +defaulted off is indistinguishable from a healthy one on the metrics surface. +Set it to `0` explicitly on a deliberately isolated +node. + +The floor is a property of the configuration as well as the network, so +`-connect=` lowers it to the number of peers configured. `-connect=` pins the +node to exactly those addresses and suppresses both DNS seeding and the fixed +seeds, so a node given one upstream can never reach a floor of `3` — the +warning would stand forever in `getblockchaininfo.warnings`, which wallet +software renders to end users. Capping rather than disabling keeps the alert +doing the one useful thing it still can here: reporting a `-connect=` node that +had all its configured peers and then lost one. An explicit `alertpeerfloor` +overrides this in either direction. A node that has never seen a peer gets a 90 s startup grace, and the hold +begins when the grace expires or the first peer arrives, whichever is sooner. + +`tip_stall` is **not** suppressed during initial block download, which is worth +stating plainly because the opposite would be the obvious design. +`is_initial_block_download()` is a function of the tip's *age*, not a sync-progress +flag, so a node that is fully caught up and then stops re-enters it a day later — +exactly when you want to hear from it — and a node restarted while already wedged +never leaves it, so gating on the flag would silence the alert permanently at the +worst possible moment. Nothing is lost by not gating: a node that is genuinely +syncing connects blocks continuously, which keeps the age far below the threshold +and suppresses the alert on its own. A node that has connected nothing for the +whole threshold is stalled whether it is wedged mid-sync or wedged at the tip. + +The stall clock is seeded at detector start rather than from the tip timestamp, +so restarting a node that has been down for hours does not page you for a stall +that is really a restart. + +`disk_low` will not take the rest of alerting down with it. `statvfs` on a hard +NFS mount whose server has gone away does not return, and a detector that simply +awaited it would park the entire poll loop: tip stall, deep reorg, mempool and +peer checks would all stop running and every gauge would freeze at its last +value, so even an external Prometheus rule on tip age could not fire. A wedged +mount would silently switch off the whole alerting subsystem — the one condition +alerting exists for. Each poll therefore gives the outstanding probe a bounded +slice and then moves on, carrying the same probe forward rather than starting +another; a filesystem that never answers costs one stuck thread in total and +leaves `disk_low` on its last known verdict, with a log line saying so. Every +other detector keeps running normally. + +`deep_reorg` figures are exact. Depth, fork height and the new tip come from +the reorg log record that `perform_reorg` writes and fsyncs, rather than being +reconstructed by counting disconnect events off the event bus. The bus ring +holds 64 entries and a reorg of depth D emits `2D + 2` events in a single +burst, so a counting implementation loses precision — or drops the reorg +entirely — at exactly the depths this alert exists for. + +**`deep_reorg` pages, but does not become a standing warning.** It and +`ibd_complete` describe events, not states, so nothing would ever clear them. +They fire `alertnotify` (every occurrence — each reorg is its own event) and +emit their `status` event, but they never enter `getwarnings`. An entry nothing +clears would keep `getblockchaininfo.warnings` non-empty and the TUI's warning +modal open for the life of the process — and on signet and testnet4, where +reorgs several blocks deep are an ordinary part of the chain rather than an +incident, that would happen on the first one and never stop. A reorg's durable +record is the reorg log (`getreorghistory`); the warnings set is for conditions +that are true right now. + +Two of the new gauges are worth having independently of alerting: +`satd_tip_last_connect_age_seconds` and `satd_disk_free_bytes` — the latter +omitted rather than reported as zero when the filesystem cannot be +interrogated, and sampled whatever `alertdiskfreemb` is set to, so disabling the +alert does not delete the series out from under a Prometheus rule of your own. +Disk headroom previously had no alerting or metric at all, which is how a silent +disk-fill wedged a dogfood node in May. + ## Storage / Core compatibility ### Core v28+ XOR-obfuscated block files (`blocksxor`) diff --git a/mcp/src/context.rs b/mcp/src/context.rs index 7da0a30c0..3e72989ce 100644 --- a/mcp/src/context.rs +++ b/mcp/src/context.rs @@ -25,4 +25,7 @@ pub struct McpContext { /// Subscription registry handle for the active-subscribers gauge. /// `None` in tests that bypass main.rs. pub addr_subs: Option>, + /// Health-detector readings, so `get_metrics_snapshot` renders the same + /// health gauges as the HTTP scrape. `None` in tests that bypass main.rs. + pub health: Option>, } diff --git a/mcp/src/tools/ergonomics.rs b/mcp/src/tools/ergonomics.rs index eaecaa46b..9b0d1b1ac 100644 --- a/mcp/src/tools/ergonomics.rs +++ b/mcp/src/tools/ergonomics.rs @@ -46,6 +46,7 @@ pub fn get_metrics_snapshot(ctx: &McpContext) -> String { // address-index state, not a hardcoded "disabled" / zero. addr_subs: ctx.addr_subs.clone(), addr_enabled: ctx.addr_enabled, + health: ctx.health.clone(), }; let body = metrics_ctx.render_prometheus(); let result = json!({ @@ -78,6 +79,9 @@ pub fn get_readiness(ctx: &McpContext) -> String { version: env!("CARGO_PKG_VERSION"), addr_subs: None, addr_enabled: false, + // Readiness reads only chain heights; the health gauges are not + // consulted, so there is nothing to thread through here. + health: None, }; let (ready, reason) = match metrics_ctx.is_ready() { Ok(()) => (true, None), diff --git a/mcp/tests/tools.rs b/mcp/tests/tools.rs index 903564d1b..1ef2292fa 100644 --- a/mcp/tests/tools.rs +++ b/mcp/tests/tools.rs @@ -58,6 +58,7 @@ fn make_test_ctx() -> (McpContext, tempfile::TempDir) { mempool_history: None, addr_enabled: false, addr_subs: None, + health: None, }; (ctx, dir) } @@ -614,6 +615,7 @@ mod mining { mempool_history: None, addr_enabled: false, addr_subs: None, + health: None, }; let result = mine::generate_blocks(&ctx, 1, REGTEST_ADDR); diff --git a/node/src/diskspace.rs b/node/src/diskspace.rs new file mode 100644 index 000000000..70dece3b0 --- /dev/null +++ b/node/src/diskspace.rs @@ -0,0 +1,58 @@ +//! Free-space probe, shared by the deferred-index backfill guards and the +//! `disk_low` health detector. +//! +//! Each index runner grew its own private copy of this while the backfill +//! preflight checks were written; the health detector needs the same number, so +//! the three copies are consolidated here. Linux-only by design — the shipped +//! binaries are musl-static Linux builds, and a `None` on other platforms means +//! callers degrade to "unknown, don't block" rather than guessing. + +/// Bytes available to an unprivileged user under `path`, or `None` if the +/// filesystem cannot be interrogated (bad path, permission error, or a platform +/// with no `statvfs`). +/// +/// Uses `f_bavail`, not `f_bfree`: the reserved-blocks pool a root process could +/// still write into is not space satd may plan on. +#[cfg(target_os = "linux")] +pub fn free_disk_bytes(path: &std::path::Path) -> Option { + use std::ffi::CString; + use std::os::unix::ffi::OsStrExt; + let cpath = CString::new(path.as_os_str().as_bytes()).ok()?; + // SAFETY: zero-init s; libc::statvfs is the canonical free-space syscall. + unsafe { + let mut s: libc::statvfs = std::mem::zeroed(); + if libc::statvfs(cpath.as_ptr(), &mut s) != 0 { + return None; + } + Some(s.f_bavail.saturating_mul(s.f_frsize)) + } +} + +#[cfg(not(target_os = "linux"))] +pub fn free_disk_bytes(_path: &std::path::Path) -> Option { + None +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + #[cfg(target_os = "linux")] + fn reports_space_for_a_real_directory() { + let dir = tempfile::tempdir().unwrap(); + let free = free_disk_bytes(dir.path()).expect("statvfs on a temp dir"); + // Any writable temp dir has *some* space; the point is that the syscall + // succeeded and the multiply did not overflow to zero. + assert!(free > 0); + } + + #[test] + #[cfg(target_os = "linux")] + fn missing_path_is_none_not_a_panic() { + assert_eq!( + free_disk_bytes(std::path::Path::new("/nonexistent/satd/disk/probe")), + None, + ); + } +} diff --git a/node/src/health.rs b/node/src/health.rs new file mode 100644 index 000000000..ee8217b7f --- /dev/null +++ b/node/src/health.rs @@ -0,0 +1,2330 @@ +//! Node-health detectors: the emitters behind +//! [`StatusEvent`](crate::events::StatusEvent). +//! +//! One task watches six conditions the daemon can observe about *itself* and +//! publishes a status event whenever one is entered or recovers. The same +//! transitions are mirrored into the [`NodeWarnings`] registry, so +//! `getwarnings`, the Core-compatible `-alertnotify` hook, and the streaming +//! `status` category can never disagree about node state. +//! +//! # Firing model +//! +//! Detectors are **level-triggered**: a standing condition raises once on entry +//! and clears once on recovery. Every standing condition has a *hysteresis gap* +//! between its raise and clear lines (a clear threshold above the raise +//! threshold, a hold time, or both), because the alternative — clearing at the +//! same value that raises — turns a metric hovering at the line into a pager +//! storm. The gaps are fixed constants (§ [`hysteresis`]); the raise thresholds +//! are operator-configurable and SIGHUP-live. +//! +//! Two conditions have no recovered state and are emitted as one-shot edges: +//! IBD finishing, and a reorg deeper than the configured floor landing. +//! +//! # Durability +//! +//! Status events are not replayable (no cursor, not in the replay ring). What +//! makes health alerting at-least-once across a restart is that this task +//! re-evaluates from scratch: a condition that is still true when the node +//! comes back is raised again, because the detector has no memory of having +//! raised it before. A condition that both raised and fully cleared while a +//! consumer was away is stale by definition and is not reconstructed. +//! +//! # Cost +//! +//! The poll loop runs every [`POLL_INTERVAL`] and reads only atomics and +//! lock-cheap accessors plus one `statvfs`. The chain-event half is driven by +//! the existing broadcast. Publishing is best-effort: with no `status` +//! subscriber and no webhook attached, the envelope is dropped by the +//! publisher's zero-receiver path. + +use std::sync::Arc; +use std::sync::atomic::{AtomicBool, AtomicU64, Ordering}; +use std::time::Instant; + +use tokio::sync::{broadcast, watch}; +use tokio::time::{Duration, MissedTickBehavior, interval}; + +use crate::chain::events::ChainEvent; +use crate::chain::reorg_log::ReorgRecord; +use crate::chain::state::ChainState; +use crate::events::status::{StatusEvent, StatusKind, StatusSeverity}; +use crate::events::{EventPublisher, StatusState}; +use crate::mempool::pool::Mempool; +use crate::net::manager::PeerManager; +use crate::warnings::{NodeWarnings, Severity}; + +/// How often the polled detectors (`disk_low`, `mempool_congested`, +/// `peer_floor`, and the tip-stall timer) re-evaluate. +pub const POLL_INTERVAL: Duration = Duration::from_secs(15); + +/// How far back to search the reorg log on each poll. +/// +/// The whole ring, deliberately. A bounded window was a second way to lose an +/// edge permanently: `deep_reorg` is never reconstructed, so any delay of this +/// task past the window — API-runtime saturation, a `statvfs` blocking on a +/// hung mount, a VM pause, `SIGSTOP` — dropped every reorg older than it out of +/// view for good. The window bought nothing, because [`ReorgSeen`] is already +/// an exact de-duplicator; the only cost of scanning everything is cloning at +/// most `DEFAULT_RING_CAPACITY` (256) records per poll. +const REORG_LOG_LOOKBACK_SECS: u64 = u64::MAX; + +/// Fixed hysteresis constants. Deliberately not configurable: six raise +/// thresholds is already a lot of operator surface, and these ratios only need +/// to be "enough of a gap that a metric sitting on the line does not flap". +pub mod hysteresis { + use super::Duration; + + /// `disk_low` clears at 1.5× the raise floor — recovering from a disk + /// alert usually means deleting something, and clearing at exactly the + /// floor would re-raise on the next block written. + pub const DISK_CLEAR_RATIO_NUM: u64 = 3; + pub const DISK_CLEAR_RATIO_DEN: u64 = 2; + + /// `mempool_congested` clears below 0.75× the raise line. A mempool at its + /// cap evicts continuously, so occupancy oscillates around the cap by + /// design; a tight clear line would emit a raise/clear pair per block. + pub const MEMPOOL_CLEAR_RATIO_NUM: u64 = 3; + pub const MEMPOOL_CLEAR_RATIO_DEN: u64 = 4; + + /// `peer_floor` requires the condition to hold for this long in *either* + /// direction. Peer counts dip transiently during normal churn (a peer + /// disconnects, the manager dials a replacement within seconds); alerting + /// on the instantaneous count would be noise. + pub const PEER_HOLD: Duration = Duration::from_secs(60); + + /// Grace from detector start until the node's first peer, during which + /// `peer_floor` does not raise. Outbound connections are dialed + /// concurrently with the rest of startup and the poll's first tick fires + /// immediately, so without a grace every node alerts once on the way up. + /// The grace ends at the first peer, so it does not blunt the alert for a + /// node that connects and *later* loses its peers. + pub const PEER_STARTUP_GRACE: Duration = Duration::from_secs(90); +} + +/// Operator-tunable raise thresholds, shared with the SIGHUP reload path. +/// +/// Every threshold is "0 disables this detector" — an operator who does not +/// want a given alert sets it to zero rather than having to know a magic +/// sentinel. All fields are plain atomics: the reload path stores, the detector +/// loads, and a torn read is impossible for these widths. +#[derive(Debug)] +pub struct AlertThresholds { + tip_stall_secs: AtomicU64, + disk_free_bytes: AtomicU64, + mempool_full_pct: AtomicU64, + peer_floor: AtomicU64, + reorg_depth: AtomicU64, +} + +/// Default raise thresholds, mirrored by the `alert*` config-key defaults. +pub mod defaults { + /// One hour without a connected block. Not gated on IBD — see + /// `check_tip_stall_values`, which explains why at length. At mainnet's 10-minute + /// target roughly 0.25 % of blocks take longer than an hour by chance, so + /// this fires spuriously about once every few days on a healthy node — + /// deliberate: a stalled node is worth a look, and an operator who finds it + /// noisy raises the value (the manual documents the trade). + pub const TIP_STALL_SECS: u64 = 3_600; + + /// The `tip_stall` default for `network`. + /// + /// Disabled on regtest, for the same reason as the peer floor and the reorg + /// depth: regtest blocks exist only when someone calls + /// `generatetoaddress`, so an idle chain is its resting state and not a + /// stall. `last_connect` is seeded at detector start and advanced only by + /// `BlockConnected`, so a developer's node left running for an hour while + /// they write code — or a harness that mines a fixture and then sits — + /// raises a *critical* alert. That pins `getwarnings`, holds `has_errors()` + /// true, and puts up the TUI's blocking modal, on a chain that is behaving + /// exactly as designed. + /// + /// Every other network keeps the hour. The spurious-raise rate on mainnet + /// is a deliberate trade documented on [`TIP_STALL_SECS`], and a test + /// network that goes an hour without a block is worth reporting even though + /// its hashrate is thin — unlike a reorg a few blocks deep, a stall there is + /// not an ordinary property of the network. + pub fn tip_stall_for(network: bitcoin::Network) -> u64 { + match network { + bitcoin::Network::Regtest => 0, + _ => TIP_STALL_SECS, + } + } + /// 10 GiB. Enough headroom to notice before a mainnet node wedges mid-block + /// (the 2026-05-13 dogfood incident was a silent disk-fill). + pub const DISK_FREE_MB: u64 = 10_240; + /// Percent of the mempool byte cap. + pub const MEMPOOL_FULL_PCT: u64 = 90; + /// Connected peers, on a network where a node is expected to have some. + pub const PEER_FLOOR: u64 = 3; + + /// The `peer_floor` default for `network`, capped by `connect_peers` — the + /// number of `-connect=` addresses the operator configured. + /// + /// Disabled on regtest only. A regtest node is routinely run entirely + /// alone, so "fewer than 3 peers" is its normal operating state rather than + /// a fault, and the default would raise a critical warning that can never + /// clear — one that drives `getwarnings`, `-alertnotify`, and the TUI's + /// blocking modal, which is a poor greeting for every developer's first run. + /// + /// Every other network keeps the real floor, signet included. Signet is a + /// public network with real peers; a peer-starved signet node is broken in + /// exactly the way this alert exists to report, and defaulting it off would + /// make the detector's silence indistinguishable from health — + /// `satd_alert_active{kind="peer_floor"}` reads 0 either way. An operator + /// running a deliberately isolated signet can set `alertpeerfloor=0`. + /// + /// `-connect=` is the same trap as regtest wearing different clothes. It + /// pins the node to exactly the addresses given and suppresses both DNS + /// seeding and the fixed seeds, so a node with one or two of them can never + /// reach a floor of 3 — no code path is left that would add a peer. The + /// floor is a property of the configuration, not only of the network, so it + /// follows the count the operator declared. Capping rather than disabling + /// keeps the alert working for what it is actually good for here: a + /// `-connect=` node that had all its configured peers and then lost one. + pub fn peer_floor_for(network: bitcoin::Network, connect_peers: usize) -> u64 { + match network { + bitcoin::Network::Regtest => 0, + _ if connect_peers > 0 => PEER_FLOOR.min(connect_peers as u64), + _ => PEER_FLOOR, + } + } + /// Blocks rolled back, on a chain where a reorg this deep is an incident. + pub const REORG_DEPTH: u64 = 3; + + /// Blocks rolled back, on a chain where reorgs are an ordinary property of + /// the network rather than an incident. + pub const REORG_DEPTH_TEST_NETWORK: u64 = 10; + + /// The `deep_reorg` default for `network`. + /// + /// Depth 3 means completely different things on different chains, and the + /// alert is only worth having if crossing it means something is wrong. + /// + /// On mainnet a 3-block reorg is a genuine incident: it costs real hashrate + /// to produce, and it invalidates transactions that merchants have begun + /// treating as settled. Waking someone is the correct response. + /// + /// The test networks are not economically secured, and reorgs a few blocks + /// deep are a *normal operating property* of them — a consequence of thin, + /// volatile hashrate (and, on testnet, of the difficulty exception). Paging + /// on those is paging on the network working as designed, and an alert that + /// fires during normal operation is one operators learn to ignore, which + /// costs them the mainnet alert too. The floor is raised rather than + /// disabled: a reorg past the 6-confirmation convention has invalidated + /// something a wallet would have called final, and that is worth reporting + /// on any chain. An operator who wants the mainnet sensitivity sets + /// `alertreorgdepth=3` explicitly. + /// + /// Regtest is off entirely. Test harnesses reorg deliberately and + /// constantly — `invalidateblock` and competing-chain tests are the point — + /// so any threshold would fire on the suite doing its job. + /// + /// An unrecognized future network takes the test-network value: new + /// networks are overwhelmingly test networks, and the failure mode of + /// guessing that way (an alert that fires slightly less often than it + /// could) is the milder one. + pub fn reorg_depth_for(network: bitcoin::Network) -> u64 { + match network { + bitcoin::Network::Bitcoin => REORG_DEPTH, + bitcoin::Network::Regtest => 0, + _ => REORG_DEPTH_TEST_NETWORK, + } + } +} + +impl Default for AlertThresholds { + fn default() -> Self { + Self::new( + defaults::TIP_STALL_SECS, + defaults::DISK_FREE_MB, + defaults::MEMPOOL_FULL_PCT, + defaults::PEER_FLOOR, + defaults::REORG_DEPTH, + ) + } +} + +impl AlertThresholds { + /// Build from the operator's configured values. `disk_free_mb` is taken in + /// mebibytes (the config unit) and stored as bytes. + pub fn new( + tip_stall_secs: u64, + disk_free_mb: u64, + mempool_full_pct: u64, + peer_floor: u64, + reorg_depth: u64, + ) -> Self { + let s = Self { + tip_stall_secs: AtomicU64::new(0), + disk_free_bytes: AtomicU64::new(0), + mempool_full_pct: AtomicU64::new(0), + peer_floor: AtomicU64::new(0), + reorg_depth: AtomicU64::new(0), + }; + s.set_tip_stall_secs(tip_stall_secs); + s.set_disk_free_mb(disk_free_mb); + s.set_mempool_full_pct(mempool_full_pct); + s.set_peer_floor(peer_floor); + s.set_reorg_depth(reorg_depth); + s + } + + pub fn set_tip_stall_secs(&self, v: u64) { + self.tip_stall_secs.store(v, Ordering::Relaxed); + } + pub fn set_disk_free_mb(&self, mb: u64) { + self.disk_free_bytes + .store(mb.saturating_mul(1024 * 1024), Ordering::Relaxed); + } + /// Values above 100 are clamped: a percentage over 100 can never be reached, + /// which would silently disable the detector rather than doing what the + /// operator meant. + pub fn set_mempool_full_pct(&self, v: u64) { + self.mempool_full_pct.store(v.min(100), Ordering::Relaxed); + } + pub fn set_peer_floor(&self, v: u64) { + self.peer_floor.store(v, Ordering::Relaxed); + } + pub fn set_reorg_depth(&self, v: u64) { + self.reorg_depth.store(v, Ordering::Relaxed); + } + + pub fn tip_stall_secs(&self) -> u64 { + self.tip_stall_secs.load(Ordering::Relaxed) + } + pub fn disk_free_bytes(&self) -> u64 { + self.disk_free_bytes.load(Ordering::Relaxed) + } + pub fn mempool_full_pct(&self) -> u64 { + self.mempool_full_pct.load(Ordering::Relaxed) + } + pub fn peer_floor(&self) -> u64 { + self.peer_floor.load(Ordering::Relaxed) + } + pub fn reorg_depth(&self) -> u64 { + self.reorg_depth.load(Ordering::Relaxed) + } +} + +/// Live health readings, published by the detector task and read by the +/// `/metrics` renderer. Separate from [`AlertThresholds`] because these flow the +/// other way: the detector writes, everything else reads. +#[derive(Debug, Default)] +pub struct HealthState { + /// One flag per [`StatusKind`], indexed by position in [`StatusKind::ALL`]. + /// Edge kinds stay `false` — they have no standing state. + active: [AtomicBool; StatusKind::ALL.len()], + /// Seconds since the last block connected (or since the detector started, + /// whichever is more recent — see `spawn_health_detectors`). + last_connect_age_secs: AtomicU64, + /// Last observed free space under the data directory. `u64::MAX` means + /// "not yet sampled / unavailable", which the renderer skips rather than + /// reporting a misleading zero. + disk_free_bytes: AtomicU64, + /// The threshold value in force when each condition was raised, indexed as + /// `active`. Read by [`clear_if_threshold_relaxed`] to tell "the reading + /// recovered" apart from "the operator moved the line". + raised_at_threshold: [AtomicU64; StatusKind::ALL.len()], + /// Latched once the node has had at least one peer. Gates the + /// `peer_floor` hold clock so a node that has not finished dialing out yet + /// does not alert on a startup transient. + saw_first_peer: AtomicBool, +} + +/// Sentinel for "no disk reading yet" — distinct from a genuine zero-free-space +/// reading, which is exactly the situation an operator most needs to see. +const DISK_UNKNOWN: u64 = u64::MAX; + +impl HealthState { + pub fn new() -> Self { + Self { + disk_free_bytes: AtomicU64::new(DISK_UNKNOWN), + ..Default::default() + } + } + + /// Whether a standing condition is currently raised. + pub fn is_active(&self, kind: StatusKind) -> bool { + self.slot(kind).load(Ordering::Relaxed) + } + + /// Seconds since the last connected block (or since detector start). + pub fn last_connect_age_secs(&self) -> u64 { + self.last_connect_age_secs.load(Ordering::Relaxed) + } + + /// Last sampled free space under the data directory, or `None` if the + /// filesystem has not been (or cannot be) interrogated. + pub fn disk_free_bytes(&self) -> Option { + match self.disk_free_bytes.load(Ordering::Relaxed) { + DISK_UNKNOWN => None, + v => Some(v), + } + } + + /// Test hook: drive a standing flag without running a detector. + #[cfg(test)] + pub fn set_active_for_test(&self, kind: StatusKind, on: bool) { + self.set_active(kind, on); + } + + /// Test hook: seed a free-space reading (or clear it back to unknown). + #[cfg(test)] + pub fn set_disk_free_for_test(&self, free: Option) { + self.disk_free_bytes + .store(free.unwrap_or(DISK_UNKNOWN), Ordering::Relaxed); + } + + fn slot(&self, kind: StatusKind) -> &AtomicBool { + let idx = StatusKind::ALL + .iter() + .position(|k| *k == kind) + .expect("every StatusKind is in StatusKind::ALL"); + &self.active[idx] + } + + fn set_active(&self, kind: StatusKind, on: bool) { + self.slot(kind).store(on, Ordering::Relaxed); + } + + fn threshold_slot(&self, kind: StatusKind) -> &AtomicU64 { + let idx = StatusKind::ALL + .iter() + .position(|k| *k == kind) + .expect("every StatusKind is in StatusKind::ALL"); + &self.raised_at_threshold[idx] + } +} + +/// Everything the detector task reads. Grouped into a struct because the +/// spawn function would otherwise take eight positional arguments. +pub struct HealthInputs { + pub chain_state: Arc, + pub mempool: Arc, + pub peer_manager: Arc, + pub publisher: Arc, + pub warnings: Arc, + pub thresholds: Arc, + /// Directory whose free space `disk_low` watches — the blocks directory + /// when it is split out, since that is what actually grows. + pub disk_watch_path: std::path::PathBuf, +} + +/// Spawn the health-detector task and return the state handle the `/metrics` +/// renderer reads. +/// +/// Spawns on the *calling* runtime, so the daemon must call this from within +/// the isolated API runtime: a detector that shared the consensus runtime could +/// have its poll delayed by block connection, which is the exact opposite of +/// what a stall detector is for. +pub fn spawn_health_detectors( + inputs: HealthInputs, + chain_rx: broadcast::Receiver, + shutdown: watch::Receiver, +) -> Arc { + let state = Arc::new(HealthState::new()); + let task_state = state.clone(); + tokio::spawn(async move { + run_detectors(inputs, chain_rx, shutdown, task_state).await; + }); + state +} + + +async fn run_detectors( + inputs: HealthInputs, + mut chain_rx: broadcast::Receiver, + mut shutdown: watch::Receiver, + state: Arc, +) { + let HealthInputs { + chain_state, + mempool, + peer_manager, + publisher, + warnings, + thresholds, + disk_watch_path, + } = inputs; + + let mut poll = interval(POLL_INTERVAL); + // A delayed tick must not turn into a burst of catch-up ticks: each tick + // does a `statvfs`, and a runtime hiccup should cost one late poll, not N + // immediate ones. + poll.set_missed_tick_behavior(MissedTickBehavior::Delay); + + // Seed the stall clock from *now*, not from the tip's timestamp. A node + // that was down for hours has a stale tip but will connect its backlog + // within seconds of starting; seeding from the tip would page the operator + // for a stall that is really just a restart. + let mut last_connect = Instant::now(); + // IBD completion is a one-shot per process, and only meaningful for a node + // that actually started in IBD — otherwise every restart of a synced node + // would announce that it finished syncing. + let mut ibd_pending = chain_state.is_initial_block_download(); + // Which reorg-log records have already been reported. Seeded from what the + // log already holds so a restart does not re-announce reorgs that predate + // it: `deep_reorg` is an edge event, and D3 re-raises standing conditions + // across a restart, not edges. + let mut reorgs_seen = ReorgSeen::default(); + if let Some(log) = chain_state.reorg_log() { + reorgs_seen.seed(&log.history(REORG_LOG_LOOKBACK_SECS)); + } + // Hold-time trackers for `peer_floor`: the condition must persist in either + // direction before it is acted on. + let mut peers_below_since: Option = None; + let mut peers_ok_since: Option = None; + // Anchors the `peer_floor` startup grace. Distinct from `last_connect`, + // which is reset by every block. + let detector_start = Instant::now(); + // Outstanding `statvfs`, collected on the following poll. See `check_disk`. + let mut disk_probe: Option = None; + + loop { + tokio::select! { + // Unconditional, like every other shutdown handler in the tree. + // `changed()` returns `Err` immediately and forever once the last + // sender drops, while `borrow()` still reads whatever value was + // last set — so gating the return on `*shutdown.borrow()` turns a + // dropped sender into a 100%-CPU spin on an API worker instead of a + // clean exit. A sender dropped without setting `true` means nobody + // is left to ask us to stop, which is a stop. + _ = shutdown.changed() => return, + ev = chain_rx.recv() => { + match ev { + Ok(ChainEvent::BlockConnected { height, .. }) => { + last_connect = Instant::now(); + state.last_connect_age_secs.store(0, Ordering::Relaxed); + clear_if_active( + &state, &warnings, &publisher, + StatusKind::TipStall, + format!("tip advanced to height {height}"), + |e| e.with_detail("height", height), + ); + if ibd_pending && !chain_state.is_initial_block_download() { + ibd_pending = false; + emit( + &warnings, + &publisher, + StatusEvent::edge( + StatusKind::IbdComplete, + format!("initial block download complete at height {height}"), + ) + .with_detail("height", height), + ); + } + } + Ok(ChainEvent::BlockDisconnected { .. }) + | Ok(ChainEvent::Reorg { .. }) => { + // Reorg depth is read from the reorg log at poll time, + // not reconstructed from these events. See + // `scan_reorg_log`. + } + Err(broadcast::error::RecvError::Lagged(n)) => { + // Nothing here depends on a complete event run. The + // tip-stall clock is advanced by `BlockConnected`, but + // a drop only delays it: the next retained connect + // still arrives and still resets it, and a node with no + // further blocks is stalled — which is what the + // detector is for. Reorg depth comes from the durable + // log, not from these events. Lag is worth a line and + // nothing more. + tracing::debug!( + target: "health", + dropped = n, + "chain-event lag in the health detector", + ); + } + Err(broadcast::error::RecvError::Closed) => return, + } + } + _ = poll.tick() => { + // Report any reorg the log recorded since the last poll. + // Depth comes from the record, never from counting events. + scan_reorg_log( + &warnings, &publisher, &thresholds, &chain_state, &mut reorgs_seen, + ); + let age = last_connect.elapsed().as_secs(); + state.last_connect_age_secs.store(age, Ordering::Relaxed); + check_tip_stall(&state, &warnings, &publisher, &thresholds, &chain_state, age); + check_disk( + &state, &warnings, &publisher, &thresholds, &disk_watch_path, + &mut disk_probe, DISK_PROBE_BUDGET, + ) + .await; + check_mempool(&state, &warnings, &publisher, &thresholds, &mempool); + check_peers( + &state, &warnings, &publisher, &thresholds, &peer_manager, + &mut peers_below_since, &mut peers_ok_since, &detector_start, + ); + } + } + } +} + +/// Publish a status event and mirror it into the warnings registry. +/// +/// Only conditions worth operator attention become warnings: an `info` event +/// (IBD finishing) is good news, not a problem, and would otherwise sit in +/// `getwarnings` forever. A `raised` event records one, a `cleared` event +/// removes it. +/// +/// An `edge` event pages but records nothing — see the note at the call site. +/// The registry is for conditions that are true *now* and that something will +/// later clear; a deep reorg is history, and history has its own log. +fn emit(warnings: &NodeWarnings, publisher: &EventPublisher, event: StatusEvent) { + let id = event.kind.warning_id(); + match event.state { + StatusState::Cleared => warnings.clear(&id), + StatusState::Raised | StatusState::Edge => { + if event.severity >= StatusSeverity::Warning { + let severity = match event.severity { + StatusSeverity::Critical => Severity::Error, + _ => Severity::Warn, + }; + // An edge observation fires the shell hook but does **not** + // become a standing warning. + // + // `NodeWarnings` holds conditions that are currently true and + // that something will later clear; its own contract says + // history-style events keep their own logs, and that an active + // warning means a problem to go fix. A deep reorg has no + // resolved state, so nothing would ever clear it — it would pin + // `getwarnings`, hold `has_errors()` true for the life of the + // process, and keep the TUI's blocking modal up. On signet and + // testnet4, where reorgs several blocks deep are ordinary, the + // first one would do that permanently. The durable record is + // `ReorgLog` plus the `status` event; this is only the page. + if event.state == StatusState::Edge { + warnings.notify_event(&id, severity, event.message.clone()); + } else { + let context = serde_json::to_value(&event.details) + .unwrap_or(serde_json::Value::Null); + warnings.record(&id, severity, event.message.clone(), context); + } + } + } + } + tracing::info!( + target: "health", + kind = event.kind.as_str(), + state = ?event.state, + severity = event.severity.as_str(), + "{}", + event.message, + ); + publisher.publish_status(event); +} + +/// Raise a standing condition if it is not already raised. +fn raise_if_new( + state: &HealthState, + warnings: &NodeWarnings, + publisher: &EventPublisher, + kind: StatusKind, + threshold: u64, + message: String, + details: impl FnOnce(StatusEvent) -> StatusEvent, +) { + // Remember the line this is being raised against, so a later poll can tell + // a recovered reading from a retuned threshold. See + // `clear_if_threshold_relaxed`. + // + // Refreshed on every evaluation where the raise predicate holds, not only + // on the raise edge. Storing it only on the edge leaves the slot stale + // across a retune that keeps the condition raised, and the next recovery + // into the hysteresis band then reads as "the operator moved the line": + // raise at 93% against a 90% threshold, retune *down* to 80% (still + // raised, so an edge-only store keeps 90), ease to 78% — below the new + // raise line, above the new clear line — and the stale 90 ≠ 80 clears an + // alert that both the old and new hysteresis lines say to hold. + state.threshold_slot(kind).store(threshold, Ordering::Relaxed); + if state.is_active(kind) { + return; + } + state.set_active(kind, true); + emit(warnings, publisher, details(StatusEvent::raised(kind, message))); +} + +/// Clear a standing condition whose **threshold moved** rather than whose +/// reading recovered. +/// +/// Every level-triggered detector clears on a hysteresis-widened predicate: disk +/// clears at 1.5× the floor, mempool at 0.75× the raise line. That gap is there +/// to stop a value hovering at the line from flapping, and it does its job — but +/// it must not also trap the operator who retunes the threshold *because* the +/// alert is firing. Raising the threshold moves both the raise line and the +/// clear line, so the current reading can land in the new dead band where +/// neither branch runs, and the alert stays raised against a threshold it no +/// longer violates. +/// +/// For `mempool_congested` that trap is inescapable rather than merely awkward: +/// the percentage clamps at 100, so the highest reachable clear line is 75% of +/// the cap. Once occupancy is at or above that, no value of +/// `alertmempoolfullpct` can clear a raised alert — only disabling the detector +/// outright. +/// +/// So: if the raise predicate no longer holds under the *current* threshold, and +/// that threshold differs from the one the condition was raised against, clear. +/// A reading that recovers on its own still goes through the hysteresis path; +/// this only fires when the operator actually moved the line. +fn clear_if_threshold_relaxed( + state: &HealthState, + warnings: &NodeWarnings, + publisher: &EventPublisher, + kind: StatusKind, + threshold: u64, + message: String, + details: impl FnOnce(StatusEvent) -> StatusEvent, +) { + if !state.is_active(kind) { + return; + } + if state.threshold_slot(kind).load(Ordering::Relaxed) == threshold { + return; + } + clear_if_active(state, warnings, publisher, kind, message, details); +} + +/// Clear a standing condition if it is currently raised. +fn clear_if_active( + state: &HealthState, + warnings: &NodeWarnings, + publisher: &EventPublisher, + kind: StatusKind, + message: String, + details: impl FnOnce(StatusEvent) -> StatusEvent, +) { + if !state.is_active(kind) { + return; + } + state.set_active(kind, false); + emit(warnings, publisher, details(StatusEvent::cleared(kind, message))); +} + +/// A disabled detector must not leave a previously-raised condition standing +/// forever: if the operator turns the threshold off while it is raised, clear it +/// (with the reason) rather than stranding a warning nothing will ever retract. +fn clear_because_disabled( + state: &HealthState, + warnings: &NodeWarnings, + publisher: &EventPublisher, + kind: StatusKind, +) { + clear_with_reason( + state, + warnings, + publisher, + kind, + format!("{} detector disabled by configuration", kind.as_str()), + "detector_disabled", + ); +} + +/// Clear a standing condition because the detector can no longer evaluate it, +/// tagging the wire event with *why*. +/// +/// `reason` is a stable token a receiver may route on, so it has to distinguish +/// causes an operator would act on differently: "you turned this off" and "the +/// input this detector divides by went to zero" are not the same message. +fn clear_with_reason( + state: &HealthState, + warnings: &NodeWarnings, + publisher: &EventPublisher, + kind: StatusKind, + message: String, + reason: &'static str, +) { + clear_if_active( + state, + warnings, + publisher, + kind, + message, + |e| e.with_detail("reason", reason), + ); +} + +fn check_tip_stall( + state: &HealthState, + warnings: &NodeWarnings, + publisher: &EventPublisher, + thresholds: &AlertThresholds, + chain_state: &ChainState, + age_secs: u64, +) { + check_tip_stall_values( + state, + warnings, + publisher, + thresholds, + chain_state.is_initial_block_download(), + chain_state.tip_height(), + age_secs, + ); +} + +/// The tip-stall detector's decision logic, over plain readings. +/// +/// Split from [`check_tip_stall`] so the IBD latch is testable without a live +/// `ChainState` — the tests exercise this exact code rather than a +/// reimplementation of it. +fn check_tip_stall_values( + state: &HealthState, + warnings: &NodeWarnings, + publisher: &EventPublisher, + thresholds: &AlertThresholds, + in_ibd: bool, + tip_height: u32, + age_secs: u64, +) { + let threshold = thresholds.tip_stall_secs(); + if threshold == 0 { + clear_because_disabled(state, warnings, publisher, StatusKind::TipStall); + return; + } + // `in_ibd` deliberately does not suppress this alert. + // + // It is tempting to: during a genuine sync the tip advances in bursts and a + // stall alert would be noise. But `is_initial_block_download` is not a sync + // flag — it compares the tip header's timestamp against the wall clock, so + // a node that is fully caught up and then *stops* re-enters it a day later, + // exactly when the operator most needs to hear from it. + // + // Latching on "we once saw a non-IBD tip" does not close that hole, because + // the latch lives in this process and restarting is an operator's first + // move during a stall. A node restarted while already wedged never observes + // a non-IBD tip, so the latch never arms and this detector goes silent + // permanently — at precisely the moment it should be paging. + // + // `age_secs` already encodes the thing worth gating on. A node that is + // really syncing connects blocks continuously, which keeps the age far + // below any sane threshold and suppresses the alert on its own. A node that + // reads as "in IBD" but has connected nothing for the whole threshold is + // stalled whether it is wedged mid-sync or wedged at the tip, and both + // warrant the page. The message — "no block connected for Ns" — is true + // either way. + let _ = in_ibd; + if age_secs >= threshold { + raise_if_new( + state, + warnings, + publisher, + StatusKind::TipStall, + threshold, + format!("no block connected for {age_secs}s (threshold {threshold}s)"), + |e| { + e.with_detail("seconds_since_block", age_secs) + .with_detail("threshold_seconds", threshold) + .with_detail("tip_height", tip_height) + }, + ); + } else { + // The fast clear is event-driven (on `BlockConnected`), because the + // point of the alert is that it lifts the instant the chain moves. This + // is the slow one: it exists for the case where the *threshold* moved + // instead of the tip. An operator who raises `alerttipstallseconds` via + // SIGHUP to quiet a firing alert would otherwise stay raised until some + // future block connects — and on a chain that only looks stalled under + // the old threshold, that block may be a long way off. + clear_if_active( + state, + warnings, + publisher, + StatusKind::TipStall, + format!("tip age {age_secs}s is within the threshold ({threshold}s)"), + |e| { + e.with_detail("seconds_since_block", age_secs) + .with_detail("threshold_seconds", threshold) + }, + ); + } +} + +/// Sample the watched volume and evaluate `disk_low`. +/// +/// `async` purely so the `statvfs` can go to `spawn_blocking`. `disk_watch_path` +/// defaults to `blocksdir`, which operators routinely point at NFS or iSCSI, and +/// `statvfs` on a hung network mount blocks uninterruptibly. Called inline it +/// would park an API-runtime worker and — since every detector shares this one +/// task — freeze *all* of them, `tip_stall` and `deep_reorg` included, for as +/// long as the mount stayed wedged. +/// How many stalled polls between repeat warnings about an unresponsive +/// filesystem. At the 15 s poll interval this is roughly hourly. +const STALL_LOG_EVERY: u32 = 240; + +/// How long one poll will wait on an outstanding `statvfs` before giving up on +/// it for this tick. Comfortably under the poll interval, so a filesystem that +/// answers at all is read inline and the detector never falls behind; a +/// filesystem that does not answer costs this much per poll and nothing more. +const DISK_PROBE_BUDGET: std::time::Duration = std::time::Duration::from_secs(2); + +/// An outstanding `statvfs`, carried across polls so a filesystem that never +/// answers strands exactly one blocking thread instead of one per poll. +struct DiskProbe { + handle: tokio::task::JoinHandle>, + /// Consecutive polls this probe has failed to finish in. Drives the + /// log-rate limiting only. + stalled_polls: u32, +} + +async fn check_disk( + state: &HealthState, + warnings: &NodeWarnings, + publisher: &EventPublisher, + thresholds: &AlertThresholds, + path: &std::path::Path, + pending: &mut Option, + budget: std::time::Duration, +) { + // Sample before any early return, so `satd_disk_free_bytes` is populated + // whatever the detector's configuration. An operator who sets + // `alertdiskfreemb=0` has usually done so *because* they alert on the gauge + // in Prometheus instead of via satd; returning before the filesystem read + // would delete the series out from under their own rule, silently. An + // unreadable filesystem reports "unknown" rather than a zero that would + // read as "completely full". + // + // The probe is collected on the *next* poll rather than awaited on this + // one. `statvfs` on a hard NFS mount whose server has gone away does not + // return, and `spawn_blocking(..).await` inherits that wait in full: it + // moves the syscall off the detector's thread but still parks the detector + // on the `JoinHandle`. Every other detector — tip stall, deep reorg, + // mempool, peers — then stops running, `chain_rx` stops being drained, and + // every gauge freezes at its last value, so an external Prometheus rule on + // tip age cannot fire either. A wedged mount would silently disable the + // whole alerting subsystem, which is the one condition alerting exists for. + // + // A timeout around the join would not fix it: `tokio::time::timeout` + // abandons the handle but cannot cancel a blocking task, so each poll would + // strand another thread and exhaust the (bounded) blocking pool within + // hours — trading a wedged detector for a wedged runtime. Holding the + // handle across polls bounds the damage at exactly one stuck thread. + let sample = match pending.as_mut() { + Some(probe) => { + // `&mut JoinHandle` is itself a future, so a timeout here does NOT + // consume the handle — which is the whole point. `timeout(_, handle)` + // by value would abandon it, and since a blocking task cannot be + // cancelled, the next poll would spawn another and the one after + // that another, exhausting the bounded blocking pool within hours. + match tokio::time::timeout(budget, &mut probe.handle).await { + Ok(joined) => joined.unwrap_or(None), + Err(_) => { + probe.stalled_polls += 1; + let stalls = probe.stalled_polls; + // Loud once, then hourly: a hung filesystem is worth saying, + // and worth repeating for whoever reads the log later, but + // four identical lines a minute buries everything else. + if stalls == 1 || stalls.is_multiple_of(STALL_LOG_EVERY) { + tracing::warn!( + target: "health", + path = %path.display(), + stalled_polls = stalls, + budget_secs = budget.as_secs(), + "disk-space probe has not returned; the filesystem is \ + not responding. Free-space alerting is stalled until \ + it does — every other health detector keeps running.", + ); + } + // Leave the gauge and the alert verdict on their last known + // values. Overwriting with "unknown" would clear a + // `disk_low` that is very likely still true — an + // unresponsive mount is not evidence the disk drained. + return; + } + } + } + // First poll of the process: nothing outstanding to collect. Start one + // below and read it next tick. + None => { + let path = path.to_path_buf(); + *pending = Some(DiskProbe { + handle: tokio::task::spawn_blocking(move || { + crate::diskspace::free_disk_bytes(&path) + }), + stalled_polls: 0, + }); + // "Not measured yet" must not reach the gauge as "unmeasurable". + return; + } + }; + // Collected: start the next one so the following poll has something to read. + { + let path = path.to_path_buf(); + *pending = Some(DiskProbe { + handle: tokio::task::spawn_blocking(move || crate::diskspace::free_disk_bytes(&path)), + stalled_polls: 0, + }); + } + state + .disk_free_bytes + .store(sample.unwrap_or(DISK_UNKNOWN), Ordering::Relaxed); + // Log on the raise edge only. Firing this every poll while the condition + // holds is 4 identical WARN lines a minute — 5,760 a day — which buries the + // rest of the log for exactly as long as the operator has a real problem. + // Checked before `check_disk_values` runs, so `is_active` still reads the + // previous poll's verdict. + if let Some(free) = sample + && free < thresholds.disk_free_bytes() + && !state.is_active(StatusKind::DiskLow) + { + // The path is deliberately NOT a wire detail. It goes to every `status` + // subscriber, every webhook receiver, and onward into APNs/FCM push + // bodies — an absolute datadir path typically containing the operator's + // username. The node's own log records which volume this is. + tracing::warn!( + target: "health", + path = %path.display(), + free_bytes = free, + threshold_bytes = thresholds.disk_free_bytes(), + "free space below the configured floor" + ); + } + check_disk_values(state, warnings, publisher, thresholds, sample); +} + +/// The disk detector's decision logic, over a plain reading. +/// +/// Split from [`check_disk`] so the threshold and hysteresis behavior is +/// testable without driving a real filesystem to a target free-space level — +/// the tests exercise this exact code rather than a reimplementation of it. +/// `free` is `None` when the volume could not be interrogated. +fn check_disk_values( + state: &HealthState, + warnings: &NodeWarnings, + publisher: &EventPublisher, + thresholds: &AlertThresholds, + free: Option, +) { + let floor = thresholds.disk_free_bytes(); + // The disabled-clear is checked before the unreadable-filesystem return, + // not after. If the watched volume becomes uninterrogable while `disk_low` + // is raised (the blocks volume is unmounted, `-blocksdir` points somewhere + // that disappeared), an early return below would leave the alert raised + // with no way to clear it — not even by setting `alertdiskfreemb=0`, which + // is the documented escape hatch for every other detector. + if floor == 0 { + clear_because_disabled(state, warnings, publisher, StatusKind::DiskLow); + return; + } + let Some(free) = free else { + return; + }; + let clear_at = floor + .saturating_mul(hysteresis::DISK_CLEAR_RATIO_NUM) + / hysteresis::DISK_CLEAR_RATIO_DEN; + if free < floor { + raise_if_new( + state, + warnings, + publisher, + StatusKind::DiskLow, + floor, + format!( + "free space {} MiB below floor {} MiB", + free / (1024 * 1024), + floor / (1024 * 1024) + ), + |e| { + e.with_detail("free_bytes", free) + .with_detail("threshold_bytes", floor) + }, + ); + } else if free >= clear_at { + clear_if_active( + state, + warnings, + publisher, + StatusKind::DiskLow, + format!("free space recovered to {} MiB", free / (1024 * 1024)), + |e| { + e.with_detail("free_bytes", free) + .with_detail("clear_threshold_bytes", clear_at) + }, + ); + } else { + clear_if_threshold_relaxed( + state, + warnings, + publisher, + StatusKind::DiskLow, + floor, + format!( + "free space {} MiB is within the floor ({} MiB)", + free / (1024 * 1024), + floor / (1024 * 1024) + ), + |e| { + e.with_detail("free_bytes", free) + .with_detail("threshold_bytes", floor) + }, + ); + } +} + +fn check_mempool( + state: &HealthState, + warnings: &NodeWarnings, + publisher: &EventPublisher, + thresholds: &AlertThresholds, + mempool: &Mempool, +) { + check_mempool_values( + state, + warnings, + publisher, + thresholds, + mempool.max_size_bytes() as u64, + mempool.acting_bytes() as u64, + mempool.min_fee_rate(), + ); +} + +/// The mempool detector's decision logic, over plain readings. +/// +/// Split from [`check_mempool`] so the thresholds/hysteresis behavior is +/// testable without filling a real mempool to a target occupancy — the tests +/// exercise this exact code rather than a reimplementation of it. +fn check_mempool_values( + state: &HealthState, + warnings: &NodeWarnings, + publisher: &EventPublisher, + thresholds: &AlertThresholds, + cap: u64, + used: u64, + min_fee: u64, +) { + let pct = thresholds.mempool_full_pct(); + if pct == 0 { + clear_because_disabled(state, warnings, publisher, StatusKind::MempoolCongested); + return; + } + if cap == 0 { + // `maxmempool=0` is accepted and is SIGHUP-live. A bare return would + // strand a raised alert with no path back: there is no occupancy ratio + // against a zero cap, so neither the raise nor the clear branch can + // ever run again. + // + // This is *not* `detector_disabled`: `alertmempoolfullpct` is still + // armed and the operator did not turn anything off — their mempool cap + // went to zero, which is itself worth surfacing. A receiver that + // suppresses `detector_disabled` (reasonably, since it means "you asked + // for this") would otherwise swallow it. + clear_with_reason( + state, + warnings, + publisher, + StatusKind::MempoolCongested, + "mempool cap is zero; congestion cannot be evaluated".to_string(), + "mempool_cap_zero", + ); + return; + } + let raise_at = cap.saturating_mul(pct) / 100; + let clear_at = raise_at.saturating_mul(hysteresis::MEMPOOL_CLEAR_RATIO_NUM) + / hysteresis::MEMPOOL_CLEAR_RATIO_DEN; + if used >= raise_at { + raise_if_new( + state, + warnings, + publisher, + StatusKind::MempoolCongested, + pct, + format!("mempool at {}% of its {} MiB cap", used * 100 / cap, cap / (1024 * 1024)), + |e| { + e.with_detail("bytes_used", used) + .with_detail("bytes_cap", cap) + .with_detail("threshold_pct", pct) + // The floor a transaction must beat to be accepted right + // now: the actionable half of a congestion alert. + .with_detail("mempoolminfee_sat_per_kvb", min_fee) + }, + ); + } else if used < clear_at { + clear_if_active( + state, + warnings, + publisher, + StatusKind::MempoolCongested, + format!("mempool back to {}% of its cap", used * 100 / cap), + |e| { + e.with_detail("bytes_used", used) + .with_detail("bytes_cap", cap) + }, + ); + } else { + clear_if_threshold_relaxed( + state, + warnings, + publisher, + StatusKind::MempoolCongested, + pct, + format!( + "mempool at {}% is within the threshold ({}%)", + used * 100 / cap, + pct + ), + |e| { + e.with_detail("bytes_used", used) + .with_detail("bytes_cap", cap) + .with_detail("threshold_pct", pct) + }, + ); + } +} + +#[allow(clippy::too_many_arguments)] +fn check_peers( + state: &HealthState, + warnings: &NodeWarnings, + publisher: &EventPublisher, + thresholds: &AlertThresholds, + peer_manager: &PeerManager, + below_since: &mut Option, + ok_since: &mut Option, + started: &Instant, +) { + check_peers_values( + state, + warnings, + publisher, + thresholds, + peer_manager.connection_count() as u64, + peer_manager.outbound_count() as u64, + below_since, + ok_since, + started, + Instant::now(), + ); +} + +/// The peer detector's decision logic, over plain readings and an injected +/// clock. +/// +/// Split from [`check_peers`] so the startup grace and hold behavior are +/// testable without a live `PeerManager` or real elapsed time — the tests +/// exercise this exact code rather than a reimplementation of it. +#[allow(clippy::too_many_arguments)] +fn check_peers_values( + state: &HealthState, + warnings: &NodeWarnings, + publisher: &EventPublisher, + thresholds: &AlertThresholds, + total: u64, + outbound: u64, + below_since: &mut Option, + ok_since: &mut Option, + started: &Instant, + now: Instant, +) { + let floor = thresholds.peer_floor(); + if floor == 0 { + *below_since = None; + *ok_since = None; + clear_because_disabled(state, warnings, publisher, StatusKind::PeerFloor); + return; + } + let inbound = total.saturating_sub(outbound); + + if total > 0 && !state.saw_first_peer.swap(true, Ordering::Relaxed) { + // First peer of this process. Normal operation starts here, so the hold + // clock starts here too — otherwise a node whose first peer arrives + // late in the startup grace ends the grace and finds a hold that has + // already elapsed, firing the alert in that very poll while it is still + // filling its remaining outbound slots. + *below_since = Some(now); + } + + if total < floor { + *ok_since = None; + let since = *below_since.get_or_insert(now); + // Startup grace. `tokio::time::interval` fires its first tick + // immediately, so without this the hold clock starts at t≈0 and a node + // that simply has not finished dialing out yet alerts at t≈PEER_HOLD. + // + // The grace defers the *start* of the hold rather than shortening it: a + // grace that merely suppressed the raise until t=GRACE would fire the + // instant it expired, since the hold would have run out alongside it. + // Once the node has seen a peer the grace is over for good — a node + // that loses its peers later is not in a startup transient and should + // alert after an ordinary hold. + let hold_from = if state.saw_first_peer.load(Ordering::Relaxed) { + since + } else { + since.max(*started + hysteresis::PEER_STARTUP_GRACE) + }; + if now.duration_since(hold_from) >= hysteresis::PEER_HOLD { + raise_if_new( + state, + warnings, + publisher, + StatusKind::PeerFloor, + floor, + format!("only {total} peers connected (floor {floor})"), + |e| { + e.with_detail("peers", total) + .with_detail("peers_outbound", outbound) + .with_detail("peers_inbound", inbound) + .with_detail("threshold", floor) + }, + ); + } + } else { + *below_since = None; + let since = *ok_since.get_or_insert(now); + if now.duration_since(since) >= hysteresis::PEER_HOLD { + clear_if_active( + state, + warnings, + publisher, + StatusKind::PeerFloor, + format!("{total} peers connected (floor {floor})"), + |e| { + e.with_detail("peers", total) + .with_detail("peers_outbound", outbound) + .with_detail("peers_inbound", inbound) + }, + ); + } + } +} + +/// How many reported reorgs to remember. The log's own ring holds +/// [`DEFAULT_RING_CAPACITY`](crate::chain::reorg_log::DEFAULT_RING_CAPACITY) +/// (256) records, and a poll can only ever show us those, so twice that is +/// comfortably more than can be re-presented. +const REORG_SEEN_CAPACITY: usize = 512; + +/// Which reorg-log records the detector has already reported. +/// +/// A **set of identities**, deliberately not a high-water mark over the clock. +/// +/// The earlier version compared `rec.ts_unix_secs` against a wall-clock value +/// seeded at startup and dropped anything older. Both sides ride +/// `SystemTime::now()`, so a single backwards step — NTP correcting a fast RTC +/// after boot, a hypervisor resyncing after live migration, an operator running +/// `date -s` — silenced *every* reorg alert until the clock caught back up to +/// the seeded value, with no event, no warning and no `-alertnotify`. The +/// watermark never reset and `deep_reorg` is an edge, so those alerts were not +/// delayed; they were gone. +/// +/// Identity is `(ts, old_tip, new_tip)`. Every component comes from the record +/// itself, so rescanning the same record is stable, and a clock that jumps in +/// either direction changes nothing about whether a reorg is recognized as one +/// we have already reported. Including the timestamp keeps a flapping chain +/// (A→B, B→A, A→B again) from collapsing its third reorg onto its first. +#[derive(Debug, Default)] +struct ReorgSeen { + seen: std::collections::HashSet, + order: std::collections::VecDeque, +} + +impl ReorgSeen { + fn key(rec: &ReorgRecord) -> String { + format!("{}:{}:{}", rec.ts_unix_secs, rec.old_tip, rec.new_tip) + } + + /// Mark `rec` as seen; returns whether it had not been seen before. + fn mark_new(&mut self, rec: &ReorgRecord) -> bool { + let key = Self::key(rec); + if !self.seen.insert(key.clone()) { + return false; + } + self.order.push_back(key); + while self.order.len() > REORG_SEEN_CAPACITY { + if let Some(evicted) = self.order.pop_front() { + self.seen.remove(&evicted); + } + } + true + } + + /// Mark everything already in the log as seen, without reporting it. + /// + /// Called once at task start so a restart does not re-page for reorgs the + /// previous process already alerted on. This replaces the old "seed a + /// timestamp from the current clock" trick and does not depend on a clock + /// at all. + fn seed(&mut self, records: &[ReorgRecord]) { + for rec in records { + self.mark_new(rec); + } + } +} + +/// Emit `deep_reorg` for every reorg the log has recorded since the last poll. +/// +/// Depth, fork height and the reconnected chain all come from the record, +/// which `perform_reorg` writes and fsyncs before pushing it to the ring that +/// `history()` reads. This is what the design specifies — "depth from +/// `ReorgRecord`" — and it is the only source that is right by construction. +/// +/// The previous implementation reconstructed depth by counting +/// `BlockDisconnected` events off the chain broadcast. That ring holds 64 +/// entries and a reorg of depth D emits `2D + 2` events in one await-free +/// burst, so the count was truncated — or the `Reorg` marker itself dropped, +/// losing the reorg entirely — at roughly the depth where this alert starts to +/// matter. It also had to infer the new tip from the first reconnect, which is +/// `fork_height + 1` rather than the tip. +fn scan_reorg_log( + warnings: &NodeWarnings, + publisher: &EventPublisher, + thresholds: &AlertThresholds, + chain_state: &ChainState, + seen: &mut ReorgSeen, +) { + let Some(log) = chain_state.reorg_log() else { + return; + }; + report_reorgs( + warnings, + publisher, + thresholds, + log.history(REORG_LOG_LOOKBACK_SECS), + seen, + ); +} + +/// The reporting half of [`scan_reorg_log`], split from the log lookup so it +/// can be driven from a hand-built record list without a whole `ChainState`. +fn report_reorgs( + warnings: &NodeWarnings, + publisher: &EventPublisher, + thresholds: &AlertThresholds, + records: Vec, + seen: &mut ReorgSeen, +) { + let threshold = thresholds.reorg_depth(); + for rec in records { + // Mark before the threshold test: a record the current threshold + // ignores is still one we have seen, and a later SIGHUP lowering the + // threshold must not resurrect it. + if !seen.mark_new(&rec) { + continue; + } + if threshold == 0 || u64::from(rec.depth) < threshold { + continue; + } + let from_height = rec.fork_height.saturating_add(rec.depth); + // `reconnected` is fork-parent-exclusive and includes the block whose + // connection triggered the reorg, so this is the new tip exactly. + let to_height = rec + .fork_height + .saturating_add(u32::try_from(rec.reconnected.len()).unwrap_or(u32::MAX)); + emit( + warnings, + publisher, + StatusEvent::edge( + StatusKind::DeepReorg, + format!( + "reorg rolled back {} blocks (from height {from_height} to \ + {to_height}; threshold {threshold})", + rec.depth + ), + ) + .with_detail("depth", rec.depth) + .with_detail("from_height", from_height) + .with_detail("to_height", to_height) + .with_detail("fork_height", rec.fork_height) + .with_detail("threshold", threshold), + ); + } +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::events::{EdgeIdentity, NodeEventBody}; + + fn publisher() -> Arc { + EventPublisher::new( + EdgeIdentity::new([0x11; 16], None).unwrap(), + 64, + ) + } + + /// Drain the published status events from a receiver, as `(kind, state)`. + fn drained( + rx: &mut broadcast::Receiver, + ) -> Vec<(StatusKind, StatusState)> { + let mut out = Vec::new(); + while let Ok(env) = rx.try_recv() { + if let NodeEventBody::Status(s) = env.body { + out.push((s.kind, s.state)); + } + } + out + } + + #[test] + fn thresholds_round_trip_and_clamp() { + let t = AlertThresholds::new(30, 2, 150, 4, 7); + assert_eq!(t.tip_stall_secs(), 30); + assert_eq!(t.disk_free_bytes(), 2 * 1024 * 1024); + // A percentage above 100 is unreachable and would silently disable the + // detector, so it clamps instead. + assert_eq!(t.mempool_full_pct(), 100); + assert_eq!(t.peer_floor(), 4); + assert_eq!(t.reorg_depth(), 7); + } + + #[test] + fn defaults_match_the_documented_values() { + let t = AlertThresholds::default(); + assert_eq!(t.tip_stall_secs(), 3_600); + assert_eq!(t.disk_free_bytes(), 10_240 * 1024 * 1024); + assert_eq!(t.mempool_full_pct(), 90); + assert_eq!(t.peer_floor(), 3); + assert_eq!(t.reorg_depth(), 3); + } + + /// Regtest blocks exist only when a test mines them, so an idle chain is + /// its resting state — not a stall worth a critical alert that pins + /// `getwarnings` and the TUI modal. + #[test] + fn tip_stall_default_is_disabled_only_on_regtest() { + use bitcoin::Network; + assert_eq!(defaults::tip_stall_for(Network::Regtest), 0); + // A stall is not an ordinary property of a thin-hashrate chain the way + // a shallow reorg is, so unlike `reorg_depth_for` the test networks are + // not relaxed — they are simply expected to make blocks. + assert_eq!(defaults::tip_stall_for(Network::Bitcoin), defaults::TIP_STALL_SECS); + assert_eq!(defaults::tip_stall_for(Network::Signet), defaults::TIP_STALL_SECS); + assert_eq!(defaults::tip_stall_for(Network::Testnet4), defaults::TIP_STALL_SECS); + } + + #[test] + fn peer_floor_default_is_disabled_only_on_regtest() { + use bitcoin::Network; + // A regtest node normally has no peers at all, so defaulting the floor + // to 3 raises a critical warning 90s into every run that can never + // clear. + assert_eq!(defaults::peer_floor_for(Network::Regtest, 0), 0); + // Signet is a public network with real peers. A peer-starved signet + // node is broken in exactly the way this alert reports, and defaulting + // it off would make the detector's silence indistinguishable from + // health. + assert_eq!(defaults::peer_floor_for(Network::Signet, 0), defaults::PEER_FLOOR); + assert_eq!(defaults::peer_floor_for(Network::Bitcoin, 0), defaults::PEER_FLOOR); + assert_eq!(defaults::peer_floor_for(Network::Testnet4, 0), defaults::PEER_FLOOR); + + // `-connect=` is the same trap as regtest: it suppresses DNS and fixed + // seeds, so the node can never hold more peers than were named and a + // stock floor of 3 would stand in `getblockchaininfo.warnings` forever. + assert_eq!(defaults::peer_floor_for(Network::Bitcoin, 1), 1); + assert_eq!(defaults::peer_floor_for(Network::Signet, 2), 2); + // Capped, not mirrored — past the stock floor the ordinary threshold + // governs, so a node wired to eight peers is not held to needing eight. + assert_eq!(defaults::peer_floor_for(Network::Bitcoin, 8), defaults::PEER_FLOOR); + // Regtest stays off; the network already answered this. + assert_eq!(defaults::peer_floor_for(Network::Regtest, 1), 0); + } + + /// Depth 3 is an incident on mainnet and an ordinary Tuesday on a test + /// chain. Firing `-alertnotify` for the latter trains operators to ignore + /// the alert, which costs them the mainnet one too. + #[test] + fn reorg_depth_default_is_network_conditional() { + use bitcoin::Network; + // Mainnet: a 3-block reorg costs real hashrate and invalidates + // transactions merchants have started treating as settled. + assert_eq!(defaults::reorg_depth_for(Network::Bitcoin), 3); + // Test networks: reorgs a few blocks deep are the network working as + // designed. Raised, not disabled — past 6 confirmations a wallet has + // been told something false, and that is worth reporting anywhere. + for n in [Network::Signet, Network::Testnet, Network::Testnet4] { + assert_eq!( + defaults::reorg_depth_for(n), + defaults::REORG_DEPTH_TEST_NETWORK, + "{n:?} should not page on routine reorgs", + ); + assert!( + defaults::reorg_depth_for(n) > 6, + "{n:?} default must sit above the confirmation convention", + ); + } + // Regtest reorgs on purpose, constantly, as the test suite's whole job. + assert_eq!(defaults::reorg_depth_for(Network::Regtest), 0); + } + + #[test] + fn raise_is_edge_triggered_not_repeated_per_poll() { + let state = HealthState::new(); + let warnings = NodeWarnings::new(); + let pubr = publisher(); + let mut rx = pubr.subscribe(); + + for _ in 0..5 { + raise_if_new( + &state, + &warnings, + &pubr, + StatusKind::DiskLow, + 1, + "low".into(), + |e| e, + ); + } + assert_eq!( + drained(&mut rx), + vec![(StatusKind::DiskLow, StatusState::Raised)], + "a standing condition raises once, not once per evaluation", + ); + assert!(state.is_active(StatusKind::DiskLow)); + // And the warning is recorded once (repeats would only bump `count`). + assert_eq!(warnings.count(), 1); + + for _ in 0..5 { + clear_if_active( + &state, + &warnings, + &pubr, + StatusKind::DiskLow, + "ok".into(), + |e| e, + ); + } + assert_eq!( + drained(&mut rx), + vec![(StatusKind::DiskLow, StatusState::Cleared)], + ); + assert!(!state.is_active(StatusKind::DiskLow)); + assert_eq!(warnings.count(), 0, "clearing removes the warning"); + } + + #[test] + fn info_severity_does_not_create_a_warning() { + // `ibd_complete` is good news; a warning for it would sit in + // `getwarnings` forever with nothing to clear it. + let warnings = NodeWarnings::new(); + let pubr = publisher(); + emit( + &warnings, + &pubr, + StatusEvent::edge(StatusKind::IbdComplete, "done"), + ); + assert_eq!(warnings.count(), 0); + } + + #[test] + fn critical_maps_to_error_severity_warning() { + let warnings = NodeWarnings::new(); + let pubr = publisher(); + emit( + &warnings, + &pubr, + StatusEvent::raised(StatusKind::TipStall, "stalled"), + ); + let w = warnings.list(); + assert_eq!(w.len(), 1); + assert_eq!(w[0].id, "alert.tip_stall"); + assert_eq!(w[0].severity, Severity::Error); + assert!(warnings.has_errors()); + } + + #[test] + fn warning_severity_maps_to_warn() { + let warnings = NodeWarnings::new(); + let pubr = publisher(); + emit( + &warnings, + &pubr, + StatusEvent::raised(StatusKind::PeerFloor, "few peers"), + ); + assert_eq!(warnings.list()[0].severity, Severity::Warn); + assert!(!warnings.has_errors()); + } + + #[tokio::test] + async fn disabling_a_detector_clears_a_standing_condition() { + // Otherwise turning the threshold off would strand a raised alert that + // nothing will ever retract. + let state = HealthState::new(); + let warnings = NodeWarnings::new(); + let pubr = publisher(); + let mut rx = pubr.subscribe(); + let thresholds = AlertThresholds::new(0, 0, 0, 0, 0); + + state.set_active(StatusKind::DiskLow, true); + warnings.record("alert.disk_low", Severity::Error, "low", serde_json::Value::Null); + + // Two polls: the probe is started on the first and collected on the + // second (see `check_disk` on why it is never awaited inline). + let mut probe = None; + for _ in 0..2 { + check_disk( + &state, + &warnings, + &pubr, + &thresholds, + std::path::Path::new("."), + &mut probe, + DISK_PROBE_BUDGET, + ) + .await; + } + assert_eq!( + drained(&mut rx), + vec![(StatusKind::DiskLow, StatusState::Cleared)], + ); + assert_eq!(warnings.count(), 0); + } + + #[test] + fn raising_the_disk_floor_clears_a_standing_alert() { + // The operator decides the current free space is acceptable after all + // and raises the floor to quiet the pager. The reading has not moved, + // so it lands inside the *new* hysteresis band — between the new raise + // line and the new clear line — where neither branch runs. Without the + // threshold-relaxation clear the alert stays raised against a floor it + // no longer violates, and on a filling disk free space only goes down, + // so it would never clear on its own. + let state = HealthState::new(); + let warnings = NodeWarnings::new(); + let pubr = publisher(); + let mut rx = pubr.subscribe(); + + // 10 MiB floor, 5 MiB free ⇒ raised. + let thresholds = AlertThresholds::new(0, 10, 0, 0, 0); + sample_disk(&state, &warnings, &pubr, &thresholds, 5 * 1024 * 1024); + assert_eq!(drained(&mut rx), vec![(StatusKind::DiskLow, StatusState::Raised)]); + + // Retune the floor down to 4 MiB. 5 MiB free is above the new floor but + // below the new 6 MiB clear line: the dead band. + thresholds.set_disk_free_mb(4); + sample_disk(&state, &warnings, &pubr, &thresholds, 5 * 1024 * 1024); + assert_eq!( + drained(&mut rx), + vec![(StatusKind::DiskLow, StatusState::Cleared)], + "a retuned threshold must be able to clear its own alert" + ); + assert!(!state.is_active(StatusKind::DiskLow)); + } + + #[test] + fn raising_the_mempool_threshold_clears_a_standing_alert() { + // The unescapable case. `alertmempoolfullpct` clamps at 100, and the + // clear line is 0.75× the raise line, so the highest reachable clear + // line is 75% of the cap. Once occupancy is at or above that, *no* + // value of the setting can clear a raised alert — the operator's only + // escape would be disabling the detector outright. + let state = HealthState::new(); + let warnings = NodeWarnings::new(); + let pubr = publisher(); + let mut rx = pubr.subscribe(); + const CAP: u64 = 300 * 1024 * 1024; + let used = CAP * 93 / 100; + + // 90% threshold against 93% occupancy ⇒ raised. + let thresholds = AlertThresholds::new(0, 0, 90, 0, 0); + check_mempool_values(&state, &warnings, &pubr, &thresholds, CAP, used, 1000); + assert_eq!( + drained(&mut rx), + vec![(StatusKind::MempoolCongested, StatusState::Raised)] + ); + + // Raise it to 95 to quiet the alert: 93 < 95 so no raise, and + // 93 >= 0.75*95 so no hysteresis clear. Dead band. + thresholds.set_mempool_full_pct(95); + check_mempool_values(&state, &warnings, &pubr, &thresholds, CAP, used, 1000); + assert_eq!( + drained(&mut rx), + vec![(StatusKind::MempoolCongested, StatusState::Cleared)], + "no value of alertmempoolfullpct could otherwise clear this" + ); + } + + #[test] + fn disk_hysteresis_gap_prevents_flapping() { + let state = HealthState::new(); + let warnings = NodeWarnings::new(); + let pubr = publisher(); + let mut rx = pubr.subscribe(); + // 1 MiB floor ⇒ clears at 1.5 MiB. + let thresholds = AlertThresholds::new(0, 1, 0, 0, 0); + let floor = thresholds.disk_free_bytes(); + + // Just below the floor raises. + sample_disk(&state, &warnings, &pubr, &thresholds, floor - 1); + assert_eq!(drained(&mut rx), vec![(StatusKind::DiskLow, StatusState::Raised)]); + // Just *above* the floor does NOT clear — that is the hysteresis gap. + sample_disk(&state, &warnings, &pubr, &thresholds, floor + 1); + assert!(drained(&mut rx).is_empty(), "clearing at the raise line would flap"); + assert!(state.is_active(StatusKind::DiskLow)); + // Above 1.5× clears. + sample_disk(&state, &warnings, &pubr, &thresholds, floor * 3 / 2 + 1); + assert_eq!(drained(&mut rx), vec![(StatusKind::DiskLow, StatusState::Cleared)]); + } + + /// Drive the real disk detector at a synthetic free-space reading. + /// + /// A thin adapter over [`check_disk_values`] — deliberately not a + /// reimplementation of its branch structure. An earlier version of these + /// tests mirrored the detector's if/else here, which meant deleting the + /// production `clear_if_threshold_relaxed` arm left every one of them green. + fn sample_disk( + state: &HealthState, + warnings: &NodeWarnings, + publisher: &EventPublisher, + thresholds: &AlertThresholds, + free: u64, + ) { + check_disk_values(state, warnings, publisher, thresholds, Some(free)); + } + + /// The retune-down case. `clear_if_threshold_relaxed` exists to release an + /// alert whose *threshold* moved, but it must not release one whose + /// threshold moved in the direction that keeps it firing. Storing the + /// remembered line only on the raise edge left it stale across exactly that + /// retune, so the next reading inside the hysteresis band read as "the + /// operator moved the line" and cleared an alert both the old and the new + /// hysteresis say to hold. + #[test] + fn lowering_the_mempool_threshold_while_raised_does_not_defeat_hysteresis() { + let state = HealthState::new(); + let warnings = NodeWarnings::new(); + let pubr = publisher(); + let mut rx = pubr.subscribe(); + let cap = 1_000_000u64; + let thresholds = AlertThresholds::new(0, 0, 90, 0, 0); + + // 93% against a 90% line raises. + check_mempool_values(&state, &warnings, &pubr, &thresholds, cap, 930_000, 1); + assert_eq!( + drained(&mut rx), + vec![(StatusKind::MempoolCongested, StatusState::Raised)] + ); + + // Operator retunes *down* to 80%, wanting an earlier warning. Still + // over the line, so it stays raised and emits nothing. + thresholds.set_mempool_full_pct(80); + check_mempool_values(&state, &warnings, &pubr, &thresholds, cap, 930_000, 1); + assert!(drained(&mut rx).is_empty(), "still above the new line"); + assert!(state.is_active(StatusKind::MempoolCongested)); + + // Ease to 78%: below the new raise line (80%), above the new clear line + // (60%) — squarely in the hysteresis band. It must hold. + check_mempool_values(&state, &warnings, &pubr, &thresholds, cap, 780_000, 1); + assert!( + drained(&mut rx).is_empty(), + "78% is inside the hysteresis band for an 80% threshold; clearing \ + here is the stale-slot bug" + ); + assert!(state.is_active(StatusKind::MempoolCongested)); + + // Below the clear line it does clear, on the ordinary path. + check_mempool_values(&state, &warnings, &pubr, &thresholds, cap, 550_000, 1); + assert_eq!( + drained(&mut rx), + vec![(StatusKind::MempoolCongested, StatusState::Cleared)] + ); + } + + /// The retune-*up* case still works — this is what the relaxed clear is for. + #[test] + fn raising_the_mempool_threshold_still_clears_from_inside_the_band() { + let state = HealthState::new(); + let warnings = NodeWarnings::new(); + let pubr = publisher(); + let mut rx = pubr.subscribe(); + let cap = 1_000_000u64; + let thresholds = AlertThresholds::new(0, 0, 90, 0, 0); + + check_mempool_values(&state, &warnings, &pubr, &thresholds, cap, 930_000, 1); + assert_eq!( + drained(&mut rx), + vec![(StatusKind::MempoolCongested, StatusState::Raised)] + ); + // 93% now sits below a 95% raise line but above the 71.25% clear line: + // the dead band. Only the relaxed clear can release it. + thresholds.set_mempool_full_pct(95); + check_mempool_values(&state, &warnings, &pubr, &thresholds, cap, 930_000, 1); + assert_eq!( + drained(&mut rx), + vec![(StatusKind::MempoolCongested, StatusState::Cleared)] + ); + } + + /// `maxmempool=0` is not the operator switching the detector off, and a + /// receiver routing on `reason` must be able to tell the two apart. + #[test] + fn a_zero_mempool_cap_is_not_reported_as_detector_disabled() { + let state = HealthState::new(); + let warnings = NodeWarnings::new(); + let pubr = publisher(); + let mut rx = pubr.subscribe(); + let thresholds = AlertThresholds::new(0, 0, 90, 0, 0); + + check_mempool_values(&state, &warnings, &pubr, &thresholds, 1_000_000, 950_000, 1); + assert_eq!( + drained(&mut rx), + vec![(StatusKind::MempoolCongested, StatusState::Raised)] + ); + + check_mempool_values(&state, &warnings, &pubr, &thresholds, 0, 0, 1); + let env = rx.try_recv().expect("a zero cap must clear the standing alert"); + let NodeEventBody::Status(s) = env.body else { + panic!("expected a status event") + }; + assert_eq!(s.state, StatusState::Cleared); + assert_eq!( + s.details.get("reason").map(String::as_str), + Some("mempool_cap_zero"), + "`detector_disabled` means the operator turned it off; a consumer \ + suppressing that would swallow a zeroed mempool cap" + ); + } + + /// A node restarted while already wedged must still page. + /// + /// This is the case an in-process latch cannot cover. The node is synced + /// but its chain stopped; >24h later the tip-age predicate reads "in IBD" + /// again. The operator restarts — the first thing anyone does — so the + /// process never observes a non-IBD tip and a latch would never arm. The + /// detector has to raise anyway. + #[test] + fn tip_stall_raises_on_a_node_restarted_while_already_wedged() { + let state = HealthState::new(); + let warnings = NodeWarnings::new(); + let pubr = publisher(); + let mut rx = pubr.subscribe(); + + let on = AlertThresholds::new(3600, 0, 0, 0, 0); + // in_ibd = true on the very first evaluation, and never false. + check_tip_stall_values(&state, &warnings, &pubr, &on, true, 100, 108_000); + assert_eq!( + drained(&mut rx), + vec![(StatusKind::TipStall, StatusState::Raised)], + "a wedged node reads as 'in IBD' by tip age; that must not silence it" + ); + } + + /// The flip side: a node that really is syncing connects blocks, which + /// keeps the age low and suppresses the alert without any IBD predicate. + #[test] + fn a_syncing_node_connecting_blocks_does_not_raise_tip_stall() { + let state = HealthState::new(); + let warnings = NodeWarnings::new(); + let pubr = publisher(); + let mut rx = pubr.subscribe(); + + let on = AlertThresholds::new(3600, 0, 0, 0, 0); + check_tip_stall_values(&state, &warnings, &pubr, &on, true, 100, 12); + assert!( + drained(&mut rx).is_empty(), + "blocks are arriving; there is no stall to report" + ); + } + + #[test] + fn peer_startup_grace_is_followed_by_a_full_hold() { + let state = HealthState::new(); + let warnings = NodeWarnings::new(); + let pubr = publisher(); + let mut rx = pubr.subscribe(); + let thresholds = AlertThresholds::new(0, 0, 0, 3, 0); + let started = Instant::now(); + let (mut below, mut ok) = (None, None); + + let at = |d: Duration| started + d; + let poll = |now: Instant, total: u64, below: &mut _, ok: &mut _| { + check_peers_values( + &state, &warnings, &pubr, &thresholds, total, total, below, ok, &started, now, + ); + }; + + // Inside the grace, no peers: silent. + poll(at(Duration::ZERO), 0, &mut below, &mut ok); + poll(at(hysteresis::PEER_STARTUP_GRACE / 2), 0, &mut below, &mut ok); + assert!(drained(&mut rx).is_empty(), "still dialing out"); + + // The instant the grace ends, the hold must start fresh — not already + // be satisfied by time spent inside the grace. + poll(at(hysteresis::PEER_STARTUP_GRACE + Duration::from_secs(1)), 0, &mut below, &mut ok); + assert!( + drained(&mut rx).is_empty(), + "the grace must not merely delay the raise to t=GRACE" + ); + + // A full hold after the grace, still starved: now it raises. + poll( + at(hysteresis::PEER_STARTUP_GRACE + hysteresis::PEER_HOLD + Duration::from_secs(2)), + 0, + &mut below, + &mut ok, + ); + assert_eq!(drained(&mut rx), vec![(StatusKind::PeerFloor, StatusState::Raised)]); + } + + /// Latching "we have seen a peer" ends the grace, but must not fire the + /// alert in the very poll the first peer arrives while the node is still + /// filling its remaining outbound slots. + #[test] + fn the_first_peer_arriving_does_not_immediately_raise() { + let state = HealthState::new(); + let warnings = NodeWarnings::new(); + let pubr = publisher(); + let mut rx = pubr.subscribe(); + let thresholds = AlertThresholds::new(0, 0, 0, 3, 0); + let started = Instant::now(); + let (mut below, mut ok) = (None, None); + + check_peers_values( + &state, &warnings, &pubr, &thresholds, 0, 0, &mut below, &mut ok, &started, + started, + ); + // First peer lands late in the grace: 1 < floor of 3, and the latch + // drops the grace — but the hold has to start here, not at t=0. + let t = started + hysteresis::PEER_STARTUP_GRACE - Duration::from_secs(1); + check_peers_values( + &state, &warnings, &pubr, &thresholds, 1, 1, &mut below, &mut ok, &started, t, + ); + assert!( + drained(&mut rx).is_empty(), + "the node just got its first peer; it is still filling outbound slots" + ); + } + + /// An operator who sets `alertdiskfreemb=0` has usually done so because + /// they alert on the Prometheus gauge instead. Disabling the alert must not + /// delete the series out from under their own rule. + #[tokio::test] + async fn the_disk_gauge_is_sampled_even_when_the_alert_is_disabled() { + let state = HealthState::new(); + let warnings = NodeWarnings::new(); + let pubr = publisher(); + let thresholds = AlertThresholds::new(0, 0, 0, 0, 0); + assert_eq!(thresholds.disk_free_bytes(), 0, "detector off"); + + let mut probe = None; + let call = async |probe: &mut Option| { + check_disk( + &state, &warnings, &pubr, &thresholds, std::path::Path::new("."), probe, + DISK_PROBE_BUDGET, + ) + .await; + }; + // First poll starts the probe and publishes nothing: "not measured yet" + // must not reach the gauge as "unmeasurable". + call(&mut probe).await; + assert!( + state.disk_free_bytes().is_none(), + "the first poll only starts the probe", + ); + assert!(probe.is_some(), "and leaves it outstanding"); + // Second poll collects it. + call(&mut probe).await; + assert!( + state.disk_free_bytes().is_some(), + "the gauge must be populated whatever the alert's configuration" + ); + } + + /// The finding: `statvfs` on a hard NFS mount whose server has gone away + /// never returns. Awaiting it inline parked the whole detector loop — tip + /// stall, deep reorg, mempool and peer checks all stopped, and every gauge + /// froze, so even an external Prometheus rule on tip age could not fire. + /// + /// A probe that never finishes must therefore (a) not block the poll, and + /// (b) not spawn a replacement each poll, which would strand one blocking + /// thread per tick and exhaust the pool within hours. + #[tokio::test] + async fn a_wedged_filesystem_does_not_stall_the_detector_or_leak_threads() { + let state = HealthState::new(); + let warnings = NodeWarnings::new(); + let pubr = publisher(); + let thresholds = AlertThresholds::new(0, 0, 0, 0, 0); + + // Stand in for the hung syscall: a blocking task that never returns. + let (release_tx, release_rx) = std::sync::mpsc::channel::<()>(); + let mut probe = Some(DiskProbe { + handle: tokio::task::spawn_blocking(move || { + let _ = release_rx.recv(); + None + }), + stalled_polls: 0, + }); + + for expected in 1..=3u32 { + // Each call must RETURN — if this hangs, the bug is back. + tokio::time::timeout( + std::time::Duration::from_secs(5), + check_disk( + &state, + &warnings, + &pubr, + &thresholds, + std::path::Path::new("."), + &mut probe, + // Tiny budget: the point is that a stuck probe is abandoned + // for this tick, not how long we are willing to wait. + std::time::Duration::from_millis(1), + ), + ) + .await + .expect("check_disk must not block on an unresponsive filesystem"); + assert_eq!( + probe.as_ref().expect("the stuck probe is retained").stalled_polls, + expected, + "the same probe is carried forward, not replaced", + ); + } + // Let the fake syscall finish so the runtime can shut down cleanly. + let _ = release_tx.send(()); + } + + /// Build a reorg-log record with a controlled timestamp and shape. + /// + /// `reconnected` is a *count*: the number of blocks the new chain put back + /// above the fork, which is what fixes the new-tip height. + fn rec(ts: u64, fork_height: u32, depth: u32, reconnected: usize, tag: u8) -> ReorgRecord { + ReorgRecord { + ts_unix_secs: ts, + depth, + fork_height, + old_tip: format!("{tag:064x}"), + new_tip: format!("{:064x}", tag.wrapping_add(0x80)), + disconnected: (0..depth).map(|i| format!("d{i}")).collect(), + reconnected: (0..reconnected).map(|i| format!("r{i}")).collect(), + } + } + + fn detail(env: &crate::events::NodeEvent, key: &str) -> String { + let NodeEventBody::Status(s) = &env.body else { + panic!("expected a status event") + }; + s.details.get(key).cloned().unwrap_or_default() + } + + #[test] + fn deep_reorg_fires_only_at_or_above_the_threshold() { + let warnings = NodeWarnings::new(); + let pubr = publisher(); + let mut rx = pubr.subscribe(); + let thresholds = AlertThresholds::new(0, 0, 0, 0, 3); + let mut seen = ReorgSeen::default(); + + report_reorgs(&warnings, &pubr, &thresholds, vec![rec(1000, 98, 2, 3, 1)], &mut seen); + assert!(drained(&mut rx).is_empty(), "a 2-deep reorg is below the floor"); + + report_reorgs(&warnings, &pubr, &thresholds, vec![rec(1001, 97, 3, 4, 2)], &mut seen); + assert_eq!(drained(&mut rx), vec![(StatusKind::DeepReorg, StatusState::Edge)]); + } + + #[test] + fn deep_reorg_reports_true_depth_and_fork_height() { + let warnings = NodeWarnings::new(); + let pubr = publisher(); + let mut rx = pubr.subscribe(); + let thresholds = AlertThresholds::new(0, 0, 0, 0, 1); + let mut seen = ReorgSeen::default(); + + // fork at 896, 4 rolled back (old tip 900), 6 reconnected (new tip 902). + report_reorgs(&warnings, &pubr, &thresholds, vec![rec(1000, 896, 4, 6, 1)], &mut seen); + let env = rx.try_recv().unwrap(); + assert_eq!(detail(&env, "depth"), "4"); + assert_eq!(detail(&env, "from_height"), "900"); + assert_eq!(detail(&env, "to_height"), "902"); + assert_eq!(detail(&env, "fork_height"), "896"); + } + + /// The new tip is the end of the reconnected chain, not its first block. + /// + /// Reconnects are emitted oldest-first, so an implementation that reads the + /// new tip off the first `BlockConnected` after a disconnect run reports + /// `fork_height + 1` — below the *old* tip — for every reorg with a + /// replacement chain, which is every reorg deep enough to alert on. + #[test] + fn to_height_is_the_new_tip_not_the_first_reconnect() { + let warnings = NodeWarnings::new(); + let pubr = publisher(); + let mut rx = pubr.subscribe(); + let thresholds = AlertThresholds::new(0, 0, 0, 0, 1); + let mut seen = ReorgSeen::default(); + + report_reorgs(&warnings, &pubr, &thresholds, vec![rec(1000, 100, 3, 4, 1)], &mut seen); + let env = rx.try_recv().unwrap(); + assert_eq!(detail(&env, "to_height"), "104"); + assert_ne!(detail(&env, "to_height"), "101", "that is the first reconnect"); + } + + /// A reorg whose record predates the last one seen must still be reported. + /// + /// `ReorgRecord::ts_unix_secs` is `SystemTime::now()`, so a backwards clock + /// step — NTP correcting a fast RTC after boot, a hypervisor resync after + /// live migration, `date -s` — makes later reorgs carry *earlier* + /// timestamps. The previous implementation kept a high-water mark over that + /// timestamp and dropped anything below it, which silenced every reorg + /// alert until the clock caught back up. `deep_reorg` is an edge, so those + /// alerts were not delayed, they were gone. + /// + /// Control: with the old `if rec.ts_unix_secs < self.ts { return false }` + /// watermark, the second `report_reorgs` here emits nothing and the final + /// assertion fails. + #[test] + fn a_reorg_is_reported_even_when_the_clock_steps_backwards() { + let warnings = NodeWarnings::new(); + let pubr = publisher(); + let mut rx = pubr.subscribe(); + let thresholds = AlertThresholds::new(0, 0, 0, 0, 1); + let mut seen = ReorgSeen::default(); + + // A reorg at t=2000, reported normally. + report_reorgs(&warnings, &pubr, &thresholds, vec![rec(2000, 100, 4, 2, 1)], &mut seen); + assert!(rx.try_recv().is_ok(), "the first reorg reports"); + + // The clock steps back 20 minutes; the next genuine reorg is stamped + // t=800. It is a different reorg (different tips) and must still page. + report_reorgs(&warnings, &pubr, &thresholds, vec![rec(800, 200, 5, 3, 2)], &mut seen); + let env = rx + .try_recv() + .expect("a reorg after a backwards clock step must still be reported"); + assert_eq!(detail(&env, "depth"), "5"); + } + + /// The same record rescanned across polls reports exactly once — the + /// property that lets the lookback window be the whole ring. + #[test] + fn rescanning_the_log_does_not_re_report_a_reorg() { + let warnings = NodeWarnings::new(); + let pubr = publisher(); + let mut rx = pubr.subscribe(); + let thresholds = AlertThresholds::new(0, 0, 0, 0, 1); + let mut seen = ReorgSeen::default(); + + let records = vec![rec(1000, 100, 4, 2, 1), rec(1001, 200, 5, 3, 2)]; + for _ in 0..5 { + report_reorgs(&warnings, &pubr, &thresholds, records.clone(), &mut seen); + } + let mut count = 0; + while rx.try_recv().is_ok() { + count += 1; + } + assert_eq!(count, 2, "two distinct reorgs, five scans, two reports"); + } + + /// Seeding marks what the log already holds as reported, so a restart does + /// not re-page for reorgs the previous process already alerted on. + #[test] + fn seeding_suppresses_reorgs_that_predate_the_process() { + let warnings = NodeWarnings::new(); + let pubr = publisher(); + let mut rx = pubr.subscribe(); + let thresholds = AlertThresholds::new(0, 0, 0, 0, 1); + let mut seen = ReorgSeen::default(); + + let old = vec![rec(1000, 100, 9, 2, 1)]; + seen.seed(&old); + report_reorgs(&warnings, &pubr, &thresholds, old, &mut seen); + assert!(rx.try_recv().is_err(), "a seeded reorg must not re-page"); + + report_reorgs(&warnings, &pubr, &thresholds, vec![rec(1001, 300, 9, 2, 3)], &mut seen); + assert!(rx.try_recv().is_ok(), "but a new one still does"); + } + + /// A truncation reorg — rolled back with nothing to replace it — leaves the + /// tip at the fork. + #[test] + fn a_truncation_reorg_reports_the_fork_as_the_new_tip() { + let warnings = NodeWarnings::new(); + let pubr = publisher(); + let mut rx = pubr.subscribe(); + let thresholds = AlertThresholds::new(0, 0, 0, 0, 1); + let mut seen = ReorgSeen::default(); + + report_reorgs(&warnings, &pubr, &thresholds, vec![rec(1000, 99, 6, 0, 1)], &mut seen); + let env = rx.try_recv().unwrap(); + assert_eq!(detail(&env, "depth"), "6"); + assert_eq!(detail(&env, "from_height"), "105"); + assert_eq!(detail(&env, "to_height"), "99"); + } + + /// The log is rescanned on every poll and keeps records for 300 s, so + /// without a watermark one reorg would re-alert every 15 s for 5 minutes. + #[test] + fn a_reorg_is_reported_once_however_often_the_log_is_scanned() { + let warnings = NodeWarnings::new(); + let pubr = publisher(); + let mut rx = pubr.subscribe(); + let thresholds = AlertThresholds::new(0, 0, 0, 0, 1); + let mut seen = ReorgSeen::default(); + let records = vec![rec(1000, 96, 4, 5, 1)]; + + report_reorgs(&warnings, &pubr, &thresholds, records.clone(), &mut seen); + assert_eq!(drained(&mut rx).len(), 1); + for _ in 0..5 { + report_reorgs(&warnings, &pubr, &thresholds, records.clone(), &mut seen); + } + assert!(drained(&mut rx).is_empty(), "a rescan must not re-alert"); + } + + /// Two reorgs can land in the same second — a tip race, or back-to-back + /// `invalidateblock`. A watermark that stored only a timestamp would + /// swallow the second. + #[test] + fn two_reorgs_in_the_same_second_are_both_reported() { + let warnings = NodeWarnings::new(); + let pubr = publisher(); + let mut rx = pubr.subscribe(); + let thresholds = AlertThresholds::new(0, 0, 0, 0, 1); + let mut seen = ReorgSeen::default(); + + report_reorgs( + &warnings, + &pubr, + &thresholds, + vec![rec(1000, 96, 4, 5, 1), rec(1000, 90, 7, 8, 2)], + &mut seen, + ); + assert_eq!(drained(&mut rx).len(), 2); + } + + /// A record the threshold ignored is still a record we have seen. Lowering + /// `alertreorgdepth` by SIGHUP must not resurrect reorgs from the window. + #[test] + fn a_sub_threshold_reorg_is_not_resurrected_by_lowering_the_threshold() { + let warnings = NodeWarnings::new(); + let pubr = publisher(); + let mut rx = pubr.subscribe(); + let mut seen = ReorgSeen::default(); + let records = vec![rec(1000, 98, 2, 3, 1)]; + + report_reorgs(&warnings, &pubr, &AlertThresholds::new(0, 0, 0, 0, 10), records.clone(), &mut seen); + assert!(drained(&mut rx).is_empty()); + + report_reorgs(&warnings, &pubr, &AlertThresholds::new(0, 0, 0, 0, 1), records, &mut seen); + assert!(drained(&mut rx).is_empty(), "already seen at the old threshold"); + } + + #[test] + fn deep_reorg_disabled_by_zero_threshold() { + let warnings = NodeWarnings::new(); + let pubr = publisher(); + let mut rx = pubr.subscribe(); + let thresholds = AlertThresholds::new(0, 0, 0, 0, 0); + let mut seen = ReorgSeen::default(); + report_reorgs(&warnings, &pubr, &thresholds, vec![rec(1000, 50, 50, 51, 1)], &mut seen); + assert!(drained(&mut rx).is_empty()); + } + + #[test] + fn health_state_tracks_per_kind_flags_independently() { + let state = HealthState::new(); + state.set_active(StatusKind::DiskLow, true); + assert!(state.is_active(StatusKind::DiskLow)); + for k in StatusKind::ALL { + if k != StatusKind::DiskLow { + assert!(!state.is_active(k), "{k:?} must be independent"); + } + } + } + + #[test] + fn disk_free_is_unknown_until_sampled() { + // A zero here would render as "completely full" on the metrics page. + let state = HealthState::new(); + assert_eq!(state.disk_free_bytes(), None); + state.disk_free_bytes.store(42, Ordering::Relaxed); + assert_eq!(state.disk_free_bytes(), Some(42)); + } +} diff --git a/node/src/index/address/runner.rs b/node/src/index/address/runner.rs index 251c18ecc..f0cf3c909 100644 --- a/node/src/index/address/runner.rs +++ b/node/src/index/address/runner.rs @@ -686,7 +686,7 @@ pub fn preflight_disk(chain: &ChainState) -> Result<(), BackfillError> { return Ok(()); } let datadir = chain.blocks_dir(); - let have = match free_disk_bytes(datadir) { + let have = match crate::diskspace::free_disk_bytes(datadir) { Some(b) => b, None => { tracing::warn!( @@ -729,23 +729,3 @@ fn debug_delay_ms() -> u64 { .unwrap_or(0) } -#[cfg(target_os = "linux")] -fn free_disk_bytes(path: &std::path::Path) -> Option { - use std::ffi::CString; - use std::os::unix::ffi::OsStrExt; - let cpath = CString::new(path.as_os_str().as_bytes()).ok()?; - // SAFETY: we zero-init `s` and pass a valid C string; libc::statvfs - // is the canonical free-space syscall on Linux. - unsafe { - let mut s: libc::statvfs = std::mem::zeroed(); - if libc::statvfs(cpath.as_ptr(), &mut s) != 0 { - return None; - } - Some(s.f_bavail.saturating_mul(s.f_frsize)) - } -} - -#[cfg(not(target_os = "linux"))] -fn free_disk_bytes(_path: &std::path::Path) -> Option { - None -} diff --git a/node/src/index/filter/runner.rs b/node/src/index/filter/runner.rs index a54dd686e..f1e998df2 100644 --- a/node/src/index/filter/runner.rs +++ b/node/src/index/filter/runner.rs @@ -600,7 +600,7 @@ impl BackfillRunner { /// platforms where free space can't be queried. pub fn preflight_disk(chain: &ChainState) -> Result<(), BackfillError> { let datadir = chain.blocks_dir(); - let have = match free_disk_bytes(datadir) { + let have = match crate::diskspace::free_disk_bytes(datadir) { Some(b) => b, None => { tracing::warn!( @@ -626,22 +626,3 @@ fn debug_delay_ms() -> u64 { .unwrap_or(0) } -#[cfg(target_os = "linux")] -fn free_disk_bytes(path: &std::path::Path) -> Option { - use std::ffi::CString; - use std::os::unix::ffi::OsStrExt; - let cpath = CString::new(path.as_os_str().as_bytes()).ok()?; - // SAFETY: zero-init s; libc::statvfs is the canonical free-space syscall. - unsafe { - let mut s: libc::statvfs = std::mem::zeroed(); - if libc::statvfs(cpath.as_ptr(), &mut s) != 0 { - return None; - } - Some(s.f_bavail.saturating_mul(s.f_frsize)) - } -} - -#[cfg(not(target_os = "linux"))] -fn free_disk_bytes(_path: &std::path::Path) -> Option { - None -} diff --git a/node/src/index/silent_payments/runner.rs b/node/src/index/silent_payments/runner.rs index 30ccdd51e..89f6b9a7d 100644 --- a/node/src/index/silent_payments/runner.rs +++ b/node/src/index/silent_payments/runner.rs @@ -476,7 +476,7 @@ impl BackfillRunner { pub fn preflight_disk(chain: &ChainState) -> Result<(), BackfillError> { use crate::index::silent_payments::backfill::PREFLIGHT_REQUIRED_FREE_BYTES; let datadir = chain.blocks_dir(); - let have = match free_disk_bytes(datadir) { + let have = match crate::diskspace::free_disk_bytes(datadir) { Some(b) => b, None => { tracing::warn!( @@ -502,22 +502,3 @@ fn debug_delay_ms() -> u64 { .unwrap_or(0) } -#[cfg(target_os = "linux")] -fn free_disk_bytes(path: &std::path::Path) -> Option { - use std::ffi::CString; - use std::os::unix::ffi::OsStrExt; - let cpath = CString::new(path.as_os_str().as_bytes()).ok()?; - // SAFETY: zero-init s; libc::statvfs is the canonical free-space syscall. - unsafe { - let mut s: libc::statvfs = std::mem::zeroed(); - if libc::statvfs(cpath.as_ptr(), &mut s) != 0 { - return None; - } - Some(s.f_bavail.saturating_mul(s.f_frsize)) - } -} - -#[cfg(not(target_os = "linux"))] -fn free_disk_bytes(_path: &std::path::Path) -> Option { - None -} diff --git a/node/src/lib.rs b/node/src/lib.rs index 642739c92..4bceb8e3c 100644 --- a/node/src/lib.rs +++ b/node/src/lib.rs @@ -1,6 +1,8 @@ pub mod adaptive_cache; pub mod chain; +pub mod diskspace; pub mod events; +pub mod health; pub mod ibd_eta; pub mod index; pub mod memstat; diff --git a/node/src/metrics.rs b/node/src/metrics.rs index ed12b2105..83e6b546b 100644 --- a/node/src/metrics.rs +++ b/node/src/metrics.rs @@ -48,6 +48,12 @@ pub struct MetricsContext { /// so operators can confirm at a glance which DB-backed indexes /// are live. pub addr_enabled: bool, + /// Live readings from the health detectors. `None` in test backends and + /// anywhere the detector task was not spawned, in which case the health + /// gauges are omitted entirely rather than reported as zeros (a zero + /// `satd_disk_free_bytes` is exactly the alarm an operator must not be + /// shown falsely). + pub health: Option>, } impl MetricsContext { @@ -328,6 +334,9 @@ impl MetricsContext { let _ = writeln!(out, "# TYPE satd_spindex_backfill_progress_ratio gauge"); let _ = writeln!(out, "satd_spindex_backfill_progress_ratio {sp_backfill_ratio}"); + // Node-health gauges, rendered only when the detector task is running. + render_health_metrics(&mut out, self.health.as_deref()); + // Transaction-filtering policy metrics (design §10, PR 7c). Extracted to // a free function so the I8-invisibility invariant (a node with no // non-empty ruleset renders a byte-identical page) is unit-testable @@ -361,8 +370,26 @@ fn metric( labels: &[(&str, &str)], value: u64, ) { + metric_header(out, name, help, kind); + metric_sample(out, name, labels, value); +} + +/// Emit the `# HELP` / `# TYPE` pair for a metric family, once. +/// +/// The text format permits **one** such pair per family name. [`metric`] writes +/// a header with every sample, which is correct only for a single-series +/// family; calling it in a loop emits repeated headers, and a strict parser +/// (`promtool check metrics`, and the `expfmt` text parser several collectors +/// and relays use) hard-errors on the second `# HELP` and discards the *entire +/// page* — taking every unrelated satd metric down with it. For a multi-series +/// family call this once, then [`metric_sample`] per series. +fn metric_header(out: &mut String, name: &str, help: &str, kind: &str) { let _ = writeln!(out, "# HELP {name} {help}"); let _ = writeln!(out, "# TYPE {name} {kind}"); +} + +/// Emit one sample of an already-headered metric family. +fn metric_sample(out: &mut String, name: &str, labels: &[(&str, &str)], value: u64) { if labels.is_empty() { let _ = writeln!(out, "{name} {value}"); } else { @@ -377,6 +404,60 @@ fn metric( } } +/// Append the node-health gauges (§A3): tip age, free disk, and one 0/1 series +/// per standing alert condition. +/// +/// `satd_alert_active` is pre-registered for *every* kind, including ones that +/// are not currently raised, so an alerting rule can be written against a series +/// that exists from the first scrape — a gauge that only appears once the +/// condition fires is exactly the gauge you cannot alert on. Edge kinds +/// (`ibd_complete`, `deep_reorg`) have no standing state and are omitted: a +/// permanently-zero series would invite a rule that can never fire. +/// +/// Renders nothing at all when no detector task is running (`None`), rather +/// than emitting zeros that would read as real readings. +fn render_health_metrics(out: &mut String, health: Option<&crate::health::HealthState>) { + let Some(health) = health else { + return; + }; + metric( + out, + "satd_tip_last_connect_age_seconds", + "Seconds since the last block was connected to the active chain (since process start if none has been).", + "gauge", + &[], + health.last_connect_age_secs(), + ); + // Omitted rather than zeroed when the filesystem cannot be interrogated. + if let Some(free) = health.disk_free_bytes() { + metric( + out, + "satd_disk_free_bytes", + "Free space available to satd on the watched data/blocks directory.", + "gauge", + &[], + free, + ); + } + metric_header( + out, + "satd_alert_active", + "1 while a node-health condition is raised, 0 while it is clear.", + "gauge", + ); + for kind in crate::events::StatusKind::ALL { + if kind.is_edge() { + continue; + } + metric_sample( + out, + "satd_alert_active", + &[("kind", kind.as_str())], + u64::from(health.is_active(kind)), + ); + } +} + /// Append the transaction-filtering policy metrics to `out` — but ONLY when a /// non-empty ruleset is active. A node with no policy (or one whose policyfile /// is just `version 1`) appends nothing, so its `/metrics` page is byte-identical @@ -715,6 +796,101 @@ mod tests { assert_eq!(scope_label(false, false), "none"); } + #[test] + fn health_metrics_absent_without_a_detector_task() { + // Rendering zeros here would show a `satd_disk_free_bytes 0` — the + // exact alarm an operator must not be given falsely. + let mut out = String::new(); + render_health_metrics(&mut out, None); + assert!(out.is_empty(), "no detector ⇒ no health metrics:\n{out}"); + } + + /// Assert a rendered page is valid Prometheus text format on the one rule + /// that is easy to break and fatal when broken: at most one `# HELP` and + /// one `# TYPE` per family name. + /// + /// Strict parsers (`promtool check metrics`, `expfmt.TextParser`) reject + /// the whole page on a duplicate, so one careless family takes every other + /// satd metric down with it. Prometheus's own scrape parser is lenient, + /// which is exactly why this is worth a test rather than a scrape check. + fn assert_one_header_per_family(out: &str) { + use std::collections::HashMap; + let mut helps: HashMap<&str, usize> = HashMap::new(); + let mut types: HashMap<&str, usize> = HashMap::new(); + for line in out.lines() { + if let Some(rest) = line.strip_prefix("# HELP ") { + *helps.entry(rest.split(' ').next().unwrap_or("")).or_default() += 1; + } else if let Some(rest) = line.strip_prefix("# TYPE ") { + *types.entry(rest.split(' ').next().unwrap_or("")).or_default() += 1; + } + } + for (name, n) in helps { + assert_eq!(n, 1, "family {name} has {n} `# HELP` lines:\n{out}"); + } + for (name, n) in types { + assert_eq!(n, 1, "family {name} has {n} `# TYPE` lines:\n{out}"); + } + } + + #[test] + fn health_metrics_are_valid_exposition_format() { + // `satd_alert_active` is one family with a series per kind, so the + // header must be emitted once and the samples must follow it. + let health = crate::health::HealthState::new(); + health.set_disk_free_for_test(Some(1 << 30)); + let mut out = String::new(); + render_health_metrics(&mut out, Some(&health)); + assert_one_header_per_family(&out); + } + + #[test] + fn health_metrics_preregister_every_standing_kind() { + use crate::events::StatusKind; + let health = crate::health::HealthState::new(); + let mut out = String::new(); + render_health_metrics(&mut out, Some(&health)); + + // The tip-age gauge always renders. + assert!(out.contains("satd_tip_last_connect_age_seconds 0"), "{out}"); + // Disk is omitted until sampled, rather than reported as zero free. + assert!( + !out.contains("satd_disk_free_bytes"), + "an unsampled disk reading must be omitted, not zeroed:\n{out}" + ); + // Every standing condition has a series from the first scrape, so an + // alerting rule can reference one before it ever fires. + for kind in StatusKind::ALL { + let want = format!("satd_alert_active{{kind=\"{}\"}} 0", kind.as_str()); + if kind.is_edge() { + assert!( + !out.contains(&format!("kind=\"{}\"", kind.as_str())), + "edge kind {kind:?} has no standing state to report:\n{out}" + ); + } else { + assert!(out.contains(&want), "missing series {want}:\n{out}"); + } + } + } + + #[test] + fn health_metrics_reflect_a_raised_condition() { + use crate::events::StatusKind; + let health = crate::health::HealthState::new(); + let mut out = String::new(); + render_health_metrics(&mut out, Some(&health)); + assert!(out.contains("satd_alert_active{kind=\"disk_low\"} 0")); + + // Simulate the detector raising `disk_low` with a real reading. + health.set_active_for_test(StatusKind::DiskLow, true); + health.set_disk_free_for_test(Some(4096)); + let mut out = String::new(); + render_health_metrics(&mut out, Some(&health)); + assert!(out.contains("satd_alert_active{kind=\"disk_low\"} 1"), "{out}"); + assert!(out.contains("satd_disk_free_bytes 4096"), "{out}"); + // Unrelated conditions stay at 0. + assert!(out.contains("satd_alert_active{kind=\"tip_stall\"} 0"), "{out}"); + } + #[test] fn policy_metrics_invisible_until_nonempty_ruleset() { use crate::mempool::pool::Mempool; diff --git a/node/src/warnings.rs b/node/src/warnings.rs index b66a9f67f..b62cf3c55 100644 --- a/node/src/warnings.rs +++ b/node/src/warnings.rs @@ -94,12 +94,55 @@ impl NodeWarnings { /// Record a warning. If `id` is already active, increment count /// and refresh `last_seen`, `severity`, `message`, `context`. + /// + /// `-alertnotify` fires only the first time an id becomes active. For a + /// standing condition that is what you want; for a one-shot event that will + /// never be cleared use [`notify_event`](Self::notify_event). pub fn record( &self, id: &str, severity: Severity, message: impl Into, context: serde_json::Value, + ) { + self.record_inner(id, severity, message, context); + } + + /// Fire `-alertnotify` for a **one-shot event**, without recording a + /// standing warning. + /// + /// This is the right call for something that *happened* rather than + /// something that *is*: a deep reorg, for instance. Such an event has no + /// resolved state, so nothing would ever call [`clear`](Self::clear) for + /// it — and an entry that never clears is exactly what this registry must + /// not accumulate. It would pin `getwarnings`, keep + /// [`has_errors`](Self::has_errors) true for the life of the process, and + /// hold the TUI's blocking modal open forever. On signet and testnet4, + /// where reorgs several blocks deep are routine, the first one would do + /// that permanently. + /// + /// Every occurrence fires the hook, since each is a distinct event rather + /// than a restatement of a standing condition. The durable record of what + /// happened is the subsystem's own log (`ReorgLog` for reorgs) and the + /// `status` event on the streaming API; this is only the shell hook. + pub fn notify_event(&self, id: &str, severity: Severity, message: impl Into) { + let guard = self.alert_tx.lock(); + if let Some(tx) = guard.as_ref() { + let _ = tx.send(format!( + "[{}] {}: {}", + severity.as_str(), + id, + message.into() + )); + } + } + + fn record_inner( + &self, + id: &str, + severity: Severity, + message: impl Into, + context: serde_json::Value, ) { let now = unix_secs(); let message: String = message.into(); @@ -295,6 +338,45 @@ mod tests { assert!(msg2.contains("peer.stall")); } + /// A one-shot event pages, but must not become a standing warning. + /// + /// Nothing ever clears an event that has no resolved state, so recording + /// one would pin `getwarnings`, keep `has_errors()` true for the life of + /// the process, and hold the TUI's blocking modal open — permanently, from + /// the first deep reorg, on chains where those are routine. + #[test] + fn notify_event_fires_the_hook_without_recording_a_warning() { + let w = NodeWarnings::new(); + let (tx, mut rx) = tokio::sync::mpsc::unbounded_channel::(); + w.set_alert_notifier(tx); + + w.notify_event("alert.deep_reorg", Severity::Error, "reorg rolled back 4 blocks"); + + let msg = rx.try_recv().expect("a one-shot event must still page"); + assert!(msg.contains("alert.deep_reorg"), "{msg}"); + assert_eq!(w.count(), 0, "it must not become a standing warning"); + assert!(!w.has_errors(), "and must not wedge has_errors()"); + assert!(w.as_strings().is_empty(), "nor appear in getwarnings"); + } + + /// Every occurrence pages — unlike `record`, which dedupes by id. Each + /// reorg is a distinct event, not a restatement of one condition. + #[test] + fn every_occurrence_of_an_event_pages() { + let w = NodeWarnings::new(); + let (tx, mut rx) = tokio::sync::mpsc::unbounded_channel::(); + w.set_alert_notifier(tx); + + for i in 0..3 { + w.notify_event("alert.deep_reorg", Severity::Error, format!("reorg {i}")); + } + for i in 0..3 { + let msg = rx.try_recv().unwrap_or_else(|_| panic!("occurrence {i} did not page")); + assert!(msg.contains(&format!("reorg {i}")), "{msg}"); + } + assert_eq!(w.count(), 0); + } + #[test] fn record_without_alert_notifier_is_fine() { // The common path: no -alertnotify configured, no sink installed. diff --git a/satd/src/config.rs b/satd/src/config.rs index c4095cdc2..050ca497c 100644 --- a/satd/src/config.rs +++ b/satd/src/config.rs @@ -1008,6 +1008,22 @@ pub struct Config { /// node warning, with `%s` replaced by the warning text. `None` = no /// dispatcher. Read once at startup (restart to change), like Core. pub alert_notify: Option, + /// Health detector: raise `tip_stall` after this many seconds with no + /// connected block. Deliberately not gated on IBD — see + /// `node::health::check_tip_stall_values`. `0` disables. SIGHUP-live. + pub alert_tip_stall_seconds: u64, + /// Health detector: raise `disk_low` below this many MiB free on the + /// watched data/blocks directory. `0` disables. SIGHUP-live. + pub alert_disk_free_mb: u64, + /// Health detector: raise `mempool_congested` at this percentage of the + /// mempool byte cap. `0` disables. SIGHUP-live. + pub alert_mempool_full_pct: u64, + /// Health detector: raise `peer_floor` below this many connected peers. + /// `0` disables. SIGHUP-live. + pub alert_peer_floor: u64, + /// Health detector: emit `deep_reorg` for a reorg at least this many + /// blocks deep. `0` disables. SIGHUP-live. + pub alert_reorg_depth: u64, /// Bitcoin Core `-startupnotify=`: a shell command run once after /// the node finishes starting up (no `%s` substitution). `None` = no /// hook. The supported alternative is a systemd `ExecStartPost=`. @@ -2023,6 +2039,16 @@ impl Config { connect = file_get_all("connect"); } + // Computed here rather than inline in the struct below, because the + // default reads `connect`, which the struct literal has already moved + // by the time it reaches the `alert_peer_floor` field. + let alert_peer_floor = cli + .alert_peer_floor + .or_else(|| file_get("alertpeerfloor").and_then(|v| v.parse().ok())) + .unwrap_or_else(|| { + node::health::defaults::peer_floor_for(network, connect.len()) + }); + let assumevalid = cli.assumevalid.or_else(|| file_get("assumevalid")); let assumevalidage = cli @@ -3247,6 +3273,23 @@ impl Config { .or_else(|| file_get("reorgwebhooksecret")), block_notify: cli.blocknotify.clone().or_else(|| file_get("blocknotify")), alert_notify: cli.alertnotify.clone().or_else(|| file_get("alertnotify")), + alert_tip_stall_seconds: cli + .alert_tip_stall_seconds + .or_else(|| file_get("alerttipstallseconds").and_then(|v| v.parse().ok())) + .unwrap_or_else(|| node::health::defaults::tip_stall_for(network)), + alert_disk_free_mb: cli + .alert_disk_free_mb + .or_else(|| file_get("alertdiskfreemb").and_then(|v| v.parse().ok())) + .unwrap_or(node::health::defaults::DISK_FREE_MB), + alert_mempool_full_pct: cli + .alert_mempool_full_pct + .or_else(|| file_get("alertmempoolfullpct").and_then(|v| v.parse().ok())) + .unwrap_or(node::health::defaults::MEMPOOL_FULL_PCT), + alert_peer_floor, + alert_reorg_depth: cli + .alert_reorg_depth + .or_else(|| file_get("alertreorgdepth").and_then(|v| v.parse().ok())) + .unwrap_or(node::health::defaults::reorg_depth_for(network)), startup_notify: cli.startupnotify.clone().or_else(|| file_get("startupnotify")), shutdown_notify: cli .shutdownnotify @@ -5112,6 +5155,41 @@ pub struct CliArgs { )] pub reorg_webhook_secret: Option, + #[arg( + long = "alerttipstallseconds", + value_name = "SECS", + help = "Raise the tip_stall health alert after this many seconds with no connected block (default 3600, 0 on regtest; 0 disables)" + )] + pub alert_tip_stall_seconds: Option, + + #[arg( + long = "alertdiskfreemb", + value_name = "MIB", + help = "Raise the disk_low health alert below this many MiB free on the data/blocks directory (default 10240; 0 disables)" + )] + pub alert_disk_free_mb: Option, + + #[arg( + long = "alertmempoolfullpct", + value_name = "PCT", + help = "Raise the mempool_congested health alert at this percentage of the mempool byte cap (default 90; 0 disables)" + )] + pub alert_mempool_full_pct: Option, + + #[arg( + long = "alertpeerfloor", + value_name = "N", + help = "Raise the peer_floor health alert below this many connected peers (default 3; 0 disables)" + )] + pub alert_peer_floor: Option, + + #[arg( + long = "alertreorgdepth", + value_name = "BLOCKS", + help = "Emit the deep_reorg health alert for a reorg at least this many blocks deep (default 3; 0 disables)" + )] + pub alert_reorg_depth: Option, + #[arg( long = "fast-start", value_name = "URL", @@ -6135,6 +6213,12 @@ pub const KNOWN_CONFIG_KEYS: &[&str] = &[ "shutdownnotify", "reorgwebhook", "reorgwebhooksecret", + // Health-alert detector thresholds (A3) + "alerttipstallseconds", + "alertdiskfreemb", + "alertmempoolfullpct", + "alertpeerfloor", + "alertreorgdepth", // MCP "mcp", "mcpport", @@ -6769,6 +6853,172 @@ rpcport=8332 assert_eq!(cfg.stream_prefix_max_bits, 32); } + #[test] + fn health_alert_threshold_defaults_and_parsing() { + use clap::Parser; + let dir = std::env::temp_dir().join(format!("satd-alert-cfg-{}", std::process::id())); + std::fs::create_dir_all(&dir).unwrap(); + + // Defaults come from the single source of truth in `node::health`, so + // the config layer and the detector can never disagree about them. + let cli = + CliArgs::try_parse_from(["satd", "--regtest", "--datadir", dir.to_str().unwrap()]) + .unwrap(); + let cfg = Config::from_cli(cli).unwrap(); + assert_eq!(cfg.alert_disk_free_mb, 10_240); + assert_eq!(cfg.alert_mempool_full_pct, 90); + // Three defaults are network-dependent, all because the alert would + // otherwise fire on the network behaving exactly as designed. + // + // A regtest node normally runs with no peers at all, reorgs on purpose + // and constantly — competing-chain and `invalidateblock` tests are the + // point of the harness — and only has blocks when a test mines them, so + // an idle chain is its resting state rather than a stall. + assert_eq!(cfg.alert_tip_stall_seconds, 0, "regtest only has blocks on demand"); + assert_eq!(cfg.alert_peer_floor, 0, "regtest defaults the peer floor off"); + assert_eq!(cfg.alert_reorg_depth, 0, "regtest reorgs deliberately"); + + // ...all armed where the condition really does mean something is + // wrong. + let cli = + CliArgs::try_parse_from(["satd", "--datadir", dir.to_str().unwrap()]).unwrap(); + let cfg_mainnet = Config::from_cli(cli).unwrap(); + assert_eq!(cfg_mainnet.alert_tip_stall_seconds, 3_600, "mainnet keeps the hour"); + assert_eq!(cfg_mainnet.alert_peer_floor, 3, "mainnet keeps the floor"); + assert_eq!(cfg_mainnet.alert_reorg_depth, 3, "3 deep is an incident on mainnet"); + + // Test networks reorg a few blocks deep as an ordinary consequence of + // thin hashrate. Paging on that trains the operator to ignore the + // alert, which costs them the mainnet one too — so the floor is raised + // rather than defaulted to mainnet's sensitivity. + let cli = + CliArgs::try_parse_from(["satd", "--signet", "--datadir", dir.to_str().unwrap()]) + .unwrap(); + let cfg_signet = Config::from_cli(cli).unwrap(); + assert_eq!(cfg_signet.alert_reorg_depth, 10, "signet does not page on routine reorgs"); + assert_eq!(cfg_signet.alert_peer_floor, 3, "but signet is a real network with peers"); + // A stall is not an ordinary property of a thin-hashrate chain the way + // a shallow reorg is, so the hour holds on the test networks. + assert_eq!(cfg_signet.alert_tip_stall_seconds, 3_600, "and it should still make blocks"); + + // Explicit CLI values parse through, including 0 (= detector off), + // which must not be swallowed by the `unwrap_or(default)`. + let cli = CliArgs::try_parse_from([ + "satd", + "--regtest", + "--datadir", + dir.to_str().unwrap(), + "--alerttipstallseconds", + "7200", + "--alertdiskfreemb", + "0", + "--alertmempoolfullpct", + "75", + "--alertpeerfloor", + "0", + "--alertreorgdepth", + "6", + ]) + .unwrap(); + let cfg = Config::from_cli(cli).unwrap(); + assert_eq!(cfg.alert_tip_stall_seconds, 7_200); + assert_eq!(cfg.alert_disk_free_mb, 0, "0 must disable, not fall back"); + assert_eq!(cfg.alert_mempool_full_pct, 75); + assert_eq!(cfg.alert_peer_floor, 0, "0 must disable, not fall back"); + assert_eq!(cfg.alert_reorg_depth, 6); + } + + /// `-connect=` pins the node to exactly the peers named and turns off both + /// DNS seeding and the fixed seeds, so the stock floor of 3 would raise a + /// warning that nothing on the node could ever clear — and it would land in + /// `getblockchaininfo.warnings`, which wallet software renders to end users. + /// The declared peer count is the floor. + #[test] + fn the_peer_floor_default_follows_the_connect_list() { + use clap::Parser; + let dir = std::env::temp_dir().join(format!("satd-alert-connect-{}", std::process::id())); + std::fs::create_dir_all(&dir).unwrap(); + let with = |args: &[&str]| { + let mut argv = vec!["satd", "--datadir", dir.to_str().unwrap()]; + argv.extend_from_slice(args); + Config::from_cli(CliArgs::try_parse_from(argv).unwrap()).unwrap() + }; + + // One trusted upstream: a floor of 1 is the most this node can ever + // satisfy, and it still alerts if that peer goes away. + assert_eq!(with(&["--connect", "10.0.0.5:8333"]).alert_peer_floor, 1); + assert_eq!( + with(&["--connect", "10.0.0.5:8333", "--connect", "10.0.0.6:8333"]).alert_peer_floor, + 2, + ); + + // Capped, not merely mirrored: past the stock floor the ordinary + // threshold governs, so a node wired to eight peers is not held to + // needing all eight. + let many: Vec<&str> = vec![ + "--connect", "10.0.0.1:8333", "--connect", "10.0.0.2:8333", "--connect", + "10.0.0.3:8333", "--connect", "10.0.0.4:8333", + ]; + assert_eq!(with(&many).alert_peer_floor, 3, "the cap is the stock floor"); + + // An explicit value still wins over the derived default, in both + // directions. + assert_eq!( + with(&["--connect", "10.0.0.5:8333", "--alertpeerfloor", "0"]).alert_peer_floor, + 0, + "explicit 0 disables even with -connect set", + ); + assert_eq!( + with(&["--connect", "10.0.0.5:8333", "--alertpeerfloor", "5"]).alert_peer_floor, + 5, + ); + + // Regtest stays off regardless: -connect there is routine, and the + // network default already answered this question. + assert_eq!( + with(&["--regtest", "--connect", "10.0.0.5:8333"]).alert_peer_floor, + 0, + ); + + // The config file is the same path — this is where a -connect node's + // settings actually live. + let conf = dir.join("connect-floor.conf"); + std::fs::write(&conf, "connect=10.0.0.5:8333\nconnect=10.0.0.6:8333\n").unwrap(); + let cfg = Config::from_cli( + CliArgs::try_parse_from([ + "satd", + "--datadir", + dir.to_str().unwrap(), + "--conf", + conf.to_str().unwrap(), + ]) + .unwrap(), + ) + .unwrap(); + assert_eq!(cfg.connect.len(), 2, "the fixture must actually take effect"); + assert_eq!(cfg.alert_peer_floor, 2, "connect= from bitcoin.conf counts too"); + } + + #[test] + fn health_alert_keys_are_known_and_parse_from_the_config_file() { + for key in [ + "alerttipstallseconds", + "alertdiskfreemb", + "alertmempoolfullpct", + "alertpeerfloor", + "alertreorgdepth", + ] { + assert!(is_known_config_key(key), "{key} should be a known key"); + } + let cf = ConfigFile::parse("alerttipstallseconds=120\nalertpeerfloor=1\n") + .expect("alert threshold keys should parse from bitcoin.conf"); + assert_eq!( + cf.global.get("alerttipstallseconds").unwrap().last().unwrap(), + "120" + ); + assert_eq!(cf.global.get("alertpeerfloor").unwrap().last().unwrap(), "1"); + } + #[test] fn policy_engine_config_defaults_and_parsing() { use clap::Parser; @@ -7255,6 +7505,11 @@ rpcport=8332 shutdownnotify: None, reorg_webhook: None, reorg_webhook_secret: None, + alert_tip_stall_seconds: None, + alert_disk_free_mb: None, + alert_mempool_full_pct: None, + alert_peer_floor: None, + alert_reorg_depth: None, fast_start: None, fast_start_sha256: None, events_node_id: None, @@ -7534,6 +7789,11 @@ rpcport=8332 shutdownnotify: None, reorg_webhook: None, reorg_webhook_secret: None, + alert_tip_stall_seconds: None, + alert_disk_free_mb: None, + alert_mempool_full_pct: None, + alert_peer_floor: None, + alert_reorg_depth: None, fast_start: None, fast_start_sha256: None, events_node_id: None, diff --git a/satd/src/main.rs b/satd/src/main.rs index 46a9c5a85..a7a972e94 100644 --- a/satd/src/main.rs +++ b/satd/src/main.rs @@ -2102,6 +2102,68 @@ async fn main() { "RPC server listening" ); + // Node-health detectors (A3). Always on: the conditions they watch are the + // ones an operator gets paged about, the poll is a handful of atomic reads + // plus one statvfs every 15s, and nothing leaves the node unless a `status` + // subscriber or a webhook asks. Individual detectors are disabled by + // setting their threshold to 0. + // + // Spawned on the API runtime, never the consensus core: a stall detector + // whose poll can be delayed by block connection is the one thing it must + // not be. The thresholds live behind an Arc so SIGHUP can retune them + // without a restart. + let alert_thresholds = std::sync::Arc::new(node::health::AlertThresholds::new( + config.alert_tip_stall_seconds, + config.alert_disk_free_mb, + config.alert_mempool_full_pct, + config.alert_peer_floor, + config.alert_reorg_depth, + )); + let health_state = match chain_state.subscribe_chain_events() { + Some(chain_rx) => { + let _api_guard = api_handle.enter(); + let state = node::health::spawn_health_detectors( + node::health::HealthInputs { + chain_state: chain_state.clone(), + mempool: mempool.clone(), + peer_manager: peer_manager.clone(), + publisher: event_publisher.clone(), + warnings: chain_state.warnings().clone(), + thresholds: alert_thresholds.clone(), + // Watch whatever directory actually grows. With a split + // `-blocksdir` that is the blocks volume, which is the one + // that fills; the chainstate volume rarely does. + disk_watch_path: config + .blocksdir + .clone() + .unwrap_or_else(|| net_datadir.clone()), + }, + chain_rx, + shutdown_rx.clone(), + ); + tracing::info!( + target: "health", + tip_stall_seconds = config.alert_tip_stall_seconds, + disk_free_mb = config.alert_disk_free_mb, + mempool_full_pct = config.alert_mempool_full_pct, + peer_floor = config.alert_peer_floor, + reorg_depth = config.alert_reorg_depth, + "health detectors started", + ); + Some(state) + } + None => { + // No chain-event broadcast means no block-connect signal, so + // `tip_stall` could never clear and `deep_reorg` could never fire. + // Running half the detectors would be worse than running none. + tracing::warn!( + target: "health", + "chain-event broadcast unavailable; health detectors not started" + ); + None + } + }; + // Start MCP server if enabled if config.mcp { let mcp_ctx = std::sync::Arc::new(satd_mcp::McpContext { @@ -2115,6 +2177,7 @@ async fn main() { mempool_history: mempool_history.clone(), addr_enabled: config.addressindex, addr_subs: Some(address_index_concrete.subscription_registry()), + health: health_state.clone(), }); // `--mcp` only enables the feature; `--mcpport` provides the transport. @@ -2224,6 +2287,7 @@ async fn main() { version: env!("CARGO_PKG_VERSION"), addr_subs: Some(address_index_concrete.subscription_registry()), addr_enabled: config.addressindex, + health: health_state.clone(), }; let rx = shutdown_rx.clone(); api_handle.spawn(async move { @@ -3023,6 +3087,7 @@ async fn main() { webhook: reorg_webhook_handle, rpc_auth: auth.clone(), token_store: token_store.clone(), + alert_thresholds: alert_thresholds.clone(), }; loop { tokio::select! { diff --git a/satd/src/reload.rs b/satd/src/reload.rs index 8984ac834..6532afe42 100644 --- a/satd/src/reload.rs +++ b/satd/src/reload.rs @@ -181,6 +181,10 @@ pub struct ReloadHandles { /// are rotatable live (the cookie is preserved). Shared with every RPC /// listener surface, so a reload covers all of them at once. pub rpc_auth: Arc, + /// Health-detector raise thresholds. Always present (the detector task is + /// always configured, even when every threshold is 0), so a threshold + /// change always has somewhere to land. + pub alert_thresholds: Arc, /// Unified-auth bearer-token store, present only when `authfile=` is set. /// Re-read on every SIGHUP independently of the rest of the config so that /// removing a `[[token]]` revokes it live; a re-read error keeps the @@ -865,6 +869,26 @@ fn field_specs() -> Vec { restart!("shutdownnotify", shutdown_notify), live!("reorgwebhook", reorg_webhook, apply_webhook), live_secret!("reorgwebhooksecret", reorg_webhook_secret, apply_webhook), + // ---- Health-alert thresholds (A3) ---- + // Retuning an alert must not need a restart: the operator is usually + // retuning it *because* it is firing, and taking the node down to + // silence a pager is the wrong trade. Each pushes into the shared + // atomics the detector task reads on its next poll. + live!("alerttipstallseconds", alert_tip_stall_seconds, |c, h| h + .alert_thresholds + .set_tip_stall_secs(c.alert_tip_stall_seconds)), + live!("alertdiskfreemb", alert_disk_free_mb, |c, h| h + .alert_thresholds + .set_disk_free_mb(c.alert_disk_free_mb)), + live!("alertmempoolfullpct", alert_mempool_full_pct, |c, h| h + .alert_thresholds + .set_mempool_full_pct(c.alert_mempool_full_pct)), + live!("alertpeerfloor", alert_peer_floor, |c, h| h + .alert_thresholds + .set_peer_floor(c.alert_peer_floor)), + live!("alertreorgdepth", alert_reorg_depth, |c, h| h + .alert_thresholds + .set_reorg_depth(c.alert_reorg_depth)), // ---- MCP ---- restart!("mcp", mcp), restart!("mcpport", mcp_port), @@ -1076,6 +1100,13 @@ mod tests { "maxshutdownsecs", "reorgwebhook", "reorgwebhooksecret", + // Retuning an alert must not require a restart — the operator is + // usually retuning it because it is firing. + "alerttipstallseconds", + "alertdiskfreemb", + "alertmempoolfullpct", + "alertpeerfloor", + "alertreorgdepth", ] { assert!( find(key).apply.is_some(), diff --git a/satd/tests/e2e/streaming.rs b/satd/tests/e2e/streaming.rs index 7e55148d1..c3273e044 100644 --- a/satd/tests/e2e/streaming.rs +++ b/satd/tests/e2e/streaming.rs @@ -1773,3 +1773,343 @@ async fn grpc_watch_txid_unconfirmed_on_invalidateblock() { // driven by real P2P block propagation rather than `invalidateblock`) would be // additive — the event paths themselves are identical and already exercised // here — so it is intentionally left as future coverage, not a gap. + +// =========================================================================== +// Node-health status events (A3) +// =========================================================================== +// +// The detectors' raise/clear state machines, hysteresis, and warning mapping +// are unit-tested in `node::health`. What can only be checked end-to-end is +// that a real condition on a real daemon reaches a real subscriber over a +// socket, carries the right shape, and is simultaneously visible through +// `getwarnings` — the three surfaces the design requires never to disagree. +// +// `peer_floor` is deliberately not covered here: it holds for 60s in either +// direction by design, which is longer than an E2E test should sit. + +/// Retune one health threshold on a running node and wait for the reload to +/// land. The config file is the only input SIGHUP re-reads (CLI args stay +/// authoritative), so the value is written there — which means it must not have +/// been passed on the command line as anything other than its startup default. +async fn set_alert_threshold(sn: &StreamingNode, key: &str, value: &str) { + let conf = sn.node.datadir.join("bitcoin.conf"); + std::fs::write(&conf, format!("[regtest]\n{key}={value}\n")).expect("write bitcoin.conf"); + let pid = sn.node.process.id().to_string(); + let status = tokio::task::spawn_blocking(move || { + std::process::Command::new("kill") + .arg("-HUP") + .arg(pid) + .status() + .expect("spawn kill -HUP") + }) + .await + .unwrap(); + assert!(status.success(), "kill -HUP returned {status:?}"); + // The reload runs between signal-loop iterations, and the detector picks the + // new value up on its next poll. + tokio::time::sleep(Duration::from_millis(500)).await; +} + +/// A status event's envelope shape, for the assertions below. +fn assert_status_shape(ev: &serde_json::Value, kind: &str, state: &str) { + assert_eq!(ev["body"]["category"], "status", "event: {ev}"); + assert_eq!(ev["body"]["kind"], kind, "event: {ev}"); + assert_eq!(ev["body"]["state"], state, "event: {ev}"); + assert!( + ev["body"]["details"].is_object(), + "details must always be an object (possibly empty): {ev}" + ); + assert!( + ev["cursor"].is_null(), + "status events are not replayable and must carry no cursor: {ev}" + ); +} + +/// Fetch the active warning ids from `getwarnings`. +async fn warning_ids(sn: &StreamingNode) -> Vec { + let rpc = sn.node.rpc_handle(); + let resp = tokio::task::spawn_blocking(move || rpc.call("getwarnings", vec![])) + .await + .unwrap() + .unwrap(); + resp["result"]["warnings"] + .as_array() + .map(|a| { + a.iter() + .filter_map(|w| w["id"].as_str().map(String::from)) + .collect() + }) + .unwrap_or_default() +} + +/// `tip_stall` raises when no block connects inside the window and clears the +/// moment one does — and the same transition is visible in `getwarnings`. +/// +/// The threshold is 1s; the detector polls every 15s, so the raise lands on the +/// first poll after startup and the whole test fits in one poll interval plus +/// slack. +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn ws_status_tip_stall_raises_then_clears_on_next_block() { + let sn = start_streaming_async(vec![ + "--alerttipstallseconds=1", + // Silence the other detectors so the only status traffic is ours. + "--alertdiskfreemb=0", + "--alertmempoolfullpct=0", + "--alertpeerfloor=0", + "--alertreorgdepth=0", + ]) + .await; + + // Regtest's genesis tip is 24h+ old, which the IBD heuristic reads as + // "still syncing" — and `tip_stall` is suppressed during IBD. Mine one + // block so the tip timestamp is current and the node is out of IBD. + mine_n(&sn, 1).await; + + // Bit 16 = status. Explicitly requested, since it is not in the default. + let mut ws = WsClient::connect(sn.ws_port()).await; + ws.send_control(serde_json::json!({"type": "set_categories", "categories": 16})) + .await; + + let ev = ws + .next_json_matching(30, |v| v["body"]["category"] == "status") + .await; + assert_status_shape(&ev, "tip_stall", "raised"); + assert_eq!(ev["body"]["severity"], "critical", "event: {ev}"); + assert!( + ev["body"]["details"]["threshold_seconds"] == "1", + "details carry the configured threshold: {ev}" + ); + + // The same condition is visible to an operator polling RPC. + assert!( + warning_ids(&sn).await.contains(&"alert.tip_stall".to_string()), + "a raised status must also be an active warning", + ); + + // Mining clears it immediately — the clear side is event-driven, not polled. + mine_n(&sn, 1).await; + let ev = ws + .next_json_matching(15, |v| v["body"]["category"] == "status") + .await; + assert_status_shape(&ev, "tip_stall", "cleared"); + assert!( + !warning_ids(&sn).await.contains(&"alert.tip_stall".to_string()), + "clearing the status must retract the warning", + ); +} + +/// A firing `tip_stall` clears when the operator raises the threshold above the +/// current tip age — without waiting for a block. +/// +/// The fast clear is event-driven, but a node that is genuinely not receiving +/// blocks has no event to clear on. Retuning the threshold live is the +/// documented way to quiet such an alert, so the poll path has to honour the new +/// level in both directions; otherwise the warning, the gauge, and the stream +/// state stay raised until a block that may be a long way off. +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn ws_status_tip_stall_clears_when_the_threshold_is_raised() { + // `alerttipstallseconds` stays off the command line so SIGHUP can move it: + // startup CLI args remain authoritative across reloads. Its default (1h) + // does not fire inside the test. + let sn = start_streaming_async(vec![ + "--alertdiskfreemb=0", + "--alertmempoolfullpct=0", + "--alertpeerfloor=0", + "--alertreorgdepth=0", + ]) + .await; + mine_n(&sn, 1).await; + + let mut ws = WsClient::connect(sn.ws_port()).await; + ws.send_control(serde_json::json!({"type": "set_categories", "categories": 16})) + .await; + tokio::time::sleep(Duration::from_millis(300)).await; + + set_alert_threshold(&sn, "alerttipstallseconds", "1").await; + let ev = ws + .next_json_matching(45, |v| v["body"]["category"] == "status") + .await; + assert_status_shape(&ev, "tip_stall", "raised"); + + // Raise the threshold past the current age. No block is mined: the clear + // must come from the poll noticing the new level. + set_alert_threshold(&sn, "alerttipstallseconds", "86400").await; + let ev = ws + .next_json_matching(45, |v| v["body"]["category"] == "status") + .await; + assert_status_shape(&ev, "tip_stall", "cleared"); + assert_eq!(ev["body"]["details"]["threshold_seconds"], "86400", "event: {ev}"); + assert!( + !warning_ids(&sn).await.contains(&"alert.tip_stall".to_string()), + "clearing the status must retract the warning", + ); +} + +/// `disk_low` raises when free space drops under the configured floor — and +/// the floor is retunable live, without a restart. +/// +/// The node starts with the detector off and it is switched on by SIGHUP, for +/// two reasons: it makes the condition become true *after* the subscriber has +/// attached (status events are not replayable, so a condition raised during +/// startup is invisible to a client that connects later — snapshot-on-subscribe +/// is deliberately deferred), and it exercises the live-reload path that exists +/// precisely so an operator can retune a firing alert without downtime. +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn ws_status_disk_low_raises_after_a_live_threshold_change() { + // `alertdiskfreemb` is deliberately NOT passed on the command line: SIGHUP + // re-reads only the config file, with the startup CLI args still + // authoritative, so a CLI-pinned value could never be retuned. + // + // PRECONDITION: the host needs **more than 10 GiB free** on /tmp. This test + // drives a raise by moving the threshold, so it needs `disk_low` to start + // clear — and the default floor is 10 GiB. On a fuller disk the alert is + // already raised at startup, `raise_if_new` is a no-op, and this fails as a + // bare 45-second websocket timeout with nothing pointing at the cause. + let sn = start_streaming_async(vec![ + "--alerttipstallseconds=0", + "--alertmempoolfullpct=0", + "--alertpeerfloor=0", + "--alertreorgdepth=0", + ]) + .await; + + let mut ws = WsClient::connect(sn.ws_port()).await; + ws.send_control(serde_json::json!({"type": "set_categories", "categories": 16})) + .await; + tokio::time::sleep(Duration::from_millis(300)).await; + + // An unreachable floor (2^44 MiB) guarantees the condition on any real + // filesystem. Applied live — no restart. + set_alert_threshold(&sn, "alertdiskfreemb", "17592186044416").await; + + let ev = ws + .next_json_matching(45, |v| v["body"]["category"] == "status") + .await; + assert_status_shape(&ev, "disk_low", "raised"); + assert_eq!(ev["body"]["severity"], "critical", "event: {ev}"); + // The actionable numbers ride in `details`, not only in the message. + assert!(ev["body"]["details"]["free_bytes"].is_string(), "event: {ev}"); + assert!(ev["body"]["details"]["threshold_bytes"].is_string(), "event: {ev}"); + // The watched path is deliberately NOT on the wire: this event reaches + // every `status` subscriber and any push-notification body, and an absolute + // datadir path usually names the account satd runs under. It goes to the + // node's log instead. + assert!( + ev["body"]["details"]["path"].is_null(), + "the datadir path must not be exposed on the wire: {ev}" + ); + assert!(warning_ids(&sn).await.contains(&"alert.disk_low".to_string())); +} + +/// A reorg at or beyond `alertreorgdepth` emits a one-shot `deep_reorg` edge +/// carrying the true depth — which the marker alone does not provide, so this +/// also covers the disconnect-counting derivation. +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn ws_status_deep_reorg_reports_true_depth() { + let sn = start_streaming_async(vec![ + "--alertreorgdepth=2", + "--alerttipstallseconds=0", + "--alertdiskfreemb=0", + "--alertmempoolfullpct=0", + "--alertpeerfloor=0", + ]) + .await; + + let rpc = sn.node.rpc_handle(); + let addr = mine_addr(0x73); + let hashes = tokio::task::spawn_blocking(move || rpc.mine(3, &addr)) + .await + .unwrap(); + assert_eq!(hashes.len(), 3); + let height2 = hashes[1].clone(); + + let mut ws = WsClient::connect(sn.ws_port()).await; + ws.send_control(serde_json::json!({"type": "set_categories", "categories": 16})) + .await; + tokio::time::sleep(Duration::from_millis(300)).await; + + // Invalidating height 2 rolls back heights 3 and 2 — a 2-deep truncation. + let rpc2 = sn.node.rpc_handle(); + let resp = tokio::task::spawn_blocking(move || { + rpc2.call("invalidateblock", vec![serde_json::json!(height2)]) + }) + .await + .unwrap() + .unwrap(); + assert!(resp["error"].is_null(), "invalidateblock errored: {resp:?}"); + + // A truncation reorg has no replacement chain to connect, so the detector + // finalizes the depth count on its next poll (≤ one poll interval). + let ev = ws + .next_json_matching(45, |v| v["body"]["category"] == "status") + .await; + assert_status_shape(&ev, "deep_reorg", "edge"); + assert_eq!(ev["body"]["severity"], "critical", "event: {ev}"); + assert_eq!(ev["body"]["details"]["depth"], "2", "event: {ev}"); + assert_eq!(ev["body"]["details"]["from_height"], "3", "event: {ev}"); + assert_eq!(ev["body"]["details"]["fork_height"], "1", "event: {ev}"); +} + +/// The `status` category is explicit-request only: a default (`categories=0`) +/// subscriber must never receive a status event, even while one is firing. +/// This is the upgrade-safety property — a client written against an older +/// node must not start receiving a body it has no parser for. +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn ws_status_absent_from_the_default_category_mask() { + // `alertdiskfreemb` is deliberately NOT passed on the command line: SIGHUP + // re-reads only the config file, with the startup CLI args still + // authoritative, so a CLI-pinned value could never be retuned. + // + // PRECONDITION: the host needs **more than 10 GiB free** on /tmp. This test + // drives a raise by moving the threshold, so it needs `disk_low` to start + // clear — and the default floor is 10 GiB. On a fuller disk the alert is + // already raised at startup, `raise_if_new` is a no-op, and this fails as a + // bare 45-second websocket timeout with nothing pointing at the cause. + let sn = start_streaming_async(vec![ + "--alerttipstallseconds=0", + "--alertmempoolfullpct=0", + "--alertpeerfloor=0", + "--alertreorgdepth=0", + ]) + .await; + + // One subscriber on the default mask, one explicitly asking for status. + let mut default_ws = WsClient::connect(sn.ws_port()).await; + let mut status_ws = WsClient::connect(sn.ws_port()).await; + status_ws + .send_control(serde_json::json!({"type": "set_categories", "categories": 16})) + .await; + tokio::time::sleep(Duration::from_millis(300)).await; + + set_alert_threshold(&sn, "alertdiskfreemb", "17592186044416").await; + + // The explicit subscriber gets it, proving the condition really fired + // while both subscribers were attached. + let ev = status_ws + .next_json_matching(45, |v| v["body"]["category"] == "status") + .await; + assert_status_shape(&ev, "disk_low", "raised"); + + // Mine so the default subscriber definitely has traffic to read — if it + // were going to see a status event, it would be interleaved with these. + mine_n(&sn, 2).await; + // Drain the default subscriber for a fixed window rather than stopping at + // the first gap: `next_json_opt` yields `None` for a keepalive ping as well + // as for a timeout, so a break-on-`None` loop would end before reading + // anything real. + let deadline = std::time::Instant::now() + Duration::from_secs(10); + let mut seen: Vec = Vec::new(); + while std::time::Instant::now() < deadline { + if let Some(v) = default_ws.next_json_opt(1).await { + assert_ne!( + v["body"]["category"], "status", + "a categories=0 subscriber must never receive status events: {v}" + ); + seen.push(v["body"]["category"].as_str().unwrap_or("?").to_string()); + } + } + assert!( + seen.iter().any(|c| c == "chain"), + "the default subscriber should still see chain events; saw: {seen:?}" + ); +}