diff --git a/CHANGELOG.md b/CHANGELOG.md index 55e52720..72057cc3 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -6,6 +6,7 @@ - Fixed Bitcoin and other containers being forcibly stopped after ten seconds during managed updates and restarts. - Existing installations now receive the same graceful shutdown allowance as new containers, without restarting apps just to apply this setting. +- Prevented unnecessary Lightning restarts when Bitcoin has stayed running; dependency restarts now require an observed Bitcoin container change. - Includes the Cashu payment, optional Bitcoin pruning, Lightning readiness, and explorer improvements from 1.8.20. ## v1.8.20-alpha (2026-09-29) diff --git a/core/archipelago/src/container/prod_orchestrator.rs b/core/archipelago/src/container/prod_orchestrator.rs index f25d8dbd..801193ef 100644 --- a/core/archipelago/src/container/prod_orchestrator.rs +++ b/core/archipelago/src/container/prod_orchestrator.rs @@ -1174,15 +1174,21 @@ impl ReconcileReport { fn cascade_pairs_for_report<'r>( report: &'r ReconcileReport, user_stopped: &std::collections::HashSet, + changed_backends: &HashSet, ) -> Vec<(&'r str, &'static str)> { let mut pairs = Vec::new(); for (backend, action) in &report.actions { if !matches!( action, - ReconcileAction::Installed | ReconcileAction::Started + ReconcileAction::NoOp | ReconcileAction::Started | ReconcileAction::Installed ) { continue; } + // A successful systemctl start can be a no-op after a transient + // Podman inspect failure. Require a witnessed lifecycle change. + if !changed_backends.contains(backend) { + continue; + } for dep in crate::app_ops::address_caching_dependents(backend) { let dep_untouched = report .actions @@ -1196,6 +1202,25 @@ fn cascade_pairs_for_report<'r>( pairs } +/// Only positive runtime evidence permits disrupting an address-caching wallet. +/// A known absent/stopped backend becoming running, a new container ID, or a +/// changed start timestamp qualifies. A failed observation never does. +fn backend_instance_changed(before: Option<&ContainerStatus>, after: &ContainerStatus) -> bool { + if after.state != ContainerState::Running || after.id.is_empty() { + return false; + } + let Some(before) = before else { + return true; + }; + if before.id.is_empty() { + return false; + } + if before.id != after.id || before.state != ContainerState::Running { + return true; + } + matches!((&before.started_at, &after.started_at), (Some(a), Some(b)) if !a.is_empty() && !b.is_empty() && a != b) +} + #[derive(Debug, Default)] pub struct AdoptionReport { pub adopted: Vec, @@ -1909,12 +1934,33 @@ impl ProdContainerOrchestrator { _ => 2, }); // Live container names (any state), for the same recovery check. - let present_containers: std::collections::HashSet = self - .runtime - .list_containers() - .await - .map(|cs| cs.into_iter().map(|c| c.name).collect()) + let listed_containers = self.runtime.list_containers().await.ok(); + let present_containers: HashSet = listed_containers + .as_ref() + .map(|cs| cs.iter().map(|c| c.name.clone()).collect()) .unwrap_or_default(); + // Keep unknown distinct from confirmed absence. Runtime queries can + // fail under load while systemd still has a healthy running backend. + let mut backend_before: HashMap> = HashMap::new(); + for lm in &manifests { + let id = &lm.manifest.app.id; + if crate::app_ops::address_caching_dependents(id).is_empty() { + continue; + } + let name = compute_container_name(&lm.manifest); + match self.runtime.get_container_status(&name).await { + Ok(status) => { + backend_before.insert(id.clone(), Some(status)); + } + Err(_) if listed_containers.is_some() && !present_containers.contains(&name) => { + backend_before.insert(id.clone(), None); + } + Err(err) => { + tracing::warn!(backend = %id, error = %err, + "cannot observe backend before reconcile; will not infer a dependency restart from an action report"); + } + } + } let mut report = ReconcileReport::default(); let disk_gb = self.disk_gb().await; let bitcoin_pruned = disk_gb < ARCHIVAL_BITCOIN_DISK_GB @@ -2096,7 +2142,20 @@ impl ProdContainerOrchestrator { // state recovery, repair recreate, boot InstallMissing) moves the // address behind a running dependent's back — §C "restart lnd after // ANY bitcoin recreate". - for (backend, dep) in cascade_pairs_for_report(&report, &user_stopped) { + let mut changed_backends = HashSet::new(); + for (backend, before) in &backend_before { + let Some(name) = container_name_by_app_id.get(backend) else { + continue; + }; + if let Ok(after) = self.runtime.get_container_status(name).await { + if backend_instance_changed(before.as_ref(), &after) { + changed_backends.insert(backend.clone()); + } + } + } + // A user stop during a slow reconcile pass still takes precedence. + let user_stopped = crate::crash_recovery::load_user_stopped(&self.data_dir).await; + for (backend, dep) in cascade_pairs_for_report(&report, &user_stopped, &changed_backends) { // Same rule as the RPC cascade: hold the dependent's op lock // across the restart; skip when a worker is mid-sequence. let lock = crate::app_ops::op_lock(dep); @@ -6409,6 +6468,67 @@ app: ); } + #[test] + fn backend_cascade_requires_observed_instance_change() { + let running = ContainerStatus { + id: "container-1".into(), + name: "bitcoin-core".into(), + state: ContainerState::Running, + started_at: Some("start-1".into()), + health: None, + exit_code: None, + image: "bitcoin:1".into(), + created: "created-1".into(), + ports: vec![], + lan_address: None, + }; + assert!(!backend_instance_changed(Some(&running), &running)); + assert!(backend_instance_changed(None, &running)); + let mut before = running.clone(); + before.state = ContainerState::Exited; + assert!(backend_instance_changed(Some(&before), &running)); + before = running.clone(); + before.id = "old-container".into(); + assert!(backend_instance_changed(Some(&before), &running)); + before = running.clone(); + before.started_at = Some("earlier-start".into()); + assert!(backend_instance_changed(Some(&before), &running)); + before.started_at = None; + assert!(!backend_instance_changed(Some(&before), &running)); + before.id.clear(); + assert!(!backend_instance_changed(Some(&before), &running)); + let mut after = running.clone(); + after.state = ContainerState::Exited; + assert!(!backend_instance_changed(None, &after)); + after = running.clone(); + after.id.clear(); + assert!(!backend_instance_changed(None, &after)); + } + + #[test] + fn cascade_ignores_false_started_report_but_detects_real_exec_drift() { + let none = HashSet::new(); + let mut report = ReconcileReport { + actions: vec![ + ("bitcoin-core".into(), ReconcileAction::Started), + ("lnd".into(), ReconcileAction::NoOp), + ], + failures: vec![], + }; + // systemctl start of an already active unit does not move its address. + assert!(cascade_pairs_for_report(&report, &none, &none).is_empty()); + // A unit exec rewrite can restart Bitcoin while the outer reconcile + // action remains NoOp. Runtime evidence still requires LND to reconnect. + let changed = ["bitcoin-core".into()].into(); + report.actions[0].1 = ReconcileAction::NoOp; + assert_eq!( + cascade_pairs_for_report(&report, &none, &changed), + vec![("bitcoin-core", "lnd")] + ); + report.actions[0].1 = ReconcileAction::Left("lifecycle-op-in-flight".into()); + assert!(cascade_pairs_for_report(&report, &none, &changed).is_empty()); + } + #[test] fn cascade_pairs_cover_backend_recreate_with_running_dependent() { use std::collections::HashSet; @@ -6420,6 +6540,7 @@ app: failures: vec![], }; let none = HashSet::new(); + let changed: HashSet = ["bitcoin-core".into(), "bitcoin-knots".into()].into(); // Backend recreated while lnd sat running (NoOp) → cascade. let r = report(vec![ @@ -6427,7 +6548,7 @@ app: ("lnd", ReconcileAction::NoOp), ]); assert_eq!( - cascade_pairs_for_report(&r, &none), + cascade_pairs_for_report(&r, &none, &changed), vec![("bitcoin-knots", "lnd")] ); @@ -6437,7 +6558,7 @@ app: ("lnd", ReconcileAction::NoOp), ]); assert_eq!( - cascade_pairs_for_report(&r, &none), + cascade_pairs_for_report(&r, &none, &changed), vec![("bitcoin-core", "lnd")] ); @@ -6446,7 +6567,7 @@ app: ("bitcoin-knots", ReconcileAction::NoOp), ("lnd", ReconcileAction::NoOp), ]); - assert!(cascade_pairs_for_report(&r, &none).is_empty()); + assert!(cascade_pairs_for_report(&r, &none, &none).is_empty()); // Dependent itself (re)started this pass → it already resolved the // fresh address; no cascade. @@ -6454,7 +6575,7 @@ app: ("bitcoin-knots", ReconcileAction::Installed), ("lnd", ReconcileAction::Started), ]); - assert!(cascade_pairs_for_report(&r, &none).is_empty()); + assert!(cascade_pairs_for_report(&r, &none, &changed).is_empty()); // User-stopped dependent is never bounced. let r = report(vec![ @@ -6462,14 +6583,14 @@ app: ("lnd", ReconcileAction::NoOp), ]); let stopped: HashSet = ["lnd".to_string()].into(); - assert!(cascade_pairs_for_report(&r, &stopped).is_empty()); + assert!(cascade_pairs_for_report(&r, &stopped, &changed).is_empty()); // Non-backend recreates don't cascade anything. let r = report(vec![ ("grafana", ReconcileAction::Installed), ("lnd", ReconcileAction::NoOp), ]); - assert!(cascade_pairs_for_report(&r, &none).is_empty()); + assert!(cascade_pairs_for_report(&r, &none, &changed).is_empty()); } #[tokio::test] diff --git a/docs/repair-release-20260929.md b/docs/repair-release-20260929.md index 6202864a..20399e6b 100644 --- a/docs/repair-release-20260929.md +++ b/docs/repair-release-20260929.md @@ -286,3 +286,24 @@ unlocked at 08:47 UTC. The existing backend-address cascade then performed a graceful LND restart at 08:57 UTC after Bitcoin reconciliation completed; LND automatically unlocked again and reached chain-sync waiting. No manual wallet unlock or restart was used for this recovery. + +### False dependency restart exposed during monitoring — 09:08 UTC + +The initial 1.8.21 candidate passed all 1,557 isolated backend tests and 1,120 +frontend tests. Monitoring nevertheless found another managed LND restart at +09:08:32 while Bitcoin's container/start timestamp remained unchanged. Management +logs explicitly attribute it to the backend-address cascade. This also makes +the earlier 08:57 cascade suspect; it must not be described as a proven necessary +restart. These service restarts preceded the isolated test executable, whose +namespace boundaries remain intact. + +The cascade trusted Started/Installed action reports. A failed runtime inspection +followed by successful systemctl start of an already active unit can produce +Started without changing Bitcoin. Dependency restarts now require observed +container-ID, running-state, or start-time changes. Failed observations remain +unknown, not absence; a known absent backend becoming running still qualifies. +Actual exec-drift restarts are recognized even when their outer report is NoOp. +Stopped/lifecycle-in-flight dependents remain excluded, and user stop markers +are re-read after the potentially slow pass. Added runtime-observation and +false-action/real-exec-drift regression cases; full isolated rerun pending. +Stopped the first optimized build and preparing new artifacts from this correction. diff --git a/neode-ui/src/views/settings/AccountInfoSection.vue b/neode-ui/src/views/settings/AccountInfoSection.vue index c6bc1258..8c3ff974 100644 --- a/neode-ui/src/views/settings/AccountInfoSection.vue +++ b/neode-ui/src/views/settings/AccountInfoSection.vue @@ -371,6 +371,7 @@ init()

Fixed Bitcoin and other containers being forcibly stopped after ten seconds during managed updates and restarts.

Existing installations now receive the same graceful shutdown allowance as new containers, without restarting apps just to apply this setting.

+

Prevented unnecessary Lightning restarts when Bitcoin has stayed running; dependency restarts now require an observed Bitcoin container change.

Includes the Cashu payment, optional Bitcoin pruning, Lightning readiness, and explorer improvements from 1.8.20.