fix(lnd): require observed Bitcoin lifecycle change before dependency restart
This commit is contained in:
@@ -6,6 +6,7 @@
|
|||||||
|
|
||||||
- Fixed Bitcoin and other containers being forcibly stopped after ten seconds during managed updates and restarts.
|
- Fixed Bitcoin and other containers being forcibly stopped after ten seconds during managed updates and restarts.
|
||||||
- Existing installations now receive the same graceful shutdown allowance as new containers, without restarting apps just to apply this setting.
|
- Existing installations now receive the same graceful shutdown allowance as new containers, without restarting apps just to apply this setting.
|
||||||
|
- Prevented unnecessary Lightning restarts when Bitcoin has stayed running; dependency restarts now require an observed Bitcoin container change.
|
||||||
- Includes the Cashu payment, optional Bitcoin pruning, Lightning readiness, and explorer improvements from 1.8.20.
|
- Includes the Cashu payment, optional Bitcoin pruning, Lightning readiness, and explorer improvements from 1.8.20.
|
||||||
|
|
||||||
## v1.8.20-alpha (2026-09-29)
|
## v1.8.20-alpha (2026-09-29)
|
||||||
|
|||||||
@@ -1174,15 +1174,21 @@ impl ReconcileReport {
|
|||||||
fn cascade_pairs_for_report<'r>(
|
fn cascade_pairs_for_report<'r>(
|
||||||
report: &'r ReconcileReport,
|
report: &'r ReconcileReport,
|
||||||
user_stopped: &std::collections::HashSet<String>,
|
user_stopped: &std::collections::HashSet<String>,
|
||||||
|
changed_backends: &HashSet<String>,
|
||||||
) -> Vec<(&'r str, &'static str)> {
|
) -> Vec<(&'r str, &'static str)> {
|
||||||
let mut pairs = Vec::new();
|
let mut pairs = Vec::new();
|
||||||
for (backend, action) in &report.actions {
|
for (backend, action) in &report.actions {
|
||||||
if !matches!(
|
if !matches!(
|
||||||
action,
|
action,
|
||||||
ReconcileAction::Installed | ReconcileAction::Started
|
ReconcileAction::NoOp | ReconcileAction::Started | ReconcileAction::Installed
|
||||||
) {
|
) {
|
||||||
continue;
|
continue;
|
||||||
}
|
}
|
||||||
|
// A successful systemctl start can be a no-op after a transient
|
||||||
|
// Podman inspect failure. Require a witnessed lifecycle change.
|
||||||
|
if !changed_backends.contains(backend) {
|
||||||
|
continue;
|
||||||
|
}
|
||||||
for dep in crate::app_ops::address_caching_dependents(backend) {
|
for dep in crate::app_ops::address_caching_dependents(backend) {
|
||||||
let dep_untouched = report
|
let dep_untouched = report
|
||||||
.actions
|
.actions
|
||||||
@@ -1196,6 +1202,25 @@ fn cascade_pairs_for_report<'r>(
|
|||||||
pairs
|
pairs
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Only positive runtime evidence permits disrupting an address-caching wallet.
|
||||||
|
/// A known absent/stopped backend becoming running, a new container ID, or a
|
||||||
|
/// changed start timestamp qualifies. A failed observation never does.
|
||||||
|
fn backend_instance_changed(before: Option<&ContainerStatus>, after: &ContainerStatus) -> bool {
|
||||||
|
if after.state != ContainerState::Running || after.id.is_empty() {
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
let Some(before) = before else {
|
||||||
|
return true;
|
||||||
|
};
|
||||||
|
if before.id.is_empty() {
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
if before.id != after.id || before.state != ContainerState::Running {
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
matches!((&before.started_at, &after.started_at), (Some(a), Some(b)) if !a.is_empty() && !b.is_empty() && a != b)
|
||||||
|
}
|
||||||
|
|
||||||
#[derive(Debug, Default)]
|
#[derive(Debug, Default)]
|
||||||
pub struct AdoptionReport {
|
pub struct AdoptionReport {
|
||||||
pub adopted: Vec<String>,
|
pub adopted: Vec<String>,
|
||||||
@@ -1909,12 +1934,33 @@ impl ProdContainerOrchestrator {
|
|||||||
_ => 2,
|
_ => 2,
|
||||||
});
|
});
|
||||||
// Live container names (any state), for the same recovery check.
|
// Live container names (any state), for the same recovery check.
|
||||||
let present_containers: std::collections::HashSet<String> = self
|
let listed_containers = self.runtime.list_containers().await.ok();
|
||||||
.runtime
|
let present_containers: HashSet<String> = listed_containers
|
||||||
.list_containers()
|
.as_ref()
|
||||||
.await
|
.map(|cs| cs.iter().map(|c| c.name.clone()).collect())
|
||||||
.map(|cs| cs.into_iter().map(|c| c.name).collect())
|
|
||||||
.unwrap_or_default();
|
.unwrap_or_default();
|
||||||
|
// Keep unknown distinct from confirmed absence. Runtime queries can
|
||||||
|
// fail under load while systemd still has a healthy running backend.
|
||||||
|
let mut backend_before: HashMap<String, Option<ContainerStatus>> = HashMap::new();
|
||||||
|
for lm in &manifests {
|
||||||
|
let id = &lm.manifest.app.id;
|
||||||
|
if crate::app_ops::address_caching_dependents(id).is_empty() {
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
let name = compute_container_name(&lm.manifest);
|
||||||
|
match self.runtime.get_container_status(&name).await {
|
||||||
|
Ok(status) => {
|
||||||
|
backend_before.insert(id.clone(), Some(status));
|
||||||
|
}
|
||||||
|
Err(_) if listed_containers.is_some() && !present_containers.contains(&name) => {
|
||||||
|
backend_before.insert(id.clone(), None);
|
||||||
|
}
|
||||||
|
Err(err) => {
|
||||||
|
tracing::warn!(backend = %id, error = %err,
|
||||||
|
"cannot observe backend before reconcile; will not infer a dependency restart from an action report");
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
let mut report = ReconcileReport::default();
|
let mut report = ReconcileReport::default();
|
||||||
let disk_gb = self.disk_gb().await;
|
let disk_gb = self.disk_gb().await;
|
||||||
let bitcoin_pruned = disk_gb < ARCHIVAL_BITCOIN_DISK_GB
|
let bitcoin_pruned = disk_gb < ARCHIVAL_BITCOIN_DISK_GB
|
||||||
@@ -2096,7 +2142,20 @@ impl ProdContainerOrchestrator {
|
|||||||
// state recovery, repair recreate, boot InstallMissing) moves the
|
// state recovery, repair recreate, boot InstallMissing) moves the
|
||||||
// address behind a running dependent's back — §C "restart lnd after
|
// address behind a running dependent's back — §C "restart lnd after
|
||||||
// ANY bitcoin recreate".
|
// ANY bitcoin recreate".
|
||||||
for (backend, dep) in cascade_pairs_for_report(&report, &user_stopped) {
|
let mut changed_backends = HashSet::new();
|
||||||
|
for (backend, before) in &backend_before {
|
||||||
|
let Some(name) = container_name_by_app_id.get(backend) else {
|
||||||
|
continue;
|
||||||
|
};
|
||||||
|
if let Ok(after) = self.runtime.get_container_status(name).await {
|
||||||
|
if backend_instance_changed(before.as_ref(), &after) {
|
||||||
|
changed_backends.insert(backend.clone());
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
// A user stop during a slow reconcile pass still takes precedence.
|
||||||
|
let user_stopped = crate::crash_recovery::load_user_stopped(&self.data_dir).await;
|
||||||
|
for (backend, dep) in cascade_pairs_for_report(&report, &user_stopped, &changed_backends) {
|
||||||
// Same rule as the RPC cascade: hold the dependent's op lock
|
// Same rule as the RPC cascade: hold the dependent's op lock
|
||||||
// across the restart; skip when a worker is mid-sequence.
|
// across the restart; skip when a worker is mid-sequence.
|
||||||
let lock = crate::app_ops::op_lock(dep);
|
let lock = crate::app_ops::op_lock(dep);
|
||||||
@@ -6409,6 +6468,67 @@ app:
|
|||||||
);
|
);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn backend_cascade_requires_observed_instance_change() {
|
||||||
|
let running = ContainerStatus {
|
||||||
|
id: "container-1".into(),
|
||||||
|
name: "bitcoin-core".into(),
|
||||||
|
state: ContainerState::Running,
|
||||||
|
started_at: Some("start-1".into()),
|
||||||
|
health: None,
|
||||||
|
exit_code: None,
|
||||||
|
image: "bitcoin:1".into(),
|
||||||
|
created: "created-1".into(),
|
||||||
|
ports: vec![],
|
||||||
|
lan_address: None,
|
||||||
|
};
|
||||||
|
assert!(!backend_instance_changed(Some(&running), &running));
|
||||||
|
assert!(backend_instance_changed(None, &running));
|
||||||
|
let mut before = running.clone();
|
||||||
|
before.state = ContainerState::Exited;
|
||||||
|
assert!(backend_instance_changed(Some(&before), &running));
|
||||||
|
before = running.clone();
|
||||||
|
before.id = "old-container".into();
|
||||||
|
assert!(backend_instance_changed(Some(&before), &running));
|
||||||
|
before = running.clone();
|
||||||
|
before.started_at = Some("earlier-start".into());
|
||||||
|
assert!(backend_instance_changed(Some(&before), &running));
|
||||||
|
before.started_at = None;
|
||||||
|
assert!(!backend_instance_changed(Some(&before), &running));
|
||||||
|
before.id.clear();
|
||||||
|
assert!(!backend_instance_changed(Some(&before), &running));
|
||||||
|
let mut after = running.clone();
|
||||||
|
after.state = ContainerState::Exited;
|
||||||
|
assert!(!backend_instance_changed(None, &after));
|
||||||
|
after = running.clone();
|
||||||
|
after.id.clear();
|
||||||
|
assert!(!backend_instance_changed(None, &after));
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn cascade_ignores_false_started_report_but_detects_real_exec_drift() {
|
||||||
|
let none = HashSet::new();
|
||||||
|
let mut report = ReconcileReport {
|
||||||
|
actions: vec![
|
||||||
|
("bitcoin-core".into(), ReconcileAction::Started),
|
||||||
|
("lnd".into(), ReconcileAction::NoOp),
|
||||||
|
],
|
||||||
|
failures: vec![],
|
||||||
|
};
|
||||||
|
// systemctl start of an already active unit does not move its address.
|
||||||
|
assert!(cascade_pairs_for_report(&report, &none, &none).is_empty());
|
||||||
|
// A unit exec rewrite can restart Bitcoin while the outer reconcile
|
||||||
|
// action remains NoOp. Runtime evidence still requires LND to reconnect.
|
||||||
|
let changed = ["bitcoin-core".into()].into();
|
||||||
|
report.actions[0].1 = ReconcileAction::NoOp;
|
||||||
|
assert_eq!(
|
||||||
|
cascade_pairs_for_report(&report, &none, &changed),
|
||||||
|
vec![("bitcoin-core", "lnd")]
|
||||||
|
);
|
||||||
|
report.actions[0].1 = ReconcileAction::Left("lifecycle-op-in-flight".into());
|
||||||
|
assert!(cascade_pairs_for_report(&report, &none, &changed).is_empty());
|
||||||
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
fn cascade_pairs_cover_backend_recreate_with_running_dependent() {
|
fn cascade_pairs_cover_backend_recreate_with_running_dependent() {
|
||||||
use std::collections::HashSet;
|
use std::collections::HashSet;
|
||||||
@@ -6420,6 +6540,7 @@ app:
|
|||||||
failures: vec![],
|
failures: vec![],
|
||||||
};
|
};
|
||||||
let none = HashSet::new();
|
let none = HashSet::new();
|
||||||
|
let changed: HashSet<String> = ["bitcoin-core".into(), "bitcoin-knots".into()].into();
|
||||||
|
|
||||||
// Backend recreated while lnd sat running (NoOp) → cascade.
|
// Backend recreated while lnd sat running (NoOp) → cascade.
|
||||||
let r = report(vec![
|
let r = report(vec![
|
||||||
@@ -6427,7 +6548,7 @@ app:
|
|||||||
("lnd", ReconcileAction::NoOp),
|
("lnd", ReconcileAction::NoOp),
|
||||||
]);
|
]);
|
||||||
assert_eq!(
|
assert_eq!(
|
||||||
cascade_pairs_for_report(&r, &none),
|
cascade_pairs_for_report(&r, &none, &changed),
|
||||||
vec![("bitcoin-knots", "lnd")]
|
vec![("bitcoin-knots", "lnd")]
|
||||||
);
|
);
|
||||||
|
|
||||||
@@ -6437,7 +6558,7 @@ app:
|
|||||||
("lnd", ReconcileAction::NoOp),
|
("lnd", ReconcileAction::NoOp),
|
||||||
]);
|
]);
|
||||||
assert_eq!(
|
assert_eq!(
|
||||||
cascade_pairs_for_report(&r, &none),
|
cascade_pairs_for_report(&r, &none, &changed),
|
||||||
vec![("bitcoin-core", "lnd")]
|
vec![("bitcoin-core", "lnd")]
|
||||||
);
|
);
|
||||||
|
|
||||||
@@ -6446,7 +6567,7 @@ app:
|
|||||||
("bitcoin-knots", ReconcileAction::NoOp),
|
("bitcoin-knots", ReconcileAction::NoOp),
|
||||||
("lnd", ReconcileAction::NoOp),
|
("lnd", ReconcileAction::NoOp),
|
||||||
]);
|
]);
|
||||||
assert!(cascade_pairs_for_report(&r, &none).is_empty());
|
assert!(cascade_pairs_for_report(&r, &none, &none).is_empty());
|
||||||
|
|
||||||
// Dependent itself (re)started this pass → it already resolved the
|
// Dependent itself (re)started this pass → it already resolved the
|
||||||
// fresh address; no cascade.
|
// fresh address; no cascade.
|
||||||
@@ -6454,7 +6575,7 @@ app:
|
|||||||
("bitcoin-knots", ReconcileAction::Installed),
|
("bitcoin-knots", ReconcileAction::Installed),
|
||||||
("lnd", ReconcileAction::Started),
|
("lnd", ReconcileAction::Started),
|
||||||
]);
|
]);
|
||||||
assert!(cascade_pairs_for_report(&r, &none).is_empty());
|
assert!(cascade_pairs_for_report(&r, &none, &changed).is_empty());
|
||||||
|
|
||||||
// User-stopped dependent is never bounced.
|
// User-stopped dependent is never bounced.
|
||||||
let r = report(vec![
|
let r = report(vec![
|
||||||
@@ -6462,14 +6583,14 @@ app:
|
|||||||
("lnd", ReconcileAction::NoOp),
|
("lnd", ReconcileAction::NoOp),
|
||||||
]);
|
]);
|
||||||
let stopped: HashSet<String> = ["lnd".to_string()].into();
|
let stopped: HashSet<String> = ["lnd".to_string()].into();
|
||||||
assert!(cascade_pairs_for_report(&r, &stopped).is_empty());
|
assert!(cascade_pairs_for_report(&r, &stopped, &changed).is_empty());
|
||||||
|
|
||||||
// Non-backend recreates don't cascade anything.
|
// Non-backend recreates don't cascade anything.
|
||||||
let r = report(vec![
|
let r = report(vec![
|
||||||
("grafana", ReconcileAction::Installed),
|
("grafana", ReconcileAction::Installed),
|
||||||
("lnd", ReconcileAction::NoOp),
|
("lnd", ReconcileAction::NoOp),
|
||||||
]);
|
]);
|
||||||
assert!(cascade_pairs_for_report(&r, &none).is_empty());
|
assert!(cascade_pairs_for_report(&r, &none, &changed).is_empty());
|
||||||
}
|
}
|
||||||
|
|
||||||
#[tokio::test]
|
#[tokio::test]
|
||||||
|
|||||||
@@ -286,3 +286,24 @@ unlocked at 08:47 UTC. The existing backend-address cascade then performed a
|
|||||||
graceful LND restart at 08:57 UTC after Bitcoin reconciliation completed; LND
|
graceful LND restart at 08:57 UTC after Bitcoin reconciliation completed; LND
|
||||||
automatically unlocked again and reached chain-sync waiting. No manual wallet
|
automatically unlocked again and reached chain-sync waiting. No manual wallet
|
||||||
unlock or restart was used for this recovery.
|
unlock or restart was used for this recovery.
|
||||||
|
|
||||||
|
### False dependency restart exposed during monitoring — 09:08 UTC
|
||||||
|
|
||||||
|
The initial 1.8.21 candidate passed all 1,557 isolated backend tests and 1,120
|
||||||
|
frontend tests. Monitoring nevertheless found another managed LND restart at
|
||||||
|
09:08:32 while Bitcoin's container/start timestamp remained unchanged. Management
|
||||||
|
logs explicitly attribute it to the backend-address cascade. This also makes
|
||||||
|
the earlier 08:57 cascade suspect; it must not be described as a proven necessary
|
||||||
|
restart. These service restarts preceded the isolated test executable, whose
|
||||||
|
namespace boundaries remain intact.
|
||||||
|
|
||||||
|
The cascade trusted Started/Installed action reports. A failed runtime inspection
|
||||||
|
followed by successful systemctl start of an already active unit can produce
|
||||||
|
Started without changing Bitcoin. Dependency restarts now require observed
|
||||||
|
container-ID, running-state, or start-time changes. Failed observations remain
|
||||||
|
unknown, not absence; a known absent backend becoming running still qualifies.
|
||||||
|
Actual exec-drift restarts are recognized even when their outer report is NoOp.
|
||||||
|
Stopped/lifecycle-in-flight dependents remain excluded, and user stop markers
|
||||||
|
are re-read after the potentially slow pass. Added runtime-observation and
|
||||||
|
false-action/real-exec-drift regression cases; full isolated rerun pending.
|
||||||
|
Stopped the first optimized build and preparing new artifacts from this correction.
|
||||||
|
|||||||
@@ -371,6 +371,7 @@ init()
|
|||||||
<div class="space-y-3 text-sm text-white/80 pl-3 border-l border-white/10">
|
<div class="space-y-3 text-sm text-white/80 pl-3 border-l border-white/10">
|
||||||
<p>Fixed Bitcoin and other containers being forcibly stopped after ten seconds during managed updates and restarts.</p>
|
<p>Fixed Bitcoin and other containers being forcibly stopped after ten seconds during managed updates and restarts.</p>
|
||||||
<p>Existing installations now receive the same graceful shutdown allowance as new containers, without restarting apps just to apply this setting.</p>
|
<p>Existing installations now receive the same graceful shutdown allowance as new containers, without restarting apps just to apply this setting.</p>
|
||||||
|
<p>Prevented unnecessary Lightning restarts when Bitcoin has stayed running; dependency restarts now require an observed Bitcoin container change.</p>
|
||||||
<p>Includes the Cashu payment, optional Bitcoin pruning, Lightning readiness, and explorer improvements from 1.8.20.</p>
|
<p>Includes the Cashu payment, optional Bitcoin pruning, Lightning readiness, and explorer improvements from 1.8.20.</p>
|
||||||
</div>
|
</div>
|
||||||
</div>
|
</div>
|
||||||
|
|||||||
Reference in New Issue
Block a user