Merge branch 'pr/fix/native-health-readiness-20261008(6d999c39)' into work/post190-source-acceptance

This commit is contained in:
archipelago
2026-10-08 16:21:45 -04:00
3 changed files with 256 additions and 13 deletions
@@ -685,23 +685,100 @@ impl Runtime for Podman {
Ok(()) Ok(())
} }
async fn healthy(&self, id: &str) -> Result<bool> { async fn healthy(&self, id: &str) -> Result<bool> {
// Missing healthcheck is not fabricated health. Running-only containers wait_for_container_health(|| async {
// can pass runtime readiness; healthchecked members must report healthy.
for _ in 0..30 {
let raw = Self::command(&["inspect", id]).await?; let raw = Self::command(&["inspect", id]).await?;
let rows: Vec<serde_json::Value> = serde_json::from_str(&raw)?; let rows: Vec<serde_json::Value> = serde_json::from_str(&raw)?;
let row = rows.first().context("Missing updated container")?; rows.into_iter().next().context("Missing updated container")
if row.pointer("/State/Status").and_then(|v| v.as_str()) != Some("running") { })
return Ok(false); .await
}
}
// A healthcheck may not run until its interval elapses. The former 30 samples
// could reject a service immediately before its first successful scheduled check.
const HEALTH_MAX_WAIT: std::time::Duration = std::time::Duration::from_secs(300);
const HEALTH_POLL: std::time::Duration = std::time::Duration::from_secs(1);
fn health_wait_budget(row: &serde_json::Value) -> std::time::Duration {
let config = row.pointer("/Config/Healthcheck");
let positive = |field: &str, default: u64| {
config
.and_then(|v| v.get(field))
.and_then(|v| v.as_u64())
.filter(|v| *v > 0)
.unwrap_or(default) as u128
};
// Zero/missing interval and timeout use conservative 30-second defaults;
// zero/missing retries uses three. Never let malformed/huge metadata extend
// the global ceiling, and never extend the deadline on subsequent polls.
let interval = positive("Interval", 30_000_000_000);
let timeout = positive("Timeout", 30_000_000_000);
let retries = positive("Retries", 3);
let start = config
.and_then(|v| v.get("StartPeriod"))
.and_then(|v| v.as_u64())
.unwrap_or(0) as u128;
let nanos = start
.saturating_add(retries.saturating_mul(interval.saturating_add(timeout)))
.saturating_add(HEALTH_POLL.as_nanos())
.clamp(30_000_000_000, HEALTH_MAX_WAIT.as_nanos());
std::time::Duration::from_nanos(nanos as u64)
}
fn container_health(row: &serde_json::Value) -> Option<bool> {
if row.pointer("/State/Status").and_then(|v| v.as_str()) != Some("running") {
return Some(false);
}
match row.pointer("/State/Health/Status").and_then(|v| v.as_str()) {
Some("healthy") => Some(true),
Some("unhealthy") => Some(false),
None | Some("") => {
let configured = match row.pointer("/Config/Healthcheck") {
None | Some(serde_json::Value::Null) => false,
Some(config) => match config.get("Test").and_then(|v| v.as_array()) {
Some(test) => {
!test.is_empty() && test.first().and_then(|v| v.as_str()) != Some("NONE")
}
None => true, // malformed/incomplete check config is not proof of no check
},
};
if configured {
None
} else {
Some(true)
} }
match row.pointer("/State/Health/Status").and_then(|v| v.as_str()) {
None | Some("") | Some("healthy") => return Ok(true),
Some("unhealthy") => return Ok(false),
_ => {}
}
tokio::time::sleep(std::time::Duration::from_secs(1)).await;
} }
Ok(false) _ => None,
}
}
async fn wait_for_container_health<F, Fut>(mut inspect: F) -> Result<bool>
where
F: FnMut() -> Fut,
Fut: Future<Output = Result<serde_json::Value>>,
{
let started = tokio::time::Instant::now();
let mut row = match tokio::time::timeout_at(started + HEALTH_MAX_WAIT, inspect()).await {
Ok(result) => result?,
Err(_) => return Ok(false),
};
let deadline = started + health_wait_budget(&row);
loop {
if tokio::time::Instant::now() > deadline {
return Ok(false);
}
if let Some(healthy) = container_health(&row) {
return Ok(healthy);
}
let now = tokio::time::Instant::now();
if now >= deadline {
return Ok(false);
}
tokio::time::sleep_until((now + HEALTH_POLL).min(deadline)).await;
row = match tokio::time::timeout_at(deadline, inspect()).await {
Ok(result) => result?,
Err(_) => return Ok(false),
};
} }
} }
@@ -1136,3 +1213,128 @@ mod tests {
assert!(Guard::acquire(root.path()).is_ok()); assert!(Guard::acquire(root.path()).is_ok());
} }
} }
#[cfg(test)]
mod health_readiness_tests {
use super::*;
use serde_json::json;
use std::time::Duration;
fn checked(status: &str) -> serde_json::Value {
json!({"State":{"Status":"running","Health":{"Status":status}},
"Config":{"Healthcheck":{"Test":["CMD","probe"],"Interval":30_000_000_000u64,
"Timeout":5_000_000_000u64,"Retries":5}}})
}
#[tokio::test(start_paused = true)]
async fn scheduled_health_at_31_seconds_is_not_rejected_at_30() {
let started = tokio::time::Instant::now();
let result = wait_for_container_health(|| async {
Ok(checked(if started.elapsed() >= Duration::from_secs(31) {
"healthy"
} else {
"starting"
}))
})
.await
.unwrap();
assert!(result);
assert_eq!(started.elapsed(), Duration::from_secs(31));
}
#[tokio::test(start_paused = true)]
async fn unhealthy_and_exited_fail_immediately() {
for row in [checked("unhealthy"), json!({"State":{"Status":"exited"}})] {
let started = tokio::time::Instant::now();
assert!(
!wait_for_container_health(|| std::future::ready(Ok(row.clone())))
.await
.unwrap()
);
assert_eq!(started.elapsed(), Duration::ZERO);
}
}
#[tokio::test(start_paused = true)]
async fn starting_is_bounded_by_initial_config_even_if_later_config_grows() {
let started = tokio::time::Instant::now();
let mut first = true;
assert!(!wait_for_container_health(|| {
let mut row = checked("starting");
if first {
first = false;
} else {
row["Config"]["Healthcheck"]["StartPeriod"] = json!(u64::MAX);
}
std::future::ready(Ok(row))
})
.await
.unwrap());
assert_eq!(started.elapsed(), Duration::from_secs(176));
}
#[tokio::test(start_paused = true)]
async fn blocked_initial_inspect_cannot_exceed_global_ceiling() {
let started = tokio::time::Instant::now();
assert!(!wait_for_container_health(|| std::future::pending())
.await
.unwrap());
assert_eq!(started.elapsed(), HEALTH_MAX_WAIT);
}
#[tokio::test(start_paused = true)]
async fn blocked_later_inspect_cannot_exceed_initial_deadline() {
let started = tokio::time::Instant::now();
let mut calls = 0;
assert!(!wait_for_container_health(|| {
calls += 1;
let call = calls;
async move {
if call == 1 {
Ok(checked("starting"))
} else {
std::future::pending().await
}
}
})
.await
.unwrap());
assert_eq!(started.elapsed(), Duration::from_secs(176));
}
#[tokio::test(start_paused = true)]
async fn missing_status_is_not_success_for_configured_healthcheck() {
let mut row = checked("starting");
row["State"].as_object_mut().unwrap().remove("Health");
let started = tokio::time::Instant::now();
assert!(
!wait_for_container_health(|| std::future::ready(Ok(row.clone())))
.await
.unwrap()
);
assert_eq!(started.elapsed(), Duration::from_secs(176));
assert!(wait_for_container_health(|| std::future::ready(Ok(
json!({"State":{"Status":"running"}})
)))
.await
.unwrap());
}
#[test]
fn health_budget_defaults_start_period_and_extreme_values_are_bounded() {
assert_eq!(
health_wait_budget(&checked("starting")),
Duration::from_secs(176)
);
let mut row = checked("starting");
row["Config"]["Healthcheck"]["StartPeriod"] = json!(20_000_000_000u64);
assert_eq!(health_wait_budget(&row), Duration::from_secs(196));
for value in [json!(0), json!("bad"), json!(-1)] {
row["Config"]["Healthcheck"] =
json!({"Interval":value,"Timeout":value,"Retries":value});
assert_eq!(health_wait_budget(&row), Duration::from_secs(181));
}
row["Config"]["Healthcheck"] = json!({"StartPeriod":u64::MAX,"Interval":u64::MAX,"Timeout":u64::MAX,"Retries":u64::MAX});
assert_eq!(health_wait_budget(&row), HEALTH_MAX_WAIT);
}
}
+40
View File
@@ -1558,3 +1558,43 @@ announcements remain unrecovered from the sources checked. This releases the
all-project-discovery publication hold, not a claim that recovery passed. Track all-project-discovery publication hold, not a claim that recovery passed. Track
recovery separately and retain the limitation in release notes. Other artifact, recovery separately and retain the limitation in release notes. Other artifact,
upgrade, security and publication checks remain required. upgrade, security and publication checks remain required.
## 2026-10-08: final IndeeHub update restored at health-check boundary
Native operation `851483a2` automatically restored during target startup, before
the final frontend was started. This is **not successful final delivery**.
The MinIO target started at 18:24:31.489 UTC. Its immediate probe was still
`starting`; MinIO reported API readiness at 18:24:32, and the next scheduled
health check reported `healthy` at 18:25:02.851. The updater's previous 30 samples,
one second apart with no final sample after sleeping, could expire before that
healthy result. The actual health configuration uses a 30-second interval,
five-second timeout and five retries. PostgreSQL, Redis and MinIO were the only
target members started; relay, API, worker and final frontend were not started.
Independent post-restoration verification passed: all 31 running containers,
24 unrelated runtime identities preserved, the exact current 110-migration
history and original database commitments preserved, and restoration-image
proof matched. Retain the operation journal and backup history; do not treat a
restored transaction as a successful update or erase the first failed attempt.
The shared native provider (`b77…`) is deployed. The final frontend image
(`8db…`) remains pending; paired cached-account-A to chosen-identity-B acceptance
has not run. Prior clean native login success does not prove this reported
cached-account transition is fixed on the deployed frontend.
Source correction `7fe63355` replaces the sample-count deadline with a bounded
monotonic readiness deadline derived once from the first inspected health
configuration. It includes the configured start period, intervals, timeouts and
retries, with a 30-second minimum and five-minute ceiling; every inspect is
bounded too. `starting` never qualifies as healthy, configured-but-missing health
status stays pending, and unhealthy or exited containers fail immediately.
Running-only readiness remains allowed only for containers without a configured
health check. The isolated runner checkout bind is separately committed as
`1b0b119a` and preserves all existing isolation properties.
Validation at this checkpoint: source formatting and independent source review
passed; seven deterministic paused-time regressions are compiling through the
isolated runner. The full isolated suite, production backend build, another
explicit native update, post-update runtime/data proofs, and paired browser
acceptance remain pending. No wallet, payment, personal-media or profile changes
are part of this readiness correction.
+1
View File
@@ -42,6 +42,7 @@ PY
unit="archy-isolated-tests-$(date +%s)-$$" unit="archy-isolated-tests-$(date +%s)-$$"
sudo -n systemd-run --unit="$unit" --wait --pipe --collect \ sudo -n systemd-run --unit="$unit" --wait --pipe --collect \
--property="WorkingDirectory=$REPO/core" \ --property="WorkingDirectory=$REPO/core" \
--property="BindReadOnlyPaths=$REPO" \
--property=PrivateNetwork=yes --property=PrivateTmp=yes --property=PrivateDevices=yes \ --property=PrivateNetwork=yes --property=PrivateTmp=yes --property=PrivateDevices=yes \
--property=ProtectSystem=strict --property=ProtectHome=read-only \ --property=ProtectSystem=strict --property=ProtectHome=read-only \
--property=NoNewPrivileges=yes \ --property=NoNewPrivileges=yes \