diff --git a/core/archipelago/src/container/update_transaction.rs b/core/archipelago/src/container/update_transaction.rs index 418c8aea..85438fda 100644 --- a/core/archipelago/src/container/update_transaction.rs +++ b/core/archipelago/src/container/update_transaction.rs @@ -685,23 +685,100 @@ impl Runtime for Podman { Ok(()) } async fn healthy(&self, id: &str) -> Result { - // Missing healthcheck is not fabricated health. Running-only containers - // can pass runtime readiness; healthchecked members must report healthy. - for _ in 0..30 { + wait_for_container_health(|| async { let raw = Self::command(&["inspect", id]).await?; let rows: Vec = serde_json::from_str(&raw)?; - let row = rows.first().context("Missing updated container")?; - if row.pointer("/State/Status").and_then(|v| v.as_str()) != Some("running") { - return Ok(false); + rows.into_iter().next().context("Missing updated container") + }) + .await + } +} + +// A healthcheck may not run until its interval elapses. The former 30 samples +// could reject a service immediately before its first successful scheduled check. +const HEALTH_MAX_WAIT: std::time::Duration = std::time::Duration::from_secs(300); +const HEALTH_POLL: std::time::Duration = std::time::Duration::from_secs(1); + +fn health_wait_budget(row: &serde_json::Value) -> std::time::Duration { + let config = row.pointer("/Config/Healthcheck"); + let positive = |field: &str, default: u64| { + config + .and_then(|v| v.get(field)) + .and_then(|v| v.as_u64()) + .filter(|v| *v > 0) + .unwrap_or(default) as u128 + }; + // Zero/missing interval and timeout use conservative 30-second defaults; + // zero/missing retries uses three. Never let malformed/huge metadata extend + // the global ceiling, and never extend the deadline on subsequent polls. + let interval = positive("Interval", 30_000_000_000); + let timeout = positive("Timeout", 30_000_000_000); + let retries = positive("Retries", 3); + let start = config + .and_then(|v| v.get("StartPeriod")) + .and_then(|v| v.as_u64()) + .unwrap_or(0) as u128; + let nanos = start + .saturating_add(retries.saturating_mul(interval.saturating_add(timeout))) + .saturating_add(HEALTH_POLL.as_nanos()) + .clamp(30_000_000_000, HEALTH_MAX_WAIT.as_nanos()); + std::time::Duration::from_nanos(nanos as u64) +} + +fn container_health(row: &serde_json::Value) -> Option { + if row.pointer("/State/Status").and_then(|v| v.as_str()) != Some("running") { + return Some(false); + } + match row.pointer("/State/Health/Status").and_then(|v| v.as_str()) { + Some("healthy") => Some(true), + Some("unhealthy") => Some(false), + None | Some("") => { + let configured = match row.pointer("/Config/Healthcheck") { + None | Some(serde_json::Value::Null) => false, + Some(config) => match config.get("Test").and_then(|v| v.as_array()) { + Some(test) => { + !test.is_empty() && test.first().and_then(|v| v.as_str()) != Some("NONE") + } + None => true, // malformed/incomplete check config is not proof of no check + }, + }; + if configured { + None + } else { + Some(true) } - match row.pointer("/State/Health/Status").and_then(|v| v.as_str()) { - None | Some("") | Some("healthy") => return Ok(true), - Some("unhealthy") => return Ok(false), - _ => {} - } - tokio::time::sleep(std::time::Duration::from_secs(1)).await; } - Ok(false) + _ => None, + } +} + +async fn wait_for_container_health(mut inspect: F) -> Result +where + F: FnMut() -> Fut, + Fut: Future>, +{ + let started = tokio::time::Instant::now(); + let mut row = match tokio::time::timeout_at(started + HEALTH_MAX_WAIT, inspect()).await { + Ok(result) => result?, + Err(_) => return Ok(false), + }; + let deadline = started + health_wait_budget(&row); + loop { + if tokio::time::Instant::now() > deadline { + return Ok(false); + } + if let Some(healthy) = container_health(&row) { + return Ok(healthy); + } + let now = tokio::time::Instant::now(); + if now >= deadline { + return Ok(false); + } + tokio::time::sleep_until((now + HEALTH_POLL).min(deadline)).await; + row = match tokio::time::timeout_at(deadline, inspect()).await { + Ok(result) => result?, + Err(_) => return Ok(false), + }; } } @@ -1136,3 +1213,128 @@ mod tests { assert!(Guard::acquire(root.path()).is_ok()); } } + +#[cfg(test)] +mod health_readiness_tests { + use super::*; + use serde_json::json; + use std::time::Duration; + + fn checked(status: &str) -> serde_json::Value { + json!({"State":{"Status":"running","Health":{"Status":status}}, + "Config":{"Healthcheck":{"Test":["CMD","probe"],"Interval":30_000_000_000u64, + "Timeout":5_000_000_000u64,"Retries":5}}}) + } + + #[tokio::test(start_paused = true)] + async fn scheduled_health_at_31_seconds_is_not_rejected_at_30() { + let started = tokio::time::Instant::now(); + let result = wait_for_container_health(|| async { + Ok(checked(if started.elapsed() >= Duration::from_secs(31) { + "healthy" + } else { + "starting" + })) + }) + .await + .unwrap(); + assert!(result); + assert_eq!(started.elapsed(), Duration::from_secs(31)); + } + + #[tokio::test(start_paused = true)] + async fn unhealthy_and_exited_fail_immediately() { + for row in [checked("unhealthy"), json!({"State":{"Status":"exited"}})] { + let started = tokio::time::Instant::now(); + assert!( + !wait_for_container_health(|| std::future::ready(Ok(row.clone()))) + .await + .unwrap() + ); + assert_eq!(started.elapsed(), Duration::ZERO); + } + } + + #[tokio::test(start_paused = true)] + async fn starting_is_bounded_by_initial_config_even_if_later_config_grows() { + let started = tokio::time::Instant::now(); + let mut first = true; + assert!(!wait_for_container_health(|| { + let mut row = checked("starting"); + if first { + first = false; + } else { + row["Config"]["Healthcheck"]["StartPeriod"] = json!(u64::MAX); + } + std::future::ready(Ok(row)) + }) + .await + .unwrap()); + assert_eq!(started.elapsed(), Duration::from_secs(176)); + } + + #[tokio::test(start_paused = true)] + async fn blocked_initial_inspect_cannot_exceed_global_ceiling() { + let started = tokio::time::Instant::now(); + assert!(!wait_for_container_health(|| std::future::pending()) + .await + .unwrap()); + assert_eq!(started.elapsed(), HEALTH_MAX_WAIT); + } + + #[tokio::test(start_paused = true)] + async fn blocked_later_inspect_cannot_exceed_initial_deadline() { + let started = tokio::time::Instant::now(); + let mut calls = 0; + assert!(!wait_for_container_health(|| { + calls += 1; + let call = calls; + async move { + if call == 1 { + Ok(checked("starting")) + } else { + std::future::pending().await + } + } + }) + .await + .unwrap()); + assert_eq!(started.elapsed(), Duration::from_secs(176)); + } + + #[tokio::test(start_paused = true)] + async fn missing_status_is_not_success_for_configured_healthcheck() { + let mut row = checked("starting"); + row["State"].as_object_mut().unwrap().remove("Health"); + let started = tokio::time::Instant::now(); + assert!( + !wait_for_container_health(|| std::future::ready(Ok(row.clone()))) + .await + .unwrap() + ); + assert_eq!(started.elapsed(), Duration::from_secs(176)); + assert!(wait_for_container_health(|| std::future::ready(Ok( + json!({"State":{"Status":"running"}}) + ))) + .await + .unwrap()); + } + + #[test] + fn health_budget_defaults_start_period_and_extreme_values_are_bounded() { + assert_eq!( + health_wait_budget(&checked("starting")), + Duration::from_secs(176) + ); + let mut row = checked("starting"); + row["Config"]["Healthcheck"]["StartPeriod"] = json!(20_000_000_000u64); + assert_eq!(health_wait_budget(&row), Duration::from_secs(196)); + for value in [json!(0), json!("bad"), json!(-1)] { + row["Config"]["Healthcheck"] = + json!({"Interval":value,"Timeout":value,"Retries":value}); + assert_eq!(health_wait_budget(&row), Duration::from_secs(181)); + } + row["Config"]["Healthcheck"] = json!({"StartPeriod":u64::MAX,"Interval":u64::MAX,"Timeout":u64::MAX,"Retries":u64::MAX}); + assert_eq!(health_wait_budget(&row), HEALTH_MAX_WAIT); + } +} diff --git a/docs/post-1.8.22-regressions-20261001.md b/docs/post-1.8.22-regressions-20261001.md index 066c0d5e..36d21ca0 100644 --- a/docs/post-1.8.22-regressions-20261001.md +++ b/docs/post-1.8.22-regressions-20261001.md @@ -1558,3 +1558,43 @@ announcements remain unrecovered from the sources checked. This releases the all-project-discovery publication hold, not a claim that recovery passed. Track recovery separately and retain the limitation in release notes. Other artifact, upgrade, security and publication checks remain required. + +## 2026-10-08: final IndeeHub update restored at health-check boundary + +Native operation `851483a2` automatically restored during target startup, before +the final frontend was started. This is **not successful final delivery**. +The MinIO target started at 18:24:31.489 UTC. Its immediate probe was still +`starting`; MinIO reported API readiness at 18:24:32, and the next scheduled +health check reported `healthy` at 18:25:02.851. The updater's previous 30 samples, +one second apart with no final sample after sleeping, could expire before that +healthy result. The actual health configuration uses a 30-second interval, +five-second timeout and five retries. PostgreSQL, Redis and MinIO were the only +target members started; relay, API, worker and final frontend were not started. + +Independent post-restoration verification passed: all 31 running containers, +24 unrelated runtime identities preserved, the exact current 110-migration +history and original database commitments preserved, and restoration-image +proof matched. Retain the operation journal and backup history; do not treat a +restored transaction as a successful update or erase the first failed attempt. + +The shared native provider (`b77…`) is deployed. The final frontend image +(`8db…`) remains pending; paired cached-account-A to chosen-identity-B acceptance +has not run. Prior clean native login success does not prove this reported +cached-account transition is fixed on the deployed frontend. + +Source correction `7fe63355` replaces the sample-count deadline with a bounded +monotonic readiness deadline derived once from the first inspected health +configuration. It includes the configured start period, intervals, timeouts and +retries, with a 30-second minimum and five-minute ceiling; every inspect is +bounded too. `starting` never qualifies as healthy, configured-but-missing health +status stays pending, and unhealthy or exited containers fail immediately. +Running-only readiness remains allowed only for containers without a configured +health check. The isolated runner checkout bind is separately committed as +`1b0b119a` and preserves all existing isolation properties. + +Validation at this checkpoint: source formatting and independent source review +passed; seven deterministic paused-time regressions are compiling through the +isolated runner. The full isolated suite, production backend build, another +explicit native update, post-update runtime/data proofs, and paired browser +acceptance remain pending. No wallet, payment, personal-media or profile changes +are part of this readiness correction. diff --git a/scripts/test-backend-isolated.sh b/scripts/test-backend-isolated.sh index 55d0aac0..c170fda7 100755 --- a/scripts/test-backend-isolated.sh +++ b/scripts/test-backend-isolated.sh @@ -42,6 +42,7 @@ PY unit="archy-isolated-tests-$(date +%s)-$$" sudo -n systemd-run --unit="$unit" --wait --pipe --collect \ --property="WorkingDirectory=$REPO/core" \ + --property="BindReadOnlyPaths=$REPO" \ --property=PrivateNetwork=yes --property=PrivateTmp=yes --property=PrivateDevices=yes \ --property=ProtectSystem=strict --property=ProtectHome=read-only \ --property=NoNewPrivileges=yes \