fix(core): throttle volume-ownership sweep to first-seen + hourly per container
The sweep podman-execs into every running container; doing that on each 30s reconcile tick was a permanent conmon 'Failed to create container' storm on hosts where exec from the backend's cgroup context fails (Debian 13 first boot). Ownership drift is an install/OTA-time event — sweep on first pass after a container appears, then at most hourly. Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
This commit is contained in:
parent
e5f8b5d789
commit
cf4a8eef0e
@ -301,6 +301,33 @@ fn unrepairable_ownership() -> &'static std::sync::Mutex<std::collections::HashS
|
||||
SET.get_or_init(|| std::sync::Mutex::new(std::collections::HashSet::new()))
|
||||
}
|
||||
|
||||
/// Per-container timestamp of the last volume-ownership sweep. The sweep's
|
||||
/// write-probes are `podman exec`s into EVERY running container; running them
|
||||
/// on every 30s reconcile tick meant six-plus cross-context exec attempts per
|
||||
/// tick forever — a permanent conmon "Failed to create container" storm on
|
||||
/// hosts where exec from the backend's cgroup context fails (Debian 13 first
|
||||
/// boot, 2026-07-19). Ownership drift is an install/OTA-time event, not a
|
||||
/// steady-state one: sweep each container on the first pass after it appears,
|
||||
/// then at most once per hour.
|
||||
fn ownership_sweep_due(name: &str) -> bool {
|
||||
const SWEEP_INTERVAL: std::time::Duration = std::time::Duration::from_secs(60 * 60);
|
||||
static LAST: std::sync::OnceLock<
|
||||
std::sync::Mutex<std::collections::HashMap<String, std::time::Instant>>,
|
||||
> = std::sync::OnceLock::new();
|
||||
let map = LAST.get_or_init(|| std::sync::Mutex::new(std::collections::HashMap::new()));
|
||||
let Ok(mut map) = map.lock() else {
|
||||
return true;
|
||||
};
|
||||
let now = std::time::Instant::now();
|
||||
match map.get(name) {
|
||||
Some(last) if now.duration_since(*last) < SWEEP_INTERVAL => false,
|
||||
_ => {
|
||||
map.insert(name.to_string(), now);
|
||||
true
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// App-agnostic, userns-mapping-proof volume-ownership repair for a RUNNING
|
||||
/// container.
|
||||
///
|
||||
@ -1739,6 +1766,11 @@ impl ProdContainerOrchestrator {
|
||||
if crate::app_ops::lifecycle_op_in_flight(&c.name) {
|
||||
continue;
|
||||
}
|
||||
// Throttled: first pass after the container appears, then
|
||||
// hourly — not on every 30s tick (see ownership_sweep_due).
|
||||
if !ownership_sweep_due(&c.name) {
|
||||
continue;
|
||||
}
|
||||
if ensure_running_container_ownership(&c.name).await {
|
||||
tracing::info!(container = %c.name, "volume ownership repaired during reconcile — restarting to recover");
|
||||
let _ = tokio::process::Command::new("podman")
|
||||
|
||||
Loading…
x
Reference in New Issue
Block a user