From 1a409900d985a89acd8e6a13b00d7e7f1e8d4d7d Mon Sep 17 00:00:00 2001 From: archipelago Date: Thu, 8 Oct 2026 04:22:25 -0400 Subject: [PATCH] Bound fresh restore database initialization by observed startup latency --- .../managed-update-recovery-implementation.md | 27 +++++++++++++++++++ scripts/indeehub-maintenance-controller.py | 5 +++- .../test_indeehub_maintenance_controller.py | 22 +++++++++++++-- 3 files changed, 51 insertions(+), 3 deletions(-) diff --git a/docs/managed-update-recovery-implementation.md b/docs/managed-update-recovery-implementation.md index 28627d77..94a04192 100644 --- a/docs/managed-update-recovery-implementation.md +++ b/docs/managed-update-recovery-implementation.md @@ -559,3 +559,30 @@ The same guest is QMP-paused while the separately frozen worker image builds. Before another transaction, bounded read-only API-module/Redis-PING and PostgreSQL probes will record latency and unchanged container identities; production deadlines and transaction gates remain unchanged. + +### Fresh recovery-image startup budget (2026-10-08) + +Actual operation `6432d320-307f-49f6-ae93-3599a066f97a` drained all seven +members and captured backup, then safely recovered before target startup when +its fresh restore database exceeded the 90-second readiness deadline. The +separate, networkless exact-image diagnostic with a 300-second observation +window completed before the daemon restart: final TCP readiness succeeded at +approximately 104.2 seconds (105.777 seconds including final evidence capture). +The guest retained its boot identity and all seven service runtimes were running; +the diagnostic container was removed. Its sole FATAL log line was "the database +system is shutting down" during the normal bootstrap-server transition, followed +by final-server readiness. No OOM, PANIC, permission or initdb error was recorded. +Host I/O/memory pressure and repeated bounded Podman probe timeouts were present. +Private receipt remains in the guest's +`indeehub-fixture/archy-pg-ready-diagnostic-c8939db7f6f74554934bf2be27efbdb5/receipt.json`. + +Only the disposable backup-restore PostgreSQL initialization deadline is now +180 seconds. Every readiness attempt remains capped at ten seconds or remaining +budget, success after the overall deadline is rejected, and cleanup remains in +`finally`. The database restore itself is never retried. Regressions cover the +observed 104-second successful initialization, expiration after 180 seconds, +late success refusal, the remaining-budget probe cap, and a failed restore being +executed exactly once. This source change still requires matching embedded-helper +build and actual transaction acceptance; it does not close cutover or rollback. +The current guest was QMP-paused without reboot to serialize worker runtime and +backend process-fixture qualification. diff --git a/scripts/indeehub-maintenance-controller.py b/scripts/indeehub-maintenance-controller.py index 304ad875..2f0b9e1f 100644 --- a/scripts/indeehub-maintenance-controller.py +++ b/scripts/indeehub-maintenance-controller.py @@ -720,7 +720,10 @@ class Controller: self.run(['podman','start',identifier]) # The image bootstrap server accepts Unix sockets before it exits; # TCP readiness waits for the final server, avoiding interrupted restore. - deadline=time.monotonic()+90 + # Exact recovery-image qualification under host I/O pressure reached + # final TCP readiness at 104 seconds. Give only this fresh initdb + # startup 180 seconds; probes remain bounded and restore is not retried. + deadline=time.monotonic()+180 while True: remaining=deadline-time.monotonic() require(remaining>0,'Backup restore database did not become ready') diff --git a/tests/regression/test_indeehub_maintenance_controller.py b/tests/regression/test_indeehub_maintenance_controller.py index 311d7981..411d5816 100644 --- a/tests/regression/test_indeehub_maintenance_controller.py +++ b/tests/regression/test_indeehub_maintenance_controller.py @@ -625,13 +625,31 @@ console.log('process identity cases passed');''' def test_database_readiness_probe_timeout_never_extends_overall_deadline(self): from unittest.mock import patch c,attempts=self.database_readiness_fixture([module.subprocess.TimeoutExpired(['pg_isready'],10)]) - with patch.object(module.time,'monotonic',side_effect=[0,1,91]),patch.object(module.time,'sleep'): + with patch.object(module.time,'monotonic',side_effect=[0,1,181]),patch.object(module.time,'sleep'): with self.assertRaisesRegex(RuntimeError,'did not become ready'):c.verify_database_backup() self.assertEqual(len(attempts),1);self.assertNotIn('backup_restore_verified',c.record);self.assertNotIn('restore_fixture',c.record) def test_database_readiness_rejects_success_after_deadline(self): from unittest.mock import patch c,attempts=self.database_readiness_fixture([]) - with patch.object(module.time,'monotonic',side_effect=[0,89,91]),patch.object(module.time,'sleep'): + with patch.object(module.time,'monotonic',side_effect=[0,179,181]),patch.object(module.time,'sleep'): with self.assertRaisesRegex(RuntimeError,'did not become ready'):c.verify_database_backup() self.assertEqual(len(attempts),1);self.assertNotIn('backup_restore_verified',c.record);self.assertNotIn('restore_fixture',c.record) + def test_database_readiness_accepts_observed_slow_bootstrap_within_startup_budget(self): + from unittest.mock import patch + c,attempts=self.database_readiness_fixture([module.subprocess.CalledProcessError(2,['pg_isready'])]) + with patch.object(module.time,'monotonic',side_effect=[0,89,90,103,104]),patch.object(module.time,'sleep'): + c.verify_database_backup() + self.assertEqual(len(attempts),2);self.assertEqual(c.record['backup_restore_verified'],c.backup_restore_terms()) + def test_database_readiness_caps_probe_at_remaining_budget_and_never_retries_restore(self): + from unittest.mock import patch + c,attempts=self.database_readiness_fixture([]);original=c.run;timeouts=[];restores=[] + def run(argv,**kwargs): + if argv[:2]==['podman','exec'] and argv[3]=='pg_isready':timeouts.append(kwargs['timeout']) + if argv[:3]==['podman','exec','-i']: + restores.append(argv);raise module.subprocess.TimeoutExpired(['pg_restore'],1800) + return original(argv,**kwargs) + c.run=run + with patch.object(module.time,'monotonic',side_effect=[0,179,179.5]),patch.object(module.time,'sleep'): + with self.assertRaises(module.subprocess.TimeoutExpired):c.verify_database_backup() + self.assertEqual(timeouts,[1]);self.assertEqual(len(restores),1);self.assertNotIn('backup_restore_verified',c.record);self.assertNotIn('restore_fixture',c.record) if __name__=='__main__':unittest.main()