Bound fresh restore database initialization by observed startup latency
This commit is contained in:
@@ -559,3 +559,30 @@ The same guest is QMP-paused while the separately frozen worker image builds.
|
||||
Before another transaction, bounded read-only API-module/Redis-PING and
|
||||
PostgreSQL probes will record latency and unchanged container identities;
|
||||
production deadlines and transaction gates remain unchanged.
|
||||
|
||||
### Fresh recovery-image startup budget (2026-10-08)
|
||||
|
||||
Actual operation `6432d320-307f-49f6-ae93-3599a066f97a` drained all seven
|
||||
members and captured backup, then safely recovered before target startup when
|
||||
its fresh restore database exceeded the 90-second readiness deadline. The
|
||||
separate, networkless exact-image diagnostic with a 300-second observation
|
||||
window completed before the daemon restart: final TCP readiness succeeded at
|
||||
approximately 104.2 seconds (105.777 seconds including final evidence capture).
|
||||
The guest retained its boot identity and all seven service runtimes were running;
|
||||
the diagnostic container was removed. Its sole FATAL log line was "the database
|
||||
system is shutting down" during the normal bootstrap-server transition, followed
|
||||
by final-server readiness. No OOM, PANIC, permission or initdb error was recorded.
|
||||
Host I/O/memory pressure and repeated bounded Podman probe timeouts were present.
|
||||
Private receipt remains in the guest's
|
||||
`indeehub-fixture/archy-pg-ready-diagnostic-c8939db7f6f74554934bf2be27efbdb5/receipt.json`.
|
||||
|
||||
Only the disposable backup-restore PostgreSQL initialization deadline is now
|
||||
180 seconds. Every readiness attempt remains capped at ten seconds or remaining
|
||||
budget, success after the overall deadline is rejected, and cleanup remains in
|
||||
`finally`. The database restore itself is never retried. Regressions cover the
|
||||
observed 104-second successful initialization, expiration after 180 seconds,
|
||||
late success refusal, the remaining-budget probe cap, and a failed restore being
|
||||
executed exactly once. This source change still requires matching embedded-helper
|
||||
build and actual transaction acceptance; it does not close cutover or rollback.
|
||||
The current guest was QMP-paused without reboot to serialize worker runtime and
|
||||
backend process-fixture qualification.
|
||||
|
||||
@@ -720,7 +720,10 @@ class Controller:
|
||||
self.run(['podman','start',identifier])
|
||||
# The image bootstrap server accepts Unix sockets before it exits;
|
||||
# TCP readiness waits for the final server, avoiding interrupted restore.
|
||||
deadline=time.monotonic()+90
|
||||
# Exact recovery-image qualification under host I/O pressure reached
|
||||
# final TCP readiness at 104 seconds. Give only this fresh initdb
|
||||
# startup 180 seconds; probes remain bounded and restore is not retried.
|
||||
deadline=time.monotonic()+180
|
||||
while True:
|
||||
remaining=deadline-time.monotonic()
|
||||
require(remaining>0,'Backup restore database did not become ready')
|
||||
|
||||
@@ -625,13 +625,31 @@ console.log('process identity cases passed');'''
|
||||
def test_database_readiness_probe_timeout_never_extends_overall_deadline(self):
|
||||
from unittest.mock import patch
|
||||
c,attempts=self.database_readiness_fixture([module.subprocess.TimeoutExpired(['pg_isready'],10)])
|
||||
with patch.object(module.time,'monotonic',side_effect=[0,1,91]),patch.object(module.time,'sleep'):
|
||||
with patch.object(module.time,'monotonic',side_effect=[0,1,181]),patch.object(module.time,'sleep'):
|
||||
with self.assertRaisesRegex(RuntimeError,'did not become ready'):c.verify_database_backup()
|
||||
self.assertEqual(len(attempts),1);self.assertNotIn('backup_restore_verified',c.record);self.assertNotIn('restore_fixture',c.record)
|
||||
def test_database_readiness_rejects_success_after_deadline(self):
|
||||
from unittest.mock import patch
|
||||
c,attempts=self.database_readiness_fixture([])
|
||||
with patch.object(module.time,'monotonic',side_effect=[0,89,91]),patch.object(module.time,'sleep'):
|
||||
with patch.object(module.time,'monotonic',side_effect=[0,179,181]),patch.object(module.time,'sleep'):
|
||||
with self.assertRaisesRegex(RuntimeError,'did not become ready'):c.verify_database_backup()
|
||||
self.assertEqual(len(attempts),1);self.assertNotIn('backup_restore_verified',c.record);self.assertNotIn('restore_fixture',c.record)
|
||||
def test_database_readiness_accepts_observed_slow_bootstrap_within_startup_budget(self):
|
||||
from unittest.mock import patch
|
||||
c,attempts=self.database_readiness_fixture([module.subprocess.CalledProcessError(2,['pg_isready'])])
|
||||
with patch.object(module.time,'monotonic',side_effect=[0,89,90,103,104]),patch.object(module.time,'sleep'):
|
||||
c.verify_database_backup()
|
||||
self.assertEqual(len(attempts),2);self.assertEqual(c.record['backup_restore_verified'],c.backup_restore_terms())
|
||||
def test_database_readiness_caps_probe_at_remaining_budget_and_never_retries_restore(self):
|
||||
from unittest.mock import patch
|
||||
c,attempts=self.database_readiness_fixture([]);original=c.run;timeouts=[];restores=[]
|
||||
def run(argv,**kwargs):
|
||||
if argv[:2]==['podman','exec'] and argv[3]=='pg_isready':timeouts.append(kwargs['timeout'])
|
||||
if argv[:3]==['podman','exec','-i']:
|
||||
restores.append(argv);raise module.subprocess.TimeoutExpired(['pg_restore'],1800)
|
||||
return original(argv,**kwargs)
|
||||
c.run=run
|
||||
with patch.object(module.time,'monotonic',side_effect=[0,179,179.5]),patch.object(module.time,'sleep'):
|
||||
with self.assertRaises(module.subprocess.TimeoutExpired):c.verify_database_backup()
|
||||
self.assertEqual(timeouts,[1]);self.assertEqual(len(restores),1);self.assertNotIn('backup_restore_verified',c.record);self.assertNotIn('restore_fixture',c.record)
|
||||
if __name__=='__main__':unittest.main()
|
||||
|
||||
Reference in New Issue
Block a user