Stop looping if all workers have died
If the workers are crashing and the restart limit has been met, we need to stop listening for events and trigger an internal error.
This commit is contained in:
1
changelog/45.bugfix
Normal file
1
changelog/45.bugfix
Normal file
@@ -0,0 +1 @@
|
|||||||
|
Fix hang when all worker nodes crash and restart limit is reached
|
||||||
@@ -701,6 +701,17 @@ class TestNodeFailure:
|
|||||||
"*2 failed*2 passed*",
|
"*2 failed*2 passed*",
|
||||||
])
|
])
|
||||||
|
|
||||||
|
def test_max_slave_restart_die(self, testdir):
|
||||||
|
f = testdir.makepyfile("""
|
||||||
|
import os
|
||||||
|
os._exit(1)
|
||||||
|
""")
|
||||||
|
res = testdir.runpytest(f, '-n4', '--max-slave-restart=0')
|
||||||
|
res.stdout.fnmatch_lines([
|
||||||
|
"*Unexpectedly no active workers*",
|
||||||
|
"*INTERNALERROR*"
|
||||||
|
])
|
||||||
|
|
||||||
def test_disable_restart(self, testdir):
|
def test_disable_restart(self, testdir):
|
||||||
f = testdir.makepyfile("""
|
f = testdir.makepyfile("""
|
||||||
import os
|
import os
|
||||||
|
|||||||
@@ -120,6 +120,10 @@ class DSession:
|
|||||||
def loop_once(self):
|
def loop_once(self):
|
||||||
"""Process one callback from one of the slaves."""
|
"""Process one callback from one of the slaves."""
|
||||||
while 1:
|
while 1:
|
||||||
|
if not self._active_nodes:
|
||||||
|
# If everything has died stop looping
|
||||||
|
self.triggershutdown()
|
||||||
|
raise RuntimeError("Unexpectedly no active workers available")
|
||||||
try:
|
try:
|
||||||
eventcall = self.queue.get(timeout=2.0)
|
eventcall = self.queue.get(timeout=2.0)
|
||||||
break
|
break
|
||||||
|
|||||||
Reference in New Issue
Block a user