e2e: add cmd probe recovery test (gost#837)

Verifies dead->alive->dead node transition: a node excluded by a failing cmd
probe resumes carrying traffic once the probe flips healthy, without restart.
This commit is contained in:
ginuerzh
2026-08-01 23:23:28 +08:00
parent f25148f2e0
commit c7d793c619
2 changed files with 96 additions and 0 deletions
+44
View File
@@ -100,6 +100,50 @@ func (s *ProbeSuite) TestCmdProbeFailover() {
}
}
// TestProbeRecovery verifies the dead→alive→dead transition that issue #837
// describes: a node that fails is excluded, and once it recovers (probe flips
// healthy) it resumes carrying traffic. Both relays stay up; only the cmd
// probe flag files in the container flip, so no restart is needed.
//
// Phase 1: node-a marked dead, node-b healthy → every request goes to node-b.
// Phase 2: recover node-a, kill node-b → every request must still succeed,
// which is only possible if node-a's probe recovered it (otherwise both nodes
// would be excluded and the request would fail).
func (s *ProbeSuite) TestProbeRecovery() {
gostC, err := RunGostContainerWithPorts(s.ctx, SharedNetworkName, "testdata/probe/recovery.yaml", "8080/tcp")
s.Require().NoError(err)
defer gostC.Terminate(s.ctx)
// The node probes fire immediately at startup, so let both settle healthy
// before we start flipping flag files.
time.Sleep(1 * time.Second)
// Phase 1: knock node-a down, leave node-b up.
_, _, err = gostC.Exec(s.ctx, []string{"touch", "/tmp/a_down"})
s.Require().NoError(err)
time.Sleep(2500 * time.Millisecond) // > probe interval
for range 10 {
code, body := s.proxyRequest(gostC, "8080")
s.Require().Equal(0, code, "phase1: dead node-a must be excluded, all traffic via node-b")
s.Require().Contains(body, "hello-gost")
}
// Phase 2: recover node-a, knock node-b down. If node-a's probe did not
// revive it, the request would fail (no healthy candidate).
_, _, err = gostC.Exec(s.ctx, []string{"rm", "-f", "/tmp/a_down", "/tmp/b_down"})
s.Require().NoError(err)
_, _, err = gostC.Exec(s.ctx, []string{"touch", "/tmp/b_down"})
s.Require().NoError(err)
time.Sleep(2500 * time.Millisecond)
for range 10 {
code, body := s.proxyRequest(gostC, "8080")
s.Require().Equal(0, code, "phase2: recovered node-a must resume carrying traffic")
s.Require().Contains(body, "hello-gost")
}
}
func TestProbeSuite(t *testing.T) {
suite.Run(t, new(ProbeSuite))
}
+52
View File
@@ -0,0 +1,52 @@
services:
- name: proxy
addr: :8080
handler:
type: http
chain: my-chain
listener:
type: tcp
# Two live relays — both forward to the echo server. Relays never go down;
# liveness is driven solely by the cmd probes reading flag files in the
# container, so the test can flip node health without restarting anything.
- name: relay-a
addr: 127.0.0.1:18081
handler:
type: http
listener:
type: tcp
- name: relay-b
addr: 127.0.0.1:18082
handler:
type: http
listener:
type: tcp
chains:
- name: my-chain
hops:
- name: hop-1
selector:
strategy: lowestlatency
maxFails: 1
failTimeout: 10s
nodes:
- name: node-a
addr: 127.0.0.1:18081
connector:
type: http
probe:
# exits 1 while /tmp/a_down exists → node marked dead by probe.
type: cmd
command: "test -f /tmp/a_down && exit 1 || exit 0"
interval: 2s
- name: node-b
addr: 127.0.0.1:18082
connector:
type: http
probe:
type: cmd
command: "test -f /tmp/b_down && exit 1 || exit 0"
interval: 2s