From c7d793c619e3690f091dd98ac9b4085e212db26f Mon Sep 17 00:00:00 2001 From: ginuerzh Date: Sat, 1 Aug 2026 23:23:28 +0800 Subject: [PATCH] e2e: add cmd probe recovery test (gost#837) Verifies dead->alive->dead node transition: a node excluded by a failing cmd probe resumes carrying traffic once the probe flips healthy, without restart. --- tests/e2e/probe_test.go | 44 ++++++++++++++++++++++ tests/e2e/testdata/probe/recovery.yaml | 52 ++++++++++++++++++++++++++ 2 files changed, 96 insertions(+) create mode 100644 tests/e2e/testdata/probe/recovery.yaml diff --git a/tests/e2e/probe_test.go b/tests/e2e/probe_test.go index d211d4a..fa41f3d 100644 --- a/tests/e2e/probe_test.go +++ b/tests/e2e/probe_test.go @@ -100,6 +100,50 @@ func (s *ProbeSuite) TestCmdProbeFailover() { } } +// TestProbeRecovery verifies the dead→alive→dead transition that issue #837 +// describes: a node that fails is excluded, and once it recovers (probe flips +// healthy) it resumes carrying traffic. Both relays stay up; only the cmd +// probe flag files in the container flip, so no restart is needed. +// +// Phase 1: node-a marked dead, node-b healthy → every request goes to node-b. +// Phase 2: recover node-a, kill node-b → every request must still succeed, +// which is only possible if node-a's probe recovered it (otherwise both nodes +// would be excluded and the request would fail). +func (s *ProbeSuite) TestProbeRecovery() { + gostC, err := RunGostContainerWithPorts(s.ctx, SharedNetworkName, "testdata/probe/recovery.yaml", "8080/tcp") + s.Require().NoError(err) + defer gostC.Terminate(s.ctx) + + // The node probes fire immediately at startup, so let both settle healthy + // before we start flipping flag files. + time.Sleep(1 * time.Second) + + // Phase 1: knock node-a down, leave node-b up. + _, _, err = gostC.Exec(s.ctx, []string{"touch", "/tmp/a_down"}) + s.Require().NoError(err) + time.Sleep(2500 * time.Millisecond) // > probe interval + + for range 10 { + code, body := s.proxyRequest(gostC, "8080") + s.Require().Equal(0, code, "phase1: dead node-a must be excluded, all traffic via node-b") + s.Require().Contains(body, "hello-gost") + } + + // Phase 2: recover node-a, knock node-b down. If node-a's probe did not + // revive it, the request would fail (no healthy candidate). + _, _, err = gostC.Exec(s.ctx, []string{"rm", "-f", "/tmp/a_down", "/tmp/b_down"}) + s.Require().NoError(err) + _, _, err = gostC.Exec(s.ctx, []string{"touch", "/tmp/b_down"}) + s.Require().NoError(err) + time.Sleep(2500 * time.Millisecond) + + for range 10 { + code, body := s.proxyRequest(gostC, "8080") + s.Require().Equal(0, code, "phase2: recovered node-a must resume carrying traffic") + s.Require().Contains(body, "hello-gost") + } +} + func TestProbeSuite(t *testing.T) { suite.Run(t, new(ProbeSuite)) } diff --git a/tests/e2e/testdata/probe/recovery.yaml b/tests/e2e/testdata/probe/recovery.yaml new file mode 100644 index 0000000..8d1b26f --- /dev/null +++ b/tests/e2e/testdata/probe/recovery.yaml @@ -0,0 +1,52 @@ +services: +- name: proxy + addr: :8080 + handler: + type: http + chain: my-chain + listener: + type: tcp + +# Two live relays — both forward to the echo server. Relays never go down; +# liveness is driven solely by the cmd probes reading flag files in the +# container, so the test can flip node health without restarting anything. +- name: relay-a + addr: 127.0.0.1:18081 + handler: + type: http + listener: + type: tcp + +- name: relay-b + addr: 127.0.0.1:18082 + handler: + type: http + listener: + type: tcp + +chains: +- name: my-chain + hops: + - name: hop-1 + selector: + strategy: lowestlatency + maxFails: 1 + failTimeout: 10s + nodes: + - name: node-a + addr: 127.0.0.1:18081 + connector: + type: http + probe: + # exits 1 while /tmp/a_down exists → node marked dead by probe. + type: cmd + command: "test -f /tmp/a_down && exit 1 || exit 0" + interval: 2s + - name: node-b + addr: 127.0.0.1:18082 + connector: + type: http + probe: + type: cmd + command: "test -f /tmp/b_down && exit 1 || exit 0" + interval: 2s