forked from celestiaorg/celestia-core
/
perturb.go
121 lines (108 loc) · 3.53 KB
/
perturb.go
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
package main
import (
"fmt"
"path/filepath"
"time"
"github.com/badrootd/celestia-core/libs/log"
rpctypes "github.com/badrootd/celestia-core/rpc/core/types"
e2e "github.com/badrootd/celestia-core/test/e2e/pkg"
)
// Perturbs a running testnet.
func Perturb(testnet *e2e.Testnet) error {
for _, node := range testnet.Nodes {
for _, perturbation := range node.Perturbations {
_, err := PerturbNode(node, perturbation)
if err != nil {
return err
}
time.Sleep(3 * time.Second) // give network some time to recover between each
}
}
return nil
}
// PerturbNode perturbs a node with a given perturbation, returning its status
// after recovering.
func PerturbNode(node *e2e.Node, perturbation e2e.Perturbation) (*rpctypes.ResultStatus, error) {
testnet := node.Testnet
baseDir := filepath.Base(testnet.Dir)
testnetName := fmt.Sprintf("%s_%s", baseDir, testnet.Name)
out, err := execComposeOutput(testnet.Dir, "ps", "-q", node.Name)
if err != nil {
return nil, err
}
name := node.Name
upgraded := false
if len(out) == 0 {
name = name + "_u"
upgraded = true
logger.Info("perturb node", "msg",
log.NewLazySprintf("Node %v already upgraded, operating on alternate container %v",
node.Name, name))
}
switch perturbation {
case e2e.PerturbationDisconnect:
logger.Info("perturb node", "msg", log.NewLazySprintf("Disconnecting node %v...", node.Name))
if err := execDocker("network", "disconnect", testnetName, name); err != nil {
return nil, err
}
time.Sleep(10 * time.Second)
if err := execDocker("network", "connect", testnetName, name); err != nil {
return nil, err
}
case e2e.PerturbationKill:
logger.Info("perturb node", "msg", log.NewLazySprintf("Killing node %v...", node.Name))
if err := execCompose(testnet.Dir, "kill", "-s", "SIGKILL", name); err != nil {
return nil, err
}
if err := execCompose(testnet.Dir, "start", name); err != nil {
return nil, err
}
case e2e.PerturbationPause:
logger.Info("perturb node", "msg", log.NewLazySprintf("Pausing node %v...", node.Name))
if err := execCompose(testnet.Dir, "pause", name); err != nil {
return nil, err
}
time.Sleep(10 * time.Second)
if err := execCompose(testnet.Dir, "unpause", name); err != nil {
return nil, err
}
case e2e.PerturbationRestart:
logger.Info("perturb node", "msg", log.NewLazySprintf("Restarting node %v...", node.Name))
if err := execCompose(testnet.Dir, "restart", name); err != nil {
return nil, err
}
case e2e.PerturbationUpgrade:
oldV := node.Version
newV := node.Testnet.UpgradeVersion
if upgraded {
return nil, fmt.Errorf("node %v can't be upgraded twice from version '%v' to version '%v'",
node.Name, oldV, newV)
}
if oldV == newV {
logger.Info("perturb node", "msg",
log.NewLazySprintf("Skipping upgrade of node %v to version '%v'; versions are equal.",
node.Name, newV))
break
}
logger.Info("perturb node", "msg",
log.NewLazySprintf("Upgrading node %v from version '%v' to version '%v'...",
node.Name, oldV, newV))
if err := execCompose(testnet.Dir, "stop", name); err != nil {
return nil, err
}
time.Sleep(10 * time.Second)
if err := execCompose(testnet.Dir, "up", "-d", name+"_u"); err != nil {
return nil, err
}
default:
return nil, fmt.Errorf("unexpected perturbation %q", perturbation)
}
status, err := waitForNode(node, 0, 20*time.Second)
if err != nil {
return nil, err
}
logger.Info("perturb node",
"msg",
log.NewLazySprintf("Node %v recovered at height %v", node.Name, status.SyncInfo.LatestBlockHeight))
return status, nil
}