valkey chaos
This commit is contained in:
+152
-229
@@ -139,17 +139,6 @@ func applyFlagOverrides(cfg *Config, host string, port int, password string, tls
|
||||
return nil
|
||||
}
|
||||
|
||||
func (c *Config) primary() (*redis.Options, error) {
|
||||
if c.Host != "" && c.Port != 0 {
|
||||
return optionsFromConn(c.Host, c.Port, c.Password, c.DB, c.TLS), nil
|
||||
}
|
||||
if len(c.Workers) > 0 {
|
||||
w := c.Workers[0]
|
||||
return optionsFromConn(w.Host, w.Port, w.Password, w.DB, w.TLS), nil
|
||||
}
|
||||
return nil, fmt.Errorf("config has neither top-level host nor workers")
|
||||
}
|
||||
|
||||
func optionsFromConn(host string, port int, password string, db int, useTLS bool) *redis.Options {
|
||||
opts := &redis.Options{
|
||||
Addr: fmt.Sprintf("%s:%d", host, port),
|
||||
@@ -219,20 +208,19 @@ func envInt(name string, def int) int {
|
||||
}
|
||||
|
||||
func main() {
|
||||
configPath := flag.String("config", "config.yaml", "path to YAML config file")
|
||||
continuous := flag.Bool("continuous", false, "single-worker rolling smoke loop")
|
||||
chaosLocal := flag.Bool("chaos", false, "multi-worker chaos TUI (local, in-process)")
|
||||
workerMode := flag.Bool("worker", false, "distributed worker — runs the chaos workload and ships events to --collector-url")
|
||||
collectorMode := flag.Bool("collector", false, "distributed collector — HTTP dashboard that ingests events from workers")
|
||||
configPath := flag.String("config", "config.yaml", "path to YAML config file (chaos --worker/--collector)")
|
||||
workerMode := flag.Bool("worker", false, "chaos worker — runs the config.yaml workload and ships events to --collector-url")
|
||||
collectorMode := flag.Bool("collector", false, "chaos collector — HTTP dashboard that ingests events from workers")
|
||||
collectorURL := flag.String("collector-url", os.Getenv("COLLECTOR_URL"), "collector base URL (worker mode); env: COLLECTOR_URL")
|
||||
port := flag.Int("collector-port", envInt("PORT", 8080), "HTTP port (collector mode); env: PORT")
|
||||
interval := flag.Duration("interval", time.Second, "tick interval for --continuous")
|
||||
hostOverride := flag.String("host", "", "override host for all workers (local testing)")
|
||||
hostOverride := flag.String("host", "", "override host for the cli tool (--seed/--verify/etc and the default test suite)")
|
||||
portOverride := flag.Int("port", 0, "override port for all workers (local testing)")
|
||||
passwordOverride := flag.String("password", "", "override password for all workers (local testing)")
|
||||
tlsEnable := flag.Bool("tls", false, "enable TLS for all workers using system root CAs (local testing)")
|
||||
tlsCAPath := flag.String("tls-ca", "", "path to TLS CA cert PEM; enables TLS for all workers (local testing)")
|
||||
seedConn := flag.String("conn", "", "Valkey connection string for --seed/--verify/--flush/--inspect (e.g. redis://default:pw@valkey1:6379 or rediss://...:6380 for TLS); overrides --host/--port/--password/--tls")
|
||||
seedConn := flag.String("conn", "", "Valkey connection string for the cli tool (default test suite, --watch, --seed/--verify/--flush/--inspect) — e.g. redis://default:pw@valkey1:6379 or rediss://...:6380 for TLS; overrides --host/--port/--password/--tls")
|
||||
watchMode := flag.Bool("watch", false, "PING the target on --interval until Ctrl+C, printing each outage's start/recovery/duration and a summary on exit")
|
||||
watchInterval := flag.Duration("interval", 500*time.Millisecond, "probe interval for --watch")
|
||||
seedMode := flag.Bool("seed", false, "bulk-load one Valkey with deterministic data (local; use --host valkey.zerops over zcli VPN)")
|
||||
verifyMode := flag.Bool("verify", false, "re-read seed:* keys and check they're 1:1 with the seed (same --seed-* flags)")
|
||||
flushMode := flag.Bool("flush", false, "FLUSHDB the selected --db (clears the whole database)")
|
||||
@@ -246,54 +234,19 @@ func main() {
|
||||
seedTimeout := flag.Duration("timeout", 5*time.Second, "dial/read/write timeout in --seed/--verify/--flush/--inspect mode (raise for slow VPN links)")
|
||||
flag.Parse()
|
||||
|
||||
// ── Regime 1: chaos ──────────────────────────────────────────────────
|
||||
// Driven solely by config.yaml. loadConfig applies the env templating that
|
||||
// lets one config file drive the distributed fleet: ${var} expansion pulls
|
||||
// secrets from the container env, and $ZEROPS_Number gives each replica a
|
||||
// distinct DB + pub/sub channel suffix. No connection-flag overrides here.
|
||||
switch {
|
||||
case *collectorMode:
|
||||
os.Exit(runCollector(*port))
|
||||
case *seedMode, *verifyMode, *flushMode, *inspectMode:
|
||||
opts, err := buildConnOptions(*seedConn, *hostOverride, *portOverride, *passwordOverride, *tlsEnable, *tlsCAPath)
|
||||
if err != nil {
|
||||
log.Fatalf("connection error: %v", err)
|
||||
}
|
||||
// --db overrides whatever the conn string / flags resolved to, so the
|
||||
// same target can be pointed at any logical database without editing
|
||||
// the conn URL. -1 means "leave as-is".
|
||||
if *dbOverride >= 0 {
|
||||
opts.DB = *dbOverride
|
||||
}
|
||||
// Generous timeouts: the zcli VPN adds latency and can stall a dial
|
||||
// well past go-redis's 5s default mid-run. PoolTimeout > DialTimeout
|
||||
// so a slow dial doesn't surface as pool exhaustion first.
|
||||
opts.DialTimeout = *seedTimeout
|
||||
opts.ReadTimeout = *seedTimeout
|
||||
opts.WriteTimeout = *seedTimeout
|
||||
opts.PoolTimeout = *seedTimeout + 5*time.Second
|
||||
switch {
|
||||
case *inspectMode:
|
||||
os.Exit(runInspect(opts))
|
||||
case *flushMode:
|
||||
os.Exit(runFlush(opts))
|
||||
case *verifyMode:
|
||||
os.Exit(runVerify(opts, *seedSeed, *seedMB*1024*1024, *seedValueBytes, *seedBatch))
|
||||
default:
|
||||
os.Exit(runSeed(opts, *seedSeed, *seedMB*1024*1024, *seedValueBytes, *seedBatch, *seedThrottleMB*1024*1024))
|
||||
}
|
||||
}
|
||||
|
||||
cfg, err := loadConfig(*configPath)
|
||||
if err != nil {
|
||||
log.Fatalf("config error: %v", err)
|
||||
}
|
||||
if err := applyFlagOverrides(cfg, *hostOverride, *portOverride, *passwordOverride, *tlsEnable, *tlsCAPath); err != nil {
|
||||
log.Fatalf("flag override error: %v", err)
|
||||
}
|
||||
|
||||
switch {
|
||||
case *chaosLocal:
|
||||
if len(cfg.Workers) == 0 {
|
||||
log.Fatalf("--chaos requires workers: list in %s", *configPath)
|
||||
}
|
||||
os.Exit(runChaos(cfg.Workers))
|
||||
case *workerMode:
|
||||
cfg, err := loadConfig(*configPath)
|
||||
if err != nil {
|
||||
log.Fatalf("config error: %v", err)
|
||||
}
|
||||
if len(cfg.Workers) == 0 {
|
||||
log.Fatalf("--worker requires workers: list in %s", *configPath)
|
||||
}
|
||||
@@ -303,20 +256,146 @@ func main() {
|
||||
os.Exit(runWorker(cfg.Workers, *collectorURL))
|
||||
}
|
||||
|
||||
primary, err := cfg.primary()
|
||||
// ── Regime 2: cli tool ───────────────────────────────────────────────
|
||||
// One-time commands against a single Valkey, resolved from --conn or the
|
||||
// split --host/--port/--password/--tls[-ca] flags. Covers the default test
|
||||
// suite plus --seed/--verify/--flush/--inspect.
|
||||
opts, err := buildConnOptions(*seedConn, *hostOverride, *portOverride, *passwordOverride, *tlsEnable, *tlsCAPath)
|
||||
if err != nil {
|
||||
log.Fatalf("config error: %v", err)
|
||||
log.Fatalf("connection error: %v", err)
|
||||
}
|
||||
fmt.Printf("redis config: addr=%s db=%d tls=%t password=%s\n",
|
||||
primary.Addr, primary.DB, primary.TLSConfig != nil, redactPassword(primary.Password))
|
||||
rdb := redis.NewClient(primary)
|
||||
// --db overrides whatever the conn string / flags resolved to, so the same
|
||||
// target can be pointed at any logical database without editing the conn
|
||||
// URL. -1 means "leave as-is".
|
||||
if *dbOverride >= 0 {
|
||||
opts.DB = *dbOverride
|
||||
}
|
||||
// Generous timeouts: the zcli VPN adds latency and can stall a dial well
|
||||
// past go-redis's 5s default mid-run. PoolTimeout > DialTimeout so a slow
|
||||
// dial doesn't surface as pool exhaustion first.
|
||||
opts.DialTimeout = *seedTimeout
|
||||
opts.ReadTimeout = *seedTimeout
|
||||
opts.WriteTimeout = *seedTimeout
|
||||
opts.PoolTimeout = *seedTimeout + 5*time.Second
|
||||
|
||||
switch {
|
||||
case *inspectMode:
|
||||
os.Exit(runInspect(opts))
|
||||
case *flushMode:
|
||||
os.Exit(runFlush(opts))
|
||||
case *verifyMode:
|
||||
os.Exit(runVerify(opts, *seedSeed, *seedMB*1024*1024, *seedValueBytes, *seedBatch))
|
||||
case *seedMode:
|
||||
os.Exit(runSeed(opts, *seedSeed, *seedMB*1024*1024, *seedValueBytes, *seedBatch, *seedThrottleMB*1024*1024))
|
||||
case *watchMode:
|
||||
os.Exit(runWatch(opts, *watchInterval))
|
||||
default:
|
||||
os.Exit(runBasicTests(opts))
|
||||
}
|
||||
}
|
||||
|
||||
// runWatch PINGs the target every interval until SIGINT/SIGTERM, reporting the
|
||||
// up→down and down→up edges so you can measure failover/outage windows. The
|
||||
// per-probe timeout is opts.ReadTimeout (set from --timeout); lower it for
|
||||
// tighter outage-edge resolution. Detection granularity ≈ max(interval, timeout).
|
||||
func runWatch(opts *redis.Options, interval time.Duration) int {
|
||||
const tsLayout = "15:04:05.000"
|
||||
fmt.Printf("watch: addr=%s db=%d tls=%t password=%s interval=%s timeout=%s (Ctrl+C to stop)\n",
|
||||
opts.Addr, opts.DB, opts.TLSConfig != nil, redactPassword(opts.Password), interval, opts.ReadTimeout)
|
||||
|
||||
rdb := redis.NewClient(opts)
|
||||
defer rdb.Close()
|
||||
|
||||
if *continuous {
|
||||
runContinuous(rdb, *interval)
|
||||
return
|
||||
ctx, stop := signal.NotifyContext(context.Background(), os.Interrupt, syscall.SIGTERM)
|
||||
defer stop()
|
||||
|
||||
ticker := time.NewTicker(interval)
|
||||
defer ticker.Stop()
|
||||
|
||||
probe := func() error {
|
||||
pctx, cancel := context.WithTimeout(ctx, opts.ReadTimeout)
|
||||
defer cancel()
|
||||
return rdb.Ping(pctx).Err()
|
||||
}
|
||||
|
||||
var (
|
||||
started = time.Now()
|
||||
up = true
|
||||
first = true
|
||||
outageStart time.Time
|
||||
lastBeat = time.Now()
|
||||
outages int
|
||||
totalDown time.Duration
|
||||
longest time.Duration
|
||||
)
|
||||
// endOutage closes the current outage, accumulates it, and prints recovery.
|
||||
endOutage := func(at time.Time, recovered bool) {
|
||||
d := at.Sub(outageStart)
|
||||
totalDown += d
|
||||
if d > longest {
|
||||
longest = d
|
||||
}
|
||||
if recovered {
|
||||
fmt.Printf("%s RECOVERED — outage lasted %s\n", at.Format(tsLayout), d.Round(time.Millisecond))
|
||||
} else {
|
||||
fmt.Printf("%s OUTAGE ONGOING at exit — %s so far\n", at.Format(tsLayout), d.Round(time.Millisecond))
|
||||
}
|
||||
}
|
||||
|
||||
for {
|
||||
select {
|
||||
case <-ctx.Done():
|
||||
now := time.Now()
|
||||
if !up {
|
||||
endOutage(now, false)
|
||||
}
|
||||
fmt.Printf("\nstopped after %s: outages=%d total_down=%s longest=%s\n",
|
||||
time.Since(started).Round(time.Millisecond), outages,
|
||||
totalDown.Round(time.Millisecond), longest.Round(time.Millisecond))
|
||||
return 0
|
||||
case <-ticker.C:
|
||||
}
|
||||
|
||||
err := probe()
|
||||
// SIGINT can land mid-probe; treat a cancelled probe as shutdown, not an outage.
|
||||
if ctx.Err() != nil {
|
||||
continue
|
||||
}
|
||||
nowUp := err == nil
|
||||
now := time.Now()
|
||||
|
||||
switch {
|
||||
case first:
|
||||
first = false
|
||||
up = nowUp
|
||||
if nowUp {
|
||||
fmt.Printf("%s up\n", now.Format(tsLayout))
|
||||
} else {
|
||||
outageStart, outages = now, outages+1
|
||||
fmt.Printf("%s DOWN at start — %s\n", now.Format(tsLayout), trim(err.Error(), 120))
|
||||
}
|
||||
case up && !nowUp:
|
||||
outageStart, outages = now, outages+1
|
||||
fmt.Printf("%s OUTAGE START — %s\n", now.Format(tsLayout), trim(err.Error(), 120))
|
||||
case !up && nowUp:
|
||||
endOutage(now, true)
|
||||
case up && nowUp && now.Sub(lastBeat) >= 10*time.Second:
|
||||
fmt.Printf("%s … up (elapsed %s, outages %d)\n",
|
||||
now.Format(tsLayout), time.Since(started).Round(time.Second), outages)
|
||||
lastBeat = now
|
||||
}
|
||||
up = nowUp
|
||||
}
|
||||
}
|
||||
|
||||
// runBasicTests runs the one-shot smoke suite against a single Valkey resolved
|
||||
// from --conn or the split connection flags.
|
||||
func runBasicTests(opts *redis.Options) int {
|
||||
fmt.Printf("redis config: addr=%s db=%d tls=%t password=%s\n",
|
||||
opts.Addr, opts.DB, opts.TLSConfig != nil, redactPassword(opts.Password))
|
||||
rdb := redis.NewClient(opts)
|
||||
defer rdb.Close()
|
||||
|
||||
ctx, cancel := context.WithTimeout(context.Background(), 30*time.Second)
|
||||
defer cancel()
|
||||
|
||||
@@ -352,8 +431,9 @@ func main() {
|
||||
|
||||
fmt.Printf("\n%d/%d tests passed\n", len(tests)-failed, len(tests))
|
||||
if failed > 0 {
|
||||
os.Exit(1)
|
||||
return 1
|
||||
}
|
||||
return 0
|
||||
}
|
||||
|
||||
func testPing(ctx context.Context, r *redis.Client) error {
|
||||
@@ -560,160 +640,3 @@ func testCleanup(ctx context.Context, r *redis.Client) error {
|
||||
}
|
||||
return r.Del(ctx, keys...).Err()
|
||||
}
|
||||
|
||||
func runContinuous(r *redis.Client, interval time.Duration) {
|
||||
stopCtx, stop := signal.NotifyContext(context.Background(), os.Interrupt, syscall.SIGTERM)
|
||||
defer stop()
|
||||
|
||||
subCtx, cancelSub := context.WithCancel(stopCtx)
|
||||
defer cancelSub()
|
||||
sub := r.Subscribe(subCtx, "vk:cont:channel")
|
||||
defer sub.Close()
|
||||
if _, err := sub.Receive(subCtx); err != nil {
|
||||
log.Fatalf("subscribe: %v", err)
|
||||
}
|
||||
subCh := sub.Channel()
|
||||
|
||||
ops := []struct {
|
||||
name string
|
||||
fn func(context.Context, *redis.Client, int64) error
|
||||
}{
|
||||
{"PING", opPing},
|
||||
{"SET/GET", opSetGet},
|
||||
{"INCR", opIncr},
|
||||
{"LIST", opList},
|
||||
{"INFO", opInfo},
|
||||
{"PUB/SUB", func(ctx context.Context, r *redis.Client, i int64) error {
|
||||
return opPubSub(ctx, r, i, subCh)
|
||||
}},
|
||||
}
|
||||
|
||||
var cycle, okCount, failCount int64
|
||||
ticker := time.NewTicker(interval)
|
||||
defer ticker.Stop()
|
||||
|
||||
fmt.Printf("continuous mode: addr=%s interval=%s (Ctrl+C to stop)\n", r.Options().Addr, interval)
|
||||
|
||||
for {
|
||||
select {
|
||||
case <-stopCtx.Done():
|
||||
fmt.Printf("\nstopped: cycles=%d ok=%d failed=%d\n", cycle, okCount, failCount)
|
||||
return
|
||||
case <-ticker.C:
|
||||
}
|
||||
cycle++
|
||||
|
||||
results := make([]string, len(ops))
|
||||
var failures []string
|
||||
cycleStart := time.Now()
|
||||
for i, op := range ops {
|
||||
ctx, cancel := context.WithTimeout(stopCtx, interval)
|
||||
start := time.Now()
|
||||
err := op.fn(ctx, r, cycle)
|
||||
cancel()
|
||||
dur := time.Since(start)
|
||||
if err != nil {
|
||||
results[i] = fmt.Sprintf("%s=FAIL", op.name)
|
||||
failures = append(failures, fmt.Sprintf(" %s (%s): %v", op.name, dur, err))
|
||||
} else {
|
||||
results[i] = fmt.Sprintf("%s=ok(%s)", op.name, dur.Round(time.Microsecond))
|
||||
}
|
||||
}
|
||||
|
||||
if len(failures) == 0 {
|
||||
okCount++
|
||||
} else {
|
||||
failCount++
|
||||
}
|
||||
|
||||
fmt.Printf("[%s] #%d %s | total=%s\n",
|
||||
time.Now().Format("15:04:05"), cycle, joinResults(results), time.Since(cycleStart).Round(time.Microsecond))
|
||||
for _, f := range failures {
|
||||
fmt.Println(f)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func joinResults(parts []string) string {
|
||||
out := ""
|
||||
for i, p := range parts {
|
||||
if i > 0 {
|
||||
out += " "
|
||||
}
|
||||
out += p
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
func opPing(ctx context.Context, r *redis.Client, _ int64) error {
|
||||
pong, err := r.Ping(ctx).Result()
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if pong != "PONG" {
|
||||
return fmt.Errorf("expected PONG, got %q", pong)
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
func opSetGet(ctx context.Context, r *redis.Client, i int64) error {
|
||||
val := fmt.Sprintf("v-%d", i)
|
||||
if err := r.Set(ctx, "vk:cont:str", val, 10*time.Second).Err(); err != nil {
|
||||
return err
|
||||
}
|
||||
got, err := r.Get(ctx, "vk:cont:str").Result()
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if got != val {
|
||||
return fmt.Errorf("set %q, got %q", val, got)
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
func opIncr(ctx context.Context, r *redis.Client, _ int64) error {
|
||||
_, err := r.Incr(ctx, "vk:cont:counter").Result()
|
||||
return err
|
||||
}
|
||||
|
||||
func opList(ctx context.Context, r *redis.Client, i int64) error {
|
||||
if err := r.RPush(ctx, "vk:cont:list", fmt.Sprintf("item-%d", i)).Err(); err != nil {
|
||||
return err
|
||||
}
|
||||
if err := r.LTrim(ctx, "vk:cont:list", -10, -1).Err(); err != nil {
|
||||
return err
|
||||
}
|
||||
_, err := r.LRange(ctx, "vk:cont:list", 0, -1).Result()
|
||||
return err
|
||||
}
|
||||
|
||||
func opInfo(ctx context.Context, r *redis.Client, _ int64) error {
|
||||
info, err := r.Info(ctx, "server").Result()
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if len(info) == 0 {
|
||||
return fmt.Errorf("empty INFO")
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
func opPubSub(ctx context.Context, r *redis.Client, i int64, ch <-chan *redis.Message) error {
|
||||
payload := fmt.Sprintf("ping-%d", i)
|
||||
if err := r.Publish(ctx, "vk:cont:channel", payload).Err(); err != nil {
|
||||
return err
|
||||
}
|
||||
for {
|
||||
select {
|
||||
case msg := <-ch:
|
||||
if msg == nil {
|
||||
return fmt.Errorf("channel closed")
|
||||
}
|
||||
if msg.Payload == payload {
|
||||
return nil
|
||||
}
|
||||
case <-ctx.Done():
|
||||
return fmt.Errorf("timeout waiting for %q", payload)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user