valkey chaos

This commit is contained in:
2026-06-11 10:52:17 +02:00
parent ec88214a2e
commit 6a43ba022c
7 changed files with 287 additions and 1005 deletions
+152 -229
View File
@@ -139,17 +139,6 @@ func applyFlagOverrides(cfg *Config, host string, port int, password string, tls
return nil
}
func (c *Config) primary() (*redis.Options, error) {
if c.Host != "" && c.Port != 0 {
return optionsFromConn(c.Host, c.Port, c.Password, c.DB, c.TLS), nil
}
if len(c.Workers) > 0 {
w := c.Workers[0]
return optionsFromConn(w.Host, w.Port, w.Password, w.DB, w.TLS), nil
}
return nil, fmt.Errorf("config has neither top-level host nor workers")
}
func optionsFromConn(host string, port int, password string, db int, useTLS bool) *redis.Options {
opts := &redis.Options{
Addr: fmt.Sprintf("%s:%d", host, port),
@@ -219,20 +208,19 @@ func envInt(name string, def int) int {
}
func main() {
configPath := flag.String("config", "config.yaml", "path to YAML config file")
continuous := flag.Bool("continuous", false, "single-worker rolling smoke loop")
chaosLocal := flag.Bool("chaos", false, "multi-worker chaos TUI (local, in-process)")
workerMode := flag.Bool("worker", false, "distributed worker — runs the chaos workload and ships events to --collector-url")
collectorMode := flag.Bool("collector", false, "distributed collector — HTTP dashboard that ingests events from workers")
configPath := flag.String("config", "config.yaml", "path to YAML config file (chaos --worker/--collector)")
workerMode := flag.Bool("worker", false, "chaos worker — runs the config.yaml workload and ships events to --collector-url")
collectorMode := flag.Bool("collector", false, "chaos collector — HTTP dashboard that ingests events from workers")
collectorURL := flag.String("collector-url", os.Getenv("COLLECTOR_URL"), "collector base URL (worker mode); env: COLLECTOR_URL")
port := flag.Int("collector-port", envInt("PORT", 8080), "HTTP port (collector mode); env: PORT")
interval := flag.Duration("interval", time.Second, "tick interval for --continuous")
hostOverride := flag.String("host", "", "override host for all workers (local testing)")
hostOverride := flag.String("host", "", "override host for the cli tool (--seed/--verify/etc and the default test suite)")
portOverride := flag.Int("port", 0, "override port for all workers (local testing)")
passwordOverride := flag.String("password", "", "override password for all workers (local testing)")
tlsEnable := flag.Bool("tls", false, "enable TLS for all workers using system root CAs (local testing)")
tlsCAPath := flag.String("tls-ca", "", "path to TLS CA cert PEM; enables TLS for all workers (local testing)")
seedConn := flag.String("conn", "", "Valkey connection string for --seed/--verify/--flush/--inspect (e.g. redis://default:pw@valkey1:6379 or rediss://...:6380 for TLS); overrides --host/--port/--password/--tls")
seedConn := flag.String("conn", "", "Valkey connection string for the cli tool (default test suite, --watch, --seed/--verify/--flush/--inspect) — e.g. redis://default:pw@valkey1:6379 or rediss://...:6380 for TLS; overrides --host/--port/--password/--tls")
watchMode := flag.Bool("watch", false, "PING the target on --interval until Ctrl+C, printing each outage's start/recovery/duration and a summary on exit")
watchInterval := flag.Duration("interval", 500*time.Millisecond, "probe interval for --watch")
seedMode := flag.Bool("seed", false, "bulk-load one Valkey with deterministic data (local; use --host valkey.zerops over zcli VPN)")
verifyMode := flag.Bool("verify", false, "re-read seed:* keys and check they're 1:1 with the seed (same --seed-* flags)")
flushMode := flag.Bool("flush", false, "FLUSHDB the selected --db (clears the whole database)")
@@ -246,54 +234,19 @@ func main() {
seedTimeout := flag.Duration("timeout", 5*time.Second, "dial/read/write timeout in --seed/--verify/--flush/--inspect mode (raise for slow VPN links)")
flag.Parse()
// ── Regime 1: chaos ──────────────────────────────────────────────────
// Driven solely by config.yaml. loadConfig applies the env templating that
// lets one config file drive the distributed fleet: ${var} expansion pulls
// secrets from the container env, and $ZEROPS_Number gives each replica a
// distinct DB + pub/sub channel suffix. No connection-flag overrides here.
switch {
case *collectorMode:
os.Exit(runCollector(*port))
case *seedMode, *verifyMode, *flushMode, *inspectMode:
opts, err := buildConnOptions(*seedConn, *hostOverride, *portOverride, *passwordOverride, *tlsEnable, *tlsCAPath)
if err != nil {
log.Fatalf("connection error: %v", err)
}
// --db overrides whatever the conn string / flags resolved to, so the
// same target can be pointed at any logical database without editing
// the conn URL. -1 means "leave as-is".
if *dbOverride >= 0 {
opts.DB = *dbOverride
}
// Generous timeouts: the zcli VPN adds latency and can stall a dial
// well past go-redis's 5s default mid-run. PoolTimeout > DialTimeout
// so a slow dial doesn't surface as pool exhaustion first.
opts.DialTimeout = *seedTimeout
opts.ReadTimeout = *seedTimeout
opts.WriteTimeout = *seedTimeout
opts.PoolTimeout = *seedTimeout + 5*time.Second
switch {
case *inspectMode:
os.Exit(runInspect(opts))
case *flushMode:
os.Exit(runFlush(opts))
case *verifyMode:
os.Exit(runVerify(opts, *seedSeed, *seedMB*1024*1024, *seedValueBytes, *seedBatch))
default:
os.Exit(runSeed(opts, *seedSeed, *seedMB*1024*1024, *seedValueBytes, *seedBatch, *seedThrottleMB*1024*1024))
}
}
cfg, err := loadConfig(*configPath)
if err != nil {
log.Fatalf("config error: %v", err)
}
if err := applyFlagOverrides(cfg, *hostOverride, *portOverride, *passwordOverride, *tlsEnable, *tlsCAPath); err != nil {
log.Fatalf("flag override error: %v", err)
}
switch {
case *chaosLocal:
if len(cfg.Workers) == 0 {
log.Fatalf("--chaos requires workers: list in %s", *configPath)
}
os.Exit(runChaos(cfg.Workers))
case *workerMode:
cfg, err := loadConfig(*configPath)
if err != nil {
log.Fatalf("config error: %v", err)
}
if len(cfg.Workers) == 0 {
log.Fatalf("--worker requires workers: list in %s", *configPath)
}
@@ -303,20 +256,146 @@ func main() {
os.Exit(runWorker(cfg.Workers, *collectorURL))
}
primary, err := cfg.primary()
// ── Regime 2: cli tool ───────────────────────────────────────────────
// One-time commands against a single Valkey, resolved from --conn or the
// split --host/--port/--password/--tls[-ca] flags. Covers the default test
// suite plus --seed/--verify/--flush/--inspect.
opts, err := buildConnOptions(*seedConn, *hostOverride, *portOverride, *passwordOverride, *tlsEnable, *tlsCAPath)
if err != nil {
log.Fatalf("config error: %v", err)
log.Fatalf("connection error: %v", err)
}
fmt.Printf("redis config: addr=%s db=%d tls=%t password=%s\n",
primary.Addr, primary.DB, primary.TLSConfig != nil, redactPassword(primary.Password))
rdb := redis.NewClient(primary)
// --db overrides whatever the conn string / flags resolved to, so the same
// target can be pointed at any logical database without editing the conn
// URL. -1 means "leave as-is".
if *dbOverride >= 0 {
opts.DB = *dbOverride
}
// Generous timeouts: the zcli VPN adds latency and can stall a dial well
// past go-redis's 5s default mid-run. PoolTimeout > DialTimeout so a slow
// dial doesn't surface as pool exhaustion first.
opts.DialTimeout = *seedTimeout
opts.ReadTimeout = *seedTimeout
opts.WriteTimeout = *seedTimeout
opts.PoolTimeout = *seedTimeout + 5*time.Second
switch {
case *inspectMode:
os.Exit(runInspect(opts))
case *flushMode:
os.Exit(runFlush(opts))
case *verifyMode:
os.Exit(runVerify(opts, *seedSeed, *seedMB*1024*1024, *seedValueBytes, *seedBatch))
case *seedMode:
os.Exit(runSeed(opts, *seedSeed, *seedMB*1024*1024, *seedValueBytes, *seedBatch, *seedThrottleMB*1024*1024))
case *watchMode:
os.Exit(runWatch(opts, *watchInterval))
default:
os.Exit(runBasicTests(opts))
}
}
// runWatch PINGs the target every interval until SIGINT/SIGTERM, reporting the
// up→down and down→up edges so you can measure failover/outage windows. The
// per-probe timeout is opts.ReadTimeout (set from --timeout); lower it for
// tighter outage-edge resolution. Detection granularity ≈ max(interval, timeout).
func runWatch(opts *redis.Options, interval time.Duration) int {
const tsLayout = "15:04:05.000"
fmt.Printf("watch: addr=%s db=%d tls=%t password=%s interval=%s timeout=%s (Ctrl+C to stop)\n",
opts.Addr, opts.DB, opts.TLSConfig != nil, redactPassword(opts.Password), interval, opts.ReadTimeout)
rdb := redis.NewClient(opts)
defer rdb.Close()
if *continuous {
runContinuous(rdb, *interval)
return
ctx, stop := signal.NotifyContext(context.Background(), os.Interrupt, syscall.SIGTERM)
defer stop()
ticker := time.NewTicker(interval)
defer ticker.Stop()
probe := func() error {
pctx, cancel := context.WithTimeout(ctx, opts.ReadTimeout)
defer cancel()
return rdb.Ping(pctx).Err()
}
var (
started = time.Now()
up = true
first = true
outageStart time.Time
lastBeat = time.Now()
outages int
totalDown time.Duration
longest time.Duration
)
// endOutage closes the current outage, accumulates it, and prints recovery.
endOutage := func(at time.Time, recovered bool) {
d := at.Sub(outageStart)
totalDown += d
if d > longest {
longest = d
}
if recovered {
fmt.Printf("%s RECOVERED — outage lasted %s\n", at.Format(tsLayout), d.Round(time.Millisecond))
} else {
fmt.Printf("%s OUTAGE ONGOING at exit — %s so far\n", at.Format(tsLayout), d.Round(time.Millisecond))
}
}
for {
select {
case <-ctx.Done():
now := time.Now()
if !up {
endOutage(now, false)
}
fmt.Printf("\nstopped after %s: outages=%d total_down=%s longest=%s\n",
time.Since(started).Round(time.Millisecond), outages,
totalDown.Round(time.Millisecond), longest.Round(time.Millisecond))
return 0
case <-ticker.C:
}
err := probe()
// SIGINT can land mid-probe; treat a cancelled probe as shutdown, not an outage.
if ctx.Err() != nil {
continue
}
nowUp := err == nil
now := time.Now()
switch {
case first:
first = false
up = nowUp
if nowUp {
fmt.Printf("%s up\n", now.Format(tsLayout))
} else {
outageStart, outages = now, outages+1
fmt.Printf("%s DOWN at start — %s\n", now.Format(tsLayout), trim(err.Error(), 120))
}
case up && !nowUp:
outageStart, outages = now, outages+1
fmt.Printf("%s OUTAGE START — %s\n", now.Format(tsLayout), trim(err.Error(), 120))
case !up && nowUp:
endOutage(now, true)
case up && nowUp && now.Sub(lastBeat) >= 10*time.Second:
fmt.Printf("%s … up (elapsed %s, outages %d)\n",
now.Format(tsLayout), time.Since(started).Round(time.Second), outages)
lastBeat = now
}
up = nowUp
}
}
// runBasicTests runs the one-shot smoke suite against a single Valkey resolved
// from --conn or the split connection flags.
func runBasicTests(opts *redis.Options) int {
fmt.Printf("redis config: addr=%s db=%d tls=%t password=%s\n",
opts.Addr, opts.DB, opts.TLSConfig != nil, redactPassword(opts.Password))
rdb := redis.NewClient(opts)
defer rdb.Close()
ctx, cancel := context.WithTimeout(context.Background(), 30*time.Second)
defer cancel()
@@ -352,8 +431,9 @@ func main() {
fmt.Printf("\n%d/%d tests passed\n", len(tests)-failed, len(tests))
if failed > 0 {
os.Exit(1)
return 1
}
return 0
}
func testPing(ctx context.Context, r *redis.Client) error {
@@ -560,160 +640,3 @@ func testCleanup(ctx context.Context, r *redis.Client) error {
}
return r.Del(ctx, keys...).Err()
}
func runContinuous(r *redis.Client, interval time.Duration) {
stopCtx, stop := signal.NotifyContext(context.Background(), os.Interrupt, syscall.SIGTERM)
defer stop()
subCtx, cancelSub := context.WithCancel(stopCtx)
defer cancelSub()
sub := r.Subscribe(subCtx, "vk:cont:channel")
defer sub.Close()
if _, err := sub.Receive(subCtx); err != nil {
log.Fatalf("subscribe: %v", err)
}
subCh := sub.Channel()
ops := []struct {
name string
fn func(context.Context, *redis.Client, int64) error
}{
{"PING", opPing},
{"SET/GET", opSetGet},
{"INCR", opIncr},
{"LIST", opList},
{"INFO", opInfo},
{"PUB/SUB", func(ctx context.Context, r *redis.Client, i int64) error {
return opPubSub(ctx, r, i, subCh)
}},
}
var cycle, okCount, failCount int64
ticker := time.NewTicker(interval)
defer ticker.Stop()
fmt.Printf("continuous mode: addr=%s interval=%s (Ctrl+C to stop)\n", r.Options().Addr, interval)
for {
select {
case <-stopCtx.Done():
fmt.Printf("\nstopped: cycles=%d ok=%d failed=%d\n", cycle, okCount, failCount)
return
case <-ticker.C:
}
cycle++
results := make([]string, len(ops))
var failures []string
cycleStart := time.Now()
for i, op := range ops {
ctx, cancel := context.WithTimeout(stopCtx, interval)
start := time.Now()
err := op.fn(ctx, r, cycle)
cancel()
dur := time.Since(start)
if err != nil {
results[i] = fmt.Sprintf("%s=FAIL", op.name)
failures = append(failures, fmt.Sprintf(" %s (%s): %v", op.name, dur, err))
} else {
results[i] = fmt.Sprintf("%s=ok(%s)", op.name, dur.Round(time.Microsecond))
}
}
if len(failures) == 0 {
okCount++
} else {
failCount++
}
fmt.Printf("[%s] #%d %s | total=%s\n",
time.Now().Format("15:04:05"), cycle, joinResults(results), time.Since(cycleStart).Round(time.Microsecond))
for _, f := range failures {
fmt.Println(f)
}
}
}
func joinResults(parts []string) string {
out := ""
for i, p := range parts {
if i > 0 {
out += " "
}
out += p
}
return out
}
func opPing(ctx context.Context, r *redis.Client, _ int64) error {
pong, err := r.Ping(ctx).Result()
if err != nil {
return err
}
if pong != "PONG" {
return fmt.Errorf("expected PONG, got %q", pong)
}
return nil
}
func opSetGet(ctx context.Context, r *redis.Client, i int64) error {
val := fmt.Sprintf("v-%d", i)
if err := r.Set(ctx, "vk:cont:str", val, 10*time.Second).Err(); err != nil {
return err
}
got, err := r.Get(ctx, "vk:cont:str").Result()
if err != nil {
return err
}
if got != val {
return fmt.Errorf("set %q, got %q", val, got)
}
return nil
}
func opIncr(ctx context.Context, r *redis.Client, _ int64) error {
_, err := r.Incr(ctx, "vk:cont:counter").Result()
return err
}
func opList(ctx context.Context, r *redis.Client, i int64) error {
if err := r.RPush(ctx, "vk:cont:list", fmt.Sprintf("item-%d", i)).Err(); err != nil {
return err
}
if err := r.LTrim(ctx, "vk:cont:list", -10, -1).Err(); err != nil {
return err
}
_, err := r.LRange(ctx, "vk:cont:list", 0, -1).Result()
return err
}
func opInfo(ctx context.Context, r *redis.Client, _ int64) error {
info, err := r.Info(ctx, "server").Result()
if err != nil {
return err
}
if len(info) == 0 {
return fmt.Errorf("empty INFO")
}
return nil
}
func opPubSub(ctx context.Context, r *redis.Client, i int64, ch <-chan *redis.Message) error {
payload := fmt.Sprintf("ping-%d", i)
if err := r.Publish(ctx, "vk:cont:channel", payload).Err(); err != nil {
return err
}
for {
select {
case msg := <-ch:
if msg == nil {
return fmt.Errorf("channel closed")
}
if msg.Payload == payload {
return nil
}
case <-ctx.Done():
return fmt.Errorf("timeout waiting for %q", payload)
}
}
}