diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md index 1052857..b7d2029 100644 --- a/CONTRIBUTING.md +++ b/CONTRIBUTING.md @@ -75,6 +75,14 @@ make soak # 5 minutes of load with a node restarted every 30 seconds make soak SOAK=4h # the full run the numbers in the docs come from ``` +Benchmarks are opt-in too. `make memtier` needs `memtier_benchmark` and `redis-cli` on `PATH` +(`brew install memtier_benchmark redis`): + +```sh +make bench # Go benchmarks for the store, RESP, and a three-node cluster +make memtier BENCH_MODE=cluster # 60 seconds of memtier load; memory, durable, or cluster +``` + ### Pre-commit Hooks (optional) ```sh diff --git a/Makefile b/Makefile index 25bb7ac..ffe9139 100644 --- a/Makefile +++ b/Makefile @@ -1,4 +1,4 @@ -.PHONY: all setup clean build test lint lint-fix vet coverage race soak notice licenses verify-release-artifacts +.PHONY: all setup clean build test lint lint-fix vet coverage race soak bench memtier notice licenses verify-release-artifacts # How long `make soak` runs for. The full run behind the numbers in the docs is SOAK=4h. SOAK ?= 5m @@ -7,6 +7,16 @@ SOAK ?= 5m # back instantly, which is the crash-loop condition the heap numbers in the docs were taken under. SOAK_DOWN ?= 10s +# How many times `make bench` repeats each benchmark. Comparing two builds with benchstat wants +# ten or so; one is enough to see a number. +BENCH_COUNT ?= 1 + +# What `make memtier` starts: memory, durable, or cluster. The performance page has all three. +BENCH_MODE ?= durable + +# How many seconds `make memtier` keeps the load on. +BENCH_TIME ?= 60 + all: vet lint test build notice: @@ -54,4 +64,14 @@ race: # test binary and falls back to the current directory. -timeout 0 leaves the deadline to the # test, so raising SOAK never means recalculating a second number. soak: - @go test . ./internal/cluster -run TestSoak -soak $(SOAK) -soak-down $(SOAK_DOWN) -timeout 0 -v + @go test ./pkg/kvs ./internal/cluster -run TestSoak -soak $(SOAK) -soak-down $(SOAK_DOWN) -timeout 0 -v + +# Go benchmarks for the store, the RESP path, and a three-node cluster. -run '^$' skips the +# tests in the same packages, which would otherwise run first and cost more than the benchmarks. +bench: + @go test ./pkg/kvs ./internal/server ./internal/cluster -run '^$$' -bench . -benchmem -count $(BENCH_COUNT) + +# Real load from memtier_benchmark against a kvs built from this tree. Needs memtier_benchmark +# and redis-cli on PATH. +memtier: build + @./scripts/memtier.sh $(BENCH_MODE) $(BENCH_TIME) diff --git a/changes/unreleased/Added-20260929-234842.yaml b/changes/unreleased/Added-20260929-234842.yaml new file mode 100644 index 0000000..4a15d12 --- /dev/null +++ b/changes/unreleased/Added-20260929-234842.yaml @@ -0,0 +1,5 @@ +kind: Added +body: '`INFO` now reports `total_connections_received`, `total_commands_processed`, and `used_memory`, and `make bench` and `make memtier` measure the store, the RESP path, and a three-node cluster.' +time: 2026-09-29T23:48:42.733703+09:00 +custom: + Issue: "270" diff --git a/changes/unreleased/Fixed-20260929-234843.yaml b/changes/unreleased/Fixed-20260929-234843.yaml new file mode 100644 index 0000000..37af4c5 --- /dev/null +++ b/changes/unreleased/Fixed-20260929-234843.yaml @@ -0,0 +1,5 @@ +kind: Fixed +body: '`make soak` runs again; it named the repository root, which has held no Go files since the layout change.' +time: 2026-09-29T23:48:43.760181+09:00 +custom: + Issue: "270" diff --git a/internal/cluster/bench_test.go b/internal/cluster/bench_test.go new file mode 100644 index 0000000..d5f8ec0 --- /dev/null +++ b/internal/cluster/bench_test.go @@ -0,0 +1,22 @@ +package cluster + +import ( + "strconv" + "testing" +) + +// BenchmarkClusterPut is one write through consensus on a three-node cluster on this machine: +// a round trip to a majority and an fsync on each of them. It is the number the durable +// single-node figure is compared against. +func BenchmarkClusterPut(b *testing.B) { + leader := waitForLeader(b, startCluster(b, 3)) + + i := 0 + for b.Loop() { + key := "bench-" + strconv.Itoa(i%1000) + if err := leader.store.Put(key, "value"); err != nil { + b.Fatalf("Put(%q) error = %v", key, err) + } + i++ + } +} diff --git a/internal/cluster/cluster_test.go b/internal/cluster/cluster_test.go index 0d1dc0e..57d5bd7 100644 --- a/internal/cluster/cluster_test.go +++ b/internal/cluster/cluster_test.go @@ -144,7 +144,7 @@ func newStore() *kvs.Store { return store } -func startCluster(t *testing.T, size int) []*testNode { +func startCluster(t testing.TB, size int) []*testNode { t.Helper() nodes := make([]*testNode, 0, size) @@ -171,7 +171,7 @@ func startCluster(t *testing.T, size int) []*testNode { return nodes } -func (n *testNode) start(t *testing.T, bootstrap bool) { +func (n *testNode) start(t testing.TB, bootstrap bool) { t.Helper() n.store = newStore() @@ -195,7 +195,7 @@ func (n *testNode) start(t *testing.T, bootstrap bool) { }) } -func (n *testNode) stop(t *testing.T) { +func (n *testNode) stop(t testing.TB) { t.Helper() if err := n.node.Close(); err != nil { @@ -206,13 +206,13 @@ func (n *testNode) stop(t *testing.T) { // restart brings the node back on the same address with the same log, which is what a process // that was killed and started again looks like to the rest of the cluster. -func (n *testNode) restart(t *testing.T) { +func (n *testNode) restart(t testing.TB) { t.Helper() n.start(t, false) } -func (n *testNode) mustEventuallyHold(t *testing.T, key, want string) { +func (n *testNode) mustEventuallyHold(t testing.TB, key, want string) { t.Helper() eventually(t, n.id+" to hold "+key, func() bool { @@ -222,7 +222,7 @@ func (n *testNode) mustEventuallyHold(t *testing.T, key, want string) { }) } -func waitForLeader(t *testing.T, nodes []*testNode) *testNode { +func waitForLeader(t testing.TB, nodes []*testNode) *testNode { t.Helper() var leader *testNode @@ -244,7 +244,7 @@ func waitForLeader(t *testing.T, nodes []*testNode) *testNode { // waitForWrite offers the write to every node until one takes it, and reports which did. It is // what a client retrying a rejected write does, so timing it measures the outage a client sees // rather than the moment the cluster privately agreed on a leader. -func waitForWrite(t *testing.T, nodes []*testNode, key, value string) *testNode { +func waitForWrite(t testing.TB, nodes []*testNode, key, value string) *testNode { t.Helper() var taken *testNode @@ -276,7 +276,7 @@ func without(nodes []*testNode, excluded *testNode) []*testNode { // reserveAddr picks a free port and lets go of it, so Raft can bind it and a restarted node can // bind it again. -func reserveAddr(t *testing.T) string { +func reserveAddr(t testing.TB) string { t.Helper() var lc net.ListenConfig @@ -295,7 +295,7 @@ func reserveAddr(t *testing.T) string { // eventually waits for consensus to settle. Elections take as long as they take, so polling is // the only honest way to wait for one. -func eventually(t *testing.T, what string, check func() bool) { +func eventually(t testing.TB, what string, check func() bool) { t.Helper() deadline := time.Now().Add(30 * time.Second) diff --git a/internal/server/resp.go b/internal/server/resp.go index f760139..50669a2 100644 --- a/internal/server/resp.go +++ b/internal/server/resp.go @@ -98,6 +98,12 @@ type RESPServer struct { cursors respCursors scripts respScripts lastID atomic.Int64 + // commands counts every command that reached its handler, for INFO. Queued commands count + // when EXEC runs them and calls made inside a script count too, which is how Redis counts. + // + // ponytail: one shared atomic per command; per-connection counters summed at INFO if it ever + // shows in a profile. + commands atomic.Int64 mu sync.Mutex conns map[*respConn]struct{} @@ -208,7 +214,6 @@ func (s *RESPServer) newConn(netConn net.Conn) *respConn { server: s, netConn: netConn, writer: resp.NewWriter(netConn), - id: s.lastID.Add(1), authed: s.password == "", pushes: make(chan respPush, respPushDepth), done: make(chan struct{}), @@ -229,6 +234,9 @@ func (s *RESPServer) track(conn *respConn) bool { return false } + // Only once admitted, so INFO's total_connections_received leaves out the refused ones, as + // Redis does. + conn.id = s.lastID.Add(1) s.conns[conn] = struct{}{} return true @@ -579,6 +587,8 @@ func (c *respConn) dispatch(args [][]byte) error { return c.writer.WriteSimple("QUEUED") } + c.server.commands.Add(1) + return cmd.run(c, args) } diff --git a/internal/server/resp_bench_test.go b/internal/server/resp_bench_test.go new file mode 100644 index 0000000..2dd0a23 --- /dev/null +++ b/internal/server/resp_bench_test.go @@ -0,0 +1,80 @@ +package server + +import ( + "strconv" + "testing" + + "github.com/redis/go-redis/v9" +) + +// These go through a real client library over loopback TCP, so they price the whole RESP path: +// parsing, dispatch, the store, and the reply. The network is the machine's own and costs little. + +// benchKeys is a fixed keyspace overwritten in place, so a run measures the command path rather +// than a map that keeps growing. The store's benchmarks use the same shape. +const benchKeys = 1000 + +const benchValue = "value" + +// benchKeyNames is built once so that strconv is not what the benchmarks end up measuring. +var benchKeyNames = func() []string { + names := make([]string, benchKeys) + for i := range names { + names[i] = "bench-" + strconv.Itoa(i) + } + + return names +}() + +func benchKey(i int) string { return benchKeyNames[i%benchKeys] } + +func BenchmarkRESPSet(b *testing.B) { + client := newGoRedisClient(b, &redis.Options{}) + ctx := b.Context() + + i := 0 + for b.Loop() { + if err := client.Set(ctx, benchKey(i), benchValue, 0).Err(); err != nil { + b.Fatalf("Set(%q) error = %v", benchKey(i), err) + } + i++ + } +} + +func BenchmarkRESPGet(b *testing.B) { + client := newGoRedisClient(b, &redis.Options{}) + ctx := b.Context() + + for i := range benchKeys { + if err := client.Set(ctx, benchKey(i), benchValue, 0).Err(); err != nil { + b.Fatalf("Set(%q) error = %v", benchKey(i), err) + } + } + + i := 0 + for b.Loop() { + if err := client.Get(ctx, benchKey(i)).Err(); err != nil { + b.Fatalf("Get(%q) error = %v", benchKey(i), err) + } + i++ + } +} + +// BenchmarkRESPSetParallel shares one client, and so its connection pool, across goroutines, +// which is how a service using go-redis sends writes. +func BenchmarkRESPSetParallel(b *testing.B) { + client := newGoRedisClient(b, &redis.Options{}) + ctx := b.Context() + + // RunParallel, unlike b.Loop, would otherwise time the server starting. + b.ResetTimer() + b.RunParallel(func(pb *testing.PB) { + for i := 0; pb.Next(); i++ { + if err := client.Set(ctx, benchKey(i), benchValue, 0).Err(); err != nil { + b.Errorf("Set(%q) error = %v", benchKey(i), err) + + return + } + } + }) +} diff --git a/internal/server/resp_client_test.go b/internal/server/resp_client_test.go index c1fa739..37f4e7e 100644 --- a/internal/server/resp_client_test.go +++ b/internal/server/resp_client_test.go @@ -14,7 +14,7 @@ import ( // newGoRedisClient starts a RESP server and connects a real client library to it. Hand // written bytes cover the wire format elsewhere; this exists to check the parts a client // drives on its own, such as protocol negotiation, pipelining, and subscribe bookkeeping. -func newGoRedisClient(t *testing.T, opts *redis.Options) *redis.Client { +func newGoRedisClient(t testing.TB, opts *redis.Options) *redis.Client { t.Helper() var lc net.ListenConfig diff --git a/internal/server/resp_commands.go b/internal/server/resp_commands.go index 104bc83..4ec9629 100644 --- a/internal/server/resp_commands.go +++ b/internal/server/resp_commands.go @@ -4,6 +4,7 @@ import ( "fmt" "maps" "runtime" + "runtime/metrics" "strconv" "strings" "sync" @@ -484,6 +485,14 @@ func (c *respConn) cmdInfo(_ [][]byte) error { "# Clients", "connected_clients:" + strconv.Itoa(c.server.connCount()), "", + "# Stats", + // Connection ids are handed out one per accepted connection, so the last one is the count. + "total_connections_received:" + strconv.FormatInt(c.server.lastID.Load(), 10), + "total_commands_processed:" + strconv.FormatInt(c.server.commands.Load(), 10), + "", + "# Memory", + "used_memory:" + strconv.FormatUint(respUsedMemory(), 10), + "", } lines = append(lines, c.server.replicationInfo()...) lines = append(lines, @@ -503,6 +512,16 @@ func (c *respConn) cmdInfo(_ [][]byte) error { return c.writer.WriteBulkString(strings.Join(lines, respCRLF) + respCRLF) } +// respUsedMemory is the heap held by objects, live ones and dead ones not yet swept, the closest +// Go has to what Redis reports as used_memory. runtime/metrics reads it without stopping the +// world, which ReadMemStats would. +func respUsedMemory() uint64 { + sample := []metrics.Sample{{Name: "/memory/classes/heap/objects:bytes"}} + metrics.Read(sample) + + return sample[0].Value.Uint64() +} + func (c *respConn) cmdConfig(args [][]byte) error { switch respUpper(args[1]) { case "GET": diff --git a/internal/server/resp_test.go b/internal/server/resp_test.go index e9c60a3..05b5d6d 100644 --- a/internal/server/resp_test.go +++ b/internal/server/resp_test.go @@ -280,8 +280,55 @@ func TestRESPClientAndInfo(t *testing.T) { t.Fatalf("INFO = %q, want it to contain %q", info, want) } } - if !strings.Contains(info, "connected_clients:1") { - t.Fatalf("INFO = %q, want connected_clients:1", info) + // Five CLIENT calls and the INFO itself, which is counted before it reports. + for _, want := range []string{ + "connected_clients:1\r\n", "total_connections_received:1\r\n", + "total_commands_processed:6\r\n", "used_memory:", + } { + if !strings.Contains(info, want) { + t.Fatalf("INFO = %q, want it to contain %q", info, want) + } + } +} + +// A queued command is counted when EXEC runs it, not when it is queued, so a transaction is not +// counted twice. +func TestRESPInfoCountsQueuedCommandsOnExec(t *testing.T) { + client := newRESPClient(t, kvs.NewStore()) + + client.do("+OK"+respCRLF, "MULTI") + client.do("+QUEUED"+respCRLF, "SET", "k", "v") + client.do("+QUEUED"+respCRLF, "GET", "k") + client.do("*2"+respCRLF+"+OK"+respCRLF+"$1"+respCRLF+"v"+respCRLF, "EXEC") + + client.send("INFO") + if info := client.readBulk(); !strings.Contains(info, "total_commands_processed:5\r\n") { + t.Fatalf("INFO = %q, want total_commands_processed:5", info) + } +} + +// A redis.call is a command of its own, so a script that makes one counts twice. +func TestRESPInfoCountsScriptCalls(t *testing.T) { + client := newRESPClient(t, kvs.NewStore()) + + client.do("+OK"+respCRLF, respCmdEval, "return redis.call('SET','k','v')", "0") + + client.send("INFO") + if info := client.readBulk(); !strings.Contains(info, "total_commands_processed:3\r\n") { + t.Fatalf("INFO = %q, want total_commands_processed:3", info) + } +} + +// A command refused before its handler runs is not a processed command. +func TestRESPInfoSkipsRejectedCommands(t *testing.T) { + client := newRESPClient(t, kvs.NewStore()) + + client.do("-ERR unknown command 'NOPE'"+respCRLF, "NOPE") + client.do("-ERR wrong number of arguments for 'echo' command"+respCRLF, "ECHO") + + client.send("INFO") + if info := client.readBulk(); !strings.Contains(info, "total_commands_processed:1\r\n") { + t.Fatalf("INFO = %q, want total_commands_processed:1", info) } } diff --git a/pkg/kvs/bench_test.go b/pkg/kvs/bench_test.go new file mode 100644 index 0000000..d6e42db --- /dev/null +++ b/pkg/kvs/bench_test.go @@ -0,0 +1,102 @@ +package kvs + +import ( + "strconv" + "testing" +) + +// benchKeys is a fixed keyspace overwritten in place, the same shape the soak uses, so a run +// measures the write path rather than a map that keeps growing. +const benchKeys = 1000 + +// benchValue is small on purpose: the question is what a write costs, not how fast bytes copy. +const benchValue = "value" + +// benchKeyNames is built once so that strconv is not what the benchmarks end up measuring. +var benchKeyNames = func() []string { + names := make([]string, benchKeys) + for i := range names { + names[i] = "bench-" + strconv.Itoa(i) + } + + return names +}() + +func benchKey(i int) string { return benchKeyNames[i%benchKeys] } + +func BenchmarkPut(b *testing.B) { + benchPut(b, NewStore()) +} + +// BenchmarkPutDurable is the same write with the append log on. Every write is flushed to disk +// under the store's write lock, so this is the disk's sync rate more than anything kvs does. +func BenchmarkPutDurable(b *testing.B) { + benchPut(b, openTestStore(b, b.TempDir())) +} + +func benchPut(b *testing.B, store *Store) { + i := 0 + for b.Loop() { + if err := store.Put(benchKey(i), benchValue); err != nil { + b.Fatalf("Put(%q) error = %v", benchKey(i), err) + } + i++ + } +} + +func BenchmarkPutParallel(b *testing.B) { + store := NewStore() + b.ResetTimer() + b.RunParallel(func(pb *testing.PB) { + for i := 0; pb.Next(); i++ { + if err := store.Put(benchKey(i), benchValue); err != nil { + b.Errorf("Put(%q) error = %v", benchKey(i), err) + + return + } + } + }) +} + +func BenchmarkGet(b *testing.B) { + store := NewStore() + preload(b, store) + + i := 0 + for b.Loop() { + if _, err := store.Get(benchKey(i)); err != nil { + b.Fatalf("Get(%q) error = %v", benchKey(i), err) + } + i++ + } +} + +// BenchmarkGetParallel is the read path under contention: every reader shares the store's one +// read lock, so this is what that lock costs once there are enough cores to fight over it. +func BenchmarkGetParallel(b *testing.B) { + store := NewStore() + preload(b, store) + + // RunParallel, unlike b.Loop, times everything since the function started. + b.ResetTimer() + b.RunParallel(func(pb *testing.PB) { + for i := 0; pb.Next(); i++ { + if _, err := store.Get(benchKey(i)); err != nil { + b.Errorf("Get(%q) error = %v", benchKey(i), err) + + return + } + } + }) +} + +// preload fills every benchmark key, so a Get never takes the not-found path. +func preload(b *testing.B, store *Store) { + b.Helper() + + for i := range benchKeys { + if err := store.Put(benchKey(i), benchValue); err != nil { + b.Fatalf("Put(%q) error = %v", benchKey(i), err) + } + } +} diff --git a/pkg/kvs/log_test.go b/pkg/kvs/log_test.go index 6924c32..c383c4a 100644 --- a/pkg/kvs/log_test.go +++ b/pkg/kvs/log_test.go @@ -263,7 +263,7 @@ func TestNewStoreWritesNothing(t *testing.T) { } } -func openTestStore(t *testing.T, dir string) *Store { +func openTestStore(t testing.TB, dir string) *Store { t.Helper() store, err := Open(dir, StringCodec{}) diff --git a/scripts/memtier.sh b/scripts/memtier.sh new file mode 100755 index 0000000..e0edbe3 --- /dev/null +++ b/scripts/memtier.sh @@ -0,0 +1,92 @@ +#!/usr/bin/env sh +# Run memtier_benchmark against the kvs in dist/, started in one of three shapes: memory (no +# data dir), durable (append log, one fsync per write), or cluster (three Raft nodes on this +# machine, load on the leader). Usage: memtier.sh [memory|durable|cluster] [seconds] +set -eu + +mode=${1:-durable} +seconds=${2:-60} +root=$(CDPATH= cd -- "$(dirname -- "$0")/.." && pwd) +bin="$root/dist/kvs" +# Away from 6379 so a local Redis does not end up being what gets measured. +port=${KVS_BENCH_PORT:-16379} +work=$(mktemp -d) +pids="" + +cleanup() { + for pid in $pids; do kill "$pid" 2>/dev/null || true; done + wait 2>/dev/null || true + rm -rf "$work" +} +trap cleanup EXIT +# A trapped signal only runs its handler and then carries on, so exit here and let EXIT clean up. +trap 'exit 1' INT TERM + +for tool in memtier_benchmark redis-cli; do + command -v "$tool" >/dev/null || { echo "$tool not found on PATH" >&2; exit 1; } +done + +# kvs reads KVS_* from the environment for any flag not passed, which would change the mode +# (a data dir in memory mode) or lock redis-cli out (a password). +unset KVS_DATA_DIR KVS_RAFT_ADDR KVS_JOIN KVS_NODE_ID KVS_RESP_PASSWORD + +# Whatever already answers here would be what gets measured once the new node fails to bind. +if redis-cli -p "$port" PING >/dev/null 2>&1; then + echo "something already answers on port $port; stop it or set KVS_BENCH_PORT" >&2 + exit 1 +fi + +# start runs node $1 on port+$1 with the remaining arguments. HTTP and gRPC take any free port: +# nothing here talks to them, and fixed ones would collide between nodes. +start() { + i=$1 + shift + "$bin" serve --resp-addr "127.0.0.1:$((port + i))" \ + --http-addr 127.0.0.1:0 --grpc-addr 127.0.0.1:0 "$@" >"$work/node$i.log" 2>&1 & + pids="$pids $!" +} + +# await polls INFO on the load port until it carries the given line, which is how a cluster +# says both that node 0 won the election and that the other two joined. +await() { + tries=0 + until redis-cli -p "$port" INFO replication 2>/dev/null | tr -d '\r' | grep -qx "$1"; do + tries=$((tries + 1)) + if [ "$tries" -gt 300 ]; then + echo "kvs never reported $1" >&2 + cat "$work"/node*.log >&2 + exit 1 + fi + sleep 0.1 + done +} + +case $mode in +memory) + start 0 + await role:master + ;; +durable) + start 0 --data-dir "$work/0" + await role:master + ;; +cluster) + for i in 0 1 2; do + set -- --data-dir "$work/$i" --raft-addr "127.0.0.1:$((port + 100 + i))" + [ "$i" -eq 0 ] || set -- "$@" --join "127.0.0.1:$port" + start "$i" "$@" + done + await connected_slaves:2 + ;; +*) + echo "usage: $0 [memory|durable|cluster] [seconds]" >&2 + exit 2 + ;; +esac + +# Mostly reads, the way a cache or a config store is used; 32 byte values over 100,000 keys. +memtier_benchmark --server 127.0.0.1 --port "$port" --protocol redis \ + --threads 4 --clients 50 --ratio 1:10 --data-size 32 \ + --key-pattern R:R --key-maximum 100000 --distinct-client-seed \ + --test-time "$seconds" --hide-histogram \ + --json-out-file "$root/dist/memtier-$mode.json" diff --git a/website/content/docs/compatibility.md b/website/content/docs/compatibility.md index fb07827..e67a4e6 100644 --- a/website/content/docs/compatibility.md +++ b/website/content/docs/compatibility.md @@ -1,6 +1,6 @@ --- title: "Compatibility" -weight: 6 +weight: 7 --- A `v1` tag is a promise: what this page lists will not break until `v2`, and what it does not @@ -84,7 +84,8 @@ upgrade that raises the format means: drain, start the new version against an em load the data again. **Performance.** Throughput, latency, and memory are not part of the promise. kvs is built for -not losing writes, not for being fast, and a release may trade one for the other. +not losing writes, not for being fast, and a release may trade one for the other. Current +numbers, and how to reproduce them, are on the [performance page](../performance/). **The Go version.** kvs builds with the Go release named in `go.mod`. A minor release may require a newer one. diff --git a/website/content/docs/contributing.md b/website/content/docs/contributing.md index e508e11..f8df880 100644 --- a/website/content/docs/contributing.md +++ b/website/content/docs/contributing.md @@ -1,6 +1,6 @@ --- title: "Contributing" -weight: 7 +weight: 8 --- Contributions are welcome! See [`CONTRIBUTING.md`](https://github.com/skyoo2003/kvs/blob/main/CONTRIBUTING.md) for the full guide. @@ -29,6 +29,14 @@ make soak # 5 minutes of load with a node restarted every 30 seconds make soak SOAK=4h # the full run the numbers on the clustering page come from ``` +Benchmarks are opt-in too. `make memtier` needs `memtier_benchmark` and `redis-cli` on `PATH` +(`brew install memtier_benchmark redis`): + +```sh +make bench # Go benchmarks for the store, RESP, and a three-node cluster +make memtier BENCH_MODE=cluster # 60 seconds of memtier load; memory, durable, or cluster +``` + ## PR Guidelines - Keep PRs small and focused on a single concern diff --git a/website/content/docs/performance.md b/website/content/docs/performance.md new file mode 100644 index 0000000..299c703 --- /dev/null +++ b/website/content/docs/performance.md @@ -0,0 +1,79 @@ +--- +title: "Performance" +weight: 6 +--- + +These numbers are a baseline to compare later changes against, not a promise: +[performance is outside the compatibility promise](../compatibility/), and kvs is built for not +losing writes before it is built for being fast. What the page does promise is that every number +on it can be reproduced with one command from a checkout. + +## How it was measured + +One machine, nothing else under load: an Apple M4 (10 cores, 16 GB, internal SSD) on macOS +26.6.2, running on **battery power**, with Go 1.26.7 and memtier_benchmark 2.5.1, on the change +that added this page, on top of `2a8076a`. Laptop numbers move with temperature and power +source; the two runs per mode below are there to show by how much. + +The load is `make memtier`: 4 threads × 50 connections, one `SET` to every ten `GET`s, 32-byte +values over 100,000 random keys, 60 seconds. Latencies are in milliseconds and are what the client +saw, queueing included. + +## Under load + +| Mode | Run | Ops/sec | SET p50 | SET p99 | GET p50 | GET p99 | +|---|---|---:|---:|---:|---:|---:| +| memory | 1 | 213,819 | 0.87 | 2.90 | 0.85 | 2.66 | +| memory | 2 | 205,800 | 0.91 | 2.93 | 0.89 | 2.74 | +| durable | 1 | 2,289 | 758 | 2,015 | 3.98 | 12.1 | +| durable | 2 | 2,786 | 758 | 774 | 3.98 | 4.67 | +| cluster | 1 | 472 | 4,555 | 4,653 | 0.087 | 0.143 | +| cluster | 2 | 469 | 4,522 | 4,686 | 0.079 | 0.143 | + +**memory** is `kvs serve` with no `--data-dir`: about **210,000 operations a second** at under a +millisecond at the median. + +**durable** adds `--data-dir`, and every write is flushed to disk before it is acknowledged, +under the store's one write lock. On this machine a flush takes about 3.7ms, so writes top out +near 270 a second however many clients send them, and two hundred connections queue behind +each other for **three quarters of a second** at the median. Reads are cheap but wait for +whichever flush holds the lock, which is the 4ms at their median. The first run's 2-second SET +tail did not repeat; the machine was on battery. + +**cluster** is three nodes on this one machine with the load on the leader. Every write is a +consensus round, one at a time, at about 21ms each (below) — roughly **45 writes a second**, and +four and a half seconds of queueing at the median under this many connections. Reads never touch +consensus, which is why they are the fastest reads in the table; it is also why they +[may be behind](../clustering/). The total is low because each connection waits for its own write +before sending its next read. + +## Per operation + +`make bench BENCH_COUNT=6`, median of the six runs: + +| Benchmark | What it is | ns/op | B/op | allocs/op | +|---|---|---:|---:|---:| +| `BenchmarkPut` | in-memory write | 101 | 144 | 2 | +| `BenchmarkPutParallel` | the same, from every core | 166 | 144 | 2 | +| `BenchmarkGet` | in-memory read | 61 | 32 | 1 | +| `BenchmarkGetParallel` | the same, from every core | 102 | 32 | 1 | +| `BenchmarkPutDurable` | write with the append log | 3,700,000 | 4,633 | 8 | +| `BenchmarkRESPSet` | `SET` through go-redis over loopback | 14,990 | 5,469 | 32 | +| `BenchmarkRESPGet` | `GET` through go-redis over loopback | 14,490 | 3,560 | 22 | +| `BenchmarkRESPSetParallel` | `SET` from every core, one client pool | 7,326 | 5,480 | 32 | +| `BenchmarkClusterPut` | write through a three-node cluster | 21,000,000 | 149,000 | 624 | + +The store's parallel runs are slower per operation than its serial ones because every caller +shares its one lock; they are there so that sharding it has something to beat. + +## Reproducing it + +```sh +brew install memtier_benchmark redis # memtier, and redis-cli for the readiness check +make memtier BENCH_MODE=memory # or durable, or cluster; BENCH_TIME=60 by default +make bench BENCH_COUNT=10 # then compare two builds with benchstat +``` + +`make memtier` builds `dist/kvs`, starts it on port 16379 (`KVS_BENCH_PORT` moves it), waits +until it answers — in cluster mode, until the other two nodes have joined — and leaves the full +result in `dist/memtier-.json`. It stops every node it started when it exits. diff --git a/website/content/docs/release.md b/website/content/docs/release.md index d90e414..2d29644 100644 --- a/website/content/docs/release.md +++ b/website/content/docs/release.md @@ -1,6 +1,6 @@ --- title: "Release Process" -weight: 8 +weight: 9 --- A release is one tag push. Everything else — binaries, checksums, the container image, the