diff --git a/client/ccstatusline-ipc b/client/ccstatusline-ipc index 2c38f4eae..5b53fbf12 100755 --- a/client/ccstatusline-ipc +++ b/client/ccstatusline-ipc @@ -15,6 +15,14 @@ # CCSTATUSLINE_RUNTIME_DIR override the daemon runtime directory # CCSTATUSLINE_IPC_CONNECT_TIMEOUT seconds to connect (default 2) # CCSTATUSLINE_IPC_TIMEOUT total seconds (default 10) +# CCSTATUSLINE_DAEMON_START override the lazy-start command (see +# start_daemon below) +# CCSTATUSLINE_NO_AUTOSTART when set, never start a daemon lazily +# +# On-demand lifecycle (#53): this wrapper runs only in installed shared mode +# (`daemon install`), so its lazy start is opt-in by construction — one-shot +# users never execute this file. On a missing daemon it starts one (`daemon +# start`, serialized by the #47 cold-start lock) and retries into it. # # Exit status: 0 on a rendered status line, 1 on any failure (stdout empty). @@ -38,21 +46,57 @@ else fi discovery=$runtime_dir/daemon.env -[ -r "$discovery" ] || die "no readable discovery file at $discovery; is the daemon running?" - # Endpoint discovery: flat KEY=VALUE lines written by the daemon. Prefix # matching keeps values (socket paths may contain '=') intact. -protocol='' -socket='' -token='' -while IFS= read -r line || [ -n "$line" ]; do - case $line in - '#'*) continue ;; - protocol=*) protocol=${line#protocol=} ;; - socket=*) socket=${line#socket=} ;; - token=*) token=${line#token=} ;; - esac -done < "$discovery" +read_discovery() { + protocol='' + socket='' + token='' + [ -r "$discovery" ] || return 1 + while IFS= read -r line || [ -n "$line" ]; do + case $line in + '#'*) continue ;; + protocol=*) protocol=${line#protocol=} ;; + socket=*) socket=${line#socket=} ;; + token=*) token=${line#token=} ;; + esac + done < "$discovery" + return 0 +} + +# Lazy start (#53): run `ccstatusline daemon start` — the #47 cold-start lock +# serializes concurrent starters onto one server, so racing clients each run +# this and converge on the same daemon. Tried at most once per invocation, +# bounded by the starter's own startup timeout. Opt out with +# CCSTATUSLINE_NO_AUTOSTART; override the command with +# CCSTATUSLINE_DAEMON_START (a full command line, split on whitespace). +autostart_attempted=0 +start_daemon() { + [ -z "${CCSTATUSLINE_NO_AUTOSTART:-}" ] || return 1 + [ "$autostart_attempted" -eq 0 ] || return 1 + autostart_attempted=1 + if [ -n "${CCSTATUSLINE_DAEMON_START:-}" ]; then + $CCSTATUSLINE_DAEMON_START >/dev/null 2>&1 + return $? + fi + # Same install root as this wrapper: the npm package ships + # client/ccstatusline-ipc next to dist/ccstatusline.js (node), a dev + # checkout next to src/ccstatusline.ts (bun). + install_root=$(CDPATH= cd -- "$(dirname -- "$0")/.." && pwd) || return 1 + if [ -f "$install_root/dist/ccstatusline.js" ]; then + node "$install_root/dist/ccstatusline.js" daemon start >/dev/null 2>&1 + elif [ -f "$install_root/src/ccstatusline.ts" ]; then + bun "$install_root/src/ccstatusline.ts" daemon start >/dev/null 2>&1 + else + return 1 + fi +} + +if ! read_discovery; then + # No live daemon: start one on demand and re-read what it published. + start_daemon || die "no readable discovery file at $discovery; is the daemon running?" + read_discovery || die "lazy daemon start left no discovery file at $discovery" +fi [ "$protocol" = "1" ] || die "unsupported daemon protocol '${protocol:-none}'; want 1" [ -n "$socket" ] || die "discovery file has no socket path" @@ -114,6 +158,14 @@ CONFIG ) curl_rc=$? [ "$curl_rc" -eq 0 ] && break + # Connect failure (curl exit 7): the daemon died after discovery was + # read. Transparently lazy-start a fresh one and retry into it — the + # autostart guard inside start_daemon bounds this to one restart. + if [ "$curl_rc" -eq 7 ]; then + if start_daemon && read_discovery && [ "$protocol" = "1" ]; then + continue + fi + fi attempt=$((attempt + 1)) # 503 busy: the render queue is momentarily full (a burst of repaints # across sessions). Back off on a growing ladder — 8s total budget — so diff --git a/docs/USAGE.md b/docs/USAGE.md index 4dc941aec..6c004644f 100644 --- a/docs/USAGE.md +++ b/docs/USAGE.md @@ -475,9 +475,16 @@ ccstatusline daemon stop # stop the daemon (status line goes quiet) was no status line before, `uninstall` removes ours again. The wrapper command is `sh /client/ccstatusline-ipc`, POSIX `sh` + `curl` only — no Node on the repaint path. -- If the daemon is down, the client fails with empty stdout and exit 1: the - status line goes quiet, nothing else breaks, and **no daemon is - auto-started**. Start it explicitly with `daemon start` (or `install`). +- The lifecycle is on-demand (#53): with the daemon down, the shared-mode + client lazily starts one (`ccstatusline daemon start`, serialized by the + cold-start lock) and transparently retries the render into it; after + `daemonIdleStopMinutes` (settings.json, default 10, `0` disables) with zero + requests the daemon exits by itself. The first render after an idle period + pays the cold start; busy periods keep it alive. Set + `CCSTATUSLINE_NO_AUTOSTART=1` to make the client fail fast instead of + starting a daemon, or `CCSTATUSLINE_DAEMON_START=""` to override + the lazy-start command — with autostart disabled that failure is empty + stdout and exit 1: the status line goes quiet, nothing else breaks. - `daemon status` prints what `/v1/health` exposes — protocol and build identity, pid, uptime, last render time, in-flight renders and aggregate request counters. It never prints the auth token. Exit code is 1 unless a diff --git a/docs/daemon-53-bench.md b/docs/daemon-53-bench.md new file mode 100644 index 000000000..8b7356656 --- /dev/null +++ b/docs/daemon-53-bench.md @@ -0,0 +1,84 @@ +# ccstatusline on-demand lifecycle bench (issue #53) + +Lazy start + idle auto-stop under a 20–30 concurrent-session burst, versus the +same burst rendered one-shot. Companion to [daemon-19-bench.md](daemon-19-bench.md); +raw numbers in [daemon-53-results.json](daemon-53-results.json), harness in +`scripts/benchmark-lifecycle-53.py`. + +- runtime: `bun src/ccstatusline.ts` (dev checkout; the wrapper's lazy start + resolves the entry from its own install root) +- machine: Apple M5, 10 cores, macOS-26.6.2-arm64-arm-64bit +- generated: 2026-10-01T14:08:47+0800 + +## Workload + +Each scenario runs N concurrent clients over two waves. Sessions alternate +between a 500-row and a 10 000-row transcript and between a dirty and a clean +git repo, so the daemon's caches hold a cold/warm mix (both transcript sizes +parse; the two repos render distinct git-changes lines). The lazy scenarios +start with **no daemon process at all** — the first wave includes the +on-demand start raced by all N clients. + +## Per-scenario cost + +| scenario | renders | CPU s/render | p50 ms | p95 ms | p99 ms | failures | +|---|---|---|---|---|---|---| +| oneshot-20-cold | 20 | 188.0 | 434 | 481 | 481 | 0 | +| oneshot-20-warm | 20 | 182.1 | 415 | 451 | 453 | 0 | +| lazy-20-cold (incl. daemon boot) | 20 | 237.0 | 799 | 902 | 905 | 0 | +| lazy-20-warm | 20 | 42.4 | 151 | 271 | 276 | 0 | +| oneshot-30-cold | 30 | 186.4 | 930 | 1035 | 1040 | 0 | +| oneshot-30-warm | 30 | 194.1 | 808 | 934 | 950 | 0 | +| lazy-30-cold (incl. daemon boot) | 30 | 229.7 | 1023 | 1149 | 1202 | 0 | +| lazy-30-warm | 30 | 49.4 | 185 | 481 | 528 | 0 | + +## One-shot vs shared (lazy) + +| sessions | phase | one-shot CPU s/render | shared CPU s/render | CPU reduction | one-shot p95 | shared p95 | hashes equal | +|---|---|---|---|---|---|---|---| +| 20 | cold | 188.0 ms | 237.0 ms | −26.0% | 481 ms | 902 ms | yes | +| 20 | warm | 182.1 ms | 42.4 ms | **76.7%** | 451 ms | 271 ms | yes | +| 30 | cold | 186.4 ms | 229.7 ms | −23.2% | 1035 ms | 1149 ms | yes | +| 30 | warm | 194.1 ms | 49.4 ms | **74.5%** | 934 ms | 481 ms | yes | + +The cold-wave regression is the on-demand start itself, paid once per idle +period: every client that finds no discovery runs `daemon start`, so a burst of +N simultaneous first renders spawns N short-lived starters (one wins the #47 +cold-start lock, the rest converge on its server). Warm bursts — the steady +state while sessions are open — keep the 75–80% CPU reduction of #19/#50 and +cut p95 roughly in half. + +## Burst behavior (acceptance: no 503 storm) + +- lazy-30 cold wave: all 30 clients raced the lazy start; the discovery pid was + identical before and after both waves — exactly one daemon served the whole + scenario (no second server, no restart). +- Daemon counters, lazy-30: `requests=176, ok=91, busy=85`, client failures 0. + The bounded queue (`MAX_IN_FLIGHT_RENDERS = 4`) rejected concurrent cold + renders with 503s exactly as designed, and the client's retry ladder + (0.05 s → 1 s, 8 s budget) absorbed all of them. +- Note from an earlier (unrecorded) run: under the cold 30-burst the + dirty-repo renders once collapsed to the clean-repo line (git reads failing + under load); the recorded run shows the expected two distinct lines with + identical output between one-shot and shared. Watch for it when re-running. + +## Idle auto-stop (acceptance: daemon disappears, render restarts it) + +With `daemonIdleStopMinutes: 1` and zero requests: + +- the daemon exited by itself after 72.2 s observed (1 minute configured; the + idle check runs at `idle/5` cadence, here 12 s, so up to one interval of + lateness), +- its discovery file was cleaned up (no stale state for the next start), +- the next client render lazily started a fresh daemon (new pid) and returned + the rendered line with rc 0. + +Default is 10 minutes; `0` disables auto-stop; any request (health included) +resets the clock, so busy periods keep the daemon alive without observers. + +## Opt-in gating + +The lazy start lives in the shared-mode client wrapper only — the file Claude +Code executes exclusively after `daemon install`. The one-shot render path +never spawns anything (covered by a test: one-shot render leaves the runtime +dir empty). `CCSTATUSLINE_NO_AUTOSTART=1` restores the fail-fast behavior. diff --git a/docs/daemon-53-results.json b/docs/daemon-53-results.json new file mode 100644 index 000000000..27807d546 --- /dev/null +++ b/docs/daemon-53-results.json @@ -0,0 +1,474 @@ +{ + "meta": { + "runtime": "bun", + "entry": "/Users/axisrow/.ao/data/worktrees/ccstatusline/ccstatusline-28/src/ccstatusline.ts", + "machine": { + "cpu_model": "Apple M5", + "cpu_count": 10, + "os": "macOS-26.6.2-arm64-arm-64bit", + "python": "3.12.10" + }, + "generated": "2026-10-01T14:08:47+0800", + "workload": "sessions alternate small(500-row)/large(10000-row) transcripts and dirty/clean repos; two waves per scenario (cold includes the lazy start)", + "notes": [ + "Lazy/shared CPU covers the daemon AND every client: clients via getrusage(RUSAGE_CHILDREN), the detached daemon via ps cumulative CPU (it escapes RUSAGE_CHILDREN after reparenting).", + "Cold wave of the lazy scenario includes the on-demand daemon start paid by the first clients." + ] + }, + "runs": { + "lazy-20": { + "cold": { + "label": "lazy-20-cold", + "renders": 20, + "failures": 0, + "cpu_total_s": 4.7399, + "cpu_per_render_ms": 236.997, + "latency_ms": { + "p50": 799.05, + "p95": 901.87, + "p99": 904.8 + }, + "latencies_ms": [ + 702.42, + 706.89, + 709.01, + 711.32, + 714.37, + 722.33, + 730.71, + 747.02, + 765.02, + 799.05, + 809.67, + 812.39, + 822.29, + 843.82, + 880.33, + 881.77, + 886.31, + 896.87, + 901.87, + 904.8 + ], + "output_hashes": [ + "3d8c1c16e8a92518e4c15b7a995f7d1a426a52eb2d8d5c28a9815896f3804047", + "69aa1f345fd61e159aea5c99074045fd5acfacc44fccd344b9b3dd56d6955b8d" + ], + "wall_s": 0.909, + "daemon_pids_cold_wave": [ + "60351" + ], + "daemon_pids_warm_wave": [ + "60351" + ], + "single_daemon_across_waves": true, + "daemon_cpu_total_s": 0.66, + "daemon_counters": { + "requests": 97, + "ok": 61, + "busy": 36 + } + }, + "warm": { + "label": "lazy-20-warm", + "renders": 20, + "failures": 0, + "cpu_total_s": 0.8478, + "cpu_per_render_ms": 42.388, + "latency_ms": { + "p50": 150.7, + "p95": 271.16, + "p99": 275.73 + }, + "latencies_ms": [ + 66.34, + 69.88, + 75.74, + 107.95, + 109.31, + 110.79, + 113.95, + 142.72, + 147.83, + 150.7, + 162.76, + 173.9, + 175.49, + 178.97, + 180.91, + 193.02, + 256.23, + 267.42, + 271.16, + 275.73 + ], + "output_hashes": [ + "3d8c1c16e8a92518e4c15b7a995f7d1a426a52eb2d8d5c28a9815896f3804047", + "69aa1f345fd61e159aea5c99074045fd5acfacc44fccd344b9b3dd56d6955b8d" + ], + "wall_s": 0.305 + } + }, + "oneshot-20": { + "cold": { + "label": "oneshot-20-cold", + "renders": 20, + "failures": 0, + "cpu_total_s": 3.7606, + "cpu_per_render_ms": 188.028, + "latency_ms": { + "p50": 434.18, + "p95": 480.86, + "p99": 481.43 + }, + "latencies_ms": [ + 365.57, + 367.87, + 392.97, + 397.48, + 410.8, + 418.16, + 423.42, + 429.7, + 434.02, + 434.18, + 453.72, + 454.46, + 457.88, + 459.32, + 465.55, + 475.11, + 475.13, + 480.79, + 480.86, + 481.43 + ], + "output_hashes": [ + "3d8c1c16e8a92518e4c15b7a995f7d1a426a52eb2d8d5c28a9815896f3804047", + "69aa1f345fd61e159aea5c99074045fd5acfacc44fccd344b9b3dd56d6955b8d" + ], + "wall_s": 0.492 + }, + "warm": { + "label": "oneshot-20-warm", + "renders": 20, + "failures": 0, + "cpu_total_s": 3.6415, + "cpu_per_render_ms": 182.077, + "latency_ms": { + "p50": 414.73, + "p95": 451.13, + "p99": 453.07 + }, + "latencies_ms": [ + 329.85, + 346.97, + 350.39, + 389.1, + 396.58, + 397.75, + 398.85, + 399.37, + 413.77, + 414.73, + 417.1, + 422.19, + 422.33, + 422.56, + 423.66, + 426.91, + 428.38, + 440.73, + 451.13, + 453.07 + ], + "output_hashes": [ + "3d8c1c16e8a92518e4c15b7a995f7d1a426a52eb2d8d5c28a9815896f3804047", + "69aa1f345fd61e159aea5c99074045fd5acfacc44fccd344b9b3dd56d6955b8d" + ], + "wall_s": 0.459 + } + }, + "lazy-30": { + "cold": { + "label": "lazy-30-cold", + "renders": 30, + "failures": 0, + "cpu_total_s": 6.8912, + "cpu_per_render_ms": 229.707, + "latency_ms": { + "p50": 1023.1, + "p95": 1149.48, + "p99": 1201.82 + }, + "latencies_ms": [ + 803.66, + 815.7, + 881.68, + 899.05, + 906.15, + 911.44, + 919.59, + 956.77, + 957.76, + 967.98, + 971.86, + 1005.09, + 1013.73, + 1019.44, + 1023.1, + 1031.11, + 1033.08, + 1040.74, + 1057.59, + 1081.38, + 1083.83, + 1084.23, + 1091.91, + 1093.22, + 1105.03, + 1117.52, + 1119.95, + 1138.77, + 1149.48, + 1201.82 + ], + "output_hashes": [ + "3d8c1c16e8a92518e4c15b7a995f7d1a426a52eb2d8d5c28a9815896f3804047", + "69aa1f345fd61e159aea5c99074045fd5acfacc44fccd344b9b3dd56d6955b8d" + ], + "wall_s": 1.23, + "daemon_pids_cold_wave": [ + "61325" + ], + "daemon_pids_warm_wave": [ + "61325" + ], + "single_daemon_across_waves": true, + "daemon_cpu_total_s": 0.93, + "daemon_counters": { + "requests": 176, + "ok": 91, + "busy": 85 + } + }, + "warm": { + "label": "lazy-30-warm", + "renders": 30, + "failures": 0, + "cpu_total_s": 1.4831, + "cpu_per_render_ms": 49.436, + "latency_ms": { + "p50": 184.57, + "p95": 480.82, + "p99": 528.27 + }, + "latencies_ms": [ + 68.91, + 73.26, + 83.67, + 91.88, + 93.41, + 126.59, + 127.37, + 135.87, + 136.39, + 138.67, + 142.08, + 147.55, + 165.96, + 183.63, + 184.57, + 205.18, + 227.49, + 233.04, + 233.54, + 250.68, + 257.9, + 277.48, + 277.94, + 308.7, + 372.82, + 412.95, + 422.77, + 431.73, + 480.82, + 528.27 + ], + "output_hashes": [ + "3d8c1c16e8a92518e4c15b7a995f7d1a426a52eb2d8d5c28a9815896f3804047", + "69aa1f345fd61e159aea5c99074045fd5acfacc44fccd344b9b3dd56d6955b8d" + ], + "wall_s": 0.586 + } + }, + "oneshot-30": { + "cold": { + "label": "oneshot-30-cold", + "renders": 30, + "failures": 0, + "cpu_total_s": 5.5929, + "cpu_per_render_ms": 186.429, + "latency_ms": { + "p50": 929.75, + "p95": 1035.4, + "p99": 1040.05 + }, + "latencies_ms": [ + 775.53, + 798.75, + 802.88, + 816.7, + 821.02, + 831.9, + 847.58, + 859.23, + 863.02, + 878.19, + 895.35, + 904.0, + 906.0, + 917.43, + 929.75, + 950.41, + 959.4, + 981.39, + 985.7, + 997.96, + 998.38, + 1000.59, + 1006.09, + 1020.74, + 1021.65, + 1027.36, + 1031.11, + 1031.61, + 1035.4, + 1040.05 + ], + "output_hashes": [ + "3d8c1c16e8a92518e4c15b7a995f7d1a426a52eb2d8d5c28a9815896f3804047", + "69aa1f345fd61e159aea5c99074045fd5acfacc44fccd344b9b3dd56d6955b8d" + ], + "wall_s": 1.086 + }, + "warm": { + "label": "oneshot-30-warm", + "renders": 30, + "failures": 0, + "cpu_total_s": 5.8232, + "cpu_per_render_ms": 194.106, + "latency_ms": { + "p50": 807.75, + "p95": 933.94, + "p99": 950.1 + }, + "latencies_ms": [ + 682.94, + 686.32, + 694.52, + 736.74, + 740.05, + 745.97, + 772.21, + 772.43, + 773.84, + 775.09, + 776.74, + 781.57, + 795.84, + 805.68, + 807.75, + 813.39, + 816.42, + 829.33, + 830.39, + 841.08, + 843.67, + 872.83, + 875.1, + 877.05, + 879.45, + 885.09, + 907.43, + 922.21, + 933.94, + 950.1 + ], + "output_hashes": [ + "3d8c1c16e8a92518e4c15b7a995f7d1a426a52eb2d8d5c28a9815896f3804047", + "69aa1f345fd61e159aea5c99074045fd5acfacc44fccd344b9b3dd56d6955b8d" + ], + "wall_s": 1.015 + } + }, + "idle-stop": { + "configured_minutes": 1, + "pid": 64669, + "exited_by_itself": true, + "observed_stop_after_s": 72.2, + "discovery_cleaned_up": true, + "still_alive_after_150s": false, + "lazy_restart_render_rc": 0, + "lazy_restart_rendered_line": true, + "lazy_restart_new_pid": 83669 + } + }, + "comparisons": [ + { + "sessions": 20, + "phase": "cold", + "oneshot_cpu_total_s": 3.7606, + "shared_cpu_total_s": 4.7399, + "cpu_reduction_pct": -26.0, + "oneshot_p95_ms": 480.86, + "shared_p95_ms": 901.87, + "shared_failures": 0, + "output_hashes_equal": true + }, + { + "sessions": 20, + "phase": "warm", + "oneshot_cpu_total_s": 3.6415, + "shared_cpu_total_s": 0.8478, + "cpu_reduction_pct": 76.7, + "oneshot_p95_ms": 451.13, + "shared_p95_ms": 271.16, + "shared_failures": 0, + "output_hashes_equal": true + }, + { + "sessions": 30, + "phase": "cold", + "oneshot_cpu_total_s": 5.5929, + "shared_cpu_total_s": 6.8912, + "cpu_reduction_pct": -23.2, + "oneshot_p95_ms": 1035.4, + "shared_p95_ms": 1149.48, + "shared_failures": 0, + "output_hashes_equal": true + }, + { + "sessions": 30, + "phase": "warm", + "oneshot_cpu_total_s": 5.8232, + "shared_cpu_total_s": 1.4831, + "cpu_reduction_pct": 74.5, + "oneshot_p95_ms": 933.94, + "shared_p95_ms": 480.82, + "shared_failures": 0, + "output_hashes_equal": true + } + ], + "idle_stop": { + "configured_minutes": 1, + "pid": 64669, + "exited_by_itself": true, + "observed_stop_after_s": 72.2, + "discovery_cleaned_up": true, + "still_alive_after_150s": false, + "lazy_restart_render_rc": 0, + "lazy_restart_rendered_line": true, + "lazy_restart_new_pid": 83669 + } +} \ No newline at end of file diff --git a/scripts/benchmark-lifecycle-53.py b/scripts/benchmark-lifecycle-53.py new file mode 100644 index 000000000..16f5b4286 --- /dev/null +++ b/scripts/benchmark-lifecycle-53.py @@ -0,0 +1,370 @@ +#!/usr/bin/env python3 +"""On-demand lifecycle benchmark (fork issue #53). + +Scenarios, all against the same mixed workload (sessions alternate between a +small and a large transcript and between a dirty and a clean repo, so the +daemon's caches hold a cold/warm mix): + + lazy-N N concurrent shared-mode clients, NO daemon pre-started: the + first wave includes the lazy start (#53), the second wave is + warm. Proves: one daemon serves the whole burst (stable pid, + bounded 503s, no failures), plus CPU/render and p95. + oneshot-N N concurrent one-shot renders (two waves) as the baseline. + idle-stop daemon started explicitly with daemonIdleStopMinutes=1 must + exit by itself with zero requests; the next render lazily + restarts it. + +Usage: + python3 scripts/benchmark-lifecycle-53.py --runtime bun \ + --entry src/ccstatusline.ts --sessions 20,30 --out docs +""" +import argparse, concurrent.futures, hashlib, json, math, os, pathlib, platform, resource, shutil, subprocess, sys, tempfile, time + +ROOT = pathlib.Path(tempfile.mkdtemp(prefix='ccsl-53bench-')).resolve() +CLIENT_SCRIPT = pathlib.Path(__file__).resolve().parent.parent / 'client' / 'ccstatusline-ipc' +FIXTURE_ROWS = {'small': 500, 'large': 10000} + + +def machine_context(): + try: + model = subprocess.run(['sysctl', '-n', 'machdep.cpu.brand_string'], capture_output=True, text=True).stdout.strip() + except OSError: + model = '' + return {'cpu_model': model or platform.machine(), 'cpu_count': os.cpu_count(), 'os': platform.platform(), 'python': sys.version.split()[0]} + + +def environment(home): + home.mkdir(parents=True, exist_ok=True) + return {'PATH': os.environ['PATH'], 'HOME': str(home), 'USERPROFILE': str(home), + 'CLAUDE_CONFIG_DIR': str(home / '.claude'), 'XDG_CONFIG_HOME': str(home / '.config'), + 'XDG_CACHE_HOME': str(home / '.cache'), 'TERM': 'xterm-256color', 'LANG': 'en_US.UTF-8', + 'TMPDIR': str(ROOT), 'CCSTATUSLINE_WIDTH': '120', 'CCSL_FORK': '1'} + + +def make_fixture(name): + path = ROOT / 'fixtures' / (name + '.jsonl') + if not path.exists(): + path.parent.mkdir(parents=True, exist_ok=True) + with path.open('w') as f: + for i in range(FIXTURE_ROWS[name]): + row = {'type': 'assistant' if i % 2 else 'user', 'timestamp': '2026-10-01T01:%02d:%02dZ' % ((i // 60) % 60, i % 60), + 'message': {'role': 'assistant' if i % 2 else 'user', 'content': [{'type': 'text', 'text': 'x' * 1105}]}} + if i % 2: + row['message'].update(id='msg-%d' % i, stop_reason='end_turn', + usage={'input_tokens': 100, 'output_tokens': 50, 'cache_read_input_tokens': 200, 'cache_creation_input_tokens': 10}) + f.write(json.dumps(row, separators=(',', ':')) + '\n') + return path + + +def make_repo(name, dirty): + repo = ROOT / ('repo-' + name) + if not (repo / '.git').exists(): + repo.mkdir(parents=True, exist_ok=True) + def git(*a): subprocess.run(['git', *a], cwd=repo, check=True, capture_output=True) + git('init') + git('config', 'user.email', 'bench@local') + git('config', 'user.name', 'bench') + (repo / 'app.js').write_text('const x = 1;\n' * 50) + git('add', '.') + git('commit', '-m', 'init') + if dirty: + (repo / 'app.js').write_text('const x = 2;\n' * 50) + (repo / 'notes.txt').write_text('untracked\n') + return repo + + +def read_discovery(path): + try: + text = path.read_text() + except OSError: + return None + info = {} + for line in text.splitlines(): + if line.startswith('#') or '=' not in line: + continue + key, _, value = line.partition('=') + info[key] = value + return info if info.get('socket') and info.get('token') else None + + +def daemon_health(info): + out = subprocess.run(['curl', '-q', '-sS', '--fail', '--noproxy', '*', '--max-time', '5', + '--unix-socket', info['socket'], '-H', 'Authorization: Bearer %s' % info['token'], + 'http://localhost/v1/health'], capture_output=True, text=True) + if out.returncode: + return None + try: + return json.loads(out.stdout) + except ValueError: + return None + + +def process_cpu_seconds(pid): + # macOS has no /proc; ps gives the cumulative CPU time of the process. + out = subprocess.run(['ps', '-p', str(pid), '-o', 'time='], capture_output=True, text=True) + if out.returncode or not out.stdout.strip(): + return 0.0 + parts = out.stdout.strip().split(':') + try: + if len(parts) == 3: + h, m, s = parts + return int(h) * 3600 + int(m) * 60 + float(s) + m, s = parts + return int(m) * 60 + float(s) + except ValueError: + return 0.0 + + +def pid_alive(pid): + try: + os.kill(pid, 0) + return True + except OSError: + return False + + +def pctl(sorted_xs, q): + return sorted_xs[min(len(sorted_xs) - 1, max(0, math.ceil(q * len(sorted_xs)) - 1))] + + +def mixed_sessions(sessions): + """(payload, cwd) per session: alternating transcript size and repo dirt.""" + small, large = make_fixture('small'), make_fixture('large') + dirty, clean = make_repo('dirty', True), make_repo('clean', False) + out = [] + for i in range(sessions): + fixture, repo = (small if i % 2 == 0 else large), (dirty if i % 2 == 0 else clean) + tx = ROOT / 'tx' / ('s%d.jsonl' % i) + tx.parent.mkdir(parents=True, exist_ok=True) + if not tx.exists(): + try: + os.link(fixture, tx) + except OSError: + import shutil + shutil.copy(fixture, tx) + payload = json.dumps({'model': {'id': 'claude-sonnet-4-5', 'display_name': 'Sonnet 4.5'}, + 'session_id': 'bench53-%d' % i, 'transcript_path': str(tx), + 'cwd': str(repo), 'workspace': {'current_dir': str(repo)}}) + out.append((payload, str(repo))) + return out + + +def wave(sessions, run_once): + """Fire every session concurrently once; returns per-render records.""" + with concurrent.futures.ThreadPoolExecutor(max_workers=len(sessions)) as pool: + return list(pool.map(lambda i: run_once(i, sessions[i]), range(len(sessions)))) + + +def summarize(label, records, cpu_total, extra=None): + latencies = sorted(r['latency_ms'] for r in records) + result = { + 'label': label, 'renders': len(records), 'failures': sum(1 for r in records if r['rc'] != 0), + 'cpu_total_s': round(cpu_total, 4), 'cpu_per_render_ms': round(cpu_total * 1000 / len(records), 3), + 'latency_ms': {'p50': round(pctl(latencies, 0.5), 2), 'p95': round(pctl(latencies, 0.95), 2), 'p99': round(pctl(latencies, 0.99), 2)}, + 'latencies_ms': [round(x, 2) for x in latencies], + 'output_hashes': sorted({r['hash'] for r in records if r['rc'] == 0}), + } + if extra: + result.update(extra) + return result + + +def scenario_oneshot(args, sessions_n): + work = mixed_sessions(sessions_n) + home = ROOT / 'homes' / ('oneshot-%d' % sessions_n) + env = environment(home) + entry = str(pathlib.Path(args.entry).resolve()) + + def once(i, item): + payload, cwd = item + start = time.perf_counter() + p = subprocess.run([args.runtime, entry], input=payload, text=True, stdout=subprocess.PIPE, + stderr=subprocess.PIPE, cwd=cwd, env=env, timeout=120) + return {'rc': p.returncode, 'latency_ms': (time.perf_counter() - start) * 1000, + 'hash': hashlib.sha256(p.stdout.encode()).hexdigest(), 'stderr': p.stderr[:200] if p.returncode else ''} + + wave(work, once) # wave 1: cold caches + before = resource.getrusage(resource.RUSAGE_CHILDREN) + start = time.perf_counter() + cold = wave(work, once) + mid = resource.getrusage(resource.RUSAGE_CHILDREN) + cold_cpu = mid.ru_utime + mid.ru_stime - before.ru_utime - before.ru_stime + cold_wall = time.perf_counter() - start + warm = wave(work, once) + after = resource.getrusage(resource.RUSAGE_CHILDREN) + warm_cpu = after.ru_utime + after.ru_stime - mid.ru_utime - mid.ru_stime + warm_wall = time.perf_counter() - start - cold_wall + # Wave 1 is cold for the fresh HOME; wave 2 is warm. Report both; the + # comparison table uses the warm pair plus the cold pair. + return { + 'cold': summarize('oneshot-%d-cold' % sessions_n, cold, cold_cpu, {'wall_s': round(cold_wall, 3)}), + 'warm': summarize('oneshot-%d-warm' % sessions_n, warm, warm_cpu, {'wall_s': round(warm_wall, 3)}), + } + + +def scenario_lazy(args, sessions_n): + work = mixed_sessions(sessions_n) + home = ROOT / 'homes' / ('lazy-%d' % sessions_n) + env = environment(home) + runtime_dir = pathlib.Path('/tmp') / ('ccsl-53bench-%d' % os.getpid()) + if runtime_dir.exists(): + shutil.rmtree(runtime_dir) + client_env = dict(env, CCSTATUSLINE_RUNTIME_DIR=str(runtime_dir)) + discovery = runtime_dir / 'daemon.env' + + def once(i, item): + payload, cwd = item + start = time.perf_counter() + p = subprocess.run(['/bin/sh', str(CLIENT_SCRIPT)], input=payload, text=True, stdout=subprocess.PIPE, + stderr=subprocess.PIPE, cwd=cwd, env=client_env, timeout=120) + return {'rc': p.returncode, 'latency_ms': (time.perf_counter() - start) * 1000, + 'hash': hashlib.sha256(p.stdout.encode()).hexdigest(), 'stderr': p.stderr[:200] if p.returncode else ''} + + # Wave 1 (cold): no daemon exists. First clients race the lazy start. + before = resource.getrusage(resource.RUSAGE_CHILDREN) + start = time.perf_counter() + lazy_ready_at = None + cold = wave(work, once) + cold_wall = time.perf_counter() - start + mid = resource.getrusage(resource.RUSAGE_CHILDREN) + cold_cpu = mid.ru_utime + mid.ru_stime - before.ru_utime - before.ru_stime + info = read_discovery(discovery) + cold_pids = [info['pid']] if info else [] + # The detached daemon escapes RUSAGE_CHILDREN (reparented when the + # starting wrapper exits) — account for its CPU via ps. + daemon_pid = int(info['pid']) if info else None + daemon_cpu_after_cold = process_cpu_seconds(daemon_pid) if daemon_pid else 0.0 + + warm = wave(work, once) + after = resource.getrusage(resource.RUSAGE_CHILDREN) + warm_client_cpu = after.ru_utime + after.ru_stime - mid.ru_utime - mid.ru_stime + warm_wall = time.perf_counter() - start - cold_wall + info2 = read_discovery(discovery) + warm_pids = [info2['pid']] if info2 else [] + daemon_cpu_total = process_cpu_seconds(daemon_pid) if daemon_pid else 0.0 + health = daemon_health(info2) if info2 else None + + cold_total = cold_cpu + daemon_cpu_after_cold + warm_total = warm_client_cpu + (daemon_cpu_total - daemon_cpu_after_cold) + extra = { + 'wall_s': round(cold_wall, 3), + 'daemon_pids_cold_wave': cold_pids, 'daemon_pids_warm_wave': warm_pids, + 'single_daemon_across_waves': bool(cold_pids and warm_pids and cold_pids == warm_pids), + 'daemon_cpu_total_s': round(daemon_cpu_total, 4), + 'daemon_counters': health.get('counters') if health else None, + } + # Terminate the daemon so later scenarios start clean. + if daemon_pid and pid_alive(daemon_pid): + os.kill(daemon_pid, 15) + deadline = time.time() + 10 + while pid_alive(daemon_pid) and time.time() < deadline: + time.sleep(0.05) + shutil.rmtree(runtime_dir, ignore_errors=True) + return { + 'cold': summarize('lazy-%d-cold' % sessions_n, cold, cold_total, extra), + 'warm': summarize('lazy-%d-warm' % sessions_n, warm, warm_total, {'wall_s': round(warm_wall, 3)}), + } + + +def scenario_idle_stop(args): + """daemonIdleStopMinutes=1: with zero requests the daemon must exit by itself.""" + home = ROOT / 'homes' / 'idle' + env = environment(home) + config_dir = home / '.config' / 'ccstatusline' + config_dir.mkdir(parents=True, exist_ok=True) + # Minimal settings: the schema fills defaults; the idle bound is the point. + (config_dir / 'settings.json').write_text(json.dumps( + {'version': 4, 'lines': [[{'id': '1', 'type': 'model'}]], 'daemonIdleStopMinutes': 1})) + runtime_dir = pathlib.Path('/tmp') / ('ccsl-53bench-idle-%d' % os.getpid()) + run_env = dict(env, CCSTATUSLINE_RUNTIME_DIR=str(runtime_dir)) + subprocess.run([args.runtime, str(pathlib.Path(args.entry).resolve()), 'daemon', 'start'], + env=run_env, capture_output=True, text=True, timeout=60) + discovery = runtime_dir / 'daemon.env' + info = read_discovery(discovery) + if not info: + return {'error': 'daemon did not start'} + pid = int(info['pid']) + started = time.perf_counter() + while time.perf_counter() - started < 150: + if not pid_alive(pid): + break + time.sleep(1) + exited = not pid_alive(pid) + result = {'configured_minutes': 1, 'pid': pid, 'exited_by_itself': exited, + 'observed_stop_after_s': round(time.perf_counter() - started, 1), + 'discovery_cleaned_up': read_discovery(discovery) is None, + 'still_alive_after_150s': not exited} + # The next render lazily restarts it (acceptance: transparent restart). + payload, cwd = mixed_sessions(1)[0] + p = subprocess.run(['/bin/sh', str(CLIENT_SCRIPT)], input=payload, text=True, stdout=subprocess.PIPE, + stderr=subprocess.PIPE, cwd=cwd, env=run_env, timeout=120) + restarted = read_discovery(discovery) + result['lazy_restart_render_rc'] = p.returncode + result['lazy_restart_rendered_line'] = bool(p.stdout.strip()) + result['lazy_restart_new_pid'] = int(restarted['pid']) if restarted else None + new_pid = int(restarted['pid']) if restarted else None + if new_pid and pid_alive(new_pid): + os.kill(new_pid, 15) + shutil.rmtree(runtime_dir, ignore_errors=True) + return result + + +def main(): + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument('--runtime', default='bun') + parser.add_argument('--entry', default='src/ccstatusline.ts') + parser.add_argument('--sessions', default='20,30') + parser.add_argument('--out', default='docs') + args = parser.parse_args() + + runs = {} + for n in [int(s) for s in args.sessions.split(',')]: + print('[53-bench] lazy-%d: running...' % n, flush=True) + runs['lazy-%d' % n] = scenario_lazy(args, n) + print('[53-bench] lazy-%d done: %s' % (n, json.dumps({k: {kk: vv for kk, vv in v.items() if kk != 'latencies_ms'} for k, v in runs['lazy-%d' % n].items()})), flush=True) + print('[53-bench] oneshot-%d: running...' % n, flush=True) + runs['oneshot-%d' % n] = scenario_oneshot(args, n) + print('[53-bench] oneshot-%d done: %s' % (n, json.dumps({k: {kk: vv for kk, vv in v.items() if kk != 'latencies_ms'} for k, v in runs['oneshot-%d' % n].items()})), flush=True) + print('[53-bench] idle-stop: running (waits ~1 minute)...', flush=True) + runs['idle-stop'] = scenario_idle_stop(args) + print('[53-bench] idle-stop done: %s' % json.dumps(runs['idle-stop']), flush=True) + + comparisons = [] + for n in [int(s) for s in args.sessions.split(',')]: + for phase in ('cold', 'warm'): + one = runs['oneshot-%d' % n][phase] + shared = runs['lazy-%d' % n][phase] + comparisons.append({ + 'sessions': n, 'phase': phase, + 'oneshot_cpu_total_s': one['cpu_total_s'], 'shared_cpu_total_s': shared['cpu_total_s'], + 'cpu_reduction_pct': round((one['cpu_total_s'] - shared['cpu_total_s']) / one['cpu_total_s'] * 100, 1) if one['cpu_total_s'] else None, + 'oneshot_p95_ms': one['latency_ms']['p95'], 'shared_p95_ms': shared['latency_ms']['p95'], + 'shared_failures': shared['failures'], + 'output_hashes_equal': one['output_hashes'] == shared['output_hashes'], + }) + doc = {'meta': {'runtime': args.runtime, 'entry': str(pathlib.Path(args.entry).resolve()), + 'machine': machine_context(), 'generated': time.strftime('%Y-%m-%dT%H:%M:%S%z'), + 'workload': 'sessions alternate small(500-row)/large(10000-row) transcripts and dirty/clean repos; two waves per scenario (cold includes the lazy start)', + 'notes': [ + 'Lazy/shared CPU covers the daemon AND every client: clients via getrusage(RUSAGE_CHILDREN), the detached daemon via ps cumulative CPU (it escapes RUSAGE_CHILDREN after reparenting).', + 'Cold wave of the lazy scenario includes the on-demand daemon start paid by the first clients.', + ]}, + 'runs': runs, 'comparisons': comparisons, 'idle_stop': runs['idle-stop']} + out = pathlib.Path(args.out) / 'daemon-53-results.json' + out.write_text(json.dumps(doc, indent=2)) + print('[53-bench] results: %s' % out, flush=True) + for c in comparisons: + print('[53-bench] s%d %-4s cpu/render one-shot=%sms shared=%sms (%s%%) p95 %s→%sms failures=%s hashes_equal=%s' % ( + c['sessions'], c['phase'], + round(runs['oneshot-%d' % c['sessions']][c['phase']]['cpu_per_render_ms'], 1), + round(runs['lazy-%d' % c['sessions']][c['phase']]['cpu_per_render_ms'], 1), + c['cpu_reduction_pct'], c['oneshot_p95_ms'], c['shared_p95_ms'], + c['shared_failures'], c['output_hashes_equal']), flush=True) + failures = sum(r[phase]['failures'] for r in runs.values() if isinstance(r, dict) and 'cold' in r for phase in ('cold', 'warm')) + ok = failures == 0 and runs['idle-stop'].get('exited_by_itself') and runs['idle-stop'].get('lazy_restart_render_rc') == 0 + print('[53-bench] %s' % ('PASS' if ok else 'FAIL'), flush=True) + return 0 if ok else 1 + + +if __name__ == '__main__': + sys.exit(main()) diff --git a/src/daemon/__tests__/client.test.ts b/src/daemon/__tests__/client.test.ts index e89e37626..8f6614147 100644 --- a/src/daemon/__tests__/client.test.ts +++ b/src/daemon/__tests__/client.test.ts @@ -12,6 +12,7 @@ import { import type { StartedTestDaemon } from './test-daemon'; import { + MODEL_ONLY_SETTINGS, startTestDaemon, stopTestDaemon } from './test-daemon'; @@ -22,10 +23,15 @@ import { const clientPath = fileURLToPath(new URL('../../../client/ccstatusline-ipc', import.meta.url)); +let payloadCounter = 0; + const started: StartedTestDaemon[] = []; -async function start(overrides: Parameters[0] = {}): Promise { - const handle = await startTestDaemon(overrides); +async function start( + overrides: Parameters[0] = {}, + options: Parameters[1] = {} +): Promise { + const handle = await startTestDaemon(overrides, options); started.push(handle); return handle; } @@ -47,7 +53,9 @@ async function runClient(runtimeDir: string, payload: string, extraEnv: Record { const child = spawn('/bin/sh', [ @@ -137,10 +145,10 @@ describe('shell client end to end', () => { expect(result.stderr).toBe(''); }); - it('fails with empty stdout when the daemon is not running', async () => { + it('fails with empty stdout when the daemon is not running and autostart is disabled', async () => { const runtimeDir = fs.mkdtempSync(path.join(os.tmpdir(), 'ccsd-client-none-')); try { - const result = await runClient(runtimeDir, '{}'); + const result = await runClient(runtimeDir, '{}', { CCSTATUSLINE_NO_AUTOSTART: '1' }); expect(result.status).not.toBe(0); expect(result.stdout).toBe(''); @@ -150,6 +158,23 @@ describe('shell client end to end', () => { } }); + it('attempts a lazy start when no daemon is running and autostart is on', async () => { + const runtimeDir = fs.mkdtempSync(path.join(os.tmpdir(), 'ccsd-client-lazy-')); + try { + // The lazy-start command records that it ran, then fails to + // produce a daemon: the client must still die with empty stdout + // (never a partial line), but the start was attempted. + const marker = path.join(runtimeDir, 'started'); + const result = await runClient(runtimeDir, '{}', { CCSTATUSLINE_DAEMON_START: `touch ${marker}` }); + + expect(fs.existsSync(marker)).toBe(true); + expect(result.status).not.toBe(0); + expect(result.stdout).toBe(''); + } finally { + fs.rmSync(runtimeDir, { recursive: true, force: true }); + } + }); + it('fails with empty stdout on a wrong token (HTTP 401) instead of printing the error body', async () => { const { daemon } = await start(); fs.writeFileSync(daemon.discoveryPath, fs.readFileSync(daemon.discoveryPath, 'utf8').replace(daemon.token, 'a'.repeat(64))); @@ -172,7 +197,7 @@ describe('shell client end to end', () => { `token=${daemon.token}` ].join('\n') + '\n', { mode: 0o600 }); - const result = await runClient(daemon.runtimeDir, JSON.stringify({ model: { id: 'm' }, cwd: '/tmp' })); + const result = await runClient(daemon.runtimeDir, JSON.stringify({ model: { id: 'm' }, cwd: '/tmp' }), { CCSTATUSLINE_NO_AUTOSTART: '1' }); expect(result.status).not.toBe(0); expect(result.stdout).toBe(''); @@ -189,3 +214,161 @@ describe('shell client end to end', () => { expect(result.stderr).toContain('protocol'); }); }); + +// On-demand lifecycle (#53): the shared-mode client lazily starts a daemon +// (serialized by the #47 cold-start lock inside `daemon start`) and retries +// the render into it. The "daemon start" command is replaced with a stub that +// republishes the in-process test daemon's discovery — what is under test is +// the client's retry logic, not the server itself. +describe('shell client on-demand start (#53)', () => { + /** A start stub: after a cold-start delay, publish $CCSD_DISCOVERY as the discovery. */ + function writeStartStub(dir: string, delayMs: number): string { + const stubPath = path.join(dir, 'start-stub.sh'); + fs.writeFileSync(stubPath, [ + '#!/bin/sh', + `sleep ${(delayMs / 1000).toFixed(2)}`, + 'printf \'%s\' "$CCSD_DISCOVERY" > "$CCSTATUSLINE_RUNTIME_DIR/daemon.env"' + ].join('\n'), { mode: 0o700 }); + return stubPath; + } + + async function runLazyClient(dir: string, payload: string, stubPath: string, discovery: string): Promise<{ + status: number; + stdout: string; + stderr: string; + }> { + return runClient(dir, payload, { + CCSTATUSLINE_DAEMON_START: `sh ${stubPath}`, + CCSD_DISCOVERY: discovery + }); + } + + it('lazily starts the daemon and retries the render into it', async () => { + const { daemon } = await start(); + const discovery = fs.readFileSync(daemon.discoveryPath, 'utf8'); + const stub = writeStartStub(daemon.runtimeDir, 300); + + // Simulate the idle-stopped state: no daemon endpoints on disk. + fs.unlinkSync(daemon.discoveryPath); + const result = await runLazyClient(daemon.runtimeDir, JSON.stringify({ model: { id: 'claude-lazy-model' }, cwd: '/tmp' }), stub, discovery); + + expect(result.status).toBe(0); + expect(result.stdout).toContain('claude-lazy-model'); + // The stub republished exactly the discovery it was given. + expect(fs.readFileSync(daemon.discoveryPath, 'utf8')).toBe(discovery); + }); + + it('concurrent cold clients all render through the lazily started daemon', async () => { + const { daemon, dependencies } = await start(); + const discovery = fs.readFileSync(daemon.discoveryPath, 'utf8'); + const stub = writeStartStub(daemon.runtimeDir, 400); + fs.unlinkSync(daemon.discoveryPath); + + const dirs = Array.from({ length: 5 }, () => fs.mkdtempSync(path.join(os.tmpdir(), 'ccsd-client-race-'))); + try { + const results = await Promise.all(dirs.map(dir => runLazyClient( + dir, + JSON.stringify({ model: { id: `claude-race-${dir.slice(-6)}` }, cwd: '/tmp' }), + stub, + discovery + ))); + + for (const result of results) { + expect(result.status).toBe(0); + expect(result.stdout).not.toBe(''); + } + // One daemon served every racing client — no second server, no + // lost renders. + expect(dependencies.invocations).toHaveLength(5); + expect(daemon.counters.ok).toBe(5); + } finally { + for (const dir of dirs) { + fs.rmSync(dir, { recursive: true, force: true }); + } + } + }); + + it('retries into a fresh daemon when the discovered one died mid-flight', async () => { + // Daemon A: dead, but its discovery still names the dead socket (the + // client must hit the connect failure and restart). Daemon B: live, + // published by the start stub into A's runtime dir. + const dead = await start(); + await dead.daemon.stop(); + fs.writeFileSync(dead.daemon.discoveryPath, [ + 'protocol=1', + `socket=${dead.daemon.socketPath}`, + `token=${dead.daemon.token}` + ].join('\n') + '\n', { mode: 0o600 }); + + const live = await start(); + const liveDiscovery = fs.readFileSync(live.daemon.discoveryPath, 'utf8'); + const stub = writeStartStub(live.daemon.runtimeDir, 100); + + const result = await runClient(dead.daemon.runtimeDir, JSON.stringify({ model: { id: 'claude-restart-model' }, cwd: '/tmp' }), { + CCSTATUSLINE_DAEMON_START: `sh ${stub}`, + CCSD_DISCOVERY: liveDiscovery + }); + + expect(result.status).toBe(0); + expect(result.stdout).toContain('claude-restart-model'); + expect(live.dependencies.invocations).toHaveLength(1); + }); + + it('rides out a 503 burst through the retry ladder and renders every client', async () => { + const { daemon } = await start({ + loadSettings: async () => { + await new Promise((resolve) => { setTimeout(resolve, 250); }); + return { settings: { ...MODEL_ONLY_SETTINGS }, loadError: null }; + } + }, { maxInFlightRenders: 1 }); + const first = runClient(daemon.runtimeDir, JSON.stringify({ model: { id: 'claude-burst-a' }, cwd: '/tmp' })); + const second = runClient(daemon.runtimeDir, JSON.stringify({ model: { id: 'claude-burst-b' }, cwd: '/tmp' })); + const [a, b] = await Promise.all([first, second]); + + expect(a.status).toBe(0); + expect(b.status).toBe(0); + expect(a.stdout).toContain('claude-burst-a'); + expect(b.stdout).toContain('claude-burst-b'); + }); + + it('never starts a daemon from the one-shot render path (opt-in gating)', async () => { + const runtimeDir = fs.mkdtempSync(path.join(os.tmpdir(), 'ccsd-gate-')); + const repoRoot = path.resolve(path.dirname(clientPath), '..'); + const payload = JSON.stringify({ model: { id: 'claude-gate-model' }, cwd: repoRoot }); + try { + const result = await new Promise<{ status: number | null; stdout: string; stderr: string }>((resolve, reject) => { + const child = spawn(process.execPath, [path.join(repoRoot, 'src', 'ccstatusline.ts')], { + cwd: repoRoot, + env: { + ...process.env, + CCSTATUSLINE_RUNTIME_DIR: runtimeDir, + CLAUDE_CONFIG_DIR: path.join(runtimeDir, 'claude'), + XDG_CONFIG_HOME: path.join(runtimeDir, 'config') + } + }); + const timeout = setTimeout(() => { + child.kill('SIGKILL'); + reject(new Error('one-shot render timed out')); + }, 20000); + let stdout = ''; + let stderr = ''; + child.stdout.on('data', (chunk: Buffer) => { stdout += chunk.toString('utf8'); }); + child.stderr.on('data', (chunk: Buffer) => { stderr += chunk.toString('utf8'); }); + child.on('error', reject); + child.on('close', (code) => { + clearTimeout(timeout); + resolve({ status: code, stdout, stderr }); + }); + // Piped payload, then EOF — how Claude Code drives one-shot mode. + child.stdin.end(payload); + }); + + expect(result.status).toBe(0); + expect(result.stdout).toContain('claude-gate-model'); + // No daemon was spawned: no discovery, no startup lock, no socket. + expect(fs.readdirSync(runtimeDir)).toEqual([]); + } finally { + fs.rmSync(runtimeDir, { recursive: true, force: true }); + } + }); +}); diff --git a/src/daemon/__tests__/idle-stop.test.ts b/src/daemon/__tests__/idle-stop.test.ts new file mode 100644 index 000000000..34ba10edb --- /dev/null +++ b/src/daemon/__tests__/idle-stop.test.ts @@ -0,0 +1,83 @@ +import * as fs from 'node:fs'; +import * as http from 'node:http'; +import { + afterEach, + describe, + expect, + it +} from 'vitest'; + +import type { StartedTestDaemon } from './test-daemon'; +import { + renderRequest, + startTestDaemon, + stopTestDaemon, + waitFor +} from './test-daemon'; + +// Idle auto-stop (#53): after the configured window with zero requests the +// daemon takes itself down — endpoints released, onIdleStop fired (the host +// process exits there in production). Any request resets the clock, and 0 +// disables the mechanism outright. + +const started: StartedTestDaemon[] = []; + +async function start(options: { idleStopMs?: number; onIdleStop?: () => void } = {}): Promise { + const handle = await startTestDaemon({}, options); + started.push(handle); + return handle; +} + +afterEach(async () => { + for (const handle of started.splice(0)) { + await stopTestDaemon(handle); + } +}); + +function socketAnswers(handle: StartedTestDaemon): Promise { + return new Promise((resolve) => { + const request = http.request({ socketPath: handle.daemon.socketPath, method: 'GET', path: '/v1/health' }); + request.on('error', () => { resolve(false); }); + // Any response at all means something still serves the endpoint. + request.on('response', () => { resolve(true); }); + request.end(); + }); +} + +describe('daemon idle auto-stop (#53)', () => { + it('stops itself after the idle window with zero requests', async () => { + let fired = false; + const handle = await start({ idleStopMs: 300, onIdleStop: () => { fired = true; } }); + + await waitFor(() => !fs.existsSync(handle.daemon.discoveryPath), 4000); + + expect(fired).toBe(true); + expect(await socketAnswers(handle)).toBe(false); + }); + + it('a request keeps the daemon alive past the idle window', async () => { + const handle = await start({ idleStopMs: 500 }); + + // A render at t=300ms resets the idle clock; the daemon must still be + // serving at t=650ms (past the original deadline) and stop only after + // its own idle window restarts. + await new Promise((resolve) => { setTimeout(resolve, 300); }); + const response = await renderRequest(handle.daemon, handle.daemon.token); + expect(response.status).toBe(200); + + await new Promise((resolve) => { setTimeout(resolve, 350); }); + expect(fs.existsSync(handle.daemon.discoveryPath)).toBe(true); + + await waitFor(() => !fs.existsSync(handle.daemon.discoveryPath), 4000); + }); + + it('idleStopMs 0 disables the auto-stop', async () => { + const handle = await start({ idleStopMs: 0 }); + + await new Promise((resolve) => { setTimeout(resolve, 400); }); + + expect(fs.existsSync(handle.daemon.discoveryPath)).toBe(true); + const response = await renderRequest(handle.daemon, handle.daemon.token); + expect(response.status).toBe(200); + }); +}); diff --git a/src/daemon/__tests__/test-daemon.ts b/src/daemon/__tests__/test-daemon.ts index 1607d3605..2bfebabb4 100644 --- a/src/daemon/__tests__/test-daemon.ts +++ b/src/daemon/__tests__/test-daemon.ts @@ -68,7 +68,7 @@ export interface StartedTestDaemon { export async function startTestDaemon( overrides: Partial = {}, - options: { maxInFlightRenders?: number } = {} + options: { maxInFlightRenders?: number; idleStopMs?: number; onIdleStop?: () => void } = {} ): Promise { const runtimeDir = makeTempRuntimeDir(); const previousRuntimeDir = process.env.CCSTATUSLINE_RUNTIME_DIR; @@ -76,7 +76,9 @@ export async function startTestDaemon( const dependencies = hermeticDependencies(overrides); const daemon = createDaemonServer({ dependencies, - ...(options.maxInFlightRenders === undefined ? {} : { maxInFlightRenders: options.maxInFlightRenders }) + ...(options.maxInFlightRenders === undefined ? {} : { maxInFlightRenders: options.maxInFlightRenders }), + ...(options.idleStopMs === undefined ? {} : { idleStopMs: options.idleStopMs }), + ...(options.onIdleStop === undefined ? {} : { onIdleStop: options.onIdleStop }) }); await daemon.start(); return { daemon, dependencies, runtimeDir, previousRuntimeDir }; diff --git a/src/daemon/server.ts b/src/daemon/server.ts index 10079d088..408a1a8c8 100644 --- a/src/daemon/server.ts +++ b/src/daemon/server.ts @@ -70,6 +70,13 @@ export interface DaemonServerOptions { maxInFlightRenders?: number; /** Test seam (#17): reported build identity in health and the discovery file. */ versionOverride?: string; + /** + * Idle auto-stop (#53): exit after this many milliseconds with zero + * requests (any request, health included, resets the clock). 0 disables. + */ + idleStopMs?: number; + /** Called after the idle stop shut the server down (the host exits here). */ + onIdleStop?: () => void; } export interface DaemonCounters { @@ -144,6 +151,8 @@ interface RenderJob { // gives the memory back after quiet periods. const IDLE_SWEEP_INTERVAL_MS = 60_000; const IDLE_SWEEP_AFTER_MS = 5 * 60_000; +/** Default idle auto-stop (#53): 10 minutes with zero requests. */ +export const DEFAULT_IDLE_STOP_MS = 10 * 60_000; function sweepProviderCaches(): void { clearGitCache(); @@ -170,15 +179,31 @@ export function createDaemonServer(options: DaemonServerOptions): DaemonServerHa const prefetchState: PrefetchState = createPrefetchState(); const renderJobs = new Map(); + const idleStopMs = options.idleStopMs ?? DEFAULT_IDLE_STOP_MS; + // The sweep cadence adapts to a short idle bound so tests (and tiny + // configured values) do not wait a full minute for the first check. + const idleCheckIntervalMs = idleStopMs > 0 + ? Math.min(IDLE_SWEEP_INTERVAL_MS, Math.max(100, Math.floor(idleStopMs / 5))) + : IDLE_SWEEP_INTERVAL_MS; + let idleStopPending = false; const idleSweeper = setInterval(() => { - if (Date.now() - lastActivityAt > IDLE_SWEEP_AFTER_MS && renderJobs.size === 0) { + const idleFor = Date.now() - lastActivityAt; + if (idleFor > IDLE_SWEEP_AFTER_MS && renderJobs.size === 0) { sweepProviderCaches(); for (const cache of prefetchState.usageCaches.values()) { cache.data = null; cache.identity = undefined; } } - }, IDLE_SWEEP_INTERVAL_MS); + // Idle auto-stop (#53): zero requests for the whole window and no + // render in flight — the daemon takes itself down. Busy periods keep + // it alive for free: every request bumps lastActivityAt. + if (idleStopMs > 0 && !idleStopPending && idleFor > idleStopMs && renderJobs.size === 0) { + idleStopPending = true; + clearInterval(idleSweeper); + void stopServer().then(() => { options.onIdleStop?.(); }); + } + }, idleCheckIntervalMs); idleSweeper.unref(); /** Join an in-flight render for `key`, or start one. */ @@ -486,6 +511,40 @@ export function createDaemonServer(options: DaemonServerOptions): DaemonServerHa fs.renameSync(tempPath, discoveryPath); } + async function stopServer(): Promise { + clearInterval(idleSweeper); + await new Promise((resolve) => { + server.closeIdleConnections(); + server.close(() => { resolve(); }); + }); + // Remove the endpoints only if this instance still owns them: + // prepareSocketPath lets a same-user daemon take over the socket + // path, and the takeover rewrites the discovery file in the same + // breath. A superseded instance shutting down must not delete the + // live one's socket or discovery. (Inode comparison is not an + // option: the freed inode is routinely reused by the new socket.) + let owned = false; + try { + owned = fs.readFileSync(discoveryPath, 'utf8').includes(`token=${token}\n`); + } catch { + // Gone already; nothing to clean up either way. + } + if (owned) { + for (const stalePath of [socketPath, discoveryPath]) { + try { + fs.unlinkSync(stalePath); + } catch { + // Already gone. + } + } + } + try { + fs.unlinkSync(`${discoveryPath}.${process.pid}.tmp`); + } catch { + // Already gone. + } + } + return { socketPath, discoveryPath, @@ -513,46 +572,13 @@ export function createDaemonServer(options: DaemonServerOptions): DaemonServerHa fs.chmodSync(socketPath, 0o600); writeDiscoveryFile(); }, - async stop(): Promise { - clearInterval(idleSweeper); - await new Promise((resolve) => { - server.closeIdleConnections(); - server.close(() => { resolve(); }); - }); - // Remove the endpoints only if this instance still owns them: - // prepareSocketPath lets a same-user daemon take over the socket - // path, and the takeover rewrites the discovery file in the same - // breath. A superseded instance shutting down must not delete the - // live one's socket or discovery. (Inode comparison is not an - // option: the freed inode is routinely reused by the new socket.) - let owned = false; - try { - owned = fs.readFileSync(discoveryPath, 'utf8').includes(`token=${token}\n`); - } catch { - // Gone already; nothing to clean up either way. - } - if (owned) { - for (const stalePath of [socketPath, discoveryPath]) { - try { - fs.unlinkSync(stalePath); - } catch { - // Already gone. - } - } - } - try { - fs.unlinkSync(`${discoveryPath}.${process.pid}.tmp`); - } catch { - // Already gone. - } - } + stop: stopServer }; } /** * Daemon host entry (`ccstatusline daemon`, #16): starts the transport and - * blocks until SIGINT/SIGTERM. Lifecycle (auto-start, respawn) is the - * sibling issue's scope; this is the foreground server. The bearer token is + * blocks until SIGINT/SIGTERM or the idle auto-stop (#53). The bearer token is * never logged — it is shared only through the discovery file. */ export async function runDaemonServer(): Promise { @@ -561,7 +587,16 @@ export async function runDaemonServer(): Promise { process.exit(1); } - const daemon = createDaemonServer({ dependencies: createProcessDaemonDependencies() }); + const daemon = createDaemonServer({ + dependencies: createProcessDaemonDependencies(), + // Idle auto-stop (#53): the configured minutes with zero requests end + // the process; an unreadable config keeps the 10-minute default. + idleStopMs: (await loadSettingsFrom(getConfigPath())).settings.daemonIdleStopMinutes * 60_000, + onIdleStop: () => { + console.error('ccstatusline daemon: idle timeout reached, shutting down'); + process.exit(0); + } + }); try { await daemon.start(); } catch (error) { diff --git a/src/types/Settings.ts b/src/types/Settings.ts index 812a4d8f3..83df98eff 100644 --- a/src/types/Settings.ts +++ b/src/types/Settings.ts @@ -117,6 +117,10 @@ export const SettingsSchema = z.object({ remaining: z.number().nullable().optional() }).optional(), installation: InstallationMetadataSchema.optional(), + // On-demand daemon lifecycle (#53): minutes with zero requests before the + // daemon exits by itself. 0 disables idle auto-stop. Additive key with a + // default, so configs written before it behave identically. + daemonIdleStopMinutes: z.number().min(0).max(1440).default(10), // Daemon shared mode (#19): present while the Claude Code statusLine // points at the IPC client wrapper. Remembers the exact previous statusLine // so returning to one-shot mode restores it verbatim; null previous means diff --git a/src/utils/claude-settings.ts b/src/utils/claude-settings.ts index 030a7e9bd..24ab301eb 100644 --- a/src/utils/claude-settings.ts +++ b/src/utils/claude-settings.ts @@ -517,10 +517,11 @@ export async function getExistingStatusLine(): Promise { // Daemon shared mode (#19): strictly opt-in switch of the Claude Code // statusLine command to the IPC client wrapper, with a tested return path. // -// The wrapper (`client/ccstatusline-ipc`, shipped in the npm package) never -// starts a daemon: with the daemon down it fails with empty stdout and the -// status line simply does not render. Starting one is always an explicit -// `ccstatusline daemon start|install` — nothing on the render path spawns it. +// The wrapper (`client/ccstatusline-ipc`, shipped in the npm package) is the +// only path that may start a daemon, and it runs exclusively in installed +// shared mode (#53 on-demand lifecycle): with the daemon down it lazily runs +// `daemon start` and retries into the fresh daemon. The one-shot render path +// never spawns anything — opting in is still only `daemon install`. // --------------------------------------------------------------------------- /** Locate the shipped IPC client wrapper, or null when this install has none. */ diff --git a/src/widgets/__tests__/CurrentWorkingDir.test.ts b/src/widgets/__tests__/CurrentWorkingDir.test.ts index 27ce7283b..b2c93b798 100644 --- a/src/widgets/__tests__/CurrentWorkingDir.test.ts +++ b/src/widgets/__tests__/CurrentWorkingDir.test.ts @@ -37,6 +37,7 @@ describe('CurrentWorkingDirWidget', () => { lines: [], flexMode: 'full', compactThreshold: 60, + daemonIdleStopMinutes: 10, colorLevel: 2, defaultPadding: ' ', defaultPaddingSide: 'both', diff --git a/src/widgets/__tests__/CustomCommand.test.ts b/src/widgets/__tests__/CustomCommand.test.ts index 841fac97e..bd329eaec 100644 --- a/src/widgets/__tests__/CustomCommand.test.ts +++ b/src/widgets/__tests__/CustomCommand.test.ts @@ -68,6 +68,7 @@ describe('CustomCommandWidget', () => { lines: [], flexMode: 'full', compactThreshold: 60, + daemonIdleStopMinutes: 10, colorLevel: 2, defaultPadding: ' ', defaultPaddingSide: 'both',