From dc9a8213efbb871b71dd20de2081cb76ac486c45 Mon Sep 17 00:00:00 2001 From: Artjoms Stukans Date: Fri, 25 Sep 2026 15:55:48 +0200 Subject: [PATCH 01/40] feat(ipc): add multi-client sync messages and output hold flag Adds the wire types phase 1 of the multi-client work needs: a client id on attach, batched resize and pane-size frames, client geometry, take control, list clients, and broadcast dismiss/seen marks. A conn can now be held off pane_output, which the daemon uses to deliver a new client's replay and live output exactly once. --- internal/ipc/protocol.go | 119 ++++++++++++- internal/ipc/protocol_test.go | 210 +++++++++++++++++++++++ internal/ipc/server.go | 26 ++- internal/ipc/subscribe_broadcast_test.go | 115 +++++++++++++ internal/ipc/subscribe_test.go | 24 +++ 5 files changed, 489 insertions(+), 5 deletions(-) diff --git a/internal/ipc/protocol.go b/internal/ipc/protocol.go index f9f3d980..f6960b02 100644 --- a/internal/ipc/protocol.go +++ b/internal/ipc/protocol.go @@ -11,7 +11,7 @@ import ( const ( // Lifecycle MsgAttach = "attach" - MsgDetach = "detach" + MsgDetach = "detach" // multi-client sync: also means a clean client exit MsgShutdown = "shutdown" MsgHeartbeat = "heartbeat" // MsgSubscribe lets a client narrow what the daemon broadcasts to it. @@ -282,6 +282,24 @@ const ( MsgWaitTaskResp = "wait_task_resp" MsgListTasksReq = "list_tasks_req" MsgListTasksResp = "list_tasks_resp" + + // Multi-client sync: several TUIs attached to the same daemon, one of + // them elected master and the rest following its geometry. + // + // MsgDetach, declared above, gains a second meaning here: a follower or + // the master sending it now also means "clean client exit", so the + // daemon can drop it from master election and the client list without + // waiting on the conn to close. + MsgPaneSizes = "pane_sizes" // daemon → follower clients (must-deliver) + MsgResizePanes = "resize_panes" // client → daemon, batched resize + MsgClientGeometry = "client_geometry" // client → daemon, raw window size + MsgTakeControl = "take_control" // client → daemon, no payload + + MsgListClientsReq = "list_clients_req" + MsgListClientsResp = "list_clients_resp" + + MsgEventDismissed = "event_dismissed" // daemon → clients + MsgPaneSeen = "pane_seen" // daemon → clients ) // Message is the wire format for IPC communication. @@ -303,6 +321,12 @@ type AttachPayload struct { Cols int `json:"cols"` Rows int `json:"rows"` CWD string `json:"cwd,omitempty"` + // ClientID identifies this client across reconnects, for multi-client + // sync: master election, the client list and per-client geometry all key + // on it. Empty on an older client, which the daemon treats as a client + // that cannot participate in election — it is neither offered control + // nor handed a follower's resize_panes stream. + ClientID string `json:"client_id,omitempty"` } type CreatePanePayload struct { @@ -450,6 +474,33 @@ type ResizePanePayload struct { Cols uint16 `json:"cols"` } +// ResizePanesPayload batches a whole resize pass — a window resize or a +// split-drag release across every pane in a tab — into ONE client → daemon +// frame, instead of one MsgResizePane per pane. A resize burst from the +// master with a follower attached must not overflow that follower's +// must-deliver queue with one pane_sizes echo per pane. +type ResizePanesPayload struct { + Panes []ResizePanePayload `json:"panes"` +} + +// PaneSizesPayload is the daemon's must-deliver echo of a resize batch to +// every OTHER attached client (the master already has the sizes it sent). +// Mirrors ResizePanesPayload's shape rather than reusing the name, because the +// two travel in opposite directions and are never decoded as the same type. +type PaneSizesPayload struct { + Panes []ResizePanePayload `json:"panes"` +} + +// ClientGeometryPayload reports a client's own window size, in terminal +// cells, independent of any pane. The master's geometry decides whether it +// stays eligible: a window shrunk below the paintable floor reports 0x0 here +// and loses master eligibility immediately, handing off to the next eligible +// client. +type ClientGeometryPayload struct { + Cols int `json:"cols"` + Rows int `json:"rows"` +} + type PaneInputPayload struct { PaneID string `json:"pane_id"` Data []byte `json:"data"` @@ -688,6 +739,14 @@ type UpdatePanePayload struct { type UpdateLayoutPayload struct { TabID string `json:"tab_id"` Layout json.RawMessage `json:"layout"` + // BaseRev is the layout revision this update was built against, for + // conflict detection between clients editing the same tab's tree. A + // POINTER so an unset field (an older client, or one that has not adopted + // revisions yet) is distinguishable from an explicit base of revision 0 — + // the daemon's very first assigned revision is 0, and collapsing that to + // "absent" would make the daemon unable to tell "no base known" from "based + // on the initial revision". + BaseRev *uint64 `json:"base_rev,omitempty"` } type PluginErrorPayload struct { @@ -1058,12 +1117,27 @@ type DestroyPaneRespPayload struct { type SetActivePanePayload struct { PaneID string `json:"pane_id"` + // Client names which attached client this applies to. Empty keeps the + // historical broadcast-to-every-TUI behavior, which is what every + // existing producer (MCP's set_active_pane) sends and what a headless + // daemon with no attached client needs: with nobody attached, this only + // switches the tab and there is no client to target. + Client string `json:"client,omitempty"` } type HighlightPanePayload struct { PaneID string `json:"pane_id"` } +// CloseTUIPayload asks one specific client to exit, for multi-client sync +// (e.g. the master asking a follower to close, or an admin action against one +// client in the list). Client empty keeps the historical behavior of +// MsgCloseTUI: broadcast to every attached TUI. A headless daemon with no +// attached client sends nothing and must not panic. +type CloseTUIPayload struct { + Client string `json:"client,omitempty"` +} + // Notification center payloads (M12) // ContextTokensCompacting is the sentinel value for a pane's context-token @@ -1094,6 +1168,22 @@ type DismissEventPayload struct { EventID string `json:"event_id"` // empty = dismiss all } +// EventDismissedPayload is the daemon's broadcast of a dismissal to every +// OTHER attached client, so a notification acted on in one client's sidebar +// does not also sit there in a second one. Mirrors DismissEventPayload's +// "" = all convention rather than reusing the type, because the two travel in +// opposite directions and one is a request while the other is a fact. +type EventDismissedPayload struct { + EventID string `json:"event_id"` // "" = all +} + +// PaneSeenPayload is the daemon's broadcast marking a pane as looked-at by +// some client, so every OTHER client's sidebar clears the same "finished +// while you were away" mark rather than each client tracking it alone. +type PaneSeenPayload struct { + PaneID string `json:"pane_id"` +} + type GetNotificationsRespPayload struct { Events []PaneEventPayload `json:"events"` } @@ -1143,7 +1233,7 @@ type VersionRespPayload struct { // GatedRequests are the request types a daemon advertises in // VersionRespPayload.Requests. Add a type here when it is new enough that an // older daemon would drop it silently. -var GatedRequests = []string{MsgCreateFromTemplateReq} +var GatedRequests = []string{MsgCreateFromTemplateReq, MsgListClientsReq} // Memory reporting payloads @@ -1296,6 +1386,31 @@ type ResourceReportRespPayload struct { CPUSupported bool `json:"cpu_supported,omitempty"` } +// ClientInfo is one attached client's row in the multi-client list — who is +// attached, since when, at what size, and whether it currently holds master +// (the client whose geometry sizes every pane's PTY). +type ClientInfo struct { + Client string `json:"client"` + AttachedAt string `json:"attached_at"` // RFC 3339 + Cols int `json:"cols"` + Rows int `json:"rows"` + Master bool `json:"master"` + // LastInputAt is when this client last sent pane input, RFC 3339. Empty + // means never — a follower that has only watched, not typed. + LastInputAt string `json:"last_input_at,omitempty"` + // Role distinguishes a TUI from an MCP bridge sharing the same attach + // path, mirroring ClientHelloPayload.Role. Empty for a client that never + // sent one. + Role string `json:"role,omitempty"` + PID int `json:"pid,omitempty"` + Exe string `json:"exe,omitempty"` +} + +// ListClientsRespPayload answers MsgListClientsReq with every attached client. +type ListClientsRespPayload struct { + Clients []ClientInfo `json:"clients"` +} + // ClientHelloPayload is a durable client's self-description. type ClientHelloPayload struct { Role string `json:"role"` // "tui" | "bridge" diff --git a/internal/ipc/protocol_test.go b/internal/ipc/protocol_test.go index 11b40708..760b5a4a 100644 --- a/internal/ipc/protocol_test.go +++ b/internal/ipc/protocol_test.go @@ -110,6 +110,14 @@ func TestMessageTypes(t *testing.T) { ipc.MsgPaneHistoryEntryResp, ipc.MsgPaneSearchReq, ipc.MsgPaneSearchResp, + ipc.MsgPaneSizes, + ipc.MsgResizePanes, + ipc.MsgClientGeometry, + ipc.MsgTakeControl, + ipc.MsgListClientsReq, + ipc.MsgListClientsResp, + ipc.MsgEventDismissed, + ipc.MsgPaneSeen, } for _, typ := range types { if typ == "" { @@ -480,3 +488,205 @@ func TestOverlayPolicyPayload_RoundTrip(t *testing.T) { t.Errorf("payload = %+v, want {7 3}", out) } } + +// Multi-client sync payload round trips. + +// TestAttachPayload_ClientIDRoundTrip locks in the wire-format contract for +// the client id multi-client sync needs for master election and per-client +// state. An older client that never sets it must decode to an empty string, +// not an error. +func TestAttachPayload_ClientIDRoundTrip(t *testing.T) { + cases := []struct { + name string + in ipc.AttachPayload + want string + }{ + {"new client with id", ipc.AttachPayload{Cols: 80, Rows: 24, ClientID: "client-abc"}, "client-abc"}, + {"old client omits id", ipc.AttachPayload{Cols: 80, Rows: 24}, ""}, + } + for _, tc := range cases { + t.Run(tc.name, func(t *testing.T) { + msg, err := ipc.NewMessage(ipc.MsgAttach, tc.in) + if err != nil { + t.Fatalf("NewMessage: %v", err) + } + var got ipc.AttachPayload + if err := msg.DecodePayload(&got); err != nil { + t.Fatalf("DecodePayload: %v", err) + } + if got.ClientID != tc.want { + t.Errorf("ClientID: got %q, want %q", got.ClientID, tc.want) + } + }) + } +} + +func TestResizePanesPayload_RoundTrip(t *testing.T) { + req := ipc.ResizePanesPayload{Panes: []ipc.ResizePanePayload{ + {PaneID: "p1", Cols: 80, Rows: 24}, + {PaneID: "p2", Cols: 40, Rows: 12}, + }} + msg, err := ipc.NewMessage(ipc.MsgResizePanes, req) + if err != nil { + t.Fatalf("NewMessage: %v", err) + } + var got ipc.ResizePanesPayload + if err := msg.DecodePayload(&got); err != nil { + t.Fatalf("DecodePayload: %v", err) + } + if len(got.Panes) != 2 || got.Panes[0].PaneID != "p1" || got.Panes[1].Cols != 40 { + t.Fatalf("round-trip mismatch: %+v", got) + } +} + +func TestPaneSizesPayload_RoundTrip(t *testing.T) { + resp := ipc.PaneSizesPayload{Panes: []ipc.ResizePanePayload{{PaneID: "p1", Cols: 100, Rows: 30}}} + msg, err := ipc.NewMessage(ipc.MsgPaneSizes, resp) + if err != nil { + t.Fatalf("NewMessage: %v", err) + } + var got ipc.PaneSizesPayload + if err := msg.DecodePayload(&got); err != nil { + t.Fatalf("DecodePayload: %v", err) + } + if len(got.Panes) != 1 || got.Panes[0].PaneID != "p1" || got.Panes[0].Rows != 30 { + t.Fatalf("round-trip mismatch: %+v", got) + } +} + +func TestClientGeometryPayload_RoundTrip(t *testing.T) { + msg, err := ipc.NewMessage(ipc.MsgClientGeometry, ipc.ClientGeometryPayload{Cols: 120, Rows: 40}) + if err != nil { + t.Fatal(err) + } + var got ipc.ClientGeometryPayload + if err := msg.DecodePayload(&got); err != nil { + t.Fatal(err) + } + if got.Cols != 120 || got.Rows != 40 { + t.Errorf("payload = %+v, want {120 40}", got) + } +} + +// TestUpdateLayoutPayload_BaseRevNilVsZero is load-bearing for Task 5: an +// absent base must stay nil (no base known) and an explicit revision 0 must +// round-trip as a non-nil zero (based on the daemon's very first revision). +// Collapsing the two would make the daemon unable to tell them apart. +func TestUpdateLayoutPayload_BaseRevNilVsZero(t *testing.T) { + t.Run("absent stays nil", func(t *testing.T) { + b, err := json.Marshal(ipc.UpdateLayoutPayload{TabID: "t1", Layout: json.RawMessage(`{}`)}) + if err != nil { + t.Fatal(err) + } + if bytes.Contains(b, []byte("base_rev")) { + t.Errorf("nil BaseRev must not appear on the wire: %s", b) + } + var got ipc.UpdateLayoutPayload + if err := json.Unmarshal(b, &got); err != nil { + t.Fatal(err) + } + if got.BaseRev != nil { + t.Errorf("BaseRev = %v, want nil", got.BaseRev) + } + }) + t.Run("explicit zero round-trips non-nil", func(t *testing.T) { + var zero uint64 + b, err := json.Marshal(ipc.UpdateLayoutPayload{TabID: "t1", Layout: json.RawMessage(`{}`), BaseRev: &zero}) + if err != nil { + t.Fatal(err) + } + var got ipc.UpdateLayoutPayload + if err := json.Unmarshal(b, &got); err != nil { + t.Fatal(err) + } + if got.BaseRev == nil { + t.Fatal("BaseRev = nil, want a non-nil zero") + } + if *got.BaseRev != 0 { + t.Errorf("BaseRev = %d, want 0", *got.BaseRev) + } + }) +} + +func TestSetActivePanePayload_ClientRoundTrip(t *testing.T) { + msg, err := ipc.NewMessage(ipc.MsgSetActivePane, ipc.SetActivePanePayload{PaneID: "p1", Client: "client-a"}) + if err != nil { + t.Fatal(err) + } + var got ipc.SetActivePanePayload + if err := msg.DecodePayload(&got); err != nil { + t.Fatal(err) + } + if got.PaneID != "p1" || got.Client != "client-a" { + t.Errorf("payload = %+v, want {p1 client-a}", got) + } +} + +func TestSetActivePanePayload_ClientOmittedWhenEmpty(t *testing.T) { + b, err := json.Marshal(ipc.SetActivePanePayload{PaneID: "p1"}) + if err != nil { + t.Fatal(err) + } + if bytes.Contains(b, []byte("client")) { + t.Errorf("empty Client must not appear on the wire: %s", b) + } +} + +func TestCloseTUIPayload_RoundTrip(t *testing.T) { + msg, err := ipc.NewMessage(ipc.MsgCloseTUI, ipc.CloseTUIPayload{Client: "client-a"}) + if err != nil { + t.Fatal(err) + } + var got ipc.CloseTUIPayload + if err := msg.DecodePayload(&got); err != nil { + t.Fatal(err) + } + if got.Client != "client-a" { + t.Errorf("Client = %q, want %q", got.Client, "client-a") + } +} + +func TestListClientsRespPayload_RoundTrip(t *testing.T) { + resp := ipc.ListClientsRespPayload{Clients: []ipc.ClientInfo{ + {Client: "c1", AttachedAt: "2026-09-25T00:00:00Z", Cols: 80, Rows: 24, Master: true, Role: "tui", PID: 123, Exe: "quil"}, + }} + msg, err := ipc.NewMessage(ipc.MsgListClientsResp, resp) + if err != nil { + t.Fatal(err) + } + var got ipc.ListClientsRespPayload + if err := msg.DecodePayload(&got); err != nil { + t.Fatal(err) + } + if len(got.Clients) != 1 || !got.Clients[0].Master || got.Clients[0].Client != "c1" { + t.Fatalf("round-trip mismatch: %+v", got) + } +} + +func TestEventDismissedPayload_RoundTrip(t *testing.T) { + msg, err := ipc.NewMessage(ipc.MsgEventDismissed, ipc.EventDismissedPayload{EventID: "evt-1"}) + if err != nil { + t.Fatal(err) + } + var got ipc.EventDismissedPayload + if err := msg.DecodePayload(&got); err != nil { + t.Fatal(err) + } + if got.EventID != "evt-1" { + t.Errorf("EventID = %q, want %q", got.EventID, "evt-1") + } +} + +func TestPaneSeenPayload_RoundTrip(t *testing.T) { + msg, err := ipc.NewMessage(ipc.MsgPaneSeen, ipc.PaneSeenPayload{PaneID: "pane-1"}) + if err != nil { + t.Fatal(err) + } + var got ipc.PaneSeenPayload + if err := msg.DecodePayload(&got); err != nil { + t.Fatal(err) + } + if got.PaneID != "pane-1" { + t.Errorf("PaneID = %q, want %q", got.PaneID, "pane-1") + } +} diff --git a/internal/ipc/server.go b/internal/ipc/server.go index cb6e8d24..95883315 100644 --- a/internal/ipc/server.go +++ b/internal/ipc/server.go @@ -120,6 +120,18 @@ type Conn struct { // Atomic because the opt-out arrives on the conn's read goroutine while // Broadcast reads it from whichever goroutine is emitting. noPaneOutput atomic.Bool + // holdPaneOutput is a second, independent gate on the live MsgPaneOutput + // stream, for multi-client sync rather than the subscribe opt-out above: + // the daemon holds a newly-attaching client off live output while it + // delivers that client's replay, so replay bytes and live bytes cannot + // interleave into a torn screen. Released once the replay is queued. + // + // A second field rather than reusing noPaneOutput because the two are + // set by different actors for different reasons and must not be + // confused: noPaneOutput is the client's own durable choice (MsgSubscribe), + // while holdPaneOutput is the daemon's own transient bookkeeping around one + // attach. wantsFrame filters MsgPaneOutput when EITHER is set. + holdPaneOutput atomic.Bool // pending counts must-deliver frames accepted by Send but not yet written // to the socket. Send is non-blocking — it hands the frame to sendLoop — // so an empty critCh does NOT mean the peer has it. Flush needs to know @@ -690,15 +702,23 @@ func (c *Conn) setPaneOutputWanted(want bool) { c.noPaneOutput.Store(!want) } func (c *Conn) wantsPaneOutput() bool { return !c.noPaneOutput.Load() } +// SetHoldPaneOutput holds this conn off the live MsgPaneOutput stream (on) +// or releases it (off), independent of the MsgSubscribe opt-out above. The +// daemon sets this while it delivers a newly-attaching client's replay, so a +// live frame cannot be interleaved into the middle of it, and clears it once +// the replay is queued. +func (c *Conn) SetHoldPaneOutput(on bool) { c.holdPaneOutput.Store(on) } + // wantsFrame reports whether a frame of this type should be delivered to this // conn. Only the live pane-output stream is ever filtered: everything else is -// must-deliver, and a client excusing itself from PTY bytes still needs -// workspace state, its own responses, and lifecycle frames. +// must-deliver, and a client excusing itself from PTY bytes — or held off +// them for a moment — still needs workspace state, its own responses, and +// lifecycle frames. func (c *Conn) wantsFrame(msgType string) bool { if msgType != MsgPaneOutput { return true } - return c.wantsPaneOutput() + return c.wantsPaneOutput() && !c.holdPaneOutput.Load() } func (s *Server) acceptLoop() { diff --git a/internal/ipc/subscribe_broadcast_test.go b/internal/ipc/subscribe_broadcast_test.go index 8fe4a0c4..a95bb7c4 100644 --- a/internal/ipc/subscribe_broadcast_test.go +++ b/internal/ipc/subscribe_broadcast_test.go @@ -112,3 +112,118 @@ func TestBroadcast_SkipsPaneOutputForOptedOutConnOnly(t *testing.T) { got.Type, ipc.MsgWorkspaceState) } } + +// TestBroadcast_SkipsPaneOutputForHeldConnOnly is the SetHoldPaneOutput analog +// of TestBroadcast_SkipsPaneOutputForOptedOutConnOnly above: the daemon holds +// a newly-attaching client off live pane_output (to deliver its replay without +// interleaving), and that hold must be as real on the wire as the client's own +// MsgSubscribe opt-out — filtering pane_output only, never must-deliver +// traffic, and never touching any other conn. +// +// Unlike the opt-out, holding is daemon-internal bookkeeping with no wire +// message of its own (a later task wires the daemon side), so this drives it +// directly through the *Conn the server hands back via ConnsSnapshot rather +// than through a client-sent message. +func TestBroadcast_SkipsPaneOutputForHeldConnOnly(t *testing.T) { + t.Parallel() + sockPath := filepath.Join(t.TempDir(), "hold.sock") + + srv := ipc.NewServer(sockPath, func(c *ipc.Conn, m *ipc.Message) {}, nil) + if err := srv.Start(); err != nil { + t.Fatalf("server start: %v", err) + } + defer srv.Stop() + + unheld, err := ipc.NewClient(sockPath) + if err != nil { + t.Fatalf("unheld client connect: %v", err) + } + defer unheld.Close() + waitForConnCount(t, srv, 1, 2*time.Second) + + held, err := ipc.NewClient(sockPath) + if err != nil { + t.Fatalf("held client connect: %v", err) + } + defer held.Close() + waitForConnCount(t, srv, 2, 2*time.Second) + + conns := srv.ConnsSnapshot() + if len(conns) != 2 { + t.Fatalf("ConnsSnapshot: got %d conns, want 2", len(conns)) + } + // Accept order matches dial order: the first entry is unheld's conn, the + // second is held's — hold the second one. + conns[1].SetHoldPaneOutput(true) + + // The held client must receive no pane output while held. + deadline := time.Now().Add(2 * time.Second) + for { + out, err := ipc.NewMessage(ipc.MsgPaneOutput, ipc.PaneOutputPayload{PaneID: "pane-probe", Data: []byte("x")}) + if err != nil { + t.Fatalf("build probe: %v", err) + } + srv.Broadcast(out) + + held.SetReadDeadline(time.Now().Add(150 * time.Millisecond)) + if _, err := held.Receive(); err != nil { + break // no pane output delivered — the hold has landed + } + if time.Now().After(deadline) { + t.Fatal("held client still receives pane output while held") + } + } + held.SetReadDeadline(time.Time{}) + + // The unheld client must still get pane output... + payload := ipc.PaneOutputPayload{PaneID: "pane-1", Data: []byte("hello")} + out, err := ipc.NewMessage(ipc.MsgPaneOutput, payload) + if err != nil { + t.Fatalf("build pane output: %v", err) + } + srv.Broadcast(out) + + unheld.SetReadDeadline(time.Now().Add(2 * time.Second)) + got, err := unheld.Receive() + if err != nil { + t.Fatalf("unheld client got no pane output: %v", err) + } + if got.Type != ipc.MsgPaneOutput { + t.Fatalf("unheld client got %q, want %q", got.Type, ipc.MsgPaneOutput) + } + + // ...and the held client must still get everything else, or the hold is + // silencing more than must-deliver traffic should allow. + state, err := ipc.NewMessage(ipc.MsgWorkspaceState, map[string]any{"tabs": []any{}}) + if err != nil { + t.Fatalf("build workspace state: %v", err) + } + srv.Broadcast(state) + + held.SetReadDeadline(time.Now().Add(2 * time.Second)) + got, err = held.Receive() + if err != nil { + t.Fatalf("held client lost must-deliver traffic: %v", err) + } + if got.Type != ipc.MsgWorkspaceState { + t.Fatalf("held client got %q, want %q — only pane output may be filtered", + got.Type, ipc.MsgWorkspaceState) + } + + // Releasing the hold must restore the live stream. + conns[1].SetHoldPaneOutput(false) + out2, err := ipc.NewMessage(ipc.MsgPaneOutput, ipc.PaneOutputPayload{PaneID: "pane-2", Data: []byte("world")}) + if err != nil { + t.Fatalf("build pane output: %v", err) + } + srv.Broadcast(out2) + + held.SetReadDeadline(time.Now().Add(2 * time.Second)) + got, err = held.Receive() + if err != nil { + t.Fatalf("released client got no pane output: %v", err) + } + if got.Type != ipc.MsgPaneOutput { + t.Fatalf("released client got %q, want %q", got.Type, ipc.MsgPaneOutput) + } +} diff --git a/internal/ipc/subscribe_test.go b/internal/ipc/subscribe_test.go index daa3a4fa..07e35993 100644 --- a/internal/ipc/subscribe_test.go +++ b/internal/ipc/subscribe_test.go @@ -61,3 +61,27 @@ func TestConn_SubscribedConnWantsEverything(t *testing.T) { } } } + +// TestConn_HoldPaneOutputSkipsOnlyPaneOutput covers SetHoldPaneOutput, the +// daemon's own transient gate used to keep live output from interleaving into +// a newly-attaching client's replay. It is independent of the client's own +// MsgSubscribe opt-out above (setPaneOutputWanted): only pane_output is ever +// filtered, and releasing the hold must restore it. +func TestConn_HoldPaneOutputSkipsOnlyPaneOutput(t *testing.T) { + local, remote := net.Pipe() + t.Cleanup(func() { local.Close(); remote.Close() }) + c := newConn(local) + c.SetHoldPaneOutput(true) + if c.wantsFrame(MsgPaneOutput) { + t.Fatal("held conn must skip pane_output") + } + for _, typ := range []string{MsgWorkspaceState, MsgPaneSizes, MsgPaneEvent} { + if !c.wantsFrame(typ) { + t.Fatalf("held conn must still want %s", typ) + } + } + c.SetHoldPaneOutput(false) + if !c.wantsFrame(MsgPaneOutput) { + t.Fatal("released conn must want pane_output") + } +} From 5ebbe34217d5c4b0aa4a65210d1a080d184fc6da Mon Sep 17 00:00:00 2001 From: Artjoms Stukans Date: Fri, 25 Sep 2026 16:18:19 +0200 Subject: [PATCH 02/40] feat(daemon): elect a size master among attached clients The attached-conn set becomes a client registry (clients.go), keyed by conn, with each client's id, first-attach time, RAW window size, cwd, last input time and overlay claims. A size master is elected over it: - The oldest attached client whose raw geometry is paintable (the TUI's 40x10 floor) wins; ties go to the smaller id. A 0x0 or 1x1 attach is never elected, even though handleAttach still defaults clientSize to 80x24. A master shrunk below the floor hands over at once. - A master whose link is LOST keeps its slot for master_grace_minutes (default 3, clamped 0-60), but only while a client attached at the loss is still attached; a lone master is replaced at once. The same id reattaching inside grace takes the slot back with no change. - detach now means a clean exit: the client is removed and the election runs with no reservation. - take_control makes an eligible attached sender the master. - The master id is written to workspace.json as size_master and restored as a min(grace, 30s) reservation, so a daemon restart resizes nothing when the previous master reattaches. Disconnects caused by our own shutdown are not treated as lost links, so the final snapshot still records the master. A master change broadcasts state once, outside the registry lock. pane_input, switch_tab, create_tab, create_pane, update_layout, update_pane and take_control stamp the sender's last input time. Part of #235 --- internal/config/config.go | 22 +- internal/config/config_test.go | 36 ++ internal/daemon/clients.go | 581 +++++++++++++++++++++++++ internal/daemon/clients_test.go | 501 +++++++++++++++++++++ internal/daemon/clients_wiring_test.go | 129 ++++++ internal/daemon/daemon.go | 109 ++--- internal/daemon/overlay.go | 28 +- internal/daemon/procreport.go | 14 +- 8 files changed, 1353 insertions(+), 67 deletions(-) create mode 100644 internal/daemon/clients.go create mode 100644 internal/daemon/clients_test.go create mode 100644 internal/daemon/clients_wiring_test.go diff --git a/internal/config/config.go b/internal/config/config.go index b0cc49d4..f3449ad3 100644 --- a/internal/config/config.go +++ b/internal/config/config.go @@ -370,6 +370,21 @@ type DaemonConfig struct { SnapshotInterval string `toml:"snapshot_interval"` AutoStart bool `toml:"auto_start"` WarmShellPoolSize int `toml:"warm_shell_pool_size"` // 0 disables pooling. + // MasterGraceMinutes is how long a size master whose link dropped keeps + // its slot while another client is attached. Read it through MasterGrace, + // which clamps it. + MasterGraceMinutes int `toml:"master_grace_minutes"` +} + +// MaxMasterGraceMinutes bounds master_grace_minutes. A longer grace would +// freeze every PTY size behind a client that is not coming back. +const MaxMasterGraceMinutes = 60 + +// MasterGrace is MasterGraceMinutes clamped to 0–MaxMasterGraceMinutes, as a +// duration. Zero means no grace: a lost master is replaced at once. +func (c DaemonConfig) MasterGrace() time.Duration { + m := min(max(c.MasterGraceMinutes, 0), MaxMasterGraceMinutes) + return time.Duration(m) * time.Minute } type GhostBufferConfig struct { @@ -613,9 +628,10 @@ type KeybindingsConfig struct { func Default() Config { return Config{ Daemon: DaemonConfig{ - SnapshotInterval: "30s", - AutoStart: true, - WarmShellPoolSize: 1, + SnapshotInterval: "30s", + AutoStart: true, + WarmShellPoolSize: 1, + MasterGraceMinutes: 3, }, GhostBuffer: GhostBufferConfig{ MaxLines: 500, diff --git a/internal/config/config_test.go b/internal/config/config_test.go index 0502de92..c405632d 100644 --- a/internal/config/config_test.go +++ b/internal/config/config_test.go @@ -105,6 +105,42 @@ func TestLoad_WarmShellPoolSize(t *testing.T) { } } +// The grace time a lost size master keeps its slot. An old config file has no +// key and must get the default, and a hand-edited value outside 0–60 is +// clamped rather than honoured: a day-long grace would freeze every PTY size +// behind a client that is never coming back. +func TestConfig_MasterGraceDefaultAndClamp(t *testing.T) { + if got := config.Default().Daemon.MasterGraceMinutes; got != 3 { + t.Errorf("Default MasterGraceMinutes = %d, want 3", got) + } + tests := []struct { + name string + toml string + want time.Duration + }{ + {"absent key", "[daemon]\nauto_start = true\n", 3 * time.Minute}, + {"explicit", "[daemon]\nmaster_grace_minutes = 7\n", 7 * time.Minute}, + {"zero means no grace", "[daemon]\nmaster_grace_minutes = 0\n", 0}, + {"above the ceiling", "[daemon]\nmaster_grace_minutes = 99\n", 60 * time.Minute}, + {"negative", "[daemon]\nmaster_grace_minutes = -5\n", 0}, + } + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + path := filepath.Join(t.TempDir(), "config.toml") + if err := os.WriteFile(path, []byte(tt.toml), 0o600); err != nil { + t.Fatal(err) + } + cfg, err := config.Load(path) + if err != nil { + t.Fatalf("Load: %v", err) + } + if got := cfg.Daemon.MasterGrace(); got != tt.want { + t.Errorf("MasterGrace() = %v, want %v", got, tt.want) + } + }) + } +} + func TestQuilDir(t *testing.T) { dir := config.QuilDir() if dir == "" { diff --git a/internal/daemon/clients.go b/internal/daemon/clients.go new file mode 100644 index 00000000..4033003b --- /dev/null +++ b/internal/daemon/clients.go @@ -0,0 +1,581 @@ +package daemon + +import ( + "log" + "sort" + "sync" + "time" + + "github.com/google/uuid" + + "github.com/artyomsv/quil/internal/ipc" +) + +// Multi-client sync: the registry of ATTACHED clients and the size-master +// election over it. +// +// Several TUIs can attach to one daemon. Each PTY has one size, so exactly one +// client may set it: the size master. The master is the OLDEST attached client +// whose raw window is paintable. A master whose link is LOST keeps its slot for +// a grace time, but only while another client that saw it leave is still +// attached, because the grace protects those followers from a resize. A clean +// exit (MsgDetach) skips the grace. A daemon restart keeps the previous +// master's slot for a short reserve, so the reattach ladder resizes nothing. + +const ( + // daemonMinClientCols and daemonMinClientRows are the TUI's minTermWidth + // and minTermHeight (internal/tui/model.go), the threshold of its + // terminalPaintable gate. The two pairs MUST stay in step: the TUI reports + // 0x0 below its own floor, and a daemon floor lower than the TUI's would + // elect a window the TUI itself refuses to size panes from. The daemon + // cannot import internal/tui. + daemonMinClientCols = 40 + daemonMinClientRows = 10 + + // restartReserveCap bounds the slot a restored size_master is kept for. + // TUIs reattach within seconds of a daemon restart, so a longer reserve + // only delays the election when the previous master is not coming back. + restartReserveCap = 30 * time.Second + + // maxClientIDLen bounds a client's self-reported id. A UUID is 36 bytes; + // the id is used only as a map key and a display value. + maxClientIDLen = 64 +) + +// clientRecord is one attached client. The registry is keyed by conn, and a +// conn that never sent MsgAttach (an MCP bridge) has no record. +type clientRecord struct { + id string + conn *ipc.Conn + attachedAt time.Time // first attach of this id; kept across a graced reconnect + cols, rows int // RAW, never defaulted + cwd string + // lastInputAt is stamped by every user-originated message on this conn. + // Zero means the client never sent one. + lastInputAt time.Time + // overlays is the set of overlay panes this client has ON SCREEN — see + // setOverlayClaim for why visibility is per client. + overlays map[string]bool +} + +// reservation keeps a lost master's slot until `until`. +type reservation struct { + id string + until time.Time + // attachedAt is the lost master's first attach, handed back to it when it + // returns, so it stays the oldest client. Zero for a restart reserve. + attachedAt time.Time + // protects holds the ids attached at the loss. The slot is kept only while + // one of them is still attached. nil for a restart reserve, which has no + // such condition. + protects map[string]bool +} + +// clientRegistry is the set of attached clients plus the master bookkeeping. +// +// Its mutex is a LEAF: never sm.mu, never a pane's PluginMu, and never held +// while broadcasting or sending. It is written from every conn's dispatch +// goroutine, the disconnect callback and the grace timer. +type clientRegistry struct { + mu sync.Mutex + byConn map[*ipc.Conn]*clientRecord + masterID string + reserved *reservation + + // now and afterFn are seams over time.Now and time.AfterFunc. New sets + // the real ones; a zero registry falls back to them too. + now func() time.Time + afterFn func(time.Duration, func()) (stop func() bool) + timerStop func() bool + grace time.Duration + + // onChange runs, outside mu, when a timer expiry changes the master. The + // other elections run on a dispatch goroutine, which broadcasts itself. + onChange func() +} + +func realAfterFunc(d time.Duration, f func()) func() bool { + return time.AfterFunc(d, f).Stop +} + +func (r *clientRegistry) clock() time.Time { + if r.now == nil { + return time.Now() + } + return r.now() +} + +// eligible reports whether a client's RAW window is paintable. The raw value +// matters: handleAttach defaults a 0x0 attach to 80x24 for clientSize, and a +// console-less client electing itself on that default is the 1x1 incident. +func eligible(rec *clientRecord) bool { + return rec.cols >= daemonMinClientCols && rec.rows >= daemonMinClientRows +} + +func (r *clientRegistry) recordByID(id string) *clientRecord { + if id == "" { + return nil + } + for _, rec := range r.byConn { + if rec.id == id { + return rec + } + } + return nil +} + +func (r *clientRegistry) reserveStillProtects(res *reservation) bool { + if res.protects == nil { + return true + } + for _, rec := range r.byConn { + if res.protects[rec.id] { + return true + } + } + return false +} + +// oldestEligibleLocked returns the eligible client with the smallest +// attachedAt. A tie goes to the smaller id, so the result is deterministic. +func (r *clientRegistry) oldestEligibleLocked() *clientRecord { + var best *clientRecord + for _, rec := range r.byConn { + if !eligible(rec) { + continue + } + if best == nil || rec.attachedAt.Before(best.attachedAt) || + (rec.attachedAt.Equal(best.attachedAt) && rec.id < best.id) { + best = rec + } + } + return best +} + +// electLocked runs the election. Called with r.mu held. It returns whether +// masterID changed; the caller broadcasts once, after releasing r.mu. +func (r *clientRegistry) electLocked() bool { + now := r.clock() + if rec := r.recordByID(r.masterID); rec != nil && eligible(rec) { + return false // rule 1: a connected, eligible master keeps the slot + } + if res := r.reserved; res != nil && now.Before(res.until) && r.reserveStillProtects(res) { + if r.masterID != res.id { + r.masterID = res.id + return true + } + return false // rule 2: the reserved slot is kept + } + r.clearReservationLocked() + best := r.oldestEligibleLocked() // rule 3 + id := "" // rule 4: nobody eligible, no master + if best != nil { + id = best.id + } + changed := id != r.masterID + r.masterID = id + return changed +} + +func (r *clientRegistry) clearReservationLocked() { + r.reserved = nil + if r.timerStop != nil { + r.timerStop() + r.timerStop = nil + } +} + +// reserveLocked replaces any reservation with res and arms the timer that +// re-runs the election when it lapses. +func (r *clientRegistry) reserveLocked(res *reservation, d time.Duration) { + r.clearReservationLocked() + r.reserved = res + after := r.afterFn + if after == nil { + after = realAfterFunc + } + r.timerStop = after(d, r.expire) +} + +// expire is the timer callback. A stale fire (the timer was replaced or the +// master returned) is harmless: the election is idempotent. +func (r *clientRegistry) expire() { + r.mu.Lock() + changed := r.electLocked() + onChange := r.onChange + r.mu.Unlock() + if changed && onChange != nil { + onChange() + } +} + +// attach records conn as the client id at a RAW geometry and elects. An empty +// id is minted as "anon-", scoped to the conn: a re-attach on the same +// conn keeps it. +func (r *clientRegistry) attach(conn *ipc.Conn, id string, cols, rows int, cwd string) bool { + r.mu.Lock() + defer r.mu.Unlock() + if r.byConn == nil { + r.byConn = make(map[*ipc.Conn]*clientRecord) + } + rec, existed := r.byConn[conn] + if id == "" { + if existed { + id = rec.id + } else { + id = "anon-" + uuid.NewString() + } + } + if !existed { + rec = &clientRecord{conn: conn, overlays: map[string]bool{}} + r.byConn[conn] = rec + } + if !existed || rec.id != id { + rec.attachedAt = r.clock() + // The id belongs to one process, which holds one conn per daemon. A + // second conn with the same id means the first is a dead link the + // server has not reaped yet: this record replaces it, and keeps its age. + for c, other := range r.byConn { + if c != conn && other.id == id { + rec.attachedAt = other.attachedAt + delete(r.byConn, c) + } + } + if res := r.reserved; res != nil && res.id == id && !res.attachedAt.IsZero() { + rec.attachedAt = res.attachedAt + } + } + rec.id = id + rec.cols, rec.rows = cols, rows + rec.cwd = cwd + if res := r.reserved; res != nil && res.id == id { + // The reserved client is back, so the slot has nothing left to wait + // for. The election below keeps it when the client is eligible. + r.clearReservationLocked() + } + return r.electLocked() +} + +// lose drops conn after a LOST link: the conn closed with no MsgDetach. A +// master that leaves this way keeps its slot for the grace time, but only +// while another client is attached. With nobody to protect, the grace would +// only make a relaunched TUI (which has a new id) wait. +func (r *clientRegistry) lose(conn *ipc.Conn) bool { + r.mu.Lock() + defer r.mu.Unlock() + rec, ok := r.byConn[conn] + if !ok { + return false + } + delete(r.byConn, conn) + if rec.id == r.masterID && r.grace > 0 && r.recordByID(rec.id) == nil { + protects := make(map[string]bool, len(r.byConn)) + for _, other := range r.byConn { + protects[other.id] = true + } + if len(protects) > 0 { + r.reserveLocked(&reservation{ + id: rec.id, + until: r.clock().Add(r.grace), + attachedAt: rec.attachedAt, + protects: protects, + }, r.grace) + } + } + return r.electLocked() +} + +// detach drops conn after a clean exit and elects with NO reservation. The +// disconnect that follows finds no record and does nothing more. +func (r *clientRegistry) detach(conn *ipc.Conn) bool { + r.mu.Lock() + defer r.mu.Unlock() + if _, ok := r.byConn[conn]; !ok { + return false + } + delete(r.byConn, conn) + return r.electLocked() +} + +// setGeometry records a client's new RAW window size and elects: a master +// shrunk below the paintable floor loses the slot at once. +func (r *clientRegistry) setGeometry(conn *ipc.Conn, cols, rows int) bool { + r.mu.Lock() + defer r.mu.Unlock() + rec, ok := r.byConn[conn] + if !ok { + return false + } + rec.cols, rec.rows = cols, rows + return r.electLocked() +} + +// takeControl makes conn's client the master when it is attached and +// eligible. accepted is false when the request was ignored. +func (r *clientRegistry) takeControl(conn *ipc.Conn) (changed, accepted bool) { + r.mu.Lock() + defer r.mu.Unlock() + rec, ok := r.byConn[conn] + if !ok || !eligible(rec) { + return false, false + } + // An explicit request overrides a slot kept for someone else. + r.clearReservationLocked() + changed = r.masterID != rec.id + r.masterID = rec.id + return changed, true +} + +// reserveAfterRestart keeps a restored size_master's slot for +// min(grace, restartReserveCap), with no follower condition: after a restart +// nobody is attached yet, and the TUIs reattach with their same ids. +func (r *clientRegistry) reserveAfterRestart(id string) { + id = truncateField(id, maxClientIDLen) + d := min(r.grace, restartReserveCap) + if id == "" || d <= 0 { + return + } + r.mu.Lock() + defer r.mu.Unlock() + r.reserveLocked(&reservation{id: id, until: r.clock().Add(d)}, d) + r.electLocked() +} + +// registerClient records conn as an attached client, from its attach payload. +// +// ATTACHMENT, not connection, is what "a client is here" means, and the +// difference is not academic: every live MCP bridge holds an IPC conn for its +// whole lifetime (cmd/quil/mcp.go dials once and closes on exit), and a bridge +// is a child of the claude process in a PANE — so bridges routinely outlive the +// TUI. Counting raw conns therefore answered "is anything connected", which in +// any session with a claude pane wired to `quil mcp` is permanently yes (21 +// conns in the session that reported 7 live overlays), and the detached-session +// overlay stamp never fired in exactly the configuration it was designed for. +// Re-attaching on the same conn keeps that client's existing overlay claims. +// +// The geometry is the RAW one from the payload, taken before handleAttach +// defaults it. It returns whether the master changed. +func (d *Daemon) registerClient(conn *ipc.Conn, attach ipc.AttachPayload) bool { + if conn == nil { + return false + } + id := truncateField(attach.ClientID, maxClientIDLen) + return d.clients.attach(conn, id, attach.Cols, attach.Rows, attach.CWD) +} + +// forgetAttachedClient drops a conn whose link was lost, and with it every +// overlay that client claimed visible. A conn that never attached, or that +// already sent MsgDetach, is not in the set, so dropping it changes nothing. +// It returns whether the master changed. +func (d *Daemon) forgetAttachedClient(conn *ipc.Conn) bool { + return d.clients.lose(conn) +} + +// detachClient handles a clean client exit. It returns whether the master +// changed. +func (d *Daemon) detachClient(conn *ipc.Conn) bool { + return d.clients.detach(conn) +} + +// setClientGeometry records a client's RAW window size. It returns whether the +// master changed. +func (d *Daemon) setClientGeometry(conn *ipc.Conn, cols, rows int) bool { + return d.clients.setGeometry(conn, cols, rows) +} + +// takeControl makes the sender the master if it is attached and eligible. It +// returns whether the master changed. +func (d *Daemon) takeControl(conn *ipc.Conn) bool { + changed, accepted := d.clients.takeControl(conn) + if !accepted { + log.Printf("take_control: ignored (sender not attached or not paintable)") + } + return changed +} + +// shuttingDown reports whether Stop or MsgShutdown has begun. +func (d *Daemon) shuttingDown() bool { + select { + case <-d.shutdown: + return true + default: + return false + } +} + +func (d *Daemon) handleDetach(conn *ipc.Conn) { + if d.detachClient(conn) { + d.broadcastState() + } +} + +func (d *Daemon) handleClientGeometry(conn *ipc.Conn, msg *ipc.Message) { + var p ipc.ClientGeometryPayload + if err := msg.DecodePayload(&p); err != nil { + return + } + if d.setClientGeometry(conn, p.Cols, p.Rows) { + d.broadcastState() + } +} + +func (d *Daemon) handleTakeControl(conn *ipc.Conn) { + if d.takeControl(conn) { + d.broadcastState() + } +} + +// masterConn returns the master's conn, or nil when there is no connected +// master (none elected, or its slot is reserved while its link is lost). +func (d *Daemon) masterConn() *ipc.Conn { + d.clients.mu.Lock() + defer d.clients.mu.Unlock() + if rec := d.clients.recordByID(d.clients.masterID); rec != nil { + return rec.conn + } + return nil +} + +func (d *Daemon) isMasterConn(c *ipc.Conn) bool { + if c == nil { + return false + } + d.clients.mu.Lock() + defer d.clients.mu.Unlock() + rec, ok := d.clients.byConn[c] + return ok && d.clients.masterID != "" && rec.id == d.clients.masterID +} + +// sizeAuthorityOpen reports the legacy state in which any attached client may +// resize: no master elected and no slot reserved. +func (d *Daemon) sizeAuthorityOpen() bool { + d.clients.mu.Lock() + defer d.clients.mu.Unlock() + return d.clients.masterID == "" && d.clients.reserved == nil +} + +func (d *Daemon) masterID() string { + d.clients.mu.Lock() + defer d.clients.mu.Unlock() + return d.clients.masterID +} + +func (d *Daemon) clientCount() int { + d.clients.mu.Lock() + defer d.clients.mu.Unlock() + return len(d.clients.byConn) +} + +// sortedRecordsLocked returns the records oldest first, ties by id. +func (r *clientRegistry) sortedRecordsLocked() []*clientRecord { + recs := make([]*clientRecord, 0, len(r.byConn)) + for _, rec := range r.byConn { + recs = append(recs, rec) + } + sort.Slice(recs, func(i, j int) bool { + if !recs[i].attachedAt.Equal(recs[j].attachedAt) { + return recs[i].attachedAt.Before(recs[j].attachedAt) + } + return recs[i].id < recs[j].id + }) + return recs +} + +// followerConns returns the attached conns that are not the master, oldest +// first, leaving out except. +func (d *Daemon) followerConns(except *ipc.Conn) []*ipc.Conn { + d.clients.mu.Lock() + defer d.clients.mu.Unlock() + var out []*ipc.Conn + for _, rec := range d.clients.sortedRecordsLocked() { + if rec.conn == except || (d.clients.masterID != "" && rec.id == d.clients.masterID) { + continue + } + out = append(out, rec.conn) + } + return out +} + +// mostRecentlyActiveConn returns the attached client with the latest input. +// When nobody has typed yet, it is the most recently attached client. nil when +// no client is attached. +func (d *Daemon) mostRecentlyActiveConn() *ipc.Conn { + d.clients.mu.Lock() + defer d.clients.mu.Unlock() + var best *clientRecord + for _, rec := range d.clients.sortedRecordsLocked() { + if best == nil || !rec.lastInputAt.Before(best.lastInputAt) { + best = rec + } + } + if best == nil { + return nil + } + return best.conn +} + +// clientByConn returns a copy of conn's record. +func (d *Daemon) clientByConn(c *ipc.Conn) (clientRecord, bool) { + d.clients.mu.Lock() + defer d.clients.mu.Unlock() + rec, ok := d.clients.byConn[c] + if !ok { + return clientRecord{}, false + } + cp := *rec + cp.overlays = make(map[string]bool, len(rec.overlays)) + for k, v := range rec.overlays { + cp.overlays[k] = v + } + return cp, true +} + +// touchClientInput stamps a user-originated message on conn. A conn that is +// not an attached client (an MCP bridge) is not stamped. +func (d *Daemon) touchClientInput(c *ipc.Conn) { + d.clients.mu.Lock() + defer d.clients.mu.Unlock() + if rec, ok := d.clients.byConn[c]; ok { + rec.lastInputAt = d.clients.clock() + } +} + +// listClients describes every attached client, oldest first. The hello +// registry supplies role, pid and exe; it is read after clients.mu is +// released, because each registry keeps its own leaf lock. +func (d *Daemon) listClients() []ipc.ClientInfo { + type row struct { + conn *ipc.Conn + info ipc.ClientInfo + } + d.clients.mu.Lock() + recs := d.clients.sortedRecordsLocked() + rows := make([]row, 0, len(recs)) + for _, rec := range recs { + info := ipc.ClientInfo{ + Client: rec.id, + AttachedAt: rec.attachedAt.UTC().Format(time.RFC3339), + Cols: rec.cols, + Rows: rec.rows, + Master: d.clients.masterID != "" && rec.id == d.clients.masterID, + } + if !rec.lastInputAt.IsZero() { + info.LastInputAt = rec.lastInputAt.UTC().Format(time.RFC3339) + } + rows = append(rows, row{conn: rec.conn, info: info}) + } + d.clients.mu.Unlock() + + out := make([]ipc.ClientInfo, 0, len(rows)) + for _, r := range rows { + if d.hellos != nil { + if h, ok := d.hellos.helloOf(r.conn); ok { + r.info.Role = h.Role + r.info.PID = h.PID + r.info.Exe = h.ExeName + } + } + out = append(out, r.info) + } + return out +} diff --git a/internal/daemon/clients_test.go b/internal/daemon/clients_test.go new file mode 100644 index 00000000..88f52c67 --- /dev/null +++ b/internal/daemon/clients_test.go @@ -0,0 +1,501 @@ +package daemon + +import ( + "testing" + "time" + + "github.com/artyomsv/quil/internal/ipc" +) + +// fakeGraceTimer is one time.AfterFunc the registry armed. Tests fire it by +// hand after moving the fake clock, so no test waits on real time. +type fakeGraceTimer struct { + d time.Duration + f func() + stopped bool +} + +// clientsHarness drives a daemon's client registry with a fake clock and a +// fake timer seam, and counts the elections a timer expiry changed. +type clientsHarness struct { + t *testing.T + d *Daemon + now time.Time + timers []*fakeGraceTimer + changes int +} + +func newClientsHarness(t *testing.T, grace time.Duration) *clientsHarness { + t.Helper() + h := &clientsHarness{t: t} + h.install(&Daemon{}, grace) + return h +} + +// install points d's registry at the harness's clock and timers. Used on a +// zero Daemon, and on one built by New for the snapshot and restore test. +// +// The seams are written under the registry's lock, because on a daemon with a +// live server the dispatch goroutines read them under that same lock. +func (h *clientsHarness) install(d *Daemon, grace time.Duration) { + h.d = d + h.now = time.Unix(1_800_000_000, 0) + d.clients.mu.Lock() + defer d.clients.mu.Unlock() + d.clients.now = func() time.Time { return h.now } + d.clients.afterFn = func(dur time.Duration, f func()) func() bool { + tm := &fakeGraceTimer{d: dur, f: f} + h.timers = append(h.timers, tm) + return func() bool { + was := !tm.stopped + tm.stopped = true + return was + } + } + d.clients.grace = grace + d.clients.onChange = func() { h.changes++ } +} + +func (h *clientsHarness) advance(d time.Duration) { h.now = h.now.Add(d) } + +// armed returns the live timer, or nil when none is armed. +func (h *clientsHarness) armed() *fakeGraceTimer { + var live *fakeGraceTimer + for _, tm := range h.timers { + if !tm.stopped { + live = tm + } + } + return live +} + +// fire runs the armed timer's callback, as time.AfterFunc would at expiry. +func (h *clientsHarness) fire() { + h.t.Helper() + tm := h.armed() + if tm == nil { + h.t.Fatal("no grace timer is armed") + } + tm.stopped = true + tm.f() +} + +// attach registers a new conn as the client id, at a RAW geometry. +func (h *clientsHarness) attach(id string, cols, rows int) (*ipc.Conn, bool) { + c := new(ipc.Conn) + changed := h.d.registerClient(c, ipc.AttachPayload{ClientID: id, Cols: cols, Rows: rows}) + return c, changed +} + +func (h *clientsHarness) wantMaster(want string) { + h.t.Helper() + if got := h.d.masterID(); got != want { + h.t.Fatalf("masterID = %q, want %q", got, want) + } +} + +const testGrace = 3 * time.Minute + +func TestElect_OldestPaintableWins(t *testing.T) { + h := newClientsHarness(t, testGrace) + a, changed := h.attach("A", 200, 50) + if !changed { + t.Fatal("the first paintable client must be elected, which is a change") + } + h.wantMaster("A") + + h.advance(time.Second) + b, changed := h.attach("B", 100, 30) + if changed { + t.Error("a younger client attaching must not change the master") + } + h.wantMaster("A") + if !h.d.isMasterConn(a) || h.d.isMasterConn(b) { + t.Error("isMasterConn must name A's conn only") + } + if h.d.masterConn() != a { + t.Error("masterConn must be A's conn") + } + if got := h.d.followerConns(nil); len(got) != 1 || got[0] != b { + t.Errorf("followerConns = %v, want only B's conn", got) + } + if h.d.clientCount() != 2 { + t.Errorf("clientCount = %d, want 2", h.d.clientCount()) + } + if h.d.sizeAuthorityOpen() { + t.Error("with a master elected, size authority is not open") + } +} + +// A console-less client attaches at 0x0 and a tiny one at 1x1. Electing +// either would resize every pane to one column: the 1x1 incident. +func TestElect_ZeroGeometryNeverElected(t *testing.T) { + h := newClientsHarness(t, testGrace) + if _, changed := h.attach("zero", 0, 0); changed { + t.Error("a 0x0 attach must not change the master") + } + if _, changed := h.attach("one", 1, 1); changed { + t.Error("a 1x1 attach must not change the master") + } + h.wantMaster("") + if !h.d.sizeAuthorityOpen() { + t.Error("with no master and no reservation, size authority must be open") + } + if h.d.masterConn() != nil { + t.Error("masterConn must be nil with no master") + } +} + +// The threshold is the TUI's own paintable floor, inclusive. +func TestElect_ThresholdIsTUIPaintableFloor(t *testing.T) { + h := newClientsHarness(t, testGrace) + h.attach("narrow", daemonMinClientCols-1, daemonMinClientRows) + h.attach("short", daemonMinClientCols, daemonMinClientRows-1) + h.wantMaster("") + h.attach("floor", daemonMinClientCols, daemonMinClientRows) + h.wantMaster("floor") +} + +// Review focus 1: a master whose window becomes unpaintable loses the slot at +// once, and with no eligible client there is no master at all. +func TestElect_MasterGoingUnpaintableHandsOver(t *testing.T) { + h := newClientsHarness(t, testGrace) + a, _ := h.attach("A", 200, 50) + h.advance(time.Second) + b, _ := h.attach("B", 100, 30) + h.wantMaster("A") + + if !h.d.setClientGeometry(a, 0, 0) { + t.Error("the master going unpaintable must change the master") + } + h.wantMaster("B") + + if !h.d.setClientGeometry(b, 0, 0) { + t.Error("the last eligible client going unpaintable must change the master") + } + h.wantMaster("") + if !h.d.sizeAuthorityOpen() { + t.Error("with nobody eligible, size authority must be open") + } + if h.armed() != nil { + t.Error("a geometry change is not a lost link and must arm no grace timer") + } +} + +func TestGrace_LostMasterReservedWhileFollowerAttached(t *testing.T) { + h := newClientsHarness(t, testGrace) + a, _ := h.attach("A", 200, 50) + h.advance(time.Second) + b, _ := h.attach("B", 100, 30) + + if h.d.forgetAttachedClient(a) { + t.Error("a lost master with a follower attached keeps its slot, which is no change") + } + h.wantMaster("A") + if h.d.isMasterConn(b) { + t.Error("the follower must not become master inside the grace time") + } + if h.d.sizeAuthorityOpen() { + t.Error("a reserved slot must keep size authority closed") + } + if h.d.masterConn() != nil { + t.Error("masterConn must be nil while the master's link is lost") + } + tm := h.armed() + if tm == nil || tm.d != testGrace { + t.Fatalf("grace timer = %+v, want one armed for %v", tm, testGrace) + } + + h.advance(testGrace) + h.fire() + h.wantMaster("B") + if !h.d.isMasterConn(b) { + t.Error("after the grace time the follower must be the master") + } + if h.changes != 1 { + t.Errorf("expiry changes = %d, want exactly 1", h.changes) + } +} + +// The reservation protects followers. Once every follower that was attached +// at the loss has gone too, it protects nobody and the election proceeds. +func TestGrace_ReservationLapsesWhenProtectedFollowersLeave(t *testing.T) { + h := newClientsHarness(t, testGrace) + a, _ := h.attach("A", 200, 50) + b, _ := h.attach("B", 100, 30) + h.d.forgetAttachedClient(a) + h.wantMaster("A") + + h.d.forgetAttachedClient(b) + h.advance(time.Second) + if _, changed := h.attach("C", 120, 40); !changed { + t.Error("a new client with nobody left to protect must be elected") + } + h.wantMaster("C") +} + +func TestGrace_LoneMasterNotReserved(t *testing.T) { + h := newClientsHarness(t, testGrace) + a, _ := h.attach("A", 200, 50) + if !h.d.forgetAttachedClient(a) { + t.Error("a lone master leaving must clear the master") + } + h.wantMaster("") + if h.armed() != nil { + t.Error("a lone master's loss must arm no grace timer") + } + + h.advance(time.Second) + c, changed := h.attach("C", 120, 40) + if !changed || !h.d.isMasterConn(c) { + t.Error("a relaunched TUI with a new id must be the master at once") + } +} + +func TestGrace_ReattachSameIDRestoresWithoutChange(t *testing.T) { + h := newClientsHarness(t, testGrace) + a, _ := h.attach("A", 200, 50) + firstAttach := h.now + h.advance(time.Second) + h.attach("B", 100, 30) + h.d.forgetAttachedClient(a) + tm := h.armed() + + h.advance(time.Minute) + a2, changed := h.attach("A", 200, 50) + if changed { + t.Error("the master returning inside grace must not be a master change") + } + if !h.d.isMasterConn(a2) { + t.Error("the returning master must hold the slot on its new conn") + } + if !tm.stopped || h.armed() != nil { + t.Error("the grace timer must be stopped when the master returns") + } + rec, ok := h.d.clientByConn(a2) + if !ok || !rec.attachedAt.Equal(firstAttach) { + t.Errorf("attachedAt = %v, want the first attach %v", rec.attachedAt, firstAttach) + } + if h.changes != 0 { + t.Errorf("changes = %d, want 0", h.changes) + } +} + +func TestDetach_SkipsGrace(t *testing.T) { + h := newClientsHarness(t, testGrace) + a, _ := h.attach("A", 200, 50) + h.advance(time.Second) + b, _ := h.attach("B", 100, 30) + + if !h.d.detachClient(a) { + t.Error("a detaching master must hand over at once") + } + if h.d.forgetAttachedClient(a) { + t.Error("the disconnect after a detach must find no record and change nothing") + } + if !h.d.isMasterConn(b) { + t.Error("after a clean exit the follower must be the master at once") + } + if h.armed() != nil { + t.Error("a detach must arm no grace timer") + } + if h.d.clientCount() != 1 { + t.Errorf("clientCount = %d, want 1", h.d.clientCount()) + } +} + +// A daemon restart keeps the previous master's slot for min(grace, 30 s), +// written to and read back from a real workspace.json. +func TestRestartReserve_PreviousMasterReclaims(t *testing.T) { + restartedWithMaster := func(t *testing.T) *clientsHarness { + t.Helper() + home := t.TempDir() + d1 := newTestDaemonInDir(t, home) + h1 := &clientsHarness{t: t} + h1.install(d1, testGrace) + h1.attach("A", 200, 50) + h1.wantMaster("A") + d1.snapshot() + + d2 := newTestDaemonInDir(t, home) + h := &clientsHarness{t: t} + h.install(d2, testGrace) + if err := d2.restoreWorkspace(); err != nil { + t.Fatalf("restoreWorkspace: %v", err) + } + h.wantMaster("A") + if tm := h.armed(); tm == nil || tm.d != restartReserveCap { + t.Fatalf("restart timer = %+v, want one armed for %v", tm, restartReserveCap) + } + return h + } + + t.Run("previous master reattaches after another client", func(t *testing.T) { + h := restartedWithMaster(t) + b, changed := h.attach("B", 100, 30) + if changed || h.d.isMasterConn(b) { + t.Error("a client attaching first after a restart must not take the reserved slot") + } + if h.d.sizeAuthorityOpen() { + t.Error("the restart reservation must keep size authority closed") + } + h.advance(2 * time.Second) + a, changed := h.attach("A", 200, 50) + if changed { + t.Error("the previous master reclaiming its slot must not be a change") + } + if !h.d.isMasterConn(a) { + t.Error("the previous master must be the master again") + } + if h.armed() != nil { + t.Error("the restart timer must be stopped once the master is back") + } + }) + + t.Run("previous master never comes", func(t *testing.T) { + h := restartedWithMaster(t) + b, _ := h.attach("B", 100, 30) + h.advance(restartReserveCap) + h.fire() + if !h.d.isMasterConn(b) { + t.Error("after the restart reserve lapses, the oldest attached client must win") + } + if h.changes != 1 { + t.Errorf("changes = %d, want 1", h.changes) + } + }) +} + +// Stop closes every conn, and each close runs onClientDisconnect concurrently +// with the final snapshot. Those disconnects are not lost links: treating them +// as such clears a lone master (nobody left to protect) before the snapshot +// writes size_master, and the restart reserve then has nothing to restore. +func TestRestartReserve_ShutdownDisconnectKeepsMaster(t *testing.T) { + home := t.TempDir() + d := newTestDaemonInDir(t, home) + h := &clientsHarness{t: t} + h.install(d, testGrace) + a, _ := h.attach("A", 200, 50) + + d.shutdownOnce.Do(func() { close(d.shutdown) }) + d.onClientDisconnect(a) + h.wantMaster("A") + d.snapshot() + + d2 := newTestDaemonInDir(t, home) + h2 := &clientsHarness{t: t} + h2.install(d2, testGrace) + if err := d2.restoreWorkspace(); err != nil { + t.Fatalf("restoreWorkspace: %v", err) + } + h2.wantMaster("A") +} + +// A shorter configured grace also shortens the restart reserve. +func TestRestartReserve_BoundedByGrace(t *testing.T) { + h := newClientsHarness(t, 10*time.Second) + h.d.clients.reserveAfterRestart("A") + if tm := h.armed(); tm == nil || tm.d != 10*time.Second { + t.Fatalf("restart timer = %+v, want 10s", tm) + } + + h = newClientsHarness(t, 0) + h.d.clients.reserveAfterRestart("A") + h.wantMaster("") + if h.armed() != nil || !h.d.sizeAuthorityOpen() { + t.Error("with no grace there is no restart reserve") + } +} + +func TestTakeControl_EligibleFollowerBecomesMaster(t *testing.T) { + h := newClientsHarness(t, testGrace) + h.attach("A", 200, 50) + h.advance(time.Second) + b, _ := h.attach("B", 100, 30) + + if !h.d.takeControl(b) { + t.Error("take_control from an eligible follower must change the master") + } + if !h.d.isMasterConn(b) { + t.Error("the follower must be the master after take_control") + } + if h.d.takeControl(b) { + t.Error("take_control from the master itself changes nothing") + } +} + +func TestTakeControl_IneligibleIgnored(t *testing.T) { + h := newClientsHarness(t, testGrace) + a, _ := h.attach("A", 200, 50) + b, _ := h.attach("B", 0, 0) + + if h.d.takeControl(b) { + t.Error("take_control from an unpaintable follower must be ignored") + } + if h.d.takeControl(new(ipc.Conn)) { + t.Error("take_control from a conn that never attached must be ignored") + } + if !h.d.isMasterConn(a) { + t.Error("the master must be unchanged") + } +} + +func TestClients_AnonymousIDAndTruncation(t *testing.T) { + h := newClientsHarness(t, testGrace) + c, _ := h.attach("", 200, 50) + rec, ok := h.d.clientByConn(c) + if !ok || len(rec.id) <= len("anon-") || rec.id[:5] != "anon-" { + t.Fatalf("id = %q, want a daemon-minted anon-", rec.id) + } + // A re-attach on the same conn with no id keeps the minted one. + h.d.registerClient(c, ipc.AttachPayload{Cols: 200, Rows: 50}) + if again, _ := h.d.clientByConn(c); again.id != rec.id { + t.Errorf("re-attach minted a new id %q, want %q", again.id, rec.id) + } + + long := make([]byte, 500) + for i := range long { + long[i] = 'x' + } + c2, _ := h.attach(string(long), 200, 50) + if rec2, _ := h.d.clientByConn(c2); len(rec2.id) != maxClientIDLen { + t.Errorf("id length = %d, want %d", len(rec2.id), maxClientIDLen) + } +} + +func TestClients_ListAndMostRecentlyActive(t *testing.T) { + h := newClientsHarness(t, testGrace) + if h.d.mostRecentlyActiveConn() != nil { + t.Error("with no client, there is no most recently active conn") + } + a, _ := h.attach("A", 200, 50) + h.advance(time.Second) + b, _ := h.attach("B", 100, 30) + if h.d.mostRecentlyActiveConn() != b { + t.Error("with no input yet, the most recently attached client is the most active") + } + + h.advance(time.Second) + h.d.touchClientInput(a) + typedAt := h.now + if h.d.mostRecentlyActiveConn() != a { + t.Error("the client that typed last must be the most active") + } + h.d.touchClientInput(new(ipc.Conn)) // a bridge: not a client, never stamped + + list := h.d.listClients() + if len(list) != 2 { + t.Fatalf("listClients = %d rows, want 2", len(list)) + } + byID := map[string]ipc.ClientInfo{} + for _, ci := range list { + byID[ci.Client] = ci + } + if ci := byID["A"]; !ci.Master || ci.Cols != 200 || ci.Rows != 50 || + ci.LastInputAt != typedAt.UTC().Format(time.RFC3339) || ci.AttachedAt == "" { + t.Errorf("A = %+v", ci) + } + if ci := byID["B"]; ci.Master || ci.LastInputAt != "" { + t.Errorf("B = %+v, want a follower that never typed", ci) + } +} diff --git a/internal/daemon/clients_wiring_test.go b/internal/daemon/clients_wiring_test.go new file mode 100644 index 00000000..797d3b2a --- /dev/null +++ b/internal/daemon/clients_wiring_test.go @@ -0,0 +1,129 @@ +package daemon + +import ( + "testing" + + "github.com/artyomsv/quil/internal/ipc" +) + +// attachClientAs dials the daemon and attaches as the client id at a RAW +// window size, like a TUI does. +func attachClientAs(t *testing.T, sock, id string, cols, rows int) *ipc.Client { + t.Helper() + c, err := ipc.NewClient(sock) + if err != nil { + t.Fatalf("dial: %v", err) + } + t.Cleanup(func() { c.Close() }) + sendClientMsg(t, c, ipc.MsgAttach, ipc.AttachPayload{ClientID: id, Cols: cols, Rows: rows}) + return c +} + +func sendClientMsg(t *testing.T, c *ipc.Client, typ string, payload any) { + t.Helper() + msg, err := ipc.NewMessage(typ, payload) + if err != nil { + t.Fatalf("build %s: %v", typ, err) + } + if err := c.Send(msg); err != nil { + t.Fatalf("send %s: %v", typ, err) + } +} + +// clientRecordByID reads one client's record by id, for tests that hold only +// the client side of a conn. +func clientRecordByID(d *Daemon, id string) (clientRecord, bool) { + d.clients.mu.Lock() + defer d.clients.mu.Unlock() + for _, rec := range d.clients.byConn { + if rec.id == id { + cp := *rec + return cp, true + } + } + return clientRecord{}, false +} + +// Proves the dispatch arms are WIRED: every step goes over a real socket, so a +// missing case arm or a stamp in the wrong arm fails here even when the +// registry's own tests pass. +func TestClientDispatch_ArmsAreWired(t *testing.T) { + d, sock := overlayServerDaemon(t) + + a := attachClientAs(t, sock, "A", 200, 50) + waitUntil(t, "A registered", func() bool { return d.masterID() == "A" }) + b := attachClientAs(t, sock, "B", 100, 30) + waitUntil(t, "B registered", func() bool { return d.clientCount() == 2 }) + if d.masterID() != "A" { + t.Fatalf("masterID = %q, want A (the oldest paintable client)", d.masterID()) + } + + // pane_input stamps lastInputAt even when the pane does not exist: the + // user typed, and the stamp is about the client, not the delivery. + if rec, _ := clientRecordByID(d, "B"); !rec.lastInputAt.IsZero() { + t.Fatal("B has input before it typed anything") + } + sendClientMsg(t, b, ipc.MsgPaneInput, ipc.PaneInputPayload{PaneID: "pane-none", Data: []byte("x")}) + waitUntil(t, "B's lastInputAt stamped", func() bool { + rec, _ := clientRecordByID(d, "B") + return !rec.lastInputAt.IsZero() + }) + if rec, _ := clientRecordByID(d, "A"); !rec.lastInputAt.IsZero() { + t.Error("B's input stamped A") + } + + sendClientMsg(t, b, ipc.MsgTakeControl, nil) + waitUntil(t, "take_control makes B master", func() bool { return d.masterID() == "B" }) + + sendClientMsg(t, b, ipc.MsgClientGeometry, ipc.ClientGeometryPayload{Cols: 0, Rows: 0}) + waitUntil(t, "B going unpaintable hands back to A", func() bool { return d.masterID() == "A" }) + + sendClientMsg(t, a, ipc.MsgDetach, nil) + waitUntil(t, "A's detach removes it at once", func() bool { return d.clientCount() == 1 }) + if d.masterID() != "" { + t.Errorf("masterID = %q, want none: B is unpaintable and A detached", d.masterID()) + } + if !d.sizeAuthorityOpen() { + t.Error("a detach must leave no reserved slot") + } +} + +// The attach geometry the registry records is the RAW one. handleAttach +// defaults clientSize to 80x24 for a 0x0 attach, and eligibility must not see +// that default: it is the 1x1 incident coming back through a new door. +func TestClientDispatch_ZeroAttachNeverElectedDespiteDefault(t *testing.T) { + d, sock := overlayServerDaemon(t) + attachClientAs(t, sock, "headless", 0, 0) + waitUntil(t, "client registered", func() bool { return d.clientCount() == 1 }) + waitUntil(t, "clientSize defaulted", func() bool { return d.clientSize.Load() != nil }) + + if sz := d.clientSize.Load(); sz.cols != 80 || sz.rows != 24 { + t.Fatalf("clientSize = %dx%d, want the 80x24 default", sz.cols, sz.rows) + } + if d.masterID() != "" { + t.Errorf("masterID = %q, want none for a 0x0 attach", d.masterID()) + } + rec, _ := clientRecordByID(d, "headless") + if rec.cols != 0 || rec.rows != 0 { + t.Errorf("recorded geometry = %dx%d, want the raw 0x0", rec.cols, rec.rows) + } +} + +// A lost link through the real server keeps the master's slot while a +// follower is attached. +func TestClientDispatch_LostLinkReservesSlot(t *testing.T) { + d, sock := overlayServerDaemon(t) + // Fake timers, so no real grace timer outlives the test. Installed before + // any client attaches; nothing below reads the harness's own fields. + (&clientsHarness{t: t}).install(d, testGrace) + a := attachClientAs(t, sock, "A", 200, 50) + waitUntil(t, "A master", func() bool { return d.masterID() == "A" }) + attachClientAs(t, sock, "B", 100, 30) + waitUntil(t, "B registered", func() bool { return d.clientCount() == 2 }) + + a.Close() + waitUntil(t, "A's disconnect processed", func() bool { return d.clientCount() == 1 }) + if d.masterID() != "A" || d.sizeAuthorityOpen() { + t.Errorf("masterID = %q, open = %v; want A's slot reserved", d.masterID(), d.sizeAuthorityOpen()) + } +} diff --git a/internal/daemon/daemon.go b/internal/daemon/daemon.go index 2d3381bb..4377db84 100644 --- a/internal/daemon/daemon.go +++ b/internal/daemon/daemon.go @@ -255,9 +255,10 @@ type Daemon struct { // pushes runtime updates via MsgOverlayPolicy without a daemon restart. overlayPolicyState overlayPolicyState - // attachedConns maps each conn that has sent MsgAttach — the clients, as - // distinct from every conn (see markClientAttached) — to the set of overlay - // panes that client currently has ON SCREEN. + // clients holds one record for each conn that has sent MsgAttach — the + // clients, as distinct from every conn (see registerClient) — with the + // size-master election over them (clients.go). Each record also carries + // the set of overlay panes that client currently has ON SCREEN. // // Visibility is per client rather than one daemon-wide field because // otherwise whichever conn spoke last defines it: with two TUIs attached, @@ -267,13 +268,12 @@ type Daemon struct { // claims it, which is also what makes a detached session fall out for free // — no clients, no claims, everything hidden. // - // Written from each conn's own dispatch goroutine and from the disconnect - // callback, so it carries its own mutex: sm.mu is the wrong lock here, - // since a reader parked behind an RWMutex writer is the failure mode this - // package keeps being bitten by. Nothing that takes PluginMu may be called - // while it is held. - attachedMu sync.Mutex - attachedConns map[*ipc.Conn]map[string]bool + // Written from each conn's own dispatch goroutine, the disconnect callback + // and the grace timer, so it carries its own mutex: sm.mu is the wrong + // lock here, since a reader parked behind an RWMutex writer is the failure + // mode this package keeps being bitten by. Nothing that takes PluginMu may + // be called while it is held, and nothing broadcasts while it is held. + clients clientRegistry } func New(cfg config.Config) *Daemon { @@ -310,6 +310,10 @@ func New(cfg config.Config) *Daemon { d.memReport = memreport.NewCollector(d.session, 5*time.Second) d.procReport = newProcCollector(d.session, memreport.ProcRSSBatch) d.hellos = newHelloRegistry() + d.clients.now = time.Now + d.clients.afterFn = realAfterFunc + d.clients.grace = cfg.Daemon.MasterGrace() + d.clients.onChange = d.broadcastState d.startedAt = time.Now() // Clamped like a pushed policy: config.toml is hand-edited, so it can carry // exactly the values the IPC path is bounded against. @@ -600,50 +604,25 @@ func (d *Daemon) Stop() { }) } -// markClientAttached records a connection that has sent MsgAttach. -// -// ATTACHMENT, not connection, is what "a client is here" means, and the -// difference is not academic: every live MCP bridge holds an IPC conn for its -// whole lifetime (cmd/quil/mcp.go dials once and closes on exit), and a bridge -// is a child of the claude process in a PANE — so bridges routinely outlive the -// TUI. Counting raw conns therefore answered "is anything connected", which in -// any session with a claude pane wired to `quil mcp` is permanently yes (21 -// conns in the session that reported 7 live overlays), and the detached-session -// stamp below never fired in exactly the configuration it was designed for. -// Re-attaching on the same conn keeps that client's existing overlay claims: -// the entry is created only when absent. -func (d *Daemon) markClientAttached(conn *ipc.Conn) { - if conn == nil { - return - } - d.attachedMu.Lock() - if d.attachedConns == nil { - d.attachedConns = make(map[*ipc.Conn]map[string]bool) - } - if _, ok := d.attachedConns[conn]; !ok { - d.attachedConns[conn] = map[string]bool{} - } - d.attachedMu.Unlock() -} - -// forgetAttachedClient drops a disconnecting conn, and with it every overlay -// that client claimed visible. A conn that never attached is not in the set, so -// dropping it changes nothing. -func (d *Daemon) forgetAttachedClient(conn *ipc.Conn) { - d.attachedMu.Lock() - delete(d.attachedConns, conn) - d.attachedMu.Unlock() -} - // onClientDisconnect is ipc.Server's disconnect callback. // // handleConn's defer removes the disconnecting conn (removeConn) before // invoking this, and the attached set is keyed on that same conn — so the state // here is already exclusive of the client that just left. +// +// A client that sent MsgDetach first is already gone from the registry, so +// only a LOST link reaches the grace logic here. +// +// A disconnect caused by our own shutdown is not a lost link, and the record is +// kept. Stop closes every conn while the final snapshot runs, and dropping the +// records there would clear a lone master (nobody left to protect) before the +// snapshot writes size_master — so the restart would have no reserve. func (d *Daemon) onClientDisconnect(conn *ipc.Conn) { d.requestSnapshot() d.events.RemoveWatchersByConn(conn) - d.forgetAttachedClient(conn) + if !d.shuttingDown() && d.forgetAttachedClient(conn) { + d.broadcastState() + } // Drop this conn's identity with it: the process it described is gone, // and a retained entry would be listed as running. d.hellos.forget(conn) @@ -673,6 +652,12 @@ func (d *Daemon) snapshot() { // N±1, surfacing as the "snapshot pane count oscillation" bug. activeTab, tabs, panesByTab, projects, activeProject := d.session.SnapshotState() state := d.workspaceStateFromSnapshot(activeTab, tabs, panesByTab, projects, activeProject, false) + // Disk only, never on the broadcast: restoreWorkspace turns it into a short + // reservation so the previous size master gets its slot back after a + // restart, and the reattach resizes nothing. + if id := d.masterID(); id != "" { + state["size_master"] = id + } if err := persist.Save(config.WorkspacePath(), state); err != nil { log.Printf("snapshot workspace: %v", err) @@ -832,6 +817,8 @@ func (d *Daemon) restoreWorkspace() error { tabs, _ := state["tabs"].([]any) panes, _ := state["panes"].([]any) activeProject, _ := state["active_project"].(string) + sizeMaster, _ := state["size_master"].(string) + d.clients.reserveAfterRestart(sizeMaster) d.session.RestoreProjects(parseRestoredProjects(state["projects"]), activeProject) @@ -1383,16 +1370,28 @@ func (d *Daemon) handleMessage(conn *ipc.Conn, msg *ipc.Message) { // several lines a second forever. Logging it would churn quild.log through // its rotation and bury the lifecycle lines this log exists for. switch msg.Type { - case ipc.MsgPaneInput, ipc.MsgResizePane, ipc.MsgUpdateLayout, ipc.MsgClientStat: + case ipc.MsgPaneInput, ipc.MsgResizePane, ipc.MsgUpdateLayout, ipc.MsgClientStat, + ipc.MsgClientGeometry: // skip logging — too noisy default: log.Printf("ipc recv: %s", msg.Type) } + // touchClientInput marks the user-originated messages below (input, tab + // switch, create, layout, pane update, take control). The latest one picks + // which client an untargeted MCP close_tui or set_active_pane reaches. switch msg.Type { case ipc.MsgAttach: d.handleAttach(conn, msg) + case ipc.MsgDetach: + d.handleDetach(conn) + case ipc.MsgClientGeometry: + d.handleClientGeometry(conn, msg) + case ipc.MsgTakeControl: + d.touchClientInput(conn) + d.handleTakeControl(conn) case ipc.MsgCreateTab: + d.touchClientInput(conn) d.handleCreateTab(conn, msg) case ipc.MsgDestroyTab: // The existence check happens HERE, before the handler, because the @@ -1401,6 +1400,7 @@ func (d *Daemon) handleMessage(conn *ipc.Conn, msg *ipc.Message) { d.handleDestroyTab(msg) answerOp(conn, msg, ipc.MsgTabOpResp, id, known, opErrUnless(known, "no such tab")) case ipc.MsgSwitchTab: + d.touchClientInput(conn) d.handleSwitchTab(msg) case ipc.MsgUpdateTab: id, known := tabIDKnown(d, msg, "tab_id") @@ -1411,18 +1411,22 @@ func (d *Daemon) handleMessage(conn *ipc.Conn, msg *ipc.Message) { case ipc.MsgMoveTab: d.handleMoveTab(conn, msg) case ipc.MsgCreatePane: + d.touchClientInput(conn) d.handleCreatePane(conn, msg) case ipc.MsgDestroyPane: d.handleDestroyPane(msg) case ipc.MsgUpdatePane: + d.touchClientInput(conn) id, known := paneIDKnown(d, msg) d.handleUpdatePane(conn, msg) answerOp(conn, msg, ipc.MsgPaneOpResp, id, known, opErrUnless(known, "no such pane")) case ipc.MsgMovePane: d.handleMovePane(conn, msg) case ipc.MsgUpdateLayout: + d.touchClientInput(conn) d.handleUpdateLayout(msg) case ipc.MsgPaneInput: + d.touchClientInput(conn) d.handlePaneInput(conn, msg) case ipc.MsgResizePane: d.handleResizePane(msg) @@ -1770,8 +1774,15 @@ func (d *Daemon) handleAttach(conn *ipc.Conn, msg *ipc.Message) { // This is what makes the conn a CLIENT rather than just a connection — the // distinction the detached-session overlay stamp turns on. Recorded before - // any of the work below, which has early returns of its own. - d.markClientAttached(conn) + // any of the work below, which has early returns of its own, and before + // the 80x24 defaulting: the election reads the RAW geometry. + // + // A master change is broadcast only once this attach is answered, so the + // new client's first workspace state is its own full one rather than a + // broadcast of a workspace this attach may be about to create. + if d.registerClient(conn, attach) { + defer d.broadcastState() + } cols, rows := attach.Cols, attach.Rows if cols <= 0 { diff --git a/internal/daemon/overlay.go b/internal/daemon/overlay.go index 49af5d5b..75793e46 100644 --- a/internal/daemon/overlay.go +++ b/internal/daemon/overlay.go @@ -189,27 +189,27 @@ func (d *Daemon) setOverlayClaim(conn *ipc.Conn, paneID string, visible bool) { if conn == nil { return } - d.attachedMu.Lock() - defer d.attachedMu.Unlock() - claims, ok := d.attachedConns[conn] + d.clients.mu.Lock() + defer d.clients.mu.Unlock() + rec, ok := d.clients.byConn[conn] if !ok { return } if visible { - claims[paneID] = true + rec.overlays[paneID] = true return } - delete(claims, paneID) + delete(rec.overlays, paneID) } // overlayClaimed reports whether ANY attached client currently has this overlay // on screen. The attached set holds one entry per client (one, occasionally // two), so the walk is trivially cheap. func (d *Daemon) overlayClaimed(paneID string) bool { - d.attachedMu.Lock() - defer d.attachedMu.Unlock() - for _, claims := range d.attachedConns { - if claims[paneID] { + d.clients.mu.Lock() + defer d.clients.mu.Unlock() + for _, rec := range d.clients.byConn { + if rec.overlays[paneID] { return true } } @@ -221,10 +221,10 @@ func (d *Daemon) overlayClaimed(paneID string) bool { // owes, so a destroyed overlay cannot leave its id in a live client's claim set // for the life of a daemon that runs for weeks. func (d *Daemon) forgetOverlayClaimsFor(paneID string) { - d.attachedMu.Lock() - defer d.attachedMu.Unlock() - for _, claims := range d.attachedConns { - delete(claims, paneID) + d.clients.mu.Lock() + defer d.clients.mu.Unlock() + for _, rec := range d.clients.byConn { + delete(rec.overlays, paneID) } } @@ -237,7 +237,7 @@ func (d *Daemon) forgetOverlayClaimsFor(paneID string) { // model — two TUIs used to mean the second one's tab switch started a // five-minute countdown on the first one's visible lazygit. // -// The claim is resolved BEFORE PluginMu is taken. attachedMu must never nest +// The claim is resolved BEFORE PluginMu is taken. clients.mu must never nest // inside a pane lock. func (d *Daemon) applyOverlayVisibility(conn *ipc.Conn, pane *Pane, visible bool) { d.setOverlayClaim(conn, pane.ID, visible) diff --git a/internal/daemon/procreport.go b/internal/daemon/procreport.go index 9b1b8c9b..b0d46c5c 100644 --- a/internal/daemon/procreport.go +++ b/internal/daemon/procreport.go @@ -246,7 +246,7 @@ type helloRecord struct { // helloRegistry tracks which connections have identified themselves. // -// Its own mutex, never sm.mu — the same rule attachedConns follows, and for the +// Its own mutex, never sm.mu — the same rule the client registry follows, and for the // same reason: this is read while assembling a report and written from every // dispatch goroutine, and coupling it to the session lock would put a second // writer in front of the snapshot loop. @@ -301,6 +301,18 @@ func (r *helloRegistry) roleOf(conn *ipc.Conn) string { return r.byConn[conn].payload.Role } +// helloOf returns a connection's self-description, and false when it never +// said hello. +func (r *helloRegistry) helloOf(conn *ipc.Conn) (ipc.ClientHelloPayload, bool) { + if conn == nil { + return ipc.ClientHelloPayload{}, false + } + r.mu.RLock() + defer r.mu.RUnlock() + rec, ok := r.byConn[conn] + return rec.payload, ok +} + // putStat records a client's latest self-measurement. // // A stat for a connection that never said hello is DROPPED, not stored. Rows From 3a3a704e0b4ad41cfb59c9cabecdc0657124d59a Mon Sep 17 00:00:00 2001 From: Artjoms Stukans Date: Fri, 25 Sep 2026 16:45:04 +0200 Subject: [PATCH 03/40] feat(daemon): apply pane resizes from the size master only resize_pane and the new batched resize_panes share one implementation, applyResizes. It applies a resize only when the sender is the size master, or when no master is elected and no slot is reserved (the single-client behaviour). A refused resize is dropped with no log line. Before any PTY is resized, every follower gets ONE pane_sizes frame for the whole batch on its must-deliver queue, which is drained ahead of pane output, so its VT holds the new size before the child's repaint arrives. A failed Resize sends a second frame with the previous size. A pane named twice in one batch is resized once, to its last size. Workspace-state broadcasts now carry size_master and clients; the workspace.json map still carries neither. A new pane starts at the master's raw window size instead of the last client to attach. Also: attach and client_geometry sizes are clamped to 1000x1000, Stop disarms the grace/reserve timer, and an ignored take_control logs at debug level. --- internal/daemon/clients.go | 32 +- internal/daemon/daemon.go | 216 ++++++--- internal/daemon/pane_initial_size.go | 28 +- internal/daemon/redraw_kick_test.go | 2 +- internal/daemon/resize_authority_test.go | 543 +++++++++++++++++++++++ internal/daemon/resize_guard_test.go | 26 +- internal/daemon/resize_repaint_test.go | 8 +- 7 files changed, 771 insertions(+), 84 deletions(-) create mode 100644 internal/daemon/resize_authority_test.go diff --git a/internal/daemon/clients.go b/internal/daemon/clients.go index 4033003b..c6971f27 100644 --- a/internal/daemon/clients.go +++ b/internal/daemon/clients.go @@ -1,7 +1,6 @@ package daemon import ( - "log" "sort" "sync" "time" @@ -9,6 +8,7 @@ import ( "github.com/google/uuid" "github.com/artyomsv/quil/internal/ipc" + "github.com/artyomsv/quil/internal/logger" ) // Multi-client sync: the registry of ATTACHED clients and the size-master @@ -40,6 +40,11 @@ const ( // maxClientIDLen bounds a client's self-reported id. A UUID is 36 bytes; // the id is used only as a map key and a display value. maxClientIDLen = 64 + + // maxClientDim bounds a client's self-reported window size, in cells, on + // each axis. The size feeds a new pane's PTY, so a value above it is taken + // as the ceiling rather than trusted verbatim. + maxClientDim = 1000 ) // clientRecord is one attached client. The registry is keyed by conn, and a @@ -197,6 +202,18 @@ func (r *clientRegistry) reserveLocked(res *reservation, d time.Duration) { r.timerStop = after(d, r.expire) } +// stopTimer disarms the grace or reserve timer at daemon shutdown, so it never +// fires into a stopped daemon. The reservation itself is kept: the final +// snapshot still writes the reserved id as size_master. +func (r *clientRegistry) stopTimer() { + r.mu.Lock() + defer r.mu.Unlock() + if r.timerStop != nil { + r.timerStop() + r.timerStop = nil + } +} + // expire is the timer callback. A stale fire (the timer was replaced or the // master returned) is harmless: the election is idempotent. func (r *clientRegistry) expire() { @@ -360,7 +377,14 @@ func (d *Daemon) registerClient(conn *ipc.Conn, attach ipc.AttachPayload) bool { return false } id := truncateField(attach.ClientID, maxClientIDLen) - return d.clients.attach(conn, id, attach.Cols, attach.Rows, attach.CWD) + return d.clients.attach(conn, id, clampClientDim(attach.Cols), clampClientDim(attach.Rows), attach.CWD) +} + +// clampClientDim bounds one axis of a self-reported window size to +// [0, maxClientDim]. It never defaults: 0 stays 0, which eligibility reads as +// "not paintable". +func clampClientDim(v int) int { + return max(0, min(v, maxClientDim)) } // forgetAttachedClient drops a conn whose link was lost, and with it every @@ -388,7 +412,7 @@ func (d *Daemon) setClientGeometry(conn *ipc.Conn, cols, rows int) bool { func (d *Daemon) takeControl(conn *ipc.Conn) bool { changed, accepted := d.clients.takeControl(conn) if !accepted { - log.Printf("take_control: ignored (sender not attached or not paintable)") + logger.Debug("take_control: ignored (sender not attached or not paintable)") } return changed } @@ -414,7 +438,7 @@ func (d *Daemon) handleClientGeometry(conn *ipc.Conn, msg *ipc.Message) { if err := msg.DecodePayload(&p); err != nil { return } - if d.setClientGeometry(conn, p.Cols, p.Rows) { + if d.setClientGeometry(conn, clampClientDim(p.Cols), clampClientDim(p.Rows)) { d.broadcastState() } } diff --git a/internal/daemon/daemon.go b/internal/daemon/daemon.go index 4377db84..3a8517b7 100644 --- a/internal/daemon/daemon.go +++ b/internal/daemon/daemon.go @@ -576,6 +576,8 @@ func (d *Daemon) Stop() { if d.server != nil { d.server.Stop() } + // No client can attach any more, so nothing can re-arm it. + d.clients.stopTimer() d.collectorWG.Wait() // Pull the latest hook-recorded session ids into PluginState so // the final snapshot survives even if the hook files are lost. @@ -652,9 +654,10 @@ func (d *Daemon) snapshot() { // N±1, surfacing as the "snapshot pane count oscillation" bug. activeTab, tabs, panesByTab, projects, activeProject := d.session.SnapshotState() state := d.workspaceStateFromSnapshot(activeTab, tabs, panesByTab, projects, activeProject, false) - // Disk only, never on the broadcast: restoreWorkspace turns it into a short - // reservation so the previous size master gets its slot back after a - // restart, and the reattach resizes nothing. + // Written here explicitly, because workspaceStateFromSnapshot leaves it out + // (the broadcast adds its own in buildWorkspaceState): restoreWorkspace + // turns it into a short reservation so the previous size master gets its + // slot back after a restart, and the reattach resizes nothing. if id := d.masterID(); id != "" { state["size_master"] = id } @@ -1370,8 +1373,8 @@ func (d *Daemon) handleMessage(conn *ipc.Conn, msg *ipc.Message) { // several lines a second forever. Logging it would churn quild.log through // its rotation and bury the lifecycle lines this log exists for. switch msg.Type { - case ipc.MsgPaneInput, ipc.MsgResizePane, ipc.MsgUpdateLayout, ipc.MsgClientStat, - ipc.MsgClientGeometry: + case ipc.MsgPaneInput, ipc.MsgResizePane, ipc.MsgResizePanes, ipc.MsgUpdateLayout, + ipc.MsgClientStat, ipc.MsgClientGeometry: // skip logging — too noisy default: log.Printf("ipc recv: %s", msg.Type) @@ -1429,7 +1432,9 @@ func (d *Daemon) handleMessage(conn *ipc.Conn, msg *ipc.Message) { d.touchClientInput(conn) d.handlePaneInput(conn, msg) case ipc.MsgResizePane: - d.handleResizePane(msg) + d.handleResizePane(conn, msg) + case ipc.MsgResizePanes: + d.handleResizePanes(conn, msg) case ipc.MsgReloadPlugins: d.handleReloadPlugins() case ipc.MsgOverlayPolicy: @@ -1784,7 +1789,9 @@ func (d *Daemon) handleAttach(conn *ipc.Conn, msg *ipc.Message) { defer d.broadcastState() } - cols, rows := attach.Cols, attach.Rows + // clientSize sizes the first pane of an empty workspace below, and new + // panes whenever no master is elected (initialPaneSize). + cols, rows := clampClientDim(attach.Cols), clampClientDim(attach.Rows) if cols <= 0 { cols = 80 } @@ -3552,68 +3559,159 @@ func (d *Daemon) notifyDegenerateResize(pane *Pane, cols, rows uint16) { log.Printf("pane %s: refusing degenerate resize to %dx%d", pane.ID, cols, rows) } -func (d *Daemon) handleResizePane(msg *ipc.Message) { +// handleResizePane applies one pane's resize from conn. See applyResizes. +func (d *Daemon) handleResizePane(conn *ipc.Conn, msg *ipc.Message) { var payload ipc.ResizePanePayload if err := msg.DecodePayload(&payload); err != nil { return } + d.applyResizes(conn, []ipc.ResizePanePayload{payload}) +} - pane := d.session.Pane(payload.PaneID) - if pane == nil { +// handleResizePanes applies a batch of resizes from conn: a window resize or a +// split-drag release, which moves many panes at once. See applyResizes. +func (d *Daemon) handleResizePanes(conn *ipc.Conn, msg *ipc.Message) { + var payload ipc.ResizePanesPayload + if err := msg.DecodePayload(&payload); err != nil { return } - // Degenerate-geometry floor — see degenerateSize for why BOTH dimensions - // must be at the floor. A client with no console attached is reported by - // Bubble Tea as 1x1, and the TUI's own floors (paneVTSize) turn that into a - // request that looks perfectly legal by the time it lands here. Applied, it - // reflows every child to one column and each transcript re-wraps - // permanently — seen twice in production against a 48-tab workspace. - // Model.terminalPaintable now refuses to send it; this is the same refusal - // for an older or third-party client. - // - // BELOW the pane lookup, not above it, so the log names a pane that exists. - // PaneID is bounded only by the 10 MB IPC frame cap while quild.log's whole - // budget is 5 MB x 10 files, so an echo on a pre-lookup path lets a - // malformed payload evict the history an operator needs to diagnose this - // very incident. A resolved pane's id is one the daemon minted itself. - if degenerateSize(int(payload.Cols), int(payload.Rows)) { - d.notifyDegenerateResize(pane, payload.Cols, payload.Rows) + d.applyResizes(conn, payload.Panes) +} + +// applyResizes is the one implementation behind resize_pane and resize_panes. +// +// Only the size master may resize: each PTY has one size, and several clients +// sizing it to their own windows would reflow the child on every broadcast. +// While there is no master and no reserved slot, any client may, which is the +// single-client behaviour and keeps an older client working. A refused resize +// is dropped with no log line, because a follower on an older build sends one +// for every pane on every broadcast. +// +// Every follower is told the new sizes BEFORE any PTY is resized, in ONE frame +// for the whole batch. The frame goes on the must-deliver queue, which each +// conn drains ahead of pane output, so a follower's VT holds the new size +// before the child's repaint at that size arrives. Without it the repaint lands +// in the old-sized VT and is then reflowed. One frame per batch rather than per +// pane, because a window resize across 40+ panes would otherwise put 40+ +// must-deliver frames on a follower's 64-slot queue at once. +func (d *Daemon) applyResizes(conn *ipc.Conn, items []ipc.ResizePanePayload) { + if !d.isMasterConn(conn) && !d.sizeAuthorityOpen() { + return // a follower or a stale sender: dropped silently (spec §4.1) + } + type todo struct { + pane *Pane + pty apty.Session + typ string + cols, rows uint16 + prevC, prevR int + } + var work []todo + seen := make(map[string]int, len(items)) + for _, it := range items { + pane := d.session.Pane(it.PaneID) + if pane == nil { + continue + } + // Degenerate-geometry floor — see degenerateSize for why BOTH dimensions + // must be at the floor. A client with no console attached is reported by + // Bubble Tea as 1x1, and the TUI's own floors (paneVTSize) turn that into a + // request that looks perfectly legal by the time it lands here. Applied, it + // reflows every child to one column and each transcript re-wraps + // permanently — seen twice in production against a 48-tab workspace. + // Model.terminalPaintable now refuses to send it; this is the same refusal + // for an older or third-party client. + // + // BELOW the pane lookup, not above it, so the log names a pane that exists. + // PaneID is bounded only by the 10 MB IPC frame cap while quild.log's whole + // budget is 5 MB x 10 files, so an echo on a pre-lookup path lets a + // malformed payload evict the history an operator needs to diagnose this + // very incident. A resolved pane's id is one the daemon minted itself. + if degenerateSize(int(it.Cols), int(it.Rows)) { + d.notifyDegenerateResize(pane, it.Cols, it.Rows) + continue + } + // Same-size guard: skip when this exact size was already applied to + // the current PTY (the TUI re-sends all pane sizes on every workspace + // broadcast). Guard fields are PluginMu-protected; the Resize syscall + // runs outside the lock. + pane.PluginMu.Lock() + pty, typ := pane.PTY, pane.Type + same := pane.appliedCols == int(it.Cols) && pane.appliedRows == int(it.Rows) + prevC, prevR := pane.appliedCols, pane.appliedRows + pane.PluginMu.Unlock() + if pty == nil || same { + continue + } + w := todo{pane, pty, typ, it.Cols, it.Rows, prevC, prevR} + // A pane named twice in one batch is resized once, to its last size. + // Otherwise each copy passes the guard above, which reads the size + // applied BEFORE this batch. + if i, dup := seen[pane.ID]; dup { + work[i] = w + continue + } + seen[pane.ID] = len(work) + work = append(work, w) + } + if len(work) == 0 { return } - // Same-size guard: skip when this exact size was already applied to - // the current PTY (the TUI re-sends all pane sizes on every workspace - // broadcast). Guard fields are PluginMu-protected; the Resize syscall - // runs outside the lock. - pane.PluginMu.Lock() - pty := pane.PTY - typ := pane.Type - same := pane.appliedCols == int(payload.Cols) && pane.appliedRows == int(payload.Rows) - pane.PluginMu.Unlock() - if pty == nil || same { + sizes := make([]ipc.ResizePanePayload, len(work)) + for i, w := range work { + sizes[i] = ipc.ResizePanePayload{PaneID: w.pane.ID, Cols: w.cols, Rows: w.rows} + } + d.sendPaneSizes(conn, sizes) + + var failed []ipc.ResizePanePayload + for _, w := range work { + if err := w.pty.Resize(w.rows, w.cols); err != nil { + // Record nothing on failure: a transient Resize error must not make + // the guard believe this size was applied, or the TUI's next + // identical re-send would be skipped and the failed resize never + // retried. Leaving appliedCols/Rows unchanged lets the next + // broadcast retry. + log.Printf("resize pane %s to %dx%d: %v", w.pane.ID, w.cols, w.rows, err) + // The followers were already told the new size, and the child is + // still at the old one: tell them the old one again. A pane that + // never had a size applied has nothing to go back to. + if w.prevC > 0 && w.prevR > 0 { + failed = append(failed, ipc.ResizePanePayload{PaneID: w.pane.ID, Cols: uint16(w.prevC), Rows: uint16(w.prevR)}) + } + continue + } + // Record only after the syscall succeeds. Cols/Rows are written INSIDE the + // lock with the applied* guards: they used to be set just below it, which + // made them a genuine data race — this runs on the resizing conn's dispatch + // goroutine while handleAttach (another conn), the PTY output goroutine's + // resizeKick, and snapshot() all read them concurrently. + w.pane.PluginMu.Lock() + w.pane.appliedCols, w.pane.appliedRows = int(w.cols), int(w.rows) + w.pane.Cols, w.pane.Rows = int(w.cols), int(w.rows) + w.pane.PluginMu.Unlock() + + d.repaintAfterResize(w.pane, w.typ) + } + if len(failed) > 0 { + d.sendPaneSizes(conn, failed) + } +} + +// sendPaneSizes queues one pane_sizes frame on every follower's must-deliver +// queue, leaving out the sender. Send never blocks: a follower too wedged to +// take the frame is disconnected by the transport, never waited for. +func (d *Daemon) sendPaneSizes(sender *ipc.Conn, sizes []ipc.ResizePanePayload) { + followers := d.followerConns(sender) + if len(followers) == 0 { return } - if err := pty.Resize(payload.Rows, payload.Cols); err != nil { - // Record nothing on failure: a transient Resize error must not make - // the guard believe this size was applied, or the TUI's next - // identical re-send would be skipped and the failed resize never - // retried. Leaving appliedCols/Rows unchanged lets the next - // broadcast retry. - log.Printf("resize pane %s to %dx%d: %v", payload.PaneID, payload.Cols, payload.Rows, err) + msg, err := ipc.NewMessage(ipc.MsgPaneSizes, ipc.PaneSizesPayload{Panes: sizes}) + if err != nil { + log.Printf("pane_sizes: encode: %v", err) return } - // Record only after the syscall succeeds. Cols/Rows are written INSIDE the - // lock with the applied* guards: they used to be set just below it, which - // made them a genuine data race — this runs on the resizing conn's dispatch - // goroutine while handleAttach (another conn), the PTY output goroutine's - // resizeKick, and snapshot() all read them concurrently. - pane.PluginMu.Lock() - pane.appliedCols = int(payload.Cols) - pane.appliedRows = int(payload.Rows) - pane.Cols = int(payload.Cols) - pane.Rows = int(payload.Rows) - pane.PluginMu.Unlock() - - d.repaintAfterResize(pane, typ) + for _, c := range followers { + c.Send(msg) + } } // repaintAfterResize nudges a pane that has just been resized into repainting, @@ -4329,6 +4427,12 @@ func (d *Daemon) buildWorkspaceState() map[string]any { if info := d.currentUpdateInfo(); info != nil { state["update"] = info } + // Broadcast-only as well: the size master's client id ("" for none) and + // the attached-client count. Each TUI reads them to tell whether it is the + // master or a follower. snapshot() writes size_master to disk by itself, + // for the restart reserve; the count means nothing after a restart. + state["size_master"] = d.masterID() + state["clients"] = d.clientCount() return state } diff --git a/internal/daemon/pane_initial_size.go b/internal/daemon/pane_initial_size.go index 62fedad1..c87b3ba7 100644 --- a/internal/daemon/pane_initial_size.go +++ b/internal/daemon/pane_initial_size.go @@ -8,12 +8,7 @@ type terminalSize struct{ cols, rows int } // MCP can create panes in hidden tabs, so waiting for a TUI resize lets a child // build its first screen for the default 80x24 terminal. func (d *Daemon) newPaneSession(pane *Pane) apty.Session { - cols, rows := 80, 24 - // Older or console-less clients can attach at 1x1. Do not inherit that - // unusable geometry and permanently reflow the child's first screen. - if size := d.clientSize.Load(); size != nil && !degenerateSize(size.cols, size.rows) { - cols, rows = size.cols, size.rows - } + cols, rows := d.initialPaneSize() for _, sibling := range d.session.Panes(pane.CurrentTabID()) { if sibling.ID == pane.ID { continue @@ -33,3 +28,24 @@ func (d *Daemon) newPaneSession(pane *Pane) apty.Session { pane.PluginMu.Unlock() return newSessionFn(cols, rows) } + +// initialPaneSize is the size a new pane starts at when no sibling in its tab +// has one. It is the size master's window, because the master is the client +// that will size the pane next; the last client to attach may be a follower +// whose window says nothing about the PTYs. With no master it falls back to +// the stored attach size, and then to 80x24. +func (d *Daemon) initialPaneSize() (cols, rows int) { + if c := d.masterConn(); c != nil { + // The RAW geometry. A master is paintable by construction, but the + // record can change between the two lookups, so it is checked again. + if rec, ok := d.clientByConn(c); ok && eligible(&rec) { + return rec.cols, rec.rows + } + } + // Older or console-less clients can attach at 1x1. Do not inherit that + // unusable geometry and permanently reflow the child's first screen. + if size := d.clientSize.Load(); size != nil && !degenerateSize(size.cols, size.rows) { + return size.cols, size.rows + } + return 80, 24 +} diff --git a/internal/daemon/redraw_kick_test.go b/internal/daemon/redraw_kick_test.go index d7f1fd92..71568272 100644 --- a/internal/daemon/redraw_kick_test.go +++ b/internal/daemon/redraw_kick_test.go @@ -268,7 +268,7 @@ func TestPaneSize_ConcurrentResizeAndRead(t *testing.T) { go func() { defer close(done) for _, m := range msgs { - d.handleResizePane(m) + d.handleResizePane(nil, m) } }() diff --git a/internal/daemon/resize_authority_test.go b/internal/daemon/resize_authority_test.go new file mode 100644 index 00000000..a757b299 --- /dev/null +++ b/internal/daemon/resize_authority_test.go @@ -0,0 +1,543 @@ +package daemon + +import ( + "errors" + "os" + "path/filepath" + "sync" + "testing" + "time" + + "github.com/artyomsv/quil/internal/config" + "github.com/artyomsv/quil/internal/ipc" +) + +// Size authority: one PTY has one size, so only the size master may set it, +// and every follower learns a new size BEFORE the child repaints at it. Every +// test here drives real conns through ipc.Server, because the gate reads which +// CONN sent the resize and a direct handler call has no conn to read. + +// resizeProbeSession is a PTY that records its Resize calls and can run a hook +// inside one, the way a real child answers SIGWINCH with a repaint. +type resizeProbeSession struct { + fakeSession + mu sync.Mutex + calls [][2]uint16 // (rows, cols) + onResize func() + fail bool +} + +func (s *resizeProbeSession) Resize(rows, cols uint16) error { + s.mu.Lock() + s.calls = append(s.calls, [2]uint16{rows, cols}) + hook, fail := s.onResize, s.fail + s.mu.Unlock() + if hook != nil { + hook() + } + if fail { + return errors.New("simulated resize failure") + } + return nil +} + +// Write accepts the redraw key repaintAfterResize enqueues. +func (s *resizeProbeSession) Write(b []byte) (int, error) { return len(b), nil } + +func (s *resizeProbeSession) resizeCount() int { + s.mu.Lock() + defer s.mu.Unlock() + return len(s.calls) +} + +// resizeAuthorityDaemon is a daemon behind a real IPC server with a fake +// spawn path, fake grace timers, and one empty tab. The tab keeps attach from +// creating the default workspace, whose shell is a real PTY. +func resizeAuthorityDaemon(t *testing.T) (*Daemon, string, string) { + t.Helper() + d, sock := overlayServerDaemonWithConfig(t, config.Default()) + // Fake timers, so no real grace timer outlives the test when the conns + // close. The fake clock is fixed, so attach order ties are broken by id: + // "A" is the older client. + (&clientsHarness{t: t}).install(d, testGrace) + tab := d.session.CreateTab("T") + return d, sock, tab.ID +} + +// attachAB attaches A at 200x50 and then B at 100x30. It waits for A's +// election before B dials, because two conns' attaches race and a connected +// master keeps its slot: B winning the race would stay the master. +func attachAB(t *testing.T, d *Daemon, sock string) (a, b *ipc.Client) { + t.Helper() + a = attachClientAs(t, sock, "A", 200, 50) + waitUntil(t, "A elected", func() bool { return d.masterID() == "A" }) + b = attachClientAs(t, sock, "B", 100, 30) + waitUntil(t, "B attached", func() bool { return d.clientCount() == 2 }) + return a, b +} + +// addProbePanes adds n panes of type typ to the tab, each with a probe PTY. +// Added AFTER the clients attach, so attach's redraw kick resizes none. +func addProbePanes(t *testing.T, d *Daemon, tabID, typ string, n int) ([]*Pane, []*resizeProbeSession) { + t.Helper() + panes := make([]*Pane, n) + probes := make([]*resizeProbeSession, n) + for i := range panes { + p, err := d.session.CreatePane(tabID, "") + if err != nil { + t.Fatalf("create pane: %v", err) + } + s := &resizeProbeSession{} + p.PluginMu.Lock() + p.Type = typ + p.PTY = s + p.PluginMu.Unlock() + t.Cleanup(p.StopInput) + panes[i], probes[i] = p, s + } + return panes, probes +} + +func appliedSize(p *Pane) (int, int) { + p.PluginMu.Lock() + defer p.PluginMu.Unlock() + return p.appliedCols, p.appliedRows +} + +// loadRedrawKeyPlugin registers a claude-like plugin: one that declares a +// redraw_key, so a resize makes repaintAfterResize send it input. +func loadRedrawKeyPlugin(t *testing.T, d *Daemon, name string) { + t.Helper() + dir := t.TempDir() + toml := "[plugin]\n" + + "name = \"" + name + "\"\n" + + "display_name = \"" + name + "\"\n" + + "category = \"test\"\n" + + "schema_version = 1\n" + + "[command]\n" + + "cmd = \"echo\"\n" + + "[persistence]\n" + + "ghost_buffer = false\n" + + "redraw_key = \"\\f\"\n" + if err := os.WriteFile(filepath.Join(dir, name+".toml"), []byte(toml), 0o600); err != nil { + t.Fatalf("write plugin toml: %v", err) + } + if err := d.registry.LoadFromDir(dir); err != nil { + t.Fatalf("LoadFromDir: %v", err) + } + if p := d.registry.Get(name); p == nil || p.Persistence.RedrawKey == "" { + t.Fatalf("plugin %q not loaded with a redraw_key", name) + } +} + +// readUntil reads c's frames until one matches, returning every frame read, +// the match last. It fails the test when none matches in time. +func readUntil(t *testing.T, c *ipc.Client, what string, match func(*ipc.Message) bool) []*ipc.Message { + t.Helper() + var got []*ipc.Message + if err := c.SetReadDeadline(time.Now().Add(3 * time.Second)); err != nil { + t.Fatalf("set read deadline: %v", err) + } + for { + m, err := c.Receive() + if err != nil { + t.Fatalf("waiting for %s: %v", what, err) + } + got = append(got, m) + if match(m) { + return got + } + } +} + +// readFor reads every frame c receives within d. It is the last read on c: +// a deadline that expires mid-frame leaves the stream unusable. +func readFor(c *ipc.Client, d time.Duration) []*ipc.Message { + var got []*ipc.Message + if err := c.SetReadDeadline(time.Now().Add(d)); err != nil { + return nil + } + for { + m, err := c.Receive() + if err != nil { + return got + } + got = append(got, m) + } +} + +func isType(typ string) func(*ipc.Message) bool { + return func(m *ipc.Message) bool { return m.Type == typ } +} + +func paneSizesOf(t *testing.T, m *ipc.Message) []ipc.ResizePanePayload { + t.Helper() + var p ipc.PaneSizesPayload + if err := m.DecodePayload(&p); err != nil { + t.Fatalf("decode pane_sizes: %v", err) + } + return p.Panes +} + +func countType(msgs []*ipc.Message, typ string) int { + n := 0 + for _, m := range msgs { + if m.Type == typ { + n++ + } + } + return n +} + +// barrier returns once every frame c sent before it has been dispatched. A +// conn's frames run in order, so a pane_input stamping the client's input time +// is a fence for the resizes queued ahead of it. +// The registry clock is moved to a fresh instant first, so the stamp this +// barrier waits for cannot be one an earlier message left. +func barrier(t *testing.T, d *Daemon, c *ipc.Client, id string) { + t.Helper() + d.clients.mu.Lock() + stamp := d.clients.clock().Add(time.Minute) + d.clients.now = func() time.Time { return stamp } + d.clients.mu.Unlock() + sendClientMsg(t, c, ipc.MsgPaneInput, ipc.PaneInputPayload{PaneID: "pane-none", Data: []byte("x")}) + waitUntil(t, id+"'s barrier", func() bool { + rec, _ := clientRecordByID(d, id) + return rec.lastInputAt.Equal(stamp) + }) +} + +// Spec 9.1.1: two clients, and only the master's resize reaches the PTY. +func TestResizeAuthority_OnlyMasterResizes(t *testing.T) { + d, sock, tabID := resizeAuthorityDaemon(t) + a, b := attachAB(t, d, sock) + if d.masterID() != "A" { + t.Fatalf("masterID = %q, want A", d.masterID()) + } + panes, probes := addProbePanes(t, d, tabID, "terminal", 1) + + sendClientMsg(t, b, ipc.MsgResizePane, ipc.ResizePanePayload{PaneID: panes[0].ID, Cols: 90, Rows: 20}) + barrier(t, d, b, "B") + if c, r := appliedSize(panes[0]); c != 0 || r != 0 || probes[0].resizeCount() != 0 { + t.Fatalf("a follower's resize applied: applied=%dx%d, Resize calls=%d", c, r, probes[0].resizeCount()) + } + + sendClientMsg(t, a, ipc.MsgResizePane, ipc.ResizePanePayload{PaneID: panes[0].ID, Cols: 150, Rows: 40}) + waitUntil(t, "the master's resize applied", func() bool { + c, r := appliedSize(panes[0]) + return c == 150 && r == 40 + }) +} + +// With no master and no reserved slot, any attached client's resize applies: +// the legacy single-client behaviour, which keeps an older client working. +func TestResizeAuthority_OpenWhenNoMaster(t *testing.T) { + d, sock, tabID := resizeAuthorityDaemon(t) + c := attachClientAs(t, sock, "headless", 0, 0) + waitUntil(t, "attached", func() bool { return d.clientCount() == 1 }) + if d.masterID() != "" || !d.sizeAuthorityOpen() { + t.Fatalf("masterID = %q, open = %v; want no master and open authority", d.masterID(), d.sizeAuthorityOpen()) + } + panes, _ := addProbePanes(t, d, tabID, "terminal", 1) + + sendClientMsg(t, c, ipc.MsgResizePane, ipc.ResizePanePayload{PaneID: panes[0].ID, Cols: 100, Rows: 40}) + waitUntil(t, "the resize applied", func() bool { + cols, rows := appliedSize(panes[0]) + return cols == 100 && rows == 40 + }) +} + +// Review Focus 2: a resize burst across 48 panes must reach a follower as ONE +// frame. One frame per pane is 48 must-deliver frames on a 64-slot queue, on +// top of whatever else that follower is owed. +func TestResizePanes_BatchSendsOnePaneSizesFramePerFollower(t *testing.T) { + d, sock, tabID := resizeAuthorityDaemon(t) + a, b := attachAB(t, d, sock) + // B's own attach frames first, so the count below sees only the burst. + readUntil(t, b, "B's attach state", isType(ipc.MsgWorkspaceState)) + + const n = 48 + panes, probes := addProbePanes(t, d, tabID, "terminal", n) + batch := make([]ipc.ResizePanePayload, n) + for i, p := range panes { + batch[i] = ipc.ResizePanePayload{PaneID: p.ID, Cols: 120, Rows: 40} + } + sendClientMsg(t, a, ipc.MsgResizePanes, ipc.ResizePanesPayload{Panes: batch}) + waitUntil(t, "every pane resized", func() bool { + for _, s := range probes { + if s.resizeCount() != 1 { + return false + } + } + return true + }) + + got := readFor(b, 300*time.Millisecond) + var frames []*ipc.Message + for _, m := range got { + if m.Type == ipc.MsgPaneSizes { + frames = append(frames, m) + } + } + if len(frames) != 1 { + t.Fatalf("B received %d pane_sizes frames, want exactly 1", len(frames)) + } + if entries := paneSizesOf(t, frames[0]); len(entries) != n { + t.Errorf("the pane_sizes frame has %d entries, want %d", len(entries), n) + } + if d.clientCount() != 2 { + t.Errorf("clientCount = %d after the burst, want 2: B was disconnected", d.clientCount()) + } +} + +// A batch naming one pane twice resizes it once, to the last size, and tells +// the follower that size alone. +func TestResizePanes_DuplicatePaneResizedOnceToLastSize(t *testing.T) { + d, sock, tabID := resizeAuthorityDaemon(t) + a, b := attachAB(t, d, sock) + readUntil(t, b, "B's attach state", isType(ipc.MsgWorkspaceState)) + + panes, probes := addProbePanes(t, d, tabID, "terminal", 1) + id := panes[0].ID + sendClientMsg(t, a, ipc.MsgResizePanes, ipc.ResizePanesPayload{Panes: []ipc.ResizePanePayload{ + {PaneID: id, Cols: 100, Rows: 30}, + {PaneID: id, Cols: 110, Rows: 35}, + }}) + got := readUntil(t, b, "the size frame", isType(ipc.MsgPaneSizes)) + if e := paneSizesOf(t, got[len(got)-1]); len(e) != 1 || e[0].Cols != 110 || e[0].Rows != 35 { + t.Errorf("pane_sizes = %+v, want one entry at 110x35", e) + } + waitUntil(t, "the resize applied", func() bool { + c, r := appliedSize(panes[0]) + return c == 110 && r == 35 + }) + if n := probes[0].resizeCount(); n != 1 { + t.Errorf("Resize called %d times, want 1", n) + } +} + +// Spec 5d: the follower must hold the new size before the child's repaint +// arrives, or the repaint lands in the old-sized VT and is reflowed. The probe +// repaints INSIDE Resize, then pauses, so a size frame sent after Resize would +// reach B behind the repaint. +func TestPaneSizes_ArriveBeforeRepaintOutput(t *testing.T) { + d, sock, tabID := resizeAuthorityDaemon(t) + loadRedrawKeyPlugin(t, d, "claude-like") + a, b := attachAB(t, d, sock) + readUntil(t, b, "B's attach state", isType(ipc.MsgWorkspaceState)) + + panes, probes := addProbePanes(t, d, tabID, "claude-like", 1) + pane := panes[0] + probes[0].mu.Lock() + probes[0].onResize = func() { + d.flushPaneOutput(pane.ID, []byte("REPAINT")) + time.Sleep(50 * time.Millisecond) + } + probes[0].mu.Unlock() + + sendClientMsg(t, a, ipc.MsgResizePane, ipc.ResizePanePayload{PaneID: pane.ID, Cols: 150, Rows: 40}) + got := readUntil(t, b, "the REPAINT output", func(m *ipc.Message) bool { + if m.Type != ipc.MsgPaneOutput { + return false + } + var p ipc.PaneOutputPayload + return m.DecodePayload(&p) == nil && p.PaneID == pane.ID && string(p.Data) == "REPAINT" + }) + sawSize := false + for _, m := range got { + if m.Type != ipc.MsgPaneSizes { + continue + } + for _, e := range paneSizesOf(t, m) { + if e.PaneID == pane.ID && e.Cols == 150 && e.Rows == 40 { + sawSize = true + } + } + } + if !sawSize { + t.Fatal("B received the repaint before the pane_sizes frame carrying 150x40") + } +} + +// A refused resize (a follower's, or a degenerate one) and a same-size one +// change nothing, so they must send no size frame. The last step proves the +// read would have seen one. +func TestPaneSizes_NoneForRefusedOrSameSize(t *testing.T) { + d, sock, tabID := resizeAuthorityDaemon(t) + a, b := attachAB(t, d, sock) + readUntil(t, b, "B's attach state", isType(ipc.MsgWorkspaceState)) + + panes, _ := addProbePanes(t, d, tabID, "terminal", 1) + pane := panes[0] + pane.PluginMu.Lock() + pane.appliedCols, pane.appliedRows = 120, 40 + pane.PluginMu.Unlock() + + sendClientMsg(t, b, ipc.MsgResizePane, ipc.ResizePanePayload{PaneID: pane.ID, Cols: 90, Rows: 20}) // follower + sendClientMsg(t, a, ipc.MsgResizePane, ipc.ResizePanePayload{PaneID: pane.ID, Cols: 120, Rows: 40}) // same size + sendClientMsg(t, a, ipc.MsgResizePane, ipc.ResizePanePayload{PaneID: pane.ID, Cols: 1, Rows: 1}) // degenerate + barrier(t, d, b, "B") + barrier(t, d, a, "A") + sendClientMsg(t, a, ipc.MsgResizePane, ipc.ResizePanePayload{PaneID: pane.ID, Cols: 130, Rows: 40}) // applies + + got := readUntil(t, b, "the applied resize's size frame", isType(ipc.MsgPaneSizes)) + entries := paneSizesOf(t, got[len(got)-1]) + if len(entries) != 1 || entries[0].Cols != 130 || entries[0].Rows != 40 { + t.Errorf("first pane_sizes on B = %+v, want only the applied 130x40", entries) + } +} + +// A resize the PTY refused leaves the child at its previous size, and the +// follower was already told the new one. A second frame puts it back. +func TestPaneSizes_FailedResizeSendsPreviousSize(t *testing.T) { + d, sock, tabID := resizeAuthorityDaemon(t) + a, b := attachAB(t, d, sock) + readUntil(t, b, "B's attach state", isType(ipc.MsgWorkspaceState)) + + panes, probes := addProbePanes(t, d, tabID, "terminal", 1) + pane := panes[0] + pane.PluginMu.Lock() + pane.appliedCols, pane.appliedRows = 120, 40 + pane.PluginMu.Unlock() + probes[0].mu.Lock() + probes[0].fail = true + probes[0].mu.Unlock() + + sendClientMsg(t, a, ipc.MsgResizePane, ipc.ResizePanePayload{PaneID: pane.ID, Cols: 150, Rows: 50}) + first := readUntil(t, b, "the new size", isType(ipc.MsgPaneSizes)) + if e := paneSizesOf(t, first[len(first)-1]); len(e) != 1 || e[0].Cols != 150 || e[0].Rows != 50 { + t.Fatalf("first pane_sizes = %+v, want 150x50", e) + } + second := readUntil(t, b, "the rollback", isType(ipc.MsgPaneSizes)) + if e := paneSizesOf(t, second[len(second)-1]); len(e) != 1 || e[0].Cols != 120 || e[0].Rows != 40 { + t.Fatalf("second pane_sizes = %+v, want the previous 120x40", e) + } + if c, r := appliedSize(pane); c != 120 || r != 40 { + t.Errorf("applied = %dx%d after a failed resize, want 120x40 unchanged", c, r) + } +} + +// The broadcast carries who the master is and how many clients are attached; +// the map written to workspace.json carries neither. +func TestWorkspaceState_CarriesSizeMasterAndClients(t *testing.T) { + d, sock, _ := resizeAuthorityDaemon(t) + attachAB(t, d, sock) + + state := d.buildWorkspaceState() + if got, _ := state["size_master"].(string); got != "A" { + t.Errorf("size_master = %v, want A", state["size_master"]) + } + if got, _ := state["clients"].(int); got != 2 { + t.Errorf("clients = %v, want 2", state["clients"]) + } + + activeTab, tabs, panesByTab, projects, activeProject := d.session.SnapshotState() + for _, overlays := range []bool{false, true} { + m := d.workspaceStateFromSnapshot(activeTab, tabs, panesByTab, projects, activeProject, overlays) + for _, k := range []string{"size_master", "clients"} { + if _, ok := m[k]; ok { + t.Errorf("workspaceStateFromSnapshot(includeOverlays=%v) has %q", overlays, k) + } + } + } +} + +// A master change reaches every client as a broadcast carrying the new +// size_master. A same-id reattach inside the grace changes nothing, so it +// costs the follower no frame. +func TestMasterChange_IsBroadcastAndReattachIsNot(t *testing.T) { + d, sock, _ := resizeAuthorityDaemon(t) + a, b := attachAB(t, d, sock) + readUntil(t, b, "B's attach state", isType(ipc.MsgWorkspaceState)) + + sendClientMsg(t, b, ipc.MsgTakeControl, nil) + readUntil(t, b, "a state naming B the master", func(m *ipc.Message) bool { + if m.Type != ipc.MsgWorkspaceState { + return false + } + var s map[string]any + return m.DecodePayload(&s) == nil && s["size_master"] == "B" + }) + + // Hand the slot back, then lose A's link: its slot is reserved. + sendClientMsg(t, a, ipc.MsgTakeControl, nil) + readUntil(t, b, "a state naming A the master", func(m *ipc.Message) bool { + if m.Type != ipc.MsgWorkspaceState { + return false + } + var s map[string]any + return m.DecodePayload(&s) == nil && s["size_master"] == "A" + }) + a.Close() + waitUntil(t, "A's link lost", func() bool { return d.clientCount() == 1 }) + if d.masterID() != "A" || d.sizeAuthorityOpen() { + t.Fatalf("masterID = %q, open = %v; want A's slot reserved", d.masterID(), d.sizeAuthorityOpen()) + } + + a2 := attachClientAs(t, sock, "A", 200, 50) + readUntil(t, a2, "A's own attach state", isType(ipc.MsgWorkspaceState)) + if n := countType(readFor(b, 300*time.Millisecond), ipc.MsgWorkspaceState); n != 0 { + t.Errorf("B received %d workspace_state frames for a same-id reattach, want 0", n) + } + if d.masterID() != "A" { + t.Errorf("masterID = %q after the reattach, want A", d.masterID()) + } +} + +// clientSize used to be whichever client attached LAST. A new pane now starts +// at the master's window, whatever order the clients came in. +func TestInitialPaneSize_UsesMasterGeometry(t *testing.T) { + d, sock, tabID := resizeAuthorityDaemon(t) + _, b := attachAB(t, d, sock) + if d.masterID() != "A" { + t.Fatalf("masterID = %q, want A", d.masterID()) + } + + sendClientMsg(t, b, ipc.MsgCreatePane, ipc.CreatePanePayload{TabID: tabID}) + waitUntil(t, "the pane created", func() bool { return len(d.session.Panes(tabID)) == 1 }) + pane := d.session.Panes(tabID)[0] + if c, r := paneSize(pane); c != 200 || r != 50 { + t.Errorf("new pane size = %dx%d, want the master's 200x50", c, r) + } +} + +// A self-reported window size is bounded before it feeds a new pane's size or +// the election: a value above the ceiling is taken as the ceiling. +func TestClientGeometry_ClampedToMax(t *testing.T) { + d, sock, _ := resizeAuthorityDaemon(t) + c := attachClientAs(t, sock, "A", 5000, 4000) + waitUntil(t, "attached", func() bool { return d.clientCount() == 1 }) + if rec, _ := clientRecordByID(d, "A"); rec.cols != maxClientDim || rec.rows != maxClientDim { + t.Errorf("attach geometry = %dx%d, want %dx%d", rec.cols, rec.rows, maxClientDim, maxClientDim) + } + if sz := d.clientSize.Load(); sz == nil || sz.cols != maxClientDim || sz.rows != maxClientDim { + t.Errorf("clientSize = %+v, want %dx%d", sz, maxClientDim, maxClientDim) + } + + sendClientMsg(t, c, ipc.MsgClientGeometry, ipc.ClientGeometryPayload{Cols: 3000, Rows: 70}) + waitUntil(t, "geometry recorded", func() bool { + rec, _ := clientRecordByID(d, "A") + return rec.rows == 70 + }) + if rec, _ := clientRecordByID(d, "A"); rec.cols != maxClientDim { + t.Errorf("client_geometry cols = %d, want %d", rec.cols, maxClientDim) + } +} + +// Stop must disarm the grace timer, so it never fires into a stopped daemon. +func TestStop_DisarmsTheGraceTimer(t *testing.T) { + t.Setenv("QUIL_HOME", t.TempDir()) + d := New(config.Default()) + h := &clientsHarness{t: t} + h.install(d, testGrace) + a, _ := h.attach("A", 200, 50) + h.attach("B", 100, 30) + d.forgetAttachedClient(a) + if h.armed() == nil { + t.Fatal("setup: losing the master armed no grace timer") + } + d.Stop() + if h.armed() != nil { + t.Error("the grace timer is still armed after Stop") + } +} diff --git a/internal/daemon/resize_guard_test.go b/internal/daemon/resize_guard_test.go index f4164113..e2bfa50a 100644 --- a/internal/daemon/resize_guard_test.go +++ b/internal/daemon/resize_guard_test.go @@ -35,12 +35,12 @@ func TestHandleResizePane_DuplicateSize_SkipsPTYResize(t *testing.T) { pane := &Pane{ID: "p1", PTY: fake} d.session.panes["p1"] = pane - d.handleResizePane(resizeMsg(t, "p1", 100, 40)) - d.handleResizePane(resizeMsg(t, "p1", 100, 40)) + d.handleResizePane(nil, resizeMsg(t, "p1", 100, 40)) + d.handleResizePane(nil, resizeMsg(t, "p1", 100, 40)) if len(fake.resizes) != 1 { t.Fatalf("PTY.Resize called %d times, want 1 (duplicate must be skipped)", len(fake.resizes)) } - d.handleResizePane(resizeMsg(t, "p1", 120, 40)) + d.handleResizePane(nil, resizeMsg(t, "p1", 120, 40)) if len(fake.resizes) != 2 { t.Fatalf("PTY.Resize called %d times, want 2 (changed size must apply)", len(fake.resizes)) } @@ -55,7 +55,7 @@ func TestHandleResizePane_FreshPTY_AcceptsSameSize(t *testing.T) { pane := &Pane{ID: "p1", PTY: fake} d.session.panes["p1"] = pane - d.handleResizePane(resizeMsg(t, "p1", 100, 40)) + d.handleResizePane(nil, resizeMsg(t, "p1", 100, 40)) // Simulate restart: new PTY installed the way spawnPane does it. fake2 := &fakeSession{} @@ -64,7 +64,7 @@ func TestHandleResizePane_FreshPTY_AcceptsSameSize(t *testing.T) { pane.appliedCols, pane.appliedRows = 0, 0 pane.PluginMu.Unlock() - d.handleResizePane(resizeMsg(t, "p1", 100, 40)) + d.handleResizePane(nil, resizeMsg(t, "p1", 100, 40)) if len(fake2.resizes) != 1 { t.Fatalf("fresh PTY got %d resizes, want 1 (guard must reset on PTY install)", len(fake2.resizes)) } @@ -74,7 +74,7 @@ func TestHandleResizePane_NilPTY_NoApply(t *testing.T) { d := &Daemon{session: NewSessionManager(4096)} pane := &Pane{ID: "p1"} d.session.panes["p1"] = pane - d.handleResizePane(resizeMsg(t, "p1", 100, 40)) // must not panic + d.handleResizePane(nil, resizeMsg(t, "p1", 100, 40)) // must not panic if pane.Cols != 0 { t.Errorf("pane.Cols = %d, want 0 (no PTY, nothing applied)", pane.Cols) } @@ -84,7 +84,7 @@ func TestHandleResizePane_NilPTY_NoApply(t *testing.T) { pane.PluginMu.Lock() pane.PTY = fake pane.PluginMu.Unlock() - d.handleResizePane(resizeMsg(t, "p1", 100, 40)) + d.handleResizePane(nil, resizeMsg(t, "p1", 100, 40)) if len(fake.resizes) != 1 { t.Fatalf("PTY got %d resizes, want 1 (nil-PTY request must not poison the guard)", len(fake.resizes)) } @@ -121,7 +121,7 @@ func TestSpawnPane_ResetsResizeGuard(t *testing.T) { pane.appliedCols, pane.appliedRows) } // A resize at the old size now goes through to the new PTY. - d.handleResizePane(resizeMsg(t, "p-reset", 100, 40)) + d.handleResizePane(nil, resizeMsg(t, "p-reset", 100, 40)) if len(fake.resizes) < 1 { t.Errorf("fresh PTY got %d resizes at the old size, want at least 1", len(fake.resizes)) } @@ -136,12 +136,12 @@ func TestHandleResizePane_FailedResizeDoesNotStickGuard(t *testing.T) { pane := &Pane{ID: "p-fail", PTY: fake} d.session.panes["p-fail"] = pane - d.handleResizePane(resizeMsg(t, "p-fail", 90, 30)) // fails + d.handleResizePane(nil, resizeMsg(t, "p-fail", 90, 30)) // fails if pane.appliedCols != 0 || pane.appliedRows != 0 { t.Fatalf("failed resize stuck the guard at %dx%d, want 0x0", pane.appliedCols, pane.appliedRows) } fake.fail = false - d.handleResizePane(resizeMsg(t, "p-fail", 90, 30)) // retry succeeds + d.handleResizePane(nil, resizeMsg(t, "p-fail", 90, 30)) // retry succeeds if pane.appliedCols != 90 || pane.appliedRows != 30 { t.Errorf("retry after failure did not apply: guard %dx%d, want 90x30", pane.appliedCols, pane.appliedRows) } @@ -203,7 +203,7 @@ func TestHandleResizePane_DegenerateSize_IsRefused(t *testing.T) { appliedCols: 200, appliedRows: 50, } - d.handleResizePane(resizeMsg(t, "p1", tt.cols, tt.rows)) + d.handleResizePane(nil, resizeMsg(t, "p1", tt.cols, tt.rows)) if len(fake.resizes) != tt.want { t.Fatalf("PTY.Resize called %d times for %dx%d, want %d", @@ -237,8 +237,8 @@ func TestHandleResizePane_DegenerateResize_LogsOncePerCooldownWindow(t *testing. appliedCols: 200, appliedRows: 50, } - d.handleResizePane(resizeMsg(t, "pane-1", 1, 1)) - d.handleResizePane(resizeMsg(t, "pane-1", 1, 1)) + d.handleResizePane(nil, resizeMsg(t, "pane-1", 1, 1)) + d.handleResizePane(nil, resizeMsg(t, "pane-1", 1, 1)) got := strings.Count(buf.String(), "refusing degenerate resize") if got != 1 { diff --git a/internal/daemon/resize_repaint_test.go b/internal/daemon/resize_repaint_test.go index 087642bc..97767afc 100644 --- a/internal/daemon/resize_repaint_test.go +++ b/internal/daemon/resize_repaint_test.go @@ -46,7 +46,7 @@ func TestHandleResizePane_RedrawsAPluginThatIgnoresSIGWINCH(t *testing.T) { // The spawn-vs-attach disagreement: the pane came up at the persisted 91 // columns and the first client resize narrows it. pane.appliedCols, pane.appliedRows = 91, 54 - d.handleResizePane(resizeMsg(t, "p1", 80, 52)) + d.handleResizePane(nil, resizeMsg(t, "p1", 80, 52)) if !waitForInput(t, pty, "\f") { t.Errorf("child received %q after a resize, want %q (Ctrl+L)", pty.got(), "\f") @@ -64,7 +64,7 @@ func TestHandleResizePane_NoInjectedInputWithoutAnOptIn(t *testing.T) { d.session.panes["p1"] = pane t.Cleanup(pane.StopInput) - d.handleResizePane(resizeMsg(t, "p1", 80, 52)) + d.handleResizePane(nil, resizeMsg(t, "p1", 80, 52)) waitForNoInput(t, pty) } @@ -81,7 +81,7 @@ func TestHandleResizePane_DuplicateSizeSendsNoRedraw(t *testing.T) { // Already applied — this is the broadcast re-send, not a real resize. pane.appliedCols, pane.appliedRows = 80, 52 - d.handleResizePane(resizeMsg(t, "p1", 80, 52)) + d.handleResizePane(nil, resizeMsg(t, "p1", 80, 52)) waitForNoInput(t, pty) } @@ -96,7 +96,7 @@ func TestHandleResizePane_FailedResizeSendsNoRedraw(t *testing.T) { d.session.panes["p1"] = pane t.Cleanup(pane.StopInput) - d.handleResizePane(resizeMsg(t, "p1", 80, 52)) + d.handleResizePane(nil, resizeMsg(t, "p1", 80, 52)) waitForNoInput(t, &pty.recordingSession) } From 7b7cd5b572c6c10d649cacfe24d00698417baf29 Mon Sep 17 00:00:00 2001 From: Artjoms Stukans Date: Fri, 25 Sep 2026 17:19:21 +0200 Subject: [PATCH 04/40] fix(daemon): deliver attach replay and live output exactly once A client attaching while panes write got its OutputBuf replay and the live stream racing each other: live frames landed mid-replay, and bytes written after the replay snapshot arrived twice or out of place. handleAttach now holds the conn off live pane_output (Conn's hold flag) from before the state frame until the replay is queued. Every flush during the hold is copied into the conn's hold with its stream position (Pane.outPos, advanced in the same PluginMu span as the OutputBuf write). The release sends the held bytes through SendBlocking, cutting what an OutputBuf replay already covered, then clears the flag. A ghostsnap or skipped replay records no end, so all held bytes follow it. A pane whose held bytes pass 4 MiB loses them and gets a redraw kick instead; the conn is never closed for it. onClientDisconnect and any early return from handleAttach drop the hold. Two ordering gaps closed beyond the plain hold: - holdGate (RWMutex) makes a flush's hold append and broadcast one step against a hold starting or ending, so a straddling flush is neither sent twice nor lost. - Conn.QueuedOutput and Conn.Done let the hold wait (up to 2 s) for live frames queued before it; the critical-first sendLoop would otherwise write them behind the state frame, mid-replay. --- internal/daemon/daemon.go | 58 ++ internal/daemon/outputhold.go | 253 +++++++++ .../daemon/outputhold_integration_test.go | 228 ++++++++ internal/daemon/outputhold_test.go | 502 ++++++++++++++++++ internal/daemon/session.go | 7 + internal/ipc/queued_output_test.go | 60 +++ internal/ipc/server.go | 12 + 7 files changed, 1120 insertions(+) create mode 100644 internal/daemon/outputhold.go create mode 100644 internal/daemon/outputhold_integration_test.go create mode 100644 internal/daemon/outputhold_test.go create mode 100644 internal/ipc/queued_output_test.go diff --git a/internal/daemon/daemon.go b/internal/daemon/daemon.go index 3a8517b7..f8a88da2 100644 --- a/internal/daemon/daemon.go +++ b/internal/daemon/daemon.go @@ -274,6 +274,18 @@ type Daemon struct { // mode this package keeps being bitten by. Nothing that takes PluginMu may // be called while it is held, and nothing broadcasts while it is held. clients clientRegistry + + // holds keeps each attaching conn's live pane output while its replay is + // sent, keyed by conn (outputhold.go). holdMu is a leaf guarding it; + // holdGate orders a flush's hold append and broadcast against a conn's + // hold flag changing. Order: holdGate, then holdMu. + holdGate sync.RWMutex + holdMu sync.Mutex + holds map[*ipc.Conn]*outputHold + // afterHoldOutput is a test seam: when set, a flush calls it between its + // hold append and its broadcast. Set before any flush runs; nil in + // production. + afterHoldOutput func(paneID string) } func New(cfg config.Config) *Daemon { @@ -620,6 +632,7 @@ func (d *Daemon) Stop() { // records there would clear a lone master (nobody left to protect) before the // snapshot writes size_master — so the restart would have no reserve. func (d *Daemon) onClientDisconnect(conn *ipc.Conn) { + d.dropOutputHold(conn) d.requestSnapshot() d.events.RemoveWatchersByConn(conn) if !d.shuttingDown() && d.forgetAttachedClient(conn) { @@ -1789,6 +1802,23 @@ func (d *Daemon) handleAttach(conn *ipc.Conn, msg *ipc.Message) { defer d.broadcastState() } + // Hold this conn off live pane output until its replay is sent, BEFORE + // the state frame is built: from here every flush is either already in + // the OutputBuf bytes the replay sends, or held and sent after it — the + // replay and the live stream reach this client once each, in order + // (outputhold.go). Every return below must end the hold, or the conn + // stays off live output for good. + d.beginOutputHold(conn) + holdReleased := false + defer func() { + if !holdReleased { + d.dropOutputHold(conn) + } + }() + // end records, per pane, the stream position its OutputBuf replay ended + // at — the point up to which this client already has the held bytes. + end := make(map[string]uint64) + // clientSize sizes the first pane of an empty workspace below, and new // panes whenever no master is elected (initialPaneSize). cols, rows := clampClientDim(attach.Cols), clampClientDim(attach.Rows) @@ -1942,6 +1972,13 @@ func (d *Daemon) handleAttach(conn *ipc.Conn, msg *ipc.Message) { } } } + // Only a replay of OutputBuf's own bytes has a stream position, + // read in this span with the Bytes() snapshot. A ghostsnap replay + // is a PREVIOUS session's bytes, so every held byte of the new + // child comes after it; a skipped replay covers nothing. + if len(ghost) > 0 && (source == "outputbuf" || source == "child-stream") { + end[pane.ID] = pane.outPos + } // Captured in the same span as Type/GhostSnap: the redraw kick below // needs a live PTY, and reading it separately would race a restart. // Same discipline as handleResizePane — pointer under the lock, the @@ -1977,6 +2014,11 @@ func (d *Daemon) handleAttach(conn *ipc.Conn, msg *ipc.Message) { } } + // The replay is queued: send what was held behind it, then let live + // output through. + d.releaseOutputHold(conn, end) + holdReleased = true + // Replay pending notification events, OLDEST FIRST — this is a replay of // state transitions, not a listing. The TUI rebuilds each pane's work // state by applying these in order, so the newest-first storage order has @@ -4249,6 +4291,11 @@ func (d *Daemon) flushPaneOutputGeneration(paneID string, data []byte, generatio } pane.OutputBuf.Write(data) } + // The stream position, in the SAME span as the OutputBuf write: an attach + // reads outPos beside the buffer bytes it replays, so the two always + // describe the same instant. Counted even with no OutputBuf. + start := pane.outPos + pane.outPos += uint64(len(data)) // Update idle tracking + mouse-mode state (guarded by PluginMu). now := time.Now() @@ -4306,7 +4353,18 @@ func (d *Daemon) flushPaneOutputGeneration(paneID string, data []byte, generatio Data: data, Generation: generation, }) + // Held conns get a copy through their hold; the broadcast skips them. The + // gate makes the pair one step against a hold starting or ending, or a + // conn could get these bytes twice or not at all (outputhold.go). + // Outside PluginMu, and Broadcast only enqueues, so the gate is held + // across no I/O. + d.holdGate.RLock() + d.holdOutput(paneID, start, data, generation) + if d.afterHoldOutput != nil { + d.afterHoldOutput(paneID) + } d.broadcast(msg) + d.holdGate.RUnlock() } // detectBellEvent checks for standalone bell characters (not OSC terminators). diff --git a/internal/daemon/outputhold.go b/internal/daemon/outputhold.go new file mode 100644 index 00000000..695c32a3 --- /dev/null +++ b/internal/daemon/outputhold.go @@ -0,0 +1,253 @@ +package daemon + +import ( + "log" + "time" + + "github.com/artyomsv/quil/internal/ipc" + "github.com/artyomsv/quil/internal/logger" +) + +// Clean attach: a client that attaches while panes are writing must receive +// each pane's history replay and its live output EXACTLY ONCE, in order. +// +// Without a hold, the two streams race. The replay is a snapshot of OutputBuf +// sent frame by frame through the conn's must-deliver queue, while live bytes +// are broadcast through its droppable queue as they arrive. A live frame could +// land in the middle of the replay, and bytes written after the snapshot but +// before the replay finished arrived twice or out of place. +// +// So handleAttach holds the conn off live pane_output (ipc.Conn's +// holdPaneOutput flag) for the whole replay. Every flush during the hold is +// COPIED into the conn's hold here, with its stream position. When the replay +// is done, the held bytes are sent in order, minus whatever the replay already +// covered, and the flag is cleared. The daemon owns the stream, so it owns the +// dedupe: no position goes on the wire and no client repeats this logic. +// +// Two locks, both only ever taken in the order holdGate → holdMu: +// +// - holdMu is a LEAF guarding the holds map and each hold's contents. It is +// never held with PluginMu or the client registry mutex, and never across +// SendBlocking. +// - holdGate makes "append to the holds, then broadcast" ONE step against +// setting or clearing a conn's flag. A flush holds it for READ from its +// hold append through its broadcast (Broadcast only enqueues, so this +// blocks on nothing); beginOutputHold and the release's final clear hold +// it for WRITE. Without it, a flush could append its chunk, the release +// could send that chunk and clear the flag, and then the flush's broadcast +// would reach the now-unheld conn — the same bytes twice. At the other +// end, a flush could miss the new hold and then broadcast to a conn whose +// flag was set in between — the bytes never reach it at all. + +// outputHoldLimit bounds one conn's held bytes. A pane whose held bytes would +// pass it loses them all and gets a repaint kick at release instead. +const outputHoldLimit = 4 << 20 + +// heldChunk is one flush held for one conn. start is the pane's stream +// position (Pane.outPos) of data[0]. data is shared read-only between every +// hold that took the same flush. +type heldChunk struct { + paneID string + start uint64 + data []byte + gen uint64 +} + +type outputHold struct { + chunks []heldChunk + bytes int + lost map[string]bool // panes whose held bytes overflowed +} + +// dropPane removes paneID's held chunks and marks the pane lost, so its later +// chunks are skipped too and the gap is whole rather than partial. +func (h *outputHold) dropPane(paneID string) { + kept := h.chunks[:0] + for _, ch := range h.chunks { + if ch.paneID == paneID { + h.bytes -= len(ch.data) + continue + } + kept = append(kept, ch) + } + // Clear the tail so dropped data is not pinned by the backing array. + for i := len(kept); i < len(h.chunks); i++ { + h.chunks[i] = heldChunk{} + } + h.chunks = kept + if h.lost == nil { + h.lost = make(map[string]bool) + } + h.lost[paneID] = true +} + +// holdDrainTimeout bounds beginOutputHold's wait for live frames queued before +// the hold. Past it the attach proceeds anyway, an ACCEPTED degradation: a +// frame still queued can land behind the replay as a duplicate, which is the +// behaviour before the hold existed, and only on a client too slow to take +// 64 frames in this long. Waiting longer would stall its attach instead. +const holdDrainTimeout = 2 * time.Second + +// beginOutputHold starts holding c's live pane output. From the moment it +// returns, every flush either was delivered to c before it, or is held. +// +// "Delivered" includes the wait at the end. A conn receives live output from +// the moment it connects, and frames still in its droppable queue are written +// AFTER any must-deliver frame, since sendLoop drains that queue first. Their +// bytes are already in the replay, so without the wait a busy client got them +// again, behind the state frame and in the middle of its replay. No new frame +// joins that queue once the flag is set, so it only drains. +func (d *Daemon) beginOutputHold(c *ipc.Conn) { + d.holdGate.Lock() + d.holdMu.Lock() + if d.holds == nil { + d.holds = make(map[*ipc.Conn]*outputHold) + } + d.holds[c] = &outputHold{} + c.SetHoldPaneOutput(true) + d.holdMu.Unlock() + d.holdGate.Unlock() + + // Outside both locks: flushes must not wait on a client's socket. + deadline := time.Now().Add(holdDrainTimeout) + for c.QueuedOutput() > 0 { + if time.Now().After(deadline) { + logger.Debug("attach: %d live frames still queued after %v; proceeding (they may repeat replayed bytes)", + c.QueuedOutput(), holdDrainTimeout) + return + } + select { + case <-d.shutdown: + return + case <-c.Done(): + return // dead conn: the queue never drains + case <-time.After(2 * time.Millisecond): + } + } +} + +// holdOutput appends one flush to every hold. The caller holds holdGate for +// read and must NOT hold the pane's PluginMu. data is copied, because the +// caller's read buffer is reused. +func (d *Daemon) holdOutput(paneID string, start uint64, data []byte, gen uint64) { + d.holdMu.Lock() + defer d.holdMu.Unlock() + var cp []byte + for _, h := range d.holds { + if h.lost[paneID] { + continue + } + if h.bytes+len(data) > outputHoldLimit { + h.dropPane(paneID) + continue + } + if cp == nil { + cp = append([]byte(nil), data...) + } + h.chunks = append(h.chunks, heldChunk{paneID: paneID, start: start, data: cp, gen: gen}) + h.bytes += len(data) + } +} + +// releaseOutputHold delivers c's held output and ends the hold. end maps each +// pane whose replay was OutputBuf's bytes to the stream position that replay +// ended at; held bytes before it are already on the client. +// +// Batches are taken under holdMu and sent outside it. Flushes that happen +// during the drain append to the hold and go out in a later batch, in order. +// Each held frame goes through SendBlocking, i.e. the must-deliver queue that +// sendLoop drains first, so no live frame sent after the flag clears can +// overtake one. +func (d *Daemon) releaseOutputHold(c *ipc.Conn, end map[string]uint64) { + for { + d.holdMu.Lock() + h := d.holds[c] + if h == nil { + d.holdMu.Unlock() + return + } + batch, lost := h.chunks, h.lost + h.chunks, h.bytes, h.lost = nil, 0, nil + d.holdMu.Unlock() + + if len(batch) == 0 && len(lost) == 0 { + if d.finishOutputHold(c) { + return + } + continue + } + for _, ch := range batch { + data := ch.data + if e, ok := end[ch.paneID]; ok { + if ch.start+uint64(len(data)) <= e { + continue // already replayed + } + if ch.start < e { + data = data[e-ch.start:] // partial overlap + } + } + msg, _ := ipc.NewMessage(ipc.MsgPaneOutput, ipc.PaneOutputPayload{PaneID: ch.paneID, Data: data, Generation: ch.gen}) + if err := c.SendBlocking(msg, d.shutdown); err != nil { + d.dropOutputHold(c) + return + } + } + for paneID := range lost { + p := d.session.Pane(paneID) + if p == nil { + continue + } + log.Printf("attach: held output for pane %s passed %d bytes and was dropped; asking it to repaint", + paneID, outputHoldLimit) + d.redrawKickPane(p) + } + } +} + +// finishOutputHold ends c's hold when nothing more is held, and reports +// whether it did. Taking holdGate for write is what stops a flush sitting +// between its hold append and its broadcast: that flush either finished (its +// chunk is in the hold, so this refuses and the caller drains again) or has +// not appended yet (it will see no hold and an unheld conn). +func (d *Daemon) finishOutputHold(c *ipc.Conn) bool { + d.holdGate.Lock() + defer d.holdGate.Unlock() + d.holdMu.Lock() + defer d.holdMu.Unlock() + h := d.holds[c] + if h == nil { + return true + } + if len(h.chunks) > 0 || len(h.lost) > 0 { + return false + } + delete(d.holds, c) + c.SetHoldPaneOutput(false) + return true +} + +// dropOutputHold discards c's hold without sending it: the conn is gone, or +// its attach stopped part way. The flag is cleared too, so a conn that is +// still alive is not left off live output forever. +func (d *Daemon) dropOutputHold(c *ipc.Conn) { + d.holdMu.Lock() + defer d.holdMu.Unlock() + if _, ok := d.holds[c]; !ok { + return + } + delete(d.holds, c) + c.SetHoldPaneOutput(false) +} + +// redrawKickPane asks a pane whose held output was dropped to repaint. Type +// and "running" are read in one PluginMu span, as handleAttach does; the kick +// itself runs outside it. +func (d *Daemon) redrawKickPane(p *Pane) { + p.PluginMu.Lock() + typ := p.Type + running := p.PTY != nil && p.ExitCode == nil + p.PluginMu.Unlock() + if running { + d.redrawKick(p, typ) + } +} diff --git a/internal/daemon/outputhold_integration_test.go b/internal/daemon/outputhold_integration_test.go new file mode 100644 index 00000000..edb01f24 --- /dev/null +++ b/internal/daemon/outputhold_integration_test.go @@ -0,0 +1,228 @@ +//go:build integration + +package daemon + +import ( + "bytes" + "encoding/binary" + "sync/atomic" + "testing" + "time" + + "github.com/artyomsv/quil/internal/ipc" +) + +// Spec 9.1.6: a client attaching while a pane is writing receives the pane's +// replay and its live output as ONE stream — every byte once, in order. +// +// Only frames after the attach's workspace_state count: a conn receives live +// output from the moment it connects, and those earlier frames are for panes +// the client does not know yet. +func TestHold_HeldConnReceivesReplayThenLiveExactlyOnce(t *testing.T) { + h := newHoldHarness(t) + tab := h.d.session.CreateTab("T") + p := h.pane(tab.ID, "terminal") + + // 2000 chunks x 64 bytes stays well inside the ring buffer, so the replay + // starts at counter 0. Paced, so the live stream never outruns a reading + // client: a full droppable queue would drop frames by design. + const chunks = 2000 + var written atomic.Int64 + writerDone := make(chan struct{}) + go func() { + defer close(writerDone) + for i := 0; i < chunks; i++ { + h.d.flushPaneOutput(p.ID, counterBytes(i)) + written.Store(int64(i + 1)) + if i%4 == 0 { + time.Sleep(200 * time.Microsecond) + } + } + }() + + client, conn := h.dial("B") + type result struct { + data []byte + err string + } + res := make(chan result, 1) + go func() { + var out []byte + attached := false + last := uint64(chunks*8 - 1) + _ = client.SetReadDeadline(time.Now().Add(20 * time.Second)) + for { + m, err := client.Receive() + if err != nil { + res <- result{out, "read: " + err.Error()} + return + } + switch m.Type { + case ipc.MsgWorkspaceState: + attached = true + case ipc.MsgPaneOutput: + if !attached { + continue + } + var f ipc.PaneOutputPayload + if err := m.DecodePayload(&f); err != nil { + res <- result{out, "decode: " + err.Error()} + return + } + if f.PaneID != p.ID { + continue + } + if len(f.Data) == 0 { + res <- result{out, "empty pane_output frame"} + return + } + out = append(out, f.Data...) + if len(out) >= 8 && binary.LittleEndian.Uint64(out[len(out)-8:]) == last { + res <- result{out, ""} + return + } + } + } + }() + + waitUntil(t, "writer mid-stream", func() bool { return written.Load() >= chunks/4 }) + sendClientMsg(t, client, ipc.MsgAttach, ipc.AttachPayload{ClientID: "B", Cols: 80, Rows: 24}) + if written.Load() >= chunks { + t.Fatal("setup: the writer finished before the attach; the test covers nothing") + } + <-writerDone + + r := <-res + if r.err != "" { + t.Fatalf("after %d bytes: %s", len(r.data), r.err) + } + if n := conn.Dropped(); n != 0 { + t.Fatalf("setup: %d live frames dropped by a full queue; the writer is paced too fast", n) + } + checkCounterRun(t, r.data, 0, uint64(chunks*8-1)) +} + +// A conn receives live output from the moment it connects. Frames still queued +// for it when it attaches — a busy client with a full socket — must not be +// written after the attach's state and replay: their bytes are already in the +// replay, and the must-deliver queue is drained first, so they would land as +// stale duplicates in the middle of it. +func TestHold_LiveFramesQueuedBeforeAttachStayBeforeTheState(t *testing.T) { + h := newHoldHarness(t) + tab := h.d.session.CreateTab("T") + p := h.pane(tab.ID, "terminal") + client, conn := h.dial("B") + + // Nobody reads: the socket fills and live frames back up in the conn's + // droppable queue. + const perChunk = 1024 // counters per 8 KiB chunk + next := uint64(0) + flush := func() { + h.d.flushPaneOutput(p.ID, counterRun(next, perChunk)) + next += perChunk + } + for conn.Dropped() == 0 && next < 1024*perChunk { + flush() + } + if conn.Dropped() == 0 { + t.Fatal("setup: the droppable queue never filled") + } + + // The live chunk flushed once the attach is done; the read stops at it. + last := next + perChunk - 1 + sendClientMsg(t, client, ipc.MsgAttach, ipc.AttachPayload{ClientID: "B", Cols: 80, Rows: 24}) + go func() { + deadline := time.Now().Add(10 * time.Second) + for time.Now().Before(deadline) { + if held, _, _ := h.holdOf(conn); !held && h.d.clientCount() == 1 { + break + } + time.Sleep(2 * time.Millisecond) + } + flush() + }() + // Give the attach time to start before draining the backlog. + time.Sleep(50 * time.Millisecond) + + var out []byte + attached := false + _ = client.SetReadDeadline(time.Now().Add(15 * time.Second)) + for { + m, err := client.Receive() + if err != nil { + t.Fatalf("read after %d bytes: %v", len(out), err) + } + if m.Type == ipc.MsgWorkspaceState { + attached = true + continue + } + if m.Type != ipc.MsgPaneOutput || !attached { + continue + } + var f ipc.PaneOutputPayload + if err := m.DecodePayload(&f); err != nil { + t.Fatalf("decode: %v", err) + } + out = append(out, f.Data...) + if len(f.Data) >= 8 && binary.LittleEndian.Uint64(f.Data[len(f.Data)-8:]) == last && !f.Ghost { + break + } + } + // The ring keeps whole counters (its size and every chunk are multiples + // of 8 bytes), so the replay starts on one. + checkCounterRun(t, out, binary.LittleEndian.Uint64(out), last) +} + +// A ghostsnap replay is a PREVIOUS session's bytes, so it records no end: every +// byte the new child wrote during the hold arrives after the replay and its +// scroll-out. Recording an end there would cut them. +func TestHold_GhostSnapReplayKeepsHeldBytes(t *testing.T) { + h := newHoldHarness(t) + tab := h.d.session.CreateTab("T") + // 1 MiB of replay ahead of the ghostsnap pane, with the client not + // reading, parks the attach before it reaches that pane. + for i := 0; i < 4; i++ { + x := h.pane(tab.ID, "terminal") + x.OutputBuf.Write(bytes.Repeat([]byte{'x'}, 256000)) + } + g := h.pane(tab.ID, "terminal") + g.PluginMu.Lock() + g.GhostSnap = []byte("SNAP") + g.PluginMu.Unlock() + + client, conn := h.dial("B") + sendClientMsg(t, client, ipc.MsgAttach, ipc.AttachPayload{ClientID: "B", Cols: 80, Rows: 24}) + waitUntil(t, "attach hold", func() bool { held, _, _ := h.holdOf(conn); return held }) + h.d.flushPaneOutput(g.ID, []byte("held-1")) + h.d.flushPaneOutput(g.ID, []byte("held-2")) + go func() { + deadline := time.Now().Add(10 * time.Second) + for h.holdCount() != 0 && time.Now().Before(deadline) { + time.Sleep(2 * time.Millisecond) + } + h.d.flushPaneOutput(g.ID, []byte("tail")) + }() + + frames := readOutput(t, client, isChunk(g.ID, "tail")) + var ghost, live []byte + for _, f := range frames { + if f.PaneID != g.ID { + continue + } + if f.Ghost { + if len(live) > 0 { + t.Fatal("a ghost frame arrived after live output") + } + ghost = append(ghost, f.Data...) + continue + } + live = append(live, f.Data...) + } + _, rows := paneSize(g) + if want := append([]byte("SNAP"), ghostScrollOut(rows)...); !bytes.Equal(ghost, want) { + t.Errorf("ghost = %q, want %q", ghost, want) + } + if got, want := string(live), "held-1held-2tail"; got != want { + t.Errorf("live = %q, want %q", got, want) + } +} diff --git a/internal/daemon/outputhold_test.go b/internal/daemon/outputhold_test.go new file mode 100644 index 00000000..e3813290 --- /dev/null +++ b/internal/daemon/outputhold_test.go @@ -0,0 +1,502 @@ +package daemon + +import ( + "bytes" + "encoding/binary" + "path/filepath" + "sync" + "testing" + "time" + + "github.com/artyomsv/quil/internal/config" + "github.com/artyomsv/quil/internal/ipc" +) + +// The output hold (outputhold.go) is driven through a real ipc.Server: the +// hold is a flag on the SERVER side of a conn, and what matters is which frames +// reach the client's socket, in which order. ipc.Conn has no exported +// constructor, so the harness learns each server-side conn from a probe frame +// the test client sends first. + +const msgHoldProbe = "hold_test_probe" + +type holdProbePayload struct { + Tag string `json:"tag"` +} + +type holdHarness struct { + t *testing.T + d *Daemon + sock string + + mu sync.Mutex + conns map[string]*ipc.Conn +} + +func newHoldHarness(t *testing.T) *holdHarness { + t.Helper() + home := t.TempDir() + t.Setenv("QUIL_HOME", home) + h := &holdHarness{t: t, d: New(config.Default()), sock: filepath.Join(home, "s.sock"), conns: map[string]*ipc.Conn{}} + h.d.server = ipc.NewServer(h.sock, func(c *ipc.Conn, m *ipc.Message) { + if m.Type == msgHoldProbe { + var p holdProbePayload + if err := m.DecodePayload(&p); err == nil { + h.mu.Lock() + h.conns[p.Tag] = c + h.mu.Unlock() + } + return + } + h.d.handleMessage(c, m) + }, h.d.onClientDisconnect) + if err := h.d.server.Start(); err != nil { + t.Fatalf("start IPC server: %v", err) + } + t.Cleanup(func() { h.d.server.Stop() }) + return h +} + +// dial connects a client and returns it with its server-side conn. +func (h *holdHarness) dial(tag string) (*ipc.Client, *ipc.Conn) { + h.t.Helper() + c, err := ipc.NewClient(h.sock) + if err != nil { + h.t.Fatalf("dial: %v", err) + } + h.t.Cleanup(func() { c.Close() }) + sendClientMsg(h.t, c, msgHoldProbe, holdProbePayload{Tag: tag}) + var conn *ipc.Conn + waitUntil(h.t, "server conn for "+tag, func() bool { + h.mu.Lock() + defer h.mu.Unlock() + conn = h.conns[tag] + return conn != nil + }) + return c, conn +} + +func (h *holdHarness) pane(tabID, typ string) *Pane { + h.t.Helper() + p, err := h.d.session.CreatePane(tabID, "") + if err != nil { + h.t.Fatalf("create pane: %v", err) + } + p.PluginMu.Lock() + p.Type = typ + p.PluginMu.Unlock() + return p +} + +func (h *holdHarness) holdOf(c *ipc.Conn) (held bool, chunks int, lost map[string]bool) { + h.d.holdMu.Lock() + defer h.d.holdMu.Unlock() + hold := h.d.holds[c] + if hold == nil { + return false, 0, nil + } + lost = map[string]bool{} + for k, v := range hold.lost { + lost[k] = v + } + return true, len(hold.chunks), lost +} + +func (h *holdHarness) holdCount() int { + h.d.holdMu.Lock() + defer h.d.holdMu.Unlock() + return len(h.d.holds) +} + +// readOutput reads c's pane_output frames until stop accepts one, returning +// them all (the accepted one last). Other frame types are skipped. +func readOutput(t *testing.T, c *ipc.Client, stop func(ipc.PaneOutputPayload) bool) []ipc.PaneOutputPayload { + t.Helper() + var got []ipc.PaneOutputPayload + if err := c.SetReadDeadline(time.Now().Add(10 * time.Second)); err != nil { + t.Fatalf("set read deadline: %v", err) + } + for { + m, err := c.Receive() + if err != nil { + t.Fatalf("read after %d pane_output frames: %v", len(got), err) + } + if m.Type != ipc.MsgPaneOutput { + continue + } + var p ipc.PaneOutputPayload + if err := m.DecodePayload(&p); err != nil { + t.Fatalf("decode pane_output: %v", err) + } + got = append(got, p) + if stop(p) { + return got + } + } +} + +func isChunk(paneID, data string) func(ipc.PaneOutputPayload) bool { + return func(p ipc.PaneOutputPayload) bool { return p.PaneID == paneID && string(p.Data) == data } +} + +// paneData concatenates one pane's frames and fails on an empty frame: a +// release that re-encodes a fully replayed chunk as zero bytes is a bug even +// though it adds no bytes. +func paneData(t *testing.T, frames []ipc.PaneOutputPayload, paneID string) []byte { + t.Helper() + var b []byte + for _, f := range frames { + if f.PaneID != paneID { + continue + } + if len(f.Data) == 0 { + t.Fatalf("empty pane_output frame for pane %s", paneID) + } + b = append(b, f.Data...) + } + return b +} + +// counterRun is n consecutive little-endian uint64 counters starting at first. +func counterRun(first uint64, n int) []byte { + b := make([]byte, 8*n) + for i := 0; i < n; i++ { + binary.LittleEndian.PutUint64(b[8*i:], first+uint64(i)) + } + return b +} + +// counterBytes is chunk i of a counter stream: 64 bytes, counters 8i..8i+7. +func counterBytes(i int) []byte { return counterRun(uint64(i)*8, 8) } + +// checkCounterRun asserts b is exactly the counters first..last, in order: +// no duplicate, no gap, nothing extra at either end. +func checkCounterRun(t *testing.T, b []byte, first, last uint64) { + t.Helper() + if len(b)%8 != 0 { + t.Fatalf("stream is %d bytes, not a whole number of counters", len(b)) + } + if len(b) == 0 { + t.Fatal("stream is empty") + } + for i := 0; i < len(b); i += 8 { + want := first + uint64(i/8) + if got := binary.LittleEndian.Uint64(b[i:]); got != want { + t.Fatalf("counter #%d = %d, want %d (a duplicate or a gap)", i/8, got, want) + } + } + if got := binary.LittleEndian.Uint64(b[len(b)-8:]); got != last { + t.Fatalf("stream ends at %d, want %d", got, last) + } +} + +// A pane whose held bytes pass the bound loses them and is kicked once to +// repaint; the other pane's held bytes still arrive, and the conn stays open. +func TestHold_OverflowDropsPaneAndKicks(t *testing.T) { + h := newHoldHarness(t) + loadRedrawKeyPlugin(t, h.d, "kicky") + tab := h.d.session.CreateTab("T") + p := h.pane(tab.ID, "kicky") + p.PluginMu.Lock() + p.PTY = &resizeProbeSession{} + p.PluginMu.Unlock() + t.Cleanup(p.StopInput) + q := h.pane(tab.ID, "terminal") + + client, conn := h.dial("B") + h.d.beginOutputHold(conn) + + h.d.flushPaneOutput(q.ID, []byte("q1")) + big := bytes.Repeat([]byte{'p'}, 64<<10) + for sent := 0; sent <= outputHoldLimit; sent += len(big) { + h.d.flushPaneOutput(p.ID, big) + } + h.d.flushPaneOutput(p.ID, []byte("after-overflow")) + h.d.flushPaneOutput(q.ID, []byte("q2")) + + if _, _, lost := h.holdOf(conn); !lost[p.ID] { + t.Fatalf("pane P not marked lost after passing %d held bytes", outputHoldLimit) + } + if n := p.inputEnqueued.Load(); n != 0 { + t.Fatalf("P kicked %d times before the release, want 0", n) + } + + h.d.releaseOutputHold(conn, nil) + h.d.flushPaneOutput(p.ID, []byte("live")) + + frames := readOutput(t, client, isChunk(p.ID, "live")) + if got := string(paneData(t, frames, q.ID)); got != "q1q2" { + t.Errorf("Q received %q, want %q", got, "q1q2") + } + if got := string(paneData(t, frames, p.ID)); got != "live" { + t.Errorf("P received %d held bytes, want only the live frame after release", len(got)-len("live")) + } + waitUntil(t, "one redraw kick for P", func() bool { return p.inputEnqueued.Load() >= 1 }) + time.Sleep(50 * time.Millisecond) + if n := p.inputEnqueued.Load(); n != 1 { + t.Errorf("P kicked %d times, want 1", n) + } + if n := h.holdCount(); n != 0 { + t.Errorf("%d holds left after release, want 0", n) + } +} + +// Review Focus 3: a restart during the hold. Held chunks of both generations +// arrive in order with their own generation, and the cut at the replay's end +// is by stream position only — the old generation's bytes past it are kept. +func TestHold_RestartDuringHoldKeepsGenerations(t *testing.T) { + h := newHoldHarness(t) + tab := h.d.session.CreateTab("T") + p := h.pane(tab.ID, "terminal") + p.PluginMu.Lock() + p.PTY = &fakeSession{} + p.ptyGen = 1 + p.PluginMu.Unlock() + + client, conn := h.dial("B") + h.d.beginOutputHold(conn) + + h.d.flushPaneOutputGeneration(p.ID, []byte("a1"), 1) + p.PluginMu.Lock() + end := p.outPos // the replay ended after a1 + p.PluginMu.Unlock() + h.d.flushPaneOutputGeneration(p.ID, []byte("a2"), 1) + p.PluginMu.Lock() + p.ptyGen = 2 // restarted + p.PluginMu.Unlock() + h.d.flushPaneOutputGeneration(p.ID, []byte("b1"), 2) + + h.d.releaseOutputHold(conn, map[string]uint64{p.ID: end}) + h.d.flushPaneOutputGeneration(p.ID, []byte("b2"), 2) + + frames := readOutput(t, client, isChunk(p.ID, "b2")) + type fr struct { + data string + gen uint64 + } + var got []fr + for _, f := range frames { + got = append(got, fr{string(f.Data), f.Generation}) + } + want := []fr{{"a2", 1}, {"b1", 2}, {"b2", 2}} + if len(got) != len(want) { + t.Fatalf("frames = %v, want %v", got, want) + } + for i := range want { + if got[i] != want[i] { + t.Fatalf("frames = %v, want %v", got, want) + } + } +} + +// The cut at the replay's end: a chunk wholly before it is dropped (no empty +// frame), one straddling it is trimmed, one after it is kept whole. +func TestHold_ReleaseCutsWhatTheReplayCovered(t *testing.T) { + for _, tc := range []struct { + name string + end uint64 + want []string + }{ + {"end on a chunk boundary", 8, []string{"ijkl"}}, + {"end inside a chunk", 6, []string{"gh", "ijkl"}}, + {"end at the start", 0, []string{"abcd", "efgh", "ijkl"}}, + } { + t.Run(tc.name, func(t *testing.T) { + h := newHoldHarness(t) + tab := h.d.session.CreateTab("T") + p := h.pane(tab.ID, "terminal") + client, conn := h.dial("B") + h.d.beginOutputHold(conn) + for _, s := range []string{"abcd", "efgh", "ijkl"} { + h.d.flushPaneOutput(p.ID, []byte(s)) + } + h.d.releaseOutputHold(conn, map[string]uint64{p.ID: tc.end}) + h.d.flushPaneOutput(p.ID, []byte("Z")) + + frames := readOutput(t, client, isChunk(p.ID, "Z")) + var got []string + for _, f := range frames[:len(frames)-1] { + got = append(got, string(f.Data)) + } + if len(got) != len(tc.want) { + t.Fatalf("held frames = %q, want %q", got, tc.want) + } + for i := range got { + if got[i] != tc.want[i] { + t.Fatalf("held frames = %q, want %q", got, tc.want) + } + } + }) + } +} + +// Flushes that land while the release is still draining go out after the +// earlier held bytes, and exactly once: the conn stays held until the hold is +// empty, so no flush reaches it live AND through the hold. +func TestHold_FlushDuringDrainArrivesOnceInOrder(t *testing.T) { + h := newHoldHarness(t) + tab := h.d.session.CreateTab("T") + p := h.pane(tab.ID, "terminal") + client, conn := h.dial("B") + h.d.beginOutputHold(conn) + + // 1 MiB held and nobody reading: more than the socket buffer plus the + // must-deliver queue's headroom, so the release blocks mid-drain. + const perChunk = 1024 // counters per chunk (8 KiB) + next := uint64(0) + flush := func() { + h.d.flushPaneOutput(p.ID, counterRun(next, perChunk)) + next += perChunk + } + for i := 0; i < 128; i++ { + flush() + } + released := make(chan struct{}) + go func() { + h.d.releaseOutputHold(conn, nil) + close(released) + }() + waitUntil(t, "release to take the first batch", func() bool { + held, chunks, _ := h.holdOf(conn) + return held && chunks == 0 + }) + for i := 0; i < 20; i++ { + flush() + } + last := next - 1 + + frames := readOutput(t, client, func(f ipc.PaneOutputPayload) bool { + return f.PaneID == p.ID && len(f.Data) >= 8 && + binary.LittleEndian.Uint64(f.Data[len(f.Data)-8:]) == last + }) + checkCounterRun(t, paneData(t, frames, p.ID), 0, last) + select { + case <-released: + case <-time.After(5 * time.Second): + t.Fatal("release did not return") + } + if n := h.holdCount(); n != 0 { + t.Errorf("%d holds left, want 0", n) + } +} + +// pauseFirstFlush makes the first flush of paneID stop between its hold append +// and its broadcast until resume is closed. paused closes when it stops. +func pauseFirstFlush(d *Daemon, paneID string) (paused, resume chan struct{}) { + paused, resume = make(chan struct{}), make(chan struct{}) + var once sync.Once + d.afterHoldOutput = func(id string) { + if id != paneID { + return + } + once.Do(func() { + close(paused) + <-resume + }) + } + return paused, resume +} + +// A flush caught between its hold append and its broadcast while the release +// ends the hold: its bytes must arrive once, through the hold, and not a +// second time through a broadcast that finds the conn already unheld. +func TestHold_FlushStraddlingTheReleaseArrivesOnce(t *testing.T) { + h := newHoldHarness(t) + tab := h.d.session.CreateTab("T") + p := h.pane(tab.ID, "terminal") + client, conn := h.dial("B") + paused, resume := pauseFirstFlush(h.d, p.ID) + + h.d.beginOutputHold(conn) + flushed := make(chan struct{}) + go func() { + h.d.flushPaneOutput(p.ID, []byte("s")) + close(flushed) + }() + <-paused + released := make(chan struct{}) + go func() { + h.d.releaseOutputHold(conn, nil) + close(released) + }() + waitUntil(t, "release to take the held chunk", func() bool { + _, chunks, _ := h.holdOf(conn) + return chunks == 0 + }) + time.Sleep(50 * time.Millisecond) // let the release reach its final clear + close(resume) + <-flushed + <-released + + h.d.flushPaneOutput(p.ID, []byte("Z")) + if got := string(paneData(t, readOutput(t, client, isChunk(p.ID, "Z")), p.ID)); got != "sZ" { + t.Fatalf("received %q, want %q", got, "sZ") + } +} + +// The same straddle as a hold begins: a flush that appended before the hold +// existed must still reach the conn through its broadcast, not find the conn +// held and be skipped with nothing held. +func TestHold_FlushStraddlingTheBeginArrivesOnce(t *testing.T) { + h := newHoldHarness(t) + tab := h.d.session.CreateTab("T") + p := h.pane(tab.ID, "terminal") + client, conn := h.dial("B") + paused, resume := pauseFirstFlush(h.d, p.ID) + + flushed := make(chan struct{}) + go func() { + h.d.flushPaneOutput(p.ID, []byte("s")) + close(flushed) + }() + <-paused + begun := make(chan struct{}) + go func() { + h.d.beginOutputHold(conn) + close(begun) + }() + time.Sleep(50 * time.Millisecond) // let the begin set the flag, if it can + close(resume) + <-flushed + <-begun + h.d.releaseOutputHold(conn, nil) + + h.d.flushPaneOutput(p.ID, []byte("Z")) + if got := string(paneData(t, readOutput(t, client, isChunk(p.ID, "Z")), p.ID)); got != "sZ" { + t.Fatalf("received %q, want %q", got, "sZ") + } +} + +func TestHold_DisconnectDropsHold(t *testing.T) { + h := newHoldHarness(t) + tab := h.d.session.CreateTab("T") + p := h.pane(tab.ID, "terminal") + client, conn := h.dial("B") + h.d.beginOutputHold(conn) + h.d.flushPaneOutput(p.ID, []byte("held")) + if held, chunks, _ := h.holdOf(conn); !held || chunks != 1 { + t.Fatalf("hold = %v with %d chunks, want a hold with 1", held, chunks) + } + client.Close() + waitUntil(t, "hold dropped on disconnect", func() bool { return h.holdCount() == 0 }) +} + +// Only the attaching conn is held: another client keeps its live output. +func TestHold_OtherClientsUnaffected(t *testing.T) { + h := newHoldHarness(t) + tab := h.d.session.CreateTab("T") + p := h.pane(tab.ID, "terminal") + a, _ := h.dial("A") + b, connB := h.dial("B") + h.d.beginOutputHold(connB) + + h.d.flushPaneOutput(p.ID, []byte("x")) + if got := string(paneData(t, readOutput(t, a, isChunk(p.ID, "x")), p.ID)); got != "x" { + t.Fatalf("A received %q while B was held, want %q", got, "x") + } + for _, m := range readFor(b, 200*time.Millisecond) { + if m.Type == ipc.MsgPaneOutput { + t.Fatal("held client B received live pane_output") + } + } +} diff --git a/internal/daemon/session.go b/internal/daemon/session.go index 79865c35..77eef2ab 100644 --- a/internal/daemon/session.go +++ b/internal/daemon/session.go @@ -94,6 +94,13 @@ type Pane struct { // the saved buffer grows by a screenful on every restart. // PluginMu-protected. ghostSeeded bool + // outPos is the total number of bytes ever appended to this pane's output + // stream: the stream position of the next byte. It never resets — not on + // OutputBuf.Reset, not on a restart (the generation marks the new run). + // An attach reads it beside the OutputBuf bytes it replays, so the bytes + // it held back can be cut where that replay ended (outputhold.go). + // Runtime-only, never on the wire. PluginMu-protected. + outPos uint64 // WorktreeOwned marks a pane created into a linked worktree Quil made for // it. PERSISTED, and it is the only thing that lets restore tell a missing // WORKTREE from a missing browsed directory — the snapshot stores just CWD diff --git a/internal/ipc/queued_output_test.go b/internal/ipc/queued_output_test.go new file mode 100644 index 00000000..93e874d2 --- /dev/null +++ b/internal/ipc/queued_output_test.go @@ -0,0 +1,60 @@ +package ipc + +import ( + "io" + "net" + "testing" + "time" +) + +func waitForConn(t *testing.T, what string, cond func() bool) { + t.Helper() + deadline := time.Now().Add(3 * time.Second) + for !cond() { + if time.Now().After(deadline) { + t.Fatalf("timed out waiting for %s", what) + } + time.Sleep(time.Millisecond) + } +} + +// QueuedOutput counts live frames waiting in the droppable queue — the ones a +// must-deliver frame queued now would overtake — and reaches zero once the +// peer reads. Done closes when the conn dies. +func TestConn_QueuedOutputCountsWaitingLiveFramesAndDoneCloses(t *testing.T) { + local, remote := net.Pipe() + t.Cleanup(func() { remote.Close() }) + c := newConn(local) + + out, err := NewMessage(MsgPaneOutput, PaneOutputPayload{PaneID: "p", Data: []byte("x")}) + if err != nil { + t.Fatalf("build pane_output: %v", err) + } + frame, err := EncodeFrame(out) + if err != nil { + t.Fatalf("encode: %v", err) + } + // Nobody reads the pipe: sendLoop takes the first frame and blocks in + // its write, so the other two wait in the queue. + for i := 0; i < 3; i++ { + if err := c.enqueue(frame, true); err != nil { + t.Fatalf("enqueue: %v", err) + } + } + waitForConn(t, "two frames waiting", func() bool { return c.QueuedOutput() == 2 }) + + go func() { _, _ = io.Copy(io.Discard, remote) }() + waitForConn(t, "the queue to drain", func() bool { return c.QueuedOutput() == 0 }) + + select { + case <-c.Done(): + t.Fatal("Done closed on a live conn") + default: + } + c.Close() + select { + case <-c.Done(): + case <-time.After(3 * time.Second): + t.Fatal("Done not closed after Close") + } +} diff --git a/internal/ipc/server.go b/internal/ipc/server.go index 95883315..03b235b8 100644 --- a/internal/ipc/server.go +++ b/internal/ipc/server.go @@ -709,6 +709,18 @@ func (c *Conn) wantsPaneOutput() bool { return !c.noPaneOutput.Load() } // the replay is queued. func (c *Conn) SetHoldPaneOutput(on bool) { c.holdPaneOutput.Store(on) } +// QueuedOutput reports how many live pane_output frames wait in this conn's +// droppable queue, not yet taken by sendLoop. sendLoop drains the must-deliver +// queue first, so a frame still waiting here is written AFTER any +// must-deliver frame queued now. The daemon waits for zero after holding a +// conn, so live frames queued before an attach cannot land behind its replay. +func (c *Conn) QueuedOutput() int { return len(c.outCh) } + +// Done is closed once the conn is dead (closed, or its sendLoop gone). After +// that the droppable queue never drains, so a wait on QueuedOutput selects on +// it to stop. +func (c *Conn) Done() <-chan struct{} { return c.done } + // wantsFrame reports whether a frame of this type should be delivered to this // conn. Only the live pane-output stream is ever filtered: everything else is // must-deliver, and a client excusing itself from PTY bytes — or held off From d11db908ad57677dc320e8e278ab7c5024c29912 Mon Sep 17 00:00:00 2001 From: Artjoms Stukans Date: Fri, 25 Sep 2026 17:28:29 +0200 Subject: [PATCH 05/40] fix(daemon): send an attach's master change to other clients only An attach that changed the size master broadcast the workspace state after answering the attach. The attaching conn therefore received two state frames back to back: its own attach state, which is built after registration and already names the new master, and the broadcast, which said nothing new. That was must-deliver queue pressure for no information, and it made two broadcast-counting tests flaky under -race (the second frame landed inside their counting window). The master change now goes only to the other attached clients. A conn that never attached (an MCP bridge) has no use for it either. Part of #235 --- internal/daemon/clients.go | 29 ++++++++++++++++++++++++++ internal/daemon/clients_wiring_test.go | 25 ++++++++++++++++++++++ internal/daemon/daemon.go | 8 +++---- 3 files changed, 58 insertions(+), 4 deletions(-) diff --git a/internal/daemon/clients.go b/internal/daemon/clients.go index c6971f27..1a89c1d1 100644 --- a/internal/daemon/clients.go +++ b/internal/daemon/clients.go @@ -427,6 +427,35 @@ func (d *Daemon) shuttingDown() bool { } } +// sendStateToOtherClients sends the workspace state to every attached client +// except the one given, for a master change an attach made. +// +// Not a broadcast. A broadcast also reached the attaching conn, which then got +// two state frames back to back: its own attach state and this one, which says +// nothing new. That is pressure on its must-deliver queue for no information. +// A conn that never attached (an MCP bridge) has no use for the master either. +func (d *Daemon) sendStateToOtherClients(except *ipc.Conn) { + d.clients.mu.Lock() + var conns []*ipc.Conn + for _, rec := range d.clients.sortedRecordsLocked() { + if rec.conn != except { + conns = append(conns, rec.conn) + } + } + d.clients.mu.Unlock() + if len(conns) == 0 { + return + } + msg, err := ipc.NewMessage(ipc.MsgWorkspaceState, d.buildWorkspaceState()) + if err != nil { + logger.Error("attach: build state for other clients: %v", err) + return + } + for _, c := range conns { + c.Send(msg) + } +} + func (d *Daemon) handleDetach(conn *ipc.Conn) { if d.detachClient(conn) { d.broadcastState() diff --git a/internal/daemon/clients_wiring_test.go b/internal/daemon/clients_wiring_test.go index 797d3b2a..4d8d4323 100644 --- a/internal/daemon/clients_wiring_test.go +++ b/internal/daemon/clients_wiring_test.go @@ -2,6 +2,7 @@ package daemon import ( "testing" + "time" "github.com/artyomsv/quil/internal/ipc" ) @@ -109,6 +110,30 @@ func TestClientDispatch_ZeroAttachNeverElectedDespiteDefault(t *testing.T) { } } +// An attach that changes the master tells every OTHER attached client, and +// sends the attaching client only its own state. That state already names the +// new master; a broadcast on top of it was a second frame with nothing new. +func TestClientDispatch_AttachMasterChangeReachesOthersOnly(t *testing.T) { + d, sock, _ := resizeAuthorityDaemon(t) + small := attachClientAs(t, sock, "small", 0, 0) + readUntil(t, small, "small's attach state", isType(ipc.MsgWorkspaceState)) + if d.masterID() != "" { + t.Fatalf("masterID = %q, want none for a 0x0 client", d.masterID()) + } + + big := attachClientAs(t, sock, "big", 200, 50) + readUntil(t, small, "a state naming big the master", func(m *ipc.Message) bool { + if m.Type != ipc.MsgWorkspaceState { + return false + } + var s map[string]any + return m.DecodePayload(&s) == nil && s["size_master"] == "big" + }) + if n := countType(readFor(big, 300*time.Millisecond), ipc.MsgWorkspaceState); n != 1 { + t.Errorf("the attaching client received %d workspace_state frames, want only its own 1", n) + } +} + // A lost link through the real server keeps the master's slot while a // follower is attached. func TestClientDispatch_LostLinkReservesSlot(t *testing.T) { diff --git a/internal/daemon/daemon.go b/internal/daemon/daemon.go index f8a88da2..b798c339 100644 --- a/internal/daemon/daemon.go +++ b/internal/daemon/daemon.go @@ -1795,11 +1795,11 @@ func (d *Daemon) handleAttach(conn *ipc.Conn, msg *ipc.Message) { // any of the work below, which has early returns of its own, and before // the 80x24 defaulting: the election reads the RAW geometry. // - // A master change is broadcast only once this attach is answered, so the - // new client's first workspace state is its own full one rather than a - // broadcast of a workspace this attach may be about to create. + // A master change reaches the OTHER attached clients once this attach is + // answered. The attaching conn gets none: its own state below is built + // after this registration, so it already names the new master. if d.registerClient(conn, attach) { - defer d.broadcastState() + defer d.sendStateToOtherClients(conn) } // Hold this conn off live pane output until its replay is sent, BEFORE From 65536321c57d70085157f99160282732c1db6c29 Mon Sep 17 00:00:00 2001 From: Artjoms Stukans Date: Fri, 25 Sep 2026 17:37:17 +0200 Subject: [PATCH 06/40] fix(daemon): pin the output hold's finish recheck and drain abort Review follow-ups for the attach output hold. - A flush can append after the release sees its last empty batch but before the hold ends; finishOutputHold's recheck sends it round again. A beforeFinishHold seam pins it: without the recheck those bytes were lost. - beginOutputHold's drain wait ends on conn close, pinned by closing a client with a backed-up queue; it uses a ticker and one timer. - Flushes skip holdMu when nothing is held (holdCount, changed only under holdGate for write). - dropOutputHold takes holdGate like every other hold change. --- internal/daemon/daemon.go | 13 ++-- internal/daemon/outputhold.go | 48 +++++++++++--- .../daemon/outputhold_integration_test.go | 10 ++- internal/daemon/outputhold_test.go | 63 +++++++++++++++++++ 4 files changed, 119 insertions(+), 15 deletions(-) diff --git a/internal/daemon/daemon.go b/internal/daemon/daemon.go index b798c339..53830679 100644 --- a/internal/daemon/daemon.go +++ b/internal/daemon/daemon.go @@ -282,10 +282,15 @@ type Daemon struct { holdGate sync.RWMutex holdMu sync.Mutex holds map[*ipc.Conn]*outputHold - // afterHoldOutput is a test seam: when set, a flush calls it between its - // hold append and its broadcast. Set before any flush runs; nil in - // production. - afterHoldOutput func(paneID string) + // holdCount is len(holds), written under holdGate (write) so a flush + // holding it for read can skip holdMu when nothing is held. + holdCount atomic.Int32 + // afterHoldOutput and beforeFinishHold are test seams: a flush calls the + // first between its hold append and its broadcast; the release calls the + // second between seeing an empty batch and ending the hold. Set before + // any flush runs; nil in production. + afterHoldOutput func(paneID string) + beforeFinishHold func(c *ipc.Conn) } func New(cfg config.Config) *Daemon { diff --git a/internal/daemon/outputhold.go b/internal/daemon/outputhold.go index 695c32a3..a253d5ff 100644 --- a/internal/daemon/outputhold.go +++ b/internal/daemon/outputhold.go @@ -32,8 +32,9 @@ import ( // - holdGate makes "append to the holds, then broadcast" ONE step against // setting or clearing a conn's flag. A flush holds it for READ from its // hold append through its broadcast (Broadcast only enqueues, so this -// blocks on nothing); beginOutputHold and the release's final clear hold -// it for WRITE. Without it, a flush could append its chunk, the release +// blocks on nothing). Every hold that starts or ends — beginOutputHold, +// the release's final clear, dropOutputHold — holds it for WRITE, which +// also keeps holdCount exact for a flush holding it for read. Without it, a flush could append its chunk, the release // could send that chunk and clear the flag, and then the flush's broadcast // would reach the now-unheld conn — the same bytes twice. At the other // end, a flush could miss the new hold and then broadcast to a conn whose @@ -103,25 +104,33 @@ func (d *Daemon) beginOutputHold(c *ipc.Conn) { if d.holds == nil { d.holds = make(map[*ipc.Conn]*outputHold) } + if _, ok := d.holds[c]; !ok { + d.holdCount.Add(1) + } d.holds[c] = &outputHold{} c.SetHoldPaneOutput(true) d.holdMu.Unlock() d.holdGate.Unlock() // Outside both locks: flushes must not wait on a client's socket. - deadline := time.Now().Add(holdDrainTimeout) + if c.QueuedOutput() == 0 { + return + } + deadline := time.NewTimer(holdDrainTimeout) + defer deadline.Stop() + tick := time.NewTicker(2 * time.Millisecond) + defer tick.Stop() for c.QueuedOutput() > 0 { - if time.Now().After(deadline) { - logger.Debug("attach: %d live frames still queued after %v; proceeding (they may repeat replayed bytes)", - c.QueuedOutput(), holdDrainTimeout) - return - } select { case <-d.shutdown: return case <-c.Done(): return // dead conn: the queue never drains - case <-time.After(2 * time.Millisecond): + case <-deadline.C: + logger.Debug("attach: %d live frames still queued after %v; proceeding (they may repeat replayed bytes)", + c.QueuedOutput(), holdDrainTimeout) + return + case <-tick.C: } } } @@ -130,6 +139,11 @@ func (d *Daemon) beginOutputHold(c *ipc.Conn) { // read and must NOT hold the pane's PluginMu. data is copied, because the // caller's read buffer is reused. func (d *Daemon) holdOutput(paneID string, start uint64, data []byte, gen uint64) { + // Nothing held, the common case: skip holdMu. Exact, because holdCount + // only changes under holdGate for write and the caller holds it for read. + if d.holdCount.Load() == 0 { + return + } d.holdMu.Lock() defer d.holdMu.Unlock() var cp []byte @@ -171,6 +185,11 @@ func (d *Daemon) releaseOutputHold(c *ipc.Conn, end map[string]uint64) { d.holdMu.Unlock() if len(batch) == 0 && len(lost) == 0 { + // A flush can append between the check above and the finish; + // finishOutputHold's recheck is what sends it round again. + if d.beforeFinishHold != nil { + d.beforeFinishHold(c) + } if d.finishOutputHold(c) { return } @@ -218,24 +237,33 @@ func (d *Daemon) finishOutputHold(c *ipc.Conn) bool { if h == nil { return true } + // The recheck: a flush may have appended since the caller saw an empty + // batch. Ending the hold now would lose those bytes — the conn would be + // unheld and the flush's broadcast already skipped it. if len(h.chunks) > 0 || len(h.lost) > 0 { return false } delete(d.holds, c) + d.holdCount.Add(-1) c.SetHoldPaneOutput(false) return true } // dropOutputHold discards c's hold without sending it: the conn is gone, or // its attach stopped part way. The flag is cleared too, so a conn that is -// still alive is not left off live output forever. +// still alive is not left off live output forever. Its held bytes are lost to +// it, which is the point; taking holdGate makes the clear atomic against +// flushes like every other hold change, so none straddles it. func (d *Daemon) dropOutputHold(c *ipc.Conn) { + d.holdGate.Lock() + defer d.holdGate.Unlock() d.holdMu.Lock() defer d.holdMu.Unlock() if _, ok := d.holds[c]; !ok { return } delete(d.holds, c) + d.holdCount.Add(-1) c.SetHoldPaneOutput(false) } diff --git a/internal/daemon/outputhold_integration_test.go b/internal/daemon/outputhold_integration_test.go index edb01f24..d1a9ddaf 100644 --- a/internal/daemon/outputhold_integration_test.go +++ b/internal/daemon/outputhold_integration_test.go @@ -132,9 +132,17 @@ func TestHold_LiveFramesQueuedBeforeAttachStayBeforeTheState(t *testing.T) { last := next + perChunk - 1 sendClientMsg(t, client, ipc.MsgAttach, ipc.AttachPayload{ClientID: "B", Cols: 80, Rows: 24}) go func() { + // Seen held first, so "not held" below means released rather than + // not begun yet. deadline := time.Now().Add(10 * time.Second) for time.Now().Before(deadline) { - if held, _, _ := h.holdOf(conn); !held && h.d.clientCount() == 1 { + if held, _, _ := h.holdOf(conn); held { + break + } + time.Sleep(time.Millisecond) + } + for time.Now().Before(deadline) { + if held, _, _ := h.holdOf(conn); !held { break } time.Sleep(2 * time.Millisecond) diff --git a/internal/daemon/outputhold_test.go b/internal/daemon/outputhold_test.go index e3813290..ce1ee148 100644 --- a/internal/daemon/outputhold_test.go +++ b/internal/daemon/outputhold_test.go @@ -434,6 +434,69 @@ func TestHold_FlushStraddlingTheReleaseArrivesOnce(t *testing.T) { } } +// The LOSS half of the release straddle: a flush that appends after the +// release saw its last, empty batch but before the hold ends. finishOutputHold +// must notice it and send it; ending the hold anyway would lose it, since the +// flush's broadcast skipped the still-held conn. +func TestHold_FlushAfterLastBatchBeforeFinishArrivesOnce(t *testing.T) { + h := newHoldHarness(t) + tab := h.d.session.CreateTab("T") + p := h.pane(tab.ID, "terminal") + client, conn := h.dial("B") + + var once sync.Once + h.d.beforeFinishHold = func(*ipc.Conn) { + once.Do(func() { h.d.flushPaneOutput(p.ID, []byte("s")) }) + } + h.d.beginOutputHold(conn) + h.d.releaseOutputHold(conn, nil) + if n := h.holdCount(); n != 0 { + t.Fatalf("%d holds left after release, want 0", n) + } + + h.d.flushPaneOutput(p.ID, []byte("Z")) + if got := string(paneData(t, readOutput(t, client, isChunk(p.ID, "Z")), p.ID)); got != "sZ" { + t.Fatalf("received %q, want %q", got, "sZ") + } +} + +// A conn that dies while its droppable queue is backed up never drains it, so +// beginOutputHold's wait must end on the close rather than sit out its bound. +func TestHold_BeginDrainAbortsOnConnClose(t *testing.T) { + h := newHoldHarness(t) + tab := h.d.session.CreateTab("T") + p := h.pane(tab.ID, "terminal") + client, conn := h.dial("B") + + // Nobody reads: the socket fills and the droppable queue backs up. + chunk := bytes.Repeat([]byte{'x'}, 8<<10) + for i := 0; conn.Dropped() == 0 && i < 4096; i++ { + h.d.flushPaneOutput(p.ID, chunk) + } + if conn.QueuedOutput() == 0 { + t.Fatal("setup: the droppable queue is empty") + } + + returned := make(chan time.Time, 1) + go func() { + h.d.beginOutputHold(conn) + returned <- time.Now() + }() + waitUntil(t, "the hold to begin", func() bool { held, _, _ := h.holdOf(conn); return held }) + time.Sleep(20 * time.Millisecond) // let the drain wait start + closedAt := time.Now() + client.Close() + + select { + case at := <-returned: + if waited := at.Sub(closedAt); waited > holdDrainTimeout/2 { + t.Fatalf("beginOutputHold returned %v after the close, want well under %v", waited, holdDrainTimeout) + } + case <-time.After(2 * holdDrainTimeout): + t.Fatal("beginOutputHold never returned") + } +} + // The same straddle as a hold begins: a flush that appended before the hold // existed must still reach the conn through its broadcast, not find the conn // held and be skipped with nothing held. From 95675116bd602abe914bb212fcfee44fff324daa Mon Sep 17 00:00:00 2001 From: Artjoms Stukans Date: Fri, 25 Sep 2026 17:50:30 +0200 Subject: [PATCH 07/40] fix(daemon): send the attached-client count when it changes The workspace state carries the attached-client count, and each TUI shows [master]/[follower] only while that count is 2 or more. But the other clients were sent a state only when the MASTER changed, so an attach, a detach or a lost link that left the master alone left the count stale on every other TUI: a second TUI attaching did not make the first one show its role. The registry now reports whether the count changed as well as the master. An attach of a new client, a detach and a lost link each send one state frame to the other attached clients, also when the master changed in the same event. Bridges never attach, so they change nothing. A disconnect during shutdown still leaves the registry alone. A same-id reattach inside the grace now sends the follower a state with the new count; it still never changes size_master. Part of #235 --- internal/daemon/clients.go | 51 ++++++++++++------ internal/daemon/clients_wiring_test.go | 68 ++++++++++++++++++++++++ internal/daemon/daemon.go | 14 ++--- internal/daemon/resize_authority_test.go | 25 +++++++-- 4 files changed, 132 insertions(+), 26 deletions(-) diff --git a/internal/daemon/clients.go b/internal/daemon/clients.go index 1a89c1d1..ba9de203 100644 --- a/internal/daemon/clients.go +++ b/internal/daemon/clients.go @@ -76,6 +76,16 @@ type reservation struct { protects map[string]bool } +// clientChange is what one attach, detach or lost link changed. The state +// carries both the master id and the attached-client count, so either change +// is news to the other clients. +type clientChange struct { + master bool // masterID changed + count bool // the number of attached clients changed +} + +func (c clientChange) any() bool { return c.master || c.count } + // clientRegistry is the set of attached clients plus the master bookkeeping. // // Its mutex is a LEAF: never sm.mu, never a pane's PluginMu, and never held @@ -229,12 +239,13 @@ func (r *clientRegistry) expire() { // attach records conn as the client id at a RAW geometry and elects. An empty // id is minted as "anon-", scoped to the conn: a re-attach on the same // conn keeps it. -func (r *clientRegistry) attach(conn *ipc.Conn, id string, cols, rows int, cwd string) bool { +func (r *clientRegistry) attach(conn *ipc.Conn, id string, cols, rows int, cwd string) clientChange { r.mu.Lock() defer r.mu.Unlock() if r.byConn == nil { r.byConn = make(map[*ipc.Conn]*clientRecord) } + before := len(r.byConn) rec, existed := r.byConn[conn] if id == "" { if existed { @@ -270,19 +281,19 @@ func (r *clientRegistry) attach(conn *ipc.Conn, id string, cols, rows int, cwd s // for. The election below keeps it when the client is eligible. r.clearReservationLocked() } - return r.electLocked() + return clientChange{master: r.electLocked(), count: len(r.byConn) != before} } // lose drops conn after a LOST link: the conn closed with no MsgDetach. A // master that leaves this way keeps its slot for the grace time, but only // while another client is attached. With nobody to protect, the grace would // only make a relaunched TUI (which has a new id) wait. -func (r *clientRegistry) lose(conn *ipc.Conn) bool { +func (r *clientRegistry) lose(conn *ipc.Conn) clientChange { r.mu.Lock() defer r.mu.Unlock() rec, ok := r.byConn[conn] if !ok { - return false + return clientChange{} } delete(r.byConn, conn) if rec.id == r.masterID && r.grace > 0 && r.recordByID(rec.id) == nil { @@ -299,19 +310,19 @@ func (r *clientRegistry) lose(conn *ipc.Conn) bool { }, r.grace) } } - return r.electLocked() + return clientChange{master: r.electLocked(), count: true} } // detach drops conn after a clean exit and elects with NO reservation. The // disconnect that follows finds no record and does nothing more. -func (r *clientRegistry) detach(conn *ipc.Conn) bool { +func (r *clientRegistry) detach(conn *ipc.Conn) clientChange { r.mu.Lock() defer r.mu.Unlock() if _, ok := r.byConn[conn]; !ok { - return false + return clientChange{} } delete(r.byConn, conn) - return r.electLocked() + return clientChange{master: r.electLocked(), count: true} } // setGeometry records a client's new RAW window size and elects: a master @@ -373,8 +384,14 @@ func (r *clientRegistry) reserveAfterRestart(id string) { // The geometry is the RAW one from the payload, taken before handleAttach // defaults it. It returns whether the master changed. func (d *Daemon) registerClient(conn *ipc.Conn, attach ipc.AttachPayload) bool { + return d.attachClient(conn, attach).master +} + +// attachClient is registerClient reporting the count change too, for +// handleAttach. +func (d *Daemon) attachClient(conn *ipc.Conn, attach ipc.AttachPayload) clientChange { if conn == nil { - return false + return clientChange{} } id := truncateField(attach.ClientID, maxClientIDLen) return d.clients.attach(conn, id, clampClientDim(attach.Cols), clampClientDim(attach.Rows), attach.CWD) @@ -392,13 +409,13 @@ func clampClientDim(v int) int { // already sent MsgDetach, is not in the set, so dropping it changes nothing. // It returns whether the master changed. func (d *Daemon) forgetAttachedClient(conn *ipc.Conn) bool { - return d.clients.lose(conn) + return d.clients.lose(conn).master } // detachClient handles a clean client exit. It returns whether the master // changed. func (d *Daemon) detachClient(conn *ipc.Conn) bool { - return d.clients.detach(conn) + return d.clients.detach(conn).master } // setClientGeometry records a client's RAW window size. It returns whether the @@ -428,12 +445,14 @@ func (d *Daemon) shuttingDown() bool { } // sendStateToOtherClients sends the workspace state to every attached client -// except the one given, for a master change an attach made. +// except the one given, after an attach, detach or lost link changed the +// master or the attached-client count. Each TUI shows [master]/[follower] only +// while that count is 2 or more, so a count left stale is a wrong status bar. // // Not a broadcast. A broadcast also reached the attaching conn, which then got // two state frames back to back: its own attach state and this one, which says // nothing new. That is pressure on its must-deliver queue for no information. -// A conn that never attached (an MCP bridge) has no use for the master either. +// A conn that never attached (an MCP bridge) has no use for either value. func (d *Daemon) sendStateToOtherClients(except *ipc.Conn) { d.clients.mu.Lock() var conns []*ipc.Conn @@ -456,9 +475,11 @@ func (d *Daemon) sendStateToOtherClients(except *ipc.Conn) { } } +// handleDetach removes a cleanly exiting client. The detach always changes +// the attached count, so the other clients always get one state frame. func (d *Daemon) handleDetach(conn *ipc.Conn) { - if d.detachClient(conn) { - d.broadcastState() + if d.clients.detach(conn).any() { + d.sendStateToOtherClients(conn) } } diff --git a/internal/daemon/clients_wiring_test.go b/internal/daemon/clients_wiring_test.go index 4d8d4323..d0e16ad4 100644 --- a/internal/daemon/clients_wiring_test.go +++ b/internal/daemon/clients_wiring_test.go @@ -1,6 +1,7 @@ package daemon import ( + "fmt" "testing" "time" @@ -134,6 +135,73 @@ func TestClientDispatch_AttachMasterChangeReachesOthersOnly(t *testing.T) { } } +// stateWith matches a workspace_state carrying this attached count. +func stateWith(clients int) func(*ipc.Message) bool { + return func(m *ipc.Message) bool { + if m.Type != ipc.MsgWorkspaceState { + return false + } + var s map[string]any + return m.DecodePayload(&s) == nil && s["clients"] == float64(clients) + } +} + +// wantOneCountFrame reads c until a state with this count arrives, checks it +// kept the master, and checks no second state frame follows. +func wantOneCountFrame(t *testing.T, c *ipc.Client, who string, clients int, master string) { + t.Helper() + got := readUntil(t, c, fmt.Sprintf("%s: a state with clients=%d", who, clients), stateWith(clients)) + var s map[string]any + if err := got[len(got)-1].DecodePayload(&s); err != nil { + t.Fatalf("%s: decode state: %v", who, err) + } + if s["size_master"] != master { + t.Errorf("%s: size_master = %v, want %q unchanged", who, s["size_master"], master) + } + if n := countType(readFor(c, 200*time.Millisecond), ipc.MsgWorkspaceState); n != 0 { + t.Errorf("%s: %d more workspace_state frames after the count change, want 1 frame per event", who, n) + } +} + +// The attached count rides the state, and each TUI shows [master]/[follower] +// only while it is 2 or more. So an attach, a detach or a lost link that leaves +// the master alone must still reach the other clients, once each. A bridge, +// which never attaches, is not a client and changes nothing. +func TestClientDispatch_CountChangeReachesOtherClients(t *testing.T) { + d, sock, _ := resizeAuthorityDaemon(t) + a, b := attachAB(t, d, sock) + readUntil(t, b, "B's attach state", isType(ipc.MsgWorkspaceState)) + readUntil(t, a, "A: B's arrival", stateWith(2)) + + c := attachClientAs(t, sock, "C", 100, 30) + readUntil(t, c, "C's attach state", isType(ipc.MsgWorkspaceState)) + wantOneCountFrame(t, a, "A after C attached", 3, "A") + wantOneCountFrame(t, b, "B after C attached", 3, "A") + + sendClientMsg(t, c, ipc.MsgDetach, nil) + wantOneCountFrame(t, a, "A after C detached", 2, "A") + wantOneCountFrame(t, b, "B after C detached", 2, "A") + + lost := attachClientAs(t, sock, "L", 100, 30) + readUntil(t, a, "A: L's arrival", stateWith(3)) + readUntil(t, b, "B: L's arrival", stateWith(3)) + lost.Close() + wantOneCountFrame(t, a, "A after L's link dropped", 2, "A") + wantOneCountFrame(t, b, "B after L's link dropped", 2, "A") + + bridge, err := ipc.NewClient(sock) + if err != nil { + t.Fatalf("dial bridge: %v", err) + } + sendClientMsg(t, bridge, ipc.MsgClientHello, ipc.ClientHelloPayload{Role: "bridge", PID: 1}) + bridge.Close() + for who, cl := range map[string]*ipc.Client{"A": a, "B": b} { + if n := countType(readFor(cl, 300*time.Millisecond), ipc.MsgWorkspaceState); n != 0 { + t.Errorf("%s received %d workspace_state frames for a bridge coming and going, want 0", who, n) + } + } +} + // A lost link through the real server keeps the master's slot while a // follower is attached. func TestClientDispatch_LostLinkReservesSlot(t *testing.T) { diff --git a/internal/daemon/daemon.go b/internal/daemon/daemon.go index 53830679..36818040 100644 --- a/internal/daemon/daemon.go +++ b/internal/daemon/daemon.go @@ -640,8 +640,10 @@ func (d *Daemon) onClientDisconnect(conn *ipc.Conn) { d.dropOutputHold(conn) d.requestSnapshot() d.events.RemoveWatchersByConn(conn) - if !d.shuttingDown() && d.forgetAttachedClient(conn) { - d.broadcastState() + // A lost link always changes the attached count, whether or not the + // master changed with it, so the other clients get one state frame. + if !d.shuttingDown() && d.clients.lose(conn).any() { + d.sendStateToOtherClients(conn) } // Drop this conn's identity with it: the process it described is gone, // and a retained entry would be listed as running. @@ -1800,10 +1802,10 @@ func (d *Daemon) handleAttach(conn *ipc.Conn, msg *ipc.Message) { // any of the work below, which has early returns of its own, and before // the 80x24 defaulting: the election reads the RAW geometry. // - // A master change reaches the OTHER attached clients once this attach is - // answered. The attaching conn gets none: its own state below is built - // after this registration, so it already names the new master. - if d.registerClient(conn, attach) { + // A new client or a master change reaches the OTHER attached clients once + // this attach is answered. The attaching conn gets none: its own state + // below is built after this registration, so it already carries both. + if d.attachClient(conn, attach).any() { defer d.sendStateToOtherClients(conn) } diff --git a/internal/daemon/resize_authority_test.go b/internal/daemon/resize_authority_test.go index a757b299..aee41c9f 100644 --- a/internal/daemon/resize_authority_test.go +++ b/internal/daemon/resize_authority_test.go @@ -443,9 +443,10 @@ func TestWorkspaceState_CarriesSizeMasterAndClients(t *testing.T) { } // A master change reaches every client as a broadcast carrying the new -// size_master. A same-id reattach inside the grace changes nothing, so it -// costs the follower no frame. -func TestMasterChange_IsBroadcastAndReattachIsNot(t *testing.T) { +// size_master. A same-id reattach inside the grace never changes size_master: +// the follower's frames for the lost link and the reattach carry only the new +// attached count. +func TestMasterChange_IsBroadcastAndReattachKeepsMaster(t *testing.T) { d, sock, _ := resizeAuthorityDaemon(t) a, b := attachAB(t, d, sock) readUntil(t, b, "B's attach state", isType(ipc.MsgWorkspaceState)) @@ -476,8 +477,22 @@ func TestMasterChange_IsBroadcastAndReattachIsNot(t *testing.T) { a2 := attachClientAs(t, sock, "A", 200, 50) readUntil(t, a2, "A's own attach state", isType(ipc.MsgWorkspaceState)) - if n := countType(readFor(b, 300*time.Millisecond), ipc.MsgWorkspaceState); n != 0 { - t.Errorf("B received %d workspace_state frames for a same-id reattach, want 0", n) + var last map[string]any + for _, m := range readFor(b, 300*time.Millisecond) { + if m.Type != ipc.MsgWorkspaceState { + continue + } + var s map[string]any + if err := m.DecodePayload(&s); err != nil { + t.Fatalf("decode state: %v", err) + } + if s["size_master"] != "A" { + t.Errorf("B received size_master = %v around a same-id reattach, want A throughout", s["size_master"]) + } + last = s + } + if last == nil || last["clients"] != float64(2) { + t.Errorf("B's last state = %v, want one carrying clients = 2 after the reattach", last) } if d.masterID() != "A" { t.Errorf("masterID = %q after the reattach, want A", d.masterID()) From 2a6ad62a6d0e5e9ecd34522efcfbfcc0903044d1 Mon Sep 17 00:00:00 2001 From: Artjoms Stukans Date: Fri, 25 Sep 2026 18:20:39 +0200 Subject: [PATCH 08/40] feat(daemon): version tab layouts and broadcast accepted changes Give each tab a LayoutRev counter and gate SetTabLayout on it: a write's BaseRev must match the tab's current revision (or be absent, for an older client) or it is refused outright, with no change to either the layout or the revision. An accepted write bumps the revision, persists in workspace.json, and is broadcast so every other attached client can adopt it. The broadcast is coalesced (broadcast_coalesce.go): a burst of accepted writes - several tabs re-sent after a split-drag release, or several clients editing at once - produces exactly one workspace-state frame at the end of a 50ms window instead of one per write, which is the queue-pressure shape multi-client sync has to avoid on every broadcast path. This also removes the old "no broadcast, to avoid a feedback loop" restriction: with per-client revision tracking a client only ever writes after its own change and only adopts a broadcast that is strictly newer than what it already holds, so echoing an accepted write back to its own sender cannot make it send again. --- internal/daemon/broadcast_coalesce.go | 63 +++++++ internal/daemon/daemon.go | 40 ++++- internal/daemon/layout_rev_test.go | 228 ++++++++++++++++++++++++++ internal/daemon/session.go | 30 +++- 4 files changed, 352 insertions(+), 9 deletions(-) create mode 100644 internal/daemon/broadcast_coalesce.go create mode 100644 internal/daemon/layout_rev_test.go diff --git a/internal/daemon/broadcast_coalesce.go b/internal/daemon/broadcast_coalesce.go new file mode 100644 index 00000000..df77ace1 --- /dev/null +++ b/internal/daemon/broadcast_coalesce.go @@ -0,0 +1,63 @@ +package daemon + +import "time" + +// broadcastCoalesceWindow is requestBroadcast's trailing-edge coalescing +// window. Every call that lands within this window of the one that opened it +// rides the same timer; the broadcast fires once, at the end of the window, +// not once per call. +const broadcastCoalesceWindow = 50 * time.Millisecond + +// requestBroadcast coalesces a burst of accepted layout writes into one +// broadcastState() call. A burst of SetTabLayout acceptances — several tabs +// re-sent after a split-drag release, or several clients editing at once — +// would otherwise put one full workspace-state frame per call onto every +// attached client's must-deliver queue, which is the documented 2026-08-09 +// critical-queue shape (see "A per-pane broadcast is a queue-pressure +// decision, not a detail" in daemon-lifecycle.md). +// +// It is single-flighted under broadcastMu: the first call in a burst arms a +// timer and every call arriving while that timer is still pending returns +// immediately, folded into the same eventual broadcast for free. +func (d *Daemon) requestBroadcast() { + d.broadcastMu.Lock() + defer d.broadcastMu.Unlock() + if d.broadcastTimerStop != nil { + // A window is already open — this call rides it rather than arming + // a second timer. + return + } + after := d.broadcastAfterFn + if after == nil { + after = realAfterFunc + } + d.broadcastTimerStop = after(broadcastCoalesceWindow, d.fireCoalescedBroadcast) +} + +// fireCoalescedBroadcast is requestBroadcast's timer callback. It clears the +// in-flight marker BEFORE calling broadcastState, so a requestBroadcast call +// racing this one opens a fresh window instead of silently riding a timer +// that has already fired. +func (d *Daemon) fireCoalescedBroadcast() { + d.broadcastMu.Lock() + d.broadcastTimerStop = nil + d.broadcastMu.Unlock() + // broadcastState nil-guards on d.server itself, which covers a Daemon + // built for tests with no server at all — that guard is what makes this + // safe to fire after Stop() has already run everything except the final + // server teardown ordering. stopBroadcastCoalescer below is what stops it + // from firing AT ALL once shutdown has begun disarming timers. + d.broadcastState() +} + +// stopBroadcastCoalescer disarms a pending coalesced broadcast at daemon +// shutdown, so it never fires into a stopped daemon. Mirrors +// clientRegistry.stopTimer, which the same Stop() call disarms alongside it. +func (d *Daemon) stopBroadcastCoalescer() { + d.broadcastMu.Lock() + defer d.broadcastMu.Unlock() + if d.broadcastTimerStop != nil { + d.broadcastTimerStop() + d.broadcastTimerStop = nil + } +} diff --git a/internal/daemon/daemon.go b/internal/daemon/daemon.go index 36818040..a46b955b 100644 --- a/internal/daemon/daemon.go +++ b/internal/daemon/daemon.go @@ -275,6 +275,17 @@ type Daemon struct { // be called while it is held, and nothing broadcasts while it is held. clients clientRegistry + // broadcastMu guards the single in-flight timer requestBroadcast arms + // (broadcast_coalesce.go), coalescing a burst of accepted layout writes + // into one broadcastState() call. broadcastAfterFn is the same seam + // shape as clientRegistry's afterFn — real time.AfterFunc from New(), a + // fake that fires by hand in tests — kept as its own field rather than + // reused because the two timers are armed and stopped independently + // (Stop() calls both stopBroadcastCoalescer() and clients.stopTimer()). + broadcastMu sync.Mutex + broadcastTimerStop func() bool + broadcastAfterFn func(time.Duration, func()) (stop func() bool) + // holds keeps each attaching conn's live pane output while its replay is // sent, keyed by conn (outputhold.go). holdMu is a leaf guarding it; // holdGate orders a flush's hold append and broadcast against a conn's @@ -331,6 +342,7 @@ func New(cfg config.Config) *Daemon { d.clients.afterFn = realAfterFunc d.clients.grace = cfg.Daemon.MasterGrace() d.clients.onChange = d.broadcastState + d.broadcastAfterFn = realAfterFunc d.startedAt = time.Now() // Clamped like a pushed policy: config.toml is hand-edited, so it can carry // exactly the values the IPC path is bounded against. @@ -595,6 +607,7 @@ func (d *Daemon) Stop() { } // No client can attach any more, so nothing can re-arm it. d.clients.stopTimer() + d.stopBroadcastCoalescer() d.collectorWG.Wait() // Pull the latest hook-recorded session ids into PluginState so // the final snapshot survives even if the hook files are lost. @@ -897,6 +910,16 @@ func (d *Daemon) restoreWorkspace() error { } } + // Restore layout revision. Absent means 0 — a workspace.json written + // before layout revisioning existed, or a tab that never had a + // layout write. JSON numbers decode to float64 through this + // map[string]any, never a uint64 directly. + if revRaw, ok := tabMap["layout_rev"]; ok { + if rev, ok := revRaw.(float64); ok && rev >= 0 { + tab.LayoutRev = uint64(rev) + } + } + // Restore panes for this tab var tabPanes []*Pane if paneIDs, ok := tabMap["panes"].([]any); ok { @@ -4014,12 +4037,18 @@ func (d *Daemon) handleUpdateLayout(msg *ipc.Message) { // Under sm.mu: SnapshotState copies Layout under the same lock, and // MovePane reads it there for its template check. - if !d.session.SetTabLayout(payload.TabID, payload.Layout) { + if !d.session.SetTabLayout(payload.TabID, payload.Layout, payload.BaseRev) { + logger.Debug("update_layout: refused tab=%s (unknown tab or stale base_rev)", payload.TabID) return } - // No broadcastState() — avoids feedback loop. - // Snapshot ensures layout is persisted to disk. + // Every client now sends an update only after ITS OWN change and adopts + // a broadcast whose layout_rev is newer than what it holds (spec §7.1), + // so echoing this accepted write back to the sender cannot make it send + // again — the feedback loop this comment used to warn about is + // structurally impossible now, not merely avoided by omission. + // requestBroadcast coalesces a burst of these (see broadcast_coalesce.go). d.requestSnapshot() + d.requestBroadcast() } // resizeKick re-applies a pane's last known size to its PTY, with a @@ -4546,6 +4575,11 @@ func (d *Daemon) workspaceStateFromSnapshot(activeTab string, tabs []*Tab, panes "color": tab.Color, "panes": paneIDs, "project_id": tab.ProjectID, + // Unconditional, unlike "layout" below: every client compares + // this against its own copy on every broadcast to decide whether + // to adopt (spec §7.1), including a tab whose layout has never + // been written, so it must be on the wire even at its zero value. + "layout_rev": tab.LayoutRev, } if len(tab.Layout) > 0 { tabData["layout"] = tab.Layout diff --git a/internal/daemon/layout_rev_test.go b/internal/daemon/layout_rev_test.go new file mode 100644 index 00000000..f77e6615 --- /dev/null +++ b/internal/daemon/layout_rev_test.go @@ -0,0 +1,228 @@ +package daemon + +import ( + "encoding/json" + "testing" + "time" + + "github.com/artyomsv/quil/internal/config" + "github.com/artyomsv/quil/internal/ipc" +) + +// TestSetTabLayout_CAS pins the compare-and-store: a matching base (including +// the special case of no base at all) accepts and bumps the revision; a stale +// base is refused with NO write, to either the layout or the revision. +func TestSetTabLayout_CAS(t *testing.T) { + d := newTestDaemon(t) + tab := d.session.CreateTab("layout") + + // nil base: accepted unconditionally (an older client, or the tab's very + // first write, for which nothing has a base to send). + if !d.session.SetTabLayout(tab.ID, json.RawMessage(`{"v":0}`), nil) { + t.Fatal("nil base should be accepted") + } + if got := d.session.Tab(tab.ID).LayoutRev; got != 1 { + t.Fatalf("LayoutRev after nil-base write = %d, want 1", got) + } + + // Current base: accepted, revision bumps again. + cur := uint64(1) + if !d.session.SetTabLayout(tab.ID, json.RawMessage(`{"v":1}`), &cur) { + t.Fatal("a base matching the current revision should be accepted") + } + if got := d.session.Tab(tab.ID).LayoutRev; got != 2 { + t.Fatalf("LayoutRev after matching-base write = %d, want 2", got) + } + + // Stale base: refused. Neither the layout nor the revision moves. + beforeLayout := string(d.session.Tab(tab.ID).Layout) + stale := uint64(0) + if d.session.SetTabLayout(tab.ID, json.RawMessage(`{"v":99}`), &stale) { + t.Fatal("a stale base should be refused") + } + after := d.session.Tab(tab.ID) + if after.LayoutRev != 2 { + t.Errorf("LayoutRev after a refused write = %d, want unchanged 2", after.LayoutRev) + } + if string(after.Layout) != beforeLayout { + t.Errorf("layout changed after a refused write: got %s, want unchanged %s", after.Layout, beforeLayout) + } + + // Unknown tab: refused. + if d.session.SetTabLayout("tab-deadbeef", json.RawMessage(`{}`), nil) { + t.Error("an unknown tab should be refused") + } +} + +// callUpdateLayout builds and dispatches an UpdateLayoutPayload directly at +// handleUpdateLayout, the way the movepane_race_test.go race tests already +// drive it — the PRODUCTION handler, not SetTabLayout in isolation. +func callUpdateLayout(t *testing.T, d *Daemon, tabID string, layout json.RawMessage, baseRev *uint64) { + t.Helper() + msg, err := ipc.NewMessage(ipc.MsgUpdateLayout, ipc.UpdateLayoutPayload{TabID: tabID, Layout: layout, BaseRev: baseRev}) + if err != nil { + t.Fatalf("NewMessage: %v", err) + } + d.handleUpdateLayout(msg) +} + +// fakeCoalesceTimer is one time.AfterFunc requestBroadcast armed. Tests fire +// it by hand instead of waiting on the real 50ms window — the same pattern +// clientsHarness.afterFn uses in clients_test.go. +type fakeCoalesceTimer struct { + f func() + stopped bool +} + +// stubBroadcastTimer points d's coalescer at a fake afterFn and returns the +// list it appends every arm to. One arm corresponds to exactly one eventual +// fireCoalescedBroadcast → broadcastState() call (that 1:1 pairing is itself +// pinned by TestRequestBroadcast_CoalescesBurst), so counting arms is +// counting coalesced broadcasts. +func stubBroadcastTimer(d *Daemon) *[]*fakeCoalesceTimer { + timers := &[]*fakeCoalesceTimer{} + d.broadcastAfterFn = func(_ time.Duration, f func()) func() bool { + tm := &fakeCoalesceTimer{f: f} + *timers = append(*timers, tm) + return func() bool { + was := !tm.stopped + tm.stopped = true + return was + } + } + return timers +} + +// TestHandleUpdateLayout_BroadcastsOnAcceptOnly: an accepted write arms a +// coalesced broadcast; a refused one (stale base_rev) never calls +// requestBroadcast at all, so no new window opens while the first is still +// pending. +func TestHandleUpdateLayout_BroadcastsOnAcceptOnly(t *testing.T) { + d := newTestDaemon(t) + tab := d.session.CreateTab("layout") + timers := stubBroadcastTimer(d) + + // Accepted (nil base): should arm exactly one coalescer window. + callUpdateLayout(t, d, tab.ID, json.RawMessage(`{"v":1}`), nil) + if len(*timers) != 1 { + t.Fatalf("arms after accepted write = %d, want 1", len(*timers)) + } + + // Refused (stale base, current revision is now 1): must not arm a + // second window — SetTabLayout never even calls requestBroadcast. + stale := uint64(0) + callUpdateLayout(t, d, tab.ID, json.RawMessage(`{"v":2}`), &stale) + if len(*timers) != 1 { + t.Fatalf("arms after a refused write = %d, want still 1", len(*timers)) + } + + // Fire the pending window by hand, then confirm the layout the ACCEPTED + // call wrote is what is live — the refused call above never touched it. + (*timers)[0].f() + if got := string(d.session.Tab(tab.ID).Layout); got != `{"v":1}` { + t.Errorf("live layout = %s, want the accepted write's {\"v\":1}", got) + } + + // A fresh accepted write after the window closed opens a NEW window. + rev := uint64(1) + callUpdateLayout(t, d, tab.ID, json.RawMessage(`{"v":3}`), &rev) + if len(*timers) != 2 { + t.Fatalf("arms after a post-fire accepted write = %d, want 2", len(*timers)) + } +} + +// TestLayoutRev_SurvivesSnapshotRoundTrip: the revision a tab reaches through +// several accepted writes is exactly what a fresh daemon reads back after +// snapshot -> restore, not reset to 0 and not an off-by-one. +func TestLayoutRev_SurvivesSnapshotRoundTrip(t *testing.T) { + tmp := t.TempDir() + t.Setenv("QUIL_HOME", tmp) + + d := New(config.Default()) + tab := d.session.CreateTab("layout") + if !d.session.SetTabLayout(tab.ID, json.RawMessage(`{"v":1}`), nil) { + t.Fatal("first SetTabLayout should be accepted") + } + rev := uint64(1) + if !d.session.SetTabLayout(tab.ID, json.RawMessage(`{"v":2}`), &rev) { + t.Fatal("second SetTabLayout should be accepted") + } + wantRev := d.session.Tab(tab.ID).LayoutRev + if wantRev != 2 { + t.Fatalf("setup: LayoutRev = %d, want 2", wantRev) + } + + d.snapshot() + + d2 := New(config.Default()) + if err := d2.restoreWorkspace(); err != nil { + t.Fatalf("restoreWorkspace: %v", err) + } + restored := d2.session.Tab(tab.ID) + if restored == nil { + t.Fatalf("tab %s was not restored", tab.ID) + } + if restored.LayoutRev != wantRev { + t.Errorf("restored LayoutRev = %d, want %d", restored.LayoutRev, wantRev) + } +} + +// TestRequestBroadcast_CoalescesBurst: 5 calls inside the coalescing window +// give exactly 1 armed timer, and firing it starts the cycle over. +func TestRequestBroadcast_CoalescesBurst(t *testing.T) { + d := &Daemon{} + timers := stubBroadcastTimer(d) + + for i := 0; i < 5; i++ { + d.requestBroadcast() + } + if len(*timers) != 1 { + t.Fatalf("arms after a burst of 5 = %d, want 1", len(*timers)) + } + + // Firing clears the in-flight marker, so the NEXT call opens a fresh + // window rather than being folded into the one that just fired. + (*timers)[0].f() + d.requestBroadcast() + if len(*timers) != 2 { + t.Fatalf("arms after firing + one more call = %d, want 2", len(*timers)) + } +} + +// TestWorkspaceState_TabCarriesLayoutRev: layout_rev is on the wire for every +// tab, including one that has never had a layout written (rev 0), and it +// tracks accepted writes. +func TestWorkspaceState_TabCarriesLayoutRev(t *testing.T) { + d := newTestDaemon(t) + tab := d.session.CreateTab("layout") + + findTab := func(state map[string]any) map[string]any { + t.Helper() + tabsOut, _ := state["tabs"].([]map[string]any) + for _, tb := range tabsOut { + if tb["id"] == tab.ID { + return tb + } + } + t.Fatalf("tab %s missing from workspace state", tab.ID) + return nil + } + + tb := findTab(d.buildWorkspaceState()) + rev, ok := tb["layout_rev"] + if !ok { + t.Fatal("layout_rev missing from a tab that has never had a layout written") + } + if rev != uint64(0) { + t.Errorf("layout_rev = %v, want 0", rev) + } + + if !d.session.SetTabLayout(tab.ID, json.RawMessage(`{"v":1}`), nil) { + t.Fatal("SetTabLayout should be accepted") + } + tb = findTab(d.buildWorkspaceState()) + rev, ok = tb["layout_rev"] + if !ok || rev != uint64(1) { + t.Errorf("layout_rev after one accepted write = %v (present=%v), want 1", rev, ok) + } +} diff --git a/internal/daemon/session.go b/internal/daemon/session.go index 77eef2ab..b604a171 100644 --- a/internal/daemon/session.go +++ b/internal/daemon/session.go @@ -25,7 +25,16 @@ type Tab struct { Color string Panes []string // Pane IDs in order Layout json.RawMessage // Opaque layout tree from TUI - ProjectID string // Project this tab belongs to (see project.go) + // LayoutRev is bumped on every accepted SetTabLayout. It is the + // compare-and-store base a client sends back on its NEXT write + // (UpdateLayoutPayload.BaseRev) and the value every client compares its + // own copy against on a broadcast, to tell an update it should adopt from + // one it should ignore (spec §7.1). Zero both for a tab that has never + // had a layout written and for one restored from a workspace.json + // written before this field existed — the two are indistinguishable and + // that is fine, since both start the CAS from the same place. + LayoutRev uint64 + ProjectID string // Project this tab belongs to (see project.go) } type Pane struct { @@ -1090,18 +1099,27 @@ func (sm *SessionManager) UpdateTab(tabID, name, color string, clearColor bool) return true } -// SetTabLayout replaces a tab's opaque layout under sm.mu. False for an -// unknown tab. handleUpdateLayout used to write tab.Layout through the live -// pointer with no lock, racing SnapshotState's copy — the handleUpdateTab -// shape #229 fixed. -func (sm *SessionManager) SetTabLayout(tabID string, layout json.RawMessage) bool { +// SetTabLayout replaces a tab's opaque layout under sm.mu, gated by a +// compare-and-store on baseRev: nil accepts unconditionally (an older client, +// or one that has not adopted revisions yet), a value equal to the tab's +// current LayoutRev accepts, and anything else is refused with NO write at +// all — not to Layout, not to LayoutRev. False also for an unknown tab. An +// accepted write bumps LayoutRev, which is the new base the caller's NEXT +// update carries. handleUpdateLayout used to write tab.Layout through the +// live pointer with no lock, racing SnapshotState's copy — the +// handleUpdateTab shape #229 fixed. +func (sm *SessionManager) SetTabLayout(tabID string, layout json.RawMessage, baseRev *uint64) bool { sm.mu.Lock() defer sm.mu.Unlock() tab, ok := sm.tabs[tabID] if !ok { return false } + if baseRev != nil && *baseRev != tab.LayoutRev { + return false + } tab.Layout = layout + tab.LayoutRev++ return true } From 644c4066f072324f5e135733bbb5183a0af351ea Mon Sep 17 00:00:00 2001 From: Artjoms Stukans Date: Fri, 25 Sep 2026 19:09:44 +0200 Subject: [PATCH 09/40] feat(daemon): target MCP focus and close at one client set_active_pane and close_tui used to broadcast to every attached TUI, which was fine before several TUIs could share a daemon and became "steal another window's focus" or "close somebody else's window" the moment they could. Both now resolve a target conn: an explicit client id from the new list_clients MCP tool, or, implicitly, the client with the most recent input (falling back to the oldest attached client when nobody has typed yet). A headless daemon with no attached client drops the command with a log line instead of sending anything. defaultCWD is now per-client: each attached client's own directory is tried first, then the size master's, then the most-recently-active client's, before falling back to the daemon's own working directory. The old single clientCWD field could not express "which client is asking" once more than one TUI could be attached at once. Dismissing a notification and clearing a pane's unseen mark now broadcast to every attached client, so acting on either in one TUI's sidebar is reflected in a second one instead of leaving a stale card or mark behind. update_pane's automatic reports (an OSC 7 CWD change, overlay visibility, the unseen mark) no longer count as user input for picking the implicit MCP target - only a field the user actually touched does. --- cmd/quil/mcp.go | 2 + cmd/quil/mcp_tools.go | 55 ++- cmd/quil/mcp_tools_router_test.go | 10 +- cmd/quil/mcp_tools_test.go | 153 +++++++++ cmd/quil/mcp_version.go | 7 + internal/daemon/browse.go | 2 +- internal/daemon/browse_test.go | 2 +- internal/daemon/create_req.go | 4 +- internal/daemon/daemon.go | 173 +++++++--- internal/daemon/daemon_test.go | 63 +--- internal/daemon/discover.go | 2 +- internal/daemon/mcp_targets.go | 91 +++++ internal/daemon/mcp_targets_test.go | 422 ++++++++++++++++++++++++ internal/daemon/pane_cleanup_test.go | 2 +- internal/daemon/project_req.go | 4 +- internal/daemon/project_test.go | 10 +- internal/daemon/worktree.go | 2 +- internal/daemon/worktree_remove_test.go | 2 +- 18 files changed, 888 insertions(+), 118 deletions(-) create mode 100644 internal/daemon/mcp_targets.go create mode 100644 internal/daemon/mcp_targets_test.go diff --git a/cmd/quil/mcp.go b/cmd/quil/mcp.go index 7a99d8c5..09d1cbb3 100644 --- a/cmd/quil/mcp.go +++ b/cmd/quil/mcp.go @@ -260,6 +260,8 @@ const mcpInstructions = "Quil is a terminal multiplexer with projects, tabs and "- Projects: list_projects, create_project, update_project, switch_project, destroy_project; tabs: create_tab, rename_tab, destroy_tab.\n" + "- Hosts: list_hosts shows the remote daemons this bridge reaches. Ids you discovered route to their host automatically; " + "pass host explicitly to create things on a remote. delegate_task with notify only works when requester and target share a host.\n" + + "- Several TUIs can share one daemon: list_clients shows them, and its client id targets set_active_pane or close_tui at " + + "one of them instead of the one that typed most recently.\n" + "- Destructive tools (restart_pane, destroy_pane, destroy_tab, destroy_project, close_tui): always confirm with the user before using.\n" + "- watch_notifications: blocks until an event fires on specified panes (replaces polling). Use after starting long-running tasks.\n" + "- get_notifications: returns all pending notification events without blocking.\n" + diff --git a/cmd/quil/mcp_tools.go b/cmd/quil/mcp_tools.go index bfddb422..4d6a124c 100644 --- a/cmd/quil/mcp_tools.go +++ b/cmd/quil/mcp_tools.go @@ -32,6 +32,7 @@ func registerMCPTools(s *mcp.Server, r *mcpRouter, mcpLog *mcpLogger) { registerDestroyPaneTool(s, r, mcpLog) registerSetActivePaneTool(s, r, mcpLog) registerCloseTUITool(s, r, mcpLog) + registerListClientsTool(s, r, mcpLog) // Notification tools registerGetNotificationsTool(s, r, mcpLog) registerWatchNotificationsTool(s, r, mcpLog) @@ -625,6 +626,7 @@ func registerDestroyPaneTool(s *mcp.Server, r *mcpRouter, mcpLog *mcpLogger) { func registerSetActivePaneTool(s *mcp.Server, r *mcpRouter, mcpLog *mcpLogger) { type Input struct { PaneID string `json:"pane_id" jsonschema:"pane to focus (switches tab if needed)"` + Client string `json:"client,omitempty" jsonschema:"optional client id from list_clients; default: the client with the most recent input"` Host string `json:"host,omitempty" jsonschema:"daemon host from list_hosts (empty = the host the id was discovered on, else local)"` } @@ -636,7 +638,7 @@ func registerSetActivePaneTool(s *mcp.Server, r *mcpRouter, mcpLog *mcpLogger) { if err != nil { return nil, nil, fmt.Errorf("set_active_pane: %w", err) } - msg, err := ipc.NewMessage(ipc.MsgSetActivePane, ipc.SetActivePanePayload{PaneID: input.PaneID}) + msg, err := ipc.NewMessage(ipc.MsgSetActivePane, ipc.SetActivePanePayload{PaneID: input.PaneID, Client: input.Client}) if err != nil { return nil, nil, fmt.Errorf("set_active_pane: %w", err) } @@ -650,7 +652,8 @@ func registerSetActivePaneTool(s *mcp.Server, r *mcpRouter, mcpLog *mcpLogger) { func registerCloseTUITool(s *mcp.Server, r *mcpRouter, mcpLog *mcpLogger) { type Input struct { - Host string `json:"host,omitempty" jsonschema:"daemon host from list_hosts (default: local)"` + Client string `json:"client,omitempty" jsonschema:"optional client id from list_clients; default: the client with the most recent input"` + Host string `json:"host,omitempty" jsonschema:"daemon host from list_hosts (default: local)"` } mcp.AddTool(s, &mcp.Tool{ @@ -662,7 +665,7 @@ func registerCloseTUITool(s *mcp.Server, r *mcpRouter, mcpLog *mcpLogger) { if err != nil { return nil, nil, fmt.Errorf("close_tui: %w", err) } - msg, err := ipc.NewMessage(ipc.MsgCloseTUI, nil) + msg, err := ipc.NewMessage(ipc.MsgCloseTUI, ipc.CloseTUIPayload{Client: input.Client}) if err != nil { return nil, nil, fmt.Errorf("close_tui: %w", err) } @@ -673,6 +676,52 @@ func registerCloseTUITool(s *mcp.Server, r *mcpRouter, mcpLog *mcpLogger) { }) } +// registerListClientsTool lists every attached client (TUI or MCP bridge) +// so an agent can target one explicitly with set_active_pane or close_tui +// instead of landing on the implicit most-recently-active one. +func registerListClientsTool(s *mcp.Server, r *mcpRouter, mcpLog *mcpLogger) { + type Input struct { + Host string `json:"host,omitempty" jsonschema:"limit to one host (default: every connected host)"` + } + type hostedClient struct { + ipc.ClientInfo + Host string `json:"host,omitempty"` + } + + mcp.AddTool(s, &mcp.Tool{ + Name: "list_clients", + Description: "List every attached TUI/bridge client: id, attach time, window size, whether it holds size master " + + "(the client whose geometry sizes every pane), and its last input time. Pass a client id to set_active_pane " + + "or close_tui to target that client instead of the one that typed most recently.", + }, func(_ context.Context, _ *mcp.CallToolRequest, input Input) (*mcp.CallToolResult, any, error) { + var out []hostedClient + err := r.forEachHost(input.Host, func(hb hostBridge) error { + if err := hb.bridge.requireRequest("list_clients", ipc.MsgListClientsReq, listClientsMinVersion); err != nil { + return err + } + resp, err := hb.bridge.request(ipc.MsgListClientsReq, nil) + if err != nil { + return fmt.Errorf("list_clients%s: %w", hostSuffix(hb.host), err) + } + var payload ipc.ListClientsRespPayload + if err := resp.DecodePayload(&payload); err != nil { + return fmt.Errorf("list_clients decode: %w", err) + } + for _, c := range payload.Clients { + out = append(out, hostedClient{ClientInfo: c, Host: hb.host}) + } + return nil + }) + if err != nil { + return nil, nil, fmt.Errorf("list_clients: %w", err) + } + if out == nil { + out = []hostedClient{} + } + return jsonResult(out), nil, nil + }) +} + // Notification tools // hostedEvent is a PaneEventPayload plus the host it came from, which diff --git a/cmd/quil/mcp_tools_router_test.go b/cmd/quil/mcp_tools_router_test.go index 9887120c..feb6a6c8 100644 --- a/cmd/quil/mcp_tools_router_test.go +++ b/cmd/quil/mcp_tools_router_test.go @@ -69,6 +69,10 @@ func newFakeIPCDaemonVersion(t *testing.T, paneID, version string) *fakeIPCDaemo resp, _ = ipc.NewMessage(ipc.MsgDelegateTaskResp, ipc.DelegateTaskRespPayload{Task: ipc.TaskInfo{ID: "task-1", ToPane: req.ToPane, FromPane: req.FromPane, State: "sent"}}) case ipc.MsgListProjectsReq: resp, _ = ipc.NewMessage(ipc.MsgListProjectsResp, ipc.ListProjectsRespPayload{Projects: []ipc.ProjectInfo{{ID: "proj-" + f.paneID, Name: f.paneID}}}) + case ipc.MsgListClientsReq: + resp, _ = ipc.NewMessage(ipc.MsgListClientsResp, ipc.ListClientsRespPayload{Clients: []ipc.ClientInfo{ + {Client: "tui-" + f.paneID, AttachedAt: "2026-01-01T00:00:00Z", Cols: 200, Rows: 50, Master: true}, + }}) case ipc.MsgCreateFromTemplateReq: var req ipc.CreateFromTemplateReqPayload if err := m.DecodePayload(&req); err != nil { @@ -309,7 +313,7 @@ func TestCreatePaneSchema_ExposesDialogOptions(t *testing.T) { } for _, want := range []string{"create_pane", "create_tab", "create_from_template", "list_projects", "create_project", "update_project", "destroy_project", "switch_project", "rename_tab", "destroy_tab", "rename_pane", "list_plugins", "list_sessions", "list_hosts", - "delegate_task", "get_task", "wait_task", "list_tasks"} { + "delegate_task", "get_task", "wait_task", "list_tasks", "list_clients"} { if byName[want] == nil { t.Errorf("tool %s not registered", want) } @@ -320,7 +324,7 @@ func TestCreatePaneSchema_ExposesDialogOptions(t *testing.T) { t.Errorf("create_pane schema lacks %s:\n%s", prop, schema) } } - if len(tools.Tools) != 35 { - t.Errorf("tool count = %d, want 35 (update docs/mcp.md and CLAUDE.md if this changed on purpose)", len(tools.Tools)) + if len(tools.Tools) != 36 { + t.Errorf("tool count = %d, want 36 (update docs/mcp.md and CLAUDE.md if this changed on purpose)", len(tools.Tools)) } } diff --git a/cmd/quil/mcp_tools_test.go b/cmd/quil/mcp_tools_test.go index 811350df..7379e6e4 100644 --- a/cmd/quil/mcp_tools_test.go +++ b/cmd/quil/mcp_tools_test.go @@ -1,8 +1,11 @@ package main import ( + "encoding/json" "reflect" + "strings" "testing" + "time" "github.com/artyomsv/quil/internal/ipc" ) @@ -130,3 +133,153 @@ func TestBuildTabMemSummaries_EmptyMemKeepsTabsAsZeroRows(t *testing.T) { } } } + +// TestSetActivePane_ClientFieldReachesThePayload: the tool's optional +// client input has to survive the trip onto the wire, or targeting one +// specific TUI (multi-client sync) silently degrades to the implicit +// most-recently-active target on every call. +func TestSetActivePane_ClientFieldReachesThePayload(t *testing.T) { + t.Setenv("QUIL_HOME", t.TempDir()) + local := newFakeIPCDaemon(t, "pane-local") + session, _ := toolHarness(t, local, nil) + + if _, err := callTool(t, session, "set_active_pane", map[string]any{ + "pane_id": "pane-local", "client": "tui-B", + }); err != nil { + t.Fatalf("set_active_pane: %v", err) + } + + // sendRaw only guarantees the frame is ENQUEUED by the time the tool call + // returns; the actual write happens on the bridge's own sendLoop + // goroutine (same reason mcp_hosts_test.go's waitUntil exists). + var got *ipc.SetActivePanePayload + waitUntil(func() bool { + local.mu.Lock() + defer local.mu.Unlock() + for _, m := range local.received { + if m.Type == ipc.MsgSetActivePane { + var p ipc.SetActivePanePayload + if err := m.DecodePayload(&p); err != nil { + t.Fatal(err) + } + got = &p + } + } + return got != nil + }, 500*time.Millisecond) + if got == nil { + t.Fatal("no set_active_pane frame reached the daemon") + } + if got.Client != "tui-B" { + t.Errorf("SetActivePanePayload.Client = %q, want %q", got.Client, "tui-B") + } +} + +// closeTUIReceived polls the fake daemon for the LATEST close_tui frame it +// received, up to 500ms. sendRaw only guarantees the frame is ENQUEUED by the +// time the tool call returns — the actual socket write happens on the +// bridge's own sendLoop goroutine, same as mcp_hosts_test.go's waitUntil +// exists for. +func closeTUIReceived(t *testing.T, f *fakeIPCDaemon) *ipc.CloseTUIPayload { + t.Helper() + var got *ipc.CloseTUIPayload + waitUntil(func() bool { + f.mu.Lock() + defer f.mu.Unlock() + for _, m := range f.received { + if m.Type == ipc.MsgCloseTUI { + var p ipc.CloseTUIPayload + if err := m.DecodePayload(&p); err != nil { + t.Fatal(err) + } + got = &p + } + } + return got != nil + }, 500*time.Millisecond) + return got +} + +// TestCloseTUI_ClientFieldReachesThePayload is the close_tui half of the +// same wiring check. +func TestCloseTUI_ClientFieldReachesThePayload(t *testing.T) { + t.Setenv("QUIL_HOME", t.TempDir()) + local := newFakeIPCDaemon(t, "pane-local") + session, _ := toolHarness(t, local, nil) + + if _, err := callTool(t, session, "close_tui", map[string]any{"client": "tui-A"}); err != nil { + t.Fatalf("close_tui: %v", err) + } + + got := closeTUIReceived(t, local) + if got == nil { + t.Fatal("no close_tui frame reached the daemon") + } + if got.Client != "tui-A" { + t.Errorf("CloseTUIPayload.Client = %q, want %q", got.Client, "tui-A") + } +} + +// TestCloseTUI_NoClientSendsAnEmptyPayload: an older-style call with no +// client argument must still reach the daemon as a valid (empty) payload — +// the historical broadcast-to-every-TUI shape a headless daemon and every +// existing caller depend on. +func TestCloseTUI_NoClientSendsAnEmptyPayload(t *testing.T) { + t.Setenv("QUIL_HOME", t.TempDir()) + local := newFakeIPCDaemon(t, "pane-local") + session, _ := toolHarness(t, local, nil) + + if _, err := callTool(t, session, "close_tui", map[string]any{}); err != nil { + t.Fatalf("close_tui: %v", err) + } + + got := closeTUIReceived(t, local) + if got == nil { + t.Fatal("no close_tui frame reached the daemon") + } + if got.Client != "" { + t.Errorf("CloseTUIPayload.Client = %q, want empty", got.Client) + } +} + +// TestListClients_ReturnsTheDaemonsList exercises the new list_clients tool +// end to end: it must gate on the daemon advertising list_clients_req (or a +// version at least listClientsMinVersion), and decode the daemon's answer. +func TestListClients_ReturnsTheDaemonsList(t *testing.T) { + t.Setenv("QUIL_HOME", t.TempDir()) + local := newFakeIPCDaemonVersion(t, "pane-local", listClientsMinVersion) + session, _ := toolHarness(t, local, nil) + + text, err := callTool(t, session, "list_clients", map[string]any{}) + if err != nil { + t.Fatalf("list_clients: %v", err) + } + var out []struct { + ipc.ClientInfo + Host string `json:"host"` + } + if err := json.Unmarshal([]byte(text), &out); err != nil { + t.Fatalf("decode: %v\n%s", err, text) + } + if len(out) != 1 || out[0].Client != "tui-pane-local" || !out[0].Master { + t.Fatalf("list_clients result = %+v", out) + } +} + +// TestListClients_RefusedBelowItsOwnFloor: list_clients_req is new with +// multi-client sync, so a daemon that neither advertises it nor clears +// listClientsMinVersion must be refused with a named error — not a silent +// drop and a timeout (the daemon simply ignores an unknown request type). +func TestListClients_RefusedBelowItsOwnFloor(t *testing.T) { + t.Setenv("QUIL_HOME", t.TempDir()) + local := newFakeIPCDaemonRequests(t, "pane-local", "9.9.9", ipc.MsgCreateTabReq) + session, _ := toolHarness(t, local, nil) + + _, err := callTool(t, session, "list_clients", map[string]any{}) + if err == nil || !strings.Contains(err.Error(), ipc.MsgListClientsReq) { + t.Fatalf("expected a refusal naming %s, got %v", ipc.MsgListClientsReq, err) + } + if !local.sawNo(ipc.MsgListClientsReq) { + t.Fatal("refused request reached the daemon") + } +} diff --git a/cmd/quil/mcp_version.go b/cmd/quil/mcp_version.go index cbc082e6..b626268b 100644 --- a/cmd/quil/mcp_version.go +++ b/cmd/quil/mcp_version.go @@ -26,6 +26,13 @@ import ( const mcpDaemonMinVersion = "1.72.0" const createFromTemplateMinVersion = "1.74.0" +// listClientsMinVersion is list_clients' own floor: list_clients_req is new +// with multi-client sync, and the daemon-side handler does not exist before +// it. A separate constant rather than raising mcpDaemonMinVersion — every +// OTHER existing tool must keep working against a daemon that predates this +// feature. +const listClientsMinVersion = "1.80.0" + // daemonVersionProbeTimeout bounds remote version probes. A pre-versioning // daemon drops the request silently. Local startup uses handshakeTimeout; // this is a package var so remote-bridge tests can keep it short. diff --git a/internal/daemon/browse.go b/internal/daemon/browse.go index 253fc330..06f15aa7 100644 --- a/internal/daemon/browse.go +++ b/internal/daemon/browse.go @@ -51,7 +51,7 @@ func (d *Daemon) handleBrowseDirReq(conn *ipc.Conn, msg *ipc.Message) { respondTo(conn, msg.ID, ipc.MsgBrowseDirResp, rejection) return } - fallback := d.defaultCWD() + fallback := d.defaultCWD(conn) go func() { defer d.browseScanning.Store(false) respondTo(conn, msg.ID, ipc.MsgBrowseDirResp, browseDirResponse(req, fallback)) diff --git a/internal/daemon/browse_test.go b/internal/daemon/browse_test.go index 1a13478e..28f9ddd4 100644 --- a/internal/daemon/browse_test.go +++ b/internal/daemon/browse_test.go @@ -1001,7 +1001,7 @@ func TestProjectCWD_DoesNotWedgeOnAnUnreachableRoot(t *testing.T) { t.Cleanup(func() { restoreSeam(t, block, func() { statPath = orig }) }) done := make(chan string, 1) - go func() { done <- d.projectCWD(p.ID) }() + go func() { done <- d.projectCWD(nil, p.ID) }() select { case got := <-done: diff --git a/internal/daemon/create_req.go b/internal/daemon/create_req.go index 5f004439..4c5a64b6 100644 --- a/internal/daemon/create_req.go +++ b/internal/daemon/create_req.go @@ -187,7 +187,7 @@ func (d *Daemon) handleCreatePaneReq(conn *ipc.Conn, msg *ipc.Message) { respondTo(conn, msg.ID, ipc.MsgCreatePaneResp, ipc.CreatePaneRespPayload{TabID: tabID, Error: "no such tab: " + tabID}) return } - payload, cwd, err := d.buildCreatePayload(req, tabID, d.defaultCWD()) + payload, cwd, err := d.buildCreatePayload(req, tabID, d.defaultCWD(conn)) if err != nil { respondTo(conn, msg.ID, ipc.MsgCreatePaneResp, ipc.CreatePaneRespPayload{TabID: tabID, Error: err.Error()}) return @@ -251,7 +251,7 @@ func (d *Daemon) handleCreateTabReq(conn *ipc.Conn, msg *ipc.Message) { if projectID == "" { projectID = d.session.ActiveProject() } - payload, cwd, err := d.buildCreatePayload(first, "", d.projectCWD(projectID)) + payload, cwd, err := d.buildCreatePayload(first, "", d.projectCWD(conn, projectID)) if err != nil { respondTo(conn, msg.ID, ipc.MsgCreateTabResp, ipc.CreateTabRespPayload{Error: err.Error()}) return diff --git a/internal/daemon/daemon.go b/internal/daemon/daemon.go index a46b955b..6ecb4a8e 100644 --- a/internal/daemon/daemon.go +++ b/internal/daemon/daemon.go @@ -64,11 +64,6 @@ type Daemon struct { tasks *taskRegistry tasksOnce sync.Once gitCache *gitCache // per-checkout branch/worktree/divergence, refreshed on a ticker - // clientCWD is the last-known CWD from a TUI client, used as the - // default working directory for new panes/tabs. Read by defaultCWD() - // from any IPC dispatch goroutine and written by handleAttach on each - // connect — atomic.Pointer is what keeps that race-free. - clientCWD atomic.Pointer[string] // Last attached terminal size, used before a client can resize a new pane. clientSize atomic.Pointer[terminalSize] @@ -1443,7 +1438,7 @@ func (d *Daemon) handleMessage(conn *ipc.Conn, msg *ipc.Message) { // The existence check happens HERE, before the handler, because the // handler reports nothing and the tab is gone afterwards either way. id, known := tabIDKnown(d, msg, "tab_id") - d.handleDestroyTab(msg) + d.handleDestroyTab(conn, msg) answerOp(conn, msg, ipc.MsgTabOpResp, id, known, opErrUnless(known, "no such tab")) case ipc.MsgSwitchTab: d.touchClientInput(conn) @@ -1462,7 +1457,14 @@ func (d *Daemon) handleMessage(conn *ipc.Conn, msg *ipc.Message) { case ipc.MsgDestroyPane: d.handleDestroyPane(msg) case ipc.MsgUpdatePane: - d.touchClientInput(conn) + // update_pane also carries automatic reports — an OSC 7 CWD change, + // overlay visibility, the unseen mark — that are not the user doing + // anything just now; only a field the user actually touched (rename, + // mute, eager, the two attention marks) should count as this + // client's input for targetConn's implicit-client fallback. + if updatePaneIsUserInput(msg) { + d.touchClientInput(conn) + } id, known := paneIDKnown(d, msg) d.handleUpdatePane(conn, msg) answerOp(conn, msg, ipc.MsgPaneOpResp, id, known, opErrUnless(known, "no such pane")) @@ -1509,7 +1511,7 @@ func (d *Daemon) handleMessage(conn *ipc.Conn, msg *ipc.Message) { // empty one renders as a blank screen the moment the user switches to // it, and there is no in-band way out: Ctrl+T files its tab against // the daemon's ACTIVE project, which a just-created one is not. - d.recoverEmptyProject(proj.ID) + d.recoverEmptyProject(conn, proj.ID) d.broadcastState() d.requestSnapshot() @@ -1536,7 +1538,7 @@ func (d *Daemon) handleMessage(conn *ipc.Conn, msg *ipc.Message) { releasePanes(detached) // Destroying the last project leaves nothing to render, and destroying // the active one can promote a project that is itself empty. - d.recoverEmptyProject(d.session.ActiveProject()) + d.recoverEmptyProject(conn, d.session.ActiveProject()) d.broadcastState() d.requestSnapshot() answerOp(conn, msg, ipc.MsgProjectOpResp, p.ProjectID, true, "") @@ -1591,7 +1593,7 @@ func (d *Daemon) handleMessage(conn *ipc.Conn, msg *ipc.Message) { // which is a blank screen with no in-band way out — Ctrl+T files against // the ACTIVE project, and that is the empty one. Create and destroy both // recover here for the same reason. - d.recoverEmptyProject(p.ProjectID) + d.recoverEmptyProject(conn, p.ProjectID) d.broadcastState() // Reassigns tabs and DROPS project records, so a daemon killed inside // the 30 s ticker window comes back holding the duplicates the user @@ -1704,7 +1706,9 @@ func (d *Daemon) handleMessage(conn *ipc.Conn, msg *ipc.Message) { case ipc.MsgSetActivePane: d.handleSetActivePane(conn, msg) case ipc.MsgCloseTUI: - d.broadcast(msg) + d.handleCloseTUI(conn, msg) + case ipc.MsgListClientsReq: + d.handleListClientsReq(conn, msg) // Notification center case ipc.MsgDismissEvent: @@ -1863,18 +1867,13 @@ func (d *Daemon) handleAttach(conn *ipc.Conn, msg *ipc.Message) { log.Printf("attach: client connected (%dx%d), tabs=%d, restored=%v", cols, rows, len(d.session.Tabs()), d.restored) - // Remember client CWD so new tabs/panes default to the TUI's directory - // instead of the daemon's (which is frozen at daemon start time). An - // empty value resets to "use daemon CWD" — preferable to retaining a - // stale value from a previous client. - cwd := attach.CWD - d.clientCWD.Store(&cwd) - // Create default workspace if empty (no tabs — neither fresh nor restored) if len(d.session.Tabs()) == 0 { log.Print("attach: creating default workspace (no tabs)") tab := d.session.CreateTab("Shell") - pane, _ := d.session.CreatePane(tab.ID, d.defaultCWD()) + // attachClient (above) already recorded attach.CWD on this conn's + // client record, so defaultCWD's first candidate is this very attach. + pane, _ := d.session.CreatePane(tab.ID, d.defaultCWD(conn)) setPaneType(pane, "terminal") ptySession := apty.NewWithSize(cols, rows) @@ -2218,7 +2217,7 @@ func (d *Daemon) handleCreateTab(conn *ipc.Conn, msg *ipc.Message) { spec.Worktree = nil } } - cwd := d.resolveRequestedCWD(spec.CWD, d.projectCWD(tab.ProjectID)) + cwd := d.resolveRequestedCWD(spec.CWD, d.projectCWD(conn, tab.ProjectID)) // The two construction paths are built SEPARATELY and share nothing but the // type and the directory. `create` and its plugin-field block used to sit @@ -2428,7 +2427,7 @@ func (d *Daemon) failPreparingPane(paneID, reason string) { d.broadcastState() } -func (d *Daemon) handleDestroyTab(msg *ipc.Message) { +func (d *Daemon) handleDestroyTab(conn *ipc.Conn, msg *ipc.Message) { var payload ipc.DestroyTabPayload if err := msg.DecodePayload(&payload); err != nil { return @@ -2455,7 +2454,7 @@ func (d *Daemon) handleDestroyTab(msg *ipc.Message) { d.cleanupPaneArtifacts(p.ID) } - d.recoverEmptyProject(projectID) + d.recoverEmptyProject(conn, projectID) d.broadcastState() d.requestSnapshot() @@ -2517,12 +2516,12 @@ func (d *Daemon) handleDestroyTab(msg *ipc.Message) { // whose project a racing DestroyProject removed. projectIsEmpty falls back to // the workspace-wide test there, and createTabLocked resolves the empty ID to // the active project or bootstraps one, so the workspace still recovers. -func (d *Daemon) recoverEmptyProject(projectID string) { +func (d *Daemon) recoverEmptyProject(conn *ipc.Conn, projectID string) { if !d.projectIsEmpty(projectID) { return } tab := d.session.CreateTabInProject(projectID, "Shell") - pane, err := d.session.CreatePane(tab.ID, d.projectCWD(tab.ProjectID)) + pane, err := d.session.CreatePane(tab.ID, d.projectCWD(conn, tab.ProjectID)) if err != nil { log.Printf("recover empty project %q: create pane: %v", projectID, err) return @@ -2565,9 +2564,9 @@ func (d *Daemon) projectIsEmpty(projectID string) bool { // value falls back rather than failing the spawn: a snapshot can outlive the // directory it names, and can be restored on a machine where that path never // existed. -func (d *Daemon) projectCWD(projectID string) string { +func (d *Daemon) projectCWD(conn *ipc.Conn, projectID string) string { if projectID == "" { - return d.defaultCWD() + return d.defaultCWD(conn) } for _, p := range d.session.Projects() { if p.ID != projectID { @@ -2583,7 +2582,7 @@ func (d *Daemon) projectCWD(projectID string) string { } break } - return d.defaultCWD() + return d.defaultCWD(conn) } func (d *Daemon) handleSwitchTab(msg *ipc.Message) { @@ -2661,7 +2660,7 @@ func (d *Daemon) handleMoveTab(conn *ipc.Conn, msg *ipc.Message) { // Moving the source project's LAST tab out leaves it exactly as empty as // DestroyTab leaves one, and owes the same replacement Shell tab. - d.recoverEmptyProject(from) + d.recoverEmptyProject(conn, from) // Each project keeps its OWN ActiveTab, and that is independent of the // daemon's single GLOBAL active project/tab (sm.activeProject/activeTab): // several clients can each be looking at a different project, so a @@ -2715,7 +2714,7 @@ func (d *Daemon) handleCreatePane(conn *ipc.Conn, msg *ipc.Message) { } logger.Debug("create pane: received payload cwd=%q type=%s", payload.CWD, payload.Type) - cwd := d.resolveRequestedCWD(payload.CWD, d.defaultCWD()) + cwd := d.resolveRequestedCWD(payload.CWD, d.defaultCWD(conn)) // Determine pane type paneType := payload.Type @@ -3128,7 +3127,9 @@ func (d *Daemon) recoverEmptyTab(tabID, reason string) { // The overlay is left in place, UNLIKE ensureTabNotEmpty's orphan sweep: // there the tab is losing its last pane for good, here it is getting a // normal one back on the next line. - pane, err := d.session.CreatePane(tabID, d.defaultCWD()) + // No conn: this runs from the destroy path with no requesting client in + // hand (a background recovery, like every other caller here). + pane, err := d.session.CreatePane(tabID, d.defaultCWD(nil)) if err != nil { log.Printf("tab %s: could not recover an empty tab: %v", tabID, err) return @@ -3226,7 +3227,7 @@ func (d *Daemon) handleMovePane(conn *ipc.Conn, msg *ipc.Message) { if destroyed { // The source may have been its project's only tab — the same emptiness // DestroyTab leaves, owed the same replacement Shell tab. - d.recoverEmptyProject(srcProject) + d.recoverEmptyProject(conn, srcProject) } // Spawn every selection this move touched; ensureTabSpawned is idempotent @@ -3338,7 +3339,9 @@ func (d *Daemon) ensureTabNotEmpty(tabID string) { d.cleanupPaneArtifacts(op.ID) d.session.DestroyPane(op.ID) } - if newPane, err := d.session.CreatePane(tabID, d.defaultCWD()); err == nil { + // No conn: ensureTabNotEmpty runs from destroy and exit paths with no + // requesting client in hand. + if newPane, err := d.session.CreatePane(tabID, d.defaultCWD(nil)); err == nil { setPaneType(newPane, "terminal") ptySession := apty.New() if err := d.spawnPane(newPane, ptySession, false); err != nil { @@ -3825,6 +3828,22 @@ func (d *Daemon) repaintAfterResize(pane *Pane, typ string) { d.sendRedrawKey(pane, typ, p.Persistence.RedrawKey) } +// updatePaneIsUserInput reports whether an update_pane payload carries a field +// the user just acted on (rename, mute, eager, pin/unpin attention, mark/unmark +// deletion) — as opposed to the automatic reports this same message also +// carries (an OSC 7 CWD change, overlay visibility, the unseen mark). Only the +// former should stamp the sending client's last-input time: a pane silently +// reporting its own CWD, or a TUI clearing an unseen mark on focus, must not +// make an idle client look like the one somebody is driving. +func updatePaneIsUserInput(msg *ipc.Message) bool { + var p ipc.UpdatePanePayload + if err := msg.DecodePayload(&p); err != nil { + return false + } + return p.Name != "" || p.Muted != nil || p.Eager != nil || + p.PinnedAttention != nil || p.MarkedForDeletion != nil +} + // handleUpdatePane applies a PARTIAL pane update. conn identifies the client // that sent it, which only the overlay-visibility field needs: that field is a // claim about one client's screen, not a daemon-wide fact. @@ -3963,8 +3982,22 @@ func (d *Daemon) handleUpdatePane(conn *ipc.Conn, msg *ipc.Message) { // from the replayed event history — a line per report would churn // quild.log for no diagnostic gain. pane.PluginMu.Lock() + wasUnseen := pane.Unseen pane.Unseen = *payload.Unseen pane.PluginMu.Unlock() + // Every OTHER attached client's sidebar carries the same "finished + // while you were away" mark for this pane, so the FALLING edge — and + // only the falling edge, never a re-affirmed true or an unchanged + // false — has to reach them too, or looking at the pane in one TUI + // leaves it marked in a second one. Sent before the quiet-field + // return below, which this field is one of. + if wasUnseen && !*payload.Unseen { + if seen, err := ipc.NewMessage(ipc.MsgPaneSeen, ipc.PaneSeenPayload{ + PaneID: pane.ID, + }); err == nil { + d.broadcast(seen) + } + } } if payload.OverlayVisible != nil { d.applyOverlayVisibility(conn, pane, *payload.OverlayVisible) @@ -5708,16 +5741,48 @@ func resolveSpawnArgs(p *plugin.PanePlugin, pane *Pane, restoring, ownsRecord bo return args } -// defaultCWD returns the best working directory for a new pane: the last -// known client CWD (from the most recent TUI attach) if it still points at -// an existing directory, falling back to the daemon's own working -// directory. Symlinks are resolved so all callers see the canonical path. -func (d *Daemon) defaultCWD() string { - if p := d.clientCWD.Load(); p != nil && *p != "" { - if dir := resolveSpawnDirWithin(*p, spawnDirProbeTimeout); dir != "" { - return dir +// defaultCWD returns the best working directory for a new pane, for a +// requesting client's conn. Multi-client sync gives each attached client its +// OWN cwd (the directory its own TUI was launched from), so "the last known +// client CWD" is no longer a single value the daemon can read off one field — +// it depends on which client is asking, and an MCP bridge (no attach at all) +// asks on behalf of nobody in particular. +// +// Checked in order, each candidate validated exactly the same way +// (resolveSpawnDirWithin: os.Stat + EvalSymlinks, so a stale or unreachable +// directory falls through rather than being trusted): +// +// 1. conn's own attached-client cwd, when conn names an attached client; +// 2. the size master's cwd — the client whose window sizes every PTY, +// the closest thing multi-client sync has to "the" TUI; +// 3. the most recently active client's cwd; +// 4. the daemon's own working directory. +// +// conn is nil for every restore and recovery caller (respawnPanes, +// recoverEmptyTab, ensureTabNotEmpty, …), which has no requesting client at +// all — those start at step 2. Symlinks are resolved so all callers see the +// canonical path. +func (d *Daemon) defaultCWD(conn *ipc.Conn) string { + if conn != nil { + if rec, ok := d.clientByConn(conn); ok { + if dir := resolveSpawnDirWithin(rec.cwd, spawnDirProbeTimeout); dir != "" { + return dir + } + } + } + if mc := d.masterConn(); mc != nil { + if rec, ok := d.clientByConn(mc); ok { + if dir := resolveSpawnDirWithin(rec.cwd, spawnDirProbeTimeout); dir != "" { + return dir + } + } + } + if ac := d.mostRecentlyActiveConn(); ac != nil { + if rec, ok := d.clientByConn(ac); ok { + if dir := resolveSpawnDirWithin(rec.cwd, spawnDirProbeTimeout); dir != "" { + return dir + } } - // stale (directory removed since attach), or unreachable — fall through } // Best-effort; if Getwd fails we return "" and the spawn will fail // with a clear error from os/exec rather than silently land somewhere. @@ -7379,11 +7444,19 @@ func (d *Daemon) handleSetActivePane(conn *ipc.Conn, msg *ipc.Message) { // Switch to the pane's tab d.session.SwitchTab(pane.CurrentTabID()) - // Broadcast to TUI clients so they can set focus - broadcast, _ := ipc.NewMessage(ipc.MsgSetActivePane, ipc.SetActivePanePayload{ - PaneID: req.PaneID, - }) - d.broadcast(broadcast) + // Focus reaches ONE client — req.Client, or the implicit target — never + // every attached TUI: a second TUI looking at something else must not + // have its focus yanked by a command aimed at the first. A headless + // daemon (nobody attached) has nothing to focus, so this send is + // nil-guarded rather than answered with a broadcast; the tab still + // switches for whoever attaches next. + if target := d.targetConn(req.Client); target != nil { + if focus, err := ipc.NewMessage(ipc.MsgSetActivePane, ipc.SetActivePanePayload{ + PaneID: req.PaneID, + }); err == nil { + target.Send(focus) + } + } d.broadcastState() d.requestSnapshot() @@ -7401,6 +7474,14 @@ func (d *Daemon) handleDismissEvent(msg *ipc.Message) { } else { d.events.Dismiss(payload.EventID) } + // Every attached client's sidebar is showing the same event(s); without + // this a card dismissed in one TUI keeps sitting in a second one until + // something else happens to refresh it. + if dismissed, err := ipc.NewMessage(ipc.MsgEventDismissed, ipc.EventDismissedPayload{ + EventID: payload.EventID, + }); err == nil { + d.broadcast(dismissed) + } } func (d *Daemon) handleGetNotificationsReq(conn *ipc.Conn, msg *ipc.Message) { diff --git a/internal/daemon/daemon_test.go b/internal/daemon/daemon_test.go index c80a3113..bc100eac 100644 --- a/internal/daemon/daemon_test.go +++ b/internal/daemon/daemon_test.go @@ -4,7 +4,6 @@ import ( "encoding/json" "errors" "os" - "path/filepath" "reflect" "testing" @@ -141,61 +140,23 @@ func TestWorkspaceStateFromSnapshot(t *testing.T) { } } -// TestDaemon_DefaultCWD covers the three branches of (*Daemon).defaultCWD: -// (1) clientCWD set and valid → returns the resolved client path, -// (2) clientCWD set but stale → falls back to os.Getwd(), -// (3) clientCWD unset → falls back to os.Getwd(). +// TestDaemon_DefaultCWD_NilConnFallsBackToGetwd covers defaultCWD's last +// candidate: a nil conn (every restore/recovery caller) on a daemon with no +// attached client at all falls back to the daemon's own working directory. // -// We bypass New() and build a minimal Daemon literal because defaultCWD only -// depends on the atomic.Pointer field, not on session/registry/etc. -func TestDaemon_DefaultCWD(t *testing.T) { +// The per-client, per-master and most-recently-active candidates need a real +// client registry with real conns attached over a socket — see +// TestDefaultCWD_PerClientAndBridge in mcp_targets_test.go, which a zero +// Daemon literal cannot exercise. +func TestDaemon_DefaultCWD_NilConnFallsBackToGetwd(t *testing.T) { hostCWD, err := os.Getwd() if err != nil { t.Fatalf("Getwd: %v", err) } - - t.Run("client CWD set and valid", func(t *testing.T) { - dir := t.TempDir() - d := &Daemon{} - d.clientCWD.Store(&dir) - got := d.defaultCWD() - // EvalSymlinks is applied; on macOS t.TempDir() lives under - // /var/folders/... which symlinks to /private/var/folders/..., - // so we compare the resolved form. - want, err := filepath.EvalSymlinks(dir) - if err != nil { - t.Fatalf("EvalSymlinks: %v", err) - } - if got != want { - t.Errorf("defaultCWD = %q, want %q", got, want) - } - }) - - t.Run("client CWD set but stale", func(t *testing.T) { - dir := t.TempDir() - stale := dir + "/does-not-exist" - d := &Daemon{} - d.clientCWD.Store(&stale) - if got := d.defaultCWD(); got != hostCWD { - t.Errorf("stale path should fall back to os.Getwd(); got %q, want %q", got, hostCWD) - } - }) - - t.Run("client CWD unset", func(t *testing.T) { - d := &Daemon{} - if got := d.defaultCWD(); got != hostCWD { - t.Errorf("unset should fall back to os.Getwd(); got %q, want %q", got, hostCWD) - } - }) - - t.Run("client CWD empty string", func(t *testing.T) { - empty := "" - d := &Daemon{} - d.clientCWD.Store(&empty) - if got := d.defaultCWD(); got != hostCWD { - t.Errorf("empty string should fall back to os.Getwd(); got %q, want %q", got, hostCWD) - } - }) + d := &Daemon{} + if got := d.defaultCWD(nil); got != hostCWD { + t.Errorf("defaultCWD(nil) = %q, want os.Getwd() %q", got, hostCWD) + } } // saveHookStubs captures and restores the package-level hook reader vars diff --git a/internal/daemon/discover.go b/internal/daemon/discover.go index d0af0315..83326e60 100644 --- a/internal/daemon/discover.go +++ b/internal/daemon/discover.go @@ -30,7 +30,7 @@ func (d *Daemon) handleGitReposReq(conn *ipc.Conn, msg *ipc.Message) { respondTo(conn, msg.ID, ipc.MsgGitReposResp, rejection) return } - fallback := d.defaultCWD() + fallback := d.defaultCWD(conn) go func() { defer d.gitDiscovering.Store(false) respondTo(conn, msg.ID, ipc.MsgGitReposResp, gitReposResponse(gitReposReq(msg), fallback)) diff --git a/internal/daemon/mcp_targets.go b/internal/daemon/mcp_targets.go new file mode 100644 index 00000000..fdf7457f --- /dev/null +++ b/internal/daemon/mcp_targets.go @@ -0,0 +1,91 @@ +package daemon + +import ( + "log" + + "github.com/artyomsv/quil/internal/ipc" +) + +// Multi-client sync: aiming an MCP command that used to broadcast to every +// attached TUI at exactly ONE client instead, plus the list_clients_req that +// lets an agent name one explicitly. + +// targetConn resolves which attached client's conn an untargeted MCP command +// (set_active_pane, close_tui) should reach. +// +// A non-empty id names an EXACT client: attached, its conn; not attached, +// nil — a stale id (the client detached, or was never real) must not +// silently redirect the command to whichever other window happens to be +// open. Empty means the IMPLICIT target: whichever client typed most +// recently, or — when nobody has typed anything at all — the OLDEST attached +// client, so an untargeted command against a freshly-attached workspace +// lands on a predictable client rather than on whichever TUI happened to +// dial last. +// +// That specific tie-break is NOT mostRecentlyActiveConn's own "nobody typed" +// answer (the newest attached client, per its own doc comment) — this +// function has its own logic rather than delegating, because the two +// callers want different defaults for the same "nobody has typed" state. +// With no client attached at all, nil. +func (d *Daemon) targetConn(clientID string) *ipc.Conn { + if clientID != "" { + d.clients.mu.Lock() + rec := d.clients.recordByID(clientID) + d.clients.mu.Unlock() + if rec == nil { + return nil + } + return rec.conn + } + d.clients.mu.Lock() + defer d.clients.mu.Unlock() + recs := d.clients.sortedRecordsLocked() + if len(recs) == 0 { + return nil + } + best := recs[0] // oldest attached — the default while nobody has input + for _, rec := range recs[1:] { + if rec.lastInputAt.After(best.lastInputAt) { + best = rec + } + } + return best.conn +} + +// handleCloseTUI answers MsgCloseTUI: ask ONE client to exit — the one named +// by CloseTUIPayload.Client, or the implicit target when it is empty or the +// payload is absent entirely (an older bridge sends nil, per the payload's +// own doc comment). +// +// This used to be a bare d.broadcast(msg), which was fine before several +// TUIs could share a daemon and became "close somebody else's window" the +// moment they could. With no attached client at all — a headless daemon — +// there is nothing to close; the command is dropped with a log line rather +// than panicking on a nil conn. +func (d *Daemon) handleCloseTUI(conn *ipc.Conn, msg *ipc.Message) { + var payload ipc.CloseTUIPayload + // A decode failure is treated exactly like an absent payload: older + // bridges send nil, and refusing the command over a malformed-but-empty + // frame would be a regression for every one of them. + _ = msg.DecodePayload(&payload) + target := d.targetConn(payload.Client) + if target == nil { + if payload.Client != "" { + log.Printf("close_tui: no attached client %q; dropping", payload.Client) + } else { + log.Printf("close_tui: no attached client; nothing to close") + } + return + } + target.Send(msg) +} + +// handleListClientsReq answers with every attached client — who is attached, +// since when, at what size, whether they hold size master, and when they +// last typed. listClients (clients.go) already merges the hello registry by +// conn and formats times as RFC 3339 UTC. +func (d *Daemon) handleListClientsReq(conn *ipc.Conn, msg *ipc.Message) { + respondTo(conn, msg.ID, ipc.MsgListClientsResp, ipc.ListClientsRespPayload{ + Clients: d.listClients(), + }) +} diff --git a/internal/daemon/mcp_targets_test.go b/internal/daemon/mcp_targets_test.go new file mode 100644 index 00000000..b918c796 --- /dev/null +++ b/internal/daemon/mcp_targets_test.go @@ -0,0 +1,422 @@ +package daemon + +import ( + "os" + "path/filepath" + "testing" + "time" + + "github.com/artyomsv/quil/internal/ipc" +) + +// Multi-client sync: targeting one client with an MCP command that used to +// broadcast to every attached TUI, per-client default directories, and the +// shared dismiss/seen marks. Every test drives real conns through +// ipc.Server, for the reason clients_wiring_test.go gives: the gate reads +// which CONN sent or should receive a message, and a direct handler call has +// no conn to read. + +// attachClientWithCWD attaches like attachClientAs but also carries a CWD, +// for defaultCWD's per-client candidate. +func attachClientWithCWD(t *testing.T, sock, id string, cols, rows int, cwd string) *ipc.Client { + t.Helper() + c, err := ipc.NewClient(sock) + if err != nil { + t.Fatalf("dial: %v", err) + } + t.Cleanup(func() { c.Close() }) + sendClientMsg(t, c, ipc.MsgAttach, ipc.AttachPayload{ClientID: id, Cols: cols, Rows: rows, CWD: cwd}) + return c +} + +// dialBridge dials the socket without ever attaching — an MCP bridge's shape: +// connected, but not a client the registry counts. +func dialBridge(t *testing.T, sock string) *ipc.Client { + t.Helper() + c, err := ipc.NewClient(sock) + if err != nil { + t.Fatalf("dial: %v", err) + } + t.Cleanup(func() { c.Close() }) + return c +} + +// resolvedTemp returns a real temp directory, symlink-resolved the same way +// resolveSpawnDirWithin resolves it — t.TempDir() on some platforms lives +// under a symlink, and defaultCWD's candidates are always compared in +// resolved form. +func resolvedTemp(t *testing.T) string { + t.Helper() + dir := t.TempDir() + resolved, err := filepath.EvalSymlinks(dir) + if err != nil { + t.Fatalf("EvalSymlinks: %v", err) + } + return resolved +} + +// TestCloseTUI_ReachesMostRecentlyActiveOnly: three attached clients, B typed +// last. Only B's conn receives close_tui; A and C get none. +func TestCloseTUI_ReachesMostRecentlyActiveOnly(t *testing.T) { + d, sock := overlayServerDaemon(t) + d.session.CreateTab("T") // skip the real-PTY default workspace on first attach + + a := attachClientAs(t, sock, "A", 200, 50) + waitUntil(t, "A attached", func() bool { return d.clientCount() == 1 }) + c := attachClientAs(t, sock, "C", 100, 30) + waitUntil(t, "C attached", func() bool { return d.clientCount() == 2 }) + b := attachClientAs(t, sock, "B", 100, 30) + waitUntil(t, "B attached", func() bool { return d.clientCount() == 3 }) + + barrier(t, d, a, "A") + barrier(t, d, c, "C") + barrier(t, d, b, "B") // B typed most recently + + bridge := dialBridge(t, sock) + sendClientMsg(t, bridge, ipc.MsgCloseTUI, nil) + + bGot := readFor(b, 500*time.Millisecond) + if countType(bGot, ipc.MsgCloseTUI) != 1 { + t.Fatalf("B (most recently active) got close_tui %d times, want 1: %v", countType(bGot, ipc.MsgCloseTUI), bGot) + } + aGot := readFor(a, 200*time.Millisecond) + if n := countType(aGot, ipc.MsgCloseTUI); n != 0 { + t.Errorf("A got close_tui %d times, want 0", n) + } + cGot := readFor(c, 200*time.Millisecond) + if n := countType(cGot, ipc.MsgCloseTUI); n != 0 { + t.Errorf("C got close_tui %d times, want 0", n) + } +} + +// TestCloseTUI_ExplicitClient: an explicit Client always wins over the +// implicit most-recently-active target. +func TestCloseTUI_ExplicitClient(t *testing.T) { + d, sock := overlayServerDaemon(t) + d.session.CreateTab("T") + + a := attachClientAs(t, sock, "A", 200, 50) + waitUntil(t, "A attached", func() bool { return d.clientCount() == 1 }) + b := attachClientAs(t, sock, "B", 100, 30) + waitUntil(t, "B attached", func() bool { return d.clientCount() == 2 }) + // B is the most-recently-active client (typed after A attached). + barrier(t, d, a, "A") + barrier(t, d, b, "B") + + bridge := dialBridge(t, sock) + sendClientMsg(t, bridge, ipc.MsgCloseTUI, ipc.CloseTUIPayload{Client: "A"}) + + aGot := readFor(a, 500*time.Millisecond) + if countType(aGot, ipc.MsgCloseTUI) != 1 { + t.Fatalf("A (explicit target) got close_tui %d times, want 1: %v", countType(aGot, ipc.MsgCloseTUI), aGot) + } + bGot := readFor(b, 200*time.Millisecond) + if n := countType(bGot, ipc.MsgCloseTUI); n != 0 { + t.Errorf("B got close_tui %d times, want 0 (A was named explicitly)", n) + } +} + +// TestSetActivePane_FocusFrameToOneConn: the tab-switch broadcast reaches +// every attached conn, and the set_active_pane focus frame reaches only the +// named client. +func TestSetActivePane_FocusFrameToOneConn(t *testing.T) { + d, sock := overlayServerDaemon(t) + tab := d.session.CreateTab("T") + pane, err := d.session.CreatePane(tab.ID, t.TempDir()) + if err != nil { + t.Fatalf("create pane: %v", err) + } + + a := attachClientAs(t, sock, "A", 200, 50) + waitUntil(t, "A attached", func() bool { return d.clientCount() == 1 }) + b := attachClientAs(t, sock, "B", 100, 30) + waitUntil(t, "B attached", func() bool { return d.clientCount() == 2 }) + // Drain each conn's own attach replay / other-client state frame before + // the assertions below, so the counts are about THIS set_active_pane. + readFor(a, 300*time.Millisecond) + readFor(b, 300*time.Millisecond) + + bridge := dialBridge(t, sock) + sendClientMsg(t, bridge, ipc.MsgSetActivePane, ipc.SetActivePanePayload{PaneID: pane.ID, Client: "B"}) + + bGot := readFor(b, 500*time.Millisecond) + if countType(bGot, ipc.MsgSetActivePane) != 1 { + t.Fatalf("B (named client) got set_active_pane %d times, want 1: %v", countType(bGot, ipc.MsgSetActivePane), bGot) + } + if countType(bGot, ipc.MsgWorkspaceState) == 0 { + t.Error("B never saw the tab-switch broadcast") + } + aGot := readFor(a, 200*time.Millisecond) + if n := countType(aGot, ipc.MsgSetActivePane); n != 0 { + t.Errorf("A got set_active_pane %d times, want 0", n) + } + if countType(aGot, ipc.MsgWorkspaceState) == 0 { + t.Error("A (not the named client) never saw the tab-switch broadcast — it must reach every attached conn") + } +} + +// TestMCPTargets_NoAttachedClient: Review Focus 4. A headless daemon (no +// attached client at all) must not panic on any of these, close_tui sends +// nothing, set_active_pane only switches the tab, and create_pane_req with +// an empty CWD falls all the way back to os.Getwd(). +func TestMCPTargets_NoAttachedClient(t *testing.T) { + d, sock := overlayServerDaemon(t) + tab := d.session.CreateTab("T") + pane, err := d.session.CreatePane(tab.ID, t.TempDir()) + if err != nil { + t.Fatalf("create pane: %v", err) + } + other := d.session.CreateTab("Other") + d.session.SwitchTab(other.ID) // so switching back to tab.ID below is an observable change + + bridge := dialBridge(t, sock) + + sendClientMsg(t, bridge, ipc.MsgCloseTUI, nil) + sendClientMsg(t, bridge, ipc.MsgSetActivePane, ipc.SetActivePanePayload{PaneID: pane.ID}) + waitUntil(t, "the tab switched with nobody attached", func() bool { + return d.session.ActiveTabID() == tab.ID + }) + + hostCWD, err := os.Getwd() + if err != nil { + t.Fatalf("Getwd: %v", err) + } + resp := decodeInto[ipc.CreatePaneRespPayload](t, roundTrip(t, bridge, ipc.MsgCreatePaneReq, ipc.MsgCreatePaneResp, + ipc.CreatePaneReqPayload{TabID: tab.ID})) + if resp.Error != "" { + t.Fatalf("create_pane_req: %s", resp.Error) + } + created := d.session.Pane(resp.PaneID) + if created == nil { + t.Fatal("pane not created") + } + created.PluginMu.Lock() + gotCWD := created.CWD + created.PluginMu.Unlock() + if gotCWD != hostCWD { + t.Errorf("headless create_pane_req CWD = %q, want os.Getwd() %q", gotCWD, hostCWD) + } + + // Neither command may have wedged the daemon: it must still answer. + // roundTrip itself fails the test if no response arrives in time. + roundTrip(t, bridge, ipc.MsgVersionReq, ipc.MsgVersionResp, struct{}{}) +} + +// TestDefaultCWD_PerClientAndBridge: each attached client gets ITS OWN +// default directory, and a bridge conn (no attach at all) falls back to the +// size master's directory. Driven through handleBrowseDirReq — an empty Path +// asks for defaultCWD(conn) and echoes it back as Resolved, with no PTY +// spawn — rather than through create_pane_req, whose behavior for the same +// resolver is separately covered by TestMCPTargets_NoAttachedClient. +func TestDefaultCWD_PerClientAndBridge(t *testing.T) { + d, sock := overlayServerDaemon(t) + d.session.CreateTab("T") + + dirA := resolvedTemp(t) + dirB := resolvedTemp(t) + + a := attachClientWithCWD(t, sock, "A", 200, 50, dirA) + waitUntil(t, "A attached", func() bool { return d.clientCount() == 1 }) + b := attachClientWithCWD(t, sock, "B", 100, 30, dirB) + waitUntil(t, "B attached", func() bool { return d.clientCount() == 2 }) + if d.masterID() != "A" { + t.Fatalf("masterID = %q, want A (the oldest eligible client)", d.masterID()) + } + + if got := decodeInto[ipc.BrowseDirRespPayload](t, roundTrip(t, a, ipc.MsgBrowseDirReq, ipc.MsgBrowseDirResp, ipc.BrowseDirReqPayload{})); got.Resolved != dirA { + t.Errorf("A's default dir = %q, want its own cwd %q", got.Resolved, dirA) + } + if got := decodeInto[ipc.BrowseDirRespPayload](t, roundTrip(t, b, ipc.MsgBrowseDirReq, ipc.MsgBrowseDirResp, ipc.BrowseDirReqPayload{})); got.Resolved != dirB { + t.Errorf("B's default dir = %q, want its own cwd %q", got.Resolved, dirB) + } + + bridge := dialBridge(t, sock) + if got := decodeInto[ipc.BrowseDirRespPayload](t, roundTrip(t, bridge, ipc.MsgBrowseDirReq, ipc.MsgBrowseDirResp, ipc.BrowseDirReqPayload{})); got.Resolved != dirA { + t.Errorf("a bridge's default dir = %q, want the master's (A's) cwd %q", got.Resolved, dirA) + } +} + +// TestDismiss_BroadcastsEventDismissed: a dismissal reaches every attached +// client, so a card dismissed through one TUI's sidebar disappears from a +// second one too. +func TestDismiss_BroadcastsEventDismissed(t *testing.T) { + d, sock := overlayServerDaemon(t) + d.session.CreateTab("T") + a := attachClientAs(t, sock, "A", 200, 50) + waitUntil(t, "A attached", func() bool { return d.clientCount() == 1 }) + readFor(a, 300*time.Millisecond) // drain the attach state frame + + bridge := dialBridge(t, sock) + sendClientMsg(t, bridge, ipc.MsgDismissEvent, ipc.DismissEventPayload{EventID: "evt-1"}) + + got := readFor(a, 500*time.Millisecond) + var found *ipc.EventDismissedPayload + for _, m := range got { + if m.Type == ipc.MsgEventDismissed { + var p ipc.EventDismissedPayload + if err := m.DecodePayload(&p); err != nil { + t.Fatal(err) + } + found = &p + } + } + if found == nil { + t.Fatalf("no event_dismissed reached A: %v", got) + } + if found.EventID != "evt-1" { + t.Errorf("EventDismissedPayload.EventID = %q, want %q", found.EventID, "evt-1") + } +} + +// TestPaneSeen_OnlyOnTrueToFalse: false→false and true→true send no +// pane_seen frame; only a true→false transition does, exactly once. +func TestPaneSeen_OnlyOnTrueToFalse(t *testing.T) { + d, sock := overlayServerDaemon(t) + tab := d.session.CreateTab("T") + pane, err := d.session.CreatePane(tab.ID, t.TempDir()) + if err != nil { + t.Fatalf("create pane: %v", err) + } + a := attachClientAs(t, sock, "A", 200, 50) + waitUntil(t, "A attached", func() bool { return d.clientCount() == 1 }) + readFor(a, 300*time.Millisecond) // drain the attach state frame + + setUnseen := func(v bool) { + pane.PluginMu.Lock() + pane.Unseen = v + pane.PluginMu.Unlock() + } + sendUnseen := func(v bool) []*ipc.Message { + sendClientMsg(t, a, ipc.MsgUpdatePane, ipc.UpdatePanePayload{PaneID: pane.ID, Unseen: &v}) + return readFor(a, 300*time.Millisecond) + } + + setUnseen(false) + if got := countType(sendUnseen(false), ipc.MsgPaneSeen); got != 0 { + t.Errorf("false->false sent pane_seen %d times, want 0", got) + } + + setUnseen(true) + if got := countType(sendUnseen(true), ipc.MsgPaneSeen); got != 0 { + t.Errorf("true->true sent pane_seen %d times, want 0", got) + } + + setUnseen(true) + msgs := sendUnseen(false) + if got := countType(msgs, ipc.MsgPaneSeen); got != 1 { + t.Fatalf("true->false sent pane_seen %d times, want 1: %v", got, msgs) + } + for _, m := range msgs { + if m.Type != ipc.MsgPaneSeen { + continue + } + var p ipc.PaneSeenPayload + if err := m.DecodePayload(&p); err != nil { + t.Fatal(err) + } + if p.PaneID != pane.ID { + t.Errorf("PaneSeenPayload.PaneID = %q, want %q", p.PaneID, pane.ID) + } + } +} + +// TestListClients_Fields covers list_clients_req's field-level contract: the +// oldest client first, Master set on exactly the size master, empty +// LastInputAt for a client that has not typed, and a populated one — parsing +// as RFC 3339 — for one that has. +func TestListClients_Fields(t *testing.T) { + d, sock := overlayServerDaemon(t) + d.session.CreateTab("T") + a := attachClientAs(t, sock, "A", 200, 50) + waitUntil(t, "A attached", func() bool { return d.clientCount() == 1 }) + b := attachClientAs(t, sock, "B", 100, 30) + waitUntil(t, "B attached", func() bool { return d.clientCount() == 2 }) + + resp := decodeInto[ipc.ListClientsRespPayload](t, roundTrip(t, a, ipc.MsgListClientsReq, ipc.MsgListClientsResp, nil)) + if len(resp.Clients) != 2 { + t.Fatalf("clients = %+v, want 2", resp.Clients) + } + if resp.Clients[0].Client != "A" || !resp.Clients[0].Master { + t.Errorf("clients[0] = %+v, want A as master (oldest attached)", resp.Clients[0]) + } + if resp.Clients[1].Client != "B" || resp.Clients[1].Master { + t.Errorf("clients[1] = %+v, want B, not master", resp.Clients[1]) + } + for _, c := range resp.Clients { + if c.LastInputAt != "" { + t.Errorf("client %s LastInputAt = %q, want empty before any input", c.Client, c.LastInputAt) + } + if _, err := time.Parse(time.RFC3339, c.AttachedAt); err != nil { + t.Errorf("client %s AttachedAt = %q, not RFC 3339: %v", c.Client, c.AttachedAt, err) + } + } + + barrier(t, d, b, "B") + resp = decodeInto[ipc.ListClientsRespPayload](t, roundTrip(t, a, ipc.MsgListClientsReq, ipc.MsgListClientsResp, nil)) + byID := map[string]ipc.ClientInfo{} + for _, c := range resp.Clients { + byID[c.Client] = c + } + if byID["A"].LastInputAt != "" { + t.Errorf("A LastInputAt = %q, want still empty", byID["A"].LastInputAt) + } + if byID["B"].LastInputAt == "" { + t.Fatal("B LastInputAt is still empty after it typed") + } + if _, err := time.Parse(time.RFC3339, byID["B"].LastInputAt); err != nil { + t.Errorf("B LastInputAt = %q, not RFC 3339: %v", byID["B"].LastInputAt, err) + } +} + +// TestUpdatePane_LastInputStampsOnlyUserFields is the carried-over Task 2 +// review fix: update_pane also reports automatic changes (an OSC 7 CWD +// change, overlay visibility, the unseen mark), and only a field the user +// actually acted on may stamp the sending client's last-input time — +// otherwise a pane silently reporting its own CWD makes an idle client look +// like the one somebody is driving, which is exactly what targetConn's +// implicit fallback reads to pick a client. +func TestUpdatePane_LastInputStampsOnlyUserFields(t *testing.T) { + d, sock := overlayServerDaemon(t) + tab := d.session.CreateTab("T") + pane, err := d.session.CreatePane(tab.ID, t.TempDir()) + if err != nil { + t.Fatalf("create pane: %v", err) + } + a := attachClientAs(t, sock, "A", 200, 50) + waitUntil(t, "A attached", func() bool { return d.clientCount() == 1 }) + + fenceMarker := 40 // eligible geometry, and distinct from A's 200x50 attach + fence := func() { + t.Helper() + fenceMarker++ + sendClientMsg(t, a, ipc.MsgClientGeometry, ipc.ClientGeometryPayload{Cols: fenceMarker, Rows: fenceMarker}) + waitUntil(t, "fence processed", func() bool { + rec, _ := clientRecordByID(d, "A") + return rec.cols == fenceMarker && rec.rows == fenceMarker + }) + } + + notStamped := func(name string, payload ipc.UpdatePanePayload) { + t.Helper() + payload.PaneID = pane.ID + sendClientMsg(t, a, ipc.MsgUpdatePane, payload) + // client_geometry never touches lastInputAt (see the dispatch table in + // daemon.go), so fencing with it proves the update_pane above already + // ran WITHOUT itself being able to stamp anything. + fence() + rec, _ := clientRecordByID(d, "A") + if !rec.lastInputAt.IsZero() { + t.Errorf("%s stamped lastInputAt, want it to stay unstamped", name) + } + } + + notStamped("CWD-only", ipc.UpdatePanePayload{CWD: t.TempDir()}) + notStamped("OverlayVisible-only", ipc.UpdatePanePayload{OverlayVisible: boolPtr(true)}) + notStamped("Unseen-only", ipc.UpdatePanePayload{Unseen: boolPtr(true)}) + + sendClientMsg(t, a, ipc.MsgUpdatePane, ipc.UpdatePanePayload{PaneID: pane.ID, Name: "renamed"}) + waitUntil(t, "Name stamps lastInputAt", func() bool { + rec, _ := clientRecordByID(d, "A") + return !rec.lastInputAt.IsZero() + }) +} diff --git a/internal/daemon/pane_cleanup_test.go b/internal/daemon/pane_cleanup_test.go index e7ef4270..39a7b8df 100644 --- a/internal/daemon/pane_cleanup_test.go +++ b/internal/daemon/pane_cleanup_test.go @@ -59,7 +59,7 @@ func TestHandleDestroyTab_CleansHookArtifacts(t *testing.T) { spoolFile, sessFile := seedPaneArtifacts(t, pane.ID) msg, _ := ipc.NewMessage(ipc.MsgDestroyTab, ipc.DestroyTabPayload{TabID: tab.ID}) - d.handleDestroyTab(msg) + d.handleDestroyTab(nil, msg) assertGone(t, spoolFile, sessFile) } diff --git a/internal/daemon/project_req.go b/internal/daemon/project_req.go index ab9f1601..04190ed5 100644 --- a/internal/daemon/project_req.go +++ b/internal/daemon/project_req.go @@ -56,9 +56,9 @@ func (d *Daemon) handleCreateProjectReq(conn *ipc.Conn, msg *ipc.Message) { respondTo(conn, msg.ID, ipc.MsgCreateProjectResp, ipc.CreateProjectRespPayload{Error: "name is required"}) return } - rootDir := d.resolveRequestedCWD(req.RootDir, d.defaultCWD()) + rootDir := d.resolveRequestedCWD(req.RootDir, d.defaultCWD(conn)) proj := d.session.CreateProject(req.Name, rootDir) - d.recoverEmptyProject(proj.ID) + d.recoverEmptyProject(conn, proj.ID) d.broadcastState() d.requestSnapshot() log.Printf("project created over IPC request: %s %q", proj.ID, proj.Name) diff --git a/internal/daemon/project_test.go b/internal/daemon/project_test.go index de093fa7..b41b2bdd 100644 --- a/internal/daemon/project_test.go +++ b/internal/daemon/project_test.go @@ -500,19 +500,19 @@ func TestHandleCreateProjectShipsATabRootedAtTheProjectDir(t *testing.T) { // Neither may fail the spawn. func TestProjectCWDFallsBackForAnUnusableRoot(t *testing.T) { d := newTestDaemon(t) - fallback := d.defaultCWD() + fallback := d.defaultCWD(nil) gone := d.session.CreateProject("gone", filepath.Join(t.TempDir(), "never-existed")) - if got := d.projectCWD(gone.ID); got != fallback { + if got := d.projectCWD(nil, gone.ID); got != fallback { t.Errorf("projectCWD(stale root) = %q, want the daemon default %q", got, fallback) } blank := d.session.CreateProject("blank", "") - if got := d.projectCWD(blank.ID); got != fallback { + if got := d.projectCWD(nil, blank.ID); got != fallback { t.Errorf("projectCWD(no root) = %q, want the daemon default %q", got, fallback) } - if got := d.projectCWD("proj-does-not-exist"); got != fallback { + if got := d.projectCWD(nil, "proj-does-not-exist"); got != fallback { t.Errorf("projectCWD(unknown project) = %q, want the daemon default %q", got, fallback) } @@ -521,7 +521,7 @@ func TestProjectCWDFallsBackForAnUnusableRoot(t *testing.T) { t.Fatalf("write file: %v", err) } notADir := d.session.CreateProject("file", file) - if got := d.projectCWD(notADir.ID); got != fallback { + if got := d.projectCWD(nil, notADir.ID); got != fallback { t.Errorf("projectCWD(file as root) = %q, want the daemon default %q", got, fallback) } } diff --git a/internal/daemon/worktree.go b/internal/daemon/worktree.go index 0bb1603f..45bb138b 100644 --- a/internal/daemon/worktree.go +++ b/internal/daemon/worktree.go @@ -36,7 +36,7 @@ func (d *Daemon) handleWorktreeListReq(conn *ipc.Conn, msg *ipc.Message) { respondTo(conn, msg.ID, ipc.MsgWorktreeListResp, rejection) return } - fallback := d.defaultCWD() + fallback := d.defaultCWD(conn) go func() { defer d.worktreeScanning.Store(false) respondTo(conn, msg.ID, ipc.MsgWorktreeListResp, diff --git a/internal/daemon/worktree_remove_test.go b/internal/daemon/worktree_remove_test.go index 1f9660d1..547ef00d 100644 --- a/internal/daemon/worktree_remove_test.go +++ b/internal/daemon/worktree_remove_test.go @@ -258,7 +258,7 @@ func TestHandleDestroyTab_RemovesEveryOwnedWorktreeInTheTab(t *testing.T) { if err != nil { t.Fatalf("NewMessage: %v", err) } - d.handleDestroyTab(msg) + d.handleDestroyTab(nil, msg) got := map[string]bool{} for range 2 { From e928ce544e8ce7133b6d52c8c1c299e44636badb Mon Sep 17 00:00:00 2001 From: Artjoms Stukans Date: Fri, 25 Sep 2026 19:47:08 +0200 Subject: [PATCH 10/40] fix(daemon): dedupe defaultCWD probes and firm up MCP target tests defaultCWD probed the same directory up to three times in the ordinary single-TUI case, since the requesting conn, the size master and the most-recently-active client all name the same client there - each probe paid its own spawnDirProbeTimeout and abandoned its own claimBlockingFSCall permit against a dead directory. Candidates are now deduped by path and share one deadline, so three different unreachable candidates together cost no more than one spawnDirProbeTimeout instead of three paid serially. set_active_pane now logs when it is given an explicit client id that is not attached, matching close_tui's existing log line for the same case. Test fixes: the close_tui target test previously let "typed last" and "attached last" coincide on the same client, so it could not tell which one the daemon was actually choosing by; it now separates them, and a new test covers the case where nobody has typed at all. A new test drives create_pane_req from a non-master client to prove the create path (not just the read-only browse path) resolves against the requesting conn, not the master's. The headless-daemon test now reads back the sending conn itself to confirm neither command is echoed to its own sender. list_clients' version-floor test is renamed to match what it actually covers (the daemon capability list, not the floor number), with a new test pinning the floor itself and its allowed-below counterpart. update_pane's stamping test now also covers Muted, Eager, PinnedAttention and MarkedForDeletion, not just Name. Several tests called the deadline-based readFor/roundTrip readers more than once on one conn; a call whose deadline lapses mid-frame discards whatever it had already read, corrupting every later read on that conn. Tests needing more than one checkpoint on a conn now either read once with a decoded assertion that cannot be satisfied by earlier noise, or use a new no-deadline matched reader safe to call repeatedly. Doc comments for SetActivePanePayload.Client, CloseTUIPayload, EventDismissedPayload and PaneSeenPayload no longer describe the old broadcast-to-every-TUI behavior these replaced. --- cmd/quil/mcp_tools.go | 9 +- cmd/quil/mcp_tools_test.go | 56 +++- internal/daemon/daemon.go | 38 ++- internal/daemon/mcp_targets_test.go | 415 +++++++++++++++++++++++++--- internal/ipc/protocol.go | 43 +-- internal/tui/pane.go | 2 +- 6 files changed, 489 insertions(+), 74 deletions(-) diff --git a/cmd/quil/mcp_tools.go b/cmd/quil/mcp_tools.go index 4d6a124c..0a89d108 100644 --- a/cmd/quil/mcp_tools.go +++ b/cmd/quil/mcp_tools.go @@ -676,9 +676,10 @@ func registerCloseTUITool(s *mcp.Server, r *mcpRouter, mcpLog *mcpLogger) { }) } -// registerListClientsTool lists every attached client (TUI or MCP bridge) -// so an agent can target one explicitly with set_active_pane or close_tui -// instead of landing on the implicit most-recently-active one. +// registerListClientsTool lists every ATTACHED client — every TUI sharing +// this daemon, never an MCP bridge, which is a connected conn but never +// attaches — so an agent can target one explicitly with set_active_pane or +// close_tui instead of landing on the implicit most-recently-active one. func registerListClientsTool(s *mcp.Server, r *mcpRouter, mcpLog *mcpLogger) { type Input struct { Host string `json:"host,omitempty" jsonschema:"limit to one host (default: every connected host)"` @@ -690,7 +691,7 @@ func registerListClientsTool(s *mcp.Server, r *mcpRouter, mcpLog *mcpLogger) { mcp.AddTool(s, &mcp.Tool{ Name: "list_clients", - Description: "List every attached TUI/bridge client: id, attach time, window size, whether it holds size master " + + Description: "List every attached TUI client: id, attach time, window size, whether it holds size master " + "(the client whose geometry sizes every pane), and its last input time. Pass a client id to set_active_pane " + "or close_tui to target that client instead of the one that typed most recently.", }, func(_ context.Context, _ *mcp.CallToolRequest, input Input) (*mcp.CallToolResult, any, error) { diff --git a/cmd/quil/mcp_tools_test.go b/cmd/quil/mcp_tools_test.go index 7379e6e4..2c5ce9bb 100644 --- a/cmd/quil/mcp_tools_test.go +++ b/cmd/quil/mcp_tools_test.go @@ -222,8 +222,9 @@ func TestCloseTUI_ClientFieldReachesThePayload(t *testing.T) { // TestCloseTUI_NoClientSendsAnEmptyPayload: an older-style call with no // client argument must still reach the daemon as a valid (empty) payload — -// the historical broadcast-to-every-TUI shape a headless daemon and every -// existing caller depend on. +// which the daemon reads as "the implicit target" (Daemon.targetConn), not +// as a refusal. Every caller that predates this field sends exactly this +// shape, and a headless daemon (nothing attached) must still accept it. func TestCloseTUI_NoClientSendsAnEmptyPayload(t *testing.T) { t.Setenv("QUIL_HOME", t.TempDir()) local := newFakeIPCDaemon(t, "pane-local") @@ -266,11 +267,15 @@ func TestListClients_ReturnsTheDaemonsList(t *testing.T) { } } -// TestListClients_RefusedBelowItsOwnFloor: list_clients_req is new with -// multi-client sync, so a daemon that neither advertises it nor clears -// listClientsMinVersion must be refused with a named error — not a silent -// drop and a timeout (the daemon simply ignores an unknown request type). -func TestListClients_RefusedBelowItsOwnFloor(t *testing.T) { +// TestListClients_DaemonWithoutTheRequest_IsRefusedHoweverNewItReads: a +// daemon that ANSWERED and did not list list_clients_req is refused however +// new its version number reads — the same reasoning +// TestCreateFromTemplate_DaemonWithoutTheRequest_IsRefusedHoweverNewItReads +// (mcp_version_test.go) pins for that tool. quil-debug.exe attaches to the +// production daemon by design, so a released build wearing a version well +// past listClientsMinVersion is reachable, and must still get the named +// refusal rather than a silent drop and a timeout. +func TestListClients_DaemonWithoutTheRequest_IsRefusedHoweverNewItReads(t *testing.T) { t.Setenv("QUIL_HOME", t.TempDir()) local := newFakeIPCDaemonRequests(t, "pane-local", "9.9.9", ipc.MsgCreateTabReq) session, _ := toolHarness(t, local, nil) @@ -283,3 +288,40 @@ func TestListClients_RefusedBelowItsOwnFloor(t *testing.T) { t.Fatal("refused request reached the daemon") } } + +// TestListClients_DaemonAdvertisingTheRequest_IsAllowedBelowTheFloor is the +// other direction: a daemon whose OWN reported version is below +// listClientsMinVersion is still allowed once it says it handles +// list_clients_req — a branch build and a release can report the same +// number (dev.sh stamps VERSION into every variant), so the daemon's own +// capability list decides ahead of the number wherever it can say. +func TestListClients_DaemonAdvertisingTheRequest_IsAllowedBelowTheFloor(t *testing.T) { + t.Setenv("QUIL_HOME", t.TempDir()) + local := newFakeIPCDaemonRequests(t, "pane-local", "1.73.0", ipc.MsgListClientsReq) + session, _ := toolHarness(t, local, nil) + + if _, err := callTool(t, session, "list_clients", map[string]any{}); err != nil { + t.Fatalf("daemon advertising the request was refused on its number: %v", err) + } + if local.sawNo(ipc.MsgListClientsReq) { + t.Fatal("allowed request was not sent") + } +} + +// TestListClients_RefusedBelowItsOwnVersionFloor tests listClientsMinVersion +// itself: a daemon reporting NO capability list (every daemon built before +// that field existed) falls back to the plain version compare, and a version +// one patch below the floor must be refused by NUMBER alone. +func TestListClients_RefusedBelowItsOwnVersionFloor(t *testing.T) { + t.Setenv("QUIL_HOME", t.TempDir()) + local := newFakeIPCDaemonVersion(t, "pane-local", "1.79.9") + session, _ := toolHarness(t, local, nil) + + _, err := callTool(t, session, "list_clients", map[string]any{}) + if err == nil || !strings.Contains(err.Error(), listClientsMinVersion) { + t.Fatalf("expected a refusal naming the floor %s, got %v", listClientsMinVersion, err) + } + if !local.sawNo(ipc.MsgListClientsReq) { + t.Fatal("refused request reached the daemon") + } +} diff --git a/internal/daemon/daemon.go b/internal/daemon/daemon.go index 6ecb4a8e..4cfbaa6f 100644 --- a/internal/daemon/daemon.go +++ b/internal/daemon/daemon.go @@ -5748,7 +5748,7 @@ func resolveSpawnArgs(p *plugin.PanePlugin, pane *Pane, restoring, ownsRecord bo // it depends on which client is asking, and an MCP bridge (no attach at all) // asks on behalf of nobody in particular. // -// Checked in order, each candidate validated exactly the same way +// Checked in order, each DISTINCT candidate validated exactly the same way // (resolveSpawnDirWithin: os.Stat + EvalSymlinks, so a stale or unreachable // directory falls through rather than being trusted): // @@ -5758,28 +5758,48 @@ func resolveSpawnArgs(p *plugin.PanePlugin, pane *Pane, restoring, ownsRecord bo // 3. the most recently active client's cwd; // 4. the daemon's own working directory. // -// conn is nil for every restore and recovery caller (respawnPanes, -// recoverEmptyTab, ensureTabNotEmpty, …), which has no requesting client at -// all — those start at step 2. Symlinks are resolved so all callers see the -// canonical path. +// In the ORDINARY case — one attached TUI — steps 1 through 3 all name the +// SAME client, so a candidate string already tried is skipped rather than +// probed again: without that, a single dead directory cost this dispatch +// goroutine up to three separate spawnDirProbeTimeout waits (6s) and +// abandoned three claimBlockingFSCall permits instead of one. The remaining +// probes also share ONE deadline rather than a fresh spawnDirProbeTimeout +// each, so even three genuinely DIFFERENT unreachable candidates cost this +// call no more than spawnDirProbeTimeout in total. +// +// conn is nil for every restore and recovery caller (recoverEmptyTab, +// ensureTabNotEmpty, …), which has no requesting client at all — those start +// at step 2. Symlinks are resolved so all callers see the canonical path. func (d *Daemon) defaultCWD(conn *ipc.Conn) string { + deadline := time.Now().Add(spawnDirProbeTimeout) + tried := make(map[string]bool, 3) + // tryCandidate skips a cwd already attempted (by value — the ordinary + // single-TUI case names the same directory at every step) and spends + // only what is left of the shared deadline. + tryCandidate := func(cwd string) string { + if cwd == "" || tried[cwd] { + return "" + } + tried[cwd] = true + return resolveSpawnDirWithin(cwd, time.Until(deadline)) + } if conn != nil { if rec, ok := d.clientByConn(conn); ok { - if dir := resolveSpawnDirWithin(rec.cwd, spawnDirProbeTimeout); dir != "" { + if dir := tryCandidate(rec.cwd); dir != "" { return dir } } } if mc := d.masterConn(); mc != nil { if rec, ok := d.clientByConn(mc); ok { - if dir := resolveSpawnDirWithin(rec.cwd, spawnDirProbeTimeout); dir != "" { + if dir := tryCandidate(rec.cwd); dir != "" { return dir } } } if ac := d.mostRecentlyActiveConn(); ac != nil { if rec, ok := d.clientByConn(ac); ok { - if dir := resolveSpawnDirWithin(rec.cwd, spawnDirProbeTimeout); dir != "" { + if dir := tryCandidate(rec.cwd); dir != "" { return dir } } @@ -7456,6 +7476,8 @@ func (d *Daemon) handleSetActivePane(conn *ipc.Conn, msg *ipc.Message) { }); err == nil { target.Send(focus) } + } else if req.Client != "" { + log.Printf("set_active_pane: no attached client %q; dropping the focus frame", req.Client) } d.broadcastState() diff --git a/internal/daemon/mcp_targets_test.go b/internal/daemon/mcp_targets_test.go index b918c796..c1acf3e7 100644 --- a/internal/daemon/mcp_targets_test.go +++ b/internal/daemon/mcp_targets_test.go @@ -1,8 +1,10 @@ package daemon import ( + "fmt" "os" "path/filepath" + "sync/atomic" "testing" "time" @@ -55,33 +57,159 @@ func resolvedTemp(t *testing.T) string { return resolved } -// TestCloseTUI_ReachesMostRecentlyActiveOnly: three attached clients, B typed -// last. Only B's conn receives close_tui; A and C get none. +// readUntilID reads c's frames — COLLECTING every one — until a response of +// respType with the given envelope id arrives, with NO SetReadDeadline. +// +// readFor and readUntil (resize_authority_test.go) both call +// SetReadDeadline, and ipc.ReadMessage's io.ReadFull DISCARDS whatever +// partial length-prefix or payload bytes it already consumed the instant +// that deadline fires mid-read — the next call on the same conn then +// misreads leftover bytes as a fresh frame header. That is fine for a +// single terminal read, but a test that calls a deadline-based reader +// MORE THAN ONCE on the same conn risks exactly that corruption on a slow +// CI run. This helper is safe to call repeatedly on one conn: each call's +// background goroutine reads until ITS match (or the conn closes, or the +// caller's own timeout fires), and two calls never race because the first +// one's goroutine has already returned by the time the caller sees its +// result. +func readUntilID(t *testing.T, c *ipc.Client, respType, id string, within time.Duration) []*ipc.Message { + t.Helper() + type result struct { + got []*ipc.Message + err error + } + ch := make(chan result, 1) + go func() { + var got []*ipc.Message + for { + m, err := c.Receive() + if err != nil { + ch <- result{got, err} + return + } + got = append(got, m) + if m.Type == respType && m.ID == id { + ch <- result{got, nil} + return + } + } + }() + select { + case r := <-ch: + if r.err != nil { + t.Fatalf("waiting for %s(%s): %v", respType, id, r.err) + } + return r.got + case <-time.After(within): + t.Fatalf("timed out waiting for %s(%s)", respType, id) + return nil + } +} + +// sendWithID sends payload as typ on c, stamping the envelope id so a +// caller can correlate it with the resulting pane_op_resp/etc via +// readUntilID. +func sendWithID(t *testing.T, c *ipc.Client, typ, id string, payload any) { + t.Helper() + msg, err := ipc.NewMessage(typ, payload) + if err != nil { + t.Fatalf("build %s: %v", typ, err) + } + msg.ID = id + if err := c.Send(msg); err != nil { + t.Fatalf("send %s: %v", typ, err) + } +} + +// workspaceStateActiveTab decodes a workspace_state frame's top-level +// active_tab field. +func workspaceStateActiveTab(t *testing.T, m *ipc.Message) string { + t.Helper() + var s struct { + ActiveTab string `json:"active_tab"` + } + if err := m.DecodePayload(&s); err != nil { + t.Fatal(err) + } + return s.ActiveTab +} + +// sawActiveTab reports whether any workspace_state frame in msgs names +// tabID as the active tab. +func sawActiveTab(t *testing.T, msgs []*ipc.Message, tabID string) bool { + t.Helper() + for _, m := range msgs { + if m.Type == ipc.MsgWorkspaceState && workspaceStateActiveTab(t, m) == tabID { + return true + } + } + return false +} + +// TestCloseTUI_ReachesMostRecentlyActiveOnly: three attached clients. C is +// the NEWEST attached, but A — the OLDEST — types LAST, so the implicit +// target must be A. This is what proves the choice is driven by input +// recency and not merely by "the newest attached client" (which happens to +// coincide with "typed last" unless a test goes out of its way to separate +// them). func TestCloseTUI_ReachesMostRecentlyActiveOnly(t *testing.T) { d, sock := overlayServerDaemon(t) d.session.CreateTab("T") // skip the real-PTY default workspace on first attach a := attachClientAs(t, sock, "A", 200, 50) waitUntil(t, "A attached", func() bool { return d.clientCount() == 1 }) - c := attachClientAs(t, sock, "C", 100, 30) - waitUntil(t, "C attached", func() bool { return d.clientCount() == 2 }) b := attachClientAs(t, sock, "B", 100, 30) - waitUntil(t, "B attached", func() bool { return d.clientCount() == 3 }) + waitUntil(t, "B attached", func() bool { return d.clientCount() == 2 }) + c := attachClientAs(t, sock, "C", 100, 30) + waitUntil(t, "C attached", func() bool { return d.clientCount() == 3 }) // C is newest attached - barrier(t, d, a, "A") + barrier(t, d, b, "B") barrier(t, d, c, "C") - barrier(t, d, b, "B") // B typed most recently + barrier(t, d, a, "A") // A — the OLDEST attached — types last bridge := dialBridge(t, sock) sendClientMsg(t, bridge, ipc.MsgCloseTUI, nil) - bGot := readFor(b, 500*time.Millisecond) - if countType(bGot, ipc.MsgCloseTUI) != 1 { - t.Fatalf("B (most recently active) got close_tui %d times, want 1: %v", countType(bGot, ipc.MsgCloseTUI), bGot) + aGot := readFor(a, 500*time.Millisecond) + if countType(aGot, ipc.MsgCloseTUI) != 1 { + t.Fatalf("A (typed last, though oldest attached) got close_tui %d times, want 1: %v", countType(aGot, ipc.MsgCloseTUI), aGot) } - aGot := readFor(a, 200*time.Millisecond) - if n := countType(aGot, ipc.MsgCloseTUI); n != 0 { - t.Errorf("A got close_tui %d times, want 0", n) + bGot := readFor(b, 200*time.Millisecond) + if n := countType(bGot, ipc.MsgCloseTUI); n != 0 { + t.Errorf("B got close_tui %d times, want 0", n) + } + cGot := readFor(c, 200*time.Millisecond) + if n := countType(cGot, ipc.MsgCloseTUI); n != 0 { + t.Errorf("C (newest attached, but did not type last) got close_tui %d times, want 0", n) + } +} + +// TestCloseTUI_NobodyTypedYetReachesOldestAttached: with no input at all, +// the implicit target is the OLDEST attached client — not the newest, which +// is mostRecentlyActiveConn's OWN "nobody typed" answer (see targetConn's +// doc comment for why the two deliberately disagree there). +func TestCloseTUI_NobodyTypedYetReachesOldestAttached(t *testing.T) { + d, sock := overlayServerDaemon(t) + d.session.CreateTab("T") + + a := attachClientAs(t, sock, "A", 200, 50) + waitUntil(t, "A attached", func() bool { return d.clientCount() == 1 }) + b := attachClientAs(t, sock, "B", 100, 30) + waitUntil(t, "B attached", func() bool { return d.clientCount() == 2 }) + c := attachClientAs(t, sock, "C", 100, 30) + waitUntil(t, "C attached", func() bool { return d.clientCount() == 3 }) + // Nobody has sent any input. + + bridge := dialBridge(t, sock) + sendClientMsg(t, bridge, ipc.MsgCloseTUI, nil) + + aGot := readFor(a, 500*time.Millisecond) + if countType(aGot, ipc.MsgCloseTUI) != 1 { + t.Fatalf("A (oldest attached, nobody has typed) got close_tui %d times, want 1: %v", countType(aGot, ipc.MsgCloseTUI), aGot) + } + bGot := readFor(b, 200*time.Millisecond) + if n := countType(bGot, ipc.MsgCloseTUI); n != 0 { + t.Errorf("B got close_tui %d times, want 0", n) } cGot := readFor(c, 200*time.Millisecond) if n := countType(cGot, ipc.MsgCloseTUI); n != 0 { @@ -118,7 +246,10 @@ func TestCloseTUI_ExplicitClient(t *testing.T) { // TestSetActivePane_FocusFrameToOneConn: the tab-switch broadcast reaches // every attached conn, and the set_active_pane focus frame reaches only the -// named client. +// named client. The workspace is switched to a DIFFERENT tab before either +// client attaches, so a workspace_state naming tab.ID as active can only be +// the broadcast this set_active_pane triggers — never attach-time noise — +// which lets this test read each conn exactly ONCE (no drain needed). func TestSetActivePane_FocusFrameToOneConn(t *testing.T) { d, sock := overlayServerDaemon(t) tab := d.session.CreateTab("T") @@ -126,15 +257,13 @@ func TestSetActivePane_FocusFrameToOneConn(t *testing.T) { if err != nil { t.Fatalf("create pane: %v", err) } + other := d.session.CreateTab("Other") + d.session.SwitchTab(other.ID) a := attachClientAs(t, sock, "A", 200, 50) waitUntil(t, "A attached", func() bool { return d.clientCount() == 1 }) b := attachClientAs(t, sock, "B", 100, 30) waitUntil(t, "B attached", func() bool { return d.clientCount() == 2 }) - // Drain each conn's own attach replay / other-client state frame before - // the assertions below, so the counts are about THIS set_active_pane. - readFor(a, 300*time.Millisecond) - readFor(b, 300*time.Millisecond) bridge := dialBridge(t, sock) sendClientMsg(t, bridge, ipc.MsgSetActivePane, ipc.SetActivePanePayload{PaneID: pane.ID, Client: "B"}) @@ -143,22 +272,24 @@ func TestSetActivePane_FocusFrameToOneConn(t *testing.T) { if countType(bGot, ipc.MsgSetActivePane) != 1 { t.Fatalf("B (named client) got set_active_pane %d times, want 1: %v", countType(bGot, ipc.MsgSetActivePane), bGot) } - if countType(bGot, ipc.MsgWorkspaceState) == 0 { - t.Error("B never saw the tab-switch broadcast") + if !sawActiveTab(t, bGot, tab.ID) { + t.Error("B never saw the tab-switch broadcast (no workspace_state named tab.ID active)") } - aGot := readFor(a, 200*time.Millisecond) + aGot := readFor(a, 300*time.Millisecond) if n := countType(aGot, ipc.MsgSetActivePane); n != 0 { t.Errorf("A got set_active_pane %d times, want 0", n) } - if countType(aGot, ipc.MsgWorkspaceState) == 0 { + if !sawActiveTab(t, aGot, tab.ID) { t.Error("A (not the named client) never saw the tab-switch broadcast — it must reach every attached conn") } } // TestMCPTargets_NoAttachedClient: Review Focus 4. A headless daemon (no -// attached client at all) must not panic on any of these, close_tui sends +// attached client at all) must not panic on any of these: close_tui sends // nothing, set_active_pane only switches the tab, and create_pane_req with -// an empty CWD falls all the way back to os.Getwd(). +// an empty CWD falls all the way back to os.Getwd(). The bridge conn that +// SENT close_tui/set_active_pane is itself read afterward to confirm the +// daemon did not echo either command back to its own sender. func TestMCPTargets_NoAttachedClient(t *testing.T) { d, sock := overlayServerDaemon(t) tab := d.session.CreateTab("T") @@ -177,6 +308,22 @@ func TestMCPTargets_NoAttachedClient(t *testing.T) { return d.session.ActiveTabID() == tab.ID }) + // Nobody is attached to receive either command, and the sender itself + // (an MCP bridge, never attached) must not have it echoed back either. + bridgeGot := readFor(bridge, 300*time.Millisecond) + if n := countType(bridgeGot, ipc.MsgCloseTUI); n != 0 { + t.Errorf("close_tui echoed back to its own sender %d times, want 0", n) + } + if n := countType(bridgeGot, ipc.MsgSetActivePane); n != 0 { + t.Errorf("set_active_pane echoed back to its own sender %d times, want 0", n) + } + // readFor's SetReadDeadline is still armed and has already elapsed — + // clear it, or the roundTrip calls below inherit that stale deadline and + // fail with a spurious i/o timeout on their very first Receive. + if err := bridge.SetReadDeadline(time.Time{}); err != nil { + t.Fatalf("clear read deadline: %v", err) + } + hostCWD, err := os.Getwd() if err != nil { t.Fatalf("Getwd: %v", err) @@ -206,8 +353,7 @@ func TestMCPTargets_NoAttachedClient(t *testing.T) { // default directory, and a bridge conn (no attach at all) falls back to the // size master's directory. Driven through handleBrowseDirReq — an empty Path // asks for defaultCWD(conn) and echoes it back as Resolved, with no PTY -// spawn — rather than through create_pane_req, whose behavior for the same -// resolver is separately covered by TestMCPTargets_NoAttachedClient. +// spawn. func TestDefaultCWD_PerClientAndBridge(t *testing.T) { d, sock := overlayServerDaemon(t) d.session.CreateTab("T") @@ -236,15 +382,177 @@ func TestDefaultCWD_PerClientAndBridge(t *testing.T) { } } +// TestCreatePaneReq_FromNonMasterClientUsesItsOwnCWD (Important 3): the +// CREATE path — not just the read-only browse path above — resolves against +// the REQUESTING conn's own cwd, even when that conn is not the size +// master. B is a follower here (A, the oldest attached, is master with +// dirA); a create sent on B's own conn must land in dirB, not in A's. +func TestCreatePaneReq_FromNonMasterClientUsesItsOwnCWD(t *testing.T) { + d, sock := overlayServerDaemon(t) + tab := d.session.CreateTab("T") + + dirA := resolvedTemp(t) + dirB := resolvedTemp(t) + attachClientWithCWD(t, sock, "A", 200, 50, dirA) + waitUntil(t, "A attached", func() bool { return d.clientCount() == 1 }) + b := attachClientWithCWD(t, sock, "B", 100, 30, dirB) + waitUntil(t, "B attached", func() bool { return d.clientCount() == 2 }) + if d.masterID() != "A" { + t.Fatalf("masterID = %q, want A", d.masterID()) + } + + resp := decodeInto[ipc.CreatePaneRespPayload](t, roundTrip(t, b, ipc.MsgCreatePaneReq, ipc.MsgCreatePaneResp, + ipc.CreatePaneReqPayload{TabID: tab.ID})) + if resp.Error != "" { + t.Fatalf("create_pane_req: %s", resp.Error) + } + pane := d.session.Pane(resp.PaneID) + if pane == nil { + t.Fatal("pane not created") + } + pane.PluginMu.Lock() + gotCWD := pane.CWD + pane.PluginMu.Unlock() + if gotCWD != dirB { + t.Errorf("create from B (a follower) landed in %q, want its own cwd %q (not A's master cwd %q)", gotCWD, dirB, dirA) + } +} + +// TestDefaultCWD_SameClientProbedOnce (Important 1): in the single-TUI case +// the conn, master and most-recently-active steps all name the SAME client, +// so a dead directory must be probed exactly once — not three times over, +// each paying its own spawnDirProbeTimeout and abandoning its own +// claimBlockingFSCall permit. +func TestDefaultCWD_SameClientProbedOnce(t *testing.T) { + d, sock := overlayServerDaemon(t) + d.session.CreateTab("T") + + var calls atomic.Int64 + block := make(chan struct{}) + orig := statPath + statPath = func(string) (os.FileInfo, error) { + calls.Add(1) + <-block + return nil, os.ErrNotExist + } + t.Cleanup(func() { restoreSeam(t, block, func() { statPath = orig }) }) + + attachClientWithCWD(t, sock, "A", 200, 50, "/mnt/dead-share/work") + waitUntil(t, "A attached", func() bool { return d.clientCount() == 1 }) + master := d.masterConn() + if master == nil { + t.Fatal("A did not become master") + } + + done := make(chan string, 1) + go func() { done <- d.defaultCWD(master) }() + + select { + case got := <-done: + if got == "/mnt/dead-share/work" { + t.Errorf("defaultCWD returned the unreachable path %q", got) + } + case <-time.After(10 * time.Second): + t.Fatal("defaultCWD did not return within 10s — probably retrying the same dead path serially") + } + + if n := calls.Load(); n != 1 { + t.Errorf("statPath called %d times, want exactly 1 (conn, master and most-recently-active all name the same client)", n) + } +} + +// TestDefaultCWD_DedupesIdenticalPathsEvenWithBudgetToSpare isolates the +// DEDUP half of the Important-1 fix from the shared-deadline half above: a +// FAST-failing stat leaves the shared deadline almost entirely unspent, so +// only the "skip a candidate already tried" check — not the deadline +// running out — can be what stops a second and third identical probe here. +func TestDefaultCWD_DedupesIdenticalPathsEvenWithBudgetToSpare(t *testing.T) { + d, sock := overlayServerDaemon(t) + d.session.CreateTab("T") + + var calls atomic.Int64 + orig := statPath + statPath = func(string) (os.FileInfo, error) { + calls.Add(1) + return nil, os.ErrNotExist + } + t.Cleanup(func() { statPath = orig }) + + attachClientWithCWD(t, sock, "A", 200, 50, "/mnt/dead-share/work") + waitUntil(t, "A attached", func() bool { return d.clientCount() == 1 }) + master := d.masterConn() + if master == nil { + t.Fatal("A did not become master") + } + + got := d.defaultCWD(master) + if got == "/mnt/dead-share/work" { + t.Errorf("defaultCWD returned the unreachable path %q", got) + } + if n := calls.Load(); n != 1 { + t.Errorf("statPath called %d times, want exactly 1 — the conn, master and most-recently-active steps all name A's own path", n) + } +} + +// TestDefaultCWD_SharesOneDeadlineAcrossDifferentDeadCandidates isolates the +// SHARED-DEADLINE half: three attached clients with three DIFFERENT +// unreachable directories (so dedup-by-path cannot collapse them) must still +// cost this call no more than roughly ONE spawnDirProbeTimeout, not three +// paid serially. +func TestDefaultCWD_SharesOneDeadlineAcrossDifferentDeadCandidates(t *testing.T) { + d, sock := overlayServerDaemon(t) + d.session.CreateTab("T") + + block := make(chan struct{}) + orig := statPath + statPath = func(string) (os.FileInfo, error) { + <-block + return nil, os.ErrNotExist + } + t.Cleanup(func() { restoreSeam(t, block, func() { statPath = orig }) }) + + attachClientWithCWD(t, sock, "A", 200, 50, "/mnt/dead-a") // oldest attached: becomes master + waitUntil(t, "A attached", func() bool { return d.clientCount() == 1 }) + attachClientWithCWD(t, sock, "B", 100, 30, "/mnt/dead-b") + waitUntil(t, "B attached", func() bool { return d.clientCount() == 2 }) + c := attachClientWithCWD(t, sock, "C", 100, 30, "/mnt/dead-c") + waitUntil(t, "C attached", func() bool { return d.clientCount() == 3 }) + barrier(t, d, c, "C") // C is now the most-recently-active client + + // The requesting conn is B: neither the master (A) nor the + // most-recently-active client (C), so all three steps of defaultCWD name + // three DIFFERENT dead directories. + rec, ok := clientRecordByID(d, "B") + if !ok { + t.Fatal("B not found in the registry") + } + + start := time.Now() + got := d.defaultCWD(rec.conn) + elapsed := time.Since(start) + + for _, dead := range []string{"/mnt/dead-a", "/mnt/dead-b", "/mnt/dead-c"} { + if got == dead { + t.Errorf("defaultCWD returned the unreachable path %q", got) + } + } + // Worst case without a shared deadline is 3 * spawnDirProbeTimeout (6s); + // this bound sits well below that and comfortably above the ~1x a shared + // deadline costs, so it separates the two without being timing-fragile. + if elapsed > 3*time.Second { + t.Errorf("defaultCWD took %s across 3 different dead candidates, want well under 3x spawnDirProbeTimeout (the deadline must be shared)", elapsed) + } +} + // TestDismiss_BroadcastsEventDismissed: a dismissal reaches every attached // client, so a card dismissed through one TUI's sidebar disappears from a -// second one too. +// second one too. event_dismissed never appears as ordinary attach noise, so +// one read after the send is enough. func TestDismiss_BroadcastsEventDismissed(t *testing.T) { d, sock := overlayServerDaemon(t) d.session.CreateTab("T") a := attachClientAs(t, sock, "A", 200, 50) waitUntil(t, "A attached", func() bool { return d.clientCount() == 1 }) - readFor(a, 300*time.Millisecond) // drain the attach state frame bridge := dialBridge(t, sock) sendClientMsg(t, bridge, ipc.MsgDismissEvent, ipc.DismissEventPayload{EventID: "evt-1"}) @@ -270,6 +578,13 @@ func TestDismiss_BroadcastsEventDismissed(t *testing.T) { // TestPaneSeen_OnlyOnTrueToFalse: false→false and true→true send no // pane_seen frame; only a true→false transition does, exactly once. +// +// Each update_pane is sent WITH an envelope id and checked via readUntilID, +// which waits for that update's own pane_op_resp (sent, unconditionally, +// AFTER handleUpdatePane returns — so any pane_seen broadcast the same +// update triggered is already queued ahead of it on this same conn). That +// makes three checkpoints on ONE conn safe: readUntilID sets no read +// deadline, unlike calling readFor three times over on the same conn. func TestPaneSeen_OnlyOnTrueToFalse(t *testing.T) { d, sock := overlayServerDaemon(t) tab := d.session.CreateTab("T") @@ -279,16 +594,18 @@ func TestPaneSeen_OnlyOnTrueToFalse(t *testing.T) { } a := attachClientAs(t, sock, "A", 200, 50) waitUntil(t, "A attached", func() bool { return d.clientCount() == 1 }) - readFor(a, 300*time.Millisecond) // drain the attach state frame setUnseen := func(v bool) { pane.PluginMu.Lock() pane.Unseen = v pane.PluginMu.Unlock() } + step := 0 sendUnseen := func(v bool) []*ipc.Message { - sendClientMsg(t, a, ipc.MsgUpdatePane, ipc.UpdatePanePayload{PaneID: pane.ID, Unseen: &v}) - return readFor(a, 300*time.Millisecond) + step++ + id := fmt.Sprintf("pane-seen-step-%d", step) + sendWithID(t, a, ipc.MsgUpdatePane, id, ipc.UpdatePanePayload{PaneID: pane.ID, Unseen: &v}) + return readUntilID(t, a, ipc.MsgPaneOpResp, id, 3*time.Second) } setUnseen(false) @@ -374,7 +691,9 @@ func TestListClients_Fields(t *testing.T) { // actually acted on may stamp the sending client's last-input time — // otherwise a pane silently reporting its own CWD makes an idle client look // like the one somebody is driving, which is exactly what targetConn's -// implicit fallback reads to pick a client. +// implicit fallback reads to pick a client. Covers every user-originated +// field (Name, Muted, Eager, PinnedAttention, MarkedForDeletion) and every +// automatic one (CWD, OverlayVisible, Unseen). func TestUpdatePane_LastInputStampsOnlyUserFields(t *testing.T) { d, sock := overlayServerDaemon(t) tab := d.session.CreateTab("T") @@ -385,7 +704,7 @@ func TestUpdatePane_LastInputStampsOnlyUserFields(t *testing.T) { a := attachClientAs(t, sock, "A", 200, 50) waitUntil(t, "A attached", func() bool { return d.clientCount() == 1 }) - fenceMarker := 40 // eligible geometry, and distinct from A's 200x50 attach + fenceMarker := 40 // eligible geometry, distinct from A's 200x50 attach fence := func() { t.Helper() fenceMarker++ @@ -395,9 +714,19 @@ func TestUpdatePane_LastInputStampsOnlyUserFields(t *testing.T) { return rec.cols == fenceMarker && rec.rows == fenceMarker }) } + clearStamp := func() { + d.clients.mu.Lock() + defer d.clients.mu.Unlock() + for _, rec := range d.clients.byConn { + if rec.id == "A" { + rec.lastInputAt = time.Time{} + } + } + } notStamped := func(name string, payload ipc.UpdatePanePayload) { t.Helper() + clearStamp() payload.PaneID = pane.ID sendClientMsg(t, a, ipc.MsgUpdatePane, payload) // client_geometry never touches lastInputAt (see the dispatch table in @@ -409,14 +738,24 @@ func TestUpdatePane_LastInputStampsOnlyUserFields(t *testing.T) { t.Errorf("%s stamped lastInputAt, want it to stay unstamped", name) } } + stamped := func(name string, payload ipc.UpdatePanePayload) { + t.Helper() + clearStamp() + payload.PaneID = pane.ID + sendClientMsg(t, a, ipc.MsgUpdatePane, payload) + waitUntil(t, name+" stamps lastInputAt", func() bool { + rec, _ := clientRecordByID(d, "A") + return !rec.lastInputAt.IsZero() + }) + } notStamped("CWD-only", ipc.UpdatePanePayload{CWD: t.TempDir()}) notStamped("OverlayVisible-only", ipc.UpdatePanePayload{OverlayVisible: boolPtr(true)}) notStamped("Unseen-only", ipc.UpdatePanePayload{Unseen: boolPtr(true)}) - sendClientMsg(t, a, ipc.MsgUpdatePane, ipc.UpdatePanePayload{PaneID: pane.ID, Name: "renamed"}) - waitUntil(t, "Name stamps lastInputAt", func() bool { - rec, _ := clientRecordByID(d, "A") - return !rec.lastInputAt.IsZero() - }) + stamped("Name", ipc.UpdatePanePayload{Name: "renamed"}) + stamped("Muted", ipc.UpdatePanePayload{Muted: boolPtr(true)}) + stamped("Eager", ipc.UpdatePanePayload{Eager: boolPtr(true)}) + stamped("PinnedAttention", ipc.UpdatePanePayload{PinnedAttention: boolPtr(true)}) + stamped("MarkedForDeletion", ipc.UpdatePanePayload{MarkedForDeletion: boolPtr(true)}) } diff --git a/internal/ipc/protocol.go b/internal/ipc/protocol.go index f6960b02..129c09a8 100644 --- a/internal/ipc/protocol.go +++ b/internal/ipc/protocol.go @@ -1117,11 +1117,14 @@ type DestroyPaneRespPayload struct { type SetActivePanePayload struct { PaneID string `json:"pane_id"` - // Client names which attached client this applies to. Empty keeps the - // historical broadcast-to-every-TUI behavior, which is what every - // existing producer (MCP's set_active_pane) sends and what a headless - // daemon with no attached client needs: with nobody attached, this only - // switches the tab and there is no client to target. + // Client names which ONE attached client the focus frame reaches — never + // every attached TUI. Empty means the IMPLICIT target: the client with + // the most recent input, or the oldest attached client while nobody has + // typed yet (Daemon.targetConn). Every existing producer (MCP's + // set_active_pane before this field existed) sent it empty, so an older + // caller keeps landing on a reasonable client rather than nothing. A + // headless daemon with no attached client only switches the tab; there + // is no client to target and nothing is sent. Client string `json:"client,omitempty"` } @@ -1129,11 +1132,15 @@ type HighlightPanePayload struct { PaneID string `json:"pane_id"` } -// CloseTUIPayload asks one specific client to exit, for multi-client sync -// (e.g. the master asking a follower to close, or an admin action against one -// client in the list). Client empty keeps the historical behavior of -// MsgCloseTUI: broadcast to every attached TUI. A headless daemon with no -// attached client sends nothing and must not panic. +// CloseTUIPayload asks ONE client to exit — never every attached TUI, for the +// same reason SetActivePanePayload.Client is scoped: closing a window is an +// action against a specific client, not a daemon-wide broadcast. Client +// empty means the IMPLICIT target: the client with the most recent input, or +// the oldest attached client while nobody has typed yet (Daemon.targetConn). +// Client naming an id that is not attached reaches nobody, logged rather +// than guessed at. A headless daemon with no attached client sends nothing +// and must not panic. Absent entirely (nil payload) is treated the same as +// empty, because older bridges send nil. type CloseTUIPayload struct { Client string `json:"client,omitempty"` } @@ -1168,18 +1175,22 @@ type DismissEventPayload struct { EventID string `json:"event_id"` // empty = dismiss all } -// EventDismissedPayload is the daemon's broadcast of a dismissal to every -// OTHER attached client, so a notification acted on in one client's sidebar -// does not also sit there in a second one. Mirrors DismissEventPayload's +// EventDismissedPayload is the daemon's broadcast of a dismissal to EVERY +// attached client, the sender included, so a notification acted on in one +// client's sidebar does not also sit there in a second one. The sender's own +// echo is harmless — its sidebar already dismissed the same event locally, +// and applying the mark again is idempotent. Mirrors DismissEventPayload's // "" = all convention rather than reusing the type, because the two travel in // opposite directions and one is a request while the other is a fact. type EventDismissedPayload struct { EventID string `json:"event_id"` // "" = all } -// PaneSeenPayload is the daemon's broadcast marking a pane as looked-at by -// some client, so every OTHER client's sidebar clears the same "finished -// while you were away" mark rather than each client tracking it alone. +// PaneSeenPayload is the daemon's broadcast marking a pane as looked-at, +// reaching EVERY attached client including the one that cleared the mark, so +// every client's sidebar clears the same "finished while you were away" mark +// rather than each one tracking it alone. The sender's own echo is a no-op — +// its sidebar already cleared the mark locally before reporting it. type PaneSeenPayload struct { PaneID string `json:"pane_id"` } diff --git a/internal/tui/pane.go b/internal/tui/pane.go index 3ebd5609..d685ac29 100644 --- a/internal/tui/pane.go +++ b/internal/tui/pane.go @@ -451,7 +451,7 @@ const minAdaptiveScrollbackLines = 2000 // goroutine in production, but a plain int makes every parallel test in the // package racy against any other that builds a pane, which the detector reports // as a failure of whichever pair it happens to catch. Same reasoning as -// Daemon.clientCWD's atomic.Pointer. +// Daemon.clientSize's atomic.Pointer. var explicitScrollback atomic.Int64 // knownPaneCount is the workspace size the adaptive depth divides. Published by From 9731a1c9fa6455d42df124734bb6167052ce6eda Mon Sep 17 00:00:00 2001 From: Artjoms Stukans Date: Fri, 25 Sep 2026 20:44:54 +0200 Subject: [PATCH 11/40] feat(tui): follow the daemon's size master MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Give the TUI a stable per-process client id (sent on every attach) and track each destination's size master and attached-client count from its broadcasts. A follower stops driving PTY sizes: resizeAllPanes, diffResizes and overlayResizeCmd — the three producers of pane resizes — each gate on isFollower and send nothing for a destination this client does not own, while diffResizes also leaves sizedOnce untouched so a later election still owes every pane its first-resize kick. Batch every resize pass into one MsgResizePanes frame per destination instead of one MsgResizePane per pane, so a window resize or a split-drag release across dozens of panes can never put one must-deliver frame per pane on a follower's queue. Report this client's own raw window size (client_geometry) after every resize pass, to every connected destination, so a master that shrinks below the paintable floor is noticed and handed off. On becoming a destination's master, clear its sizedOnce entries and resize every pane at once, since the sizes a previous master or follower state left behind are not this client's own. Add Take control (client.take_control): an early-tier keymap action with no default binding, plus a palette command, that asks the active destination's daemon to make this client the master immediately. Send a clean-exit detach before closing each connection, so a normal quit does not cost the next master a lost-link grace period. Show a [master]/[follower] marker in the status bar once a destination has more than one attached client. --- internal/keymap/action.go | 9 + internal/keymap/action_test.go | 15 +- internal/tui/broadcast_echo_test.go | 45 +-- internal/tui/keyspecs_test.go | 3 + internal/tui/mcp_paint_test.go | 22 +- internal/tui/model.go | 251 +++++++++++++++- internal/tui/multiclient_role_test.go | 409 ++++++++++++++++++++++++++ internal/tui/overlay.go | 19 +- internal/tui/palette.go | 5 + internal/tui/reconnect.go | 27 ++ internal/tui/sidebar_test.go | 14 +- internal/tui/splitdrag_test.go | 10 +- internal/tui/tinyterm_test.go | 28 +- 13 files changed, 785 insertions(+), 72 deletions(-) create mode 100644 internal/tui/multiclient_role_test.go diff --git a/internal/keymap/action.go b/internal/keymap/action.go index 5b756b0c..b95dd9bf 100644 --- a/internal/keymap/action.go +++ b/internal/keymap/action.go @@ -58,6 +58,15 @@ var registry = []Action{ // project. Early tier with the rest of the project keys. {ID: "project.move_up", Label: "Move project up", Group: "Projects", Tier: TierEarly, Order: 1800, Default: "alt+shift+up"}, {ID: "project.move_down", Label: "Move project down", Group: "Projects", Tier: TierEarly, Order: 1900, Default: "alt+shift+down"}, + // Multi-client sync (D6): makes this client the size master of the active + // destination at once. Grouped with System, where the other daemon/window + // actions live, rather than with Projects or Panes — it names no project + // or pane, it names a CLIENT. Early tier so it wins over a plugin's + // raw_keys claim on whatever chord a user binds it to, like every other + // early action; no default chord, because taking control from a running + // follower resizes every one of its panes, which must never happen to a + // key a fresh install already owns. + {ID: "client.take_control", Label: "Take control (size master)", Group: "System", Tier: TierEarly, Order: 1950, Default: ""}, // --- Late tier: handleKey's second switch, after tryPluginRawKey --- {ID: "app.quit", Label: "Quit", Group: "System", Tier: TierLate, Order: 2000, Default: "ctrl+q"}, diff --git a/internal/keymap/action_test.go b/internal/keymap/action_test.go index fc4a375f..be3845bc 100644 --- a/internal/keymap/action_test.go +++ b/internal/keymap/action_test.go @@ -4,8 +4,8 @@ import "testing" func TestActions_RegistryIntegrity(t *testing.T) { acts := Actions() - if len(acts) != 66 { - t.Fatalf("registry has %d actions, want 66 (42 config-backed + 12 promoted from the reserved-key switch + 4 reorder + 6 tab layout + 2 project groups)", len(acts)) + if len(acts) != 67 { + t.Fatalf("registry has %d actions, want 67 (42 config-backed + 12 promoted from the reserved-key switch + 4 reorder + 6 tab layout + 2 project groups + 1 multi-client sync)", len(acts)) } seen := make(map[ActionID]bool, len(acts)) orders := make(map[int]ActionID, len(acts)) @@ -40,6 +40,12 @@ func TestActions_TierSplitMatchesLegacySwitches(t *testing.T) { // and command_history, which is the early switch. The two overlay toggles // share one slot per tab, so splitting them across the seam would let a // plugin's raw_keys claim one of the pair and not the other. + // + // client.take_control has no pre-rewrite position either — it is a + // multi-client-sync action with no legacy handleKey switch at all. It is + // early tier by design (see its registry comment in action.go): a chord + // bound to it must win over a plugin's raw_keys claim, like every other + // early action. early := map[ActionID]bool{ "notification.toggle": true, "notification.focus": true, "sidebar.toggle": true, "pane.go_back": true, "pane.mute": true, @@ -51,9 +57,10 @@ func TestActions_TierSplitMatchesLegacySwitches(t *testing.T) { "project.next": true, "project.prev": true, "project.toggle": true, "project.attention_queue": true, "project.move_up": true, "project.move_down": true, + "client.take_control": true, } - if len(early) != 20 { - t.Fatalf("expected-early table has %d entries, want 20", len(early)) + if len(early) != 21 { + t.Fatalf("expected-early table has %d entries, want 21", len(early)) } for _, a := range Actions() { want := TierLate diff --git a/internal/tui/broadcast_echo_test.go b/internal/tui/broadcast_echo_test.go index c5fe6b45..732df783 100644 --- a/internal/tui/broadcast_echo_test.go +++ b/internal/tui/broadcast_echo_test.go @@ -203,8 +203,11 @@ func sentCounts(fs *echoRecorder) (layouts, resizes int) { switch msg.Type { case ipc.MsgUpdateLayout: layouts++ - case ipc.MsgResizePane: - resizes++ + case ipc.MsgResizePanes: + var p ipc.ResizePanesPayload + if err := json.Unmarshal(msg.Payload, &p); err == nil { + resizes += len(p.Panes) + } } } return layouts, resizes @@ -266,7 +269,7 @@ func TestWorkspaceState_FirstResizeAfterAttach_IsAlwaysSent(t *testing.T) { _, resizes := sentCounts(fs) if resizes != 3 { - t.Errorf("MsgResizePane count = %d, want 3 (one per pane) — the daemon "+ + t.Errorf("resize count = %d, want 3 (one per pane) — the daemon "+ "zeroes its applied-size guard on PTY install and needs the first "+ "client resize to kick a repaint", resizes) } @@ -294,7 +297,7 @@ func TestWorkspaceState_UnchangedSizes_SendNoResizeOnRepeat(t *testing.T) { _, resizes := sentCounts(fs) if resizes != 0 { - t.Errorf("MsgResizePane count = %d, want 0 on a repeat broadcast — "+ + t.Errorf("resize count = %d, want 0 on a repeat broadcast — "+ "every pane already has the size the broadcast reports", resizes) } } @@ -468,17 +471,19 @@ func TestWorkspaceState_PendingPane_IsNotResized(t *testing.T) { runCmd(cmd) for _, msg := range fs.sent { - if msg.Type != ipc.MsgResizePane { + if msg.Type != ipc.MsgResizePanes { continue } - var p ipc.ResizePanePayload - if err := json.Unmarshal(msg.Payload, &p); err != nil { - t.Fatalf("decode resize payload: %v", err) + var batch ipc.ResizePanesPayload + if err := json.Unmarshal(msg.Payload, &batch); err != nil { + t.Fatalf("decode resize_panes payload: %v", err) } - if p.PaneID == echo.Panes[2].ID { - t.Errorf("resized deferred pane %s — the daemon drops it (nil PTY) "+ - "and never records the size, so this repeats every broadcast "+ - "forever", p.PaneID) + for _, p := range batch.Panes { + if p.PaneID == echo.Panes[2].ID { + t.Errorf("resized deferred pane %s — the daemon drops it (nil PTY) "+ + "and never records the size, so this repeats every broadcast "+ + "forever", p.PaneID) + } } } @@ -493,15 +498,17 @@ func TestWorkspaceState_PendingPane_IsNotResized(t *testing.T) { var sawSpawned bool for _, msg := range fs2.sent { - if msg.Type != ipc.MsgResizePane { + if msg.Type != ipc.MsgResizePanes { continue } - var p ipc.ResizePanePayload - if err := json.Unmarshal(msg.Payload, &p); err != nil { - t.Fatalf("decode resize payload: %v", err) + var batch ipc.ResizePanesPayload + if err := json.Unmarshal(msg.Payload, &batch); err != nil { + t.Fatalf("decode resize_panes payload: %v", err) } - if p.PaneID == echo.Panes[2].ID { - sawSpawned = true + for _, p := range batch.Panes { + if p.PaneID == echo.Panes[2].ID { + sawSpawned = true + } } } if !sawSpawned { @@ -593,7 +600,7 @@ func TestReattach_ReArmsTheFirstResizeKick(t *testing.T) { _, resizes := sentCounts(fs) if resizes != 3 { - t.Errorf("MsgResizePane count = %d after reattach, want 3 — the daemon "+ + t.Errorf("resize count = %d after reattach, want 3 — the daemon "+ "zeroed its guard on PTY install, so the suppression state from "+ "before the outage describes a daemon that no longer exists", resizes) } diff --git a/internal/tui/keyspecs_test.go b/internal/tui/keyspecs_test.go index 7828def8..e0f1b2ec 100644 --- a/internal/tui/keyspecs_test.go +++ b/internal/tui/keyspecs_test.go @@ -30,6 +30,9 @@ var promotedActions = map[keymap.ActionID]bool{ "tab.layout_grid": true, "tab.layout_main": true, "tab.layout_spiral": true, // The project-group actions arrived after bindings.toml too. "project.group_toggle": true, "project.groups_collapse_all": true, + // client.take_control is new with multi-client sync and never had a + // [keybindings] field either. + "client.take_control": true, } func TestKeySpecsFromConfig_MapsEveryConfigBackedAction(t *testing.T) { diff --git a/internal/tui/mcp_paint_test.go b/internal/tui/mcp_paint_test.go index dbef55d3..7bf30f4d 100644 --- a/internal/tui/mcp_paint_test.go +++ b/internal/tui/mcp_paint_test.go @@ -46,20 +46,22 @@ func TestUpdate_MCPHiddenPaneDimensions(t *testing.T) { } resized := false for _, msg := range fs.sent { - if msg.Type != ipc.MsgResizePane { + if msg.Type != ipc.MsgResizePanes { continue } - var it ipc.ResizePanePayload - if err := msg.DecodePayload(&it); err != nil { + var batch ipc.ResizePanesPayload + if err := msg.DecodePayload(&batch); err != nil { t.Fatal(err) } - if it.PaneID != p.ID { - continue - } - resized = true - t.Logf("hidden: broadcast=80x24 emulator=%dx%d resize=%dx%d", p.vt.Width(), p.vt.Height(), it.Cols, it.Rows) - if p.vt.Width() != int(it.Cols) || p.vt.Height() != int(it.Rows) { - t.Fatal("emulator/PTY resize split") + for _, it := range batch.Panes { + if it.PaneID != p.ID { + continue + } + resized = true + t.Logf("hidden: broadcast=80x24 emulator=%dx%d resize=%dx%d", p.vt.Width(), p.vt.Height(), it.Cols, it.Rows) + if p.vt.Width() != int(it.Cols) || p.vt.Height() != int(it.Rows) { + t.Fatal("emulator/PTY resize split") + } } } if !resized { diff --git a/internal/tui/model.go b/internal/tui/model.go index 57429292..05a4698b 100644 --- a/internal/tui/model.go +++ b/internal/tui/model.go @@ -20,6 +20,7 @@ import ( tea "charm.land/bubbletea/v2" "charm.land/lipgloss/v2" "github.com/charmbracelet/x/ansi" + "github.com/google/uuid" "github.com/artyomsv/quil/internal/changelog" "github.com/artyomsv/quil/internal/claudesessions" @@ -67,6 +68,13 @@ type WorkspaceStateMsg struct { Dest string // Update is the daemon's announced newer release (nil when up to date). Update *ipc.UpdateInfo + // SizeMaster is this destination's size-master client id, or "" when it + // has none. Clients is the number of clients currently attached to it + // (bridges excluded). Both ride every broadcast (buildWorkspaceState), + // which is what keeps Model.sizeMaster/clientCount current with no + // dedicated round trip — see isFollower. + SizeMaster string + Clients int } // ProjectInfo is one daemon-side project as broadcast. TabIDs carries the @@ -465,7 +473,22 @@ type Model struct { // daemon's pane consume another's kick and let armReattachReset for one // dest clear the other's flag. Two daemons minting the same UUID is not a // realistic accident, but the invariant should not rest on that. - sizedOnce map[string]bool + sizedOnce map[string]bool + // clientID identifies this PROCESS across reconnects — minted once in + // NewModel with uuid.NewString() and sent on every attach (attachMessage). + // It is never persisted to disk: two TUIs on one machine would then share + // it, and each is a distinct client to the daemon's master election. + clientID string + // sizeMaster records, per destination, the master client's id reported by + // the last broadcast ("" = no master on that destination). isFollower + // derives from it: this client is a follower of dest whenever sizeMaster + // names someone else. Updated in applyWorkspaceState from + // WorkspaceStateMsg.SizeMaster. + sizeMaster map[string]string + // clientCount records, per destination, the last broadcast's attached- + // client count (bridges excluded) — what renderStatusBar's role marker + // and D9's "most recent input" default both key off of at the TUI layer. + clientCount map[string]int renaming bool renameInput string renamingPane bool @@ -1146,10 +1169,14 @@ func (m *Model) SetRecentCWDs(list []string) { m.recentCWDs = list } // what had to move.) Nil when there is nothing to show. func NewModel(client Client, cfg config.Config, version string, registry *plugin.Registry, stalePlugins []plugin.StalePlugin, whatsNew *changelog.Window) Model { m := Model{ - client: client, - cfg: cfg, - version: version, - devMode: os.Getenv("QUIL_HOME") != "", + client: client, + cfg: cfg, + // Minted once per process, per D3: stable across this process's own + // reconnects (it never changes after this), but a NEW process — a + // closed and relaunched TUI — is a new client with a new id. + clientID: uuid.NewString(), + version: version, + devMode: os.Getenv("QUIL_HOME") != "", // See the field comment: a terminal with no focus reporting never // corrects this, and assuming focused is the quiet failure. termFocused: true, @@ -1198,6 +1225,23 @@ func (m *Model) initKeymap() { m.keymap, m.keyConflicts = buildKeymap(m.cfg.Keybindings) } +// SetClientID overrides the process-minted client id. A test seam: production +// never needs a stable id across separate NewModel calls, but a test driving +// two Models as two "clients" of one daemon needs to give them distinct, +// known ids rather than two random UUIDs it cannot assert against. +func (m *Model) SetClientID(id string) { m.clientID = id } + +// isFollower reports whether this client is NOT the size master of dest, and +// there IS a master — see D4. A dest this client has never seen a broadcast +// for (m.sizeMaster is nil, or holds no entry for it) answers false: the zero +// value of "no master reported yet" must behave exactly like "no follower +// gate applies", which is what every pre-multi-client-sync test and every +// single-client session already assumes. +func (m *Model) isFollower(dest string) bool { + master := m.sizeMaster[dest] + return master != "" && master != m.clientID +} + // WindowSize returns the last known window dimensions for persistence. func (m Model) WindowSize() (width, height int) { return m.lastWidth, m.lastHeight @@ -1538,7 +1582,8 @@ func (m Model) Update(msg tea.Msg) (retModel tea.Model, retCmd tea.Cmd) { // statement. The same hazard the ledger comment above documents. m.promptNextUpgrade() resize, attach, wake := m.resizeAllPanes(), m.attachAllDests(), m.wakeOfflineDests() - return m, tea.Batch(resize, attach, wake) + geom := m.clientGeometryCmd() + return m, tea.Batch(resize, attach, wake, geom) } // A destination can join the router after the first resize (a host that @@ -1726,6 +1771,12 @@ func (m Model) Update(msg tea.Msg) (retModel tea.Model, retCmd tea.Cmd) { // Also resize an active overlay pane so the daemon's PTY tracks the new size. var overlayCmds []tea.Cmd overlayCmds = append(overlayCmds, m.resizeAllPanes()) + // The debounced report of THIS client's own window size, to every + // connected destination — see clientGeometryCmd. Riding the same + // debounce point as resizeAllPanes rather than the raw tea.WindowSizeMsg + // is deliberate: a resize burst (dragging the window edge) would + // otherwise cost one client_geometry per intermediate size report. + overlayCmds = append(overlayCmds, m.clientGeometryCmd()) // EVERY tab, not just the active one. An overlay pane sits outside the // layout tree, so resizeAllPanes never walks it (it iterates // tab.Leaves()) and diffResizes keeps no sizedOnce ledger for it — @@ -5316,6 +5367,8 @@ func (m Model) handleKey(msg tea.KeyPressMsg) (tea.Model, tea.Cmd) { // jumpToNextBlocked mutates m through a pointer receiver. cmd := m.jumpToNextBlocked() return m, cmd + case "client.take_control": + return m, m.sendTakeControl(m.activeDest()) } // Everything from here to the late-tier lookup is skipped for a completed @@ -6021,6 +6074,40 @@ func (m *Model) applyWorkspaceState(state WorkspaceStateMsg, dest string) ([]str // destination, and it self-skips when nothing changed. m.cacheRemoteProjects(dest) + // Multi-client sync (§4.2): this destination's size master and attached- + // client count. Read the PREVIOUS master before overwriting it — becoming + // master is a TRANSITION, not a state, and the resize kick below must fire + // once, on the broadcast that flips it, never on every later broadcast + // that merely reconfirms it. + var prevMaster string + if m.sizeMaster != nil { + prevMaster = m.sizeMaster[dest] + } + if m.sizeMaster == nil { + m.sizeMaster = make(map[string]string) + } + m.sizeMaster[dest] = state.SizeMaster + if m.clientCount == nil { + m.clientCount = make(map[string]int) + } + m.clientCount[dest] = state.Clients + // The leading state.SizeMaster != "" guard matters on its own: without it, + // an empty m.clientID (never set — every real Model gets one from + // NewModel, but a bare Model literal in a test does not) would equal an + // empty state.SizeMaster ("no master on dest"), and losing a master would + // misread as this client becoming one. + if state.SizeMaster != "" && state.SizeMaster == m.clientID && prevMaster != state.SizeMaster { + // This client just became dest's size master. Every pane's last + // resize on dest was sent by whoever was master before (or by nobody, + // if there was none) — sizedOnce still reports "already sized" for + // sizes THIS client never sent, so the diff in resizeAllPanes/ + // diffResizes would suppress the very sizes that just became + // authoritative. Clearing it re-arms the same first-resize kick + // armReattachReset re-arms after a reattach. + m.clearSizedOnceForDest(dest) + overlayResizeCmds = append(overlayResizeCmds, m.resizeAllPanes()) + } + // Dispose panes that did not survive reconciliation — both panes pruned // from surviving tabs and every pane of tabs the daemon dropped. Without // this, each removed pane leaks its VT emulator (drain goroutine + @@ -7158,6 +7245,19 @@ func (m Model) renderStatusBar() string { if m.devMode { right = "[dev] " + right } + // Multi-client sync (§4.3): the role marker sits beside [dev], in the same + // style, because it says something about how THIS process relates to the + // workspace rather than about any one pane or project. Shown only once a + // second client exists — with one client the question "who is the master" + // has exactly one uninteresting answer, and showing it on every ordinary + // single-client session would be noise nobody asked for. See isFollower. + if dest := m.activeDest(); m.clientCount[dest] >= 2 { + if m.isFollower(dest) { + right = "[follower] " + right + } else { + right = "[master] " + right + } + } // Placed after [dev] so it renders leftmost of the two, i.e. first in // reading order: which MACHINE you are driving outranks which build you // are running. Without it the status bar is identical whether the panes @@ -7393,9 +7493,10 @@ func (m Model) attachMessage(dest string) *ipc.Message { // Best-effort; if Getwd fails the daemon falls back to its own CWD. localCWD, _ := os.Getwd() msg, _ := ipc.NewMessage(ipc.MsgAttach, ipc.AttachPayload{ - Cols: cols, - Rows: rows, - CWD: attachCWD(dest, localCWD), + Cols: cols, + Rows: rows, + CWD: attachCWD(dest, localCWD), + ClientID: m.clientID, }) return msg } @@ -7739,6 +7840,16 @@ func parseWorkspaceState(raw map[string]any) WorkspaceStateMsg { if ap, ok := raw["active_project"].(string); ok { state.ActiveProject = ap } + // Multi-client sync: size_master ("" = no master) and clients (attached + // count, bridges excluded) — see buildWorkspaceState on the daemon side. + // Absent on an older daemon, which leaves both at their zero values, and + // isFollower already treats "" as "not a follower". + if sm, ok := raw["size_master"].(string); ok { + state.SizeMaster = sm + } + if c, ok := raw["clients"].(float64); ok { + state.Clients = int(c) + } if projects, ok := raw["projects"].([]any); ok { for _, p := range projects { pm, ok := p.(map[string]any) @@ -9044,15 +9155,90 @@ func (m *Model) terminalPaintable() bool { return m.width >= minTermWidth && m.height >= minTermHeight } +// clientGeometryCmd reports this client's own RAW window size to every +// connected destination (§3.5's MsgClientGeometry) — 0x0 when +// !terminalPaintable(), matching attachMessage's own zero-below-the-floor +// rule, and the true m.width/m.height otherwise, never a pane or a rect. It +// keeps a destination's master-eligibility test (and clientSize, the default +// size for a new pane) current between broadcasts, in particular a MASTER +// whose window shrinks below the paintable floor: that daemon must see this +// arrive so it can hand off at once, per D4/Review-Focus-1. +// +// Connections are resolved HERE, on the Update goroutine — Router.Conns() +// walks its own lock, so this is not required for safety, but every other +// "resolve destinations, then send inside the Cmd" site in this file does it +// this way, and a fire-and-forget geometry report is not the place to invent +// a second convention. +func (m Model) clientGeometryCmd() tea.Cmd { + cols, rows := m.width, m.height + if !m.terminalPaintable() { + cols, rows = 0, 0 + } + var conns []Client + if r, ok := m.client.(*Router); ok { + conns = r.Conns() + } else if m.client != nil { + conns = []Client{m.client} + } + return func() tea.Msg { + msg, err := ipc.NewMessage(ipc.MsgClientGeometry, ipc.ClientGeometryPayload{Cols: cols, Rows: rows}) + if err != nil { + return nil + } + for _, c := range conns { + if c == nil { + continue + } + if err := c.Send(msg); err != nil { + log.Printf("client_geometry: send: %v", err) + } + } + return nil + } +} + +// sendTakeControl asks dest's daemon to make this client the size master at +// once (D6). Fire-and-forget: MsgTakeControl carries no payload, and the +// daemon either promotes an eligible, attached sender or logs and ignores an +// ineligible one — there is nothing for the client to wait on, and the next +// broadcast is what tells it whether the request took. +func (m Model) sendTakeControl(dest string) tea.Cmd { + return func() tea.Msg { + msg, err := ipc.NewMessage(ipc.MsgTakeControl, nil) + if err != nil { + return nil + } + if err := m.sendForDest(dest, msg); err != nil { + log.Printf("take control: send: %v", err) + } + return nil + } +} + // resizeAllPanes walks the projects rather than allTabs() so each pane's // message can carry its own daemon: this is a broadcast over EVERY project, so // the active dest would be the right answer for at most one of them. +// +// Every destination's panes are batched into ONE MsgResizePanes frame rather +// than one MsgResizePane per pane: a window resize or a split-drag release +// across dozens of panes must not put one must-deliver frame per pane on a +// follower's 64-slot queue (spec §4.1's "at most one frame per applied resize +// per follower" bound assumes the master itself never sent more than one +// batch to begin with). +// +// A FOLLOWER destination sends nothing at all — see isFollower. The master +// already owns every pane's size there, and this client's own idea of what +// size a pane should be is stale by construction once someone else is master. func (m Model) resizeAllPanes() tea.Cmd { if !m.terminalPaintable() { return nil // see terminalPaintable } return func() tea.Msg { + batches := make(map[string][]ipc.ResizePanePayload) for _, proj := range m.projects { + if m.isFollower(proj.Dest) { + continue + } for _, tab := range proj.tabs { if tab.Root == nil { continue @@ -9065,15 +9251,21 @@ func (m Model) resizeAllPanes() tea.Cmd { // the VT, so the mode this reproduces cannot disagree with // the one already applied. cols, rows := paneVTSize(pane.WideCanvas, pane.MinNativeCols, pane.Width, pane.Height, pane.NativeW, tab.CanvasW, tab.CanvasH) - msg, _ := ipc.NewMessage(ipc.MsgResizePane, ipc.ResizePanePayload{ + batches[proj.Dest] = append(batches[proj.Dest], ipc.ResizePanePayload{ PaneID: pane.ID, Cols: uint16(cols), Rows: uint16(rows), }) - m.sendForDest(proj.Dest, msg) } } } + for dest, panes := range batches { + msg, err := ipc.NewMessage(ipc.MsgResizePanes, ipc.ResizePanesPayload{Panes: panes}) + if err != nil { + continue + } + m.sendForDest(dest, msg) + } return nil } } @@ -9325,6 +9517,20 @@ func (m *Model) diffLayouts(state WorkspaceStateMsg) []layoutSend { // of distinct inputs can collide on one key. func sizedKey(dest, paneID string) string { return dest + "\x00" + paneID } +// clearSizedOnceForDest drops every sizedOnce entry recorded for dest — used +// when this client becomes dest's size master (applyWorkspaceState), so the +// first-resize kick resizeAllPanes/diffResizes rely on is owed again for +// every one of dest's panes, exactly as a reattach re-arms it +// (armReattachReset). +func (m *Model) clearSizedOnceForDest(dest string) { + prefix := dest + "\x00" + for key := range m.sizedOnce { + if strings.HasPrefix(key, prefix) { + delete(m.sizedOnce, key) + } + } +} + // hasProjectForDest reports whether any project belongs to dest. func (m *Model) hasProjectForDest(dest string) bool { for _, proj := range m.projects { @@ -9343,6 +9549,16 @@ func (m *Model) diffResizes(state WorkspaceStateMsg) []resizeSend { if !m.terminalPaintable() { return nil } + // A follower sends nothing for this destination, and — unlike the pending + // branch below — marks nothing in sizedOnce either: the whole function is + // scoped to state.Dest (every project this loop can reach has + // proj.Dest == state.Dest), so one check here covers it. Leaving sizedOnce + // untouched means a later election that hands this client the master role + // still owes every pane its first-resize kick, exactly as if it had never + // been diffed at all. + if m.isFollower(state.Dest) { + return nil + } type size struct { cols, rows uint16 pending bool @@ -9417,22 +9633,29 @@ func (m Model) sendDiffedLayouts(items []layoutSend) tea.Cmd { } } -// sendDiffedResizes ships an already-decided resize list. +// sendDiffedResizes ships an already-decided resize list, batched into one +// MsgResizePanes per destination — diffResizes can name several panes across +// one broadcast, and each must reach its daemon as a single must-deliver +// frame rather than one per pane (see resizeAllPanes' doc comment for why). func (m Model) sendDiffedResizes(items []resizeSend) tea.Cmd { if len(items) == 0 { return nil } return func() tea.Msg { + batches := make(map[string][]ipc.ResizePanePayload, len(items)) for _, it := range items { - msg, err := ipc.NewMessage(ipc.MsgResizePane, ipc.ResizePanePayload{ + batches[it.dest] = append(batches[it.dest], ipc.ResizePanePayload{ PaneID: it.paneID, Cols: it.cols, Rows: it.rows, }) + } + for dest, panes := range batches { + msg, err := ipc.NewMessage(ipc.MsgResizePanes, ipc.ResizePanesPayload{Panes: panes}) if err != nil { continue } - m.sendForDest(it.dest, msg) + m.sendForDest(dest, msg) } return nil } diff --git a/internal/tui/multiclient_role_test.go b/internal/tui/multiclient_role_test.go new file mode 100644 index 00000000..ac9abc20 --- /dev/null +++ b/internal/tui/multiclient_role_test.go @@ -0,0 +1,409 @@ +package tui + +import ( + "fmt" + "strings" + "testing" + + tea "charm.land/bubbletea/v2" + + "github.com/artyomsv/quil/internal/config" + "github.com/artyomsv/quil/internal/ipc" + "github.com/artyomsv/quil/internal/keymap" +) + +// Multi-client sync: TUI client identity, role, resize gates, Take control. +// +// isFollower(dest) = sizeMaster[dest] != "" && sizeMaster[dest] != clientID. +// resizeAllPanes, diffResizes and overlayResizeCmd are the three producers of +// pane resizes, and each gates on it — see model.go/overlay.go for the gates +// themselves. These tests drive Model.Update (or the literal Update dispatch +// target, e.g. finishSplitDrag), never the follower gate directly. + +// sawResizePanes reports whether any MsgResizePanes batch reached the wire. +func sawResizePanes(conn *fakeConn) bool { + conn.mu.Lock() + defer conn.mu.Unlock() + for _, msg := range conn.sent { + if msg.Type == ipc.MsgResizePanes { + return true + } + } + return false +} + +// sawClientGeometry reports whether a MsgClientGeometry reached the wire. +func sawClientGeometry(conn *fakeConn) bool { + conn.mu.Lock() + defer conn.mu.Unlock() + for _, msg := range conn.sent { + if msg.Type == ipc.MsgClientGeometry { + return true + } + } + return false +} + +// followerWorkspaceState builds the smallest broadcast that makes dest "" a +// follower of "other-client", differing in reported size from what a +// non-follower diff would compute (0x0 versus a real pane rect) — so a model +// that ignored the follower gate would have something to send. +func followerWorkspaceState() WorkspaceStateMsg { + return WorkspaceStateMsg{ + Dest: "", SizeMaster: "other-client", Clients: 2, + ActiveProject: "proj-1", ActiveTab: "tab-1", + Projects: []ProjectInfo{{ID: "proj-1", Name: "Default", TabIDs: []string{"tab-1"}}}, + Tabs: []TabInfo{{ID: "tab-1", Name: "Shell", ProjectID: "proj-1", Panes: []string{"pane-1"}}}, + Panes: []PaneInfo{{ID: "pane-1", TabID: "tab-1", Type: "terminal"}}, + } +} + +// TestFollower_SendsNoResizeOnBroadcast is the first of the three mutation +// checks: a broadcast reporting a size master other than this client, with a +// size that genuinely differs from what this client would compute, must +// produce zero resize sends. +func TestFollower_SendsNoResizeOnBroadcast(t *testing.T) { + t.Parallel() + m, conn := tinyTermModel(t) + m.SetClientID("me") + + // A healthy first size, so the pane has a real rect and a non-follower + // diff would find something to send. + next, cmd := m.Update(tea.WindowSizeMsg{Width: 172, Height: 48}) + runCmd(cmd) + m = next.(Model) + clearSent(conn) + + // Receive() must not park the listen command the broadcast re-arms. + close(conn.recv) + _, cmd = m.Update(followerWorkspaceState()) + runCmd(cmd) + + if n := countResizes(t, conn); n != 0 { + t.Errorf("resize count = %d, want 0 — a follower must never resize a "+ + "pane the master owns", n) + } +} + +// TestFollower_SendsNoResizeOnWindowResize: second mutation check. The client +// still reports its OWN window size (client_geometry) — that is what lets the +// daemon notice a master that became unpaintable — but sends no pane resize. +func TestFollower_SendsNoResizeOnWindowResize(t *testing.T) { + t.Parallel() + m, conn := tinyTermModel(t) + m.SetClientID("me") + + next, cmd := m.Update(tea.WindowSizeMsg{Width: 172, Height: 48}) + runCmd(cmd) + m = next.(Model) + + // Make this client a follower without running the broadcast's own cmds + // (which would re-arm the listen loop against an empty, unclosed channel). + m.sizeMaster = map[string]string{"": "other-client"} + m.clientCount = map[string]int{"": 2} + clearSent(conn) + + // The tea.Tick is deliberately not run — the arm under test is the + // resizeTickMsg one, delivered here directly through Update, matching + // every other debounced-resize test in this package (tinyterm_test.go). + next, _ = m.Update(tea.WindowSizeMsg{Width: 180, Height: 50}) + m = next.(Model) + _, tickCmd := m.Update(resizeTickMsg{seq: m.resizeSeq}) + runCmd(tickCmd) + + if n := countResizes(t, conn); n != 0 { + t.Errorf("resize count = %d, want 0 — a follower must not resize a "+ + "pane on its own window change", n) + } + if !sawClientGeometry(conn) { + t.Error("a follower must still report its own window size via " + + "client_geometry, or the daemon can never notice it became " + + "unpaintable and hand off") + } +} + +// TestFollower_SendsNoResizeOnSplitDragRelease: third mutation check, for +// diffResizes/resizeAllPanes' shared call path — finishSplitDrag is the exact +// method Update's MouseReleaseMsg case dispatches a split-border release to. +func TestFollower_SendsNoResizeOnSplitDragRelease(t *testing.T) { + t.Parallel() + m := newSplitDragTestModel(t) + m.SetClientID("me") + m.sizeMaster = map[string]string{"": "other-client"} + m.clientCount = map[string]int{"": 2} + conn := newFakeConn() + m.client = conn + + // Arm and drag, exactly as TestModel_FinishSplitDrag_CommitsToDaemon does + // for the master case. + hit := m.hitTestSplitBorder(50, 10) + if hit == nil { + t.Fatal("setup: no split border hit at (50,10)") + } + m.splitDragNode = hit.Node + m.splitDragRect = *hit + m.dragSplitBorder(70, 10) + + cmd := m.finishSplitDrag() + if cmd == nil { + t.Fatal("finishSplitDrag returned nil — the layout commit must still happen") + } + runCmd(cmd) + + if sawResizePanes(conn) { + t.Error("a follower's split-drag release must not resize any pane") + } +} + +// TestFollower_OverlayResizeGated: the fourth site, and the one no tree walk +// reaches — an overlay pane sits outside the layout tree, so this sweep +// (resizeTickMsg) is its only resize producer. +func TestFollower_OverlayResizeGated(t *testing.T) { + t.Parallel() + m, conn := tinyTermModel(t) + m.SetClientID("me") + + next, cmd := m.Update(tea.WindowSizeMsg{Width: 172, Height: 48}) + runCmd(cmd) + m = next.(Model) + + overlay := NewPaneModel("overlay-1", testRingBufSize) + t.Cleanup(overlay.Dispose) + tab := m.projects[0].tabs[0] + tab.overlayPane = overlay + tab.overlayVisible = true + m.sizeMaster = map[string]string{"": "other-client"} + m.clientCount = map[string]int{"": 2} + clearSent(conn) + + next, _ = m.Update(tea.WindowSizeMsg{Width: 180, Height: 50}) + m = next.(Model) + _, tickCmd := m.Update(resizeTickMsg{seq: m.resizeSeq}) + runCmd(tickCmd) + + if sawResizeFor(t, conn, "overlay-1") { + t.Error("a follower must not resize its overlay pane either") + } +} + +// TestMaster_ResizeAllPanesSendsOneBatchPerDest is Review Focus 2: a resize +// burst across many panes must not put one must-deliver frame per pane on a +// follower's queue. 40 panes, one destination, one MsgResizePanes frame. +func TestMaster_ResizeAllPanesSendsOneBatchPerDest(t *testing.T) { + t.Parallel() + const n = 40 + tabs := make([]*TabModel, n) + for i := 0; i < n; i++ { + id := fmt.Sprintf("pane-%d", i) + tab := NewTabModel(fmt.Sprintf("tab-%d", i), fmt.Sprintf("T%d", i)) + pane := NewPaneModel(id, testRingBufSize) + t.Cleanup(pane.Dispose) + tab.Root = NewLeaf(pane) + tab.ActivePane = id + tab.Resize(80, 24) + tabs[i] = tab + } + conn := newFakeConn() + m := Model{ + client: conn, sized: true, width: 172, height: 48, + tabDragFromIdx: -1, + projects: []*ProjectModel{{ID: "proj-1", Name: "Default", tabs: tabs}}, + } + + cmd := m.resizeAllPanes() + if cmd == nil { + t.Fatal("resizeAllPanes returned nil") + } + runCmd(cmd) + + var frames, panes int + for _, msg := range conn.sent { + if msg.Type != ipc.MsgResizePanes { + continue + } + frames++ + var p ipc.ResizePanesPayload + if err := msg.DecodePayload(&p); err != nil { + t.Fatalf("decode resize_panes: %v", err) + } + panes += len(p.Panes) + } + if frames != 1 { + t.Errorf("resize_panes frame count = %d, want 1 — a resize burst must "+ + "not put one must-deliver frame per pane on the wire", frames) + } + if panes != n { + t.Errorf("panes carried across all frames = %d, want %d", panes, n) + } +} + +// TestBecomingMaster_ResendsSizes: on the broadcast that hands this client +// the master role, every pane must be resized at once — even one diffResizes +// alone would have suppressed, because the reported size already agrees with +// what this client would compute AND sizedOnce already says "already sent". +func TestBecomingMaster_ResendsSizes(t *testing.T) { + t.Parallel() + m, conn := tinyTermModel(t) + m.SetClientID("me") + + next, cmd := m.Update(tea.WindowSizeMsg{Width: 172, Height: 48}) + runCmd(cmd) + m = next.(Model) + + tab := m.projects[0].tabs[0] + pane := tab.Leaves()[0] + cols, rows := paneVTSize(pane.WideCanvas, pane.MinNativeCols, + pane.Width, pane.Height, pane.NativeW, tab.CanvasW, tab.CanvasH) + + // A follower that already sent this exact size once before (e.g. from an + // earlier mastership) — diffResizes' own diff would find nothing to send. + m.sizeMaster = map[string]string{"": "other-client"} + m.sizedOnce = map[string]bool{sizedKey("", "pane-1"): true} + clearSent(conn) + + close(conn.recv) + _, cmd = m.Update(WorkspaceStateMsg{ + Dest: "", SizeMaster: "me", Clients: 2, + ActiveProject: "proj-1", ActiveTab: "tab-1", + Projects: []ProjectInfo{{ID: "proj-1", Name: "Default", TabIDs: []string{"tab-1"}}}, + Tabs: []TabInfo{{ID: "tab-1", Name: "Shell", ProjectID: "proj-1", Panes: []string{"pane-1"}}}, + Panes: []PaneInfo{{ID: "pane-1", TabID: "tab-1", Type: "terminal", Cols: uint16(cols), Rows: uint16(rows)}}, + }) + runCmd(cmd) + + if !sawResizeFor(t, conn, "pane-1") { + t.Error("becoming the size master must resize every pane at once, " + + "even one diffResizes alone would have suppressed") + } +} + +// TestAttach_CarriesClientID: attachMessage must carry the process-minted id +// on every attach and reattach. +func TestAttach_CarriesClientID(t *testing.T) { + t.Parallel() + m := Model{cfg: config.Default(), width: 172, height: 48} + m.SetClientID("my-client-id") + + var p ipc.AttachPayload + if err := m.attachMessage("").DecodePayload(&p); err != nil { + t.Fatalf("decode attach payload: %v", err) + } + if p.ClientID != "my-client-id" { + t.Errorf("ClientID = %q, want %q", p.ClientID, "my-client-id") + } +} + +// TestCloseClient_SendsDetachBeforeClose: the recording client must see +// detach queued BEFORE the conn is released — Close discards whatever is +// still in the send queue, so the order is the whole point (D3/3.3). +func TestCloseClient_SendsDetachBeforeClose(t *testing.T) { + t.Parallel() + check := func(t *testing.T, conn *fakeConn) { + t.Helper() + conn.mu.Lock() + defer conn.mu.Unlock() + if len(conn.sent) != 1 || conn.sent[0].Type != ipc.MsgDetach { + t.Errorf("at close time, sent = %+v, want exactly one queued "+ + "MsgDetach — Close discards frames still in the send queue, "+ + "so detach must be queued before Close runs, not after", conn.sent) + } + } + + t.Run("single connection", func(t *testing.T) { + conn := newFakeConn() + m := Model{client: conn} + m.SetClientCloser(func(c Client) { check(t, c.(*fakeConn)) }) + m.CloseClient() + }) + + t.Run("router", func(t *testing.T) { + local, gpu := newFakeConn(), newFakeConn() + r := NewRouter(map[string]Client{"": local, "gpu01": gpu}) + m := Model{client: r} + m.SetClientCloser(func(c Client) { check(t, c.(*fakeConn)) }) + m.CloseClient() + }) +} + +// TestStatusBar_RoleMarkerOnlyWithTwoClients: the marker is shown only once a +// second client exists, and names the right role. +func TestStatusBar_RoleMarkerOnlyWithTwoClients(t *testing.T) { + t.Setenv("QUIL_HOME", t.TempDir()) + m := Model{ + cfg: config.Default(), width: 100, height: 40, + notifications: NewNotificationCenter(30, 50), + projects: []*ProjectModel{{ID: "proj-1", Dest: ""}}, + } + m.SetClientID("me") + + // One client: no marker, whichever role this client would otherwise hold. + m.clientCount = map[string]int{"": 1} + m.sizeMaster = map[string]string{"": "me"} + if got := m.renderStatusBar(); strings.Contains(got, "[master]") || strings.Contains(got, "[follower]") { + t.Errorf("status bar shows a role marker with one client: %q", got) + } + + // Two clients, this one is master. + m.clientCount[""] = 2 + if got := m.renderStatusBar(); !strings.Contains(got, "[master]") { + t.Errorf("status bar = %q, want [master]", got) + } + + // Two clients, this one is a follower. + m.sizeMaster[""] = "other-client" + if got := m.renderStatusBar(); !strings.Contains(got, "[follower]") { + t.Errorf("status bar = %q, want [follower]", got) + } +} + +// TestTakeControlAction_SendsTakeControl covers both front doors: the keymap +// action and the palette command, each sending MsgTakeControl to the active +// destination. +func TestTakeControlAction_SendsTakeControl(t *testing.T) { + t.Parallel() + + t.Run("key", func(t *testing.T) { + t.Parallel() + m := newModelForTest([]string{"T"}, 0) + m.SetBindings(config.Bindings{ + Preset: keymap.DefaultPresetName, + Overrides: map[keymap.ActionID]string{"client.take_control": "ctrl+g"}, + }) + conn := newFakeConn() + m.client = conn + + _, cmd := m.Update(tea.KeyPressMsg{Code: 'g', Mod: tea.ModCtrl}) + runCmd(cmd) + + var saw bool + for _, msg := range conn.sent { + if msg.Type == ipc.MsgTakeControl { + saw = true + } + } + if !saw { + t.Error("bound key did not send take_control") + } + }) + + t.Run("palette", func(t *testing.T) { + t.Parallel() + m := newModelForTest([]string{"T"}, 0) + conn := newFakeConn() + m.client = conn + m.dialog = dialogCommandPalette + + _, cmd := m.executePaletteCommand(paletteCommand{action: palActTakeControl, enabled: true}) + runCmd(cmd) + + var saw bool + for _, msg := range conn.sent { + if msg.Type == ipc.MsgTakeControl { + saw = true + } + } + if !saw { + t.Error("palette command did not send take_control") + } + }) +} diff --git a/internal/tui/overlay.go b/internal/tui/overlay.go index 8b9f3b1b..ee453fd6 100644 --- a/internal/tui/overlay.go +++ b/internal/tui/overlay.go @@ -481,9 +481,9 @@ func (m *Model) createOverlay(tab *TabModel, repo, pluginName string) tea.Cmd { return tea.Batch(cmds...) } -// overlayResizeCmd sends MsgResizePane for the overlay pane so the daemon's -// PTY tracks the current tab dimensions. Cols/Rows subtract the 2-cell border; -// each dimension is clamped to at least 1. +// overlayResizeCmd sends a (single-entry) MsgResizePanes batch for the overlay +// pane so the daemon's PTY tracks the current tab dimensions. Cols/Rows +// subtract the 2-cell border; each dimension is clamped to at least 1. func (m *Model) overlayResizeCmd(tab *TabModel) tea.Cmd { if tab.overlayPane == nil { return nil @@ -493,6 +493,13 @@ func (m *Model) overlayResizeCmd(tab *TabModel) tea.Cmd { if !m.terminalPaintable() { return nil } + // Third of the three follower gates: an overlay pane sits outside the + // layout tree, so resizeAllPanes/diffResizes never reach it, and this is + // its only resize producer. A follower must stay silent here too, or its + // overlay's size fights the master's the same way a tree pane's would. + if m.isFollower(tab.Dest) { + return nil + } paneID, dest := tab.overlayPane.ID, tab.Dest cols := tab.Width - 2 rows := tab.Height - 2 @@ -505,10 +512,8 @@ func (m *Model) overlayResizeCmd(tab *TabModel) tea.Cmd { c := uint16(cols) r := uint16(rows) return func() tea.Msg { - msg, err := ipc.NewMessage(ipc.MsgResizePane, ipc.ResizePanePayload{ - PaneID: paneID, - Cols: c, - Rows: r, + msg, err := ipc.NewMessage(ipc.MsgResizePanes, ipc.ResizePanesPayload{ + Panes: []ipc.ResizePanePayload{{PaneID: paneID, Cols: c, Rows: r}}, }) if err != nil { log.Printf("overlay: resize pane encode: %v", err) diff --git a/internal/tui/palette.go b/internal/tui/palette.go index e9dbf48c..bd9e8e7d 100644 --- a/internal/tui/palette.go +++ b/internal/tui/palette.go @@ -97,6 +97,8 @@ const ( palActMoveProjectDown palActNewTemplate palActTabLayout // arg = the keymap action id, e.g. "tab.layout_grid" + // Multi-client sync (D6): sends MsgTakeControl to the ACTIVE destination. + palActTakeControl ) // paletteCommand is one row of the palette. Disabled rows render greyed and are @@ -566,6 +568,7 @@ func (m *Model) buildPaletteCommands() []paletteCommand { paletteCommand{action: palActDaemonLog, enabled: true, label: "View daemon log", keywords: []string{"log", "daemon"}}, paletteCommand{action: palActMCPLog, enabled: true, label: "View MCP logs", keywords: []string{"log", "mcp"}}, paletteCommand{action: palActRedraw, enabled: true, label: "Force redraw", detail: m.keymap.Display("app.redraw"), keywords: []string{"redraw", "repaint", "refresh"}}, + paletteCommand{action: palActTakeControl, enabled: true, label: "Take control (size master)", detail: m.keymap.Display("client.take_control"), keywords: []string{"master", "control", "resize", "follower"}}, ) // --- Appearance -------------------------------------------------------- @@ -1237,6 +1240,8 @@ func (m Model) executePaletteCommand(c paletteCommand) (tea.Model, tea.Cmd) { return m.openMCPLogsViewer() case palActRedraw: return m.forceRedraw() + case palActTakeControl: + return m, m.sendTakeControl(m.activeDest()) // --- Appearance -------------------------------------------------------- case palActDimToggle: diff --git a/internal/tui/reconnect.go b/internal/tui/reconnect.go index 5b654708..0ce92cb1 100644 --- a/internal/tui/reconnect.go +++ b/internal/tui/reconnect.go @@ -10,6 +10,8 @@ import ( tea "charm.land/bubbletea/v2" "charm.land/lipgloss/v2" + + "github.com/artyomsv/quil/internal/ipc" ) // linkLostMsg reports that the connection to the daemon died. @@ -375,6 +377,29 @@ func (m Model) closeClient(c Client) { m.closeClientFn(c) } +// sendDetach tells a conn's daemon that this client is exiting cleanly (§3.3's +// "a clean exit skips the grace time"): the daemon drops it from master +// election and the client list at once, rather than waiting out the lost-link +// grace timer for a link that is not actually lost. Best-effort and silent — +// this runs on the exit path, where there is nobody left to report a failure +// to, and an older daemon simply ignores an unknown message type. +// +// Sent here, immediately before closeClient, rather than as a tea.Cmd: the +// Update loop is already gone on the exit path (main.go), and closeClient's +// own Flush is what actually carries a just-queued Send to the socket. +func sendDetach(c Client) { + if c == nil { + return + } + msg, err := ipc.NewMessage(ipc.MsgDetach, nil) + if err != nil { + return + } + if err := c.Send(msg); err != nil { + log.Printf("detach: send: %v", err) + } +} + // CloseClient releases every connection the Model currently holds. Called by // cmd/quil on exit, after the Bubble Tea program has returned. // @@ -387,10 +412,12 @@ func (m Model) closeClient(c Client) { func (m Model) CloseClient() { if r, ok := m.client.(*Router); ok { for _, c := range r.Conns() { + sendDetach(c) m.closeClient(c) } return } + sendDetach(m.client) m.closeClient(m.client) } diff --git a/internal/tui/sidebar_test.go b/internal/tui/sidebar_test.go index 0554143d..0c292b1d 100644 --- a/internal/tui/sidebar_test.go +++ b/internal/tui/sidebar_test.go @@ -233,7 +233,7 @@ func TestSwitchProjectNotifiesDaemonAndResyncsGeometry(t *testing.T) { // was current when it went to the background. var sawResize bool for _, msg := range fake.sent { - if msg.Type == ipc.MsgResizePane { + if msg.Type == ipc.MsgResizePanes { sawResize = true } } @@ -1794,15 +1794,17 @@ func TestSidebarDoesNotRemodeAWideCanvasPane(t *testing.T) { wireCols := func(t *testing.T, conn *fakeConn, paneID string) int { t.Helper() for _, msg := range conn.sent { - if msg.Type != ipc.MsgResizePane { + if msg.Type != ipc.MsgResizePanes { continue } - var p ipc.ResizePanePayload + var p ipc.ResizePanesPayload if err := msg.DecodePayload(&p); err != nil { - t.Fatalf("decode resize: %v", err) + t.Fatalf("decode resize_panes: %v", err) } - if p.PaneID == paneID { - return int(p.Cols) + for _, rp := range p.Panes { + if rp.PaneID == paneID { + return int(rp.Cols) + } } } t.Fatalf("no resize sent for pane %s", paneID) diff --git a/internal/tui/splitdrag_test.go b/internal/tui/splitdrag_test.go index 12c88fec..0badf98b 100644 --- a/internal/tui/splitdrag_test.go +++ b/internal/tui/splitdrag_test.go @@ -184,14 +184,18 @@ func TestModel_FinishSplitDrag_CommitsToDaemon(t *testing.T) { var resizes, layouts int for _, sent := range fs.sent { switch sent.Type { - case ipc.MsgResizePane: - resizes++ + case ipc.MsgResizePanes: + var p ipc.ResizePanesPayload + if err := sent.DecodePayload(&p); err != nil { + t.Fatalf("decode resize_panes: %v", err) + } + resizes += len(p.Panes) case ipc.MsgUpdateLayout: layouts++ } } if resizes != 2 { - t.Errorf("MsgResizePane count = %d, want 2", resizes) + t.Errorf("resize count = %d, want 2", resizes) } if layouts != 1 { t.Errorf("MsgUpdateLayout count = %d, want 1", layouts) diff --git a/internal/tui/tinyterm_test.go b/internal/tui/tinyterm_test.go index 951ee1d4..aee53a97 100644 --- a/internal/tui/tinyterm_test.go +++ b/internal/tui/tinyterm_test.go @@ -47,16 +47,23 @@ func tinyTermModel(t *testing.T) (Model, *fakeConn) { return m, conn } -// countResizes reports how many MsgResizePane frames reached the wire. +// countResizes reports how many individual pane resizes reached the wire, +// across every MsgResizePanes batch (resizeAllPanes/diffResizes/ +// overlayResizeCmd all batch into one frame per destination now). func countResizes(t *testing.T, conn *fakeConn) int { t.Helper() n := 0 conn.mu.Lock() defer conn.mu.Unlock() for _, msg := range conn.sent { - if msg.Type == ipc.MsgResizePane { - n++ + if msg.Type != ipc.MsgResizePanes { + continue + } + var p ipc.ResizePanesPayload + if err := msg.DecodePayload(&p); err != nil { + t.Fatalf("decode resize_panes payload: %v", err) } + n += len(p.Panes) } return n } @@ -225,21 +232,24 @@ func TestUpdate_GrowBackAboveMinimum_ResizesABackgroundTabsOverlay(t *testing.T) } } -// sawResizeFor reports whether a MsgResizePane for paneID reached the wire. +// sawResizeFor reports whether a resize for paneID reached the wire, inside +// any MsgResizePanes batch. func sawResizeFor(t *testing.T, conn *fakeConn, paneID string) bool { t.Helper() conn.mu.Lock() defer conn.mu.Unlock() for _, msg := range conn.sent { - if msg.Type != ipc.MsgResizePane { + if msg.Type != ipc.MsgResizePanes { continue } - var p ipc.ResizePanePayload + var p ipc.ResizePanesPayload if err := msg.DecodePayload(&p); err != nil { - t.Fatalf("decode resize payload: %v", err) + t.Fatalf("decode resize_panes payload: %v", err) } - if p.PaneID == paneID { - return true + for _, rp := range p.Panes { + if rp.PaneID == paneID { + return true + } } } return false From 3880aa9256b31e154744fd0ec2ef5971de9291a3 Mon Sep 17 00:00:00 2001 From: Artjoms Stukans Date: Fri, 25 Sep 2026 21:30:28 +0200 Subject: [PATCH 12/40] fix(tui): stop resizeAllPanes racing the size-master map resizeAllPanes' returned tea.Cmd read m.sizeMaster (via isFollower) from inside the closure Bubble Tea runs on its own goroutine, while applyWorkspaceState mutates that same map in place on the Update goroutine. Go treats a concurrent map read/write as fatal, not merely a -race finding, and this could crash the TUI outright. Compute the per-destination resize batches synchronously before returning the closure, so it only ever touches a fresh, unshared local map. Drop the redundant resizeAllPanes() call from the becoming-master branch in applyWorkspaceState: clearing sizedOnce for that destination is enough on its own, since diffResizes runs immediately afterward, scoped to the same destination, and resends every one of its panes. Calling resizeAllPanes() there was wrong rather than merely redundant, since it walks every destination and would resize ones this broadcast never mentioned. Send each connection's clean-exit detach concurrently and bound the wait at 500ms before closing, instead of sending them one at a time. ipc.Client.Send can block up to 5s against a wedged peer, so detaching several dead remote hosts in sequence could turn quitting the TUI into a multi-second hang; a conn still wedged past the budget is closed anyway. Also: sentCounts (broadcast_echo_test.go) now fails the test on a decode error instead of silently counting it as zero, and a couple of stale comments still naming the pre-batching MsgResizePane are corrected. Tests: a regression test drives resizeAllPanes' returned closure against concurrent size-master writes under -race; a wedged-connection test proves CloseClient no longer blocks on a dead peer; the becoming-master test now asserts exactly one resize frame, addressed to the destination that actually changed; CloseClient's detach test asserts the closer actually ran; a new test covers client_geometry reporting 0x0 below the paintable floor; and a mixed-session test covers being master on one destination while following another. --- internal/tui/broadcast_echo_test.go | 24 ++- internal/tui/model.go | 98 +++++---- internal/tui/multiclient_role_test.go | 293 +++++++++++++++++++++++++- internal/tui/reconnect.go | 61 +++++- internal/tui/template_layout_test.go | 10 +- 5 files changed, 423 insertions(+), 63 deletions(-) diff --git a/internal/tui/broadcast_echo_test.go b/internal/tui/broadcast_echo_test.go index 732df783..0ceb32b6 100644 --- a/internal/tui/broadcast_echo_test.go +++ b/internal/tui/broadcast_echo_test.go @@ -117,7 +117,7 @@ func TestLayoutAgrees_MalformedStoredLayoutResends(t *testing.T) { _, cmd := m.Update(echo) runCmd(cmd) - layouts, _ := sentCounts(fs) + layouts, _ := sentCounts(t, fs) if layouts != 1 { t.Errorf("MsgUpdateLayout count = %d, want 1 — an unparseable stored "+ "layout must be replaced, not accepted", layouts) @@ -198,16 +198,18 @@ func (e *echoRecorder) Receive() (*ipc.Message, error) { return &ipc.Message{Type: "test-inert"}, nil } -func sentCounts(fs *echoRecorder) (layouts, resizes int) { +func sentCounts(t *testing.T, fs *echoRecorder) (layouts, resizes int) { + t.Helper() for _, msg := range fs.sent { switch msg.Type { case ipc.MsgUpdateLayout: layouts++ case ipc.MsgResizePanes: var p ipc.ResizePanesPayload - if err := json.Unmarshal(msg.Payload, &p); err == nil { - resizes += len(p.Panes) + if err := json.Unmarshal(msg.Payload, &p); err != nil { + t.Fatalf("decode resize_panes payload: %v", err) } + resizes += len(p.Panes) } } return layouts, resizes @@ -223,7 +225,7 @@ func TestWorkspaceState_UnchangedSplitLayout_SendsNoLayoutUpdate(t *testing.T) { _ = next runCmd(cmd) - layouts, _ := sentCounts(fs) + layouts, _ := sentCounts(t, fs) if layouts != 0 { t.Errorf("MsgUpdateLayout count = %d, want 0 — the broadcast already "+ "carried these layouts, so echoing them back is pure queue "+ @@ -267,7 +269,7 @@ func TestWorkspaceState_FirstResizeAfterAttach_IsAlwaysSent(t *testing.T) { _ = next runCmd(cmd) - _, resizes := sentCounts(fs) + _, resizes := sentCounts(t, fs) if resizes != 3 { t.Errorf("resize count = %d, want 3 (one per pane) — the daemon "+ "zeroes its applied-size guard on PTY install and needs the first "+ @@ -295,7 +297,7 @@ func TestWorkspaceState_UnchangedSizes_SendNoResizeOnRepeat(t *testing.T) { _, cmd = m.Update(echo) runCmd(cmd) - _, resizes := sentCounts(fs) + _, resizes := sentCounts(t, fs) if resizes != 0 { t.Errorf("resize count = %d, want 0 on a repeat broadcast — "+ "every pane already has the size the broadcast reports", resizes) @@ -357,7 +359,7 @@ func TestWorkspaceState_ReportedCrashConfiguration_SettlesToSilence(t *testing.T _, cmd = m.Update(echo) runCmd(cmd) - layouts, resizes := sentCounts(fs) + layouts, resizes := sentCounts(t, fs) if layouts+resizes != 0 { t.Errorf("a repeat broadcast at 33 tabs/36 panes produced %d layout + %d "+ "resize frames, want 0 — 69 of these on a %d-slot must-deliver queue "+ @@ -435,7 +437,7 @@ func TestWorkspaceState_LazyRestoreAtScale_SettlesToSilence(t *testing.T) { _, cmd = m.Update(echo) runCmd(cmd) - layouts, resizes := sentCounts(fs) + layouts, resizes := sentCounts(t, fs) if layouts+resizes != 0 { t.Errorf("a repeat broadcast over a lazily-restored 33-tab workspace "+ "produced %d layout + %d resize frames, want 0 — deferred panes "+ @@ -598,7 +600,7 @@ func TestReattach_ReArmsTheFirstResizeKick(t *testing.T) { _, cmd = m.Update(echo) runCmd(cmd) - _, resizes := sentCounts(fs) + _, resizes := sentCounts(t, fs) if resizes != 3 { t.Errorf("resize count = %d after reattach, want 3 — the daemon "+ "zeroed its guard on PTY install, so the suppression state from "+ @@ -686,7 +688,7 @@ func TestWorkspaceState_AbsentLayout_StillSends(t *testing.T) { _, cmd := m.Update(echo) runCmd(cmd) - layouts, _ := sentCounts(fs) + layouts, _ := sentCounts(t, fs) if layouts != 1 { t.Errorf("MsgUpdateLayout count = %d, want 1 — a tab the daemon has no "+ "layout for must be sent, or the arrangement is never persisted", diff --git a/internal/tui/model.go b/internal/tui/model.go index 05a4698b..123c38fe 100644 --- a/internal/tui/model.go +++ b/internal/tui/model.go @@ -459,14 +459,15 @@ type Model struct { // the failure it guards — a SECOND reader of the router's channel — has no // error to assert on, only reordering. listenCountFn func() - // sizedOnce records panes this connection has sent at least one - // MsgResizePane for. A broadcast-driven resize is suppressed when the - // reported size already matches, but the FIRST one is always sent: the - // daemon's duplicate guard is appliedCols/appliedRows, which it zeroes on - // every PTY install, and repaintAfterResize's redraw kick for a restored - // pane rides that first client resize. Cleared by armReattachReset, since - // a reattach is exactly when the daemon's guard may have been zeroed, and - // pruned by applyWorkspaceState when a pane stops existing. + // sizedOnce records panes this connection has sent at least one resize + // for (batched into MsgResizePanes). A broadcast-driven resize is + // suppressed when the reported size already matches, but the FIRST one + // is always sent: the daemon's duplicate guard is + // appliedCols/appliedRows, which it zeroes on every PTY install, and + // repaintAfterResize's redraw kick for a restored pane rides that first + // client resize. Cleared by armReattachReset, since a reattach is + // exactly when the daemon's guard may have been zeroed, and pruned by + // applyWorkspaceState when a pane stops existing. // // Keyed by sizedKey(dest, paneID), not by pane id alone: this decides // whether a pane's FIRST resize ships, so a shared key would let one @@ -6100,12 +6101,20 @@ func (m *Model) applyWorkspaceState(state WorkspaceStateMsg, dest string) ([]str // This client just became dest's size master. Every pane's last // resize on dest was sent by whoever was master before (or by nobody, // if there was none) — sizedOnce still reports "already sized" for - // sizes THIS client never sent, so the diff in resizeAllPanes/ - // diffResizes would suppress the very sizes that just became - // authoritative. Clearing it re-arms the same first-resize kick - // armReattachReset re-arms after a reattach. + // sizes THIS client never sent, so diffResizes' diff would suppress + // the very sizes that just became authoritative. + // + // Clearing it is enough on its own, and calling resizeAllPanes() here + // as well — an earlier version of this did — is wrong, not merely + // redundant: diffResizes runs immediately after this function returns + // (the WorkspaceStateMsg arm in Update) and is scoped to this SAME + // dest, so with sizedOnce empty for it, every one of dest's panes + // fails the "already sized" check and rides that ONE + // sendDiffedResizes batch — exactly the same re-arm armReattachReset + // performs after a reattach. resizeAllPanes walks EVERY destination, + // so appending it here would also resize panes on other destinations + // this broadcast never mentioned. m.clearSizedOnceForDest(dest) - overlayResizeCmds = append(overlayResizeCmds, m.resizeAllPanes()) } // Dispose panes that did not survive reconciliation — both panes pruned @@ -9133,10 +9142,11 @@ func keyToBytes(keyMsg tea.KeyPressMsg) []byte { // minTermWidth x minTermHeight and renders no panes at all, so a pane size // derived from a smaller geometry describes nothing that is on screen. // -// It gates EVERY MsgResizePane this client produces — resizeAllPanes, -// diffResizes and overlayResizeCmd — and it has to sit at those fan-outs rather -// than further down, because by the time a size reaches the wire the degenerate -// case is indistinguishable from a legal one: paneVTSize floors both dimensions +// It gates every pane resize this client produces — resizeAllPanes, +// diffResizes and overlayResizeCmd, batched into MsgResizePanes — and it has +// to sit at those fan-outs rather than further down, because by the time a +// size reaches the wire the degenerate case is indistinguishable from a +// legal one: paneVTSize floors both dimensions // at 1 on purpose, since a genuinely narrow SPLIT pane needs that floor. A // client started with no console attached (`quil.exe --version` from a // non-interactive shell) is reported by Bubble Tea as 1x1, and the fan-out then @@ -9233,32 +9243,42 @@ func (m Model) resizeAllPanes() tea.Cmd { if !m.terminalPaintable() { return nil // see terminalPaintable } - return func() tea.Msg { - batches := make(map[string][]ipc.ResizePanePayload) - for _, proj := range m.projects { - if m.isFollower(proj.Dest) { + // Computed HERE, on the Update goroutine, and handed to the closure as a + // plain local map — never read live from inside the closure. Every + // tea.Cmd Bubble Tea returns runs on its OWN goroutine, concurrently with + // whatever Update call comes next, and m.sizeMaster is a map MUTATED IN + // PLACE by applyWorkspaceState (m.sizeMaster[dest] = state.SizeMaster), + // never reassigned wholesale — so a later isFollower read from inside the + // closure races that write. Go's runtime treats a concurrent map + // read/write as a fatal error, not merely a -race finding: it can crash + // the TUI outright. batches is allocated fresh by this call and shared + // with nobody, so the closure reading it from another goroutine is safe. + batches := make(map[string][]ipc.ResizePanePayload) + for _, proj := range m.projects { + if m.isFollower(proj.Dest) { + continue + } + for _, tab := range proj.tabs { + if tab.Root == nil { continue } - for _, tab := range proj.tabs { - if tab.Root == nil { - continue - } - for _, pane := range tab.Leaves() { - // paneVTSize keeps the PTY in lockstep with the VT: rect - // size for normal panes, tab canvas for wide-canvas panes. - // The daemon drops exact duplicates (same-size guard). - // pane.NativeW comes from the same resize pass that sized - // the VT, so the mode this reproduces cannot disagree with - // the one already applied. - cols, rows := paneVTSize(pane.WideCanvas, pane.MinNativeCols, pane.Width, pane.Height, pane.NativeW, tab.CanvasW, tab.CanvasH) - batches[proj.Dest] = append(batches[proj.Dest], ipc.ResizePanePayload{ - PaneID: pane.ID, - Cols: uint16(cols), - Rows: uint16(rows), - }) - } + for _, pane := range tab.Leaves() { + // paneVTSize keeps the PTY in lockstep with the VT: rect + // size for normal panes, tab canvas for wide-canvas panes. + // The daemon drops exact duplicates (same-size guard). + // pane.NativeW comes from the same resize pass that sized + // the VT, so the mode this reproduces cannot disagree with + // the one already applied. + cols, rows := paneVTSize(pane.WideCanvas, pane.MinNativeCols, pane.Width, pane.Height, pane.NativeW, tab.CanvasW, tab.CanvasH) + batches[proj.Dest] = append(batches[proj.Dest], ipc.ResizePanePayload{ + PaneID: pane.ID, + Cols: uint16(cols), + Rows: uint16(rows), + }) } } + } + return func() tea.Msg { for dest, panes := range batches { msg, err := ipc.NewMessage(ipc.MsgResizePanes, ipc.ResizePanesPayload{Panes: panes}) if err != nil { diff --git a/internal/tui/multiclient_role_test.go b/internal/tui/multiclient_role_test.go index ac9abc20..093e6fe6 100644 --- a/internal/tui/multiclient_role_test.go +++ b/internal/tui/multiclient_role_test.go @@ -3,7 +3,9 @@ package tui import ( "fmt" "strings" + "sync" "testing" + "time" tea "charm.land/bubbletea/v2" @@ -255,6 +257,18 @@ func TestBecomingMaster_ResendsSizes(t *testing.T) { cols, rows := paneVTSize(pane.WideCanvas, pane.MinNativeCols, pane.Width, pane.Height, pane.NativeW, tab.CanvasW, tab.CanvasH) + // A second destination, UNTOUCHED by the coming broadcast and with no + // master of its own reported (isFollower("gpu01") is therefore false — + // the state a lone-client destination is in). If the fix that removed + // the extra resizeAllPanes() call from the becoming-master branch ever + // regresses, resizeAllPanes walks every destination and would resize + // this one too, even though nothing about it changed. + otherTab := tabWithPane("gpu-tab", "gpu-pane") + otherTab.Resize(80, 24) + m.projects = append(m.projects, &ProjectModel{ + ID: "proj-gpu", Name: "gpu", Dest: "gpu01", tabs: []*TabModel{otherTab}, + }) + // A follower that already sent this exact size once before (e.g. from an // earlier mastership) — diffResizes' own diff would find nothing to send. m.sizeMaster = map[string]string{"": "other-client"} @@ -271,12 +285,169 @@ func TestBecomingMaster_ResendsSizes(t *testing.T) { }) runCmd(cmd) + var frames int + for _, msg := range conn.sent { + if msg.Type != ipc.MsgResizePanes { + continue + } + frames++ + if msg.Origin != destLocal { + t.Errorf("resize_panes frame addressed to origin %q, want %q — "+ + "this broadcast is scoped to dest \"\", and becoming its "+ + "master must not resize an unrelated destination", + msg.Origin, destLocal) + } + } + if frames != 1 { + t.Errorf("resize_panes frame count = %d, want exactly 1 — diffResizes' "+ + "own re-send (from the cleared sizedOnce) must be the only "+ + "producer here, not a second resizeAllPanes() batch", frames) + } if !sawResizeFor(t, conn, "pane-1") { t.Error("becoming the size master must resize every pane at once, " + "even one diffResizes alone would have suppressed") } } +// TestResizeAllPanes_NoRaceAgainstConcurrentMasterChanges is the regression +// test for a real crash, not just a -race finding: resizeAllPanes' returned +// tea.Cmd used to read m.sizeMaster (via isFollower) from INSIDE the +// closure — which Bubble Tea runs on its own goroutine, concurrently with +// whatever Update call comes next — while applyWorkspaceState writes +// m.sizeMaster[dest] = ... IN PLACE on the Update goroutine. Go's runtime +// treats a concurrent map read/write as FATAL, unconditionally: it killed +// this very test with "fatal error: concurrent map read and map write" the +// first time this test was written wrong (calling resizeAllPanes() itself, +// not just its returned closure, from the "Cmd" goroutine — see below). +// +// resizeAllPanes() is called EXACTLY ONCE here, synchronously, before either +// goroutine starts — that single call is what production does too: Update +// calls it on its own one goroutine, so THAT read of m.sizeMaster races +// nothing. What must never race is the RETURNED CLOSURE, which Bubble Tea +// runs on a separate goroutine and which this test then re-invokes many +// times concurrently with further broadcasts — exactly the shape "a Cmd +// still running when the next Update call lands" takes in production. +// +// forUpdate is a SEPARATE Model copy from base, exactly matching Bubble +// Tea's real shape: a Cmd closure holds the Model value from the Update call +// that created it, and the NEXT Update call produces its own separate +// value — no goroutine ever shares a Model STRUCT instance. A plain field +// (m.projects, say) is therefore never raced by this test. sizeMaster is +// different: it is a MAP, a reference type, so copying the struct copies the +// map HEADER only — both copies still point at the same underlying data, +// which is the entire bug. +// +// Run with dev.sh test-race internal/tui. To confirm it catches the OLD +// code: move the `if m.isFollower(proj.Dest) { continue }` line from before +// `return func() tea.Msg {` in resizeAllPanes back inside the closure, and +// re-run — -race fails (and often crashes outright, with no -race needed). +func TestResizeAllPanes_NoRaceAgainstConcurrentMasterChanges(t *testing.T) { + t.Parallel() + base, _ := tinyTermModel(t) + base.SetClientID("me") + next, cmd0 := base.Update(tea.WindowSizeMsg{Width: 172, Height: 48}) + runCmd(cmd0) + base = next.(Model) + // This client is its OWN reported master at the moment resizeAllPanes() + // is called — a real, pre-existing, non-nil map (so it is genuinely + // SHARED with forUpdate below, the way a live sizeMaster map always is), + // and NOT a follower yet, so the pane below is actually included in the + // batch this call computes. A follower start would make the (correct) + // outer gate skip the destination entirely, leaving the closure nothing + // to iterate and the whole test vacuous. + base.sizeMaster = map[string]string{"": "me"} + + forUpdate := base + + cmd := base.resizeAllPanes() + if cmd == nil { + t.Fatal("resizeAllPanes returned nil") + } + + state := WorkspaceStateMsg{ + Dest: "", SizeMaster: "client-a", + ActiveProject: "proj-1", ActiveTab: "tab-1", + Projects: []ProjectInfo{{ID: "proj-1", Name: "Default", TabIDs: []string{"tab-1"}}}, + Tabs: []TabInfo{{ID: "tab-1", Name: "Shell", ProjectID: "proj-1", Panes: []string{"pane-1"}}}, + Panes: []PaneInfo{{ID: "pane-1", TabID: "tab-1", Type: "terminal"}}, + } + + const iterations = 300 + var wg sync.WaitGroup + wg.Add(2) + + // The Cmd-executor goroutine: re-runs the SAME returned closure many + // times — a stress-test proxy for "however late this closure actually + // runs, and however many broadcasts have landed by then, it must still + // touch nothing shared." + go func() { + defer wg.Done() + for i := 0; i < iterations; i++ { + cmd() + } + }() + + // The Update goroutine: applies broadcasts that flip the size master — + // applyWorkspaceState's real write path, called directly (as Update + // does), so this test isolates the sizeMaster race from the unrelated + // pointer-graph writes resizeTabs makes elsewhere in the WorkspaceStateMsg + // arm — this test's business is the map, not that graph. + go func() { + defer wg.Done() + for i := 0; i < iterations; i++ { + s := state + if i%2 == 0 { + s.SizeMaster = "client-a" + } else { + s.SizeMaster = "client-b" + } + forUpdate.applyWorkspaceState(s, "") + } + }() + + wg.Wait() +} + +// TestResizeAllPanes_MixedSession_FollowerOnOneDestMasterOnAnother covers D8: +// a TUI can be the master on one daemon and a follower on another, and each +// destination's resize decision must be independent. +func TestResizeAllPanes_MixedSession_FollowerOnOneDestMasterOnAnother(t *testing.T) { + t.Parallel() + localTab := tabWithPane("tab-local", "pane-local") + localTab.Resize(80, 24) + gpuTab := tabWithPane("tab-gpu", "pane-gpu") + gpuTab.Resize(80, 24) + + local, gpu := newFakeConn(), newFakeConn() + r := NewRouter(map[string]Client{"": local, "gpu01": gpu}) + + m := Model{ + client: r, sized: true, width: 172, height: 48, + tabDragFromIdx: -1, + projects: []*ProjectModel{ + {ID: "proj-local", Name: "Local", Dest: "", tabs: []*TabModel{localTab}}, + {ID: "proj-gpu", Name: "GPU", Dest: "gpu01", tabs: []*TabModel{gpuTab}}, + }, + } + m.SetClientID("me") + // No master reported locally (this client acts as master by default); + // someone else is master on gpu01. + m.sizeMaster = map[string]string{"gpu01": "other-client"} + + cmd := m.resizeAllPanes() + if cmd == nil { + t.Fatal("resizeAllPanes returned nil") + } + runCmd(cmd) + + if !sawResizeFor(t, local, "pane-local") { + t.Error("the destination this client masters must be resized") + } + if sawResizePanes(gpu) { + t.Error("the destination this client FOLLOWS must not be resized") + } +} + // TestAttach_CarriesClientID: attachMessage must carry the process-minted id // on every attach and reattach. func TestAttach_CarriesClientID(t *testing.T) { @@ -293,6 +464,63 @@ func TestAttach_CarriesClientID(t *testing.T) { } } +// TestClientGeometryCmd_ReportsZeroWhenUnpaintable mirrors +// TestAttachMessage_ReportsNoGeometryBelowTheMinimum: a terminal too small to +// paint must report 0x0, never the raw sub-floor size. The daemon uses this +// exact rule for master eligibility (§3.3: "the raw values are used, never +// the 80x24 default... a console-less client attaches at 0x0 and must never +// be elected") — client_geometry has to honour the same floor after attach, +// or a window that shrank below it would still look eligible. +func TestClientGeometryCmd_ReportsZeroWhenUnpaintable(t *testing.T) { + t.Parallel() + for _, tt := range []struct { + name string + width, height int + wantZero bool + }{ + {"console-less client", 1, 1, true}, + {"one column short", minTermWidth - 1, 40, true}, + {"one row short", 120, minTermHeight - 1, true}, + {"exactly the minimum", minTermWidth, minTermHeight, false}, + {"ordinary terminal", 172, 48, false}, + } { + t.Run(tt.name, func(t *testing.T) { + conn := newFakeConn() + m := Model{client: conn, width: tt.width, height: tt.height} + cmd := m.clientGeometryCmd() + if cmd == nil { + t.Fatal("clientGeometryCmd returned nil") + } + runCmd(cmd) + + var p ipc.ClientGeometryPayload + var found bool + for _, msg := range conn.sent { + if msg.Type == ipc.MsgClientGeometry { + if err := msg.DecodePayload(&p); err != nil { + t.Fatalf("decode client_geometry: %v", err) + } + found = true + } + } + if !found { + t.Fatal("no client_geometry sent") + } + if tt.wantZero { + if p.Cols != 0 || p.Rows != 0 { + t.Errorf("geometry = %dx%d at %dx%d, want 0x0", + p.Cols, p.Rows, tt.width, tt.height) + } + return + } + if p.Cols != tt.width || p.Rows != tt.height { + t.Errorf("geometry = %dx%d, want %dx%d (the raw window size)", + p.Cols, p.Rows, tt.width, tt.height) + } + }) + } +} + // TestCloseClient_SendsDetachBeforeClose: the recording client must see // detach queued BEFORE the conn is released — Close discards whatever is // still in the send queue, so the order is the whole point (D3/3.3). @@ -311,20 +539,81 @@ func TestCloseClient_SendsDetachBeforeClose(t *testing.T) { t.Run("single connection", func(t *testing.T) { conn := newFakeConn() + var closed int m := Model{client: conn} - m.SetClientCloser(func(c Client) { check(t, c.(*fakeConn)) }) + m.SetClientCloser(func(c Client) { + closed++ + check(t, c.(*fakeConn)) + }) m.CloseClient() + // The closer running is what makes check() mean anything — a mutation + // that dropped the closeClient(c) call entirely would leave check() + // never invoked, and this test would pass having asserted nothing. + if closed != 1 { + t.Errorf("closer ran %d times, want 1", closed) + } }) t.Run("router", func(t *testing.T) { local, gpu := newFakeConn(), newFakeConn() r := NewRouter(map[string]Client{"": local, "gpu01": gpu}) + var closed int m := Model{client: r} - m.SetClientCloser(func(c Client) { check(t, c.(*fakeConn)) }) + m.SetClientCloser(func(c Client) { + closed++ + check(t, c.(*fakeConn)) + }) m.CloseClient() + if closed != 2 { + t.Errorf("closer ran %d times, want 2 (one per conn)", closed) + } }) } +// TestCloseClient_DoesNotHangOnAWedgedConn covers the fix for a real +// usability bug: ipc.Client.Send can block up to clientSendTimeout (5s) +// against a peer whose must-deliver queue never drains, and CloseClient used +// to detach each conn SEQUENTIALLY — so a handful of dead remote hosts could +// turn Ctrl+Q into a many-second hang. detachTimeout bounds the wait; a conn +// whose Send never returns must still be closed once the budget expires. +func TestCloseClient_DoesNotHangOnAWedgedConn(t *testing.T) { + old := detachTimeout + detachTimeout = 20 * time.Millisecond + t.Cleanup(func() { detachTimeout = old }) + + wedged := &blockingSendConn{block: make(chan struct{})} // never closed + t.Cleanup(func() { close(wedged.block) }) // let the goroutine finish, don't leak it + + var closed int + m := Model{client: wedged} + m.SetClientCloser(func(Client) { closed++ }) + + done := make(chan struct{}) + go func() { + m.CloseClient() + close(done) + }() + + select { + case <-done: + case <-time.After(2 * time.Second): + t.Fatal("CloseClient hung on a wedged conn's detach send") + } + if closed != 1 { + t.Errorf("closer ran %d times, want 1 — a wedged detach must not stop the conn from being closed", closed) + } +} + +// blockingSendConn's Send never returns until block is closed, simulating a +// peer whose must-deliver queue is permanently full. +type blockingSendConn struct{ block chan struct{} } + +func (b *blockingSendConn) Send(*ipc.Message) error { + <-b.block + return nil +} +func (b *blockingSendConn) Receive() (*ipc.Message, error) { return nil, nil } + // TestStatusBar_RoleMarkerOnlyWithTwoClients: the marker is shown only once a // second client exists, and names the right role. func TestStatusBar_RoleMarkerOnlyWithTwoClients(t *testing.T) { diff --git a/internal/tui/reconnect.go b/internal/tui/reconnect.go index 0ce92cb1..25229647 100644 --- a/internal/tui/reconnect.go +++ b/internal/tui/reconnect.go @@ -6,6 +6,7 @@ import ( "log" "math/rand" "strings" + "sync" "time" tea "charm.land/bubbletea/v2" @@ -400,6 +401,23 @@ func sendDetach(c Client) { } } +// detachTimeout bounds how long CloseClient waits for every conn's detach +// send to complete before moving on to the flush/close path. +// +// ipc.Client.Send is NOT a plain enqueue — it routes through SendBlocking and +// can wait up to clientSendTimeout (5s, internal/ipc) against a peer whose +// must-deliver queue stays full, e.g. a remote host whose link died without +// the reconnect ladder having noticed yet. CloseClient can be releasing +// several such conns at once, and detaching them ONE AT A TIME would let a +// handful of dead hosts turn quitting the TUI itself into a multi-second (or +// multi-ten-second) hang. 500ms is generous for the healthy case — an +// ordinary Send returns in microseconds — and short enough that a wedged +// host costs the user nothing beyond it. +// +// A var, not a const, mirroring clientSendTimeout's own reasoning: a test +// that wants to prove the bound without actually waiting 500ms shrinks it. +var detachTimeout = 500 * time.Millisecond + // CloseClient releases every connection the Model currently holds. Called by // cmd/quil on exit, after the Bubble Tea program has returned. // @@ -409,16 +427,47 @@ func sendDetach(c Client) { // child and every remote `quil --stdio` outlived the client, on top of the // per-reconnect leak retire used to cause. cmd/quil's own `defer client.Close()` // cannot cover this either: it captured the startup conn of ONE destination. +// +// Every conn's detach is sent CONCURRENTLY and the whole batch is bounded at +// detachTimeout — see its doc comment. Order is still preserved for every +// conn that finishes within the budget: detach is queued (Send) before +// closeClient runs for that conn, and closeClient's own Flush is what +// actually carries it to the socket. A conn that is still wedged past the +// budget is closed anyway; its detach send may or may not have reached the +// socket, which is no worse than the lost-link grace period it would +// otherwise have cost the next election. Any straggling sendDetach goroutine +// left running past the budget dies with the process — CloseClient runs on +// the exit path, with nothing left to join it. func (m Model) CloseClient() { + var conns []Client if r, ok := m.client.(*Router); ok { - for _, c := range r.Conns() { + conns = r.Conns() + } else { + conns = []Client{m.client} + } + + var wg sync.WaitGroup + for _, c := range conns { + c := c + wg.Add(1) + go func() { + defer wg.Done() sendDetach(c) - m.closeClient(c) - } - return + }() + } + allSent := make(chan struct{}) + go func() { + wg.Wait() + close(allSent) + }() + select { + case <-allSent: + case <-time.After(detachTimeout): + } + + for _, c := range conns { + m.closeClient(c) } - sendDetach(m.client) - m.closeClient(m.client) } // canReconnect reports whether a dropped link to dest should be retried rather diff --git a/internal/tui/template_layout_test.go b/internal/tui/template_layout_test.go index 633e8757..c617150b 100644 --- a/internal/tui/template_layout_test.go +++ b/internal/tui/template_layout_test.go @@ -112,17 +112,17 @@ func TestTemplateWorkspace_PreparingSwapAndCompletion_BuildsAndReportsOnlyComple preparing.Panes[0].PreparingWorktree = "feat/template" update(preparing) runCmd(m.sendAllLayouts()) // Even an unrelated resize/action must not persist preparation. - if layouts, _ := sentCounts(recorder); layouts != 0 { + if layouts, _ := sentCounts(t, recorder); layouts != 0 { t.Fatal("saved preparing placeholder") } update(templateWorkspace([]string{"first"}, "placeholder")) - if layouts, _ := sentCounts(recorder); layouts != 0 { + if layouts, _ := sentCounts(t, recorder); layouts != 0 { t.Fatal("saved intermediate swap") } incomplete := templateWorkspace([]string{"first", "second", "main"}, "main") incomplete.Panes = incomplete.Panes[:2] update(incomplete) - if layouts, _ := sentCounts(recorder); layouts != 0 { + if layouts, _ := sentCounts(t, recorder); layouts != 0 { t.Fatal("saved before every declared pane existed") } complete := templateWorkspace([]string{"first", "second", "main"}, "main") @@ -132,7 +132,7 @@ func TestTemplateWorkspace_PreparingSwapAndCompletion_BuildsAndReportsOnlyComple if got := templateShape(root); got != "H(main,V(first,second))" { t.Fatal(got) } - if layouts, _ := sentCounts(recorder); layouts != 1 { + if layouts, _ := sentCounts(t, recorder); layouts != 1 { t.Fatalf("layout sends=%d want=1", layouts) } var saved ipc.UpdateLayoutPayload @@ -151,7 +151,7 @@ func TestTemplateWorkspace_PreparingSwapAndCompletion_BuildsAndReportsOnlyComple if tab.Root != root { t.Fatal("rebuilt the tree on the second broadcast") } - if layouts, _ := sentCounts(recorder); layouts != 1 { + if layouts, _ := sentCounts(t, recorder); layouts != 1 { t.Fatal("echoed the saved layout") } // User changes survive subsequent completed frames too. From c2440c5d7fb3159980cda89b13ea0b54c7b8beb9 Mon Sep 17 00:00:00 2001 From: Artjoms Stukans Date: Fri, 25 Sep 2026 21:54:03 +0200 Subject: [PATCH 13/40] feat(tui): render follower panes at the master's size A follower TUI now sizes each pane's VT to the size the daemon's master chose for it, never to its own box. targetVTSize is the single decision point, used by resizeNode and sizePaneFull (layout leaves, focus mode and overlays); the resize producers keep using paneVTSize, since a master sends its own rect size and a follower sends nothing. A pane with no daemon size yet falls back to its rect. The follower flag and the daemon size reach PaneModel through syncPaneMeta (a new follower parameter) and, between broadcasts, through the pane_sizes frame. The listener decodes MsgPaneSizes beside set_active_pane, and Update applies it synchronously, so the repaint that follows the master's resize lands in a VT of the new size. applyWorkspaceState now records a destination's size master before it rebuilds the panes, or the first broadcast naming another master would still size the VTs by the previous one. previewMode is true for a follower whose grid exceeds its box in either dimension, reusing the preview's left-edge crop and bottom anchor. A grid that fits takes the native path, drawn top-left and padded. The top border marks a cut with a corner replaced by an ellipsis: top-left for hidden rows, top-right for hidden columns. Wheel forwarding to a tracking app maps box coordinates to grid coordinates, and a notch over the padding sends nothing. toggle_wrap now reaches follower panes too. --- internal/tui/deletion_mark_test.go | 4 +- internal/tui/follower_render_test.go | 547 ++++++++++++++++++++++++ internal/tui/gitworktreename_test.go | 4 +- internal/tui/layout.go | 5 +- internal/tui/model.go | 142 ++++-- internal/tui/modelinfo_test.go | 4 +- internal/tui/pane.go | 98 ++++- internal/tui/pane_preview.go | 16 +- internal/tui/pinned_attention_test.go | 4 +- internal/tui/restore_indicator_test.go | 6 +- internal/tui/spawnerror_test.go | 4 +- internal/tui/tab.go | 5 +- internal/tui/unseen_mark_test.go | 10 +- internal/tui/wheel_forward_test.go | 2 +- internal/tui/workstate.go | 10 +- internal/tui/workstate_test.go | 8 +- internal/tui/worktree_preparing_test.go | 4 +- 17 files changed, 810 insertions(+), 63 deletions(-) create mode 100644 internal/tui/follower_render_test.go diff --git a/internal/tui/deletion_mark_test.go b/internal/tui/deletion_mark_test.go index 6da37fb1..ffb6d69e 100644 --- a/internal/tui/deletion_mark_test.go +++ b/internal/tui/deletion_mark_test.go @@ -52,11 +52,11 @@ func TestParseWorkspaceState_ReadsTheDeletionMarkWireKey(t *testing.T) { func TestSyncPaneMeta_AdoptsTheDaemonsDeletionMark(t *testing.T) { t.Parallel() pane := &PaneModel{ID: "p1"} - syncPaneMeta(pane, &PaneInfo{ID: "p1", MarkedForDeletion: true}, false, 0, false) + syncPaneMeta(pane, &PaneInfo{ID: "p1", MarkedForDeletion: true}, false, 0, false, false) if !pane.markedForDeletion { t.Fatal("a daemon-reported deletion mark was not adopted") } - syncPaneMeta(pane, &PaneInfo{ID: "p1", MarkedForDeletion: false}, false, 0, false) + syncPaneMeta(pane, &PaneInfo{ID: "p1", MarkedForDeletion: false}, false, 0, false, false) if pane.markedForDeletion { t.Error("a daemon-reported CLEAR was not adopted — the copy must be unconditional, " + "or an unmark from another client can never reach this one") diff --git a/internal/tui/follower_render_test.go b/internal/tui/follower_render_test.go new file mode 100644 index 00000000..ecf7c59b --- /dev/null +++ b/internal/tui/follower_render_test.go @@ -0,0 +1,547 @@ +package tui + +import ( + "fmt" + "strings" + "testing" + "time" + + tea "charm.land/bubbletea/v2" + "charm.land/lipgloss/v2" + + "github.com/artyomsv/quil/internal/ipc" +) + +// Multi-client sync, follower rendering (spec §5). A follower shows each pane +// at the size the MASTER chose — its VT takes the daemon's cols/rows, never its +// own box — and cuts or pads that grid into the box it has. Every decision is +// driven through Model.Update: the follower flag reaches the pane only through +// a broadcast, so a test that set it directly could pass against a syncPaneMeta +// that never copies it. + +// followerFixture lays out paneIDs in one tab with NO master (so the rects +// settle exactly as they do today), then applies a second broadcast naming +// `master` as the size master and reporting, per pane, the cols/rows sizeFn +// derives from that pane's inner box. mutate may adjust the second broadcast +// before it is applied. The returned conn has an empty send log. +func followerFixture(t *testing.T, master string, paneIDs []string, + sizeFn func(id string, innerW, innerH int) (cols, rows int), + mutate func(*WorkspaceStateMsg)) (Model, *fakeConn) { + t.Helper() + t.Setenv("QUIL_HOME", t.TempDir()) + m, conn := tinyTermModel(t) + m.SetClientID("me") + m.notifications = NewNotificationCenter(30, 50) // the mouse and View paths read it + // Receive() must not park the listen command each broadcast re-arms. + close(conn.recv) + + next, cmd := m.Update(tea.WindowSizeMsg{Width: 124, Height: 44}) + runCmd(cmd) + m = next.(Model) + + st := WorkspaceStateMsg{ + Dest: "", Clients: 2, + ActiveProject: "proj-1", ActiveTab: "tab-1", + Projects: []ProjectInfo{{ID: "proj-1", Name: "Default", TabIDs: []string{"tab-1"}}}, + Tabs: []TabInfo{{ID: "tab-1", Name: "Shell", ProjectID: "proj-1", Panes: paneIDs}}, + } + for _, id := range paneIDs { + st.Panes = append(st.Panes, PaneInfo{ID: id, TabID: "tab-1", Type: "terminal"}) + } + next, cmd = m.Update(st) + runCmd(cmd) + m = next.(Model) + + st.SizeMaster = master + for i := range st.Panes { + p := paneByID(t, m, st.Panes[i].ID) + c, r := sizeFn(p.ID, p.Width-2, p.Height-2) + st.Panes[i].Cols, st.Panes[i].Rows = uint16(c), uint16(r) + } + if mutate != nil { + mutate(&st) + } + next, cmd = m.Update(st) + runCmd(cmd) + m = next.(Model) + clearSent(conn) + return m, conn +} + +// paneByID resolves a tree pane or an overlay pane of the fixture's model. +func paneByID(t *testing.T, m Model, id string) *PaneModel { + t.Helper() + for _, tab := range m.allTabs() { + if tab.overlayPane != nil && tab.overlayPane.ID == id { + return tab.overlayPane + } + if tab.Root != nil { + if leaf := tab.Root.FindLeaf(id); leaf != nil { + return leaf.Pane + } + } + } + t.Fatalf("pane %s not found", id) + return nil +} + +func fixedSize(cols, rows int) func(string, int, int) (int, int) { + return func(string, int, int) (int, int) { return cols, rows } +} + +func vtSize(p *PaneModel) (int, int) { return p.vt.Width(), p.vt.Height() } + +// feedOutput delivers bytes to a pane through Update, as the listener would. +func feedOutput(t *testing.T, m Model, paneID, data string) Model { + t.Helper() + next, cmd := m.Update(PaneOutputMsg{PaneID: paneID, Data: []byte(data)}) + _ = cmd // settle ticks and the listen re-arm are irrelevant here + return next.(Model) +} + +// numberedRows writes n rows "L00".."L", the last without a newline so a +// screen of exactly n rows does not scroll. +func numberedRows(n int) string { + var b strings.Builder + for i := 0; i < n; i++ { + if i > 0 { + b.WriteString("\r\n") + } + fmt.Fprintf(&b, "L%02d", i) + } + return b.String() +} + +// assertExactBox checks every rendered line of a pane is exactly the pane's +// width and that there are exactly Height lines — a follower's grid is never +// allowed to widen, narrow or lengthen the box it is drawn in. +func assertExactBox(t *testing.T, p *PaneModel, view string) { + t.Helper() + lines := strings.Split(view, "\n") + if len(lines) != p.Height { + t.Errorf("rendered %d lines, want %d (the pane height)", len(lines), p.Height) + } + for i, l := range lines { + if w := lipgloss.Width(l); w != p.Width { + t.Errorf("line %d width = %d, want %d: %q", i, w, p.Width, stripANSI(l)) + } + } +} + +func TestFollower_VTTakesDaemonSize(t *testing.T) { + m, _ := followerFixture(t, "other", []string{"pane-1"}, fixedSize(200, 50), nil) + p := paneByID(t, m, "pane-1") + if p.Width-2 >= 200 || p.Height-2 >= 50 { + t.Fatalf("setup: box %dx%d must be smaller than the daemon size", p.Width, p.Height) + } + if c, r := vtSize(p); c != 200 || r != 50 { + t.Fatalf("follower VT = %dx%d, want the daemon's 200x50", c, r) + } + + next, _ := m.Update(paneSizesMsg{dest: "", sizes: []ipc.ResizePanePayload{{PaneID: "pane-1", Cols: 180, Rows: 45}}}) + m = next.(Model) + if c, r := vtSize(paneByID(t, m, "pane-1")); c != 180 || r != 45 { + t.Fatalf("after pane_sizes VT = %dx%d, want 180x45", c, r) + } +} + +// A pane_sizes frame for a DIFFERENT destination must not touch a pane that +// merely shares an id — sizes describe one daemon's panes. +func TestFollower_PaneSizesMsgScopedToDest(t *testing.T) { + m, _ := followerFixture(t, "other", []string{"pane-1"}, fixedSize(200, 50), nil) + next, _ := m.Update(paneSizesMsg{dest: "remote-a", sizes: []ipc.ResizePanePayload{{PaneID: "pane-1", Cols: 180, Rows: 45}}}) + m = next.(Model) + if c, r := vtSize(paneByID(t, m, "pane-1")); c != 200 || r != 50 { + t.Fatalf("VT = %dx%d, want 200x50 — another destination's sizes applied", c, r) + } +} + +// The listener decodes MsgPaneSizes into a paneSizesMsg stamped with the +// frame's origin, and the Update arm renders (it changes the frame). +func TestFollower_ListenerDecodesPaneSizes(t *testing.T) { + m, conn := tinyTermModel(t) + msg, err := ipc.NewMessage(ipc.MsgPaneSizes, ipc.PaneSizesPayload{ + Panes: []ipc.ResizePanePayload{{PaneID: "pane-1", Cols: 90, Rows: 30}}, + }) + if err != nil { + t.Fatal(err) + } + msg.Origin = "host-b" + conn.recv <- msg + got := m.listenForMessages()() + ps, ok := got.(paneSizesMsg) + if !ok { + t.Fatalf("listener returned %T, want paneSizesMsg", got) + } + if ps.dest != "host-b" || len(ps.sizes) != 1 || ps.sizes[0].Cols != 90 || ps.sizes[0].Rows != 30 { + t.Fatalf("decoded %+v", ps) + } + next, _ := m.Update(ps) + if next.(Model).skipRender { + t.Fatal("a pane_sizes frame changes the frame and must not be skipRender") + } +} + +// §4.1 / 8a: the size arrives BEFORE the repaint, so the repaint must land in +// the new-sized VT. 185 cells in a 180-column grid wrap to row 1 col 5; in the +// old 200-column grid they would not wrap at all. +func TestFollower_PaneSizesMsgResizesBeforeOutput(t *testing.T) { + m, _ := followerFixture(t, "other", []string{"pane-1"}, fixedSize(200, 50), nil) + next, _ := m.Update(paneSizesMsg{dest: "", sizes: []ipc.ResizePanePayload{{PaneID: "pane-1", Cols: 180, Rows: 45}}}) + m = next.(Model) + m = feedOutput(t, m, "pane-1", strings.Repeat("x", 185)) + pos := paneByID(t, m, "pane-1").vt.CursorPosition() + if pos.X != 5 || pos.Y != 1 { + t.Fatalf("cursor at (%d,%d), want (5,1) — output landed in a VT of the old width", pos.X, pos.Y) + } +} + +func TestFollower_RenderTooWideCropsLeftWithMarker(t *testing.T) { + var innerW int + m, _ := followerFixture(t, "other", []string{"pane-1"}, func(_ string, w, h int) (int, int) { + innerW = w + return w + 30, h + }, nil) + row := strings.Repeat("a", innerW) + "BBBB" + m = feedOutput(t, m, "pane-1", row) + p := paneByID(t, m, "pane-1") + view := p.View() + lines := strings.Split(view, "\n") + first := stripANSI(lines[1]) + if !strings.Contains(first, strings.Repeat("a", innerW)) || strings.Contains(first, "B") { + t.Errorf("first row %q: want the LEFT %d columns only", first, innerW) + } + top := []rune(stripANSI(lines[0])) + if top[len(top)-1] != '…' { + t.Errorf("top border %q: want a … at the right end for a width cut", string(top)) + } + if top[0] == '…' { + t.Errorf("top border %q: no height cut, so no … at the left end", string(top)) + } + assertExactBox(t, p, view) +} + +func TestFollower_RenderTooTallShowsBottomRowsWithMarker(t *testing.T) { + var innerH int + m, _ := followerFixture(t, "other", []string{"pane-1"}, func(_ string, w, h int) (int, int) { + innerH = h + return w, h + 10 + }, nil) + m = feedOutput(t, m, "pane-1", numberedRows(innerH+10)) + p := paneByID(t, m, "pane-1") + view := p.View() + lines := strings.Split(view, "\n") + if got := stripANSI(lines[1]); !strings.Contains(got, "L10") { + t.Errorf("first visible row %q: want L10 (bottom-anchored, 10 rows cut)", got) + } + if got := stripANSI(lines[innerH]); !strings.Contains(got, fmt.Sprintf("L%02d", innerH+9)) { + t.Errorf("last visible row %q: want L%02d", got, innerH+9) + } + top := []rune(stripANSI(lines[0])) + if top[0] != '…' { + t.Errorf("top border %q: want a … at the left end for a height cut", string(top)) + } + if top[len(top)-1] == '…' { + t.Errorf("top border %q: no width cut, so no … at the right end", string(top)) + } + assertExactBox(t, p, view) +} + +func TestFollower_RenderSmallerIsPadded(t *testing.T) { + m, _ := followerFixture(t, "other", []string{"pane-1"}, fixedSize(20, 5), nil) + m = feedOutput(t, m, "pane-1", numberedRows(5)) + p := paneByID(t, m, "pane-1") + if p.previewMode() { + t.Fatal("a grid that fits must use the native renderer, not the preview") + } + view := p.View() + lines := strings.Split(view, "\n") + if got := stripANSI(lines[1]); !strings.HasPrefix(got, "│L00") { + t.Errorf("first row %q: want the grid drawn top-left", got) + } + for i := 6; i < len(lines)-1; i++ { + if got := strings.Trim(stripANSI(lines[i]), "│ "); got != "" { + t.Errorf("row %d %q: want blank padding below the grid", i, got) + } + } + top := stripANSI(lines[0]) + if strings.Contains(top, "…") { + t.Errorf("top border %q: nothing is cut, so no marker", top) + } + assertExactBox(t, p, view) +} + +func TestFollower_RenderEveryRowExactWidth(t *testing.T) { + cases := map[string]func(string, int, int) (int, int){ + "wide": func(_ string, w, h int) (int, int) { return w + 40, h }, + "tall": func(_ string, w, h int) (int, int) { return w, h + 7 }, + "both": func(_ string, w, h int) (int, int) { return w + 40, h + 7 }, + "smaller": func(_ string, w, h int) (int, int) { return w / 2, h / 2 }, + } + for name, fn := range cases { + t.Run(name, func(t *testing.T) { + m, _ := followerFixture(t, "other", []string{"pane-1"}, fn, nil) + p := paneByID(t, m, "pane-1") + c, r := vtSize(p) + var b strings.Builder + for i := 0; i < r; i++ { + if i > 0 { + b.WriteString("\r\n") + } + b.WriteString(strings.Repeat(string(rune('a'+i%26)), c)) + } + m = feedOutput(t, m, "pane-1", b.String()) + assertExactBox(t, p, p.View()) + // Scrolled back too: the scrollbar column must not widen a row. + p.ScrollUp(2) + assertExactBox(t, p, p.View()) + }) + } +} + +// Spec §5.2: scrollback sits ABOVE the visible rows, so the wheel first +// reveals the screen rows the box cut, then the history. +func TestFollower_WheelRevealsCutRowsFirst(t *testing.T) { + var innerH int + m, _ := followerFixture(t, "other", []string{"pane-1"}, func(_ string, w, h int) (int, int) { + innerH = h + return w, h + 10 + }, nil) + m = feedOutput(t, m, "pane-1", numberedRows(innerH+10)) + rect := m.activePaneRect() + if rect == nil { + t.Fatal("no active pane rect") + } + next, _ := m.Update(tea.MouseWheelMsg{X: rect.OX + 3, Y: rect.OY + 3, Button: tea.MouseWheelUp}) + m = next.(Model) + p := paneByID(t, m, "pane-1") + lines := strings.Split(p.View(), "\n") + lines = lines[1:] // drop the top border + notch := m.cfg.UI.MouseScrollLines + if notch < 1 { + notch = 3 // the handler's own default + } + want := fmt.Sprintf("L%02d", 10-notch) + if got := stripANSI(lines[0]); !strings.Contains(got, want) { + t.Errorf("after one notch the top row is %q, want %s (a cut screen row)", got, want) + } +} + +// wheelForwarded returns the bytes the input queue received. +func wheelForwarded(m Model) string { + var got strings.Builder + for { + select { + case in := <-m.inputCh: + got.Write(in.data) + default: + return got.String() + } + } +} + +func trackingMutate(st *WorkspaceStateMsg) { + for i := range st.Panes { + st.Panes[i].MouseTracking, st.Panes[i].MouseSGR = true, true + } +} + +// Spec §5.3: box row r of a too-tall follower is grid row r + (vtH - innerH). +func TestFollower_WheelForwardTranslatedToGridRow(t *testing.T) { + var innerH int + m, _ := followerFixture(t, "other", []string{"pane-1"}, func(_ string, w, h int) (int, int) { + innerH = h + return w, h + 10 + }, trackingMutate) + m.inputCh = make(chan paneInput, inputForwardBuffer) + rect := m.activePaneRect() + relX, relY := 4, 3 + next, _ := m.Update(tea.MouseWheelMsg{X: rect.OX + 1 + relX, Y: rect.OY + 1 + relY, Button: tea.MouseWheelUp}) + m = next.(Model) + want := fmt.Sprintf("\x1b[<64;%d;%dM", relX+1, relY+10+1) + if got := wheelForwarded(m); got != want { + t.Fatalf("forwarded %q, want %q (box row %d + %d cut rows)", got, want, relY, 10) + } + _ = innerH +} + +// Spec §5.3: a position in the padding of a grid smaller than the box sends +// nothing — there is no grid cell there for the app to receive. +func TestFollower_WheelInPaddingNotForwarded(t *testing.T) { + m, _ := followerFixture(t, "other", []string{"pane-1"}, fixedSize(20, 5), trackingMutate) + m.inputCh = make(chan paneInput, inputForwardBuffer) + rect := m.activePaneRect() + for _, pos := range [][2]int{{25, 2}, {3, 8}} { + next, _ := m.Update(tea.MouseWheelMsg{X: rect.OX + 1 + pos[0], Y: rect.OY + 1 + pos[1], Button: tea.MouseWheelUp}) + m = next.(Model) + if got := wheelForwarded(m); got != "" { + t.Errorf("wheel at padding (%d,%d) forwarded %q, want nothing", pos[0], pos[1], got) + } + } + // Inside the grid it is forwarded unchanged: the grid is top-left. + next, _ := m.Update(tea.MouseWheelMsg{X: rect.OX + 1 + 3, Y: rect.OY + 1 + 2, Button: tea.MouseWheelUp}) + m = next.(Model) + if got, want := wheelForwarded(m), "\x1b[<64;4;3M"; got != want { + t.Errorf("wheel inside the grid forwarded %q, want %q", got, want) + } +} + +// Spec 8a: the two sizing sites outside the layout walk — sizePaneFull for +// focus mode and for an overlay — take the daemon size too. +func TestFollower_OverlayAndFocusModeUseDaemonSize(t *testing.T) { + sizes := map[string][2]int{"pane-1": {50, 12}, "pane-2": {52, 13}, "ov-1": {70, 25}} + m, _ := followerFixture(t, "other", []string{"pane-1", "pane-2"}, + func(id string, _, _ int) (int, int) { return sizes[id][0], sizes[id][1] }, + nil) + + next, cmd := m.toggleFocusForActiveTab() + runCmd(cmd) + m = next.(Model) + tab := m.activeTabModel() + if !tab.FocusMode() { + t.Fatal("setup: focus mode did not engage") + } + active := tab.ActivePaneModel() + if c, r := vtSize(active); c != sizes[active.ID][0] || r != sizes[active.ID][1] { + t.Errorf("focus-mode VT = %dx%d, want the daemon's %v (box is %dx%d)", c, r, sizes[active.ID], active.Width, active.Height) + } + next, cmd = m.toggleFocusForActiveTab() + runCmd(cmd) + m = next.(Model) + + // Overlay: announced by this client's Alt+G (pendingOverlayShow), so it + // arrives visible and is sized by sizePaneFull. + m.pendingOverlayShow = map[string]bool{"tab-1": true} + st := WorkspaceStateMsg{ + Dest: "", SizeMaster: "other", Clients: 2, + ActiveProject: "proj-1", ActiveTab: "tab-1", + Projects: []ProjectInfo{{ID: "proj-1", Name: "Default", TabIDs: []string{"tab-1"}}}, + Tabs: []TabInfo{{ID: "tab-1", Name: "Shell", ProjectID: "proj-1", Panes: []string{"pane-1", "pane-2", "ov-1"}}}, + } + for _, id := range []string{"pane-1", "pane-2", "ov-1"} { + st.Panes = append(st.Panes, PaneInfo{ID: id, TabID: "tab-1", Type: "terminal", + Overlay: id == "ov-1", Cols: uint16(sizes[id][0]), Rows: uint16(sizes[id][1])}) + } + next, cmd = m.Update(st) + runCmd(cmd) + m = next.(Model) + tab = m.activeTabModel() + if !tab.overlayVisible || tab.overlayPane == nil { + t.Fatal("setup: overlay not visible") + } + if c, r := vtSize(tab.overlayPane); c != 70 || r != 25 { + t.Errorf("overlay VT = %dx%d, want the daemon's 70x25 (box is %dx%d)", c, r, tab.overlayPane.Width, tab.overlayPane.Height) + } +} + +// runCmdNoWait runs cmd like runCmd but abandons a leaf that has not returned +// within a short budget: a tea.Tick sleeps its whole interval (the notes +// autosave tick is seconds), and nothing a tick produces matters here. Every +// send under test happens synchronously inside its own leaf. +func runCmdNoWait(cmd tea.Cmd) { + if cmd == nil { + return + } + done := make(chan tea.Msg, 1) + go func() { done <- cmd() }() + select { + case msg := <-done: + if batch, ok := msg.(tea.BatchMsg); ok { + for _, c := range batch { + runCmdNoWait(c) + } + } + case <-time.After(200 * time.Millisecond): + } +} + +// Review Focus 5: a follower's own rect changes — focus mode, notes mode, the +// notification sidebar, the project sidebar, its window — neither resize its +// VTs nor send a single resize frame. +func TestFollower_LocalRectChangesNeverResizeTheVT(t *testing.T) { + sizes := map[string][2]int{"pane-1": {61, 17}, "pane-2": {63, 11}} + m, conn := followerFixture(t, "other", []string{"pane-1", "pane-2"}, + func(id string, _, _ int) (int, int) { return sizes[id][0], sizes[id][1] }, + nil) + check := func(step string) { + t.Helper() + for id, want := range sizes { + if c, r := vtSize(paneByID(t, m, id)); c != want[0] || r != want[1] { + t.Errorf("%s: %s VT = %dx%d, want %dx%d", step, id, c, r, want[0], want[1]) + } + } + if n := countResizes(t, conn); n != 0 { + t.Errorf("%s: %d resize(s) sent, want 0", step, n) + } + } + + next, cmd := m.toggleFocusForActiveTab() + runCmd(cmd) + m = next.(Model) + m.View() + check("focus on") + next, cmd = m.toggleFocusForActiveTab() + runCmd(cmd) + m = next.(Model) + m.View() + check("focus off") + + next, cmd = m.toggleNotesMode() + runCmdNoWait(cmd) + m = next.(Model) + if !m.notesMode { + t.Fatal("setup: notes mode did not open") + } + m.View() + check("notes on") + next, cmd = m.toggleNotesMode() + runCmdNoWait(cmd) + m = next.(Model) + m.View() + check("notes off") + + m.notifications.visible = true + m.View() + check("notification sidebar") + m.notifications.visible = false + + next, cmd = m.toggleProjectSidebar() + runCmd(cmd) + m = next.(Model) + m.View() + check("project sidebar") + + next, _ = m.Update(tea.WindowSizeMsg{Width: 150, Height: 50}) + m = next.(Model) + next, cmd = m.Update(resizeTickMsg{seq: m.resizeSeq}) + runCmd(cmd) + m = next.(Model) + m.View() + check("window resize") +} + +// A master's pane renders exactly as before this feature: its VT follows its +// own rect (paneVTSize), whatever cols/rows the broadcast carries, and its +// frame is byte-identical to a pane sized the pre-change way. +func TestMaster_RenderUnchanged(t *testing.T) { + m, _ := followerFixture(t, "me", []string{"pane-1"}, fixedSize(200, 50), nil) + out := numberedRows(8) + "\r\n" + strings.Repeat("w", 300) + m = feedOutput(t, m, "pane-1", out) + p := paneByID(t, m, "pane-1") + wantC, wantR := paneVTSize(false, 0, p.Width, p.Height, p.NativeW, 0, 0) + if c, r := vtSize(p); c != wantC || r != wantR { + t.Fatalf("master VT = %dx%d, want its own rect's %dx%d", c, r, wantC, wantR) + } + + ref := NewPaneModel("pane-1", testRingBufSize) + t.Cleanup(ref.Dispose) + ref.Type, ref.Name, ref.CWD = p.Type, p.Name, p.CWD + ref.Active, ref.liveOutputSeen, ref.unseen = p.Active, p.liveOutputSeen, p.unseen + ref.ghost, ref.resuming, ref.preparing = p.ghost, p.resuming, p.preparing + ref.Width, ref.Height = p.Width, p.Height + ref.ResizeVT(wantC, wantR) + ref.AppendOutput([]byte(out)) + if got, want := p.View(), ref.View(); got != want { + t.Fatalf("master frame differs from the pre-change rendering:\n got: %q\nwant: %q", got, want) + } +} diff --git a/internal/tui/gitworktreename_test.go b/internal/tui/gitworktreename_test.go index 9b737a5d..bfe3142c 100644 --- a/internal/tui/gitworktreename_test.go +++ b/internal/tui/gitworktreename_test.go @@ -13,7 +13,7 @@ import ( // one it had — the same reason the block around it does not guard on empty. func TestSyncPaneMeta_ClearsTheWorktreeName(t *testing.T) { pane := &PaneModel{ID: "p1", GitWorktree: true, GitWorktreeName: "feat-x"} - syncPaneMeta(pane, &PaneInfo{ID: "p1"}, false, 0, false) + syncPaneMeta(pane, &PaneInfo{ID: "p1"}, false, 0, false, false) if pane.GitWorktreeName != "" { t.Errorf("GitWorktreeName = %q after an update with no git keys, want it cleared", pane.GitWorktreeName) @@ -22,7 +22,7 @@ func TestSyncPaneMeta_ClearsTheWorktreeName(t *testing.T) { func TestSyncPaneMeta_CarriesTheWorktreeName(t *testing.T) { pane := &PaneModel{ID: "p1"} - syncPaneMeta(pane, &PaneInfo{ID: "p1", GitWorktree: true, GitWorktreeName: "feat-x"}, false, 0, false) + syncPaneMeta(pane, &PaneInfo{ID: "p1", GitWorktree: true, GitWorktreeName: "feat-x"}, false, 0, false, false) if pane.GitWorktreeName != "feat-x" { t.Errorf("GitWorktreeName = %q, want \"feat-x\"", pane.GitWorktreeName) diff --git a/internal/tui/layout.go b/internal/tui/layout.go index 9e77c183..2dc4a998 100644 --- a/internal/tui/layout.go +++ b/internal/tui/layout.go @@ -634,7 +634,8 @@ func (n *LayoutNode) FindPaneRectAt(x, y, ox, oy, w, h int) *PaneRect { // resizeNode recursively assigns dimensions to each node. canvasW/canvasH // are the full tab-area dimensions — wide-canvas panes size their VT to -// the canvas (via paneVTSize) while their rect keeps following the tree. +// the canvas (via paneVTSize) while their rect keeps following the tree, +// and a follower pane's VT keeps the master's size (targetVTSize). // // fullW is w plus whatever the project sidebar reserved, split by the SAME // ratios all the way down, so each leaf learns the width it would have had @@ -667,7 +668,7 @@ func resizeNode(n *LayoutNode, w, h, fullW, canvasW, canvasH int) { n.Pane.Width = w n.Pane.Height = h n.Pane.NativeW = fullW - n.Pane.ResizeVT(paneVTSize(n.Pane.WideCanvas, n.Pane.MinNativeCols, w, h, fullW, canvasW, canvasH)) + n.Pane.ResizeVT(n.Pane.targetVTSize(w, h, fullW, canvasW, canvasH)) return } diff --git a/internal/tui/model.go b/internal/tui/model.go index 123c38fe..9fc78df4 100644 --- a/internal/tui/model.go +++ b/internal/tui/model.go @@ -244,6 +244,15 @@ type setActivePaneMsg struct { PaneID string } +// paneSizesMsg carries a daemon's pane_sizes frame: the sizes its master just +// applied, sent to this follower BEFORE the PTY resize, so it precedes the +// child's repaint on the same ordered connection. dest is the frame's Origin +// — sizes describe one daemon's panes and nobody else's. +type paneSizesMsg struct { + dest string + sizes []ipc.ResizePanePayload +} + // paneEventMsg delivers a notification event from the daemon. type paneEventMsg ipc.PaneEventPayload @@ -2467,6 +2476,13 @@ func (m Model) Update(msg tea.Msg) (retModel tea.Model, retCmd tea.Cmd) { } relX := msg.X - rect.OX - 1 relY := msg.Y - rect.OY - 1 + // A follower's grid is cut or padded into this box, + // so box coordinates are not grid coordinates; a notch + // over the padding has no cell to send. + relX, relY, inGrid := followerGridPos(pane, relX, relY) + if !inGrid { + return m, nil + } if seq := pane.wheelForwardSeq(up, relX, relY); seq != nil { logger.Debug("wheel: forward pane=%s type=%s btn=%v rel=(%d,%d) seq=%q (local n=%v b=%v a=%v sgr=%v daemonTrack=%v)", pane.ID, pane.Type, msg.Button, relX, relY, string(seq), @@ -2920,6 +2936,15 @@ func (m Model) Update(msg tea.Msg) (retModel tea.Model, retCmd tea.Cmd) { } return m, tea.Batch(overlayCmd, m.listenForMessages()) + case paneSizesMsg: + // Applied HERE, synchronously, never in a Cmd: the frame arrived on + // the must-deliver queue ahead of the repaint the master's resize + // triggers, and the very next PaneOutputMsg from this connection may + // be that repaint — it must land in a VT that already has the new + // size. Deliberately not skipRender: a resized VT is a changed frame. + m.applyPaneSizes(msg) + return m, m.listenForMessages() + case highlightPaneMsg: m.mcpHighlights[msg.PaneID] = true m.mcpHighlightSeq[msg.PaneID]++ @@ -5263,9 +5288,11 @@ func (m Model) handleKey(msg tea.KeyPressMsg) (tea.Model, tea.Cmd) { case "pane.toggle_wrap": // Flip the active wide-canvas pane's preview between left-edge // crop (default) and soft-wrap. View-only state — no IPC, no PTY - // touch; the preview layout cache re-keys on the flag. + // touch; the preview layout cache re-keys on the flag. A follower + // pane renders the same preview when its grid is cut (spec §5.2), + // so the toggle reaches it too. if tab := m.activeTabModel(); tab != nil { - if pane := tab.ActivePaneModel(); pane != nil && pane.WideCanvas { + if pane := tab.ActivePaneModel(); pane != nil && (pane.WideCanvas || pane.follower) { pane.previewWrap = !pane.previewWrap } } @@ -5977,6 +6004,30 @@ func (m *Model) applyWorkspaceState(state WorkspaceStateMsg, dest string) ([]str // the depth the budget allows. m.setDestPaneCount(dest, len(state.Panes)) + // Multi-client sync (§4.2): this destination's size master and attached- + // client count. Read the PREVIOUS master before overwriting it — becoming + // master is a TRANSITION, not a state, and the resize kick at the end of + // this function must fire once, on the broadcast that flips it, never on + // every later broadcast that merely reconfirms it. + // + // Recorded BEFORE the rebuild, not after it: every syncPaneMeta below + // copies isFollower(dest) onto its pane, and that flag decides which size + // the pane's VT takes (targetVTSize). Recording it afterwards would size + // every pane by the PREVIOUS broadcast's master — the first broadcast that + // makes this client a follower would still size its VTs to its own boxes. + var prevMaster string + if m.sizeMaster != nil { + prevMaster = m.sizeMaster[dest] + } + if m.sizeMaster == nil { + m.sizeMaster = make(map[string]string) + } + m.sizeMaster[dest] = state.SizeMaster + if m.clientCount == nil { + m.clientCount = make(map[string]int) + } + m.clientCount[dest] = state.Clients + paneMap := make(map[string]*PaneInfo) for i := range state.Panes { paneMap[state.Panes[i].ID] = &state.Panes[i] @@ -6075,23 +6126,6 @@ func (m *Model) applyWorkspaceState(state WorkspaceStateMsg, dest string) ([]str // destination, and it self-skips when nothing changed. m.cacheRemoteProjects(dest) - // Multi-client sync (§4.2): this destination's size master and attached- - // client count. Read the PREVIOUS master before overwriting it — becoming - // master is a TRANSITION, not a state, and the resize kick below must fire - // once, on the broadcast that flips it, never on every later broadcast - // that merely reconfirms it. - var prevMaster string - if m.sizeMaster != nil { - prevMaster = m.sizeMaster[dest] - } - if m.sizeMaster == nil { - m.sizeMaster = make(map[string]string) - } - m.sizeMaster[dest] = state.SizeMaster - if m.clientCount == nil { - m.clientCount = make(map[string]int) - } - m.clientCount[dest] = state.Clients // The leading state.SizeMaster != "" guard matters on its own: without it, // an empty m.clientID (never set — every real Model gets one from // NewModel, but a bare Model literal in a test does not) would equal an @@ -6256,14 +6290,14 @@ func (m *Model) rebuildTabs(info ProjectInfo, state WorkspaceStateMsg, existingT tab, exists := existingTabs[tabInfo.ID] if exists && tab.templateLayoutPending && len(tabInfo.Layout) > 0 { // Another client may have already saved the completed tree. - tab = m.restoreTabLayout(tab, tabInfo, paneMap, existingPanes) + tab = m.restoreTabLayout(tab, tabInfo, paneMap, existingPanes, dest) } if !exists { tab = NewTabModel(tabInfo.ID, tabInfo.Name) // New tab that doesn't exist locally — try to restore layout from daemon. if len(tabInfo.Layout) > 0 { - tab = m.restoreTabLayout(tab, tabInfo, paneMap, existingPanes) + tab = m.restoreTabLayout(tab, tabInfo, paneMap, existingPanes, dest) tab.Dest = dest // All non-overlay panes in a restored tab are new. for _, pid := range tabInfo.Panes { @@ -6345,7 +6379,7 @@ func (m *Model) rebuildTabs(info ProjectInfo, state WorkspaceStateMsg, existingT if info, ok := paneMap[paneID]; ok { if leaf := tab.Root.FindLeaf(paneID); leaf != nil { wasPending := leaf.Pane.Pending - syncPaneMeta(leaf.Pane, info, m.pluginWideCanvas(info.Type), m.pluginMinNativeCols(info.Type), m.pluginRestoresViaSession(info.Type)) + syncPaneMeta(leaf.Pane, info, m.pluginWideCanvas(info.Type), m.pluginMinNativeCols(info.Type), m.pluginRestoresViaSession(info.Type), m.isFollower(dest)) // A deferred pane that just lazy-spawned (Pending→running, // e.g. on tab switch): arm the restore indicator NOW so it // covers the real boot, and enroll it for spinner ticks. @@ -6412,7 +6446,7 @@ func (m *Model) rebuildTabs(info ProjectInfo, state WorkspaceStateMsg, existingT newPaneIDs = append(newPaneIDs, paneID) } if info != nil { - syncPaneMeta(pane, info, m.pluginWideCanvas(info.Type), m.pluginMinNativeCols(info.Type), m.pluginRestoresViaSession(info.Type)) + syncPaneMeta(pane, info, m.pluginWideCanvas(info.Type), m.pluginMinNativeCols(info.Type), m.pluginRestoresViaSession(info.Type), m.isFollower(dest)) } // Try to fill a pending split placeholder first. @@ -6574,7 +6608,7 @@ func (m *Model) rebuildTabs(info ProjectInfo, state WorkspaceStateMsg, existingT } // restoreTabLayout rebuilds a tab's layout tree from serialized daemon state. -func (m *Model) restoreTabLayout(tab *TabModel, tabInfo TabInfo, paneMap map[string]*PaneInfo, existingPanes map[string]*PaneModel) *TabModel { +func (m *Model) restoreTabLayout(tab *TabModel, tabInfo TabInfo, paneMap map[string]*PaneInfo, existingPanes map[string]*PaneModel, dest string) *TabModel { tab.templateLayoutApplied, tab.templateLayoutPending = true, false log.Printf("restoreLayout: tab %s %q with %d panes", tab.ID, tabInfo.Name, len(tabInfo.Panes)) tab.Name = tabInfo.Name @@ -6596,7 +6630,7 @@ func (m *Model) restoreTabLayout(tab *TabModel, tabInfo TabInfo, paneMap map[str pane.resumeStart = time.Now() } if info, ok := paneMap[paneID]; ok { - syncPaneMeta(pane, info, m.pluginWideCanvas(info.Type), m.pluginMinNativeCols(info.Type), m.pluginRestoresViaSession(info.Type)) + syncPaneMeta(pane, info, m.pluginWideCanvas(info.Type), m.pluginMinNativeCols(info.Type), m.pluginRestoresViaSession(info.Type), m.isFollower(dest)) } paneModels[paneID] = pane } @@ -6709,7 +6743,7 @@ func (m *Model) reconcileOverlayPane( } newPaneIDs = append(newPaneIDs, overlayInfo.ID) } - syncPaneMeta(pane, overlayInfo, m.pluginWideCanvas(overlayInfo.Type), m.pluginMinNativeCols(overlayInfo.Type), m.pluginRestoresViaSession(overlayInfo.Type)) + syncPaneMeta(pane, overlayInfo, m.pluginWideCanvas(overlayInfo.Type), m.pluginMinNativeCols(overlayInfo.Type), m.pluginRestoresViaSession(overlayInfo.Type), m.isFollower(tab.Dest)) tab.overlayPane = pane // Show the overlay immediately when this TUI's Alt+G triggered its // creation (pendingOverlayShow entry). On plain reattach, default hidden. @@ -6721,7 +6755,7 @@ func (m *Model) reconcileOverlayPane( } default: // Same overlay pane — refresh metadata only. - syncPaneMeta(tab.overlayPane, overlayInfo, m.pluginWideCanvas(overlayInfo.Type), m.pluginMinNativeCols(overlayInfo.Type), m.pluginRestoresViaSession(overlayInfo.Type)) + syncPaneMeta(tab.overlayPane, overlayInfo, m.pluginWideCanvas(overlayInfo.Type), m.pluginMinNativeCols(overlayInfo.Type), m.pluginRestoresViaSession(overlayInfo.Type), m.isFollower(tab.Dest)) } return newPaneIDs, false, nil @@ -6785,6 +6819,50 @@ func (m *Model) resizeTabs() { } } +// applyPaneSizes records a daemon's pane_sizes frame on its panes and resizes +// every follower pane's VT to match, at once (spec §4.1, §5.1). Scoped to +// msg.dest: pane ids are only unique within one daemon. A pane that is not a +// follower still records the size — it is what targetVTSize reads should this +// client become a follower before the next broadcast — but its VT follows its +// own box. ResizeVT is a no-op on an unchanged size and bumps contentGen on a +// changed one, which is what marks the pane dirty for its render cache. +func (m *Model) applyPaneSizes(msg paneSizesMsg) { + // The tab rides along for its canvas: targetVTSize falls back to + // paneVTSize for a size of 0x0, and a wide-canvas pane needs the canvas + // there to come out the same as the resize pass would make it. + type located struct { + pane *PaneModel + tab *TabModel + } + byID := make(map[string]located) + for _, proj := range m.projects { + if proj.Dest != msg.dest { + continue + } + for _, tab := range proj.tabs { + if tab.Root != nil { + for _, p := range tab.Leaves() { + byID[p.ID] = located{p, tab} + } + } + if tab.overlayPane != nil { + byID[tab.overlayPane.ID] = located{tab.overlayPane, tab} + } + } + } + for _, s := range msg.sizes { + at, ok := byID[s.PaneID] + if !ok { + continue + } + p := at.pane + p.daemonCols, p.daemonRows = int(s.Cols), int(s.Rows) + if p.follower { + p.ResizeVT(p.targetVTSize(p.Width, p.Height, p.NativeW, at.tab.CanvasW, at.tab.CanvasH)) + } + } +} + // isActivePane reports whether paneID is the pane the user is currently // focused on (active pane of the active tab). Used by the notification // dispatcher to suppress redundant idle events for the pane the user is @@ -7630,6 +7708,16 @@ func (m Model) listenForMessages() tea.Cmd { log.Printf("ipc recv: set_active_pane %s", payload.PaneID) return setActivePaneMsg{PaneID: payload.PaneID} + case ipc.MsgPaneSizes: + var payload ipc.PaneSizesPayload + if err := msg.DecodePayload(&payload); err != nil { + log.Printf("decode pane_sizes: %v", err) + return listenContinueMsg{} + } + // Origin, like workspace_state's Dest: sizes name one daemon's + // panes, and it is not on the wire. + return paneSizesMsg{dest: msg.Origin, sizes: payload.Panes} + case ipc.MsgCloseTUI: log.Print("ipc recv: close_tui") return tea.QuitMsg{} diff --git a/internal/tui/modelinfo_test.go b/internal/tui/modelinfo_test.go index 158f9d59..2d0d3d12 100644 --- a/internal/tui/modelinfo_test.go +++ b/internal/tui/modelinfo_test.go @@ -36,7 +36,7 @@ func TestSyncPaneMeta_ModelFollowsSnapshot(t *testing.T) { t.Parallel() pane := &PaneModel{ID: "p1"} // A snapshot carrying values applies them. - syncPaneMeta(pane, &PaneInfo{ID: "p1", Model: "claude-sonnet-5", ContextTokens: 42}, false, 0, false) + syncPaneMeta(pane, &PaneInfo{ID: "p1", Model: "claude-sonnet-5", ContextTokens: 42}, false, 0, false, false) if pane.Model != "claude-sonnet-5" || pane.ContextTokens != 42 { t.Fatalf("snapshot values not applied: model=%q tokens=%d", pane.Model, pane.ContextTokens) } @@ -44,7 +44,7 @@ func TestSyncPaneMeta_ModelFollowsSnapshot(t *testing.T) { // daemon-side restart-clear (handleRestartPaneReq zeroes LastModel) // reaches the status bar; keeping the old value would show the // pre-restart model until the next completed turn. - syncPaneMeta(pane, &PaneInfo{ID: "p1", CWD: "/tmp"}, false, 0, false) + syncPaneMeta(pane, &PaneInfo{ID: "p1", CWD: "/tmp"}, false, 0, false, false) if pane.Model != "" || pane.ContextTokens != 0 { t.Fatalf("restart-clear did not propagate: model=%q tokens=%d", pane.Model, pane.ContextTokens) } diff --git a/internal/tui/pane.go b/internal/tui/pane.go index d685ac29..9737c6db 100644 --- a/internal/tui/pane.go +++ b/internal/tui/pane.go @@ -169,6 +169,16 @@ type PaneModel struct { daemonMouseSGR bool daemonBracketedPaste bool + // Multi-client sync (spec §5). follower mirrors Model.isFollower for this + // pane's destination: another client is the size master, so this pane's + // VT takes the size the master chose (daemonCols x daemonRows, the last + // size the daemon accepted) instead of its own box, and View cuts or pads + // that grid into the box. Set by syncPaneMeta from every broadcast; the + // daemon size is also updated between broadcasts by a pane_sizes frame, + // which arrives BEFORE the child's repaint at that size. See targetVTSize. + follower bool + daemonCols, daemonRows int + // Render cache: View() output is reused while renderKey() is unchanged. // contentGen covers VT-grid/raw-buffer mutations (the grid itself has no // public change counter; PaneModel mediates all writes via AppendOutput/ @@ -238,6 +248,7 @@ type paneRenderKey struct { name, cwd string paneType, sessionID string wideCanvas, previewWrap bool + follower bool // previewMode and the cut markers read it historyLines int selActive bool sel Selection @@ -279,6 +290,7 @@ func (p *PaneModel) renderKey() paneRenderKey { sessionID: p.SessionID, wideCanvas: p.WideCanvas, previewWrap: p.previewWrap, + follower: p.follower, historyLines: p.HistoryLines, } // renderContent only honors a selection whose PaneID matches this pane; @@ -628,6 +640,24 @@ func (p *PaneModel) acceptOutputGeneration(generation uint64) bool { return true } +// targetVTSize is the single decision point for the size a pane's VT +// emulator takes. A follower pane (spec §5.1) takes the size the master chose +// for it, whatever box this client draws it in — resizing the VT to its own +// box would rewrap the master's output with no PTY redraw to pair it, the +// unpaired-resize corruption ResizeVT's contract forbids. A follower with no +// daemon size yet (never sized, or pending) falls back to its rect, and sends +// nothing either way. Everyone else uses paneVTSize. +// +// Only EMULATOR sizing goes through here. The resize producers +// (resizeAllPanes, diffResizes) keep calling paneVTSize: a master sends its +// own rect's size, and a follower sends nothing at all. +func (p *PaneModel) targetVTSize(rectW, rectH, nativeW, canvasW, canvasH int) (cols, rows int) { + if p.follower && p.daemonCols > 0 && p.daemonRows > 0 { + return p.daemonCols, p.daemonRows + } + return paneVTSize(p.WideCanvas, p.MinNativeCols, rectW, rectH, nativeW, canvasW, canvasH) +} + func (p *PaneModel) ResizeVT(cols, rows int) { if cols <= 0 || rows <= 0 || (cols == p.vt.Width() && rows == p.vt.Height()) { return @@ -785,6 +815,44 @@ func (p *PaneModel) wheelForwardSeq(up bool, relX, relY int) []byte { return []byte(ansi.MouseX10(b, relX, relY)) } +// followerGridPos maps a box-relative mouse position (0-based, inside the +// border) to the follower's grid (spec §5.3), for forwarding to a tracking +// app. The grid is cropped at the left, so gx is relX; rows follow what the +// renderer shows — bottom-anchored in the preview, so a live too-tall view +// maps box row r to grid row r + (vtH - innerH). ok is false where no grid +// cell is drawn: the padding right of or below a grid smaller than the box, +// and a scrollback row, which the app has no coordinate for. A non-follower +// pane is returned unchanged, exactly as it was forwarded before. +func followerGridPos(p *PaneModel, relX, relY int) (gx, gy int, ok bool) { + if !p.follower { + return relX, relY, true + } + if relX < 0 || relY < 0 { + return 0, 0, false + } + if p.previewMode() { + innerW, innerH := max(1, p.Width-2), max(1, p.Height-2) + l := p.previewLayoutFor(innerW) + total := l.totalVisual() + // renderPreview's own viewStart: the inverse must agree with it. + viewStart := max(0, total-innerH-p.scrollBack) + v := viewStart + relY + if v >= total { + return 0, 0, false + } + absRow, s := l.locate(v) + gx, gy = s.start+relX, absRow-p.vt.ScrollbackLen() + } else { + // Native: drawn top-left, scrollBack counting emulator scrollback + // lines above the screen. + gx, gy = relX, relY-p.scrollBack + } + if gy < 0 || gx >= p.vt.Width() || gy >= p.vt.Height() { + return 0, 0, false + } + return gx, gy, true +} + // ScrollToRelY positions the scrollback so that the scrollbar thumb's TOP // row lands at relY (relative to the content area, 0..innerH-1). Inverse // of the thumb-position formula in renderScrollback — a click at row R @@ -1271,8 +1339,14 @@ func (p *PaneModel) View() string { // The worktree wait reuses the `preparing` label slot rather than adding a // third: "preparing..." is what it is, and the branch is already in the // pane body, where there is room to elide it honestly. - topLine := buildTopBorder(p.Width, p.CWD, rightLabel, borderColor, p.ghost, p.resuming, - p.preparing || p.PreparingWorktree != "", p.focusMode, p.spinnerFrame, p.working, p.workFrame) + // Follower cut markers (spec §5.2): only a follower's grid can exceed its + // box in height, and a wide-canvas pane's width crop is its normal state, + // not a cut someone else's size imposed — so both are follower-only. + cutRows := p.follower && p.vt.Height() > innerH + cutCols := p.follower && p.vt.Width() > innerW + topLine := buildTopBorderCut(p.Width, p.CWD, rightLabel, borderColor, p.ghost, p.resuming, + p.preparing || p.PreparingWorktree != "", p.focusMode, p.spinnerFrame, p.working, p.workFrame, + cutRows, cutCols) out := topLine + "\n" + body p.cachedKey, p.cachedView, p.hasCache = key, out, true @@ -1280,6 +1354,20 @@ func (p *PaneModel) View() string { } func buildTopBorder(width int, cwd, name string, color color.Color, ghost, resuming, preparing, focus bool, spinnerFrame int, working bool, workFrame int) string { + return buildTopBorderCut(width, cwd, name, color, ghost, resuming, preparing, focus, spinnerFrame, working, workFrame, false, false) +} + +// followerCutMark replaces a top-border corner to say the box cut part of a +// follower's grid (spec §5.2). One cell for one cell, so the border keeps its +// exact width, and it never touches the labels between the corners. +const followerCutMark = "…" + +// buildTopBorderCut is buildTopBorder with the follower cut markers: cutRows +// puts followerCutMark in place of the top-left corner (rows above the box +// are hidden — the view is bottom-anchored), cutCols in place of the +// top-right corner, the right border's cell on the title row (columns right +// of the box are hidden — the view is cropped at the left edge). +func buildTopBorderCut(width int, cwd, name string, color color.Color, ghost, resuming, preparing, focus bool, spinnerFrame int, working bool, workFrame int, cutRows, cutCols bool) string { if ghost { if name == "" { name = "restored" @@ -1300,6 +1388,12 @@ func buildTopBorder(width int, cwd, name string, color color.Color, ghost, resum style := lipgloss.NewStyle().Foreground(color) b := lipgloss.RoundedBorder() + if cutRows { + b.TopLeft = followerCutMark + } + if cutCols { + b.TopRight = followerCutMark + } innerW := width - 2 if innerW < 1 { return style.Render(b.TopLeft + b.TopRight) diff --git a/internal/tui/pane_preview.go b/internal/tui/pane_preview.go index 889b6870..efd8bca8 100644 --- a/internal/tui/pane_preview.go +++ b/internal/tui/pane_preview.go @@ -29,13 +29,21 @@ type previewLayout struct { } // previewMode reports whether this pane renders the wrapped preview: a -// wide-canvas pane whose viewport is narrower than its emulator. +// wide-canvas pane whose viewport is narrower than its emulator, or a +// follower pane whose master-sized grid exceeds its box in EITHER dimension +// (spec §5.2). The preview already crops at the left edge and bottom-anchors +// with scrollback above, which is exactly the follower's cut. A follower +// grid that FITS takes the native path, which draws it top-left and lets the +// body style pad the rest of the box. func (p *PaneModel) previewMode() bool { - if !p.WideCanvas { + innerW, innerH := p.Width-2, p.Height-2 + if innerW < 1 { return false } - innerW := p.Width - 2 - return innerW >= 1 && innerW < p.vt.Width() + if p.follower { + return innerW < p.vt.Width() || innerH < p.vt.Height() + } + return p.WideCanvas && innerW < p.vt.Width() } // previewLayoutFor returns the preview layout for innerW, rebuilding only diff --git a/internal/tui/pinned_attention_test.go b/internal/tui/pinned_attention_test.go index ef3fbe4e..8fba4539 100644 --- a/internal/tui/pinned_attention_test.go +++ b/internal/tui/pinned_attention_test.go @@ -358,11 +358,11 @@ func TestParseWorkspaceState_ReadsThePinWireKey(t *testing.T) { func TestSyncPaneMeta_AdoptsTheDaemonsPin(t *testing.T) { t.Parallel() pane := &PaneModel{ID: "p1"} - syncPaneMeta(pane, &PaneInfo{ID: "p1", PinnedAttention: true}, false, 0, false) + syncPaneMeta(pane, &PaneInfo{ID: "p1", PinnedAttention: true}, false, 0, false, false) if !pane.pinnedAttention { t.Fatal("a daemon-reported pin was not adopted") } - syncPaneMeta(pane, &PaneInfo{ID: "p1", PinnedAttention: false}, false, 0, false) + syncPaneMeta(pane, &PaneInfo{ID: "p1", PinnedAttention: false}, false, 0, false, false) if pane.pinnedAttention { t.Error("a daemon-reported CLEAR was not adopted — the copy must be unconditional, " + "or an unmark from another client can never reach this one") diff --git a/internal/tui/restore_indicator_test.go b/internal/tui/restore_indicator_test.go index 2702d475..1318594f 100644 --- a/internal/tui/restore_indicator_test.go +++ b/internal/tui/restore_indicator_test.go @@ -99,11 +99,11 @@ func TestSyncPaneMeta_PropagatesPending(t *testing.T) { p := NewPaneModel("p", testRingBufSize) defer p.Dispose() p.Pending = true - syncPaneMeta(p, &PaneInfo{ID: "p", Pending: false}, false, 0, false) + syncPaneMeta(p, &PaneInfo{ID: "p", Pending: false}, false, 0, false, false) if p.Pending { t.Error("syncPaneMeta should clear Pending when the daemon reports it spawned") } - syncPaneMeta(p, &PaneInfo{ID: "p", Pending: true}, false, 0, false) + syncPaneMeta(p, &PaneInfo{ID: "p", Pending: true}, false, 0, false, false) if !p.Pending { t.Error("syncPaneMeta should set Pending when the daemon reports deferred") } @@ -290,7 +290,7 @@ func TestSyncPaneMeta_PropagatesSessionAndHistory(t *testing.T) { t.Parallel() p := NewPaneModel("p", testRingBufSize) defer p.Dispose() - syncPaneMeta(p, &PaneInfo{ID: "p", Type: "claude-code", SessionID: "abc123", HistoryLines: 42}, false, 0, false) + syncPaneMeta(p, &PaneInfo{ID: "p", Type: "claude-code", SessionID: "abc123", HistoryLines: 42}, false, 0, false, false) if p.SessionID != "abc123" { t.Errorf("SessionID = %q, want abc123", p.SessionID) } diff --git a/internal/tui/spawnerror_test.go b/internal/tui/spawnerror_test.go index 7ad5d82f..837bd52f 100644 --- a/internal/tui/spawnerror_test.go +++ b/internal/tui/spawnerror_test.go @@ -52,7 +52,7 @@ func TestPaneView_SanitizesTheSpawnError(t *testing.T) { // guarded copy would keep showing the old one. func TestSyncPaneMeta_ClearsTheSpawnError(t *testing.T) { pane := &PaneModel{ID: "p1", SpawnError: "worktree is gone: /wt/feat-x"} - syncPaneMeta(pane, &PaneInfo{ID: "p1"}, false, 0, false) + syncPaneMeta(pane, &PaneInfo{ID: "p1"}, false, 0, false, false) if pane.SpawnError != "" { t.Errorf("SpawnError = %q after a clean update, want it cleared", pane.SpawnError) @@ -61,7 +61,7 @@ func TestSyncPaneMeta_ClearsTheSpawnError(t *testing.T) { func TestSyncPaneMeta_CarriesTheSpawnError(t *testing.T) { pane := &PaneModel{ID: "p1"} - syncPaneMeta(pane, &PaneInfo{ID: "p1", SpawnError: "worktree is gone: /wt/feat-x"}, false, 0, false) + syncPaneMeta(pane, &PaneInfo{ID: "p1", SpawnError: "worktree is gone: /wt/feat-x"}, false, 0, false, false) if pane.SpawnError != "worktree is gone: /wt/feat-x" { t.Errorf("SpawnError = %q, want the daemon's message", pane.SpawnError) diff --git a/internal/tui/tab.go b/internal/tui/tab.go index feca8ea6..3ef93c0e 100644 --- a/internal/tui/tab.go +++ b/internal/tui/tab.go @@ -349,13 +349,14 @@ func (t *TabModel) activeIndex(leaves []*PaneModel) int { // sizePaneFull sizes a pane to fill the entire tab area (used by the // full-tab modes: focus and overlay). Wide-canvas panes resolve to the tab // canvas via paneVTSize — for them focus mode is a pure viewport change, -// never a PTY/emulator resize. +// never a PTY/emulator resize. A follower pane keeps the master's size +// (targetVTSize), padded or cut into the full-tab box. func sizePaneFull(t *TabModel, p *PaneModel, w, h int) { nativeW := w + t.ChromeW p.Width = w p.Height = h p.NativeW = nativeW - p.ResizeVT(paneVTSize(p.WideCanvas, p.MinNativeCols, w, h, nativeW, t.CanvasW, t.CanvasH)) + p.ResizeVT(p.targetVTSize(w, h, nativeW, t.CanvasW, t.CanvasH)) } // Resize recomputes dimensions for the entire layout tree. diff --git a/internal/tui/unseen_mark_test.go b/internal/tui/unseen_mark_test.go index 818937dd..2fe9b221 100644 --- a/internal/tui/unseen_mark_test.go +++ b/internal/tui/unseen_mark_test.go @@ -61,22 +61,22 @@ func TestParseWorkspaceState_ReadsTheUnseenWireKey(t *testing.T) { func TestSyncPaneMeta_SeedsUnseenOnceThenLeavesItToTheClient(t *testing.T) { t.Parallel() pane := NewPaneModel("p1", 1024) - syncPaneMeta(pane, &PaneInfo{ID: "p1", Unseen: true}, false, 0, false) + syncPaneMeta(pane, &PaneInfo{ID: "p1", Unseen: true}, false, 0, false, false) if !pane.unseen { t.Fatal("the first sync must seed the mark from the daemon") } - syncPaneMeta(pane, &PaneInfo{ID: "p1", Unseen: false}, false, 0, false) + syncPaneMeta(pane, &PaneInfo{ID: "p1", Unseen: false}, false, 0, false, false) if !pane.unseen { t.Error("a later broadcast must not overwrite the client's live mark") } fresh := NewPaneModel("p2", 1024) - syncPaneMeta(fresh, &PaneInfo{ID: "p2"}, false, 0, false) + syncPaneMeta(fresh, &PaneInfo{ID: "p2"}, false, 0, false, false) if fresh.unseen { t.Error("a pane the daemon holds no mark for must seed unmarked") } fresh.unseen = true - syncPaneMeta(fresh, &PaneInfo{ID: "p2"}, false, 0, false) + syncPaneMeta(fresh, &PaneInfo{ID: "p2"}, false, 0, false, false) if !fresh.unseen { t.Error("a later broadcast must not clear a mark the client set") } @@ -201,7 +201,7 @@ func TestFinishReconnect_RestatesUnseenMarks(t *testing.T) { // still does. func TestRaiseDeferredToasts_SkipsASeededMark(t *testing.T) { m, f, pane := toastModel(t) - syncPaneMeta(pane, &PaneInfo{ID: pane.ID, Unseen: true}, false, 0, false) + syncPaneMeta(pane, &PaneInfo{ID: pane.ID, Unseen: true}, false, 0, false, false) if !pane.unseen { t.Fatal("setup: the seed must mark the pane") } diff --git a/internal/tui/wheel_forward_test.go b/internal/tui/wheel_forward_test.go index 1d05a80f..aff2565d 100644 --- a/internal/tui/wheel_forward_test.go +++ b/internal/tui/wheel_forward_test.go @@ -117,7 +117,7 @@ func TestUpdate_MouseWheel_ForwardsViaDaemonFlagOnReattach(t *testing.T) { pane.Type = "opencode" // Simulate the daemon snapshot reconciliation: no local emulator modes were // ever observed (reattach), only the daemon flags. - syncPaneMeta(pane, &PaneInfo{ID: "oc2", Type: "opencode", MouseTracking: true, MouseSGR: true}, false, 0, false) + syncPaneMeta(pane, &PaneInfo{ID: "oc2", Type: "opencode", MouseTracking: true, MouseSGR: true}, false, 0, false, false) if !pane.MouseTracking() { t.Fatal("MouseTracking() = false with daemon flag set, want true") } diff --git a/internal/tui/workstate.go b/internal/tui/workstate.go index afdbfdda..9e03cf21 100644 --- a/internal/tui/workstate.go +++ b/internal/tui/workstate.go @@ -878,7 +878,11 @@ func (m Model) workSpinnerTick() tea.Cmd { // suppresses the visible notification card (see emitEvent) — so the normal // completion edge keeps `working` accurate across the whole mute/unmute // window instead of going stale the instant the pane is muted. -func syncPaneMeta(pane *PaneModel, info *PaneInfo, wideCanvas bool, minNativeCols int, restoresViaSession bool) { +// +// follower is Model.isFollower for the pane's destination, passed in for the +// same reason wideCanvas is: this is a free function with no Model. It and +// the daemon's size for the pane decide the pane's VT size (targetVTSize). +func syncPaneMeta(pane *PaneModel, info *PaneInfo, wideCanvas bool, minNativeCols int, restoresViaSession bool, follower bool) { pane.Name = info.Name pane.CWD = info.CWD pane.Type = info.Type @@ -924,6 +928,10 @@ func syncPaneMeta(pane *PaneModel, info *PaneInfo, wideCanvas bool, minNativeCol pane.daemonMouseTracking = info.MouseTracking pane.daemonMouseSGR = info.MouseSGR pane.daemonBracketedPaste = info.BracketedPaste + // Unconditional, like the rest: the last size the daemon ACCEPTED is the + // size a follower's VT takes, and 0x0 (never sized) must fall back. + pane.follower = follower + pane.daemonCols, pane.daemonRows = int(info.Cols), int(info.Rows) // Unconditional copy, like the other daemon-authoritative fields: the // daemon writes LastModel BEFORE broadcasting the hook event and IPC // delivery is ordered per connection, so a snapshot can never lag behind diff --git a/internal/tui/workstate_test.go b/internal/tui/workstate_test.go index 477e99c8..1ab47bd5 100644 --- a/internal/tui/workstate_test.go +++ b/internal/tui/workstate_test.go @@ -1431,11 +1431,11 @@ func TestSyncPaneMeta_SetsWideCanvas(t *testing.T) { // the caller) so every reconciliation path re-evaluates it — a plugin // migration mid-session must be able to flip it in both directions. pane := NewPaneModel("p", 1024) - syncPaneMeta(pane, &PaneInfo{Type: "claude-code"}, true, 0, false) + syncPaneMeta(pane, &PaneInfo{Type: "claude-code"}, true, 0, false, false) if !pane.WideCanvas { t.Error("syncPaneMeta must set WideCanvas from the passed flag (true)") } - syncPaneMeta(pane, &PaneInfo{Type: "claude-code"}, false, 0, false) + syncPaneMeta(pane, &PaneInfo{Type: "claude-code"}, false, 0, false, false) if pane.WideCanvas { t.Error("syncPaneMeta must clear WideCanvas when the flag flips to false") } @@ -1450,14 +1450,14 @@ func TestSyncPaneMeta_MuteDoesNotDisturbWorking(t *testing.T) { // the spinner would never reappear after unmuting a still-working pane. pane := NewPaneModel("p1", 1024) pane.working = true - syncPaneMeta(pane, &PaneInfo{Muted: true}, false, 0, false) + syncPaneMeta(pane, &PaneInfo{Muted: true}, false, 0, false, false) if !pane.working { t.Error("a mute metadata sync must not clear working") } pane2 := NewPaneModel("p2", 1024) pane2.working = true - syncPaneMeta(pane2, &PaneInfo{Muted: false}, false, 0, false) + syncPaneMeta(pane2, &PaneInfo{Muted: false}, false, 0, false, false) if !pane2.working { t.Error("a non-mute metadata sync must not clear working") } diff --git a/internal/tui/worktree_preparing_test.go b/internal/tui/worktree_preparing_test.go index 121707a9..a479a32e 100644 --- a/internal/tui/worktree_preparing_test.go +++ b/internal/tui/worktree_preparing_test.go @@ -78,12 +78,12 @@ func TestPaneView_PreparingNeverOutgrowsThePane(t *testing.T) { // carries no branch, and a guarded copy would leave a finished checkout spinning. func TestSyncPaneMeta_CarriesAndClearsPreparingWorktree(t *testing.T) { pane := &PaneModel{ID: "p1"} - syncPaneMeta(pane, &PaneInfo{ID: "p1", PreparingWorktree: "feat/x"}, false, 0, false) + syncPaneMeta(pane, &PaneInfo{ID: "p1", PreparingWorktree: "feat/x"}, false, 0, false, false) if pane.PreparingWorktree != "feat/x" { t.Fatalf("PreparingWorktree = %q, want the daemon's branch", pane.PreparingWorktree) } - syncPaneMeta(pane, &PaneInfo{ID: "p1"}, false, 0, false) + syncPaneMeta(pane, &PaneInfo{ID: "p1"}, false, 0, false, false) if pane.PreparingWorktree != "" { t.Errorf("PreparingWorktree = %q after a clean update, want it cleared", pane.PreparingWorktree) } From ab2086ebdd6d1229870f66a10bf9d067d4d080b6 Mon Sep 17 00:00:00 2001 From: Artjoms Stukans Date: Fri, 25 Sep 2026 21:55:48 +0200 Subject: [PATCH 14/40] test(tui): cover both follower cut markers at once A follower grid cut in both dimensions must carry the marker in both top-border corners, exactly two of them, with the border still the pane's exact width and the right columns still cropped. --- internal/tui/follower_render_test.go | 26 ++++++++++++++++++++++++++ 1 file changed, 26 insertions(+) diff --git a/internal/tui/follower_render_test.go b/internal/tui/follower_render_test.go index ecf7c59b..1f1bef52 100644 --- a/internal/tui/follower_render_test.go +++ b/internal/tui/follower_render_test.go @@ -247,6 +247,32 @@ func TestFollower_RenderTooTallShowsBottomRowsWithMarker(t *testing.T) { assertExactBox(t, p, view) } +// Both cuts at once: both corners carry the marker, the labels between them +// are untouched, and the border keeps the pane's exact width. +func TestFollower_RenderTooWideAndTallShowsBothMarkers(t *testing.T) { + var innerW, innerH int + m, _ := followerFixture(t, "other", []string{"pane-1"}, func(_ string, w, h int) (int, int) { + innerW, innerH = w, h + return w + 30, h + 10 + }, nil) + rows := numberedRows(innerH + 10) + m = feedOutput(t, m, "pane-1", rows+strings.Repeat("a", innerW)+"BBBB") + p := paneByID(t, m, "pane-1") + view := p.View() + lines := strings.Split(view, "\n") + top := []rune(stripANSI(lines[0])) + if top[0] != '…' || top[len(top)-1] != '…' { + t.Errorf("top border %q: want … at BOTH ends for a width and a height cut", string(top)) + } + if strings.Count(string(top), "…") != 2 { + t.Errorf("top border %q: want exactly the two corner markers", string(top)) + } + if got := stripANSI(lines[innerH]); strings.Contains(got, "B") { + t.Errorf("last visible row %q: want the right columns cut", got) + } + assertExactBox(t, p, view) +} + func TestFollower_RenderSmallerIsPadded(t *testing.T) { m, _ := followerFixture(t, "other", []string{"pane-1"}, fixedSize(20, 5), nil) m = feedOutput(t, m, "pane-1", numberedRows(5)) From 2b58949687843457ee1d7c26d6e08b6a4ed975cb Mon Sep 17 00:00:00 2001 From: Artjoms Stukans Date: Fri, 25 Sep 2026 22:17:28 +0200 Subject: [PATCH 15/40] fix: number pane sizes so a stale broadcast cannot undo them applyResizes announces a batch in pane_sizes before any PTY resize and records Cols/Rows only after each one. A workspace broadcast built in that window carries the old sizes but reaches a follower after the frame, so the follower resized its VT back to the old size with no PTY redraw to pair it, and kept it until an unrelated broadcast. The daemon now numbers every announced size per pane (sizeSeq, taken before the frame leaves and carried on each pane_sizes entry as size_seq) and records which number Cols/Rows hold (colsSeq), which the broadcast reports as size_seq. The two are separate because they are written at different moments: a broadcast built mid-batch carries the old size with the old number, never the new number. A failed resize's rollback takes a new number; a spawn size from newPaneSession takes one too. The TUI adopts a daemon size only when its number is not lower than the one it holds, and forgets the number on reattach, since a restarted daemon counts from 1 again. Tests: the frame carries the number and a mid-batch broadcast is older; the rollback is numbered above the size it undoes; a stale broadcast and a stale pane_sizes leave a follower's VT alone while newer ones still apply; a reattach accepts a lower number. Also: the row-width test's scrolled case now has scrollback and a too-wide but short grid, the sidebar steps run through their real keys, a dead assignment is gone and the applyPaneSizes comment says what happens. --- internal/daemon/daemon.go | 36 +++++- internal/daemon/pane_initial_size.go | 5 + internal/daemon/resize_authority_test.go | 67 ++++++++++ internal/daemon/session.go | 17 ++- internal/ipc/protocol.go | 6 + internal/tui/follower_render_test.go | 151 +++++++++++++++++++---- internal/tui/model.go | 27 +++- internal/tui/pane.go | 18 +++ internal/tui/reconnect.go | 3 + internal/tui/workstate.go | 9 +- 10 files changed, 304 insertions(+), 35 deletions(-) diff --git a/internal/daemon/daemon.go b/internal/daemon/daemon.go index 4cfbaa6f..adc68797 100644 --- a/internal/daemon/daemon.go +++ b/internal/daemon/daemon.go @@ -3731,14 +3731,22 @@ func (d *Daemon) applyResizes(conn *ipc.Conn, items []ipc.ResizePanePayload) { if len(work) == 0 { return } + // Each pane's announcement is numbered BEFORE the frame leaves, so the + // frame carries it and the Cols/Rows recorded after the resize can be + // stamped with the same number (see Pane.sizeSeq). sizes := make([]ipc.ResizePanePayload, len(work)) + seqs := make([]uint64, len(work)) for i, w := range work { - sizes[i] = ipc.ResizePanePayload{PaneID: w.pane.ID, Cols: w.cols, Rows: w.rows} + w.pane.PluginMu.Lock() + w.pane.sizeSeq++ + seqs[i] = w.pane.sizeSeq + w.pane.PluginMu.Unlock() + sizes[i] = ipc.ResizePanePayload{PaneID: w.pane.ID, Cols: w.cols, Rows: w.rows, SizeSeq: seqs[i]} } d.sendPaneSizes(conn, sizes) var failed []ipc.ResizePanePayload - for _, w := range work { + for i, w := range work { if err := w.pty.Resize(w.rows, w.cols); err != nil { // Record nothing on failure: a transient Resize error must not make // the guard believe this size was applied, or the TUI's next @@ -3749,8 +3757,15 @@ func (d *Daemon) applyResizes(conn *ipc.Conn, items []ipc.ResizePanePayload) { // The followers were already told the new size, and the child is // still at the old one: tell them the old one again. A pane that // never had a size applied has nothing to go back to. + // The rollback is a newer announcement than the one it undoes, + // so it takes a new number; Cols/Rows and colsSeq are left + // alone, so a broadcast still carrying them is older than both. if w.prevC > 0 && w.prevR > 0 { - failed = append(failed, ipc.ResizePanePayload{PaneID: w.pane.ID, Cols: uint16(w.prevC), Rows: uint16(w.prevR)}) + w.pane.PluginMu.Lock() + w.pane.sizeSeq++ + seq := w.pane.sizeSeq + w.pane.PluginMu.Unlock() + failed = append(failed, ipc.ResizePanePayload{PaneID: w.pane.ID, Cols: uint16(w.prevC), Rows: uint16(w.prevR), SizeSeq: seq}) } continue } @@ -3762,6 +3777,13 @@ func (d *Daemon) applyResizes(conn *ipc.Conn, items []ipc.ResizePanePayload) { w.pane.PluginMu.Lock() w.pane.appliedCols, w.pane.appliedRows = int(w.cols), int(w.rows) w.pane.Cols, w.pane.Rows = int(w.cols), int(w.rows) + // Never backwards. Should another batch's newer announcement already + // be recorded, this resize still ran LAST, so Cols/Rows above are the + // PTY's real size — and keeping the newer number is what lets a + // broadcast carry that truth past the newer, now-wrong frame. + if seqs[i] > w.pane.colsSeq { + w.pane.colsSeq = seqs[i] + } w.pane.PluginMu.Unlock() d.repaintAfterResize(w.pane, w.typ) @@ -4761,7 +4783,15 @@ func (d *Daemon) workspaceStateFromSnapshot(activeTab string, tabs []*Tab, panes // goroutine while handleResizePane writes them from a conn dispatch // goroutine. snapCols, snapRows := pane.Cols, pane.Rows + // In the same span as Cols/Rows: the number must describe exactly + // the size read beside it (see Pane.sizeSeq). + snapSizeSeq := pane.colsSeq pane.PluginMu.Unlock() + // Broadcast-only, runtime: the counter restarts with the daemon, + // so a persisted one would mean nothing. + if includeOverlays && snapSizeSeq > 0 { + paneData["size_seq"] = snapSizeSeq + } // Pending (deferred, not yet lazy-spawned) is spawnMu-guarded — // read it the same way list_panes does. The TUI uses it to show the // restore indicator on deferred panes and to re-arm the indicator diff --git a/internal/daemon/pane_initial_size.go b/internal/daemon/pane_initial_size.go index c87b3ba7..97bee98e 100644 --- a/internal/daemon/pane_initial_size.go +++ b/internal/daemon/pane_initial_size.go @@ -25,6 +25,11 @@ func (d *Daemon) newPaneSession(pane *Pane) apty.Session { } pane.PluginMu.Lock() pane.Cols, pane.Rows = cols, rows + // A new announcement (see Pane.sizeSeq): the spawn size reaches followers + // only through the broadcast, and a restart reuses the Pane, so it must + // outrank every size announced for the previous child. + pane.sizeSeq++ + pane.colsSeq = pane.sizeSeq pane.PluginMu.Unlock() return newSessionFn(cols, rows) } diff --git a/internal/daemon/resize_authority_test.go b/internal/daemon/resize_authority_test.go index aee41c9f..5813626b 100644 --- a/internal/daemon/resize_authority_test.go +++ b/internal/daemon/resize_authority_test.go @@ -415,6 +415,73 @@ func TestPaneSizes_FailedResizeSendsPreviousSize(t *testing.T) { if c, r := appliedSize(pane); c != 120 || r != 40 { t.Errorf("applied = %dx%d after a failed resize, want 120x40 unchanged", c, r) } + // The rollback is a newer announcement than the size it undoes. + e1, e2 := paneSizesOf(t, first[len(first)-1]), paneSizesOf(t, second[len(second)-1]) + if e2[0].SizeSeq <= e1[0].SizeSeq { + t.Errorf("rollback size_seq %d, want above the failed announcement's %d", e2[0].SizeSeq, e1[0].SizeSeq) + } +} + +// broadcastSizeSeq reads pane id's size_seq out of a broadcast state map; 0 +// when absent. No t here: it also runs inside a PTY hook, off the test +// goroutine, where t.Fatal is not allowed. +func broadcastSizeSeq(state map[string]any, id string) uint64 { + panes, _ := state["panes"].([]map[string]any) + for _, p := range panes { + if p["id"] == id { + n, _ := p["size_seq"].(uint64) + return n + } + } + return 0 +} + +// Each announcement is numbered before the pane_sizes frame leaves, and the +// frame carries the number. A broadcast built INSIDE the batch window — after +// the frame, before Cols/Rows are recorded — still reports the old size, so it +// must carry an OLDER number than the frame, or a follower would adopt the old +// size over the new one with no PTY redraw to pair it. Once recorded, the +// broadcast carries the frame's own number. +func TestPaneSizes_FrameCarriesSizeSeqAndMidBatchBroadcastIsOlder(t *testing.T) { + d, sock, tabID := resizeAuthorityDaemon(t) + a, b := attachAB(t, d, sock) + readUntil(t, b, "B's attach state", isType(ipc.MsgWorkspaceState)) + + panes, probes := addProbePanes(t, d, tabID, "terminal", 1) + pane := panes[0] + var midSeq, midBroadcast uint64 + var midCols int + probes[0].mu.Lock() + probes[0].onResize = func() { + pane.PluginMu.Lock() + midSeq, midCols = pane.sizeSeq, pane.Cols + pane.PluginMu.Unlock() + midBroadcast = broadcastSizeSeq(d.buildWorkspaceState(), pane.ID) + } + probes[0].mu.Unlock() + + sendClientMsg(t, a, ipc.MsgResizePane, ipc.ResizePanePayload{PaneID: pane.ID, Cols: 150, Rows: 40}) + got := readUntil(t, b, "the size frame", isType(ipc.MsgPaneSizes)) + e := paneSizesOf(t, got[len(got)-1]) + if len(e) != 1 || e[0].SizeSeq == 0 { + t.Fatalf("pane_sizes = %+v, want one numbered entry", e) + } + waitUntil(t, "the resize applied", func() bool { + c, r := appliedSize(pane) + return c == 150 && r == 40 + }) + if midSeq != e[0].SizeSeq { + t.Errorf("sizeSeq during the PTY resize = %d, want the frame's %d: the number must be taken before the frame leaves", midSeq, e[0].SizeSeq) + } + if midCols == 150 { + t.Fatal("setup: Cols was already recorded inside Resize, so no window was probed") + } + if midBroadcast >= e[0].SizeSeq { + t.Errorf("a broadcast built mid-batch carries size_seq %d with the OLD size; want below the frame's %d", midBroadcast, e[0].SizeSeq) + } + if after := broadcastSizeSeq(d.buildWorkspaceState(), pane.ID); after != e[0].SizeSeq { + t.Errorf("broadcast after the resize carries size_seq %d, want the frame's %d", after, e[0].SizeSeq) + } } // The broadcast carries who the master is and how many clients are attached; diff --git a/internal/daemon/session.go b/internal/daemon/session.go index b604a171..4e794dee 100644 --- a/internal/daemon/session.go +++ b/internal/daemon/session.go @@ -243,8 +243,21 @@ type Pane struct { // workspace broadcast; this guard turns the duplicates into no-ops. // Zeroed when a new PTY is installed (spawnPane) so a fresh PTY // always accepts its first resize. Guarded by PluginMu. - appliedCols int - appliedRows int + appliedCols int + appliedRows int + // sizeSeq numbers every size this pane's followers are TOLD, and colsSeq + // is the announcement Cols/Rows currently record. Two counters because + // the two are written at different moments: applyResizes announces a + // batch (pane_sizes) BEFORE the PTY resize and records Cols/Rows only + // AFTER it, so a broadcast built in between still carries the old + // Cols/Rows — stamped with the old colsSeq, which the TUI then rejects as + // older than the pane_sizes it already applied. Stamping the broadcast + // with sizeSeq instead would pass the old size off as the new one. + // Monotonic for the Pane's life (a restart keeps the struct); a daemon + // restart restarts both, which is why the TUI forgets them on reattach. + // Guarded by PluginMu. + sizeSeq uint64 + colsSeq uint64 LastOutputAt time.Time // Updated on every flushPaneOutput IdleNotified bool // Prevents re-firing for same idle period LastIdleEventAt time.Time // Cooldown: last time a idle event was emitted diff --git a/internal/ipc/protocol.go b/internal/ipc/protocol.go index 129c09a8..cb68015d 100644 --- a/internal/ipc/protocol.go +++ b/internal/ipc/protocol.go @@ -472,6 +472,12 @@ type ResizePanePayload struct { PaneID string `json:"pane_id"` Rows uint16 `json:"rows"` Cols uint16 `json:"cols"` + // SizeSeq numbers a size the daemon announces in pane_sizes (daemon → + // follower only; a client → daemon resize leaves it zero). The broadcast's + // per-pane size_seq is the same counter, so a follower can tell a stale + // broadcast from a newer frame. omitempty keeps the client → daemon wire + // byte-identical, and an older client ignores the field. + SizeSeq uint64 `json:"size_seq,omitempty"` } // ResizePanesPayload batches a whole resize pass — a window resize or a diff --git a/internal/tui/follower_render_test.go b/internal/tui/follower_render_test.go index 1f1bef52..6af32b03 100644 --- a/internal/tui/follower_render_test.go +++ b/internal/tui/follower_render_test.go @@ -9,6 +9,7 @@ import ( tea "charm.land/bubbletea/v2" "charm.land/lipgloss/v2" + "github.com/artyomsv/quil/internal/config" "github.com/artyomsv/quil/internal/ipc" ) @@ -32,6 +33,9 @@ func followerFixture(t *testing.T, master string, paneIDs []string, m, conn := tinyTermModel(t) m.SetClientID("me") m.notifications = NewNotificationCenter(30, 50) // the mouse and View paths read it + // The shipped keymap, so a step can be driven through its real key. + km, _ := buildKeymap(config.Default().Keybindings) + m.keymap = km // Receive() must not park the listen command each broadcast re-arms. close(conn.recv) @@ -39,15 +43,7 @@ func followerFixture(t *testing.T, master string, paneIDs []string, runCmd(cmd) m = next.(Model) - st := WorkspaceStateMsg{ - Dest: "", Clients: 2, - ActiveProject: "proj-1", ActiveTab: "tab-1", - Projects: []ProjectInfo{{ID: "proj-1", Name: "Default", TabIDs: []string{"tab-1"}}}, - Tabs: []TabInfo{{ID: "tab-1", Name: "Shell", ProjectID: "proj-1", Panes: paneIDs}}, - } - for _, id := range paneIDs { - st.Panes = append(st.Panes, PaneInfo{ID: id, TabID: "tab-1", Type: "terminal"}) - } + st := fixtureState("", paneIDs) next, cmd = m.Update(st) runCmd(cmd) m = next.(Model) @@ -68,6 +64,29 @@ func followerFixture(t *testing.T, master string, paneIDs []string, return m, conn } +// fixtureState is the fixture's one-tab broadcast, naming master (or no +// master) and carrying no pane sizes. +func fixtureState(master string, paneIDs []string) WorkspaceStateMsg { + st := WorkspaceStateMsg{ + Dest: "", SizeMaster: master, Clients: 2, + ActiveProject: "proj-1", ActiveTab: "tab-1", + Projects: []ProjectInfo{{ID: "proj-1", Name: "Default", TabIDs: []string{"tab-1"}}}, + Tabs: []TabInfo{{ID: "tab-1", Name: "Shell", ProjectID: "proj-1", Panes: paneIDs}}, + } + for _, id := range paneIDs { + st.Panes = append(st.Panes, PaneInfo{ID: id, TabID: "tab-1", Type: "terminal"}) + } + return st +} + +// sizedBroadcast is a follower broadcast reporting pane-1 at cols x rows, +// numbered seq. +func sizedBroadcast(cols, rows int, seq uint64) WorkspaceStateMsg { + st := fixtureState("other", []string{"pane-1"}) + st.Panes[0].Cols, st.Panes[0].Rows, st.Panes[0].SizeSeq = uint16(cols), uint16(rows), seq + return st +} + // paneByID resolves a tree pane or an overlay pane of the fixture's model. func paneByID(t *testing.T, m Model, id string) *PaneModel { t.Helper() @@ -196,6 +215,71 @@ func TestFollower_PaneSizesMsgResizesBeforeOutput(t *testing.T) { } } +// A broadcast the daemon built while a resize batch was in flight carries the +// size from BEFORE the batch but reaches the follower AFTER the batch's +// pane_sizes frame. Adopting it would resize the VT back with no PTY redraw to +// pair it. Its lower size_seq is what marks it stale. +func TestFollower_StaleBroadcastCannotUndoPaneSizes(t *testing.T) { + m, _ := followerFixture(t, "other", []string{"pane-1"}, fixedSize(180, 45), func(st *WorkspaceStateMsg) { + st.Panes[0].SizeSeq = 4 + }) + pane := func() *PaneModel { return paneByID(t, m, "pane-1") } + if c, r := vtSize(pane()); c != 180 || r != 45 { + t.Fatalf("setup: VT = %dx%d, want 180x45", c, r) + } + + next, _ := m.Update(paneSizesMsg{dest: "", sizes: []ipc.ResizePanePayload{{PaneID: "pane-1", Cols: 200, Rows: 50, SizeSeq: 5}}}) + m = next.(Model) + if c, r := vtSize(pane()); c != 200 || r != 50 { + t.Fatalf("after pane_sizes seq 5: VT = %dx%d, want 200x50", c, r) + } + + next, cmd := m.Update(sizedBroadcast(180, 45, 4)) + runCmd(cmd) + m = next.(Model) + if c, r := vtSize(pane()); c != 200 || r != 50 { + t.Fatalf("a stale broadcast (seq 4) resized the VT to %dx%d; want it kept at 200x50", c, r) + } + + next, cmd = m.Update(sizedBroadcast(200, 50, 5)) + runCmd(cmd) + m = next.(Model) + if c, r := vtSize(pane()); c != 200 || r != 50 { + t.Fatalf("the broadcast recording seq 5 moved the VT to %dx%d; want 200x50", c, r) + } + + // A genuinely newer broadcast is still adopted: the counter guards order, + // it does not freeze the size. + next, cmd = m.Update(sizedBroadcast(190, 48, 6)) + runCmd(cmd) + m = next.(Model) + if c, r := vtSize(pane()); c != 190 || r != 48 { + t.Fatalf("newer broadcast (seq 6): VT = %dx%d, want 190x48", c, r) + } + + // And a stale pane_sizes entry is ignored the same way. + next, _ = m.Update(paneSizesMsg{dest: "", sizes: []ipc.ResizePanePayload{{PaneID: "pane-1", Cols: 170, Rows: 40, SizeSeq: 5}}}) + m = next.(Model) + if c, r := vtSize(pane()); c != 190 || r != 48 { + t.Fatalf("a stale pane_sizes (seq 5) resized the VT to %dx%d; want 190x48", c, r) + } +} + +// A reattach may be to a RESTARTED daemon, whose counter starts again from 1: +// the reset must let the lower number through. +func TestFollower_ReattachAcceptsLowerSizeSeq(t *testing.T) { + m, _ := followerFixture(t, "other", []string{"pane-1"}, fixedSize(200, 50), func(st *WorkspaceStateMsg) { + st.Panes[0].SizeSeq = 9 + }) + m.armReattachReset("") + next, cmd := m.Update(sizedBroadcast(170, 40, 1)) + runCmd(cmd) + m = next.(Model) + if c, r := vtSize(paneByID(t, m, "pane-1")); c != 170 || r != 40 { + t.Fatalf("after reattach, VT = %dx%d; want the restarted daemon's 170x40 (seq 1)", c, r) + } +} + func TestFollower_RenderTooWideCropsLeftWithMarker(t *testing.T) { var innerW int m, _ := followerFixture(t, "other", []string{"pane-1"}, func(_ string, w, h int) (int, int) { @@ -303,23 +387,34 @@ func TestFollower_RenderEveryRowExactWidth(t *testing.T) { "tall": func(_ string, w, h int) (int, int) { return w, h + 7 }, "both": func(_ string, w, h int) (int, int) { return w + 40, h + 7 }, "smaller": func(_ string, w, h int) (int, int) { return w / 2, h / 2 }, + // The preview path (too wide) with a grid SHORTER than the box: the + // preview bottom-anchors, so its padding comes from the short grid. + "wide-short": func(_ string, w, h int) (int, int) { return w + 40, h / 2 }, } for name, fn := range cases { t.Run(name, func(t *testing.T) { m, _ := followerFixture(t, "other", []string{"pane-1"}, fn, nil) p := paneByID(t, m, "pane-1") c, r := vtSize(p) + // r+Height full-width rows: the extra ones scroll off into scrollback, + // more than a box of it, so the scrolled-back view has history to show. var b strings.Builder - for i := 0; i < r; i++ { + for i := 0; i < r+p.Height; i++ { if i > 0 { b.WriteString("\r\n") } b.WriteString(strings.Repeat(string(rune('a'+i%26)), c)) } m = feedOutput(t, m, "pane-1", b.String()) + if p.vt.ScrollbackLen() == 0 { + t.Fatal("setup: no scrollback") + } assertExactBox(t, p, p.View()) // Scrolled back too: the scrollbar column must not widen a row. p.ScrollUp(2) + if p.scrollBack == 0 { + t.Fatal("setup: ScrollUp did not scroll") + } assertExactBox(t, p, p.View()) }) } @@ -374,9 +469,7 @@ func trackingMutate(st *WorkspaceStateMsg) { // Spec §5.3: box row r of a too-tall follower is grid row r + (vtH - innerH). func TestFollower_WheelForwardTranslatedToGridRow(t *testing.T) { - var innerH int m, _ := followerFixture(t, "other", []string{"pane-1"}, func(_ string, w, h int) (int, int) { - innerH = h return w, h + 10 }, trackingMutate) m.inputCh = make(chan paneInput, inputForwardBuffer) @@ -388,7 +481,6 @@ func TestFollower_WheelForwardTranslatedToGridRow(t *testing.T) { if got := wheelForwarded(m); got != want { t.Fatalf("forwarded %q, want %q (box row %d + %d cut rows)", got, want, relY, 10) } - _ = innerH } // Spec §5.3: a position in the padding of a grid smaller than the box sends @@ -526,15 +618,32 @@ func TestFollower_LocalRectChangesNeverResizeTheVT(t *testing.T) { m.View() check("notes off") - m.notifications.visible = true - m.View() - check("notification sidebar") - m.notifications.visible = false + // Both sidebars through their real keys (notification.toggle, + // sidebar.toggle), so the whole dispatch path is under test. + press := func(chord string) { + t.Helper() + key, ok := keyPressForChord(chord) + if !ok { + t.Fatalf("cannot build a key press for %q", chord) + } + next, cmd := m.Update(key) + runCmdNoWait(cmd) + m = next.(Model) + m.View() + } + press(m.keymap.Keys("notification.toggle")[0]) + if !m.notifications.visible { + t.Fatal("setup: the notification sidebar did not open") + } + check("notification sidebar on") + press(m.keymap.Keys("notification.toggle")[0]) + check("notification sidebar off") - next, cmd = m.toggleProjectSidebar() - runCmd(cmd) - m = next.(Model) - m.View() + wasOpen := m.sidebarOpen + press(m.keymap.Keys("sidebar.toggle")[0]) + if m.sidebarOpen == wasOpen { + t.Fatal("setup: the project sidebar did not toggle") + } check("project sidebar") next, _ = m.Update(tea.WindowSizeMsg{Width: 150, Height: 50}) diff --git a/internal/tui/model.go b/internal/tui/model.go index 9fc78df4..4fa33534 100644 --- a/internal/tui/model.go +++ b/internal/tui/model.go @@ -195,6 +195,12 @@ type PaneInfo struct { // Zero means the daemon has never applied a size — always send then. Cols uint16 Rows uint16 + // SizeSeq numbers Cols/Rows among every size the daemon announced for + // this pane (daemon Pane.colsSeq). A pane_sizes frame carries the same + // counter, so a broadcast built before a resize was recorded — older + // Cols/Rows, lower SizeSeq — cannot undo the frame a follower already + // applied. 0 from a daemon that predates it: always adopted. + SizeSeq uint64 } // paneSettleRepaintMsg fires shortly after a pane's first live output and @@ -6821,11 +6827,15 @@ func (m *Model) resizeTabs() { // applyPaneSizes records a daemon's pane_sizes frame on its panes and resizes // every follower pane's VT to match, at once (spec §4.1, §5.1). Scoped to -// msg.dest: pane ids are only unique within one daemon. A pane that is not a -// follower still records the size — it is what targetVTSize reads should this -// client become a follower before the next broadcast — but its VT follows its -// own box. ResizeVT is a no-op on an unchanged size and bumps contentGen on a -// changed one, which is what marks the pane dirty for its render cache. +// msg.dest: pane ids are only unique within one daemon. An entry older than +// the size a pane already holds is ignored (adoptDaemonSize). A pane that is +// not a follower records the size too, which keeps its sequence number +// current, but its VT keeps following its own box: the daemon sends this +// frame only to non-masters, so a non-follower receiving it is a client with +// no master to follow (or one the next broadcast is about to demote, and +// that broadcast resizes it). ResizeVT is a no-op on an unchanged size and +// bumps contentGen on a changed one, which is what marks the pane dirty for +// its render cache. func (m *Model) applyPaneSizes(msg paneSizesMsg) { // The tab rides along for its canvas: targetVTSize falls back to // paneVTSize for a size of 0x0, and a wide-canvas pane needs the canvas @@ -6856,7 +6866,9 @@ func (m *Model) applyPaneSizes(msg paneSizesMsg) { continue } p := at.pane - p.daemonCols, p.daemonRows = int(s.Cols), int(s.Rows) + if !p.adoptDaemonSize(int(s.Cols), int(s.Rows), s.SizeSeq) { + continue + } if p.follower { p.ResizeVT(p.targetVTSize(p.Width, p.Height, p.NativeW, at.tab.CanvasW, at.tab.CanvasH)) } @@ -8128,6 +8140,9 @@ func parseWorkspaceState(raw map[string]any) WorkspaceStateMsg { if n, ok := pm["rows"].(float64); ok && n >= 0 && n <= math.MaxUint16 { pi.Rows = uint16(n) } + if n, ok := pm["size_seq"].(float64); ok && n >= 0 { + pi.SizeSeq = uint64(n) + } state.Panes = append(state.Panes, pi) } } diff --git a/internal/tui/pane.go b/internal/tui/pane.go index 9737c6db..d21a79eb 100644 --- a/internal/tui/pane.go +++ b/internal/tui/pane.go @@ -178,6 +178,12 @@ type PaneModel struct { // which arrives BEFORE the child's repaint at that size. See targetVTSize. follower bool daemonCols, daemonRows int + // daemonSizeSeq is the daemon's number for daemonCols/daemonRows (see + // PaneInfo.SizeSeq). A size is adopted only when its number is not LOWER, + // so a stale broadcast cannot resize the VT back behind a newer + // pane_sizes frame with no PTY redraw to pair it. Zeroed on reattach, + // because a restarted daemon restarts the counter. + daemonSizeSeq uint64 // Render cache: View() output is reused while renderKey() is unchanged. // contentGen covers VT-grid/raw-buffer mutations (the grid itself has no @@ -651,6 +657,18 @@ func (p *PaneModel) acceptOutputGeneration(generation uint64) bool { // Only EMULATOR sizing goes through here. The resize producers // (resizeAllPanes, diffResizes) keep calling paneVTSize: a master sends its // own rect's size, and a follower sends nothing at all. +// adoptDaemonSize records a daemon size for the pane unless the pane already +// holds a NEWER one. Equal is adopted, so a repeat of the same announcement +// is idempotent and a daemon without the counter (always 0) always wins. +// Reports whether it adopted. +func (p *PaneModel) adoptDaemonSize(cols, rows int, seq uint64) bool { + if seq < p.daemonSizeSeq { + return false + } + p.daemonCols, p.daemonRows, p.daemonSizeSeq = cols, rows, seq + return true +} + func (p *PaneModel) targetVTSize(rectW, rectH, nativeW, canvasW, canvasH int) (cols, rows int) { if p.follower && p.daemonCols > 0 && p.daemonRows > 0 { return p.daemonCols, p.daemonRows diff --git a/internal/tui/reconnect.go b/internal/tui/reconnect.go index 25229647..7b8e2c57 100644 --- a/internal/tui/reconnect.go +++ b/internal/tui/reconnect.go @@ -923,6 +923,9 @@ func (m *Model) armReattachReset(dest string) { p.reattachReset = true // The daemon may have restarted too, resetting its PTY run counter. p.outputGeneration = 0 + // And its per-pane size counter: a restarted daemon numbers sizes from + // 1 again, and a kept number would refuse every one of them as stale. + p.daemonSizeSeq = 0 // Forget that this pane has been sized. The suppression in diffResizes // describes a daemon-side guard (appliedCols/appliedRows) that a PTY // reinstall zeroes, so carrying it across an outage would withhold the diff --git a/internal/tui/workstate.go b/internal/tui/workstate.go index 9e03cf21..750648ed 100644 --- a/internal/tui/workstate.go +++ b/internal/tui/workstate.go @@ -928,10 +928,13 @@ func syncPaneMeta(pane *PaneModel, info *PaneInfo, wideCanvas bool, minNativeCol pane.daemonMouseTracking = info.MouseTracking pane.daemonMouseSGR = info.MouseSGR pane.daemonBracketedPaste = info.BracketedPaste - // Unconditional, like the rest: the last size the daemon ACCEPTED is the - // size a follower's VT takes, and 0x0 (never sized) must fall back. + // The last size the daemon ACCEPTED is the size a follower's VT takes, and + // 0x0 (never sized) must fall back. NOT unconditional like the rest: a + // broadcast built while a resize batch was in flight carries the size from + // BEFORE it, and would undo the pane_sizes frame that already resized this + // pane's VT — see adoptDaemonSize. pane.follower = follower - pane.daemonCols, pane.daemonRows = int(info.Cols), int(info.Rows) + pane.adoptDaemonSize(int(info.Cols), int(info.Rows), info.SizeSeq) // Unconditional copy, like the other daemon-authoritative fields: the // daemon writes LastModel BEFORE broadcasting the hook event and IPC // delivery is ordered per connection, so a snapshot can never lag behind From b1ffd4e4dd8ee6716e1bd0f94cc849ca099f03af Mon Sep 17 00:00:00 2001 From: Artjoms Stukans Date: Fri, 25 Sep 2026 22:55:41 +0200 Subject: [PATCH 16/40] feat(tui): sync tab layouts between clients by revision Each tab now tracks the daemon's layout revision. A client sends its tree only when its own user changed it (split fill, own close, arrange, pane-drag drop, border-drag release, a template's first tree), with BaseRev set to the revision it was built on, and adopts any broadcast carrying a higher revision, reusing PaneModels by id. It never sends because the stored tree merely disagrees, so two clients can no longer re-send each other's trees. Arrivals and prunes nobody on this client asked for are placed locally and awaited; only when the next broadcast's stored tree still lacks them does the client send. Adoption cancels a drag armed on the tab and re-seats this client's pendingSplit beside its original sibling. A change made while a write is in flight is held and sent on that write's echo. armReattachReset zeroes the revisions so the daemon's tree wins after a reattach, even at a lower revision. diffLayouts, layoutAgrees, sendAllLayouts and sendTabLayout are gone; every write is marshalled on the Update goroutine. --- internal/tui/arrange_apply.go | 23 +- internal/tui/arrange_apply_test.go | 2 +- internal/tui/broadcast_echo_test.go | 54 +-- internal/tui/dialog.go | 6 + internal/tui/layout_sync_test.go | 514 +++++++++++++++++++++++++++ internal/tui/layoutsync.go | 305 ++++++++++++++++ internal/tui/model.go | 230 ++++++------ internal/tui/move_pane_apply_test.go | 46 ++- internal/tui/project_merge_test.go | 2 +- internal/tui/projectdialog.go | 2 +- internal/tui/reconnect.go | 4 + internal/tui/router.go | 4 +- internal/tui/router_test.go | 2 +- internal/tui/splitdrag_test.go | 2 +- internal/tui/tab.go | 73 +++- internal/tui/template_layout.go | 13 +- internal/tui/template_layout_test.go | 8 +- 17 files changed, 1080 insertions(+), 210 deletions(-) create mode 100644 internal/tui/layout_sync_test.go create mode 100644 internal/tui/layoutsync.go diff --git a/internal/tui/arrange_apply.go b/internal/tui/arrange_apply.go index efa7e3c9..68f05c3d 100644 --- a/internal/tui/arrange_apply.go +++ b/internal/tui/arrange_apply.go @@ -1,8 +1,6 @@ package tui import ( - "log" - tea "charm.land/bubbletea/v2" "github.com/artyomsv/quil/internal/keymap" @@ -82,8 +80,8 @@ func (m *Model) arrangeTab(tab *TabModel, kind layoutKind) tea.Cmd { // the tab being arranged may not be the one on screen. // // On success: the tree, the active pane (and every Active flag in the tab), -// focus mode off, one resize at the canonical geometry, and one layout send -// for THIS tab. +// focus mode off, one resize at the canonical geometry, and one layout write +// for THIS tab (markLayoutChanged). func (m *Model) applyTabArrangement(tab *TabModel, newRoot *LayoutNode, active *PaneModel) tea.Cmd { if tab == nil || newRoot == nil || m.notesMode { return nil @@ -116,20 +114,5 @@ func (m *Model) applyTabArrangement(tab *TabModel, newRoot *LayoutNode, active * tab.SetCanvas(w, h) tab.SetChrome(m.projectSidebarWidth()) tab.Resize(w, h) - return tea.Batch(m.resizeAllPanes(), m.sendTabLayout(tab)) -} - -// sendTabLayout persists ONE tab's tree. It marshals here, on the Update -// goroutine, and ships through sendDiffedLayouts — sendAllLayouts would re-send -// every tab of every project, reading the trees from a Cmd goroutine. -func (m *Model) sendTabLayout(tab *TabModel) tea.Cmd { - if tab == nil || tab.Root == nil { - return nil - } - data, err := MarshalLayout(tab.Root) - if err != nil { - log.Printf("tab layout: marshal %s: %v", tab.ID, err) - return nil - } - return m.sendDiffedLayouts([]layoutSend{{dest: m.destOfTab(tab.ID), tabID: tab.ID, data: data}}) + return tea.Batch(m.resizeAllPanes(), m.markLayoutChanged(m.destOfTab(tab.ID), tab)) } diff --git a/internal/tui/arrange_apply_test.go b/internal/tui/arrange_apply_test.go index 19f187e5..7d2b613c 100644 --- a/internal/tui/arrange_apply_test.go +++ b/internal/tui/arrange_apply_test.go @@ -172,7 +172,7 @@ func TestTabLayoutMenu_ColumnsArrangesTheTabItWasOpenedOn(t *testing.T) { if sends[0].TabID != "tab-b" { t.Errorf("layout sent for %q, want tab-b", sends[0].TabID) } - if !layoutAgrees(sends[0].Layout, tb.Root) { + if !sentLayoutIs(sends[0].Layout, tb.Root) { t.Error("the sent layout does not match tab-b's new tree") } } diff --git a/internal/tui/broadcast_echo_test.go b/internal/tui/broadcast_echo_test.go index 0ceb32b6..d870dbc1 100644 --- a/internal/tui/broadcast_echo_test.go +++ b/internal/tui/broadcast_echo_test.go @@ -72,37 +72,6 @@ func TestBroadcastLayoutBytesDifferForASplit(t *testing.T) { } } -func TestLayoutAgrees(t *testing.T) { - t.Parallel() - leaf := NewLeaf(NewPaneModel("p1", 1024)) - split := NewLeaf(NewPaneModel("p1", 1024)) - split.SplitLeaf("p1", SplitHorizontal) - split.Right.Pane = NewPaneModel("p2", 1024) - - tests := []struct { - name string - stored json.RawMessage - root *LayoutNode - want bool - }{ - {"empty stored means the daemon holds nothing", nil, leaf, false}, - {"zero-length stored", json.RawMessage{}, leaf, false}, - {"malformed stored", json.RawMessage(`{"split":`), leaf, false}, - {"matching leaf", broadcastLayout(t, leaf), leaf, true}, - {"matching split", broadcastLayout(t, split), split, true}, - {"leaf stored against a split tree", broadcastLayout(t, leaf), split, false}, - {"split stored against a leaf tree", broadcastLayout(t, split), leaf, false}, - } - for _, tt := range tests { - t.Run(tt.name, func(t *testing.T) { - t.Parallel() - if got := layoutAgrees(tt.stored, tt.root); got != tt.want { - t.Errorf("layoutAgrees = %v, want %v", got, tt.want) - } - }) - } -} - // A layout the daemon stores but the client cannot parse must be re-sent, not // treated as agreeing — otherwise a corrupt stored tree is never corrected. It // costs one frame per broadcast for that tab until the daemon accepts the @@ -166,13 +135,14 @@ func echoModel(t *testing.T) (Model, WorkspaceStateMsg) { } // The echo: same tabs and panes, now carrying the layout the daemon would - // have stored and broadcast back. + // have stored and broadcast back — at revision 1, since storing the + // client's first write is what bumped it from 0. echo := initial echo.Tabs = []TabInfo{ {ID: "t-split", Name: "Split", Panes: []string{"p1", "p2"}, - Layout: broadcastLayout(t, m.curTabs()[0].Root)}, + Layout: broadcastLayout(t, m.curTabs()[0].Root), LayoutRev: 1}, {ID: "t-solo", Name: "Solo", Panes: []string{"p3"}, - Layout: broadcastLayout(t, m.curTabs()[1].Root)}, + Layout: broadcastLayout(t, m.curTabs()[1].Root), LayoutRev: 1}, } return m, echo } @@ -698,6 +668,10 @@ func TestWorkspaceState_AbsentLayout_StillSends(t *testing.T) { // The mirror of the suppression test: a real divergence must still reach the // daemon, and only for the tab that diverged. +// +// Under layout sync (layoutsync.go) the stale tree is ADOPTED first, and the +// pane it lacks is placed locally and awaited — nobody on this client asked +// for it. Only the next broadcast, still lacking it, sends. func TestWorkspaceState_ChangedLayout_SendsOnlyThatTab(t *testing.T) { t.Parallel() m, echo := echoModel(t) @@ -706,10 +680,18 @@ func TestWorkspaceState_ChangedLayout_SendsOnlyThatTab(t *testing.T) { stale := NewLeaf(NewPaneModel("p1", 1024)) echo.Tabs[0].Layout = broadcastLayout(t, stale) + first := &echoRecorder{} + m.client = first + next, cmd := m.Update(echo) + runCmd(cmd) + m = next.(Model) + if layouts, _ := sentCounts(t, first); layouts != 0 { + t.Fatalf("the adopting broadcast sent %d layouts, want 0", layouts) + } + fs := &echoRecorder{} m.client = fs - - _, cmd := m.Update(echo) + _, cmd = m.Update(echo) runCmd(cmd) var gotTabs []string diff --git a/internal/tui/dialog.go b/internal/tui/dialog.go index c4947105..583d1e09 100644 --- a/internal/tui/dialog.go +++ b/internal/tui/dialog.go @@ -1359,6 +1359,10 @@ func (m Model) handleConfirmKey(msg tea.KeyPressMsg) (tea.Model, tea.Cmd) { switch kind { case "pane": dest = m.destOfPane(id) + if m.closeRequested == nil { + m.closeRequested = make(map[string]bool) + } + m.closeRequested[id] = true case "tab": dest = m.destOfTab(id) } @@ -2748,6 +2752,7 @@ func (m Model) handleCreatePaneSplit() (tea.Model, tea.Cmd) { m.pendingSplit = make(map[string]*LayoutNode) } m.pendingSplit[tab.ID] = leaf + tab.noteReservation(oldPaneID, 0, true) // What the reserved leaf is waiting for, recorded on the node so it // dies with it. A replace disposes the pane it stands in for at send // time, so on a single-pane tab this placeholder IS the tab. @@ -2819,6 +2824,7 @@ func (m Model) handleCreatePaneSplit() (tea.Model, tea.Cmd) { m.pendingSplit = make(map[string]*LayoutNode) } m.pendingSplit[tab.ID] = placeholder + tab.noteReservation(pane.ID, dir, false) // See the replace arm: recorded on the node, so it needs no unwinding. placeholder.phType = pluginName diff --git a/internal/tui/layout_sync_test.go b/internal/tui/layout_sync_test.go new file mode 100644 index 00000000..ebd6c44f --- /dev/null +++ b/internal/tui/layout_sync_test.go @@ -0,0 +1,514 @@ +package tui + +import ( + "encoding/json" + "reflect" + "strconv" + "testing" + + tea "charm.land/bubbletea/v2" + + "github.com/artyomsv/quil/internal/ipc" +) + +// Layout sync between several TUIs on one daemon (spec §7.2). Every decision +// is driven through Update with a workspace_state broadcast, the only thing a +// client ever learns another client's tree from. + +const lsProject = "proj-ls" + +// lsLeaf and lsSplit build a stored tree the way the daemon holds it. +func lsLeaf(id string) *SerializedNode { return &SerializedNode{PaneID: id} } + +func lsSplit(dir SplitDir, ratio float64, l, r *SerializedNode) *SerializedNode { + return &SerializedNode{Split: &dir, Ratio: ratio, Left: l, Right: r} +} + +// lsWire renders a stored tree as it reaches a client: parseWorkspaceState +// decodes the broadcast into map[string]any and re-marshals the layout, so the +// bytes are map-ordered (see broadcastLayout). +func lsWire(t *testing.T, s *SerializedNode) json.RawMessage { + t.Helper() + if s == nil { + return nil + } + raw, err := json.Marshal(s) + if err != nil { + t.Fatalf("marshal: %v", err) + } + var generic any + if err := json.Unmarshal(raw, &generic); err != nil { + t.Fatalf("decode: %v", err) + } + out, err := json.Marshal(generic) + if err != nil { + t.Fatalf("re-marshal: %v", err) + } + return out +} + +// lsState is a one-project, one-tab ("t1") broadcast at layout revision rev. +func lsState(rev uint64, layout json.RawMessage, panes ...string) WorkspaceStateMsg { + st := WorkspaceStateMsg{ + ActiveProject: lsProject, + ActiveTab: "t1", + Projects: []ProjectInfo{{ID: lsProject, Name: "ls", TabIDs: []string{"t1"}, ActiveTab: "t1"}}, + Tabs: []TabInfo{{ID: "t1", Name: "t1", ProjectID: lsProject, Panes: panes, Layout: layout, LayoutRev: rev}}, + } + for _, p := range panes { + st.Panes = append(st.Panes, PaneInfo{ID: p, TabID: "t1", Type: "terminal"}) + } + return st +} + +func newLayoutSyncModel(t *testing.T) Model { + t.Helper() + m := newModelForTest(nil, 0) + m.width, m.height = 120, 40 + m.client = &echoRecorder{} + m.notifications = NewNotificationCenter(30, 200) + m.mcpHighlights = make(map[string]bool) + return m +} + +// lsApply delivers st through Update, runs every command it returned, and +// returns the update_layout frames those commands sent. +func lsApply(t *testing.T, m Model, st WorkspaceStateMsg) (Model, []ipc.UpdateLayoutPayload) { + t.Helper() + rec := &echoRecorder{} + m.client = rec + next, cmd := m.Update(st) + runCmd(cmd) + return next.(Model), lsLayouts(t, rec) +} + +// lsRun runs a user action's command against a fresh recorder and returns the +// update_layout frames it sent. +func lsRun(t *testing.T, m *Model, action func() tea.Cmd) []ipc.UpdateLayoutPayload { + t.Helper() + rec := &echoRecorder{} + m.client = rec + runCmd(action()) + return lsLayouts(t, rec) +} + +func lsLayouts(t *testing.T, rec *echoRecorder) []ipc.UpdateLayoutPayload { + t.Helper() + var out []ipc.UpdateLayoutPayload + for _, msg := range rec.sent { + if msg.Type != ipc.MsgUpdateLayout { + continue + } + var p ipc.UpdateLayoutPayload + if err := msg.DecodePayload(&p); err != nil { + t.Fatalf("decode update_layout: %v", err) + } + out = append(out, p) + } + return out +} + +func lsTab(t *testing.T, m *Model) *TabModel { + t.Helper() + tab := m.tabByID("t1") + if tab == nil { + t.Fatal("tab t1 is missing") + } + return tab +} + +// lsTree is the tab's tree in its stored form, placeholders omitted. +func lsTree(t *testing.T, m *Model) *SerializedNode { + t.Helper() + return SerializeLayout(lsTab(t, m).Root) +} + +// sentLayoutIs reports whether a sent layout describes root, structurally +// (the wire bytes may be struct- or map-ordered). +func sentLayoutIs(raw json.RawMessage, root *LayoutNode) bool { + s, err := UnmarshalLayout(raw) + return err == nil && s != nil && reflect.DeepEqual(s, SerializeLayout(root)) +} + +func lsSentTree(t *testing.T, p ipc.UpdateLayoutPayload) *SerializedNode { + t.Helper() + s, err := UnmarshalLayout(p.Layout) + if err != nil { + t.Fatalf("sent layout does not parse: %v", err) + } + return s +} + +func lsBase(p ipc.UpdateLayoutPayload) string { + if p.BaseRev == nil { + return "nil" + } + return strconv.FormatUint(*p.BaseRev, 10) +} + +// lsParentOf returns node's parent in root and whether node is its left child. +func lsParentOf(root, node *LayoutNode) (*LayoutNode, bool) { + if root == nil || root.IsLeaf() { + return nil, false + } + if root.Left == node { + return root, true + } + if root.Right == node { + return root, false + } + if p, l := lsParentOf(root.Left, node); p != nil { + return p, l + } + return lsParentOf(root.Right, node) +} + +func TestLayoutSync_HigherRevAdoptedKeepsPaneModels(t *testing.T) { + t.Setenv("QUIL_HOME", t.TempDir()) + m := newLayoutSyncModel(t) + m, _ = lsApply(t, m, lsState(3, lsWire(t, lsSplit(SplitHorizontal, 0.5, lsLeaf("p1"), lsLeaf("p2"))), "p1", "p2")) + tab := lsTab(t, &m) + p1, p2 := mpPane(t, tab, "p1"), mpPane(t, tab, "p2") + + theirs := lsSplit(SplitVertical, 0.3, lsLeaf("p2"), lsLeaf("p1")) + m, sent := lsApply(t, m, lsState(4, lsWire(t, theirs), "p1", "p2")) + + if got := lsTree(t, &m); !reflect.DeepEqual(got, theirs) { + t.Fatalf("tree = %s, want the adopted %s", layoutString(got), layoutString(theirs)) + } + tab = lsTab(t, &m) + if mpPane(t, tab, "p1") != p1 || mpPane(t, tab, "p2") != p2 { + t.Error("adoption built new PaneModels — the panes' emulators and scrollback were lost") + } + if tab.layoutRev != 4 { + t.Errorf("layoutRev = %d, want 4", tab.layoutRev) + } + if len(sent) != 0 { + t.Errorf("adoption sent %d update_layout frames, want 0", len(sent)) + } +} + +func TestLayoutSync_EqualRevDirtyKeepsLocal(t *testing.T) { + t.Setenv("QUIL_HOME", t.TempDir()) + m := newLayoutSyncModel(t) + stored := lsSplit(SplitHorizontal, 0.5, lsLeaf("p1"), lsLeaf("p2")) + m, _ = lsApply(t, m, lsState(3, lsWire(t, stored), "p1", "p2")) + + sent := lsRun(t, &m, func() tea.Cmd { return m.arrangeTab(lsTab(t, &m), layoutRows) }) + if len(sent) != 1 || lsBase(sent[0]) != "3" { + t.Fatalf("the arrangement sent %d frames, want 1 with base 3", len(sent)) + } + local := lsTree(t, &m) + + // Our write is in flight: a broadcast still at rev 3 carries the OLD tree. + m, sent = lsApply(t, m, lsState(3, lsWire(t, stored), "p1", "p2")) + if got := lsTree(t, &m); !reflect.DeepEqual(got, local) { + t.Errorf("tree = %s, want the local %s kept while our write is in flight", layoutString(got), layoutString(local)) + } + if len(sent) != 0 { + t.Errorf("sent %d frames, want 0", len(sent)) + } +} + +// Two local changes inside one round trip: the second would carry the same +// base as the first and be refused behind it, then lost to the first one's +// echo. It is held instead, and sent on that echo with the new base. +func TestLayoutSync_ChangeWhileInFlightIsSentOnTheEcho(t *testing.T) { + t.Setenv("QUIL_HOME", t.TempDir()) + m := newLayoutSyncModel(t) + m, _ = lsApply(t, m, lsState(3, lsWire(t, lsSplit(SplitHorizontal, 0.5, lsLeaf("p1"), lsLeaf("p2"))), "p1", "p2")) + + first := lsRun(t, &m, func() tea.Cmd { return m.arrangeTab(lsTab(t, &m), layoutRows) }) + if len(first) != 1 || lsBase(first[0]) != "3" { + t.Fatalf("the first change sent %d frames, want 1 with base 3", len(first)) + } + lsTab(t, &m).Root.Ratio = 0.7 // a second change, as a border drag leaves it + if sent := lsRun(t, &m, func() tea.Cmd { return m.markLayoutChanged("", lsTab(t, &m)) }); len(sent) != 0 { + t.Fatalf("the second change sent %d frames while the first was in flight, want 0", len(sent)) + } + second := lsTree(t, &m) + + m, sent := lsApply(t, m, lsState(4, lsWire(t, lsSentTree(t, first[0])), "p1", "p2")) + if len(sent) != 1 || lsBase(sent[0]) != "4" { + t.Fatalf("the echo sent %d frames, want the held change once with base 4", len(sent)) + } + if got := lsSentTree(t, sent[0]); !reflect.DeepEqual(got, second) { + t.Errorf("sent %s, want the held %s", layoutString(got), layoutString(second)) + } + if got := lsTree(t, &m); !reflect.DeepEqual(got, second) { + t.Errorf("tree = %s, want the local %s kept through our own echo", layoutString(got), layoutString(second)) + } +} + +func TestLayoutSync_LocalSplitSendsOneUpdateWithBase(t *testing.T) { + t.Setenv("QUIL_HOME", t.TempDir()) + m := newLayoutSyncModel(t) + m, _ = lsApply(t, m, lsState(5, lsWire(t, lsLeaf("p1")), "p1")) + + if sent := lsRun(t, &m, func() tea.Cmd { return m.splitPane(SplitHorizontal) }); len(sent) != 0 { + t.Fatalf("the split sent %d layouts before its pane existed, want 0", len(sent)) + } + m, sent := lsApply(t, m, lsState(5, lsWire(t, lsLeaf("p1")), "p1", "p-new")) + + want := lsSplit(SplitHorizontal, 0.5, lsLeaf("p1"), lsLeaf("p-new")) + if len(sent) != 1 { + t.Fatalf("sent %d update_layout frames, want exactly 1", len(sent)) + } + if sent[0].TabID != "t1" || lsBase(sent[0]) != "5" { + t.Errorf("sent tab %q base %s, want t1 base 5", sent[0].TabID, lsBase(sent[0])) + } + if got := lsSentTree(t, sent[0]); !reflect.DeepEqual(got, want) { + t.Errorf("sent %s, want %s", layoutString(got), layoutString(want)) + } + + // The echo of our own write: adopted without a word. + m, sent = lsApply(t, m, lsState(6, lsWire(t, want), "p1", "p-new")) + if len(sent) != 0 { + t.Errorf("the echo produced %d frames, want 0", len(sent)) + } + if tab := lsTab(t, &m); tab.layoutRev != 6 || tab.layoutDirty { + t.Errorf("after the echo layoutRev=%d dirty=%v, want 6 and clean", tab.layoutRev, tab.layoutDirty) + } +} + +func TestLayoutSync_RatioOnlyDifferenceSendsNothing(t *testing.T) { + t.Setenv("QUIL_HOME", t.TempDir()) + m := newLayoutSyncModel(t) + m, _ = lsApply(t, m, lsState(7, lsWire(t, lsSplit(SplitHorizontal, 0.5, lsLeaf("p1"), lsLeaf("p2"))), "p1", "p2")) + + differs := lsState(7, lsWire(t, lsSplit(SplitHorizontal, 0.6, lsLeaf("p1"), lsLeaf("p2"))), "p1", "p2") + for i := 0; i < 3; i++ { + var sent []ipc.UpdateLayoutPayload + m, sent = lsApply(t, m, differs) + if len(sent) != 0 { + t.Fatalf("broadcast %d sent %d update_layout frames, want 0 — a clean tab at the "+ + "stored revision never sends on a mere disagreement", i+1, len(sent)) + } + } +} + +func TestLayoutSync_AdoptionCancelsDrag(t *testing.T) { + t.Setenv("QUIL_HOME", t.TempDir()) + stored := lsSplit(SplitHorizontal, 0.5, lsLeaf("p1"), lsLeaf("p2")) + theirs := lsSplit(SplitVertical, 0.5, lsLeaf("p1"), lsLeaf("p2")) + + t.Run("split border", func(t *testing.T) { + m := newLayoutSyncModel(t) + m, _ = lsApply(t, m, lsState(1, lsWire(t, stored), "p1", "p2")) + next, _ := m.Update(tea.MouseClickMsg{X: 60, Y: 10, Button: tea.MouseLeft}) + m = next.(Model) + if m.splitDragNode == nil { + t.Fatal("setup: the press did not arm a border drag") + } + m, _ = lsApply(t, m, lsState(2, lsWire(t, theirs), "p1", "p2")) + if m.splitDragNode != nil { + t.Error("the border drag survived adoption of another tree") + } + next, cmd := m.Update(tea.MouseReleaseMsg{X: 90, Y: 10, Button: tea.MouseLeft}) + m = next.(Model) + rec := &echoRecorder{} + m.client = rec + runCmd(cmd) + if sent := lsLayouts(t, rec); len(sent) != 0 { + t.Errorf("the release after adoption sent %d layouts, want 0", len(sent)) + } + if got := lsTree(t, &m); !reflect.DeepEqual(got, theirs) { + t.Errorf("tree = %s, want the adopted %s", layoutString(got), layoutString(theirs)) + } + }) + + t.Run("pane drag", func(t *testing.T) { + m := newLayoutSyncModel(t) + m, _ = lsApply(t, m, lsState(1, lsWire(t, stored), "p1", "p2")) + next, _ := m.Update(tea.MouseClickMsg{X: 20, Y: 10, Button: tea.MouseLeft, Mod: tea.ModAlt}) + m = next.(Model) + if !m.paneDrag.active() { + t.Fatal("setup: Alt+press did not arm a pane drag") + } + m, _ = lsApply(t, m, lsState(2, lsWire(t, theirs), "p1", "p2")) + if m.paneDrag.active() { + t.Error("the pane drag survived adoption of another tree") + } + }) +} + +func TestLayoutSync_PendingSplitSurvivesBesideSibling(t *testing.T) { + t.Setenv("QUIL_HOME", t.TempDir()) + m := newLayoutSyncModel(t) + m, _ = lsApply(t, m, lsState(1, lsWire(t, lsSplit(SplitHorizontal, 0.5, lsLeaf("p1"), lsLeaf("p2"))), "p1", "p2")) + tab := lsTab(t, &m) + tab.ActivePane = "p2" + _ = m.splitPane(SplitVertical) + if m.pendingSplit["t1"] == nil { + t.Fatal("setup: the split armed no reservation") + } + // Held open the way a worktree create holds it, so a broadcast that + // carries no new pane does not prune it. + m.worktreeCreates = map[string]string{"t1": "feat/x"} + + // Another client swapped the two panes. + m, _ = lsApply(t, m, lsState(2, lsWire(t, lsSplit(SplitHorizontal, 0.5, lsLeaf("p2"), lsLeaf("p1"))), "p1", "p2")) + + tab = lsTab(t, &m) + ph := m.pendingSplit["t1"] + if ph == nil || !treeHoldsNode(tab.Root, ph) { + t.Fatal("the reservation was lost on adoption — the requested pane would land nowhere") + } + parent, isLeft := lsParentOf(tab.Root, ph) + if parent == nil || isLeft || parent.Split != SplitVertical || + parent.Left == nil || !parent.Left.IsLeaf() || parent.Left.Pane.ID != "p2" { + t.Fatalf("reservation is not below its sibling p2: tree %s", layoutString(SerializeLayout(tab.Root))) + } + + // The requested pane lands in it, and this client — the requester — sends. + m, sent := lsApply(t, m, lsState(2, lsWire(t, lsSplit(SplitHorizontal, 0.5, lsLeaf("p2"), lsLeaf("p1"))), "p1", "p2", "p-new")) + want := lsSplit(SplitHorizontal, 0.5, lsSplit(SplitVertical, 0.5, lsLeaf("p2"), lsLeaf("p-new")), lsLeaf("p1")) + if got := lsTree(t, &m); !reflect.DeepEqual(got, want) { + t.Errorf("tree = %s, want %s", layoutString(got), layoutString(want)) + } + if len(sent) != 1 || lsBase(sent[0]) != "2" { + t.Errorf("sent %d frames, want 1 with base 2", len(sent)) + } +} + +// Spec 7a: only the requester of a pane sends the tree that places it. +func TestLayoutSync_ArrivalRace(t *testing.T) { + t.Setenv("QUIL_HOME", t.TempDir()) + before := lsState(3, lsWire(t, lsLeaf("p1")), "p1") + arrived := lsState(3, lsWire(t, lsLeaf("p1")), "p1", "px") + + a, b := newLayoutSyncModel(t), newLayoutSyncModel(t) + a, _ = lsApply(t, a, before) + b, _ = lsApply(t, b, before) + _ = a.splitPane(SplitHorizontal) + + a, sentA := lsApply(t, a, arrived) + b, sentB := lsApply(t, b, arrived) + if len(sentA) != 1 || lsBase(sentA[0]) != "3" { + t.Fatalf("the requester sent %d frames, want 1 with base 3", len(sentA)) + } + if len(sentB) != 0 { + t.Fatalf("a bystander sent %d frames on the arrival, want 0 — its placement "+ + "would race the requester's on the same base", len(sentB)) + } + + next := lsState(4, lsWire(t, lsSentTree(t, sentA[0])), "p1", "px") + a, sentA = lsApply(t, a, next) + b, sentB = lsApply(t, b, next) + if len(sentA)+len(sentB) != 0 { + t.Errorf("rev 4 produced %d+%d frames, want none", len(sentA), len(sentB)) + } + if ta, tb := lsTree(t, &a), lsTree(t, &b); !reflect.DeepEqual(ta, tb) { + t.Errorf("trees differ after adoption: A=%s B=%s", layoutString(ta), layoutString(tb)) + } +} + +// An MCP-created pane nobody requested: every client places it locally, and +// on the next broadcast that still lacks it, each sends once. +func TestLayoutSync_UnrequestedArrivalSendsOnceIfStillMissing(t *testing.T) { + t.Setenv("QUIL_HOME", t.TempDir()) + before := lsState(3, lsWire(t, lsLeaf("p1")), "p1") + arrived := lsState(3, lsWire(t, lsLeaf("p1")), "p1", "p-mcp") + + a, b := newLayoutSyncModel(t), newLayoutSyncModel(t) + a, _ = lsApply(t, a, before) + b, _ = lsApply(t, b, before) + + a, sentA := lsApply(t, a, arrived) + b, sentB := lsApply(t, b, arrived) + if len(sentA)+len(sentB) != 0 { + t.Fatalf("the arrival itself sent %d+%d frames, want none", len(sentA), len(sentB)) + } + + // The stored tree still lacks it: nobody asked, so each client sends once. + a, sentA = lsApply(t, a, arrived) + b, sentB = lsApply(t, b, arrived) + if len(sentA) != 1 || len(sentB) != 1 || lsBase(sentA[0]) != "3" || lsBase(sentB[0]) != "3" { + t.Fatalf("still missing: sent %d and %d frames, want 1 each with base 3", len(sentA), len(sentB)) + } + + // A's write won; B's was refused. Both adopt, and it stays quiet. + winner := lsState(4, lsWire(t, lsSentTree(t, sentA[0])), "p1", "p-mcp") + for i := 0; i < 2; i++ { + a, sentA = lsApply(t, a, winner) + b, sentB = lsApply(t, b, winner) + if len(sentA)+len(sentB) != 0 { + t.Errorf("rev 4 broadcast %d produced %d+%d frames, want none", i+1, len(sentA), len(sentB)) + } + } + if ta, tb := lsTree(t, &a), lsTree(t, &b); !reflect.DeepEqual(ta, tb) { + t.Errorf("trees differ: A=%s B=%s", layoutString(ta), layoutString(tb)) + } +} + +// Spec 7b: after a reattach the daemon is the authority, even when its +// revision is LOWER — a restarted daemon lost the debounced snapshot. +func TestLayoutSync_ReattachAdoptsLowerRev(t *testing.T) { + t.Setenv("QUIL_HOME", t.TempDir()) + m := newLayoutSyncModel(t) + m, _ = lsApply(t, m, lsState(9, lsWire(t, lsSplit(SplitHorizontal, 0.5, lsLeaf("p1"), lsLeaf("p2"))), "p1", "p2")) + + m.armReattachReset("") + + restored := lsSplit(SplitVertical, 0.4, lsLeaf("p2"), lsLeaf("p1")) + m, sent := lsApply(t, m, lsState(2, lsWire(t, restored), "p1", "p2")) + if got := lsTree(t, &m); !reflect.DeepEqual(got, restored) { + t.Errorf("tree = %s, want the daemon's %s adopted after the reattach", layoutString(got), layoutString(restored)) + } + if tab := lsTab(t, &m); tab.layoutRev != 2 { + t.Errorf("layoutRev = %d, want 2", tab.layoutRev) + } + if len(sent) != 0 { + t.Errorf("sent %d frames, want 0", len(sent)) + } +} + +func TestLayoutSync_EmptyStoredLayoutSendsOnce(t *testing.T) { + t.Setenv("QUIL_HOME", t.TempDir()) + m := newLayoutSyncModel(t) + m, _ = lsApply(t, m, lsState(4, lsWire(t, lsSplit(SplitHorizontal, 0.5, lsLeaf("p1"), lsLeaf("p2"))), "p1", "p2")) + local := lsTree(t, &m) + + empty := lsState(4, nil, "p1", "p2") + m, sent := lsApply(t, m, empty) + if len(sent) != 1 || lsBase(sent[0]) != "4" { + t.Fatalf("an empty stored layout sent %d frames, want 1 with base 4", len(sent)) + } + if got := lsSentTree(t, sent[0]); !reflect.DeepEqual(got, local) { + t.Errorf("sent %s, want the local %s", layoutString(got), layoutString(local)) + } + m, sent = lsApply(t, m, empty) + if len(sent) != 0 { + t.Errorf("a second empty broadcast sent %d more frames, want 0 — the first is in flight", len(sent)) + } +} + +// A close THIS client asked for is a user mutation: the requester sends, the +// bystander only prunes and waits. +func TestLayoutSync_OwnCloseSendsBystanderWaits(t *testing.T) { + t.Setenv("QUIL_HOME", t.TempDir()) + before := lsState(2, lsWire(t, lsSplit(SplitHorizontal, 0.5, lsLeaf("p1"), lsLeaf("p2"))), "p1", "p2") + closed := lsState(2, lsWire(t, lsSplit(SplitHorizontal, 0.5, lsLeaf("p1"), lsLeaf("p2"))), "p1") + + a, b := newLayoutSyncModel(t), newLayoutSyncModel(t) + a, _ = lsApply(t, a, before) + b, _ = lsApply(t, b, before) + lsTab(t, &a).ActivePane = "p2" + next, _ := a.openClosePaneConfirm() + a = next.(Model) + next, cmd := a.Update(tea.KeyPressMsg{Code: 'y', Text: "y"}) + a = next.(Model) + runCmd(cmd) + + a, sentA := lsApply(t, a, closed) + b, sentB := lsApply(t, b, closed) + if len(sentA) != 1 || lsBase(sentA[0]) != "2" { + t.Fatalf("the closing client sent %d frames, want 1 with base 2", len(sentA)) + } + if len(sentB) != 0 { + t.Fatalf("the bystander sent %d frames on the prune, want 0", len(sentB)) + } + if got := lsSentTree(t, sentA[0]); !reflect.DeepEqual(got, lsLeaf("p1")) { + t.Errorf("sent %s, want p1", layoutString(got)) + } +} diff --git a/internal/tui/layoutsync.go b/internal/tui/layoutsync.go new file mode 100644 index 00000000..e596f797 --- /dev/null +++ b/internal/tui/layoutsync.go @@ -0,0 +1,305 @@ +package tui + +import ( + "encoding/json" + "log" + "reflect" + "time" + + tea "charm.land/bubbletea/v2" +) + +// Layout sync between several TUIs on one daemon (spec §7.2). +// +// The daemon numbers every stored write of a tab's tree (layout_rev) and +// refuses a write whose base_rev is not the current one. A client therefore +// sends only when ITS user changed the tree (markLayoutChanged), and adopts +// any broadcast carrying a higher revision than the tree it holds. It never +// sends because the stored tree merely disagrees with its own — two clients +// doing that re-send each other's trees forever. +// +// A change nobody on this client asked for (an MCP create, another client's +// split or close, a moved pane) is placed or pruned locally and remembered in +// awaitingPanes/awaitingGone. The requester's write normally arrives with the +// next broadcast and is adopted; only when that broadcast's stored tree still +// lacks the change does every client send, and the first write wins. + +// markLayoutChanged records a tree change this client's user made and returns +// the command that stores it on dest. It marshals HERE, on the Update +// goroutine: the command holds only the bytes, never the tab. +// +// A write already in flight for the tab defers this one (layoutResend): both +// would carry the same base, so the second would be refused behind the first +// and then lost to the first one's echo. The echo sends it instead +// (syncTabLayout). +func (m *Model) markLayoutChanged(dest string, tab *TabModel) tea.Cmd { + if tab == nil || tab.Root == nil || tab.templateLayoutPending { + return nil + } + if tab.layoutDirty { + tab.layoutResend = true + return nil + } + s := m.layoutForSend(tab) + if s == nil { + return nil + } + data, err := json.Marshal(s) + if err != nil { + log.Printf("tab layout: marshal %s: %v", tab.ID, err) + return nil + } + tab.layoutDirty, tab.layoutSent, tab.layoutResend = true, s, false + tab.clearAwaiting() + return m.sendDiffedLayouts([]layoutSend{{dest: dest, tabID: tab.ID, data: data, baseRev: tab.layoutRev}}) +} + +// layoutForSend is the tab's tree as it may be stored: no placeholders, since +// a reservation is this client's runtime state and means nothing to another. +// The one exception is a worktree REPLACE, whose placeholder stands in for a +// pane that is still live on the daemon — it is written under that pane's id, +// so another client adopting the tree keeps the pane where it is. +func (m *Model) layoutForSend(tab *TabModel) *SerializedNode { + var keep *LayoutNode + var keepID string + if held := m.worktreeReplaced[tab.ID]; held != nil { + keep, keepID = m.pendingSplit[tab.ID], held.ID + } + return serializeForSend(tab.Root, keep, keepID) +} + +func serializeForSend(n, keep *LayoutNode, keepID string) *SerializedNode { + if n == nil { + return nil + } + if n.IsLeaf() { + return &SerializedNode{PaneID: n.Pane.ID} + } + if n.Left == nil && n.Right == nil { + if n == keep && keepID != "" { + return &SerializedNode{PaneID: keepID} + } + return nil + } + l, r := serializeForSend(n.Left, keep, keepID), serializeForSend(n.Right, keep, keepID) + if l == nil { + return r + } + if r == nil { + return l + } + split := n.Split + return &SerializedNode{Split: &split, Ratio: n.Ratio, Left: l, Right: r} +} + +// serializedIDs is the set of pane ids a stored tree names. +func serializedIDs(s *SerializedNode) map[string]bool { + ids := make(map[string]bool) + var walk func(*SerializedNode) + walk = func(n *SerializedNode) { + if n == nil { + return + } + if n.PaneID != "" { + ids[n.PaneID] = true + return + } + walk(n.Left) + walk(n.Right) + } + walk(s) + return ids +} + +// storedLayout parses a broadcast's layout, or nil when there is none or it +// does not parse — both mean the daemon holds nothing this client can use. +func storedLayout(raw json.RawMessage) *SerializedNode { + s, err := UnmarshalLayout(raw) + if err != nil { + return nil + } + return s +} + +// layoutPass is what syncTabLayout decided for one tab of one broadcast. +type layoutPass struct { + // send: this pass must end in markLayoutChanged. + send bool + // oldTree holds the ids the tab's tree had BEFORE adoption, which is what + // tells a moved pane (another tab's) from one this tab already held. + oldTree map[string]bool + // created: pane ids adoption built a new PaneModel for. + created []string +} + +// syncTabLayout runs for an EXISTING tab before its panes are reconciled, and +// decides what the broadcast's revision means for the tree it holds. +func (m *Model) syncTabLayout(tab *TabModel, ti TabInfo, paneSet map[string]bool, paneMap map[string]*PaneInfo, existingPanes map[string]*PaneModel) layoutPass { + lp := layoutPass{oldTree: map[string]bool{}} + if tab.Root != nil { + lp.oldTree = tab.Root.PaneIDs() + } + stored := storedLayout(ti.Layout) + + // Every tree comparison here is STRUCTURAL (parsed SerializedNode), never + // bytes: the daemon stores the struct-ordered bytes a client sent, but + // parseWorkspaceState re-marshals the layout from map[string]any, which + // sorts keys — so the same split tree arrives as different bytes. + switch { + case ti.LayoutRev > tab.layoutRev: + // Our own write coming back: the tree we hold already contains it, + // plus anything done since, which the deferred resend now carries. + echo := tab.layoutDirty && stored != nil && reflect.DeepEqual(stored, tab.layoutSent) + resend := echo && tab.layoutResend + tab.layoutRev = ti.LayoutRev + tab.layoutDirty, tab.layoutSent, tab.layoutResend = false, nil, false + if stored != nil && !echo && !reflect.DeepEqual(stored, m.layoutForSend(tab)) { + lp.created, lp.send = m.adoptTabLayout(tab, stored, paneSet, paneMap, existingPanes) + return lp + } + lp.send = resend + case ti.LayoutRev < tab.layoutRev || tab.layoutDirty: + // A dirty tab keeps its tree: our write is in flight, and whichever + // write the daemon stores next arrives with a higher revision. + return lp + } + + // The tab holds the stored revision and nothing of ours is in flight: the + // last broadcast's local placements are settled now. Anything the stored + // tree still has not caught up with was not written by the client that + // asked for it, so this one sends. + if len(tab.awaitingPanes) > 0 || len(tab.awaitingGone) > 0 { + ids := serializedIDs(stored) + for id := range tab.awaitingPanes { + if paneSet[id] && !ids[id] { + lp.send = true + } + } + for id := range tab.awaitingGone { + if !paneSet[id] && ids[id] { + lp.send = true + } + } + tab.clearAwaiting() + } + return lp +} + +// adoptTabLayout replaces tab's tree with the stored one. Pane models are +// reused by id, so no emulator or scrollback is lost; ids the broadcast no +// longer lists are dropped, and panes the stored tree lacks are left for the +// caller's arrival loop to place. Returns the ids it built new models for, and +// whether a pane THIS client closed was still in the stored tree (a user +// change the requester must store). +func (m *Model) adoptTabLayout(tab *TabModel, stored *SerializedNode, paneSet map[string]bool, paneMap map[string]*PaneInfo, existingPanes map[string]*PaneModel) (created []string, send bool) { + // A drag armed on this tab describes a tree that is about to go. + if (m.splitDragNode != nil && treeContains(tab.Root, m.splitDragNode)) || + (m.paneDrag.active() && m.paneDrag.srcTabID == tab.ID) { + m.clearDragState() + } + + held := m.worktreeReplaced[tab.ID] + panes := make(map[string]*PaneModel, len(paneSet)) + gone := make([]string, 0) + for id := range serializedIDs(stored) { + if !paneSet[id] { + // Gone from the daemon but still in the stored tree. + if m.closeRequested[id] { + delete(m.closeRequested, id) + send = true + } else { + gone = append(gone, id) + } + continue + } + if p, ok := existingPanes[id]; ok { + panes[id] = p + continue + } + if held != nil && held.ID == id { + continue // the worktree replace's pane: its leaf is the reservation + } + p := NewPaneModel(id, m.replayBufSize()) + p.resumeStart = time.Now() + if info := paneMap[id]; info == nil || !info.Pending { + p.preparing = true + } + panes[id] = p + created = append(created, id) + } + + // The stored tree names the pane a REPLACE reservation stands in for; + // a sentinel keeps that leaf through the prune so the reservation can + // take its place. + ph := m.pendingSplit[tab.ID] + var sentinel *PaneModel + if ph != nil && tab.reserveReplace && tab.reserveSibling != "" && panes[tab.reserveSibling] == nil { + sentinel = &PaneModel{ID: tab.reserveSibling} + panes[tab.reserveSibling] = sentinel + } + + root := DeserializeLayout(stored, panes) + if root != nil { + root.PrunePlaceholders() + if root.Pane == nil && root.Left == nil && root.Right == nil { + root = nil + } + } + tab.Root = root + tab.invalidateLeaves() + tab.clearAwaiting() + for _, id := range gone { + tab.awaitGone(id) + } + + if ph != nil { + m.reseatReservation(tab, ph, sentinel) + } + return created, send +} + +// reseatReservation puts this client's pendingSplit placeholder back into an +// adopted tree: beside its original sibling in the original direction (the +// right/bottom half, where SplitLeaf put it), or in place of the pane a +// replace stands in for; failing both, where the spiral would place an +// arrival. +func (m *Model) reseatReservation(tab *TabModel, old *LayoutNode, sentinel *PaneModel) { + var ph *LayoutNode + switch { + case sentinel != nil: + if leaf := tab.Root.FindLeaf(sentinel.ID); leaf != nil && leaf.Pane == sentinel { + leaf.Pane = nil + ph = leaf + } + case !tab.reserveReplace && tab.reserveSibling != "" && tab.Root.FindLeaf(tab.reserveSibling) != nil: + ph = tab.SplitAtPane(tab.reserveSibling, tab.reserveDir) + } + if ph == nil { + ph = tab.spiralSlot(m.paneAreaWidth(), m.height-chromeHeight) + } + if ph == nil { + // Nothing left to sit beside: the reservation is the whole tab. + ph = &LayoutNode{Ratio: 0.5} + tab.Root = ph + } + ph.phType = old.phType + tab.invalidateLeaves() + m.pendingSplit[tab.ID] = ph +} + +// resetLayoutSync forgets every tab's revision on dest, so the first broadcast +// after a reattach is adopted: the daemon's stored tree is the authority, and +// after a daemon restart its revision can be LOWER than ours (the snapshot is +// debounced), which the ordinary higher-wins rule would ignore forever. +func (m *Model) resetLayoutSync(dest string) { + for _, proj := range m.projects { + if proj.Dest != dest { + continue + } + for _, tab := range proj.tabs { + tab.layoutRev = 0 + tab.layoutDirty, tab.layoutSent, tab.layoutResend = false, nil, false + tab.clearAwaiting() + } + } +} diff --git a/internal/tui/model.go b/internal/tui/model.go index 4fa33534..e723a072 100644 --- a/internal/tui/model.go +++ b/internal/tui/model.go @@ -12,7 +12,6 @@ import ( "math/rand" "os" "path/filepath" - "reflect" "strconv" "strings" "time" @@ -107,6 +106,9 @@ type TabInfo struct { Layout json.RawMessage TemplateLayout string TemplateMain string + // LayoutRev is the daemon's revision of Layout: bumped on every stored + // write, 0 for a tab no client has described. See layoutsync.go. + LayoutRev uint64 } type PaneInfo struct { @@ -984,6 +986,11 @@ type Model struct { // drag. Rides clearDragState like every other drag. paneDrag paneDragState + // closeRequested holds the panes THIS client's user confirmed closing. + // The broadcast that prunes one is a user change this client stores; any + // other prune waits for whoever asked (layoutsync.go). + closeRequested map[string]bool + // Project-sidebar edge drag. sidebarDragging is set while a drag is in // flight; sidebarDragW is the PENDING width, painted as a preview rule and // committed to sidebarWidth only on release. @@ -2888,14 +2895,15 @@ func (m Model) Update(msg tea.Msg) (retModel tea.Model, retCmd tea.Cmd) { // used to cost one must-deliver frame per tab PLUS one per pane — 69 // frames on a 64-slot queue at 33 tabs/36 panes, which overflowed and // made the client's own IPC layer close the connection (2026-08-09). - // Both diffs run here, on the Update goroutine, because they read - // m.projects, which applyWorkspaceState has just rebuilt. + // The diff runs here, on the Update goroutine, because it reads + // m.projects, which applyWorkspaceState has just rebuilt. Layout + // writes are not diffed at all: rebuildTabs returns the few this + // client owes among overlayResizeCmds (layoutsync.go). cmds := []tea.Cmd{ templateFocusCmd, pickerVanishCmd, m.listenForMessages(), m.sendDiffedResizes(m.diffResizes(msg)), - m.sendDiffedLayouts(m.diffLayouts(msg)), groupsCmd, } // Resize overlay PTYs that just became visible on initial creation. @@ -3903,20 +3911,24 @@ func (m *Model) dragSplitBorder(x, y int) { // finishSplitDrag commits an in-progress border drag: the daemon gets the // final pane sizes (one PTY resize per pane — children reflow once, per -// the on-release-only design) and every tab's layout blob (persists the -// new Ratio). resizeAllPanes/sendAllLayouts cover all panes/tabs; the -// daemon's same-size guard drops the untouched panes' resizes, and layout -// updates are stored opaquely without broadcast, so the extra breadth is -// harmless and reuses tested plumbing. +// the on-release-only design) and the dragged tab's layout (persists the +// new Ratio). resizeAllPanes covers all panes; the daemon's same-size guard +// drops the untouched panes' resizes. Only the active tab's tree moved, so +// only it is stored (markLayoutChanged). func (m *Model) finishSplitDrag() tea.Cmd { // The one VT resize of the whole drag: old size → final size, paired // with the PTY resize below so the child's SIGWINCH redraw lands in a // matching grid (mid-drag only rects moved — see resizeNodeRects). - if tab := m.activeTabModel(); tab != nil { + tab := m.activeTabModel() + if tab != nil { tab.Resize(tab.Width, tab.Height) } m.clearDragState() - return tea.Batch(m.resizeAllPanes(), m.sendAllLayouts()) + var layout tea.Cmd + if tab != nil { + layout = m.markLayoutChanged(m.destOfTab(tab.ID), tab) + } + return tea.Batch(m.resizeAllPanes(), layout) } // moveTab repositions the active project's tab at `from` to ordinal `to`, @@ -6268,7 +6280,8 @@ func (m *Model) applyWorkspaceState(state WorkspaceStateMsg, dest string) ([]str // skipped rather than materialised as an empty tab. // // Returns the project's tabs, the pane IDs it created (the caller arms a -// spinner per ID) and the overlay resize commands the caller must batch. +// spinner per ID) and the commands the caller must batch: overlay resizes and +// the layout writes this client owes (layoutsync.go). func (m *Model) rebuildTabs(info ProjectInfo, state WorkspaceStateMsg, existingTabs map[string]*TabModel, existingPanes map[string]*PaneModel, paneMap map[string]*PaneInfo, dest string) ([]*TabModel, []string, []tea.Cmd) { var newPaneIDs []string var overlayResizeCmds []tea.Cmd @@ -6294,12 +6307,15 @@ func (m *Model) rebuildTabs(info ProjectInfo, state WorkspaceStateMsg, existingT } // Reuse existing tab if possible (preserves layout tree). tab, exists := existingTabs[tabInfo.ID] + restored := false if exists && tab.templateLayoutPending && len(tabInfo.Layout) > 0 { // Another client may have already saved the completed tree. tab = m.restoreTabLayout(tab, tabInfo, paneMap, existingPanes, dest) + restored = true } if !exists { tab = NewTabModel(tabInfo.ID, tabInfo.Name) + tab.layoutRev = tabInfo.LayoutRev // New tab that doesn't exist locally — try to restore layout from daemon. if len(tabInfo.Layout) > 0 { @@ -6342,6 +6358,19 @@ func (m *Model) rebuildTabs(info ProjectInfo, state WorkspaceStateMsg, existingT daemonPaneSet[pid] = true } + // Layout sync (layoutsync.go): adopt a newer stored tree, or settle + // the previous broadcast's local placements. A template tab still + // waiting for its panes has no tree worth either; it takes the + // broadcast's revision as the base of the write that builds it. + lp := layoutPass{oldTree: map[string]bool{}} + switch { + case tab.templateLayoutPending: + tab.layoutRev = tabInfo.LayoutRev + case exists && !restored: + lp = m.syncTabLayout(tab, tabInfo, daemonPaneSet, paneMap, existingPanes) + newPaneIDs = append(newPaneIDs, lp.created...) + } + // Prune panes the daemon removed. if tab.Root != nil { for id := range tab.Root.PaneIDs() { @@ -6356,6 +6385,14 @@ func (m *Model) rebuildTabs(info ProjectInfo, state WorkspaceStateMsg, existingT tab.ExitFocus() } tab.RemovePane(id) + // A close THIS client confirmed is its user's change to + // store; any other prune waits for the requester's write. + if m.closeRequested[id] { + delete(m.closeRequested, id) + lp.send = true + } else { + tab.awaitGone(id) + } } } } @@ -6427,8 +6464,13 @@ func (m *Model) rebuildTabs(info ProjectInfo, state WorkspaceStateMsg, existingT // skipped just before. Every client of this daemon holds every tab // of it, so every client sees the same reuse and places the pane // the same way. + // + // Except after ADOPTION: a pane this tab held that the adopted tree + // lacks (lp.oldTree) also reuses its model, and it did not move — + // it takes the ordinary arrival rule, and fills no reservation. pane, ok := existingPanes[paneID] - migrated := ok + migrated := ok && !lp.oldTree[paneID] + fresh := !ok info := paneMap[paneID] if !ok { pane = NewPaneModel(paneID, m.replayBufSize()) @@ -6470,11 +6512,14 @@ func (m *Model) rebuildTabs(info ProjectInfo, state WorkspaceStateMsg, existingT if migrated && m.pendingSplit[tab.ID] != nil { sparedReservation = true } - if m.pendingSplit != nil && !migrated { + if m.pendingSplit != nil && fresh { if placeholder, ok := m.pendingSplit[tab.ID]; ok { placeholder.fill(pane) tab.invalidateLeaves() delete(m.pendingSplit, tab.ID) + // The pane this client asked for: its user's split, so + // this client — and only this one — stores the tree. + lp.send = true // The pane LANDING is what retires a worktree create's // prune exemption — not the daemon saying it succeeded. // Dropping it on the success response instead would trust @@ -6558,9 +6603,16 @@ func (m *Model) rebuildTabs(info ProjectInfo, state WorkspaceStateMsg, existingT if migrated && tab != m.activeTabModel() { adoptMovedPane(tab, pane) } + // Placed by this client alone, for an arrival nobody here asked + // for: the requester stores it (see layoutsync.go). + tab.awaitPane(paneID) } - applyTemplateLayout(tab, tabInfo, paneMap) + if applyTemplateLayout(tab, tabInfo, paneMap) { + // The template's first real tree, built from the broadcast's + // revision (set above while the tab was pending). + lp.send = true + } // Clean up any unfilled placeholders (e.g., rapid double-splits). // @@ -6607,6 +6659,17 @@ func (m *Model) rebuildTabs(info ProjectInfo, state WorkspaceStateMsg, existingT m.finalizeTabPanes(tab) log.Printf("apply: tab %s finalized", tab.ID) + + // The daemon holds no usable tree for this tab (fresh, restored before + // any client described it, or unparseable): describe it, once — a + // dirty tab's write is already on its way. + if !tab.templateLayoutPending && tab.Root != nil && !tab.layoutDirty && + tabInfo.LayoutRev == tab.layoutRev && storedLayout(tabInfo.Layout) == nil { + lp.send = true + } + if lp.send { + overlayResizeCmds = append(overlayResizeCmds, m.markLayoutChanged(dest, tab)) + } out = append(out, tab) } @@ -6614,8 +6677,14 @@ func (m *Model) rebuildTabs(info ProjectInfo, state WorkspaceStateMsg, existingT } // restoreTabLayout rebuilds a tab's layout tree from serialized daemon state. +// The tree it builds IS the stored revision, so the tab holds that revision, +// clean; panes the stored tree lacks are placed and awaited like any other +// arrival nobody here asked for (layoutsync.go). func (m *Model) restoreTabLayout(tab *TabModel, tabInfo TabInfo, paneMap map[string]*PaneInfo, existingPanes map[string]*PaneModel, dest string) *TabModel { tab.templateLayoutApplied, tab.templateLayoutPending = true, false + tab.layoutRev = tabInfo.LayoutRev + tab.layoutDirty, tab.layoutSent, tab.layoutResend = false, nil, false + tab.clearAwaiting() log.Printf("restoreLayout: tab %s %q with %d panes", tab.ID, tabInfo.Name, len(tabInfo.Panes)) tab.Name = tabInfo.Name tab.Color = tabInfo.Color @@ -6671,6 +6740,7 @@ func (m *Model) restoreTabLayout(tab *TabModel, tabInfo TabInfo, paneMap map[str } else { splitForNewPane(tab, tab.Leaves(), pane) } + tab.awaitPane(paneID) } m.finalizeTabPanes(tab) @@ -8026,6 +8096,9 @@ func parseWorkspaceState(raw map[string]any) WorkspaceStateMsg { ti.Layout = data } } + if n, ok := tm["layout_rev"].(float64); ok && n >= 0 { + ti.LayoutRev = uint64(n) + } state.Tabs = append(state.Tabs, ti) } } @@ -8183,6 +8256,7 @@ func (m *Model) splitPane(dir SplitDir) tea.Cmd { m.pendingSplit = make(map[string]*LayoutNode) } m.pendingSplit[tab.ID] = placeholder + tab.noteReservation(pane.ID, dir, false) // The same node-local label the dialog path records. The payload below // carries no Type and the daemon normalises an empty one to terminal, so // this is what the pane will be rather than a guess about it. @@ -9550,14 +9624,15 @@ func (m Model) toggleActivePaneEager() tea.Cmd { } } -// layoutSend and resizeSend are one decided frame each. The diff that produces -// them runs on the Update goroutine and the command only ships the result — -// the walk reads m.projects, which Update rebuilds on every broadcast, so -// deciding inside the command would read a list that is being replaced. +// layoutSend and resizeSend are one decided frame each. The decision runs on +// the Update goroutine and the command only ships the result — it reads +// m.projects and the tab trees, which Update rebuilds on every broadcast, so +// deciding inside the command would read state that is being replaced. type layoutSend struct { - dest string - tabID string - data json.RawMessage + dest string + tabID string + data json.RawMessage + baseRev uint64 // the revision the tree was built on (layoutsync.go) } type resizeSend struct { @@ -9567,74 +9642,6 @@ type resizeSend struct { rows uint16 } -// layoutAgrees reports whether the daemon's stored layout for a tab already -// describes the tree we hold. -// -// The comparison is STRUCTURAL, and that is not a style preference. The daemon -// stores MarshalLayout's bytes — a struct, so Go emits its fields in -// declaration order — but parseWorkspaceState decodes the whole broadcast into -// map[string]any and re-marshals the layout sub-map, and Go sorts map keys -// alphabetically. For any node with more than one key the two encodings differ -// for the identical tree, so a byte comparison reports "changed" forever on -// every tab containing a split, while still matching single-leaf tabs. That -// asymmetry is invisible in a workspace of single-pane tabs, which is exactly -// what the crash was reported from. -// -// Empty means the daemon holds nothing for this tab (fresh, or restored before -// any client described it) — the caller must send, or the arrangement is never -// persisted. -func layoutAgrees(stored json.RawMessage, root *LayoutNode) bool { - if len(stored) == 0 { - return false - } - theirs, err := UnmarshalLayout(stored) - if err != nil { - return false - } - return reflect.DeepEqual(theirs, SerializeLayout(root)) -} - -// diffLayouts decides which tabs need their layout pushed after a broadcast. -// -// Scoped to the broadcast's OWN destination: a broadcast is the full state of -// one daemon, so it says nothing about another daemon's tabs and cannot be -// diffed against them. Before this scoping, any daemon's broadcast re-sent -// every daemon's layouts. -func (m *Model) diffLayouts(state WorkspaceStateMsg) []layoutSend { - stored := make(map[string]json.RawMessage, len(state.Tabs)) - for _, ti := range state.Tabs { - stored[ti.ID] = ti.Layout - } - // A broadcast whose dest matches no project means every layout and every - // resize below is silently skipped, and the failure has no other symptom: - // splits revert on restart, panes keep a stale PTY width, and nothing logs. - // The invariant holds structurally today — ProjectModel.Dest is only ever - // assigned from a broadcast's own dest — so this line exists to make a - // future break greppable rather than a multi-hour hunt. - if len(m.projects) > 0 && !m.hasProjectForDest(state.Dest) { - log.Printf("apply: broadcast dest %q matches no project — no layout or "+ - "resize will be sent for it", state.Dest) - } - - var out []layoutSend - for _, proj := range m.projects { - if proj.Dest != state.Dest { - continue - } - for _, tab := range proj.tabs { - if tab.templateLayoutPending || tab.Root == nil || layoutAgrees(stored[tab.ID], tab.Root) { - continue - } - data, err := MarshalLayout(tab.Root) - if err != nil { - continue - } - out = append(out, layoutSend{dest: proj.Dest, tabID: tab.ID, data: data}) - } - } - return out -} - // sizedKey scopes a sizedOnce entry to its owning destination. NUL separates // the halves because it cannot occur in either a dest or a pane id, so no pair // of distinct inputs can collide on one key. @@ -9667,6 +9674,16 @@ func (m *Model) hasProjectForDest(dest string) bool { // diffResizes decides which panes need a resize pushed after a broadcast. // See Model.sizedOnce for why the first send per pane is never suppressed. func (m *Model) diffResizes(state WorkspaceStateMsg) []resizeSend { + // A broadcast whose dest matches no project means every resize below is + // silently skipped, and the failure has no other symptom: panes keep a + // stale PTY width, and nothing logs. The invariant holds structurally + // today — ProjectModel.Dest is only ever assigned from a broadcast's own + // dest — so this line exists to make a future break greppable rather than + // a multi-hour hunt. + if len(m.projects) > 0 && !m.hasProjectForDest(state.Dest) { + log.Printf("apply: broadcast dest %q matches no project — no resize "+ + "will be sent for it", state.Dest) + } // Ahead of every sizedOnce write, so the first-resize kick each pane is // owed survives until the terminal is paintable again. See terminalPaintable. if !m.terminalPaintable() { @@ -9736,16 +9753,19 @@ func (m *Model) diffResizes(state WorkspaceStateMsg) []resizeSend { return out } -// sendDiffedLayouts ships an already-decided layout list. +// sendDiffedLayouts ships an already-decided layout list, each write carrying +// the revision it was built on so the daemon can refuse a stale one. func (m Model) sendDiffedLayouts(items []layoutSend) tea.Cmd { if len(items) == 0 { return nil } return func() tea.Msg { for _, it := range items { + base := it.baseRev msg, err := ipc.NewMessage(ipc.MsgUpdateLayout, ipc.UpdateLayoutPayload{ - TabID: it.tabID, - Layout: it.data, + TabID: it.tabID, + Layout: it.data, + BaseRev: &base, }) if err != nil { continue @@ -9783,27 +9803,3 @@ func (m Model) sendDiffedResizes(items []resizeSend) tea.Cmd { return nil } } - -// sendAllLayouts walks the projects for the same reason resizeAllPanes does — -// every tab's layout has to reach the daemon that owns that tab. -func (m Model) sendAllLayouts() tea.Cmd { - return func() tea.Msg { - for _, proj := range m.projects { - for _, tab := range proj.tabs { - if tab.templateLayoutPending || tab.Root == nil { - continue - } - data, err := MarshalLayout(tab.Root) - if err != nil { - continue - } - msg, _ := ipc.NewMessage(ipc.MsgUpdateLayout, ipc.UpdateLayoutPayload{ - TabID: tab.ID, - Layout: data, - }) - m.sendForDest(proj.Dest, msg) - } - } - return nil - } -} diff --git a/internal/tui/move_pane_apply_test.go b/internal/tui/move_pane_apply_test.go index dc0a777c..5d157b72 100644 --- a/internal/tui/move_pane_apply_test.go +++ b/internal/tui/move_pane_apply_test.go @@ -1,6 +1,7 @@ package tui import ( + "encoding/json" "reflect" "testing" "time" @@ -674,12 +675,28 @@ func TestNewPaneFromElsewhere_StillUsesLegacyPlacement(t *testing.T) { // Two clients with the same prior state and geometry place the moved pane // identically, and once one of them has pushed its tree the other agrees with // what the daemon stores — no layout thrash. +// +// Nobody on either client asked for the move's placement, so neither sends on +// the move itself; the next broadcast, whose stored trees still predate it, +// makes each send once (layoutsync.go). func TestMovedPane_TwoClientsConverge(t *testing.T) { t.Parallel() - before := mpState("tab-src", - mpTab{"tab-src", []string{"p1", "p2"}}, mpTab{"tab-tgt", []string{"p3", "p4"}}) - after := mpState("tab-src", - mpTab{"tab-src", []string{"p1"}}, mpTab{"tab-tgt", []string{"p3", "p4", "p2"}}) + stored := func(st WorkspaceStateMsg) WorkspaceStateMsg { + st.Tabs = append([]TabInfo(nil), st.Tabs...) + for i := range st.Tabs { + switch st.Tabs[i].ID { + case "tab-src": + st.Tabs[i].Layout = lsWire(t, lsSplit(SplitVertical, 0.5, lsLeaf("p1"), lsLeaf("p2"))) + case "tab-tgt": + st.Tabs[i].Layout = lsWire(t, lsSplit(SplitVertical, 0.5, lsLeaf("p3"), lsLeaf("p4"))) + } + } + return st + } + before := stored(mpState("tab-src", + mpTab{"tab-src", []string{"p1", "p2"}}, mpTab{"tab-tgt", []string{"p3", "p4"}})) + after := stored(mpState("tab-src", + mpTab{"tab-src", []string{"p1"}}, mpTab{"tab-tgt", []string{"p3", "p4", "p2"}})) a := newMovePaneModel(t, 120, 40) b := newMovePaneModel(t, 120, 40) @@ -692,10 +709,11 @@ func TestMovedPane_TwoClientsConverge(t *testing.T) { t.Fatalf("the two clients placed the moved pane differently:\nA=%+v\nB=%+v", treeA, treeB) } - var pushed *layoutSend - for _, s := range a.diffLayouts(after) { - if s.tabID == "tab-tgt" { - pushed = &s + a, sentA := lsApply(t, a, after) + var pushed json.RawMessage + for _, s := range sentA { + if s.TabID == "tab-tgt" { + pushed = s.Layout } } if pushed == nil { @@ -706,13 +724,17 @@ func TestMovedPane_TwoClientsConverge(t *testing.T) { third.Tabs = append([]TabInfo(nil), after.Tabs...) for i := range third.Tabs { if third.Tabs[i].ID == "tab-tgt" { - third.Tabs[i].Layout = pushed.data + third.Tabs[i].Layout = pushed + third.Tabs[i].LayoutRev = 1 } } - b = mpApply(t, b, third) - for _, s := range b.diffLayouts(third) { - if s.tabID == "tab-tgt" { + b, sentB := lsApply(t, b, third) + for _, s := range sentB { + if s.TabID == "tab-tgt" { t.Error("client B disagrees with the tree client A stored — the two would re-send each other's layout forever") } } + if got := SerializeLayout(mpTabOf(t, &b, "tab-tgt").Root); !reflect.DeepEqual(got, treeA) { + t.Errorf("client B holds %+v, want A's stored %+v", got, treeA) + } } diff --git a/internal/tui/project_merge_test.go b/internal/tui/project_merge_test.go index 1aaaa6c4..8cb0f5ef 100644 --- a/internal/tui/project_merge_test.go +++ b/internal/tui/project_merge_test.go @@ -146,7 +146,7 @@ func TestSendForDestStrict_ReportsAnUnreachableDest(t *testing.T) { // depend on it, which is why the strict one had to be added beside it rather // than changing Router.Send. if sendErr := m.sendForDest("gpu01", msg); sendErr != nil { - t.Errorf("sendForDest err = %v, want nil — resizeAllPanes and sendAllLayouts "+ + t.Errorf("sendForDest err = %v, want nil — resizeAllPanes and sendDiffedLayouts "+ "must not break mid-iteration", sendErr) } } diff --git a/internal/tui/projectdialog.go b/internal/tui/projectdialog.go index f00713bf..5c0a07c2 100644 --- a/internal/tui/projectdialog.go +++ b/internal/tui/projectdialog.go @@ -246,7 +246,7 @@ func (m *Model) submitNewProject(name, rootDir string) tea.Cmd { } // Reachability is checked HERE because the send cannot report it. Router.Send // drops a message aimed at a dest it has no connection for, logs, and returns - // NIL — deliberately, so resizeAllPanes and sendAllLayouts cannot break + // NIL — deliberately, so resizeAllPanes and sendDiffedLayouts cannot break // mid-iteration and leave other daemons unsynced. Every `if err := send(…)` // below is therefore blind to the likeliest failure of all: the host going // away while this dialog is open. diff --git a/internal/tui/reconnect.go b/internal/tui/reconnect.go index 7b8e2c57..71003017 100644 --- a/internal/tui/reconnect.go +++ b/internal/tui/reconnect.go @@ -933,6 +933,10 @@ func (m *Model) armReattachReset(dest string) { // delete on a nil map is a no-op. delete(m.sizedOnce, sizedKey(dest, p.ID)) }) + // And every tab's layout revision: the daemon's stored tree is the + // authority after a reattach, and a restarted daemon's revision can be + // LOWER than ours (resetLayoutSync). + m.resetLayoutSync(dest) // Selection is Model-level and anchors to row/column coordinates that any // replay invalidates. Dropped now rather than armed: there is no per-pane // chunk to hang it off, and a selection surviving an outage is worth nothing. diff --git a/internal/tui/router.go b/internal/tui/router.go index 8ae8ad6d..b102b1bf 100644 --- a/internal/tui/router.go +++ b/internal/tui/router.go @@ -303,7 +303,7 @@ func (r *Router) Send(m *ipc.Message) error { if !ok { // Drop with a log. Returning an error would break resizeAllPanes and - // sendAllLayouts mid-iteration and leave other daemons unsynced. + // sendDiffedLayouts mid-iteration and leave other daemons unsynced. log.Printf("router: dropping %s for unreachable dest %q", m.Type, dest) return nil } @@ -455,7 +455,7 @@ var ErrDestUnreachable = errors.New("no connection for that destination") // // Router.Send DROPS a message for a dest it has no conn for, logs, and returns // nil. That is right for the bulk iterators it was written for — resizeAllPanes -// and sendAllLayouts must not break mid-iteration and leave other daemons +// and sendDiffedLayouts must not break mid-iteration and leave other daemons // unsynced — and wrong for an action a user confirmed, which would otherwise be // reported as done. // diff --git a/internal/tui/router_test.go b/internal/tui/router_test.go index 394b104d..d8bb0ea0 100644 --- a/internal/tui/router_test.go +++ b/internal/tui/router_test.go @@ -224,7 +224,7 @@ func TestPaneInputForALocalPaneWhileARemoteProjectIsActive(t *testing.T) { } } -// The same hole on the broadcast paths: resizeAllPanes, sendAllLayouts and +// The same hole on the broadcast paths: resizeAllPanes, sendDiffedLayouts and // overlayResizeCmd all stamp a project's or tab's dest, which is "" for local. func TestSendForDestLocalIsHonouredWhileARemoteProjectIsActive(t *testing.T) { local, gpu := newFakeConn(), newFakeConn() diff --git a/internal/tui/splitdrag_test.go b/internal/tui/splitdrag_test.go index 0badf98b..c6147849 100644 --- a/internal/tui/splitdrag_test.go +++ b/internal/tui/splitdrag_test.go @@ -171,7 +171,7 @@ func TestModel_FinishSplitDrag_CommitsToDaemon(t *testing.T) { if cmd == nil { t.Fatal("finishSplitDrag must return the commit command") } - // Execute the batch: resizeAllPanes + sendAllLayouts. + // Execute the batch: resizeAllPanes + markLayoutChanged. if batch, ok := cmd().(tea.BatchMsg); ok { for _, c := range batch { if c != nil { diff --git a/internal/tui/tab.go b/internal/tui/tab.go index 3ef93c0e..245285fb 100644 --- a/internal/tui/tab.go +++ b/internal/tui/tab.go @@ -56,8 +56,56 @@ type TabModel struct { // leavesCache memoizes Root.Leaves(); nil = invalid. The tab bar alone // walks every tab's tree twice per render without it. leavesCache []*PaneModel + + // Layout sync between clients (layoutsync.go). layoutRev is the daemon's + // revision of the tree this tab holds. layoutDirty means a write built on + // that revision is in flight; layoutSent is what it carried, so its echo + // is recognised, and layoutResend is a local change made while it was in + // flight, sent once the echo lands. + layoutRev uint64 + layoutDirty bool + layoutSent *SerializedNode + layoutResend bool + // awaitingPanes / awaitingGone are ids this client placed or pruned on + // its own, for a change nobody here asked for (an MCP create, another + // client's split or close, a moved pane). The next broadcast says whether + // the requester already stored a tree covering them; only when it did not + // does this client send. + awaitingPanes map[string]bool + awaitingGone map[string]bool + // reserveSibling/reserveDir describe this client's pendingSplit + // reservation when it was made — the pane it was split from and the + // direction, or (reserveReplace) the pane it stands in for — so adopting + // another client's tree can put the placeholder back where it was. + reserveSibling string + reserveDir SplitDir + reserveReplace bool +} + +// noteReservation records where this client's pendingSplit placeholder sits. +// Called at every site that arms one. +func (t *TabModel) noteReservation(sibling string, dir SplitDir, replace bool) { + t.reserveSibling, t.reserveDir, t.reserveReplace = sibling, dir, replace +} + +// awaitPane and awaitGone record a local placement or prune nobody on this +// client asked for (see awaitingPanes). +func (t *TabModel) awaitPane(id string) { + if t.awaitingPanes == nil { + t.awaitingPanes = make(map[string]bool) + } + t.awaitingPanes[id] = true +} + +func (t *TabModel) awaitGone(id string) { + if t.awaitingGone == nil { + t.awaitingGone = make(map[string]bool) + } + t.awaitingGone[id] = true } +func (t *TabModel) clearAwaiting() { t.awaitingPanes, t.awaitingGone = nil, nil } + func NewTabModel(id, name string) *TabModel { return &TabModel{ ID: id, @@ -460,12 +508,25 @@ func (t *TabModel) SplitAtPane(paneID string, dir SplitDir) *LayoutNode { // when the tree is nil or holds no pane leaf. The caller guarantees pane is // not already in the tree. func (t *TabModel) placeArrivingPane(pane *PaneModel, w, h int) bool { - if t.Root == nil { + ph := t.spiralSlot(w, h) + if ph == nil { return false } + ph.fill(pane) + t.invalidateLeaves() + return true +} + +// spiralSlot is placeArrivingPane without the fill: it splits the spiral leaf +// and returns the empty placeholder, or nil when the tree holds no pane leaf. +// Adoption uses it to re-seat a reservation whose sibling is gone. +func (t *TabModel) spiralSlot(w, h int) *LayoutNode { + if t.Root == nil { + return nil + } leaf, parentSplit, hasParent := t.Root.spiralLeaf() if leaf == nil { - return false + return nil } var rects []PaneRect @@ -479,13 +540,7 @@ func (t *TabModel) placeArrivingPane(pane *PaneModel, w, h int) bool { } dir := arrivalSplitDir(spiralSplitDir(parentSplit, hasParent), rectW, rectH) - ph := t.SplitAtPane(leaf.Pane.ID, dir) - if ph == nil { - return false - } - ph.fill(pane) - t.invalidateLeaves() - return true + return t.SplitAtPane(leaf.Pane.ID, dir) } // RemovePane removes the pane with the given ID, promoting its sibling. diff --git a/internal/tui/template_layout.go b/internal/tui/template_layout.go index 50e78f6f..df955aa6 100644 --- a/internal/tui/template_layout.go +++ b/internal/tui/template_layout.go @@ -53,20 +53,22 @@ func templateStack(panes []*PaneModel, dir SplitDir) *LayoutNode { return &LayoutNode{Split: dir, Ratio: 1 / float64(len(panes)), Left: NewLeaf(panes[0]), Right: templateStack(panes[1:], dir)} } -func applyTemplateLayout(tab *TabModel, info TabInfo, paneMap map[string]*PaneInfo) { +// applyTemplateLayout reports whether it built the tree — the one moment a +// template tab's layout is first worth storing. +func applyTemplateLayout(tab *TabModel, info TabInfo, paneMap map[string]*PaneInfo) bool { if !tab.templateLayoutPending { - return + return false } panes := make([]*PaneModel, 0, len(info.Panes)) main := -1 for i, id := range info.Panes { meta := paneMap[id] if meta == nil || meta.TabID != tab.ID || meta.PreparingWorktree != "" || meta.Overlay { - return + return false } leaf := tab.Root.FindLeaf(id) if leaf == nil || leaf.Pane == nil { - return + return false } panes = append(panes, leaf.Pane) if id == info.TemplateMain { @@ -76,9 +78,10 @@ func applyTemplateLayout(tab *TabModel, info TabInfo, paneMap map[string]*PaneIn // The preparing and swap frames lack a final anchor. Do not consume the // keyword or publish a layout containing their soon-to-be-replaced pane. if main < 0 { - return + return false } tab.Root = templateLayout(info.TemplateLayout, panes, main) tab.invalidateLeaves() tab.templateLayoutApplied, tab.templateLayoutPending = true, false + return true } diff --git a/internal/tui/template_layout_test.go b/internal/tui/template_layout_test.go index c617150b..3a4d20c4 100644 --- a/internal/tui/template_layout_test.go +++ b/internal/tui/template_layout_test.go @@ -98,8 +98,8 @@ func templateWorkspace(ids []string, main string) WorkspaceStateMsg { func TestTemplateWorkspace_PreparingSwapAndCompletion_BuildsAndReportsOnlyCompletedTree(t *testing.T) { t.Setenv("QUIL_HOME", t.TempDir()) m := paletteModelWithProjects(t) - // Count this workspace's sends only; sendAllLayouts also legitimately - // saves unrelated remote tabs in the shared palette fixture. + // Count this workspace's sends only; the fixture's other tabs have no + // stored layout and would legitimately describe theirs. m.projects = m.projects[:1] recorder := &echoRecorder{} m.client = recorder @@ -111,7 +111,7 @@ func TestTemplateWorkspace_PreparingSwapAndCompletion_BuildsAndReportsOnlyComple preparing := templateWorkspace([]string{"placeholder"}, "placeholder") preparing.Panes[0].PreparingWorktree = "feat/template" update(preparing) - runCmd(m.sendAllLayouts()) // Even an unrelated resize/action must not persist preparation. + runCmd(m.resizeAllPanes()) // Even an unrelated resize/action must not persist preparation. if layouts, _ := sentCounts(t, recorder); layouts != 0 { t.Fatal("saved preparing placeholder") } @@ -143,7 +143,7 @@ func TestTemplateWorkspace_PreparingSwapAndCompletion_BuildsAndReportsOnlyComple } } } - if saved.TabID != tab.ID || !layoutAgrees(saved.Layout, root) { + if saved.TabID != tab.ID || !sentLayoutIs(saved.Layout, root) { t.Fatal("reported wrong tree", saved) } complete.Tabs[0].Layout = saved.Layout From 73573d32281639277f28395dd1e6ed5a762b543d Mon Sep 17 00:00:00 2001 From: Artjoms Stukans Date: Fri, 25 Sep 2026 23:17:35 +0200 Subject: [PATCH 17/40] fix(tui): keep a re-seated reservation through its adopting pass Adopting another client's tree re-seated this client's pendingSplit placeholder, but the same rebuild pass then pruned it as an unfilled placeholder: pendingSplit pointed at a detached node, the requested pane filled it invisibly and its model leaked. Adoption now reports the re-seat and the pass spares the placeholder, as it does for a moved pane. Also: - the first broadcast after a reattach is adopted whatever its revision (adoptNext), so a rev-0 stored tree is not ignored; - a border click released without motion stores nothing; - close requests are keyed by destination and dropped on reattach; - tests for the replace and spiral-fallback re-seats, and release tests that install their recorder before Update. --- internal/tui/broadcast_echo_test.go | 10 +- internal/tui/dialog.go | 2 +- internal/tui/layout_sync_test.go | 167 ++++++++++++++++++++++++++-- internal/tui/layoutsync.go | 63 ++++++++--- internal/tui/model.go | 29 +++-- internal/tui/tab.go | 3 + 6 files changed, 234 insertions(+), 40 deletions(-) diff --git a/internal/tui/broadcast_echo_test.go b/internal/tui/broadcast_echo_test.go index d870dbc1..335e7210 100644 --- a/internal/tui/broadcast_echo_test.go +++ b/internal/tui/broadcast_echo_test.go @@ -72,11 +72,11 @@ func TestBroadcastLayoutBytesDifferForASplit(t *testing.T) { } } -// A layout the daemon stores but the client cannot parse must be re-sent, not -// treated as agreeing — otherwise a corrupt stored tree is never corrected. It -// costs one frame per broadcast for that tab until the daemon accepts the -// replacement, which is bounded by the tab count and self-healing. -func TestLayoutAgrees_MalformedStoredLayoutResends(t *testing.T) { +// A layout the daemon stores but the client cannot parse counts as no layout +// at all: the client describes the tab once, with the current base revision, +// rather than keeping a tree nobody can restore. Treating it as agreeing +// would leave a corrupt stored tree uncorrected forever. +func TestWorkspaceState_MalformedStoredLayout_IsReplacedOnce(t *testing.T) { t.Parallel() m, echo := echoModel(t) echo.Tabs[0].Layout = json.RawMessage(`{"split": "not-a-direction"`) diff --git a/internal/tui/dialog.go b/internal/tui/dialog.go index 583d1e09..942a764f 100644 --- a/internal/tui/dialog.go +++ b/internal/tui/dialog.go @@ -1362,7 +1362,7 @@ func (m Model) handleConfirmKey(msg tea.KeyPressMsg) (tea.Model, tea.Cmd) { if m.closeRequested == nil { m.closeRequested = make(map[string]bool) } - m.closeRequested[id] = true + m.closeRequested[closeKey(dest, id)] = true case "tab": dest = m.destOfTab(id) } diff --git a/internal/tui/layout_sync_test.go b/internal/tui/layout_sync_test.go index ebd6c44f..ce34591e 100644 --- a/internal/tui/layout_sync_test.go +++ b/internal/tui/layout_sync_test.go @@ -304,10 +304,12 @@ func TestLayoutSync_AdoptionCancelsDrag(t *testing.T) { if m.splitDragNode != nil { t.Error("the border drag survived adoption of another tree") } - next, cmd := m.Update(tea.MouseReleaseMsg{X: 90, Y: 10, Button: tea.MouseLeft}) - m = next.(Model) + // The recorder goes in BEFORE Update: the command captures the + // client it was built with. rec := &echoRecorder{} m.client = rec + next, cmd := m.Update(tea.MouseReleaseMsg{X: 90, Y: 10, Button: tea.MouseLeft}) + m = next.(Model) runCmd(cmd) if sent := lsLayouts(t, rec); len(sent) != 0 { t.Errorf("the release after adoption sent %d layouts, want 0", len(sent)) @@ -338,21 +340,21 @@ func TestLayoutSync_PendingSplitSurvivesBesideSibling(t *testing.T) { m, _ = lsApply(t, m, lsState(1, lsWire(t, lsSplit(SplitHorizontal, 0.5, lsLeaf("p1"), lsLeaf("p2"))), "p1", "p2")) tab := lsTab(t, &m) tab.ActivePane = "p2" + // An ORDINARY split (Alt+Shift+V): no worktree create holds the + // placeholder open, so only the adoption's own spare keeps it. _ = m.splitPane(SplitVertical) if m.pendingSplit["t1"] == nil { t.Fatal("setup: the split armed no reservation") } - // Held open the way a worktree create holds it, so a broadcast that - // carries no new pane does not prune it. - m.worktreeCreates = map[string]string{"t1": "feat/x"} - // Another client swapped the two panes. - m, _ = lsApply(t, m, lsState(2, lsWire(t, lsSplit(SplitHorizontal, 0.5, lsLeaf("p2"), lsLeaf("p1"))), "p1", "p2")) + // Another client swapped the two panes; the requested pane has not come. + swapped := lsSplit(SplitHorizontal, 0.5, lsLeaf("p2"), lsLeaf("p1")) + m, _ = lsApply(t, m, lsState(2, lsWire(t, swapped), "p1", "p2")) tab = lsTab(t, &m) ph := m.pendingSplit["t1"] if ph == nil || !treeHoldsNode(tab.Root, ph) { - t.Fatal("the reservation was lost on adoption — the requested pane would land nowhere") + t.Fatal("the reservation was pruned in the adopting pass — the requested pane would land nowhere") } parent, isLeft := lsParentOf(tab.Root, ph) if parent == nil || isLeft || parent.Split != SplitVertical || @@ -360,14 +362,86 @@ func TestLayoutSync_PendingSplitSurvivesBesideSibling(t *testing.T) { t.Fatalf("reservation is not below its sibling p2: tree %s", layoutString(SerializeLayout(tab.Root))) } - // The requested pane lands in it, and this client — the requester — sends. - m, sent := lsApply(t, m, lsState(2, lsWire(t, lsSplit(SplitHorizontal, 0.5, lsLeaf("p2"), lsLeaf("p1"))), "p1", "p2", "p-new")) + // The requested pane lands in it, visibly, and this client — the + // requester — sends. + m, sent := lsApply(t, m, lsState(2, lsWire(t, swapped), "p1", "p2", "p-new")) + tab = lsTab(t, &m) + if ph.Pane == nil || ph.Pane.ID != "p-new" || !treeHoldsNode(tab.Root, ph) { + t.Fatalf("the requested pane did not fill the reservation in the tree (it holds %v)", ph.Pane) + } + if tab.ActivePane != "p-new" { + t.Errorf("ActivePane = %q, want p-new", tab.ActivePane) + } want := lsSplit(SplitHorizontal, 0.5, lsSplit(SplitVertical, 0.5, lsLeaf("p2"), lsLeaf("p-new")), lsLeaf("p1")) if got := lsTree(t, &m); !reflect.DeepEqual(got, want) { t.Errorf("tree = %s, want %s", layoutString(got), layoutString(want)) } if len(sent) != 1 || lsBase(sent[0]) != "2" { - t.Errorf("sent %d frames, want 1 with base 2", len(sent)) + t.Fatalf("sent %d frames, want 1 with base 2", len(sent)) + } + if got := lsSentTree(t, sent[0]); !reflect.DeepEqual(got, want) { + t.Errorf("sent %s, want %s", layoutString(got), layoutString(want)) + } +} + +// The sibling is gone from the adopted tree: the reservation falls back to +// where the spiral would place an arrival. +func TestLayoutSync_PendingSplitWithoutSiblingTakesTheSpiralSlot(t *testing.T) { + t.Setenv("QUIL_HOME", t.TempDir()) + m := newLayoutSyncModel(t) + m, _ = lsApply(t, m, lsState(1, lsWire(t, lsSplit(SplitHorizontal, 0.5, lsLeaf("p1"), lsLeaf("p2"))), "p1", "p2")) + lsTab(t, &m).ActivePane = "p2" + _ = m.splitPane(SplitVertical) + + // Another client closed p2, the pane this client split. + m, _ = lsApply(t, m, lsState(2, lsWire(t, lsLeaf("p1")), "p1")) + + tab := lsTab(t, &m) + ph := m.pendingSplit["t1"] + if ph == nil || !treeHoldsNode(tab.Root, ph) || tab.Root.Right != ph || + tab.Root.Split != SplitHorizontal || tab.Root.Left.Pane == nil || tab.Root.Left.Pane.ID != "p1" { + t.Fatalf("tree = %s, want p1 left|right beside the reservation", layoutString(SerializeLayout(tab.Root))) + } + + m, _ = lsApply(t, m, lsState(2, lsWire(t, lsLeaf("p1")), "p1", "p-new")) + want := lsSplit(SplitHorizontal, 0.5, lsLeaf("p1"), lsLeaf("p-new")) + if got := lsTree(t, &m); !reflect.DeepEqual(got, want) { + t.Errorf("tree = %s, want %s", layoutString(got), layoutString(want)) + } +} + +// A worktree REPLACE's reservation is the leaf of the pane it stands in for. +// Adopting a tree that moved that pane puts the reservation where the pane now +// is, and the replacing pane lands there. +func TestLayoutSync_ReplaceReservationFollowsItsPane(t *testing.T) { + m := newLayoutSyncModel(t) + m, _ = lsApply(t, m, lsState(1, lsWire(t, lsSplit(SplitHorizontal, 0.5, lsLeaf("p1"), lsLeaf("p2"))), "p1", "p2")) + lsTab(t, &m).ActivePane = "p2" + m = armOwnWorktreeCreate(t, m, 2) // sets QUIL_HOME + if held := m.worktreeReplaced["t1"]; held == nil || held.ID != "p2" { + t.Fatal("setup: the replace did not hold p2") + } + + // Another client stacked the panes, p2 on top; the daemon still has p2. + m, _ = lsApply(t, m, lsState(2, lsWire(t, lsSplit(SplitVertical, 0.5, lsLeaf("p2"), lsLeaf("p1"))), "p1", "p2")) + tab := lsTab(t, &m) + ph := m.pendingSplit["t1"] + if ph == nil || tab.Root.Left != ph || ph.Pane != nil || tab.Root.Split != SplitVertical { + t.Fatalf("tree = %s, want the reservation on top where p2 now is", layoutString(SerializeLayout(tab.Root))) + } + // What this client would store still names the live pane there. + if got, want := m.layoutForSend(tab), lsSplit(SplitVertical, 0.5, lsLeaf("p2"), lsLeaf("p1")); !reflect.DeepEqual(got, want) { + t.Errorf("layoutForSend = %s, want %s", layoutString(got), layoutString(want)) + } + + // The swap lands: p2 is gone, p-wt fills the reservation. + m, sent := lsApply(t, m, lsState(2, lsWire(t, lsSplit(SplitVertical, 0.5, lsLeaf("p2"), lsLeaf("p1"))), "p-wt", "p1")) + want := lsSplit(SplitVertical, 0.5, lsLeaf("p-wt"), lsLeaf("p1")) + if got := lsTree(t, &m); !reflect.DeepEqual(got, want) { + t.Errorf("tree = %s, want %s", layoutString(got), layoutString(want)) + } + if len(sent) != 1 || lsBase(sent[0]) != "2" { + t.Fatalf("sent %d frames, want 1 with base 2", len(sent)) } } @@ -463,6 +537,54 @@ func TestLayoutSync_ReattachAdoptsLowerRev(t *testing.T) { } } +// A daemon restored from an older workspace.json (or one that lost every +// debounced write) reports revision 0 with a stored tree. The first broadcast +// after a reattach is still adopted — even a zeroed revision would call that +// "equal" and keep the local tree. +func TestLayoutSync_ReattachAdoptsRevZeroStoredTree(t *testing.T) { + t.Setenv("QUIL_HOME", t.TempDir()) + m := newLayoutSyncModel(t) + m, _ = lsApply(t, m, lsState(5, lsWire(t, lsSplit(SplitHorizontal, 0.5, lsLeaf("p1"), lsLeaf("p2"))), "p1", "p2")) + + m.armReattachReset("") + + restored := lsSplit(SplitVertical, 0.5, lsLeaf("p2"), lsLeaf("p1")) + m, sent := lsApply(t, m, lsState(0, lsWire(t, restored), "p1", "p2")) + if got := lsTree(t, &m); !reflect.DeepEqual(got, restored) { + t.Errorf("tree = %s, want the daemon's %s adopted after the reattach", layoutString(got), layoutString(restored)) + } + if len(sent) != 0 { + t.Errorf("sent %d frames, want 0", len(sent)) + } + + // Only the FIRST broadcast: after it, rev 0 is an ordinary equal rev. + m, _ = lsApply(t, m, lsState(0, lsWire(t, lsSplit(SplitHorizontal, 0.5, lsLeaf("p1"), lsLeaf("p2"))), "p1", "p2")) + if got := lsTree(t, &m); !reflect.DeepEqual(got, restored) { + t.Errorf("a second rev-0 broadcast replaced the tree with %s — only the first after a reattach is adopted", layoutString(got)) + } +} + +// A press on a border and a release without motion changed nothing, and +// storing it would bump the revision under every other client. +func TestLayoutSync_BorderClickWithoutMotionSendsNothing(t *testing.T) { + t.Setenv("QUIL_HOME", t.TempDir()) + m := newLayoutSyncModel(t) + m, _ = lsApply(t, m, lsState(1, lsWire(t, lsSplit(SplitHorizontal, 0.5, lsLeaf("p1"), lsLeaf("p2"))), "p1", "p2")) + next, _ := m.Update(tea.MouseClickMsg{X: 60, Y: 10, Button: tea.MouseLeft}) + m = next.(Model) + if m.splitDragNode == nil { + t.Fatal("setup: the press did not arm a border drag") + } + rec := &echoRecorder{} + m.client = rec // before Update: the command captures this client + next, cmd := m.Update(tea.MouseReleaseMsg{X: 60, Y: 10, Button: tea.MouseLeft}) + m = next.(Model) + runCmd(cmd) + if sent := lsLayouts(t, rec); len(sent) != 0 { + t.Errorf("a release that moved nothing sent %d layouts, want 0", len(sent)) + } +} + func TestLayoutSync_EmptyStoredLayoutSendsOnce(t *testing.T) { t.Setenv("QUIL_HOME", t.TempDir()) m := newLayoutSyncModel(t) @@ -512,3 +634,26 @@ func TestLayoutSync_OwnCloseSendsBystanderWaits(t *testing.T) { t.Errorf("sent %s, want p1", layoutString(got)) } } + +// A close request is scoped to the daemon it went to, and a reattach to that +// daemon forgets it: the answering daemon may be a restarted one that never +// saw the request. +func TestLayoutSync_ReattachForgetsCloseRequests(t *testing.T) { + t.Setenv("QUIL_HOME", t.TempDir()) + m := newLayoutSyncModel(t) + m, _ = lsApply(t, m, lsState(2, lsWire(t, lsSplit(SplitHorizontal, 0.5, lsLeaf("p1"), lsLeaf("p2"))), "p1", "p2")) + lsTab(t, &m).ActivePane = "p2" + next, _ := m.openClosePaneConfirm() + m = next.(Model) + next, cmd := m.Update(tea.KeyPressMsg{Code: 'y', Text: "y"}) + m = next.(Model) + runCmd(cmd) + if !m.closeRequested[closeKey("", "p2")] { + t.Fatal("setup: the confirmed close was not recorded under its destination") + } + + m.armReattachReset("") + if len(m.closeRequested) != 0 { + t.Errorf("closeRequested = %v after the reattach, want empty", m.closeRequested) + } +} diff --git a/internal/tui/layoutsync.go b/internal/tui/layoutsync.go index e596f797..4dd751a4 100644 --- a/internal/tui/layoutsync.go +++ b/internal/tui/layoutsync.go @@ -4,6 +4,7 @@ import ( "encoding/json" "log" "reflect" + "strings" "time" tea "charm.land/bubbletea/v2" @@ -130,11 +131,31 @@ type layoutPass struct { oldTree map[string]bool // created: pane ids adoption built a new PaneModel for. created []string + // reseated: adoption put this client's reservation back into the new + // tree. The caller must spare it from this pass's placeholder prune, or + // pendingSplit is left pointing at a detached node and the requested pane + // lands in it invisibly. + reseated bool +} + +// closeKey scopes a closeRequested entry to its destination, as sizedKey does +// for sizedOnce: pane ids are per daemon. +func closeKey(dest, paneID string) string { return sizedKey(dest, paneID) } + +// takeCloseRequest reports whether THIS client's user asked to close paneID +// on dest, consuming the request. +func (m *Model) takeCloseRequest(dest, paneID string) bool { + k := closeKey(dest, paneID) + if !m.closeRequested[k] { + return false + } + delete(m.closeRequested, k) + return true } // syncTabLayout runs for an EXISTING tab before its panes are reconciled, and // decides what the broadcast's revision means for the tree it holds. -func (m *Model) syncTabLayout(tab *TabModel, ti TabInfo, paneSet map[string]bool, paneMap map[string]*PaneInfo, existingPanes map[string]*PaneModel) layoutPass { +func (m *Model) syncTabLayout(tab *TabModel, ti TabInfo, paneSet map[string]bool, paneMap map[string]*PaneInfo, existingPanes map[string]*PaneModel, dest string) layoutPass { lp := layoutPass{oldTree: map[string]bool{}} if tab.Root != nil { lp.oldTree = tab.Root.PaneIDs() @@ -146,15 +167,20 @@ func (m *Model) syncTabLayout(tab *TabModel, ti TabInfo, paneSet map[string]bool // parseWorkspaceState re-marshals the layout from map[string]any, which // sorts keys — so the same split tree arrives as different bytes. switch { - case ti.LayoutRev > tab.layoutRev: + case ti.LayoutRev > tab.layoutRev || tab.adoptNext: + // Higher revision, or the first broadcast after a reattach, whose + // revision can be anything — even the 0 of a restored workspace that + // no client has written since (resetLayoutSync). + // // Our own write coming back: the tree we hold already contains it, // plus anything done since, which the deferred resend now carries. echo := tab.layoutDirty && stored != nil && reflect.DeepEqual(stored, tab.layoutSent) resend := echo && tab.layoutResend tab.layoutRev = ti.LayoutRev + tab.adoptNext = false tab.layoutDirty, tab.layoutSent, tab.layoutResend = false, nil, false if stored != nil && !echo && !reflect.DeepEqual(stored, m.layoutForSend(tab)) { - lp.created, lp.send = m.adoptTabLayout(tab, stored, paneSet, paneMap, existingPanes) + lp.created, lp.send, lp.reseated = m.adoptTabLayout(tab, stored, paneSet, paneMap, existingPanes, dest) return lp } lp.send = resend @@ -188,10 +214,11 @@ func (m *Model) syncTabLayout(tab *TabModel, ti TabInfo, paneSet map[string]bool // adoptTabLayout replaces tab's tree with the stored one. Pane models are // reused by id, so no emulator or scrollback is lost; ids the broadcast no // longer lists are dropped, and panes the stored tree lacks are left for the -// caller's arrival loop to place. Returns the ids it built new models for, and +// caller's arrival loop to place. Returns the ids it built new models for, // whether a pane THIS client closed was still in the stored tree (a user -// change the requester must store). -func (m *Model) adoptTabLayout(tab *TabModel, stored *SerializedNode, paneSet map[string]bool, paneMap map[string]*PaneInfo, existingPanes map[string]*PaneModel) (created []string, send bool) { +// change the requester must store), and whether it re-seated this client's +// reservation (which the caller must spare from this pass's prune). +func (m *Model) adoptTabLayout(tab *TabModel, stored *SerializedNode, paneSet map[string]bool, paneMap map[string]*PaneInfo, existingPanes map[string]*PaneModel, dest string) (created []string, send, reseated bool) { // A drag armed on this tab describes a tree that is about to go. if (m.splitDragNode != nil && treeContains(tab.Root, m.splitDragNode)) || (m.paneDrag.active() && m.paneDrag.srcTabID == tab.ID) { @@ -204,8 +231,7 @@ func (m *Model) adoptTabLayout(tab *TabModel, stored *SerializedNode, paneSet ma for id := range serializedIDs(stored) { if !paneSet[id] { // Gone from the daemon but still in the stored tree. - if m.closeRequested[id] { - delete(m.closeRequested, id) + if m.takeCloseRequest(dest, id) { send = true } else { gone = append(gone, id) @@ -254,8 +280,9 @@ func (m *Model) adoptTabLayout(tab *TabModel, stored *SerializedNode, paneSet ma if ph != nil { m.reseatReservation(tab, ph, sentinel) + reseated = true } - return created, send + return created, send, reseated } // reseatReservation puts this client's pendingSplit placeholder back into an @@ -287,10 +314,13 @@ func (m *Model) reseatReservation(tab *TabModel, old *LayoutNode, sentinel *Pane m.pendingSplit[tab.ID] = ph } -// resetLayoutSync forgets every tab's revision on dest, so the first broadcast -// after a reattach is adopted: the daemon's stored tree is the authority, and -// after a daemon restart its revision can be LOWER than ours (the snapshot is -// debounced), which the ordinary higher-wins rule would ignore forever. +// resetLayoutSync forgets every tab's revision on dest and marks the first +// broadcast after a reattach for adoption whatever its revision: the daemon's +// stored tree is the authority, and after a daemon restart its revision can be +// LOWER than ours (the snapshot is debounced) — as low as the 0 of a tree no +// client has written since restore, which even a zeroed revision would not +// adopt. Pending close requests for dest are dropped too: the daemon that +// answers may not be the one they were sent to. func (m *Model) resetLayoutSync(dest string) { for _, proj := range m.projects { if proj.Dest != dest { @@ -298,8 +328,15 @@ func (m *Model) resetLayoutSync(dest string) { } for _, tab := range proj.tabs { tab.layoutRev = 0 + tab.adoptNext = true tab.layoutDirty, tab.layoutSent, tab.layoutResend = false, nil, false tab.clearAwaiting() } } + prefix := dest + "\x00" + for k := range m.closeRequested { + if strings.HasPrefix(k, prefix) { + delete(m.closeRequested, k) + } + } } diff --git a/internal/tui/model.go b/internal/tui/model.go index e723a072..75ed13fd 100644 --- a/internal/tui/model.go +++ b/internal/tui/model.go @@ -981,6 +981,9 @@ type Model struct { // (finishSplitDrag) — mid-drag only the local tree and VT change. splitDragNode *LayoutNode splitDragRect BorderHit + // splitDragRatio is the node's Ratio when the drag armed, so a release + // that moved nothing stores nothing (finishSplitDrag). + splitDragRatio float64 // paneDrag is an Alt+drag of a whole pane (panedrag.go). Zero value = no // drag. Rides clearDragState like every other drag. @@ -988,7 +991,8 @@ type Model struct { // closeRequested holds the panes THIS client's user confirmed closing. // The broadcast that prunes one is a user change this client stores; any - // other prune waits for whoever asked (layoutsync.go). + // other prune waits for whoever asked (layoutsync.go). Keyed by + // closeKey(dest, paneID); a reattach drops that dest's entries. closeRequested map[string]bool // Project-sidebar edge drag. sidebarDragging is set while a drag is in @@ -2147,6 +2151,7 @@ func (m Model) Update(msg tea.Msg) (retModel tea.Model, retCmd tea.Cmd) { m.clearDragState() m.splitDragNode = hit.Node m.splitDragRect = *hit + m.splitDragRatio = hit.Node.Ratio m.selection = nil m.setSplitDragHighlight(hit, true) return m, nil @@ -3751,6 +3756,7 @@ func (m *Model) clearDragState() { m.viewerMouseDown = false m.splitDragNode = nil m.splitDragRect = BorderHit{} + m.splitDragRatio = 0 m.sidebarDragging = false m.sidebarDragW = 0 m.paneDrag = paneDragState{} @@ -3914,7 +3920,9 @@ func (m *Model) dragSplitBorder(x, y int) { // the on-release-only design) and the dragged tab's layout (persists the // new Ratio). resizeAllPanes covers all panes; the daemon's same-size guard // drops the untouched panes' resizes. Only the active tab's tree moved, so -// only it is stored (markLayoutChanged). +// only it is stored (markLayoutChanged) — and not at all when the release +// left the ratio where the press found it: a click on a border is not a +// change, and storing it would bump the revision for every other client. func (m *Model) finishSplitDrag() tea.Cmd { // The one VT resize of the whole drag: old size → final size, paired // with the PTY resize below so the child's SIGWINCH redraw lands in a @@ -3923,9 +3931,10 @@ func (m *Model) finishSplitDrag() tea.Cmd { if tab != nil { tab.Resize(tab.Width, tab.Height) } + moved := m.splitDragNode == nil || m.splitDragNode.Ratio != m.splitDragRatio m.clearDragState() var layout tea.Cmd - if tab != nil { + if tab != nil && moved { layout = m.markLayoutChanged(m.destOfTab(tab.ID), tab) } return tea.Batch(m.resizeAllPanes(), layout) @@ -6367,7 +6376,7 @@ func (m *Model) rebuildTabs(info ProjectInfo, state WorkspaceStateMsg, existingT case tab.templateLayoutPending: tab.layoutRev = tabInfo.LayoutRev case exists && !restored: - lp = m.syncTabLayout(tab, tabInfo, daemonPaneSet, paneMap, existingPanes) + lp = m.syncTabLayout(tab, tabInfo, daemonPaneSet, paneMap, existingPanes, dest) newPaneIDs = append(newPaneIDs, lp.created...) } @@ -6387,8 +6396,7 @@ func (m *Model) rebuildTabs(info ProjectInfo, state WorkspaceStateMsg, existingT tab.RemovePane(id) // A close THIS client confirmed is its user's change to // store; any other prune waits for the requester's write. - if m.closeRequested[id] { - delete(m.closeRequested, id) + if m.takeCloseRequest(dest, id) { lp.send = true } else { tab.awaitGone(id) @@ -6408,9 +6416,10 @@ func (m *Model) rebuildTabs(info ProjectInfo, state WorkspaceStateMsg, existingT treePaneIDs = tab.Root.PaneIDs() } // Set when a moved pane arrived while this client held a reservation - // in this tab; the placeholder prune below then spares it for this - // pass (see there). - sparedReservation := false + // in this tab, or when adoption just re-seated that reservation in a + // new tree; the placeholder prune below then spares it for this pass + // (see there). + sparedReservation := lp.reseated for _, paneID := range tabInfo.Panes { // Overlay panes are reconciled separately — never insert into the tree. if isOverlayPane(paneMap, paneID) { @@ -6682,7 +6691,7 @@ func (m *Model) rebuildTabs(info ProjectInfo, state WorkspaceStateMsg, existingT // arrival nobody here asked for (layoutsync.go). func (m *Model) restoreTabLayout(tab *TabModel, tabInfo TabInfo, paneMap map[string]*PaneInfo, existingPanes map[string]*PaneModel, dest string) *TabModel { tab.templateLayoutApplied, tab.templateLayoutPending = true, false - tab.layoutRev = tabInfo.LayoutRev + tab.layoutRev, tab.adoptNext = tabInfo.LayoutRev, false tab.layoutDirty, tab.layoutSent, tab.layoutResend = false, nil, false tab.clearAwaiting() log.Printf("restoreLayout: tab %s %q with %d panes", tab.ID, tabInfo.Name, len(tabInfo.Panes)) diff --git a/internal/tui/tab.go b/internal/tui/tab.go index 245285fb..d0cf2b9b 100644 --- a/internal/tui/tab.go +++ b/internal/tui/tab.go @@ -66,6 +66,9 @@ type TabModel struct { layoutDirty bool layoutSent *SerializedNode layoutResend bool + // adoptNext: the next broadcast is adopted whatever its revision — set by + // a reattach, when the daemon's stored tree is the authority. + adoptNext bool // awaitingPanes / awaitingGone are ids this client placed or pruned on // its own, for a change nobody here asked for (an MCP create, another // client's split or close, a moved pane). The next broadcast says whether From 81c50de9456aee1dcdf6f814f6fb5220a203ffa8 Mon Sep 17 00:00:00 2001 From: Artjoms Stukans Date: Fri, 25 Sep 2026 23:25:19 +0200 Subject: [PATCH 18/40] fix(tui): drop a reservation once its placeholder is pruned An ordinary pendingSplit entry was cleared only by a fill, so when a pass pruned its placeholder the entry stayed: tabLayoutBusy reported the tab busy for the rest of the session, the next fresh arrival filled a node no tree held, and an adoption re-seated the abandoned placeholder as an empty slot. The prune now deletes an entry whose placeholder it detached, with its sibling/direction record, and adoption re-seats only a reservation still in the tree it replaces. Worktree creates keep their exemption. --- internal/tui/layout_sync_test.go | 69 ++++++++++++++++++++++++++++++++ internal/tui/layoutsync.go | 8 ++++ internal/tui/model.go | 17 ++++++-- 3 files changed, 91 insertions(+), 3 deletions(-) diff --git a/internal/tui/layout_sync_test.go b/internal/tui/layout_sync_test.go index ce34591e..6097a924 100644 --- a/internal/tui/layout_sync_test.go +++ b/internal/tui/layout_sync_test.go @@ -384,6 +384,75 @@ func TestLayoutSync_PendingSplitSurvivesBesideSibling(t *testing.T) { } } +// An ordinary reservation whose pane never came is pruned by the next pass, +// and with it the reservation itself: the tab is not busy any more, a later +// adoption re-seats nothing, and an unrelated arrival is placed by the normal +// rule rather than into a node no tree holds. +func TestLayoutSync_AbandonedReservationIsForgotten(t *testing.T) { + t.Setenv("QUIL_HOME", t.TempDir()) + m := newLayoutSyncModel(t) + stored := lsSplit(SplitHorizontal, 0.5, lsLeaf("p1"), lsLeaf("p2")) + m, _ = lsApply(t, m, lsState(1, lsWire(t, stored), "p1", "p2")) + lsTab(t, &m).ActivePane = "p2" + _ = m.splitPane(SplitVertical) + if !m.tabLayoutBusy(lsTab(t, &m)) { + t.Fatal("setup: the reservation did not make the tab busy") + } + + // A broadcast with no new pane: the placeholder is pruned. + m, _ = lsApply(t, m, lsState(1, lsWire(t, stored), "p1", "p2")) + tab := lsTab(t, &m) + if _, armed := m.pendingSplit["t1"]; armed { + t.Error("pendingSplit still armed after its placeholder was pruned") + } + if m.tabLayoutBusy(tab) { + t.Error("the tab still reads busy after its reservation was pruned") + } + + // A later adoption re-seats nothing. + theirs := lsSplit(SplitVertical, 0.5, lsLeaf("p2"), lsLeaf("p1")) + m, _ = lsApply(t, m, lsState(2, lsWire(t, theirs), "p1", "p2")) + if got := lsTree(t, &m); !reflect.DeepEqual(got, theirs) { + t.Errorf("tree = %s, want exactly the adopted %s", layoutString(got), layoutString(theirs)) + } + + // An unrelated arrival takes the ordinary rule (first leaf, top|bottom), + // visibly, and without taking the tab's active pane. + m, sent := lsApply(t, m, lsState(2, lsWire(t, theirs), "p1", "p2", "p-mcp")) + tab = lsTab(t, &m) + want := lsSplit(SplitVertical, 0.5, lsSplit(SplitVertical, 0.5, lsLeaf("p2"), lsLeaf("p-mcp")), lsLeaf("p1")) + if got := lsTree(t, &m); !reflect.DeepEqual(got, want) { + t.Errorf("tree = %s, want %s", layoutString(got), layoutString(want)) + } + if tab.ActivePane == "p-mcp" { + t.Error("the unrelated arrival took the tab's active pane, as a reservation fill would") + } + if len(sent) != 0 { + t.Errorf("the unrelated arrival sent %d frames, want 0 — nobody here asked for it", len(sent)) + } +} + +// Adoption re-seats only a reservation still in the tree it replaces; a +// detached one is dropped rather than brought back as an empty slot. +func TestLayoutSync_AdoptionDropsADetachedReservation(t *testing.T) { + t.Setenv("QUIL_HOME", t.TempDir()) + m := newLayoutSyncModel(t) + m, _ = lsApply(t, m, lsState(1, lsWire(t, lsSplit(SplitHorizontal, 0.5, lsLeaf("p1"), lsLeaf("p2"))), "p1", "p2")) + tab := lsTab(t, &m) + // A stale entry pointing at a node no tree holds. + m.pendingSplit = map[string]*LayoutNode{"t1": {Ratio: 0.5}} + tab.noteReservation("p2", SplitVertical, false) + + theirs := lsSplit(SplitVertical, 0.5, lsLeaf("p2"), lsLeaf("p1")) + m, _ = lsApply(t, m, lsState(2, lsWire(t, theirs), "p1", "p2")) + if got := lsTree(t, &m); !reflect.DeepEqual(got, theirs) { + t.Errorf("tree = %s, want exactly the adopted %s — a detached reservation was re-seated", layoutString(got), layoutString(theirs)) + } + if _, armed := m.pendingSplit["t1"]; armed { + t.Error("the detached reservation is still armed after adoption") + } +} + // The sibling is gone from the adopted tree: the reservation falls back to // where the spiral would place an arrival. func TestLayoutSync_PendingSplitWithoutSiblingTakesTheSpiralSlot(t *testing.T) { diff --git a/internal/tui/layoutsync.go b/internal/tui/layoutsync.go index 4dd751a4..d6484c84 100644 --- a/internal/tui/layoutsync.go +++ b/internal/tui/layoutsync.go @@ -257,7 +257,15 @@ func (m *Model) adoptTabLayout(tab *TabModel, stored *SerializedNode, paneSet ma // The stored tree names the pane a REPLACE reservation stands in for; // a sentinel keeps that leaf through the prune so the reservation can // take its place. + // + // Only a reservation still IN the tree being replaced is re-seated; one + // whose placeholder is already detached was abandoned, and is dropped. ph := m.pendingSplit[tab.ID] + if ph != nil && !treeContains(tab.Root, ph) { + delete(m.pendingSplit, tab.ID) + tab.noteReservation("", 0, false) + ph = nil + } var sentinel *PaneModel if ph != nil && tab.reserveReplace && tab.reserveSibling != "" && panes[tab.reserveSibling] == nil { sentinel = &PaneModel{ID: tab.reserveSibling} diff --git a/internal/tui/model.go b/internal/tui/model.go index 75ed13fd..bdfee44b 100644 --- a/internal/tui/model.go +++ b/internal/tui/model.go @@ -6648,9 +6648,20 @@ func (m *Model) rebuildTabs(info ProjectInfo, state WorkspaceStateMsg, existingT // way. That pane normally rides the very next broadcast; if it never // comes, the next pass prunes as it always did. tab.CreatingBranch = m.worktreeCreates[tab.ID] - if tab.Root != nil && tab.CreatingBranch == "" && !sparedReservation { - tab.Root.PrunePlaceholders() - tab.invalidateLeaves() + if tab.CreatingBranch == "" && !sparedReservation { + if tab.Root != nil { + tab.Root.PrunePlaceholders() + tab.invalidateLeaves() + } + // A reservation whose placeholder that prune detached is over: its + // pane is not coming into it. Left armed, it reports the tab busy + // for the rest of the session (tabLayoutBusy), the next fresh + // arrival fills a node no tree holds, and an adoption re-seats an + // empty slot nobody is waiting for. + if ph := m.pendingSplit[tab.ID]; ph != nil && !treeContains(tab.Root, ph) { + delete(m.pendingSplit, tab.ID) + tab.noteReservation("", 0, false) + } } if tab.Root != nil { log.Printf("apply: tab %s panes reconciled (n=%d leaves)", tab.ID, len(tab.Leaves())) From e39e838b64437c4bc6c80786842ae9861c4ce9f0 Mon Sep 17 00:00:00 2001 From: Artjoms Stukans Date: Sat, 26 Sep 2026 00:06:41 +0200 Subject: [PATCH 19/40] feat(tui): guard typing across a remote tab switch MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit When another attached client switches this client's active tab, a keystroke or paste already "in flight" must not land in whatever pane the switch happened to make active. Model.requestedTab records the tab id THIS client asked for (per dest+project), via switchTab/ switchTabBy and sendCreateTab's pendingTabCreateToken; applyWorkspaceState compares the daemon's adopted ActiveTab against it to tell a local switch's own echo from a genuinely remote one. A remote change arms remoteSwitchAt/guardPaneID (the previously active pane) and flashes "Tab switched by another client"; enqueueKeyInput (the entry point for typed keys and both paste paths, never wheel input) redirects to that pane for remoteSwitchGuardWindow (250ms) or until it stops existing. A pane focused only by a remote switch also must not read as "seen" on every attached client the instant one of them switches — ackFocusedPane skips its unseen-clearing report while remoteFocusUnacked is set, which Update's prologue clears the moment local input (a key or a mouse click) actually arrives. Also wires the client side of the two small daemon broadcasts this depends on: event_dismissed removes a notification card (or all, when empty) from the sidebar, and pane_seen clears a pane's local unseen mark without echoing anything back. --- internal/tui/model.go | 238 ++++++++++++++++++++++- internal/tui/notification.go | 21 +++ internal/tui/typing_guard_test.go | 301 ++++++++++++++++++++++++++++++ internal/tui/workstate.go | 9 + 4 files changed, 565 insertions(+), 4 deletions(-) create mode 100644 internal/tui/typing_guard_test.go diff --git a/internal/tui/model.go b/internal/tui/model.go index bdfee44b..12cb1c8b 100644 --- a/internal/tui/model.go +++ b/internal/tui/model.go @@ -264,6 +264,26 @@ type paneSizesMsg struct { // paneEventMsg delivers a notification event from the daemon. type paneEventMsg ipc.PaneEventPayload +// eventDismissedMsg is the daemon's broadcast that a notification was +// dismissed (event_dismissed, spec §8.4), reaching every attached client — +// this client's own dismissal included. dest is carried for parity with the +// wire event; the notification sidebar is not scoped per destination (a +// pre-existing property this task does not change), so the removal applies to +// the one shared list regardless of which daemon reported it. +type eventDismissedMsg struct { + dest string + eventID string // "" = dismiss every card +} + +// paneSeenMsg is the daemon's broadcast that a pane's unseen mark was cleared +// by SOME attached client (pane_seen, spec §8.4). This client clears its own +// local copy of the mark and reports nothing back — reporting would echo the +// clear it was just told about, forever. +type paneSeenMsg struct { + dest string + paneID string +} + // pasteRefreshMsg triggers a re-render after paste so the cursor updates. type pasteRefreshMsg struct{} @@ -506,7 +526,39 @@ type Model struct { // clientCount records, per destination, the last broadcast's attached- // client count (bridges excluded) — what renderStatusBar's role marker // and D9's "most recent input" default both key off of at the TUI layer. - clientCount map[string]int + clientCount map[string]int + // requestedTab records, per (dest, project) key (requestedTabKey), the tab + // id THIS client last asked for via switchTab/switchTabBy or sendCreateTab. + // applyWorkspaceState consults it to tell this client's own switch landing + // apart from a change some OTHER client made (typing guard, spec §8.1): + // a broadcast whose adopted ActiveTab equals the recorded value is this + // client's own request landing (the entry is cleared); anything else is + // remote and arms remoteSwitchAt/guardPaneID/remoteFocusUnacked below. + requestedTab map[string]string + // remoteSwitchAt is m.clock() at the last REMOTE active-tab change this + // client observed for its active project — the typing guard's window + // (remoteSwitchGuardWindow) is measured from here. + remoteSwitchAt time.Time + // guardPaneID is the pane that was active, on THIS client, immediately + // before a remote tab switch moved focus elsewhere. Key-originated input + // (typed keys and paste — never mouse) arriving within the guard window + // retargets here instead of the pane the remote switch made active, since + // the user was mid-keystroke in THIS pane, not the one another client + // picked. Cleared implicitly once the window elapses or the pane stops + // existing (guardedInputTarget checks both, live, rather than expiring the + // field itself). + guardPaneID string + // remoteFocusUnacked is true from the moment a remote switch focuses a + // pane on this client until this client's own next key or mouse click. + // While set, ackFocusedPane skips its unseen-clearing report — a pane + // must not read as "seen" on every attached client just because one of + // them switched tabs, whether or not anyone actually looked. + remoteFocusUnacked bool + // now is the typing guard's clock seam: time.Now in NewModel, a fixed + // function in tests. Read through Model.clock(), never directly, since a + // Model literal built by a test (the common shape in this package) leaves + // it nil. + now func() time.Time renaming bool renameInput string renamingPane bool @@ -1222,6 +1274,8 @@ func NewModel(client Client, cfg config.Config, version string, registry *plugin inputCh: make(chan paneInput, inputForwardBuffer), inputDone: make(chan struct{}), inputIdle: make(chan struct{}), + requestedTab: make(map[string]string), + now: time.Now, } // Startup dialog priority: migration > what's-new > update-notice > // disclaimer. Migration blocks startup until every stale plugin is @@ -1399,6 +1453,17 @@ func (m Model) Update(msg tea.Msg) (retModel tea.Model, retCmd tea.Cmd) { // means the flag describes THIS message and can never leak into the next. m.skipRender = false m.skipHidden = false + // Local input answers the typing guard's ack hold (spec §8.1): a pane that + // became focused only because ANOTHER client switched tabs must not read + // as "seen" until the user actually looks at it, which a key or a mouse + // click proves and a spinner tick or a PTY chunk does not. Cleared BEFORE + // ackFocusedPane runs below, so the very keystroke that answers the guard + // also acks the pane in the same Update call — there is no reason to make + // the user press twice. + switch msg.(type) { + case tea.KeyPressMsg, tea.MouseClickMsg: + m.remoteFocusUnacked = false + } // Acknowledge the focused pane of the active tab before processing the // message — focusing is the acknowledgement; see ackFocusedPane. // @@ -3037,6 +3102,23 @@ func (m Model) Update(msg tea.Msg) (retModel tea.Model, retCmd tea.Cmd) { } return m, tea.Batch(cmds...) + case eventDismissedMsg: + // Applied locally, exactly like the daemon-side dismissal this mirrors: + // remove the card (or every card, "" = all) and report nothing — this + // broadcast IS the report, whether it originated here or on another + // attached client (spec §8.4). + m.notifications.DismissByID(msg.eventID) + return m, m.listenForMessages() + + case paneSeenMsg: + // Some attached client (this one included) cleared the pane's unseen + // mark; mirror it locally without sending anything back, or every + // client would echo the clear at each other forever (spec §8.4). + if pane, _, _ := m.findPaneAndTab(msg.paneID); pane != nil { + pane.unseen = false + } + return m, m.listenForMessages() + case sidebarTickMsg: // Re-render sidebar to update relative timestamps; schedule next tick if still visible. // @@ -4379,6 +4461,16 @@ func (m Model) handleNewTab() (tea.Model, tea.Cmd) { // windows regressed exactly that way when this function stamped unconditionally. func (m Model) sendCreateTab(spec *ipc.FirstPaneSpec) tea.Cmd { dest := m.createPaneDest + // Typing guard (spec §8.1): this client is about to become the reason its + // active project's ActiveTab changes, so the landing broadcast must not + // read as another client's switch. There is no tab id to record yet — the + // daemon mints one — so pendingTabCreateToken stands in for it. Value + // receiver: this only reaches an existing map (every production Model's, + // from NewModel), matching switchTab's synchronous recording so the guard + // can never observe a create that is already in flight. + if proj := m.cur(); proj != nil && m.requestedTab != nil { + m.requestedTab[requestedTabKey(dest, proj.ID)] = pendingTabCreateToken + } return func() tea.Msg { msg, err := ipc.NewMessage(ipc.MsgCreateTab, ipc.CreateTabPayload{ Name: "New Tab", @@ -5983,6 +6075,37 @@ func (m *Model) handlePaneOutput(msg PaneOutputMsg) (tea.Cmd, bool) { return nil, false } +// applyTabMoveGuard decides whether an active-tab change applyWorkspaceState +// just observed for the ACTIVE project is this client's own switch landing or +// one another attached client made, and arms the typing guard for the latter +// (spec §8.1). +// +// "Not requested" is decided with the requestedTab TOKEN, never with a time +// window: a local switch followed quickly by an unrelated remote one must +// still be guarded, which a window alone cannot tell apart from the local +// switch's own delayed echo. +// +// fromTab is the tab THIS client was showing right before the change — the +// caller has already established it is non-nil. Its ActivePane is the pane +// that owns the guard: the user was looking at THAT pane, not whatever the +// new active tab's pane happens to be. +func (m *Model) applyTabMoveGuard(dest, projectID, newActiveTab string, fromTab *TabModel) { + key := requestedTabKey(dest, projectID) + if req, ok := m.requestedTab[key]; ok { + // An exact match is an ordinary switchTab landing. pendingTabCreateToken + // matches whatever tab the daemon makes active next, since a create_tab + // request has no id to compare by equality — the daemon mints one. + if req == newActiveTab || req == pendingTabCreateToken { + delete(m.requestedTab, key) + return + } + } + m.remoteSwitchAt = m.clock() + m.guardPaneID = fromTab.ActivePane + m.remoteFocusUnacked = true + m.setFlash("Tab switched by another client") +} + // applyWorkspaceState rebuilds the TUI state from one daemon's broadcast. // dest names the destination that broadcast arrived on (empty = the local // daemon) and scopes the merge: a broadcast is the FULL state of ONE daemon, @@ -6126,6 +6249,14 @@ func (m *Model) applyWorkspaceState(state WorkspaceStateMsg, dest string) ([]str proj.activeTab = indexOfTab(proj.tabs, info.ActiveTab) if targetTab := tabAt(proj.tabs, proj.activeTab); fromTab != targetTab { tabMoves = append(tabMoves, activeTabMove{from: fromTab, target: targetTab}) + // Typing guard (spec §8.1), scoped to the ACTIVE project only: a + // background project's active tab moving under it is not something + // anyone is typing into right now. ok (the project already existed) + // and fromTab != nil rule out this project's very first broadcast, + // which has no "before" for the guard to protect. + if info.ID == activeID && ok && fromTab != nil { + m.applyTabMoveGuard(dest, info.ID, info.ActiveTab, fromTab) + } } newPaneIDs = append(newPaneIDs, projPaneIDs...) overlayResizeCmds = append(overlayResizeCmds, projResizeCmds...) @@ -7035,6 +7166,13 @@ func (m *Model) switchTab(idx int) tea.Cmd { from := m.activeTabModel() tabID, dest := target.ID, target.Dest m.setActiveTabIdx(idx) + // Typing guard (spec §8.1): this client asked for tabID, so the broadcast + // that lands it must not be mistaken for another client's switch. Recorded + // against the CURRENT project — target and its project share one Dest, so + // this is the same key applyWorkspaceState looks up. + if proj := m.cur(); proj != nil { + m.recordRequestedTab(dest, proj.ID, tabID) + } cmds := []tea.Cmd{func() tea.Msg { msg, _ := ipc.NewMessage(ipc.MsgSwitchTab, ipc.SwitchTabPayload{ TabID: tabID, @@ -7054,6 +7192,54 @@ func (m *Model) switchTab(idx int) tea.Cmd { return tea.Batch(cmds...) } +// remoteSwitchGuardWindow is the typing guard's window (spec §8.1, table in +// global-constraints.md): key-originated input arriving this soon after a +// REMOTE active-tab change still goes to the pane the user was looking at +// before it, not the pane the switch made active. +const remoteSwitchGuardWindow = 250 * time.Millisecond + +// pendingTabCreateToken marks requestedTab[key] while this client's own +// create_tab is in flight for that project. Unlike switchTab, sendCreateTab +// has no id to record ahead of time — the daemon mints the new tab's id — so +// there is nothing to compare the eventual broadcast's ActiveTab against by +// equality. The token can never collide with a real tab id (daemon-issued ids +// never carry a NUL byte), and applyTabMoveGuard treats it as a match for +// whatever tab the daemon makes active next. +const pendingTabCreateToken = "\x00pending-create" + +// requestedTabKey identifies one project on one destination for +// Model.requestedTab. Dest alone is not unique (each daemon mints its own +// project ids independently) and a project id alone is not unique across +// daemons either — the same (Dest, ID) pairing projectgroups.go uses for +// exactly this reason. +func requestedTabKey(dest, projectID string) string { + return dest + "\x00" + projectID +} + +// recordRequestedTab notes that THIS client is the one asking for tabID to +// become the active tab of (dest, projectID). Called by switchTab/ +// switchTabBy with the tab they are switching TO, and by sendCreateTab with +// pendingTabCreateToken. See requestedTab's field comment. +func (m *Model) recordRequestedTab(dest, projectID, tabID string) { + if m.requestedTab == nil { + m.requestedTab = make(map[string]string) + } + m.requestedTab[requestedTabKey(dest, projectID)] = tabID +} + +// clock returns the typing guard's current time: m.now when the Model has +// one (every production Model, via NewModel), else the real wall clock. The +// fallback is what keeps the ~46 Model literals other tests build directly +// safe to pass through applyWorkspaceState — none of them care about this +// feature, and a nil-func-call panic on an unrelated broadcast would be a +// surprising way to learn they now do. +func (m Model) clock() time.Time { + if m.now != nil { + return m.now() + } + return time.Now() +} + // eagerTabMarker is a single-width BMP glyph (deliberately not an emoji — wide // glyphs drift conhost columns; see pane_widechar_test.go). Shown on any tab // containing at least one eager-restore pane. @@ -7835,6 +8021,16 @@ func (m Model) listenForMessages() tea.Cmd { log.Printf("ipc recv: pane_event %s %s %s", payload.Type, payload.PaneID, payload.Title) return paneEventMsg(payload) + case ipc.MsgEventDismissed: + var payload ipc.EventDismissedPayload + msg.DecodePayload(&payload) + return eventDismissedMsg{dest: msg.Origin, eventID: payload.EventID} + + case ipc.MsgPaneSeen: + var payload ipc.PaneSeenPayload + msg.DecodePayload(&payload) + return paneSeenMsg{dest: msg.Origin, paneID: payload.PaneID} + case ipc.MsgResourceReportResp: var payload ipc.ResourceReportRespPayload if err := msg.DecodePayload(&payload); err != nil { @@ -8492,7 +8688,7 @@ func (m Model) forwardInputBytes(data []byte) tea.Cmd { if pane == nil { return nil } - m.enqueueInput(pane.ID, data) + m.enqueueKeyInput(pane.ID, data) return nil } @@ -8560,6 +8756,40 @@ func (m Model) enqueueInput(paneID string, data []byte) { m.inputCh <- paneInput{dest: dest, paneID: paneID, data: data} } +// enqueueKeyInput is the entry point for KEY-originated input — typed +// keystrokes (forwardInputBytes) and both paste paths (sendClipboardToPane, +// sendClipboardToPaneID). Paste counts as a key here (spec §8.1): it is the +// same user action the guard protects, delivered as one chunk instead of one +// byte at a time. Mouse-originated input (a wheel notch, sendInputToPane) +// must never call this — it goes straight to enqueueInput, because a remote +// tab switch says nothing about where the pointer is now. +// +// The guard only RETARGETS the pane id; dest is still resolved by +// enqueueInput's own destOfPane(paneID) for whichever id wins here, exactly +// as today. +func (m Model) enqueueKeyInput(paneID string, data []byte) { + m.enqueueInput(m.guardedInputTarget(paneID), data) +} + +// guardedInputTarget applies the typing guard (spec §8.1): within +// remoteSwitchGuardWindow of a remote tab switch, key-originated input still +// goes to guardPaneID — the pane the user was mid-keystroke in — rather than +// wherever the remote switch moved focus. It falls through to paneID once the +// window has elapsed or the guarded pane no longer exists (closed, moved, +// destroyed — there is nowhere left to redirect to). +func (m Model) guardedInputTarget(paneID string) string { + if m.guardPaneID == "" { + return paneID + } + if m.clock().Sub(m.remoteSwitchAt) >= remoteSwitchGuardWindow { + return paneID + } + if pane, _, _ := m.findPaneAndTab(m.guardPaneID); pane == nil { + return paneID + } + return m.guardPaneID +} + // sendPaneInput marshals and sends one MsgPaneInput frame to an ALREADY-RESOLVED // destination. It deliberately takes dest rather than calling sendForPane: the // forwarder goroutine must not walk m.projects (see paneInput). client.Send is @@ -9199,7 +9429,7 @@ func (m Model) sendClipboardToPane(text string) { // Pasted text is the user acting on the pane, so it answers a parked one // exactly as a typed key does. pane.answerBlockedByInput() - m.enqueueInput(pane.ID, pastePayload(pane, text)) + m.enqueueKeyInput(pane.ID, pastePayload(pane, text)) } // sendClipboardToPaneID pastes into a NAMED pane rather than whichever is @@ -9224,7 +9454,7 @@ func (m Model) sendClipboardToPaneID(paneID, text string) { // active pane any more — which is exactly why the answer is keyed to input // reaching a pane rather than to which pane holds focus. pane.answerBlockedByInput() - m.enqueueInput(paneID, pastePayload(pane, text)) + m.enqueueKeyInput(paneID, pastePayload(pane, text)) } func keyToBytes(keyMsg tea.KeyPressMsg) []byte { diff --git a/internal/tui/notification.go b/internal/tui/notification.go index de775793..c6681fd0 100644 --- a/internal/tui/notification.go +++ b/internal/tui/notification.go @@ -290,6 +290,27 @@ func (nc *NotificationCenter) DismissSelected() string { return id } +// DismissByID removes the event with the given id, or every event when id is +// empty — the client-side application of the daemon's event_dismissed +// broadcast (spec §8.4), which mirrors DismissEventPayload's own "" = all +// convention. Unlike DismissSelected/DismissAll, this is driven by a REPORT +// of what was already dismissed (this client's own action, or another +// attached client's), so it must never send anything back — that would echo +// the dismissal the broadcast just delivered. +func (nc *NotificationCenter) DismissByID(id string) { + if id == "" { + nc.DismissAll() + return + } + for i, e := range nc.events { + if e.ID == id { + nc.events = append(nc.events[:i], nc.events[i+1:]...) + break + } + } + nc.clampCursor() +} + // DismissAll removes all events. // // Every stored event, not just the visible ones: the key is documented as diff --git a/internal/tui/typing_guard_test.go b/internal/tui/typing_guard_test.go new file mode 100644 index 00000000..cb5349ae --- /dev/null +++ b/internal/tui/typing_guard_test.go @@ -0,0 +1,301 @@ +package tui + +import ( + "testing" + "time" + + tea "charm.land/bubbletea/v2" + + "github.com/artyomsv/quil/internal/config" + "github.com/artyomsv/quil/internal/ipc" +) + +// typingGuardModel builds a Model holding ONE project (the interim project +// oneProject seeds) with one tab per entry in tabIDs, each holding a single +// pane named "p"+tabID's own suffix ("t1" -> "p1"). Tab 0 starts active. The +// clock is a fake fixed at t0 so tests can advance it deterministically +// around remoteSwitchGuardWindow. +func typingGuardModel(t *testing.T, t0 time.Time, tabIDs ...string) (*Model, *fakeSender) { + t.Helper() + tabs := make([]*TabModel, 0, len(tabIDs)) + for _, id := range tabIDs { + paneID := "p" + id[1:] // "t1" -> "p1", "t2" -> "p2", ... + pane := NewPaneModel(paneID, 1024) + tab := NewTabModel(id, id) + tab.Root = NewLeaf(pane) + tab.ActivePane = paneID + tabs = append(tabs, tab) + } + fake := &fakeSender{} + m := &Model{ + cfg: config.Default(), + client: fake, + termFocused: true, + notifications: NewNotificationCenter(30, 50), + inputCh: make(chan paneInput, inputForwardBuffer), + tabDragFromIdx: -1, + now: func() time.Time { return t0 }, + } + m.projects = oneProject(tabs...) + m.setActiveTabIdx(0) + m.initKeymap() + return m, fake +} + +// typingGuardBroadcast returns a WorkspaceStateMsg naming exactly the tabs +// typingGuardModel built, with activeTab as the daemon's reported active tab +// — the shape a real broadcast carries (mirrors broadcast_echo_test.go's +// echoModel, minus the layout round trip this task's tests don't need). +func typingGuardBroadcast(activeTab string, tabIDs ...string) WorkspaceStateMsg { + state := WorkspaceStateMsg{ActiveTab: activeTab} + for _, id := range tabIDs { + paneID := "p" + id[1:] + state.Tabs = append(state.Tabs, TabInfo{ID: id, Name: id, Panes: []string{paneID}}) + state.Panes = append(state.Panes, PaneInfo{ID: paneID, TabID: id}) + } + return state +} + +// drainOneInput pops the single entry a test expects to already be queued, +// mirroring input_order_test.go's drainQueued (unavailable here for a count +// of 1 with a custom failure message). +func drainOneInput(t *testing.T, m *Model) paneInput { + t.Helper() + select { + case in := <-m.inputCh: + return in + default: + t.Fatal("nothing enqueued") + return paneInput{} + } +} + +// TestTypingGuard_KeyWithinWindowGoesToOldPane pins the core of spec §8.1: a +// remote switch (nobody on this client requested it) redirects key-originated +// input to the pane that was active before it, for as long as the guard +// window holds. +func TestTypingGuard_KeyWithinWindowGoesToOldPane(t *testing.T) { + t.Parallel() + t0 := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + m, _ := typingGuardModel(t, t0, "t1", "t2") + + // Another client switches the active tab to t2 — this client never asked + // for it (m.requestedTab is empty), so this is a REMOTE switch. + m.applyWorkspaceState(typingGuardBroadcast("t2", "t1", "t2"), "") + + if m.guardPaneID != "p1" { + t.Fatalf("guardPaneID = %q, want p1 (the pane active before the remote switch)", m.guardPaneID) + } + if m.activeTabModel().ID != "t2" { + t.Fatalf("active tab = %q, want t2 (the daemon's own switch still lands)", m.activeTabModel().ID) + } + + // Within the window: a keystroke goes to p1, not the now-active p2. + m.now = func() time.Time { return t0.Add(100 * time.Millisecond) } + if cmd := m.forwardInputBytes([]byte("x")); cmd != nil { + t.Fatalf("forwardInputBytes returned a non-nil cmd; keystrokes must enqueue synchronously") + } + in := drainOneInput(t, m) + if in.paneID != "p1" { + t.Errorf("queued paneID = %q, want p1", in.paneID) + } +} + +// TestTypingGuard_AfterWindowGoesToNewPane is the other side of the window: +// once remoteSwitchGuardWindow has elapsed, a keystroke goes to whatever pane +// is active now, not the guarded one. This is also the "window of 0" mutation +// check: mutating remoteSwitchGuardWindow to 0 makes THIS test's "within the +// window" sibling fail instead, by expiring the guard immediately. +func TestTypingGuard_AfterWindowGoesToNewPane(t *testing.T) { + t.Parallel() + t0 := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + m, _ := typingGuardModel(t, t0, "t1", "t2") + + m.applyWorkspaceState(typingGuardBroadcast("t2", "t1", "t2"), "") + if m.guardPaneID != "p1" { + t.Fatalf("guardPaneID = %q, want p1", m.guardPaneID) + } + + // Past the window: the keystroke goes to p2, the pane the remote switch + // actually made active. + m.now = func() time.Time { return t0.Add(300 * time.Millisecond) } + if cmd := m.forwardInputBytes([]byte("x")); cmd != nil { + t.Fatalf("forwardInputBytes returned a non-nil cmd") + } + in := drainOneInput(t, m) + if in.paneID != "p2" { + t.Errorf("queued paneID = %q, want p2 (guard window elapsed)", in.paneID) + } +} + +// TestTypingGuard_LocalSwitchSetsNoGuard: this client's OWN switchTab, echoed +// back by the daemon unchanged, must never arm the guard — every attached +// client sees its own switch land as a broadcast, and treating that as +// "someone else switched" would guard every ordinary tab change. +func TestTypingGuard_LocalSwitchSetsNoGuard(t *testing.T) { + t.Parallel() + t0 := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + m, _ := typingGuardModel(t, t0, "t1", "t2") + + m.switchTab(1) // requests t2 + + m.applyWorkspaceState(typingGuardBroadcast("t2", "t1", "t2"), "") + + if m.guardPaneID != "" { + t.Errorf("guardPaneID = %q, want empty — a local switch's own echo must not arm the guard", m.guardPaneID) + } + if m.remoteFocusUnacked { + t.Errorf("remoteFocusUnacked = true, want false") + } +} + +// TestTypingGuard_LocalThenDifferentRemoteIsGuarded pins the requestedTab +// TOKEN compare (spec §8.1): a local switch (to t2) followed quickly by a +// DIFFERENT client's switch (to t3, not t2) must still be guarded — a time +// window alone could not tell that apart from the local switch's own +// (slightly late) echo. Removing the token compare — always treating this as +// the local switch landing — is the mutation this test exists to catch. +func TestTypingGuard_LocalThenDifferentRemoteIsGuarded(t *testing.T) { + t.Parallel() + t0 := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + m, _ := typingGuardModel(t, t0, "t1", "t2", "t3") + + m.switchTab(1) // requests t2; this client's active tab is now t2 (pane p2) + + // A DIFFERENT client's switch lands first, to t3 — not what this client + // asked for. + m.applyWorkspaceState(typingGuardBroadcast("t3", "t1", "t2", "t3"), "") + + if m.guardPaneID != "p2" { + t.Fatalf("guardPaneID = %q, want p2 (this client's own tab when the remote switch arrived)", m.guardPaneID) + } + if !m.remoteFocusUnacked { + t.Error("remoteFocusUnacked = false, want true") + } + if m.activeTabModel().ID != "t3" { + t.Fatalf("active tab = %q, want t3 — the daemon's switch still lands, only typing is redirected", m.activeTabModel().ID) + } + + // The originally-requested "t2" then lands late (a delayed confirmation + // arriving after the intervening remote switch). requestedTab still holds + // "t2" — untouched by the remote switch above, which did not match it — + // so THIS broadcast must be recognised as the local request finally + // landing, and must NOT re-arm the guard against t3's own pane. This is + // the token compare's MATCH branch: mutating it away (always treating a + // move as remote) would re-arm here and change guardPaneID to p3. + m.applyWorkspaceState(typingGuardBroadcast("t2", "t1", "t2", "t3"), "") + if m.guardPaneID != "p2" { + t.Errorf("guardPaneID = %q after the delayed local confirmation, want it unchanged at p2 — "+ + "the token match must be recognised as this client's own request, not a second remote switch", m.guardPaneID) + } +} + +// TestTypingGuard_RemoteSwitchDoesNotAckUnseenUntilInput pins the other half +// of spec §8.1: a pane focused only because of a remote switch keeps its +// unseen mark through messages that are not local input, and only a key or a +// mouse click acknowledges it. Removing the remoteFocusUnacked check in +// ackFocusedPane is the mutation this test exists to catch. +func TestTypingGuard_RemoteSwitchDoesNotAckUnseenUntilInput(t *testing.T) { + t.Parallel() + t0 := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + m, fake := typingGuardModel(t, t0, "t1", "t2") + + m.applyWorkspaceState(typingGuardBroadcast("t2", "t1", "t2"), "") + if !m.remoteFocusUnacked { + t.Fatal("remoteFocusUnacked should be set by the remote switch") + } + + // Set AFTER the broadcast: syncPaneMeta seeds a pane's unseen mark from + // the (unset, so false) daemon copy exactly once per PaneModel, and that + // first sync already happened inside applyWorkspaceState above — setting + // it before would just be overwritten back to false. + p2, _, _ := m.findPaneAndTab("p2") + p2.unseen = true + + // A message that is not local input — the shared spinner tick — must not + // ack the mark, however many of them arrive. + updated, _ := m.Update(workSpinnerTickMsg{}) + got := updated.(Model) + p2after, _, _ := got.findPaneAndTab("p2") + if p2after == nil || !p2after.unseen { + t.Fatal("unseen was acknowledged by a non-input message; it must survive until local input arrives") + } + if !got.remoteFocusUnacked { + t.Fatal("remoteFocusUnacked was cleared by a non-input message") + } + + // A real keystroke both clears the hold and acknowledges the pane, in the + // same Update call. + updated, _ = got.Update(tea.KeyPressMsg{Text: "x"}) + got = updated.(Model) + p2final, _, _ := got.findPaneAndTab("p2") + if p2final == nil || p2final.unseen { + t.Error("unseen should be cleared once local input arrives") + } + if got.remoteFocusUnacked { + t.Error("remoteFocusUnacked should be cleared by the keystroke") + } + _ = fake +} + +// TestEventDismissed_RemovesCard pins the client-side half of spec §8.4's +// dismissal broadcast: a named id removes that one card, and an empty id +// clears the whole sidebar list — mirroring the daemon's own DismissEventPayload +// "" = all convention. +func TestEventDismissed_RemovesCard(t *testing.T) { + t.Parallel() + m := Model{ + cfg: config.Default(), + client: &fakeSender{}, + notifications: NewNotificationCenter(30, 50), + } + m.notifications.AddEvent(ipc.PaneEventPayload{ID: "evt-1", Type: "bell", Title: "one"}) + m.notifications.AddEvent(ipc.PaneEventPayload{ID: "evt-2", Type: "bell", Title: "two"}) + + updated, _ := m.Update(eventDismissedMsg{eventID: "evt-1"}) + got := updated.(Model) + if n := got.notifications.Count(); n != 1 { + t.Fatalf("count after dismissing evt-1 = %d, want 1", n) + } + if got.notifications.events[0].ID != "evt-2" { + t.Errorf("remaining event id = %q, want evt-2", got.notifications.events[0].ID) + } + + updated, _ = got.Update(eventDismissedMsg{eventID: ""}) + got = updated.(Model) + if n := got.notifications.Count(); n != 0 { + t.Errorf("count after dismissing all = %d, want 0", n) + } +} + +// TestPaneSeen_ClearsMarkWithoutEcho pins spec §8.4's unseen-clear broadcast: +// this client clears its own local mark and sends nothing back — echoing +// would have every attached client answer each other's clears forever. +func TestPaneSeen_ClearsMarkWithoutEcho(t *testing.T) { + t.Parallel() + pane := NewPaneModel("p1", 1024) + pane.unseen = true + tab := NewTabModel("t1", "One") + tab.Root = NewLeaf(pane) + tab.ActivePane = "p1" + fake := &fakeSender{} + m := Model{ + cfg: config.Default(), + client: fake, + projects: oneProject(tab), + } + + updated, _ := m.Update(paneSeenMsg{paneID: "p1"}) + got := updated.(Model) + + target, _, _ := got.findPaneAndTab("p1") + if target == nil { + t.Fatal("pane p1 vanished") + } + if target.unseen { + t.Error("unseen should be cleared") + } + if len(fake.sent) != 0 { + t.Errorf("pane_seen must not echo anything back to the daemon; got %d sends", len(fake.sent)) + } +} diff --git a/internal/tui/workstate.go b/internal/tui/workstate.go index 750648ed..d452535d 100644 --- a/internal/tui/workstate.go +++ b/internal/tui/workstate.go @@ -683,6 +683,15 @@ func (m *Model) ackFocusedPane() bool { if !m.termFocused { return false } + // A pane focused only because ANOTHER client switched tabs is not one this + // user has looked at yet (spec §8.1) — Update's prologue clears the flag + // the moment local input (a key or a mouse click) actually arrives, so + // skipping the ack here does not mean skipping it forever, only until then. + // Without this, every attached client would clear the mark the instant one + // of them switched, whether or not anyone was watching that screen. + if m.remoteFocusUnacked { + return false + } tab := m.activeTabModel() if tab == nil || tab.Root == nil || tab.ActivePane == "" { return false From c98a4f17958f0310a8202474d1030eb5b73d11ae Mon Sep 17 00:00:00 2001 From: Artjoms Stukans Date: Sat, 26 Sep 2026 00:45:29 +0200 Subject: [PATCH 20/40] fix(tui): correct four typing-guard review findings MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Review round 1 on the typing-guard commit found four correctness gaps: - requestedTab's token was only retired inside a "moved" branch, so an ORDINARY echo of a local switchTab (fromTab already equals targetTab, since the client's own index moves synchronously) never cleared it. The stale token could then wrongly match a later, unrelated remote switch to the same tab id and suppress the guard. applyTabMoveGuard now runs once per broadcast for the active project, move or not, and retires a matching token unconditionally. - Destroying, moving, or dissolving the active tab locally (Ctrl+W, Move to project, a last-pane dissolve) looked identical to a remote switch away from it, arming a false "Tab switched by another client" flash and an unseen-ack hold on ordinary tab actions. The guard now only arms when the FROM tab is still part of the broadcast's tab list for that project — a vanished source was taken away by this client, not switched away from by another. - A redirected keystroke or paste encoded and answered against the PRE-redirect pane (ResetScroll, answerBlockedByInput, interruptWorkingPane, and pastePayload's bracketed-paste mode) while the bytes themselves went to the guarded pane — crediting the wrong pane with input it never received, and risking an unbracketed multi-line paste into a shell. guardedInputPane resolves the actual target pane once, before any of that runs. - event_dismissed with an empty id (dismiss all) cleared every stored notification regardless of which destination broadcast it, so one daemon's dismiss-all wiped every other attached daemon's cards too. DismissByID now scopes the empty-id case to events whose pane resolves to the broadcasting destination. Also: the create_tab pending token now matches only a genuinely new tab id and is spent after one broadcast either way; sendCreateTab records the token only when the create's destination matches the active project's own; armReattachReset retires a reattaching destination's requestedTab entries and guard state; a lazygit-style overlay's own pane is guarded instead of the tree pane behind it; and a listener decode failure for either new message type falls back to listenContinueMsg instead of propagating a zero-value message. --- internal/tui/model.go | 194 ++++++++-- internal/tui/notification.go | 33 +- internal/tui/reconnect.go | 17 + internal/tui/typing_guard_test.go | 578 ++++++++++++++++++++++++++---- 4 files changed, 712 insertions(+), 110 deletions(-) diff --git a/internal/tui/model.go b/internal/tui/model.go index 12cb1c8b..b4ca4b6c 100644 --- a/internal/tui/model.go +++ b/internal/tui/model.go @@ -3104,10 +3104,11 @@ func (m Model) Update(msg tea.Msg) (retModel tea.Model, retCmd tea.Cmd) { case eventDismissedMsg: // Applied locally, exactly like the daemon-side dismissal this mirrors: - // remove the card (or every card, "" = all) and report nothing — this - // broadcast IS the report, whether it originated here or on another - // attached client (spec §8.4). - m.notifications.DismissByID(msg.eventID) + // remove the card (or every card FROM THIS DEST, "" = all) and report + // nothing — this broadcast IS the report, whether it originated here or + // on another attached client (spec §8.4). destOfPane scopes "all" to + // msg.dest, since the sidebar holds cards from every attached daemon. + m.notifications.DismissByID(msg.eventID, msg.dest, m.destOfPane) return m, m.listenForMessages() case paneSeenMsg: @@ -4468,7 +4469,15 @@ func (m Model) sendCreateTab(spec *ipc.FirstPaneSpec) tea.Cmd { // receiver: this only reaches an existing map (every production Model's, // from NewModel), matching switchTab's synchronous recording so the guard // can never observe a create that is already in flight. - if proj := m.cur(); proj != nil && m.requestedTab != nil { + // + // proj.Dest == dest is required, not assumed: createPaneDest pins the + // destination at dialog OPEN, and the active project can move to a + // DIFFERENT destination while the dialog sits open. Recording under + // (dest, m.cur().ID) then would pair a foreign dest with the wrong + // project's id — a key applyTabMoveGuard could never legitimately match, + // since it always looks up (dest, THAT dest's own active project). Better + // to record nothing than to record a key that can only ever be wrong. + if proj := m.cur(); proj != nil && proj.Dest == dest && m.requestedTab != nil { m.requestedTab[requestedTabKey(dest, proj.ID)] = pendingTabCreateToken } return func() tea.Msg { @@ -5561,7 +5570,12 @@ func (m Model) handleKey(msg tea.KeyPressMsg) (tea.Model, tea.Cmd) { if data := m.rawKeyFor(seqAction, key, msg); data != nil { m.selection = nil if tab := m.activeTabModel(); tab != nil { - if pane := tab.ActivePaneModel(); pane != nil { + // guardedInputPane, not ActivePaneModel directly: within the typing + // guard window (spec §8.1) the bytes below are headed at guardPaneID, + // not whatever tab.ActivePaneModel() now returns, and the scroll + // reset / blocked-answer must land on the pane that actually + // receives them. + if pane := m.guardedInputPane(tab.ActivePaneModel()); pane != nil { pane.ResetScroll() // A typed key is the answer a parked pane was waiting for; // approving a permission prompt fires no hook of its own. @@ -5799,7 +5813,11 @@ func (m Model) handleKey(msg tea.KeyPressMsg) (tea.Model, tea.Cmd) { } m.selection = nil if tab := m.activeTabModel(); tab != nil { - if pane := tab.ActivePaneModel(); pane != nil { + // guardedInputPane: within the typing guard window (spec §8.1) + // forwardInputBytes below redirects these bytes to guardPaneID, so + // the scroll reset, the blocked-answer and ESC's interrupt must act + // on THAT pane, not whatever tab.ActivePaneModel() now returns. + if pane := m.guardedInputPane(tab.ActivePaneModel()); pane != nil { pane.ResetScroll() // Same trigger as the scroll reset above — the user acted on // this pane — and the answer a parked pane never otherwise @@ -6075,35 +6093,93 @@ func (m *Model) handlePaneOutput(msg PaneOutputMsg) (tea.Cmd, bool) { return nil, false } +// tabsContainID reports whether tabs (a project's REBUILT tab list) still +// holds id. applyTabMoveGuard needs this rather than trusting a non-nil +// fromTab: a tab this client just destroyed, moved to another project, or +// dissolved (its last pane moved out) is a tab THIS client took away from +// itself, not one it was "switched away from" by another client. +func tabsContainID(tabs []*TabModel, id string) bool { + for _, t := range tabs { + if t != nil && t.ID == id { + return true + } + } + return false +} + +// tabInputPaneID returns the pane id actually receiving keyboard input for +// tab — the overlay's, while one is visible, matching ActivePaneModel's own +// rule, else the tree's active pane. applyTabMoveGuard needs this rather than +// the bare ActivePane field: a user typing into a lazygit overlay when a +// remote switch lands must get the overlay back, not the tree pane sitting +// behind it. +func tabInputPaneID(tab *TabModel) string { + if tab.overlayVisible && tab.overlayPane != nil { + return tab.overlayPane.ID + } + return tab.ActivePane +} + // applyTabMoveGuard decides whether an active-tab change applyWorkspaceState -// just observed for the ACTIVE project is this client's own switch landing or -// one another attached client made, and arms the typing guard for the latter -// (spec §8.1). +// just observed for the ACTIVE project is this client's own request landing +// or one another attached client made, and arms the typing guard for the +// latter (spec §8.1). Returns the flash-expiry cmd when it arms, nil +// otherwise — the caller must batch it, or the flash never clears itself. +// +// Called UNCONDITIONALLY for the active project on every broadcast, not only +// when the tab actually moved: a token must be retired on an ORDINARY echo +// too (fromTab == targetTab already, because switchTab updates the client's +// own index synchronously before any broadcast can land) — leaving a matched +// token in place would let it silently satisfy some LATER, unrelated +// broadcast that happens to name the same tab id. // // "Not requested" is decided with the requestedTab TOKEN, never with a time // window: a local switch followed quickly by an unrelated remote one must // still be guarded, which a window alone cannot tell apart from the local // switch's own delayed echo. // -// fromTab is the tab THIS client was showing right before the change — the -// caller has already established it is non-nil. Its ActivePane is the pane -// that owns the guard: the user was looking at THAT pane, not whatever the -// new active tab's pane happens to be. -func (m *Model) applyTabMoveGuard(dest, projectID, newActiveTab string, fromTab *TabModel) { +// existedBefore reports whether newActiveTab was already one of this +// client's tabs (any project) before this broadcast — the create-token's +// only use for it, since a create_tab request has no id to compare by +// equality ahead of time (the daemon mints one). +func (m *Model) applyTabMoveGuard(dest, projectID, newActiveTab string, fromTab, targetTab *TabModel, tabs []*TabModel, existedBefore bool) tea.Cmd { key := requestedTabKey(dest, projectID) if req, ok := m.requestedTab[key]; ok { - // An exact match is an ordinary switchTab landing. pendingTabCreateToken - // matches whatever tab the daemon makes active next, since a create_tab - // request has no id to compare by equality — the daemon mints one. - if req == newActiveTab || req == pendingTabCreateToken { + switch { + case req == pendingTabCreateToken: + // One-shot: spent against the very next active-tab change either + // way, so a leftover token can never outlive the create it was + // minted for and silently swallow some LATER remote switch. delete(m.requestedTab, key) - return + if !existedBefore { + return nil // the create's own tab landing + } + // Not the create landing — an unrelated change beat it there. + // Fall through to the ordinary remote-switch handling below. + case req == newActiveTab: + // An exact match is this client's own switchTab landing, echo or + // not — see the function comment for why this compare must run + // unconditionally rather than only inside a "moved" branch. + delete(m.requestedTab, key) + return nil } } + if fromTab == nil || fromTab == targetTab { + return nil // nothing moved, or this project's very first broadcast + } + // A vanished source tab (this client's own Ctrl+W, Move to project, or a + // dissolve/recovery that moved its last pane out) is not "switched away + // from" by another client — it is this client's own local action taking + // the tab away from under itself. Only arm when fromTab is still part of + // the broadcast's tab list for this project. + if !tabsContainID(tabs, fromTab.ID) { + return nil + } m.remoteSwitchAt = m.clock() - m.guardPaneID = fromTab.ActivePane + m.guardPaneID = tabInputPaneID(fromTab) m.remoteFocusUnacked = true m.setFlash("Tab switched by another client") + return m.flashCmd() } // applyWorkspaceState rebuilds the TUI state from one daemon's broadcast. @@ -6247,16 +6323,22 @@ func (m *Model) applyWorkspaceState(state WorkspaceStateMsg, dest string) ([]str tabs, projPaneIDs, projResizeCmds := m.rebuildTabs(info, state, existingTabs, existingPanes, paneMap, dest) proj.tabs = tabs proj.activeTab = indexOfTab(proj.tabs, info.ActiveTab) - if targetTab := tabAt(proj.tabs, proj.activeTab); fromTab != targetTab { + targetTab := tabAt(proj.tabs, proj.activeTab) + // Typing guard (spec §8.1), scoped to the ACTIVE project only: a + // background project's active tab moving under it is not something + // anyone is typing into right now. Called even when fromTab == targetTab + // (an ordinary echo) — see applyTabMoveGuard's own comment for why the + // requestedTab token must be retired on that path too, not only inside + // the "moved" branch below. ok (the project already existed) is what + // makes the token lookup meaningful; a brand new project has none. + if info.ID == activeID && ok { + _, existedBefore := existingTabs[info.ActiveTab] + if cmd := m.applyTabMoveGuard(dest, info.ID, info.ActiveTab, fromTab, targetTab, proj.tabs, existedBefore); cmd != nil { + overlayResizeCmds = append(overlayResizeCmds, cmd) + } + } + if fromTab != targetTab { tabMoves = append(tabMoves, activeTabMove{from: fromTab, target: targetTab}) - // Typing guard (spec §8.1), scoped to the ACTIVE project only: a - // background project's active tab moving under it is not something - // anyone is typing into right now. ok (the project already existed) - // and fromTab != nil rule out this project's very first broadcast, - // which has no "before" for the guard to protect. - if info.ID == activeID && ok && fromTab != nil { - m.applyTabMoveGuard(dest, info.ID, info.ActiveTab, fromTab) - } } newPaneIDs = append(newPaneIDs, projPaneIDs...) overlayResizeCmds = append(overlayResizeCmds, projResizeCmds...) @@ -8023,12 +8105,18 @@ func (m Model) listenForMessages() tea.Cmd { case ipc.MsgEventDismissed: var payload ipc.EventDismissedPayload - msg.DecodePayload(&payload) + if err := msg.DecodePayload(&payload); err != nil { + log.Printf("decode event_dismissed: %v", err) + return listenContinueMsg{} + } return eventDismissedMsg{dest: msg.Origin, eventID: payload.EventID} case ipc.MsgPaneSeen: var payload ipc.PaneSeenPayload - msg.DecodePayload(&payload) + if err := msg.DecodePayload(&payload); err != nil { + log.Printf("decode pane_seen: %v", err) + return listenContinueMsg{} + } return paneSeenMsg{dest: msg.Origin, paneID: payload.PaneID} case ipc.MsgResourceReportResp: @@ -8790,6 +8878,30 @@ func (m Model) guardedInputTarget(paneID string) string { return m.guardPaneID } +// guardedInputPane resolves the PaneModel a keystroke or paste should +// actually reach — guardedInputTarget's id, looked up live — for callers that +// must act on the pane OBJECT itself rather than just its id: ResetScroll, +// answerBlockedByInput, interruptWorkingPane, and pastePayload's bracketed- +// paste encoding. Encoding or answering against the PRE-redirect pane while +// the bytes themselves go to the guarded one credits and decodes for the +// wrong pane — an unbracketed multi-line paste sent to a plain shell, say, +// because it was encoded against a claude-code pane's bracketed-paste mode. +// Returns from unchanged (including nil) when there is nothing to redirect +// to, so callers can keep using their existing nil check. +func (m Model) guardedInputPane(from *PaneModel) *PaneModel { + if from == nil { + return nil + } + targetID := m.guardedInputTarget(from.ID) + if targetID == from.ID { + return from + } + if pane, _, _ := m.findPaneAndTab(targetID); pane != nil { + return pane + } + return from +} + // sendPaneInput marshals and sends one MsgPaneInput frame to an ALREADY-RESOLVED // destination. It deliberately takes dest rather than calling sendForPane: the // forwarder goroutine must not walk m.projects (see paneInput). client.Send is @@ -9422,7 +9534,12 @@ func (m Model) sendClipboardToPane(text string) { if tab == nil { return } - pane := tab.ActivePaneModel() + // guardedInputPane, not ActivePaneModel directly: within the typing guard + // window (spec §8.1) the paste is headed at guardPaneID, and encoding + // against the wrong pane's bracketed-paste mode (pastePayload) sends an + // unbracketed multi-line paste to a plain shell, or a bracketed one to an + // app that never asked for it. + pane := m.guardedInputPane(tab.ActivePaneModel()) if pane == nil { return } @@ -9445,16 +9562,19 @@ func (m Model) sendClipboardToPaneID(paneID, text string) { if text == "" || paneID == "" { return } - pane, _, _ := m.findPaneAndTab(paneID) - if pane == nil { + bound, _, _ := m.findPaneAndTab(paneID) + if bound == nil { logger.Debug("paste: pane %s vanished during the clipboard read — dropping", paneID) return } // The target was bound when the user asked to paste, so it need not be the // active pane any more — which is exactly why the answer is keyed to input - // reaching a pane rather than to which pane holds focus. + // reaching a pane rather than to which pane holds focus. guardedInputPane + // applies the SAME typing guard on top: the bound pane can itself be the + // one a remote switch just moved away from. + pane := m.guardedInputPane(bound) pane.answerBlockedByInput() - m.enqueueKeyInput(paneID, pastePayload(pane, text)) + m.enqueueKeyInput(pane.ID, pastePayload(pane, text)) } func keyToBytes(keyMsg tea.KeyPressMsg) []byte { diff --git a/internal/tui/notification.go b/internal/tui/notification.go index c6681fd0..8a60629c 100644 --- a/internal/tui/notification.go +++ b/internal/tui/notification.go @@ -290,16 +290,31 @@ func (nc *NotificationCenter) DismissSelected() string { return id } -// DismissByID removes the event with the given id, or every event when id is -// empty — the client-side application of the daemon's event_dismissed -// broadcast (spec §8.4), which mirrors DismissEventPayload's own "" = all -// convention. Unlike DismissSelected/DismissAll, this is driven by a REPORT -// of what was already dismissed (this client's own action, or another -// attached client's), so it must never send anything back — that would echo -// the dismissal the broadcast just delivered. -func (nc *NotificationCenter) DismissByID(id string) { +// DismissByID removes the event with the given id, or every event whose pane +// belongs to dest when id is empty — the client-side application of the +// daemon's event_dismissed broadcast (spec §8.4). Empty mirrors +// DismissEventPayload's own "" = all convention, but "all" is scoped to the +// DEST that broadcast it: the notification list holds cards from every +// destination this client is attached to, and a dismiss-all from ONE daemon +// must not clear another daemon's cards too. destOf resolves an event's pane +// to its destination (Model.destOfPane) — the stored event carries no +// destination of its own. A named id needs no such scoping: an id can only +// ever match the one event the dismissing daemon reported. +// +// Unlike DismissSelected/DismissAll, this is driven by a REPORT of what was +// already dismissed (this client's own action, or another attached client's), +// so it must never send anything back — that would echo the dismissal the +// broadcast just delivered. +func (nc *NotificationCenter) DismissByID(id, dest string, destOf func(paneID string) string) { if id == "" { - nc.DismissAll() + kept := nc.events[:0] + for _, e := range nc.events { + if destOf(e.PaneID) != dest { + kept = append(kept, e) + } + } + nc.events = kept + nc.clampCursor() return } for i, e := range nc.events { diff --git a/internal/tui/reconnect.go b/internal/tui/reconnect.go index 71003017..b536a162 100644 --- a/internal/tui/reconnect.go +++ b/internal/tui/reconnect.go @@ -948,6 +948,23 @@ func (m *Model) armReattachReset(dest string) { if m.selection != nil && m.destOfPane(m.selection.PaneID) == dest { m.selection = nil } + // Typing guard state (spec §8.1) is stale the moment its destination + // reattaches: a reattach replaces that daemon's whole state, so a pending + // requestedTab token can never land the broadcast it was waiting for (the + // tab it named may not even exist any more), and a guardPaneID pointing at + // one of its panes is redirecting input toward a pane about to be rebuilt + // out from under it. Requests for OTHER destinations are untouched — one + // daemon reconnecting says nothing about another's in-flight switches. + prefix := dest + "\x00" + for key := range m.requestedTab { + if strings.HasPrefix(key, prefix) { + delete(m.requestedTab, key) + } + } + if m.guardPaneID != "" && m.destOfPane(m.guardPaneID) == dest { + m.guardPaneID = "" + m.remoteFocusUnacked = false + } } // resetWorkStateForReattach zeroes in-flight execution state on every pane. diff --git a/internal/tui/typing_guard_test.go b/internal/tui/typing_guard_test.go index cb5349ae..e4cea22d 100644 --- a/internal/tui/typing_guard_test.go +++ b/internal/tui/typing_guard_test.go @@ -1,6 +1,7 @@ package tui import ( + "strings" "testing" "time" @@ -14,7 +15,8 @@ import ( // oneProject seeds) with one tab per entry in tabIDs, each holding a single // pane named "p"+tabID's own suffix ("t1" -> "p1"). Tab 0 starts active. The // clock is a fake fixed at t0 so tests can advance it deterministically -// around remoteSwitchGuardWindow. +// around remoteSwitchGuardWindow. Sized and resized like newSplitDragTestModel +// so mouse-driven tests (clicks) have real geometry to hit-test against. func typingGuardModel(t *testing.T, t0 time.Time, tabIDs ...string) (*Model, *fakeSender) { t.Helper() tabs := make([]*TabModel, 0, len(tabIDs)) @@ -35,17 +37,23 @@ func typingGuardModel(t *testing.T, t0 time.Time, tabIDs ...string) (*Model, *fa inputCh: make(chan paneInput, inputForwardBuffer), tabDragFromIdx: -1, now: func() time.Time { return t0 }, + width: 100, + height: 40, } m.projects = oneProject(tabs...) m.setActiveTabIdx(0) m.initKeymap() + m.resizeTabs() return m, fake } // typingGuardBroadcast returns a WorkspaceStateMsg naming exactly the tabs -// typingGuardModel built, with activeTab as the daemon's reported active tab -// — the shape a real broadcast carries (mirrors broadcast_echo_test.go's -// echoModel, minus the layout round trip this task's tests don't need). +// listed, with activeTab as the daemon's reported active tab — the shape a +// real broadcast carries (mirrors broadcast_echo_test.go's echoModel, minus +// the layout round trip this task's tests don't need). tabIDs need not match +// what typingGuardModel built: a shorter list simulates tabs this client no +// longer holds (destroyed, moved), and an id absent from BOTH the model and +// every prior broadcast simulates a brand new tab. func typingGuardBroadcast(activeTab string, tabIDs ...string) WorkspaceStateMsg { state := WorkspaceStateMsg{ActiveTab: activeTab} for _, id := range tabIDs { @@ -56,6 +64,13 @@ func typingGuardBroadcast(activeTab string, tabIDs ...string) WorkspaceStateMsg return state } +// altKey builds the Alt+ keypress that drives tab.switch_N — the real +// local-switch entry point (Update -> handleKey -> switchTab), matching how +// tabbar_scroll_test.go drives the same action. +func altKey(digit rune) tea.KeyPressMsg { + return tea.KeyPressMsg{Code: digit, Mod: tea.ModAlt} +} + // drainOneInput pops the single entry a test expects to already be queued, // mirroring input_order_test.go's drainQueued (unavailable here for a count // of 1 with a custom failure message). @@ -73,7 +88,9 @@ func drainOneInput(t *testing.T, m *Model) paneInput { // TestTypingGuard_KeyWithinWindowGoesToOldPane pins the core of spec §8.1: a // remote switch (nobody on this client requested it) redirects key-originated // input to the pane that was active before it, for as long as the guard -// window holds. +// window holds. Driven entirely through Update: the broadcast arrives as a +// WorkspaceStateMsg and the keystroke as a KeyPressMsg, exactly as production +// delivers both. func TestTypingGuard_KeyWithinWindowGoesToOldPane(t *testing.T) { t.Parallel() t0 := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) @@ -81,21 +98,22 @@ func TestTypingGuard_KeyWithinWindowGoesToOldPane(t *testing.T) { // Another client switches the active tab to t2 — this client never asked // for it (m.requestedTab is empty), so this is a REMOTE switch. - m.applyWorkspaceState(typingGuardBroadcast("t2", "t1", "t2"), "") + updated, _ := m.Update(typingGuardBroadcast("t2", "t1", "t2")) + got := updated.(Model) - if m.guardPaneID != "p1" { - t.Fatalf("guardPaneID = %q, want p1 (the pane active before the remote switch)", m.guardPaneID) + if got.guardPaneID != "p1" { + t.Fatalf("guardPaneID = %q, want p1 (the pane active before the remote switch)", got.guardPaneID) } - if m.activeTabModel().ID != "t2" { - t.Fatalf("active tab = %q, want t2 (the daemon's own switch still lands)", m.activeTabModel().ID) + if got.activeTabModel().ID != "t2" { + t.Fatalf("active tab = %q, want t2 (the daemon's own switch still lands)", got.activeTabModel().ID) } // Within the window: a keystroke goes to p1, not the now-active p2. - m.now = func() time.Time { return t0.Add(100 * time.Millisecond) } - if cmd := m.forwardInputBytes([]byte("x")); cmd != nil { - t.Fatalf("forwardInputBytes returned a non-nil cmd; keystrokes must enqueue synchronously") - } - in := drainOneInput(t, m) + got.now = func() time.Time { return t0.Add(100 * time.Millisecond) } + updated, _ = got.Update(tea.KeyPressMsg{Text: "x"}) + got2 := updated.(Model) + + in := drainOneInput(t, &got2) if in.paneID != "p1" { t.Errorf("queued paneID = %q, want p1", in.paneID) } @@ -111,40 +129,44 @@ func TestTypingGuard_AfterWindowGoesToNewPane(t *testing.T) { t0 := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) m, _ := typingGuardModel(t, t0, "t1", "t2") - m.applyWorkspaceState(typingGuardBroadcast("t2", "t1", "t2"), "") - if m.guardPaneID != "p1" { - t.Fatalf("guardPaneID = %q, want p1", m.guardPaneID) + updated, _ := m.Update(typingGuardBroadcast("t2", "t1", "t2")) + got := updated.(Model) + if got.guardPaneID != "p1" { + t.Fatalf("guardPaneID = %q, want p1", got.guardPaneID) } // Past the window: the keystroke goes to p2, the pane the remote switch // actually made active. - m.now = func() time.Time { return t0.Add(300 * time.Millisecond) } - if cmd := m.forwardInputBytes([]byte("x")); cmd != nil { - t.Fatalf("forwardInputBytes returned a non-nil cmd") - } - in := drainOneInput(t, m) + got.now = func() time.Time { return t0.Add(300 * time.Millisecond) } + updated, _ = got.Update(tea.KeyPressMsg{Text: "x"}) + got2 := updated.(Model) + + in := drainOneInput(t, &got2) if in.paneID != "p2" { t.Errorf("queued paneID = %q, want p2 (guard window elapsed)", in.paneID) } } -// TestTypingGuard_LocalSwitchSetsNoGuard: this client's OWN switchTab, echoed -// back by the daemon unchanged, must never arm the guard — every attached -// client sees its own switch land as a broadcast, and treating that as -// "someone else switched" would guard every ordinary tab change. +// TestTypingGuard_LocalSwitchSetsNoGuard: this client's OWN switchTab (driven +// via the real Alt+2 keypress), echoed back by the daemon unchanged, must +// never arm the guard — every attached client sees its own switch land as a +// broadcast, and treating that as "someone else switched" would guard every +// ordinary tab change. func TestTypingGuard_LocalSwitchSetsNoGuard(t *testing.T) { t.Parallel() t0 := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) m, _ := typingGuardModel(t, t0, "t1", "t2") - m.switchTab(1) // requests t2 + updated, _ := m.Update(altKey('2')) // requests t2 (tab.switch_2) + got := updated.(Model) - m.applyWorkspaceState(typingGuardBroadcast("t2", "t1", "t2"), "") + updated, _ = got.Update(typingGuardBroadcast("t2", "t1", "t2")) + got2 := updated.(Model) - if m.guardPaneID != "" { - t.Errorf("guardPaneID = %q, want empty — a local switch's own echo must not arm the guard", m.guardPaneID) + if got2.guardPaneID != "" { + t.Errorf("guardPaneID = %q, want empty — a local switch's own echo must not arm the guard", got2.guardPaneID) } - if m.remoteFocusUnacked { + if got2.remoteFocusUnacked { t.Errorf("remoteFocusUnacked = true, want false") } } @@ -153,55 +175,122 @@ func TestTypingGuard_LocalSwitchSetsNoGuard(t *testing.T) { // TOKEN compare (spec §8.1): a local switch (to t2) followed quickly by a // DIFFERENT client's switch (to t3, not t2) must still be guarded — a time // window alone could not tell that apart from the local switch's own -// (slightly late) echo. Removing the token compare — always treating this as -// the local switch landing — is the mutation this test exists to catch. +// (slightly late) echo. func TestTypingGuard_LocalThenDifferentRemoteIsGuarded(t *testing.T) { t.Parallel() t0 := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) m, _ := typingGuardModel(t, t0, "t1", "t2", "t3") - m.switchTab(1) // requests t2; this client's active tab is now t2 (pane p2) + updated, _ := m.Update(altKey('2')) // requests t2; active tab is now t2 (pane p2) + got := updated.(Model) // A DIFFERENT client's switch lands first, to t3 — not what this client // asked for. - m.applyWorkspaceState(typingGuardBroadcast("t3", "t1", "t2", "t3"), "") + updated, _ = got.Update(typingGuardBroadcast("t3", "t1", "t2", "t3")) + got2 := updated.(Model) - if m.guardPaneID != "p2" { - t.Fatalf("guardPaneID = %q, want p2 (this client's own tab when the remote switch arrived)", m.guardPaneID) + if got2.guardPaneID != "p2" { + t.Fatalf("guardPaneID = %q, want p2 (this client's own tab when the remote switch arrived)", got2.guardPaneID) } - if !m.remoteFocusUnacked { + if !got2.remoteFocusUnacked { t.Error("remoteFocusUnacked = false, want true") } - if m.activeTabModel().ID != "t3" { - t.Fatalf("active tab = %q, want t3 — the daemon's switch still lands, only typing is redirected", m.activeTabModel().ID) + if got2.activeTabModel().ID != "t3" { + t.Fatalf("active tab = %q, want t3 — the daemon's switch still lands, only typing is redirected", got2.activeTabModel().ID) } // The originally-requested "t2" then lands late (a delayed confirmation // arriving after the intervening remote switch). requestedTab still holds // "t2" — untouched by the remote switch above, which did not match it — // so THIS broadcast must be recognised as the local request finally - // landing, and must NOT re-arm the guard against t3's own pane. This is - // the token compare's MATCH branch: mutating it away (always treating a - // move as remote) would re-arm here and change guardPaneID to p3. - m.applyWorkspaceState(typingGuardBroadcast("t2", "t1", "t2", "t3"), "") - if m.guardPaneID != "p2" { + // landing, and must NOT re-arm the guard against t3's own pane. + updated, _ = got2.Update(typingGuardBroadcast("t2", "t1", "t2", "t3")) + got3 := updated.(Model) + if got3.guardPaneID != "p2" { t.Errorf("guardPaneID = %q after the delayed local confirmation, want it unchanged at p2 — "+ - "the token match must be recognised as this client's own request, not a second remote switch", m.guardPaneID) + "the token match must be recognised as this client's own request, not a second remote switch", got3.guardPaneID) + } +} + +// TestTypingGuard_StaleTokenDoesNotSuppressALaterRemoteSwitch pins the fix for +// review round 1's Important 1: requestedTab must be retired on an ORDINARY +// echo (fromTab == targetTab, no "move"), not only inside the moved branch — +// switchTab updates this client's own active-tab index synchronously, so the +// confirming broadcast for a local switch is never itself a "move". Leaving +// the token in place after that echo lets it wrongly satisfy a LATER, +// unrelated broadcast that happens to name the same tab id. +func TestTypingGuard_StaleTokenDoesNotSuppressALaterRemoteSwitch(t *testing.T) { + t.Parallel() + t0 := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + m, _ := typingGuardModel(t, t0, "t1", "t2") + + updated, _ := m.Update(altKey('2')) // requests t2 + got := updated.(Model) + + // The ordinary echo: fromTab == targetTab == t2 already (switchTab moved + // it synchronously), so this is not a "move" — the token must still be + // retired here. + updated, _ = got.Update(typingGuardBroadcast("t2", "t1", "t2")) + got2 := updated.(Model) + if got2.guardPaneID != "" { + t.Fatalf("guardPaneID = %q after the echo, want empty", got2.guardPaneID) + } + + // A different client switches to t1 — a genuine remote switch. + updated, _ = got2.Update(typingGuardBroadcast("t1", "t1", "t2")) + got3 := updated.(Model) + if got3.guardPaneID != "p2" { + t.Fatalf("guardPaneID = %q after switching to t1, want p2", got3.guardPaneID) + } + + // A different client switches BACK to t2 — a SECOND, unrelated remote + // change. If the stale "t2" token from the very first local switch had + // survived the echo above, it would wrongly match here and suppress this + // genuine remote switch. + updated, _ = got3.Update(typingGuardBroadcast("t2", "t1", "t2")) + got4 := updated.(Model) + if got4.guardPaneID != "p1" { + t.Errorf("guardPaneID = %q after the second remote switch back to t2, want p1 "+ + "(armed by a genuine remote switch, not suppressed by a stale token)", got4.guardPaneID) + } +} + +// TestTypingGuard_LocalTabDestroyIsNotARemoteSwitch pins the fix for review +// round 1's Important 2: closing, moving, or dissolving the ACTIVE tab is +// this client's OWN action taking the tab away from under itself, not a +// switch "by another client" — the false positive read as a bogus flash and +// an unseen-ack hold on the most common tab actions there are. +func TestTypingGuard_LocalTabDestroyIsNotARemoteSwitch(t *testing.T) { + t.Parallel() + t0 := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + m, _ := typingGuardModel(t, t0, "t1", "t2") + + // This client destroyed t1 (Ctrl+W): the very next broadcast no longer + // mentions it at all, and its neighbour t2 becomes active. + updated, _ := m.Update(typingGuardBroadcast("t2", "t2")) + got := updated.(Model) + + if got.guardPaneID != "" { + t.Errorf("guardPaneID = %q, want empty — destroying the active tab must not arm the typing guard", got.guardPaneID) + } + if got.remoteFocusUnacked { + t.Error("remoteFocusUnacked = true, want false") } } // TestTypingGuard_RemoteSwitchDoesNotAckUnseenUntilInput pins the other half // of spec §8.1: a pane focused only because of a remote switch keeps its // unseen mark through messages that are not local input, and only a key or a -// mouse click acknowledges it. Removing the remoteFocusUnacked check in -// ackFocusedPane is the mutation this test exists to catch. +// mouse click acknowledges it — and while unacked, nothing is reported back +// to the daemon (no pane_seen echo, no MsgUpdatePane{Unseen}). func TestTypingGuard_RemoteSwitchDoesNotAckUnseenUntilInput(t *testing.T) { t.Parallel() t0 := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) m, fake := typingGuardModel(t, t0, "t1", "t2") - m.applyWorkspaceState(typingGuardBroadcast("t2", "t1", "t2"), "") - if !m.remoteFocusUnacked { + updated, _ := m.Update(typingGuardBroadcast("t2", "t1", "t2")) + got := updated.(Model) + if !got.remoteFocusUnacked { t.Fatal("remoteFocusUnacked should be set by the remote switch") } @@ -209,39 +298,365 @@ func TestTypingGuard_RemoteSwitchDoesNotAckUnseenUntilInput(t *testing.T) { // the (unset, so false) daemon copy exactly once per PaneModel, and that // first sync already happened inside applyWorkspaceState above — setting // it before would just be overwritten back to false. - p2, _, _ := m.findPaneAndTab("p2") + p2, _, _ := got.findPaneAndTab("p2") p2.unseen = true // A message that is not local input — the shared spinner tick — must not - // ack the mark, however many of them arrive. - updated, _ := m.Update(workSpinnerTickMsg{}) - got := updated.(Model) - p2after, _, _ := got.findPaneAndTab("p2") + // ack the mark, however many of them arrive, and must not report anything. + updated, _ = got.Update(workSpinnerTickMsg{}) + got2 := updated.(Model) + p2after, _, _ := got2.findPaneAndTab("p2") if p2after == nil || !p2after.unseen { t.Fatal("unseen was acknowledged by a non-input message; it must survive until local input arrives") } - if !got.remoteFocusUnacked { + if !got2.remoteFocusUnacked { t.Fatal("remoteFocusUnacked was cleared by a non-input message") } + if len(fake.sent) != 0 { + t.Fatalf("%d message(s) sent to the daemon while unacked, want 0 (no pane_seen echo, no unseen report)", len(fake.sent)) + } // A real keystroke both clears the hold and acknowledges the pane, in the - // same Update call. - updated, _ = got.Update(tea.KeyPressMsg{Text: "x"}) - got = updated.(Model) - p2final, _, _ := got.findPaneAndTab("p2") + // same Update call — which reports the clear back (MsgUpdatePane{Unseen:false}). + updated, _ = got2.Update(tea.KeyPressMsg{Text: "x"}) + got3 := updated.(Model) + p2final, _, _ := got3.findPaneAndTab("p2") if p2final == nil || p2final.unseen { t.Error("unseen should be cleared once local input arrives") } - if got.remoteFocusUnacked { + if got3.remoteFocusUnacked { t.Error("remoteFocusUnacked should be cleared by the keystroke") } - _ = fake + if len(fake.sent) != 1 { + t.Fatalf("sends after the acknowledging keystroke = %d, want 1 (the unseen report)", len(fake.sent)) + } + var payload ipc.UpdatePanePayload + if err := fake.sent[0].DecodePayload(&payload); err != nil { + t.Fatalf("decode sent payload: %v", err) + } + if payload.PaneID != "p2" || payload.Unseen == nil || *payload.Unseen { + t.Errorf("sent payload = %+v, want {PaneID: p2, Unseen: false}", payload) + } +} + +// TestTypingGuard_WheelNeverGuarded pins the key-vs-mouse ruling: wheel input +// goes straight to enqueueInput (sendInputToPane), bypassing the typing guard +// entirely, because a remote tab switch says nothing about where the mouse +// pointer is now. Calls sendInputToPane directly, matching this file's own +// precedent for the wheel path (TestSendInputToPane_SharesTheKeystrokeQueue +// in input_order_test.go) rather than driving a full mouse-rect Update. +func TestTypingGuard_WheelNeverGuarded(t *testing.T) { + t.Parallel() + t0 := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + m, _ := typingGuardModel(t, t0, "t1", "t2") + + updated, _ := m.Update(typingGuardBroadcast("t2", "t1", "t2")) + got := updated.(Model) + if got.guardPaneID != "p1" { + t.Fatalf("guardPaneID = %q, want p1", got.guardPaneID) + } + got.now = func() time.Time { return t0.Add(100 * time.Millisecond) } // well within the window + + got.sendInputToPane("p2", []byte("\x1b[<64;1;1M")) // one wheel-up notch + + in := drainOneInput(t, &got) + if in.paneID != "p2" { + t.Errorf("wheel notch queued paneID = %q, want p2 (unredirected) — the typing guard must never apply to mouse input", in.paneID) + } +} + +// TestTypingGuard_PasteWithinWindowGoesToOldPane drives the paste path +// through Update (tea.PasteMsg), the second key-originated producer besides +// typed keys (spec §8.1: "paste counts as a key here"). +func TestTypingGuard_PasteWithinWindowGoesToOldPane(t *testing.T) { + t.Parallel() + t0 := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + m, _ := typingGuardModel(t, t0, "t1", "t2") + + updated, _ := m.Update(typingGuardBroadcast("t2", "t1", "t2")) + got := updated.(Model) + if got.guardPaneID != "p1" { + t.Fatalf("guardPaneID = %q, want p1", got.guardPaneID) + } + got.now = func() time.Time { return t0.Add(100 * time.Millisecond) } + + updated, _ = got.Update(tea.PasteMsg{Content: "pasted"}) + got2 := updated.(Model) + + in := drainOneInput(t, &got2) + if in.paneID != "p1" { + t.Errorf("pasted into %q, want p1 (the pane active before the remote switch)", in.paneID) + } +} + +// TestTypingGuard_PasteEncodesForTheRedirectedPane pins review round 1's +// Important 3: the paste must be ENCODED against the pane it actually reaches +// (bracketed-paste mode), not the pane that was active when the paste was +// requested. Encoding against the wrong pane's mode and then redirecting the +// bytes sends an unbracketed multi-line paste into a shell character by +// character (executing every line but the last), or a bracketed one into an +// app that never enabled the mode. +func TestTypingGuard_PasteEncodesForTheRedirectedPane(t *testing.T) { + t.Parallel() + t0 := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + m, _ := typingGuardModel(t, t0, "t1", "t2") + + p1, _, _ := m.findPaneAndTab("p1") + p1.bracketedPasteSeen = true + p1.bracketedPaste = true // the REDIRECTED pane wants bracketed paste + // p2 (the pane superficially "active" when the paste is requested) is + // left at its zero value — bracketed paste NOT enabled. + + updated, _ := m.Update(typingGuardBroadcast("t2", "t1", "t2")) + got := updated.(Model) + if got.guardPaneID != "p1" { + t.Fatalf("guardPaneID = %q, want p1", got.guardPaneID) + } + got.now = func() time.Time { return t0.Add(100 * time.Millisecond) } + + updated, _ = got.Update(tea.PasteMsg{Content: "hello\nworld"}) + got2 := updated.(Model) + + in := drainOneInput(t, &got2) + if in.paneID != "p1" { + t.Fatalf("pasted into %q, want p1", in.paneID) + } + if !strings.HasPrefix(string(in.data), pasteStart) { + t.Errorf("paste data = %q, want bracketed (p1's own mode) — "+ + "encoding against p2's mode instead sent it unbracketed", string(in.data)) + } +} + +// TestTypingGuard_KeySideEffectsActOnTheRedirectedPane pins the rest of +// Important 3: ResetScroll, answerBlockedByInput and interruptWorkingPane +// must act on whichever pane the keystroke actually reaches, not on +// tab.ActivePaneModel() — crediting the wrong pane with input it never +// received (a parked prompt on the redirected pane stays parked; the WRONG +// pane's scrollback resets instead). +func TestTypingGuard_KeySideEffectsActOnTheRedirectedPane(t *testing.T) { + t.Parallel() + t0 := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + m, _ := typingGuardModel(t, t0, "t1", "t2") + + p1, _, _ := m.findPaneAndTab("p1") + p1.blockedSince = t0.Add(-time.Hour) + p1.blockedReason = "Bash" + p1.scrollBack = 42 + + updated, _ := m.Update(typingGuardBroadcast("t2", "t1", "t2")) + got := updated.(Model) + if got.guardPaneID != "p1" { + t.Fatalf("guardPaneID = %q, want p1", got.guardPaneID) + } + got.now = func() time.Time { return t0.Add(100 * time.Millisecond) } + + updated, _ = got.Update(tea.KeyPressMsg{Text: "x"}) + got2 := updated.(Model) + + p1after, _, _ := got2.findPaneAndTab("p1") + if !p1after.blockedSince.IsZero() { + t.Error("p1.blockedSince should be cleared — the keystroke was redirected to it") + } + if p1after.blockedReason != "" { + t.Errorf("p1.blockedReason = %q, want empty", p1after.blockedReason) + } + if p1after.scrollBack != 0 { + t.Errorf("p1.scrollBack = %d, want 0 (ResetScroll should act on the redirected pane)", p1after.scrollBack) + } +} + +// TestTypingGuard_ClickClearsFlag pins the mouse half of "local input answers +// the ack hold" (spec §8.1): a mouse click, not only a keystroke, clears +// remoteFocusUnacked. Driven through Update with a real tea.MouseClickMsg. +func TestTypingGuard_ClickClearsFlag(t *testing.T) { + t.Parallel() + t0 := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + m, _ := typingGuardModel(t, t0, "t1", "t2") + + updated, _ := m.Update(typingGuardBroadcast("t2", "t1", "t2")) + got := updated.(Model) + if !got.remoteFocusUnacked { + t.Fatal("remoteFocusUnacked should be set by the remote switch") + } + + updated, _ = got.Update(tea.MouseClickMsg{X: 10, Y: 5, Button: tea.MouseLeft}) + got2 := updated.(Model) + + if got2.remoteFocusUnacked { + t.Error("remoteFocusUnacked should be cleared by a mouse click") + } +} + +// TestTypingGuard_OverlayVisibleGuardsTheOverlayPane pins review round 1's +// Important 4/minor 4: when the FROM tab has a visible overlay (lazygit, say) +// the user was typing into, the guard must protect the OVERLAY pane, not the +// tree pane sitting behind it — ActivePaneModel's own rule for who owns +// keyboard input while an overlay is up. +func TestTypingGuard_OverlayVisibleGuardsTheOverlayPane(t *testing.T) { + t.Parallel() + t0 := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + m, _ := typingGuardModel(t, t0, "t1", "t2") + + tab1 := m.curTabs()[0] // t1, the FROM tab + overlay := NewPaneModel("overlay-1", 1024) + tab1.overlayPane = overlay + tab1.overlayVisible = true + + state := WorkspaceStateMsg{ + ActiveTab: "t2", + Tabs: []TabInfo{ + {ID: "t1", Name: "t1", Panes: []string{"p1", "overlay-1"}}, + {ID: "t2", Name: "t2", Panes: []string{"p2"}}, + }, + Panes: []PaneInfo{ + {ID: "p1", TabID: "t1"}, + {ID: "overlay-1", TabID: "t1", Overlay: true}, + {ID: "p2", TabID: "t2"}, + }, + } + updated, _ := m.Update(state) + got := updated.(Model) + + if got.guardPaneID != "overlay-1" { + t.Errorf("guardPaneID = %q, want overlay-1 — the user was typing into the overlay, not the tree pane behind it", got.guardPaneID) + } +} + +// TestTypingGuard_CreateTabTokenSpentOnAnExistingTab pins review round 1's +// minor 1: pendingTabCreateToken must match ONLY a genuinely new tab id (one +// absent before this broadcast). A switch to an EXISTING tab beating the +// create's own landing to the wire must be treated as remote, and the +// one-shot token must be spent regardless — checked directly against +// requestedTab, since a leftover token could otherwise wrongly satisfy some +// LATER broadcast. +func TestTypingGuard_CreateTabTokenSpentOnAnExistingTab(t *testing.T) { + t.Parallel() + t0 := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + m, _ := typingGuardModel(t, t0, "t1", "t2") + proj := m.cur() + key := requestedTabKey("", proj.ID) + m.recordRequestedTab("", proj.ID, pendingTabCreateToken) + + updated, _ := m.Update(typingGuardBroadcast("t2", "t1", "t2")) + got := updated.(Model) + + if got.guardPaneID != "p1" { + t.Fatalf("guardPaneID = %q, want p1 — a switch to an EXISTING tab must not be mistaken for the pending create", got.guardPaneID) + } + if _, stillSet := got.requestedTab[key]; stillSet { + t.Error("pendingTabCreateToken survived a broadcast it did not match — " + + "it must be spent either way, or it could wrongly satisfy a LATER broadcast") + } +} + +// TestTypingGuard_CreateTabTokenMatchesTheNewTab is the token's intended +// match: a genuinely new tab id landing while a create is pending is +// recognised as that create's own landing, and does not arm the guard. +func TestTypingGuard_CreateTabTokenMatchesTheNewTab(t *testing.T) { + t.Parallel() + t0 := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + m, _ := typingGuardModel(t, t0, "t1") + proj := m.cur() + m.recordRequestedTab("", proj.ID, pendingTabCreateToken) + + // A brand new tab (never seen before) becomes active — this client's own + // create_tab landing. + updated, _ := m.Update(typingGuardBroadcast("t2", "t1", "t2")) + got := updated.(Model) + + if got.guardPaneID != "" { + t.Errorf("guardPaneID = %q, want empty — a genuinely new tab must be recognised as the pending create's own landing", got.guardPaneID) + } +} + +// TestArmReattachReset_ClearsTypingGuardStateForThatDest pins review round +// 1's minor 3: a reattach replaces its destination's whole state, so a +// pending requestedTab token can never land the broadcast it was waiting for, +// and a guardPaneID naming one of that destination's panes is redirecting +// input toward a pane about to be rebuilt out from under it. +func TestArmReattachReset_ClearsTypingGuardStateForThatDest(t *testing.T) { + t.Parallel() + t0 := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + m, _ := typingGuardModel(t, t0, "t1", "t2") + + updated, _ := m.Update(typingGuardBroadcast("t2", "t1", "t2")) + got := updated.(Model) + if got.guardPaneID != "p1" { + t.Fatalf("setup: guardPaneID = %q, want p1", got.guardPaneID) + } + got.recordRequestedTab("", "some-other-project", "some-tab") + + got.armReattachReset("") + + if got.guardPaneID != "" { + t.Errorf("guardPaneID = %q after reattach reset, want empty", got.guardPaneID) + } + if got.remoteFocusUnacked { + t.Error("remoteFocusUnacked should be cleared by a reattach of its own destination") + } + if _, ok := got.requestedTab[requestedTabKey("", "some-other-project")]; ok { + t.Error("requestedTab entries for the reattaching dest should be cleared") + } +} + +// TestListenForMessages_DecodesEventDismissedWithOrigin pins the wire decode +// for spec §8.4's dismissal broadcast, including that Origin (the sending +// daemon, stamped by the router) survives into eventDismissedMsg.dest. +func TestListenForMessages_DecodesEventDismissedWithOrigin(t *testing.T) { + t.Parallel() + wire, err := ipc.NewMessage(ipc.MsgEventDismissed, ipc.EventDismissedPayload{EventID: "evt-9"}) + if err != nil { + t.Fatalf("build message: %v", err) + } + wire.Origin = "gpu01" + c := &scriptedConn{msgs: make(chan *ipc.Message, 1)} + c.msgs <- wire + + m := Model{client: c} + got := m.listenForMessages()() + msg, ok := got.(eventDismissedMsg) + if !ok { + t.Fatalf("msg is %T, want eventDismissedMsg", got) + } + if msg.dest != "gpu01" { + t.Errorf("dest = %q, want gpu01", msg.dest) + } + if msg.eventID != "evt-9" { + t.Errorf("eventID = %q, want evt-9", msg.eventID) + } +} + +// TestListenForMessages_DecodesPaneSeenWithOrigin is the pane_seen twin. +func TestListenForMessages_DecodesPaneSeenWithOrigin(t *testing.T) { + t.Parallel() + wire, err := ipc.NewMessage(ipc.MsgPaneSeen, ipc.PaneSeenPayload{PaneID: "pane-7"}) + if err != nil { + t.Fatalf("build message: %v", err) + } + wire.Origin = "gpu01" + c := &scriptedConn{msgs: make(chan *ipc.Message, 1)} + c.msgs <- wire + + m := Model{client: c} + got := m.listenForMessages()() + msg, ok := got.(paneSeenMsg) + if !ok { + t.Fatalf("msg is %T, want paneSeenMsg", got) + } + if msg.dest != "gpu01" { + t.Errorf("dest = %q, want gpu01", msg.dest) + } + if msg.paneID != "pane-7" { + t.Errorf("paneID = %q, want pane-7", msg.paneID) + } } // TestEventDismissed_RemovesCard pins the client-side half of spec §8.4's // dismissal broadcast: a named id removes that one card, and an empty id // clears the whole sidebar list — mirroring the daemon's own DismissEventPayload -// "" = all convention. +// "" = all convention. Both events resolve to the same ("") dest here (no +// projects are set up, so destOfPane falls back to activeDest, "" either +// way) — TestEventDismissed_DismissAllScopedToDest covers the multi-dest case. func TestEventDismissed_RemovesCard(t *testing.T) { t.Parallel() m := Model{ @@ -268,6 +683,41 @@ func TestEventDismissed_RemovesCard(t *testing.T) { } } +// TestEventDismissed_DismissAllScopedToDest pins review round 1's Important +// 4: a dismiss-all ("" event id) from ONE destination must remove only that +// destination's cards, not every attached daemon's. destOfPane resolves each +// stored event's pane through the client's projects, so this test sets up +// one project per destination, each holding the pane the matching event +// names. +func TestEventDismissed_DismissAllScopedToDest(t *testing.T) { + t.Parallel() + tabA := NewTabModel("t-a", "A") + tabA.Root = NewLeaf(NewPaneModel("pane-a", 1024)) + tabB := NewTabModel("t-b", "B") + tabB.Root = NewLeaf(NewPaneModel("pane-b", 1024)) + m := Model{ + cfg: config.Default(), + client: &fakeSender{}, + notifications: NewNotificationCenter(30, 50), + projects: []*ProjectModel{ + {ID: "proj-a", Dest: "hostA", tabs: []*TabModel{tabA}}, + {ID: "proj-b", Dest: "hostB", tabs: []*TabModel{tabB}}, + }, + } + m.notifications.AddEvent(ipc.PaneEventPayload{ID: "evt-a", PaneID: "pane-a", Type: "bell"}) + m.notifications.AddEvent(ipc.PaneEventPayload{ID: "evt-b", PaneID: "pane-b", Type: "bell"}) + + updated, _ := m.Update(eventDismissedMsg{dest: "hostA", eventID: ""}) + got := updated.(Model) + + if n := got.notifications.Count(); n != 1 { + t.Fatalf("count after dismiss-all from hostA = %d, want 1", n) + } + if got.notifications.events[0].ID != "evt-b" { + t.Errorf("remaining event = %q, want evt-b (hostB's card must survive hostA's dismiss-all)", got.notifications.events[0].ID) + } +} + // TestPaneSeen_ClearsMarkWithoutEcho pins spec §8.4's unseen-clear broadcast: // this client clears its own local mark and sends nothing back — echoing // would have every attached client answer each other's clears forever. From 954ec4a834ca2c59fbbd7075d5d19cc0a4520fbb Mon Sep 17 00:00:00 2001 From: Artjoms Stukans Date: Sat, 26 Sep 2026 01:16:16 +0200 Subject: [PATCH 21/40] fix(tui): retire the create token on no-op broadcasts, reject stale from-tab echoes MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Two more typing-guard correctness gaps found in review round 2, both in applyTabMoveGuard now that it runs on every broadcast rather than only on a detected move (round 1's fix for the stale-token bug): - The pending create_tab token was deleted unconditionally the moment it was inspected, then checked against existedBefore. An ordinary broadcast that changes nothing (the git ticker, an OSC 7 CWD update, another client's unrelated action) reports the SAME active tab this client was already on, which trivially "existed before" — so it silently spent the token, and the create's own tab landing moments later read as a stranger's remote switch: a false flash and 250ms of redirected typing right after Ctrl+T. The token is now kept whenever the active tab hasn't actually changed, and spent only against a genuine change either way (its own landing, or something else that beat it there). - requestedTab's value is now a small struct (pendingSwitch: target, from, at) instead of a bare tab id. A broadcast already in flight when switchTab runs still names the tab this client just left; with no way to tell that apart from a genuine switch back to it, the tab visibly jumped to the old one for the width of one round trip before the requester's own echo corrected it. Recording the pre-switch tab and a timestamp lets applyTabMoveGuard reject a broadcast naming it outright — holding the active tab at what was requested, arming no guard, keeping the token — for up to requestedSwitchStaleWindow (2s). Past that the local switch is assumed lost and such a broadcast is adopted normally, guard included. applyTabMoveGuard now returns the tab id to actually treat as active alongside the guard cmd, since a rejected stale broadcast must not have its reported ActiveTab adopted into proj.activeTab at all. --- internal/tui/model.go | 174 +++++++++++++++++++++--------- internal/tui/typing_guard_test.go | 125 ++++++++++++++++++++- 2 files changed, 247 insertions(+), 52 deletions(-) diff --git a/internal/tui/model.go b/internal/tui/model.go index b4ca4b6c..7d53473a 100644 --- a/internal/tui/model.go +++ b/internal/tui/model.go @@ -527,14 +527,16 @@ type Model struct { // client count (bridges excluded) — what renderStatusBar's role marker // and D9's "most recent input" default both key off of at the TUI layer. clientCount map[string]int - // requestedTab records, per (dest, project) key (requestedTabKey), the tab - // id THIS client last asked for via switchTab/switchTabBy or sendCreateTab. - // applyWorkspaceState consults it to tell this client's own switch landing - // apart from a change some OTHER client made (typing guard, spec §8.1): - // a broadcast whose adopted ActiveTab equals the recorded value is this - // client's own request landing (the entry is cleared); anything else is - // remote and arms remoteSwitchAt/guardPaneID/remoteFocusUnacked below. - requestedTab map[string]string + // requestedTab records, per (dest, project) key (requestedTabKey), THIS + // client's own in-flight switchTab/switchTabBy/sendCreateTab request. + // applyTabMoveGuard consults it to tell this client's own switch landing + // apart from a change some OTHER client made (typing guard, spec §8.1): a + // broadcast whose adopted ActiveTab equals pendingSwitch.target is this + // client's own request landing (the entry is cleared); one naming + // pendingSwitch.from within the stale window is a network-ordering + // leftover from before the request and is rejected outright; anything + // else is remote and arms remoteSwitchAt/guardPaneID/remoteFocusUnacked. + requestedTab map[string]pendingSwitch // remoteSwitchAt is m.clock() at the last REMOTE active-tab change this // client observed for its active project — the typing guard's window // (remoteSwitchGuardWindow) is measured from here. @@ -1274,7 +1276,7 @@ func NewModel(client Client, cfg config.Config, version string, registry *plugin inputCh: make(chan paneInput, inputForwardBuffer), inputDone: make(chan struct{}), inputIdle: make(chan struct{}), - requestedTab: make(map[string]string), + requestedTab: make(map[string]pendingSwitch), now: time.Now, } // Startup dialog priority: migration > what's-new > update-notice > @@ -4478,7 +4480,7 @@ func (m Model) sendCreateTab(spec *ipc.FirstPaneSpec) tea.Cmd { // since it always looks up (dest, THAT dest's own active project). Better // to record nothing than to record a key that can only ever be wrong. if proj := m.cur(); proj != nil && proj.Dest == dest && m.requestedTab != nil { - m.requestedTab[requestedTabKey(dest, proj.ID)] = pendingTabCreateToken + m.requestedTab[requestedTabKey(dest, proj.ID)] = pendingSwitch{target: pendingTabCreateToken, at: m.clock()} } return func() tea.Msg { msg, err := ipc.NewMessage(ipc.MsgCreateTab, ipc.CreateTabPayload{ @@ -6120,52 +6122,81 @@ func tabInputPaneID(tab *TabModel) string { return tab.ActivePane } -// applyTabMoveGuard decides whether an active-tab change applyWorkspaceState -// just observed for the ACTIVE project is this client's own request landing -// or one another attached client made, and arms the typing guard for the -// latter (spec §8.1). Returns the flash-expiry cmd when it arms, nil -// otherwise — the caller must batch it, or the flash never clears itself. +// applyTabMoveGuard decides, for the ACTIVE project, what active tab this +// broadcast should actually settle on — the daemon's own report, this +// client's pending request held in place instead (a stale echo), or the +// daemon's report adopted with the typing guard armed (spec §8.1) — and +// returns that tab id alongside the flash-expiry cmd when the guard armed +// (nil otherwise; the caller must batch a non-nil one, or the flash never +// clears itself). // // Called UNCONDITIONALLY for the active project on every broadcast, not only // when the tab actually moved: a token must be retired on an ORDINARY echo -// too (fromTab == targetTab already, because switchTab updates the client's -// own index synchronously before any broadcast can land) — leaving a matched -// token in place would let it silently satisfy some LATER, unrelated -// broadcast that happens to name the same tab id. +// too (fromTab.ID == newActiveTab already, because switchTab updates the +// client's own index synchronously before any broadcast can land) — leaving +// a matched token in place would let it silently satisfy some LATER, +// unrelated broadcast that happens to name the same tab id. The SAME reason +// is why the create-token case below must not spend itself on a broadcast +// that changes nothing (review round 2): applyTabMoveGuard now runs on every +// broadcast including the ordinary ones a pending create sits through (the +// git ticker, an OSC 7 CWD update, another client's unrelated action) before +// its own tab ever lands. // // "Not requested" is decided with the requestedTab TOKEN, never with a time // window: a local switch followed quickly by an unrelated remote one must // still be guarded, which a window alone cannot tell apart from the local -// switch's own delayed echo. +// switch's own delayed echo. The one exception is pendingSwitch.from within +// requestedSwitchStaleWindow (review round 2's pre-existing issue): a +// broadcast already in flight when switchTab ran still names the tab this +// client just left, and adopting it would jump the tab visibly back for the +// width of one round trip before the requester's own echo corrects it — so +// that report is rejected outright rather than merely un-guarded. // // existedBefore reports whether newActiveTab was already one of this // client's tabs (any project) before this broadcast — the create-token's // only use for it, since a create_tab request has no id to compare by // equality ahead of time (the daemon mints one). -func (m *Model) applyTabMoveGuard(dest, projectID, newActiveTab string, fromTab, targetTab *TabModel, tabs []*TabModel, existedBefore bool) tea.Cmd { +func (m *Model) applyTabMoveGuard(dest, projectID, newActiveTab string, fromTab *TabModel, tabs []*TabModel, existedBefore bool) (string, tea.Cmd) { key := requestedTabKey(dest, projectID) if req, ok := m.requestedTab[key]; ok { switch { - case req == pendingTabCreateToken: - // One-shot: spent against the very next active-tab change either - // way, so a leftover token can never outlive the create it was - // minted for and silently swallow some LATER remote switch. + case req.target == pendingTabCreateToken: + if fromTab != nil && fromTab.ID == newActiveTab { + // Nothing has changed yet: an ordinary broadcast landed + // before the pending create's own tab. Keep waiting — the + // token must not be spent on a broadcast that names the + // tab we were ALREADY on. + return newActiveTab, nil + } + // One-shot from here: spent against this active-tab change + // either way, so a leftover token can never outlive the create + // it was minted for and silently swallow some LATER remote + // switch. delete(m.requestedTab, key) if !existedBefore { - return nil // the create's own tab landing + return newActiveTab, nil // the create's own tab landing } // Not the create landing — an unrelated change beat it there. // Fall through to the ordinary remote-switch handling below. - case req == newActiveTab: + case req.target == newActiveTab: // An exact match is this client's own switchTab landing, echo or // not — see the function comment for why this compare must run // unconditionally rather than only inside a "moved" branch. delete(m.requestedTab, key) - return nil + return newActiveTab, nil + case req.from != "" && req.from == newActiveTab && m.clock().Sub(req.at) < requestedSwitchStaleWindow: + // This broadcast reports the tab we just switched AWAY FROM — + // network-ordering leftover from before the request, not a + // switch back to it (a genuine one arrives, if it happens at + // all, only after this client's own confirmation, by which + // point the token above is gone and this case cannot match). + // Reject it outright: hold the tab at what we asked for, keep + // the token, arm no guard. + return req.target, nil } } - if fromTab == nil || fromTab == targetTab { - return nil // nothing moved, or this project's very first broadcast + if fromTab == nil || fromTab.ID == newActiveTab { + return newActiveTab, nil // nothing moved, or this project's very first broadcast } // A vanished source tab (this client's own Ctrl+W, Move to project, or a // dissolve/recovery that moved its last pane out) is not "switched away @@ -6173,13 +6204,13 @@ func (m *Model) applyTabMoveGuard(dest, projectID, newActiveTab string, fromTab, // the tab away from under itself. Only arm when fromTab is still part of // the broadcast's tab list for this project. if !tabsContainID(tabs, fromTab.ID) { - return nil + return newActiveTab, nil } m.remoteSwitchAt = m.clock() m.guardPaneID = tabInputPaneID(fromTab) m.remoteFocusUnacked = true m.setFlash("Tab switched by another client") - return m.flashCmd() + return newActiveTab, m.flashCmd() } // applyWorkspaceState rebuilds the TUI state from one daemon's broadcast. @@ -6322,21 +6353,28 @@ func (m *Model) applyWorkspaceState(state WorkspaceStateMsg, dest string) ([]str proj.Offline = nil tabs, projPaneIDs, projResizeCmds := m.rebuildTabs(info, state, existingTabs, existingPanes, paneMap, dest) proj.tabs = tabs - proj.activeTab = indexOfTab(proj.tabs, info.ActiveTab) - targetTab := tabAt(proj.tabs, proj.activeTab) // Typing guard (spec §8.1), scoped to the ACTIVE project only: a // background project's active tab moving under it is not something - // anyone is typing into right now. Called even when fromTab == targetTab - // (an ordinary echo) — see applyTabMoveGuard's own comment for why the - // requestedTab token must be retired on that path too, not only inside - // the "moved" branch below. ok (the project already existed) is what + // anyone is typing into right now. Called even when nothing moved (an + // ordinary echo, or an unrelated broadcast while a request is pending) + // — see applyTabMoveGuard's own comment for why the requestedTab token + // must be retired (or deliberately KEPT) on those paths too, not only + // inside a "moved" branch. ok (the project already existed) is what // makes the token lookup meaningful; a brand new project has none. + // effectiveActiveTab may differ from info.ActiveTab: a rejected stale + // broadcast (pendingSwitch.from) holds the tab at this client's own + // pending request instead of adopting the daemon's report. + effectiveActiveTab := info.ActiveTab if info.ID == activeID && ok { _, existedBefore := existingTabs[info.ActiveTab] - if cmd := m.applyTabMoveGuard(dest, info.ID, info.ActiveTab, fromTab, targetTab, proj.tabs, existedBefore); cmd != nil { + var cmd tea.Cmd + effectiveActiveTab, cmd = m.applyTabMoveGuard(dest, info.ID, info.ActiveTab, fromTab, proj.tabs, existedBefore) + if cmd != nil { overlayResizeCmds = append(overlayResizeCmds, cmd) } } + proj.activeTab = indexOfTab(proj.tabs, effectiveActiveTab) + targetTab := tabAt(proj.tabs, proj.activeTab) if fromTab != targetTab { tabMoves = append(tabMoves, activeTabMove{from: fromTab, target: targetTab}) } @@ -7251,9 +7289,16 @@ func (m *Model) switchTab(idx int) tea.Cmd { // Typing guard (spec §8.1): this client asked for tabID, so the broadcast // that lands it must not be mistaken for another client's switch. Recorded // against the CURRENT project — target and its project share one Dest, so - // this is the same key applyWorkspaceState looks up. + // this is the same key applyWorkspaceState looks up. fromID lets a + // broadcast still describing the PRE-switch state (in flight when this + // ran) be recognised as stale rather than adopted as a switch back — see + // requestedSwitchStaleWindow. if proj := m.cur(); proj != nil { - m.recordRequestedTab(dest, proj.ID, tabID) + fromID := "" + if from != nil { + fromID = from.ID + } + m.recordRequestedTab(dest, proj.ID, tabID, fromID) } cmds := []tea.Cmd{func() tea.Msg { msg, _ := ipc.NewMessage(ipc.MsgSwitchTab, ipc.SwitchTabPayload{ @@ -7280,7 +7325,19 @@ func (m *Model) switchTab(idx int) tea.Cmd { // before it, not the pane the switch made active. const remoteSwitchGuardWindow = 250 * time.Millisecond -// pendingTabCreateToken marks requestedTab[key] while this client's own +// requestedSwitchStaleWindow bounds how long a broadcast naming the tab a +// pending LOCAL switch moved AWAY FROM (pendingSwitch.from) is rejected as a +// stale echo of the pre-switch state, rather than adopted as a genuine switch +// back to it. review round 2: a broadcast already in flight when switchTab +// runs still names the OLD active tab; without this bound, that broadcast is +// indistinguishable from another client genuinely switching back, and the +// tab visibly jumped to the old one for the width of one round trip before +// the requester's own echo corrected it. Past the bound the local switch is +// assumed lost (never reached the daemon, or was overtaken) and a broadcast +// naming `from` is adopted normally, guard included. +const requestedSwitchStaleWindow = 2 * time.Second + +// pendingTabCreateToken marks pendingSwitch.target while this client's own // create_tab is in flight for that project. Unlike switchTab, sendCreateTab // has no id to record ahead of time — the daemon mints the new tab's id — so // there is nothing to compare the eventual broadcast's ActiveTab against by @@ -7289,6 +7346,23 @@ const remoteSwitchGuardWindow = 250 * time.Millisecond // whatever tab the daemon makes active next. const pendingTabCreateToken = "\x00pending-create" +// pendingSwitch is requestedTab's value — see that field's comment for the +// decisions it drives. +type pendingSwitch struct { + // target is the tab id this client asked for (switchTab), or + // pendingTabCreateToken when the daemon has not minted one yet + // (sendCreateTab). + target string + // from is the tab this project was showing right before the request — + // "" for a create, which never moves this client's own active tab ahead + // of the daemon's answer, so there is no "pre-request state" a broadcast + // could stale-echo. See requestedSwitchStaleWindow. + from string + // at is m.clock() when the request was recorded — requestedSwitchStaleWindow + // is measured from here. + at time.Time +} + // requestedTabKey identifies one project on one destination for // Model.requestedTab. Dest alone is not unique (each daemon mints its own // project ids independently) and a project id alone is not unique across @@ -7298,15 +7372,17 @@ func requestedTabKey(dest, projectID string) string { return dest + "\x00" + projectID } -// recordRequestedTab notes that THIS client is the one asking for tabID to -// become the active tab of (dest, projectID). Called by switchTab/ -// switchTabBy with the tab they are switching TO, and by sendCreateTab with -// pendingTabCreateToken. See requestedTab's field comment. -func (m *Model) recordRequestedTab(dest, projectID, tabID string) { +// recordRequestedTab notes that THIS client is the one asking for target to +// become the active tab of (dest, projectID), moving there FROM the tab this +// project was showing before (from — "" when not applicable, e.g. a create). +// Called by switchTab/switchTabBy with the tab they are switching TO and the +// tab they are leaving, and by sendCreateTab with pendingTabCreateToken and +// no from. See requestedTab's field comment. +func (m *Model) recordRequestedTab(dest, projectID, target, from string) { if m.requestedTab == nil { - m.requestedTab = make(map[string]string) + m.requestedTab = make(map[string]pendingSwitch) } - m.requestedTab[requestedTabKey(dest, projectID)] = tabID + m.requestedTab[requestedTabKey(dest, projectID)] = pendingSwitch{target: target, from: from, at: m.clock()} } // clock returns the typing guard's current time: m.now when the Model has diff --git a/internal/tui/typing_guard_test.go b/internal/tui/typing_guard_test.go index e4cea22d..bc97bb6a 100644 --- a/internal/tui/typing_guard_test.go +++ b/internal/tui/typing_guard_test.go @@ -535,7 +535,7 @@ func TestTypingGuard_CreateTabTokenSpentOnAnExistingTab(t *testing.T) { m, _ := typingGuardModel(t, t0, "t1", "t2") proj := m.cur() key := requestedTabKey("", proj.ID) - m.recordRequestedTab("", proj.ID, pendingTabCreateToken) + m.recordRequestedTab("", proj.ID, pendingTabCreateToken, "") updated, _ := m.Update(typingGuardBroadcast("t2", "t1", "t2")) got := updated.(Model) @@ -557,7 +557,7 @@ func TestTypingGuard_CreateTabTokenMatchesTheNewTab(t *testing.T) { t0 := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) m, _ := typingGuardModel(t, t0, "t1") proj := m.cur() - m.recordRequestedTab("", proj.ID, pendingTabCreateToken) + m.recordRequestedTab("", proj.ID, pendingTabCreateToken, "") // A brand new tab (never seen before) becomes active — this client's own // create_tab landing. @@ -569,6 +569,125 @@ func TestTypingGuard_CreateTabTokenMatchesTheNewTab(t *testing.T) { } } +// TestTypingGuard_CreateTabTokenSurvivesANoOpBroadcast pins review round 2's +// new Important finding: applyTabMoveGuard now runs on EVERY broadcast for +// the active project (fix round 1's Important 1), including ordinary ones +// that change nothing — the git ticker, an OSC 7 CWD update, another +// client's unrelated action. Before this fix, the pending-create branch +// deleted its token unconditionally and only THEN checked existedBefore, so +// a no-op broadcast (existedBefore trivially true — it's the tab we were +// already on) spent it — and the create's own tab, landing moments later, +// read as a stranger's remote switch: a false flash and 250 ms of redirected +// typing right after Ctrl+T. +func TestTypingGuard_CreateTabTokenSurvivesANoOpBroadcast(t *testing.T) { + t.Parallel() + t0 := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + m, _ := typingGuardModel(t, t0, "t1") + proj := m.cur() + m.recordRequestedTab("", proj.ID, pendingTabCreateToken, "") + + // An ordinary broadcast, same active tab, lands before the create's own. + updated, _ := m.Update(typingGuardBroadcast("t1", "t1")) + got := updated.(Model) + if got.guardPaneID != "" { + t.Fatalf("guardPaneID = %q after a no-op broadcast, want empty", got.guardPaneID) + } + key := requestedTabKey("", proj.ID) + if _, ok := got.requestedTab[key]; !ok { + t.Fatal("pendingTabCreateToken was spent by a broadcast that changed nothing") + } + + // The create's own tab lands next — the token must still be there to + // recognise it. + updated, _ = got.Update(typingGuardBroadcast("t2", "t1", "t2")) + got2 := updated.(Model) + if got2.guardPaneID != "" { + t.Errorf("guardPaneID = %q, want empty — the surviving token must still recognise its own create's landing", got2.guardPaneID) + } +} + +// TestTypingGuard_StaleFromTabBroadcastIsRejectedWhilePending pins review +// round 2's pre-existing issue: a broadcast already in flight when switchTab +// runs still names the tab this client just left. Adopting it — as the +// pre-fix code did, since it had no way to tell "the daemon really switched +// back" from "this is leftover from before my own switch" — visibly jumped +// the tab back to the old one for the width of one round trip before the +// requester's own echo corrected it, and armed a false guard/flash on top. +func TestTypingGuard_StaleFromTabBroadcastIsRejectedWhilePending(t *testing.T) { + t.Parallel() + t0 := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + m, _ := typingGuardModel(t, t0, "t1", "t2") + + updated, _ := m.Update(altKey('2')) // requests t2, leaving t1 + got := updated.(Model) + if got.activeTabModel().ID != "t2" { + t.Fatalf("setup: active tab = %q, want t2", got.activeTabModel().ID) + } + + // A broadcast already in flight before the switch still names t1 — the + // PRE-switch state, not a real switch back. Well within the stale bound. + got.now = func() time.Time { return t0.Add(500 * time.Millisecond) } + updated, _ = got.Update(typingGuardBroadcast("t1", "t1", "t2")) + got2 := updated.(Model) + + if got2.activeTabModel().ID != "t2" { + t.Errorf("active tab = %q after the stale t1 broadcast, want t2 (unchanged)", got2.activeTabModel().ID) + } + if got2.guardPaneID != "" { + t.Errorf("guardPaneID = %q, want empty — a stale echo of the pre-switch state must not arm the guard", got2.guardPaneID) + } + if got2.flashText != "" { + t.Errorf("flashText = %q, want empty — no false flash for a stale broadcast", got2.flashText) + } + + // The real confirmation lands next and must still be recognised — the + // stale broadcast above must not have consumed the token. + updated, _ = got2.Update(typingGuardBroadcast("t2", "t1", "t2")) + got3 := updated.(Model) + if got3.activeTabModel().ID != "t2" { + t.Fatalf("active tab = %q after the echo, want t2", got3.activeTabModel().ID) + } + if got3.guardPaneID != "" { + t.Errorf("guardPaneID = %q after the echo, want empty", got3.guardPaneID) + } + + // NOW a genuine remote switch to t1 (the token is spent, so nothing can + // mistake this for another stale echo) must be honoured and guarded. + updated, _ = got3.Update(typingGuardBroadcast("t1", "t1", "t2")) + got4 := updated.(Model) + if got4.activeTabModel().ID != "t1" { + t.Fatalf("active tab = %q after the genuine remote switch, want t1", got4.activeTabModel().ID) + } + if got4.guardPaneID != "p2" { + t.Errorf("guardPaneID = %q, want p2 (this client's own tab when the remote switch arrived)", got4.guardPaneID) + } +} + +// TestTypingGuard_StaleFromTabBroadcastIsAdoptedPastTheBound is the other +// side of requestedSwitchStaleWindow: once it elapses, the local switch is +// assumed lost (never reached the daemon, or was overtaken), and a broadcast +// naming the tab this client asked to leave is adopted normally — guard +// included — rather than held forever. +func TestTypingGuard_StaleFromTabBroadcastIsAdoptedPastTheBound(t *testing.T) { + t.Parallel() + t0 := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + m, _ := typingGuardModel(t, t0, "t1", "t2") + + updated, _ := m.Update(altKey('2')) // requests t2, leaving t1 + got := updated.(Model) + + got.now = func() time.Time { return t0.Add(3 * time.Second) } // past requestedSwitchStaleWindow + updated, _ = got.Update(typingGuardBroadcast("t1", "t1", "t2")) + got2 := updated.(Model) + + if got2.activeTabModel().ID != "t1" { + t.Errorf("active tab = %q, want t1 — past the bound the broadcast must be adopted, not held", got2.activeTabModel().ID) + } + if got2.guardPaneID != "p2" { + t.Errorf("guardPaneID = %q, want p2", got2.guardPaneID) + } +} + // TestArmReattachReset_ClearsTypingGuardStateForThatDest pins review round // 1's minor 3: a reattach replaces its destination's whole state, so a // pending requestedTab token can never land the broadcast it was waiting for, @@ -584,7 +703,7 @@ func TestArmReattachReset_ClearsTypingGuardStateForThatDest(t *testing.T) { if got.guardPaneID != "p1" { t.Fatalf("setup: guardPaneID = %q, want p1", got.guardPaneID) } - got.recordRequestedTab("", "some-other-project", "some-tab") + got.recordRequestedTab("", "some-other-project", "some-tab", "") got.armReattachReset("") From ba74b758c0d8ea9ad6ebcfb7bdf22be5d2fd48fe Mon Sep 17 00:00:00 2001 From: Artjoms Stukans Date: Sat, 26 Sep 2026 01:45:57 +0200 Subject: [PATCH 22/40] docs: describe multi-client sync and the size master Document the multi-client sync feature (Tasks 1-10 on this branch): the client registry and size-master election, the resize batching and per-pane size generation, the output hold on attach, layout sync by revision, the typing guard across a remote tab switch, and the MCP unicast targets and list_clients tool. - .claude/rules/daemon-lifecycle.md: new "Multi-client" section; name the registry (clientRegistry) in "ATTACHED clients vs CONNECTED conns", which used to describe a bare attachedConns set. - .claude/rules/tui-rendering.md: new "Multi-client" section (resize gates and batching, follower rendering, layout sync, typing guard); replace the two "Known limit" passages that described the bug this fixes with what actually landed. - .claude/CLAUDE.md: rewrite the layout-persistence and MsgResizePane invariant bullets for the new behaviour; bump the MCP tool count to 36 and add list_clients and the client field; fix a stale reference to the removed Daemon.clientCWD. - docs/features.md, docs/configuration.md (master_grace_minutes), docs/keybindings.md (client.take_control), docs/mcp.md (list_clients, the client field): user-facing documentation of the same feature. - changelog.d/added-multi-client-sync.md: release notes fragment. Verification: dev.sh test (internal/ipc, internal/daemon, internal/tui, internal/keymap, internal/config, cmd/quil), dev.sh test-race (internal/daemon, internal/tui), dev.sh vet, the integration-tagged daemon suite under golang:1.25, and the changelog fragment gate all pass. dev.sh docs-size reports every file within its limit. --- .claude/CLAUDE.md | 12 +- .claude/rules/daemon-lifecycle.md | 181 +++++++++++++++++++- .claude/rules/tui-rendering.md | 218 ++++++++++++++++++++++++- changelog.d/added-multi-client-sync.md | 9 + docs/configuration.md | 1 + docs/features.md | 59 ++++++- docs/keybindings.md | 10 ++ docs/mcp.md | 19 ++- 8 files changed, 487 insertions(+), 22 deletions(-) create mode 100644 changelog.d/added-multi-client-sync.md diff --git a/.claude/CLAUDE.md b/.claude/CLAUDE.md index 3f7366eb..04d026f1 100644 --- a/.claude/CLAUDE.md +++ b/.claude/CLAUDE.md @@ -125,7 +125,7 @@ Architecture: thin bridge between MCP JSON-RPC (stdio) and daemon IPC (socket). MCP SDK: `github.com/modelcontextprotocol/go-sdk` (official SDK, v1.4+). Typed tool handlers with struct-based input schemas. -35 MCP tools: `list_panes` (marks the caller's own pane `self`, reports `agent_state`), `read_pane_output` (ANSI-stripped), `send_to_pane` (`paste` for multi-line), `get_pane_status`, `create_pane` (the dialog's options: `name`, `toggles` by NAME, `resume_session_id`, `worktree_branch`, `sandbox_image`/`sandbox_auth`), `send_keys` (named key sequences), `restart_pane`, `screenshot_pane` (VT-emulated text screenshot), `switch_tab`, `list_tabs`, `destroy_pane`, `set_active_pane` (TUI cooperation), `close_tui` (TUI cooperation), `get_notifications` (non-blocking; carries `data.excerpt` with the triggering lines), `watch_notifications` (blocking, replaces polling; optional `since_timestamp` closes the race-on-registration window), `dismiss_notifications` (ack handled events from the agent side), `get_memory_report` (per-tab totals + Go-heap + PTY RSS), `get_pane_memory` (single pane detail); projects and tabs: `list_projects`, `create_project`, `update_project`, `switch_project`, `destroy_project`, `create_tab` (any first pane, in a named project, no focus steal), `create_from_template` (ordered panes, frozen args and prompts from templates.toml), `rename_tab`, `destroy_tab`, `rename_pane`; discovery: `list_plugins` (toggle names, `sandbox_available`), `list_sessions`, `list_hosts`; tasking: `delegate_task`, `get_task`, `wait_task`, `list_tasks`. +36 MCP tools: `list_panes` (marks the caller's own pane `self`, reports `agent_state`), `read_pane_output` (ANSI-stripped), `send_to_pane` (`paste` for multi-line), `get_pane_status`, `create_pane` (the dialog's options: `name`, `toggles` by NAME, `resume_session_id`, `worktree_branch`, `sandbox_image`/`sandbox_auth`), `send_keys` (named key sequences), `restart_pane`, `screenshot_pane` (VT-emulated text screenshot), `switch_tab`, `list_tabs`, `destroy_pane`, `set_active_pane` (TUI cooperation; optional `client` targets one attached TUI, default the most recently active), `close_tui` (TUI cooperation; same optional `client`, unicast rather than broadcast), `list_clients` (every attached client's id, `attached_at`, `cols`/`rows`, master flag, `last_input_at`, plus role/pid/exe when known — own version floor 1.80.0), `get_notifications` (non-blocking; carries `data.excerpt` with the triggering lines), `watch_notifications` (blocking, replaces polling; optional `since_timestamp` closes the race-on-registration window), `dismiss_notifications` (ack handled events from the agent side), `get_memory_report` (per-tab totals + Go-heap + PTY RSS), `get_pane_memory` (single pane detail); projects and tabs: `list_projects`, `create_project`, `update_project`, `switch_project`, `destroy_project`, `create_tab` (any first pane, in a named project, no focus steal), `create_from_template` (ordered panes, frozen args and prompts from templates.toml), `rename_tab`, `destroy_tab`, `rename_pane`; discovery: `list_plugins` (toggle names, `sandbox_available`), `list_sessions`, `list_hosts`; tasking: `delegate_task`, `get_task`, `wait_task`, `list_tasks`. IPC request-response: `Message.ID` field (omitempty, backward compatible) correlates requests with responses. Daemon responds to the requesting connection when `ID` is set, broadcasts when empty. **The six project mutations, `update_tab`, `destroy_tab` and `update_pane` answer an `OpRespPayload` ONLY when the request carries an ID** (`answerOp`, `internal/daemon/project_req.go`) — the TUI sets none and keeps getting nothing; answering its id-less sends would put a critical frame per keystroke-class message on its 64-slot queue, the 2026-08-09 disconnect shape. @@ -200,11 +200,11 @@ These hold regardless of which file you open. Violating one breaks something. - Platform-specific code uses `//go:build` tags (not `// +build`) - ConPTY API: `conpty.New(width, height, flags)` — 3 args, uses `Spawn()`, reads/writes directly on ConPty object - Bubble Tea v2 / Lipgloss v2 — import paths: `charm.land/bubbletea/v2`, `charm.land/lipgloss/v2`. View() returns `tea.View` struct (not string). KeyMsg is `tea.KeyPressMsg`. MouseMsg split into `tea.MouseClickMsg`, `tea.MouseWheelMsg`, `tea.MouseMotionMsg`, `tea.MouseReleaseMsg`. Clipboard via `internal/clipboard` (platform-native: Win32/pbcopy/xclip). Paste wraps in bracketed paste sequences (`\x1b[200~...\x1b[201~`). Mouse modifiers: `msg.Mod.Contains(tea.ModCtrl)`. Quit: `tea.Quit` (function value, not call) -- IPC protocol: 4-byte big-endian length prefix + JSON payload. Optional `ID` field for request-response correlation (MCP bridge). When `ID` is set, daemon responds to specific connection; when empty, broadcasts to all. `AttachPayload` carries an optional `CWD` field (omitempty) — the TUI sends `os.Getwd()` on attach so the daemon can spawn new panes/tabs in the client's directory rather than the daemon's frozen-at-spawn-time CWD. Stored in `Daemon.clientCWD` (atomic.Pointer for race-free cross-goroutine access) and consumed via `defaultCWD()` which validates with `os.Stat` + `EvalSymlinks` and falls back to the daemon's own `os.Getwd()` if the client value is empty/stale +- IPC protocol: 4-byte big-endian length prefix + JSON payload. Optional `ID` field for request-response correlation (MCP bridge). When `ID` is set, daemon responds to specific connection; when empty, broadcasts to all. `AttachPayload` carries an optional `CWD` field (omitempty) — the TUI sends `os.Getwd()` on attach so the daemon can spawn new panes/tabs in the client's directory rather than the daemon's frozen-at-spawn-time CWD. Recorded per attached client in the registry (`clientRecord.cwd`, `.claude/rules/daemon-lifecycle.md`'s "Multi-client" section) rather than one global `Daemon.clientCWD` — several TUIs on one daemon each have their own directory. `defaultCWD(conn)` resolves it through a chain (the requesting conn's own client record → the size master's → the most recently active client's → the daemon's own `os.Getwd()`), validating each candidate with `os.Stat` + `EvalSymlinks` before accepting it - `.gitignore` uses root-anchored patterns (`/quil`, `/quild`) to avoid matching `cmd/` directories - Pane layout uses a binary split tree (`LayoutNode` in `internal/tui/layout.go`) — each internal node has its own `SplitDir`, enabling mixed H/V splits (tmux-style). The tree is serialized to JSON and persisted in the daemon's `Tab.Layout` field for reconnect restoration -- Layout persistence: the TUI sends `MsgUpdateLayout` only when a tab's tree DIFFERS from the one the broadcast just reported (`diffLayouts`/`sendDiffedLayouts`, `internal/tui/model.go`); the same broadcast arm diffs pane sizes (`diffResizes`). Both are scoped to the broadcasting `Dest` — a broadcast is one daemon's full state and says nothing about another's tabs. Daemon stores the layout opaquely (no broadcast, to avoid a feedback loop). On reconnect, `applyWorkspaceState()` deserializes the tree and prunes missing panes. **The comparison is structural, never by bytes**: `MarshalLayout` emits a struct (declaration order) while `parseWorkspaceState` re-marshals a `map[string]any` (alphabetical order), so identical trees encode differently and a byte-diff matches only single-leaf tabs — which is invisible in a workspace that has none. Sending unconditionally is what overflowed the client's own 64-slot critical queue at 33 tabs + 36 panes and made the TUI close its connection and exit (2026-08-09); split-drag release still does a full sweep (`sendAllLayouts`/`resizeAllPanes`), where everything genuinely changed. The FIRST resize per pane is never suppressed (`Model.sizedOnce`, cleared by `armReattachReset`) because the daemon's own guard is `appliedCols/appliedRows`, which a PTY install zeroes -- **Every `MsgResizePane` this client produces is gated on `Model.terminalPaintable()`** (`m.width >= minTermWidth && m.height >= minTermHeight`) — `resizeAllPanes`, `diffResizes` and `overlayResizeCmd` are the only three producers, and `attachMessage` reports 0x0 below that threshold so `handleAttach`'s own 80x24 default stands. The gate belongs at the FAN-OUTS: `paneVTSize` floors both dimensions at 1 on purpose for genuinely narrow SPLIT panes, so a degenerate geometry is indistinguishable from a legal one by the time it reaches the wire. A console-less client attaches at 1x1 (Bubble Tea reports that for a process with no console), and without the gate that resized every pane in a 48-tab workspace to one column and permanently re-wrapped every child's transcript — observed 2026-08-25 and 2026-09-05. It sits AHEAD of every `sizedOnce` write, so each pane is still owed its first-resize kick once a usable geometry arrives. `handleResizePane` repeats the refusal daemon-side, but only when BOTH dimensions are at the floor together — a narrow split is narrow in one dimension and wide in the other. Regression tests: `internal/tui/tinyterm_test.go` (driven through `Update`, because the defect was that the call sites were unconditional), `internal/daemon/resize_guard_test.go` +- Layout persistence: the TUI sends `MsgUpdateLayout` only when THIS CLIENT'S OWN USER changed a tab's tree (`markLayoutChanged`, `internal/tui/layoutsync.go`), never merely because the stored tree disagrees with the local one. Every tab's tree is numbered (`Tab.LayoutRev`, `layout_rev` on the wire and in `workspace.json`, omitempty); a write carries `BaseRev`, the daemon accepts it only when that matches the tab's current revision (else refuses, no store, no broadcast), and now DOES broadcast an accepted write, through a 50 ms coalescer so a burst of tabs in one pass produces one frame. A client adopts any broadcast whose revision is higher than the tree it holds, reusing `*PaneModel`s by id; an arrival nobody on this client asked for (MCP-created, another client's split, a moved pane) is placed or pruned locally and sent only if the NEXT broadcast for that tab still lacks it. Reattach resets every tab's revision to 0 and force-adopts the next broadcast (`armReattachReset` → `resetLayoutSync`), because a restored daemon's revision can be lower than a client's own. Full detail, including the reservation re-seat and the retired feedback-loop concern, in `.claude/rules/tui-rendering.md`'s "Layout sync between clients" and `.claude/rules/daemon-lifecycle.md`'s "Multi-client" section. The same broadcast arm still diffs pane sizes (`diffResizes`), scoped to the broadcasting `Dest` — a broadcast is one daemon's full state and says nothing about another's tabs. On reconnect, `applyWorkspaceState()` deserializes the tree and prunes missing panes. **The comparison is structural, never by bytes**: `MarshalLayout` emits a struct (declaration order) while `parseWorkspaceState` re-marshals a `map[string]any` (alphabetical order), so identical trees encode differently and a byte-diff matches only single-leaf tabs — which is invisible in a workspace that has none. Sending unconditionally used to overflow the client's own 64-slot critical queue at 33 tabs + 36 panes and made the TUI close its connection and exit (2026-08-09); split-drag release still does a full sweep (`sendAllLayouts`/`resizeAllPanes`), where everything genuinely changed. The FIRST resize per pane is never suppressed (`Model.sizedOnce`, cleared by `armReattachReset`) because the daemon's own guard is `appliedCols/appliedRows`, which a PTY install zeroes +- **Every `MsgResizePane`/`MsgResizePanes` this client produces is gated on `Model.terminalPaintable()`** (`m.width >= minTermWidth && m.height >= minTermHeight`) **and, per destination, on NOT being a follower** (`isFollower(dest)`, true while another attached client holds size master there) — `resizeAllPanes`, `diffResizes` and `overlayResizeCmd` are the only three producers, and `attachMessage` reports 0x0 below the paintable threshold so `handleAttach`'s own 80x24 default stands. The gate belongs at the FAN-OUTS: `paneVTSize` floors both dimensions at 1 on purpose for genuinely narrow SPLIT panes, so a degenerate geometry is indistinguishable from a legal one by the time it reaches the wire. A console-less client attaches at 1x1 (Bubble Tea reports that for a process with no console), and without the gate that resized every pane in a 48-tab workspace to one column and permanently re-wrapped every child's transcript — observed 2026-08-25 and 2026-09-05. It sits AHEAD of every `sizedOnce` write, so each pane is still owed its first-resize kick once a usable geometry arrives (a follower's `sizedOnce` is deliberately left unmarked for the same reason). `handleResizePane`/`handleResizePanes` repeat the refusal daemon-side — the degenerate-geometry check for BOTH dimensions at the floor together, and now also a master-only check (`applyResizes`): a follower's resize is dropped silently, with no log line, because an older-build follower sends one on every broadcast. A window resize or split-drag release batches every pane into ONE `resize_panes` frame, and the daemon answers with at most one `pane_sizes` frame per batch to each follower — never one per pane — sent before `pty.Resize` runs, so a follower's VT holds the new size before the child's own repaint at it arrives. See `.claude/rules/daemon-lifecycle.md`'s "Multi-client" section for the size-master election this all serves, and `.claude/rules/tui-rendering.md`'s "The resize gates and batching" / "Follower rendering" for the client side. Regression tests: `internal/tui/tinyterm_test.go` (driven through `Update`, because the defect was that the call sites were unconditional), `internal/daemon/resize_guard_test.go` - Pane restart boundary: `PaneOutputPayload.Generation` carries the daemon's PTY run counter on every live chunk. The TUI resets its VT before a replacement run's output and drops older frames; reattach forgets the counter because the daemon may have restarted. `newPaneSession` picks sibling/client dimensions before spawn, rejecting a degenerate client size via `degenerateSize` and falling back to 80x24. Keep the IPC fast-frame decoder and its wire-shape test in sync with this payload. - Pane naming: `MsgUpdatePane` IPC message, `Pane.Name` field in daemon, Alt+F2 keybinding to rename active pane (mirrors F2 tab rename pattern) - Hand-started agents (#221): typing `claude`/`codex`/`opencode` at a terminal pane's shell opens the pane as that agent. The seam is a shell FUNCTION that shadows the binary, not a hook on a running process — **Quil never attaches to a running agent**. Detail in [`.claude/rules/daemon-lifecycle.md`](./rules/daemon-lifecycle.md) @@ -260,7 +260,7 @@ Project docs are now organized as a navigable tree under `docs/` (with the index - `docs/keybindings.md` — Full keymap + customization syntax - `docs/configuration.md` — `~/.quil/config.toml` reference - `docs/workspace-templates.md` — Template fields, layouts, prompts, creation, settings and adapter limits -- `docs/mcp.md` — User-facing MCP guide (client wiring, all 35 tools, redaction model) +- `docs/mcp.md` — User-facing MCP guide (client wiring, all 36 tools, redaction model) - `docs/plugin-reference.md` — TOML plugin schema (every field, every strategy, examples) - `docs/troubleshooting.md` — Daemon won't start, MCP not detected, log file locations, reset - `docs/sandbox-panes.md` — Docker sandbox panes: building the image, signing in, what the sandbox does and does not bound @@ -292,7 +292,7 @@ Cached reference repos: | M6 | Done | Pane focus — Ctrl+E full-screen active pane | | M7 | Done | Pane notes — Alt+E editor bound per pane, three save safety nets | | M8 | Done | Bubble Tea v2 + Lipgloss v2 migration | -| M10 | Done | MCP server — `quil mcp`, 35 tools; projects, hosts and task delegation added September 2026; request-response IPC via `Message.ID` | +| M10 | Done | MCP server — `quil mcp`, 36 tools; projects, hosts and task delegation added September 2026; request-response IPC via `Message.ID` | | M11 | Done | Command palette — Alt+Shift+P, fuzzy find, unified content search | | M12 | Done | Notification center — daemon event queue, per-pane mute, sidebar, 3 MCP tools | | M13 | Done | Memory reporting — 5s collector, per-pane Go-heap + PTY RSS, dialog + 2 MCP tools | diff --git a/.claude/rules/daemon-lifecycle.md b/.claude/rules/daemon-lifecycle.md index 72b6658f..a6c1407c 100644 --- a/.claude/rules/daemon-lifecycle.md +++ b/.claude/rules/daemon-lifecycle.md @@ -56,18 +56,187 @@ transport half by `TestBroadcast_SkipsPaneOutputForOptedOutConnOnly`. Every live MCP bridge holds a conn for its whole lifetime and those outlive the TUI, so in an ordinary session (21 conns observed, mostly bridges) `ConnCount() == 0` essentially never happens. Anything that means "nobody is driving this -daemon any more" must therefore ask a different question. `attachedConns` -(`daemon.go`, its own `attachedMu`, never `sm.mu`) is the set of conns that -have sent `MsgAttach`; `markClientAttached` adds, `forgetAttachedClient` removes -from `onClientDisconnect`. A bridge that never attaches is correctly not a -client. Shipped once against the raw conn count and it was inert in exactly the -case it was written for. +daemon any more" must therefore ask a different question. `Daemon.clients` +(`internal/daemon/clients.go`, a `clientRegistry` with its own leaf mutex, never +`sm.mu`) is the set of conns that have sent `MsgAttach` — grown from a bare +`attachedConns` set into a full record per client (id, conn, `attachedAt`, raw +`cols`/`rows`, `cwd`, `lastInputAt`, overlay claims) once several TUIs on one +daemon needed to elect a size master among them; see "Multi-client" below. +`registerClient`/`attachClient` add on `handleAttach`, `forgetAttachedClient` +removes a LOST link from `onClientDisconnect`, `detachClient` removes a clean +exit (`MsgDetach`, before the conn closes). A bridge that never attaches is +correctly not a client. Shipped once against the raw conn count and it was +inert in exactly the case it was written for. `onClientDisconnect` runs from `handleConn`'s defer, which calls `removeConn` BEFORE invoking it — so the set is already exclusive of the client that just left and needs no self-filtering. That ordering is load-bearing; check it before relying on a count read inside the callback. +### Multi-client: registry, size master, output hold, layout rev + +Several TUIs can attach to one daemon and see the same workspace (projects, +tabs, panes, layout, the active tab of each project). What changes is only +where clients would otherwise conflict: who sets a PTY's size, what a freshly +attaching client's live output does to its history replay, and which client an +untargeted MCP command reaches. + +**Client identity.** `AttachPayload.ClientID` is minted once per TUI PROCESS +(`uuid.NewString()`) and sent on every attach and reattach to every +destination — never persisted, so two TUIs on one machine never share it. An +attach with no id (an older client, a test) gets `anon-`, scoped to that +conn. Ids are bounded (`maxClientIDLen`, `truncateField`) and used only as a +map key and a display value. + +**Master election (`clientRegistry.electLocked`, `internal/daemon/clients.go`).** +Each PTY has one size, so exactly one attached client — the size master — may +set it: the OLDEST attached client with a PAINTABLE RAW geometry +(`eligible`: `cols >= daemonMinClientCols && rows >= daemonMinClientRows`, +40×10, the daemon-side mirror of the TUI's `terminalPaintable` floor — the two +MUST move together, since a daemon floor below the TUI's would elect a window +the TUI itself refuses to size panes from). The RAW value matters: a +console-less client attaches at 0×0, and electing it on `handleAttach`'s +80×24-defaulted `clientSize` is the 1×1 incident +([[headless-attach-reflows-all-panes]]) coming back through a new door — a +shrink below the floor (`setGeometry`, fed by `MsgClientGeometry`) drops +eligibility and re-elects AT ONCE, with no grace, because the client is still +attached and the daemon knows immediately. + +A master whose LINK IS LOST (no `MsgDetach`) keeps its slot for +`master_grace_minutes` (`[daemon]`, default 3, clamped 0–60, `internal/config`; +0 = no grace) — but ONLY while another client that was attached at the moment +of loss is still attached (`reservation.protects`): the grace exists to protect +FOLLOWERS from a resize while the master might still come back, and with no +follower left to protect it would only make a relaunched TUI (a new id, since +the id is per-process) wait for nothing. A clean exit sends `MsgDetach` (from +`closeClient`, riding the existing `Flush`, not a `tea.Cmd` — the Update loop +is already gone on the exit path) and skips the grace entirely: `detach` elects +with no reservation. `take_control` (`MsgTakeControl`, no payload, keymap +action `client.take_control`, no default key, plus a palette command) makes +the sender master at once if it is attached and eligible, overriding any +reserved slot; an ineligible or unattached sender is ignored with a debug log, +never an error. + +**After a daemon restart**, the registry is runtime-only, so `size_master` +(the master's id) is also written to `workspace.json` (top level, omitempty) +on every snapshot. Restore turns it into a reservation with no `protects` +condition, for `min(grace, 30s)` (`restartReserveCap`) — TUIs reattach with +their SAME process id within seconds of a restart, so the previous master +reclaims its slot (`attach` hands the reservation's `attachedAt` back to the +returning record, so it stays the oldest) and nothing resizes. + +**Size authority (`applyResizes`, `internal/daemon/daemon.go`).** `resize_pane` +and `resize_panes` share one implementation. A resize applies only from the +master conn (`isMasterConn`), with one exception: while there is no master +AND no reserved slot (`sizeAuthorityOpen`), any attached client's resize +applies — today's single-client behaviour. A refused resize is dropped with no +log line, because a follower on an older build sends one on every broadcast. +`resize_panes` is the BATCHED form clients send for a window resize or a +split-drag release across many panes; the plan's per-resize `pane_size` frame +was replaced with one `pane_sizes` frame per APPLIED batch, sent to every +OTHER attached conn on the must-deliver queue BEFORE any `pty.Resize` call — +`sendLoop` drains that queue ahead of pane output, so a follower's VT holds the +new size before the child's own repaint at that size arrives; without it the +repaint lands in the OLD-sized VT and is reflowed, the unpaired-resize +corruption `ResizeVT`'s contract forbids. One frame per BATCH, never per pane, +because a resize burst across 40+ panes would otherwise put 40+ must-deliver +frames on a follower's 64-slot queue at once (the 2026-08-09 shape, again). + +**Stale-broadcast races are closed with a per-pane size generation, not a +lock across record-and-send.** Every applied resize numbers the pane +(`Pane.sizeSeq`, bumped before the `pane_sizes` frame leaves, so the frame and +the later-recorded `Cols`/`Rows` share the number) and stamps it on the wire in +both `PaneInfo.SizeSeq` (workspace-state broadcasts) and `ResizePanePayload` +(`pane_sizes` frames, field `SizeSeq`); a follower TUI's `PaneModel.adoptDaemonSize` +adopts only `seq >= daemonSizeSeq`, so a workspace-state broadcast racing a +`pane_sizes` frame from the same batch can never undo it, and reattach resets +the counter (a fresh daemon incarnation renumbers from zero). A failed +`pty.Resize` sends a second `pane_sizes` with the PREVIOUS size and a newer +seq — a rollback is a newer announcement, not an undo, and `Cols`/`Rows` stay +at whatever the syscall actually left the PTY at. + +**Output hold (`internal/daemon/outputhold.go`).** A client attaching while +panes are writing must receive each pane's history replay and its live output +EXACTLY ONCE, in order — without a hold, a live broadcast frame can land in +the middle of the replay, or bytes written between the snapshot and the replay +finishing arrive twice. `handleAttach` calls `beginOutputHold(conn)` BEFORE +building the state frame: it sets `ipc.Conn.holdPaneOutput` (an atomic bool +`Broadcast`'s `wantsFrame` already checks, the same shape as `noPaneOutput`) +and then waits (`holdDrainTimeout`, 2s, polled every 2ms) for any live frames +already queued on the conn's droppable `outCh` to drain — those bytes are +already covered by the replay snapshot about to be taken, so without the wait +a busy client received them again, behind the state frame. `Pane.outPos` +(runtime-only, never on the wire) is the total bytes ever appended to a pane's +stream; `flushPaneOutputGeneration` reads `start := pane.outPos` in the SAME +`PluginMu` span as the `OutputBuf` write, so a hold entry's position and the +replay's `OutputBuf` snapshot can never disagree about what has been sent. +Every flush during a hold is copied into that conn's hold (`holdOutput`, +keyed by `*ipc.Conn` under the daemon's `holdMu` leaf lock) BEFORE the ordinary +broadcast. Two locks, always taken `holdGate` → `holdMu`: `holdGate` (an +`RWMutex`) makes "append to the hold, then broadcast" one step against setting +or clearing the conn's flag (a flush holding it for READ; every hold that +starts or ends holding it for WRITE) — without it a flush could append and +broadcast to a conn whose flag flipped in between, either duplicating the +bytes or losing them. + +On release, held chunks are deduped against `end[paneID]` (the replay's own +stream-position snapshot, read in the same `PluginMu` span as the `OutputBuf` +bytes it replayed): a chunk fully covered by the replay is dropped, one +straddling `end` is re-encoded from the overlap point, and the rest are sent +in order through `SendBlocking` (must-deliver, so nothing sent after the flag +clears can overtake them). Each hold is bounded at 4 MiB +(`outputHoldLimit`) per conn; a pane that would pass it loses ALL its held +bytes for that conn (never a partial pane) and gets one `redrawKick` at +release instead — the conn itself is never closed for this. `finishOutputHold` +rechecks under both locks before clearing the flag, because a flush can append +between an empty-batch check and the finish; the recheck is what sends it +round again rather than losing it. `onClientDisconnect` (`dropOutputHold`) +discards a conn's hold without sending — `SendBlocking` would return +`ErrConnClosed` and the drain loop stops anyway. + +**Layout revision (`Tab.LayoutRev`, spec §7).** Each tab's stored tree carries +a `uint64` revision, persisted as `layout_rev` (omitempty) and put on the wire +in every broadcast's per-tab entry. `SetTabLayout` (`internal/daemon/project.go`, +under `sm.mu`, unchanged lock) now takes `UpdateLayoutPayload.BaseRev +*uint64`: a nil `BaseRev` (an older client) or one equal to the tab's current +revision stores, bumps the revision and broadcasts; a stale one is refused — +no store, no broadcast, a debug log only. **The `// No broadcastState() — +avoids feedback loop` note is retired**: a layout write now broadcasts, +because clients no longer re-send on mere disagreement (they only send a +change their OWN user made — see `tui-rendering.md`'s Layout sync), so the +loop the old comment guarded against cannot occur any more. The broadcast +itself goes through the SAME `requestBroadcast` 50 ms coalescer the snapshot +debounce already uses, so several tabs written in one burst (`sendAllLayouts` +after a border-drag release) still produce ONE frame, keeping the +must-deliver queue safe. + +**MCP unicast targets (`internal/daemon/mcp_targets.go`).** `close_tui` and +`set_active_pane`'s focus frame used to broadcast to every attached TUI; both +now reach exactly ONE conn, via `targetConn(clientID)` — an explicit, +attached `client` id, or (empty) whichever client typed most recently, falling +back to the OLDEST attached client when nobody has typed yet. `targetConn` +does not delegate to `mostRecentlyActiveConn`'s own "nobody typed" answer (the +newest attached client) because the two callers want different defaults for +that state. With no attached client at all (a headless daemon), `close_tui` +sends nothing and logs, and `set_active_pane` still switches the shared active +tab and broadcasts state — only the unicast focus frame has nobody to reach. +`list_clients` (`handleListClientsReq` → `listClients`, `clients.go`) reports +every attached client — id, `attached_at` (RFC 3339), raw `cols`/`rows`, +whether it holds size master, `last_input_at` — merged with the hello +registry's `role`/`pid`/`exe` by conn, read AFTER `clients.mu` is released +since each registry keeps its own leaf lock. + +**`defaultCWD(conn)` chain** (`daemon.go`): the requesting conn's own recorded +`cwd` (when it is an attached client) → the master's `cwd` → the most +recently active client's `cwd` → the daemon's own `os.Getwd()`. Every step +shares ONE probe deadline (`spawnDirProbeTimeout`) and skips a candidate +directory string already tried, so the ordinary single-TUI case — where every +step names the same client — cannot pay the dead-directory timeout more than +once. `conn` is nil for every restore/recovery caller, which starts at step 2 +(the master). An MCP bridge never attaches, so `create_pane` with no CWD +resolves through steps 2–3, keeping today's behaviour of opening in a TUI's +directory rather than the daemon's frozen one. + ### A per-pane broadcast is a queue-pressure decision, not a detail Two rules the overlay retention work had to learn, both instances of the diff --git a/.claude/rules/tui-rendering.md b/.claude/rules/tui-rendering.md index 598b4e30..8feab078 100644 --- a/.claude/rules/tui-rendering.md +++ b/.claude/rules/tui-rendering.md @@ -19,6 +19,10 @@ paths: - "**/internal/tui/splitdrag*.go" - "**/internal/tui/perf*.go" - "**/internal/tui/frame_*_test.go" + - "**/internal/tui/layoutsync*.go" + - "**/internal/tui/layout_sync_test.go" + - "**/internal/tui/typing_guard_test.go" + - "**/internal/tui/multiclient_role_test.go" - "**/internal/clipboard/**" --- @@ -90,7 +94,7 @@ Release re-tracks at the release cell (a release with no motion still resolves, A pane the daemon moved (`move_pane`) reaches the TUI only as a broadcast listing it under its new tab; there is no optimistic move and no mover-side pre-split. `rebuildTabs` reconciles it in the same pass as every other arrival, in `internal/tui/model.go`. -**Detection is `existingPanes` reuse** (`migrated := ok` in the add loop). A hit there can only be a pane in ANOTHER tab's tree: panes already in this tree take the `treePaneIDs` branch, overlays never reach the loop, and the worktree-held pane is skipped before it. Every attached client of a daemon holds every tab of that daemon, so every client observes the same reuse — which is what lets a CLIENT-side rule stay in agreement across clients. A mover-only pre-split was rejected: other clients would place the pane differently, existing tabs never adopt the stored layout (`diffLayouts` only sends), so the clients would re-send each other's trees on every broadcast; and a pre-armed placeholder is pruned by any broadcast landing first (the 5 s git ticker), while a daemon refusal sends nothing that could unwind it. +**Detection is `existingPanes` reuse** (`migrated := ok` in the add loop). A hit there can only be a pane in ANOTHER tab's tree: panes already in this tree take the `treePaneIDs` branch, overlays never reach the loop, and the worktree-held pane is skipped before it. Every attached client of a daemon holds every tab of that daemon, so every client observes the same reuse — which is what lets a CLIENT-side rule stay in agreement across clients. A mover-only pre-split is still rejected, but not for the reason it once was: layout sync (below) now lets every OTHER client adopt the daemon's stored tree by revision, so two clients no longer re-send each other's trees on disagreement. What survives is the arrival race a revision cannot arbitrate — the requesting client's own `pendingSplit` reservation is placed LOCALLY, before `create_pane`/`MovePane` ever reaches the daemon, so it carries no revision of its own and a bystander's broadcast can still land inside that window. See "A migrated pane never fills a reservation" below for the guard that covers exactly that race. **Placement: a spiral ("dwindle") into the last pane.** `placeArrivingPane` splits the LAST pane leaf in tree order (`spiralLeaf` — a descent preferring Right, skipping placeholder leaves) AGAINST its parent's direction (`spiralSplitDir`: parent left|right → top|bottom, parent top|bottom → left|right, a root leaf → left|right), and installs the pane in the right/bottom half at Ratio 0.5. Successive arrivals therefore spiral into the bottom-right corner — `A` → `A|new`, `A|B` → `A|(B/new)`, `A|(B/C)` → `A|(B/(C|new))`, `(A/B)` → `(A/(B|new))`. It replaced a largest-leaf rule that turned `p3|p4` into three thin columns. Both halves of the choice are properties of the tree alone, so equal trees choose the same leaf and direction whatever each client's window size. `arrivalSplitDir(pref, w, h)` then flips the direction only when the preferred one would leave a half under `minPaneW` (left|right) or `minPaneH` (top|bottom) AND the other one fits; neither fitting, or unknown geometry, keeps `pref`. The rect comes from the canonical `paneAreaWidth() × (height - chromeHeight)`, not the notes-squeezed width. The only client-dependent input is that fallback, which needs one client's leaf below the minimum and another's not. **"Equal trees" does not hold for a client with its own reservation** — its tab carries an extra placeholder split (`pendingSplit`) that no other client's copy has, since placeholders are pure client-local runtime state and are never broadcast, so THAT client's walk can choose a different leaf than everyone else (the spiral skips the placeholder, which may be the last leaf); see "A migrated pane never fills a reservation" below for why the divergence is contained rather than a thrash. @@ -118,7 +122,217 @@ A pane the daemon moved (`move_pane`) reaches the TUI only as a broadcast listin `applyTabArrangement(tab, root, active)` (`arrange_apply.go`) is the ONE apply step for the six menu/palette/key actions and both in-tab drops. In order: notes mode → silent no-op; `tabLayoutBusy` → flash `Tab is busy — try again in a moment`; fewer than two panes → no-op; `fitsMinSize` against the CANONICAL geometry (`paneAreaWidth()` × `height-chromeHeight` — never the notes-squeezed width, and the menu may be arranging a BACKGROUND tab) → flash `Not enough room for that layout`; then the tree, `invalidateLeaves`, `ActivePane` plus every `Active` flag in the tab, `ExitFocus`, `SetCanvas`/`SetChrome`/`Resize`, and `tea.Batch(resizeAllPanes(), sendTabLayout(tab))`. `sendTabLayout` marshals on the Update goroutine and ships ONE `MsgUpdateLayout` for that tab through `sendDiffedLayouts`. -**`tabLayoutBusy` = `tabInFlight` + this client's own `pendingSplit` reservation.** `tabInFlight` deliberately did not grow the reservation: it also gates Move to tab…, which has its own rule for reservations (a moved pane never fills one). For an arrangement the reservation is the dangerous case — the new tree is a copy, so `pendingSplit` would be left pointing at a placeholder no tree holds. **Known limit:** other attached TUIs keep their own tree for that tab until the separate layout-sync item lands — existing tabs never adopt the stored layout (`diffLayouts` only sends). Worse than "stale": on its next broadcast such a TUI sees the stored layout disagree with its tree and RE-SENDS its own, overwriting the arrangement; the layout a restart restores is whichever client sent last. Tests: `arrange_test.go`, `arrange_apply_test.go`. +**`tabLayoutBusy` = `tabInFlight` + this client's own `pendingSplit` reservation.** `tabInFlight` deliberately did not grow the reservation: it also gates Move to tab…, which has its own rule for reservations (a moved pane never fills one). For an arrangement the reservation is the dangerous case — the new tree is a copy, so `pendingSplit` would be left pointing at a placeholder no tree holds. **Fixed by layout sync (below):** `sendTabLayout` now ships `MsgUpdateLayout{BaseRev: &tab.layoutRev}`, and every OTHER attached TUI adopts the daemon's higher-revision tree in `syncTabLayout`/`adoptTabLayout` instead of re-sending its own. Before that, other attached TUIs kept their own tree for the tab indefinitely — existing tabs never adopted the stored layout — so a second TUI's next broadcast disagreed with its OWN tree and RE-SENT it, overwriting the arrangement; the layout a restart restored was whichever client had sent last. Tests: `arrange_test.go`, `arrange_apply_test.go`, `layout_sync_test.go`. + +## Multi-client + +Several TUIs can attach to one daemon and share its workspace. `Model.clientID` +(minted once per process, sent on every attach) and, per destination, +`Model.sizeMaster[dest]`/`Model.clientCount[dest]` (kept from each broadcast's +`SizeMaster`/`Clients` fields, `internal/tui/model.go`) are what a client uses +to tell whether it is the size master. `isFollower(dest)` is +`sizeMaster[dest] != "" && sizeMaster[dest] != m.clientID` — the zero value (no +entry yet) answers false, so a client that has not heard from a destination +behaves as it always did. See `.claude/rules/daemon-lifecycle.md`'s +"Multi-client" section for the daemon-side registry and election this reads. + +### The resize gates and batching + +**A follower sends no `MsgResizePane`/`MsgResizePanes`.** The three resize +producers named by the `terminalPaintable` invariant in `.claude/CLAUDE.md` — +`resizeAllPanes`, `diffResizes`, `overlayResizeCmd` — each gate on +`isFollower(dest)` in addition to `terminalPaintable()`; the gate sits at the +same three fan-outs because those are the only three producers. A follower's +`sizedOnce` is deliberately NOT marked, so if this client later becomes master +the pane still gets its first-resize kick rather than reading as +already-sized for a size it never sent. A local rect change that never reaches +the daemon — entering focus mode, opening notes, toggling the notification +sidebar — resizes nothing on a follower for the same reason it resizes nothing +today: none of those three producers fires for it, follower or not. + +**Becoming master clears `sizedOnce` for that destination rather than calling +`resizeAllPanes` directly** (`applyWorkspaceState`, guarded on +`state.SizeMaster == m.clientID && prevMaster != state.SizeMaster`). Every +pane's last-applied size was sent by whoever was master before (or by +nobody), so `sizedOnce` still reads "already sized" for sizes this client +never sent — `clearSizedOnceForDest` clears that, and `diffResizes` (which +Update runs immediately afterward, scoped to the same `dest`) picks up every +pane in ONE batch, the same re-arm `armReattachReset` performs after a +reattach. Calling `resizeAllPanes()` here as well would walk every OTHER +destination too, resizing panes this broadcast never mentioned. + +**Batching, daemon and client.** A window resize or a split-drag release +sends `MsgResizePanes` (one frame for the whole destination) instead of one +`MsgResizePane` per pane; `sendDiffedResizes` and `resizeAllPanes` both build +one batch per dest. The daemon answers with at most one `pane_sizes` frame per +applied batch to each follower (`applyResizes`/`sendPaneSizes`, +`internal/daemon/daemon.go`), never one per pane — see the daemon-lifecycle +note for why the frame goes out before `pty.Resize` runs. Every pane in a +`pane_sizes` frame, and every `PaneInfo` in a workspace-state broadcast, +carries `size_seq`; `PaneModel.adoptDaemonSize` adopts only `seq >= +daemonSizeSeq`, so a workspace-state broadcast that raced a `pane_sizes` frame +from the same resize cannot undo it. + +### Follower rendering + +**Grid size.** `PaneModel.targetVTSize` (`internal/tui/pane.go`) is the single +decision point for a pane's EMULATOR size: a follower pane with a known +daemon size (`p.follower && p.daemonCols > 0 && p.daemonRows > 0`) takes that +size regardless of the box this client draws it in, at every site that sizes +a VT — `TabModel.Resize`/`resizeNode` for layout leaves, `sizePaneFull` for +focus mode and for overlay panes (lazygit), which sit outside `Leaves()`. +Everyone else, and a follower pane with no daemon size yet (never sized, or +pending — it falls back to `paneVTSize(rect)` for drawing and sends nothing), +uses `paneVTSize`. Resizing a follower's VT to its own box instead would +rewrap the master's output with no PTY redraw to pair it — the unpaired-resize +corruption `ResizeVT`'s contract forbids for split drags applies here too. +`p.follower` and `p.daemonCols`/`p.daemonRows` reach `PaneModel` through +`syncPaneMeta` (`internal/tui/workstate.go`), the path that already copies +every other daemon-derived field onto pane models; `adoptDaemonSize` is the +seq-gated write into `daemonCols`/`daemonRows` described above. + +**Viewport (`pane_preview.go`).** A follower pane reuses the wide-canvas +preview renderer instead of a second one. `previewMode()` is true for a +follower whose grid exceeds its box in EITHER dimension (`innerW < +vt.Width() || innerH < vt.Height()`), on top of its existing wide-canvas +condition — a grid that FITS renders NATIVELY (top-left, padded), same as a +non-follower pane. The preview already bottom-anchors with scrollback above +(`renderPreview`, which shows `total - innerH - scrollBack`) and left-edge +crops, which is exactly the follower's cut: width keeps the LEFT columns, +height keeps the BOTTOM rows. Every rendered row stays exactly the pane's +width, as for any preview pane. + +**Corner markers (`followerCutMark`, `internal/tui/pane.go`).** +`buildTopBorderCut` swaps one top-border CORNER for `"…"` — one cell for one +cell, so the border keeps its exact width and the label between the corners +is untouched — top-left for a HEIGHT cut (rows above the box are hidden, the +view is bottom-anchored) and top-right for a WIDTH cut (columns right of the +box are hidden, the view is cropped at the left edge). Both can show at once. + +**Mouse.** There is no click forwarder in the TUI at all today, so follower +grid translation applies only to the wheel forwarder +(`wheelForwardSeq`/`sendInputToPane`): a notch translates box row `relY` to +grid row `relY + max(0, vtH - innerH)` while the view is not scrolled back, +with no horizontal offset (the crop is at the left already). A position past +`vtW`/`vtH` (the padding) sends nothing. Mouse selection uses the existing +`previewPosAt` mapping; keyboard selection works only when the grid fits the +box, as for wide-canvas panes today. + +**Status bar.** While `clientCount[activeDest] >= 2`, the status bar shows +`[master]` or `[follower]` for the active destination, next to `[dev]` +(`internal/tui/model.go`, the status-bar assembly). With one client, nothing +is shown — a single TUI on a daemon looks exactly as it always has. `Take +control` (`client.take_control`, no default key, plus a palette command, +"Take control (size master)") sends `MsgTakeControl` and makes this client +master at once when the daemon accepts it. + +### Layout sync between clients + +`internal/tui/layoutsync.go`. The daemon numbers every stored write of a +tab's tree (`layout_rev`, spec §7.1) and refuses a write whose `base_rev` is +not the tab's current revision. A client sends a tab's tree only when ITS OWN +USER changed it, and adopts any broadcast carrying a higher revision than the +tree it holds — it never sends merely because the stored tree disagrees with +its own, which is what let two clients re-send each other's trees forever +before this landed (see the retired "Known limit" notes above). + +**`markLayoutChanged(dest, tab)`** is the one place a user-caused mutation — +split, close, arrange, pane drag drop, split-border drag release, a client's +own `pendingSplit` reservation being filled — records the change: it +marshals the tree HERE, on the Update goroutine (the `tea.Cmd` it returns +holds only bytes, never the tab), sets `tab.layoutDirty`, and sends +`MsgUpdateLayout{TabID, Layout, BaseRev: &tab.layoutRev}` through +`sendDiffedLayouts`. **A write already in flight defers the next one** +(`tab.layoutResend`) instead of sending a second write on the same base — +that would be refused behind the first and lost — and the deferred change +rides the first write's own echo instead. + +**Arrivals nobody on this client asked for — an MCP-created pane, another +client's split as a bystander sees it, a moved pane, a pane pruned because +another client closed it — are placed or pruned LOCALLY with the ordinary +arrival rules, so the screen is right at once, and recorded in +`tab.awaitingPanes`/`awaitingGone` without sending.** `syncTabLayout` checks, +on the NEXT broadcast for that tab, whether the stored tree still lacks any +awaited id or still holds a pruned one; only then does this client send its +tree with the current base rev, and the first such send from any client wins +— the rest are refused and adopt. Otherwise the requester's own write already +arrived at a higher revision and this client adopts it. + +**`syncTabLayout`** runs per existing tab, per broadcast, BEFORE panes are +reconciled, and every comparison is STRUCTURAL (parsed `SerializedNode`), +never by bytes — the daemon's stored bytes and a client's own re-marshal of +the same tree encode differently (declaration order vs. `map[string]any`'s +alphabetical order), the same caveat the layout-persistence invariant in +`.claude/CLAUDE.md` states. A higher `layout_rev` (or `tab.adoptNext`, set +after a reattach — see below) triggers `adoptTabLayout`, UNLESS the stored +tree is this client's own write echoing back (`reflect.DeepEqual(stored, +tab.layoutSent)`), which is adopted as a no-op and clears `layoutDirty`, +resending only a deferred `layoutResend`. A LOWER revision, or a dirty tab, +keeps the local tree — the write in flight will come back with a higher one. +A refused write is simply superseded by the winner's broadcast and its local +change is lost; that needs two users editing the same tab at the same moment. + +**`adoptTabLayout`** replaces the tab's tree with the stored one, reusing +`*PaneModel`s by id (no lost emulator or scrollback), dropping ids the +broadcast no longer lists, and leaving panes the stored tree lacks for the +ordinary arrival loop to place (which then follows the "arrivals nobody asked +for" rule above). It cancels an in-progress drag whose node is in this tab. +**This client's own `pendingSplit` reservation survives adoption** +(`reseatReservation`): it is put back BESIDE its original sibling pane, in +the original direction and half, when that sibling is still in the adopted +tree; only when the sibling is gone does it fall back to the spiral arrival +slot, and failing that it becomes the whole tab. A reservation whose +placeholder is no longer even IN the pre-adoption tree (abandoned) is +forgotten instead of re-seated. `tabLayoutBusy` still gates arrangements +throughout. + +**Reattach resets the revs.** `armReattachReset` calls `resetLayoutSync(dest)`, +which zeros every tab's `layoutRev` on that destination and sets +`tab.adoptNext = true`: the daemon's stored tree is authoritative, and its +revision can be LOWER than this client's after a restart (the snapshot is +debounced 500 ms) — as low as the 0 of a tree nobody has written since +restore, which even a zeroed local revision would not otherwise adopt. +`adoptNext` is what makes `syncTabLayout` adopt the very next broadcast +whatever its revision. A tab whose stored layout is EMPTY (fresh, or restored +before any client described it) is sent by the client with the current base +rev — this replaces the retired `diffLayouts`' `len(stored)==0` branch. + +**Older clients during dev** (release builds refuse a version mismatch, so +this matters only for dev builds): a client with no `BaseRev` support always +has its write accepted and never adopts; new clients adopt its tree, so +everyone converges on the old client's tree with no loop. + +Tests: `layout_sync_test.go`. Mutation-checked: the send-on-change gate, the +rev compare-and-store (daemon side), the adopt condition, the arrival +"only the requester sends" rule, and the reattach rev reset. + +### Typing guard across a remote tab switch + +When a broadcast changes THIS client's active tab for the active project, +and this client did not itself request that switch, `Model.remoteSwitchAt` +is stamped and `Model.guardPaneID` records the pane that was active +immediately before the switch (`internal/tui/model.go`). "Requested" is +decided by a TOKEN, never a time window: `Model.requestedTab`, keyed by +`requestedTabKey(dest, projectID)`, records the tab this client asked for on +every local `switchTab`/`create_tab`; +a broadcast whose active tab equals it is this client's own switch landing +and clears the token, and any other change is remote. + +**Keys reaching `enqueueInput` within `remoteSwitchGuardWindow` (250 ms) of +`remoteSwitchAt` are redirected to `guardPaneID`**, if that pane still exists +— otherwise they go to the new active pane. Only typed input is redirected; +mouse input always targets whatever is under the pointer now, since a click +is inherently aimed at what is on screen. A flash shows `Tab switched by +another client`. + +**Unseen is not acknowledged until local input arrives.** A pane that became +focused only because of a remote switch is skipped by `ackFocusedPane` (no +`pane_seen` is sent) until this client receives a real key or mouse click — +otherwise every attached client would clear the mark on every remote switch +whether or not anyone actually looked at the pane. + +Tests: `typing_guard_test.go`. Mutation-checked: the 250 ms window itself and +the `requestedTab` token gate. ### Mouse-wheel forwarding to tracking apps diff --git a/changelog.d/added-multi-client-sync.md b/changelog.d/added-multi-client-sync.md new file mode 100644 index 00000000..f366d9a7 --- /dev/null +++ b/changelog.d/added-multi-client-sync.md @@ -0,0 +1,9 @@ +--- +headline: Two Quil windows can now share one daemon's workspace +--- +- **Two or more Quil windows can attach to the same daemon and see one shared workspace** — the same projects, tabs, panes, names, colours and layout. A pane either window creates appears in both, and closing it in one closes it in the other. +- One window is the size master (the oldest attached, with a large enough terminal of its own) and sets pane sizes; the status bar shows `[master]` or `[follower]` while two or more windows are attached. A follower's panes are cropped to fit its own box rather than resized, with a `…` marker on the cut edge, and pad rather than crop when they are smaller than it. **Take control** (unbound by default — bind `client.take_control`, or run it from the command palette) makes the current window the master immediately. If the master's window closes normally the next one takes over at once; if its connection merely drops, its slot is held for a few minutes (`[daemon] master_grace_minutes`, default 3) so a following window is not resized out from under it. +- Splitting, closing, dragging a pane or a split border, and arranging a tab in one window is mirrored in every other attached window immediately, and survives closing and reopening both. +- Switching the shared active tab from one window no longer steals keystrokes out from under someone typing in another — a short guard keeps your next few keys in the pane you were in and shows a flash saying another client switched. +- Dismissing a notification, or clearing a pane's unseen mark, updates every attached window's sidebar. +- A new MCP tool, `list_clients`, lists every attached window; `set_active_pane` and `close_tui` gain an optional `client` field to target one window instead of whichever typed most recently. diff --git a/docs/configuration.md b/docs/configuration.md index 81a52ed3..9e0c85bc 100644 --- a/docs/configuration.md +++ b/docs/configuration.md @@ -147,6 +147,7 @@ sidebar_toggle = "alt+shift+s" # collapse/expand the PROJECT sidebar (not the n | `snapshot_interval` | duration | `"30s"` | Periodic safety-net write of `workspace.json` + ghost buffers. Event-driven snapshots (pane create/destroy, etc.) still fire 500 ms after the trigger. | | `auto_start` | bool | `true` | The TUI auto-starts `quild --background` when it can't find an existing daemon. Set `false` if you manage `quild` yourself (systemd, launchd, etc.) — the TUI will error instead of auto-spawning. | | `warm_shell_pool_size` | int | `1` | Number of pre-spawned shells kept in a pool for faster Ctrl+N; capped at `8`, and `0` or negative values disable pooling. | +| `master_grace_minutes` | int | `3` | With two or more Quil windows attached to this daemon (see [Multi-client sync](features.md#multi-client-sync)), how long the size master's slot is kept after its *connection* drops (not a normal quit) before another attached window takes over. Clamped to `0`–`60`; `0` means no grace — the next window is elected immediately. Only matters while another window is still attached; a lone window relaunching always takes over at once. | ## `[ghost_buffer]` diff --git a/docs/features.md b/docs/features.md index 70f4f88d..82cb64ea 100644 --- a/docs/features.md +++ b/docs/features.md @@ -2,7 +2,7 @@ A capability-by-capability tour of what Quil does. For configuration knobs, see [Configuration](configuration.md). For keystrokes, see [Keybindings](keybindings.md). For AI integration, see [MCP](mcp.md). -Quil exposes **35 MCP tools**: agents can manage [projects and tabs](mcp.md#projects-and-tabs), discover and route work across [remote hosts](mcp.md#remote-hosts), and create AI panes with the TUI dialog's options. [`delegate_task`](mcp.md#delegating-work-to-another-pane) tracks pane-to-pane work and can notify the requester after completion, when it is ready to receive input. +Quil exposes **36 MCP tools**: agents can manage [projects and tabs](mcp.md#projects-and-tabs), discover and route work across [remote hosts](mcp.md#remote-hosts), and create AI panes with the TUI dialog's options. [`delegate_task`](mcp.md#delegating-work-to-another-pane) tracks pane-to-pane work and can notify the requester after completion, when it is ready to receive input. ## Table of contents @@ -48,6 +48,7 @@ Quil exposes **35 MCP tools**: agents can manage [projects and tabs](mcp.md#proj - [Projects](#projects) - [Project groups](#project-groups) - [Projects on another machine](#projects-on-another-machine) +- [Multi-client sync](#multi-client-sync) - [Pane notes](#pane-notes) - [Operations](#operations) - [Self-healing daemon](#self-healing-daemon) @@ -669,6 +670,62 @@ A configured remote host that's unreachable when you launch no longer drops out --- +## Multi-client sync + +Two or more Quil windows can attach to the same daemon and see one shared +workspace: the same projects, tabs, panes, names, colours and layout. A pane +either of them creates appears in both; closing one in either window closes it +for both. There is no per-client copy of the workspace to fall out of sync. + +**Size master.** Each pane's PTY has one size, so exactly one attached client +sets it — the **size master**, the oldest attached window whose own terminal +is large enough to draw panes at all. While two or more clients are attached, +the status bar shows `[master]` or `[follower]` next to `[dev]`; with a single +window attached, nothing is shown, and it behaves exactly as it always has. +A follower's panes are sized to whatever the master last set: too wide or too +tall for the follower's own box, they are cropped rather than reflowed — width +keeps the left columns, height keeps the bottom rows, each with a `…` marker +on the cut border edge — and too small, they are simply padded. Scrolling the +wheel into a cropped pane reveals the cut rows first, then real scrollback. + +If the master's window closes normally, the next-oldest attached window +becomes master immediately, and its panes resize once to fit it. If the +master's *connection* merely drops (a network hiccup, not a quit) its slot is +held for a few minutes — `master_grace_minutes` in +[Configuration](configuration.md#daemon), default 3 — so a following window's +panes are not resized out from under it while the master might still come +back; with nobody left to protect, a relaunched window takes over at once +instead of waiting. **Take control** (unbound by default — bind +`client.take_control` in `bindings.toml`, or run it from the command palette) +makes the window you are typing in the master immediately, whatever the +election above would otherwise pick. + +**Typing guard.** If another window switches your shared active tab while you +are mid-keystroke, your next 250 ms of typing still lands in the pane you were +in — a flash reads `Tab switched by another client` — rather than being +redirected into whatever the switch brought to the front. A pane that becomes +focused only because of someone else's switch does not clear its unseen mark +until you actually type or click in this window; otherwise every attached +window would silently mark a pane "seen" the moment anyone else looked at it. + +**Layout.** Splitting, closing, dragging a pane, dragging a split border or +arranging a tab in one window is mirrored in every other attached window at +once, and survives quitting and reopening both. Two windows editing the same +tab's layout at the exact same moment is resolved by "first write wins" — the +losing window's change is superseded by the other's, which is expected only +when two people are actively rearranging one tab together. + +**Marks.** Dismissing a notification, or clearing a pane's unseen mark by +viewing it, updates every attached window's sidebar — not just the one that +did it. + +**MCP.** `set_active_pane` and `close_tui` each act on one attached window +rather than every one of them: by default the window you last typed in, or +name one explicitly with the `client` field (see [`list_clients`](mcp.md#tui-cooperation) +for the ids to choose from). With no window attached at all, `close_tui` sends +nothing rather than erroring. + +--- ## Pane notes diff --git a/docs/keybindings.md b/docs/keybindings.md index dc047621..a866fa25 100644 --- a/docs/keybindings.md +++ b/docs/keybindings.md @@ -14,6 +14,7 @@ Quil's full keymap. Every binding is configurable in `~/.quil/bindings.toml`, ei - [Clipboard](#clipboard) - [Text selection](#text-selection) - [Scrolling](#scrolling) +- [Multi-client](#multi-client) - [Dialogs (F1 menus)](#dialogs-f1-menus) - [Keys that pass through to the PTY](#keys-that-pass-through-to-the-pty) @@ -257,6 +258,15 @@ If the clipboard has no text but contains an image, Quil decodes the DIB, saves | Click + drag on scrollbar | Continuous scroll — drag follows cursor Y, even off-pane | | `Alt+Up` / `Alt+Down` *(in log viewer)* | Jump cursor by `[ui] log_viewer_page_lines` (default 40) | +## Multi-client + +Keys for when two or more Quil windows are attached to the same daemon — see +[Multi-client sync](features.md#multi-client-sync). + +| Key | Action | +|---|---| +| *(unbound)* | `client.take_control` — "Take control (size master)". Makes this window the size master immediately. Also in the command palette's **System** group. Bind it in `bindings.toml`, e.g. `"client.take_control" = "alt+shift+m"`. | + ## Command palette | Key | Action | diff --git a/docs/mcp.md b/docs/mcp.md index 0339eac4..8e34fa4e 100644 --- a/docs/mcp.md +++ b/docs/mcp.md @@ -14,7 +14,7 @@ The result: your AI can **see what's in your build pane and react**, instead of - [VS Code (GitHub Copilot Chat)](#vs-code-github-copilot-chat) - [Any MCP-capable client](#any-mcp-capable-client) - [Verify the connection](#verify-the-connection) -- [The 35 tools](#the-35-tools) +- [The 36 tools](#the-36-tools) - [Discovery](#discovery) - [Reading pane output](#reading-pane-output) - [Interacting with panes](#interacting-with-panes) @@ -70,7 +70,7 @@ Edit `~/Library/Application Support/Claude/claude_desktop_config.json` (macOS) o } ``` -Restart Claude Desktop. The 🔌 icon in the input bar should show Quil with 35 tools. +Restart Claude Desktop. The 🔌 icon in the input bar should show Quil with 36 tools. ### Claude Code (CLI) @@ -139,7 +139,7 @@ In your AI client, ask: The AI should call `list_panes` and return a JSON array with each pane's `id`, `type`, `tab_id`, `cwd`, etc. If you see "no Quil panes" or an error, check [Troubleshooting](#troubleshooting). -## The 35 tools +## The 36 tools Tools are grouped below by purpose. Every tool returns a `text` content block; many return JSON-formatted payloads. @@ -296,13 +296,18 @@ The daemon does not parse the target's reply. The excerpt is raw output; the req ### TUI cooperation -These steer the live TUI window (if one is attached). +These steer the live TUI window(s) attached to the daemon — see +[Multi-client sync](features.md#multi-client-sync) for what it means to have +more than one. | Tool | Input | Returns | Notes | |---|---|---|---| -| `switch_tab` | `tab_id` (required) | "Switched to tab " | Brings a different tab into view in the TUI. | -| `set_active_pane` | `pane_id` (required) | "Set active pane to " | Focuses the pane. Auto-switches tab if needed. | -| `close_tui` | — | "TUI close signal sent. Daemon continues running." | Closes the TUI window. Daemon and all pane processes keep running — reattach by running `quil` again. | +| `switch_tab` | `tab_id` (required) | "Switched to tab " | Brings a different tab into view — in every attached window, since the active tab is shared. | +| `set_active_pane` | `pane_id` (required), `client` (optional) | "Set active pane to " | Switches the shared active tab if needed, then focuses the pane in ONE window: the one named by `client`, or, when omitted, whichever window you last typed in. | +| `close_tui` | `client` (optional) | "TUI close signal sent. Daemon continues running." | Closes ONE window — named by `client`, or the one you last typed in — not every attached window. Daemon and all pane processes keep running regardless; reattach by running `quil` again. With no window attached at all, nothing is sent. | +| `list_clients` | `host` (optional) | JSON array: `{client, attached_at, cols, rows, master, last_input_at, role, pid, exe}` | Every attached window (and any other attached client), oldest first. `client` is the id to pass to `set_active_pane`/`close_tui`; `master` marks which one currently sets pane sizes. `attached_at`/`last_input_at` are RFC 3339, and `last_input_at` is absent for a window that has only watched. **Requires daemon 1.80.0+.** | + +`client` picks an EXACT window and never falls back to another one — but both tools are fire-and-forget, so a `client` that names no attached window still returns success; the daemon simply has nothing to send it to (logged daemon-side). Use `list_clients` right before to confirm the id is current. Omit `client` to get today's single-window behaviour unchanged — with one window attached, there is only ever one to choose. ### Event observation From e32250177e5ec6b60116117bd7816d107ec4998f Mon Sep 17 00:00:00 2001 From: Artjoms Stukans Date: Sat, 26 Sep 2026 01:58:31 +0200 Subject: [PATCH 23/40] docs: fix SetTabLayout location and stale layout-sync claims Review found two errors in the previous docs commit and a scope gap: - SetTabLayout lives in internal/daemon/session.go (SessionManager method), not project.go. - sendAllLayouts no longer exists; a split-drag release or an arrangement action only ever marks the ONE tab it changed (finishSplitDrag/applyTabArrangement -> markLayoutChanged). Rewrote the coalescer paragraph in daemon-lifecycle.md, the layout-persistence bullet in CLAUDE.md, and the Split-border drag-resize paragraph in tui-rendering.md around what actually produces a burst: several tabs sent from one client's own broadcast-reconciliation pass, or several clients accepting a write inside the same 50ms window. - Bumped the remaining "35 tools"/"35 MCP" mentions to 36 in docs/roadmap.md, docs/quick-start.md, docs/README.md, docs/prd.md, docs/competitive-analysis.md and .claude/rules/templates.md. --- .claude/CLAUDE.md | 2 +- .claude/rules/daemon-lifecycle.md | 40 +++++++++++++++++++++---------- .claude/rules/templates.md | 2 +- .claude/rules/tui-rendering.md | 2 +- docs/README.md | 2 +- docs/competitive-analysis.md | 6 ++--- docs/prd.md | 2 +- docs/quick-start.md | 2 +- docs/roadmap.md | 2 +- 9 files changed, 37 insertions(+), 23 deletions(-) diff --git a/.claude/CLAUDE.md b/.claude/CLAUDE.md index 04d026f1..f8d55b5e 100644 --- a/.claude/CLAUDE.md +++ b/.claude/CLAUDE.md @@ -203,7 +203,7 @@ These hold regardless of which file you open. Violating one breaks something. - IPC protocol: 4-byte big-endian length prefix + JSON payload. Optional `ID` field for request-response correlation (MCP bridge). When `ID` is set, daemon responds to specific connection; when empty, broadcasts to all. `AttachPayload` carries an optional `CWD` field (omitempty) — the TUI sends `os.Getwd()` on attach so the daemon can spawn new panes/tabs in the client's directory rather than the daemon's frozen-at-spawn-time CWD. Recorded per attached client in the registry (`clientRecord.cwd`, `.claude/rules/daemon-lifecycle.md`'s "Multi-client" section) rather than one global `Daemon.clientCWD` — several TUIs on one daemon each have their own directory. `defaultCWD(conn)` resolves it through a chain (the requesting conn's own client record → the size master's → the most recently active client's → the daemon's own `os.Getwd()`), validating each candidate with `os.Stat` + `EvalSymlinks` before accepting it - `.gitignore` uses root-anchored patterns (`/quil`, `/quild`) to avoid matching `cmd/` directories - Pane layout uses a binary split tree (`LayoutNode` in `internal/tui/layout.go`) — each internal node has its own `SplitDir`, enabling mixed H/V splits (tmux-style). The tree is serialized to JSON and persisted in the daemon's `Tab.Layout` field for reconnect restoration -- Layout persistence: the TUI sends `MsgUpdateLayout` only when THIS CLIENT'S OWN USER changed a tab's tree (`markLayoutChanged`, `internal/tui/layoutsync.go`), never merely because the stored tree disagrees with the local one. Every tab's tree is numbered (`Tab.LayoutRev`, `layout_rev` on the wire and in `workspace.json`, omitempty); a write carries `BaseRev`, the daemon accepts it only when that matches the tab's current revision (else refuses, no store, no broadcast), and now DOES broadcast an accepted write, through a 50 ms coalescer so a burst of tabs in one pass produces one frame. A client adopts any broadcast whose revision is higher than the tree it holds, reusing `*PaneModel`s by id; an arrival nobody on this client asked for (MCP-created, another client's split, a moved pane) is placed or pruned locally and sent only if the NEXT broadcast for that tab still lacks it. Reattach resets every tab's revision to 0 and force-adopts the next broadcast (`armReattachReset` → `resetLayoutSync`), because a restored daemon's revision can be lower than a client's own. Full detail, including the reservation re-seat and the retired feedback-loop concern, in `.claude/rules/tui-rendering.md`'s "Layout sync between clients" and `.claude/rules/daemon-lifecycle.md`'s "Multi-client" section. The same broadcast arm still diffs pane sizes (`diffResizes`), scoped to the broadcasting `Dest` — a broadcast is one daemon's full state and says nothing about another's tabs. On reconnect, `applyWorkspaceState()` deserializes the tree and prunes missing panes. **The comparison is structural, never by bytes**: `MarshalLayout` emits a struct (declaration order) while `parseWorkspaceState` re-marshals a `map[string]any` (alphabetical order), so identical trees encode differently and a byte-diff matches only single-leaf tabs — which is invisible in a workspace that has none. Sending unconditionally used to overflow the client's own 64-slot critical queue at 33 tabs + 36 panes and made the TUI close its connection and exit (2026-08-09); split-drag release still does a full sweep (`sendAllLayouts`/`resizeAllPanes`), where everything genuinely changed. The FIRST resize per pane is never suppressed (`Model.sizedOnce`, cleared by `armReattachReset`) because the daemon's own guard is `appliedCols/appliedRows`, which a PTY install zeroes +- Layout persistence: the TUI sends `MsgUpdateLayout` only when THIS CLIENT'S OWN USER changed a tab's tree (`markLayoutChanged`, `internal/tui/layoutsync.go`), never merely because the stored tree disagrees with the local one. Every tab's tree is numbered (`Tab.LayoutRev`, `layout_rev` on the wire and in `workspace.json`, omitempty); a write carries `BaseRev`, the daemon accepts it only when that matches the tab's current revision (else refuses, no store, no broadcast), and now DOES broadcast an accepted write, through a 50 ms coalescer so a burst of tabs in one pass produces one frame. A client adopts any broadcast whose revision is higher than the tree it holds, reusing `*PaneModel`s by id; an arrival nobody on this client asked for (MCP-created, another client's split, a moved pane) is placed or pruned locally and sent only if the NEXT broadcast for that tab still lacks it. Reattach resets every tab's revision to 0 and force-adopts the next broadcast (`armReattachReset` → `resetLayoutSync`), because a restored daemon's revision can be lower than a client's own. Full detail, including the reservation re-seat and the retired feedback-loop concern, in `.claude/rules/tui-rendering.md`'s "Layout sync between clients" and `.claude/rules/daemon-lifecycle.md`'s "Multi-client" section. The same broadcast arm still diffs pane sizes (`diffResizes`), scoped to the broadcasting `Dest` — a broadcast is one daemon's full state and says nothing about another's tabs. On reconnect, `applyWorkspaceState()` deserializes the tree and prunes missing panes. **The comparison is structural, never by bytes**: `MarshalLayout` emits a struct (declaration order) while `parseWorkspaceState` re-marshals a `map[string]any` (alphabetical order), so identical trees encode differently and a byte-diff matches only single-leaf tabs — which is invisible in a workspace that has none. Sending unconditionally used to overflow the client's own 64-slot critical queue at 33 tabs + 36 panes and made the TUI close its connection and exit (2026-08-09); a split-drag release still resizes every pane in one sweep (`resizeAllPanes`), where every pane's rect genuinely changed, but sends a layout update for only the ONE tab whose border moved (`finishSplitDrag` → `markLayoutChanged`) — there is no equivalent all-tabs layout sweep any more. The FIRST resize per pane is never suppressed (`Model.sizedOnce`, cleared by `armReattachReset`) because the daemon's own guard is `appliedCols/appliedRows`, which a PTY install zeroes - **Every `MsgResizePane`/`MsgResizePanes` this client produces is gated on `Model.terminalPaintable()`** (`m.width >= minTermWidth && m.height >= minTermHeight`) **and, per destination, on NOT being a follower** (`isFollower(dest)`, true while another attached client holds size master there) — `resizeAllPanes`, `diffResizes` and `overlayResizeCmd` are the only three producers, and `attachMessage` reports 0x0 below the paintable threshold so `handleAttach`'s own 80x24 default stands. The gate belongs at the FAN-OUTS: `paneVTSize` floors both dimensions at 1 on purpose for genuinely narrow SPLIT panes, so a degenerate geometry is indistinguishable from a legal one by the time it reaches the wire. A console-less client attaches at 1x1 (Bubble Tea reports that for a process with no console), and without the gate that resized every pane in a 48-tab workspace to one column and permanently re-wrapped every child's transcript — observed 2026-08-25 and 2026-09-05. It sits AHEAD of every `sizedOnce` write, so each pane is still owed its first-resize kick once a usable geometry arrives (a follower's `sizedOnce` is deliberately left unmarked for the same reason). `handleResizePane`/`handleResizePanes` repeat the refusal daemon-side — the degenerate-geometry check for BOTH dimensions at the floor together, and now also a master-only check (`applyResizes`): a follower's resize is dropped silently, with no log line, because an older-build follower sends one on every broadcast. A window resize or split-drag release batches every pane into ONE `resize_panes` frame, and the daemon answers with at most one `pane_sizes` frame per batch to each follower — never one per pane — sent before `pty.Resize` runs, so a follower's VT holds the new size before the child's own repaint at it arrives. See `.claude/rules/daemon-lifecycle.md`'s "Multi-client" section for the size-master election this all serves, and `.claude/rules/tui-rendering.md`'s "The resize gates and batching" / "Follower rendering" for the client side. Regression tests: `internal/tui/tinyterm_test.go` (driven through `Update`, because the defect was that the call sites were unconditional), `internal/daemon/resize_guard_test.go` - Pane restart boundary: `PaneOutputPayload.Generation` carries the daemon's PTY run counter on every live chunk. The TUI resets its VT before a replacement run's output and drops older frames; reattach forgets the counter because the daemon may have restarted. `newPaneSession` picks sibling/client dimensions before spawn, rejecting a degenerate client size via `degenerateSize` and falling back to 80x24. Keep the IPC fast-frame decoder and its wire-shape test in sync with this payload. - Pane naming: `MsgUpdatePane` IPC message, `Pane.Name` field in daemon, Alt+F2 keybinding to rename active pane (mirrors F2 tab rename pattern) diff --git a/.claude/rules/daemon-lifecycle.md b/.claude/rules/daemon-lifecycle.md index a6c1407c..6c43997b 100644 --- a/.claude/rules/daemon-lifecycle.md +++ b/.claude/rules/daemon-lifecycle.md @@ -196,19 +196,33 @@ discards a conn's hold without sending — `SendBlocking` would return **Layout revision (`Tab.LayoutRev`, spec §7).** Each tab's stored tree carries a `uint64` revision, persisted as `layout_rev` (omitempty) and put on the wire -in every broadcast's per-tab entry. `SetTabLayout` (`internal/daemon/project.go`, -under `sm.mu`, unchanged lock) now takes `UpdateLayoutPayload.BaseRev -*uint64`: a nil `BaseRev` (an older client) or one equal to the tab's current -revision stores, bumps the revision and broadcasts; a stale one is refused — -no store, no broadcast, a debug log only. **The `// No broadcastState() — -avoids feedback loop` note is retired**: a layout write now broadcasts, -because clients no longer re-send on mere disagreement (they only send a -change their OWN user made — see `tui-rendering.md`'s Layout sync), so the -loop the old comment guarded against cannot occur any more. The broadcast -itself goes through the SAME `requestBroadcast` 50 ms coalescer the snapshot -debounce already uses, so several tabs written in one burst (`sendAllLayouts` -after a border-drag release) still produce ONE frame, keeping the -must-deliver queue safe. +in every broadcast's per-tab entry. `SetTabLayout` (`SessionManager.SetTabLayout`, +`internal/daemon/session.go:1124`, under `sm.mu`, unchanged lock) now takes +`UpdateLayoutPayload.BaseRev *uint64`: a nil `BaseRev` (an older client) or one +equal to the tab's current revision stores, bumps the revision and +broadcasts; a stale one is refused — no store, no broadcast, a debug log +only. **The `// No broadcastState() — avoids feedback loop` note is +retired**: a layout write now broadcasts, because clients no longer re-send +on mere disagreement (they only send a change their OWN user made — see +`tui-rendering.md`'s Layout sync), so the loop the old comment guarded +against is now structurally impossible, not merely avoided by omission. + +The broadcast goes through its OWN trailing-edge coalescer, +`requestBroadcast` (`internal/daemon/broadcast_coalesce.go`, 50 ms, +single-flighted under its own `broadcastMu` — modeled on the snapshot +debounce's shape, but a separate timer, not a reuse of it). A burst it needs +to fold does not come from a border-drag release or an arrangement action — +`finishSplitDrag` and `applyTabArrangement` each call `markLayoutChanged` for +the ONE tab they changed, and there is no `sendAllLayouts` sweep any more. +It comes from `rebuildTabs` calling `markLayoutChanged` for SEVERAL tabs +within one broadcast-processing pass on the SAME client (each an empty +stored layout describing itself, or an awaiting-panes condition just +satisfied — see "Layout sync between clients" in `tui-rendering.md`), and +from several different clients each accepting a write inside the same +window. `sendDiffedLayouts` sends one `MsgUpdateLayout` frame PER TAB (unlike +the resize batching above), so those land as separate `handleUpdateLayout` +calls moments apart; the coalescer is what turns them back into one +broadcast rather than one per accepted write. **MCP unicast targets (`internal/daemon/mcp_targets.go`).** `close_tui` and `set_active_pane`'s focus frame used to broadcast to every attached TUI; both diff --git a/.claude/rules/templates.md b/.claude/rules/templates.md index 425289ef..dcd153ce 100644 --- a/.claude/rules/templates.md +++ b/.claude/rules/templates.md @@ -22,4 +22,4 @@ No branch: one completed state frame. Branch: preparing, existing swap, complete The template dialog routes all text/paste by `templateTextTarget`; selector rows swallow paste. Character input comes from `msg.Text`. The directory row is the shared `cwdBrowse*` browser, reset through `resetDirBrowseState` on open, Esc and a successful reply, and committed as `m.cwdBrowseDir` — never a typed field, so it is absent from `templateTextTarget`. Submission is blocked while `m.browse.pending`, or a create lands on the daemon default rather than the directory on screen. Enter creates from every row except the task editor (newline) and the directory row (descend); Ctrl+S creates from all four. Browsing and creation are stamped with the destination pinned on open. Only the requesting client's correlated response arms focus. Prompts are queued in listed order after all panes exist; substitution is single-pass. No prompt is delivered at all once any pane failed to be created: `{{panes}}` lists only panes that EXIST, so briefing a short roster starts a team without a worker it was told to use. The failed panes stay visible rather than rolling back a checkout that took minutes. A version number cannot gate a request type added on a branch: `dev.sh` stamps `VERSION` into every variant, dev included, so a locally built pair and a released daemon can report the same string. `VersionRespPayload.Requests` carries the gated types the daemon actually handles and `requireRequest` reads it; an absent list means "cannot say" and the version floor still decides there. Never re-derive this from the number in either direction — too strict refuses the daemon the client was built beside, too loose sends a released daemon a request it drops in silence. Codex output is not a reliable answer channel: request a file in prompts that need results. -The ordinary MCP server exposes 35 tools. `create_from_template` requires daemon 1.74.0; older project/tab/task tools retain 1.72.0. See `docs/workspace-templates.md` and `docs/mcp.md`. +The ordinary MCP server exposes 36 tools. `create_from_template` requires daemon 1.74.0; older project/tab/task tools retain 1.72.0. See `docs/workspace-templates.md` and `docs/mcp.md`. diff --git a/.claude/rules/tui-rendering.md b/.claude/rules/tui-rendering.md index 8feab078..2a656c2b 100644 --- a/.claude/rules/tui-rendering.md +++ b/.claude/rules/tui-rendering.md @@ -78,7 +78,7 @@ keys resolve to ACTIONS, not to config strings. `internal/keymap` owns: `ParseCh ### Split-border drag-resize -`CollectBorders`/`BorderHit` (`internal/tui/layout.go`) enumerate split lines (hit zone per line = the two drawn border glyphs + `splitBorderHitPadding` widening — symmetric for V-split rows, right-only for H-split columns so the zone never reaches the left neighbour's drawn scrollbar column at bd-2; reverse scan = deepest node wins at T-junctions). The border check runs BEFORE the scrollbar check in `MouseClickMsg` — the drawn split line always arms the drag (a scrollbar-first order silently ate the left glyph via scrollbar padding, with zero feedback on panes without scrollback), while thumb clicks on scrollbarX keep working because the border zone stops at bd-1; `hitTestSplitBorder`/`dragSplitBorder`/`finishSplitDrag` (`model.go`) arm/move/commit the drag — the ratio is clamped in cells against subtree minimums (`minWidth`/`minHeight`: leaves are 10×4, H-splits sum widths, V-splits sum heights) then derived, so boundaries are exact; PTY resize + layout persistence are deferred to mouse release (`resizeAllPanes` + `sendAllLayouts` — the on-release-only design avoids mid-drag PTY churn). Mid-drag only `Ratio` + pane RECTS move (`resizeNodeRects` in layout.go) — the VT emulator must NOT resize mid-drag: `ResizeVT`'s contract pairs every emulator resize with a PTY redraw, so unpaired intermediate-width rewraps permanently garble content at the narrowest width crossed (2026-07-15 corruption bug); the single VT+PTY resize pair fires together in `finishSplitDrag`. Panes whose rect touches the dragged line get a transient `splitDragHighlight` border (color 39, included in `renderKey`), set/cleared via `setSplitDragHighlight`. Disabled in focus mode, notes mode, and single-pane tabs; scrollbar hit test keeps priority. Drag state (`splitDragNode`/`splitDragRect`) rides `clearDragState()`; a node pruned mid-drag (workspace reconciliation) drops the drag via `treeContains`. Tests in `splitdrag_test.go` +`CollectBorders`/`BorderHit` (`internal/tui/layout.go`) enumerate split lines (hit zone per line = the two drawn border glyphs + `splitBorderHitPadding` widening — symmetric for V-split rows, right-only for H-split columns so the zone never reaches the left neighbour's drawn scrollbar column at bd-2; reverse scan = deepest node wins at T-junctions). The border check runs BEFORE the scrollbar check in `MouseClickMsg` — the drawn split line always arms the drag (a scrollbar-first order silently ate the left glyph via scrollbar padding, with zero feedback on panes without scrollback), while thumb clicks on scrollbarX keep working because the border zone stops at bd-1; `hitTestSplitBorder`/`dragSplitBorder`/`finishSplitDrag` (`model.go`) arm/move/commit the drag — the ratio is clamped in cells against subtree minimums (`minWidth`/`minHeight`: leaves are 10×4, H-splits sum widths, V-splits sum heights) then derived, so boundaries are exact; PTY resize + layout persistence are deferred to mouse release (`resizeAllPanes` + `markLayoutChanged` for the one tab that moved — the on-release-only design avoids mid-drag PTY churn). Mid-drag only `Ratio` + pane RECTS move (`resizeNodeRects` in layout.go) — the VT emulator must NOT resize mid-drag: `ResizeVT`'s contract pairs every emulator resize with a PTY redraw, so unpaired intermediate-width rewraps permanently garble content at the narrowest width crossed (2026-07-15 corruption bug); the single VT+PTY resize pair fires together in `finishSplitDrag`. Panes whose rect touches the dragged line get a transient `splitDragHighlight` border (color 39, included in `renderKey`), set/cleared via `setSplitDragHighlight`. Disabled in focus mode, notes mode, and single-pane tabs; scrollbar hit test keeps priority. Drag state (`splitDragNode`/`splitDragRect`) rides `clearDragState()`; a node pruned mid-drag (workspace reconciliation) drops the drag via `treeContains`. Tests in `splitdrag_test.go` ### Pane drag (Alt+drag) diff --git a/docs/README.md b/docs/README.md index a653e0d9..e4225857 100644 --- a/docs/README.md +++ b/docs/README.md @@ -20,7 +20,7 @@ | [Keybindings](keybindings.md) | Full keymap, customization syntax, what to bind and what to leave for the PTY | | [tmux comparison](tmux-comparison.md) | The `tmux` preset against tmux's own default prefix table, and how to switch between keymaps | | [Configuration](configuration.md) | `~/.quil/config.toml` reference — every section + every key | -| [MCP](mcp.md) | **Let your AI assistant drive Quil.** Wiring for Claude Desktop / Claude Code / Cursor / VS Code Copilot + all 35 tools documented + redaction model | +| [MCP](mcp.md) | **Let your AI assistant drive Quil.** Wiring for Claude Desktop / Claude Code / Cursor / VS Code Copilot + all 36 tools documented + redaction model | | [Workspace templates](workspace-templates.md) | Create named panes, layouts and starting prompts from a file; edit templates through F1 | ## Customization diff --git a/docs/competitive-analysis.md b/docs/competitive-analysis.md index 817120f6..01d40500 100644 --- a/docs/competitive-analysis.md +++ b/docs/competitive-analysis.md @@ -38,7 +38,7 @@ architectural bet. | VT emulation | `charmbracelet/x/vt` | Vendored libghostty-vt (Ghostty engine) | `vt100` crate over `tmux pipe-pane` | | Client/server | daemon + TUI client | server + thin client(s) | tmux + TUI + optional HTTP daemon | | **Windows** | ✅ **Native** (bundled ConPTY/OpenConsole) | ✅ Native, **GA** (ConPTY) as of 2026-09-10 — no terminal attach, no live handoff, no clipboard image bridge | ❌ **WSL2 only** | -| Agent-drives-it API | **MCP server** (35 tools, native protocol) | Socket API + full CLI + agent skill | HTTP REST API (130 routes) + CLI | +| Agent-drives-it API | **MCP server** (36 tools, native protocol) | Socket API + full CLI + agent skill | HTTP REST API (130 routes) + CLI | | Web/browser UI | ❌ TUI only | ❌ (responsive TUI) | ✅ **React PWA dashboard** | | Container sandbox | ✅ **Docker** (per-pane, user-supplied image) | ❌ | ✅ Docker/Podman/Apple | | Remote phone access | ❌ | via SSH TUI | ✅ Tunnel + PWA + Web Push | @@ -157,7 +157,7 @@ Legend: ✅ full · 🟡 partial/different · ❌ absent · ❓ not re-verified | Executable plugins (any language) | ✅ | ✅ (design) | ❌ | | Plugin actions / event hooks / link handlers | ✅ | ✅ | ❌ | | Plugin marketplace (GitHub topic index) | ✅ | ✅ (featured + hash) | ❌ | -| Agent-drives-multiplexer API | ✅ socket+CLI | ✅ HTTP+CLI | ✅ **MCP (35 tools)** | +| Agent-drives-multiplexer API | ✅ socket+CLI | ✅ HTTP+CLI | ✅ **MCP (36 tools)** | | Subscribable event stream (`events.subscribe`) | ✅ (workspace/tab/pane/layout/worktree lifecycle) | 🟡 | 🟡 (`watch_notifications` + task completion only) | | Wait on *semantic* agent state (`--until done/blocked`) | ✅ | 🟡 | ✅ (`wait_task`, daemon-side ledger) | | Portable layout export / declarative apply | ✅ (`layout.*`) | ❌ | ❌ (layout persists, but is not exportable) | @@ -343,7 +343,7 @@ jobs, and Quil is not an emulator. executable cannot be replaced in place). It is a shared limitation of the category, not a wedge, and it was written into the site copy once during the 2026-09-10 pass before being caught. -- **First-class MCP server.** Quil exposes 35 MCP tools that Claude Desktop / +- **First-class MCP server.** Quil exposes 36 MCP tools that Claude Desktop / Cursor / VS Code consume *natively*, including projects, remote hosts and pane-to-pane task delegation. The competitors built bespoke socket/HTTP APIs that need an "agent skill" to teach. For the *AI-agent-as-operator* use diff --git a/docs/prd.md b/docs/prd.md index 29346b37..841d454a 100644 --- a/docs/prd.md +++ b/docs/prd.md @@ -633,7 +633,7 @@ The PRD captures the original v1 plan. The product has shipped past M5 in tightl | **M6: Pane Focus Mode** | Done | `Ctrl+E` toggles the active pane to fill the tab — other panes keep running, layout tree intact, focus state not persisted | | **M7: Pane Notes** | Done | `Alt+E` opens a plain-text notes editor next to the active pane; one file per pane (`~/.quil/notes/.md`); 30 s debounced auto-save + explicit `Ctrl+S` + flush on exit. Notes outlive the pane. See [roadmap/pane-notes.md](roadmap/pane-notes.md) | | **M8: Bubble Tea v2 Migration** | Done | Migrated to Bubble Tea v2 (`charm.land/bubbletea/v2`) and Lipgloss v2 with declarative `View` / typed mouse events / `KeyPressMsg`. Added platform-native clipboard, terminal text selection, editor selection / clipboard / word jumps, beta disclaimer dialog, runtime `config.Save()` | -| **M10: MCP Server** | Done | `quil mcp` exposes 35 tools over Model Context Protocol stdio so any MCP-capable client (Claude Desktop, Claude Code, Cursor) can drive the live workspace, manage projects across remote hosts, and delegate tasks between AI panes. See [roadmap/mcp-server.md](roadmap/mcp-server.md) | +| **M10: MCP Server** | Done | `quil mcp` exposes 36 tools over Model Context Protocol stdio so any MCP-capable client (Claude Desktop, Claude Code, Cursor) can drive the live workspace, manage projects across remote hosts, and delegate tasks between AI panes. See [roadmap/mcp-server.md](roadmap/mcp-server.md) | | **M12: Notification Center** | Done | Daemon event queue with process-exit / OSC 133 / bell / smart-idle detection; non-modal `Alt+N` sidebar with severity colours and a pane-history stack (`Alt+Backspace`); blocking and non-blocking MCP tools. See [roadmap/notification-center.md](roadmap/notification-center.md) | | **M13: Memory Reporting** | Done | Per-pane Go-heap + PTY RSS surfaced in a status-bar `mem ` segment and two MCP tools (`get_memory_report`, `get_pane_memory`). Cross-platform RSS via `/proc//status` / `ps` / `GetProcessMemoryInfo`. The F1 → Memory dialog was replaced in v1.63.0 by F1 → Processes, which adds the full process tree and CPU | | **v1.8.0+ patch milestones** | Done | Client/daemon version handshake (auto-restart on mismatch), VT-emulator drain goroutine + Update watchdog, claude-code SessionStart hook for session-id rotation, notes editor soft-wrap. See [CHANGELOG.md](../CHANGELOG.md) | diff --git a/docs/quick-start.md b/docs/quick-start.md index a139f2e1..6ae73a89 100644 --- a/docs/quick-start.md +++ b/docs/quick-start.md @@ -102,7 +102,7 @@ Restart the client. Then ask the AI: You should see a JSON array of every pane with its id, type, tab, and CWD. If you don't, see [MCP → Troubleshooting](mcp.md#troubleshooting). -The full [MCP guide](mcp.md) covers all 35 tools, wiring for Claude Code / Cursor / VS Code, the redaction model for secrets, and example prompts. +The full [MCP guide](mcp.md) covers all 36 tools, wiring for Claude Code / Cursor / VS Code, the redaction model for secrets, and example prompts. ## Where to go next diff --git a/docs/roadmap.md b/docs/roadmap.md index 0a1bcf4d..4c21c701 100644 --- a/docs/roadmap.md +++ b/docs/roadmap.md @@ -106,7 +106,7 @@ Key capabilities: `quil mcp` subcommand bridges MCP JSON-RPC (stdio) to daemon IPC (socket). AI assistants can read pane output, send commands, take screenshots, navigate tabs, restart panes, and control the TUI. No other terminal multiplexer offers this. Key capabilities: -- **35 MCP tools** — Phase A (workspace control): `list_panes`, `read_pane_output`, `send_to_pane`, `get_pane_status`, `create_pane`. Phase B (interaction + introspection): `send_keys`, `restart_pane`, `screenshot_pane`, `switch_tab`, `list_tabs`, `destroy_pane`, `set_active_pane`, `close_tui`. Notification Center (M12): `get_notifications`, `watch_notifications`, `dismiss_notifications`. Memory Reporting (M13): `get_memory_report`, `get_pane_memory`. Projects, hosts and tasks (2026-09): `list_projects`, `create_project`, `update_project`, `switch_project`, `destroy_project`, `create_tab`, `create_from_template`, `rename_tab`, `destroy_tab`, `rename_pane`, `list_plugins`, `list_sessions`, `list_hosts`, `delegate_task`, `get_task`, `wait_task`, `list_tasks`. Workspace templates add ordered pane creation with frozen arguments, prompts and layouts +- **36 MCP tools** — Phase A (workspace control): `list_panes`, `read_pane_output`, `send_to_pane`, `get_pane_status`, `create_pane`. Phase B (interaction + introspection): `send_keys`, `restart_pane`, `screenshot_pane`, `switch_tab`, `list_tabs`, `destroy_pane`, `set_active_pane`, `close_tui`. Notification Center (M12): `get_notifications`, `watch_notifications`, `dismiss_notifications`. Memory Reporting (M13): `get_memory_report`, `get_pane_memory`. Projects, hosts and tasks (2026-09): `list_projects`, `create_project`, `update_project`, `switch_project`, `destroy_project`, `create_tab`, `create_from_template`, `rename_tab`, `destroy_tab`, `rename_pane`, `list_plugins`, `list_sessions`, `list_hosts`, `delegate_task`, `get_task`, `wait_task`, `list_tasks`. Workspace templates add ordered pane creation with frozen arguments, prompts and layouts - **Official MCP SDK** — `modelcontextprotocol/go-sdk` v1.4+, typed tool handlers with struct-based input schemas - **Request-response IPC** — backward-compatible `Message.ID` field for correlation; daemon responds to specific connection - **VT-emulated screenshots** — `charmbracelet/x/vt` renders ring buffer into text grid showing actual screen state From 5441512f4448f237aa6e4a147159ea62f15fcbc6 Mon Sep 17 00:00:00 2001 From: Artjoms Stukans Date: Sat, 26 Sep 2026 02:17:48 +0200 Subject: [PATCH 24/40] fix(daemon): let a cold start take a restart-reserved master slot A daemon restart kept the previous master's slot for min(grace, 30 s) with no condition. After an unclean stop (reboot, kill) the next TUI is a new process with a new id, so it sat as a follower for the whole reserve with nobody else attached to protect. AttachPayload gains Reattach. The TUI sets it only on the attach its reconnect path sends. While a restart reserve is active, a FIRST attach from a different id that is the only attached client clears the reserve and is elected at once. Reconnecting clients still respect the reserve, and the reserved id reattaching still reclaims it. --- internal/daemon/clients.go | 16 ++++++-- internal/daemon/clients_test.go | 38 +++++++++++++++++-- internal/ipc/protocol.go | 7 ++++ internal/tui/model.go | 11 ++++-- internal/tui/multiclient_role_test.go | 54 ++++++++++++++++++++++++++- internal/tui/tinyterm_test.go | 2 +- 6 files changed, 117 insertions(+), 11 deletions(-) diff --git a/internal/daemon/clients.go b/internal/daemon/clients.go index ba9de203..351c601d 100644 --- a/internal/daemon/clients.go +++ b/internal/daemon/clients.go @@ -239,7 +239,10 @@ func (r *clientRegistry) expire() { // attach records conn as the client id at a RAW geometry and elects. An empty // id is minted as "anon-", scoped to the conn: a re-attach on the same // conn keeps it. -func (r *clientRegistry) attach(conn *ipc.Conn, id string, cols, rows int, cwd string) clientChange { +// +// reattach is the payload's Reattach flag: false on a process's first attach +// to this daemon, which a restart reserve yields to when it is alone. +func (r *clientRegistry) attach(conn *ipc.Conn, id string, cols, rows int, cwd string, reattach bool) clientChange { r.mu.Lock() defer r.mu.Unlock() if r.byConn == nil { @@ -280,6 +283,12 @@ func (r *clientRegistry) attach(conn *ipc.Conn, id string, cols, rows int, cwd s // The reserved client is back, so the slot has nothing left to wait // for. The election below keeps it when the client is eligible. r.clearReservationLocked() + } else if res != nil && res.protects == nil && !reattach && len(r.byConn) == 1 { + // A restart reserve waits for TUIs RECONNECTING after the restart. A + // new process attaching alone is a cold start after an unclean stop + // (reboot, kill): the previous master went with its process, and + // waiting would make the only client a follower for the reserve. + r.clearReservationLocked() } return clientChange{master: r.electLocked(), count: len(r.byConn) != before} } @@ -356,7 +365,8 @@ func (r *clientRegistry) takeControl(conn *ipc.Conn) (changed, accepted bool) { // reserveAfterRestart keeps a restored size_master's slot for // min(grace, restartReserveCap), with no follower condition: after a restart -// nobody is attached yet, and the TUIs reattach with their same ids. +// nobody is attached yet, and the TUIs reattach with their same ids. A new +// process's FIRST attach, alone on the daemon, ends it early (see attach). func (r *clientRegistry) reserveAfterRestart(id string) { id = truncateField(id, maxClientIDLen) d := min(r.grace, restartReserveCap) @@ -394,7 +404,7 @@ func (d *Daemon) attachClient(conn *ipc.Conn, attach ipc.AttachPayload) clientCh return clientChange{} } id := truncateField(attach.ClientID, maxClientIDLen) - return d.clients.attach(conn, id, clampClientDim(attach.Cols), clampClientDim(attach.Rows), attach.CWD) + return d.clients.attach(conn, id, clampClientDim(attach.Cols), clampClientDim(attach.Rows), attach.CWD, attach.Reattach) } // clampClientDim bounds one axis of a self-reported window size to diff --git a/internal/daemon/clients_test.go b/internal/daemon/clients_test.go index 88f52c67..73bfee33 100644 --- a/internal/daemon/clients_test.go +++ b/internal/daemon/clients_test.go @@ -87,6 +87,13 @@ func (h *clientsHarness) attach(id string, cols, rows int) (*ipc.Conn, bool) { return c, changed } +// reattach is attach from a client's reconnect path: Reattach is set. +func (h *clientsHarness) reattach(id string, cols, rows int) (*ipc.Conn, bool) { + c := new(ipc.Conn) + changed := h.d.registerClient(c, ipc.AttachPayload{ClientID: id, Cols: cols, Rows: rows, Reattach: true}) + return c, changed +} + func (h *clientsHarness) wantMaster(want string) { h.t.Helper() if got := h.d.masterID(); got != want { @@ -332,9 +339,9 @@ func TestRestartReserve_PreviousMasterReclaims(t *testing.T) { t.Run("previous master reattaches after another client", func(t *testing.T) { h := restartedWithMaster(t) - b, changed := h.attach("B", 100, 30) + b, changed := h.reattach("B", 100, 30) if changed || h.d.isMasterConn(b) { - t.Error("a client attaching first after a restart must not take the reserved slot") + t.Error("a client reattaching first after a restart must not take the reserved slot") } if h.d.sizeAuthorityOpen() { t.Error("the restart reservation must keep size authority closed") @@ -354,7 +361,7 @@ func TestRestartReserve_PreviousMasterReclaims(t *testing.T) { t.Run("previous master never comes", func(t *testing.T) { h := restartedWithMaster(t) - b, _ := h.attach("B", 100, 30) + b, _ := h.reattach("B", 100, 30) h.advance(restartReserveCap) h.fire() if !h.d.isMasterConn(b) { @@ -364,6 +371,31 @@ func TestRestartReserve_PreviousMasterReclaims(t *testing.T) { t.Errorf("changes = %d, want 1", h.changes) } }) + + // After an unclean stop (reboot, kill) the next TUI is a new process with + // a new id. Kept waiting on the reserve, it would be a follower for 30 s + // with nobody else attached to protect. + t.Run("fresh single first attach takes the slot at once", func(t *testing.T) { + h := restartedWithMaster(t) + b, changed := h.attach("B", 100, 30) + if !changed || !h.d.isMasterConn(b) { + t.Error("a new process attaching alone during the restart reserve must be the master at once") + } + if h.armed() != nil { + t.Error("the restart timer must be stopped once the reserve yields") + } + h.wantMaster("B") + }) + + t.Run("fresh first attach beside a reattached client keeps the reserve", func(t *testing.T) { + h := restartedWithMaster(t) + h.reattach("B", 100, 30) + c, changed := h.attach("C", 100, 30) + if changed || h.d.isMasterConn(c) { + t.Error("a first attach that is not the only client must not take the reserved slot") + } + h.wantMaster("A") + }) } // Stop closes every conn, and each close runs onClientDisconnect concurrently diff --git a/internal/ipc/protocol.go b/internal/ipc/protocol.go index cb68015d..64189e42 100644 --- a/internal/ipc/protocol.go +++ b/internal/ipc/protocol.go @@ -327,6 +327,13 @@ type AttachPayload struct { // that cannot participate in election — it is neither offered control // nor handed a follower's resize_panes stream. ClientID string `json:"client_id,omitempty"` + // Reattach is true on an attach sent from the client's reconnect path, + // and false on the process's first attach to this daemon. A daemon + // restart keeps the previous master's slot for a short reserve so the + // reconnecting TUIs resize nothing; a FIRST attach from a new process, + // alone on the daemon, is a cold start after an unclean stop, and the + // reserve yields to it rather than making it a follower for 30 s. + Reattach bool `json:"reattach,omitempty"` } type CreatePanePayload struct { diff --git a/internal/tui/model.go b/internal/tui/model.go index 7d53473a..81cce48d 100644 --- a/internal/tui/model.go +++ b/internal/tui/model.go @@ -7934,7 +7934,7 @@ func (m *Model) attachAllDests() tea.Cmd { if m.attached[dest] { continue } - if err := m.sendForDest(dest, m.attachMessage(dest)); err != nil { + if err := m.sendForDest(dest, m.attachMessage(dest, false)); err != nil { log.Printf("attach to %q failed, retrying on the next resize: %v", dest, err) continue } @@ -7997,7 +7997,11 @@ func (m *Model) attachAllDests() tea.Cmd { // attachMessage builds the MsgAttach describing this client's geometry for one // destination. -func (m Model) attachMessage(dest string) *ipc.Message { +// +// reattach is true only from the reconnect path (attachToDest). A daemon that +// just restarted keeps the previous master's slot for its reconnecting TUIs, +// and yields it to a first attach from a new process that is alone there. +func (m Model) attachMessage(dest string, reattach bool) *ipc.Message { // Subtract chrome (tab bar + status bar), the project sidebar (if open), // then pane border (2) — the same reservation resizeTabs applies, so the // very first spawned pane isn't sized wider than what's about to be @@ -8030,6 +8034,7 @@ func (m Model) attachMessage(dest string) *ipc.Message { Rows: rows, CWD: attachCWD(dest, localCWD), ClientID: m.clientID, + Reattach: reattach, }) return msg } @@ -8055,7 +8060,7 @@ func (m Model) attachMessage(dest string) *ipc.Message { // sets m.attached[dest] — so its copy of this report does not cover it. func (m Model) attachToDest(dest string) tea.Cmd { attachCmd := func() tea.Msg { - m.sendForDest(dest, m.attachMessage(dest)) + m.sendForDest(dest, m.attachMessage(dest, true)) return nil } return tea.Batch(attachCmd, m.requestPluginListFor(dest), m.overlayTruthDestCmd(dest), diff --git a/internal/tui/multiclient_role_test.go b/internal/tui/multiclient_role_test.go index 093e6fe6..1ca27523 100644 --- a/internal/tui/multiclient_role_test.go +++ b/internal/tui/multiclient_role_test.go @@ -456,7 +456,7 @@ func TestAttach_CarriesClientID(t *testing.T) { m.SetClientID("my-client-id") var p ipc.AttachPayload - if err := m.attachMessage("").DecodePayload(&p); err != nil { + if err := m.attachMessage("", false).DecodePayload(&p); err != nil { t.Fatalf("decode attach payload: %v", err) } if p.ClientID != "my-client-id" { @@ -464,6 +464,58 @@ func TestAttach_CarriesClientID(t *testing.T) { } } +// attachPayloads decodes every MsgAttach a conn received, in order. +func attachPayloads(t *testing.T, conn *fakeConn) []ipc.AttachPayload { + t.Helper() + conn.mu.Lock() + defer conn.mu.Unlock() + var out []ipc.AttachPayload + for _, msg := range conn.sent { + if msg.Type != ipc.MsgAttach { + continue + } + var p ipc.AttachPayload + if err := msg.DecodePayload(&p); err != nil { + t.Fatalf("decode attach payload: %v", err) + } + out = append(out, p) + } + return out +} + +// TestAttach_ReattachFlagMarksOnlyTheReconnectPath: the daemon's restart +// reserve yields to a FIRST attach from a new process and keeps waiting for a +// reconnecting one, so the flag must be false on the attach a WindowSizeMsg +// sends and true on the one a completed redial sends. +func TestAttach_ReattachFlagMarksOnlyTheReconnectPath(t *testing.T) { + t.Setenv("QUIL_HOME", t.TempDir()) + first, fresh := newFakeConn(), newFakeConn() + r := NewRouter(map[string]Client{"gpu01": first}) + var m tea.Model = Model{client: r, cfg: config.Default()} + + m, cmd := m.Update(tea.WindowSizeMsg{Width: 120, Height: 40}) + runCmdNoWait(cmd) + got := attachPayloads(t, first) + if len(got) != 1 { + t.Fatalf("first attach count = %d, want 1", len(got)) + } + if got[0].Reattach { + t.Error("the process's first attach must not set reattach") + } + + mm := m.(Model) + mm.links = map[string]*reconnectState{"gpu01": {active: true, gen: 1}} + m, cmd = mm.Update(redialResultMsg{gen: 1, dest: "gpu01", client: fresh}) + runCmdNoWait(cmd) + got = attachPayloads(t, fresh) + if len(got) != 1 { + t.Fatalf("reconnect attach count = %d, want 1", len(got)) + } + if !got[0].Reattach { + t.Error("the attach sent from the reconnect path must set reattach") + } +} + // TestClientGeometryCmd_ReportsZeroWhenUnpaintable mirrors // TestAttachMessage_ReportsNoGeometryBelowTheMinimum: a terminal too small to // paint must report 0x0, never the raw sub-floor size. The daemon uses this diff --git a/internal/tui/tinyterm_test.go b/internal/tui/tinyterm_test.go index aee53a97..2b0cd76c 100644 --- a/internal/tui/tinyterm_test.go +++ b/internal/tui/tinyterm_test.go @@ -274,7 +274,7 @@ func TestAttachMessage_ReportsNoGeometryBelowTheMinimum(t *testing.T) { m := Model{cfg: config.Default(), width: tt.width, height: tt.height} var p ipc.AttachPayload - if err := m.attachMessage("").DecodePayload(&p); err != nil { + if err := m.attachMessage("", false).DecodePayload(&p); err != nil { t.Fatalf("decode attach payload: %v", err) } From 68e09b87d72b073dc0d1aae41543be217693c40c Mon Sep 17 00:00:00 2001 From: Artjoms Stukans Date: Sat, 26 Sep 2026 02:20:38 +0200 Subject: [PATCH 25/40] fix(tui): record a requested-tab token on every pane jump jumpToPane moves this client's active tab for MCP set_active_pane, sidebar clicks, Alt+Backspace, the palette and attention jumps, but it recorded no requestedTab token. A broadcast in flight from before the jump named the old tab and read as another client switching back: the tab jumped back, the typing guard armed and the flash showed. The jump now records the same token switchTab does, with the project's own previous tab as the stale-echo candidate. A jump that keeps the project's tab records nothing, so it cannot overwrite a pending token. --- internal/tui/typing_guard_test.go | 38 +++++++++++++++++++++++++++++++ internal/tui/workstate.go | 16 +++++++++++++ 2 files changed, 54 insertions(+) diff --git a/internal/tui/typing_guard_test.go b/internal/tui/typing_guard_test.go index bc97bb6a..6e712796 100644 --- a/internal/tui/typing_guard_test.go +++ b/internal/tui/typing_guard_test.go @@ -663,6 +663,44 @@ func TestTypingGuard_StaleFromTabBroadcastIsRejectedWhilePending(t *testing.T) { } } +// TestTypingGuard_JumpRecordsARequestedTab: jumpToPane (MCP set_active_pane, +// sidebar clicks, Alt+Backspace, the palette, attention jumps) moves this +// client's active tab exactly as switchTab does, so it must record the same +// token. Without one, a broadcast still in flight from before the jump names +// the old tab and reads as another client switching back: the tab jumps +// back, the guard arms and the flash shows. +func TestTypingGuard_JumpRecordsARequestedTab(t *testing.T) { + t.Parallel() + t0 := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + m, _ := typingGuardModel(t, t0, "t1", "t2") + + updated, _ := m.Update(setActivePaneMsg{PaneID: "p2"}) + got := updated.(Model) + if got.activeTabModel().ID != "t2" { + t.Fatalf("setup: active tab = %q, want t2", got.activeTabModel().ID) + } + + got.now = func() time.Time { return t0.Add(100 * time.Millisecond) } + updated, _ = got.Update(typingGuardBroadcast("t1", "t1", "t2")) + got2 := updated.(Model) + if got2.activeTabModel().ID != "t2" { + t.Errorf("active tab = %q after a stale t1 broadcast, want t2 (no jump back)", got2.activeTabModel().ID) + } + if got2.guardPaneID != "" { + t.Errorf("guardPaneID = %q, want empty — the jump was this client's own", got2.guardPaneID) + } + if got2.flashText != "" { + t.Errorf("flashText = %q, want empty", got2.flashText) + } + + // The jump's own echo lands and retires the token. + updated, _ = got2.Update(typingGuardBroadcast("t2", "t1", "t2")) + got3 := updated.(Model) + if got3.guardPaneID != "" || got3.flashText != "" { + t.Errorf("the jump's echo armed the guard (%q) or flashed (%q)", got3.guardPaneID, got3.flashText) + } +} + // TestTypingGuard_StaleFromTabBroadcastIsAdoptedPastTheBound is the other // side of requestedSwitchStaleWindow: once it elapses, the local switch is // assumed lost (never reached the daemon, or was overtaken), and a broadcast diff --git a/internal/tui/workstate.go b/internal/tui/workstate.go index d452535d..a8e8a6a2 100644 --- a/internal/tui/workstate.go +++ b/internal/tui/workstate.go @@ -194,9 +194,25 @@ func (m *Model) jumpToPane(paneID string) (bool, tea.Cmd) { break } } + // The tab this project showed before the jump, for the typing guard's + // token below. It is the project's OWN previous tab, not `from`: after a + // cross-project jump `from` is in another project, while a broadcast in + // flight from before the jump names this project's old active tab. + prevID := "" + if prev := proj.activeTab; prev >= 0 && prev < len(proj.tabs) { + prevID = proj.tabs[prev].ID + } proj.activeTab = tabIdx target := proj.tabs[tabIdx] target.ActivePane = paneID + // Typing guard (spec §8.1): this jump is THIS client's own switch, exactly + // as switchTab's is, so its broadcast must not read as another client's. + // See switchTab for the token and requestedSwitchStaleWindow for prevID. + // A jump that keeps the project's tab records nothing, so it cannot + // overwrite a token an earlier switch is still waiting on. + if prevID != target.ID { + m.recordRequestedTab(target.Dest, proj.ID, target.ID, prevID) + } // The Active FLAG is set here, not left to the caller. ActivePane alone // routes keystrokes, but Active is what draws the pane's cursor // (renderPane) and its focused border — so a pane raised without it looks From 2601955587a497bcf5c7e381920c6b6e1ff2d9d4 Mon Sep 17 00:00:00 2001 From: Artjoms Stukans Date: Sat, 26 Sep 2026 02:22:31 +0200 Subject: [PATCH 26/40] fix(tui): adopt the daemon's tab when a held stale-reject tab is gone A broadcast naming the tab this client just left is rejected as a stale echo, holding the requested tab. If another client destroyed that tab inside the window, the held id matched nothing and the project fell back to tab index 0, a tab nobody chose. The reject now holds the tab this client is actually on, and only while the broadcast still lists it. Otherwise the token is retired and the daemon's tab is adopted. --- internal/tui/model.go | 19 ++++++++++++----- internal/tui/typing_guard_test.go | 34 +++++++++++++++++++++++++++++++ 2 files changed, 48 insertions(+), 5 deletions(-) diff --git a/internal/tui/model.go b/internal/tui/model.go index 81cce48d..f1cb3f80 100644 --- a/internal/tui/model.go +++ b/internal/tui/model.go @@ -6190,9 +6190,18 @@ func (m *Model) applyTabMoveGuard(dest, projectID, newActiveTab string, fromTab // switch back to it (a genuine one arrives, if it happens at // all, only after this client's own confirmation, by which // point the token above is gone and this case cannot match). - // Reject it outright: hold the tab at what we asked for, keep - // the token, arm no guard. - return req.target, nil + // Reject it outright: hold the tab this client is on, keep the + // token, arm no guard. + // + // Only while that tab still exists. Another client can destroy + // it inside the window, and holding a missing id resolves to + // index 0 rather than to anything anyone chose: then the + // daemon's tab is adopted, and the token has nothing left to + // wait for. + if fromTab != nil && tabsContainID(tabs, fromTab.ID) { + return fromTab.ID, nil + } + delete(m.requestedTab, key) } } if fromTab == nil || fromTab.ID == newActiveTab { @@ -6362,8 +6371,8 @@ func (m *Model) applyWorkspaceState(state WorkspaceStateMsg, dest string) ([]str // inside a "moved" branch. ok (the project already existed) is what // makes the token lookup meaningful; a brand new project has none. // effectiveActiveTab may differ from info.ActiveTab: a rejected stale - // broadcast (pendingSwitch.from) holds the tab at this client's own - // pending request instead of adopting the daemon's report. + // broadcast (pendingSwitch.from) holds the tab this client is on, + // while that tab still exists, instead of adopting the daemon's report. effectiveActiveTab := info.ActiveTab if info.ID == activeID && ok { _, existedBefore := existingTabs[info.ActiveTab] diff --git a/internal/tui/typing_guard_test.go b/internal/tui/typing_guard_test.go index 6e712796..077e0336 100644 --- a/internal/tui/typing_guard_test.go +++ b/internal/tui/typing_guard_test.go @@ -701,6 +701,40 @@ func TestTypingGuard_JumpRecordsARequestedTab(t *testing.T) { } } +// TestTypingGuard_StaleRejectAdoptsTheDaemonTabWhenTheTargetIsGone: a +// broadcast naming the tab this client just left is normally a stale echo and +// is rejected, holding the tab the client is on. When another client +// DESTROYED that tab inside the window, there is nothing left to hold: the +// broadcast must be adopted, landing on the daemon's tab rather than whatever +// index a missing id resolves to. +func TestTypingGuard_StaleRejectAdoptsTheDaemonTabWhenTheTargetIsGone(t *testing.T) { + t.Parallel() + t0 := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + m, _ := typingGuardModel(t, t0, "t1", "t2", "t3") + + updated, _ := m.Update(altKey('2')) + got := updated.(Model) + updated, _ = got.Update(typingGuardBroadcast("t2", "t1", "t2", "t3")) // echo retires the token + got = updated.(Model) + updated, _ = got.Update(altKey('3')) // requests t3, leaving t2 + got = updated.(Model) + if got.activeTabModel().ID != "t3" { + t.Fatalf("setup: active tab = %q, want t3", got.activeTabModel().ID) + } + + // Another client destroyed t3 inside the stale window; the daemon's + // active tab for the project is t2. + got.now = func() time.Time { return t0.Add(500 * time.Millisecond) } + updated, _ = got.Update(typingGuardBroadcast("t2", "t1", "t2")) + got2 := updated.(Model) + if got2.activeTabModel().ID != "t2" { + t.Errorf("active tab = %q, want t2 (the daemon's tab, since t3 is gone)", got2.activeTabModel().ID) + } + if got2.guardPaneID != "" { + t.Errorf("guardPaneID = %q, want empty — the tab this client was on is gone", got2.guardPaneID) + } +} + // TestTypingGuard_StaleFromTabBroadcastIsAdoptedPastTheBound is the other // side of requestedSwitchStaleWindow: once it elapses, the local switch is // assumed lost (never reached the daemon, or was overtaken), and a broadcast From 89da31436a91d492e142a7e7522700b2aed2b074 Mon Sep 17 00:00:00 2001 From: Artjoms Stukans Date: Sat, 26 Sep 2026 02:24:29 +0200 Subject: [PATCH 27/40] test(daemon): probe the hold-begin park instead of sleeping TestHold_FlushStraddlingTheBeginArrivesOnce slept 50 ms and hoped beginOutputHold had reached holdGate.Lock by then. Under load it may not have, and the test then passed without exercising the straddle. Poll until holdGate.TryRLock fails instead: the paused flush holds the gate for read, and a pending writer blocks new readers, so a failed TryRLock proves the begin is parked. A read lock the probe does get is released at once. --- internal/daemon/outputhold_test.go | 12 +++++++++++- 1 file changed, 11 insertions(+), 1 deletion(-) diff --git a/internal/daemon/outputhold_test.go b/internal/daemon/outputhold_test.go index ce1ee148..9fe1197d 100644 --- a/internal/daemon/outputhold_test.go +++ b/internal/daemon/outputhold_test.go @@ -518,7 +518,17 @@ func TestHold_FlushStraddlingTheBeginArrivesOnce(t *testing.T) { h.d.beginOutputHold(conn) close(begun) }() - time.Sleep(50 * time.Millisecond) // let the begin set the flag, if it can + // The paused flush holds holdGate for read, so the begin parks in + // holdGate.Lock. A PENDING writer blocks new readers, so TryRLock failing + // is the proof it got there; a read lock the probe does get is released + // at once so the writer is never held up by it. + waitUntil(t, "beginOutputHold parked on holdGate", func() bool { + if h.d.holdGate.TryRLock() { + h.d.holdGate.RUnlock() + return false + } + return true + }) close(resume) <-flushed <-begun From f01322a4016c0f0cb162d319752812b4a6ec11dc Mon Sep 17 00:00:00 2001 From: Artjoms Stukans Date: Sat, 26 Sep 2026 02:25:37 +0200 Subject: [PATCH 28/40] docs: correct four stale multi-client comments - eventDismissedMsg: DismissByID now scopes a dismiss-all to the sending daemon; the comment still said the sidebar was unscoped. - ClientInfo.Role: MCP bridges never attach, so they are never listed; the comment said they shared the attach path. - sendStateToOtherClients logged "attach:" from the detach and lost-link paths too; the log prefix now names the caller. - defaultCWD: state the accepted shared-deadline trade-off, where a dead earlier candidate can make a later live one fall through to the daemon's own directory. --- internal/daemon/clients.go | 9 ++++++--- internal/daemon/daemon.go | 10 ++++++++-- internal/ipc/protocol.go | 6 +++--- internal/tui/model.go | 7 +++---- 4 files changed, 20 insertions(+), 12 deletions(-) diff --git a/internal/daemon/clients.go b/internal/daemon/clients.go index 351c601d..fc94d40c 100644 --- a/internal/daemon/clients.go +++ b/internal/daemon/clients.go @@ -463,7 +463,10 @@ func (d *Daemon) shuttingDown() bool { // two state frames back to back: its own attach state and this one, which says // nothing new. That is pressure on its must-deliver queue for no information. // A conn that never attached (an MCP bridge) has no use for either value. -func (d *Daemon) sendStateToOtherClients(except *ipc.Conn) { +// +// cause names the event that called it ("attach", "detach", "lost link"), +// for the log line. +func (d *Daemon) sendStateToOtherClients(except *ipc.Conn, cause string) { d.clients.mu.Lock() var conns []*ipc.Conn for _, rec := range d.clients.sortedRecordsLocked() { @@ -477,7 +480,7 @@ func (d *Daemon) sendStateToOtherClients(except *ipc.Conn) { } msg, err := ipc.NewMessage(ipc.MsgWorkspaceState, d.buildWorkspaceState()) if err != nil { - logger.Error("attach: build state for other clients: %v", err) + logger.Error("%s: build state for other clients: %v", cause, err) return } for _, c := range conns { @@ -489,7 +492,7 @@ func (d *Daemon) sendStateToOtherClients(except *ipc.Conn) { // the attached count, so the other clients always get one state frame. func (d *Daemon) handleDetach(conn *ipc.Conn) { if d.clients.detach(conn).any() { - d.sendStateToOtherClients(conn) + d.sendStateToOtherClients(conn, "detach") } } diff --git a/internal/daemon/daemon.go b/internal/daemon/daemon.go index adc68797..c85eefaf 100644 --- a/internal/daemon/daemon.go +++ b/internal/daemon/daemon.go @@ -651,7 +651,7 @@ func (d *Daemon) onClientDisconnect(conn *ipc.Conn) { // A lost link always changes the attached count, whether or not the // master changed with it, so the other clients get one state frame. if !d.shuttingDown() && d.clients.lose(conn).any() { - d.sendStateToOtherClients(conn) + d.sendStateToOtherClients(conn, "lost link") } // Drop this conn's identity with it: the process it described is gone, // and a retained entry would be listed as running. @@ -1833,7 +1833,7 @@ func (d *Daemon) handleAttach(conn *ipc.Conn, msg *ipc.Message) { // this attach is answered. The attaching conn gets none: its own state // below is built after this registration, so it already carries both. if d.attachClient(conn, attach).any() { - defer d.sendStateToOtherClients(conn) + defer d.sendStateToOtherClients(conn, "attach") } // Hold this conn off live pane output until its replay is sent, BEFORE @@ -5797,6 +5797,12 @@ func resolveSpawnArgs(p *plugin.PanePlugin, pane *Pane, restoring, ownsRecord bo // each, so even three genuinely DIFFERENT unreachable candidates cost this // call no more than spawnDirProbeTimeout in total. // +// The shared deadline has a known cost, accepted: a DEAD earlier candidate +// can spend the whole budget, and a later candidate that is live and distinct +// then gets no time and falls through to step 4, so the pane opens in the +// daemon's own directory. It needs one client's recorded cwd on a dead mount +// first; bounding the total wait on dead mounts is worth more than that case. +// // conn is nil for every restore and recovery caller (recoverEmptyTab, // ensureTabNotEmpty, …), which has no requesting client at all — those start // at step 2. Symlinks are resolved so all callers see the canonical path. diff --git a/internal/ipc/protocol.go b/internal/ipc/protocol.go index 64189e42..e964a81e 100644 --- a/internal/ipc/protocol.go +++ b/internal/ipc/protocol.go @@ -1422,9 +1422,9 @@ type ClientInfo struct { // LastInputAt is when this client last sent pane input, RFC 3339. Empty // means never — a follower that has only watched, not typed. LastInputAt string `json:"last_input_at,omitempty"` - // Role distinguishes a TUI from an MCP bridge sharing the same attach - // path, mirroring ClientHelloPayload.Role. Empty for a client that never - // sent one. + // Role is the role this client declared in its hello, mirroring + // ClientHelloPayload.Role — "tui" in practice, since an MCP bridge never + // attaches and so is never listed. Empty for a client that sent no hello. Role string `json:"role,omitempty"` PID int `json:"pid,omitempty"` Exe string `json:"exe,omitempty"` diff --git a/internal/tui/model.go b/internal/tui/model.go index f1cb3f80..10a220fa 100644 --- a/internal/tui/model.go +++ b/internal/tui/model.go @@ -266,10 +266,9 @@ type paneEventMsg ipc.PaneEventPayload // eventDismissedMsg is the daemon's broadcast that a notification was // dismissed (event_dismissed, spec §8.4), reaching every attached client — -// this client's own dismissal included. dest is carried for parity with the -// wire event; the notification sidebar is not scoped per destination (a -// pre-existing property this task does not change), so the removal applies to -// the one shared list regardless of which daemon reported it. +// this client's own dismissal included. dest is the daemon that sent it: +// DismissByID removes one card by id, and a dismiss-all (empty eventID) +// removes only that daemon's cards from the shared sidebar list. type eventDismissedMsg struct { dest string eventID string // "" = dismiss every card From 39871c7d150cb357080d3353eefa5fc7a9018275 Mon Sep 17 00:00:00 2001 From: Artjoms Stukans Date: Sat, 26 Sep 2026 02:27:05 +0200 Subject: [PATCH 29/40] docs: close the multi-client doc gaps from the branch review - daemon-lifecycle.md: SetTabLayout writes now broadcast through the 50 ms coalescer; document the Reattach flag and the cold-start exit from the restart reserve; record the accepted one-time unpaired VT correction for a follower attaching inside a resize batch. - projects.md: sendAllLayouts is gone; name markLayoutChanged. - tui-rendering.md: a new TUI against an older dev daemon sends only resize_panes, which that daemon drops. - site: the MCP tool count is 36, not 35, in all five places. - features.md and the changelog fragment: note the single-window costs (one coalesced state frame per layout change, and an attach that can wait up to 2 s for a busy output queue to drain). --- .claude/rules/daemon-lifecycle.md | 18 +++++++++++++++++- .claude/rules/projects.md | 2 +- .claude/rules/tui-rendering.md | 4 +++- changelog.d/added-multi-client-sync.md | 1 + docs/features.md | 10 +++++++++- site/src/data/competitors.ts | 2 +- site/src/data/faq.ts | 2 +- site/src/data/features.ts | 2 +- site/src/pages/docs.astro | 2 +- site/src/pages/index.astro | 2 +- 10 files changed, 36 insertions(+), 9 deletions(-) diff --git a/.claude/rules/daemon-lifecycle.md b/.claude/rules/daemon-lifecycle.md index 6c43997b..92f23ab7 100644 --- a/.claude/rules/daemon-lifecycle.md +++ b/.claude/rules/daemon-lifecycle.md @@ -124,6 +124,12 @@ condition, for `min(grace, 30s)` (`restartReserveCap`) — TUIs reattach with their SAME process id within seconds of a restart, so the previous master reclaims its slot (`attach` hands the reservation's `attachedAt` back to the returning record, so it stays the oldest) and nothing resizes. +`AttachPayload.Reattach` separates those reconnects from a COLD start: the +TUI sets it only on the attach its reconnect path sends. After an unclean +stop (reboot, kill) the next TUI is a new process with a new id, and a first +attach (`Reattach` false) from a different id that is the ONLY attached client +clears the restart reserve and is elected at once — otherwise it would be a +follower for 30 s with nobody to protect. Reconnecting clients still wait. **Size authority (`applyResizes`, `internal/daemon/daemon.go`).** `resize_pane` and `resize_panes` share one implementation. A resize applies only from the @@ -155,6 +161,15 @@ the counter (a fresh daemon incarnation renumbers from zero). A failed seq — a rollback is a newer announcement, not an undo, and `Cols`/`Rows` stay at whatever the syscall actually left the PTY at. +**Accepted gap: a follower attaching INSIDE a batch.** `sendPaneSizes` picks +its recipients before the `pty.Resize` calls, and `Cols`/`Rows` are recorded +after them. A follower whose FIRST attach lands between the two misses the +frame, and its attach state can still carry the pre-batch size. Its VT then +takes the new size from the next broadcast (seconds later) with no PTY redraw +to pair it: one unpaired correction, once, on that pane. Accepted as rare +(it needs a master resize and a new follower in the same instant); the cost +is a one-time garble, not a lasting one. + **Output hold (`internal/daemon/outputhold.go`).** A client attaching while panes are writing must receive each pane's history replay and its live output EXACTLY ONCE, in order — without a hold, a live broadcast frame can land in @@ -431,7 +446,8 @@ destroys nothing and `recoverEmptyProject` does not run. **`SetTabLayout`** replaces `handleUpdateLayout`'s unlocked write through the live `*Tab`, which raced `SnapshotState`'s copy (the `handleUpdateTab` shape -#229 fixed) and now also `MovePane`'s template check. Still no broadcast. +#229 fixed) and now also `MovePane`'s template check. An accepted write now +broadcasts, through the 50 ms coalescer (see "Layout revision" above). Events already queued keep the `TabID` they were emitted with; TUI navigation is by pane id, so only MCP `get_notifications` reports a stale tab. diff --git a/.claude/rules/projects.md b/.claude/rules/projects.md index e967c479..4ee830a5 100644 --- a/.claude/rules/projects.md +++ b/.claude/rules/projects.md @@ -238,7 +238,7 @@ unnamed and its root IS the daemon's default, so writing it back is a no-op. **Reachability is checked in the client, because the send cannot report it.** `Router.Send` DROPS a message aimed at a dest it has no conn for, logs, and -returns nil — deliberately, so `resizeAllPanes`/`sendAllLayouts` cannot break +returns nil — deliberately, so `resizeAllPanes` (or `markLayoutChanged` for the changed tab) cannot break mid-iteration. Every `if err := send(…)` in the dialog is therefore blind to the likeliest failure of all. `destReachable` guards the whole `projectFormDest != ""` branch rather than the fold alone, because a host that disconnects between the diff --git a/.claude/rules/tui-rendering.md b/.claude/rules/tui-rendering.md index 2a656c2b..a9c3ae57 100644 --- a/.claude/rules/tui-rendering.md +++ b/.claude/rules/tui-rendering.md @@ -171,7 +171,9 @@ note for why the frame goes out before `pty.Resize` runs. Every pane in a `pane_sizes` frame, and every `PaneInfo` in a workspace-state broadcast, carries `size_seq`; `PaneModel.adoptDaemonSize` adopts only `seq >= daemonSizeSeq`, so a workspace-state broadcast that raced a `pane_sizes` frame -from the same resize cannot undo it. +from the same resize cannot undo it. A new TUI against an OLDER daemon (dev +builds only; release builds are version-gated) sends only `resize_panes`, +which that daemon drops as an unknown type, so its panes are never resized. ### Follower rendering diff --git a/changelog.d/added-multi-client-sync.md b/changelog.d/added-multi-client-sync.md index f366d9a7..7582440b 100644 --- a/changelog.d/added-multi-client-sync.md +++ b/changelog.d/added-multi-client-sync.md @@ -7,3 +7,4 @@ headline: Two Quil windows can now share one daemon's workspace - Switching the shared active tab from one window no longer steals keystrokes out from under someone typing in another — a short guard keeps your next few keys in the pane you were in and shows a flash saying another client switched. - Dismissing a notification, or clearing a pane's unseen mark, updates every attached window's sidebar. - A new MCP tool, `list_clients`, lists every attached window; `set_active_pane` and `close_tui` gain an optional `client` field to target one window instead of whichever typed most recently. +- With a single window attached, two small costs remain: each layout change now costs one workspace-state frame back from the daemon (coalesced over 50 ms), and attaching can wait up to 2 s for a busy live-output queue to drain so each pane's history and live output arrive exactly once. diff --git a/docs/features.md b/docs/features.md index 82cb64ea..3a7a390e 100644 --- a/docs/features.md +++ b/docs/features.md @@ -698,7 +698,9 @@ back; with nobody left to protect, a relaunched window takes over at once instead of waiting. **Take control** (unbound by default — bind `client.take_control` in `bindings.toml`, or run it from the command palette) makes the window you are typing in the master immediately, whatever the -election above would otherwise pick. +election above would otherwise pick. After a daemon restart, the previous +master's window gets its slot back when it reconnects, so nothing resizes; a +freshly started window that is alone on the daemon becomes master at once. **Typing guard.** If another window switches your shared active tab while you are mid-keystroke, your next 250 ms of typing still lands in the pane you were @@ -725,6 +727,12 @@ name one explicitly with the `client` field (see [`list_clients`](mcp.md#tui-coo for the ids to choose from). With no window attached at all, `close_tui` sends nothing rather than erroring. +**Cost with a single window.** Two small costs apply even when only one window +is attached. Each layout change (a split, a close, an arrangement, a border +drag) now costs one workspace-state frame back from the daemon, coalesced over +50 ms. And attaching can wait up to 2 s for a busy live-output queue to drain, +so a pane's history replay and its live output arrive exactly once, in order. + --- ## Pane notes diff --git a/site/src/data/competitors.ts b/site/src/data/competitors.ts index 3a222346..f5e753a8 100644 --- a/site/src/data/competitors.ts +++ b/site/src/data/competitors.ts @@ -493,7 +493,7 @@ export const competitors: Record = { { question: "How do agents control each tool?", answer: - "herdr exposes a Unix-socket API plus a CLI, and ships an installable 'skill' so an agent learns to call it. Quil exposes 35 tools for panes, projects, remote hosts and task delegation over the Model Context Protocol, which Claude Desktop, Cursor, and VS Code speak natively with no glue.", + "herdr exposes a Unix-socket API plus a CLI, and ships an installable 'skill' so an agent learns to call it. Quil exposes 36 tools for panes, projects, remote hosts and task delegation over the Model Context Protocol, which Claude Desktop, Cursor, and VS Code speak natively with no glue.", }, ], }, diff --git a/site/src/data/faq.ts b/site/src/data/faq.ts index 91f7bda8..3c5aa351 100644 --- a/site/src/data/faq.ts +++ b/site/src/data/faq.ts @@ -30,7 +30,7 @@ export const homeFaq: FaqItem[] = [ { question: "Which AI tools does Quil support today?", answer: - "Claude Code has first-class support via the built-in Claude Code pane type, with auto-resume on daemon restart, a setup dialog that pre-fills the active pane's working directory (so the project's `.claude/` context is preserved), and a one-click `Dangerously skip permissions` toggle for unattended runs. Quil also runs an MCP server (`quil mcp`) that exposes 35 tools so any MCP-capable client can read pane output, send keystrokes, snapshot a workspace, query per-pane memory usage, create tabs and AI panes with the dialog's options, manage projects across remote hosts, and delegate work between AI panes. Any other AI tool can be wrapped in a custom TOML plugin that defines its spawn command, resume strategy, and error patterns.", + "Claude Code has first-class support via the built-in Claude Code pane type, with auto-resume on daemon restart, a setup dialog that pre-fills the active pane's working directory (so the project's `.claude/` context is preserved), and a one-click `Dangerously skip permissions` toggle for unattended runs. Quil also runs an MCP server (`quil mcp`) that exposes 36 tools so any MCP-capable client can read pane output, send keystrokes, snapshot a workspace, query per-pane memory usage, create tabs and AI panes with the dialog's options, manage projects across remote hosts, and delegate work between AI panes. Any other AI tool can be wrapped in a custom TOML plugin that defines its spawn command, resume strategy, and error patterns.", }, { question: "Can I paste a screenshot into Claude Code on Windows?", diff --git a/site/src/data/features.ts b/site/src/data/features.ts index abfb794f..aa5b8163 100644 --- a/site/src/data/features.ts +++ b/site/src/data/features.ts @@ -128,7 +128,7 @@ export const features: Feature[] = [ "Run `quil mcp` and an AI agent can manage projects across remote hosts, create AI panes, and delegate work between them.", category: "ai", detail: [ - "35 tools exposed over the Model Context Protocol (Anthropic's open standard for AI tool use).", + "36 tools exposed over the Model Context Protocol (Anthropic's open standard for AI tool use).", "Create tabs and AI panes with named toggles, session resume, worktree and sandbox options; manage projects across configured remote hosts.", "Delegate tasks between panes, track completion, and notify the requester when it is ready. Read output, send keys, inspect screens, watch events, and query memory use.", "Lets any MCP-capable client (Claude Desktop, Claude Code, Cursor) reach directly into your running Quil session.", diff --git a/site/src/pages/docs.astro b/site/src/pages/docs.astro index 75ed4aa7..0b033ee3 100644 --- a/site/src/pages/docs.astro +++ b/site/src/pages/docs.astro @@ -50,7 +50,7 @@ const sections: DocSection[] = [ { name: "Features", path: "docs/features.md", tag: "← TOUR", desc: "Capability tour grouped by area: persistence, remote attach over SSH, the command palette, layout, typed panes, clipboard, observability, pane notes." }, { name: "Keybindings", path: "docs/keybindings.md", tag: "← KEYS", desc: "Full keymap, customization syntax, what to bind and what to leave passing through to the PTY." }, { name: "Configuration", path: "docs/configuration.md", tag: "← CONFIG", desc: "~/.quil/config.toml reference — every section and every key with defaults and effects." }, - { name: "MCP", path: "docs/mcp.md", tag: "← AI", desc: "Let your AI assistant drive Quil. Client wiring for Claude / Cursor / VS Code, all 35 tools, redaction model." }, + { name: "MCP", path: "docs/mcp.md", tag: "← AI", desc: "Let your AI assistant drive Quil. Client wiring for Claude / Cursor / VS Code, all 36 tools, redaction model." }, { name: "Workspace templates", path: "docs/workspace-templates.md", tag: "← SETUP", desc: "Describe a tab once — its panes, their layout, and each one's opening prompt — and create it from the palette. Every pane field, the five layouts, the prompt placeholders, and the limits worth knowing before writing a team prompt." }, ], }, diff --git a/site/src/pages/index.astro b/site/src/pages/index.astro index 9c398350..cfbbf7b9 100644 --- a/site/src/pages/index.astro +++ b/site/src/pages/index.astro @@ -74,7 +74,7 @@ const topFeatures = [ number: "◇ 04", title: "MCP server for agents", tag: "← AGENT LAYER", - body: `Run quil mcp and any MCP-capable client — Claude Desktop, Cursor, Claude Code — can manage projects across remote hosts, read output, create AI panes, and delegate work between them. 35 tools. Your terminal becomes addressable by AI.`, + body: `Run quil mcp and any MCP-capable client — Claude Desktop, Cursor, Claude Code — can manage projects across remote hosts, read output, create AI panes, and delegate work between them. 36 tools. Your terminal becomes addressable by AI.`, }, { number: "◇ 05", From d442bc2a33dee40f5aa534d58afe611b5cc37045 Mon Sep 17 00:00:00 2001 From: Artjoms Stukans Date: Sat, 26 Sep 2026 10:59:15 +0200 Subject: [PATCH 30/40] fix(tui): record a requested-tab token on the attention-queue jump Alt+Shift+A (jumpToNextBlocked) moves activeTab by hand instead of through jumpToPane, because switchProject does work jumpToPane does not. It recorded no requestedTab token, so a broadcast in flight from before the jump named the old tab and read as another client switching back: the tab jumped back, the typing guard armed and "Tab switched by another client" flashed. Record the token exactly as jumpToPane does, keyed on the project's own previous tab. Correct the two comments that claimed the attention queue already routed through jumpToPane. --- internal/tui/attention.go | 15 +++++++++++ internal/tui/typing_guard_test.go | 42 +++++++++++++++++++++++++++---- internal/tui/workstate.go | 6 +++-- 3 files changed, 56 insertions(+), 7 deletions(-) diff --git a/internal/tui/attention.go b/internal/tui/attention.go index d4ab5148..91bd9407 100644 --- a/internal/tui/attention.go +++ b/internal/tui/attention.go @@ -85,12 +85,27 @@ func (m *Model) jumpToNextBlocked() tea.Cmd { m.exitNotesModeInPlace() } cmd := m.switchProject(i) + // The project's OWN previous tab, for the typing guard's token + // below — jumpToPane's prevID, for the same reason. + prevID := "" + if prev := target.Project.activeTab; prev >= 0 && prev < len(target.Project.tabs) { + prevID = target.Project.tabs[prev].ID + } target.Project.activeTab = target.TabIndex // TabModel.ActivePane is the pane-ID field; ActivePaneModel() // repairs a stale value on next read, so assigning it is the // whole focus change. targetTab := target.Project.tabs[target.TabIndex] targetTab.ActivePane = target.Pane.ID + // Typing guard (spec §8.1): recorded exactly as jumpToPane records + // it, because this function moves activeTab by hand rather than + // routing through it (switchProject does work jumpToPane does not). + // Without the token a broadcast in flight from before the jump + // names the old tab and reads as another client switching back: + // the tab jumps back, the guard arms and the flash shows. + if prevID != targetTab.ID { + m.recordRequestedTab(targetTab.Dest, target.Project.ID, targetTab.ID, prevID) + } // MsgSwitchProject only reaches the project's REMEMBERED tab; the // blocked pane is routinely in a different one, whose panes are // still Pending after a lazy restore. Without this the queue lands diff --git a/internal/tui/typing_guard_test.go b/internal/tui/typing_guard_test.go index 077e0336..1d56486f 100644 --- a/internal/tui/typing_guard_test.go +++ b/internal/tui/typing_guard_test.go @@ -664,11 +664,12 @@ func TestTypingGuard_StaleFromTabBroadcastIsRejectedWhilePending(t *testing.T) { } // TestTypingGuard_JumpRecordsARequestedTab: jumpToPane (MCP set_active_pane, -// sidebar clicks, Alt+Backspace, the palette, attention jumps) moves this -// client's active tab exactly as switchTab does, so it must record the same -// token. Without one, a broadcast still in flight from before the jump names -// the old tab and reads as another client switching back: the tab jumps -// back, the guard arms and the flash shows. +// sidebar clicks, Alt+Backspace, the palette) moves this client's active tab +// exactly as switchTab does, so it must record the same token. Without one, a +// broadcast still in flight from before the jump names the old tab and reads +// as another client switching back: the tab jumps back, the guard arms and +// the flash shows. The attention queue does not route through jumpToPane — +// TestTypingGuard_AttentionJumpRecordsARequestedTab covers it. func TestTypingGuard_JumpRecordsARequestedTab(t *testing.T) { t.Parallel() t0 := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) @@ -701,6 +702,37 @@ func TestTypingGuard_JumpRecordsARequestedTab(t *testing.T) { } } +// TestTypingGuard_AttentionJumpRecordsARequestedTab: the attention queue +// (Alt+Shift+A, jumpToNextBlocked) moves activeTab by hand instead of through +// jumpToPane, so it records its own token. Without it a broadcast in flight +// from before the jump names the old tab: the tab jumps back, the guard arms +// and "Tab switched by another client" flashes. +func TestTypingGuard_AttentionJumpRecordsARequestedTab(t *testing.T) { + t.Parallel() + t0 := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + m, _ := typingGuardModel(t, t0, "t1", "t2") + m.projects[0].tabs[1].Root.FindLeaf("p2").Pane.blockedSince = t0.Add(-time.Minute) + + updated, _ := m.Update(tea.KeyPressMsg{Mod: tea.ModAlt | tea.ModShift, Code: 'a'}) + got := updated.(Model) + if got.activeTabModel().ID != "t2" { + t.Fatalf("setup: active tab = %q after Alt+Shift+A, want t2 (the blocked pane's tab)", got.activeTabModel().ID) + } + + got.now = func() time.Time { return t0.Add(100 * time.Millisecond) } + updated, _ = got.Update(typingGuardBroadcast("t1", "t1", "t2")) + got2 := updated.(Model) + if got2.activeTabModel().ID != "t2" { + t.Errorf("active tab = %q after a stale t1 broadcast, want t2 (no jump back)", got2.activeTabModel().ID) + } + if got2.guardPaneID != "" { + t.Errorf("guardPaneID = %q, want empty — the attention jump was this client's own", got2.guardPaneID) + } + if got2.flashText != "" { + t.Errorf("flashText = %q, want empty", got2.flashText) + } +} + // TestTypingGuard_StaleRejectAdoptsTheDaemonTabWhenTheTargetIsGone: a // broadcast naming the tab this client just left is normally a stale echo and // is rejected, holding the tab the client is on. When another client diff --git a/internal/tui/workstate.go b/internal/tui/workstate.go index a8e8a6a2..774b0850 100644 --- a/internal/tui/workstate.go +++ b/internal/tui/workstate.go @@ -166,8 +166,10 @@ func (m *Model) jumpToPane(paneID string) (bool, tea.Cmd) { // active when it runs, so calling it afterwards reverts the wrong one. // // This is the choke point for every cross-tab jump that is not switchTab — - // MCP set_active_pane, the notification sidebar, pane-history back, the - // palette and the attention queue all arrive here. They each moved the + // MCP set_active_pane, the notification sidebar, pane-history back and the + // palette all arrive here. The attention queue (jumpToNextBlocked) does + // NOT: it moves activeTab by hand, so it repeats this teardown and the + // typing-guard token itself. They each moved the // active tab with the editor still open, bound to a pane in the tab being // left, still claiming its share of the width. notesKeyExempt does not // cover it: that branch only runs while the EDITOR has focus, and notes From 9ac06fac14a33601292a086f8f2801d7b53d4320 Mon Sep 17 00:00:00 2001 From: Artjoms Stukans Date: Sat, 26 Sep 2026 10:59:15 +0200 Subject: [PATCH 31/40] fix(daemon): only an attach with a client id clears the restart reserve A TUI older than multi-client sync sends neither ClientID nor Reattach, so its reconnect after a daemon restart read exactly like a cold start from a new process. Alone on the daemon, it cleared the restart reserve and lost the previous master's slot. The reserve now yields only when the attach carried its own id; an id-less attach waits out the reserve like any reconnecting client. --- internal/daemon/clients.go | 14 ++++++++++++-- internal/daemon/clients_test.go | 18 ++++++++++++++++++ 2 files changed, 30 insertions(+), 2 deletions(-) diff --git a/internal/daemon/clients.go b/internal/daemon/clients.go index fc94d40c..553770a8 100644 --- a/internal/daemon/clients.go +++ b/internal/daemon/clients.go @@ -241,7 +241,8 @@ func (r *clientRegistry) expire() { // conn keeps it. // // reattach is the payload's Reattach flag: false on a process's first attach -// to this daemon, which a restart reserve yields to when it is alone. +// to this daemon, which a restart reserve yields to when it is alone — but +// only when the attach carried its own id (an older client sends neither). func (r *clientRegistry) attach(conn *ipc.Conn, id string, cols, rows int, cwd string, reattach bool) clientChange { r.mu.Lock() defer r.mu.Unlock() @@ -249,6 +250,9 @@ func (r *clientRegistry) attach(conn *ipc.Conn, id string, cols, rows int, cwd s r.byConn = make(map[*ipc.Conn]*clientRecord) } before := len(r.byConn) + // Taken before an empty id is minted below: only a client that sent its + // own id can also have sent a meaningful Reattach flag. + sentID := id != "" rec, existed := r.byConn[conn] if id == "" { if existed { @@ -283,11 +287,17 @@ func (r *clientRegistry) attach(conn *ipc.Conn, id string, cols, rows int, cwd s // The reserved client is back, so the slot has nothing left to wait // for. The election below keeps it when the client is eligible. r.clearReservationLocked() - } else if res != nil && res.protects == nil && !reattach && len(r.byConn) == 1 { + } else if res != nil && res.protects == nil && sentID && !reattach && len(r.byConn) == 1 { // A restart reserve waits for TUIs RECONNECTING after the restart. A // new process attaching alone is a cold start after an unclean stop // (reboot, kill): the previous master went with its process, and // waiting would make the only client a follower for the reserve. + // + // sentID: a client older than the multi-client branch sends neither a + // ClientID nor Reattach, so its RECONNECT after the restart reads + // exactly like a cold start. Absence of the flag says nothing there, + // so such an attach never clears the reserve; it waits it out like + // any reconnecting client. r.clearReservationLocked() } return clientChange{master: r.electLocked(), count: len(r.byConn) != before} diff --git a/internal/daemon/clients_test.go b/internal/daemon/clients_test.go index 73bfee33..57930924 100644 --- a/internal/daemon/clients_test.go +++ b/internal/daemon/clients_test.go @@ -387,6 +387,24 @@ func TestRestartReserve_PreviousMasterReclaims(t *testing.T) { h.wantMaster("B") }) + // A TUI older than this branch sends no ClientID and no Reattach, so its + // reconnect looks like a cold start. Missing fields say nothing: it must + // not clear the reserve, even alone. + t.Run("attach with no client id alone keeps the reserve", func(t *testing.T) { + h := restartedWithMaster(t) + c := new(ipc.Conn) + if changed := h.d.registerClient(c, ipc.AttachPayload{Cols: 100, Rows: 30}); changed { + t.Error("an attach with no ClientID must not change the master during the restart reserve") + } + if h.d.isMasterConn(c) { + t.Error("an attach with no ClientID must not take the reserved slot") + } + if h.armed() == nil { + t.Error("the restart timer must still be armed") + } + h.wantMaster("A") + }) + t.Run("fresh first attach beside a reattached client keeps the reserve", func(t *testing.T) { h := restartedWithMaster(t) h.reattach("B", 100, 30) From 6b102037c46e8ee56ad7a796910e688a4893bb9b Mon Sep 17 00:00:00 2001 From: Artjoms Stukans Date: Sat, 26 Sep 2026 10:59:16 +0200 Subject: [PATCH 32/40] fix(tui): set reattach only when this process attached there before A destination unreachable at launch gets its FIRST attach through finishReconnect, which always sent Reattach=true. After that host's unclean restart, the lone TUI then waited out the whole restart reserve as a follower. The Model now keeps attachedOnce, a per-destination set written on the Update goroutine by both attach paths (attachAllDests, which adoptDest also uses, and finishReconnect) and never cleared. Reattach is attachedOnce[dest], read before the mark. The AttachPayload comment now states the rule and the ClientID requirement the daemon applies. --- internal/ipc/protocol.go | 16 +++++---- internal/tui/model.go | 33 +++++++++++++++--- internal/tui/multiclient_role_test.go | 49 ++++++++++++++++++++++++++- internal/tui/reconnect.go | 9 +++-- 4 files changed, 93 insertions(+), 14 deletions(-) diff --git a/internal/ipc/protocol.go b/internal/ipc/protocol.go index e964a81e..fe0327db 100644 --- a/internal/ipc/protocol.go +++ b/internal/ipc/protocol.go @@ -327,12 +327,16 @@ type AttachPayload struct { // that cannot participate in election — it is neither offered control // nor handed a follower's resize_panes stream. ClientID string `json:"client_id,omitempty"` - // Reattach is true on an attach sent from the client's reconnect path, - // and false on the process's first attach to this daemon. A daemon - // restart keeps the previous master's slot for a short reserve so the - // reconnecting TUIs resize nothing; a FIRST attach from a new process, - // alone on the daemon, is a cold start after an unclean stop, and the - // reserve yields to it rather than making it a follower for 30 s. + // Reattach is false on the process's first attach to this daemon and + // true on every later one, whichever client path sends it (a daemon + // unreachable at launch gets its first attach from the reconnect path). + // A daemon restart keeps the previous master's slot for a short reserve + // so the reconnecting TUIs resize nothing; a FIRST attach from a new + // process, alone on the daemon, is a cold start after an unclean stop, + // and the reserve yields to it rather than making it a follower for + // 30 s. Only an attach that carries a ClientID can clear the reserve: an + // older client sends neither field, so its reconnect looks like a cold + // start. Reattach bool `json:"reattach,omitempty"` } diff --git a/internal/tui/model.go b/internal/tui/model.go index 10a220fa..3ab7f8b8 100644 --- a/internal/tui/model.go +++ b/internal/tui/model.go @@ -488,6 +488,13 @@ type Model struct { version string sized bool // the terminal has reported its geometry at least once attached map[string]bool // destinations already attached — see attachAllDests + // attachedOnce records every destination this PROCESS has sent an attach + // to, and is never cleared (attached is, on disconnect). It is the + // AttachPayload.Reattach flag: a destination unreachable at launch gets its + // first attach through finishReconnect, and that one must still say + // "first", or the lone TUI after that host's unclean restart waits out the + // restart reserve as a follower. + attachedOnce map[string]bool // offlineWoken records which offline destinations have had their ladder // started, so the wake-up fires once rather than on every resize. offlineWoken map[string]bool @@ -7942,11 +7949,12 @@ func (m *Model) attachAllDests() tea.Cmd { if m.attached[dest] { continue } - if err := m.sendForDest(dest, m.attachMessage(dest, false)); err != nil { + if err := m.sendForDest(dest, m.attachMessage(dest, m.attachedOnce[dest])); err != nil { log.Printf("attach to %q failed, retrying on the next resize: %v", dest, err) continue } m.attached[dest] = true + m.markAttachedOnce(dest) newlyAttached[dest] = true // Batched per destination so each daemon is asked about its OWN // registry; see requestPluginListFor. @@ -8006,9 +8014,10 @@ func (m *Model) attachAllDests() tea.Cmd { // attachMessage builds the MsgAttach describing this client's geometry for one // destination. // -// reattach is true only from the reconnect path (attachToDest). A daemon that -// just restarted keeps the previous master's slot for its reconnecting TUIs, -// and yields it to a first attach from a new process that is alone there. +// reattach is m.attachedOnce[dest] at the call site: true once this process +// has attached to dest before, whichever path sends it. A daemon that just +// restarted keeps the previous master's slot for its reconnecting TUIs, and +// yields it to a first attach from a new process that is alone there. func (m Model) attachMessage(dest string, reattach bool) *ipc.Message { // Subtract chrome (tab bar + status bar), the project sidebar (if open), // then pane border (2) — the same reservation resizeTabs applies, so the @@ -8066,9 +8075,14 @@ func (m Model) attachMessage(dest string, reattach bool) *ipc.Message { // truth here the sweep destroys an overlay the user is looking at five minutes // after a link blip. attachAllDests never runs for this flow — finishReconnect // sets m.attached[dest] — so its copy of this report does not cover it. +// +// The Reattach flag is read HERE, on the Update goroutine, not inside the +// command: the caller marks dest in attachedOnce right after this returns, and +// the command runs later on its own goroutine. func (m Model) attachToDest(dest string) tea.Cmd { + reattach := m.attachedOnce[dest] attachCmd := func() tea.Msg { - m.sendForDest(dest, m.attachMessage(dest, true)) + m.sendForDest(dest, m.attachMessage(dest, reattach)) return nil } return tea.Batch(attachCmd, m.requestPluginListFor(dest), m.overlayTruthDestCmd(dest), @@ -8078,6 +8092,15 @@ func (m Model) attachToDest(dest string) tea.Cmd { m.requestSandboxCap(dest)) } +// markAttachedOnce records that this process has sent dest an attach. Called +// on the Update goroutine by both attach paths — see attachedOnce. +func (m *Model) markAttachedOnce(dest string) { + if m.attachedOnce == nil { + m.attachedOnce = map[string]bool{} + } + m.attachedOnce[dest] = true +} + // listenContinueMsg signals the TUI to keep listening for daemon messages. type listenContinueMsg struct{} diff --git a/internal/tui/multiclient_role_test.go b/internal/tui/multiclient_role_test.go index 1ca27523..d2567f9b 100644 --- a/internal/tui/multiclient_role_test.go +++ b/internal/tui/multiclient_role_test.go @@ -486,7 +486,7 @@ func attachPayloads(t *testing.T, conn *fakeConn) []ipc.AttachPayload { // TestAttach_ReattachFlagMarksOnlyTheReconnectPath: the daemon's restart // reserve yields to a FIRST attach from a new process and keeps waiting for a // reconnecting one, so the flag must be false on the attach a WindowSizeMsg -// sends and true on the one a completed redial sends. +// sends and true on the one a completed redial sends after it. func TestAttach_ReattachFlagMarksOnlyTheReconnectPath(t *testing.T) { t.Setenv("QUIL_HOME", t.TempDir()) first, fresh := newFakeConn(), newFakeConn() @@ -516,6 +516,53 @@ func TestAttach_ReattachFlagMarksOnlyTheReconnectPath(t *testing.T) { } } +// TestAttach_FirstAttachThroughReconnectIsNotAReattach: a destination +// unreachable at launch gets its FIRST attach from finishReconnect. The flag +// is "has this process attached there before", not "which path sent it", so +// that attach must say first — else the lone TUI after that host's unclean +// restart is a follower for the whole restart reserve. Each destination keeps +// its own answer. +func TestAttach_FirstAttachThroughReconnectIsNotAReattach(t *testing.T) { + t.Setenv("QUIL_HOME", t.TempDir()) + up := newFakeConn() + r := NewRouter(map[string]Client{"gpu01": up}) // gpu02 was unreachable at launch + var m tea.Model = Model{client: r, cfg: config.Default()} + + m, cmd := m.Update(tea.WindowSizeMsg{Width: 120, Height: 40}) + runCmdNoWait(cmd) + if got := attachPayloads(t, up); len(got) != 1 || got[0].Reattach { + t.Fatalf("gpu01 launch attach = %+v, want one with reattach=false", got) + } + + redial := func(m tea.Model, dest string, c *fakeConn) tea.Model { + t.Helper() + mm := m.(Model) + mm.links = map[string]*reconnectState{dest: {active: true, gen: 1}} + m, cmd := mm.Update(redialResultMsg{gen: 1, dest: dest, client: c}) + runCmdNoWait(cmd) + return m + } + + first := newFakeConn() + m = redial(m, "gpu02", first) + if got := attachPayloads(t, first); len(got) != 1 || got[0].Reattach { + t.Errorf("gpu02 first attach (through reconnect) = %+v, want one with reattach=false", got) + } + + second := newFakeConn() + m = redial(m, "gpu02", second) + if got := attachPayloads(t, second); len(got) != 1 || !got[0].Reattach { + t.Errorf("gpu02 second attach = %+v, want one with reattach=true", got) + } + + // gpu02's history must not leak into a destination attached only once. + third := newFakeConn() + redial(m, "gpu03", third) + if got := attachPayloads(t, third); len(got) != 1 || got[0].Reattach { + t.Errorf("gpu03 first attach = %+v, want one with reattach=false", got) + } +} + // TestClientGeometryCmd_ReportsZeroWhenUnpaintable mirrors // TestAttachMessage_ReportsNoGeometryBelowTheMinimum: a terminal too small to // paint must report 0x0, never the raw sub-floor size. The daemon uses this diff --git a/internal/tui/reconnect.go b/internal/tui/reconnect.go index b536a162..9566eac9 100644 --- a/internal/tui/reconnect.go +++ b/internal/tui/reconnect.go @@ -1195,8 +1195,13 @@ func (m Model) finishReconnect(dest string, c Client) (tea.Model, tea.Cmd) { } }) + // attachToDest reads attachedOnce for the Reattach flag, so the mark comes + // after it: a destination unreachable at launch attaches here for the + // first time, and must say so. + attach := m.attachToDest(dest) + m.markAttachedOnce(dest) if isRouter { - return m, m.attachToDest(dest) + return m, attach } - return m, tea.Batch(m.attachToDest(dest), m.listenForMessages()) + return m, tea.Batch(attach, m.listenForMessages()) } From 48d789a9a7f5a23244eeb3eee4789d1a763b512a Mon Sep 17 00:00:00 2001 From: Artjoms Stukans Date: Sat, 26 Sep 2026 10:59:16 +0200 Subject: [PATCH 33/40] docs: count 36 MCP tools in the MCP server roadmap page The page still said 35 in its summary, its section heading and its acceptance list, and its TUI cooperation table did not list list_clients. Add the row and describe set_active_pane and close_tui as acting on one attached window. --- docs/roadmap/mcp-server.md | 13 +++++++------ 1 file changed, 7 insertions(+), 6 deletions(-) diff --git a/docs/roadmap/mcp-server.md b/docs/roadmap/mcp-server.md index ea750e52..3e70bd5c 100644 --- a/docs/roadmap/mcp-server.md +++ b/docs/roadmap/mcp-server.md @@ -17,7 +17,7 @@ make this harder: the relevant project may be on another machine. ## Implemented solution -`quil mcp` exposes 35 Model Context Protocol tools over stdio. An MCP-capable +`quil mcp` exposes 36 Model Context Protocol tools over stdio. An MCP-capable client can discover the workspace, create tabs and AI panes using the same validated options as the TUI, manage projects across configured remote hosts, and delegate work between panes with completion tracking and notify-back. @@ -25,7 +25,7 @@ and delegate work between panes with completion tracking and notify-back. This PRD records the capability and its constraints. The [MCP guide](../mcp.md) is the reference for client configuration, input schemas, responses and examples. -## Tools by purpose (35 total) +## Tools by purpose (36 total) ### Discovery (6) @@ -97,12 +97,13 @@ task finishes on shell command completion. Notify-back waits until the requester can receive input and is limited to panes on the same daemon. Tasks are bounded and runtime-only. See [task delegation](../mcp.md#delegating-work-to-another-pane). -### TUI cooperation (2) +### TUI cooperation (3) | Tool | Purpose | |------|---------| -| `set_active_pane` | Focus a pane, including across tabs | -| `close_tui` | Close the frontend while the daemon and panes remain alive | +| `set_active_pane` | Focus a pane, including across tabs, in one attached window | +| `close_tui` | Close one attached window while the daemon and panes remain alive | +| `list_clients` | List attached windows, which one sets pane sizes, and their ids (daemon 1.80.0+) | ### Event observation (3) @@ -153,7 +154,7 @@ old terminal state and discard late output from the replaced process. ## Acceptance -- MCP clients can connect through stdio and discover all 35 registered tools. +- MCP clients can connect through stdio and discover all 36 registered tools. - Agents can manage projects, tabs and panes on local and configured remote daemons. - Unscoped discovery recovers hosts after the retry backoff without a named call. - Pane creation honors TUI-equivalent options and reports validation or spawn errors. From 9d1aa9cf873ca72a8713f2b4556a08a9d2ae61d0 Mon Sep 17 00:00:00 2001 From: Artjoms Stukans Date: Sat, 26 Sep 2026 10:59:32 +0200 Subject: [PATCH 34/40] docs: state the attachedOnce and client-id rules for the restart reserve daemon-lifecycle.md said the TUI sets Reattach only on its reconnect path. It now sets it once the process has attached to that destination before, whichever path sends it, and the daemon lets only an attach that carries a ClientID clear the restart reserve. --- .claude/rules/daemon-lifecycle.md | 9 +++++++-- 1 file changed, 7 insertions(+), 2 deletions(-) diff --git a/.claude/rules/daemon-lifecycle.md b/.claude/rules/daemon-lifecycle.md index 92f23ab7..fdd02d63 100644 --- a/.claude/rules/daemon-lifecycle.md +++ b/.claude/rules/daemon-lifecycle.md @@ -125,11 +125,16 @@ their SAME process id within seconds of a restart, so the previous master reclaims its slot (`attach` hands the reservation's `attachedAt` back to the returning record, so it stays the oldest) and nothing resizes. `AttachPayload.Reattach` separates those reconnects from a COLD start: the -TUI sets it only on the attach its reconnect path sends. After an unclean +TUI sets it once THIS PROCESS has attached to that destination before +(`Model.attachedOnce`, written by both attach paths and never cleared) — not +by which path sends it, because a destination unreachable at launch gets its +first attach from `finishReconnect`. After an unclean stop (reboot, kill) the next TUI is a new process with a new id, and a first attach (`Reattach` false) from a different id that is the ONLY attached client clears the restart reserve and is elected at once — otherwise it would be a -follower for 30 s with nobody to protect. Reconnecting clients still wait. +follower for 30 s with nobody to protect. Reconnecting clients still wait, and +so does an attach with NO `ClientID`: a TUI older than this feature sends +neither field, so its reconnect is indistinguishable from a cold start. **Size authority (`applyResizes`, `internal/daemon/daemon.go`).** `resize_pane` and `resize_panes` share one implementation. A resize applies only from the From fe08807066a6777d9b4b92ede4f1ca2797f8f14a Mon Sep 17 00:00:00 2001 From: Artjoms Stukans Date: Sat, 26 Sep 2026 11:06:05 +0200 Subject: [PATCH 35/40] chore(dev): stop the dev daemon and TUIs before a build A dev daemon or dev TUI left running made `dev.sh build` refuse, and each refusal cost a round trip. build and clean now stop the dev variant from this directory first, then run refuse_if_binaries_held unchanged, which keeps the last word. Scope is exactly quil-dev and quild-dev in the project directory, matched by full executable path (Win32_Process via powershell.exe on Windows, /proc//exe on Linux, lsof on macOS), never by name, as every worktree has its own quil-dev. The dev daemon stops first and gracefully through `quil-dev daemon stop` with QUIL_HOME set to the project .quil, and only when .quil/quild.pid names a live dev daemon from this directory. What remains is stopped by pid alone, never as a tree. Production quil/quild and quil-debug/quild-debug are never stopped, and ~/.quil is never read. A fast path skips the process listing when both dev files are free, so an ordinary build pays nothing. After a stop, the build waits up to 5 s for Windows to release the image files. --- .claude/CLAUDE.md | 2 + .claude/rules/dev-environment.md | 2 +- scripts/dev.sh | 171 ++++++++++++++++++++++++++++++- 3 files changed, 171 insertions(+), 4 deletions(-) diff --git a/.claude/CLAUDE.md b/.claude/CLAUDE.md index f8d55b5e..5578c287 100644 --- a/.claude/CLAUDE.md +++ b/.claude/CLAUDE.md @@ -98,6 +98,8 @@ Go module cache is persisted in a Docker volume (`quil-gomod`) for fast repeated `build` and `clean` refuse to run while any binary they would write is held by a running process (`refuse_if_binaries_held` in `scripts/dev.sh`). Neither platform can overwrite a running executable — Windows fails the open with a sharing violation, Linux returns ETXTBSY — and the ORDER is what makes it dangerous rather than the failure: the six builds are chained with `&&`, so a holder on the Nth leaves the first N-1 freshly built and the rest stale, and a new TUI beside a stale daemon fails the version gate at launch, reading as a bug in whatever you were working on. Detection is a **non-destructive probe, not a process query**: `(exec 3>>"$target")` asks the OS the exact question the build is about to ask, and append never truncates. The first version instead read `.quil/quild.pid` and looked for a daemon — it missed a dev TUI holding `quil-dev.exe` and let the half-build through on its first real use, which is why this checks the condition rather than a proxy for it. Only files in the project directory are probed, so a production install elsewhere is never touched, and a stale or malformed pid file cannot block a build because no pid file is consulted. There is deliberately no override flag: "build anyway" produces exactly the mismatched set it exists to prevent. +**Before that probe, `build` and `clean` stop the DEV variant themselves** (`stop_dev_processes`), because a dev daemon or dev TUI left running was the usual reason for the refusal, and each refusal cost a round trip. Scope is exactly `quil-dev` and `quild-dev` in THIS project directory, matched by FULL executable path (`Win32_Process.ExecutablePath` via `powershell.exe` on Windows, `/proc//exe` on Linux, `lsof` on macOS) — never by name, since every worktree has its own `quil-dev`. The dev daemon goes first and gracefully, through `quil-dev daemon stop` with `QUIL_HOME` set to the project `.quil/`, and only when `.quil/quild.pid` names a live process running this directory's `quild-dev`; whatever remains is stopped by PID alone, never as a tree (`taskkill /T` would reach pane children and lingering wrapper parents). `quil`/`quild` (production) and `quil-debug`/`quild-debug` (which serve the production `~/.quil`) are NEVER stopped, and `~/.quil` is never read. The probe then runs unchanged after a ≤5 s wait for Windows to release the image files, so it still has the last word: a production or debug binary here held by a process refuses the build exactly as before. + ### Windows Icon `build` and `cross` embed the Quil brand mark as a Windows executable icon via `go-winres` (v0.3.3). Build assets live in `winres/` (icon PNGs + `winres.json` manifest with `RT_GROUP_ICON` + `RT_VERSION`). The build script installs `go-winres` inside the Docker container and generates `.syso` files in `cmd/quil/` and `cmd/quild/` before `go build`. The Go linker picks up `.syso` files automatically (Windows only — ignored on Linux/Darwin). Generated `.syso` files are gitignored. diff --git a/.claude/rules/dev-environment.md b/.claude/rules/dev-environment.md index 55acf7b0..218850a2 100644 --- a/.claude/rules/dev-environment.md +++ b/.claude/rules/dev-environment.md @@ -38,7 +38,7 @@ Every code change in this repo follows the same loop: 1. Edit code (`internal/…`, `cmd/…`). 2. Rebuild: `./scripts/dev.sh build`. -3. If a dev daemon from a previous iteration is running, stop it by PID from `./.quil/quild.pid` (or manually via Task Manager / `kill`). +3. Nothing to do by hand: `./scripts/dev.sh build` (and `clean`) now stops this directory's dev daemon (gracefully, via `quil-dev daemon stop`) and any dev TUI first, matching `quil-dev`/`quild-dev` by full executable path. It never stops `quil`/`quild` or `quil-debug`/`quild-debug`, and a dev TUI you had open is closed by design. If the build still refuses, something else holds a binary here — stop a stray dev daemon by PID from `./.quil/quild.pid` only, never from `~/.quil/quild.pid`. 4. Launch: `./scripts/quil-dev.ps1` (Windows) or `./scripts/quil-dev.sh` (Unix). 5. Verify `[dev]` is visible in the status bar before testing. 6. Test the change. When done, close the dev TUI — do NOT run any `kill-daemon` / `reset-daemon` helper scripts. diff --git a/scripts/dev.sh b/scripts/dev.sh index f06f86df..22ff2407 100644 --- a/scripts/dev.sh +++ b/scripts/dev.sh @@ -141,9 +141,9 @@ refuse_if_binaries_held() { leaving a mismatched set — typically a new TUI against a stale daemon, which then fails the version gate at launch. - Close any Quil started from this directory. If a dev daemon is running: - - QUIL_HOME="$PROJECT_DIR/.quil" "$PROJECT_DIR/quil-dev$EXE" daemon stop + The dev daemon and dev TUIs from this directory were already stopped. + Close whatever else is running these files: a quil or quil-debug started + from this directory, or a scanner holding them. Only files in $PROJECT_DIR were checked. A production install elsewhere is untouched. @@ -153,11 +153,175 @@ EOF } } +# stop_dev_processes stops the DEV variant started from this directory, so a +# dev daemon or dev TUI left running no longer turns `build` into a refusal +# and an extra round trip. refuse_if_binaries_held still runs afterwards and +# still has the last word. +# +# Scope is exactly two files: quil-dev and quild-dev in $PROJECT_DIR. A +# process is matched by its FULL executable path, never by name — every +# worktree has its own quil-dev, and the name alone would stop another +# checkout's session. quil/quild (production) and quil-debug/quild-debug +# (which serve the production ~/.quil) are never stopped, and ~/.quil is +# never read. +# +# Order: the dev daemon first, gracefully, through the dev binary itself — +# `daemon stop` writes the final snapshot and closes the panes' PTYs, which a +# kill does not. Only when .quil/quild.pid names a live process running this +# directory's quild-dev. Then every remaining matching process is stopped by +# PID alone, never as a tree (taskkill /T), because a tree kill reaches the +# panes' children and wrapper parents that are not ours to end. +stop_dev_processes() { + case "$(host_goos)" in windows) host_exe=".exe" ;; *) host_exe="" ;; esac + dev_tui="$PROJECT_DIR/quil-dev$host_exe" + dev_daemon="$PROJECT_DIR/quild-dev$host_exe" + # Fast path: a file the probe can open for writing has no process running + # it — the assumption refuse_if_binaries_held already rests on — so an + # ordinary build never pays for a process listing (seconds of PowerShell + # startup on Windows). + dev_binaries_held || return 0 + + running="$(dev_processes list)" + [ -n "$running" ] || return 0 + + pid_file="$PROJECT_DIR/.quil/quild.pid" + if [ -f "$pid_file" ] && [ -f "$dev_tui" ]; then + daemon_pid="$(tr -dc '0-9' < "$pid_file")" + if [ -n "$daemon_pid" ] && printf '%s\n' "$running" | grep -q "^$daemon_pid "; then + # QUIL_HOME is explicit rather than left to the dev build's own default, + # so the stop cannot reach any daemon but the one this pid file names. + if out="$(QUIL_HOME="$PROJECT_DIR/.quil" "$dev_tui" daemon stop 2>&1)"; then + echo "stopped dev daemon (pid $daemon_pid, graceful): $dev_daemon" >&2 + else + printf 'dev daemon did not stop gracefully, stopping it by pid:\n%s\n' "$out" >&2 + fi + fi + fi + + stopped="$(dev_processes stop)" + if [ -n "$stopped" ]; then + printf '%s\n' "$stopped" | while IFS= read -r line; do + echo "stopped dev process (pid ${line%% *}): ${line#* }" >&2 + done + fi + + # Windows releases an image file a moment AFTER its process exits. Poll the + # same probe refuse_if_binaries_held uses, for at most 5 s; anything still + # held after that is left for it to report. Reached only when something was + # running, so an ordinary build pays nothing here. + tries=0 + while dev_binaries_held && [ "$tries" -lt 10 ]; do + sleep 0.5 + tries=$((tries + 1)) + done +} + +# dev_binaries_held succeeds when $dev_tui or $dev_daemon cannot be opened for +# append — refuse_if_binaries_held's probe, over the two dev files only. +dev_binaries_held() { + for f in "$dev_tui" "$dev_daemon"; do + [ -f "$f" ] || continue + (exec 3>>"$f") 2>/dev/null || return 0 + done + return 1 +} + +# dev_processes list prints " " for every process whose executable +# is $dev_tui or $dev_daemon; dev_processes stop stops each of them and prints +# the ones it stopped, in the same format. +dev_processes() { + mode="$1" + if [ "$(host_goos)" = "windows" ]; then + dev_processes_windows "$mode" + else + dev_processes_unix "$mode" + fi +} + +# Win32_Process.ExecutablePath is the image path the loader opened, so the +# comparison is exact; case and slashes are normalised because Windows paths +# are case-insensitive and PROJECT_DIR comes from `pwd -W` with forward +# slashes. Paths travel in the environment rather than in the command text, so +# no quoting of the project path can break the script. A process whose path +# cannot be read (another user's, an elevated one) is skipped: it is not +# stopped, and the probe then reports its file as held. +dev_processes_windows() { + ps_exe="$(command -v powershell.exe 2>/dev/null || true)" + if [ -z "$ps_exe" ]; then + # shellcheck disable=SC1003 # a literal backslash, not an escaped quote + ps_exe="$(printf '%s' "${SYSTEMROOT:-C:/Windows}" | tr '\\' '/')/System32/WindowsPowerShell/v1.0/powershell.exe" + fi + [ -f "$ps_exe" ] || { echo "powershell.exe not found; dev processes not stopped" >&2; return 0; } + # shellcheck disable=SC2016 # PowerShell's $ variables, not the shell's + QUIL_DEV_PATHS="$dev_tui|$dev_daemon" QUIL_DEV_MODE="$1" MSYS_NO_PATHCONV=1 \ + "$ps_exe" -NoProfile -NonInteractive -Command ' + $want = $env:QUIL_DEV_PATHS.Split("|") | ForEach-Object { $_.Replace("/", "\").ToLowerInvariant() } + $names = ($want | ForEach-Object { "Name=`"" + (Split-Path $_ -Leaf) + "`"" }) -join " OR " + Get-CimInstance Win32_Process -Filter $names | + Where-Object { $_.ExecutablePath -and ($want -contains $_.ExecutablePath.ToLowerInvariant()) } | + ForEach-Object { + if ($env:QUIL_DEV_MODE -eq "stop") { + Stop-Process -Id $_.ProcessId -Force -ErrorAction SilentlyContinue + Wait-Process -Id $_.ProcessId -Timeout 5 -ErrorAction SilentlyContinue + if (Get-Process -Id $_.ProcessId -ErrorAction SilentlyContinue) { return } + } + "{0} {1}" -f $_.ProcessId, $_.ExecutablePath + }' | tr -d '\r' || true +} + +# Linux reads /proc//exe, which the kernel resolves (a binary replaced +# since it started reads " (deleted)" and is still ours). macOS has no +# /proc: `ps` names the candidates and lsof resolves each one's executable, as +# `ps -o comm` shows the path the process was STARTED with, which is relative +# for `./quil-dev`. Both compare against the physical project path. +dev_processes_unix() { + mode="$1" + phys="$(cd "$PROJECT_DIR" && pwd -P)" + want_tui="$phys/$(basename "$dev_tui")" + want_daemon="$phys/$(basename "$dev_daemon")" + found="" + if [ -e /proc/self/exe ]; then + for d in /proc/[0-9]*; do + exe="$(readlink "$d/exe" 2>/dev/null)" || continue + case "$exe" in + "$want_tui" | "$want_tui (deleted)" | "$want_daemon" | "$want_daemon (deleted)") + found="$found${d#/proc/} ${exe% (deleted)} +" ;; + esac + done + else + candidates="$(ps -axo pid=,comm= | awk '$NF ~ /(^|\/)quild?-dev$/ { print $1 }')" + for pid in $candidates; do + exe="$(lsof -a -p "$pid" -d txt -Fn 2>/dev/null | sed -n 's/^n//p' | head -n 1)" + case "$exe" in + "$want_tui" | "$want_daemon") found="$found$pid $exe +" ;; + esac + done + fi + printf '%s' "$found" | while IFS= read -r line; do + [ -n "$line" ] || continue + pid="${line%% *}" + if [ "$mode" = "stop" ]; then + kill "$pid" 2>/dev/null || continue + n=0 + while kill -0 "$pid" 2>/dev/null && [ "$n" -lt 20 ]; do + sleep 0.1 + n=$((n + 1)) + done + kill -0 "$pid" 2>/dev/null && kill -9 "$pid" 2>/dev/null + kill -0 "$pid" 2>/dev/null && continue + fi + printf '%s\n' "$line" + done +} + case "${1:-help}" in build) # Cheap, host-side, and it fails BEFORE the Docker run so a docs-size # problem costs a second rather than a full build. sh "$PROJECT_DIR/scripts/check-claude-md-size.sh" + stop_dev_processes refuse_if_binaries_held echo "building for $TARGET_GOOS/$TARGET_GOARCH (override: QUIL_BUILD_GOOS / QUIL_BUILD_GOARCH)" >&2 @@ -314,6 +478,7 @@ case "${1:-help}" in clean) # Same reason as build: rm cannot remove a held executable, and `set -e` # would abort the cleanup partway through. + stop_dev_processes refuse_if_binaries_held # Driven off BUILT_BINARIES so the two lists cannot drift. The hand-written # list this replaces named only the .exe spellings plus bare quil/quild, so From 6e4587cf4763f9035b4071170beb6b36ab555a91 Mon Sep 17 00:00:00 2001 From: Artjoms Stukans Date: Sat, 26 Sep 2026 11:28:43 +0200 Subject: [PATCH 36/40] docs: tidy auto-stop and client-id wording before the push - dev.sh refusal text no longer claims the dev processes were stopped; it says any found were, and keeps the manual daemon stop command. - AttachPayload.ClientID: an anon client with a paintable geometry can be elected; it only cannot clear a restart reserve. - CLAUDE.md and dev-environment.md: the auto-stop also ends quil-dev mcp bridges, can leave a killed dev TUI window in mouse-tracking or alt-screen mode, ends a shell in a pane of this folder's dev daemon, and its macOS branch is best-effort and untested. - Rename the attach test to state the rule: reattach only after this process already attached there. --- .claude/CLAUDE.md | 2 +- .claude/rules/dev-environment.md | 2 +- internal/ipc/protocol.go | 8 +++++--- internal/tui/multiclient_role_test.go | 9 +++++---- scripts/dev.sh | 12 +++++++++--- 5 files changed, 21 insertions(+), 12 deletions(-) diff --git a/.claude/CLAUDE.md b/.claude/CLAUDE.md index 5578c287..78e291ba 100644 --- a/.claude/CLAUDE.md +++ b/.claude/CLAUDE.md @@ -98,7 +98,7 @@ Go module cache is persisted in a Docker volume (`quil-gomod`) for fast repeated `build` and `clean` refuse to run while any binary they would write is held by a running process (`refuse_if_binaries_held` in `scripts/dev.sh`). Neither platform can overwrite a running executable — Windows fails the open with a sharing violation, Linux returns ETXTBSY — and the ORDER is what makes it dangerous rather than the failure: the six builds are chained with `&&`, so a holder on the Nth leaves the first N-1 freshly built and the rest stale, and a new TUI beside a stale daemon fails the version gate at launch, reading as a bug in whatever you were working on. Detection is a **non-destructive probe, not a process query**: `(exec 3>>"$target")` asks the OS the exact question the build is about to ask, and append never truncates. The first version instead read `.quil/quild.pid` and looked for a daemon — it missed a dev TUI holding `quil-dev.exe` and let the half-build through on its first real use, which is why this checks the condition rather than a proxy for it. Only files in the project directory are probed, so a production install elsewhere is never touched, and a stale or malformed pid file cannot block a build because no pid file is consulted. There is deliberately no override flag: "build anyway" produces exactly the mismatched set it exists to prevent. -**Before that probe, `build` and `clean` stop the DEV variant themselves** (`stop_dev_processes`), because a dev daemon or dev TUI left running was the usual reason for the refusal, and each refusal cost a round trip. Scope is exactly `quil-dev` and `quild-dev` in THIS project directory, matched by FULL executable path (`Win32_Process.ExecutablePath` via `powershell.exe` on Windows, `/proc//exe` on Linux, `lsof` on macOS) — never by name, since every worktree has its own `quil-dev`. The dev daemon goes first and gracefully, through `quil-dev daemon stop` with `QUIL_HOME` set to the project `.quil/`, and only when `.quil/quild.pid` names a live process running this directory's `quild-dev`; whatever remains is stopped by PID alone, never as a tree (`taskkill /T` would reach pane children and lingering wrapper parents). `quil`/`quild` (production) and `quil-debug`/`quild-debug` (which serve the production `~/.quil`) are NEVER stopped, and `~/.quil` is never read. The probe then runs unchanged after a ≤5 s wait for Windows to release the image files, so it still has the last word: a production or debug binary here held by a process refuses the build exactly as before. +**Before that probe, `build` and `clean` stop the DEV variant themselves** (`stop_dev_processes`), because a dev daemon or dev TUI left running was the usual reason for the refusal, and each refusal cost a round trip. Scope is exactly `quil-dev` and `quild-dev` in THIS project directory, matched by FULL executable path (`Win32_Process.ExecutablePath` via `powershell.exe` on Windows, `/proc//exe` on Linux, `lsof` on macOS) — never by name, since every worktree has its own `quil-dev`. The dev daemon goes first and gracefully, through `quil-dev daemon stop` with `QUIL_HOME` set to the project `.quil/`, and only when `.quil/quild.pid` names a live process running this directory's `quild-dev`; whatever remains is stopped by PID alone, never as a tree (`taskkill /T` would reach pane children and lingering wrapper parents). `quil`/`quild` (production) and `quil-debug`/`quild-debug` (which serve the production `~/.quil`) are NEVER stopped, and `~/.quil` is never read. Side effects, all by design: a `quil-dev mcp` bridge from this folder is stopped too (it runs `quil-dev.exe`, so it holds the file); a dev TUI is killed without restoring its console, so its window can be left in mouse-tracking / alt-screen mode and should just be closed; and running the build from a pane hosted by this folder's dev daemon ends that pane's shell along with the daemon. The macOS branch (`ps` + `lsof`) is best-effort and untested. The probe then runs unchanged after a ≤5 s wait for Windows to release the image files, so it still has the last word: a production or debug binary here held by a process refuses the build exactly as before. ### Windows Icon diff --git a/.claude/rules/dev-environment.md b/.claude/rules/dev-environment.md index 218850a2..f754a008 100644 --- a/.claude/rules/dev-environment.md +++ b/.claude/rules/dev-environment.md @@ -38,7 +38,7 @@ Every code change in this repo follows the same loop: 1. Edit code (`internal/…`, `cmd/…`). 2. Rebuild: `./scripts/dev.sh build`. -3. Nothing to do by hand: `./scripts/dev.sh build` (and `clean`) now stops this directory's dev daemon (gracefully, via `quil-dev daemon stop`) and any dev TUI first, matching `quil-dev`/`quild-dev` by full executable path. It never stops `quil`/`quild` or `quil-debug`/`quild-debug`, and a dev TUI you had open is closed by design. If the build still refuses, something else holds a binary here — stop a stray dev daemon by PID from `./.quil/quild.pid` only, never from `~/.quil/quild.pid`. +3. Nothing to do by hand: `./scripts/dev.sh build` (and `clean`) now stops this directory's dev daemon (gracefully, via `quil-dev daemon stop`) and any dev TUI first, matching `quil-dev`/`quild-dev` by full executable path. It never stops `quil`/`quil-debug` or their daemons. By design it also stops `quil-dev mcp` bridges from this folder (they hold `quil-dev.exe`) and any dev TUI you had open — that window may be left in mouse-tracking / alt-screen mode, so close it. Running the build from a pane hosted by this folder's dev daemon ends that pane's shell. The macOS branch is best-effort and untested. If the build still refuses, something else holds a binary here — stop a stray dev daemon by PID from `./.quil/quild.pid` only, never from `~/.quil/quild.pid`. 4. Launch: `./scripts/quil-dev.ps1` (Windows) or `./scripts/quil-dev.sh` (Unix). 5. Verify `[dev]` is visible in the status bar before testing. 6. Test the change. When done, close the dev TUI — do NOT run any `kill-daemon` / `reset-daemon` helper scripts. diff --git a/internal/ipc/protocol.go b/internal/ipc/protocol.go index fe0327db..8f4bb9ea 100644 --- a/internal/ipc/protocol.go +++ b/internal/ipc/protocol.go @@ -323,9 +323,11 @@ type AttachPayload struct { CWD string `json:"cwd,omitempty"` // ClientID identifies this client across reconnects, for multi-client // sync: master election, the client list and per-client geometry all key - // on it. Empty on an older client, which the daemon treats as a client - // that cannot participate in election — it is neither offered control - // nor handed a follower's resize_panes stream. + // on it. Empty on an older client: the daemon mints an "anon-" id + // scoped to that conn, and the client is otherwise treated like any + // other — with a paintable geometry it CAN be elected master. The one + // difference is that an attach with no ClientID never clears a restart + // reserve (see Reattach). ClientID string `json:"client_id,omitempty"` // Reattach is false on the process's first attach to this daemon and // true on every later one, whichever client path sends it (a daemon diff --git a/internal/tui/multiclient_role_test.go b/internal/tui/multiclient_role_test.go index d2567f9b..97475ebe 100644 --- a/internal/tui/multiclient_role_test.go +++ b/internal/tui/multiclient_role_test.go @@ -483,11 +483,12 @@ func attachPayloads(t *testing.T, conn *fakeConn) []ipc.AttachPayload { return out } -// TestAttach_ReattachFlagMarksOnlyTheReconnectPath: the daemon's restart +// TestAttach_ReattachOnlyAfterThisProcessAttachedThere: the daemon's restart // reserve yields to a FIRST attach from a new process and keeps waiting for a -// reconnecting one, so the flag must be false on the attach a WindowSizeMsg -// sends and true on the one a completed redial sends after it. -func TestAttach_ReattachFlagMarksOnlyTheReconnectPath(t *testing.T) { +// reconnecting one, so the flag is false on this process's first attach to a +// destination (here the one a WindowSizeMsg sends) and true on every later +// one (here the one a completed redial sends). +func TestAttach_ReattachOnlyAfterThisProcessAttachedThere(t *testing.T) { t.Setenv("QUIL_HOME", t.TempDir()) first, fresh := newFakeConn(), newFakeConn() r := NewRouter(map[string]Client{"gpu01": first}) diff --git a/scripts/dev.sh b/scripts/dev.sh index 22ff2407..f72bbb96 100644 --- a/scripts/dev.sh +++ b/scripts/dev.sh @@ -141,9 +141,15 @@ refuse_if_binaries_held() { leaving a mismatched set — typically a new TUI against a stale daemon, which then fails the version gate at launch. - The dev daemon and dev TUIs from this directory were already stopped. - Close whatever else is running these files: a quil or quil-debug started - from this directory, or a scanner holding them. + Any dev daemon, dev TUI or dev MCP bridge found running from this + directory was stopped first. Close whatever else holds these files: a + quil or quil-debug started from this directory, a dev process that could + not be stopped (another user's, an elevated one), or a scanner. For a dev + daemon that is still running: + + QUIL_HOME="$PROJECT_DIR/.quil" "$PROJECT_DIR/quil-dev$EXE" daemon stop + + and close any dev TUI window by hand. Only files in $PROJECT_DIR were checked. A production install elsewhere is untouched. From 1987b44c3f655132329aafd5fabca71088966e13 Mon Sep 17 00:00:00 2001 From: Artjoms Stukans Date: Sat, 26 Sep 2026 19:38:14 +0200 Subject: [PATCH 37/40] fix(daemon): publish pane output and broadcast it under one gate A flush wrote OutputBuf and advanced outPos under PluginMu but took holdGate only later, after the detectors and plugin handlers. A whole handleAttach could run in that gap: its replay already carried the new bytes (end = outPos), its hold never saw the flush, and the resumed broadcast sent the same bytes again to the now-unheld conn. The flush now takes holdGate for read before PluginMu and releases it right after the hold append and broadcast, so publication and delivery are one step against an attach's hold. Lock order is holdGate then PluginMu everywhere. The mouse-mode broadcastState (still decided inside the PluginMu span) and the bell, hand-start, OSC 133 and plugin detectors run after the gate, so it is never held across a spawn. Regression: TestHold_AttachInsideAPublishedFlushGetsItsBytesOnce, via the new afterFlushPublish seam, received "ONEONEEND" before the fix. --- .claude/rules/daemon-lifecycle.md | 26 ++++++-- internal/daemon/daemon.go | 73 ++++++++++++++-------- internal/daemon/outputhold.go | 14 +++-- internal/daemon/outputhold_test.go | 99 ++++++++++++++++++++++++++++++ 4 files changed, 179 insertions(+), 33 deletions(-) diff --git a/.claude/rules/daemon-lifecycle.md b/.claude/rules/daemon-lifecycle.md index fdd02d63..587b0627 100644 --- a/.claude/rules/daemon-lifecycle.md +++ b/.claude/rules/daemon-lifecycle.md @@ -193,11 +193,27 @@ replay's `OutputBuf` snapshot can never disagree about what has been sent. Every flush during a hold is copied into that conn's hold (`holdOutput`, keyed by `*ipc.Conn` under the daemon's `holdMu` leaf lock) BEFORE the ordinary broadcast. Two locks, always taken `holdGate` → `holdMu`: `holdGate` (an -`RWMutex`) makes "append to the hold, then broadcast" one step against setting -or clearing the conn's flag (a flush holding it for READ; every hold that -starts or ends holding it for WRITE) — without it a flush could append and -broadcast to a conn whose flag flipped in between, either duplicating the -bytes or losing them. +`RWMutex`) makes "publish to `OutputBuf`, append to the hold, then broadcast" +one step against setting or clearing the conn's flag (a flush holding it for +READ; every hold that starts or ends holding it for WRITE) — without it a flush +could append and broadcast to a conn whose flag flipped in between, either +duplicating the bytes or losing them. **The read lock is taken BEFORE the +`OutputBuf` write and `outPos` advance, not after them**: taken after, a whole +`handleAttach` fitted between publication and broadcast — its replay already +carried the bytes (`end = outPos`), its hold never saw the flush, and the +resumed broadcast sent them again to the now-unheld conn (`ONEONEEND`, +`TestHold_AttachInsideAPublishedFlushGetsItsBytesOnce`, seam +`afterFlushPublish`). So the flush takes `holdGate` → `PluginMu`, and nothing +anywhere takes `holdGate` while holding a `PluginMu` (`beginOutputHold`, +`finishOutputHold`, `dropOutputHold` and `handleAttach`'s replay span all keep +the two apart). The gated span does no I/O and spawns nothing, which is why +the mouse-mode `broadcastState` (decided inside the `PluginMu` span, sent +after the gate — the state frame rides the must-deliver queue, which +`sendLoop` drains ahead of pane output, so its enqueue order against this +chunk never decided delivery order) and the bell / hand-start / OSC 133 / +plugin detectors run AFTER it: the hand-start conversion restarts the pane, +and a flush re-entered from there would ask for the read lock again behind a +waiting writer. On release, held chunks are deduped against `end[paneID]` (the replay's own stream-position snapshot, read in the same `PluginMu` span as the `OutputBuf` diff --git a/internal/daemon/daemon.go b/internal/daemon/daemon.go index c85eefaf..a806e2a1 100644 --- a/internal/daemon/daemon.go +++ b/internal/daemon/daemon.go @@ -291,12 +291,15 @@ type Daemon struct { // holdCount is len(holds), written under holdGate (write) so a flush // holding it for read can skip holdMu when nothing is held. holdCount atomic.Int32 - // afterHoldOutput and beforeFinishHold are test seams: a flush calls the - // first between its hold append and its broadcast; the release calls the - // second between seeing an empty batch and ending the hold. Set before - // any flush runs; nil in production. - afterHoldOutput func(paneID string) - beforeFinishHold func(c *ipc.Conn) + // afterHoldOutput, beforeFinishHold and afterFlushPublish are test seams: + // a flush calls the first between its hold append and its broadcast; the + // release calls the second between seeing an empty batch and ending the + // hold; a flush calls the third right after its OutputBuf write and + // outPos advance, before its hold append. Set before any flush runs; nil + // in production. + afterHoldOutput func(paneID string) + beforeFinishHold func(c *ipc.Conn) + afterFlushPublish func(paneID string) } func New(cfg config.Config) *Daemon { @@ -4355,9 +4358,29 @@ func (d *Daemon) flushPaneOutputGeneration(paneID string, data []byte, generatio if pane == nil { return } + // holdGate is held for read from BEFORE the OutputBuf write until after + // the hold append and broadcast, so publishing these bytes and delivering + // them are one step against an attach's hold (outputhold.go). Taken any + // later, a whole handleAttach fits between the two: its replay already + // carries the bytes (end = outPos), its hold never sees them, and the + // broadcast then sends them again to the now-unheld conn. + // + // Order holdGate → PluginMu, as everywhere: nothing takes holdGate while + // holding a PluginMu. The span below does no I/O and spawns nothing — + // Broadcast only enqueues — which is why the detectors and plugin + // handlers run after it: the hand-start conversion restarts the pane, + // and a flush re-entered from there would ask for the read lock again + // behind a waiting writer. + msg, _ := ipc.NewMessage(ipc.MsgPaneOutput, ipc.PaneOutputPayload{ + PaneID: paneID, + Data: data, + Generation: generation, + }) + d.holdGate.RLock() pane.PluginMu.Lock() if generation != 0 && (generation != pane.ptyGen || pane.PTY == nil) { pane.PluginMu.Unlock() + d.holdGate.RUnlock() return } if pane.OutputBuf != nil { @@ -4421,11 +4444,31 @@ func (d *Daemon) flushPaneOutputGeneration(paneID string, data []byte, generatio doMouseBroadcast = true } pane.PluginMu.Unlock() + if d.afterFlushPublish != nil { + d.afterFlushPublish(paneID) + } + + // Held conns get a copy through their hold; the broadcast skips them. + // Outside PluginMu, and Broadcast only enqueues, so the gate is held + // across no I/O. + d.holdOutput(paneID, start, data, generation) + if d.afterHoldOutput != nil { + d.afterHoldOutput(paneID) + } + d.broadcast(msg) + d.holdGate.RUnlock() + + // After the gate, not inside it: broadcastState builds the whole + // workspace, and the state frame travels on the must-deliver queue, which + // sendLoop drains ahead of pane output anyway — so enqueueing it before or + // after this chunk never decided which the client saw first. if doMouseBroadcast { logger.Debug("pane %s: mouse-mode change tracking=%v sgr=%v", paneID, newModes.tracking(), newModes.sgr) d.broadcastState() } + // The detectors read the same bytes the clients were just sent; none of + // them changes what this chunk is or which generation it carries. d.detectBellEvent(pane, paneID, data) // Before the OSC 133 detector: a conversion restarts the pane, and the // shell's own prompt hooks emit a D for the command it never ran. Running @@ -4438,24 +4481,6 @@ func (d *Daemon) flushPaneOutputGeneration(paneID string, data []byte, generatio d.detectHandStart(pane, paneID, data, flushedAt) d.detectOSC133Exit(pane, paneID, data) d.applyPluginHandlers(pane, paneID, data) - - msg, _ := ipc.NewMessage(ipc.MsgPaneOutput, ipc.PaneOutputPayload{ - PaneID: paneID, - Data: data, - Generation: generation, - }) - // Held conns get a copy through their hold; the broadcast skips them. The - // gate makes the pair one step against a hold starting or ending, or a - // conn could get these bytes twice or not at all (outputhold.go). - // Outside PluginMu, and Broadcast only enqueues, so the gate is held - // across no I/O. - d.holdGate.RLock() - d.holdOutput(paneID, start, data, generation) - if d.afterHoldOutput != nil { - d.afterHoldOutput(paneID) - } - d.broadcast(msg) - d.holdGate.RUnlock() } // detectBellEvent checks for standalone bell characters (not OSC terminators). diff --git a/internal/daemon/outputhold.go b/internal/daemon/outputhold.go index a253d5ff..3aa21954 100644 --- a/internal/daemon/outputhold.go +++ b/internal/daemon/outputhold.go @@ -29,10 +29,16 @@ import ( // - holdMu is a LEAF guarding the holds map and each hold's contents. It is // never held with PluginMu or the client registry mutex, and never across // SendBlocking. -// - holdGate makes "append to the holds, then broadcast" ONE step against -// setting or clearing a conn's flag. A flush holds it for READ from its -// hold append through its broadcast (Broadcast only enqueues, so this -// blocks on nothing). Every hold that starts or ends — beginOutputHold, +// - holdGate makes "publish to OutputBuf, append to the holds, then +// broadcast" ONE step against setting or clearing a conn's flag. A flush +// holds it for READ from BEFORE its OutputBuf write (and outPos advance) +// through its broadcast, taking the pane's PluginMu inside it — so the +// order is holdGate → PluginMu, and nothing takes holdGate while holding +// a PluginMu. Publishing outside the gate let a whole attach run between +// the write and the broadcast: its replay carried the bytes, its hold +// never saw them, and the broadcast sent them a second time. The span +// does no I/O (Broadcast only enqueues) and spawns nothing; the flush's +// detectors run after it. Every hold that starts or ends — beginOutputHold, // the release's final clear, dropOutputHold — holds it for WRITE, which // also keeps holdCount exact for a flush holding it for read. Without it, a flush could append its chunk, the release // could send that chunk and clear the flag, and then the flush's broadcast diff --git a/internal/daemon/outputhold_test.go b/internal/daemon/outputhold_test.go index 9fe1197d..1b7b89cb 100644 --- a/internal/daemon/outputhold_test.go +++ b/internal/daemon/outputhold_test.go @@ -540,6 +540,105 @@ func TestHold_FlushStraddlingTheBeginArrivesOnce(t *testing.T) { } } +// An attach arriving while a flush sits between publishing its bytes to +// OutputBuf and broadcasting them. From the attach's state frame on, the +// client must see those bytes once: in the replay or live, never both. +// +// Publishing outside holdGate let the whole attach run in that gap: its +// replay carried "ONE" (end = outPos), its hold never saw the flush, and the +// resumed broadcast sent "ONE" again to the now-unheld conn — "ONEONEEND". +// With publication inside the gate the attach parks in beginOutputHold until +// the flush finishes; the flush's live frame then lands BEFORE the state frame +// (the client was not attached yet), and after it come the replay and "END". +func TestHold_AttachInsideAPublishedFlushGetsItsBytesOnce(t *testing.T) { + h := newHoldHarness(t) + tab := h.d.session.CreateTab("T") + p := h.pane(tab.ID, "terminal") + client, conn := h.dial("B") + + paused, resume := make(chan struct{}), make(chan struct{}) + var once sync.Once + h.d.afterFlushPublish = func(id string) { + if id != p.ID { + return + } + once.Do(func() { + close(paused) + <-resume + }) + } + flushed := make(chan struct{}) + go func() { + h.d.flushPaneOutput(p.ID, []byte("ONE")) + close(flushed) + }() + <-paused + + attach, err := ipc.NewMessage(ipc.MsgAttach, ipc.AttachPayload{Cols: 120, Rows: 40, ClientID: "B"}) + if err != nil { + t.Fatalf("attach message: %v", err) + } + attached := make(chan struct{}) + go func() { + h.d.handleAttach(conn, attach) + close(attached) + }() + // Either the attach finished inside the gap (the defect) or it parked + // on holdGate behind the paused flush. A PENDING writer blocks new + // readers, so TryRLock failing is the proof it parked. + waitUntil(t, "the attach to finish or park on holdGate", func() bool { + select { + case <-attached: + return true + default: + } + if h.d.holdGate.TryRLock() { + h.d.holdGate.RUnlock() + return false + } + return true + }) + close(resume) + <-flushed + <-attached + h.d.flushPaneOutput(p.ID, []byte("END")) + + if err := client.SetReadDeadline(time.Now().Add(10 * time.Second)); err != nil { + t.Fatalf("set read deadline: %v", err) + } + var afterState []byte + stateSeen := false + for { + m, err := client.Receive() + if err != nil { + t.Fatalf("read: %v (after the state frame so far: %q)", err, afterState) + } + if m.Type == ipc.MsgWorkspaceState { + stateSeen = true + continue + } + if m.Type != ipc.MsgPaneOutput { + continue + } + var f ipc.PaneOutputPayload + if err := m.DecodePayload(&f); err != nil { + t.Fatalf("decode pane_output: %v", err) + } + if f.PaneID != p.ID { + continue + } + if stateSeen { + afterState = append(afterState, f.Data...) + } + if string(f.Data) == "END" { + break + } + } + if got := string(afterState); got != "ONEEND" { + t.Fatalf("from the attach's state frame on, received %q, want %q", got, "ONEEND") + } +} + func TestHold_DisconnectDropsHold(t *testing.T) { h := newHoldHarness(t) tab := h.d.session.CreateTab("T") From 305026f53cf4bb292beb46754e6f8c0c04be9abd Mon Sep 17 00:00:00 2001 From: Artjoms Stukans Date: Sat, 26 Sep 2026 19:44:06 +0200 Subject: [PATCH 38/40] fix(tui): retire the typing guard on explicit local navigation A remote tab switch arms a 250 ms guard that keeps typed keys and pastes on the pane the user was in. It survived explicit local navigation: after a remote switch, Alt+3 moved focus to t3 while the next key still went to the pane on t1, and a pane click cleared only the unseen-ack hold. retireTypingGuard clears guardPaneID and remoteSwitchAt at three sites: any mouse click (a click on the remote-focused pane changes no focus, so only this tells that choice apart); a key press whose handling changed the local focus (dest, project, tab, input pane), measured by a defer on Update's named return, which covers pane-navigation keys, the palette, the attention queue and pane history; and switchTab, so choosing the already-focused tab counts. Uninterrupted typing across the remote switch stays guarded. Regression: TestTypingGuard_*RetiresTheGuard (tab key, tab key to the remote tab, pane click on either pane, Alt+Right, project.next), each checking a typed key and a paste. --- .claude/rules/tui-rendering.md | 20 +++- internal/tui/model.go | 61 +++++++++++- internal/tui/typing_guard_test.go | 154 ++++++++++++++++++++++++++++++ 3 files changed, 232 insertions(+), 3 deletions(-) diff --git a/.claude/rules/tui-rendering.md b/.claude/rules/tui-rendering.md index a9c3ae57..31335312 100644 --- a/.claude/rules/tui-rendering.md +++ b/.claude/rules/tui-rendering.md @@ -327,14 +327,30 @@ mouse input always targets whatever is under the pointer now, since a click is inherently aimed at what is on screen. A flash shows `Tab switched by another client`. +**Explicit local navigation retires the guard** (`retireTypingGuard` clears +`guardPaneID` and `remoteSwitchAt`). The guard protects keys typed straight +through a remote switch; once the user CHOOSES a tab, pane or project, what +they type next is meant for that choice. Three sites, each needed: any +`tea.MouseClickMsg` (in `Update`'s prologue, beside the `remoteFocusUnacked` +clear — a click on the pane the remote switch focused changes no focus, so no +diff could tell that choice apart); a `tea.KeyPressMsg` whose handling changed +`localFocus()` (dest, project, active tab, input pane — a defer on `Update`'s +named return, armed only while a guard is live, covering pane-navigation +keys, the palette, the attention queue, pane history and opening an overlay +without a retire in each); and `switchTab` itself, so Alt+N naming the tab +the remote switch already focused still counts. Before this, Alt+3 or a click +inside the window moved focus while the next key still went to the pane the +user had left. + **Unseen is not acknowledged until local input arrives.** A pane that became focused only because of a remote switch is skipped by `ackFocusedPane` (no `pane_seen` is sent) until this client receives a real key or mouse click — otherwise every attached client would clear the mark on every remote switch whether or not anyone actually looked at the pane. -Tests: `typing_guard_test.go`. Mutation-checked: the 250 ms window itself and -the `requestedTab` token gate. +Tests: `typing_guard_test.go`. Mutation-checked: the 250 ms window itself, +the `requestedTab` token gate, and each of the three retire sites +(`TestTypingGuard_*RetiresTheGuard`). ### Mouse-wheel forwarding to tracking apps diff --git a/internal/tui/model.go b/internal/tui/model.go index 3ab7f8b8..2548dc94 100644 --- a/internal/tui/model.go +++ b/internal/tui/model.go @@ -1450,6 +1450,25 @@ func (m Model) Update(msg tea.Msg) (retModel tea.Model, retCmd tea.Cmd) { retModel = mm } }() + // Typing guard (spec §8.1): a key that MOVES this client's own focus — a + // pane-navigation key, the palette, the attention queue, pane history, + // opening an overlay — retires the guard, so the next key reaches the pane + // the user just chose instead of the one they left. Measured on the named + // return, for the reason the defer above gives: every navigation path + // returns through it, and a retire sprinkled into each is one a future path + // forgets. switchTab also retires it itself, so choosing the tab focus is + // already on still counts (a project switch always moves focus, since it + // refuses the active project). Armed only while a guard is live, so an + // ordinary keystroke pays nothing. + if _, isKey := msg.(tea.KeyPressMsg); isKey && m.guardPaneID != "" { + before := m.localFocus() + defer func() { + if mm, ok := retModel.(Model); ok && mm.guardPaneID != "" && mm.localFocus() != before { + mm.retireTypingGuard() + retModel = mm + } + }() + } start := time.Now() markUpdateStart(start) defer func() { @@ -1468,9 +1487,17 @@ func (m Model) Update(msg tea.Msg) (retModel tea.Model, retCmd tea.Cmd) { // ackFocusedPane runs below, so the very keystroke that answers the guard // also acks the pane in the same Update call — there is no reason to make // the user press twice. + // + // A click also retires the typing guard outright, wherever it lands: the + // user has left the keyboard, so nothing they type next is continuing + // through the remote switch — and a click on the pane the remote switch + // focused changes no focus, so only this can tell that choice apart. switch msg.(type) { - case tea.KeyPressMsg, tea.MouseClickMsg: + case tea.KeyPressMsg: m.remoteFocusUnacked = false + case tea.MouseClickMsg: + m.remoteFocusUnacked = false + m.retireTypingGuard() } // Acknowledge the focused pane of the active tab before processing the // message — focusing is the acknowledgement; see ackFocusedPane. @@ -7301,6 +7328,9 @@ func (m *Model) switchTab(idx int) tea.Cmd { from := m.activeTabModel() tabID, dest := target.ID, target.Dest m.setActiveTabIdx(idx) + // The user picked this tab, even when it is the one a remote switch just + // focused: the typing guard no longer describes where they are typing. + m.retireTypingGuard() // Typing guard (spec §8.1): this client asked for tabID, so the broadcast // that lands it must not be mistaken for another client's switch. Recorded // against the CURRENT project — target and its project share one Dest, so @@ -8977,6 +9007,35 @@ func (m Model) enqueueKeyInput(paneID string, data []byte) { // wherever the remote switch moved focus. It falls through to paneID once the // window has elapsed or the guarded pane no longer exists (closed, moved, // destroyed — there is nowhere left to redirect to). +// retireTypingGuard ends the typing guard at once. The guard is for keys typed +// straight through a remote switch; the moment this client's own user picks a +// tab, a project or a pane, what they type next is meant for that choice, and +// redirecting it to the pane they left would put it somewhere they are no +// longer looking. +func (m *Model) retireTypingGuard() { + m.guardPaneID = "" + m.remoteSwitchAt = time.Time{} +} + +// localFocusKey names where this client's keystrokes go without the guard: +// the active project, its active tab and that tab's input pane. The typing +// guard compares it across one key press to tell a key that navigated from one +// that was typed. +type localFocusKey struct { + dest, project, tab, pane string +} + +func (m Model) localFocus() localFocusKey { + var k localFocusKey + if p := m.cur(); p != nil { + k.dest, k.project = p.Dest, p.ID + } + if tab := m.activeTabModel(); tab != nil { + k.tab, k.pane = tab.ID, tabInputPaneID(tab) + } + return k +} + func (m Model) guardedInputTarget(paneID string) string { if m.guardPaneID == "" { return paneID diff --git a/internal/tui/typing_guard_test.go b/internal/tui/typing_guard_test.go index 1d56486f..84bd9311 100644 --- a/internal/tui/typing_guard_test.go +++ b/internal/tui/typing_guard_test.go @@ -792,6 +792,160 @@ func TestTypingGuard_StaleFromTabBroadcastIsAdoptedPastTheBound(t *testing.T) { } } +// splitGuardTab turns tab idx of m's current project into a horizontal split +// of its own pane (left) and a second pane "b" (right), leaving the left +// one active, and re-lays out the tabs so clicks have real geometry. +func splitGuardTab(t *testing.T, m *Model, idx int) (left, right string) { + t.Helper() + tab := m.curTabs()[idx] + left = tab.ActivePane + right = left + "b" + tab.Root.SplitLeaf(left, SplitHorizontal) + tab.Root.Right.Pane = NewPaneModel(right, 1024) + tab.ActivePane = left + m.resizeTabs() + return left, right +} + +// splitGuardBroadcast is typingGuardBroadcast for a model whose tab splitTab +// also holds splitPane (splitGuardTab's right pane). +func splitGuardBroadcast(activeTab, splitTab, splitPane string, tabIDs ...string) WorkspaceStateMsg { + state := typingGuardBroadcast(activeTab, tabIDs...) + for i := range state.Tabs { + if state.Tabs[i].ID == splitTab { + state.Tabs[i].Panes = append(state.Tabs[i].Panes, splitPane) + } + } + state.Panes = append(state.Panes, PaneInfo{ID: splitPane, TabID: splitTab}) + return state +} + +// assertKeyAndPasteReach types one key and then pastes, both through Update, +// and asserts each reached want. +func assertKeyAndPasteReach(t *testing.T, m Model, want string) { + t.Helper() + updated, _ := m.Update(tea.KeyPressMsg{Text: "x"}) + got := updated.(Model) + if in := drainOneInput(t, &got); in.paneID != want { + t.Errorf("typed key queued for %q, want %q (the pane the user just chose)", in.paneID, want) + } + updated, _ = got.Update(tea.PasteMsg{Content: "pasted"}) + got = updated.(Model) + if in := drainOneInput(t, &got); in.paneID != want { + t.Errorf("paste queued for %q, want %q (the pane the user just chose)", in.paneID, want) + } +} + +// remoteSwitchThenWait delivers a remote switch, checks it armed the guard for +// wantGuard, and moves the clock 100 ms on — well inside the guard window, so +// only the navigation under test can explain input escaping the guard. +func remoteSwitchThenWait(t *testing.T, m *Model, t0 time.Time, state WorkspaceStateMsg, wantGuard string) Model { + t.Helper() + updated, _ := m.Update(state) + got := updated.(Model) + if got.guardPaneID != wantGuard { + t.Fatalf("setup: guardPaneID = %q, want %q", got.guardPaneID, wantGuard) + } + got.now = func() time.Time { return t0.Add(100 * time.Millisecond) } + return got +} + +// Review finding M-1: explicit local navigation inside the guard window retires +// the guard. Uninterrupted typing across the remote switch stays guarded +// (TestTypingGuard_KeyWithinWindowGoesToOldPane); a user who then CHOOSES a +// tab, pane or project is typing into that choice, not the pane they left. + +func TestTypingGuard_LocalTabKeyRetiresTheGuard(t *testing.T) { + t.Parallel() + t0 := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + m, _ := typingGuardModel(t, t0, "t1", "t2", "t3") + got := remoteSwitchThenWait(t, m, t0, typingGuardBroadcast("t2", "t1", "t2", "t3"), "p1") + + updated, _ := got.Update(altKey('3')) + got = updated.(Model) + if got.activeTabModel().ID != "t3" { + t.Fatalf("active tab = %q after Alt+3, want t3", got.activeTabModel().ID) + } + assertKeyAndPasteReach(t, got, "p3") +} + +// Alt+N naming the tab the remote switch already focused changes no focus, and +// is still the user's choice of that tab. +func TestTypingGuard_LocalTabKeyToTheRemoteTabRetiresTheGuard(t *testing.T) { + t.Parallel() + t0 := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + m, _ := typingGuardModel(t, t0, "t1", "t2") + got := remoteSwitchThenWait(t, m, t0, typingGuardBroadcast("t2", "t1", "t2"), "p1") + + updated, _ := got.Update(altKey('2')) + assertKeyAndPasteReach(t, updated.(Model), "p2") +} + +func TestTypingGuard_PaneClickRetiresTheGuard(t *testing.T) { + t.Parallel() + for _, tc := range []struct { + name string + x int + want string + }{ + {"click on the other pane", 75, "p2b"}, + {"click on the pane the remote switch focused", 20, "p2"}, + } { + t.Run(tc.name, func(t *testing.T) { + t.Parallel() + t0 := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + m, _ := typingGuardModel(t, t0, "t1", "t2") + _, right := splitGuardTab(t, m, 1) + got := remoteSwitchThenWait(t, m, t0, splitGuardBroadcast("t2", "t2", right, "t1", "t2"), "p1") + + updated, _ := got.Update(tea.MouseClickMsg{X: tc.x, Y: 10, Button: tea.MouseLeft}) + got = updated.(Model) + updated, _ = got.Update(tea.MouseReleaseMsg{X: tc.x, Y: 10, Button: tea.MouseLeft}) + got = updated.(Model) + if active := got.activeTabModel().ActivePane; active != tc.want { + t.Fatalf("active pane = %q after the click, want %q", active, tc.want) + } + assertKeyAndPasteReach(t, got, tc.want) + }) + } +} + +func TestTypingGuard_PaneNavigationKeyRetiresTheGuard(t *testing.T) { + t.Parallel() + t0 := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + m, _ := typingGuardModel(t, t0, "t1", "t2") + _, right := splitGuardTab(t, m, 1) + got := remoteSwitchThenWait(t, m, t0, splitGuardBroadcast("t2", "t2", right, "t1", "t2"), "p1") + + updated, _ := got.Update(tea.KeyPressMsg{Code: tea.KeyRight, Mod: tea.ModAlt}) // pane.right + got = updated.(Model) + if active := got.activeTabModel().ActivePane; active != right { + t.Fatalf("active pane = %q after Alt+Right, want %q", active, right) + } + assertKeyAndPasteReach(t, got, right) +} + +func TestTypingGuard_ProjectSwitchRetiresTheGuard(t *testing.T) { + t.Parallel() + t0 := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + m, _ := typingGuardModel(t, t0, "t1", "t2") + // A second project on another host: the local broadcast below replaces + // only the local daemon's projects, so this one survives it. + t3 := NewTabModel("t3", "t3") + t3.Root = NewLeaf(NewPaneModel("p3", 1024)) + t3.ActivePane = "p3" + m.projects = append(m.projects, &ProjectModel{ID: "proj-b", Name: "B", Dest: "hostB", tabs: []*TabModel{t3}}) + m.resizeTabs() + got := remoteSwitchThenWait(t, m, t0, typingGuardBroadcast("t2", "t1", "t2"), "p1") + + updated, _ := got.Update(tea.KeyPressMsg{Code: tea.KeyRight, Mod: tea.ModAlt | tea.ModShift}) // project.next + got = updated.(Model) + if p := got.cur(); p == nil || p.ID != "proj-b" { + t.Fatalf("active project = %v after project.next, want proj-b", p) + } + assertKeyAndPasteReach(t, got, "p3") +} + // TestArmReattachReset_ClearsTypingGuardStateForThatDest pins review round // 1's minor 3: a reattach replaces its destination's whole state, so a // pending requestedTab token can never land the broadcast it was waiting for, From aee41f8aa4064b50df7fb59ac04590c4f817ee0e Mon Sep 17 00:00:00 2001 From: Artjoms Stukans Date: Sat, 26 Sep 2026 19:49:03 +0200 Subject: [PATCH 39/40] fix(daemon): elect the size master on the raw window size at attach attachClient read AttachPayload.Cols/Rows as the raw window for master eligibility, but attachMessage sends the pane interior (the window less the sidebar, chrome and borders): a paintable 80x12 window sent 78x8 and was ineligible against the 40x10 floor. The first WindowSizeMsg's client_geometry hid it at launch; a reconnect or runtime host attach sends no geometry after it, so a returning master near the floor lost its slot to a follower, and list_clients showed interior sizes. AttachPayload gains WinCols/WinRows (win_cols/win_rows, omitempty): the raw window, 0x0 below the paintable floor exactly like client_geometry. Cols/Rows stay the first pane's spawn size. The daemon feeds the registry from the raw pair when present and falls back to Cols/Rows for an older client. Regression: TestClientDispatch_ReattachNearTheFloorKeepsTheSlot (master went to B, list_clients showed 78x8) and TestAttach_ReconnectCarriesTheRawWindow. --- .claude/rules/daemon-lifecycle.md | 15 ++++++ .claude/rules/tui-rendering.md | 8 +++ internal/daemon/clients.go | 18 ++++++- internal/daemon/clients_wiring_test.go | 68 ++++++++++++++++++++++++++ internal/ipc/protocol.go | 11 +++++ internal/tui/model.go | 10 ++++ internal/tui/multiclient_role_test.go | 48 ++++++++++++++++++ 7 files changed, 177 insertions(+), 1 deletion(-) diff --git a/.claude/rules/daemon-lifecycle.md b/.claude/rules/daemon-lifecycle.md index 587b0627..737312f2 100644 --- a/.claude/rules/daemon-lifecycle.md +++ b/.claude/rules/daemon-lifecycle.md @@ -102,6 +102,21 @@ shrink below the floor (`setGeometry`, fed by `MsgClientGeometry`) drops eligibility and re-elects AT ONCE, with no grace, because the client is still attached and the daemon knows immediately. +**Raw means the WINDOW, and the attach carries it separately.** +`AttachPayload.Cols`/`Rows` are the first pane's SPAWN size — the pane +interior `attachMessage` computes (window minus sidebar, chrome and border). +The raw window rides beside them in `WinCols`/`WinRows` (`win_cols`/`win_rows`, +omitempty; 0x0 below the paintable floor exactly like `client_geometry`), and +`attachWindowSize` (`clients.go`) feeds the registry — eligibility and +`list_clients` — from that pair, falling back to `Cols`/`Rows` only when it is +absent (an older client). Electing on the interior made a paintable window +near the floor ineligible (80x12 sends a 78x8 interior, under 40x10); the +first `WindowSizeMsg`'s `client_geometry` hid that on launch, but a reconnect +or runtime host attach (`attachToDest`) sends no geometry after it, so a +returning master lost its slot to a follower +(`TestClientDispatch_ReattachNearTheFloorKeepsTheSlot`, +`TestAttach_ReconnectCarriesTheRawWindow`). + A master whose LINK IS LOST (no `MsgDetach`) keeps its slot for `master_grace_minutes` (`[daemon]`, default 3, clamped 0–60, `internal/config`; 0 = no grace) — but ONLY while another client that was attached at the moment diff --git a/.claude/rules/tui-rendering.md b/.claude/rules/tui-rendering.md index 31335312..d907f2a3 100644 --- a/.claude/rules/tui-rendering.md +++ b/.claude/rules/tui-rendering.md @@ -136,6 +136,14 @@ entry yet) answers false, so a client that has not heard from a destination behaves as it always did. See `.claude/rules/daemon-lifecycle.md`'s "Multi-client" section for the daemon-side registry and election this reads. +**`attachMessage` sends two sizes.** `Cols`/`Rows` stay the pane interior the +daemon spawns an empty workspace's first pane at; `WinCols`/`WinRows` are the +raw window (`m.width`/`m.height`, 0x0 when `!terminalPaintable()` — the pair +`clientGeometryCmd` reports), which the daemon elects on. Both attach paths +carry it, and the reconnect one (`attachToDest`) must: no `client_geometry` +follows it, so a raw size missing there left an 80x12 master judged on its +78x8 interior until the next resize. + ### The resize gates and batching **A follower sends no `MsgResizePane`/`MsgResizePanes`.** The three resize diff --git a/internal/daemon/clients.go b/internal/daemon/clients.go index 553770a8..bb007397 100644 --- a/internal/daemon/clients.go +++ b/internal/daemon/clients.go @@ -414,7 +414,23 @@ func (d *Daemon) attachClient(conn *ipc.Conn, attach ipc.AttachPayload) clientCh return clientChange{} } id := truncateField(attach.ClientID, maxClientIDLen) - return d.clients.attach(conn, id, clampClientDim(attach.Cols), clampClientDim(attach.Rows), attach.CWD, attach.Reattach) + cols, rows := attachWindowSize(attach) + return d.clients.attach(conn, id, clampClientDim(cols), clampClientDim(rows), attach.CWD, attach.Reattach) +} + +// attachWindowSize is the RAW window size an attach reports, which is what +// eligibility and list_clients mean by a client's geometry. A current client +// sends it in WinCols/WinRows; Cols/Rows are its pane interior, smaller by the +// sidebar, the chrome and the border, so electing on them made a paintable +// window near the floor (80x12 → 78x8) ineligible on every reattach, where no +// client_geometry follows to correct it. An older client sends only Cols/Rows, +// and they are its best answer. Both pairs are 0x0 below the paintable floor, +// so an absent raw pair and an unpaintable one read the same. +func attachWindowSize(attach ipc.AttachPayload) (cols, rows int) { + if attach.WinCols > 0 || attach.WinRows > 0 { + return attach.WinCols, attach.WinRows + } + return attach.Cols, attach.Rows } // clampClientDim bounds one axis of a self-reported window size to diff --git a/internal/daemon/clients_wiring_test.go b/internal/daemon/clients_wiring_test.go index d0e16ad4..d94c0fb8 100644 --- a/internal/daemon/clients_wiring_test.go +++ b/internal/daemon/clients_wiring_test.go @@ -202,6 +202,74 @@ func TestClientDispatch_CountChangeReachesOtherClients(t *testing.T) { } } +// attachTUIAs attaches the way a current TUI does: Cols/Rows are the pane +// interior (the window minus the chrome and the border, as attachMessage +// computes it) and WinCols/WinRows the raw window. +func attachTUIAs(t *testing.T, sock, id string, winCols, winRows int, reattach bool) *ipc.Client { + t.Helper() + c, err := ipc.NewClient(sock) + if err != nil { + t.Fatalf("dial: %v", err) + } + t.Cleanup(func() { c.Close() }) + sendClientMsg(t, c, ipc.MsgAttach, ipc.AttachPayload{ + ClientID: id, Cols: winCols - 2, Rows: winRows - 4, + WinCols: winCols, WinRows: winRows, Reattach: reattach, + }) + return c +} + +// Review finding M-2: eligibility is judged on the RAW window. An 80x12 TUI +// sends a 78x8 interior, below the 40x10 floor; judged on that it was never +// master, and after a lost link its reattach (which no client_geometry +// follows) handed the slot to the follower. list_clients reports the raw +// size too. +func TestClientDispatch_ReattachNearTheFloorKeepsTheSlot(t *testing.T) { + d, sock := overlayServerDaemon(t) + (&clientsHarness{t: t}).install(d, testGrace) + a := attachTUIAs(t, sock, "A", 80, 12, false) + // A TUI follows its first attach with client_geometry (its first + // WindowSizeMsg), which is what used to hide the interior-size bug here. + sendClientMsg(t, a, ipc.MsgClientGeometry, ipc.ClientGeometryPayload{Cols: 80, Rows: 12}) + waitUntil(t, "A master", func() bool { return d.masterID() == "A" }) + attachTUIAs(t, sock, "B", 120, 40, false) + waitUntil(t, "B registered", func() bool { return d.clientCount() == 2 }) + + a.Close() + waitUntil(t, "A's disconnect processed", func() bool { return d.clientCount() == 1 }) + attachTUIAs(t, sock, "A", 80, 12, true) + waitUntil(t, "A reattached", func() bool { return d.clientCount() == 2 }) + if got := d.masterID(); got != "A" { + t.Errorf("masterID = %q after A's reattach, want A (the returning master keeps its slot)", got) + } + + var sawA bool + for _, c := range d.listClients() { + if c.Client != "A" { + continue + } + sawA = true + if c.Cols != 80 || c.Rows != 12 || !c.Master { + t.Errorf("list_clients A = %dx%d master=%v, want 80x12 master=true (the raw window)", c.Cols, c.Rows, c.Master) + } + } + if !sawA { + t.Error("list_clients does not list A") + } +} + +// An older client sends no raw pair: its Cols/Rows are all it has, and they +// are used as they always were. +func TestAttachWindowSize_FallsBackToColsRowsForAnOlderClient(t *testing.T) { + d, sock := overlayServerDaemon(t) + attachClientAs(t, sock, "old", 100, 30) + waitUntil(t, "old registered", func() bool { return d.clientCount() == 1 }) + rec, ok := clientRecordByID(d, "old") + if !ok || rec.cols != 100 || rec.rows != 30 { + t.Errorf("record = %+v (found %v), want 100x30 from Cols/Rows", rec, ok) + } +} + // A lost link through the real server keeps the master's slot while a // follower is attached. func TestClientDispatch_LostLinkReservesSlot(t *testing.T) { diff --git a/internal/ipc/protocol.go b/internal/ipc/protocol.go index 8f4bb9ea..a3faf3c5 100644 --- a/internal/ipc/protocol.go +++ b/internal/ipc/protocol.go @@ -318,9 +318,20 @@ type Message struct { // Payload types type AttachPayload struct { + // Cols and Rows are the size to spawn the first pane of an empty + // workspace at: the pane INTERIOR (the window minus the sidebar, the + // chrome and the pane border), or 0x0 below the paintable floor. Cols int `json:"cols"` Rows int `json:"rows"` CWD string `json:"cwd,omitempty"` + // WinCols and WinRows are the RAW window size, exactly what + // MsgClientGeometry reports: 0x0 below the paintable floor (so both are + // omitted), the true window otherwise. Master eligibility and + // list_clients read these, never the interior above — an 80x12 window + // has a 78x8 interior, which is below the daemon's 40x10 floor. Absent + // from an older client, whose Cols/Rows the daemon falls back to. + WinCols int `json:"win_cols,omitempty"` + WinRows int `json:"win_rows,omitempty"` // ClientID identifies this client across reconnects, for multi-client // sync: master election, the client list and per-client geometry all key // on it. Empty on an older client: the daemon mints an "anon-" id diff --git a/internal/tui/model.go b/internal/tui/model.go index 2548dc94..c326e8ea 100644 --- a/internal/tui/model.go +++ b/internal/tui/model.go @@ -8074,11 +8074,21 @@ func (m Model) attachMessage(dest string, reattach bool) *ipc.Message { rows = 1 } } + // The RAW window as well, which is what the daemon elects a master on — + // the same pair clientGeometryCmd reports. The interior above is only the + // first pane's spawn size; electing on it made a paintable window near the + // floor ineligible, and on a reattach no client_geometry follows to fix it. + winCols, winRows := m.width, m.height + if !m.terminalPaintable() { + winCols, winRows = 0, 0 + } // Best-effort; if Getwd fails the daemon falls back to its own CWD. localCWD, _ := os.Getwd() msg, _ := ipc.NewMessage(ipc.MsgAttach, ipc.AttachPayload{ Cols: cols, Rows: rows, + WinCols: winCols, + WinRows: winRows, CWD: attachCWD(dest, localCWD), ClientID: m.clientID, Reattach: reattach, diff --git a/internal/tui/multiclient_role_test.go b/internal/tui/multiclient_role_test.go index 97475ebe..9a0a0c94 100644 --- a/internal/tui/multiclient_role_test.go +++ b/internal/tui/multiclient_role_test.go @@ -517,6 +517,54 @@ func TestAttach_ReattachOnlyAfterThisProcessAttachedThere(t *testing.T) { } } +// TestAttach_ReconnectCarriesTheRawWindow: the daemon elects on the RAW +// window, and the reconnect attach is the one no client_geometry follows — so +// it must carry the raw window itself (WinCols/WinRows), with Cols/Rows left +// as the pane interior the first pane spawns at. Review finding M-2: an 80x12 +// window sent only its 78x8 interior and was ineligible against the 40x10 +// floor on every reattach. +func TestAttach_ReconnectCarriesTheRawWindow(t *testing.T) { + t.Setenv("QUIL_HOME", t.TempDir()) + first, fresh := newFakeConn(), newFakeConn() + r := NewRouter(map[string]Client{"gpu01": first}) + var m tea.Model = Model{client: r, cfg: config.Default()} + + m, cmd := m.Update(tea.WindowSizeMsg{Width: 80, Height: 12}) + runCmdNoWait(cmd) + mm := m.(Model) + mm.links = map[string]*reconnectState{"gpu01": {active: true, gen: 1}} + _, cmd = mm.Update(redialResultMsg{gen: 1, dest: "gpu01", client: fresh}) + runCmdNoWait(cmd) + + for name, conn := range map[string]*fakeConn{"launch": first, "reconnect": fresh} { + got := attachPayloads(t, conn) + if len(got) != 1 { + t.Fatalf("%s attach count = %d, want 1", name, len(got)) + } + p := got[0] + if p.WinCols != 80 || p.WinRows != 12 { + t.Errorf("%s attach raw window = %dx%d, want 80x12", name, p.WinCols, p.WinRows) + } + if p.Cols >= p.WinCols || p.Rows >= p.WinRows || p.Cols < 1 || p.Rows < 1 { + t.Errorf("%s attach interior = %dx%d, want the pane interior inside the 80x12 window", name, p.Cols, p.Rows) + } + } +} + +// TestAttachMessage_NoRawWindowBelowTheFloor: an unpaintable window reports +// 0x0 raw, exactly as client_geometry does, so the daemon never elects it. +func TestAttachMessage_NoRawWindowBelowTheFloor(t *testing.T) { + t.Parallel() + m := Model{cfg: config.Default(), width: minTermWidth - 1, height: 40} + var p ipc.AttachPayload + if err := m.attachMessage("", false).DecodePayload(&p); err != nil { + t.Fatalf("decode attach payload: %v", err) + } + if p.WinCols != 0 || p.WinRows != 0 { + t.Errorf("raw window = %dx%d below the floor, want 0x0", p.WinCols, p.WinRows) + } +} + // TestAttach_FirstAttachThroughReconnectIsNotAReattach: a destination // unreachable at launch gets its FIRST attach from finishReconnect. The flag // is "has this process attached there before", not "which path sent it", so From 26d75d031122a3523c0c4131861a67d79b6e3f2c Mon Sep 17 00:00:00 2001 From: Artjoms Stukans Date: Sat, 26 Sep 2026 20:09:11 +0200 Subject: [PATCH 40/40] docs: correct the mouse-mode ordering note and a merged doc comment The mouse-mode note claimed the state frame always reaches the client before the pane chunk. sendLoop prefers the must-deliver queue only when both queues hold a frame at once, so an idle one usually writes the chunk first. State the real cost instead: enabling is unaffected, and disabling leaves a few-millisecond window in which a wheel notch can reach a program that just turned tracking off, accepted as far smaller than the existing 250 ms mouseModeBroadcastCooldown gap. Also move retireTypingGuard's doc comment off guardedInputTarget's, so each function keeps its own. Comment-only; no logic change. --- .claude/rules/daemon-lifecycle.md | 11 ++++++++--- internal/daemon/daemon.go | 13 ++++++++++--- internal/tui/model.go | 12 ++++++------ 3 files changed, 24 insertions(+), 12 deletions(-) diff --git a/.claude/rules/daemon-lifecycle.md b/.claude/rules/daemon-lifecycle.md index 737312f2..e3d26933 100644 --- a/.claude/rules/daemon-lifecycle.md +++ b/.claude/rules/daemon-lifecycle.md @@ -223,9 +223,14 @@ anywhere takes `holdGate` while holding a `PluginMu` (`beginOutputHold`, `finishOutputHold`, `dropOutputHold` and `handleAttach`'s replay span all keep the two apart). The gated span does no I/O and spawns nothing, which is why the mouse-mode `broadcastState` (decided inside the `PluginMu` span, sent -after the gate — the state frame rides the must-deliver queue, which -`sendLoop` drains ahead of pane output, so its enqueue order against this -chunk never decided delivery order) and the bell / hand-start / OSC 133 / +after the gate because it builds the whole workspace — `sendLoop` prefers the +must-deliver queue only when both queues hold a frame at once, so an idle one +usually writes the chunk BEFORE the state frame; enabling is unaffected since +the TUI's emulator sees `?1000h` in the chunk and tracking is local OR daemon, +while disabling leaves a few-millisecond window where the daemon flag is still +true after the chunk cleared the local one, so a wheel notch can type SGR +mouse escapes into a program that just turned tracking off — accepted as far +smaller than the existing 250 ms `mouseModeBroadcastCooldown` gap) and the bell / hand-start / OSC 133 / plugin detectors run AFTER it: the hand-start conversion restarts the pane, and a flush re-entered from there would ask for the read lock again behind a waiting writer. diff --git a/internal/daemon/daemon.go b/internal/daemon/daemon.go index a806e2a1..9711e3f1 100644 --- a/internal/daemon/daemon.go +++ b/internal/daemon/daemon.go @@ -4459,9 +4459,16 @@ func (d *Daemon) flushPaneOutputGeneration(paneID string, data []byte, generatio d.holdGate.RUnlock() // After the gate, not inside it: broadcastState builds the whole - // workspace, and the state frame travels on the must-deliver queue, which - // sendLoop drains ahead of pane output anyway — so enqueueing it before or - // after this chunk never decided which the client saw first. + // workspace, which must not be held under holdGate. The cost is order: + // sendLoop prefers the must-deliver queue only when both queues hold a + // frame at the same moment, so an idle sendLoop usually writes this chunk + // BEFORE the state frame. Enabling is unaffected — the TUI's own emulator + // sees ?1000h in the chunk, and tracking is local OR daemon. Disabling + // opens a window of a few milliseconds where the daemon flag is still true + // after the chunk cleared the local one, so a wheel notch then can type + // SGR mouse escapes into a program that just turned tracking off. + // Accepted: it is far smaller than the 250 ms mouseModeBroadcastCooldown + // gap that already exists. if doMouseBroadcast { logger.Debug("pane %s: mouse-mode change tracking=%v sgr=%v", paneID, newModes.tracking(), newModes.sgr) d.broadcastState() diff --git a/internal/tui/model.go b/internal/tui/model.go index c326e8ea..61dc314f 100644 --- a/internal/tui/model.go +++ b/internal/tui/model.go @@ -9011,12 +9011,6 @@ func (m Model) enqueueKeyInput(paneID string, data []byte) { m.enqueueInput(m.guardedInputTarget(paneID), data) } -// guardedInputTarget applies the typing guard (spec §8.1): within -// remoteSwitchGuardWindow of a remote tab switch, key-originated input still -// goes to guardPaneID — the pane the user was mid-keystroke in — rather than -// wherever the remote switch moved focus. It falls through to paneID once the -// window has elapsed or the guarded pane no longer exists (closed, moved, -// destroyed — there is nowhere left to redirect to). // retireTypingGuard ends the typing guard at once. The guard is for keys typed // straight through a remote switch; the moment this client's own user picks a // tab, a project or a pane, what they type next is meant for that choice, and @@ -9046,6 +9040,12 @@ func (m Model) localFocus() localFocusKey { return k } +// guardedInputTarget applies the typing guard (spec §8.1): within +// remoteSwitchGuardWindow of a remote tab switch, key-originated input still +// goes to guardPaneID — the pane the user was mid-keystroke in — rather than +// wherever the remote switch moved focus. It falls through to paneID once the +// window has elapsed or the guarded pane no longer exists (closed, moved, +// destroyed — there is nowhere left to redirect to). func (m Model) guardedInputTarget(paneID string) string { if m.guardPaneID == "" { return paneID