From 10cc649a671ee0e486c51a8dd79caa025bb765b9 Mon Sep 17 00:00:00 2001 From: CMGS Date: Mon, 14 Sep 2026 19:00:21 +0800 Subject: [PATCH 01/40] wire: scan past a tag's siblings without recursing skipValue recursed once per nesting level, so a guest that put a deeply nested value ahead of the type key could grow a host goroutine stack by megabytes per frame. The scan is now a flat loop with a small depth cap; wire values nest two levels at most. --- protocol/wire/frame.go | 31 ++++++++++++++++++++----------- protocol/wire/frame_test.go | 11 +++++++++++ 2 files changed, 31 insertions(+), 11 deletions(-) diff --git a/protocol/wire/frame.go b/protocol/wire/frame.go index 474235bf..19bbee67 100644 --- a/protocol/wire/frame.go +++ b/protocol/wire/frame.go @@ -27,6 +27,9 @@ const ( // PortWriteChunk keeps a port data frame (payload x4/3 base64 plus envelope) well under MaxFrame. PortWriteChunk = 1 << 20 + // maxSkipDepth bounds a tag scan past nested siblings; wire values nest two deep. + maxSkipDepth = 16 + // GitBranch.Action values (silkd's GitBranchOp). BranchList = "list" BranchCreate = "create" @@ -796,21 +799,27 @@ func scanTag(line []byte, key string) (string, error) { return "", nil } -// skipValue consumes one JSON value, recursing into containers. +// skipValue consumes one JSON value; a guest cannot grow the host stack with nesting. func skipValue(dec *json.Decoder) error { - tok, err := dec.Token() - if err != nil { - return err - } - if d, ok := tok.(json.Delim); ok && (d == '{' || d == '[') { - for dec.More() { - if err = skipValue(dec); err != nil { - return err + depth := 0 + for { + tok, err := dec.Token() + if err != nil { + return err + } + if d, ok := tok.(json.Delim); ok { + if d == '{' || d == '[' { + if depth++; depth > maxSkipDepth { + return fmt.Errorf("value nested deeper than %d", maxSkipDepth) + } + } else { + depth-- } } - _, err = dec.Token() + if depth == 0 { + return nil + } } - return err } func decodeAs[T any](line []byte) (*T, error) { diff --git a/protocol/wire/frame_test.go b/protocol/wire/frame_test.go index 19a7e47c..1a6076a5 100644 --- a/protocol/wire/frame_test.go +++ b/protocol/wire/frame_test.go @@ -260,6 +260,17 @@ func TestTagAfterOtherKeys(t *testing.T) { } } +func TestTagScanBoundsNesting(t *testing.T) { + shallow := `{"x":` + strings.Repeat("[", maxSkipDepth) + strings.Repeat("]", maxSkipDepth) + `,"type":"done"}` + if _, err := DecodeResponse([]byte(shallow)); err != nil { + t.Fatalf("nesting at the cap rejected: %v", err) + } + deep := `{"x":` + strings.Repeat("[", 1<<20) + strings.Repeat("]", 1<<20) + `,"type":"done"}` + if _, err := DecodeResponse([]byte(deep)); err == nil || !strings.Contains(err.Error(), "nested deeper") { + t.Fatalf("deep nesting: err %v, want the depth cap", err) + } +} + func TestUnknownTagsRejected(t *testing.T) { if _, err := DecodeRequest([]byte(`{"v":1,"op":"teleport"}`)); err == nil { t.Error("unknown op accepted") From 3eac2a920f4d18459a0824c78d34788fbfb5c5a9 Mon Sep 17 00:00:00 2001 From: CMGS Date: Mon, 14 Sep 2026 19:02:17 +0800 Subject: [PATCH 02/40] mcp: keep exec output when the call fails, cap it, and reject a negative ttl A cut-off or dropped exec answered with the bare error and threw away the output the command had produced; the agent now gets the collected stdout and stderr next to the error field. Each stream keeps its first 1 MiB and flags truncated past that, so a firehose cannot grow the process. A negative ttl_seconds used to reach the node, which clamps it to its own 5-minute default; it is now a bad-argument error. --- mcp/server.go | 4 ++- mcp/server_test.go | 64 ++++++++++++++++++++++++++++++++++++++++++++++ mcp/tools.go | 35 ++++++++++++++++++++----- 3 files changed, 96 insertions(+), 7 deletions(-) diff --git a/mcp/server.go b/mcp/server.go index d50b3330..67fe2280 100644 --- a/mcp/server.go +++ b/mcp/server.go @@ -2,6 +2,7 @@ package main import ( "bufio" + "cmp" "context" "encoding/json" "fmt" @@ -17,6 +18,7 @@ import ( const ( protocolVersion = "2024-11-05" execTimeout = 5 * time.Minute + execOutputCap = 1 << 20 defaultToolTTL = time.Hour ) @@ -119,7 +121,7 @@ func (s *server) dispatch(ctx context.Context, req *rpcRequest) rpcResponse { text, err := s.callTool(ctx, call.Name, call.Arguments) if err != nil { return result(req.ID, map[string]any{ - "content": []map[string]any{{"type": "text", "text": err.Error()}}, + "content": []map[string]any{{"type": "text", "text": cmp.Or(text, err.Error())}}, "isError": true, }) } diff --git a/mcp/server_test.go b/mcp/server_test.go index 4860bef7..8fde8e55 100644 --- a/mcp/server_test.go +++ b/mcp/server_test.go @@ -5,6 +5,7 @@ import ( "bytes" "encoding/json" "fmt" + "io" "net/http" "net/http/httptest" "strings" @@ -91,6 +92,69 @@ func TestServeSpeaksMCP(t *testing.T) { } } +func TestExecKeepsOutputWhenTheGuestDrops(t *testing.T) { + node := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + switch { + case r.URL.Path == "/v1/claim": + _ = json.NewEncoder(w).Encode(map[string]any{"id": "sb_1", "token": "tok"}) + case strings.HasSuffix(r.URL.Path, "/agent"): + conn, _, err := http.NewResponseController(w).Hijack() + if err != nil { + t.Errorf("hijack: %v", err) + return + } + _, _ = io.WriteString(conn, "HTTP/1.1 101 Switching Protocols\r\nUpgrade: silkd\r\nConnection: Upgrade\r\n\r\n"+ + `{"type":"started","pid":7}`+"\n"+`{"type":"stdout","data":"cGFydGlhbA=="}`+"\n") + _ = conn.Close() + default: + http.Error(w, `{"error":"no route"}`, http.StatusNotFound) + } + })) + t.Cleanup(node.Close) + srv, err := newServer(strings.TrimPrefix(node.URL, "http://"), "", "rt:24.04") + if err != nil { + t.Fatalf("newServer: %v", err) + } + lines := `{"jsonrpc":"2.0","id":1,"method":"tools/call","params":{"name":"create_sandbox","arguments":{}}} +{"jsonrpc":"2.0","id":2,"method":"tools/call","params":{"name":"exec","arguments":{"sandbox_id":"sb_1","command":"yes"}}} +{"jsonrpc":"2.0","id":3,"method":"tools/call","params":{"name":"create_sandbox","arguments":{"ttl_seconds":-5}}} +` + var out bytes.Buffer + if err := srv.serve(t.Context(), bufio.NewReader(strings.NewReader(lines)), &out); err != nil { + t.Fatalf("serve: %v", err) + } + var replies []map[string]any + dec := json.NewDecoder(&out) + for dec.More() { + var resp map[string]any + if err := dec.Decode(&resp); err != nil { + t.Fatalf("decode reply: %v", err) + } + replies = append(replies, resp) + } + execResult := replies[1]["result"].(map[string]any) + text := toolText(t, replies[1]) + if execResult["isError"] != true || !strings.Contains(text, `"stdout":"partial"`) || !strings.Contains(text, `"error"`) { + t.Errorf("exec reply %v: want isError with the partial stdout and an error field", text) + } + if !strings.Contains(toolText(t, replies[2]), "ttl_seconds must not be negative") { + t.Errorf("negative ttl accepted: %q", toolText(t, replies[2])) + } +} + +func TestCappedOutputMarksTruncation(t *testing.T) { + var o cappedOutput + chunk := bytes.Repeat([]byte("x"), execOutputCap/2+1) + for range 3 { + if n, err := o.Write(chunk); n != len(chunk) || err != nil { + t.Fatalf("Write = %d, %v; want the full length accepted", n, err) + } + } + if o.Len() != execOutputCap || !o.truncated { + t.Errorf("kept %d bytes, truncated=%v; want the cap and the flag", o.Len(), o.truncated) + } +} + func toolText(t *testing.T, resp map[string]any) string { t.Helper() res, ok := resp["result"].(map[string]any) diff --git a/mcp/tools.go b/mcp/tools.go index 8666d272..d19dfb40 100644 --- a/mcp/tools.go +++ b/mcp/tools.go @@ -17,7 +17,7 @@ var tools = []tool{ schema(props{"template": str("template image ref, or a name published by promote; empty uses the server default"), "net": str("network lane: none (default, no NIC, vsock-only I/O) or egress (bridge NIC, outbound network)"), "size": str("resource tier: small (default), medium, large, xlarge, 2xlarge"), "ttl_seconds": integer("sandbox lifetime in seconds; 0 means one hour, and nothing renews it")}), toolCreateSandbox, }, { - "exec", "Run a shell command in a sandbox, wait for it to exit, and return stdout, stderr, and the exit code as JSON. The call is cut off after 5 minutes; for servers or long jobs use spawn instead. A hibernated sandbox wakes transparently on this call.", + "exec", "Run a shell command in a sandbox, wait for it to exit, and return stdout, stderr, and the exit code as JSON. The call is cut off after 5 minutes; for servers or long jobs use spawn instead. A failed or cut-off call still returns the output collected so far next to an error field. Each stream keeps its first 1 MiB and sets truncated beyond that. A hibernated sandbox wakes transparently on this call.", schema(props{"sandbox_id": str("id returned by create_sandbox, fork, or branch_checkpoint"), "command": str("shell command, run via sh -c"), "cwd": str("working directory; empty runs in the guest's default")}, "sandbox_id", "command"), toolExec, }, { @@ -101,8 +101,10 @@ func toolCreateSandbox(ctx context.Context, s *server, raw json.RawMessage) (str if err := parse(raw, &args); err != nil { return "", err } - // An agent session outlives the node's 5-minute default and nothing renews - // a lease, so the sandbox would vanish mid-conversation. + if args.TTLSeconds < 0 { + return "", fmt.Errorf("ttl_seconds must not be negative, got %d", args.TTLSeconds) + } + // an agent session outlives the node's 5-minute default, and nothing renews a lease ttl := cmp.Or(time.Duration(args.TTLSeconds)*time.Second, defaultToolTTL) opts := []sandbox.Option{sandbox.WithTimeout(ttl)} if args.Net != "" { @@ -144,20 +146,41 @@ type cmdArgs struct { Cwd string `json:"cwd"` } +// cappedOutput keeps the first execOutputCap bytes; an agent that tails a firehose must not grow this process. +type cappedOutput struct { + strings.Builder + truncated bool +} + +func (o *cappedOutput) Write(p []byte) (int, error) { + n := len(p) + if room := execOutputCap - o.Len(); len(p) > room { + p, o.truncated = p[:room], true + } + _, _ = o.Builder.Write(p) + return n, nil +} + func toolExec(ctx context.Context, s *server, raw json.RawMessage) (string, error) { args, sb, err := parseAndBox[cmdArgs](s, raw) if err != nil { return "", err } - var stdout, stderr strings.Builder + var stdout, stderr cappedOutput code, err := sb.Run(ctx, sandbox.Cmd{ Argv: []string{"sh", "-c", args.Command}, Cwd: args.Cwd, Stdout: &stdout, Stderr: &stderr, }) + out := map[string]any{"stdout": stdout.String(), "stderr": stderr.String()} + if stdout.truncated || stderr.truncated { + out["truncated"] = true + } if err != nil { - return "", err + out["error"] = err.Error() + } else { + out["exit_code"] = code } - return jsonText(map[string]any{"exit_code": code, "stdout": stdout.String(), "stderr": stderr.String()}), nil + return jsonText(out), err } func toolSpawn(ctx context.Context, s *server, raw json.RawMessage) (string, error) { From 0a485c23f2125a1a98c0ec44d744f3b3b7f0657e Mon Sep 17 00:00:00 2001 From: CMGS Date: Mon, 14 Sep 2026 19:05:04 +0800 Subject: [PATCH 03/40] sdk/go: release the relay when a watch ends on its own drain returned on the sandbox's error frame or a dropped connection without running the dial's cleanup, so the relay connection and its context hook lived until the caller's ctx ended; only Close released them. --- sdk/go/watch.go | 1 + sdk/go/watch_test.go | 36 ++++++++++++++++++++++++++++++++++++ 2 files changed, 37 insertions(+) diff --git a/sdk/go/watch.go b/sdk/go/watch.go index dc230f8a..a0347325 100644 --- a/sdk/go/watch.go +++ b/sdk/go/watch.go @@ -44,6 +44,7 @@ func (w *Watcher) Err() error { // closed channel keeps it from blocking on a consumer that stopped reading. func (w *Watcher) drain(ctx context.Context, conn *silkd.Conn) { defer close(w.events) + defer w.stop() for { resp, err := recv(ctx, conn) if err != nil { diff --git a/sdk/go/watch_test.go b/sdk/go/watch_test.go index a87229e7..d43b4e8d 100644 --- a/sdk/go/watch_test.go +++ b/sdk/go/watch_test.go @@ -1,9 +1,13 @@ package sandbox import ( + "bufio" "errors" + "io" + "net" "strings" "testing" + "time" "github.com/cocoonstack/sandbox/protocol/wire" ) @@ -37,6 +41,38 @@ func TestWatchDeliversEventsUntilClose(t *testing.T) { } } +func TestWatchReleasesTheRelayWhenTheSandboxEnds(t *testing.T) { + released := make(chan struct{}) + ts := newAgentServer(t, func(conn net.Conn) { + defer conn.Close() + r := bufio.NewReader(conn) + if _, err := r.ReadString('\n'); err != nil { + t.Errorf("read watch request: %v", err) + return + } + _, _ = io.WriteString(conn, `{"type":"ready"}`+"\n"+ + `{"type":"event","kind":"created","path":"/work/a"}`+"\n"+ + `{"type":"error","kind":"internal","message":"watcher died"}`+"\n") + _, _ = r.ReadByte() + close(released) + }) + sb := testSandbox(t, ts) + w, err := sb.Watch(t.Context(), "/work", true) + if err != nil { + t.Fatalf("Watch: %v", err) + } + for range w.Events() { + } + if e, ok := errors.AsType[*wire.ErrorResp](w.Err()); !ok || e.Message != "watcher died" { + t.Fatalf("Err = %v, want the sandbox's error frame", w.Err()) + } + select { + case <-released: + case <-time.After(2 * time.Second): + t.Fatal("relay connection still open after the watch ended on its own") + } +} + func TestWatchMissingPathFailsSynchronously(t *testing.T) { sb := fakeSandbox(t) _, err := sb.Watch(t.Context(), "/nope", false) From 0652d6e9aae02b409a4fd1b36477d8922aac2f22 Mon Sep 17 00:00:00 2001 From: CMGS Date: Mon, 14 Sep 2026 19:07:41 +0800 Subject: [PATCH 04/40] sdk/python: enforce the exec cutoff inside the guest Since the stream socket lost its timeout (#178), the openai adapter's wait_for could cancel the awaiting coroutine while the worker thread stayed blocked on the relay, and the langchain tool's 5-minute promise had nothing behind it. Both now run the command under timeout(1) -s KILL, which kills the process group at the deadline; the adapter maps exit 124 to the TimeoutError the Agents SDK documents, the tool reports it as exit code 124. --- docs/langchain.md | 2 +- .../cocoonsandbox_langchain/toolkit.py | 9 +++--- sdk/langchain/tests/test_toolkit.py | 6 ++-- sdk/openai/cocoonsandbox_openai/adapter.py | 18 +++++++---- sdk/openai/tests/test_adapter.py | 31 ++++++++++++++++--- 5 files changed, 48 insertions(+), 18 deletions(-) diff --git a/docs/langchain.md b/docs/langchain.md index 112a1cbd..60ce0753 100644 --- a/docs/langchain.md +++ b/docs/langchain.md @@ -19,7 +19,7 @@ schemas, sync-native with `asyncio.to_thread` async bridges): | tool | what it does | |---|---| -| `sandbox_exec` | run a shell command, cut off after 5 minutes without output; stdout/stderr/exit code; disk state persists across calls | +| `sandbox_exec` | run a shell command, cut off after 5 minutes (exit code 124); stdout/stderr/exit code; disk state persists across calls | | `sandbox_write_file` | write a text file (atomic on the guest) | | `sandbox_read_file` | read a text file | | `sandbox_list_dir` | list a directory as JSON | diff --git a/sdk/langchain/cocoonsandbox_langchain/toolkit.py b/sdk/langchain/cocoonsandbox_langchain/toolkit.py index 8c279f14..27fafa76 100644 --- a/sdk/langchain/cocoonsandbox_langchain/toolkit.py +++ b/sdk/langchain/cocoonsandbox_langchain/toolkit.py @@ -14,8 +14,8 @@ from langchain_core.tools import StructuredTool from pydantic import BaseModel, Field -# a socket inactivity bound, not a wall clock; the tool description states it -CALL_TIMEOUT = 300.0 +# one exec's wall clock, enforced by the guest's timeout(1); the tool description states it +CALL_TIMEOUT = 300 class ExecInput(BaseModel): @@ -73,7 +73,7 @@ def get_tools(self) -> list[StructuredTool]: "Returns stdout; a non-empty stderr is appended as a 'stderr:' " "line and a non-zero status as an 'exit code: N' line; a " "command that prints nothing and exits 0 returns '(no output)'. " - "The call is cut off after 5 minutes without output. " + "The call is cut off after 5 minutes (exit code 124). " "Files and installed packages persist across calls; environment " "variables and the working directory do not.", ExecInput, @@ -138,7 +138,8 @@ async def arun(**kwargs): def _exec(self, command: str, cwd: str = "") -> str: out: list[bytes] = [] errs: list[bytes] = [] - code = self.sandbox().run(["sh", "-c", command], cwd=cwd, on_stdout=out.append, on_stderr=errs.append) + argv = ["timeout", "-s", "KILL", str(CALL_TIMEOUT), "sh", "-c", command] + code = self.sandbox().run(argv, cwd=cwd, on_stdout=out.append, on_stderr=errs.append) stdout = b"".join(out).decode(errors="replace") stderr = b"".join(errs).decode(errors="replace") result = stdout diff --git a/sdk/langchain/tests/test_toolkit.py b/sdk/langchain/tests/test_toolkit.py index ab604229..a2f03732 100644 --- a/sdk/langchain/tests/test_toolkit.py +++ b/sdk/langchain/tests/test_toolkit.py @@ -61,11 +61,11 @@ def __init__(self): self.files = {} def run(self, argv, cwd="", on_stdout=None, on_stderr=None, **_): - assert argv[:2] == ["sh", "-c"] - if argv[2] == "boom": + assert argv[:6] == ["timeout", "-s", "KILL", "300", "sh", "-c"], argv + if argv[6] == "boom": on_stderr(b"kaboom\n") return 3 - on_stdout(f"ran: {argv[2]}\n".encode()) + on_stdout(f"ran: {argv[6]}\n".encode()) return 0 def write_file(self, path, data): diff --git a/sdk/openai/cocoonsandbox_openai/adapter.py b/sdk/openai/cocoonsandbox_openai/adapter.py index 989aa467..89333f34 100644 --- a/sdk/openai/cocoonsandbox_openai/adapter.py +++ b/sdk/openai/cocoonsandbox_openai/adapter.py @@ -21,6 +21,9 @@ from agents.sandbox.types import ExecResult, ExposedPortEndpoint, User from cocoonsandbox import Client, Sandbox, SandboxError, SilkdError +# timeout(1) answers 124 when the command it ran was cut off +TIMEOUT_EXIT = 124 + class CocoonSandboxClientOptions(BaseSandboxClientOptions): """Connection settings for a sandboxd node (or cluster entry node).""" @@ -62,15 +65,19 @@ async def _prepare_backend_workspace(self) -> None: await asyncio.to_thread(sb.mkdir, str(self.state.manifest.root), parents=True) async def _exec_internal(self, *command: str | Path, timeout: float | None = None) -> ExecResult: - # the socket timeout bounds the blocking SDK call too, so a wait_for cancellation unblocks the worker thread - sb = self._sandbox(timeout=timeout) + sb = self._sandbox() argv = [str(part) for part in command] + if timeout is not None: + # the guest enforces the cutoff: a stream has no socket timeout, so a cancelled wait strands the worker + argv = ["timeout", "-s", "KILL", str(timeout), *argv] stdout, stderr = bytearray(), bytearray() def run() -> int: return sb.run(argv, on_stdout=stdout.extend, on_stderr=stderr.extend) - code = await asyncio.wait_for(asyncio.to_thread(run), timeout=timeout) + code = await asyncio.to_thread(run) + if timeout is not None and code == TIMEOUT_EXIT: + raise TimeoutError(f"command did not finish within {timeout}s") return ExecResult(stdout=bytes(stdout), stderr=bytes(stderr), exit_code=code) async def read(self, path: Path, *, user: str | User | None = None) -> io.IOBase: @@ -118,10 +125,9 @@ async def _shutdown_backend(self) -> None: listener.close() self._proxies.clear() - def _sandbox(self, timeout: float | None = None) -> Sandbox: + def _sandbox(self) -> Sandbox: s = self.state - kwargs = {"timeout": timeout} if timeout is not None else {} - client = Client(s.addr, api_token=s.api_token, **kwargs) + client = Client(s.addr, api_token=s.api_token) return Sandbox(client=client, id=s.sandbox_id, token=s.sandbox_token, owner=s.owner or s.addr) def _abs(self, path: Path | str) -> Path: diff --git a/sdk/openai/tests/test_adapter.py b/sdk/openai/tests/test_adapter.py index ccf3ca9c..f8be4242 100644 --- a/sdk/openai/tests/test_adapter.py +++ b/sdk/openai/tests/test_adapter.py @@ -81,7 +81,7 @@ async def go(): client = CocoonSandboxClient() session = await client.create(options=CocoonSandboxClientOptions(addr=node)) inner = session._inner - monkeypatch.setattr(inner, "_sandbox", lambda timeout=None: FakeSandbox()) + monkeypatch.setattr(inner, "_sandbox", lambda: FakeSandbox()) result = await inner._exec_internal("echo", "hi") assert result.exit_code == 0 and result.stdout == b"hi\n" assert result.ok() @@ -89,6 +89,29 @@ async def go(): asyncio.run(go()) +def test_exec_timeout_runs_under_guest_timeout(node, monkeypatch): + seen = [] + + class FakeSandbox: + def run(self, argv, on_stdout=None, on_stderr=None): + seen.append(argv) + on_stdout(b"partial\n") + return 124 + + async def go(): + client = CocoonSandboxClient() + session = await client.create(options=CocoonSandboxClientOptions(addr=node)) + inner = session._inner + monkeypatch.setattr(inner, "_sandbox", lambda: FakeSandbox()) + with pytest.raises(TimeoutError): + await inner._exec_internal("sleep", "9", timeout=1.5) + assert seen == [["timeout", "-s", "KILL", "1.5", "sleep", "9"]] + result = await inner._exec_internal("sleep", "9") + assert seen[-1] == ["sleep", "9"] and result.exit_code == 124 + + asyncio.run(go()) + + def test_read_missing_maps_to_filenotfound(node, monkeypatch): class FakeSandbox: def read_file(self, path): @@ -98,7 +121,7 @@ async def go(): client = CocoonSandboxClient() session = await client.create(options=CocoonSandboxClientOptions(addr=node)) inner = session._inner - monkeypatch.setattr(inner, "_sandbox", lambda timeout=None: FakeSandbox()) + monkeypatch.setattr(inner, "_sandbox", lambda: FakeSandbox()) with pytest.raises(FileNotFoundError): await inner.read(Path("/nope")) @@ -114,7 +137,7 @@ async def go(): client = CocoonSandboxClient() session = await client.create(options=CocoonSandboxClientOptions(addr=node)) inner = session._inner - monkeypatch.setattr(inner, "_sandbox", lambda timeout=None: FakeSandbox()) + monkeypatch.setattr(inner, "_sandbox", lambda: FakeSandbox()) assert await inner.running() is False asyncio.run(go()) @@ -138,7 +161,7 @@ async def go(): client = CocoonSandboxClient() session = await client.create(options=CocoonSandboxClientOptions(addr=node)) inner = session._inner - monkeypatch.setattr(inner, "_sandbox", lambda timeout=None: FakeSandbox()) + monkeypatch.setattr(inner, "_sandbox", lambda: FakeSandbox()) await inner.write(Path("/workspace/a.txt"), io.BytesIO(b"body")) assert calls["write"] == ("/workspace/a.txt", b"body") tar = await inner.persist_workspace() From 3b42d68f05f3013fc8d94dbe18dabaf34d0b2a97 Mon Sep 17 00:00:00 2001 From: CMGS Date: Mon, 14 Sep 2026 19:09:11 +0800 Subject: [PATCH 05/40] config: keep an explicit warm 0 under warm_max; refuse a wildcard advertise_addr on a mesh warm defaulted through cmp.Or, so a pool written as warm 0 with a warm_max started at 4 instead of empty; the default now applies only when neither is set. advertise_addr defaulted to the listen address even when that was a wildcard, and a mesh node gossiped 0.0.0.0 or an empty host as its owner address to peers; a mesh config now fails to load until it names a routable host. --- docs/deploy.md | 4 ++-- sandboxd/config/config.go | 17 ++++++++++++++--- sandboxd/config/config_test.go | 17 +++++++++++++++++ 3 files changed, 33 insertions(+), 5 deletions(-) diff --git a/docs/deploy.md b/docs/deploy.md index 37d08780..59469bdc 100644 --- a/docs/deploy.md +++ b/docs/deploy.md @@ -99,7 +99,7 @@ sandboxd reads one JSON file (`-config`, default | `restore_mode` | unset | clone and wake-restore memory mode: `copy`, `ondemand`, or `mmap`; use `mmap` for dense pools | | `no_direct_io` | false | use buffered writable disks for Cloud Hypervisor cold boots and clones; recommended for dense ephemeral pools to avoid direct-I/O CoW journal contention | | `no_balloon` | false | boot pool and template VMs without the virtio-balloon (cocoon otherwise returns 25% of guest memory to the host); clones inherit it from the golden. A guest that thrashes before deflate-on-OOM fires — a 16G build tier running a large typecheck — needs its whole memory | -| `advertise_addr` | = `listen` | the host:port clients reach this node at; returned as a claim's owner address and gossiped to peers. Must be routable when `listen` is a wildcard | +| `advertise_addr` | = `listen` | the host:port clients reach this node at; returned as a claim's owner address and gossiped to peers. Must be routable when `listen` is a wildcard; a node with `mesh` set refuses to load while it names an unspecified host | | `bridges` / `networks` | unset | egress-lane attachment: a list of host bridge devices, or a list of CNI conflist names. Mutually exclusive; with neither set the node serves only the no-network lane. A Linux bridge holds at most 1024 ports (kernel `BR_MAX_PORTS`), so an N-entry list raises the node's egress ceiling to N×1024 — VMs spread over the list by a stable hash of the VM name, so size it with headroom (the spread is statistical, not exact). `bridges` keeps the raw TAP-on-bridge attachment (taps in the root netns, no per-VM network namespace or CNI plugin execution); `networks` runs the CNI chain per VM. [Guarded egress](egress.md) on the egress lane (an egress-lane pool policy or any tenant policy) needs `bridges` and rejects a CNI network at load; none-lane pool policies ride the proxy on either | | `volumes` | unset | node-local catalog of operator-managed dataset images: `[ {"name":"imagenet","path":"/srv/datasets/imagenet.img","directio":"off","tenants":["acme"]}, {"name":"scratch-db","path":"/srv/datasets/scratch.img","writable":true} ]`. Names match `^[a-z][a-z0-9_-]{0,19}$` and cannot start with `cocoon-`; paths are absolute; `directio` is `on`, `off`, or `auto` and defaults to `off` for both read-only and writable entries. `tenants` is an optional access list: empty means every authenticated scope, while every listed name must exist in the node's `tenants` config; root always has access. `writable` (default `false`) lets a claim request `mode: "rw"` on that entry — see [Dataset volumes](#dataset-volumes). The catalog is intentionally not part of the cluster digest | | `secrets` | unset | node-side credentials the egress proxy injects by name: `[{"name": "gh", "header": "Authorization", "value_env": "GH_TOKEN"}]`. A pool or tenant rule references the name; the value comes from the environment, never this file. See [egress](egress.md) | @@ -125,7 +125,7 @@ sandboxd reads one JSON file (`-config`, default | `archive_after_seconds` | 0 (off) | tier below hibernation: a hibernated claim idle this long is checkpointed to the store and its local VM dropped, freeing the node entirely; the next call that reaches the guest restores it transparently (a checkpoint restore's latency) with a fresh server-default 5m lease. Requires `idle_hibernate_seconds > 0` and must exceed it. Node-wide for unpooled keys; per-pool overrides for that pool | | `archive_delete_after_seconds` | 0 (keep) | purge an archived claim's store checkpoint this long after it was archived, reclaiming storage; the claim is then gone for good. On archive this retention window replaces the live claim deadline; 0 clears the deadline so the archive is kept forever. Same node-wide/per-pool split | | `mesh` | unset | join a cluster ([Clusters](cluster.md)); unset = single node | -| `pools[]` | — | warm pools, keyed by `(template, net, size)`. `warm` defaults to 4; `net` is `none` or `egress`; `size` is a tier, below. Retune online without a restart via [`PUT /v1/pools`](sandboxd-api.md#put-v1pools) — omitted pools drain. This is the **first-boot seed**: once a node takes a `PUT /v1/pools`, the applied set persists to `/pools.json` and overrides this section on every later boot (a startup log notes it); delete `pools.json` to return to config-owned pools. Egress stays config-owned either way. See [state ownership](cluster.md#state-ownership) | +| `pools[]` | — | warm pools, keyed by `(template, net, size)`. `warm` defaults to 4 unless `warm_max` is set, so `warm: 0` under a `warm_max` starts the pool empty and grows it on demand; `net` is `none` or `egress`; `size` is a tier, below. Retune online without a restart via [`PUT /v1/pools`](sandboxd-api.md#put-v1pools) — omitted pools drain. This is the **first-boot seed**: once a node takes a `PUT /v1/pools`, the applied set persists to `/pools.json` and overrides this section on every later boot (a startup log notes it); delete `pools.json` to return to config-owned pools. Egress stays config-owned either way. See [state ownership](cluster.md#state-ownership) | Size tiers (free-form CPU/memory is deliberately not accepted — it would fragment the warm pools): diff --git a/sandboxd/config/config.go b/sandboxd/config/config.go index 7f6f861b..d6ba2154 100644 --- a/sandboxd/config/config.go +++ b/sandboxd/config/config.go @@ -291,7 +291,9 @@ func (c *Config) applyDefaults() { c.RefillConcurrency = autoRefillConcurrency(runtime.NumCPU()) } for i := range c.Pools { - c.Pools[i].Warm = cmp.Or(c.Pools[i].Warm, defaultWarm) + if c.Pools[i].Warm == 0 && c.Pools[i].WarmMax == 0 { + c.Pools[i].Warm = defaultWarm + } c.Pools[i].PoolKey = c.Pools[i].Defaulted() } for i := range c.Volumes { @@ -409,8 +411,17 @@ func (c *Config) validateMesh() error { if _, _, err := c.Mesh.ParsedBind(); err != nil { return err } - _, err := c.Mesh.DecodedKey() - return err + if _, err := c.Mesh.DecodedKey(); err != nil { + return err + } + host, _, err := net.SplitHostPort(c.AdvertiseAddr) + if err != nil { + return fmt.Errorf("advertise_addr: %w", err) + } + if ip, _ := netip.ParseAddr(host); host == "" || ip.IsUnspecified() { + return fmt.Errorf("advertise_addr %q is gossiped to peers and must name a routable host", c.AdvertiseAddr) + } + return nil } func (c *Config) validateEgress(secrets map[string]struct{}) error { diff --git a/sandboxd/config/config_test.go b/sandboxd/config/config_test.go index a4cf391a..016d4b79 100644 --- a/sandboxd/config/config_test.go +++ b/sandboxd/config/config_test.go @@ -126,6 +126,8 @@ func TestLoadRejectsInvalid(t *testing.T) { {"guarded egress on cni pool", `{"networks":["cni"],"pools":[{"template":"rt:24.04","net":"egress","size":"small","egress":{"allow":[{"host":"x"}]}}]}`, "needs a bridge lane"}, {"guarded egress on cni tenant", `{"api_token":"root","networks":["cni"],"pools":[{"template":"rt:24.04","net":"egress","size":"small"}],"tenants":[{"name":"acme","token":"t1","egress":{"allow":[{"host":"x"}]}}]}`, "needs a bridge lane"}, {"cni tenant egress no egress pool", `{"api_token":"root","networks":["cni"],"pools":[{"template":"rt:24.04","net":"none","size":"small"}],"tenants":[{"name":"acme","token":"t1","egress":{"allow":[{"host":"x"}]}}]}`, "needs a bridge lane"}, + {"mesh with wildcard advertise", `{"listen":":7777","pools":[],"mesh":{"bind":"node1:7946"}}`, "routable host"}, + {"mesh with unspecified advertise", `{"advertise_addr":"0.0.0.0:7777","pools":[],"mesh":{"bind":"node1:7946"}}`, "routable host"}, {"mesh bind missing port", `{"pools":[],"mesh":{"bind":"node1"}}`, "mesh bind"}, {"mesh bind wildcard host", `{"pools":[],"mesh":{"bind":":7946"}}`, "explicit host"}, {"mesh cluster key not base64", `{"pools":[],"mesh":{"bind":"node1:7946","cluster_key":"not!base64"}}`, "not valid base64"}, @@ -262,6 +264,21 @@ func TestHasEgress(t *testing.T) { } } +func TestLoadKeepsWarmZeroUnderWarmMax(t *testing.T) { + path := writeConfig(t, `{"pools":[{"template":"rt:24.04","net":"none","size":"small","warm":0,"warm_max":8}, + {"template":"rt:24.04","net":"none","size":"medium","warm":0}]}`) + cfg, err := Load(path) + if err != nil { + t.Fatalf("Load: %v", err) + } + if cfg.Pools[0].Warm != 0 || cfg.Pools[0].WarmMax != 8 { + t.Errorf("adaptive pool %+v, want warm 0 kept under warm_max 8", cfg.Pools[0]) + } + if cfg.Pools[1].Warm != defaultWarm { + t.Errorf("static pool warm %d, want the default %d", cfg.Pools[1].Warm, defaultWarm) + } +} + func TestLoadKeepsExplicitValues(t *testing.T) { path := writeConfig(t, `{"listen":"0.0.0.0:9999","advertise_addr":"10.0.0.5:9999","max_fork_count":4, "refill_concurrency":8,"no_direct_io":true,"bridges":["br0"],"pools":[{"template":"rt:24.04","net":"egress","size":"small","warm":3}]}`) From a7a1b550b578354a1c210ea8807166398aebc7ad Mon Sep 17 00:00:00 2001 From: CMGS Date: Mon, 14 Sep 2026 19:11:43 +0800 Subject: [PATCH 06/40] server: stop a preview forward after one hop, keep encoded slashes, end WriteTo cleanly A preview token names its owner by advertise_addr; after that address changed, the node forwarded the request to its old name, which routed back to itself and forwarded again until the connections ran out. The forward now marks the request and a marked request that still misses answers 502. proxyLocal rewrote only URL.Path and left the inbound RawPath behind, so a %2F in the guest path reached the guest decoded. auditTee.WriteTo returned io.EOF where io.WriterTo promises nil, and collectExec now accumulates into builders instead of copying the whole output once more into strings. --- sandboxd/server/exec.go | 13 ++++++----- sandboxd/server/preview.go | 24 +++++++++++++++----- sandboxd/server/preview_test.go | 39 +++++++++++++++++++++++++++++++++ sandboxd/server/relay.go | 3 +++ sandboxd/server/relay_test.go | 11 ++++++++++ 5 files changed, 79 insertions(+), 11 deletions(-) diff --git a/sandboxd/server/exec.go b/sandboxd/server/exec.go index a9416fad..298ffd90 100644 --- a/sandboxd/server/exec.go +++ b/sandboxd/server/exec.go @@ -7,6 +7,7 @@ import ( "io" "net" "net/http" + "strings" "time" "github.com/projecteru2/core/log" @@ -145,7 +146,7 @@ func (s *Server) killExec(ctx context.Context, id, token string, pid uint32) err } func collectExec(guest net.Conn) (resp ExecResponse, pid uint32, err error) { - var stdout, stderr []byte + var stdout, stderr strings.Builder sc := wire.NewFrameScanner(guest) for sc.Scan() { frame, err := wire.DecodeResponse(sc.Bytes()) @@ -156,17 +157,17 @@ func collectExec(guest net.Conn) (resp ExecResponse, pid uint32, err error) { case *wire.Started: pid = f.PID case *wire.Stdout: - if len(stdout)+len(stderr)+len(f.Data) > execOutputCap { + if stdout.Len()+stderr.Len()+len(f.Data) > execOutputCap { return ExecResponse{}, pid, errExecOutputCap } - stdout = append(stdout, f.Data...) + stdout.Write(f.Data) case *wire.Stderr: - if len(stdout)+len(stderr)+len(f.Data) > execOutputCap { + if stdout.Len()+stderr.Len()+len(f.Data) > execOutputCap { return ExecResponse{}, pid, errExecOutputCap } - stderr = append(stderr, f.Data...) + stderr.Write(f.Data) case *wire.Exit: - return ExecResponse{ExitCode: f.Code, Stdout: string(stdout), Stderr: string(stderr)}, 0, nil + return ExecResponse{ExitCode: f.Code, Stdout: stdout.String(), Stderr: stderr.String()}, 0, nil case *wire.ErrorResp: return ExecResponse{}, 0, f } diff --git a/sandboxd/server/preview.go b/sandboxd/server/preview.go index 40776397..ac9dcf59 100644 --- a/sandboxd/server/preview.go +++ b/sandboxd/server/preview.go @@ -20,6 +20,9 @@ import ( "github.com/cocoonstack/sandbox/sandboxd/types" ) +// forwardedHeader marks a request one owner already forwarded; a token whose owner string no longer names this node must not bounce forever. +const forwardedHeader = "X-Sandbox-Preview-Forwarded" + type previewClaims struct { ID string `json:"id"` Port uint16 `json:"port"` @@ -93,6 +96,10 @@ func (p *PreviewServer) serve(w http.ResponseWriter, r *http.Request) { return } if claims.Owner != p.owner { + if r.Header.Get(forwardedHeader) != "" { + http.Error(w, "preview owner is not this node", http.StatusBadGateway) + return + } p.forward(w, r, claims.Owner) return } @@ -105,7 +112,9 @@ func (p *PreviewServer) proxyLocal(w http.ResponseWriter, r *http.Request, claim pr.Out.URL.Scheme = "http" // the synthetic host carries PreviewDial's target pr.Out.URL.Host = fmt.Sprintf("%s:%d", claims.ID, claims.Port) - pr.Out.URL.Path = "/" + strings.TrimPrefix(pr.In.URL.Path, "/p/"+r.PathValue("token")+"/") + prefix := "/p/" + r.PathValue("token") + "/" + pr.Out.URL.Path = "/" + strings.TrimPrefix(pr.In.URL.Path, prefix) + pr.Out.URL.RawPath = "/" + strings.TrimPrefix(pr.In.URL.EscapedPath(), prefix) // browser credentials for the preview domain must not reach guest code pr.Out.Header.Del("Cookie") pr.Out.Header.Del("Authorization") @@ -121,10 +130,15 @@ func (p *PreviewServer) proxyLocal(w http.ResponseWriter, r *http.Request, claim func (p *PreviewServer) forward(w http.ResponseWriter, r *http.Request, owner string) { target := &url.URL{Scheme: "http", Host: owner} - rp := httputil.NewSingleHostReverseProxy(target) - rp.ErrorHandler = func(w http.ResponseWriter, r *http.Request, err error) { - log.WithFunc("server.forward").Errorf(r.Context(), err, "forward to owner %s", owner) - http.Error(w, "owner node unreachable", http.StatusBadGateway) + rp := &httputil.ReverseProxy{ + Rewrite: func(pr *httputil.ProxyRequest) { + pr.SetURL(target) + pr.Out.Header.Set(forwardedHeader, p.owner) + }, + ErrorHandler: func(w http.ResponseWriter, r *http.Request, err error) { + log.WithFunc("server.forward").Errorf(r.Context(), err, "forward to owner %s", owner) + http.Error(w, "owner node unreachable", http.StatusBadGateway) + }, } rp.ServeHTTP(w, r) //nolint:gosec // owner host comes from an HMAC-signed token } diff --git a/sandboxd/server/preview_test.go b/sandboxd/server/preview_test.go index dccf2f00..271c2281 100644 --- a/sandboxd/server/preview_test.go +++ b/sandboxd/server/preview_test.go @@ -158,6 +158,45 @@ func TestPreviewForwardsToOwner(t *testing.T) { } } +func TestPreviewForwardStopsAfterOneHop(t *testing.T) { + entry := httptest.NewUnstartedServer(nil) + entryAddr := entry.Listener.Addr().String() + ps := NewPreviewServer("secret", "https://preview.example.com", "renamed:7777", &fakePreviewMgr{}) + entry.Config.Handler = ps.Handler() + entry.Start() + t.Cleanup(entry.Close) + + minter := NewPreviewServer("secret", "https://preview.example.com", entryAddr, &fakePreviewMgr{}) + resp, err := http.Get(entry.URL + "/p/" + mintToken(minter, "sb_1", 8080, time.Hour) + "/x") + if err != nil { + t.Fatalf("get: %v", err) + } + defer resp.Body.Close() + body, _ := io.ReadAll(resp.Body) + if resp.StatusCode != http.StatusBadGateway || !strings.Contains(string(body), "not this node") { + t.Errorf("status %d body %q, want one forward hop then 502", resp.StatusCode, body) + } +} + +func TestPreviewKeepsEncodedSlashes(t *testing.T) { + guestAddr := newGuestServer(t, func(r *http.Request) string { return "guest saw " + r.URL.EscapedPath() }) + ps := NewPreviewServer("secret", "node:9000", "node:7777", &fakePreviewMgr{ + dial: func(string, uint16) (net.Conn, error) { return net.Dial("tcp", guestAddr) }, + }) + ts := httptest.NewServer(ps.Handler()) + t.Cleanup(ts.Close) + + resp, err := http.Get(ts.URL + "/p/" + mintToken(ps, "sb_1", 8080, time.Hour) + "/repo/a%2Fb/tree") + if err != nil { + t.Fatalf("get: %v", err) + } + defer resp.Body.Close() + body, _ := io.ReadAll(resp.Body) + if string(body) != "guest saw /repo/a%2Fb/tree" { + t.Errorf("body %q, want the encoded slash preserved", body) + } +} + type fakePreviewMgr struct { dial func(id string, port uint16) (net.Conn, error) } diff --git a/sandboxd/server/relay.go b/sandboxd/server/relay.go index fd4fa1cf..efd0ccba 100644 --- a/sandboxd/server/relay.go +++ b/sandboxd/server/relay.go @@ -155,6 +155,9 @@ func (t *auditTee) WriteTo(w io.Writer) (int64, error) { return total, werr } } + if err == io.EOF { + return total, nil + } if err != nil { return total, err } diff --git a/sandboxd/server/relay_test.go b/sandboxd/server/relay_test.go index 08801ef3..20135de2 100644 --- a/sandboxd/server/relay_test.go +++ b/sandboxd/server/relay_test.go @@ -2,11 +2,13 @@ package server import ( "bufio" + "bytes" "context" "io" "net" "net/http" "net/http/httptest" + "strings" "testing" "time" ) @@ -125,6 +127,15 @@ func TestRelayClientDisconnectClosesGuest(t *testing.T) { } } +func TestAuditTeeWriteToEndsCleanlyBeforeALine(t *testing.T) { + tee := &auditTee{r: strings.NewReader("partial"), record: func([]byte) { t.Error("recorded a line that never ended") }} + var out bytes.Buffer + n, err := tee.WriteTo(&out) + if err != nil || n != 7 || out.String() != "partial" { + t.Errorf("WriteTo = %d, %v, %q; want 7, nil, the bytes", n, err, out.String()) + } +} + func TestCloseRelaysDrains(t *testing.T) { ts, srv := newRelayServer(t, func(c net.Conn) { _, _ = io.Copy(io.Discard, c) From a4aa2273bcf789c0c8074e0106c8087959bc6ff9 Mon Sep 17 00:00:00 2001 From: CMGS Date: Mon, 14 Sep 2026 19:14:02 +0800 Subject: [PATCH 07/40] review: take the pool-key digest off the warm claim path; small simplifications handleClaim hashed the pool key on every claim although only the redirect and error paths read it; the digest is now computed where it is used, and writeResult logs the key itself. Also: the orphan-snapshot filter is a slices.DeleteFunc, confirmGone names its logger once, SetPools and the egress dialer wrap their causes with %w, ClusterDigest reuses DecodedKey instead of decoding the cluster key again, portconn's lone constant uses the bare form, and redirectFallback is a plain loop without the retryAny predicate that existed only for it. --- sandboxd/config/config.go | 4 ++-- sandboxd/engine/portconn.go | 6 ++--- sandboxd/pool/egress.go | 2 +- sandboxd/pool/reconcile.go | 11 ++++----- sandboxd/pool/remove.go | 9 ++++--- sandboxd/pool/setpools.go | 2 +- sandboxd/server/server.go | 9 ++++--- sandboxd/server/server_http.go | 4 ++-- sdk/go/client.go | 43 ++++++++++------------------------ 9 files changed, 33 insertions(+), 57 deletions(-) diff --git a/sandboxd/config/config.go b/sandboxd/config/config.go index d6ba2154..e873fa46 100644 --- a/sandboxd/config/config.go +++ b/sandboxd/config/config.go @@ -255,8 +255,8 @@ func (c *Config) ClusterDigest(caFingerprint string) string { names[i] = t.Name } slices.Sort(names) - if c.Mesh != nil && c.Mesh.ClusterKey != "" { - if key, err := base64.StdEncoding.DecodeString(c.Mesh.ClusterKey); err == nil { + if c.Mesh != nil { + if key, _ := c.Mesh.DecodedKey(); key != nil { type auth struct{ Name, Token string } tenants := make([]auth, len(c.Tenants)) for i, t := range c.Tenants { diff --git a/sandboxd/engine/portconn.go b/sandboxd/engine/portconn.go index 5458c356..02fdaefc 100644 --- a/sandboxd/engine/portconn.go +++ b/sandboxd/engine/portconn.go @@ -12,10 +12,8 @@ import ( "github.com/cocoonstack/sandbox/protocol/wire" ) -const ( - // portReadBuf fits silkd's data frames in one buffered read. - portReadBuf = 64 << 10 -) +// portReadBuf fits silkd's data frames in one buffered read. +const portReadBuf = 64 << 10 var portDataHead = []byte(`{"type":"data","data":"`) diff --git a/sandboxd/pool/egress.go b/sandboxd/pool/egress.go index 4f64da3f..93ede798 100644 --- a/sandboxd/pool/egress.go +++ b/sandboxd/pool/egress.go @@ -322,7 +322,7 @@ func newEgressDialer(allow []netip.Prefix) *net.Dialer { } ip, err := netip.ParseAddr(host) if err != nil { - return fmt.Errorf("egress: unresolved address %q", host) + return fmt.Errorf("egress: unresolved address %q: %w", host, err) } ip = ip.Unmap() if nat64Range.Contains(ip) { diff --git a/sandboxd/pool/reconcile.go b/sandboxd/pool/reconcile.go index fd44cf4f..58467dba 100644 --- a/sandboxd/pool/reconcile.go +++ b/sandboxd/pool/reconcile.go @@ -90,13 +90,10 @@ func (m *Manager) Reconcile(ctx context.Context) error { if snapsErr != nil { logger.Warnf(ctx, "snapshot sweep skipped: %v", snapsErr) } else { - var orphans []string - for _, snap := range snaps { - orphanHib := strings.HasPrefix(snap, hibernatePrefix) && !referenced[snap] - if orphanHib || strings.HasPrefix(snap, forkPrefix) || strings.HasPrefix(snap, goldenPrefix) { - orphans = append(orphans, snap) - } - } + orphans := slices.DeleteFunc(snaps, func(snap string) bool { + return strings.HasPrefix(snap, hibernatePrefix) && referenced[snap] || + !strings.HasPrefix(snap, hibernatePrefix) && !strings.HasPrefix(snap, forkPrefix) && !strings.HasPrefix(snap, goldenPrefix) + }) m.runBounded(ctx, len(orphans), func(ctx context.Context, i int) { m.dropSnap(ctx, orphans[i]) logger.Infof(ctx, "removed orphan snapshot %s", orphans[i]) diff --git a/sandboxd/pool/remove.go b/sandboxd/pool/remove.go index 4209ec61..e7a6a54e 100644 --- a/sandboxd/pool/remove.go +++ b/sandboxd/pool/remove.go @@ -29,16 +29,15 @@ func (m *Manager) removeVM(ctx context.Context, name string) bool { func (m *Manager) confirmGone(ctx context.Context, name string) bool { ctx, cancel := context.WithTimeout(ctx, removeVerifyTimeout) defer cancel() + logger := log.WithFunc("pool.confirmGone") _, present, err := m.findVM(ctx, name) if err != nil { - // Cannot tell: a false "gone" leaks a running VM, a retry only costs a sweep. - log.WithFunc("pool.confirmGone").Warnf(ctx, "verify remove of %s: %v", name, err) + // cannot tell: a false "gone" leaks a running VM, a retry only costs a sweep + logger.Warnf(ctx, "verify remove of %s: %v", name, err) return false } if present { - log.WithFunc("pool.confirmGone").Errorf(ctx, - fmt.Errorf("vm %s survived removal", name), - "remove did not take effect; leaving it accounted for retry") + logger.Errorf(ctx, fmt.Errorf("vm %s survived removal", name), "remove did not take effect; leaving it accounted for retry") return false } return true diff --git a/sandboxd/pool/setpools.go b/sandboxd/pool/setpools.go index a45437be..48b7df62 100644 --- a/sandboxd/pool/setpools.go +++ b/sandboxd/pool/setpools.go @@ -20,7 +20,7 @@ func (m *Manager) SetPools(ctx context.Context, specs []config.PoolSpec) error { return err } if err := spec.ValidateLimits(); err != nil { - return fmt.Errorf("%w: %v", ErrBadCount, err) + return fmt.Errorf("%w: %w", ErrBadCount, err) } // egress is config-owned; accepting it here would silently drop it if spec.Egress != nil { diff --git a/sandboxd/server/server.go b/sandboxd/server/server.go index 85e254a2..810b4829 100644 --- a/sandboxd/server/server.go +++ b/sandboxd/server/server.go @@ -217,27 +217,26 @@ func (s *Server) handleClaim(w http.ResponseWriter, r *http.Request) { return } key := req.Key() - hash := key.Hash() tenant := tenantFrom(r.Context()) if len(req.Volumes) > 0 || req.VolumesAttachOnly { - s.handleVolumeClaim(w, r, req, key, hash, tenant) + s.handleVolumeClaim(w, r, req, key, key.Hash(), tenant) return } // the data plane must be direct, so a warm peer gets the claim by redirect, not proxy sb, err := s.mgr.ClaimWarm(r.Context(), key, req.TTL(), tenant, req.ClaimRef, nil) if errors.Is(err, pool.ErrNoWarm) { - if s.redirectClaim(r.Context(), w, req, key, hash, tenant) { + if s.redirectClaim(r.Context(), w, req, key, key.Hash(), tenant) { return } sb, err = s.mgr.ClaimProvision(r.Context(), key, req.TTL(), tenant, req.ClaimRef, nil) } // quota is per node, so a full node bounces the claim to a peer before answering 429 if errors.Is(err, pool.ErrQuota) && s.placer != nil && !req.NoRedirect && - writeRedirect(w, s.placer.Candidates(hash)) { + writeRedirect(w, s.placer.Candidates(key.Hash())) { return } - writeResult(w, r, "claim", hash, "provisioning failed", err, func() { + writeResult(w, r, "claim", key, "provisioning failed", err, func() { writeJSON(w, http.StatusOK, s.claimResponse(sb)) }) } diff --git a/sandboxd/server/server_http.go b/sandboxd/server/server_http.go index 3ace7f56..14457d97 100644 --- a/sandboxd/server/server_http.go +++ b/sandboxd/server/server_http.go @@ -84,11 +84,11 @@ func writePoolErr(w http.ResponseWriter, err error) bool { return false } -func writeResult(w http.ResponseWriter, r *http.Request, op, id, failMsg string, err error, ok func()) { +func writeResult(w http.ResponseWriter, r *http.Request, op string, id any, failMsg string, err error, ok func()) { switch { case writePoolErr(w, err): case err != nil: - log.WithFunc("server.writeResult").Errorf(r.Context(), err, "%s %s", op, id) + log.WithFunc("server.writeResult").Errorf(r.Context(), err, "%s %v", op, id) writeErr(w, http.StatusInternalServerError, failMsg) default: ok() diff --git a/sdk/go/client.go b/sdk/go/client.go index 8bbecd6e..89045ebf 100644 --- a/sdk/go/client.go +++ b/sdk/go/client.go @@ -281,11 +281,6 @@ func retryMiss(err error) bool { return !ok || he.Status == http.StatusNotFound } -// retryAny retries a redirect candidate's failure unconditionally: one -// candidate being wrong for the claim (no egress, a stale name) must not -// stop the walk from reaching a candidate that would still succeed. -func retryAny(error) bool { return true } - // retryTransient reports whether an origin-fallback is worth the round-trip: // true for a transport failure, a miss (404), full (429), mid-heal (503), an // engine/proxy failure (500/502/504), or a mid-rotation 401 (the origin @@ -338,14 +333,9 @@ func claimFollow(origin, verb string, encode claimEncoder, claimAt claimPoster) return addr, target, nil } -// redirectFallback walks candidates via claimAt, retrying broadly (retryAny). -// If every candidate is exhausted and the last failure was transient -// (retryTransient), it gives origin one more no_redirect attempt — the node -// that issued the redirect provisions or heals locally instead of leaving -// the claim stuck on stale gossip. A definitive last failure (a bad request, -// an auth rejection, a conflict) skips the fallback: origin would fail the -// same way. A second-level redirect (a compliant server never sends one -// once no_redirect is set) fails the candidate rather than being followed. +// redirectFallback claims at every candidate with no_redirect; when the last failure was transient +// (retryTransient) origin gets one more attempt, since it provisions or heals locally instead of +// leaving the claim on stale gossip. func redirectFallback(origin string, candidates []string, claimAt func(addr string) (claimResponse, error)) (string, claimResponse, error) { claimNoRedirect := func(target string) (claimResponse, error) { cr, err := claimAt(target) @@ -357,28 +347,21 @@ func redirectFallback(origin string, candidates []string, claimAt func(addr stri } return cr, nil } - - var won string - var cr claimResponse - tryErr := tryEach(candidates, func(addr string) error { - target, err := claimNoRedirect(addr) - if err != nil { - return err + var lastErr error + for _, addr := range candidates { + cr, err := claimNoRedirect(addr) + if err == nil { + return addr, cr, nil } - won, cr = addr, target - return nil - }, retryAny) - if tryErr == nil { - return won, cr, nil + lastErr = err } - if !retryTransient(tryErr) { - return "", claimResponse{}, fmt.Errorf("all redirect targets failed: %w", tryErr) + if !retryTransient(lastErr) { + return "", claimResponse{}, fmt.Errorf("all redirect targets failed: %w", lastErr) } cr, err := claimNoRedirect(origin) if err != nil { - // Both halves matter to whoever reads this: the peers' failures say why - // the claim left the origin, the origin's says why coming back did not help. - return "", claimResponse{}, fmt.Errorf("all redirect targets failed, origin fallback failed: %w", errors.Join(tryErr, err)) + // the peers' failures say why the claim left origin, origin's says why coming back did not help + return "", claimResponse{}, fmt.Errorf("all redirect targets failed, origin fallback failed: %w", errors.Join(lastErr, err)) } return origin, cr, nil } From 965297fc70db37b804fbf8fd4c425217158eced7 Mon Sep 17 00:00:00 2001 From: CMGS Date: Mon, 14 Sep 2026 19:15:59 +0800 Subject: [PATCH 08/40] silkd: bound the find read on the handle, fail a session without entropy, latch a present lane verdict scan_file checked the size through metadata and then read the file unbounded, so a file growing between the two calls was read whole; the read now stops at FIND_MAX_FILE on the same handle. rand_token fell back to a per-process counter when no entropy source answered, which made the session marker forgeable by the shell's own commands; session.create now fails instead. routes_directly re-read /etc/silkd-lane on every exec of a direct-lane guest; the host writes the file once per claim, so a present verdict is latched and only its absence is re-read. --- silkd/src/find.rs | 6 +++--- silkd/src/net.rs | 42 +++++++++++++++++++++++------------------- silkd/src/session.rs | 2 +- silkd/src/sysutil.rs | 10 ++++++---- 4 files changed, 33 insertions(+), 27 deletions(-) diff --git a/silkd/src/find.rs b/silkd/src/find.rs index 72cd703b..9562df3d 100644 --- a/silkd/src/find.rs +++ b/silkd/src/find.rs @@ -102,16 +102,16 @@ impl Walk<'_> { Ok(true) } - /// Scans one file; the size bound comes off the open handle, so check and read see one file. + /// Scans one file; the size check and the read share one handle, and the read itself stops at the bound. fn scan_file(&self, path: &Path, body: &mut String) -> bool { - let Ok(mut file) = std::fs::File::open(path) else { + let Ok(file) = std::fs::File::open(path) else { return true; }; if file.metadata().is_ok_and(|m| m.len() > FIND_MAX_FILE) { return true; } body.clear(); - if file.read_to_string(body).is_err() { + if file.take(FIND_MAX_FILE).read_to_string(body).is_err() { return true; } let name: Arc = path.to_string_lossy().into(); diff --git a/silkd/src/net.rs b/silkd/src/net.rs index 58ce1edf..22ba7876 100644 --- a/silkd/src/net.rs +++ b/silkd/src/net.rs @@ -3,13 +3,14 @@ use std::fs::File; use std::io::Read; use std::sync::LazyLock; -use std::sync::atomic::{AtomicBool, AtomicI8, Ordering}; +use std::sync::atomic::{AtomicI8, Ordering}; /// File the host writes with its lane verdict, `relay` or `direct`; on the root filesystem so a guest reboot keeps it. const LANE_FILE: &str = "/etc/silkd-lane"; static LANE_OVERRIDE: AtomicI8 = AtomicI8::new(-1); -static LANE_RELAY: AtomicBool = AtomicBool::new(false); +/// The lane file's verdict once read: -1 not yet present, 0 direct, 1 relay. +static LANE: AtomicI8 = AtomicI8::new(-1); /// Reports whether the guest can reach a network; only a `device`-backed interface counts, since the kernel auto-creates virtual tunnels. pub fn has_egress() -> bool { @@ -38,15 +39,18 @@ pub fn routes_directly() -> bool { if !has_egress() { return false; } - // relay latches: the host never unlocks a lane, but a late lock can land after the first exec. - if LANE_RELAY.load(Ordering::Relaxed) { - return false; - } - let relay = lane_is_relay(LANE_FILE); - if relay { - LANE_RELAY.store(true, Ordering::Relaxed); - } - !relay + // the host writes the file once per claim, so a present verdict latches; only absence re-reads. + let lane = match LANE.load(Ordering::Relaxed) { + -1 => match lane_verdict(LANE_FILE) { + Some(relay) => { + LANE.store(i8::from(relay), Ordering::Relaxed); + relay + } + None => false, + }, + v => v == 1, + }; + !lane } /// Lane override for tests: set_var would race every concurrent getenv. @@ -65,26 +69,26 @@ fn silkd_net() -> i8 { *SILKD_NET } -fn lane_is_relay(path: &str) -> bool { +/// Reads the lane file: `Some(true)` for relay, `Some(false)` for direct, `None` while the host has not written it. +fn lane_verdict(path: &str) -> Option { let mut buf = [0u8; 8]; - File::open(path) - .and_then(|mut f| f.read(&mut buf)) - .is_ok_and(|n| buf[..n].trim_ascii() == b"relay") + let n = File::open(path).and_then(|mut f| f.read(&mut buf)).ok()?; + Some(buf[..n].trim_ascii() == b"relay") } #[cfg(test)] mod tests { - use super::lane_is_relay; + use super::lane_verdict; #[test] fn lane_file_decides_relay() { let dir = tempfile::tempdir().unwrap(); let path = dir.path().join("lane"); let text = path.to_str().unwrap(); - assert!(!lane_is_relay(text)); + assert_eq!(lane_verdict(text), None); std::fs::write(&path, "direct\n").unwrap(); - assert!(!lane_is_relay(text)); + assert_eq!(lane_verdict(text), Some(false)); std::fs::write(&path, "relay\n").unwrap(); - assert!(lane_is_relay(text)); + assert_eq!(lane_verdict(text), Some(true)); } } diff --git a/silkd/src/session.rs b/silkd/src/session.rs index 4c296198..7d70276e 100644 --- a/silkd/src/session.rs +++ b/silkd/src/session.rs @@ -203,7 +203,7 @@ impl Io { c => c, }; // the marker must be unforgeable: the shell runs untrusted code that could fake a sentinel. - let marker = format!("__SILK_{}__", sysutil::rand_token()); + let marker = format!("__SILK_{}__", sysutil::rand_token()?); self.cmd_buf.clear(); // `{ …; } String { } /// A 128-bit CSPRNG hex token, used where an in-sandbox command must not forge the value. -pub fn rand_token() -> String { +pub fn rand_token() -> std::io::Result { let mut b = [0u8; 16]; if !fill_random(&mut b) { - return tmp_suffix(); + return Err(std::io::Error::other( + "no entropy source for a session marker", + )); } - b.iter() + Ok(b.iter() .flat_map(|byte| [HEX[(byte >> 4) as usize], HEX[(byte & 0xf) as usize]]) .map(char::from) - .collect() + .collect()) } /// SIGKILLs the group led by `pgid`, so a session's external command dies with its shell; a synthetic id misses with ESRCH. From a6d0fafe32429e616c351b89602c8e3f384493e8 Mon Sep 17 00:00:00 2001 From: CMGS Date: Mon, 14 Sep 2026 19:18:22 +0800 Subject: [PATCH 09/40] review: silkd and boot-init simplifications fs.replace expands capture groups through a counting Replacer instead of allocating a String per match; the pty pre_exec registration moves into sysutil so every unsafe block lives in one module; tree's stderr cap is a file-level constant; Table::get is crate-visible; boot-init renders its marks into one String, trims the MAC in place, and formats the device path only for a serial that matched. Doc lines trimmed to one fact. --- boot/init/src/boot.rs | 21 +++++++++++---------- boot/init/src/sys.rs | 2 +- silkd/src/exec.rs | 1 + silkd/src/find.rs | 34 ++++++++++++++++++++++------------ silkd/src/proc.rs | 2 +- silkd/src/pty.rs | 6 +----- silkd/src/sysutil.rs | 12 ++++++++++-- silkd/src/tree.rs | 9 +++++---- 8 files changed, 52 insertions(+), 35 deletions(-) diff --git a/boot/init/src/boot.rs b/boot/init/src/boot.rs index db2df6c2..c495bded 100644 --- a/boot/init/src/boot.rs +++ b/boot/init/src/boot.rs @@ -1,5 +1,6 @@ //! Boot sequence: early mounts, resolve disks, overlay, persist network, switch_root, exec. +use std::fmt::Write; use std::fs; use std::path::Path; use std::time::{Duration, Instant}; @@ -33,10 +34,11 @@ impl Marks { } fn render(&self) -> String { - self.points - .iter() - .map(|(label, us)| format!(" {label}@{us}us")) - .collect() + let mut out = String::new(); + for (label, us) in &self.points { + let _ = write!(out, " {label}@{us}us"); + } + out } } @@ -54,7 +56,7 @@ pub fn run() -> ! { let _ = sys::mount("proc", "/proc", Some("proc"), sys::MNT_SECURE, None); let _ = sys::mount("sysfs", "/sys", Some("sysfs"), sys::MNT_SECURE, None); - // start marker, visible at production loglevel where the kernel's own boot lines are suppressed. + // printed at the production loglevel, which suppresses the kernel's own boot lines. println!("sandbox-init: start at {}s", uptime()); let cmdline = fs::read_to_string("/proc/cmdline").unwrap_or_default(); @@ -193,9 +195,9 @@ fn poll_slots( } fn read_nic_mac(device: &str) -> Option { - let mac = fs::read_to_string(format!("/sys/class/net/{device}/address")).ok()?; - let mac = mac.trim_end(); - (!mac.is_empty() && mac != "00:00:00:00:00:00").then(|| mac.to_string()) + let mut mac = fs::read_to_string(format!("/sys/class/net/{device}/address")).ok()?; + mac.truncate(mac.trim_end().len()); + (!mac.is_empty() && mac != "00:00:00:00:00:00").then_some(mac) } fn mkdir_all(path: &str) -> Result<(), String> { @@ -233,7 +235,6 @@ fn scan_serials(ids: &[&str], found: &mut [Option]) { format!("/sys/block/{name}/serial"), format!("/sys/block/{name}/device/serial"), ]; - let device = format!("/dev/{name}"); for path in paths { let Ok(serial) = fs::read_to_string(&path) else { continue; @@ -241,7 +242,7 @@ fn scan_serials(ids: &[&str], found: &mut [Option]) { if serial.trim_end().is_empty() { continue; } - record_serial(ids, found, serial.trim_end(), &device); + record_serial(ids, found, serial.trim_end(), &format!("/dev/{name}")); break; } } diff --git a/boot/init/src/sys.rs b/boot/init/src/sys.rs index bf3848af..b2470138 100644 --- a/boot/init/src/sys.rs +++ b/boot/init/src/sys.rs @@ -5,7 +5,7 @@ use std::io; use std::ptr; use std::time::Duration; -/// nosuid|nodev|noexec for the kernel API filesystems. +/// Mount flags for the kernel API filesystems. pub const MNT_SECURE: libc::c_ulong = libc::MS_NOSUID | libc::MS_NODEV | libc::MS_NOEXEC; pub fn mount( diff --git a/silkd/src/exec.rs b/silkd/src/exec.rs index 2630323d..58da4c1f 100644 --- a/silkd/src/exec.rs +++ b/silkd/src/exec.rs @@ -77,6 +77,7 @@ where let pump = tokio::spawn(pump_out(Arc::clone(&proc), stdout, stderr, pump_fg)); let pump_abort = pump.abort_handle(); + // stdin rides its own task so a slow write cannot stall output or reaping. tokio::spawn(pump_stdin(stdin, client, req.detach)); // record the reaped code before the drain: an abort mid-drain must publish the real code, not -1. diff --git a/silkd/src/find.rs b/silkd/src/find.rs index 9562df3d..0b7855ef 100644 --- a/silkd/src/find.rs +++ b/silkd/src/find.rs @@ -4,7 +4,7 @@ use std::io::Read; use std::path::{Path, PathBuf}; use std::sync::Arc; -use regex::Regex; +use regex::{Regex, Replacer}; use tokio::fs; use tokio::io::{AsyncBufRead, AsyncBufReadExt, AsyncWrite}; use tokio::runtime::Handle; @@ -18,7 +18,7 @@ pub const FIND_MAX_FILE: u64 = 8 * 1024 * 1024; /// Match frames in flight between the walking thread and the writer. const MATCH_QUEUE: usize = 256; -/// Bytes of match content in flight: above the largest single match, so a reserve never blocks forever, and far below the gigabytes 256 size-bound frames would pin. +/// Bytes of match content in flight: above one size-bound match, so a reserve never blocks forever, far below 256 of them. const MATCH_QUEUE_BYTES: usize = 2 * FIND_MAX_FILE as usize; struct MatchBudget { @@ -132,6 +132,19 @@ impl Walk<'_> { } } +/// Counts replacements while expanding `$n` groups straight into the output, no String per match. +struct Counting<'a> { + replacement: &'a str, + count: u64, +} + +impl regex::Replacer for Counting<'_> { + fn replace_append(&mut self, caps: ®ex::Captures<'_>, dst: &mut String) { + self.count += 1; + caps.expand(self.replacement, dst); + } +} + /// Streams `match` frames for every line under `path` matching `pattern`, terminated by `done`. pub async fn find( reader: &mut R, @@ -159,20 +172,17 @@ pub async fn replace( Err(e) => return proto::error_frame(w, ErrorKind::BadRequest, e.to_string()).await, }; for file in files { - let mut count: u64 = 0; + let mut counting = Counting { + replacement: &replacement, + count: 0, + }; if !is_oversized(&file).await { let body = match fs::read_to_string(&file).await { Ok(body) => body, Err(e) => return err_frame(w, &e, "read").await, }; - // expand keeps the &str replacer's $n capture-group semantics. - let new = re.replace_all(&body, |caps: ®ex::Captures| { - count += 1; - let mut expanded = String::new(); - caps.expand(&replacement, &mut expanded); - expanded - }); - if count > 0 + let new = re.replace_all(&body, counting.by_ref()); + if counting.count > 0 && let Err(e) = crate::fs::write_atomic(Path::new(&file), new.as_bytes()).await { return err_frame(w, &e, "write").await; @@ -182,7 +192,7 @@ pub async fn replace( w, &Response::Replaced { file, - replacements: count, + replacements: counting.count, }, ) .await?; diff --git a/silkd/src/proc.rs b/silkd/src/proc.rs index 434f9763..bbdef43c 100644 --- a/silkd/src/proc.rs +++ b/silkd/src/proc.rs @@ -65,7 +65,7 @@ impl Table { proc } - pub fn get(&self, pid: u32) -> Option> { + pub(crate) fn get(&self, pid: u32) -> Option> { sysutil::lock(&self.inner).get(&pid).cloned() } diff --git a/silkd/src/pty.rs b/silkd/src/pty.rs index cbab7f0c..84949dcd 100644 --- a/silkd/src/pty.rs +++ b/silkd/src/pty.rs @@ -61,11 +61,7 @@ pub async fn open( return crate::proto::err_frame(out, &e, "dup pty slave").await; } } - // SAFETY: make_controlling_tty runs only async-signal-safe syscalls, which - // is the contract for a post-fork pre_exec hook. - unsafe { - cmd.pre_exec(|| sysutil::make_controlling_tty()); - } + sysutil::adopt_controlling_tty(&mut cmd); cmd.kill_on_drop(true); let mut child = match cmd.spawn() { diff --git a/silkd/src/sysutil.rs b/silkd/src/sysutil.rs index ade7c2e5..e1c312e4 100644 --- a/silkd/src/sysutil.rs +++ b/silkd/src/sysutil.rs @@ -1,4 +1,4 @@ -//! Small OS helpers; the crate's unsafe work lives here, except pty.rs's pre_exec registration. +//! Small OS helpers; the crate's unsafe work lives here. use std::ffi::{CStr, CString}; use std::io::Read; @@ -140,11 +140,19 @@ pub fn set_winsize(fd: RawFd, cols: u16, rows: u16) -> std::io::Result<()> { Ok(()) } +/// Makes the child a session leader with fd 0 as its controlling terminal. +pub fn adopt_controlling_tty(cmd: &mut Command) { + // SAFETY: make_controlling_tty runs only async-signal-safe syscalls, the contract for a post-fork pre_exec hook. + unsafe { + cmd.pre_exec(|| make_controlling_tty()); + } +} + /// Makes the calling process a session leader and adopts fd 0 as its controlling terminal. /// /// # Safety /// Only async-signal-safe syscalls; valid in a post-fork child. -pub unsafe fn make_controlling_tty() -> std::io::Result<()> { +unsafe fn make_controlling_tty() -> std::io::Result<()> { // SAFETY: setsid and ioctl are async-signal-safe and take no pointers; the // caller guarantees a post-fork child (see # Safety). unsafe { diff --git a/silkd/src/tree.rs b/silkd/src/tree.rs index 7c14027d..cde49f2b 100644 --- a/silkd/src/tree.rs +++ b/silkd/src/tree.rs @@ -12,6 +12,8 @@ use tokio::process::Command; use crate::proto::{self, ErrorKind, Response, err_frame}; use crate::sysutil; +const STDERR_CAP: usize = 16 * 1024; + /// Extracts a client tar stream into `dest`, creating it; a stream failure leaves `dest` unchanged. pub async fn push(mut reader: R, w: &mut W, dest: String) -> io::Result<()> where @@ -154,17 +156,16 @@ fn merge_tree(src: &Path, dst: &Path) -> io::Result<()> { Ok(()) } -/// Reads a child's stderr, capped at CAP bytes; tar's first error is the useful one. +/// Reads a child's stderr up to STDERR_CAP; tar's first error is the useful one. async fn drain(mut stderr: tokio::process::ChildStderr) -> String { - const CAP: usize = 16 * 1024; let mut out = Vec::new(); let mut buf = [0u8; 4096]; while let Ok(n) = stderr.read(&mut buf).await { if n == 0 { break; } - if out.len() < CAP { - out.extend_from_slice(&buf[..(n.min(CAP - out.len()))]); + if out.len() < STDERR_CAP { + out.extend_from_slice(&buf[..(n.min(STDERR_CAP - out.len()))]); } } String::from_utf8(out).unwrap_or_else(|e| String::from_utf8_lossy(e.as_bytes()).into_owned()) From 05b0bda16a57944b542e709d3e294e45d17a6555 Mon Sep 17 00:00:00 2001 From: CMGS Date: Mon, 14 Sep 2026 19:23:12 +0800 Subject: [PATCH 10/40] review: comment register and budget sweep; adapter method order Inline comments open lowercase in the STE register (61 lines); godocs that restated a short body are gone (checkpointURL, probeMAC, guardsEgressLane, validateArchiveWindow, synced, goldenCAMatches, two desktopsmoke steps) and multi-sentence comments in the wire package and the Go SDK are one line each. The openai adapter lists its public methods before the SDK hooks; two Python **fields parameters are typed; three test fakes close their socket, narrow a suppress, and report the observed op to the main thread instead of asserting inside it. --- e2e/cmd/desktopsmoke/main.go | 2 -- e2e/cmd/lifecycle/main.go | 2 +- e2e/cmd/rpcbench/main.go | 2 +- e2e/cmd/smoke/main.go | 10 +++--- e2e/cmd/volumesmoke/main.go | 2 +- protocol/wire/frame.go | 12 ++----- sandboxd/config/config.go | 5 +-- sandboxd/egress/ca.go | 4 +-- sandboxd/engine/engine.go | 2 +- sandboxd/engine/installca.go | 4 +-- sandboxd/engine/volume.go | 2 +- sandboxd/main.go | 8 ++--- sandboxd/pool/archive.go | 12 +++---- sandboxd/pool/checkpoint.go | 2 +- sandboxd/pool/claim.go | 6 ++-- sandboxd/pool/claims.go | 1 - sandboxd/pool/hibernate.go | 2 +- sandboxd/pool/refill.go | 1 - sandboxd/pool/remove.go | 2 +- sandboxd/pool/setpools.go | 4 +-- sandboxd/pool/template.go | 2 +- sandboxd/pool/volume.go | 8 ++--- sandboxd/server/relay.go | 3 +- sandboxd/store/dir/dir.go | 8 ++--- sandboxd/store/peer/broadcast.go | 2 +- sandboxd/store/peer/peer.go | 4 +-- sandboxd/store/peer/probe.go | 3 +- sandboxd/store/peer/transport.go | 5 ++- sandboxd/store/s3/s3.go | 4 +-- sandboxd/store/storetest/storetest.go | 2 +- sdk/go/client.go | 24 +++---------- sdk/go/info.go | 3 +- sdk/go/port.go | 3 +- sdk/go/sandbox.go | 3 +- sdk/go/silkd/silkdtest/fake.go | 5 +-- sdk/openai/cocoonsandbox_openai/adapter.py | 40 +++++++++++----------- sdk/openai/e2e.py | 1 - sdk/python/cocoonsandbox/conn.py | 4 +-- sdk/python/cocoonsandbox/frames.py | 6 ++-- sdk/python/e2e.py | 8 ++--- sdk/python/tests/test_hardening.py | 1 + sdk/python/tests/test_stream.py | 8 +++-- sdk/python/tests/test_wire_binding.py | 2 +- silkd/src/find.rs | 4 +-- silkd/src/fs.rs | 2 +- silkd/src/watch.rs | 4 +-- 46 files changed, 105 insertions(+), 139 deletions(-) diff --git a/e2e/cmd/desktopsmoke/main.go b/e2e/cmd/desktopsmoke/main.go index 766e6359..b2aa0ba2 100644 --- a/e2e/cmd/desktopsmoke/main.go +++ b/e2e/cmd/desktopsmoke/main.go @@ -143,7 +143,6 @@ func clickAndReadCursor(ctx context.Context, sb *sandbox.Sandbox) error { return nil } -// proxyReachesSession proves the session's environment carries the relay and the relay reaches an allowed origin. func proxyReachesSession(ctx context.Context, sb *sandbox.Sandbox, probe string) error { script := `systemctl is-active guest-proxy.service >&2 || exit 1; [ -n "$http_proxy" ] || { echo http_proxy unset in the session >&2; exit 1; }; exec curl -sS -m 20 -o /dev/null -w '%{http_code}' "$1"` out, err := execute(ctx, sb, []string{"sh", "-c", script, "sh", probe}, 30) @@ -157,7 +156,6 @@ func proxyReachesSession(ctx context.Context, sb *sandbox.Sandbox, probe string) return nil } -// execute runs one command through osworld-server and returns its output. func execute(ctx context.Context, sb *sandbox.Sandbox, command []string, timeout int) (string, error) { req, _ := json.Marshal(map[string]any{"command": command, "shell": false, "timeout": timeout}) body, err := harness.HTTPOverPort(ctx, sb, serverPort, "POST", "/execute", req) diff --git a/e2e/cmd/lifecycle/main.go b/e2e/cmd/lifecycle/main.go index df1ba5a1..870efeb5 100644 --- a/e2e/cmd/lifecycle/main.go +++ b/e2e/cmd/lifecycle/main.go @@ -52,7 +52,7 @@ func run(addr, token, template, netShape string, wait time.Duration) error { } fmt.Printf("idle-hibernated then archived in %.1fs (local VM dropped)\n", elapsed.Seconds()) - // The first access on an archived id is the cold wake: fetch the + // the first access on an archived id is the cold wake: fetch the // checkpoint, provision a fresh local VM, keep the id/token. wakeStart := time.Now() got, err := sb.ReadFile(ctx, markerPath) diff --git a/e2e/cmd/rpcbench/main.go b/e2e/cmd/rpcbench/main.go index ae313cc0..05e95985 100644 --- a/e2e/cmd/rpcbench/main.go +++ b/e2e/cmd/rpcbench/main.go @@ -47,7 +47,7 @@ func run(addr, token, template string, n int) error { dial := func() (net.Conn, error) { return dialAgent(ctx, sb.Owner(), sb.ID, sb.Token()) } - // Warm the path (wake resolution, page cache) before either mode. + // warm the path (wake resolution, page cache) before either mode. for range 5 { conn, err := dial() if err != nil { diff --git a/e2e/cmd/smoke/main.go b/e2e/cmd/smoke/main.go index 9e9a309a..84d61ee1 100644 --- a/e2e/cmd/smoke/main.go +++ b/e2e/cmd/smoke/main.go @@ -50,7 +50,7 @@ func run(addr, token, template, lspTemplate string, egress bool) error { ctx, cancel := context.WithTimeout(context.Background(), timeout) defer cancel() - // Explicitly the no-network lane: the git step asserts its typed error. + // explicitly the no-network lane: the git step asserts its typed error. client, sb, err := harness.Claim(ctx, addr, token, template, sandbox.WithNetwork(sandbox.NetNone)) if err != nil { return err @@ -209,7 +209,7 @@ func smokeWatch(ctx context.Context, sb *sandbox.Sandbox) error { if !ok { return fmt.Errorf("watch ended early: %v", w.Err()) } - // The atomic write surfaces as temp-file events too; any event + // the atomic write surfaces as temp-file events too; any event // whose path carries the name proves the stream. if strings.Contains(ev.Path, "w.txt") { return nil @@ -282,7 +282,7 @@ func smokeGit(ctx context.Context, sb *sandbox.Sandbox) error { return fmt.Errorf("after checkout: current=%q, want feature", br.Current) } - // This sandbox is on the no-network lane: push must fail with the typed + // this sandbox is on the no-network lane: push must fail with the typed // unimplemented error, not hang on an unreachable remote. if err := sb.GitPush(ctx, "/work", ""); !isSilkdKind(err, wire.KindUnimplemented) { return fmt.Errorf("push on none lane: %v, want unimplemented", err) @@ -553,7 +553,7 @@ func smokeTree(ctx context.Context, sb *sandbox.Sandbox) error { // (socket-activated in the base image) answers on 22, so its banner must // arrive through DialPort; a dead port must fail with the typed not_found. func smokePortForward(ctx context.Context, sb *sandbox.Sandbox) error { - // The step proves the relay, not sshd's readiness SLA: the guest's + // the step proves the relay, not sshd's readiness SLA: the guest's // socket-activated sshd can transiently refuse, so the positive probe // retries briefly; the dead-port negative below stays strict. pc, err := sb.DialPort(ctx, 22) @@ -664,7 +664,7 @@ func smokeProcs(ctx context.Context, sb *sandbox.Sandbox) error { return fmt.Errorf("logs replay %q missing mark", out.String()) } - // A long-runner dies to kill; its pid must be gone from the next attach. + // a long-runner dies to kill; its pid must be gone from the next attach. pid, err = sb.Spawn(ctx, sandbox.Cmd{Argv: []string{"sleep", "30"}}) if err != nil { return fmt.Errorf("spawn sleeper: %w", err) diff --git a/e2e/cmd/volumesmoke/main.go b/e2e/cmd/volumesmoke/main.go index 6dce910e..0696b4d7 100644 --- a/e2e/cmd/volumesmoke/main.go +++ b/e2e/cmd/volumesmoke/main.go @@ -151,7 +151,7 @@ func runWritable(ctx context.Context, client *sandbox.Client, template, volume s if _, err = writer.Exec(ctx, "sh", "-c", fmt.Sprintf("printf %%s %s > %s", stamp, path.Join(mount, file))); err != nil { return fmt.Errorf("write through the writable mount: %w", err) } - // A live writer owns the image: the second claim is refused before attach. + // a live writer owns the image: the second claim is refused before attach. busy, busyErr := client.New(ctx, template, sandbox.WithNetwork(sandbox.NetNone), sandbox.WithVolumes(sandbox.Volume{Name: volume, Mount: mount, Mode: "rw"})) if busyErr == nil { diff --git a/protocol/wire/frame.go b/protocol/wire/frame.go index 19bbee67..9a7a2ecc 100644 --- a/protocol/wire/frame.go +++ b/protocol/wire/frame.go @@ -102,9 +102,7 @@ var ( "data_end": decodeReq[DataEnd], } - // responseDecoders maps each type tag to a decoder. Two-stage dispatch is - // required because the same key differs in shape across variants (info's - // procs is a count, ps's procs is a list). + // responseDecoders dispatches on the type tag first: one key differs in shape across variants. responseDecoders = map[string]respDecoder{ "started": decodeResp[Started], "stdout": fastBulk("stdout", decodeResp[Stdout], func(d []byte) Response { return &Stdout{Data: d} }), @@ -149,10 +147,7 @@ type respPtr[T any] interface { Response } -// B64 carries request payload bytes. It exists because silkd's deserializer -// requires a base64 string and rejects null — which is exactly what -// encoding/json emits for a nil []byte. Decoding needs no counterpart: -// []byte-kinded types already base64-decode by default. +// B64 carries request payload bytes; a nil slice marshals as "" because silkd rejects null. type B64 []byte func (b B64) MarshalJSON() ([]byte, error) { @@ -419,8 +414,7 @@ type GitPull struct { func (GitPull) Op() string { return "git_pull" } -// GitBranch lists, creates, deletes, or checks out a branch. Action is -// list|create|delete|checkout ("op" is reserved by the frame tag). +// GitBranch lists, creates, deletes, or checks out a branch; Action carries the verb because "op" is the frame tag. type GitBranch struct { Path string `json:"path"` Action string `json:"action"` diff --git a/sandboxd/config/config.go b/sandboxd/config/config.go index e873fa46..5c1b4530 100644 --- a/sandboxd/config/config.go +++ b/sandboxd/config/config.go @@ -274,7 +274,6 @@ func (c *Config) ClusterDigest(caFingerprint string) string { return hex.EncodeToString(sum[:]) } -// guardsEgressLane counts any tenant policy, but only an egress-lane pool policy. func (c *Config) guardsEgressLane() bool { return slices.ContainsFunc(c.Tenants, func(t TenantSpec) bool { return t.Egress != nil }) || slices.ContainsFunc(c.Pools, func(p PoolSpec) bool { return p.Net == types.NetEgress && p.Egress != nil }) @@ -319,7 +318,6 @@ func (c *Config) validate() error { if err := c.validateAttachment(); err != nil { return err } - // a CNI network's tap lives in the VM netns, unreachable from the root-netns nft lock. if len(c.Networks) > 0 && c.guardsEgressLane() { return fmt.Errorf("guarded egress needs a bridge lane, not a CNI network: the tap lives in the VM netns and cannot be locked") } @@ -526,7 +524,7 @@ func Load(path string) (*Config, error) { return nil, fmt.Errorf("read config: %w", err) } cfg := &Config{} - // Hand-edited file: a typo must fail load, not silently change policy. + // hand-edited file: a typo must fail load, not silently change policy. if err := utils.DecodeStrictJSON(raw, cfg); err != nil { return nil, fmt.Errorf("parse config: %w", err) } @@ -556,7 +554,6 @@ func validatePolicy(p *egress.Policy, secrets map[string]struct{}) error { return nil } -// validateArchiveWindow requires archive_after past idle_hibernate, both non-negative. func validateArchiveWindow(idle, after, del int) error { if after < 0 || del < 0 { return fmt.Errorf("archive seconds must not be negative") diff --git a/sandboxd/egress/ca.go b/sandboxd/egress/ca.go index 2469b5ee..a2d1c6ea 100644 --- a/sandboxd/egress/ca.go +++ b/sandboxd/egress/ca.go @@ -68,7 +68,7 @@ func LoadCA(rootCertPEM, interCertPEM, interKeyPEM []byte) (*CA, error) { if err != nil { return nil, fmt.Errorf("generate leaf key: %w", err) } - // Fingerprint the whole file: CertPEM bakes it verbatim. + // fingerprint the whole file: CertPEM bakes it verbatim. sum := sha256.Sum256(rootCertPEM) return &CA{ rootPEM: rootCertPEM, @@ -93,7 +93,7 @@ func (c *CA) SignLeaf(host string) (*tls.Certificate, error) { return nil, err } now := time.Now() - // Guests reject a chain whose leaf outlives its intermediate. + // guests reject a chain whose leaf outlives its intermediate. notAfter := now.Add(leafValidity) if notAfter.After(c.interCert.NotAfter) { notAfter = c.interCert.NotAfter diff --git a/sandboxd/engine/engine.go b/sandboxd/engine/engine.go index 5f1b8dda..39e90312 100644 --- a/sandboxd/engine/engine.go +++ b/sandboxd/engine/engine.go @@ -190,7 +190,7 @@ func (e *Engine) SnapshotList(ctx context.Context) ([]string, error) { if err != nil { return nil, err } - // An empty store prints a human line ("No snapshots found."), not JSON. + // an empty store prints a human line ("No snapshots found."), not JSON. out = bytes.TrimSpace(out) if len(out) == 0 || out[0] != '[' { return nil, nil diff --git a/sandboxd/engine/installca.go b/sandboxd/engine/installca.go index 9749319a..8882cba8 100644 --- a/sandboxd/engine/installca.go +++ b/sandboxd/engine/installca.go @@ -36,7 +36,7 @@ func (e *Engine) silkdWriteFile(ctx context.Context, vsockSocket, path string, m return err } defer s.close() - // A send racing the guest's early answer+close hits EPIPE; prefer the buffered verdict. + // a send racing the guest's early answer+close hits EPIPE; prefer the buffered verdict. sendErr := func() error { if serr := s.send(wire.FsWrite{Path: path, Mode: &mode}); serr != nil { return serr @@ -70,7 +70,7 @@ func (e *Engine) silkdExec(ctx context.Context, vsockSocket string, argv ...stri defer s.close() sendErr := s.send(wire.Exec{Argv: argv, Env: map[string]string{"PATH": guestExecPATH}}) if sendErr == nil { - // A child exiting without reading stdin races this close; prefer the buffered exit frame. + // a child exiting without reading stdin races this close; prefer the buffered exit frame. sendErr = s.send(wire.StdinClose{}) } var out []byte diff --git a/sandboxd/engine/volume.go b/sandboxd/engine/volume.go index a5fc1a7e..62daa94f 100644 --- a/sandboxd/engine/volume.go +++ b/sandboxd/engine/volume.go @@ -126,7 +126,7 @@ func (e *Engine) findVolumeDevice(ctx context.Context, vsockSocket, name string) continue } device := "/dev/" + entry.Name - // The kernel publishes /sys/block before devtmpfs creates the node. + // the kernel publishes /sys/block before devtmpfs creates the node. if err := e.silkdStat(ctx, vsockSocket, device); err != nil { if isNotFound(err) { return "", false, nil diff --git a/sandboxd/main.go b/sandboxd/main.go index c7559bdd..8bb9ef41 100644 --- a/sandboxd/main.go +++ b/sandboxd/main.go @@ -92,7 +92,7 @@ func main() { ctx, stop := signal.NotifyContext(ctx, os.Interrupt, syscall.SIGTERM) defer stop() - // A node that cannot reconcile cannot trust its view of local VMs. + // a node that cannot reconcile cannot trust its view of local VMs. if err := mgr.Reconcile(ctx); err != nil { logger.Fatalf(ctx, err, "reconcile") } @@ -157,7 +157,7 @@ func main() { drained := make(chan struct{}) context.AfterFunc(ctx, func() { defer close(drained) - // Must outlive the canceled signal ctx to bound the drain. + // must outlive the canceled signal ctx to bound the drain. sctx, cancel := context.WithTimeout(context.WithoutCancel(ctx), shutdownGrace) defer cancel() _ = httpSrv.Shutdown(sctx) @@ -169,7 +169,7 @@ func main() { logger.Fatalf(ctx, err, "serve") } <-drained - // A detached recommit may not have converged; leave disk matching memory. + // a detached recommit may not have converged; leave disk matching memory. if err := mgr.FlushClaims(); err != nil { logger.Error(ctx, err, "flush claims") } @@ -197,7 +197,7 @@ func startMesh(ctx context.Context, cfg *config.Config, mgr *pool.Manager) (*mes if err != nil { return nil, err } - // Publish the config digest before Join, so the first gossip carries it. + // publish the config digest before Join, so the first gossip carries it. msh.SetSelfDigest(cfg.ClusterDigest(mgr.EgressCAFingerprint())) msh.UpdateSelf(ctx, mgr.WarmCounts(), mgr.TemplateHashes(), mgr.VolumeNames()) if err := msh.Join(mc.Join); err != nil { diff --git a/sandboxd/pool/archive.go b/sandboxd/pool/archive.go index 4768d74e..d1f79ede 100644 --- a/sandboxd/pool/archive.go +++ b/sandboxd/pool/archive.go @@ -79,7 +79,7 @@ func (m *Manager) archiveOnce(ctx context.Context) { // archive swaps a hibernated claim's backing from local VM to a store checkpoint. func (m *Manager) archive(ctx context.Context, sb *types.Sandbox) error { - // Must finish even if the sweep ctx is canceled, or the record diverges from the store. + // must finish even if the sweep ctx is canceled, or the record diverges from the store. ctx = context.WithoutCancel(ctx) defer m.untrack(m.archiving, sb.ID) m.mu.Lock() @@ -94,7 +94,7 @@ func (m *Manager) archive(ctx context.Context, sb *types.Sandbox) error { m.pendingCks[ckID] = struct{}{} m.mu.Unlock() defer m.untrack(m.pendingCks, ckID) - // A hibernated source is copied from its wake image, so no VM starts. + // a hibernated source is copied from its wake image, so no VM starts. ck, srcSnap, err := m.publishCheckpoint(ctx, sb, ckID, "archive", sb.Tenant, true) if err != nil { return err @@ -119,7 +119,7 @@ func (m *Manager) archive(ctx context.Context, sb *types.Sandbox) error { js := m.store.set(sb) m.mu.Unlock() if saveErr := m.store.commit(js); saveErr != nil { - // Roll back so memory matches the still-durable hibernated record; drop the orphan ck. + // roll back so memory matches the still-durable hibernated record; drop the orphan ck. m.mu.Lock() sb.ArchiveCk = "" sb.HibernateSnap, sb.VsockSocket, sb.VMName = snap, sock, vmName @@ -134,7 +134,7 @@ func (m *Manager) archive(ctx context.Context, sb *types.Sandbox) error { // disarm under Transition so a wake right after the release is not clobbered by a late disarm m.disarmEgress(sb.ID, true) sb.Transition.Unlock() - // Committed: the store ck is authoritative now; reclaim the local footprint. + // committed: the store ck is authoritative now; reclaim the local footprint. m.destroy(ctx, vmName) m.dropSnap(ctx, snap) m.counters.archives.Add(1) @@ -144,7 +144,7 @@ func (m *Manager) archive(ctx context.Context, sb *types.Sandbox) error { // wakeArchived restores an archived claim into a fresh local VM; caller holds Transition. func (m *Manager) wakeArchived(ctx context.Context, sb *types.Sandbox) (string, error) { - // Egress never archives; a corrupt archived egress claim must fail closed. + // egress never archives; a corrupt archived egress claim must fail closed. if sb.Key.Net == types.NetEgress { return "", fmt.Errorf("wake %s: egress lane cannot resume from archive", sb.ID) } @@ -186,7 +186,7 @@ func (m *Manager) wakeArchived(ctx context.Context, sb *types.Sandbox) (string, } return "", fmt.Errorf("wake %s: persist claims", sb.ID) } - // The pre-commit marker keeps a failed delete retryable after the journal update. + // the pre-commit marker keeps a failed delete retryable after the journal update. if delErr := m.ckpts.Delete(ctx, ck); delErr != nil { log.WithFunc("pool.wakeArchived").Warnf(ctx, "delete consumed archive ck %s: %v", ck, delErr) } else { diff --git a/sandboxd/pool/checkpoint.go b/sandboxd/pool/checkpoint.go index bf63918d..5bbfd7f6 100644 --- a/sandboxd/pool/checkpoint.go +++ b/sandboxd/pool/checkpoint.go @@ -132,7 +132,7 @@ func (m *Manager) DeleteCheckpoint(ctx context.Context, ckptID, tenant string, s if err := m.ckpts.Delete(ctx, ckptID); err != nil { return fmt.Errorf("delete checkpoint: %w", err) } - // A shared backend has no per-node replicas to chase. + // a shared backend has no per-node replicas to chase. if scope == DeleteFleet && m.peerDelete != nil && !m.ckptsShared { m.peerDelete(context.WithoutCancel(ctx), ckptID) } diff --git a/sandboxd/pool/claim.go b/sandboxd/pool/claim.go index 69b8c2c3..d64e7a4e 100644 --- a/sandboxd/pool/claim.go +++ b/sandboxd/pool/claim.go @@ -36,7 +36,7 @@ func (m *Manager) ClaimWarm(ctx context.Context, key types.PoolKey, ttl time.Dur if err != nil { return nil, err } - // Holds belong to this path until finalize; past it the sandbox carries them. + // holds belong to this path until finalize; past it the sandbox carries them. reserved := applied defer func() { m.unreserveVolumes(reserved) }() m.mu.Lock() @@ -169,7 +169,7 @@ func (m *Manager) releaseResolved(ctx context.Context, id string, sb *types.Sand log.WithFunc("pool.releaseResolved").Errorf(ctx, saveErr, "persist release of %s", id) return fmt.Errorf("release %s: %w", id, saveErr) } - // Cleanup must survive the caller hanging up; the claim is already dropped. + // cleanup must survive the caller hanging up; the claim is already dropped. ctx = context.WithoutCancel(ctx) if ck != "" { m.purgeArchiveCk(ctx, id, ck, sb.Tenant) // archived: no local VM @@ -343,7 +343,7 @@ func (m *Manager) reapOnce(ctx context.Context) { var expired []victim var dropped []string for id, sb := range m.claimed { - // A zero deadline means no expiry (an archived claim kept forever). + // a zero deadline means no expiry (an archived claim kept forever). if sb.Deadline.IsZero() || !now.After(sb.Deadline) { continue } diff --git a/sandboxd/pool/claims.go b/sandboxd/pool/claims.go index 9b1c2c7d..f68931c7 100644 --- a/sandboxd/pool/claims.go +++ b/sandboxd/pool/claims.go @@ -143,7 +143,6 @@ func (s *claimStore) save(claims map[string]*types.Sandbox) error { return s.commit(s.reset(claims)) } -// synced reports whether every sequenced change has reached disk. func (s *claimStore) synced() bool { s.mu.Lock() defer s.mu.Unlock() diff --git a/sandboxd/pool/hibernate.go b/sandboxd/pool/hibernate.go index 89c0b245..731f3d6e 100644 --- a/sandboxd/pool/hibernate.go +++ b/sandboxd/pool/hibernate.go @@ -97,7 +97,7 @@ func (m *Manager) hibernateLocked(ctx context.Context, sb *types.Sandbox) error m.dropSnap(ctx, snap) return ErrUnknownSandbox } - // The VM is hibernated either way, so the billing window closes here. + // the VM is hibernated either way, so the billing window closes here. m.recordHibernate(ctx, sb) m.disarmEgress(sb.ID, true) if err != nil { diff --git a/sandboxd/pool/refill.go b/sandboxd/pool/refill.go index 2cfeb713..329a9cba 100644 --- a/sandboxd/pool/refill.go +++ b/sandboxd/pool/refill.go @@ -260,7 +260,6 @@ func (m *Manager) adoptGolden(p *pool) { } } -// goldenCAMatches reports whether a golden's baked-CA state fits the pool. func (m *Manager) goldenCAMatches(final string, caNeeded bool) bool { if !caNeeded { return goldenSidecarMatches(final+caSidecarSuffix, "") diff --git a/sandboxd/pool/remove.go b/sandboxd/pool/remove.go index e7a6a54e..d5fa5fdb 100644 --- a/sandboxd/pool/remove.go +++ b/sandboxd/pool/remove.go @@ -16,7 +16,7 @@ import ( func (m *Manager) removeVM(ctx context.Context, name string) bool { m.closePrebound(name) - // Cancellation-immune: on a canceled ctx `cocoon vm rm` no-ops and orphans. + // cancellation-immune: on a canceled ctx `cocoon vm rm` no-ops and orphans. ctx = context.WithoutCancel(ctx) err := m.eng.Remove(ctx, name) if err == nil { diff --git a/sandboxd/pool/setpools.go b/sandboxd/pool/setpools.go index 48b7df62..12d11028 100644 --- a/sandboxd/pool/setpools.go +++ b/sandboxd/pool/setpools.go @@ -37,7 +37,7 @@ func (m *Manager) SetPools(ctx context.Context, specs []config.PoolSpec) error { var trim []string m.mu.Lock() - // Sequenced under the mutex (apply order) so commit drops a write a later apply superseded. + // sequenced under the mutex (apply order) so commit drops a write a later apply superseded. seq := m.poolStore.seq.Add(1) now := time.Now() for key, p := range m.pools { @@ -67,7 +67,7 @@ func (m *Manager) SetPools(ctx context.Context, specs []config.PoolSpec) error { m.destroy(ctx, trim[i]) }).Wait() m.refillOnce(runCtx) - // Persist the applied set so a restart rebuilds from it, not the config seed. + // persist the applied set so a restart rebuilds from it, not the config seed. persisted := slices.Collect(maps.Values(desired)) return m.poolStore.commit(seq, poolsFile{ConfigSeed: m.configSeedHash, Pools: persisted}) } diff --git a/sandboxd/pool/template.go b/sandboxd/pool/template.go index ce653f5c..d9f0ca6e 100644 --- a/sandboxd/pool/template.go +++ b/sandboxd/pool/template.go @@ -149,7 +149,7 @@ func (m *Manager) HasPromotedTemplate(ctx context.Context, key types.PoolKey, te owner, cached := m.tplSet[id] m.tplMu.Unlock() if !cached { - // Only a shared-store template promoted elsewhere after startup. + // only a shared-store template promoted elsewhere after startup. raw, err := m.tpls.ReadMeta(ctx, id) if err != nil { return false diff --git a/sandboxd/pool/volume.go b/sandboxd/pool/volume.go index f92893b1..0ec38f55 100644 --- a/sandboxd/pool/volume.go +++ b/sandboxd/pool/volume.go @@ -20,7 +20,7 @@ import ( const ( // sidecar, not data_dir: the marker travels with the image and survives a data_dir wipe volumeDirtySuffix = ".dirty" - // Caps the per-mount budget below, so teardown can never hang for long. + // caps the per-mount budget below, so teardown can never hang for long. volumeQuiesceMax = 10 * time.Second ) @@ -192,9 +192,9 @@ func (m *Manager) applyVolumes(ctx context.Context, sb *types.Sandbox, volumes [ // applyVolume keeps one volume's steps strictly ordered; siblings overlap freely. func (m *Manager) applyVolume(ctx context.Context, sb *types.Sandbox, volume resolvedVolume) error { - // Attach-only mounts nothing, so there is no umount to verify and no marker. + // attach-only mounts nothing, so there is no umount to verify and no marker. attachOnly := volume.applied.Mount == "" - // Write-ahead: the marker must be durable before any guest write can be. + // write-ahead: the marker must be durable before any guest write can be. if volume.disk.RW && !attachOnly { if err := markVolumeDirty(volume.disk.Path); err != nil { return fmt.Errorf("mark volume %q dirty: %w", volume.applied.Name, err) @@ -220,7 +220,7 @@ func (m *Manager) quiesceVolumes(ctx context.Context, sb *types.Sandbox) volumeT return td } logger := log.WithFunc("pool.quiesceVolumes") - // Cancellation-immune like removal: a caller hanging up must not skip the flush. + // cancellation-immune like removal: a caller hanging up must not skip the flush. ctx, cancel := context.WithTimeout(context.WithoutCancel(ctx), quiesceBudget(mounts)) defer cancel() stuck := false diff --git a/sandboxd/server/relay.go b/sandboxd/server/relay.go index efd0ccba..871e5ac1 100644 --- a/sandboxd/server/relay.go +++ b/sandboxd/server/relay.go @@ -59,8 +59,7 @@ func (s *Server) handleAgent(w http.ResponseWriter, r *http.Request) { s.relay(r.Context(), r.PathValue("id"), client, bufrw.Reader, guest) } -// wakeGuest resolves the sandbox's agent socket, waking a hibernated VM first, and dials silkd; -// on failure it has already answered the request. done ends the sandbox hold when the connection is over. +// wakeGuest dials the sandbox's silkd, waking a hibernated VM first; on failure it has already answered the request. func (s *Server) wakeGuest(ctx context.Context, w http.ResponseWriter, id, token string) (guest net.Conn, done func(), ok bool) { sock, done, err := s.mgr.WakeAgentSocket(ctx, id, token) switch { diff --git a/sandboxd/store/dir/dir.go b/sandboxd/store/dir/dir.go index 395fa521..4c3b8640 100644 --- a/sandboxd/store/dir/dir.go +++ b/sandboxd/store/dir/dir.go @@ -195,7 +195,7 @@ func (d *Store) publish(_ context.Context, staging, id, digest string) error { switch { case errors.Is(statErr, fs.ErrNotExist): stagedExport := filepath.Join(staging, store.ExportDir) - // Make the generation fresh before a peer can observe it without committed meta. + // make the generation fresh before a peer can observe it without committed meta. if err := os.Chtimes(stagedExport, now, now); err != nil { //nolint:gosec // our own staging dir return err } @@ -213,7 +213,7 @@ func (d *Store) publish(_ context.Context, staging, id, digest string) error { if readErr != nil && !errors.Is(readErr, fs.ErrNotExist) { return readErr } - // A sweep may already have selected an expired path for removal. + // a sweep may already have selected an expired path for removal. if time.Since(genInfo.ModTime()) >= generationGrace { return fmt.Errorf("generation %s expired before commit; retry after sweep", filepath.Base(genDir)) } @@ -225,7 +225,7 @@ func (d *Store) publish(_ context.Context, staging, id, digest string) error { _ = d.sweepGenerations(id) return os.RemoveAll(staging) } - // Age the previous current generation from supersession, not publication. + // age the previous current generation from supersession, not publication. if err := touchCurrentGeneration(final, now); err != nil { return err } @@ -263,7 +263,7 @@ func (d *Store) sweepGenerations(id string) (err error) { return err } defer func() { err = errors.Join(err, metaFile.Close()) }() - // The open file pins one inode across a concurrent meta rename. + // the open file pins one inode across a concurrent meta rename. meta, err := io.ReadAll(metaFile) if err != nil { return err diff --git a/sandboxd/store/peer/broadcast.go b/sandboxd/store/peer/broadcast.go index 34ae397f..6f6c173f 100644 --- a/sandboxd/store/peer/broadcast.go +++ b/sandboxd/store/peer/broadcast.go @@ -19,7 +19,7 @@ const ( // Broadcaster fans a checkpoint delete out to every peer so a healed copy does not outlive it. type Broadcaster struct { - // A nil Client uses a default with a short timeout: one wedged peer must not hold up the fan-out. + // a nil Client uses a default with a short timeout: one wedged peer must not hold up the fan-out. Client *http.Client Peers func() []string Token string diff --git a/sandboxd/store/peer/peer.go b/sandboxd/store/peer/peer.go index f049d79b..41c2929a 100644 --- a/sandboxd/store/peer/peer.go +++ b/sandboxd/store/peer/peer.go @@ -43,7 +43,7 @@ func (h *Healer) Pull(ctx context.Context, id, staging string, validate Validate return store.ErrNotFound } budget := cmp.Or(h.budget, healBudget) - // A client hanging up must not abandon a started pull; the budget bounds it. + // a client hanging up must not abandon a started pull; the budget bounds it. ctx, cancel := context.WithTimeout(context.WithoutCancel(ctx), budget) defer cancel() return h.pullFrom(ctx, id, staging, addrs, budget, validate) @@ -53,7 +53,7 @@ func (h *Healer) pullFrom(ctx context.Context, id, staging string, addrs []strin perOwner := budget / time.Duration(len(addrs)) var errs []error for _, addr := range addrs { - // One peer's rejected or partial transfer must not linger into the next's. + // one peer's rejected or partial transfer must not linger into the next's. if err := utils.RemoveDirEntries(staging, nil); err != nil { return fmt.Errorf("reset staging: %w", err) } diff --git a/sandboxd/store/peer/probe.go b/sandboxd/store/peer/probe.go index 0745cd38..3d3498d6 100644 --- a/sandboxd/store/peer/probe.go +++ b/sandboxd/store/peer/probe.go @@ -40,7 +40,7 @@ type redirectCacheEntry struct { // HTTPProber finds which peers hold a checkpoint by probing them on a cross-node miss. type HTTPProber struct { - // A nil Client uses a default 2-second timeout: a probe must never hang on a wedged peer. + // a nil Client uses a default 2-second timeout: a probe must never hang on a wedged peer. Client *http.Client Peers func() []string @@ -188,7 +188,6 @@ func VerifyProbeMAC(key []byte, id, sig string) bool { return false } -// probeMAC returns the base64 HMAC-SHA256 tag for id in bucket, keyed by key. func probeMAC(key []byte, id string, bucket int64) string { mac := hmac.New(sha256.New, key) mac.Write([]byte(id)) diff --git a/sandboxd/store/peer/transport.go b/sandboxd/store/peer/transport.go index 09e72c38..8159d014 100644 --- a/sandboxd/store/peer/transport.go +++ b/sandboxd/store/peer/transport.go @@ -20,7 +20,7 @@ import ( ) const ( - // Generous for a guest memory image, but a wedged peer must fail so the next owner is tried. + // generous for a guest memory image, but a wedged peer must fail so the next owner is tried. pullTimeout = 30 * time.Minute maxRecordBytes = 1 << 40 // 1 TiB @@ -42,7 +42,7 @@ type Puller interface { // HTTPPuller pulls records over sandboxd's control-plane HTTP port. type HTTPPuller struct { - // A nil Client uses a default with no overall timeout; the deadline comes from the context. + // a nil Client uses a default with no overall timeout; the deadline comes from the context. Client *http.Client Token string } @@ -167,7 +167,6 @@ func writeFile(target string, r io.Reader, mode os.FileMode) error { return f.Close() } -// checkpointURL builds the control-plane URL for id's record on addr, defaulting the scheme. func checkpointURL(addr, id, suffix string) string { if !strings.Contains(addr, "://") { addr = "http://" + addr diff --git a/sandboxd/store/s3/s3.go b/sandboxd/store/s3/s3.go index 9e3ad615..0c41fb62 100644 --- a/sandboxd/store/s3/s3.go +++ b/sandboxd/store/s3/s3.go @@ -179,7 +179,7 @@ func (s *Store) Delete(ctx context.Context, id string) error { if err != nil { return err } - // Keep the commit marker so a failed export cleanup remains discoverable on retry. + // keep the commit marker so a failed export cleanup remains discoverable on retry. keys = slices.DeleteFunc(keys, func(key string) bool { return key == metaKey }) if err := s.deleteKeys(ctx, keys); err != nil { return err @@ -315,7 +315,7 @@ func (s *Store) populate(ctx context.Context, id string, meta []byte, gen string return err } if len(keys) == 0 { - // Records published before per-generation prefixes. + // records published before per-generation prefixes. exportPrefix = s.key(id, store.ExportDir) + "/" if keys, err = s.list(ctx, exportPrefix); err != nil { return err diff --git a/sandboxd/store/storetest/storetest.go b/sandboxd/store/storetest/storetest.go index 50125215..e0cd3d60 100644 --- a/sandboxd/store/storetest/storetest.go +++ b/sandboxd/store/storetest/storetest.go @@ -52,7 +52,7 @@ func RunContract(t *testing.T, st store.Store) { } release() - // A half-published checkpoint (no meta) is invisible to Metas. + // a half-published checkpoint (no meta) is invisible to Metas. orphan, err := st.Stage("ck_00000000000000bb") if err != nil { t.Fatalf("Stage orphan: %v", err) diff --git a/sdk/go/client.go b/sdk/go/client.go index 89045ebf..29569058 100644 --- a/sdk/go/client.go +++ b/sdk/go/client.go @@ -139,9 +139,7 @@ func (c *Client) DeleteTemplate(ctx context.Context, template string, opts ...Op if err != nil || len(redirect) == 0 { return err } - // The entry node doesn't hold the template but gossip named its owners. - // The retry carries no_redirect, mirroring the claim protocol: the owner - // answers for itself, never a second hop. + // gossip named the template's owners; no_redirect makes the owner answer for itself, never a second hop u.Set(noRedirectQueryParam, "1") if tryErr := tryEach(redirect, func(addr string) error { _, retryErr := c.deleteTemplates(ctx, addr, u) @@ -216,9 +214,7 @@ func (c *Client) roundTrip(ctx context.Context, method, addr, path string, body return c.hc.Do(req) //nolint:gosec // dialing the caller-configured node is the SDK's purpose } -// doJSON issues one control-plane request and decodes a 200 reply into T; -// any other status maps through apiError under verb. The shared plumbing -// behind every decode-a-reply verb in this file and its siblings. +// doJSON issues one control-plane request and decodes a 200 reply into T; any other status maps through apiError under verb. func doJSON[T any](ctx context.Context, c *Client, method, addr, path string, body io.Reader, bearer, verb string) (T, error) { var out T resp, err := c.roundTrip(ctx, method, addr, path, body, bearer) @@ -256,10 +252,7 @@ func doNoContent(ctx context.Context, c *Client, method, addr, path string, body return nil } -// tryEach calls call against each candidate in turn, stopping at the first -// success. An error for which retry is true moves on to the next candidate -// and the last such error propagates; any other error returns at once. The -// per-verb retry policy mirrors the Python SDK's _try_each. +// tryEach walks candidates until call succeeds: an error retry accepts moves on (the last one propagates), any other returns at once. func tryEach(candidates []string, call func(addr string) error, retry func(error) bool) error { var lastErr error for _, addr := range candidates { @@ -281,12 +274,7 @@ func retryMiss(err error) bool { return !ok || he.Status == http.StatusNotFound } -// retryTransient reports whether an origin-fallback is worth the round-trip: -// true for a transport failure, a miss (404), full (429), mid-heal (503), an -// engine/proxy failure (500/502/504), or a mid-rotation 401 (the origin -// proved the token valid by issuing the redirect). A served 4xx like a bad -// request, a forbidden token, or an egress conflict is definitive: the -// origin would fail the same way. +// retryTransient reports whether origin can still answer differently; a served 4xx outside the list would fail there the same way. func retryTransient(err error) bool { he, ok := errors.AsType[*APIError](err) if !ok { @@ -306,9 +294,7 @@ type claimEncoder func(noRedirect, requirePromoted bool) ([]byte, error) type claimPoster func(addr string, body []byte) (claimResponse, error) -// claimFollow runs the claim protocol from origin: claim there, and on a -// redirect re-encode with no_redirect and follow via redirectFallback. Only -// the fallback error carries the verb — first-contact errors return raw. +// claimFollow claims at origin and follows a redirect through redirectFallback; only the fallback error carries the verb. func claimFollow(origin, verb string, encode claimEncoder, claimAt claimPoster) (string, claimResponse, error) { body, err := encode(false, false) if err != nil { diff --git a/sdk/go/info.go b/sdk/go/info.go index ba23c3c2..e46ed18a 100644 --- a/sdk/go/info.go +++ b/sdk/go/info.go @@ -68,8 +68,7 @@ func (c *Client) Sandboxes(ctx context.Context) ([]SandboxSummary, error) { return reply.Sandboxes, nil } -// peersOrErr fetches the node addresses, surfacing a discovery failure. It reads -// /v1/peers (tenant-accessible), so it works under a tenant token. +// peersOrErr reads /v1/peers, which a tenant token may call too, surfacing a discovery failure. func (c *Client) peersOrErr(ctx context.Context) ([]string, error) { ctx, cancel := context.WithTimeout(ctx, peersTimeout) defer cancel() diff --git a/sdk/go/port.go b/sdk/go/port.go index 81f233c5..345cc07d 100644 --- a/sdk/go/port.go +++ b/sdk/go/port.go @@ -148,8 +148,7 @@ func (s *Sandbox) openStream(ctx context.Context, req wire.Request) (*PortConn, func (s *Sandbox) proxyConn(ctx context.Context, local net.Conn, port uint16) { defer func() { _ = local.Close() }() - // The copy below can sit in local.Read forever; only closing the conn - // unblocks it, so ctx teardown must reach the local side too. + // the copy can sit in local.Read forever; only closing the conn unblocks it, so ctx teardown reaches local too unarm := context.AfterFunc(ctx, func() { _ = local.Close() }) defer unarm() guest, err := s.DialPort(ctx, port) diff --git a/sdk/go/sandbox.go b/sdk/go/sandbox.go index f786b1c0..a722d606 100644 --- a/sdk/go/sandbox.go +++ b/sdk/go/sandbox.go @@ -195,8 +195,7 @@ func (s *Sandbox) Close() error { return apiError("release", resp) } -// dial opens one relayed silkd connection and arms ctx cancellation to close -// it; the returned cleanup must be deferred. One connection carries one RPC. +// dial opens one relayed silkd connection, closed by ctx cancellation or the returned cleanup; one connection carries one RPC. func (s *Sandbox) dial(ctx context.Context) (*silkd.Conn, func(), error) { raw, err := s.c.dialAgent(ctx, s.owner, s.ID, s.token) if err != nil { diff --git a/sdk/go/silkd/silkdtest/fake.go b/sdk/go/silkd/silkdtest/fake.go index 9f5cf0ae..19b789c0 100644 --- a/sdk/go/silkd/silkdtest/fake.go +++ b/sdk/go/silkd/silkdtest/fake.go @@ -19,10 +19,7 @@ import ( "github.com/cocoonstack/sandbox/protocol/wire" ) -// Fake is a stateful silkd fake backing the fs verbs with a real directory -// and tracking sessions, so an SDK write-then-read round-trips through it. -// exec/info reuse the stateless handlers. It exists for host-side unit tests; -// the authoritative fs/session behavior is silkd's own Rust test suite. +// Fake is a stateful silkd fake: fs verbs run on a real directory and sessions are tracked, so a write-then-read round-trips. type Fake struct { Root string diff --git a/sdk/openai/cocoonsandbox_openai/adapter.py b/sdk/openai/cocoonsandbox_openai/adapter.py index 89333f34..86e67f11 100644 --- a/sdk/openai/cocoonsandbox_openai/adapter.py +++ b/sdk/openai/cocoonsandbox_openai/adapter.py @@ -60,26 +60,6 @@ def __init__(self, *, state: CocoonSandboxSessionState) -> None: def from_state(cls, state: CocoonSandboxSessionState) -> CocoonSandboxSession: return cls(state=state) - async def _prepare_backend_workspace(self) -> None: - sb = self._sandbox() - await asyncio.to_thread(sb.mkdir, str(self.state.manifest.root), parents=True) - - async def _exec_internal(self, *command: str | Path, timeout: float | None = None) -> ExecResult: - sb = self._sandbox() - argv = [str(part) for part in command] - if timeout is not None: - # the guest enforces the cutoff: a stream has no socket timeout, so a cancelled wait strands the worker - argv = ["timeout", "-s", "KILL", str(timeout), *argv] - stdout, stderr = bytearray(), bytearray() - - def run() -> int: - return sb.run(argv, on_stdout=stdout.extend, on_stderr=stderr.extend) - - code = await asyncio.to_thread(run) - if timeout is not None and code == TIMEOUT_EXIT: - raise TimeoutError(f"command did not finish within {timeout}s") - return ExecResult(stdout=bytes(stdout), stderr=bytes(stderr), exit_code=code) - async def read(self, path: Path, *, user: str | User | None = None) -> io.IOBase: _reject_user(user) sb = self._sandbox() @@ -113,6 +93,26 @@ async def hydrate_workspace(self, data: io.IOBase) -> None: sb = self._sandbox() await asyncio.to_thread(sb.push, str(self.state.manifest.root), data.read()) + async def _prepare_backend_workspace(self) -> None: + sb = self._sandbox() + await asyncio.to_thread(sb.mkdir, str(self.state.manifest.root), parents=True) + + async def _exec_internal(self, *command: str | Path, timeout: float | None = None) -> ExecResult: + sb = self._sandbox() + argv = [str(part) for part in command] + if timeout is not None: + # the guest enforces the cutoff: a stream has no socket timeout, so a cancelled wait strands the worker + argv = ["timeout", "-s", "KILL", str(timeout), *argv] + stdout, stderr = bytearray(), bytearray() + + def run() -> int: + return sb.run(argv, on_stdout=stdout.extend, on_stderr=stderr.extend) + + code = await asyncio.to_thread(run) + if timeout is not None and code == TIMEOUT_EXIT: + raise TimeoutError(f"command did not finish within {timeout}s") + return ExecResult(stdout=bytes(stdout), stderr=bytes(stderr), exit_code=code) + async def _resolve_exposed_port(self, port: int) -> ExposedPortEndpoint: sb = self._sandbox() listener = await asyncio.to_thread(sb.proxy_port, "127.0.0.1:0", port) diff --git a/sdk/openai/e2e.py b/sdk/openai/e2e.py index 6863d671..0925723d 100644 --- a/sdk/openai/e2e.py +++ b/sdk/openai/e2e.py @@ -45,7 +45,6 @@ async def main() -> int: assert tar and len(tar) > 0 print(f" persist_workspace: {len(tar)} tar bytes captured") - # Resume from the serialized state must reattach to the same sandbox. payload = client.serialize_session_state(session._inner.state) resumed = await client.resume(client.deserialize_session_state(payload)) async with resumed: diff --git a/sdk/python/cocoonsandbox/conn.py b/sdk/python/cocoonsandbox/conn.py index f267dd3e..f694190b 100644 --- a/sdk/python/cocoonsandbox/conn.py +++ b/sdk/python/cocoonsandbox/conn.py @@ -31,7 +31,7 @@ def __init__(self, sock: socket.socket, reader: BinaryIO): self._sock = sock self._reader = reader - def send(self, op: str, **fields) -> None: + def send(self, op: str, **fields: object) -> None: self._sock.sendall(encode_request(op, **fields)) def recv(self) -> dict: @@ -118,7 +118,7 @@ def dial_agent(addr: str, sandbox_id: str, token: str, timeout: float) -> Conn: except ValueError as exc: raise ProtocolError("invalid content-length in upgrade reply") from exc if code != 101: - # A bogus content-length must not buffer unbounded bytes. + # a bogus content-length must not buffer unbounded bytes. body = reader.read(min(body_len, MAX_FRAME)).decode(errors="replace") if body_len else "" raise APIError("agent upgrade", code, body.strip() or status.strip()) sock.settimeout(None) diff --git a/sdk/python/cocoonsandbox/frames.py b/sdk/python/cocoonsandbox/frames.py index 68760364..36bd39a1 100644 --- a/sdk/python/cocoonsandbox/frames.py +++ b/sdk/python/cocoonsandbox/frames.py @@ -10,12 +10,12 @@ PROTO_VERSION = 1 MAX_FRAME = 8 * 1024 * 1024 -FS_CHUNK = 256 * 1024 # silkd's per-frame chunk size, distinct from BULK_CHUNK below -# bulk streams chunk larger: fewer frames per byte, still under MAX_FRAME after base64. +FS_CHUNK = 256 * 1024 +# bulk streams chunk larger than silkd's FS_CHUNK: fewer frames per byte, still under MAX_FRAME after base64. BULK_CHUNK = 1 << 20 -def encode_request(op: str, **fields) -> bytes: +def encode_request(op: str, **fields: object) -> bytes: """Renders {"v":1,"op":...,fields} with a trailing newline; None fields are omitted and bytes values ride base64 under their field name.""" frame = {"v": PROTO_VERSION, "op": op} diff --git a/sdk/python/e2e.py b/sdk/python/e2e.py index af87f983..8f18b8b0 100644 --- a/sdk/python/e2e.py +++ b/sdk/python/e2e.py @@ -104,7 +104,7 @@ def _hibernate(): sess = sb.session() sess.exec("export", "HIB=alive") sb.hibernate() - assert sess.exec("sh", "-c", "echo $HIB") == "alive\n" # transparent wake + assert sess.exec("sh", "-c", "echo $HIB") == "alive\n" sess.close() @step("fork") @@ -128,8 +128,8 @@ def _checkpoint(): sb.write_file("/root/ck.txt", b"v2") branch = ckpt.new() try: - assert branch.read_file("/root/ck.txt") == b"v1" # captured moment - assert sb.read_file("/root/ck.txt") == b"v2" # source unaffected + assert branch.read_file("/root/ck.txt") == b"v1" + assert sb.read_file("/root/ck.txt") == b"v2" finally: branch.close() listed = [c.id for c in client.checkpoints()] @@ -173,7 +173,7 @@ def _port(): raise print(f" {name:<18} ok ({int((time.time() - start) * 1000)}ms)") finally: - # Never mask a step failure with a close error. + # never mask a step failure with a close error. with contextlib.suppress(Exception): sb.close() print("PY-E2E PASS") diff --git a/sdk/python/tests/test_hardening.py b/sdk/python/tests/test_hardening.py index a9161831..26f5dcf0 100644 --- a/sdk/python/tests/test_hardening.py +++ b/sdk/python/tests/test_hardening.py @@ -66,6 +66,7 @@ def serve(): conn, _ = server.accept() conn.recv(4096) conn.sendall(b"HTTP/1.1 500 nope\r\nContent-Length: -1\r\n\r\nboom") + conn.close() threading.Thread(target=serve, daemon=True).start() try: diff --git a/sdk/python/tests/test_stream.py b/sdk/python/tests/test_stream.py index b0873e21..6964d61b 100644 --- a/sdk/python/tests/test_stream.py +++ b/sdk/python/tests/test_stream.py @@ -13,13 +13,13 @@ TIMEOUT = 0.2 -def serve_port_forward(server: socket.socket, quiet: float) -> None: +def serve_port_forward(server: socket.socket, quiet: float, ops: list[str]) -> None: conn, _ = server.accept() reader = conn.makefile("rb") while reader.readline() not in (b"\r\n", b""): pass conn.sendall(b"HTTP/1.1 101 Switching Protocols\r\n\r\n") - assert json.loads(reader.readline())["op"] == "port_forward" + ops.append(json.loads(reader.readline())["op"]) conn.sendall(b'{"type":"ready"}\n') time.sleep(quiet) conn.sendall(b'{"type":"data","data":"bGF0ZQ=="}\n{"type":"done"}\n') @@ -36,7 +36,8 @@ def serve_silence(server: socket.socket) -> None: def test_port_stream_outlives_the_client_timeout(): server = socket.create_server(("127.0.0.1", 0)) addr = f"127.0.0.1:{server.getsockname()[1]}" - threading.Thread(target=serve_port_forward, args=(server, 3 * TIMEOUT), daemon=True).start() + ops: list[str] = [] + threading.Thread(target=serve_port_forward, args=(server, 3 * TIMEOUT, ops), daemon=True).start() sb = Sandbox(client=Client(addr, timeout=TIMEOUT), id="sb_1", token="tok", owner=addr) try: with sb.dial_port(5000) as port: @@ -44,6 +45,7 @@ def test_port_stream_outlives_the_client_timeout(): assert port.recv() == b"" finally: server.close() + assert ops == ["port_forward"] def test_dial_is_still_bounded_by_the_client_timeout(): diff --git a/sdk/python/tests/test_wire_binding.py b/sdk/python/tests/test_wire_binding.py index 0622c826..91041457 100644 --- a/sdk/python/tests/test_wire_binding.py +++ b/sdk/python/tests/test_wire_binding.py @@ -164,7 +164,7 @@ def fake_sandbox(monkeypatch, replies): def guest(): reader = guest_sock.makefile("rb") - with contextlib.suppress(Exception): + with contextlib.suppress(OSError, ValueError): line = reader.readline() if line: sent.append(json.loads(line)) diff --git a/silkd/src/find.rs b/silkd/src/find.rs index 0b7855ef..922e8bd1 100644 --- a/silkd/src/find.rs +++ b/silkd/src/find.rs @@ -221,7 +221,7 @@ where Some(Ok(re)) => Some(re), Some(Err(e)) => return proto::error_frame(w, ErrorKind::BadRequest, e.to_string()).await, }; - // One blocking-pool dispatch for the whole tree, not three per file. + // one blocking-pool dispatch for the whole tree, not three per file. let (tx, mut rx) = mpsc::channel::(MATCH_QUEUE); let budget = Arc::new(MatchBudget::new(budget_bytes, Handle::current())); let walker_budget = budget.clone(); @@ -241,7 +241,7 @@ where loop { tokio::select! { biased; - // The client sends nothing during a find, so any readable state ends the walk. + // the client sends nothing during a find, so any readable state ends the walk. _ = reader.fill_buf() => { gone = true; break; diff --git a/silkd/src/fs.rs b/silkd/src/fs.rs index 511ef93e..f989ac76 100644 --- a/silkd/src/fs.rs +++ b/silkd/src/fs.rs @@ -62,7 +62,7 @@ pub async fn read(w: &mut W, path: String) -> io::Result< /// Lists a directory as `entries` frames terminated by `done`, batched under the frame cap. pub async fn list(w: &mut W, path: String) -> io::Result<()> { - // One blocking-pool dispatch for the whole directory, not one per entry. + // one blocking-pool dispatch for the whole directory, not one per entry. let (tx, mut rx) = mpsc::channel::>(2); let scan = tokio::task::spawn_blocking(move || scan_dir(&path, &tx)); let mut failed = None; diff --git a/silkd/src/watch.rs b/silkd/src/watch.rs index a7ae2151..145891ef 100644 --- a/silkd/src/watch.rs +++ b/silkd/src/watch.rs @@ -46,13 +46,13 @@ where loop { tokio::select! { n = rx.recv_many(&mut batch, EVENT_BATCH) => { - // Only overflow drops the sender mid-watch; the buffered prefix is already out. + // only overflow drops the sender mid-watch; the buffered prefix is already out. if n == 0 { return proto::error_frame(w, ErrorKind::Internal, OVERFLOW_MESSAGE).await; } let terminal = batch.iter().position(|f| matches!(f, Response::Error { .. })); let upto = terminal.map_or(batch.len(), |i| i + 1); - // A failed write is the disconnect the EOF arm can lose the select to. + // a failed write is the disconnect the EOF arm can lose the select to. let sent = proto::write_frames(w, &mut buf, &batch[..upto]).await; batch.clear(); if sent.is_err() || terminal.is_some() { From 0bc4ded7aa3bf8aa7795b3fa685a40d5924f92f1 Mon Sep 17 00:00:00 2001 From: CMGS Date: Mon, 14 Sep 2026 19:59:43 +0800 Subject: [PATCH 11/40] pool: refuse capture verbs on an archived claim; keep rollbacks and egress doors off a released claim Checkpoint, Promote and Fork of an archived claim ran the engine with an empty VM name and Hibernate journaled a bogus intent onto it; they now answer ErrArchived (409) and Hibernate is the no-op it already is for a hibernated claim. archive() and commitWake re-set the record on a failed commit even when a concurrent Release had deleted it, which resurrected the claim in claims.json; the rollback is now gated on the claim still being live. armEgressProxy closes the listener it displaces, and resyncEgress leaves hibernated and archived claims to their wake, so a guarded claim woken after a restart no longer leaks its startup door; finalizeBatch disarms a claim released in its own arm window. Also: an oversized request frame records as op oversized instead of escaping the audit journal; the pool passes no CA to a proxy whose pool never intercepts (one Transport clone and a leaf map less per claim); Sandboxes sorts outside m.mu; idleEnabled/archiveEnabled are atomics (read off m.mu by the sweeps); ClaimWarm kicks the refill before it answers ErrNoWarm; sweepExpiredCheckpoints owns its TTL guard; DeleteCheckpoint asks archiveCkPinned; quiesce gives the fallback sync its own budget; a persisted pool that fails validation names pools.json. --- sandboxd/pool/archive.go | 21 ++++++++++++------- sandboxd/pool/archive_test.go | 39 +++++++++++++++++++++++++++++++++++ sandboxd/pool/checkpoint.go | 7 +++++-- sandboxd/pool/claim.go | 6 +++++- sandboxd/pool/egress.go | 14 +++++++++++-- sandboxd/pool/fork.go | 3 +++ sandboxd/pool/hibernate.go | 5 ++++- sandboxd/pool/idle_test.go | 2 +- sandboxd/pool/pool.go | 24 ++++++++++----------- sandboxd/pool/pool_test.go | 2 +- sandboxd/pool/poolstore.go | 4 ++-- sandboxd/pool/reconcile.go | 10 +++++++-- sandboxd/pool/telemetry.go | 7 ++++++- sandboxd/pool/template.go | 3 +++ sandboxd/pool/volume.go | 5 ++++- 15 files changed, 118 insertions(+), 34 deletions(-) diff --git a/sandboxd/pool/archive.go b/sandboxd/pool/archive.go index d1f79ede..36092669 100644 --- a/sandboxd/pool/archive.go +++ b/sandboxd/pool/archive.go @@ -39,7 +39,7 @@ func (m *Manager) archiveEnabledFor(key types.PoolKey) bool { // archiveOnce checkpoints hibernated claims idle past their archive threshold and drops their VM. func (m *Manager) archiveOnce(ctx context.Context) { - if !m.archiveEnabled { + if !m.archiveEnabled.Load() { return } if !m.archiveSweep.CompareAndSwap(false, true) { @@ -121,10 +121,14 @@ func (m *Manager) archive(ctx context.Context, sb *types.Sandbox) error { if saveErr := m.store.commit(js); saveErr != nil { // roll back so memory matches the still-durable hibernated record; drop the orphan ck. m.mu.Lock() - sb.ArchiveCk = "" - sb.HibernateSnap, sb.VsockSocket, sb.VMName = snap, sock, vmName - sb.Deadline = prevDeadline - rb := m.store.set(sb) + var rb claimSnapshot + // a Release that landed meanwhile already deleted the claim; re-setting it would resurrect it + if m.claimed[sb.ID] == sb { + sb.ArchiveCk = "" + sb.HibernateSnap, sb.VsockSocket, sb.VMName = snap, sock, vmName + sb.Deadline = prevDeadline + rb = m.store.set(sb) + } m.mu.Unlock() sb.Transition.Unlock() m.recommit(ctx, rb) @@ -222,8 +226,11 @@ func (m *Manager) commitWake(ctx context.Context, sb *types.Sandbox, vmName, soc m.mu.Unlock() if err := m.store.commit(js); err != nil { m.mu.Lock() - sb.VMName, sb.VsockSocket, sb.ArchiveCk, sb.Deadline = "", "", ck, deadline - rb := m.store.set(sb) + var rb claimSnapshot + if m.claimed[sb.ID] == sb { + sb.VMName, sb.VsockSocket, sb.ArchiveCk, sb.Deadline = "", "", ck, deadline + rb = m.store.set(sb) + } m.mu.Unlock() m.recommit(ctx, rb) log.WithFunc("pool.commitWake").Warnf(ctx, "persist claims: %v", err) diff --git a/sandboxd/pool/archive_test.go b/sandboxd/pool/archive_test.go index 19ec8a0c..f21f0b2d 100644 --- a/sandboxd/pool/archive_test.go +++ b/sandboxd/pool/archive_test.go @@ -234,6 +234,45 @@ func TestArchiveLifecycleRoundTrip(t *testing.T) { }) } +func TestArchivedClaimRefusesCaptureVerbs(t *testing.T) { + eng := newFakeEngine() + m := newTestManager(t, eng, archivePool(3600)) + sb := mustClaim(t, m, testKey) + mustArchive(t, m, sb) + cred := Cred{Token: sb.Token} + + if _, err := m.Checkpoint(t.Context(), sb.ID, cred, "", ""); !errors.Is(err, ErrArchived) { + t.Errorf("Checkpoint = %v, want ErrArchived", err) + } + if _, _, err := m.Promote(t.Context(), sb.ID, cred, "tpl", ""); !errors.Is(err, ErrArchived) { + t.Errorf("Promote = %v, want ErrArchived", err) + } + if _, err := m.Fork(t.Context(), sb.ID, cred, 1, 0); !errors.Is(err, ErrArchived) { + t.Errorf("Fork = %v, want ErrArchived", err) + } + if err := m.Hibernate(t.Context(), sb.ID, cred); err != nil { + t.Errorf("Hibernate of an archived claim = %v, want the no-op success", err) + } + if sb.PendingSnap != "" { + t.Errorf("Hibernate journaled intent %q onto an archived claim", sb.PendingSnap) + } +} + +func TestSweepExpiredCheckpointsIsANoopWithoutTTL(t *testing.T) { + eng := newFakeEngine() + m := newTestManager(t, eng, archivePool(3600)) + sb := mustClaim(t, m, testKey) + ck, err := m.Checkpoint(t.Context(), sb.ID, Cred{Token: sb.Token}, "", "") + if err != nil { + t.Fatalf("Checkpoint: %v", err) + } + m.ckptTTL = 0 + m.sweepExpiredCheckpoints(t.Context()) + if !ckExists(t, m, ck.ID) { + t.Fatal("a zero TTL sweep deleted a checkpoint") + } +} + func TestArchiveTTLSweepSparesLiveCheckpoint(t *testing.T) { eng := newFakeEngine() m := newTestManager(t, eng, archivePool(3600)) diff --git a/sandboxd/pool/checkpoint.go b/sandboxd/pool/checkpoint.go index 5bbfd7f6..6afbece4 100644 --- a/sandboxd/pool/checkpoint.go +++ b/sandboxd/pool/checkpoint.go @@ -46,6 +46,9 @@ func (m *Manager) Checkpoint(ctx context.Context, id string, cred Cred, name, te if !sb.Key.Capturable() { return types.Checkpoint{}, ErrNoEgressFork } + if sb.ArchiveCk != "" { + return types.Checkpoint{}, ErrArchived + } // See Hibernate: a started capture must finish even if the caller hangs up. ctx = context.WithoutCancel(ctx) ckpt, _, err := m.publishCheckpoint(ctx, sb, store.CheckpointID(randHex(8)), name, tenant, false) @@ -126,7 +129,7 @@ func (m *Manager) DeleteCheckpoint(ctx context.Context, ckptID, tenant string, s return ErrUnknownCheckpoint } // ckpt.Archive guards wake images store-wide; the pin set guards uncommitted local ones - if _, pinned := m.pinnedArchiveCks()[ckptID]; pinned || ckpt.Archive { + if ckpt.Archive || m.archiveCkPinned(ckptID) { return ErrUnknownCheckpoint // backs an archived sandbox, not a deletable checkpoint } if err := m.ckpts.Delete(ctx, ckptID); err != nil { @@ -334,7 +337,7 @@ func (m *Manager) pinnedArchiveCks() map[string]struct{} { // sweepExpiredCheckpoints ages out checkpoints older than the configured TTL. func (m *Manager) sweepExpiredCheckpoints(ctx context.Context) { - if !m.ckptSweeping.CompareAndSwap(false, true) { + if m.ckptTTL <= 0 || !m.ckptSweeping.CompareAndSwap(false, true) { return } defer m.ckptSweeping.Store(false) diff --git a/sandboxd/pool/claim.go b/sandboxd/pool/claim.go index d64e7a4e..490ba910 100644 --- a/sandboxd/pool/claim.go +++ b/sandboxd/pool/claim.go @@ -49,10 +49,10 @@ func (m *Manager) ClaimWarm(ctx context.Context, key types.PoolKey, ttl time.Dur } } m.mu.Unlock() + m.kickRefill() if sb == nil { return nil, ErrNoWarm } - m.kickRefill() if volumeErr := m.applyVolumes(ctx, sb, volumeSpecs, applied); volumeErr != nil { m.abortVolumeClaim(ctx, sb.VMName, &reserved) return nil, volumeErr @@ -295,6 +295,10 @@ func (m *Manager) finalizeBatch(ctx context.Context, sbs []*types.Sandbox, ttl t m.rollbackClaim(ctx, sbs) return fmt.Errorf("arm egress %s: %w", sb.ID, armErr) } + // the claim was visible before it was armed, so a release in that window must find nothing left behind + if m.guardedEgress || m.lockEgress { + m.disarmIfReleased(sb) + } } // usage lands only after the batch armed, so a rollback leaves no unterminated claim event for _, sb := range sbs { diff --git a/sandboxd/pool/egress.go b/sandboxd/pool/egress.go index 93ede798..7668416e 100644 --- a/sandboxd/pool/egress.go +++ b/sandboxd/pool/egress.go @@ -152,12 +152,18 @@ func (m *Manager) armEgressProxy(ctx context.Context, sb *types.Sandbox) error { } id, tenant := sb.ID, sb.Tenant evCtx := context.WithoutCancel(ctx) - proxy := egress.New(id, tenant, policy, m.egressSecrets, m.egressCA, m.dial, - func(ev egress.Event) { m.recordEgress(evCtx, id, tenant, ev) }, sb) m.mu.Lock() el := m.egressPrebound[sb.VMName] delete(m.egressPrebound, sb.VMName) + intercepts := m.poolEgress[sb.Key].Intercepts() m.mu.Unlock() + // only an intercepting pool pays for the TLS transport and leaf cache + ca := m.egressCA + if !intercepts { + ca = nil + } + proxy := egress.New(id, tenant, policy, m.egressSecrets, ca, m.dial, + func(ev egress.Event) { m.recordEgress(evCtx, id, tenant, ev) }, sb) if el != nil && (reached(el.ln) || (el.socks != nil && reached(el.socks))) { el.close() el = nil @@ -171,8 +177,12 @@ func (m *Manager) armEgressProxy(ctx context.Context, sb *types.Sandbox) error { el.srv = &http.Server{Handler: proxy, ReadHeaderTimeout: 30 * time.Second} el.proxy = proxy m.mu.Lock() + displaced := m.egressListeners[id] m.egressListeners[id] = el m.mu.Unlock() + if displaced != nil { + displaced.close() + } go func() { _ = el.srv.Serve(el.ln) }() if el.socks != nil { go proxy.ServeSOCKS(evCtx, el.socks) diff --git a/sandboxd/pool/fork.go b/sandboxd/pool/fork.go index 06462f42..b3ec96e8 100644 --- a/sandboxd/pool/fork.go +++ b/sandboxd/pool/fork.go @@ -25,6 +25,9 @@ func (m *Manager) Fork(ctx context.Context, id string, cred Cred, count int, ttl if !sb.Key.Capturable() { return nil, ErrNoEgressFork } + if sb.ArchiveCk != "" { + return nil, ErrArchived + } if err := m.overQuota(count, sb.Tenant); err != nil { return nil, err } diff --git a/sandboxd/pool/hibernate.go b/sandboxd/pool/hibernate.go index 731f3d6e..c67bcabd 100644 --- a/sandboxd/pool/hibernate.go +++ b/sandboxd/pool/hibernate.go @@ -41,6 +41,9 @@ func (m *Manager) WakeAgentSocket(ctx context.Context, id, token string) (string // hibernateLocked is Hibernate's body; the caller holds sb.Transition. func (m *Manager) hibernateLocked(ctx context.Context, sb *types.Sandbox) error { + if sb.ArchiveCk != "" { + return nil // already off the node + } if hasAppliedVolumes(sb) { return ErrVolumeCapture } @@ -176,7 +179,7 @@ func (m *Manager) wakeResolved(ctx context.Context, sb *types.Sandbox) (string, // idleOnce hibernates claims idle past their pool's (or the node's) threshold. func (m *Manager) idleOnce(ctx context.Context) { - if !m.idleEnabled { + if !m.idleEnabled.Load() { return } if !m.idleSweep.CompareAndSwap(false, true) { diff --git a/sandboxd/pool/idle_test.go b/sandboxd/pool/idle_test.go index 95eb6a56..6b3e109f 100644 --- a/sandboxd/pool/idle_test.go +++ b/sandboxd/pool/idle_test.go @@ -43,7 +43,7 @@ func TestIdleOncePolicyScope(t *testing.T) { eng := newFakeEngine() m := newTestManager(t, eng, config.PoolSpec{PoolKey: testKey, Warm: 1}) m.idleDefault = time.Second - m.idleEnabled = true + m.idleEnabled.Store(true) sb := mustClaim(t, m, testKey) backdate(m, sb, time.Hour) diff --git a/sandboxd/pool/pool.go b/sandboxd/pool/pool.go index 4ae885fe..41e3961f 100644 --- a/sandboxd/pool/pool.go +++ b/sandboxd/pool/pool.go @@ -94,6 +94,7 @@ var ( ErrVolumeBusy = errors.New("volume is held by another claim") // replaying a journal takes a writable mount, so readers stay out. ErrVolumeNeedsRecovery = errors.New("volume needs recovery by a writable claim") + ErrArchived = errors.New("sandbox is archived; an exec or file call wakes it first") ErrQuota = errors.New("node claim quota reached") errWokeMeanwhile = errors.New("woke between sweep and hibernate") @@ -255,13 +256,13 @@ type Manager struct { // idleDefault is the idle-hibernate threshold for unpooled keys; zero disables. idleDefault time.Duration - idleEnabled bool + idleEnabled atomic.Bool idleSweep atomic.Bool // archive*Default are the archive thresholds for unpooled keys. archiveAfterDefault time.Duration archiveDeleteDefault time.Duration - archiveEnabled bool + archiveEnabled atomic.Bool archiveSweep atomic.Bool archiveDeleteSweep atomic.Bool // archiving holds ids with an export in flight; pendingCks pins ids mid-commit; both under m.mu. @@ -480,9 +481,7 @@ func (m *Manager) Run(ctx context.Context) { // store retention is hourly: each sweep is cluster-visible I/O on a shared root storeSweep := time.NewTicker(time.Hour) defer storeSweep.Stop() - if m.ckptTTL > 0 { - m.sweepExpiredCheckpoints(ctx) - } + m.sweepExpiredCheckpoints(ctx) m.refillOnce(ctx) for { select { @@ -500,9 +499,7 @@ func (m *Manager) Run(ctx context.Context) { go m.retryArchiveDeletes(ctx) case <-storeSweep.C: m.sweepStoreGenerations(ctx) - if m.ckptTTL > 0 { - m.sweepExpiredCheckpoints(ctx) - } + m.sweepExpiredCheckpoints(ctx) } } } @@ -553,7 +550,6 @@ func (m *Manager) Info() ([]PoolInfo, Gauges) { // Sandboxes lists live claims visible to tenant; empty tenant means root. func (m *Manager) Sandboxes(tenant string) []SandboxSummary { m.mu.Lock() - defer m.mu.Unlock() out := make([]SandboxSummary, 0, len(m.claimed)) for _, sb := range m.claimed { if !tenantOwns(tenant, sb.Tenant) { @@ -561,6 +557,7 @@ func (m *Manager) Sandboxes(tenant string) []SandboxSummary { } out = append(out, summarize(sb)) } + m.mu.Unlock() slices.SortFunc(out, func(a, b SandboxSummary) int { return strings.Compare(a.ID, b.ID) }) return out } @@ -603,12 +600,13 @@ func (m *Manager) sweepStoreGenerations(ctx context.Context) { // recomputeSweepFlags derives the sweep switches from the live pool set rather than latching them: removing every idle pool turns the sweep off again. func (m *Manager) recomputeSweepFlags() { - m.idleEnabled = m.idleDefault > 0 - m.archiveEnabled = m.archiveAfterDefault > 0 + idle, archive := m.idleDefault > 0, m.archiveAfterDefault > 0 for _, p := range m.pools { - m.idleEnabled = m.idleEnabled || p.idle > 0 - m.archiveEnabled = m.archiveEnabled || p.archiveAfter > 0 + idle = idle || p.idle > 0 + archive = archive || p.archiveAfter > 0 } + m.idleEnabled.Store(idle) + m.archiveEnabled.Store(archive) } func (m *Manager) untrack(set map[string]struct{}, key string) { diff --git a/sandboxd/pool/pool_test.go b/sandboxd/pool/pool_test.go index 95712d8e..ffd29c28 100644 --- a/sandboxd/pool/pool_test.go +++ b/sandboxd/pool/pool_test.go @@ -460,7 +460,7 @@ func TestSetPoolsRemovalShedsArchivePolicy(t *testing.T) { if p.archiveAfter != 0 || p.archiveDelete != 0 { t.Errorf("archive policy survived removal: after=%v delete=%v, want 0", p.archiveAfter, p.archiveDelete) } - if m.archiveEnabled { + if m.archiveEnabled.Load() { t.Error("archiveEnabled still latched by a removed pool") } p.refilling = 0 diff --git a/sandboxd/pool/poolstore.go b/sandboxd/pool/poolstore.go index 40c86546..acd6a053 100644 --- a/sandboxd/pool/poolstore.go +++ b/sandboxd/pool/poolstore.go @@ -90,10 +90,10 @@ func (m *Manager) adoptPersistedPools(ctx context.Context) error { for _, spec := range pf.Pools { spec = normalizePoolSpec(spec) if err := m.validate(spec.PoolKey); err != nil { - return fmt.Errorf("restore pool %q: %w", spec.Template, err) + return fmt.Errorf("restore pool %q from %s: %w", spec.Template, m.poolStore.path, err) } if err := spec.ValidateLimits(); err != nil { - return fmt.Errorf("restore pool %q: %w", spec.Template, err) + return fmt.Errorf("restore pool %q from %s: %w", spec.Template, m.poolStore.path, err) } p := newPool(spec.PoolKey) p.applySpec(spec) diff --git a/sandboxd/pool/reconcile.go b/sandboxd/pool/reconcile.go index 58467dba..be9f8788 100644 --- a/sandboxd/pool/reconcile.go +++ b/sandboxd/pool/reconcile.go @@ -91,8 +91,10 @@ func (m *Manager) Reconcile(ctx context.Context) error { logger.Warnf(ctx, "snapshot sweep skipped: %v", snapsErr) } else { orphans := slices.DeleteFunc(snaps, func(snap string) bool { - return strings.HasPrefix(snap, hibernatePrefix) && referenced[snap] || - !strings.HasPrefix(snap, hibernatePrefix) && !strings.HasPrefix(snap, forkPrefix) && !strings.HasPrefix(snap, goldenPrefix) + if strings.HasPrefix(snap, hibernatePrefix) { + return referenced[snap] + } + return !strings.HasPrefix(snap, forkPrefix) && !strings.HasPrefix(snap, goldenPrefix) }) m.runBounded(ctx, len(orphans), func(ctx context.Context, i int) { m.dropSnap(ctx, orphans[i]) @@ -162,6 +164,10 @@ func (m *Manager) resyncEgress(ctx context.Context, live map[string]types.VMReco var quarantine []*types.Sandbox for _, sb := range m.claimed { sb.TouchAt(now) + // a hibernated or archived claim has no guest to serve; its wake arms the door + if sb.HibernateSnap != "" || sb.ArchiveCk != "" { + continue + } if m.locksNIC(sb.Key) { tap := m.readoptEgressTap(sb, live) if tap == "" { diff --git a/sandboxd/pool/telemetry.go b/sandboxd/pool/telemetry.go index 4ddccd33..64caa7e8 100644 --- a/sandboxd/pool/telemetry.go +++ b/sandboxd/pool/telemetry.go @@ -95,7 +95,12 @@ func (m *Manager) TenantClaims() map[string]int { // Audit records one relayed request frame against a sandbox: the op and addressing fields only. func (m *Manager) Audit(ctx context.Context, id string, line []byte) { - if m.audit == nil || len(line) > AuditLineCap { + if m.audit == nil { + return + } + if len(line) > AuditLineCap { + // the addressing fields stay unread at this size, but a padded frame must not escape the journal + m.recordAudit(ctx, id, auditFrame{Op: "oversized"}) return } var frame auditFrame diff --git a/sandboxd/pool/template.go b/sandboxd/pool/template.go index d9f0ca6e..2076d784 100644 --- a/sandboxd/pool/template.go +++ b/sandboxd/pool/template.go @@ -37,6 +37,9 @@ func (m *Manager) Promote(ctx context.Context, id string, cred Cred, template, t if !sb.Key.Capturable() { return types.PoolKey{}, "", ErrNoEgressFork } + if sb.ArchiveCk != "" { + return types.PoolKey{}, "", ErrArchived + } key := types.PoolKey{Template: template, Net: sb.Key.Net, Size: sb.Key.Size} if m.pooled(key) { // a configured pool owns this key; promoting over it would change what refills produce diff --git a/sandboxd/pool/volume.go b/sandboxd/pool/volume.go index 0ec38f55..ca38660c 100644 --- a/sandboxd/pool/volume.go +++ b/sandboxd/pool/volume.go @@ -238,7 +238,10 @@ func (m *Manager) quiesceVolumes(ctx context.Context, sb *types.Sandbox) volumeT } } if stuck { - if err := m.eng.SyncGuest(ctx, sb.VsockSocket); err != nil { + // the unmounts may have spent the whole budget; the fallback flush gets its own + syncCtx, cancelSync := context.WithTimeout(context.WithoutCancel(ctx), engine.VolumeCallTimeout) + defer cancelSync() + if err := m.eng.SyncGuest(syncCtx, sb.VsockSocket); err != nil { logger.Errorf(ctx, err, "sync guest of %s", sb.ID) } } From 87ffe99ea0a5f26fec490e6c96b4f1acb5915b3f Mon Sep 17 00:00:00 2001 From: CMGS Date: Mon, 14 Sep 2026 19:59:43 +0800 Subject: [PATCH 12/40] egress: keep the upstream connection for a closing guest; keep a clamped leaf cached r.Clone carried the guest's Connection: close into the upstream request, so an HTTP/1.0 guest or one sending the header paid a fresh upstream connect (and TLS handshake on the intercept path) per request; the relay clears out.Close like httputil.ReverseProxy does. A leaf whose end is clamped to the intermediate's own could never be renewed, so once the intermediate entered its last 24 h every intercepted session re-signed; such a leaf stays cached. --- sandboxd/egress/intercept.go | 3 ++- sandboxd/egress/intercept_test.go | 18 ++++++++++++++++++ sandboxd/egress/proxy.go | 1 + sandboxd/egress/proxy_test.go | 26 ++++++++++++++++++++++++++ 4 files changed, 47 insertions(+), 1 deletion(-) diff --git a/sandboxd/egress/intercept.go b/sandboxd/egress/intercept.go index 707368d2..8a6ecaee 100644 --- a/sandboxd/egress/intercept.go +++ b/sandboxd/egress/intercept.go @@ -64,7 +64,8 @@ func (p *Proxy) serveIntercept(w http.ResponseWriter, r *http.Request, host stri func (p *Proxy) leafFor(host string) (*tls.Certificate, error) { p.leafMu.Lock() defer p.leafMu.Unlock() - if crt, ok := p.leaves[host]; ok && time.Now().Before(crt.Leaf.NotAfter.Add(-leafRenewBefore)) { + // a leaf clamped to the intermediate's own end has nothing to renew into + if crt, ok := p.leaves[host]; ok && (time.Now().Before(crt.Leaf.NotAfter.Add(-leafRenewBefore)) || !crt.Leaf.NotAfter.Before(p.ca.interCert.NotAfter)) { return crt, nil } crt, err := p.ca.SignLeaf(host) diff --git a/sandboxd/egress/intercept_test.go b/sandboxd/egress/intercept_test.go index 32da193c..d2ecd8e3 100644 --- a/sandboxd/egress/intercept_test.go +++ b/sandboxd/egress/intercept_test.go @@ -11,6 +11,7 @@ import ( "net/http/httptest" "strings" "testing" + "time" ) func TestInterceptLeafCaches(t *testing.T) { @@ -29,6 +30,23 @@ func TestInterceptLeafCaches(t *testing.T) { } } +func TestInterceptLeafClampedToTheIntermediateStaysCached(t *testing.T) { + ca, _ := testCA(t) + ca.interCert.NotAfter = time.Now().Add(time.Hour).Truncate(time.Second) + p := New("s", "", Policy{}, nil, ca, fixedDial("127.0.0.1:1"), nil, nil) + a, err := p.leafFor("example.com") + if err != nil { + t.Fatalf("leafFor: %v", err) + } + b, err := p.leafFor("example.com") + if err != nil { + t.Fatalf("leafFor again: %v", err) + } + if a != b { + t.Error("leaf re-signed although its end is the intermediate's own") + } +} + func TestInterceptLeafCacheBounded(t *testing.T) { ca, _ := testCA(t) p := New("s", "", Policy{}, nil, ca, fixedDial("127.0.0.1:1"), nil, nil) diff --git a/sandboxd/egress/proxy.go b/sandboxd/egress/proxy.go index 49d2bcdc..0b9d5885 100644 --- a/sandboxd/egress/proxy.go +++ b/sandboxd/egress/proxy.go @@ -235,6 +235,7 @@ func (p *Proxy) relay(w http.ResponseWriter, r *http.Request, ev Event, rule Rul prepare(out) } out.RequestURI = "" + out.Close = false // the guest's Connection: close is its own; the upstream pool keeps the conn stripHop(out.Header) ev.Injected = p.inject(rule, out.Header) p.record(ev) diff --git a/sandboxd/egress/proxy_test.go b/sandboxd/egress/proxy_test.go index 7dd6a02a..6c71ef6f 100644 --- a/sandboxd/egress/proxy_test.go +++ b/sandboxd/egress/proxy_test.go @@ -52,6 +52,32 @@ func TestForwardAllowInjectsSecretAndOverwritesGuestHeader(t *testing.T) { } } +func TestRelayKeepsTheUpstreamConnForAClosingGuest(t *testing.T) { + closeSeen := make(chan bool, 1) + upstream := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + closeSeen <- r.Close + _, _ = io.WriteString(w, "hello") + })) + defer upstream.Close() + p := New("sb_1", "acme", Policy{Allow: []Rule{{Host: "api.internal"}}}, nil, nil, fixedDial(upstream.Listener.Addr().String()), nil, nil) + front := httptest.NewServer(p) + defer front.Close() + + req, err := http.NewRequestWithContext(t.Context(), http.MethodGet, "http://api.internal/x", nil) + if err != nil { + t.Fatalf("build request: %v", err) + } + req.Close = true + resp, err := proxyClient(t, front.URL).Do(req) + if err != nil { + t.Fatalf("proxied GET: %v", err) + } + _ = resp.Body.Close() + if <-closeSeen { + t.Error("the guest's Connection: close reached the upstream hop") + } +} + func TestCloseReleasesIdleUpstreamConns(t *testing.T) { closed := make(chan struct{}, 4) upstream := httptest.NewUnstartedServer(http.HandlerFunc(func(w http.ResponseWriter, _ *http.Request) { From 098804d5cccb19fd56dc7b6f74f423752a893e04 Mon Sep 17 00:00:00 2001 From: CMGS Date: Mon, 14 Sep 2026 19:59:43 +0800 Subject: [PATCH 13/40] server: serve a held claim under a stale owner name; omit an unspecified owner address A preview token names its owner by the advertise address at mint time; a node that holds the claim now serves it locally whatever that string says, and the one-hop marker stays as the backstop. The raw path suffix is derived by segments, so a %70 spelling of the prefix keeps %2F too. A claim response and /owner no longer emit a wildcard advertise address (:7777, 0.0.0.0) as owner_addr: an empty owner makes both SDKs reuse the address they dialed, which is what a default single-node config needs. The audit tee records an oversized first line as such instead of dropping it; writeResult keeps a string id (the template name), and ErrArchived maps to 409. --- sandboxd/server/preview.go | 19 ++++++-- sandboxd/server/preview_test.go | 79 ++++++++++++++++++++------------- sandboxd/server/relay.go | 1 + sandboxd/server/server.go | 14 ++++-- sandboxd/server/server_http.go | 4 +- sandboxd/server/server_test.go | 9 ++++ 6 files changed, 86 insertions(+), 40 deletions(-) diff --git a/sandboxd/server/preview.go b/sandboxd/server/preview.go index ac9dcf59..87d22a19 100644 --- a/sandboxd/server/preview.go +++ b/sandboxd/server/preview.go @@ -17,6 +17,7 @@ import ( "github.com/projecteru2/core/log" + "github.com/cocoonstack/sandbox/sandboxd/pool" "github.com/cocoonstack/sandbox/sandboxd/types" ) @@ -33,6 +34,7 @@ type previewClaims struct { // PreviewManager is the slice of the pool manager the preview path needs. type PreviewManager interface { PreviewDial(ctx context.Context, id string, port uint16) (net.Conn, error) + Sandbox(id string) (pool.SandboxSummary, bool) } // PreviewServer serves signed guest HTTP URLs and forwards requests to their owner node. @@ -95,7 +97,8 @@ func (p *PreviewServer) serve(w http.ResponseWriter, r *http.Request) { http.Error(w, "invalid or expired preview token", http.StatusForbidden) return } - if claims.Owner != p.owner { + // the token names the owner by its advertise address at mint time; holding the claim is what counts + if _, held := p.mgr.Sandbox(claims.ID); !held && claims.Owner != p.owner { if r.Header.Get(forwardedHeader) != "" { http.Error(w, "preview owner is not this node", http.StatusBadGateway) return @@ -112,9 +115,17 @@ func (p *PreviewServer) proxyLocal(w http.ResponseWriter, r *http.Request, claim pr.Out.URL.Scheme = "http" // the synthetic host carries PreviewDial's target pr.Out.URL.Host = fmt.Sprintf("%s:%d", claims.ID, claims.Port) - prefix := "/p/" + r.PathValue("token") + "/" - pr.Out.URL.Path = "/" + strings.TrimPrefix(pr.In.URL.Path, prefix) - pr.Out.URL.RawPath = "/" + strings.TrimPrefix(pr.In.URL.EscapedPath(), prefix) + pr.Out.URL.Path = "/" + strings.TrimPrefix(pr.In.URL.Path, "/p/"+r.PathValue("token")+"/") + // the raw form drops the same two segments however the client spelled them, so %2F survives + raw := pr.In.URL.EscapedPath() + for range 2 { + if i := strings.IndexByte(raw[1:], '/'); i >= 0 { + raw = raw[i+1:] + } else { + raw = "/" + } + } + pr.Out.URL.RawPath = raw // browser credentials for the preview domain must not reach guest code pr.Out.Header.Del("Cookie") pr.Out.Header.Del("Authorization") diff --git a/sandboxd/server/preview_test.go b/sandboxd/server/preview_test.go index 271c2281..a1aeec90 100644 --- a/sandboxd/server/preview_test.go +++ b/sandboxd/server/preview_test.go @@ -10,6 +10,8 @@ import ( "sync/atomic" "testing" "time" + + "github.com/cocoonstack/sandbox/sandboxd/pool" ) func TestPreviewTokenRoundTrip(t *testing.T) { @@ -40,12 +42,12 @@ func TestPreviewRejectsExpired(t *testing.T) { } func TestPreviewProxiesToGuest(t *testing.T) { - guestAddr := newGuestServer(t, func(r *http.Request) string { return "guest saw " + r.URL.Path }) + guestAddr := newGuestServer(t, func(r *http.Request) string { return "guest saw " + r.URL.EscapedPath() }) - dialed := false + dialed := 0 ps := NewPreviewServer("secret", "node:9000", "node:7777", &fakePreviewMgr{ dial: func(id string, port uint16) (net.Conn, error) { - dialed = true + dialed++ if id != "sb_1" || port != 8080 { t.Errorf("dial %s:%d", id, port) } @@ -56,15 +58,23 @@ func TestPreviewProxiesToGuest(t *testing.T) { t.Cleanup(ts.Close) token := mintToken(ps, "sb_1", 8080, time.Hour) - - resp, err := http.Get(ts.URL + "/p/" + token + "/hello/world") - if err != nil { - t.Fatalf("get: %v", err) - } - defer resp.Body.Close() - body, _ := io.ReadAll(resp.Body) - if !dialed || string(body) != "guest saw /hello/world" { - t.Errorf("body %q dialed=%v", body, dialed) + for _, tt := range []struct{ prefix, path string }{ + {"/p/", "/hello/world"}, + {"/p/", "/repo/a%2Fb/tree"}, + {"/%70/", "/repo/a%2Fb/tree"}, + } { + resp, err := http.Get(ts.URL + tt.prefix + token + tt.path) + if err != nil { + t.Fatalf("get %s: %v", tt.path, err) + } + body, _ := io.ReadAll(resp.Body) + _ = resp.Body.Close() + if string(body) != "guest saw "+tt.path { + t.Errorf("%s%s: body %q, want the guest to see %s verbatim", tt.prefix, tt.path, body, tt.path) + } + } + if dialed != 3 { + t.Errorf("dialed %d times, want one per request", dialed) } } @@ -158,6 +168,27 @@ func TestPreviewForwardsToOwner(t *testing.T) { } } +func TestPreviewServesAHeldClaimUnderAStaleOwnerName(t *testing.T) { + guestAddr := newGuestServer(t, func(r *http.Request) string { return "guest saw " + r.URL.Path }) + ps := NewPreviewServer("secret", "https://preview.example.com", "renamed:7777", &fakePreviewMgr{ + dial: func(string, uint16) (net.Conn, error) { return net.Dial("tcp", guestAddr) }, + held: map[string]bool{"sb_1": true}, + }) + ts := httptest.NewServer(ps.Handler()) + t.Cleanup(ts.Close) + + minter := NewPreviewServer("secret", "https://preview.example.com", "old:7777", &fakePreviewMgr{}) + resp, err := http.Get(ts.URL + "/p/" + mintToken(minter, "sb_1", 8080, time.Hour) + "/x") + if err != nil { + t.Fatalf("get: %v", err) + } + defer resp.Body.Close() + body, _ := io.ReadAll(resp.Body) + if string(body) != "guest saw /x" { + t.Errorf("body %q, want the held claim served locally despite the old owner name", body) + } +} + func TestPreviewForwardStopsAfterOneHop(t *testing.T) { entry := httptest.NewUnstartedServer(nil) entryAddr := entry.Listener.Addr().String() @@ -178,33 +209,19 @@ func TestPreviewForwardStopsAfterOneHop(t *testing.T) { } } -func TestPreviewKeepsEncodedSlashes(t *testing.T) { - guestAddr := newGuestServer(t, func(r *http.Request) string { return "guest saw " + r.URL.EscapedPath() }) - ps := NewPreviewServer("secret", "node:9000", "node:7777", &fakePreviewMgr{ - dial: func(string, uint16) (net.Conn, error) { return net.Dial("tcp", guestAddr) }, - }) - ts := httptest.NewServer(ps.Handler()) - t.Cleanup(ts.Close) - - resp, err := http.Get(ts.URL + "/p/" + mintToken(ps, "sb_1", 8080, time.Hour) + "/repo/a%2Fb/tree") - if err != nil { - t.Fatalf("get: %v", err) - } - defer resp.Body.Close() - body, _ := io.ReadAll(resp.Body) - if string(body) != "guest saw /repo/a%2Fb/tree" { - t.Errorf("body %q, want the encoded slash preserved", body) - } -} - type fakePreviewMgr struct { dial func(id string, port uint16) (net.Conn, error) + held map[string]bool } func (f *fakePreviewMgr) PreviewDial(_ context.Context, id string, port uint16) (net.Conn, error) { return f.dial(id, port) } +func (f *fakePreviewMgr) Sandbox(id string) (pool.SandboxSummary, bool) { + return pool.SandboxSummary{ID: id}, f.held[id] +} + func newGuestServer(t *testing.T, body func(r *http.Request) string) string { t.Helper() ts := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { diff --git a/sandboxd/server/relay.go b/sandboxd/server/relay.go index 871e5ac1..1e9bff80 100644 --- a/sandboxd/server/relay.go +++ b/sandboxd/server/relay.go @@ -181,6 +181,7 @@ func (t *auditTee) Read(p []byte) (int, error) { t.buf = append(t.buf, chunk...) if len(t.buf) > pool.AuditLineCap { t.done = true // a frame this large is payload, not addressing + t.record(t.buf) t.buf = nil } return n, err diff --git a/sandboxd/server/server.go b/sandboxd/server/server.go index 810b4829..11dc0275 100644 --- a/sandboxd/server/server.go +++ b/sandboxd/server/server.go @@ -15,6 +15,7 @@ import ( "io" "net" "net/http" + "net/netip" "slices" "sync" "time" @@ -49,6 +50,7 @@ var poolErrHTTP = []struct { {pool.ErrVolumeCapture, http.StatusConflict, ""}, {pool.ErrVolumeBusy, http.StatusConflict, ""}, {pool.ErrVolumeNeedsRecovery, http.StatusConflict, ""}, + {pool.ErrArchived, http.StatusConflict, ""}, {pool.ErrQuota, http.StatusTooManyRequests, ""}, {pool.ErrHealBusy, http.StatusServiceUnavailable, ""}, {pool.ErrPooledTemplate, http.StatusConflict, ""}, @@ -158,6 +160,12 @@ type Server struct { // New returns a Server; an empty apiToken with no tenants leaves node-level endpoints open. func New(apiToken string, tenants []config.TenantSpec, advertise string, mgr Manager, dialer Dialer, placer Placer, prober CheckpointProber, probeKey []byte, preview *PreviewServer) *Server { + // an unspecified host names nothing a remote client can dial; an empty owner makes the SDK reuse the address it reached + if host, _, err := net.SplitHostPort(advertise); err == nil { + if ip, _ := netip.ParseAddr(host); host == "" || ip.IsUnspecified() { + advertise = "" + } + } return &Server{ mgr: mgr, dialer: dialer, @@ -236,7 +244,7 @@ func (s *Server) handleClaim(w http.ResponseWriter, r *http.Request) { writeRedirect(w, s.placer.Candidates(key.Hash())) { return } - writeResult(w, r, "claim", key, "provisioning failed", err, func() { + writeResult(w, r, "claim", key.Template, "provisioning failed", err, func() { writeJSON(w, http.StatusOK, s.claimResponse(sb)) }) } @@ -250,7 +258,7 @@ func (s *Server) handleVolumeClaim(w http.ResponseWriter, r *http.Request, req t req.Volumes = volumes redirected, err := s.redirectVolumeClaim(r.Context(), w, &req, key, hash, tenant) if err != nil { - writeResult(w, r, "claim", hash, "provisioning failed", err, func() {}) + writeResult(w, r, "claim", key.Template, "provisioning failed", err, func() {}) return } if redirected { @@ -268,7 +276,7 @@ func (s *Server) handleVolumeClaim(w http.ResponseWriter, r *http.Request, req t sb, err = s.mgr.ClaimProvision(r.Context(), key, req.TTL(), tenant, req.ClaimRef, req.Volumes) } } - writeResult(w, r, "claim", hash, "provisioning failed", err, func() { + writeResult(w, r, "claim", key.Template, "provisioning failed", err, func() { writeJSON(w, http.StatusOK, s.claimResponse(sb)) }) } diff --git a/sandboxd/server/server_http.go b/sandboxd/server/server_http.go index 14457d97..3ace7f56 100644 --- a/sandboxd/server/server_http.go +++ b/sandboxd/server/server_http.go @@ -84,11 +84,11 @@ func writePoolErr(w http.ResponseWriter, err error) bool { return false } -func writeResult(w http.ResponseWriter, r *http.Request, op string, id any, failMsg string, err error, ok func()) { +func writeResult(w http.ResponseWriter, r *http.Request, op, id, failMsg string, err error, ok func()) { switch { case writePoolErr(w, err): case err != nil: - log.WithFunc("server.writeResult").Errorf(r.Context(), err, "%s %v", op, id) + log.WithFunc("server.writeResult").Errorf(r.Context(), err, "%s %s", op, id) writeErr(w, http.StatusInternalServerError, failMsg) default: ok() diff --git a/sandboxd/server/server_test.go b/sandboxd/server/server_test.go index d2f24042..627274e6 100644 --- a/sandboxd/server/server_test.go +++ b/sandboxd/server/server_test.go @@ -23,6 +23,15 @@ import ( "github.com/cocoonstack/sandbox/sandboxd/types" ) +func TestOwnerAddrOmitsAnUnspecifiedHost(t *testing.T) { + for advertise, want := range map[string]string{":7777": "", "0.0.0.0:7777": "", "[::]:7777": "", "10.0.0.5:7777": "10.0.0.5:7777"} { + srv := New("", nil, advertise, &fakeManager{}, &fakeDialer{}, nil, nil, nil, nil) + if got := srv.claimResponse(&types.Sandbox{ID: "sb_1"}).OwnerAddr; got != want { + t.Errorf("advertise %q: owner_addr %q, want %q", advertise, got, want) + } + } +} + func TestClaimHappyPath(t *testing.T) { var gotKey types.PoolKey var gotTTL time.Duration From 0cd543ccbe48d9b53a77ac3092acb14a99de55c8 Mon Sep 17 00:00:00 2001 From: CMGS Date: Mon, 14 Sep 2026 19:59:43 +0800 Subject: [PATCH 14/40] mcp: bound read_file to a regular file under the exec cap read_file buffered whatever the guest streamed, so a device or a multi-gigabyte file grew the MCP process the way exec used to; it now stats first and refuses anything but a regular file up to 1 MiB with a pointer to exec head/tail. The tool tests share one fake node and a reply decoder. --- mcp/go.mod | 2 +- mcp/server.go | 6 +- mcp/server_test.go | 154 ++++++++++++++++++++++----------------------- mcp/tools.go | 10 ++- 4 files changed, 89 insertions(+), 83 deletions(-) diff --git a/mcp/go.mod b/mcp/go.mod index f915459c..d8dc6226 100644 --- a/mcp/go.mod +++ b/mcp/go.mod @@ -3,6 +3,7 @@ module github.com/cocoonstack/sandbox/mcp go 1.27.0 require ( + github.com/cocoonstack/sandbox/protocol/wire v0.1.10 github.com/cocoonstack/sandbox/sdk/go v0.0.0 github.com/projecteru2/core v0.1.4 ) @@ -11,7 +12,6 @@ require ( github.com/cockroachdb/errors v1.14.0 // indirect github.com/cockroachdb/logtags v0.0.0-20241215232642-bb51bb14a506 // indirect github.com/cockroachdb/redact v1.1.8 // indirect - github.com/cocoonstack/sandbox/protocol/wire v0.1.10 // indirect github.com/docker/go-units v0.5.0 // indirect github.com/getsentry/sentry-go v0.48.0 // indirect github.com/gogo/protobuf v1.3.2 // indirect diff --git a/mcp/server.go b/mcp/server.go index 67fe2280..c78a2160 100644 --- a/mcp/server.go +++ b/mcp/server.go @@ -2,7 +2,6 @@ package main import ( "bufio" - "cmp" "context" "encoding/json" "fmt" @@ -120,8 +119,11 @@ func (s *server) dispatch(ctx context.Context, req *rpcRequest) rpcResponse { } text, err := s.callTool(ctx, call.Name, call.Arguments) if err != nil { + if text == "" { + text = err.Error() + } return result(req.ID, map[string]any{ - "content": []map[string]any{{"type": "text", "text": cmp.Or(text, err.Error())}}, + "content": []map[string]any{{"type": "text", "text": text}}, "isError": true, }) } diff --git a/mcp/server_test.go b/mcp/server_test.go index 8fde8e55..e8e64d57 100644 --- a/mcp/server_test.go +++ b/mcp/server_test.go @@ -13,28 +13,16 @@ import ( ) func TestServeSpeaksMCP(t *testing.T) { - node := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { - switch { - case r.URL.Path == "/v1/claim": - _ = json.NewEncoder(w).Encode(map[string]any{"id": "sb_1", "token": "tok"}) - case strings.HasSuffix(r.URL.Path, "/checkpoint"): - _ = json.NewEncoder(w).Encode(map[string]any{ - "checkpoint": map[string]any{"id": "ck_0011223344556677", "name": "s1", "sandbox_id": "sb_1"}, - }) - case strings.HasSuffix(r.URL.Path, "/release"): - w.WriteHeader(http.StatusNoContent) - default: - http.Error(w, `{"error":"no route"}`, http.StatusNotFound) + srv := newTestServer(t, func(w http.ResponseWriter, r *http.Request) bool { + if !strings.HasSuffix(r.URL.Path, "/checkpoint") { + return false } - })) - t.Cleanup(node.Close) - - srv, err := newServer(strings.TrimPrefix(node.URL, "http://"), "", "rt:24.04") - if err != nil { - t.Fatalf("newServer: %v", err) - } - - lines := []string{ + _ = json.NewEncoder(w).Encode(map[string]any{ + "checkpoint": map[string]any{"id": "ck_0011223344556677", "name": "s1", "sandbox_id": "sb_1"}, + }) + return true + }) + replies := serveLines(t, srv, `{"jsonrpc":"2.0","id":1,"method":"initialize","params":{}}`, `{"jsonrpc":"2.0","method":"notifications/initialized"}`, `{"jsonrpc":"2.0","id":2,"method":"tools/list"}`, @@ -42,23 +30,7 @@ func TestServeSpeaksMCP(t *testing.T) { `{"jsonrpc":"2.0","id":4,"method":"tools/call","params":{"name":"checkpoint","arguments":{"sandbox_id":"sb_1","name":"s1"}}}`, `{"jsonrpc":"2.0","id":5,"method":"tools/call","params":{"name":"release","arguments":{"sandbox_id":"sb_1"}}}`, `{"jsonrpc":"2.0","id":6,"method":"tools/call","params":{"name":"exec","arguments":{"sandbox_id":"sb_1","command":"true"}}}`, - } - var out bytes.Buffer - in := bufio.NewReader(strings.NewReader(strings.Join(lines, "\n") + "\n")) - if err := srv.serve(t.Context(), in, &out); err != nil { - t.Fatalf("serve: %v", err) - } - - replies := map[int]map[string]any{} - dec := json.NewDecoder(&out) - for dec.More() { - var resp map[string]any - if err := dec.Decode(&resp); err != nil { - t.Fatalf("decode reply: %v", err) - } - id := int(resp["id"].(float64)) - replies[id] = resp - } + ) if len(replies) != 6 { t.Fatalf("got %d replies, want 6 (notification unanswered)", len(replies)) } @@ -93,52 +65,36 @@ func TestServeSpeaksMCP(t *testing.T) { } func TestExecKeepsOutputWhenTheGuestDrops(t *testing.T) { - node := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { - switch { - case r.URL.Path == "/v1/claim": - _ = json.NewEncoder(w).Encode(map[string]any{"id": "sb_1", "token": "tok"}) - case strings.HasSuffix(r.URL.Path, "/agent"): - conn, _, err := http.NewResponseController(w).Hijack() - if err != nil { - t.Errorf("hijack: %v", err) - return - } - _, _ = io.WriteString(conn, "HTTP/1.1 101 Switching Protocols\r\nUpgrade: silkd\r\nConnection: Upgrade\r\n\r\n"+ - `{"type":"started","pid":7}`+"\n"+`{"type":"stdout","data":"cGFydGlhbA=="}`+"\n") - _ = conn.Close() - default: - http.Error(w, `{"error":"no route"}`, http.StatusNotFound) + srv := newTestServer(t, func(w http.ResponseWriter, r *http.Request) bool { + if !strings.HasSuffix(r.URL.Path, "/agent") { + return false } - })) - t.Cleanup(node.Close) - srv, err := newServer(strings.TrimPrefix(node.URL, "http://"), "", "rt:24.04") - if err != nil { - t.Fatalf("newServer: %v", err) - } - lines := `{"jsonrpc":"2.0","id":1,"method":"tools/call","params":{"name":"create_sandbox","arguments":{}}} -{"jsonrpc":"2.0","id":2,"method":"tools/call","params":{"name":"exec","arguments":{"sandbox_id":"sb_1","command":"yes"}}} -{"jsonrpc":"2.0","id":3,"method":"tools/call","params":{"name":"create_sandbox","arguments":{"ttl_seconds":-5}}} -` - var out bytes.Buffer - if err := srv.serve(t.Context(), bufio.NewReader(strings.NewReader(lines)), &out); err != nil { - t.Fatalf("serve: %v", err) - } - var replies []map[string]any - dec := json.NewDecoder(&out) - for dec.More() { - var resp map[string]any - if err := dec.Decode(&resp); err != nil { - t.Fatalf("decode reply: %v", err) + conn, _, err := http.NewResponseController(w).Hijack() + if err != nil { + t.Errorf("hijack: %v", err) + return true } - replies = append(replies, resp) - } - execResult := replies[1]["result"].(map[string]any) - text := toolText(t, replies[1]) + _, _ = io.WriteString(conn, "HTTP/1.1 101 Switching Protocols\r\nUpgrade: silkd\r\nConnection: Upgrade\r\n\r\n"+ + `{"type":"started","pid":7}`+"\n"+`{"type":"stdout","data":"cGFydGlhbA=="}`+"\n") + _ = conn.Close() + return true + }) + replies := serveLines(t, srv, + `{"jsonrpc":"2.0","id":1,"method":"tools/call","params":{"name":"create_sandbox","arguments":{}}}`, + `{"jsonrpc":"2.0","id":2,"method":"tools/call","params":{"name":"exec","arguments":{"sandbox_id":"sb_1","command":"yes"}}}`, + ) + execResult := replies[2]["result"].(map[string]any) + text := toolText(t, replies[2]) if execResult["isError"] != true || !strings.Contains(text, `"stdout":"partial"`) || !strings.Contains(text, `"error"`) { t.Errorf("exec reply %v: want isError with the partial stdout and an error field", text) } - if !strings.Contains(toolText(t, replies[2]), "ttl_seconds must not be negative") { - t.Errorf("negative ttl accepted: %q", toolText(t, replies[2])) +} + +func TestCreateSandboxRejectsNegativeTTL(t *testing.T) { + replies := serveLines(t, newTestServer(t, nil), + `{"jsonrpc":"2.0","id":1,"method":"tools/call","params":{"name":"create_sandbox","arguments":{"ttl_seconds":-5}}}`) + if !strings.Contains(toolText(t, replies[1]), "ttl_seconds must not be negative") { + t.Errorf("negative ttl accepted: %q", toolText(t, replies[1])) } } @@ -164,3 +120,43 @@ func toolText(t *testing.T, resp map[string]any) string { content := res["content"].([]any) return fmt.Sprintf("%v", content[0].(map[string]any)["text"]) } + +func newTestServer(t *testing.T, extra func(http.ResponseWriter, *http.Request) bool) *server { + t.Helper() + node := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + switch { + case r.URL.Path == "/v1/claim": + _ = json.NewEncoder(w).Encode(map[string]any{"id": "sb_1", "token": "tok"}) + case strings.HasSuffix(r.URL.Path, "/release"): + w.WriteHeader(http.StatusNoContent) + case extra != nil && extra(w, r): + default: + http.Error(w, `{"error":"no route"}`, http.StatusNotFound) + } + })) + t.Cleanup(node.Close) + srv, err := newServer(strings.TrimPrefix(node.URL, "http://"), "", "rt:24.04") + if err != nil { + t.Fatalf("newServer: %v", err) + } + return srv +} + +func serveLines(t *testing.T, srv *server, lines ...string) map[int]map[string]any { + t.Helper() + var out bytes.Buffer + in := bufio.NewReader(strings.NewReader(strings.Join(lines, "\n") + "\n")) + if err := srv.serve(t.Context(), in, &out); err != nil { + t.Fatalf("serve: %v", err) + } + replies := map[int]map[string]any{} + dec := json.NewDecoder(&out) + for dec.More() { + var resp map[string]any + if err := dec.Decode(&resp); err != nil { + t.Fatalf("decode reply: %v", err) + } + replies[int(resp["id"].(float64))] = resp + } + return replies +} diff --git a/mcp/tools.go b/mcp/tools.go index d19dfb40..6aefb78a 100644 --- a/mcp/tools.go +++ b/mcp/tools.go @@ -8,6 +8,7 @@ import ( "strings" "time" + "github.com/cocoonstack/sandbox/protocol/wire" sandbox "github.com/cocoonstack/sandbox/sdk/go" ) @@ -41,7 +42,7 @@ var tools = []tool{ schema(props{"sandbox_id": str("id returned by create_sandbox, fork, or branch_checkpoint"), "path": str("absolute path of the file"), "content": str("full file content; written verbatim")}, "sandbox_id", "path", "content"), toolWriteFile, }, { - "read_file", "Return the whole content of a file in a sandbox as text; bytes that are not valid UTF-8 are replaced, so binary content is lossy. For large or binary files use exec with head, tail, or a checksum instead. A missing path is an error.", + "read_file", "Return the whole content of a regular file up to 1 MiB in a sandbox as text; bytes that are not valid UTF-8 are replaced, so binary content is lossy. A larger file, a directory, or a device is an error: use exec with head, tail, or a checksum instead. A missing path is an error.", schema(props{"sandbox_id": str("id returned by create_sandbox, fork, or branch_checkpoint"), "path": str("absolute path of the file")}, "sandbox_id", "path"), toolReadFile, }, { @@ -273,6 +274,13 @@ func toolReadFile(ctx context.Context, s *server, raw json.RawMessage) (string, if err != nil { return "", err } + info, err := sb.Stat(ctx, args.Path) + if err != nil { + return "", err + } + if info.Kind != wire.FileKindFile || info.Size > execOutputCap { + return "", fmt.Errorf("%s is a %s of %d bytes; read_file returns a regular file up to %d bytes, use exec with head or tail", args.Path, info.Kind, info.Size, execOutputCap) + } data, err := sb.ReadFile(ctx, args.Path) if err != nil { return "", err From 017c97b73a605d4c5541789a95a37191b2671ab7 Mon Sep 17 00:00:00 2001 From: CMGS Date: Mon, 14 Sep 2026 19:59:43 +0800 Subject: [PATCH 15/40] sdk/go: release the relay when a pty exits Pty.drain returned on the exit or error frame without the dial's cleanup, the same leak Watcher had; the drain now owns it. --- sdk/go/pty.go | 7 ++++--- sdk/go/pty_test.go | 33 +++++++++++++++++++++++++++++++++ 2 files changed, 37 insertions(+), 3 deletions(-) diff --git a/sdk/go/pty.go b/sdk/go/pty.go index 8ebc7253..a068b2d4 100644 --- a/sdk/go/pty.go +++ b/sdk/go/pty.go @@ -81,8 +81,9 @@ func (p *Pty) ExitCode() (code int, ok bool) { return p.exitCode, p.exited } -// drain relays response frames into the output pipe and terminal state. -func (p *Pty) drain(ctx context.Context, pw *io.PipeWriter) { +// drain relays response frames into the output pipe and terminal state, releasing the relay when they end. +func (p *Pty) drain(ctx context.Context, pw *io.PipeWriter, stop func()) { + defer stop() for { resp, err := recv(ctx, p.conn) if err != nil { @@ -126,6 +127,6 @@ func (s *Sandbox) OpenPty(ctx context.Context, opts PtyOpts) (*Pty, error) { pr, pw := io.Pipe() p := &Pty{PID: started.PID, sb: s, conn: conn, stop: done, out: pr} - go p.drain(ctx, pw) + go p.drain(ctx, pw, done) return p, nil } diff --git a/sdk/go/pty_test.go b/sdk/go/pty_test.go index f7eb5547..a5a950e6 100644 --- a/sdk/go/pty_test.go +++ b/sdk/go/pty_test.go @@ -9,6 +9,7 @@ import ( "io" "net" "testing" + "time" "github.com/cocoonstack/sandbox/protocol/wire" "github.com/cocoonstack/sandbox/sdk/go/silkd" @@ -52,6 +53,38 @@ func TestPtyEchoAndExit(t *testing.T) { } } +func TestPtyReleasesTheRelayWhenTheShellExits(t *testing.T) { + released := make(chan struct{}) + ts := newAgentServer(t, func(conn net.Conn) { + defer conn.Close() + r := bufio.NewReader(conn) + if _, err := r.ReadString('\n'); err != nil { + t.Errorf("read pty request: %v", err) + return + } + _, _ = io.WriteString(conn, `{"type":"started","pid":7}`+"\n"+ + `{"type":"stdout","data":"aGk="}`+"\n"+`{"type":"exit","code":3}`+"\n") + _, _ = r.ReadByte() + close(released) + }) + pty, err := testSandbox(t, ts).OpenPty(t.Context(), PtyOpts{Cols: 80, Rows: 24}) + if err != nil { + t.Fatalf("OpenPty: %v", err) + } + out, err := io.ReadAll(pty) + if err != nil || string(out) != "hi" { + t.Fatalf("read %q, %v; want hi to EOF", out, err) + } + if code, ok := pty.ExitCode(); !ok || code != 3 { + t.Errorf("exit code %d ok=%v, want 3 true", code, ok) + } + select { + case <-released: + case <-time.After(2 * time.Second): + t.Fatal("relay connection still open after the shell exited") + } +} + func TestPtyCtxCancelIsTyped(t *testing.T) { sb := fakeSandbox(t) ctx, cancel := context.WithCancel(t.Context()) From c209a6d0e8d3eeb3a07cc372388034984a8f77cb Mon Sep 17 00:00:00 2001 From: CMGS Date: Mon, 14 Sep 2026 19:59:43 +0800 Subject: [PATCH 16/40] sdk/python: a wall-clock timeout on run and exec Sandbox.run and exec take timeout, a wall clock over the dial and the run: a watchdog cuts the connection at the deadline, which makes silkd kill the command, and TimeoutError is raised; the dial itself is bounded by the remaining time. The openai adapter passes its per-call timeout straight through, and the langchain tool carries one deadline across the lazy claim, the dial and the command, so neither depends on timeout(1) in the guest image or on exit code 124. --- .../cocoonsandbox_langchain/toolkit.py | 21 +++++-- sdk/langchain/tests/test_toolkit.py | 13 ++-- sdk/openai/cocoonsandbox_openai/adapter.py | 11 +--- sdk/openai/tests/test_adapter.py | 17 ++--- sdk/python/cocoonsandbox/conn.py | 5 ++ sdk/python/cocoonsandbox/sandbox.py | 62 ++++++++++++++++--- sdk/python/tests/test_proc.py | 4 +- sdk/python/tests/test_stream.py | 32 ++++++++++ sdk/python/tests/test_wire_binding.py | 4 +- 9 files changed, 131 insertions(+), 38 deletions(-) diff --git a/sdk/langchain/cocoonsandbox_langchain/toolkit.py b/sdk/langchain/cocoonsandbox_langchain/toolkit.py index 27fafa76..5a2a3aca 100644 --- a/sdk/langchain/cocoonsandbox_langchain/toolkit.py +++ b/sdk/langchain/cocoonsandbox_langchain/toolkit.py @@ -8,13 +8,14 @@ import asyncio import json import threading +import time from collections.abc import Callable from cocoonsandbox import Client, Sandbox from langchain_core.tools import StructuredTool from pydantic import BaseModel, Field -# one exec's wall clock, enforced by the guest's timeout(1); the tool description states it +# one call's wall clock over the claim, the dial and the command; the tool description states it CALL_TIMEOUT = 300 @@ -73,7 +74,7 @@ def get_tools(self) -> list[StructuredTool]: "Returns stdout; a non-empty stderr is appended as a 'stderr:' " "line and a non-zero status as an 'exit code: N' line; a " "command that prints nothing and exits 0 returns '(no output)'. " - "The call is cut off after 5 minutes (exit code 124). " + "The call is cut off after 5 minutes and the reply says so. " "Files and installed packages persist across calls; environment " "variables and the working directory do not.", ExecInput, @@ -136,17 +137,25 @@ async def arun(**kwargs): ) def _exec(self, command: str, cwd: str = "") -> str: + deadline = time.monotonic() + CALL_TIMEOUT out: list[bytes] = [] errs: list[bytes] = [] - argv = ["timeout", "-s", "KILL", str(CALL_TIMEOUT), "sh", "-c", command] - code = self.sandbox().run(argv, cwd=cwd, on_stdout=out.append, on_stderr=errs.append) + tail = "" + try: + sb = self.sandbox() + timeout = max(deadline - time.monotonic(), 1) + code = sb.run(["sh", "-c", command], cwd=cwd, on_stdout=out.append, on_stderr=errs.append, timeout=timeout) + if code != 0: + tail = f"exit code: {code}" + except TimeoutError: + tail = f"cut off after {CALL_TIMEOUT}s" stdout = b"".join(out).decode(errors="replace") stderr = b"".join(errs).decode(errors="replace") result = stdout if stderr: result += ("\n" if result else "") + "stderr: " + stderr - if code != 0: - result += ("\n" if result else "") + f"exit code: {code}" + if tail: + result += ("\n" if result else "") + tail return result or "(no output)" def _write_file(self, path: str, content: str) -> str: diff --git a/sdk/langchain/tests/test_toolkit.py b/sdk/langchain/tests/test_toolkit.py index a2f03732..d28b5b21 100644 --- a/sdk/langchain/tests/test_toolkit.py +++ b/sdk/langchain/tests/test_toolkit.py @@ -23,6 +23,7 @@ def test_exec_tool_output(monkeypatch): assert exec_tool.invoke({"command": "echo hi"}) == "ran: echo hi\n" out = exec_tool.invoke({"command": "boom"}) assert "kaboom" in out and "exit code: 3" in out + assert exec_tool.invoke({"command": "hang"}) == "partial\n\ncut off after 300s" def test_async_bridge(monkeypatch): @@ -60,12 +61,16 @@ def __init__(self): self.closed = 0 self.files = {} - def run(self, argv, cwd="", on_stdout=None, on_stderr=None, **_): - assert argv[:6] == ["timeout", "-s", "KILL", "300", "sh", "-c"], argv - if argv[6] == "boom": + def run(self, argv, cwd="", on_stdout=None, on_stderr=None, timeout=None, **_): + assert argv[:2] == ["sh", "-c"], argv + assert timeout is not None and 0 < timeout <= 300, timeout + if argv[2] == "boom": on_stderr(b"kaboom\n") return 3 - on_stdout(f"ran: {argv[6]}\n".encode()) + if argv[2] == "hang": + on_stdout(b"partial\n") + raise TimeoutError("cut") + on_stdout(f"ran: {argv[2]}\n".encode()) return 0 def write_file(self, path, data): diff --git a/sdk/openai/cocoonsandbox_openai/adapter.py b/sdk/openai/cocoonsandbox_openai/adapter.py index 86e67f11..5fd676a7 100644 --- a/sdk/openai/cocoonsandbox_openai/adapter.py +++ b/sdk/openai/cocoonsandbox_openai/adapter.py @@ -21,9 +21,6 @@ from agents.sandbox.types import ExecResult, ExposedPortEndpoint, User from cocoonsandbox import Client, Sandbox, SandboxError, SilkdError -# timeout(1) answers 124 when the command it ran was cut off -TIMEOUT_EXIT = 124 - class CocoonSandboxClientOptions(BaseSandboxClientOptions): """Connection settings for a sandboxd node (or cluster entry node).""" @@ -100,17 +97,13 @@ async def _prepare_backend_workspace(self) -> None: async def _exec_internal(self, *command: str | Path, timeout: float | None = None) -> ExecResult: sb = self._sandbox() argv = [str(part) for part in command] - if timeout is not None: - # the guest enforces the cutoff: a stream has no socket timeout, so a cancelled wait strands the worker - argv = ["timeout", "-s", "KILL", str(timeout), *argv] stdout, stderr = bytearray(), bytearray() def run() -> int: - return sb.run(argv, on_stdout=stdout.extend, on_stderr=stderr.extend) + # the SDK's wall clock cuts the connection, which kills the command, and raises TimeoutError + return sb.run(argv, on_stdout=stdout.extend, on_stderr=stderr.extend, timeout=timeout) code = await asyncio.to_thread(run) - if timeout is not None and code == TIMEOUT_EXIT: - raise TimeoutError(f"command did not finish within {timeout}s") return ExecResult(stdout=bytes(stdout), stderr=bytes(stderr), exit_code=code) async def _resolve_exposed_port(self, port: int) -> ExposedPortEndpoint: diff --git a/sdk/openai/tests/test_adapter.py b/sdk/openai/tests/test_adapter.py index f8be4242..d0ff30a3 100644 --- a/sdk/openai/tests/test_adapter.py +++ b/sdk/openai/tests/test_adapter.py @@ -72,8 +72,8 @@ class FakeSandbox: def __init__(self, **kw): self.id = "sb_1" - def run(self, argv, on_stdout=None, on_stderr=None): - assert argv == ["echo", "hi"] + def run(self, argv, on_stdout=None, on_stderr=None, timeout=None): + assert argv == ["echo", "hi"] and timeout is None on_stdout(b"hi\n") return 0 @@ -89,14 +89,16 @@ async def go(): asyncio.run(go()) -def test_exec_timeout_runs_under_guest_timeout(node, monkeypatch): +def test_exec_timeout_reaches_the_sdk_and_surfaces_as_timeouterror(node, monkeypatch): seen = [] class FakeSandbox: - def run(self, argv, on_stdout=None, on_stderr=None): - seen.append(argv) + def run(self, argv, on_stdout=None, on_stderr=None, timeout=None): + seen.append(timeout) on_stdout(b"partial\n") - return 124 + if timeout is not None: + raise TimeoutError("cut") + return 0 async def go(): client = CocoonSandboxClient() @@ -105,9 +107,8 @@ async def go(): monkeypatch.setattr(inner, "_sandbox", lambda: FakeSandbox()) with pytest.raises(TimeoutError): await inner._exec_internal("sleep", "9", timeout=1.5) - assert seen == [["timeout", "-s", "KILL", "1.5", "sleep", "9"]] result = await inner._exec_internal("sleep", "9") - assert seen[-1] == ["sleep", "9"] and result.exit_code == 124 + assert seen == [1.5, None] and result.exit_code == 0 asyncio.run(go()) diff --git a/sdk/python/cocoonsandbox/conn.py b/sdk/python/cocoonsandbox/conn.py index f694190b..54235f5b 100644 --- a/sdk/python/cocoonsandbox/conn.py +++ b/sdk/python/cocoonsandbox/conn.py @@ -34,6 +34,11 @@ def __init__(self, sock: socket.socket, reader: BinaryIO): def send(self, op: str, **fields: object) -> None: self._sock.sendall(encode_request(op, **fields)) + def abort(self) -> None: + """Unblocks a reader parked in recv from another thread; close() still owns the socket.""" + with contextlib.suppress(OSError): + self._sock.shutdown(socket.SHUT_RDWR) + def recv(self) -> dict: """Returns the next frame; raises SilkdError on an error frame and ProtocolError on EOF, an oversized line, or an undecodable one — every diff --git a/sdk/python/cocoonsandbox/sandbox.py b/sdk/python/cocoonsandbox/sandbox.py index df912ff2..3d3f1a20 100644 --- a/sdk/python/cocoonsandbox/sandbox.py +++ b/sdk/python/cocoonsandbox/sandbox.py @@ -8,6 +8,7 @@ import contextlib import socket import threading +import time from collections.abc import Callable, Iterator from typing import TYPE_CHECKING @@ -55,10 +56,17 @@ def __exit__(self, *exc) -> None: raise def exec( - self, *argv: str, cwd: str = "", env: dict | None = None, user: str = "", session: str = "", stdin: bytes = b"" + self, + *argv: str, + cwd: str = "", + env: dict | None = None, + user: str = "", + session: str = "", + stdin: bytes = b"", + timeout: float | None = None, ) -> str: """Runs argv to completion and returns stdout; a non-zero exit raises - ExitError carrying stderr.""" + ExitError carrying stderr. timeout is run()'s wall clock.""" out, err = bytearray(), bytearray() code = self.run( list(argv), @@ -69,6 +77,7 @@ def exec( stdin=stdin, on_stdout=out.extend, on_stderr=err.extend, + timeout=timeout, ) if code != 0: raise ExitError(code, err.decode(errors="replace"), out.decode(errors="replace")) @@ -84,17 +93,39 @@ def run( stdin: bytes = b"", on_stdout: Callable[[bytes], object] | None = None, on_stderr: Callable[[bytes], object] | None = None, + timeout: float | None = None, ) -> int: """Runs argv streaming stdio through the callbacks (raw bytes — chunk - boundaries may split multi-byte sequences); returns the exit code.""" - with self._dial() as conn: + boundaries may split multi-byte sequences); returns the exit code. + timeout is a wall clock over the dial and the run: at its end the + connection is cut, which kills the command, and TimeoutError is raised.""" + if timeout is not None and timeout <= 0: + raise ValueError("timeout must be positive") + deadline = None if timeout is None else time.monotonic() + timeout + expired = threading.Event() + try: + conn = self._dial(deadline) + except ProtocolError: + if deadline is not None and time.monotonic() >= deadline: + raise TimeoutError(f"command did not finish within {timeout}s") from None + raise + with conn: conn.send( "exec", argv=argv, cwd=cwd or None, env=env, user=user or None, detach=False, session=session or None ) # the guest stops draining stdin while blocked on stdout, so feeding it fully first deadlocks. pump = threading.Thread(target=_feed_stdin, args=(conn, stdin), daemon=True) pump.start() - code = _pump_stdio(conn, on_stdout, on_stderr) + watchdog = _arm_watchdog(conn, deadline, expired) + try: + code = _pump_stdio(conn, on_stdout, on_stderr) + except ProtocolError: + if expired.is_set(): + raise TimeoutError(f"command did not finish within {timeout}s") from None + raise + finally: + if watchdog is not None: + watchdog.cancel() pump.join() # the closed conn fails a stalled send, so this cannot hang if code is None: raise ProtocolError("exec stream ended without an exit frame") @@ -346,8 +377,11 @@ def close(self) -> None: if exc.status != 404: raise - def _dial(self) -> Conn: - return dial_agent(self.owner, self.id, self.token, self._client.timeout) + def _dial(self, deadline: float | None = None) -> Conn: + timeout = self._client.timeout + if deadline is not None: + timeout = max(min(timeout, deadline - time.monotonic()), 0.001) + return dial_agent(self.owner, self.id, self.token, timeout) def _open_stream(self, op: str, expect: str = "ready", **fields) -> tuple[Conn, dict]: """Dials, sends op, and waits for the handshake frame, closing the @@ -545,6 +579,20 @@ def _send_chunks(conn: Conn, data: bytes, op: str = "data", chunk: int = FS_CHUN conn.send(op, data=view[off : off + chunk]) +def _arm_watchdog(conn: Conn, deadline: float | None, expired: threading.Event) -> threading.Timer | None: + if deadline is None: + return None + + def cut() -> None: + expired.set() + conn.abort() + + watchdog = threading.Timer(max(deadline - time.monotonic(), 0), cut) + watchdog.daemon = True + watchdog.start() + return watchdog + + def _feed_stdin(conn: Conn, stdin: bytes) -> None: with contextlib.suppress(SandboxError, OSError): # the reader reports the real failure if stdin: diff --git a/sdk/python/tests/test_proc.py b/sdk/python/tests/test_proc.py index 8cb4bd64..d03e22dd 100644 --- a/sdk/python/tests/test_proc.py +++ b/sdk/python/tests/test_proc.py @@ -45,7 +45,7 @@ def test_attach_returns_exit_code(monkeypatch): def test_run_pumps_stdin_while_reading_output(monkeypatch): sb = Sandbox(client=Client("127.0.0.1:1"), id="sb_1", token="tok", owner="127.0.0.1:1") blocking = BlockingStdinConn([{"type": "exit", "code": 0}], buffer_frames=1) - monkeypatch.setattr(sb, "_dial", lambda: blocking) + monkeypatch.setattr(sb, "_dial", lambda deadline=None: blocking) assert sb.run(["cat"], stdin=b"x" * (FS_CHUNK * 3)) == 0 assert [op for op, _ in blocking.sent].count("stdin") == 3 @@ -105,5 +105,5 @@ def recv(self): def fake_sandbox(monkeypatch, frames): sb = Sandbox(client=Client("127.0.0.1:1"), id="sb_1", token="tok", owner="127.0.0.1:1") conn = FakeConn(frames) - monkeypatch.setattr(sb, "_dial", lambda: conn) + monkeypatch.setattr(sb, "_dial", lambda deadline=None: conn) return sb, conn diff --git a/sdk/python/tests/test_stream.py b/sdk/python/tests/test_stream.py index 6964d61b..0ec5a0de 100644 --- a/sdk/python/tests/test_stream.py +++ b/sdk/python/tests/test_stream.py @@ -26,6 +26,18 @@ def serve_port_forward(server: socket.socket, quiet: float, ops: list[str]) -> N conn.close() +def serve_started_then_hang(server: socket.socket, quiet: float) -> None: + conn, _ = server.accept() + reader = conn.makefile("rb") + while reader.readline() not in (b"\r\n", b""): + pass + conn.sendall(b"HTTP/1.1 101 Switching Protocols\r\n\r\n") + reader.readline() + conn.sendall(b'{"type":"started","pid":7}\n') + time.sleep(quiet) + conn.close() + + def serve_silence(server: socket.socket) -> None: conn, _ = server.accept() conn.recv(4096) @@ -48,6 +60,26 @@ def test_port_stream_outlives_the_client_timeout(): assert ops == ["port_forward"] +def test_run_timeout_cuts_a_silent_command(): + server = socket.create_server(("127.0.0.1", 0)) + addr = f"127.0.0.1:{server.getsockname()[1]}" + threading.Thread(target=serve_started_then_hang, args=(server, 5 * TIMEOUT), daemon=True).start() + sb = Sandbox(client=Client(addr, timeout=TIMEOUT), id="sb_1", token="tok", owner=addr) + started = time.monotonic() + try: + with pytest.raises(TimeoutError): + sb.run(["sleep", "9"], timeout=TIMEOUT) + finally: + server.close() + assert time.monotonic() - started < 3 * TIMEOUT + + +def test_run_rejects_a_non_positive_timeout(): + sb = Sandbox(client=Client("127.0.0.1:1", timeout=TIMEOUT), id="sb_1", token="tok", owner="127.0.0.1:1") + with pytest.raises(ValueError): + sb.run(["true"], timeout=0) + + def test_dial_is_still_bounded_by_the_client_timeout(): server = socket.create_server(("127.0.0.1", 0)) addr = f"127.0.0.1:{server.getsockname()[1]}" diff --git a/sdk/python/tests/test_wire_binding.py b/sdk/python/tests/test_wire_binding.py index 91041457..4327ec2a 100644 --- a/sdk/python/tests/test_wire_binding.py +++ b/sdk/python/tests/test_wire_binding.py @@ -121,7 +121,7 @@ def test_git_branch_actions_come_from_the_corpus(monkeypatch): enums = json.loads((FIXTURES / "enums.json").read_text()) sb = Sandbox(client=Client("127.0.0.1:1"), id="sb_1", token="tok", owner="127.0.0.1:1") sent = [] - monkeypatch.setattr(sb, "_dial", lambda: BranchActionConn(sent)) + monkeypatch.setattr(sb, "_dial", lambda deadline=None: BranchActionConn(sent)) sb.git_branches("/w") sb.git_create_branch("/w", "b") @@ -182,5 +182,5 @@ def guest(): thread.start() sb = Sandbox(client=Client("127.0.0.1:1"), id="sb_1", token="tok", owner="127.0.0.1:1") - monkeypatch.setattr(sb, "_dial", lambda: Conn(client_sock, client_sock.makefile("rb"))) + monkeypatch.setattr(sb, "_dial", lambda deadline=None: Conn(client_sock, client_sock.makefile("rb"))) return sb, sent, thread From bd835420c4f7c83b79e1267830e595888de72d99 Mon Sep 17 00:00:00 2001 From: CMGS Date: Mon, 14 Sep 2026 19:59:43 +0800 Subject: [PATCH 17/40] silkd: bound the replace read through one handle fs.replace still sized the file by path and then read it whole by path; both verbs now share read_bounded, which opens once, reserves the size it measured and stops at FIND_MAX_FILE. --- silkd/src/find.rs | 44 +++++++++++++++++++++++++------------------- 1 file changed, 25 insertions(+), 19 deletions(-) diff --git a/silkd/src/find.rs b/silkd/src/find.rs index 922e8bd1..e522c7e6 100644 --- a/silkd/src/find.rs +++ b/silkd/src/find.rs @@ -5,7 +5,6 @@ use std::path::{Path, PathBuf}; use std::sync::Arc; use regex::{Regex, Replacer}; -use tokio::fs; use tokio::io::{AsyncBufRead, AsyncBufReadExt, AsyncWrite}; use tokio::runtime::Handle; use tokio::sync::{Semaphore, SemaphorePermit, mpsc}; @@ -102,16 +101,9 @@ impl Walk<'_> { Ok(true) } - /// Scans one file; the size check and the read share one handle, and the read itself stops at the bound. + /// Scans one file; an unreadable or oversized file is skipped, not an error. fn scan_file(&self, path: &Path, body: &mut String) -> bool { - let Ok(file) = std::fs::File::open(path) else { - return true; - }; - if file.metadata().is_ok_and(|m| m.len() > FIND_MAX_FILE) { - return true; - } - body.clear(); - if file.take(FIND_MAX_FILE).read_to_string(body).is_err() { + if !matches!(read_bounded(path, body), Ok(true)) { return true; } let name: Arc = path.to_string_lossy().into(); @@ -176,11 +168,18 @@ pub async fn replace( replacement: &replacement, count: 0, }; - if !is_oversized(&file).await { - let body = match fs::read_to_string(&file).await { - Ok(body) => body, - Err(e) => return err_frame(w, &e, "read").await, - }; + let path = PathBuf::from(&file); + let read = tokio::task::spawn_blocking(move || { + let mut body = String::new(); + read_bounded(&path, &mut body).map(|within| within.then_some(body)) + }) + .await; + let body = match read { + Ok(Ok(body)) => body, + Ok(Err(e)) => return err_frame(w, &e, "read").await, + Err(e) => return err_frame(w, &std::io::Error::other(e), "read").await, + }; + if let Some(body) = body { let new = re.replace_all(&body, counting.by_ref()); if counting.count > 0 && let Err(e) = crate::fs::write_atomic(Path::new(&file), new.as_bytes()).await @@ -278,10 +277,17 @@ where } } -async fn is_oversized(file: &str) -> bool { - fs::metadata(file) - .await - .is_ok_and(|m| m.len() > FIND_MAX_FILE) +/// Reads a file into `body` through one handle, stopping at FIND_MAX_FILE; `Ok(false)` when the file is above the bound. +fn read_bounded(path: &Path, body: &mut String) -> std::io::Result { + let file = std::fs::File::open(path)?; + let len = file.metadata()?.len(); + if len > FIND_MAX_FILE { + return Ok(false); + } + body.clear(); + body.reserve(len as usize); + file.take(FIND_MAX_FILE).read_to_string(body)?; + Ok(true) } fn match_cost(frame: &Response) -> usize { From 5ae1fd00ef3bea974d0334276264a91f307be3e8 Mon Sep 17 00:00:00 2001 From: CMGS Date: Mon, 14 Sep 2026 19:59:43 +0800 Subject: [PATCH 18/40] review: simplify-lens follow-ups in silkd and boot-init Marks.render goes back to the iterator form (a once-per-boot path with at most seven marks), and routes_directly latches through Option::inspect. --- boot/init/src/boot.rs | 10 ++++------ silkd/src/net.rs | 14 +++++--------- 2 files changed, 9 insertions(+), 15 deletions(-) diff --git a/boot/init/src/boot.rs b/boot/init/src/boot.rs index c495bded..a69f5beb 100644 --- a/boot/init/src/boot.rs +++ b/boot/init/src/boot.rs @@ -1,6 +1,5 @@ //! Boot sequence: early mounts, resolve disks, overlay, persist network, switch_root, exec. -use std::fmt::Write; use std::fs; use std::path::Path; use std::time::{Duration, Instant}; @@ -34,11 +33,10 @@ impl Marks { } fn render(&self) -> String { - let mut out = String::new(); - for (label, us) in &self.points { - let _ = write!(out, " {label}@{us}us"); - } - out + self.points + .iter() + .map(|(label, us)| format!(" {label}@{us}us")) + .collect() } } diff --git a/silkd/src/net.rs b/silkd/src/net.rs index 22ba7876..50e13ff3 100644 --- a/silkd/src/net.rs +++ b/silkd/src/net.rs @@ -40,17 +40,13 @@ pub fn routes_directly() -> bool { return false; } // the host writes the file once per claim, so a present verdict latches; only absence re-reads. - let lane = match LANE.load(Ordering::Relaxed) { - -1 => match lane_verdict(LANE_FILE) { - Some(relay) => { - LANE.store(i8::from(relay), Ordering::Relaxed); - relay - } - None => false, - }, + let relay = match LANE.load(Ordering::Relaxed) { + -1 => lane_verdict(LANE_FILE) + .inspect(|&r| LANE.store(i8::from(r), Ordering::Relaxed)) + .unwrap_or(false), v => v == 1, }; - !lane + !relay } /// Lane override for tests: set_var would race every concurrent getenv. From 5f0f600638ea98544c99a441dacad1c3cf7f2dc0 Mon Sep 17 00:00:00 2001 From: CMGS Date: Mon, 14 Sep 2026 19:59:43 +0800 Subject: [PATCH 19/40] docs: align the pages with the code README lists the boot-init and shell workflows; index drops the network adb path the android flavor never had; the desktop pages name the whole app set; performance records the refill-time door bind; sdk-python states the peer-heal TTL requirement and the run/exec timeout; the adapter, langchain, mcp and API pages describe the exec timeout, the capped exec result, the bounded read_file, the archived 409 and the oversized audit record. --- README.md | 2 ++ docs/deploy.md | 2 +- docs/desktop.md | 3 ++- docs/index.md | 2 +- docs/langchain.md | 2 +- docs/mcp.md | 4 ++-- docs/openai-adapter.md | 2 +- docs/performance.md | 4 ++++ docs/sandboxd-api.md | 11 +++++++---- docs/sdk-python.md | 11 ++++++++--- os-image/desktop/README.md | 3 ++- 11 files changed, 31 insertions(+), 15 deletions(-) diff --git a/README.md b/README.md index 3946485e..825d6ac6 100644 --- a/README.md +++ b/README.md @@ -116,7 +116,9 @@ TEMPLATE=rt:24.04 scripts/sandboxd-e2e.sh ## CI - `silkd.yml` / `sandboxd.yml` — Rust and Go test+lint suites +- `boot-init.yml` — the boot/init crate's own fmt+clippy+test gate - `python.yml` — ruff + pytest for the three Python packages +- `shell.yml` — shellcheck over every tracked shell script - `images.yml` — the single image entry point: on a push touching `boot/**`, `silkd/**`, `protocol/**`, or `os-image/**` it builds the changed carriers (via `build-boot.yml` / `build-silkd.yml`, diff --git a/docs/deploy.md b/docs/deploy.md index 59469bdc..d722587d 100644 --- a/docs/deploy.md +++ b/docs/deploy.md @@ -120,7 +120,7 @@ sandboxd reads one JSON file (`-config`, default | `warm_max` (pool entry) | 0 (static) | turns on the demand-adaptive watermark for that pool: the warm target rises from `warm` toward `warm_max` while claims arrive faster than the measured provision lead covers, and decays back over ~a minute of silence | | `warmup` (pool entry) | unset | argv run in the golden VM after readiness and before its snapshot, so the files it touches are page-cache-resident in every clone, and again in every clone before it joins the warm pool, so those pages are already faulted into the restored VM when the first command runs — e.g. `["node", "-e", "0"]` on a Node flavor. It runs under the engine's 2-minute command timeout in silkd's base environment (`PATH`, `TERM`, and the guest image's proxy variables wherever nothing routes directly — the none lane and the locked bridge egress lane — with the relay not yet armed, since arming happens at claim); a non-zero exit or a timeout fails the golden build, so the pool stays unfilled until the config is fixed. Config-owned like `egress`: `PUT /v1/pools` rejects it, and a golden built with a different warmup is rebuilt | | `max_claims` | 0 (unlimited) | node-wide cap on live claims; claim/fork/branch requests beyond it answer 429 with the pool state unharmed (on a cluster, normal warm-candidate placement applies, with volume claims limited to candidates holding every requested volume) | -| `audit_log` | false | append every relayed request frame's op + addressing fields (never payloads) to `/audit.jsonl`, size-rotated with one `.1` backup. Records are `{t, id, op}` plus whichever addressing fields the op carries (`argv`, `path`, `dest`, `from`, `to`, `url`, `session`, `port`), plus `method` (`GET`, `CONNECT`, `SOCKS5`, …), `decision` and `secret` (the ref name, never its value) on `egress` records; preview accesses record as op `preview`, one per request. A request frame whose first line exceeds 4 KiB is skipped, never truncated | +| `audit_log` | false | append every relayed request frame's op + addressing fields (never payloads) to `/audit.jsonl`, size-rotated with one `.1` backup. Records are `{t, id, op}` plus whichever addressing fields the op carries (`argv`, `path`, `dest`, `from`, `to`, `url`, `session`, `port`), plus `method` (`GET`, `CONNECT`, `SOCKS5`, …), `decision` and `secret` (the ref name, never its value) on `egress` records; preview accesses record as op `preview`, one per request. A request frame whose first line exceeds 4 KiB records as op `oversized` with no addressing fields | | `idle_hibernate_seconds` | 0 (off) | node-wide idle policy for unpooled claims (template/checkpoint claims): a none-lane claim is hibernated once it has had no open data-plane connection (relay, buffered exec, preview) and no egress request in flight for this long; the clock restarts when the last connection closes, so a long command is never cut short. The next call that reaches the guest wakes it transparently. Per-pool `idle_hibernate_seconds` does the same for that pool's claims; pooled keys ignore the node-wide value, and egress pools reject it because they cannot resume safely. Opt in deliberately: a wake costs latency and the snapshot, so callers with their own idle logic must not pay twice | | `archive_after_seconds` | 0 (off) | tier below hibernation: a hibernated claim idle this long is checkpointed to the store and its local VM dropped, freeing the node entirely; the next call that reaches the guest restores it transparently (a checkpoint restore's latency) with a fresh server-default 5m lease. Requires `idle_hibernate_seconds > 0` and must exceed it. Node-wide for unpooled keys; per-pool overrides for that pool | | `archive_delete_after_seconds` | 0 (keep) | purge an archived claim's store checkpoint this long after it was archived, reclaiming storage; the claim is then gone for good. On archive this retention window replaces the live claim deadline; 0 clears the deadline so the archive is kept forever. Same node-wide/per-pool split | diff --git a/docs/desktop.md b/docs/desktop.md index 1918020b..a1fc2838 100644 --- a/docs/desktop.md +++ b/docs/desktop.md @@ -2,7 +2,8 @@ The `desktop` flavor boots a GNOME session (Ubuntu session on Xvfb, 1920x1080) with the OSWorld guest server on guest loopback `5000` and the OSWorld app -set (Google Chrome, LibreOffice, GIMP, VLC). A computer-use agent or the +set (Google Chrome, LibreOffice, GIMP, VLC, Thunderbird, VS Code, Zotero, +Obsidian, Shotcut, FreeCAD, WPS Office, MuseScore). A computer-use agent or the [OSWorld](https://github.com/xlang-ai/OSWorld-V2) harness claims it and drives the desktop through the same HTTP contract the OSWorld AWS and docker guests speak — screenshot, AT-SPI accessibility tree, PyAutoGUI diff --git a/docs/index.md b/docs/index.md index f934122e..7740a280 100644 --- a/docs/index.md +++ b/docs/index.md @@ -36,7 +36,7 @@ vsock-only I/O (hardened default); `net=egress` attaches a bridge/CNI NIC. - [OpenAI Agents SDK adapter](openai-adapter.md) — run Agents SDK tools inside cocoon microVMs via the custom sandbox-provider interface - [Android sandboxes](android.md) — the redroid flavor: claim shape, adb - access through the relay or the network, checkpoint/branch + access through the relay, checkpoint/branch - [Browser sandboxes](browser.md) — headless Chromium with CDP through the relay: Playwright/Puppeteer access, checkpoint/branch of a live browser diff --git a/docs/langchain.md b/docs/langchain.md index 60ce0753..06643ab1 100644 --- a/docs/langchain.md +++ b/docs/langchain.md @@ -19,7 +19,7 @@ schemas, sync-native with `asyncio.to_thread` async bridges): | tool | what it does | |---|---| -| `sandbox_exec` | run a shell command, cut off after 5 minutes (exit code 124); stdout/stderr/exit code; disk state persists across calls | +| `sandbox_exec` | run a shell command, cut off after 5 minutes with the reply saying so; stdout/stderr/exit code; disk state persists across calls | | `sandbox_write_file` | write a text file (atomic on the guest) | | `sandbox_read_file` | read a text file | | `sandbox_list_dir` | list a directory as JSON | diff --git a/docs/mcp.md b/docs/mcp.md index 2ac80f31..18cc7431 100644 --- a/docs/mcp.md +++ b/docs/mcp.md @@ -25,12 +25,12 @@ Build: `cd mcp && go build -o sandbox-mcp .` | tool | what it does | |---|---| | `create_sandbox` | claim a microVM and return its id plus deadline; optional `template`, `net` (`none` default, or `egress`), `size` (`small` default, `medium`, `large`, `xlarge`, `2xlarge`) and `ttl_seconds` (0 means one hour); warm claims take milliseconds, nothing renews the deadline | -| `exec` | run a shell command to completion (5-minute cap); returns stdout, stderr and the exit code; a hibernated sandbox wakes transparently | +| `exec` | run a shell command to completion (5-minute cap); returns stdout, stderr and the exit code, each stream keeping its first 1 MiB with `truncated` set past that; a cut-off or dropped run returns the output collected so far next to an `error` field; a hibernated sandbox wakes transparently | | `spawn` | start a command detached and return its pid; output goes to a 256 KiB ring buffer that `logs` replays | | `ps` | list tracked processes (exec, spawn, pty) with state, exit code and start time | | `logs` | replay up to 256 KiB of a tracked process's newest whole stdout/stderr chunks (+ exit code once ended) | | `kill` | signal a tracked process (0 = SIGKILL); an exited process is a no-op success | -| `write_file` / `read_file` / `list_dir` | atomic whole-file write (parent must exist); whole-file text read (invalid UTF-8 replaced, missing path is an error); one-level listing of `{name, kind, size}` entries | +| `write_file` / `read_file` / `list_dir` | atomic whole-file write (parent must exist); whole-file text read of a regular file up to 1 MiB (a larger file, a directory or a device is an error, invalid UTF-8 replaced, missing path is an error); one-level listing of `{name, kind, size}` entries | | `fork` | clone into N children (1 to the node's `max_fork_count`, default 16) carrying exact memory + disk state, all-or-nothing; the parent keeps running | | `checkpoint` | capture full state without stopping; returns a `checkpoint_id` that can be branched repeatedly | | `branch_checkpoint` | claim a fresh sandbox from a checkpoint's captured moment | diff --git a/docs/openai-adapter.md b/docs/openai-adapter.md index c0377158..47466537 100644 --- a/docs/openai-adapter.md +++ b/docs/openai-adapter.md @@ -32,7 +32,7 @@ pair over the sync [Python SDK](sdk-python.md), bridged with | SDK surface | cocoon | |---|---| | `create` | `Client.new` — one claimed sandbox per session | -| session `exec` | `Sandbox.run` (stdout/stderr/exit) | +| session `exec` | `Sandbox.run` (stdout/stderr/exit); a per-call `timeout` is `run`'s wall clock and surfaces as `TimeoutError` | | `read` / `write` | `Sandbox.read_file` / `write_file`; missing → `FileNotFoundError` | | `persist_workspace` / `hydrate_workspace` | `Sandbox.pull` / `push` (tar) | | exposed port | `Sandbox.proxy_port` | diff --git a/docs/performance.md b/docs/performance.md index fdac6472..1e92f532 100644 --- a/docs/performance.md +++ b/docs/performance.md @@ -23,6 +23,10 @@ bare metal, `small` tier: | pool miss, golden exists | **~26–39 ms** | clone from the golden snapshot + entropy/machine-id reseed + readiness probe | | cold boot (no golden yet) | **~215–400 ms** | full boot from the template image to silkd answering | +A guarded-egress claim binds its proxy doors at refill rather than at claim +(#177): on bare metal that took the warm claim with the HTTP door from 307 to +263 µs p50, and with both doors from 351 to 274 µs. + Cloud Hypervisor lifecycle latency (bare metal, vsock agent-ready): | path | latency | diff --git a/docs/sandboxd-api.md b/docs/sandboxd-api.md index ed85b9ed..60c40156 100644 --- a/docs/sandboxd-api.md +++ b/docs/sandboxd-api.md @@ -259,7 +259,8 @@ All-or-nothing: on error no child survived. 200 with one claim per child: Children inherit the parent's tenant and count against its `max_claims`, whoever calls. 400 invalid count or body, 401 bad api token, 404 unknown id or wrong sandbox token, 409 egress-lane or volume parent (neither forks, -checkpoints, or promotes; see [egress](egress.md)), 429 node or the parent's +checkpoints, or promotes; see [egress](egress.md)) or an archived parent (an +exec or file call wakes it first), 429 node or the parent's tenant at `max_claims`, or the node draining. ## POST /v1/sandboxes/{id}/promote @@ -297,7 +298,8 @@ digest; changing any exported path or bytes changes it. 400 invalid name, 401 bad api token, 409 when the name collides with a configured pool, the template is owned by another tenant, or the sandbox is -on the egress lane or has volumes attached (see [egress](egress.md)), 404 unknown id or wrong +on the egress lane, has volumes attached (see [egress](egress.md)), or is +archived (an exec or file call wakes it first), 404 unknown id or wrong sandbox token. ## DELETE /v1/templates?template=…&net=…&size=… @@ -363,8 +365,9 @@ Auth: node API token; body `{"token": "", "name": "..."}` answers `200 {"checkpoint": {id, name, sandbox_id, key, tenant?, created_at}}` — `tenant` records the calling tenant, absent for root. 400 bad body or name, 401 bad api token, 404 unknown id or wrong sandbox -token, 409 egress-lane sandbox or one with volumes attached (see -[egress](egress.md)). +token, 409 egress-lane sandbox, one with volumes attached (see +[egress](egress.md)), or an archived one (an exec or file call wakes it +first). ## POST /v1/checkpoints/{id}/claim diff --git a/docs/sdk-python.md b/docs/sdk-python.md index 9147ed47..bcd3a3a9 100644 --- a/docs/sdk-python.md +++ b/docs/sdk-python.md @@ -243,8 +243,8 @@ heal a missing record locally; `delete()` acts on the handle's bound node. heal pulled — best-effort eventual cleanup, not a fleet-wide revocation. A peer that misses the broadcast (offline, partitioned, or joined later) keeps serving branches from its replica until the node's `checkpoint_ttl_hours` ages it out; -with that TTL at its default of 0 (keep forever), an unreachable peer's replica -has no cleanup bound at all. +enabling peer heal requires that TTL to be set, so every healed replica has a +cleanup bound. ## Language servers (LSP) @@ -301,9 +301,14 @@ code = sb.run(["bash", "-c", "make test"], cwd="/work", env={"CI": "1"}, user="ubuntu", stdin=input_bytes, on_stdout=lambda b: sys.stdout.buffer.write(b), - on_stderr=lambda b: sys.stderr.buffer.write(b)) + on_stderr=lambda b: sys.stderr.buffer.write(b), + timeout=600) # seconds; TimeoutError past it ``` +`timeout` on `exec` and `run` is a wall clock over the dial and the run: at +its end the connection is cut, which makes silkd kill the command, and +`TimeoutError` is raised. + `exec` returns stdout and raises `ExitError` on a non-zero exit — carrying `code`, `stderr`, and the `stdout` produced before it failed. `run` streams raw bytes through the callbacks (chunk boundaries may split multi-byte sequences) and returns the exit code. `user` de-escalates diff --git a/os-image/desktop/README.md b/os-image/desktop/README.md index 5d81d931..d7c35c69 100644 --- a/os-image/desktop/README.md +++ b/os-image/desktop/README.md @@ -6,7 +6,8 @@ plus a GNOME session (Ubuntu session, dock, Yaru) on an Xvfb `:1` display at ([xlang-ai/osworld-server](https://github.com/xlang-ai/osworld-server), pinned commit) on guest loopback `5000`, the OSWorld Chrome CDP bridge on loopback `9222`, and the OSWorld app set: Google Chrome, LibreOffice, GIMP, -VLC, Thunderbird, VS Code, Zotero, Obsidian, Shotcut, FreeCAD and WPS Office. +VLC, Thunderbird, VS Code, Zotero, Obsidian, Shotcut, FreeCAD, WPS Office and +MuseScore. The intended pool shape is `size: 2xlarge` (8 CPU / 16G); the session idles at ~0.5 GB anonymous memory with gnome-shell around 290 MB RSS. From 64b578c6faacb6f282be30b7243fe7e17911aa48 Mon Sep 17 00:00:00 2001 From: CMGS Date: Mon, 14 Sep 2026 20:08:13 +0800 Subject: [PATCH 20/40] sdk/go: stream a file into a writer; keep reading after a failed stdin close ReadFileTo hands each data frame to an io.Writer and stops when the writer refuses, so a caller can bound a read without buffering it. Run sent stdin_close synchronously and returned its send error before reading anything, so a guest that had already answered and closed cost the caller the exit code it had sent; the frames are read regardless. --- sdk/go/files.go | 9 +++++++++ sdk/go/files_test.go | 33 +++++++++++++++++++++++++++++++++ sdk/go/sandbox.go | 5 ++--- 3 files changed, 44 insertions(+), 3 deletions(-) diff --git a/sdk/go/files.go b/sdk/go/files.go index ba23eb00..042c8ada 100644 --- a/sdk/go/files.go +++ b/sdk/go/files.go @@ -3,6 +3,7 @@ package sandbox import ( "bytes" "context" + "io" "slices" "github.com/cocoonstack/sandbox/protocol/wire" @@ -31,6 +32,14 @@ func (s *Sandbox) ReadFile(ctx context.Context, path string) ([]byte, error) { return slices.Concat(chunks...), nil } +// ReadFileTo streams path into w; an error from w ends the read there. +func (s *Sandbox) ReadFileTo(ctx context.Context, path string, w io.Writer) error { + return s.downloadRPC(ctx, &wire.FsRead{Path: path}, func(b []byte) error { + _, err := w.Write(b) + return err + }) +} + // ListDir returns the entries of a directory (batched frames are concatenated). func (s *Sandbox) ListDir(ctx context.Context, path string) ([]wire.DirEntry, error) { frames, err := collectRPC[wire.Entries](ctx, s, &wire.FsList{Path: path}) diff --git a/sdk/go/files_test.go b/sdk/go/files_test.go index e319c8ef..6f0dc24e 100644 --- a/sdk/go/files_test.go +++ b/sdk/go/files_test.go @@ -2,11 +2,15 @@ package sandbox import ( "bytes" + "errors" "testing" + "github.com/cocoonstack/sandbox/protocol/wire" "github.com/cocoonstack/sandbox/sdk/go/silkd/silkdtest" ) +var errLimit = errors.New("limit") + func TestFilesRoundTrip(t *testing.T) { sb := fakeSandbox(t) ctx := t.Context() @@ -52,6 +56,23 @@ func TestFilesRoundTrip(t *testing.T) { } } +func TestReadFileToStopsWhenTheWriterRefuses(t *testing.T) { + sb := fakeSandbox(t) + ctx := t.Context() + body := bytes.Repeat([]byte("x"), 3*wire.BulkChunk) + if err := sb.WriteFile(ctx, "/big", body, nil); err != nil { + t.Fatalf("write: %v", err) + } + var whole bytes.Buffer + if err := sb.ReadFileTo(ctx, "/big", &whole); err != nil || whole.Len() != len(body) { + t.Fatalf("ReadFileTo: %d bytes, %v; want the whole file", whole.Len(), err) + } + limit := &limitWriter{max: wire.BulkChunk} + if err := sb.ReadFileTo(ctx, "/big", limit); !errors.Is(err, errLimit) { + t.Fatalf("ReadFileTo past the limit = %v, want the writer's error", err) + } +} + func TestReadMissingFileErrors(t *testing.T) { sb := fakeSandbox(t) if _, err := sb.ReadFile(t.Context(), "/nope"); err == nil { @@ -114,3 +135,15 @@ func fakeSandbox(t *testing.T) *Sandbox { fake := silkdtest.NewFake(t.TempDir()) return testSandbox(t, newAgentServer(t, fake.ServeConn)) } + +type limitWriter struct { + n, max int +} + +func (w *limitWriter) Write(p []byte) (int, error) { + if w.n+len(p) > w.max { + return 0, errLimit + } + w.n += len(p) + return len(p), nil +} diff --git a/sdk/go/sandbox.go b/sdk/go/sandbox.go index a722d606..44aec0e8 100644 --- a/sdk/go/sandbox.go +++ b/sdk/go/sandbox.go @@ -109,9 +109,8 @@ func (s *Sandbox) Run(ctx context.Context, cmd Cmd) (int, error) { defer done() if cmd.Stdin == nil { - if err = conn.Send(wire.StdinClose{}); err != nil { - return 0, fmt.Errorf("close stdin: %w", err) - } + // a guest that already answered and closed fails this send; the frames it sent still come back below + _ = conn.Send(wire.StdinClose{}) } else { go pumpStdin(conn, cmd.Stdin) } From dc453a859350851dc651af0862e5669da8d6b1d2 Mon Sep 17 00:00:00 2001 From: CMGS Date: Mon, 14 Sep 2026 20:08:13 +0800 Subject: [PATCH 21/40] mcp: bound read_file by the bytes it reads The stat-then-read guard was not a bound: a file can grow or be swapped between the two RPCs. read_file now streams through the capped writer and stops the read at 1 MiB with the same pointer to exec head/tail. --- mcp/server_test.go | 37 +++++++++++++++++++++++++++++++++++++ mcp/tools.go | 30 +++++++++++++++++------------- 2 files changed, 54 insertions(+), 13 deletions(-) diff --git a/mcp/server_test.go b/mcp/server_test.go index e8e64d57..2d7999ee 100644 --- a/mcp/server_test.go +++ b/mcp/server_test.go @@ -3,6 +3,7 @@ package main import ( "bufio" "bytes" + "encoding/base64" "encoding/json" "fmt" "io" @@ -90,6 +91,42 @@ func TestExecKeepsOutputWhenTheGuestDrops(t *testing.T) { } } +func TestReadFileStopsAtTheCap(t *testing.T) { + chunk := `{"type":"data","data":"` + base64.StdEncoding.EncodeToString(make([]byte, 256<<10)) + `"}` + "\n" + srv := newTestServer(t, func(w http.ResponseWriter, r *http.Request) bool { + if !strings.HasSuffix(r.URL.Path, "/agent") { + return false + } + conn, _, err := http.NewResponseController(w).Hijack() + if err != nil { + t.Errorf("hijack: %v", err) + return true + } + defer conn.Close() + br := bufio.NewReader(conn) + if _, err := io.WriteString(conn, "HTTP/1.1 101 Switching Protocols\r\nUpgrade: silkd\r\nConnection: Upgrade\r\n\r\n"); err != nil { + return true + } + if _, err := br.ReadString('\n'); err != nil { + return true + } + for range 8 { + if _, err := io.WriteString(conn, chunk); err != nil { + return true + } + } + _, _ = io.WriteString(conn, `{"type":"done"}`+"\n") + return true + }) + replies := serveLines(t, srv, + `{"jsonrpc":"2.0","id":1,"method":"tools/call","params":{"name":"create_sandbox","arguments":{}}}`, + `{"jsonrpc":"2.0","id":2,"method":"tools/call","params":{"name":"read_file","arguments":{"sandbox_id":"sb_1","path":"/dev/zero"}}}`, + ) + if text := toolText(t, replies[2]); !strings.Contains(text, "read_file cap") { + t.Errorf("read_file past the cap answered %q, want the cap error", text) + } +} + func TestCreateSandboxRejectsNegativeTTL(t *testing.T) { replies := serveLines(t, newTestServer(t, nil), `{"jsonrpc":"2.0","id":1,"method":"tools/call","params":{"name":"create_sandbox","arguments":{"ttl_seconds":-5}}}`) diff --git a/mcp/tools.go b/mcp/tools.go index 6aefb78a..a4364146 100644 --- a/mcp/tools.go +++ b/mcp/tools.go @@ -4,15 +4,18 @@ import ( "cmp" "context" "encoding/json" + "errors" "fmt" "strings" "time" - "github.com/cocoonstack/sandbox/protocol/wire" sandbox "github.com/cocoonstack/sandbox/sdk/go" ) -var tools = []tool{ +var ( + errOutputCap = errors.New("output past the cap") + + tools = []tool{ { "create_sandbox", "Claim a fresh microVM sandbox and return its id plus deadline. Warm claims take milliseconds; a cold template boots in well under a second. Every sandbox-scoped tool takes the returned sandbox_id. The sandbox is destroyed at its deadline unless released earlier; nothing renews it.", schema(props{"template": str("template image ref, or a name published by promote; empty uses the server default"), "net": str("network lane: none (default, no NIC, vsock-only I/O) or egress (bridge NIC, outbound network)"), "size": str("resource tier: small (default), medium, large, xlarge, 2xlarge"), "ttl_seconds": integer("sandbox lifetime in seconds; 0 means one hour, and nothing renews it")}), toolCreateSandbox, @@ -73,7 +76,8 @@ var tools = []tool{ }, {"release", "Destroy a sandbox and free its resources; files and processes inside it are lost. This session forgets the id, so a second release of it is rejected as unknown.", schema(props{"sandbox_id": str("id returned by create_sandbox, fork, or branch_checkpoint")}, "sandbox_id"), toolRelease}, {"node_info", "Report the connected node's warm pools, live claims, drain state, capacity, and mesh peers as JSON.", schema(props{}), toolNodeInfo}, -} + } +) // tool binds one MCP tool's spec to its handler, so a tool can never exist // in the listing without a dispatch entry. @@ -148,14 +152,19 @@ type cmdArgs struct { } // cappedOutput keeps the first execOutputCap bytes; an agent that tails a firehose must not grow this process. +// An exec keeps draining past the cap so the command runs on; a read stops there instead. type cappedOutput struct { strings.Builder + stopAtCap bool truncated bool } func (o *cappedOutput) Write(p []byte) (int, error) { n := len(p) if room := execOutputCap - o.Len(); len(p) > room { + if o.stopAtCap { + return 0, errOutputCap + } p, o.truncated = p[:room], true } _, _ = o.Builder.Write(p) @@ -274,18 +283,13 @@ func toolReadFile(ctx context.Context, s *server, raw json.RawMessage) (string, if err != nil { return "", err } - info, err := sb.Stat(ctx, args.Path) - if err != nil { - return "", err - } - if info.Kind != wire.FileKindFile || info.Size > execOutputCap { - return "", fmt.Errorf("%s is a %s of %d bytes; read_file returns a regular file up to %d bytes, use exec with head or tail", args.Path, info.Kind, info.Size, execOutputCap) - } - data, err := sb.ReadFile(ctx, args.Path) - if err != nil { + out := cappedOutput{stopAtCap: true} + if err := sb.ReadFileTo(ctx, args.Path, &out); errors.Is(err, errOutputCap) { + return "", fmt.Errorf("%s exceeds the %d-byte read_file cap; use exec with head, tail, or a checksum", args.Path, execOutputCap) + } else if err != nil { return "", err } - return string(data), nil + return out.String(), nil } func toolListDir(ctx context.Context, s *server, raw json.RawMessage) (string, error) { From 7f3bdabfb709703c20dba1f54106e10fa40464b0 Mon Sep 17 00:00:00 2001 From: CMGS Date: Mon, 14 Sep 2026 20:11:28 +0800 Subject: [PATCH 22/40] review: gofumpt the mcp tools table inside its var block --- mcp/tools.go | 120 +++++++++++++++++++++++++-------------------------- 1 file changed, 60 insertions(+), 60 deletions(-) diff --git a/mcp/tools.go b/mcp/tools.go index a4364146..0facce5b 100644 --- a/mcp/tools.go +++ b/mcp/tools.go @@ -16,66 +16,66 @@ var ( errOutputCap = errors.New("output past the cap") tools = []tool{ - { - "create_sandbox", "Claim a fresh microVM sandbox and return its id plus deadline. Warm claims take milliseconds; a cold template boots in well under a second. Every sandbox-scoped tool takes the returned sandbox_id. The sandbox is destroyed at its deadline unless released earlier; nothing renews it.", - schema(props{"template": str("template image ref, or a name published by promote; empty uses the server default"), "net": str("network lane: none (default, no NIC, vsock-only I/O) or egress (bridge NIC, outbound network)"), "size": str("resource tier: small (default), medium, large, xlarge, 2xlarge"), "ttl_seconds": integer("sandbox lifetime in seconds; 0 means one hour, and nothing renews it")}), toolCreateSandbox, - }, - { - "exec", "Run a shell command in a sandbox, wait for it to exit, and return stdout, stderr, and the exit code as JSON. The call is cut off after 5 minutes; for servers or long jobs use spawn instead. A failed or cut-off call still returns the output collected so far next to an error field. Each stream keeps its first 1 MiB and sets truncated beyond that. A hibernated sandbox wakes transparently on this call.", - schema(props{"sandbox_id": str("id returned by create_sandbox, fork, or branch_checkpoint"), "command": str("shell command, run via sh -c"), "cwd": str("working directory; empty runs in the guest's default")}, "sandbox_id", "command"), toolExec, - }, - { - "spawn", "Start a shell command detached in a sandbox and return its guest pid immediately. The process keeps running across later tool calls; its output goes to a per-process ring buffer that keeps up to 256 KiB of the newest whole output chunks, which logs replays. Use exec instead when you need the result now.", - schema(props{"sandbox_id": str("id returned by create_sandbox, fork, or branch_checkpoint"), "command": str("shell command, run via sh -c"), "cwd": str("working directory; empty runs in the guest's default")}, "sandbox_id", "command"), toolSpawn, - }, - { - "ps", "List the sandbox's tracked processes (exec, spawn, and pty commands) as JSON: pid, argv, detached, state (running or exited), exit_code once exited, and start time. Processes started inside the guest by other means are not listed.", - schema(props{"sandbox_id": str("id returned by create_sandbox, fork, or branch_checkpoint")}, "sandbox_id"), toolPs, - }, - { - "kill", "Send a signal to a tracked process. Killing a process that already exited is a no-op success; the guest never re-signals a reaped pid.", - schema(props{"sandbox_id": str("id returned by create_sandbox, fork, or branch_checkpoint"), "pid": integer("guest pid from spawn or ps"), "signal": integer("signal number, e.g. 15 for SIGTERM; 0 sends SIGKILL")}, "sandbox_id", "pid"), toolKill, - }, - { - "logs", "Return a tracked process's buffered stdout and stderr, plus exit_code once it has ended. The buffer keeps up to 256 KiB of the newest whole output chunks per process, so redirect long output to a file when it must be complete.", - schema(props{"sandbox_id": str("id returned by create_sandbox, fork, or branch_checkpoint"), "pid": integer("guest pid from spawn or ps")}, "sandbox_id", "pid"), toolLogs, - }, - { - "write_file", "Write text content to a file in a sandbox, replacing any existing file. The write is atomic (temp file plus rename); an existing file keeps its permission bits. The parent directory must already exist.", - schema(props{"sandbox_id": str("id returned by create_sandbox, fork, or branch_checkpoint"), "path": str("absolute path of the file"), "content": str("full file content; written verbatim")}, "sandbox_id", "path", "content"), toolWriteFile, - }, - { - "read_file", "Return the whole content of a regular file up to 1 MiB in a sandbox as text; bytes that are not valid UTF-8 are replaced, so binary content is lossy. A larger file, a directory, or a device is an error: use exec with head, tail, or a checksum instead. A missing path is an error.", - schema(props{"sandbox_id": str("id returned by create_sandbox, fork, or branch_checkpoint"), "path": str("absolute path of the file")}, "sandbox_id", "path"), toolReadFile, - }, - { - "list_dir", "List one directory in a sandbox (not recursive) as JSON entries with name, kind (file, dir, symlink, or other), and size in bytes.", - schema(props{"sandbox_id": str("id returned by create_sandbox, fork, or branch_checkpoint"), "path": str("absolute path of the directory")}, "sandbox_id", "path"), toolListDir, - }, - { - "fork", "Clone a sandbox into N independent children that start from its exact memory and disk state, including running processes; each child gets its own id and lives one hour. N is capped by the node's max_fork_count (default 16). All-or-nothing: on failure no child survives. The parent keeps running.", - schema(props{"sandbox_id": str("id returned by create_sandbox, fork, or branch_checkpoint"), "count": integer("number of children, 1 to the node's max_fork_count (default 16)")}, "sandbox_id", "count"), toolFork, - }, - { - "checkpoint", "Capture a sandbox's full state (memory, disk, running processes) without stopping it and return a checkpoint id. branch_checkpoint claims new sandboxes from that moment any number of times; a checkpoint outlives this session until it is deleted or expires under the node's checkpoint TTL.", - schema(props{"sandbox_id": str("id returned by create_sandbox, fork, or branch_checkpoint"), "name": str("optional human label")}, "sandbox_id"), toolCheckpoint, - }, - { - "branch_checkpoint", "Claim a fresh sandbox that resumes from a checkpoint's exact captured moment and return its id; it lives one hour. The checkpoint is unchanged and can be branched again.", - schema(props{"checkpoint_id": str("id returned by checkpoint or list_checkpoints")}, "checkpoint_id"), toolBranchCheckpoint, - }, - {"list_checkpoints", "List the node's checkpoints newest first: checkpoint_id, name, source sandbox_id, created_at.", schema(props{}), toolListCheckpoints}, - {"delete_checkpoint", "Delete the node's copy of a checkpoint and tell peers to drop theirs; a peer replica that misses that broadcast stays branchable until its own TTL sweep. Sandboxes already branched from it are unaffected.", schema(props{"checkpoint_id": str("id returned by checkpoint or list_checkpoints")}, "checkpoint_id"), toolDeleteCheckpoint}, - { - "hibernate", "Snapshot a sandbox and stop its VM, freeing memory while keeping its id, files, processes, and shell state. The next call that reaches the guest (exec, spawn, ps, kill, logs, the file tools) wakes it transparently in tens of milliseconds; fork, checkpoint, promote and release act on the snapshot without waking it.", - schema(props{"sandbox_id": str("id returned by create_sandbox, fork, or branch_checkpoint")}, "sandbox_id"), toolHibernate, - }, - { - "promote", "Publish the sandbox's current state as a named template on its node; later create_sandbox calls with that name (and the same net and size) start from it. Re-promoting to the same name replaces the template.", - schema(props{"sandbox_id": str("id returned by create_sandbox, fork, or branch_checkpoint"), "template_name": str("template name to publish as")}, "sandbox_id", "template_name"), toolPromote, - }, - {"release", "Destroy a sandbox and free its resources; files and processes inside it are lost. This session forgets the id, so a second release of it is rejected as unknown.", schema(props{"sandbox_id": str("id returned by create_sandbox, fork, or branch_checkpoint")}, "sandbox_id"), toolRelease}, - {"node_info", "Report the connected node's warm pools, live claims, drain state, capacity, and mesh peers as JSON.", schema(props{}), toolNodeInfo}, + { + "create_sandbox", "Claim a fresh microVM sandbox and return its id plus deadline. Warm claims take milliseconds; a cold template boots in well under a second. Every sandbox-scoped tool takes the returned sandbox_id. The sandbox is destroyed at its deadline unless released earlier; nothing renews it.", + schema(props{"template": str("template image ref, or a name published by promote; empty uses the server default"), "net": str("network lane: none (default, no NIC, vsock-only I/O) or egress (bridge NIC, outbound network)"), "size": str("resource tier: small (default), medium, large, xlarge, 2xlarge"), "ttl_seconds": integer("sandbox lifetime in seconds; 0 means one hour, and nothing renews it")}), toolCreateSandbox, + }, + { + "exec", "Run a shell command in a sandbox, wait for it to exit, and return stdout, stderr, and the exit code as JSON. The call is cut off after 5 minutes; for servers or long jobs use spawn instead. A failed or cut-off call still returns the output collected so far next to an error field. Each stream keeps its first 1 MiB and sets truncated beyond that. A hibernated sandbox wakes transparently on this call.", + schema(props{"sandbox_id": str("id returned by create_sandbox, fork, or branch_checkpoint"), "command": str("shell command, run via sh -c"), "cwd": str("working directory; empty runs in the guest's default")}, "sandbox_id", "command"), toolExec, + }, + { + "spawn", "Start a shell command detached in a sandbox and return its guest pid immediately. The process keeps running across later tool calls; its output goes to a per-process ring buffer that keeps up to 256 KiB of the newest whole output chunks, which logs replays. Use exec instead when you need the result now.", + schema(props{"sandbox_id": str("id returned by create_sandbox, fork, or branch_checkpoint"), "command": str("shell command, run via sh -c"), "cwd": str("working directory; empty runs in the guest's default")}, "sandbox_id", "command"), toolSpawn, + }, + { + "ps", "List the sandbox's tracked processes (exec, spawn, and pty commands) as JSON: pid, argv, detached, state (running or exited), exit_code once exited, and start time. Processes started inside the guest by other means are not listed.", + schema(props{"sandbox_id": str("id returned by create_sandbox, fork, or branch_checkpoint")}, "sandbox_id"), toolPs, + }, + { + "kill", "Send a signal to a tracked process. Killing a process that already exited is a no-op success; the guest never re-signals a reaped pid.", + schema(props{"sandbox_id": str("id returned by create_sandbox, fork, or branch_checkpoint"), "pid": integer("guest pid from spawn or ps"), "signal": integer("signal number, e.g. 15 for SIGTERM; 0 sends SIGKILL")}, "sandbox_id", "pid"), toolKill, + }, + { + "logs", "Return a tracked process's buffered stdout and stderr, plus exit_code once it has ended. The buffer keeps up to 256 KiB of the newest whole output chunks per process, so redirect long output to a file when it must be complete.", + schema(props{"sandbox_id": str("id returned by create_sandbox, fork, or branch_checkpoint"), "pid": integer("guest pid from spawn or ps")}, "sandbox_id", "pid"), toolLogs, + }, + { + "write_file", "Write text content to a file in a sandbox, replacing any existing file. The write is atomic (temp file plus rename); an existing file keeps its permission bits. The parent directory must already exist.", + schema(props{"sandbox_id": str("id returned by create_sandbox, fork, or branch_checkpoint"), "path": str("absolute path of the file"), "content": str("full file content; written verbatim")}, "sandbox_id", "path", "content"), toolWriteFile, + }, + { + "read_file", "Return the whole content of a regular file up to 1 MiB in a sandbox as text; bytes that are not valid UTF-8 are replaced, so binary content is lossy. A larger file, a directory, or a device is an error: use exec with head, tail, or a checksum instead. A missing path is an error.", + schema(props{"sandbox_id": str("id returned by create_sandbox, fork, or branch_checkpoint"), "path": str("absolute path of the file")}, "sandbox_id", "path"), toolReadFile, + }, + { + "list_dir", "List one directory in a sandbox (not recursive) as JSON entries with name, kind (file, dir, symlink, or other), and size in bytes.", + schema(props{"sandbox_id": str("id returned by create_sandbox, fork, or branch_checkpoint"), "path": str("absolute path of the directory")}, "sandbox_id", "path"), toolListDir, + }, + { + "fork", "Clone a sandbox into N independent children that start from its exact memory and disk state, including running processes; each child gets its own id and lives one hour. N is capped by the node's max_fork_count (default 16). All-or-nothing: on failure no child survives. The parent keeps running.", + schema(props{"sandbox_id": str("id returned by create_sandbox, fork, or branch_checkpoint"), "count": integer("number of children, 1 to the node's max_fork_count (default 16)")}, "sandbox_id", "count"), toolFork, + }, + { + "checkpoint", "Capture a sandbox's full state (memory, disk, running processes) without stopping it and return a checkpoint id. branch_checkpoint claims new sandboxes from that moment any number of times; a checkpoint outlives this session until it is deleted or expires under the node's checkpoint TTL.", + schema(props{"sandbox_id": str("id returned by create_sandbox, fork, or branch_checkpoint"), "name": str("optional human label")}, "sandbox_id"), toolCheckpoint, + }, + { + "branch_checkpoint", "Claim a fresh sandbox that resumes from a checkpoint's exact captured moment and return its id; it lives one hour. The checkpoint is unchanged and can be branched again.", + schema(props{"checkpoint_id": str("id returned by checkpoint or list_checkpoints")}, "checkpoint_id"), toolBranchCheckpoint, + }, + {"list_checkpoints", "List the node's checkpoints newest first: checkpoint_id, name, source sandbox_id, created_at.", schema(props{}), toolListCheckpoints}, + {"delete_checkpoint", "Delete the node's copy of a checkpoint and tell peers to drop theirs; a peer replica that misses that broadcast stays branchable until its own TTL sweep. Sandboxes already branched from it are unaffected.", schema(props{"checkpoint_id": str("id returned by checkpoint or list_checkpoints")}, "checkpoint_id"), toolDeleteCheckpoint}, + { + "hibernate", "Snapshot a sandbox and stop its VM, freeing memory while keeping its id, files, processes, and shell state. The next call that reaches the guest (exec, spawn, ps, kill, logs, the file tools) wakes it transparently in tens of milliseconds; fork, checkpoint, promote and release act on the snapshot without waking it.", + schema(props{"sandbox_id": str("id returned by create_sandbox, fork, or branch_checkpoint")}, "sandbox_id"), toolHibernate, + }, + { + "promote", "Publish the sandbox's current state as a named template on its node; later create_sandbox calls with that name (and the same net and size) start from it. Re-promoting to the same name replaces the template.", + schema(props{"sandbox_id": str("id returned by create_sandbox, fork, or branch_checkpoint"), "template_name": str("template name to publish as")}, "sandbox_id", "template_name"), toolPromote, + }, + {"release", "Destroy a sandbox and free its resources; files and processes inside it are lost. This session forgets the id, so a second release of it is rejected as unknown.", schema(props{"sandbox_id": str("id returned by create_sandbox, fork, or branch_checkpoint")}, "sandbox_id"), toolRelease}, + {"node_info", "Report the connected node's warm pools, live claims, drain state, capacity, and mesh peers as JSON.", schema(props{}), toolNodeInfo}, } ) From 7659b82b3177a059b06c57d039dde4f829017fe1 Mon Sep 17 00:00:00 2001 From: CMGS Date: Mon, 14 Sep 2026 20:20:25 +0800 Subject: [PATCH 23/40] pool: check for an archived claim under the transition lock The ErrArchived prechecks in Checkpoint, Promote and Fork ran before sb.Transition was taken, and archive() clears VMName under that lock, so a capture racing an archive could still snapshot an empty VM name. The check now lives in sourceSnap, which every capture path calls under the lock; the prechecks are gone. --- sandboxd/pool/checkpoint.go | 3 --- sandboxd/pool/fork.go | 3 --- sandboxd/pool/refill.go | 6 +++++- sandboxd/pool/template.go | 3 --- 4 files changed, 5 insertions(+), 10 deletions(-) diff --git a/sandboxd/pool/checkpoint.go b/sandboxd/pool/checkpoint.go index 6afbece4..7ae45cbd 100644 --- a/sandboxd/pool/checkpoint.go +++ b/sandboxd/pool/checkpoint.go @@ -46,9 +46,6 @@ func (m *Manager) Checkpoint(ctx context.Context, id string, cred Cred, name, te if !sb.Key.Capturable() { return types.Checkpoint{}, ErrNoEgressFork } - if sb.ArchiveCk != "" { - return types.Checkpoint{}, ErrArchived - } // See Hibernate: a started capture must finish even if the caller hangs up. ctx = context.WithoutCancel(ctx) ckpt, _, err := m.publishCheckpoint(ctx, sb, store.CheckpointID(randHex(8)), name, tenant, false) diff --git a/sandboxd/pool/fork.go b/sandboxd/pool/fork.go index b3ec96e8..06462f42 100644 --- a/sandboxd/pool/fork.go +++ b/sandboxd/pool/fork.go @@ -25,9 +25,6 @@ func (m *Manager) Fork(ctx context.Context, id string, cred Cred, count int, ttl if !sb.Key.Capturable() { return nil, ErrNoEgressFork } - if sb.ArchiveCk != "" { - return nil, ErrArchived - } if err := m.overQuota(count, sb.Tenant); err != nil { return nil, err } diff --git a/sandboxd/pool/refill.go b/sandboxd/pool/refill.go index 329a9cba..0ac83c51 100644 --- a/sandboxd/pool/refill.go +++ b/sandboxd/pool/refill.go @@ -284,8 +284,12 @@ func (m *Manager) exportGolden(ctx context.Context, snap, final string) error { return os.Rename(tmp, final) } -// sourceSnap picks the snapshot to export a claimed sandbox from, with its cleanup. +// sourceSnap picks the snapshot to export a claimed sandbox from, with its cleanup; the caller holds sb.Transition. func (m *Manager) sourceSnap(ctx context.Context, sb *types.Sandbox) (string, func(), error) { + // archive() clears VMName under the same lock, so this is the check that cannot race it + if sb.ArchiveCk != "" { + return "", nil, ErrArchived + } if sb.HibernateSnap != "" { return sb.HibernateSnap, func() {}, nil } diff --git a/sandboxd/pool/template.go b/sandboxd/pool/template.go index 2076d784..d9f0ca6e 100644 --- a/sandboxd/pool/template.go +++ b/sandboxd/pool/template.go @@ -37,9 +37,6 @@ func (m *Manager) Promote(ctx context.Context, id string, cred Cred, template, t if !sb.Key.Capturable() { return types.PoolKey{}, "", ErrNoEgressFork } - if sb.ArchiveCk != "" { - return types.PoolKey{}, "", ErrArchived - } key := types.PoolKey{Template: template, Net: sb.Key.Net, Size: sb.Key.Size} if m.pooled(key) { // a configured pool owns this key; promoting over it would change what refills produce From d66e083c64ed0e4c380d176d5c1b9ac914dcb33a Mon Sep 17 00:00:00 2001 From: CMGS Date: Mon, 14 Sep 2026 20:20:25 +0800 Subject: [PATCH 24/40] sdk/langchain: refuse the command when the claim used the whole call budget A lazy claim that consumed the five minutes handed the command a one-second timeout; the tool now reports the cut-off instead. --- sdk/langchain/cocoonsandbox_langchain/toolkit.py | 4 +++- 1 file changed, 3 insertions(+), 1 deletion(-) diff --git a/sdk/langchain/cocoonsandbox_langchain/toolkit.py b/sdk/langchain/cocoonsandbox_langchain/toolkit.py index 5a2a3aca..be74a151 100644 --- a/sdk/langchain/cocoonsandbox_langchain/toolkit.py +++ b/sdk/langchain/cocoonsandbox_langchain/toolkit.py @@ -143,7 +143,9 @@ def _exec(self, command: str, cwd: str = "") -> str: tail = "" try: sb = self.sandbox() - timeout = max(deadline - time.monotonic(), 1) + timeout = deadline - time.monotonic() + if timeout <= 0: + raise TimeoutError("the claim used the whole call budget") code = sb.run(["sh", "-c", command], cwd=cwd, on_stdout=out.append, on_stderr=errs.append, timeout=timeout) if code != 0: tail = f"exit code: {code}" From f8f7d649dbaf6b83f8b3558b5a0163e0e5c4224f Mon Sep 17 00:00:00 2001 From: CMGS Date: Mon, 14 Sep 2026 20:31:31 +0800 Subject: [PATCH 25/40] silkd: read only a regular file fs_read opened whatever the path named: a device streamed forever and a FIFO without a writer parked a blocking-pool thread in open for the life of the daemon. A path that is not a regular file now answers bad_request before the open. --- docs/silkd.md | 2 +- silkd/src/fs.rs | 17 +++++++++++++++-- silkd/tests/fs_e2e.rs | 10 ++++++++++ 3 files changed, 26 insertions(+), 3 deletions(-) diff --git a/docs/silkd.md b/docs/silkd.md index 11ab2886..da83385c 100644 --- a/docs/silkd.md +++ b/docs/silkd.md @@ -34,7 +34,7 @@ a frame only one side can parse fails CI. | exec | `exec {argv, cwd?, env?, user?, detach?, session?}` | `started{pid}` → `stdout/stderr{data}`… → `exit{code}`; the client may stream `stdin{data}` / `stdin_close`. `detach` returns after `started`; the process keeps a bounded output ring for later `logs`/`attach`; not combinable with `session` | | procs | `ps` / `kill {pid, signal?}` / `attach {pid}` / `logs {pid}` | `ps` answers one `procs{procs}` frame, `kill` a bare `done`; `logs` and `attach` stream `stdout/stderr{data}`… then `exit{code}` (`logs` closes with `done`, and so does `attach` when the process is gone before its exit). Handles are guest pids; any connection can list, signal, replay, or re-attach live | | sessions | `session_create {id?, cwd?, env?}` / `session_list` / `session_rm {id}` | `session_created{id}` / `sessions{sessions}` / `done`. A session is a real persistent bash; `exec` with `session` runs inside it. Idle sessions are reaped after 30 minutes | -| fs | `fs_write {path, mode?}` (+`data`/`data_end` frames) / `fs_read` / `fs_list` / `fs_stat` / `fs_mkdir {parents?}` / `fs_rm {recursive?}` / `fs_rename {from, to}` | `fs_read` streams `data{data}`… → `done`, `fs_list` streams 4096-entry `entries{entries}` batches → `done`, `fs_stat` answers one `stat{info}`, and the mutating verbs terminate with `done`. Streaming runs both directions; write commits atomically via temp+rename and inherits an overwritten file's mode | +| fs | `fs_write {path, mode?}` (+`data`/`data_end` frames) / `fs_read` / `fs_list` / `fs_stat` / `fs_mkdir {parents?}` / `fs_rm {recursive?}` / `fs_rename {from, to}` | `fs_read` streams `data{data}`… → `done` for a regular file and answers `bad_request` for a directory, device or FIFO, `fs_list` streams 4096-entry `entries{entries}` batches → `done`, `fs_stat` answers one `stat{info}`, and the mutating verbs terminate with `done`. Streaming runs both directions; write commits atomically via temp+rename and inherits an overwritten file's mode | | tree | `fs_push {dest}` (+tar as `data` frames) / `fs_pull {path}` | whole trees as tar streams through the guest tar: `fs_pull` streams `data{data}`… → `done`, `fs_push` terminates with `done` | | search | `fs_find {path, pattern, glob?}` → `match{file, line, content}`… → `done` / `fs_replace {files, pattern, replacement}` → `replaced{file, replacements}`… → `done` | regex as data, no shell quoting; `glob` is anchored `*`/`?` wildcards over file names; find skips binary and >8 MiB files, and replace skips the same 8 MiB bound with a zero count | | watch | `fs_watch {path, recursive?}` | `ready` once armed (events after it are guaranteed captured), then `event{kind, path}` until the client disconnects; watcher or delivery-queue overflow errors arrive as a terminal `error` instead of silently losing events | diff --git a/silkd/src/fs.rs b/silkd/src/fs.rs index f989ac76..7ee61841 100644 --- a/silkd/src/fs.rs +++ b/silkd/src/fs.rs @@ -9,7 +9,7 @@ use tokio::fs; use tokio::io::{AsyncBufRead, AsyncWrite, AsyncWriteExt}; use tokio::sync::mpsc; -use crate::proto::{self, DirEntry, FileInfo, FileKind, Response, err_frame}; +use crate::proto::{self, DirEntry, ErrorKind, FileInfo, FileKind, Response, err_frame}; /// Entries per `entries` frame; a worst-case 1.6KiB entry keeps a full batch under MAX_FRAME. pub const LIST_BATCH: usize = 4096; @@ -48,8 +48,21 @@ where proto::write_frame(w, &Response::Done).await } -/// Streams the file at `path` back as `data` frames, then `done`. +/// Streams the regular file at `path` back as `data` frames, then `done`. pub async fn read(w: &mut W, path: String) -> io::Result<()> { + // a device streams forever and a writerless FIFO blocks in open, so only a regular file is read + match fs::metadata(&path).await { + Ok(meta) if !meta.is_file() => { + return proto::error_frame( + w, + ErrorKind::BadRequest, + format!("{path}: not a regular file"), + ) + .await; + } + Err(e) => return err_frame(w, &e, "stat").await, + Ok(_) => {} + } let mut file = match fs::File::open(&path).await { Ok(f) => f, Err(e) => return err_frame(w, &e, "open").await, diff --git a/silkd/tests/fs_e2e.rs b/silkd/tests/fs_e2e.rs index 2f92c712..040f428a 100644 --- a/silkd/tests/fs_e2e.rs +++ b/silkd/tests/fs_e2e.rs @@ -139,6 +139,16 @@ async fn mkdir_rename_rm_lifecycle() { assert!(!moved.exists()); } +#[tokio::test] +async fn read_of_a_non_regular_file_is_bad_request() { + let dir = tempfile::tempdir().unwrap(); + for path in [dir.path().to_str().unwrap(), "/dev/null"] { + let frames = exchange(&[json!({"op":"fs_read","path":path}).to_string()]).await; + assert_eq!(type_of(&frames[0]), "error", "{path}: {frames:?}"); + assert_eq!(frames[0]["kind"], "bad_request", "{path}"); + } +} + #[tokio::test] async fn read_missing_path_is_not_found() { let frames = exchange(&[json!({"op":"fs_read","path":"/no/such/file/xyz"}).to_string()]).await; From 235d7e4a5458e59a2130b1d03e0b600a6a745b5d Mon Sep 17 00:00:00 2001 From: CMGS Date: Mon, 14 Sep 2026 20:31:31 +0800 Subject: [PATCH 26/40] mcp: refuse a non-regular file before streaming it read_file stats first and rejects a directory, device or FIFO with the kind in the message; the streamed 1 MiB cap stays the size bound. --- mcp/server_test.go | 7 ++++++- mcp/tools.go | 9 +++++++++ 2 files changed, 15 insertions(+), 1 deletion(-) diff --git a/mcp/server_test.go b/mcp/server_test.go index 2d7999ee..616c2dd9 100644 --- a/mcp/server_test.go +++ b/mcp/server_test.go @@ -107,7 +107,12 @@ func TestReadFileStopsAtTheCap(t *testing.T) { if _, err := io.WriteString(conn, "HTTP/1.1 101 Switching Protocols\r\nUpgrade: silkd\r\nConnection: Upgrade\r\n\r\n"); err != nil { return true } - if _, err := br.ReadString('\n'); err != nil { + req, err := br.ReadString('\n') + if err != nil { + return true + } + if strings.Contains(req, `"op":"fs_stat"`) { + _, _ = io.WriteString(conn, `{"type":"stat","info":{"kind":"file","size":0}}`+"\n") return true } for range 8 { diff --git a/mcp/tools.go b/mcp/tools.go index 0facce5b..75e02e12 100644 --- a/mcp/tools.go +++ b/mcp/tools.go @@ -9,6 +9,7 @@ import ( "strings" "time" + "github.com/cocoonstack/sandbox/protocol/wire" sandbox "github.com/cocoonstack/sandbox/sdk/go" ) @@ -283,6 +284,14 @@ func toolReadFile(ctx context.Context, s *server, raw json.RawMessage) (string, if err != nil { return "", err } + // a device or a FIFO is not a file: /dev/null reads as empty and a writerless FIFO blocks in open + info, err := sb.Stat(ctx, args.Path) + if err != nil { + return "", err + } + if info.Kind != wire.FileKindFile { + return "", fmt.Errorf("%s is a %s; read_file reads a regular file", args.Path, info.Kind) + } out := cappedOutput{stopAtCap: true} if err := sb.ReadFileTo(ctx, args.Path, &out); errors.Is(err, errOutputCap) { return "", fmt.Errorf("%s exceeds the %d-byte read_file cap; use exec with head, tail, or a checksum", args.Path, execOutputCap) From 7d2b5094bf3ef3d9dbfe9d113a36837a293e7453 Mon Sep 17 00:00:00 2001 From: CMGS Date: Mon, 14 Sep 2026 20:31:31 +0800 Subject: [PATCH 27/40] pool: read the VM name for the fork and checkpoint usage events under the transition lock archive() clears VMName under sb.Transition; the two events read it after their capture released the lock. --- sandboxd/pool/checkpoint.go | 2 +- sandboxd/pool/fork.go | 9 ++++++++- 2 files changed, 9 insertions(+), 2 deletions(-) diff --git a/sandboxd/pool/checkpoint.go b/sandboxd/pool/checkpoint.go index 7ae45cbd..b3aaab23 100644 --- a/sandboxd/pool/checkpoint.go +++ b/sandboxd/pool/checkpoint.go @@ -53,7 +53,7 @@ func (m *Manager) Checkpoint(ctx context.Context, id string, cred Cred, name, te return types.Checkpoint{}, err } m.counters.checkpoints.Add(1) - m.recordUsage(ctx, usageEvent{Event: "checkpoint", ID: sb.ID, VMName: sb.VMName, Reference: ckpt.ID}) + m.recordUsage(ctx, usageEvent{Event: "checkpoint", ID: sb.ID, VMName: lockedVMName(sb), Reference: ckpt.ID}) return ckpt, nil } diff --git a/sandboxd/pool/fork.go b/sandboxd/pool/fork.go index 06462f42..f4b16d0e 100644 --- a/sandboxd/pool/fork.go +++ b/sandboxd/pool/fork.go @@ -47,10 +47,17 @@ func (m *Manager) Fork(ctx context.Context, id string, cred Cred, count int, ttl for i, c := range children { ids[i] = c.ID } - m.recordUsage(ctx, usageEvent{Event: "fork", ID: sb.ID, VMName: sb.VMName, Children: ids}) + m.recordUsage(ctx, usageEvent{Event: "fork", ID: sb.ID, VMName: lockedVMName(sb), Children: ids}) return children, nil } +// lockedVMName reads the VM name under the lock that archive() clears it under. +func lockedVMName(sb *types.Sandbox) string { + sb.Transition.Lock() + defer sb.Transition.Unlock() + return sb.VMName +} + // forkClones clones a running source from a fresh snapshot, a hibernated one from an export. func (m *Manager) forkClones(ctx context.Context, sb *types.Sandbox, count int) ([]*types.Sandbox, error) { create, cleanup, err := m.forkSource(ctx, sb) From 2a90c246ddab9a09cf0f779bee4f3058ef2b3d33 Mon Sep 17 00:00:00 2001 From: CMGS Date: Mon, 14 Sep 2026 20:37:32 +0800 Subject: [PATCH 28/40] review: lint follow-ups The MCP read test reuses the hijack's err instead of shadowing it; the pool's lockedVMName sits below the Manager method set. --- mcp/server_test.go | 2 +- sandboxd/pool/fork.go | 14 +++++++------- 2 files changed, 8 insertions(+), 8 deletions(-) diff --git a/mcp/server_test.go b/mcp/server_test.go index 616c2dd9..67521c89 100644 --- a/mcp/server_test.go +++ b/mcp/server_test.go @@ -104,7 +104,7 @@ func TestReadFileStopsAtTheCap(t *testing.T) { } defer conn.Close() br := bufio.NewReader(conn) - if _, err := io.WriteString(conn, "HTTP/1.1 101 Switching Protocols\r\nUpgrade: silkd\r\nConnection: Upgrade\r\n\r\n"); err != nil { + if _, err = io.WriteString(conn, "HTTP/1.1 101 Switching Protocols\r\nUpgrade: silkd\r\nConnection: Upgrade\r\n\r\n"); err != nil { return true } req, err := br.ReadString('\n') diff --git a/sandboxd/pool/fork.go b/sandboxd/pool/fork.go index f4b16d0e..8b525ca2 100644 --- a/sandboxd/pool/fork.go +++ b/sandboxd/pool/fork.go @@ -51,13 +51,6 @@ func (m *Manager) Fork(ctx context.Context, id string, cred Cred, count int, ttl return children, nil } -// lockedVMName reads the VM name under the lock that archive() clears it under. -func lockedVMName(sb *types.Sandbox) string { - sb.Transition.Lock() - defer sb.Transition.Unlock() - return sb.VMName -} - // forkClones clones a running source from a fresh snapshot, a hibernated one from an export. func (m *Manager) forkClones(ctx context.Context, sb *types.Sandbox, count int) ([]*types.Sandbox, error) { create, cleanup, err := m.forkSource(ctx, sb) @@ -92,3 +85,10 @@ func (m *Manager) forkSource(ctx context.Context, sb *types.Sandbox) (vmProvisio return func(name string) (types.VMRecord, error) { return m.eng.Clone(ctx, exportDir, name, sb.Key) }, func() { _ = os.RemoveAll(dir) }, nil } + +// lockedVMName reads the VM name under the lock that archive() clears it under. +func lockedVMName(sb *types.Sandbox) string { + sb.Transition.Lock() + defer sb.Transition.Unlock() + return sb.VMName +} From 3cc28e74c4621ae9061a619fa8bddc296e1d76f4 Mon Sep 17 00:00:00 2001 From: CMGS Date: Mon, 14 Sep 2026 20:38:23 +0800 Subject: [PATCH 29/40] silkd: check the file kind on the opened descriptor A path could be swapped for a FIFO between the stat and the open, and a writerless FIFO still parked a blocking-pool thread there. fs_read now opens with O_NONBLOCK and fstats the descriptor it will stream, so the kind it checks is the kind it reads; the test adds a FIFO to the directory and /dev/null cases. --- silkd/src/fs.rs | 32 +++++++++++++++++--------------- silkd/tests/fs_e2e.rs | 10 +++++++++- 2 files changed, 26 insertions(+), 16 deletions(-) diff --git a/silkd/src/fs.rs b/silkd/src/fs.rs index 7ee61841..31b090fe 100644 --- a/silkd/src/fs.rs +++ b/silkd/src/fs.rs @@ -1,7 +1,7 @@ //! Filesystem verbs: the vsock boundary is the trust boundary, so paths are taken as-is. use std::io; -use std::os::unix::fs::PermissionsExt; +use std::os::unix::fs::{OpenOptionsExt, PermissionsExt}; use std::path::{Path, PathBuf}; use std::time::UNIX_EPOCH; @@ -50,21 +50,23 @@ where /// Streams the regular file at `path` back as `data` frames, then `done`. pub async fn read(w: &mut W, path: String) -> io::Result<()> { - // a device streams forever and a writerless FIFO blocks in open, so only a regular file is read - match fs::metadata(&path).await { - Ok(meta) if !meta.is_file() => { - return proto::error_frame( - w, - ErrorKind::BadRequest, - format!("{path}: not a regular file"), - ) - .await; + // O_NONBLOCK keeps a writerless FIFO from parking the open; the kind check runs on the opened fd, so a swap + // between check and open cannot turn the read into a device stream + let opened = tokio::task::spawn_blocking(move || { + let file = std::fs::OpenOptions::new() + .read(true) + .custom_flags(libc::O_NONBLOCK) + .open(&path)?; + let regular = file.metadata()?.is_file(); + Ok::<_, io::Error>((file, regular)) + }) + .await + .map_err(io::Error::other)?; + let mut file = match opened { + Ok((file, true)) => fs::File::from_std(file), + Ok((_, false)) => { + return proto::error_frame(w, ErrorKind::BadRequest, "not a regular file").await; } - Err(e) => return err_frame(w, &e, "stat").await, - Ok(_) => {} - } - let mut file = match fs::File::open(&path).await { - Ok(f) => f, Err(e) => return err_frame(w, &e, "open").await, }; if let Err(e) = proto::stream_data_frames(&mut file, w).await? { diff --git a/silkd/tests/fs_e2e.rs b/silkd/tests/fs_e2e.rs index 040f428a..c71b39ef 100644 --- a/silkd/tests/fs_e2e.rs +++ b/silkd/tests/fs_e2e.rs @@ -142,7 +142,15 @@ async fn mkdir_rename_rm_lifecycle() { #[tokio::test] async fn read_of_a_non_regular_file_is_bad_request() { let dir = tempfile::tempdir().unwrap(); - for path in [dir.path().to_str().unwrap(), "/dev/null"] { + let fifo = dir.path().join("pipe"); + let c_fifo = std::ffi::CString::new(fifo.to_str().unwrap()).unwrap(); + // SAFETY: a valid NUL-terminated path; mkfifo touches nothing else. + assert_eq!(unsafe { libc::mkfifo(c_fifo.as_ptr(), 0o600) }, 0); + for path in [ + dir.path().to_str().unwrap(), + "/dev/null", + fifo.to_str().unwrap(), + ] { let frames = exchange(&[json!({"op":"fs_read","path":path}).to_string()]).await; assert_eq!(type_of(&frames[0]), "error", "{path}: {frames:?}"); assert_eq!(frames[0]["kind"], "bad_request", "{path}"); From 120c83a1a46d3df0835c78e4482e5c4346425e62 Mon Sep 17 00:00:00 2001 From: CMGS Date: Mon, 14 Sep 2026 21:11:34 +0800 Subject: [PATCH 30/40] egress: track both halves of a tunnel so Close ends it Close closed only the guest side of a CONNECT or SOCKS5 tunnel. When the upstream stayed silent after the half-close, the copy goroutine, the upstream socket and the sandbox hold outlived the proxy for the life of the daemon. --- sandboxd/egress/proxy.go | 6 +++- sandboxd/egress/proxy_test.go | 61 +++++++++++++++++++++++++++++++++++ sandboxd/egress/socks.go | 4 +++ sandboxd/egress/socks_test.go | 34 +++++++++++++------ 4 files changed, 95 insertions(+), 10 deletions(-) diff --git a/sandboxd/egress/proxy.go b/sandboxd/egress/proxy.go index 0b9d5885..5da3ea2e 100644 --- a/sandboxd/egress/proxy.go +++ b/sandboxd/egress/proxy.go @@ -72,7 +72,7 @@ type Proxy struct { leafMu sync.Mutex leaves map[string]*tls.Certificate - // conns tracks hijacked tunnels and SOCKS5 connections, which http.Server.Close does not reach. + // conns tracks both halves of every tunnel and SOCKS5 connection, which http.Server.Close does not reach. connMu sync.Mutex conns map[net.Conn]struct{} closed bool @@ -176,6 +176,10 @@ func (p *Proxy) serveConnect(w http.ResponseWriter, r *http.Request) { return } defer func() { _ = upstream.Close() }() + if !p.track(upstream) { + return + } + defer p.untrack(upstream) client, _, err := http.NewResponseController(w).Hijack() if err != nil { http.Error(w, "egress: connection cannot be hijacked", http.StatusInternalServerError) diff --git a/sandboxd/egress/proxy_test.go b/sandboxd/egress/proxy_test.go index 6c71ef6f..62b91d86 100644 --- a/sandboxd/egress/proxy_test.go +++ b/sandboxd/egress/proxy_test.go @@ -8,6 +8,7 @@ import ( "net/http" "net/http/httptest" "net/url" + "sync/atomic" "testing" "time" ) @@ -254,6 +255,26 @@ func TestCloseEndsSplicedTunnel(t *testing.T) { } } +func TestCloseEndsTunnelWhoseUpstreamStaysSilent(t *testing.T) { + var held holdCounter + p := New("sb_1", "", Policy{Allow: []Rule{{Host: "quiet.internal"}}}, nil, nil, fixedDial(silentServer(t)), nil, &held) + front := httptest.NewServer(p) + defer front.Close() + + conn := dialConnect(t, front.Listener.Addr().String(), "quiet.internal:443") + defer func() { _ = conn.Close() }() + if status := readStatus(t, bufio.NewReader(conn)); status != "HTTP/1.1 200 Connection Established" { + t.Fatalf("CONNECT status = %q, want 200 Connection Established", status) + } + if got := held.Load(); got != 1 { + t.Fatalf("holds = %d while the tunnel is up, want 1", got) + } + + p.Close() + + waitHolds(t, &held, 0) +} + func TestConnectDeniedIsTyped(t *testing.T) { policy := Policy{Allow: []Rule{{Host: "echo.internal"}}} p := New("sb_1", "acme", policy, nil, nil, fixedDial("127.0.0.1:1"), nil, nil) @@ -410,6 +431,12 @@ func (f fakeSecrets) Header(name string) (string, string, bool) { return hv[0], hv[1], ok } +type holdCounter struct{ atomic.Int32 } + +func (h *holdCounter) Hold() { h.Add(1) } + +func (h *holdCounter) Unhold() { h.Add(-1) } + func fixedDial(target string) DialFunc { return func(ctx context.Context, network, _ string) (net.Conn, error) { var d net.Dialer @@ -474,6 +501,29 @@ func echoServer(t *testing.T) string { return ln.Addr().String() } +func silentServer(t *testing.T) string { + t.Helper() + ln, err := net.Listen("tcp", "127.0.0.1:0") + if err != nil { + t.Fatalf("listen silent: %v", err) + } + done := make(chan struct{}) + t.Cleanup(func() { + close(done) + _ = ln.Close() + }) + go func() { + conn, err := ln.Accept() + if err != nil { + return + } + _, _ = io.Copy(io.Discard, conn) + <-done + _ = conn.Close() + }() + return ln.Addr().String() +} + func recvEvent(t *testing.T, events <-chan Event) Event { t.Helper() select { @@ -484,3 +534,14 @@ func recvEvent(t *testing.T, events <-chan Event) Event { return Event{} } } + +func waitHolds(t *testing.T, held *holdCounter, want int32) { + t.Helper() + deadline := time.Now().Add(2 * time.Second) + for held.Load() != want { + if time.Now().After(deadline) { + t.Fatalf("holds = %d after proxy Close, want %d", held.Load(), want) + } + time.Sleep(10 * time.Millisecond) + } +} diff --git a/sandboxd/egress/socks.go b/sandboxd/egress/socks.go index 996098cb..3abd9e3c 100644 --- a/sandboxd/egress/socks.go +++ b/sandboxd/egress/socks.go @@ -76,6 +76,10 @@ func (p *Proxy) serveSocks(ctx context.Context, conn net.Conn) { return } defer func() { _ = upstream.Close() }() + if !p.track(upstream) { + return + } + defer p.untrack(upstream) if socksReply(conn, socksGranted) != nil { return } diff --git a/sandboxd/egress/socks_test.go b/sandboxd/egress/socks_test.go index 6b7c43fa..4ae3c24e 100644 --- a/sandboxd/egress/socks_test.go +++ b/sandboxd/egress/socks_test.go @@ -15,7 +15,7 @@ var socksGreeting = []byte{socksVersion, 1, socksNoAuth} func TestSocksAllowTunnels(t *testing.T) { echo := echoServer(t) events := make(chan Event, 4) - _, addr := socksProxy(t, Policy{Allow: []Rule{{Host: "echo.internal"}}}, nil, fixedDial(echo), events) + _, addr := socksProxy(t, Policy{Allow: []Rule{{Host: "echo.internal"}}}, nil, fixedDial(echo), events, nil) conn, reply := socksExchange(t, addr, socksGreeting, socksNameRequest("echo.internal", 443)) if reply[1] != socksGranted { @@ -54,7 +54,7 @@ func TestSocksDecisionMirrorsConnect(t *testing.T) { for _, tt := range tests { t.Run(tt.name, func(t *testing.T) { events := make(chan Event, 4) - p, socksAddr := socksProxy(t, tt.policy, tt.ca, fixedDial(echoServer(t)), events) + p, socksAddr := socksProxy(t, tt.policy, tt.ca, fixedDial(echoServer(t)), events, nil) front := httptest.NewServer(p) defer front.Close() @@ -95,7 +95,7 @@ func TestSocksRefusesWhatItCannotServe(t *testing.T) { for _, tt := range tests { t.Run(tt.name, func(t *testing.T) { events := make(chan Event, 4) - _, addr := socksProxy(t, Policy{Allow: []Rule{{Host: "*"}}}, nil, fixedDial("127.0.0.1:1"), events) + _, addr := socksProxy(t, Policy{Allow: []Rule{{Host: "*"}}}, nil, fixedDial("127.0.0.1:1"), events, nil) _, reply := socksExchange(t, addr, tt.greeting, tt.request) if string(reply[:2]) != string(tt.want) { t.Errorf("reply = %v, want prefix %v", reply, tt.want) @@ -111,7 +111,7 @@ func TestSocksRefusesWhatItCannotServe(t *testing.T) { func TestSocksIPv4TargetMatchesLiteralRule(t *testing.T) { events := make(chan Event, 4) - _, addr := socksProxy(t, Policy{Allow: []Rule{{Host: "127.0.0.1"}}}, nil, fixedDial(echoServer(t)), events) + _, addr := socksProxy(t, Policy{Allow: []Rule{{Host: "127.0.0.1"}}}, nil, fixedDial(echoServer(t)), events, nil) request := []byte{socksVersion, socksConnect, 0, socksAtypIPv4, 127, 0, 0, 1, 1, 187} _, reply := socksExchange(t, addr, socksGreeting, request) if reply[1] != socksGranted { @@ -123,7 +123,7 @@ func TestSocksIPv4TargetMatchesLiteralRule(t *testing.T) { } func TestCloseEndsSocksTunnelAndHandshake(t *testing.T) { - p, addr := socksProxy(t, Policy{Allow: []Rule{{Host: "echo.internal"}}}, nil, fixedDial(echoServer(t)), nil) + p, addr := socksProxy(t, Policy{Allow: []Rule{{Host: "echo.internal"}}}, nil, fixedDial(echoServer(t)), nil, nil) tunnel, reply := socksExchange(t, addr, socksGreeting, socksNameRequest("echo.internal", 443)) if reply[1] != socksGranted { t.Fatalf("reply code = %#x, want granted", reply[1]) @@ -142,9 +142,25 @@ func TestCloseEndsSocksTunnelAndHandshake(t *testing.T) { } } +func TestCloseEndsSocksTunnelWhoseUpstreamStaysSilent(t *testing.T) { + var held holdCounter + p, addr := socksProxy(t, Policy{Allow: []Rule{{Host: "quiet.internal"}}}, nil, fixedDial(silentServer(t)), nil, &held) + _, reply := socksExchange(t, addr, socksGreeting, socksNameRequest("quiet.internal", 443)) + if reply[1] != socksGranted { + t.Fatalf("reply code = %#x, want granted", reply[1]) + } + if got := held.Load(); got != 1 { + t.Fatalf("holds = %d while the tunnel is up, want 1", got) + } + + p.Close() + + waitHolds(t, &held, 0) +} + func TestSocksPortRuleGatesTheTunnel(t *testing.T) { events := make(chan Event, 4) - _, addr := socksProxy(t, Policy{Allow: []Rule{{Host: "mail.internal", Ports: []uint16{993}}}}, nil, fixedDial(echoServer(t)), events) + _, addr := socksProxy(t, Policy{Allow: []Rule{{Host: "mail.internal", Ports: []uint16{993}}}}, nil, fixedDial(echoServer(t)), events, nil) conn, reply := socksExchange(t, addr, socksGreeting, socksNameRequest("mail.internal", 993)) defer func() { _ = conn.Close() }() @@ -167,7 +183,7 @@ func TestSocksPortRuleGatesTheTunnel(t *testing.T) { func TestSocksPortZeroIsDeniedEvenByABareRule(t *testing.T) { events := make(chan Event, 4) - _, addr := socksProxy(t, Policy{Allow: []Rule{{Host: "mail.internal"}}}, nil, fixedDial(echoServer(t)), events) + _, addr := socksProxy(t, Policy{Allow: []Rule{{Host: "mail.internal"}}}, nil, fixedDial(echoServer(t)), events, nil) conn, reply := socksExchange(t, addr, socksGreeting, socksNameRequest("mail.internal", 0)) defer func() { _ = conn.Close() }() if reply[1] != socksDenied { @@ -178,13 +194,13 @@ func TestSocksPortZeroIsDeniedEvenByABareRule(t *testing.T) { } } -func socksProxy(t *testing.T, policy Policy, ca *CA, dial DialFunc, events chan Event) (*Proxy, string) { +func socksProxy(t *testing.T, policy Policy, ca *CA, dial DialFunc, events chan Event, holder Holder) (*Proxy, string) { t.Helper() var audit func(Event) if events != nil { audit = func(ev Event) { events <- ev } } - p := New("sb_1", "acme", policy, nil, ca, dial, audit, nil) + p := New("sb_1", "acme", policy, nil, ca, dial, audit, holder) ln, err := net.Listen("tcp", "127.0.0.1:0") if err != nil { t.Fatalf("listen socks: %v", err) From 3fa209d8948805e2f5863a609a5cbdff4168ecf1 Mon Sep 17 00:00:00 2001 From: CMGS Date: Mon, 14 Sep 2026 21:11:41 +0800 Subject: [PATCH 31/40] pool: drop the consumed snapshot when a release beats the wake to the finish A wake whose commit landed before a concurrent Release removed the claim returned without dropping the hibernate image it had just resumed from; Release saw an empty HibernateSnap and left it too, so the image stayed on disk until the next restart's Reconcile. No cheap test: the window sits between the commit and the release check with no seam to hold it open. --- sandboxd/pool/hibernate.go | 4 ++++ 1 file changed, 4 insertions(+) diff --git a/sandboxd/pool/hibernate.go b/sandboxd/pool/hibernate.go index c67bcabd..8480b278 100644 --- a/sandboxd/pool/hibernate.go +++ b/sandboxd/pool/hibernate.go @@ -164,6 +164,10 @@ func (m *Manager) wakeResolved(ctx context.Context, sb *types.Sandbox) (string, log.WithFunc("pool.wakeResolved").Errorf(ctx, proxyErr, "arm egress proxy %s", sb.ID) } if m.disarmIfReleased(sb) { + if err == nil { + m.dropStale(ctx, sb) + m.dropSnap(ctx, snap) + } return "", ErrUnknownSandbox } if err != nil { From e8bcac7a5b0297c088afe44f2787dfd0d351013e Mon Sep 17 00:00:00 2001 From: CMGS Date: Mon, 14 Sep 2026 21:11:43 +0800 Subject: [PATCH 32/40] pool: evict a template's record lock with the record DeleteTemplate released its lock with recDone, so the per-id mutex stayed in recLocks after the record was gone, and resolveGolden created one for every key it looked up and did not find. The checkpoint paths already evict on delete and gate the lookup before locking. --- sandboxd/pool/promote_test.go | 30 ++++++++++++++++++++++++++++++ sandboxd/pool/template.go | 10 ++++++---- 2 files changed, 36 insertions(+), 4 deletions(-) diff --git a/sandboxd/pool/promote_test.go b/sandboxd/pool/promote_test.go index edaacd5a..50b9d31d 100644 --- a/sandboxd/pool/promote_test.go +++ b/sandboxd/pool/promote_test.go @@ -169,6 +169,36 @@ func TestDeleteTemplate(t *testing.T) { } } +func TestTemplateRecordLockEvictsWithTheRecord(t *testing.T) { + m := newTestManager(t, newFakeEngine(), config.PoolSpec{PoolKey: testKey, Warm: 0}) + parent := mustClaim(t, m, testKey) + key := types.PoolKey{Template: "tpl:evict", Net: testKey.Net, Size: testKey.Size} + id := store.TemplateID(key.Hash()) + base := lockCount(m) + + if _, err := claimAny(t.Context(), m, key, 0); err != nil { + t.Fatalf("Claim of an unpromoted key: %v", err) + } + if hasRecLock(m, id) { + t.Error("recLocks kept an entry for a template that was never promoted") + } + if _, _, err := m.Promote(t.Context(), parent.ID, Cred{Token: parent.Token}, key.Template, ""); err != nil { + t.Fatalf("Promote: %v", err) + } + if !hasRecLock(m, id) { + t.Error("recLocks dropped the entry of a live template") + } + if err := m.DeleteTemplate(t.Context(), key, ""); err != nil { + t.Fatalf("DeleteTemplate: %v", err) + } + if hasRecLock(m, id) { + t.Error("recLocks retained a lock for the deleted template") + } + if got := lockCount(m); got != base { + t.Errorf("recLocks grew %d->%d over claim, promote and delete, want no net growth", base, got) + } +} + func TestReconcileSweepsGoldenTmpDirs(t *testing.T) { eng := newFakeEngine() m := newTestManager(t, eng) diff --git a/sandboxd/pool/template.go b/sandboxd/pool/template.go index d9f0ca6e..386a557f 100644 --- a/sandboxd/pool/template.go +++ b/sandboxd/pool/template.go @@ -79,7 +79,7 @@ func (m *Manager) DeleteTemplate(ctx context.Context, key types.PoolKey, tenant id := store.TemplateID(key.Hash()) l := m.recLock(id) l.Lock() - defer func() { l.Unlock(); m.recDone(id) }() + defer func() { l.Unlock(); m.recDoneEvict(id) }() raw, err := m.tpls.ReadMeta(ctx, id) if err != nil { if errors.Is(err, store.ErrNotFound) { @@ -261,12 +261,14 @@ func (m *Manager) resolveGolden(ctx context.Context, key types.PoolKey, tenant s l := m.recLock(id) l.RLock() dir, meta, digest, release, err := m.tpls.Fetch(ctx, id) + if errors.Is(err, store.ErrNotFound) { + l.RUnlock() + m.recDoneEvict(id) + return goldenResolution{release: func() {}}, nil + } if err != nil { l.RUnlock() m.recDone(id) - if errors.Is(err, store.ErrNotFound) { - return goldenResolution{release: func() {}}, nil - } return goldenResolution{release: func() {}}, err } cleanup := func() { release(); l.RUnlock(); m.recDone(id) } From 1f275a0a0f3fd3eb90b7800a5e3c0b54ae17de58 Mon Sep 17 00:00:00 2001 From: CMGS Date: Mon, 14 Sep 2026 21:18:37 +0800 Subject: [PATCH 33/40] docs: align the remaining pages with the code Egress: the audit record's field names, the SSRF guard's NAT64 handling, which lane a tenant policy may run on, the CONNECT rule wording, the archive wording and the internal-range file. Security and silkd pages name the SOCKS5 door and every not_found case. API: PUT /v1/pools 400 cases, the probe replay window, archive-backed checkpoints, usage-event tenant fields. Cluster gossip digest and epoch rule; the claims-journal write lock. SDK pages: which node verbs need the root token, the idle sweep's live-connection hold, spawn bad_request kinds, the checkpoint delete peer drop (and the Go godoc's stale TTL clause), the Python lookup, proxy_port and adapter state; langchain install and write contract; MCP defaults, per-call cap and child lifetimes. README: sh-lint target and the arm64 kernel; desktop, browser, benchmarks and image READMEs match their Dockerfiles and scripts. --- README.md | 3 ++- docs/benchmarks.md | 6 ++++-- docs/browser.md | 5 +++-- docs/cluster.md | 6 ++++-- docs/deploy.md | 9 +++++---- docs/desktop.md | 24 +++++++++++++++++------- docs/egress.md | 34 ++++++++++++++++++++-------------- docs/langchain.md | 5 +++-- docs/mcp.md | 13 ++++++++----- docs/openai-adapter.md | 7 ++++--- docs/performance.md | 5 +++-- docs/sandboxd-api.md | 20 ++++++++++++-------- docs/sdk-python.md | 13 +++++++++---- docs/sdk.md | 20 ++++++++++++++++---- docs/security.md | 4 +++- docs/silkd.md | 2 +- os-image/desktop/README.md | 16 ++++++++++++---- os-image/node/README.md | 2 +- protocol/README.md | 7 ++++--- sdk/go/checkpoint.go | 5 ++--- sdk/langchain/README.md | 4 ++-- sdk/openai/README.md | 3 ++- silkd/README.md | 5 ++++- 23 files changed, 141 insertions(+), 77 deletions(-) diff --git a/README.md b/README.md index 825d6ac6..b6004ea0 100644 --- a/README.md +++ b/README.md @@ -81,6 +81,7 @@ make help # this list make lint test # Rust: boot/init + silkd (fmt --check, clippy -D warnings, tests) make go-lint # Go: protocol/wire + sandboxd + sdk/go + e2e + mcp, GOOS linux AND darwin make go-test # Go: go test -race across the Go modules +make sh-lint # shellcheck every tracked shell script make sandboxd # build dist/sandboxd make boot # kernel + initramfs artifact image (docker) # KERNEL_MIRROR=… if kernel.org tarball paths 404 locally @@ -135,7 +136,7 @@ On a fresh repo run build-boot first — images build `FROM` the boot artifact. ``` cloud-hypervisor - → vmlinux (PVH ELF, everything =y, no decompress stage) + → guest kernel (amd64: PVH ELF vmlinux, everything =y, no decompress stage; arm64: flat Image) → uncompressed ~1.5MB cpio: /init = sandbox-init (static Rust) → resolve virtio-blk serials via sysfs (2ms poll, no udev) → mount EROFS layers → overlayfs + ext4 COW → switch_root diff --git a/docs/benchmarks.md b/docs/benchmarks.md index a6680f62..d4710cf9 100644 --- a/docs/benchmarks.md +++ b/docs/benchmarks.md @@ -21,7 +21,7 @@ the stop line. | **warm pool hit** | ownership transfer of a pre-booted, probed VM; no VM lifecycle work on the request path | | **clone from golden** | restore a full VM (memory + disk) from a golden snapshot, reseed entropy/machine identity, re-probe readiness | | **cold boot** | boot from the template image: kernel + initramfs + rootfs assembly + init to a probed silkd | -| **burst** | `BURST_N` clone-tier claims issued concurrently — per-claim latency under restore contention plus the batch wall clock. Runs last so its churn cannot contaminate the RTT and throughput windows | +| **burst** | `BURST_N` clone-tier claims issued concurrently — per-claim latency under restore contention plus the batch wall clock. Runs after the RTT and throughput windows so its churn cannot contaminate them | The harness also reports **warm refill recovery**: after fully draining the warm pool it times the refill loop rebuilding to target — the number bounded @@ -72,7 +72,9 @@ table stamped with the host evidence: virtualization Knobs (environment variables): `WARM`/`WARM_N` (warm-pool depth and burst size — the burst must stay within the depth, or refill loses the race and the tail measures clones), `CLONE_N`, `COLD_N`, `RPC_N`, `PULL_MB`, -`PULL_N`, `BURST_N` (concurrent clone-claim burst; 0 skips the stage). +`PULL_N`, `BURST_N` (concurrent clone-claim burst; 0 skips the stage), +`ADDR`/`TOKEN` (the throwaway daemon's listen address and token; change them +to run beside a live sandboxd). Boot anatomy (where inside the cold tier the milliseconds go — kernel, initramfs phases, rootfs handoff) has its own harness: diff --git a/docs/browser.md b/docs/browser.md index 21650675..579d381c 100644 --- a/docs/browser.md +++ b/docs/browser.md @@ -66,8 +66,9 @@ branch from. `sandbox-id:port`, which Chrome's DevTools allowlist rejects. Use `ProxyPort`/`DialPort` for CDP; preview URLs serve human-facing HTTP the workload chooses to expose (a live-view page, a screenshot server). -- Guest env knobs on the unit: `CDP_PORT` (default 9222), - `CHROMIUM_FLAGS` (extra flags). +- Guest env knobs read by `/usr/local/bin/chromium-cdp`: `CDP_PORT` (default + 9222) and `CHROMIUM_FLAGS` (extra flags); set them with a `chromium.service` + drop-in, the unit itself declares no environment. - No stealth build: headless Chromium is fingerprintable; this flavor targets automation, not anti-bot evasion. - One browser per sandbox by design — the VM is the isolation and diff --git a/docs/cluster.md b/docs/cluster.md index dceea3f4..2b935e7e 100644 --- a/docs/cluster.md +++ b/docs/cluster.md @@ -3,7 +3,8 @@ A cluster is a set of sandboxd nodes joined through a [hashicorp/memberlist](https://github.com/hashicorp/memberlist) SWIM mesh. Gossip carries only placement hints — per-pool warm counts, promoted-template -hashes, available volume names, and each node's data-plane address. +hashes, available volume names, each node's data-plane address, and a digest +of the cluster-invariant config. Per-sandbox state never leaves its owning node, so a stale view costs at most one extra redirect, never correctness. A single node with no seeds is a valid mesh of one. @@ -65,7 +66,8 @@ membership is node-local and deliberately excluded from the cluster config digest. Nodes gossip only their currently available catalog names: host paths and access lists never leave the node. After config load the set appears on the next gossip tick; later image distribution or removal is detected the same way. -The node epoch bumps only when the advertised name set changes. +The node epoch bumps only when one of the gossiped sets — warm counts, +template hashes, volume names — actually changes. A writable name (`writable: true`) still needs its catalog entry — name, access list, and the `writable` flag — declared identically on every node diff --git a/docs/deploy.md b/docs/deploy.md index d722587d..b0231368 100644 --- a/docs/deploy.md +++ b/docs/deploy.md @@ -118,7 +118,7 @@ sandboxd reads one JSON file (`-config`, default | `checkpoint_ttl_hours` | 0 (keep forever) | ages out checkpoints older than this; the sweep runs hourly and at startup. Explicit deletes never wait for it. Must be nonzero and match fleet-wide when `checkpoint_peer_heal` is on — it is the expiry eligibility point for a healed replica a delete broadcast missed, after which its next successful hourly sweep removes it; persistent sweep failure extends retention until one succeeds, so it is not a hard ceiling | | `checkpoint_peer_heal` | false | on a cluster, lets a node pull a checkpoint it lacks from a peer — found via a live probe, not gossip — rather than failing the branch; see [placement lifecycle](cluster.md#checkpoints-on-a-cluster). Three requirements, all enforced at config load: a nonempty `api_token` (the blob transfer between peers authenticates with it; without one the raw record stream would be open), `mesh.cluster_key` set (the pull presents the fleet `api_token` to an address learned from the peer probe, so the gossip layer carrying that address must itself be authenticated), and `checkpoint_ttl_hours` nonzero (a replica a delete broadcast missed becomes eligible for expiry after it, and its next successful hourly sweep removes it — so it is the finite eligibility point, not an exact ceiling). A shared checkpoint store (`checkpoint_store` kind `s3`) ignores this setting — every node already resolves every checkpoint directly, so there is nothing to heal | | `warm_max` (pool entry) | 0 (static) | turns on the demand-adaptive watermark for that pool: the warm target rises from `warm` toward `warm_max` while claims arrive faster than the measured provision lead covers, and decays back over ~a minute of silence | -| `warmup` (pool entry) | unset | argv run in the golden VM after readiness and before its snapshot, so the files it touches are page-cache-resident in every clone, and again in every clone before it joins the warm pool, so those pages are already faulted into the restored VM when the first command runs — e.g. `["node", "-e", "0"]` on a Node flavor. It runs under the engine's 2-minute command timeout in silkd's base environment (`PATH`, `TERM`, and the guest image's proxy variables wherever nothing routes directly — the none lane and the locked bridge egress lane — with the relay not yet armed, since arming happens at claim); a non-zero exit or a timeout fails the golden build, so the pool stays unfilled until the config is fixed. Config-owned like `egress`: `PUT /v1/pools` rejects it, and a golden built with a different warmup is rebuilt | +| `warmup` (pool entry) | unset | argv run in the golden VM after readiness and before its snapshot, so the files it touches are page-cache-resident in every clone, and again in every clone before it joins the warm pool, so those pages are already faulted into the restored VM when the first command runs — e.g. `["node", "-e", "0"]` on a Node flavor. It runs under the engine's 2-minute command timeout in silkd's base environment (`PATH`, `TERM`, and the guest image's proxy variables wherever nothing routes directly — the none lane and the locked bridge egress lane — with the proxy not yet serving, since a door pre-bound at refill only starts serving at claim); a non-zero exit or a timeout fails the golden build, so the pool stays unfilled until the config is fixed. Config-owned like `egress`: `PUT /v1/pools` rejects it, and a golden built with a different warmup is rebuilt | | `max_claims` | 0 (unlimited) | node-wide cap on live claims; claim/fork/branch requests beyond it answer 429 with the pool state unharmed (on a cluster, normal warm-candidate placement applies, with volume claims limited to candidates holding every requested volume) | | `audit_log` | false | append every relayed request frame's op + addressing fields (never payloads) to `/audit.jsonl`, size-rotated with one `.1` backup. Records are `{t, id, op}` plus whichever addressing fields the op carries (`argv`, `path`, `dest`, `from`, `to`, `url`, `session`, `port`), plus `method` (`GET`, `CONNECT`, `SOCKS5`, …), `decision` and `secret` (the ref name, never its value) on `egress` records; preview accesses record as op `preview`, one per request. A request frame whose first line exceeds 4 KiB records as op `oversized` with no addressing fields | | `idle_hibernate_seconds` | 0 (off) | node-wide idle policy for unpooled claims (template/checkpoint claims): a none-lane claim is hibernated once it has had no open data-plane connection (relay, buffered exec, preview) and no egress request in flight for this long; the clock restarts when the last connection closes, so a long command is never cut short. The next call that reaches the guest wakes it transparently. Per-pool `idle_hibernate_seconds` does the same for that pool's claims; pooled keys ignore the node-wide value, and egress pools reject it because they cannot resume safely. Opt in deliberately: a wake costs latency and the snapshot, so callers with their own idle logic must not pay twice | @@ -312,9 +312,10 @@ here validates on load: "pools": [ {"template": "rt:24.04", "net": "none", "size": "small", "warm": 4, "warm_max": 12}, {"template": "rt:24.04", "net": "egress", "size": "medium", "warm": 2, - "egress": {"allow": [ + "egress": {"socks5": true, "allow": [ {"host": "api.github.com", "methods": ["GET", "POST"], "secret": "gh", "intercept": true}, - {"host": "*.googleapis.com"} + {"host": "*.googleapis.com"}, + {"host": "imap.example.com", "ports": [993]} ]}} ] } @@ -463,7 +464,7 @@ exclusion, and a clean read-only claim afterward. HTTP port under a signed, expiring shareable URL. The whole mechanism is in sandboxd: -- **Minting** (`sb.PreviewURL(port, ttl)`): the owner node signs a token +- **Minting** (`sb.PreviewURL(ctx, port, ttl)`): the owner node signs a token embedding `{sandbox, port, owner, exp}` with `preview_secret`; the URL's life is clamped to the claim's lease. - **Serving**: any node's preview listener verifies the token (no shared diff --git a/docs/desktop.md b/docs/desktop.md index a1fc2838..f224aea7 100644 --- a/docs/desktop.md +++ b/docs/desktop.md @@ -3,7 +3,9 @@ The `desktop` flavor boots a GNOME session (Ubuntu session on Xvfb, 1920x1080) with the OSWorld guest server on guest loopback `5000` and the OSWorld app set (Google Chrome, LibreOffice, GIMP, VLC, Thunderbird, VS Code, Zotero, -Obsidian, Shotcut, FreeCAD, WPS Office, MuseScore). A computer-use agent or the +Obsidian, Shotcut, FreeCAD, WPS Office, MuseScore, REAPER and the CAD, EDA +and teaching tools the task set drives — the full list is in +[`os-image/desktop/README.md`](../os-image/desktop/README.md)). A computer-use agent or the [OSWorld](https://github.com/xlang-ai/OSWorld-V2) harness claims it and drives the desktop through the same HTTP contract the OSWorld AWS and docker guests speak — screenshot, AT-SPI accessibility tree, PyAutoGUI @@ -25,12 +27,14 @@ up a few seconds later — poll `GET /screenshot` until it returns 200. with an [egress policy](egress.md) when tasks visit the OSWorld mocked websites or the real web; the session's browsers reach the web through the relay. Thunderbird's mail policy is pinned to the SOCKS5 door, so a pool that - runs mail tasks sets `"socks5": true` in its policy. `net=egress` on a + runs mail tasks sets `"socks5": true` in its policy alongside at least one + allow rule that admits CONNECT (a bare-host rule, or one whose `methods` + names it); the policy is rejected at load otherwise. `net=egress` on a guarded bridge works the same way: the session's proxy setup waits for the host's lane verdict in `/etc/silkd-lane` and follows it, so the locked NIC costs the desktop nothing (a CNI-backed lane is told `direct` and routes). -- **Size**: `2xlarge` (8 CPU / 16G) — the t3.xlarge class the OSWorld AWS - image runs on; the idle session is ~0.5 GB anonymous memory with +- **Size**: `2xlarge` (8 CPU / 16G) — the memory the OSWorld AWS image's + workload is sized for; the idle session is ~0.5 GB anonymous memory with gnome-shell around 290 MB RSS, and the headroom is for the apps. - **Template**: `ghcr.io/cocoonstack/sandbox/desktop:24.04` — `base:24.04` plus GNOME on Xvfb, `osworld-server` at a pinned commit, Chrome from @@ -51,9 +55,10 @@ Both bind guest loopback; reach them with `DialPort`/`ProxyPort`. OSWorld's `DesktopEnv` drives VMs through a `Provider` whose `get_ip_address` may return `localhost::::` with per-environment ports — the shape its docker provider uses. A cocoon -provider claims one sandbox per environment, serves the four guest ports on -loopback listeners with `proxy_port`, and implements `revert_to_snapshot` as -release + fresh claim, so a warm pool is the snapshot revert: +provider (kept with the harness, not in this repository) claims one sandbox +per environment, serves the four guest ports on loopback listeners with +`proxy_port`, and implements `revert_to_snapshot` as release + fresh claim, +so a warm pool is the snapshot revert: ``` DesktopEnv(provider_name="cocoon") ── localhost: ── sandboxd ── desktop VM :5000 @@ -89,3 +94,8 @@ Configure it with `SANDBOXD_ADDR`, `SANDBOXD_TOKEN`, `COCOON_TEMPLATE` to a bridge after the claim) reaches silkd's execs at once but the session only through a re-claim. - x86_64 only. + +The hardware acceptance is `e2e/cmd/desktopsmoke`: claim → `/screenshot` and +the AT-SPI tree over the relay → a PyAutoGUI click echoed by +`/cursor_position` → a checkpoint/branch of the warmed desktop on the none +lane, a fetch through the session's proxy environment on the egress lane. diff --git a/docs/egress.md b/docs/egress.md index a04b0083..d6fa8bf1 100644 --- a/docs/egress.md +++ b/docs/egress.md @@ -85,16 +85,16 @@ and the image's units wait for the same file instead of guessing from NIC presence; git keeps running as on any lane with a NIC. A golden built under another verdict is rebuilt, not adopted. -Egress-lane sandboxes do not hibernate, archive, fork, checkpoint, or promote: -cocoon resumes a guest before its fresh tap can be re-locked, so any resume from -a snapshot would open an unlocked-NIC window. Keeping the lane live holds the -lock unbroken from claim to release; those operations are refused (409) on the -lane. +Egress-lane sandboxes never archive and do not hibernate, fork, checkpoint, or +promote: cocoon resumes a guest before its fresh tap can be re-locked, so any +resume from a snapshot would open an unlocked-NIC window. Keeping the lane live +holds the lock unbroken from claim to release; the idle sweep skips the lane and +those four verbs are refused (409) on it. The proxy also refuses to connect to internal addresses — every IANA special-purpose range that is not globally reachable (loopback, link-local incl. cloud metadata, private, carrier-grade NAT, benchmarking, documentation, -reserved, per the registry snapshot in `egress.go`) plus the IPv4-embedding +reserved, per the registry snapshot in `sandboxd/pool/egress.go`) plus the IPv4-embedding IPv6 forms (NAT64, 6to4, Teredo, IPv4-compatible) — so an allow-listed host that resolves, or is rebound, to one cannot reach the sandboxd host or a sibling VM. @@ -107,8 +107,10 @@ whose sandboxes legitimately need internal services: ``` It is node-wide (every pool and tenant on the node gets the same re-admission) -and checked after NAT64 unwrapping, so an embedded IPv4 matches as the IPv4 it -is. Name service prefixes, never the whole private space: the guest bridges are +and checked after the well-known `64:ff9b::/96` unwrap, so an IPv4 embedded in +that prefix matches as the IPv4 it is; an address in the local-use +`64:ff9b:1::/48` is blocked as a whole and needs that prefix itself in the list. +Name service prefixes, never the whole private space: the guest bridges are themselves ULA/RFC1918, so a blanket permit would open sandbox-to-sandbox and the host's own gateway. `0.0.0.0/0` + `::/0` turns the guard off entirely — for fleets with a policy-enforcing proxy in front — and requests still pass the @@ -125,7 +127,9 @@ domain policy first; the allow-list widens the IP gate only. - **Bridge lane only (egress lane).** A CNI network's tap lives in the VM netns, out of reach of the root-netns lock, so a guarded egress *lane* needs the `bridges` form (those taps stay in the root netns) and is rejected on CNI - `networks`. None-lane policies ride the proxy and work on either. A bridge + `networks`. A none-lane *pool* policy rides the proxy and works on either; a + tenant policy does not, because it could land on an egress-lane claim, so + `networks` plus any tenant `egress` block is rejected at load. A bridge egress lane locks every NIC default-deny, even with no policy configured. - **No custom NAT64/DNS64 prefix routed to the host.** The SSRF guard folds the standard NAT64 forms (RFC 6052 well-known `64:ff9b::/96`, RFC 8215 local-use @@ -182,8 +186,8 @@ woken sandbox binds at arm time. - `host`: an exact name, a `*.`-prefixed suffix wildcard, or `*`. Case-insensitive. - `methods`: empty means any. Enforced on plaintext and on intercepted HTTPS. A non-intercepted CONNECT tunnel is opaque — the method cannot be checked — so a - methods-restricted rule without `intercept` denies CONNECT outright rather - than tunneling unchecked. + rule whose `methods` does not name `CONNECT`, and does not `intercept`, denies + the tunnel outright rather than tunneling unchecked. - `ports`: empty means any; otherwise the destination port must be listed. The port is the tunnel's target on CONNECT and SOCKS5, the URL's port (or the scheme's default) on the forward path, and the CONNECT's port for every @@ -195,8 +199,9 @@ woken sandbox binds at arm time. method and the secret injected (see below). Only a pool rule may set it. - No policy on a claim ⇒ no egress at all (the proxy is not started). -Each decision is written to `audit.jsonl` (`op:"egress"`, host, port, -allow/deny, the secret **name**) and metered as an `egress` usage event. +Each decision is written to `audit.jsonl` (`op:"egress"`, `dest`, `port`, +`decision`, and the secret **name** in `secret`) and metered as an `egress` +usage event. ## HTTPS interception @@ -273,7 +278,8 @@ every other node, is unaffected. `egress_ca` is required whenever a pool has an intercept rule. The root cert is baked into a guest **when the guest is created** — at golden build, or at a -pre-golden cold claim's provision (both via silkd, off the claim path). It is +pre-golden cold claim's provision (both via silkd; only the golden-build +install is off the claim path). It is **not** re-installed on re-claim: a clone, checkpoint restore, archive wake, or reconcile adopts the guest with whatever root it was born with. A `.cafp` sidecar ties golden adoption to the baked bytes, so a changed root rebuilds diff --git a/docs/langchain.md b/docs/langchain.md index 06643ab1..0a849414 100644 --- a/docs/langchain.md +++ b/docs/langchain.md @@ -1,7 +1,8 @@ # LangChain toolkit `cocoonstack-sandbox-langchain` turns one sandbox into a LangChain tool -set: `pip install cocoonstack-sandbox-langchain`. +set: `pip install cocoonstack-sandbox-langchain` (the example below also +needs `langgraph`, which the package does not pull in). ```python from cocoonsandbox_langchain import CocoonToolkit @@ -20,7 +21,7 @@ schemas, sync-native with `asyncio.to_thread` async bridges): | tool | what it does | |---|---| | `sandbox_exec` | run a shell command, cut off after 5 minutes with the reply saying so; stdout/stderr/exit code; disk state persists across calls | -| `sandbox_write_file` | write a text file (atomic on the guest) | +| `sandbox_write_file` | write a text file (atomic on the guest; the parent directory must exist) | | `sandbox_read_file` | read a text file | | `sandbox_list_dir` | list a directory as JSON | diff --git a/docs/mcp.md b/docs/mcp.md index 18cc7431..fbaa39b2 100644 --- a/docs/mcp.md +++ b/docs/mcp.md @@ -17,23 +17,26 @@ real microVM sandboxes with no glue code. } ``` -Flags fall back to `SANDBOXD_ADDR`, `SANDBOXD_TOKEN`, `SANDBOXD_TEMPLATE`. -Build: `cd mcp && go build -o sandbox-mcp .` +Flags fall back to `SANDBOXD_ADDR`, `SANDBOXD_TOKEN`, `SANDBOXD_TEMPLATE`, +then to `127.0.0.1:7777` and `rt:24.04`. Build: `cd mcp && go build -o +sandbox-mcp .` ## Tools +Every tool call is capped at 5 minutes. + | tool | what it does | |---|---| | `create_sandbox` | claim a microVM and return its id plus deadline; optional `template`, `net` (`none` default, or `egress`), `size` (`small` default, `medium`, `large`, `xlarge`, `2xlarge`) and `ttl_seconds` (0 means one hour); warm claims take milliseconds, nothing renews the deadline | -| `exec` | run a shell command to completion (5-minute cap); returns stdout, stderr and the exit code, each stream keeping its first 1 MiB with `truncated` set past that; a cut-off or dropped run returns the output collected so far next to an `error` field; a hibernated sandbox wakes transparently | +| `exec` | run a shell command to completion; returns stdout, stderr and the exit code, each stream keeping its first 1 MiB with `truncated` set past that; a cut-off or dropped run returns the output collected so far next to an `error` field; a hibernated sandbox wakes transparently | | `spawn` | start a command detached and return its pid; output goes to a 256 KiB ring buffer that `logs` replays | | `ps` | list tracked processes (exec, spawn, pty) with state, exit code and start time | | `logs` | replay up to 256 KiB of a tracked process's newest whole stdout/stderr chunks (+ exit code once ended) | | `kill` | signal a tracked process (0 = SIGKILL); an exited process is a no-op success | | `write_file` / `read_file` / `list_dir` | atomic whole-file write (parent must exist); whole-file text read of a regular file up to 1 MiB (a larger file, a directory or a device is an error, invalid UTF-8 replaced, missing path is an error); one-level listing of `{name, kind, size}` entries | -| `fork` | clone into N children (1 to the node's `max_fork_count`, default 16) carrying exact memory + disk state, all-or-nothing; the parent keeps running | +| `fork` | clone into N children (1 to the node's `max_fork_count`, default 16) carrying exact memory + disk state, all-or-nothing; the parent keeps running and each child lives one hour | | `checkpoint` | capture full state without stopping; returns a `checkpoint_id` that can be branched repeatedly | -| `branch_checkpoint` | claim a fresh sandbox from a checkpoint's captured moment | +| `branch_checkpoint` | claim a fresh sandbox from a checkpoint's captured moment; it lives one hour | | `list_checkpoints` / `delete_checkpoint` | checkpoint lifecycle | | `hibernate` | snapshot + stop, freeing memory while keeping id, files and processes; the next call that reaches the guest wakes it | | `promote` | publish the sandbox as a named template on its node; re-promoting replaces it | diff --git a/docs/openai-adapter.md b/docs/openai-adapter.md index 47466537..0c99c610 100644 --- a/docs/openai-adapter.md +++ b/docs/openai-adapter.md @@ -33,14 +33,15 @@ pair over the sync [Python SDK](sdk-python.md), bridged with |---|---| | `create` | `Client.new` — one claimed sandbox per session | | session `exec` | `Sandbox.run` (stdout/stderr/exit); a per-call `timeout` is `run`'s wall clock and surfaces as `TimeoutError` | -| `read` / `write` | `Sandbox.read_file` / `write_file`; missing → `FileNotFoundError` | +| `read` / `write` | `Sandbox.read_file` / `write_file`; a missing path on `read` → `FileNotFoundError` | | `persist_workspace` / `hydrate_workspace` | `Sandbox.pull` / `push` (tar) | | exposed port | `Sandbox.proxy_port` | | `delete` | `Sandbox.close` (release) | -| `resume` | reattach by id + token from the serialized session state | +| `resume` | reattach from the serialized session state: node address, api token, sandbox id, sandbox token, owner | `CocoonSandboxClientOptions` carries the node address, api token, template -ref, network lane (`none`/`egress`), and TTL. The session state is +ref, network lane (empty for the node's default, or `none`/`egress`), and +TTL. The session state is JSON-serializable, so a run can be resumed against the same sandbox after a process restart. Requires Python 3.10+ (the Agents SDK floor); the underlying `cocoonsandbox` stays 3.9+. diff --git a/docs/performance.md b/docs/performance.md index 1e92f532..31a1fe7c 100644 --- a/docs/performance.md +++ b/docs/performance.md @@ -119,8 +119,9 @@ snapshot paths. kept off the manager mutex — the lock every data-plane op contends. `set`/`del` update a projection map and bump a sequence under the store's own mutex; `commit()` clones that map under it, then marshals, writes, and renames off -every mutex, serialized and coalescing by sequence so an older snapshot never -overwrites a newer one. Only the startup `Reconcile` +both the manager and the store mutex, serialized by its own write lock and +coalescing by sequence so an older snapshot never overwrites a newer one. +Only the startup `Reconcile` pass (pre-contention) still marshals and writes in one call under the lock. `BenchmarkStorePersistContention` measures the ns a concurrent manager-mutex diff --git a/docs/sandboxd-api.md b/docs/sandboxd-api.md index 60c40156..30e03c0c 100644 --- a/docs/sandboxd-api.md +++ b/docs/sandboxd-api.md @@ -327,9 +327,10 @@ targets online — no restart, live claims untouched: Pools omitted from the list are drained: their unclaimed warm VMs are destroyed and the pool entry retires. `net`/`size` default like a claim's. Answers the fresh `GET /v1/info` payload. 400 bad key, negative warm/idle, -`warm_max` below `warm`, duplicate pool, or a config-owned `egress`/`warmup` -field; 401 bad api token; 409 egress -pool on a node without an egress attachment. +`warm_max` below `warm`, `idle_hibernate_seconds` on an egress pool, an +`archive_after_seconds` not above the pool's `idle_hibernate_seconds`, +duplicate pool, or a config-owned `egress`/`warmup` field; 401 bad api +token; 409 egress pool on a node without an egress attachment. ## POST /v1/drain @@ -418,7 +419,7 @@ part of the public API; an SDK caller has no reason to call it directly. `X-Cocoon-Probe` is absent or expired. On a mesh with `cluster_key` set the request must carry `X-Cocoon-Probe`, an HMAC over the id and a coarse time bucket keyed off a probe-specific derivation of the cluster key — verified - before any disk is touched, replayable for roughly a minute at most. On a + before any disk is touched, replayable for roughly ninety seconds at most. On a keyless mesh (redirect-only fleets have no shared secret to sign with) the id itself remains the only capability, matching the mesh's own posture. Repeat probes for one id are answered from a short positive cache on the @@ -427,13 +428,15 @@ part of the public API; an SDK caller has no reason to call it directly. ## GET /v1/checkpoints Auth: node API token. Lists this node's checkpoints, newest first. A tenant -sees only its own records; root sees everything. +sees only its own records; root sees everything. A checkpoint backing an +archived claim is that claim's wake image, not a listable record. ## DELETE /v1/checkpoints/{id} Auth: node API token. A tenant may delete only its own records — anything else is 404, never a hint the id exists; root deletes anything. 204 on -success, 404 unknown. +success, 404 unknown — including the checkpoint behind an archived claim, +which only that claim's release or retention window removes. **Delete removes the local record, then best-effort broadcasts to peers so a healed replica does not outlive it — eventual cleanup, not a fleet-wide @@ -495,8 +498,9 @@ truth is the usage journal below. Always on: every lifecycle transition appends one JSONL event to `/usage.jsonl` — `{"t": , "ev": "claim|hibernate|wake|fork|checkpoint|promote|release|reap|archive|unarchive|archive_delete|egress", -"id": "sb_…", "vm": "sbx-…"}` plus `key` and `tenant` (the pool key's stable -hash and the owning tenant, claim events), `children` (fork) and `ref` (the promoted +"id": "sb_…", "vm": "sbx-…"}` plus `key` (the pool key's stable hash, claim +events), `tenant` (the owning tenant, on claim, egress, archive, unarchive and +archive_delete), `children` (fork) and `ref` (the promoted template / checkpoint id, or the egress host). A volume claim also carries `volumes`, the applied catalog names, and — omitted when empty — `volumes_rw`, the subset of those names claimed `rw`, so billing can discriminate write diff --git a/docs/sdk-python.md b/docs/sdk-python.md index bcd3a3a9..e8e60ec5 100644 --- a/docs/sdk-python.md +++ b/docs/sdk-python.md @@ -82,7 +82,7 @@ directly. To recover a handle when only `id` + `token` survived (say, across a process restart): ```python -sb = client.lookup(id, token) # asks the entry node, then each mesh peer +sb = client.lookup(id, token) # probes the entry node and every mesh peer concurrently ``` ## Claiming @@ -166,7 +166,9 @@ running — a hibernated sandbox is still reaped at its deadline, so claim with a `ttl_seconds` that covers the idle period. When to hibernate is your policy, unless the deployment opts into `idle_hibernate_seconds` ([deploy](deploy.md#configuration)), which hibernates idle claims -automatically with the same transparent wake. +automatically with the same transparent wake. A claim with a live connection +(a relay stream, a buffered exec, a preview dial, an egress request) is never +swept mid-call; the idle clock restarts when that connection ends. If that deployment also enables `archive_after_seconds`, archiving replaces the original claim deadline with the archive-retention deadline (or no @@ -277,7 +279,8 @@ context-manager) relayed over the silkd protocol — it works on the no-network lane, where the vsock relay is the only way in. A dead port raises silkd's `not_found`. `proxy_port` serves the port on a local listener for unmodified local tools (browsers, curl); close the returned -socket to stop. `preview_url` mints a signed URL served by the node's +socket to stop, and a guest port that closes ends the local connection too. +`preview_url` mints a signed URL served by the node's preview listener, clamped to the claim's remaining lease — the URL dies with the sandbox, and a node without `preview_listen` answers 501. @@ -454,7 +457,9 @@ zero. - `APIError(verb, status, message)` — control plane (HTTP status) - `SilkdError(kind, message)` — typed guest failure; `kind` is - `bad_request` / `not_found` / `unimplemented` / `internal` + `bad_request` (including a spawn the guest cannot start: missing binary, a + cwd that is not a directory, no exec bit) / `not_found` / `unimplemented` / + `internal` - `ExitError(code, stderr, stdout)` — non-zero exit from `exec` - `ProtocolError` — broken stream (EOF, oversized or undecodable frame) diff --git a/docs/sdk.md b/docs/sdk.md index c99c91fa..87a41578 100644 --- a/docs/sdk.md +++ b/docs/sdk.md @@ -216,7 +216,9 @@ claim with a `WithTimeout` that covers the idle period. When to hibernate is your policy; the node only provides the transition — unless the deployment opts into `idle_hibernate_seconds` (see [deploy](deploy.md#configuration)), which hibernates idle claims -automatically with the same transparent wake. +automatically with the same transparent wake. A claim with a live connection +(a relay stream, a buffered exec, a preview dial, an egress request) is never +swept mid-call; the idle clock restarts when that connection ends. If that deployment also enables `archive_after_seconds`, archiving replaces the original claim deadline with the archive-retention deadline (or no @@ -297,6 +299,13 @@ heal a missing record locally; `Checkpoint.Delete` acts on the handle's bound node. Checkpoint creation is resource-creating and takes the api token, like fork. +`Checkpoint.Delete` also asks every peer that node currently sees to drop any +replica a heal pulled — best-effort eventual cleanup, not a fleet-wide +revocation. A peer that misses the broadcast (offline, partitioned, or joined +later) keeps serving branches from its replica until the node's +`checkpoint_ttl_hours` ages it out; enabling peer heal requires that TTL to be +set, so every healed replica has a cleanup bound. + ## Language servers (LSP) ```go @@ -508,7 +517,8 @@ ctx) tears the shell down. ## Node operations -Root-token verbs for operating a node, plus the reference the aggregated +Verbs for operating a node — `Drain`, `Uncordon`, `SetPools` and +`SetPoolsCluster` need the root token — plus the reference the aggregated apiserver claims under: ```go @@ -528,8 +538,10 @@ claims. `Drain` leaves live claims alone — poll `Info` until `Claimed` is zero - `*sandbox.ExitError` — non-zero exit from `Exec` (`Code`, `Stderr`) - `*wire.ErrorResp` — a typed guest-side failure; `Kind` is one of - `wire.KindBadRequest`, `KindNotFound`, `KindUnimplemented`, - `KindInternal` (import `github.com/cocoonstack/sandbox/protocol/wire`) + `wire.KindBadRequest` (including a spawn the guest cannot start: missing + binary, a cwd that is not a directory, no exec bit), `KindNotFound`, + `KindUnimplemented`, `KindInternal` (import + `github.com/cocoonstack/sandbox/protocol/wire`) ```go var e *wire.ErrorResp diff --git a/docs/security.md b/docs/security.md index 2b6576bb..b9ecd315 100644 --- a/docs/security.md +++ b/docs/security.md @@ -24,7 +24,9 @@ exposing any part of a deployment beyond a single trusted host. guest can lie to its own client about its own state, but gains nothing toward the host, siblings, or the network beyond its lanes. - **What a compromised guest can reach.** On the none lane: nothing but - vsock — the relay back to its own client and the guarded-egress proxy. + vsock — the relay back to its own client and the guarded-egress proxy, + plus, on a policy that opts into `socks5`, a SOCKS5 door to the same + allow-list that carries arbitrary TCP rather than HTTP only. On the egress lane: the same vsock paths plus a NIC whose every guest-initiated packet except IPv4 broadcast DHCP is dropped by an nftables lock in the host root netns ([egress](egress.md)); the lock is fail-closed and applied diff --git a/docs/silkd.md b/docs/silkd.md index da83385c..f47f58e7 100644 --- a/docs/silkd.md +++ b/docs/silkd.md @@ -51,7 +51,7 @@ A failed verb terminates with `error {kind, message}`: | kind | meaning | |---|---| | `bad_request` | malformed frame, empty argv, invalid pattern, an exec argv or cwd the guest cannot spawn (missing, not a directory, not executable) | -| `not_found` | unknown pid/session/path | +| `not_found` | unknown pid/session/path, a `port_forward` port nothing listens on, a language with no LSP manifest | | `unimplemented` | verb unavailable on this lane — notably git clone/push/pull on the no-network lane, whose message points at `fs.push` | | `internal` | everything else (other spawn failures, git errors, I/O) | diff --git a/os-image/desktop/README.md b/os-image/desktop/README.md index d7c35c69..9755dd00 100644 --- a/os-image/desktop/README.md +++ b/os-image/desktop/README.md @@ -6,8 +6,9 @@ plus a GNOME session (Ubuntu session, dock, Yaru) on an Xvfb `:1` display at ([xlang-ai/osworld-server](https://github.com/xlang-ai/osworld-server), pinned commit) on guest loopback `5000`, the OSWorld Chrome CDP bridge on loopback `9222`, and the OSWorld app set: Google Chrome, LibreOffice, GIMP, -VLC, Thunderbird, VS Code, Zotero, Obsidian, Shotcut, FreeCAD, WPS Office and -MuseScore. +VLC, Thunderbird, VS Code, Zotero, Obsidian, Shotcut, FreeCAD, WPS Office, +MuseScore Studio, REAPER, Blender, KiCad, SolveSpace, LabPlot, GeoGebra, +Logisim, OpenBoard and GNOME Calendar. The intended pool shape is `size: 2xlarge` (8 CPU / 16G); the session idles at ~0.5 GB anonymous memory with gnome-shell around 290 MB RSS. @@ -36,13 +37,20 @@ bridge `egress` lane alike (see `docs/desktop.md`). guest it holds the session for that verdict up to 300 s and fails visibly without it. Thunderbird's policy pins mail to the SOCKS5 door on `1080`, so a pool that runs mail tasks opts its policy into `socks5`. +- dockerd, once a task installs it, is pinned to the `vfs` storage driver and + carries its own proxy drop-in: the guest root is overlayfs, which overlay2 + cannot stack on, and dockerd does not read the guest proxy environment. + `/etc/pip.conf` sets `break-system-packages`, since the task set pip-installs + into the system interpreter. ## Build - `24.04/Dockerfile` — `FROM base:24.04`; apt the GNOME/X/AT-SPI/app set, Google Chrome from Google's apt repo, `osworld-server` at - `OSWORLD_SERVER_REF` into a system-site-packages venv; bake the `xvfb`, - `desktop-session`, `osworld-server` and `cdp-bridge` units. + `OSWORLD_SERVER_REF` into a system-site-packages venv; bake the + `guest-proxy`, `xvfb`, `desktop-session`, `osworld-server` and `cdp-bridge` + units. A second layer adds the archive- and vendor-pinned applications the + task set drives. - `platforms` — `linux/amd64`; Chrome is amd64-only. Services are product services, not readiness gates: the claim returns on diff --git a/os-image/node/README.md b/os-image/node/README.md index abda4e07..32aa9bd0 100644 --- a/os-image/node/README.md +++ b/os-image/node/README.md @@ -2,7 +2,7 @@ `base:24.04` plus Node.js 22 LTS (official tarball, sha256-pinned per architecture, unpacked into `/usr/local`) and the native-module build -toolchain — build-essential, python3, unzip — so package installs with +toolchain — build-essential, python3, python3-setuptools, unzip — so package installs with node-gyp steps need only the packages themselves from the network. `node-rt` is the same rootfs squashed to one layer for latency-sensitive pools (see `rt`). diff --git a/protocol/README.md b/protocol/README.md index 679d6104..9f663157 100644 --- a/protocol/README.md +++ b/protocol/README.md @@ -1,14 +1,15 @@ # silkd wire protocol fixtures -Golden JSON frames shared by all three implementations of the silkd -protocol: +Golden JSON frames under `wire/fixtures/v1`, shared by all three +implementations of the silkd protocol: - silkd (Rust, guest) parses/emits these in `silkd/src/proto.rs` tests. - `protocol/wire` (Go, host) round-trips the corpus in its tests; it is the one Go implementation, consumed by both the Go SDK and sandboxd. - the Python SDK round-trips it in `sdk/python/tests/test_fixtures.py`. `req_*.json` are client→server frames, `resp_*.json` server→client. A frame -that only one side can round-trip is a protocol drift bug. +that only one side can round-trip is a protocol drift bug. The Rust side also +pins the corpus size, so adding a verb without its fixture fails CI. `enums.json` pins each wire enum's full value set (error/event/file kinds, git branch actions). Frame fixtures carry only one representative value per diff --git a/sdk/go/checkpoint.go b/sdk/go/checkpoint.go index 7cd534ae..c41d8608 100644 --- a/sdk/go/checkpoint.go +++ b/sdk/go/checkpoint.go @@ -53,9 +53,8 @@ func (ck *Checkpoint) New(ctx context.Context, opts ...Option) (*Sandbox, error) // currently sees to drop any replica a heal pulled — best-effort eventual // cleanup, not a fleet-wide revocation. A peer that misses that broadcast // (offline, partitioned, or joined later) keeps serving branches from its -// replica until the node's checkpoint_ttl_hours ages it out; with that TTL -// at its default of 0 (keep forever), an unreachable peer's replica has no -// cleanup bound at all. +// replica until the node's checkpoint_ttl_hours ages it out, which enabling +// peer heal requires to be set. func (ck *Checkpoint) Delete(ctx context.Context) error { return doNoContent(ctx, ck.c, http.MethodDelete, ck.addr, "/v1/checkpoints/"+ck.ID, nil, ck.c.apiToken, "delete checkpoint") } diff --git a/sdk/langchain/README.md b/sdk/langchain/README.md index 68a2f855..665f1cda 100644 --- a/sdk/langchain/README.md +++ b/sdk/langchain/README.md @@ -19,7 +19,7 @@ with CocoonToolkit("10.0.0.5:7777", api_token="...") as kit: captured moment instead of a clean template — agents resume from prepared state (dependencies installed, repo cloned) in milliseconds. -Sync-native (`_run` calls the stdlib-only cocoonstack-sandbox SDK -directly); async agents get `_arun` bridged via `asyncio.to_thread`. +Sync-native (each tool's function calls the stdlib-only cocoonstack-sandbox +SDK directly); async agents get a coroutine bridged via `asyncio.to_thread`. Full reference: https://cocoonstack.github.io/sandbox/langchain diff --git a/sdk/openai/README.md b/sdk/openai/README.md index 04fae5ab..7c506fad 100644 --- a/sdk/openai/README.md +++ b/sdk/openai/README.md @@ -14,7 +14,8 @@ session = await CocoonSandboxClient().create(options=options) ``` One session is one claimed sandbox; `delete` releases it, `resume` -reattaches from serialized state (id + token). Exec maps to streaming +reattaches from serialized state (node address, api token, sandbox id and +token, owner). Exec maps to streaming `run`, workspace persist/hydrate to tar pull/push, exposed ports to local port proxies. Requires Python >= 3.10 (the `openai-agents` floor). diff --git a/silkd/README.md b/silkd/README.md index 0e1edcc3..aadf5803 100644 --- a/silkd/README.md +++ b/silkd/README.md @@ -9,9 +9,12 @@ attach with a bounded output ring), streaming fs + tar tree push/pull (both atomic), find/replace, watch (ready-acked), pty, structured git, guest port relay (`port_forward`), and an LSP broker that spawns the language server a flavor image ships under `/etc/silkd/lsp.d/`. +Plus two loopback egress relays — `127.0.0.1:3128` and `127.0.0.1:1080` over +vsock to the host proxy (`src/net_egress.rs`) — and the lane verdict that +decides whether an exec inherits the proxy variables (`src/net.rs`). - `src/proto.rs` — frame types + caps; the golden corpus in - `../protocol/wire/fixtures/` is round-tripped by the Rust, Go, and Python + `../protocol/wire/fixtures/v1/` is round-tripped by the Rust, Go, and Python sides, so wire drift fails CI - `src/server.rs` — dispatch; one module per verb family - `tests/` — e2e suites driving the daemon in-process From 5a82908b9462b1bee45f61e18e92a5d735353627 Mon Sep 17 00:00:00 2001 From: CMGS Date: Mon, 14 Sep 2026 21:37:24 +0800 Subject: [PATCH 34/40] docs: tighten four sentences from the last review round The idle-sweep sentence promises only what the sweep's re-check observes; a bare egress rule reads as admitting CONNECT; the checkpoint-delete paragraphs and the Go godoc carry the one reachable zero-TTL case (a node later run with healing off); the PUT /v1/pools 400 list names negative archive durations. --- docs/egress.md | 8 ++++---- docs/sandboxd-api.md | 7 ++++--- docs/sdk-python.md | 9 +++++---- docs/sdk.md | 10 ++++++---- sdk/go/checkpoint.go | 4 ++-- 5 files changed, 21 insertions(+), 17 deletions(-) diff --git a/docs/egress.md b/docs/egress.md index d6fa8bf1..fc826010 100644 --- a/docs/egress.md +++ b/docs/egress.md @@ -44,8 +44,8 @@ identity), and a `DOMAINNAME` destination is resolved host-side, so the guest still needs no resolver. A SOCKS5 tunnel takes the decision an HTTP `CONNECT` to the same host takes, -through the same code: a rule with `methods` that does not name `CONNECT` -denies it, and a rule with a `secret` opens it without injecting anything, on +through the same code: a rule with a nonempty `methods` list that omits +`CONNECT` denies it, and a rule with a `secret` opens it without injecting anything, on both ports. The one difference is `intercept`: on 3128 such a rule terminates the TLS and filters the requests inside, on 1080 there is no HTTP to filter, so the tunnel is refused. The door is opt-in: the pool policy sets @@ -186,8 +186,8 @@ woken sandbox binds at arm time. - `host`: an exact name, a `*.`-prefixed suffix wildcard, or `*`. Case-insensitive. - `methods`: empty means any. Enforced on plaintext and on intercepted HTTPS. A non-intercepted CONNECT tunnel is opaque — the method cannot be checked — so a - rule whose `methods` does not name `CONNECT`, and does not `intercept`, denies - the tunnel outright rather than tunneling unchecked. + rule with a nonempty `methods` list that omits `CONNECT`, and no `intercept`, + denies the tunnel outright rather than tunneling unchecked. - `ports`: empty means any; otherwise the destination port must be listed. The port is the tunnel's target on CONNECT and SOCKS5, the URL's port (or the scheme's default) on the forward path, and the CONNECT's port for every diff --git a/docs/sandboxd-api.md b/docs/sandboxd-api.md index 30e03c0c..859c23f7 100644 --- a/docs/sandboxd-api.md +++ b/docs/sandboxd-api.md @@ -327,9 +327,10 @@ targets online — no restart, live claims untouched: Pools omitted from the list are drained: their unclaimed warm VMs are destroyed and the pool entry retires. `net`/`size` default like a claim's. Answers the fresh `GET /v1/info` payload. 400 bad key, negative warm/idle, -`warm_max` below `warm`, `idle_hibernate_seconds` on an egress pool, an -`archive_after_seconds` not above the pool's `idle_hibernate_seconds`, -duplicate pool, or a config-owned `egress`/`warmup` field; 401 bad api +`warm_max` below `warm`, `idle_hibernate_seconds` on an egress pool, a +negative archive duration or an `archive_after_seconds` not above the pool's +`idle_hibernate_seconds`, duplicate pool, or a config-owned `egress`/`warmup` +field; 401 bad api token; 409 egress pool on a node without an egress attachment. ## POST /v1/drain diff --git a/docs/sdk-python.md b/docs/sdk-python.md index e8e60ec5..40a85940 100644 --- a/docs/sdk-python.md +++ b/docs/sdk-python.md @@ -166,9 +166,9 @@ running — a hibernated sandbox is still reaped at its deadline, so claim with a `ttl_seconds` that covers the idle period. When to hibernate is your policy, unless the deployment opts into `idle_hibernate_seconds` ([deploy](deploy.md#configuration)), which hibernates idle claims -automatically with the same transparent wake. A claim with a live connection -(a relay stream, a buffered exec, a preview dial, an egress request) is never -swept mid-call; the idle clock restarts when that connection ends. +automatically with the same transparent wake. A claim with a connection live +when the sweep checks it (a relay stream, a buffered exec, a preview dial, an +egress request) is not swept; the idle clock restarts when that connection ends. If that deployment also enables `archive_after_seconds`, archiving replaces the original claim deadline with the archive-retention deadline (or no @@ -246,7 +246,8 @@ heal pulled — best-effort eventual cleanup, not a fleet-wide revocation. A pee that misses the broadcast (offline, partitioned, or joined later) keeps serving branches from its replica until the node's `checkpoint_ttl_hours` ages it out; enabling peer heal requires that TTL to be set, so every healed replica has a -cleanup bound. +cleanup bound while healing stays on. A node later run with healing off and +that TTL back at 0 keeps such a replica until an explicit delete. ## Language servers (LSP) diff --git a/docs/sdk.md b/docs/sdk.md index 87a41578..7125ed50 100644 --- a/docs/sdk.md +++ b/docs/sdk.md @@ -216,9 +216,9 @@ claim with a `WithTimeout` that covers the idle period. When to hibernate is your policy; the node only provides the transition — unless the deployment opts into `idle_hibernate_seconds` (see [deploy](deploy.md#configuration)), which hibernates idle claims -automatically with the same transparent wake. A claim with a live connection -(a relay stream, a buffered exec, a preview dial, an egress request) is never -swept mid-call; the idle clock restarts when that connection ends. +automatically with the same transparent wake. A claim with a connection live +when the sweep checks it (a relay stream, a buffered exec, a preview dial, an +egress request) is not swept; the idle clock restarts when that connection ends. If that deployment also enables `archive_after_seconds`, archiving replaces the original claim deadline with the archive-retention deadline (or no @@ -304,7 +304,9 @@ replica a heal pulled — best-effort eventual cleanup, not a fleet-wide revocation. A peer that misses the broadcast (offline, partitioned, or joined later) keeps serving branches from its replica until the node's `checkpoint_ttl_hours` ages it out; enabling peer heal requires that TTL to be -set, so every healed replica has a cleanup bound. +set, so every healed replica has a cleanup bound while healing stays on. A node +later run with healing off and that TTL back at 0 keeps such a replica until an +explicit delete. ## Language servers (LSP) diff --git a/sdk/go/checkpoint.go b/sdk/go/checkpoint.go index c41d8608..fa331252 100644 --- a/sdk/go/checkpoint.go +++ b/sdk/go/checkpoint.go @@ -53,8 +53,8 @@ func (ck *Checkpoint) New(ctx context.Context, opts ...Option) (*Sandbox, error) // currently sees to drop any replica a heal pulled — best-effort eventual // cleanup, not a fleet-wide revocation. A peer that misses that broadcast // (offline, partitioned, or joined later) keeps serving branches from its -// replica until the node's checkpoint_ttl_hours ages it out, which enabling -// peer heal requires to be set. +// replica until the node's checkpoint_ttl_hours ages it out; a node later run +// with healing off and that TTL back at 0 keeps the replica until a delete. func (ck *Checkpoint) Delete(ctx context.Context) error { return doNoContent(ctx, ck.c, http.MethodDelete, ck.addr, "/v1/checkpoints/"+ck.ID, nil, ck.c.apiToken, "delete checkpoint") } From e85823c80c3de5ca2c4f072f70278317cb9c7ec2 Mon Sep 17 00:00:00 2001 From: CMGS Date: Mon, 14 Sep 2026 23:03:45 +0800 Subject: [PATCH 35/40] pool: take the release re-check off the claim path finalizeBatch re-took m.mu after arming every guarded claim to catch a release inside the arm window. Only a root client that lists and deletes the id within that sub-millisecond window can reach it, and the residue is one door listener and one nft table until the next restart; the wake paths keep the check, where the claimant holds the token and the window is hundreds of milliseconds. --- sandboxd/pool/claim.go | 4 ---- 1 file changed, 4 deletions(-) diff --git a/sandboxd/pool/claim.go b/sandboxd/pool/claim.go index 490ba910..b8855422 100644 --- a/sandboxd/pool/claim.go +++ b/sandboxd/pool/claim.go @@ -295,10 +295,6 @@ func (m *Manager) finalizeBatch(ctx context.Context, sbs []*types.Sandbox, ttl t m.rollbackClaim(ctx, sbs) return fmt.Errorf("arm egress %s: %w", sb.ID, armErr) } - // the claim was visible before it was armed, so a release in that window must find nothing left behind - if m.guardedEgress || m.lockEgress { - m.disarmIfReleased(sb) - } } // usage lands only after the batch armed, so a rollback leaves no unterminated claim event for _, sb := range sbs { From 4b8db0b18095b34676d829f48af93339a810c270 Mon Sep 17 00:00:00 2001 From: CMGS Date: Mon, 14 Sep 2026 23:10:30 +0800 Subject: [PATCH 36/40] review: drop the no-op Fetch release and the dead pty exit guard Both store backends returned func() {} from Fetch and every caller composed it into a no-op; the read pin is the pool's record lock. The pty finish guard tested a state only finish sets, from two mutually exclusive call sites. The other cut-list entries stay: the sweep skeletons save under ten lines for a generic helper, the two SDK options are published API, and the silkd helper the ledger named does not exist. --- sandboxd/pool/archive.go | 3 +- sandboxd/pool/archive_test.go | 8 ++--- sandboxd/pool/checkpoint.go | 7 ++--- sandboxd/pool/promote_test.go | 3 +- sandboxd/pool/template.go | 4 +-- sandboxd/store/dir/dir.go | 14 ++++----- sandboxd/store/dir/dir_test.go | 43 +++++++++------------------ sandboxd/store/s3/s3.go | 10 +++---- sandboxd/store/s3/s3_test.go | 18 ++++------- sandboxd/store/store.go | 2 +- sandboxd/store/storetest/storetest.go | 14 ++++----- silkd/src/pty.rs | 6 ++-- 12 files changed, 49 insertions(+), 83 deletions(-) diff --git a/sandboxd/pool/archive.go b/sandboxd/pool/archive.go index 36092669..146fba06 100644 --- a/sandboxd/pool/archive.go +++ b/sandboxd/pool/archive.go @@ -166,7 +166,7 @@ func (m *Manager) wakeArchived(ctx context.Context, sb *types.Sandbox) (string, m.recDone(ck) } }() - dir, _, _, release, err := m.ckpts.Fetch(ctx, ck) + dir, _, _, err := m.ckpts.Fetch(ctx, ck) if errors.Is(err, store.ErrNotFound) { return "", ErrUnknownSandbox // record disagrees with the store } @@ -174,7 +174,6 @@ func (m *Manager) wakeArchived(ctx context.Context, sb *types.Sandbox) (string, return "", fmt.Errorf("wake %s: fetch archive: %w", sb.ID, err) } built, err := m.provision(ctx, sb.Key, dir) - release() if err != nil { return "", fmt.Errorf("wake %s: %w", sb.ID, err) } diff --git a/sandboxd/pool/archive_test.go b/sandboxd/pool/archive_test.go index f21f0b2d..d610382d 100644 --- a/sandboxd/pool/archive_test.go +++ b/sandboxd/pool/archive_test.go @@ -960,12 +960,8 @@ func mustArchive(t *testing.T, m *Manager, sb *types.Sandbox) { func ckExists(t *testing.T, m *Manager, ck string) bool { t.Helper() - _, _, _, release, err := m.ckpts.Fetch(t.Context(), ck) //nolint:dogsled - if err != nil { - return false - } - release() - return true + _, _, _, err := m.ckpts.Fetch(t.Context(), ck) //nolint:dogsled + return err == nil } func archiveCkMarked(m *Manager, ck string) bool { diff --git a/sandboxd/pool/checkpoint.go b/sandboxd/pool/checkpoint.go index b3aaab23..f2c6e5dd 100644 --- a/sandboxd/pool/checkpoint.go +++ b/sandboxd/pool/checkpoint.go @@ -152,7 +152,7 @@ func (m *Manager) FetchCheckpoint(ctx context.Context, ckptID string) (string, [ } l := m.recLock(ckptID) l.RLock() - dir, meta, _, release, err := m.ckpts.Fetch(ctx, ckptID) + dir, meta, _, err := m.ckpts.Fetch(ctx, ckptID) if err != nil { l.RUnlock() m.recDone(ckptID) @@ -162,7 +162,7 @@ func (m *Manager) FetchCheckpoint(ctx context.Context, ckptID string) (string, [ return "", nil, nil, fmt.Errorf("fetch checkpoint: %w", err) } // the read lock spans the transfer so a delete cannot pull the export from under it - return dir, meta, func() { release(); l.RUnlock(); m.recDone(ckptID) }, nil + return dir, meta, func() { l.RUnlock(); m.recDone(ckptID) }, nil } // publishCheckpoint stages, writes the meta, and publishes, returning the record and source snap. @@ -203,14 +203,13 @@ func (m *Manager) claimLoaded(ctx context.Context, ckpt types.Checkpoint, ttl ti l := m.recLock(ckpt.ID) l.RLock() defer func() { l.RUnlock(); m.recDone(ckpt.ID) }() - dir, _, _, release, err := m.ckpts.Fetch(ctx, ckpt.ID) + dir, _, _, err := m.ckpts.Fetch(ctx, ckpt.ID) if errors.Is(err, store.ErrNotFound) { return nil, ErrUnknownCheckpoint // deleted between the pre-check and the lock } if err != nil { return nil, fmt.Errorf("fetch checkpoint: %w", err) } - defer release() if ckpt.Archive { return nil, ErrUnknownCheckpoint // a wake image, not a branchable checkpoint } diff --git a/sandboxd/pool/promote_test.go b/sandboxd/pool/promote_test.go index 50b9d31d..a8f07438 100644 --- a/sandboxd/pool/promote_test.go +++ b/sandboxd/pool/promote_test.go @@ -33,11 +33,10 @@ func TestPromoteThenClaimClonesFromTemplate(t *testing.T) { if gotDigest == "" { t.Fatal("Promote returned an empty content digest") } - golden, _, _, release, err := m.tpls.Fetch(t.Context(), store.TemplateID(key.Hash())) + golden, _, _, err := m.tpls.Fetch(t.Context(), store.TemplateID(key.Hash())) if err != nil { t.Fatalf("template export missing: %v", err) } - release() child, err := claimAny(t.Context(), m, key, 0) if err != nil { diff --git a/sandboxd/pool/template.go b/sandboxd/pool/template.go index 386a557f..e231becd 100644 --- a/sandboxd/pool/template.go +++ b/sandboxd/pool/template.go @@ -260,7 +260,7 @@ func (m *Manager) resolveGolden(ctx context.Context, key types.PoolKey, tenant s id := store.TemplateID(key.Hash()) l := m.recLock(id) l.RLock() - dir, meta, digest, release, err := m.tpls.Fetch(ctx, id) + dir, meta, digest, err := m.tpls.Fetch(ctx, id) if errors.Is(err, store.ErrNotFound) { l.RUnlock() m.recDoneEvict(id) @@ -271,7 +271,7 @@ func (m *Manager) resolveGolden(ctx context.Context, key types.PoolKey, tenant s m.recDone(id) return goldenResolution{release: func() {}}, err } - cleanup := func() { release(); l.RUnlock(); m.recDone(id) } + cleanup := func() { l.RUnlock(); m.recDone(id) } var rec templateRecord if err := json.Unmarshal(meta, &rec); err != nil { cleanup() diff --git a/sandboxd/store/dir/dir.go b/sandboxd/store/dir/dir.go index 4c3b8640..a56a4d2d 100644 --- a/sandboxd/store/dir/dir.go +++ b/sandboxd/store/dir/dir.go @@ -81,28 +81,28 @@ func (d *Store) PublishDigested(ctx context.Context, staging, id string) (string return digest, nil } -func (d *Store) Fetch(ctx context.Context, id string) (string, []byte, string, func(), error) { +func (d *Store) Fetch(ctx context.Context, id string) (string, []byte, string, error) { meta, err := d.ReadMeta(ctx, id) if err != nil { - return "", nil, "", nil, err + return "", nil, "", err } dir := filepath.Join(d.root, id, store.ExportGen(meta)) if _, statErr := os.Stat(dir); errors.Is(statErr, fs.ErrNotExist) { dir = filepath.Join(d.root, id, store.ExportDir) switch _, legacyErr := os.Stat(dir); { case errors.Is(legacyErr, fs.ErrNotExist): - return "", nil, "", nil, store.ErrNotFound + return "", nil, "", store.ErrNotFound case legacyErr != nil: - return "", nil, "", nil, legacyErr + return "", nil, "", legacyErr } } else if statErr != nil { - return "", nil, "", nil, statErr + return "", nil, "", statErr } digest, err := os.ReadFile(filepath.Join(d.root, id, digestName(meta))) //nolint:gosec // id pinned by the instance idRe if err != nil && !errors.Is(err, fs.ErrNotExist) { - return "", nil, "", nil, fmt.Errorf("read digest: %w", err) + return "", nil, "", fmt.Errorf("read digest: %w", err) } - return dir, meta, string(digest), func() {}, nil + return dir, meta, string(digest), nil } func (d *Store) ReadMeta(_ context.Context, id string) ([]byte, error) { diff --git a/sandboxd/store/dir/dir_test.go b/sandboxd/store/dir/dir_test.go index 69621ee9..d119e3db 100644 --- a/sandboxd/store/dir/dir_test.go +++ b/sandboxd/store/dir/dir_test.go @@ -42,7 +42,7 @@ func TestFetchPinsGenerationAcrossStoreInstances(t *testing.T) { backdate(t, firstDigestPath) backdate(t, filepath.Join(root, id, store.MetaFile)) - dir, meta, digest, release, err := reader.Fetch(t.Context(), id) + dir, meta, digest, err := reader.Fetch(t.Context(), id) if err != nil { t.Fatalf("fetch first: %v", err) } @@ -59,13 +59,11 @@ func TestFetchPinsGenerationAcrossStoreInstances(t *testing.T) { if _, statErr := os.Stat(firstDigestPath); statErr != nil { t.Fatalf("first generation digest disturbed by re-publish: %v", statErr) } - release() - _, meta, digest, release, err = reader.Fetch(t.Context(), id) + _, meta, digest, err = reader.Fetch(t.Context(), id) if err != nil { t.Fatalf("fetch second: %v", err) } - defer release() if string(meta) != metaJSON("second") { t.Fatalf("fetch meta %q, want second generation", meta) } @@ -86,11 +84,10 @@ func TestPlainPublishReplacesDigestWithEmpty(t *testing.T) { firstDigestPath := filepath.Join(root, id, digestName(firstMeta)) mustPublish(t, st, id, "plain") - _, meta, digest, release, err := st.Fetch(t.Context(), id) + _, meta, digest, err := st.Fetch(t.Context(), id) if err != nil { t.Fatalf("Fetch: %v", err) } - defer release() if string(meta) != metaJSON("plain") || digest != "" { t.Fatalf("plain replacement meta/digest = %q/%q, want plain/empty", meta, digest) } @@ -142,11 +139,10 @@ func TestPublishDigestedChunkBoundaries(t *testing.T) { if digest != tt.want { t.Errorf("digest = %q, want %q", digest, tt.want) } - _, _, fetched, release, err := st.Fetch(t.Context(), id) + _, _, fetched, err := st.Fetch(t.Context(), id) if err != nil { t.Fatalf("Fetch: %v", err) } - defer release() if fetched != tt.want { t.Errorf("fetched digest = %q, want %q", fetched, tt.want) } @@ -221,12 +217,11 @@ func TestPublishDigestedFailureAndFreshRetry(t *testing.T) { if digest, publishErr := st.PublishDigested(t.Context(), failed, id); publishErr == nil || digest != "" { t.Fatalf("PublishDigested with blocked sidecar = %q, %v, want empty/error", digest, publishErr) } - dir, meta, digest, release, err := st.Fetch(t.Context(), id) + dir, meta, digest, err := st.Fetch(t.Context(), id) if err != nil { t.Fatalf("Fetch previous generation: %v", err) } content, readErr := os.ReadFile(filepath.Join(dir, "disk.img")) - release() if string(meta) != metaJSON("first") || digest != firstDigest { t.Errorf("previous generation meta/digest = %q/%q, want first/%q", meta, digest, firstDigest) } @@ -241,11 +236,10 @@ func TestPublishDigestedFailureAndFreshRetry(t *testing.T) { t.Fatalf("remove failed staging: %v", err) } secondDigest := mustPublishDigested(t, st, id, "second") - dir, meta, digest, release, err = st.Fetch(t.Context(), id) + dir, meta, digest, err = st.Fetch(t.Context(), id) if err != nil { t.Fatalf("Fetch replacement generation: %v", err) } - defer release() content, readErr = os.ReadFile(filepath.Join(dir, "disk.img")) if string(meta) != metaJSON("second") || digest != secondDigest || digest == firstDigest { t.Errorf("replacement generation meta/digest = %q/%q, want second/%q", meta, digest, secondDigest) @@ -287,11 +281,10 @@ func TestSweepGenerationsPairsCurrentAndSupersededDigests(t *testing.T) { t.Errorf("current entry %s was swept: %v", filepath.Base(path), statErr) } } - _, meta, digest, release, err := st.Fetch(t.Context(), id) + _, meta, digest, err := st.Fetch(t.Context(), id) if err != nil { t.Fatalf("Fetch current generation: %v", err) } - defer release() if string(meta) != metaJSON("second") || digest != secondDigest { t.Fatalf("current meta/digest = %q/%q, want second/%q", meta, digest, secondDigest) } @@ -331,11 +324,10 @@ func TestPublishRetriesAfterExpiredInstall(t *testing.T) { mustPublish(t, st, id, "same") backdate(t, gen) mustPublish(t, st, id, "same") - dir, _, _, release, err := st.Fetch(t.Context(), id) + dir, _, _, err := st.Fetch(t.Context(), id) if err != nil { t.Fatalf("fetch after retried publish: %v", err) } - defer release() if _, err := os.Stat(filepath.Join(dir, "disk.img")); err != nil { t.Fatalf("export after retried publish: %v", err) } @@ -350,21 +342,19 @@ func TestFetchLegacyFlatLayout(t *testing.T) { } seedRecord(t, filepath.Join(root, id), "legacy") - dir, meta, digest, release, err := st.Fetch(t.Context(), id) + dir, meta, digest, err := st.Fetch(t.Context(), id) if err != nil { t.Fatalf("fetch legacy: %v", err) } - release() if string(meta) != metaJSON("legacy") || digest != "" || dir != filepath.Join(root, id, store.ExportDir) { t.Fatalf("legacy fetch = %q %q %q, want the flat export dir without digest", dir, meta, digest) } mustPublish(t, st, id, "modern") - dir, meta, digest, release, err = st.Fetch(t.Context(), id) + dir, meta, digest, err = st.Fetch(t.Context(), id) if err != nil { t.Fatalf("fetch after re-publish: %v", err) } - defer release() if string(meta) != metaJSON("modern") || digest != "" || dir == filepath.Join(root, id, store.ExportDir) { t.Fatalf("re-published fetch = %q %q %q, want a generation dir without digest", dir, meta, digest) } @@ -395,11 +385,10 @@ func TestPublishSweepsGenerationsBySupersessionAge(t *testing.T) { if _, statErr := os.Stat(gen2); statErr != nil { t.Fatalf("generation inside its supersession grace was reclaimed: %v", statErr) } - dir, meta, _, release, err := st.Fetch(t.Context(), id) + dir, meta, _, err := st.Fetch(t.Context(), id) if err != nil { t.Fatalf("fetch after sweep: %v", err) } - defer release() if string(meta) != metaJSON("fourth") { t.Fatalf("fetch meta %q, want the current generation", meta) } @@ -437,11 +426,10 @@ func TestSweepSparesPublishingGeneration(t *testing.T) { if renameErr := os.Rename(tmp, filepath.Join(root, id, store.MetaFile)); renameErr != nil { t.Fatalf("commit meta: %v", renameErr) } - dir, gotMeta, digest, release, err := writer.Fetch(t.Context(), id) + dir, gotMeta, digest, err := writer.Fetch(t.Context(), id) if err != nil { t.Fatalf("fetch committed generation: %v", err) } - defer release() if string(gotMeta) != metaJSON("new") { t.Fatalf("fetch meta %q, want new generation", gotMeta) } @@ -466,10 +454,8 @@ func TestSweepSparesLegacyFallback(t *testing.T) { if err := st.SweepStaging(); err != nil { t.Fatalf("sweep: %v", err) } - if _, _, _, release, err := st.Fetch(t.Context(), id); err != nil { + if _, _, _, err := st.Fetch(t.Context(), id); err != nil { t.Fatalf("legacy record swept while still current: %v", err) - } else { - release() } backdate(t, filepath.Join(root, id, store.ExportDir)) @@ -542,11 +528,10 @@ func TestSweepReclaimsOnlyAgedUncommittedGenerations(t *testing.T) { t.Fatalf("remove commit blocker: %v", removeErr) } mustPublish(t, writer, publishID, "fresh") - dir, gotMeta, digest, release, err := writer.Fetch(t.Context(), publishID) + dir, gotMeta, digest, err := writer.Fetch(t.Context(), publishID) if err != nil { t.Fatalf("fetch retried publish: %v", err) } - defer release() if string(gotMeta) != metaJSON("fresh") { t.Fatalf("fetch meta %q, want fresh", gotMeta) } diff --git a/sandboxd/store/s3/s3.go b/sandboxd/store/s3/s3.go index 0c41fb62..2800eb05 100644 --- a/sandboxd/store/s3/s3.go +++ b/sandboxd/store/s3/s3.go @@ -110,23 +110,23 @@ func (s *Store) PublishDigested(ctx context.Context, staging, id string) (string return s.publish(ctx, staging, id, true) } -func (s *Store) Fetch(ctx context.Context, id string) (string, []byte, string, func(), error) { +func (s *Store) Fetch(ctx context.Context, id string) (string, []byte, string, error) { meta, digest, err := s.readMeta(ctx, id) if err != nil { - return "", nil, "", nil, err + return "", nil, "", err } gen := filepath.Join(s.staging, "cache", id, store.ExportGenHash(meta)) export := filepath.Join(gen, store.ExportDir) if _, statErr := os.Stat(export); statErr == nil { - return export, meta, digest, func() {}, nil + return export, meta, digest, nil } _, err, _ = s.fetches.Do(gen, func() (any, error) { return nil, s.populate(ctx, id, meta, gen) }) if err != nil { - return "", nil, "", nil, err + return "", nil, "", err } - return export, meta, digest, func() {}, nil + return export, meta, digest, nil } func (s *Store) ReadMeta(ctx context.Context, id string) ([]byte, error) { diff --git a/sandboxd/store/s3/s3_test.go b/sandboxd/store/s3/s3_test.go index 5ae9cca3..29d90b3e 100644 --- a/sandboxd/store/s3/s3_test.go +++ b/sandboxd/store/s3/s3_test.go @@ -127,11 +127,10 @@ func TestFetchLegacyExportLayout(t *testing.T) { "ck/" + id + "/" + store.ExportDir + "/disk.img": []byte("legacy-bytes"), }} st := newTestStore(t, fake) - dir, _, _, release, err := st.Fetch(t.Context(), id) + dir, _, _, err := st.Fetch(t.Context(), id) if err != nil { t.Fatalf("Fetch: %v", err) } - defer release() got, err := os.ReadFile(filepath.Join(dir, "disk.img")) if err != nil || string(got) != "legacy-bytes" { t.Fatalf("fetched legacy export: %q, %v", got, err) @@ -228,33 +227,30 @@ func TestPublishDigestedRetryAndMetaRequestAccounting(t *testing.T) { } fake.resetRequests() - _, _, fetchedDigest, release, err := st.Fetch(t.Context(), id) + _, _, fetchedDigest, err := st.Fetch(t.Context(), id) if err != nil { t.Fatalf("Fetch miss: %v", err) } - release() if fetchedDigest != want { t.Errorf("Fetch miss digest = %q, want %q", fetchedDigest, want) } assertSingleMetaGet(t, fake, metaKey) fake.resetRequests() - _, _, fetchedDigest, release, err = st.Fetch(t.Context(), id) + _, _, fetchedDigest, err = st.Fetch(t.Context(), id) if err != nil { t.Fatalf("Fetch hit: %v", err) } - release() if fetchedDigest != want { t.Errorf("Fetch hit digest = %q, want %q", fetchedDigest, want) } assertSingleMetaGet(t, fake, metaKey) publishRecord(t, st, id, []byte(`{"id":"`+id+`","gen":2}`), "plain") - _, _, fetchedDigest, release, err = st.Fetch(t.Context(), id) + _, _, fetchedDigest, err = st.Fetch(t.Context(), id) if err != nil { t.Fatalf("Fetch plain replacement: %v", err) } - release() if fetchedDigest != "" { t.Errorf("plain replacement digest = %q, want empty", fetchedDigest) } @@ -291,11 +287,10 @@ func TestPublishDigestedFailurePreservesCommittedGeneration(t *testing.T) { t.Errorf("marker PUT requests after export failure = %d, want %d", got, markerPuts) } - dir, meta, fetchedDigest, release, err := st.Fetch(t.Context(), id) + dir, meta, fetchedDigest, err := st.Fetch(t.Context(), id) if err != nil { t.Fatalf("Fetch old generation: %v", err) } - defer release() content, err := os.ReadFile(filepath.Join(dir, "disk.img")) if err != nil { t.Fatalf("read old generation: %v", err) @@ -309,11 +304,10 @@ func TestPublishDigestedFailurePreservesCommittedGeneration(t *testing.T) { if err != nil { t.Fatalf("retry PublishDigested: %v", err) } - dir, meta, fetchedDigest, release, err = st.Fetch(t.Context(), id) + dir, meta, fetchedDigest, err = st.Fetch(t.Context(), id) if err != nil { t.Fatalf("Fetch replacement: %v", err) } - defer release() content, err = os.ReadFile(filepath.Join(dir, "disk.img")) if err != nil { t.Fatalf("read replacement: %v", err) diff --git a/sandboxd/store/store.go b/sandboxd/store/store.go index 76dc203d..b3974011 100644 --- a/sandboxd/store/store.go +++ b/sandboxd/store/store.go @@ -39,7 +39,7 @@ type Store interface { // PublishDigested applies Publish semantics and returns the export digest. PublishDigested(ctx context.Context, staging, id string) (string, error) // Fetch materializes a record's export locally and returns a release to hold until the clone ends. - Fetch(ctx context.Context, id string) (dir string, meta []byte, digest string, release func(), err error) + Fetch(ctx context.Context, id string) (dir string, meta []byte, digest string, err error) // ReadMeta returns a record's metadata, or an error when the record does not exist. ReadMeta(ctx context.Context, id string) ([]byte, error) // Metas lists the metadata of every record in this instance's id namespace. diff --git a/sandboxd/store/storetest/storetest.go b/sandboxd/store/storetest/storetest.go index e0cd3d60..daf0bb81 100644 --- a/sandboxd/store/storetest/storetest.go +++ b/sandboxd/store/storetest/storetest.go @@ -36,7 +36,7 @@ func RunContract(t *testing.T, st store.Store) { t.Fatalf("ReadMeta: %q, %v", raw, err) } - dir, meta, digest, release, err := st.Fetch(ctx, id) + dir, meta, digest, err := st.Fetch(ctx, id) if err != nil { t.Fatalf("Fetch: %v", err) } @@ -50,7 +50,6 @@ func RunContract(t *testing.T, st store.Store) { if err != nil || string(got) != "snapshot-bytes" { t.Fatalf("fetched export: %q, %v", got, err) } - release() // a half-published checkpoint (no meta) is invisible to Metas. orphan, err := st.Stage("ck_00000000000000bb") @@ -83,7 +82,7 @@ func RunContract(t *testing.T, st store.Store) { if err = st.Publish(ctx, second, id); err != nil { t.Fatalf("Publish second: %v", err) } - dir, meta, digest, release, err = st.Fetch(ctx, id) + dir, meta, digest, err = st.Fetch(ctx, id) if err != nil { t.Fatalf("Fetch second: %v", err) } @@ -99,7 +98,6 @@ func RunContract(t *testing.T, st store.Store) { if got, readErr := os.ReadFile(filepath.Join(dir, "disk2.img")); readErr != nil || string(got) != "second-gen" { //nolint:gosec // test path t.Errorf("second-generation export: %q, %v", got, readErr) } - release() if err = st.Delete(ctx, id); err != nil { t.Fatalf("Delete: %v", err) @@ -107,7 +105,7 @@ func RunContract(t *testing.T, st store.Store) { if _, err = st.ReadMeta(ctx, id); !errors.Is(err, store.ErrNotFound) { t.Fatalf("ReadMeta after Delete: %v, want store.ErrNotFound", err) } - if _, _, _, _, err = st.Fetch(ctx, id); !errors.Is(err, store.ErrNotFound) { + if _, _, _, err = st.Fetch(ctx, id); !errors.Is(err, store.ErrNotFound) { t.Fatalf("Fetch after Delete: %v, want store.ErrNotFound", err) } if metas, err = st.Metas(ctx); err != nil || len(metas) != 0 { @@ -137,14 +135,13 @@ func runDigestContract(t *testing.T, st store.Store) { if digest != want { t.Fatalf("PublishDigested digest %q, want %q", digest, want) } - _, _, fetchedDigest, release, err := st.Fetch(ctx, id) + _, _, fetchedDigest, err := st.Fetch(ctx, id) if err != nil { t.Fatalf("Fetch digested: %v", err) } if fetchedDigest != want { t.Errorf("Fetch digest %q, want %q", fetchedDigest, want) } - release() rejectNonRegularReplacement(t, st, id, want) if err = st.Delete(ctx, id); err != nil { t.Fatalf("Delete digested: %v", err) @@ -171,11 +168,10 @@ func rejectNonRegularReplacement(t *testing.T, st store.Store, id, wantDigest st if _, err = st.PublishDigested(t.Context(), staging, id); err == nil { t.Fatal("PublishDigested accepted a non-regular export entry") } - dir, meta, digest, release, err := st.Fetch(t.Context(), id) + dir, meta, digest, err := st.Fetch(t.Context(), id) if err != nil { t.Fatalf("Fetch after rejected replacement: %v", err) } - defer release() if string(meta) != `{"id":"`+id+`"}` || digest != wantDigest { t.Errorf("committed meta/digest after rejection = %q/%q, want original/%q", meta, digest, wantDigest) } diff --git a/silkd/src/pty.rs b/silkd/src/pty.rs index 84949dcd..3ce76c5c 100644 --- a/silkd/src/pty.rs +++ b/silkd/src/pty.rs @@ -221,10 +221,8 @@ async fn drain( /// Publishes the terminal state once so attachers always see an Exit. fn finish(proc: &Arc, code: i32) { - if proc.exit_code().is_none() { - proc.mark_exited(code); - proc.emit(&Chunk::Exit(code)); - } + proc.mark_exited(code); + proc.emit(&Chunk::Exit(code)); } async fn write_all(master: &Master, mut data: &[u8]) -> std::io::Result<()> { From e6fc0019fd070ac60c0a4cb70f6da5c063f61874 Mon Sep 17 00:00:00 2001 From: CMGS Date: Mon, 14 Sep 2026 23:10:36 +0800 Subject: [PATCH 37/40] pool: write the archive delete marker outside the manager mutex releaseResolved ran MkdirAll and WriteFile under m.mu for an archived claim; reapOnce already writes the same marker after unlocking. The marker is written first and the claim re-checked under the lock, looping if the archive checkpoint changed meanwhile, so a wake or archive racing the release still ends with a marker for the checkpoint the release removes. --- sandboxd/pool/claim.go | 21 +++++++++++++++------ 1 file changed, 15 insertions(+), 6 deletions(-) diff --git a/sandboxd/pool/claim.go b/sandboxd/pool/claim.go index b8855422..7137dd65 100644 --- a/sandboxd/pool/claim.go +++ b/sandboxd/pool/claim.go @@ -136,17 +136,26 @@ func (m *Manager) AgentSocket(id, token string) (string, error) { // releaseResolved re-checks under m.mu that sb is still the live claim: no double teardown. func (m *Manager) releaseResolved(ctx context.Context, id string, sb *types.Sandbox) error { + var marked string m.mu.Lock() - if m.claimed[id] != sb { + for { + if m.claimed[id] != sb { + m.mu.Unlock() + return ErrUnknownSandbox + } + ck := sb.ArchiveCk + if ck == "" || ck == marked { + break + } m.mu.Unlock() - return ErrUnknownSandbox - } - snap, ck, vmName := sb.HibernateSnap, sb.ArchiveCk, sb.VMName - if ck != "" { if markErr := m.markArchiveCk(ck); markErr != nil { - m.mu.Unlock() return fmt.Errorf("release %s: track archive checkpoint: %w", id, markErr) } + marked = ck + m.mu.Lock() + } + snap, ck, vmName := sb.HibernateSnap, sb.ArchiveCk, sb.VMName + if ck != "" { m.pendingCks[ck] = struct{}{} } delete(m.claimed, id) From ef9a3f6c91cb30c20ab3e41e42f6c8cb42d633d5 Mon Sep 17 00:00:00 2001 From: CMGS Date: Mon, 14 Sep 2026 23:23:24 +0800 Subject: [PATCH 38/40] docs: a bare egress rule admits the SOCKS5 door The opt-in sentence said every rule without CONNECT is rejected; a rule with an empty methods list admits it, as Validate checks. --- docs/egress.md | 5 +++-- 1 file changed, 3 insertions(+), 2 deletions(-) diff --git a/docs/egress.md b/docs/egress.md index fc826010..ad0d8ca0 100644 --- a/docs/egress.md +++ b/docs/egress.md @@ -54,8 +54,9 @@ is refused like the HTTP one with no policy. The tenant policy's rules gate every tunnel through the door as they gate `CONNECT`, so the tenant does not opt in separately; its own `socks5` counts only on a claim outside any configured pool, where the tenant policy is the whole policy. A policy that -opts in with rules that all carry `methods` without `CONNECT`, or all -`intercept`, is rejected at load. A pool whose policy does not opt in pays +opts in with rules that all carry a nonempty `methods` list without +`CONNECT`, or all `intercept`, is rejected at load. A pool whose policy does +not opt in pays nothing for the door on the claim path, whatever its tenants' policies say. Audit lines carry `"method":"SOCKS5"`. From b34ec98ec7e454b99268ff5f3883f8afc7f98276 Mon Sep 17 00:00:00 2001 From: CMGS Date: Mon, 14 Sep 2026 23:47:02 +0800 Subject: [PATCH 39/40] sdk/python: enforce run deadline through upgrade and send --- sdk/python/cocoonsandbox/conn.py | 26 ++++++++++++++++++++------ sdk/python/cocoonsandbox/sandbox.py | 21 +++++++++------------ sdk/python/tests/test_stream.py | 27 +++++++++++++++++++++++++++ 3 files changed, 56 insertions(+), 18 deletions(-) diff --git a/sdk/python/cocoonsandbox/conn.py b/sdk/python/cocoonsandbox/conn.py index 54235f5b..39a65569 100644 --- a/sdk/python/cocoonsandbox/conn.py +++ b/sdk/python/cocoonsandbox/conn.py @@ -5,6 +5,7 @@ import contextlib import socket +import time from collections.abc import Iterator from typing import BinaryIO, TypeVar @@ -35,7 +36,7 @@ def send(self, op: str, **fields: object) -> None: self._sock.sendall(encode_request(op, **fields)) def abort(self) -> None: - """Unblocks a reader parked in recv from another thread; close() still owns the socket.""" + """Unblocks a socket operation from another thread; close() still owns the socket.""" with contextlib.suppress(OSError): self._sock.shutdown(socket.SHUT_RDWR) @@ -73,23 +74,22 @@ def close_write(self) -> None: self._sock.shutdown(socket.SHUT_WR) def close(self) -> None: + self.abort() try: self._reader.close() finally: self._sock.close() -def dial_agent(addr: str, sandbox_id: str, token: str, timeout: float) -> Conn: - """Opens the data-plane connection: TCP dial plus a hand-rolled HTTP - Upgrade, both bounded by timeout, so nothing pools or proxies underneath - the byte stream; the stream itself has no idle timeout.""" +def dial_agent(addr: str, sandbox_id: str, token: str, timeout: float, deadline: float | None = None) -> Conn: + """Opens one TCP/HTTP Upgrade relay within timeout and the optional deadline.""" # id/token interpolate into the raw request line; CR/LF would inject headers. for name, value in (("sandbox id", sandbox_id), ("token", token)): if any(c in value for c in "\r\n\0"): raise APIError("agent upgrade", 0, f"{name} contains a control character") host, port = addr.rsplit(":", 1) try: - sock = socket.create_connection((host, int(port)), timeout=timeout) + sock = socket.create_connection((host, int(port)), timeout=_remaining_timeout(timeout, deadline)) except OSError as exc: raise ProtocolError(f"dial {addr}: {exc}") from exc # Nagle off: exec/write send small back-to-back frames before the first read. @@ -104,13 +104,16 @@ def dial_agent(addr: str, sandbox_id: str, token: str, timeout: float) -> Conn: f"Authorization: Bearer {token}\r\n" "\r\n" ) + sock.settimeout(_remaining_timeout(timeout, deadline)) sock.sendall(request.encode()) reader = sock.makefile("rb") + sock.settimeout(_remaining_timeout(timeout, deadline)) status = reader.readline(1024).decode(errors="replace") parts = status.split(" ", 2) code = int(parts[1]) if len(parts) > 1 and parts[1].isdigit() else 0 body_len = 0 while True: + sock.settimeout(_remaining_timeout(timeout, deadline)) header = reader.readline(4096) if header in (b"\r\n", b"\n", b""): break @@ -124,8 +127,10 @@ def dial_agent(addr: str, sandbox_id: str, token: str, timeout: float) -> Conn: raise ProtocolError("invalid content-length in upgrade reply") from exc if code != 101: # a bogus content-length must not buffer unbounded bytes. + sock.settimeout(_remaining_timeout(timeout, deadline)) body = reader.read(min(body_len, MAX_FRAME)).decode(errors="replace") if body_len else "" raise APIError("agent upgrade", code, body.strip() or status.strip()) + _remaining_timeout(timeout, deadline) sock.settimeout(None) return Conn(sock, reader) except Exception: @@ -134,3 +139,12 @@ def dial_agent(addr: str, sandbox_id: str, token: str, timeout: float) -> Conn: reader.close() sock.close() raise + + +def _remaining_timeout(timeout: float, deadline: float | None) -> float: + if deadline is None: + return timeout + remaining = deadline - time.monotonic() + if remaining <= 0: + raise TimeoutError("agent dial timed out") + return min(timeout, remaining) diff --git a/sdk/python/cocoonsandbox/sandbox.py b/sdk/python/cocoonsandbox/sandbox.py index 3d3f1a20..e944d1e5 100644 --- a/sdk/python/cocoonsandbox/sandbox.py +++ b/sdk/python/cocoonsandbox/sandbox.py @@ -105,21 +105,21 @@ def run( expired = threading.Event() try: conn = self._dial(deadline) - except ProtocolError: + except (ProtocolError, TimeoutError): if deadline is not None and time.monotonic() >= deadline: raise TimeoutError(f"command did not finish within {timeout}s") from None raise with conn: - conn.send( - "exec", argv=argv, cwd=cwd or None, env=env, user=user or None, detach=False, session=session or None - ) - # the guest stops draining stdin while blocked on stdout, so feeding it fully first deadlocks. - pump = threading.Thread(target=_feed_stdin, args=(conn, stdin), daemon=True) - pump.start() watchdog = _arm_watchdog(conn, deadline, expired) try: + conn.send( + "exec", argv=argv, cwd=cwd or None, env=env, user=user or None, detach=False, session=session or None + ) + # the guest stops draining stdin while blocked on stdout, so feeding it fully first deadlocks. + pump = threading.Thread(target=_feed_stdin, args=(conn, stdin), daemon=True) + pump.start() code = _pump_stdio(conn, on_stdout, on_stderr) - except ProtocolError: + except (ProtocolError, OSError): if expired.is_set(): raise TimeoutError(f"command did not finish within {timeout}s") from None raise @@ -378,10 +378,7 @@ def close(self) -> None: raise def _dial(self, deadline: float | None = None) -> Conn: - timeout = self._client.timeout - if deadline is not None: - timeout = max(min(timeout, deadline - time.monotonic()), 0.001) - return dial_agent(self.owner, self.id, self.token, timeout) + return dial_agent(self.owner, self.id, self.token, self._client.timeout, deadline) def _open_stream(self, op: str, expect: str = "ready", **fields) -> tuple[Conn, dict]: """Dials, sends op, and waits for the handshake frame, closing the diff --git a/sdk/python/tests/test_stream.py b/sdk/python/tests/test_stream.py index 0ec5a0de..15a3858f 100644 --- a/sdk/python/tests/test_stream.py +++ b/sdk/python/tests/test_stream.py @@ -13,6 +13,24 @@ TIMEOUT = 0.2 +class BlockedSendConn: + def __init__(self) -> None: + self.aborted = threading.Event() + + def __enter__(self): + return self + + def __exit__(self, *exc) -> None: + pass + + def send(self, op: str, **fields) -> None: + assert self.aborted.wait(5 * TIMEOUT), "exec send was not aborted" + raise OSError("connection cut") + + def abort(self) -> None: + self.aborted.set() + + def serve_port_forward(server: socket.socket, quiet: float, ops: list[str]) -> None: conn, _ = server.accept() reader = conn.makefile("rb") @@ -74,6 +92,15 @@ def test_run_timeout_cuts_a_silent_command(): assert time.monotonic() - started < 3 * TIMEOUT +def test_run_timeout_cuts_a_blocked_exec_send(monkeypatch): + sb = Sandbox(client=Client("127.0.0.1:1"), id="sb_1", token="tok", owner="127.0.0.1:1") + conn = BlockedSendConn() + monkeypatch.setattr(sb, "_dial", lambda deadline=None: conn) + with pytest.raises(TimeoutError): + sb.run(["echo", "hello"], timeout=TIMEOUT) + assert conn.aborted.is_set() + + def test_run_rejects_a_non_positive_timeout(): sb = Sandbox(client=Client("127.0.0.1:1", timeout=TIMEOUT), id="sb_1", token="tok", owner="127.0.0.1:1") with pytest.raises(ValueError): From 1616e1780eab94a034733034851f110c68e14eb9 Mon Sep 17 00:00:00 2001 From: CMGS Date: Mon, 14 Sep 2026 23:52:03 +0800 Subject: [PATCH 40/40] sdk/python: wrap the exec send call The previous commit left the call one column past the 120-character limit, which fails ruff format and E501 in CI. --- sdk/python/cocoonsandbox/sandbox.py | 8 +++++++- 1 file changed, 7 insertions(+), 1 deletion(-) diff --git a/sdk/python/cocoonsandbox/sandbox.py b/sdk/python/cocoonsandbox/sandbox.py index e944d1e5..a6668c3a 100644 --- a/sdk/python/cocoonsandbox/sandbox.py +++ b/sdk/python/cocoonsandbox/sandbox.py @@ -113,7 +113,13 @@ def run( watchdog = _arm_watchdog(conn, deadline, expired) try: conn.send( - "exec", argv=argv, cwd=cwd or None, env=env, user=user or None, detach=False, session=session or None + "exec", + argv=argv, + cwd=cwd or None, + env=env, + user=user or None, + detach=False, + session=session or None, ) # the guest stops draining stdin while blocked on stdout, so feeding it fully first deadlocks. pump = threading.Thread(target=_feed_stdin, args=(conn, stdin), daemon=True)