From 600632ad7392d82186f5a4585e92fe482e3ffae1 Mon Sep 17 00:00:00 2001
From: Harry Phan
Date: Tue, 15 Sep 2026 18:51:20 +0700
Subject: [PATCH 001/132] perf(mcp): stop memwal_remember blocking on the whole
Walrus write
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit
A remember is a background job (pending -> running -> uploaded -> done) and
the tool waited for `done` with a 90s deadline. Measured against production
that costs 30-75s for a single short fact, while the job is durably accepted
about a second in. Phase timing on one job: accepted 1.2s, pending->running
2.1s, running->uploaded 29.6s (84% of the call), uploaded->done 2.1s. Payload
size does not drive it — a 1.2 KB fact finished faster than a 97-character
one — so the agent spends half a minute waiting on a queue it cannot affect.
memwal_remember now waits MEMWAL_MCP_REMEMBER_WAIT_MS (default 10s) for the
write to land, so the fast path still returns a blob_id, and otherwise hands
back the job_id and says plainly that the fact is not saved yet. Set 0 to
always return at accept, or 90000 to restore the previous behaviour.
Returning early cannot mean losing writes: a job can still fail after it is
accepted — an outage on the SEAL encrypt sidecar lands it in `failed` with
"Memory encryption backend is unavailable" long after the tool returned. So
memwal_remember_status resolves an in-flight job by id, reporting a failed or
missing job as an error envelope rather than a status line, and a genuinely
failed job still reaches the agent as an error from memwal_remember itself.
Only our own wait expiring is treated as a non-failure.
---
.../__tests__/remember-fast-return.test.ts | 150 ++++++++++++++++++
.../mcp/__tests__/tool-annotations.test.ts | 4 +
.../scripts/mcp/__tests__/tool-scope.test.ts | 1 +
.../server/scripts/mcp/tools/annotations.ts | 4 +
services/server/scripts/mcp/tools/index.ts | 3 +
.../scripts/mcp/tools/remember-status.ts | 104 ++++++++++++
services/server/scripts/mcp/tools/remember.ts | 111 +++++++++++--
7 files changed, 362 insertions(+), 15 deletions(-)
create mode 100644 services/server/scripts/mcp/__tests__/remember-fast-return.test.ts
create mode 100644 services/server/scripts/mcp/tools/remember-status.ts
diff --git a/services/server/scripts/mcp/__tests__/remember-fast-return.test.ts b/services/server/scripts/mcp/__tests__/remember-fast-return.test.ts
new file mode 100644
index 000000000..0b91b415a
--- /dev/null
+++ b/services/server/scripts/mcp/__tests__/remember-fast-return.test.ts
@@ -0,0 +1,150 @@
+import test from "node:test";
+import assert from "node:assert/strict";
+import { McpServer } from "@modelcontextprotocol/sdk/server/mcp.js";
+import { Client } from "@modelcontextprotocol/sdk/client/index.js";
+import { InMemoryTransport } from "@modelcontextprotocol/sdk/inMemory.js";
+import type { MemWalSession } from "../auth.js";
+import { registerRememberTool } from "../tools/remember.js";
+import { registerRememberStatusTool } from "../tools/remember-status.js";
+
+/** The SDK signals "job is still running" by throwing a class the tools match
+ * on by constructor name, so the stub has to carry the same name. */
+class MemWalRememberJobTimeout extends Error {}
+class MemWalRememberJobFailed extends Error {}
+
+function sessionWith(memwal: unknown): MemWalSession {
+ return { memwal, accountId: "acct-test" } as unknown as MemWalSession;
+}
+
+async function callTool(
+ session: MemWalSession,
+ register: (s: McpServer, sess: MemWalSession) => void,
+ name: string,
+ args: Record,
+ t: { after: (fn: () => Promise) => void }
+) {
+ const [clientTransport, serverTransport] = InMemoryTransport.createLinkedPair();
+ const server = new McpServer({ name: "test", version: "1.0.0" });
+ register(server, session);
+ const client = new Client({ name: "remember-test", version: "1.0.0" });
+ t.after(async () => {
+ await client.close();
+ await server.close();
+ });
+ await server.connect(serverTransport);
+ await client.connect(clientTransport);
+ return (await client.callTool({ name, arguments: args })) as {
+ content: Array<{ type: string; text: string }>;
+ isError?: boolean;
+ };
+}
+
+test("memwal_remember returns the blob_id when the write finishes inside the wait", async (t) => {
+ const session = sessionWith({
+ rememberAsync: async () => ({ job_id: "job-1", status: "running" }),
+ waitForRememberJob: async () => ({
+ blob_id: "blob-abc",
+ namespace: "ns-1",
+ }),
+ });
+
+ const res = await callTool(session, registerRememberTool, "memwal_remember", { text: "a fact" }, t);
+
+ assert.equal(res.isError ?? false, false);
+ assert.match(res.content[0].text, /^Saved to Walrus Memory\./);
+ assert.match(res.content[0].text, /blob_id=blob-abc/);
+});
+
+test("memwal_remember hands back a job_id instead of blocking when the write runs long", async (t) => {
+ let waited = false;
+ const session = sessionWith({
+ rememberAsync: async () => ({ job_id: "job-2", status: "running" }),
+ waitForRememberJob: async () => {
+ waited = true;
+ throw new MemWalRememberJobTimeout("deadline exceeded");
+ },
+ });
+
+ const res = await callTool(session, registerRememberTool, "memwal_remember", { text: "a slow fact" }, t);
+
+ assert.ok(waited, "should have waited before giving up on the blob_id");
+ // Still in flight is not an error — the write was accepted and is running.
+ assert.equal(res.isError ?? false, false);
+ assert.match(res.content[0].text, /still writing/i);
+ assert.match(res.content[0].text, /job_id=job-2/);
+ // The agent must not report this as saved, and must be told how to confirm.
+ assert.match(res.content[0].text, /Do not tell the user it is saved/);
+ assert.match(res.content[0].text, /memwal_remember_status/);
+ assert.doesNotMatch(res.content[0].text, /blob_id=/);
+});
+
+test("memwal_remember still surfaces a genuinely failed job as an error", async (t) => {
+ const session = sessionWith({
+ rememberAsync: async () => ({ job_id: "job-3", status: "running" }),
+ waitForRememberJob: async () => {
+ throw new MemWalRememberJobFailed("Memory encryption backend is unavailable");
+ },
+ });
+
+ const res = await callTool(session, registerRememberTool, "memwal_remember", { text: "doomed" }, t);
+
+ assert.equal(res.isError, true);
+ assert.match(res.content[0].text, /Walrus Memory job failed/);
+ assert.match(res.content[0].text, /encryption backend is unavailable/);
+});
+
+test("memwal_remember_status reports the blob_id once the job lands", async (t) => {
+ const session = sessionWith({
+ waitForRememberJob: async () => ({ blob_id: "blob-xyz", namespace: "ns-2" }),
+ });
+
+ const res = await callTool(session, registerRememberStatusTool, "memwal_remember_status", { job_id: "job-4" }, t);
+
+ assert.equal(res.isError ?? false, false);
+ assert.match(res.content[0].text, /^Stored\./);
+ assert.match(res.content[0].text, /blob_id=blob-xyz/);
+});
+
+test("memwal_remember_status reports a failed job as an error so the fact is not assumed saved", async (t) => {
+ const session = sessionWith({
+ waitForRememberJob: async () => {
+ throw new MemWalRememberJobFailed("walrus upload failed");
+ },
+ });
+
+ const res = await callTool(session, registerRememberStatusTool, "memwal_remember_status", { job_id: "job-5" }, t);
+
+ assert.equal(res.isError, true);
+ assert.match(res.content[0].text, /Walrus Memory job failed/);
+});
+
+test("memwal_remember_status says still writing while the job is running", async (t) => {
+ const session = sessionWith({
+ waitForRememberJob: async () => {
+ throw new MemWalRememberJobTimeout("deadline exceeded");
+ },
+ });
+
+ const res = await callTool(session, registerRememberStatusTool, "memwal_remember_status", { job_id: "job-6", wait_seconds: 1 }, t);
+
+ assert.equal(res.isError ?? false, false);
+ assert.match(res.content[0].text, /Still writing/);
+ assert.match(res.content[0].text, /job_id=job-6/);
+});
+
+test("memwal_remember_status never asks the SDK for a deadline shorter than one poll", async (t) => {
+ let seenTimeout: number | undefined;
+ const session = sessionWith({
+ waitForRememberJob: async (_id: string, opts: { timeoutMs: number }) => {
+ seenTimeout = opts.timeoutMs;
+ throw new MemWalRememberJobTimeout("deadline exceeded");
+ },
+ });
+
+ await callTool(session, registerRememberStatusTool, "memwal_remember_status", { job_id: "job-7", wait_seconds: 0 }, t);
+
+ assert.ok(
+ seenTimeout !== undefined && seenTimeout >= 250,
+ `wait_seconds=0 must still allow one status read, got ${seenTimeout}`
+ );
+});
diff --git a/services/server/scripts/mcp/__tests__/tool-annotations.test.ts b/services/server/scripts/mcp/__tests__/tool-annotations.test.ts
index ebb9bae49..bef7caab6 100644
--- a/services/server/scripts/mcp/__tests__/tool-annotations.test.ts
+++ b/services/server/scripts/mcp/__tests__/tool-annotations.test.ts
@@ -32,6 +32,10 @@ test("tools/list publishes safe titles and behavior annotations for every remote
title: "Remember a Fact",
annotations: { readOnlyHint: false, destructiveHint: false },
},
+ memwal_remember_status: {
+ title: "Check a Remember Job",
+ annotations: { readOnlyHint: true, destructiveHint: false },
+ },
memwal_remember_bulk: {
title: "Remember Multiple Facts",
annotations: { readOnlyHint: false, destructiveHint: false },
diff --git a/services/server/scripts/mcp/__tests__/tool-scope.test.ts b/services/server/scripts/mcp/__tests__/tool-scope.test.ts
index e902001ac..1251c4425 100644
--- a/services/server/scripts/mcp/__tests__/tool-scope.test.ts
+++ b/services/server/scripts/mcp/__tests__/tool-scope.test.ts
@@ -7,6 +7,7 @@ import { createMcpServer } from "../server.js";
const WRITE_TOOLS = [
"memwal_remember",
+ "memwal_remember_status",
"memwal_remember_bulk",
"memwal_analyze",
"memwal_restore",
diff --git a/services/server/scripts/mcp/tools/annotations.ts b/services/server/scripts/mcp/tools/annotations.ts
index 12e85fbde..eda6deefb 100644
--- a/services/server/scripts/mcp/tools/annotations.ts
+++ b/services/server/scripts/mcp/tools/annotations.ts
@@ -11,6 +11,10 @@ export const TOOL_METADATA = {
title: "Remember a Fact",
annotations: { readOnlyHint: false, destructiveHint: false },
},
+ memwal_remember_status: {
+ title: "Check a Remember Job",
+ annotations: { readOnlyHint: true, destructiveHint: false },
+ },
memwal_remember_bulk: {
title: "Remember Multiple Facts",
annotations: { readOnlyHint: false, destructiveHint: false },
diff --git a/services/server/scripts/mcp/tools/index.ts b/services/server/scripts/mcp/tools/index.ts
index e230ee39c..763ed0a73 100644
--- a/services/server/scripts/mcp/tools/index.ts
+++ b/services/server/scripts/mcp/tools/index.ts
@@ -2,6 +2,7 @@ import type { McpServer } from "@modelcontextprotocol/sdk/server/mcp.js";
import type { MemWalSession } from "../auth.js";
import { registerRememberTool } from "./remember.js";
+import { registerRememberStatusTool } from "./remember-status.js";
import { registerRememberBulkTool } from "./remember-bulk.js";
import { registerRecallTool } from "./recall.js";
import { registerAnalyzeTool } from "./analyze.js";
@@ -26,6 +27,7 @@ export function registerTools(server: McpServer, session: MemWalSession): void {
if (canWrite) {
registerRememberTool(server, session);
+ registerRememberStatusTool(server, session);
registerRememberBulkTool(server, session);
registerAnalyzeTool(server, session);
registerRestoreTool(server, session);
@@ -38,6 +40,7 @@ export function registerTools(server: McpServer, session: MemWalSession): void {
export {
registerRememberTool,
+ registerRememberStatusTool,
registerRememberBulkTool,
registerRecallTool,
registerAnalyzeTool,
diff --git a/services/server/scripts/mcp/tools/remember-status.ts b/services/server/scripts/mcp/tools/remember-status.ts
new file mode 100644
index 000000000..c6848c671
--- /dev/null
+++ b/services/server/scripts/mcp/tools/remember-status.ts
@@ -0,0 +1,104 @@
+import { z } from "zod";
+import type { McpServer } from "@modelcontextprotocol/sdk/server/mcp.js";
+import type { MemWalSession } from "../auth.js";
+import { TOOL_METADATA } from "./annotations.js";
+import { wrapTool, walruscanBlobUrl } from "./util.js";
+
+/** Default settle window. Short: this tool answers "did it land yet", and an
+ * agent that wants to keep waiting can just call it again. */
+const DEFAULT_WAIT_SECONDS = 2;
+const MAX_WAIT_SECONDS = 30;
+
+const REMEMBER_STATUS_INPUT = {
+ job_id: z
+ .string()
+ .min(1)
+ .describe(
+ "The job_id returned by memwal_remember when the write was still in flight."
+ ),
+ wait_seconds: z
+ .number()
+ .int()
+ .min(0)
+ .max(MAX_WAIT_SECONDS)
+ .optional()
+ .describe(
+ `How long to wait for the job to settle before answering (default ${DEFAULT_WAIT_SECONDS}s, max ${MAX_WAIT_SECONDS}s). Use 0 for an immediate answer.`
+ ),
+} as const;
+
+/**
+ * memwal_remember_status — resolve a `memwal_remember` job that had not
+ * finished when the tool returned.
+ *
+ * This is the other half of the fast-return path in `remember.ts`. That tool
+ * stops waiting once the write is durably accepted, which means a job can
+ * still fail afterwards — a Walrus upload or SEAL encrypt outage lands the job
+ * in `failed`, long after the agent was told it was accepted. Without a way to
+ * ask after the fact, returning early would turn a slow write into a silently
+ * lost one.
+ *
+ * `waitForRememberJob` already encodes the three outcomes an agent has to tell
+ * apart, so this tool leans on it rather than re-deriving them: it resolves at
+ * `done`, throws `MemWalRememberJobFailed` / `MemWalRememberJobNotFound` for
+ * the terminal bad cases (mapped to an error envelope by `wrapTool`), and
+ * throws `MemWalRememberJobTimeout` while the job is simply still running —
+ * the one case that is not an error here.
+ */
+export function registerRememberStatusTool(
+ server: McpServer,
+ session: MemWalSession
+): void {
+ server.registerTool(
+ "memwal_remember_status",
+ {
+ ...TOOL_METADATA.memwal_remember_status,
+ description:
+ "Check whether an in-flight memwal_remember job finished. Call this with the job_id from a memwal_remember result that came back still writing, when you need to confirm the fact was actually stored. Returns the blob_id once stored, an error if the write failed, or tells you it is still running.",
+ inputSchema: REMEMBER_STATUS_INPUT,
+ },
+ wrapTool<{ job_id: string; wait_seconds?: number }>(
+ session,
+ "memwal_remember_status",
+ async ({ job_id, wait_seconds }) => {
+ const waitSeconds = Math.min(
+ wait_seconds ?? DEFAULT_WAIT_SECONDS,
+ MAX_WAIT_SECONDS
+ );
+ try {
+ const result = await session.memwal.waitForRememberJob(
+ job_id,
+ // Floor at one poll: a zero deadline would expire
+ // before the first status read and report "still
+ // running" without ever having asked.
+ { timeoutMs: Math.max(250, waitSeconds * 1000) }
+ );
+ return {
+ content: [
+ {
+ type: "text",
+ text: `Stored. job_id=${job_id} blob_id=${result.blob_id} namespace=${result.namespace}\nExplorer: ${walruscanBlobUrl(result.blob_id)}`,
+ },
+ ],
+ };
+ } catch (err: any) {
+ // Still running is the expected answer, not a failure.
+ // Everything else — failed, not_found, transport — is a
+ // real error and propagates to wrapTool's mapping, which
+ // already names those classes for the agent.
+ if (err?.constructor?.name === "MemWalRememberJobTimeout") {
+ return {
+ content: [
+ {
+ type: "text",
+ text: `Still writing after ${waitSeconds}s. job_id=${job_id}. Check again shortly; do not tell the user it is saved yet.`,
+ },
+ ],
+ };
+ }
+ throw err;
+ }
+ }
+ )
+ );
+}
diff --git a/services/server/scripts/mcp/tools/remember.ts b/services/server/scripts/mcp/tools/remember.ts
index 2e9b523e9..65e765d9c 100644
--- a/services/server/scripts/mcp/tools/remember.ts
+++ b/services/server/scripts/mcp/tools/remember.ts
@@ -3,6 +3,9 @@ import type { McpServer } from "@modelcontextprotocol/sdk/server/mcp.js";
import type { MemWalSession } from "../auth.js";
import { TOOL_METADATA } from "./annotations.js";
import { wrapTool, walruscanBlobUrl } from "./util.js";
+import { createLogger } from "../logger.js";
+
+const log = createLogger("mcp");
const REMEMBER_INPUT = {
text: z
@@ -20,8 +23,53 @@ const REMEMBER_INPUT = {
} as const;
/**
- * memwal_remember — persist a durable fact to MemWal and return only when the
- * blob is written end-to-end (embed → SEAL encrypt → Walrus upload → on-chain).
+ * How long `memwal_remember` waits for the write to finish before handing the
+ * agent a job id instead.
+ *
+ * The write itself is a background job (`pending` → `running` → `uploaded` →
+ * `done`). Blocking to `done` used to be the whole tool, which made a healthy
+ * save cost as much as the slowest step in that pipeline — measured at 30-75s
+ * against production, dominated by the Walrus upload phase. Almost none of
+ * that time tells the agent anything it can act on: the job is already durably
+ * accepted a second or so in.
+ *
+ * So this is a deadline for the *answer*, not for the write. The job keeps
+ * running past it either way; the only question is whether the tool stays
+ * parked. `memwal_remember_status` resolves the ones that run long.
+ *
+ * Set to `0` to always return as soon as the job is accepted. Set it to the
+ * old `90000` to restore the previous always-block behaviour.
+ */
+const DEFAULT_REMEMBER_WAIT_MS = 10_000;
+
+/** Hard ceiling — the relayer's own remember deadline. Waiting past it cannot
+ * observe anything the job has not already settled. */
+const MAX_REMEMBER_WAIT_MS = 90_000;
+
+const REMEMBER_WAIT_MS = (() => {
+ const raw = process.env.MEMWAL_MCP_REMEMBER_WAIT_MS;
+ if (raw === undefined || raw.trim() === "") return DEFAULT_REMEMBER_WAIT_MS;
+ const parsed = Number(raw);
+ // Zero is meaningful here (return at accept), so it is allowed while every
+ // other unusable value falls back rather than silently disabling the wait.
+ if (!Number.isFinite(parsed) || parsed < 0) {
+ log.warn("remember.wait_ms_invalid", {
+ value: raw,
+ usingMs: DEFAULT_REMEMBER_WAIT_MS,
+ });
+ return DEFAULT_REMEMBER_WAIT_MS;
+ }
+ return Math.min(parsed, MAX_REMEMBER_WAIT_MS);
+})();
+
+/**
+ * memwal_remember — persist a durable fact to MemWal.
+ *
+ * Returns once the write is durably accepted by the relayer, waiting up to
+ * `MEMWAL_MCP_REMEMBER_WAIT_MS` for it to finish so the common fast case still
+ * comes back with a `blob_id`. A job that outlives that window is reported as
+ * still writing, with its `job_id`, and is resolved by
+ * `memwal_remember_status` — the job is NOT abandoned.
*
* Call this PROACTIVELY whenever the user reveals a durable fact about
* themselves or the project (preference, decision, constraint, correction,
@@ -38,23 +86,56 @@ export function registerRememberTool(
{
...TOOL_METADATA.memwal_remember,
description:
- "Save a durable fact about the user or project to their Walrus Memory. Call this PROACTIVELY whenever the user states a preference, decision, constraint, correction, identity detail, or recurring workflow — even if they did not say 'remember this'. Skip one-off tasks, the current file or bug, and small talk. Pass the full statement; do not summarize. To save several facts at once, use memwal_remember_bulk instead.",
+ "Save a durable fact about the user or project to their Walrus Memory. Call this PROACTIVELY whenever the user states a preference, decision, constraint, correction, identity detail, or recurring workflow — even if they did not say 'remember this'. Skip one-off tasks, the current file or bug, and small talk. Pass the full statement; do not summarize. To save several facts at once, use memwal_remember_bulk instead. If the result says the write is still in flight, it carries a job_id — confirm it later with memwal_remember_status rather than telling the user it is saved.",
inputSchema: REMEMBER_INPUT,
},
wrapTool<{ text: string; namespace?: string }>(session, "memwal_remember", async ({ text, namespace }) => {
- const result = await session.memwal.rememberAndWait(
- text,
- namespace,
- { timeoutMs: 90_000 }
- );
- return {
- content: [
- {
- type: "text",
- text: `Saved to Walrus Memory. blob_id=${result.blob_id} namespace=${result.namespace}\nExplorer: ${walruscanBlobUrl(result.blob_id)}`,
- },
- ],
+ const accepted = await session.memwal.rememberAsync(text, namespace);
+
+ const stillWriting = () => {
+ log.info("remember.returned_in_flight", {
+ jobId: accepted.job_id,
+ waitedMs: REMEMBER_WAIT_MS,
+ accountId: session.accountId ?? null,
+ });
+ return {
+ content: [
+ {
+ type: "text" as const,
+ text:
+ `Accepted by Walrus Memory and still writing. job_id=${accepted.job_id}` +
+ `${namespace ? ` namespace=${namespace}` : ""}\n` +
+ `The write did not finish within ${Math.round(REMEMBER_WAIT_MS / 1000)}s, so there is no blob_id yet. ` +
+ `Do not tell the user it is saved — confirm with memwal_remember_status(job_id="${accepted.job_id}").`,
+ },
+ ],
+ };
};
+
+ if (REMEMBER_WAIT_MS === 0) return stillWriting();
+
+ try {
+ const result = await session.memwal.waitForRememberJob(
+ accepted.job_id,
+ { timeoutMs: REMEMBER_WAIT_MS }
+ );
+ return {
+ content: [
+ {
+ type: "text" as const,
+ text: `Saved to Walrus Memory. blob_id=${result.blob_id} namespace=${result.namespace}\nExplorer: ${walruscanBlobUrl(result.blob_id)}`,
+ },
+ ],
+ };
+ } catch (err: any) {
+ // Only our own wait expiring is a non-failure. A job that
+ // actually failed still has to reach the agent as an error, so
+ // everything else propagates to `wrapTool`'s mapping.
+ if (err?.constructor?.name === "MemWalRememberJobTimeout") {
+ return stillWriting();
+ }
+ throw err;
+ }
})
);
}
From 35436f259816997ed39fdb3bb85c68f6d5d76e7e Mon Sep 17 00:00:00 2001
From: Harry Phan
Date: Tue, 15 Sep 2026 19:10:18 +0700
Subject: [PATCH 002/132] fix(mcp): match the SDK's real job-timeout shape, not
a class name
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit
The fast-return path keyed "job is still running" off a
`MemWalRememberJobTimeout` constructor name. No shipped SDK throws that: both
the pinned 0.0.x and the current 0.1.x line reject with a plain Error carrying
`status` — 504 when the wait ran out with the job still going, 500 when the job
itself failed. `wrapTool`'s mapping for those names is dead code for the same
reason.
So every slow write came back as `isError: true` with "Tool error: remember job
timed out after 30000ms", telling the agent the save had broken when it was
simply still running — the exact failure mode the fast return exists to avoid.
Caught by driving the tools against the production relayer; the unit tests
missed it because the stubs threw a conveniently-named class instead of the
shape the SDK actually produces.
Match on `status` instead, and reproduce the real error shape in the stubs.
memwal_remember_status now also names a failed job (500) plainly rather than
letting it fall through to the generic "Tool error" prefix.
---
.../__tests__/remember-fast-return.test.ts | 24 ++++++-----
.../scripts/mcp/tools/remember-status.ts | 40 ++++++++++++++-----
services/server/scripts/mcp/tools/remember.ts | 20 +++++++---
3 files changed, 58 insertions(+), 26 deletions(-)
diff --git a/services/server/scripts/mcp/__tests__/remember-fast-return.test.ts b/services/server/scripts/mcp/__tests__/remember-fast-return.test.ts
index 0b91b415a..e7ddbaf98 100644
--- a/services/server/scripts/mcp/__tests__/remember-fast-return.test.ts
+++ b/services/server/scripts/mcp/__tests__/remember-fast-return.test.ts
@@ -7,10 +7,15 @@ import type { MemWalSession } from "../auth.js";
import { registerRememberTool } from "../tools/remember.js";
import { registerRememberStatusTool } from "../tools/remember-status.js";
-/** The SDK signals "job is still running" by throwing a class the tools match
- * on by constructor name, so the stub has to carry the same name. */
-class MemWalRememberJobTimeout extends Error {}
-class MemWalRememberJobFailed extends Error {}
+/** The SDK throws a plain Error carrying a `status` — 504 when the wait ran out
+ * with the job still going, 500 when the job itself failed. It ships no named
+ * error classes for these, so the stubs must reproduce that exact shape:
+ * stubbing a nicely-named class here is what let a constructor-name check pass
+ * in tests and fail against the real SDK. */
+const jobStillRunning = (msg = "remember job timed out after 1000ms") =>
+ Object.assign(new Error(msg), { status: 504 });
+const jobFailed = (msg: string) =>
+ Object.assign(new Error(`remember job failed: ${msg}`), { status: 500 });
function sessionWith(memwal: unknown): MemWalSession {
return { memwal, accountId: "acct-test" } as unknown as MemWalSession;
@@ -61,7 +66,7 @@ test("memwal_remember hands back a job_id instead of blocking when the write run
rememberAsync: async () => ({ job_id: "job-2", status: "running" }),
waitForRememberJob: async () => {
waited = true;
- throw new MemWalRememberJobTimeout("deadline exceeded");
+ throw jobStillRunning();
},
});
@@ -82,14 +87,13 @@ test("memwal_remember still surfaces a genuinely failed job as an error", async
const session = sessionWith({
rememberAsync: async () => ({ job_id: "job-3", status: "running" }),
waitForRememberJob: async () => {
- throw new MemWalRememberJobFailed("Memory encryption backend is unavailable");
+ throw jobFailed("Memory encryption backend is unavailable");
},
});
const res = await callTool(session, registerRememberTool, "memwal_remember", { text: "doomed" }, t);
assert.equal(res.isError, true);
- assert.match(res.content[0].text, /Walrus Memory job failed/);
assert.match(res.content[0].text, /encryption backend is unavailable/);
});
@@ -108,7 +112,7 @@ test("memwal_remember_status reports the blob_id once the job lands", async (t)
test("memwal_remember_status reports a failed job as an error so the fact is not assumed saved", async (t) => {
const session = sessionWith({
waitForRememberJob: async () => {
- throw new MemWalRememberJobFailed("walrus upload failed");
+ throw jobFailed("walrus upload failed");
},
});
@@ -121,7 +125,7 @@ test("memwal_remember_status reports a failed job as an error so the fact is not
test("memwal_remember_status says still writing while the job is running", async (t) => {
const session = sessionWith({
waitForRememberJob: async () => {
- throw new MemWalRememberJobTimeout("deadline exceeded");
+ throw jobStillRunning();
},
});
@@ -137,7 +141,7 @@ test("memwal_remember_status never asks the SDK for a deadline shorter than one
const session = sessionWith({
waitForRememberJob: async (_id: string, opts: { timeoutMs: number }) => {
seenTimeout = opts.timeoutMs;
- throw new MemWalRememberJobTimeout("deadline exceeded");
+ throw jobStillRunning();
},
});
diff --git a/services/server/scripts/mcp/tools/remember-status.ts b/services/server/scripts/mcp/tools/remember-status.ts
index c6848c671..af38f2794 100644
--- a/services/server/scripts/mcp/tools/remember-status.ts
+++ b/services/server/scripts/mcp/tools/remember-status.ts
@@ -4,6 +4,13 @@ import type { MemWalSession } from "../auth.js";
import { TOOL_METADATA } from "./annotations.js";
import { wrapTool, walruscanBlobUrl } from "./util.js";
+/** `waitForRememberJob` rejects with 504 when its deadline passes with the job
+ * still running, and 500 when the job itself failed. Neither is a distinct
+ * error class in any shipped SDK line, so the status code is the only stable
+ * discriminator. */
+const JOB_STILL_RUNNING_STATUS = 504;
+const JOB_FAILED_STATUS = 500;
+
/** Default settle window. Short: this tool answers "did it land yet", and an
* agent that wants to keep waiting can just call it again. */
const DEFAULT_WAIT_SECONDS = 2;
@@ -38,12 +45,12 @@ const REMEMBER_STATUS_INPUT = {
* ask after the fact, returning early would turn a slow write into a silently
* lost one.
*
- * `waitForRememberJob` already encodes the three outcomes an agent has to tell
- * apart, so this tool leans on it rather than re-deriving them: it resolves at
- * `done`, throws `MemWalRememberJobFailed` / `MemWalRememberJobNotFound` for
- * the terminal bad cases (mapped to an error envelope by `wrapTool`), and
- * throws `MemWalRememberJobTimeout` while the job is simply still running —
- * the one case that is not an error here.
+ * `waitForRememberJob` already encodes the outcomes an agent has to tell apart,
+ * so this tool leans on it rather than re-deriving them: it resolves at `done`,
+ * and otherwise rejects with a plain Error carrying a `status` — 500 when the
+ * job failed, 504 when only our wait ran out and the job is still going. The
+ * SDK ships no dedicated error classes for these, so `status` is what we match
+ * on; a constructor-name check silently never fires.
*/
export function registerRememberStatusTool(
server: McpServer,
@@ -82,11 +89,8 @@ export function registerRememberStatusTool(
],
};
} catch (err: any) {
- // Still running is the expected answer, not a failure.
- // Everything else — failed, not_found, transport — is a
- // real error and propagates to wrapTool's mapping, which
- // already names those classes for the agent.
- if (err?.constructor?.name === "MemWalRememberJobTimeout") {
+ // Still running is the expected answer here, not a failure.
+ if (err?.status === JOB_STILL_RUNNING_STATUS) {
return {
content: [
{
@@ -96,6 +100,20 @@ export function registerRememberStatusTool(
],
};
}
+ // A terminal failure is the case this tool exists for, so
+ // it is named plainly rather than left to the generic
+ // "Tool error" prefix: the fact was NOT stored.
+ if (err?.status === JOB_FAILED_STATUS) {
+ return {
+ content: [
+ {
+ type: "text",
+ text: `Walrus Memory job failed: job_id=${job_id} was not stored — ${err?.message ?? "unknown error"}. The fact is NOT in memory; save it again.`,
+ },
+ ],
+ isError: true,
+ };
+ }
throw err;
}
}
diff --git a/services/server/scripts/mcp/tools/remember.ts b/services/server/scripts/mcp/tools/remember.ts
index 65e765d9c..34632bddb 100644
--- a/services/server/scripts/mcp/tools/remember.ts
+++ b/services/server/scripts/mcp/tools/remember.ts
@@ -42,6 +42,11 @@ const REMEMBER_INPUT = {
*/
const DEFAULT_REMEMBER_WAIT_MS = 10_000;
+/** `waitForRememberJob` rejects with this HTTP status when the deadline passes
+ * without the job settling. The job itself is still running; only the wait
+ * ended. A genuinely failed job rejects with 500 instead. */
+const JOB_STILL_RUNNING_STATUS = 504;
+
/** Hard ceiling — the relayer's own remember deadline. Waiting past it cannot
* observe anything the job has not already settled. */
const MAX_REMEMBER_WAIT_MS = 90_000;
@@ -105,7 +110,9 @@ export function registerRememberTool(
text:
`Accepted by Walrus Memory and still writing. job_id=${accepted.job_id}` +
`${namespace ? ` namespace=${namespace}` : ""}\n` +
- `The write did not finish within ${Math.round(REMEMBER_WAIT_MS / 1000)}s, so there is no blob_id yet. ` +
+ (REMEMBER_WAIT_MS === 0
+ ? "The tool did not wait for the write, so there is no blob_id yet. "
+ : `The write did not finish within ${Math.round(REMEMBER_WAIT_MS / 1000)}s, so there is no blob_id yet. `) +
`Do not tell the user it is saved — confirm with memwal_remember_status(job_id="${accepted.job_id}").`,
},
],
@@ -128,10 +135,13 @@ export function registerRememberTool(
],
};
} catch (err: any) {
- // Only our own wait expiring is a non-failure. A job that
- // actually failed still has to reach the agent as an error, so
- // everything else propagates to `wrapTool`'s mapping.
- if (err?.constructor?.name === "MemWalRememberJobTimeout") {
+ // Only our own wait expiring is a non-failure. The SDK signals
+ // that as a plain Error carrying `status: 504` (a failed job
+ // carries 500) — it has no dedicated error class, in either the
+ // pinned 0.0.x or the current 0.1.x line, so matching on a
+ // constructor name would never fire and every slow write would
+ // surface as an error.
+ if (err?.status === JOB_STILL_RUNNING_STATUS) {
return stillWriting();
}
throw err;
From 63deb0e5c930b7a9bc99072925a3d9493585a277 Mon Sep 17 00:00:00 2001
From: Harry Phan
Date: Tue, 15 Sep 2026 21:29:11 +0700
Subject: [PATCH 003/132] perf(mcp): return memwal_remember at accept, not at
done
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit
The 10s wait this branch shipped was the worst of both options. Against the
measured 30-75s completion spread it lands in the pending branch on nearly
every call, so the agent pays the full 10s and still gets no guarantee. Wait
long enough to mean it (MEMWAL_MCP_REMEMBER_WAIT_MS=90000) or do not wait —
so the default is now 0 and the tool returns once the relayer has durably
accepted the job, ~1.1s.
Returning at accept is safe from a disconnect: the job is a row in
remember_jobs driven by the relayer, not work held in this process. It is not
safe from a job that fails after acceptance, which is why the result never
reads as saved and memwal_remember_status exists to settle it.
Split the shared budget parsing and job-error mapping into remember-wait.ts
rather than duplicating the 504-vs-500 discrimination in both tools.
memwal_remember_status gains waitMs=0 for an immediate read. waitForRememberJob
cannot express that — it sleeps before its first poll, so a 0ms deadline
answers "still running" without ever asking the relayer. That needs
getRememberStatus, added to the SDK in 0.0.4, hence the dependency bump. The
wider 0.1.x upgrade stays with the sidecar SDK work.
Also adds the Streamable HTTP transport behind MEMWAL_MCP_TRANSPORT (sse by
default). It answers a call on the same request, so there is no idle watchdog
and no replay-on-reconnect.
Tests: 269 run, 236 pass, 33 fail — the same 33 sandbox port-binding failures
present on dev. tsc --noEmit clean.
---
docs/reference/environment-variables.md | 1 +
packages/mcp/src/auth-required.ts | 18 +-
packages/mcp/src/bridge.ts | 50 +++-
packages/mcp/src/instructions.ts | 7 +
packages/mcp/src/streamable.ts | 178 ++++++++++++
packages/mcp/test/coldstart-init.test.mjs | 1 +
.../mcp/test/streamable-transport.test.mjs | 53 ++++
.../__tests__/remember-fast-return.test.ts | 271 ++++++++++--------
.../mcp/__tests__/tool-annotations.test.ts | 8 +-
.../scripts/mcp/__tests__/tool-scope.test.ts | 2 +-
services/server/scripts/mcp/server.ts | 7 +
.../server/scripts/mcp/tools/annotations.ts | 9 +-
.../scripts/mcp/tools/remember-status.ts | 181 ++++++------
.../server/scripts/mcp/tools/remember-wait.ts | 138 +++++++++
services/server/scripts/mcp/tools/remember.ts | 128 +++------
services/server/scripts/mcp/tools/util.ts | 10 +-
services/server/scripts/package-lock.json | 8 +-
services/server/scripts/package.json | 2 +-
18 files changed, 757 insertions(+), 315 deletions(-)
create mode 100644 packages/mcp/src/streamable.ts
create mode 100644 packages/mcp/test/streamable-transport.test.mjs
create mode 100644 services/server/scripts/mcp/tools/remember-wait.ts
diff --git a/docs/reference/environment-variables.md b/docs/reference/environment-variables.md
index cb6d095cf..5ccc2af17 100644
--- a/docs/reference/environment-variables.md
+++ b/docs/reference/environment-variables.md
@@ -70,6 +70,7 @@ The stdio MCP package reads these environment variables directly. A CLI flag tak
| `MEMWAL_CLIENT_LABEL` | `--label ` | `MCP Client` / `Walrus Memory MCP` | Friendly delegate-key label shown in the dashboard |
| `MEMWAL_MCP_DEBUG` | none | `0` | Set to `1` for verbose stderr logging |
| `MEMWAL_CREDS_DIR` | none | `~/.memwal` | Directory holding `credentials.json`. Overrides both project-local and `~/.memwal` credentials, re-read on every access so a test can redirect it after import. Mainly for tests, which must not write into the real credential directory |
+| `MEMWAL_MCP_TRANSPORT` | none | `sse` | Which relayer transport the stdio bridge dials. `sse` uses the legacy split (`POST /api/mcp/messages` + `GET /api/mcp/sse`). `http` (aliases `streamable`, `streamable-http`) uses the Streamable HTTP endpoint `/api/mcp`, where a call is answered on the same request — no idle watchdog and no replay-on-reconnect. Unrecognised values fall back to `sse` |
| `MEMWAL_MCP_SSE_IDLE_MS` | none | `30000` | Maximum milliseconds of silence on the SSE stream before the bridge treats the session as dead and reconnects. Values below `500` are ignored and fall back to the default. Mainly for tests |
| `MEMWAL_MCP_CALL_TIMEOUT_MS` | none | `240000` | Maximum milliseconds a single request might wait for its response before the bridge answers with a retryable error. Covers a reply lost while the stream itself stays healthy, which `MEMWAL_MCP_SSE_IDLE_MS` cannot detect. The default is derived in code from the slowest server-side tool deadline plus headroom, so it moves with that tool rather than being pinned here. Values below `1000` are ignored and fall back to the default |
| `MEMWAL_MCP_THROTTLE_FLOOR_MS` | none | `5000` | Minimum milliseconds the bridge waits before retrying an SSE handshake the relayer refused with HTTP 429 and no `Retry-After`. A `Retry-After` on the response wins instead. Either way the wait is capped at `60000`. Non-numeric or negative values are ignored and fall back to the default. Mainly for tests |
diff --git a/packages/mcp/src/auth-required.ts b/packages/mcp/src/auth-required.ts
index ed6bdfda2..b099d0d56 100644
--- a/packages/mcp/src/auth-required.ts
+++ b/packages/mcp/src/auth-required.ts
@@ -41,7 +41,7 @@ interface RpcMessage {
const SIGNED_OUT_REMEMBER =
"Save a fact to the user's Walrus Memory personal memory. Call ONLY when the user explicitly asks to remember/save something. Pass the full, detailed text — never summarize.";
const SIGNED_IN_REMEMBER =
- "Save a durable fact about the user or project to their Walrus Memory. Call this PROACTIVELY whenever the user states a preference, decision, constraint, correction, identity detail, or recurring workflow — even if they did not say 'remember this'. Skip one-off tasks, the current file or bug, and small talk. Pass the full statement; do not summarize. To save several facts at once, use memwal_remember_bulk instead.";
+ "Save a durable fact about the user or project to their Walrus Memory. Call this PROACTIVELY whenever the user states a preference, decision, constraint, correction, identity detail, or recurring workflow — even if they did not say 'remember this'. Skip one-off tasks, the current file or bug, and small talk. Pass the full statement; do not summarize. To save several facts at once, use memwal_remember_bulk instead. Walrus writes queue, so this may return a job_id with the fact NOT yet saved — in that case say so rather than claiming it is stored, and resolve it with memwal_remember_status.";
const SIGNED_OUT_RECALL =
"Search the user's Walrus Memory for facts relevant to a query. Returns matching memories ranked by relevance.";
const SIGNED_IN_RECALL =
@@ -88,6 +88,22 @@ function buildToolDefinitions(proactive: boolean) {
additionalProperties: false,
},
},
+ {
+ name: "memwal_remember_status",
+ title: "Check a Remember Job",
+ annotations: { readOnlyHint: true, destructiveHint: false },
+ description:
+ "Check whether an in-flight memwal_remember write has landed. Call this with the job_id memwal_remember returned when it reported the fact was NOT saved yet. Returns the blob_id once stored, reports that it is still uploading (call again), or reports that it failed \u2014 in which case the fact was never stored and you should send it again with memwal_remember.",
+ inputSchema: {
+ type: "object",
+ properties: {
+ job_id: { type: "string", minLength: 1 },
+ waitMs: { type: "integer", minimum: 0, maximum: 60000, default: 10000 },
+ },
+ required: ["job_id"],
+ additionalProperties: false,
+ },
+ },
{
name: "memwal_recall",
title: "Recall Memories",
diff --git a/packages/mcp/src/bridge.ts b/packages/mcp/src/bridge.ts
index 6b9319ee3..91b6f2a4e 100644
--- a/packages/mcp/src/bridge.ts
+++ b/packages/mcp/src/bridge.ts
@@ -33,6 +33,7 @@ import {
loginSuccessNotification,
type LoginSuccessInfo,
} from "./messages.js";
+import { openStreamableSession, resolveTransport } from "./streamable.js";
import { MEMWAL_MCP_VERSION } from "./version.js";
/** Bridge mode runtime config — the URLs / label resolved at boot from
@@ -423,6 +424,12 @@ class RelayerUnauthorizedError extends Error {
interface SseHandshakeResult {
/** Absolute URL the client must POST to for outbound JSON-RPC messages. */
postUrl: string;
+ /**
+ * Forward one message, resolving with the HTTP status. Same shape as the
+ * Streamable transport's `send`, so the forwarding path does not have to
+ * know which transport is underneath.
+ */
+ send: (msg: RpcMessage, creds: MemWalCredentials, extra: Record) => Promise;
/** Per-line iterator for incoming SSE messages (already-parsed JSON-RPC). */
iter: AsyncIterator;
/** Abort + close the SSE stream. */
@@ -440,6 +447,28 @@ function mcpAuthHeaders(
};
}
+/**
+ * Open a relayer session on the configured transport.
+ *
+ * `MEMWAL_MCP_TRANSPORT=http` dials the Streamable HTTP endpoint, which
+ * answers a call on the same request instead of splitting POST from reply.
+ * Default stays SSE until the new path has production mileage.
+ */
+async function openRelaySession(
+ relayerUrl: string,
+ creds: MemWalCredentials,
+ extraHeaders: Record = {},
+): Promise {
+ if (resolveTransport(process.env.MEMWAL_MCP_TRANSPORT) === "http") {
+ // `postUrl` is logging-only on this path; the session owns its
+ // endpoint. A plain cast, not `as unknown as` — the two shapes must
+ // stay structurally compatible, and a widening cast would hide it if
+ // they ever stopped being.
+ return (await openStreamableSession(relayerUrl, creds, extraHeaders)) as SseHandshakeResult;
+ }
+ return openSseStream(relayerUrl, creds, extraHeaders);
+}
+
async function openSseStream(
relayerUrl: string,
creds: MemWalCredentials,
@@ -712,6 +741,7 @@ async function openSseStream(
return {
postUrl,
+ send: (msg, sendCreds, extra) => postMessage(postUrl, msg, sendCreds, extra),
iter,
abort: () => {
controller.abort();
@@ -1092,7 +1122,7 @@ export async function runBridge(
}
function postIfCurrent(
epoch: number,
- postUrl: string,
+ send: SseHandshakeResult["send"],
msg: RpcMessage,
postCreds: MemWalCredentials,
): Promise {
@@ -1105,7 +1135,7 @@ export async function runBridge(
const tracked = inFlight.get(msg.id);
if (tracked) tracked.sent = true;
}
- return postMessage(postUrl, msg, postCreds, extraHeaders).then((status) => {
+ return send(msg, postCreds, extraHeaders).then((status) => {
// 404 is the relayer saying that session does not exist, so the
// message was discarded rather than routed: it provably did not
// run, and the request goes back to being never-sent.
@@ -1334,7 +1364,7 @@ export async function runBridge(
// gone, so there is nothing to authorize a new session
// with. Belt-and-braces against `loggedOut` alone.
if (!openingCreds) break;
- const candidate = await openSseStream(
+ const candidate = await openRelaySession(
openingCreds.relayerUrl,
openingCreds,
connectHeaders(),
@@ -1425,9 +1455,9 @@ export async function runBridge(
expectSuppressedReply(msg.id);
}
const epoch = sessionEpoch;
- const postUrl = sse.postUrl;
+ const send = sse.send;
const status = await enqueuePost(() =>
- postIfCurrent(epoch, postUrl, msg, openingCreds),
+ postIfCurrent(epoch, send, msg, openingCreds),
);
log.info("bridge.replayed", { id, status });
} catch (err) {
@@ -1997,10 +2027,10 @@ export async function runBridge(
return;
}
const epoch = sessionEpoch;
- const postUrl = sse.postUrl;
+ const send = sse.send;
const postCreds = creds;
const status = await enqueuePost(() =>
- postIfCurrent(epoch, postUrl, msg, postCreds),
+ postIfCurrent(epoch, send, msg, postCreds),
);
if (status === 404) {
log.warn("bridge.session_stale", { sessionUrl: sse.postUrl });
@@ -2038,10 +2068,10 @@ export async function runBridge(
const msg = pendingForward.shift()!;
try {
const epoch = sessionEpoch;
- const postUrl = sse.postUrl;
+ const send = sse.send;
const postCreds = creds;
const status = await enqueuePost(() =>
- postIfCurrent(epoch, postUrl, msg, postCreds),
+ postIfCurrent(epoch, send, msg, postCreds),
);
if (status === 404) {
// Stale session right after connect. EVERY id-bearing
@@ -2403,7 +2433,7 @@ export async function runBridge(
}
const openingGeneration = credentialGeneration;
try {
- const candidate = await openSseStream(creds.relayerUrl, creds, connectHeaders());
+ const candidate = await openRelaySession(creds.relayerUrl, creds, connectHeaders());
if (stdinClosed) {
candidate.abort();
break;
diff --git a/packages/mcp/src/instructions.ts b/packages/mcp/src/instructions.ts
index df4452e64..e964cbfc0 100644
--- a/packages/mcp/src/instructions.ts
+++ b/packages/mcp/src/instructions.ts
@@ -39,6 +39,13 @@ export const PROACTIVE_INSTRUCTIONS = [
"summary. Skip one-off tasks, the current file or bug, and small talk. Use",
"memwal_remember_bulk when several distinct facts arrived at once.",
"",
+ "A Walrus write is queued, not instant: memwal_remember normally returns a job_id with",
+ "the fact ACCEPTED but NOT YET SAVED, and storing it takes roughly another 30-60s. That",
+ "is the healthy path, not an error. Tell the user the fact is being saved rather than",
+ "that it is saved, and do not re-send it — that queues a duplicate. A job can still fail",
+ "after acceptance, so when it matters that a fact landed, resolve the job_id with",
+ "memwal_remember_status; only the blob_id it returns means the fact is stored.",
+ "",
"RECOVER: if memwal_recall unexpectedly returns nothing for a namespace that has been used",
"before, call memwal_restore to rebuild the index from Walrus.",
"",
diff --git a/packages/mcp/src/streamable.ts b/packages/mcp/src/streamable.ts
new file mode 100644
index 000000000..b203ca2bd
--- /dev/null
+++ b/packages/mcp/src/streamable.ts
@@ -0,0 +1,178 @@
+/**
+ * Streamable HTTP transport for the stdio bridge.
+ *
+ * The legacy transport splits a call in two: POST to `/api/mcp/messages`, then
+ * wait for the reply to arrive on a separate `/api/mcp/sse` stream. Everything
+ * expensive in `bridge.ts` follows from that split — the `inFlight` map, the
+ * `sent` flag, the 404-means-never-ran reset, the idle watchdog, and the
+ * replay-on-reconnect path — because a POST can succeed while its reply is
+ * lost, and the bridge cannot tell that from a call still running.
+ *
+ * Streamable HTTP (MCP 2025-06) collapses that: one endpoint, and the reply
+ * comes back on the same request. The relayer has served it since
+ * `mcp_proxy.rs:751` ("Single endpoint that supersedes the SSE+POST split");
+ * only the bridge was still on the old transport.
+ *
+ * This module wraps the MCP SDK's own client transport rather than hand-rolling
+ * the protocol: session-id round-tripping, the optional SSE upgrade on a POST
+ * response, and resumption tokens are all spec details that are easy to get
+ * subtly wrong and that the SDK already implements.
+ */
+import { StreamableHTTPClientTransport } from "@modelcontextprotocol/sdk/client/streamableHttp.js";
+import type { JSONRPCMessage } from "@modelcontextprotocol/sdk/types.js";
+
+import type { MemWalCredentials } from "./auth.js";
+import { log } from "./logger.js";
+
+/** Which relayer transport the bridge dials. */
+export type TransportKind = "sse" | "http";
+
+/**
+ * Resolve the transport from `MEMWAL_MCP_TRANSPORT`.
+ *
+ * Defaults to `sse`, the transport every released bridge has used. Streamable
+ * HTTP is opt-in until it has production mileage: this runs on users' machines
+ * against their real memories, so the new path proves itself before it becomes
+ * the one that runs by default.
+ *
+ * An unrecognised value falls back rather than throwing — a typo in a user's
+ * MCP config must not stop their memory from working.
+ */
+export function resolveTransport(raw: string | undefined): TransportKind {
+ const value = raw?.trim().toLowerCase();
+ if (!value) return "sse";
+ if (value === "http" || value === "streamable" || value === "streamable-http") {
+ return "http";
+ }
+ if (value === "sse") return "sse";
+ log.warn("bridge.transport_unrecognized", { value, using: "sse" });
+ return "sse";
+}
+
+/**
+ * The Streamable HTTP endpoint for a relayer base URL.
+ *
+ * `/api/mcp` — the same base the SSE transport hangs `/api/mcp/sse` and
+ * `/api/mcp/messages` off, minus the split.
+ */
+export function streamableUrl(relayerUrl: string): string {
+ return `${relayerUrl.replace(/\/+$/, "")}/api/mcp`;
+}
+
+/**
+ * A live relayer session. Deliberately the same shape the SSE handshake
+ * returns, so `runBridge` can hold either without branching on transport
+ * everywhere it forwards a message.
+ */
+export interface RelaySession {
+ /** Endpoint this session talks to. Logging only. */
+ postUrl: string;
+ /**
+ * Forward one JSON-RPC message. Resolves with an HTTP-ish status the
+ * caller can act on: 200 for accepted, 404 when the relayer says the
+ * session does not exist (the message provably did not run, so the
+ * caller may retry it without risking a duplicate write).
+ */
+ send(
+ msg: JSONRPCMessage,
+ /** Unused here — headers are bound when the session opens. Present so
+ * this matches the SSE handshake's `send`, which signs per POST. */
+ creds?: MemWalCredentials,
+ extra?: Record,
+ ): Promise;
+ /** Incoming messages from the relayer. */
+ iter: AsyncIterator;
+ /** Tear the session down. */
+ abort: () => void;
+}
+
+/** HTTP status carried on the SDK's transport error, when it has one. */
+function statusOf(err: unknown): number {
+ const code = (err as { code?: unknown } | null)?.code;
+ return typeof code === "number" ? code : 0;
+}
+
+export async function openStreamableSession(
+ relayerUrl: string,
+ creds: MemWalCredentials,
+ extraHeaders: Record = {},
+): Promise {
+ const url = streamableUrl(relayerUrl);
+
+ // Queue + waiter rather than an event emitter, so a message that arrives
+ // before `runBridge` pulls from the iterator is buffered instead of
+ // dropped. The SSE path does the same thing for the same reason.
+ const queue: JSONRPCMessage[] = [];
+ let wake: (() => void) | null = null;
+ let closed = false;
+ const push = (msg: JSONRPCMessage) => {
+ queue.push(msg);
+ const resume = wake;
+ wake = null;
+ resume?.();
+ };
+ const finish = () => {
+ closed = true;
+ const resume = wake;
+ wake = null;
+ resume?.();
+ };
+
+ const transport = new StreamableHTTPClientTransport(new URL(url), {
+ requestInit: {
+ headers: {
+ authorization: `Bearer ${creds.delegatePrivateKey}`,
+ "x-memwal-account-id": creds.accountId,
+ ...extraHeaders,
+ },
+ },
+ });
+
+ transport.onmessage = push;
+ transport.onclose = finish;
+ transport.onerror = (err) => {
+ log.warn("bridge.streamable_error", { err: String(err) });
+ // Not `finish()` — the SDK transport reconnects its own stream, and
+ // tearing the session down on a transient read error is what the SSE
+ // watchdog did wrong.
+ };
+
+ await transport.start();
+
+ const iter: AsyncIterator = {
+ async next() {
+ while (queue.length === 0) {
+ if (closed) return { value: undefined as never, done: true };
+ await new Promise((resolve) => (wake = resolve));
+ }
+ return { value: queue.shift()!, done: false };
+ },
+ };
+
+ return {
+ postUrl: url,
+ async send(msg) {
+ try {
+ await transport.send(msg);
+ return 200;
+ } catch (err) {
+ const status = statusOf(err);
+ log.warn("bridge.streamable_send_failed", {
+ status,
+ err: String(err),
+ });
+ // Surface the status rather than throwing: `postIfCurrent`
+ // routes on it, and a 404 specifically means the message was
+ // discarded rather than run.
+ return status;
+ }
+ },
+ iter,
+ abort: () => {
+ void transport.close().catch(() => {
+ /* already gone */
+ });
+ finish();
+ },
+ };
+}
diff --git a/packages/mcp/test/coldstart-init.test.mjs b/packages/mcp/test/coldstart-init.test.mjs
index ef400c5bc..46303cb23 100644
--- a/packages/mcp/test/coldstart-init.test.mjs
+++ b/packages/mcp/test/coldstart-init.test.mjs
@@ -44,6 +44,7 @@ const SSE_DELAY_MS = 3_000;
const UPSTREAM_TOOL_NAMES = [
"memwal_remember",
"memwal_remember_bulk",
+ "memwal_remember_status",
"memwal_recall",
"memwal_analyze",
"memwal_restore",
diff --git a/packages/mcp/test/streamable-transport.test.mjs b/packages/mcp/test/streamable-transport.test.mjs
new file mode 100644
index 000000000..03838106e
--- /dev/null
+++ b/packages/mcp/test/streamable-transport.test.mjs
@@ -0,0 +1,53 @@
+import assert from "node:assert/strict";
+import test from "node:test";
+
+import { resolveTransport, streamableUrl } from "../dist/streamable.js";
+
+/**
+ * Transport selection and endpoint derivation only — the parts that hold
+ * without a socket. The session itself needs a live relayer, so it is covered
+ * by the live suite rather than here.
+ */
+
+test("the default transport stays SSE", () => {
+ // Every released bridge dials SSE. Streamable HTTP is opt-in until it has
+ // production mileage, so an unset variable must not move users onto it.
+ assert.equal(resolveTransport(undefined), "sse");
+ assert.equal(resolveTransport(""), "sse");
+ assert.equal(resolveTransport(" "), "sse");
+});
+
+test("http is selected by any of its spellings", () => {
+ for (const value of ["http", "streamable", "streamable-http"]) {
+ assert.equal(resolveTransport(value), "http", value);
+ }
+});
+
+test("selection ignores case and surrounding whitespace", () => {
+ assert.equal(resolveTransport(" HTTP "), "http");
+ assert.equal(resolveTransport("SSE"), "sse");
+});
+
+test("an unrecognised value falls back instead of throwing", () => {
+ // A typo in a user's MCP config must not stop their memory from working.
+ assert.equal(resolveTransport("htpp"), "sse");
+ assert.equal(resolveTransport("websocket"), "sse");
+});
+
+test("the streamable endpoint sits on the same base as the SSE pair", () => {
+ assert.equal(
+ streamableUrl("https://relayer.memory.walrus.xyz"),
+ "https://relayer.memory.walrus.xyz/api/mcp"
+ );
+});
+
+test("a trailing slash on the relayer URL does not double up", () => {
+ // Users paste URLs with and without it; a `//api/mcp` path 404s.
+ assert.equal(streamableUrl("https://relayer.example/"), "https://relayer.example/api/mcp");
+ assert.equal(streamableUrl("https://relayer.example///"), "https://relayer.example/api/mcp");
+});
+
+test("a relayer on a port or subpath keeps it", () => {
+ assert.equal(streamableUrl("http://127.0.0.1:8000"), "http://127.0.0.1:8000/api/mcp");
+ assert.equal(streamableUrl("https://host/base"), "https://host/base/api/mcp");
+});
diff --git a/services/server/scripts/mcp/__tests__/remember-fast-return.test.ts b/services/server/scripts/mcp/__tests__/remember-fast-return.test.ts
index e7ddbaf98..53ec0f8a8 100644
--- a/services/server/scripts/mcp/__tests__/remember-fast-return.test.ts
+++ b/services/server/scripts/mcp/__tests__/remember-fast-return.test.ts
@@ -1,154 +1,195 @@
-import test from "node:test";
import assert from "node:assert/strict";
-import { McpServer } from "@modelcontextprotocol/sdk/server/mcp.js";
+import test, { type TestContext } from "node:test";
import { Client } from "@modelcontextprotocol/sdk/client/index.js";
import { InMemoryTransport } from "@modelcontextprotocol/sdk/inMemory.js";
import type { MemWalSession } from "../auth.js";
-import { registerRememberTool } from "../tools/remember.js";
-import { registerRememberStatusTool } from "../tools/remember-status.js";
-
-/** The SDK throws a plain Error carrying a `status` — 504 when the wait ran out
- * with the job still going, 500 when the job itself failed. It ships no named
- * error classes for these, so the stubs must reproduce that exact shape:
- * stubbing a nicely-named class here is what let a constructor-name check pass
- * in tests and fail against the real SDK. */
-const jobStillRunning = (msg = "remember job timed out after 1000ms") =>
- Object.assign(new Error(msg), { status: 504 });
-const jobFailed = (msg: string) =>
- Object.assign(new Error(`remember job failed: ${msg}`), { status: 500 });
-
-function sessionWith(memwal: unknown): MemWalSession {
- return { memwal, accountId: "acct-test" } as unknown as MemWalSession;
+import { createMcpServer } from "../server.js";
+import { REMEMBER_WAIT_MS, parseWaitBudget } from "../tools/remember-wait.js";
+
+/**
+ * `memwal_remember` no longer blocks until a Walrus write reaches `done` —
+ * that cost 30–75s per call against production. It returns at accept (~1.1s)
+ * and hands back a job_id.
+ *
+ * The risk that buys is an agent reading "accepted" as "saved", so these tests
+ * pin what keeps it honest: an accepted result must never read as success or
+ * carry a blob_id, and `memwal_remember_status` — now the only thing that can
+ * observe a job failing after acceptance — must report that failure as an
+ * error rather than as a write still in flight.
+ */
+
+interface FakeJob {
+ /** Calls to waitForRememberJob before the job reports done. */
+ pendingPolls: number;
+ blobId?: string;
+ /** When set, the job fails with this message instead of completing. */
+ failWith?: string;
}
-async function callTool(
- session: MemWalSession,
- register: (s: McpServer, sess: MemWalSession) => void,
- name: string,
- args: Record,
- t: { after: (fn: () => Promise) => void }
-) {
+function sessionWith(job: FakeJob, calls: string[] = []): MemWalSession {
+ let polls = 0;
+ return {
+ oauthScope: "memwal:read memwal:write",
+ memwal: {
+ async rememberAsync(text: string, namespace?: string) {
+ calls.push(`rememberAsync:${text}:${namespace ?? ""}`);
+ return { job_id: "job-1", status: "running" };
+ },
+ async waitForRememberJob(jobId: string) {
+ calls.push(`wait:${jobId}`);
+ if (job.failWith) {
+ throw Object.assign(
+ new Error(`remember job failed: ${job.failWith}`),
+ { status: 500, jobId }
+ );
+ }
+ if (polls++ < job.pendingPolls) {
+ throw Object.assign(
+ new Error(`remember job timed out (job_id=${jobId})`),
+ { status: 504, jobId }
+ );
+ }
+ return {
+ id: jobId,
+ job_id: jobId,
+ blob_id: job.blobId ?? "blob-abc",
+ owner: "0xowner",
+ namespace: "default",
+ };
+ },
+ async getRememberStatus(jobId: string) {
+ calls.push(`status:${jobId}`);
+ if (job.failWith) {
+ return { job_id: jobId, status: "failed", error: job.failWith };
+ }
+ return polls++ < job.pendingPolls
+ ? { job_id: jobId, status: "running" }
+ : {
+ job_id: jobId,
+ status: "done",
+ blob_id: job.blobId ?? "blob-abc",
+ namespace: "default",
+ };
+ },
+ },
+ } as unknown as MemWalSession;
+}
+
+async function clientFor(session: MemWalSession, t: TestContext): Promise {
const [clientTransport, serverTransport] = InMemoryTransport.createLinkedPair();
- const server = new McpServer({ name: "test", version: "1.0.0" });
- register(server, session);
- const client = new Client({ name: "remember-test", version: "1.0.0" });
+ const server = createMcpServer(session);
+ const client = new Client({ name: "remember-fast-return-test", version: "1.0.0" });
t.after(async () => {
await client.close();
await server.close();
});
await server.connect(serverTransport);
await client.connect(clientTransport);
- return (await client.callTool({ name, arguments: args })) as {
- content: Array<{ type: string; text: string }>;
- isError?: boolean;
- };
+ return client;
}
-test("memwal_remember returns the blob_id when the write finishes inside the wait", async (t) => {
- const session = sessionWith({
- rememberAsync: async () => ({ job_id: "job-1", status: "running" }),
- waitForRememberJob: async () => ({
- blob_id: "blob-abc",
- namespace: "ns-1",
- }),
- });
-
- const res = await callTool(session, registerRememberTool, "memwal_remember", { text: "a fact" }, t);
+function textOf(result: unknown): string {
+ return (result as { content: Array<{ text: string }> }).content
+ .map((c) => c.text)
+ .join("\n");
+}
- assert.equal(res.isError ?? false, false);
- assert.match(res.content[0].text, /^Saved to Walrus Memory\./);
- assert.match(res.content[0].text, /blob_id=blob-abc/);
+test("the default budget is zero — memwal_remember returns at accept", () => {
+ // A non-zero budget below the real completion time is the worst case:
+ // the caller pays the wait and still gets no guarantee.
+ assert.equal(parseWaitBudget(undefined), 0);
+ assert.equal(parseWaitBudget(""), 0);
+ assert.equal(REMEMBER_WAIT_MS, 0);
});
-test("memwal_remember hands back a job_id instead of blocking when the write runs long", async (t) => {
- let waited = false;
- const session = sessionWith({
- rememberAsync: async () => ({ job_id: "job-2", status: "running" }),
- waitForRememberJob: async () => {
- waited = true;
- throw jobStillRunning();
- },
- });
+test("a typo'd budget falls back to the default instead of picking one nobody asked for", () => {
+ // Number("10s") is NaN, and every NaN comparison is false — an unvalidated
+ // parse would sail past a range check.
+ assert.equal(parseWaitBudget("10s"), 0);
+ assert.equal(parseWaitBudget("abc"), 0);
+ assert.equal(parseWaitBudget("-1"), 0);
+});
- const res = await callTool(session, registerRememberTool, "memwal_remember", { text: "a slow fact" }, t);
-
- assert.ok(waited, "should have waited before giving up on the blob_id");
- // Still in flight is not an error — the write was accepted and is running.
- assert.equal(res.isError ?? false, false);
- assert.match(res.content[0].text, /still writing/i);
- assert.match(res.content[0].text, /job_id=job-2/);
- // The agent must not report this as saved, and must be told how to confirm.
- assert.match(res.content[0].text, /Do not tell the user it is saved/);
- assert.match(res.content[0].text, /memwal_remember_status/);
- assert.doesNotMatch(res.content[0].text, /blob_id=/);
+test("a budget past the ceiling is clamped, not honoured", () => {
+ assert.equal(parseWaitBudget("90000"), 90_000);
+ assert.equal(parseWaitBudget("600000"), 90_000);
});
-test("memwal_remember still surfaces a genuinely failed job as an error", async (t) => {
- const session = sessionWith({
- rememberAsync: async () => ({ job_id: "job-3", status: "running" }),
- waitForRememberJob: async () => {
- throw jobFailed("Memory encryption backend is unavailable");
- },
+test("memwal_remember returns at accept and never reads as saved", async (t) => {
+ const calls: string[] = [];
+ const client = await clientFor(sessionWith({ pendingPolls: 0 }, calls), t);
+ const result = await client.callTool({
+ name: "memwal_remember",
+ arguments: { text: "a durable fact" },
});
- const res = await callTool(session, registerRememberTool, "memwal_remember", { text: "doomed" }, t);
-
- assert.equal(res.isError, true);
- assert.match(res.content[0].text, /encryption backend is unavailable/);
+ // Accepted is the expected path, not a failure — the job is a durable row
+ // the relayer drives, and keeps going after the tool returns.
+ assert.equal(result.isError, undefined);
+
+ const text = textOf(result);
+ assert.match(text, /ACCEPTED, NOT YET SAVED/);
+ assert.match(text, /job_id=job-1/);
+ assert.match(text, /memwal_remember_status/);
+ // The whole point of the wording: an agent must not be able to read this
+ // as a completed write.
+ assert.doesNotMatch(text, /Saved to Walrus Memory/);
+ assert.ok(!text.includes("blob_id="), "an accepted result must not carry a blob_id");
+
+ // A zero budget must not poll at all — the job was accepted, and waiting
+ // zero milliseconds for it is not a thing worth a round trip.
+ assert.deepEqual(calls, ["rememberAsync:a durable fact:"]);
});
-test("memwal_remember_status reports the blob_id once the job lands", async (t) => {
- const session = sessionWith({
- waitForRememberJob: async () => ({ blob_id: "blob-xyz", namespace: "ns-2" }),
+test("the accepted message does not claim a duration it never waited", async (t) => {
+ const client = await clientFor(sessionWith({ pendingPolls: 0 }), t);
+ const result = await client.callTool({
+ name: "memwal_remember",
+ arguments: { text: "a fact" },
});
-
- const res = await callTool(session, registerRememberStatusTool, "memwal_remember_status", { job_id: "job-4" }, t);
-
- assert.equal(res.isError ?? false, false);
- assert.match(res.content[0].text, /^Stored\./);
- assert.match(res.content[0].text, /blob_id=blob-xyz/);
+ assert.doesNotMatch(textOf(result), /after 0\.0s/);
});
-test("memwal_remember_status reports a failed job as an error so the fact is not assumed saved", async (t) => {
- const session = sessionWith({
- waitForRememberJob: async () => {
- throw jobFailed("walrus upload failed");
- },
+test("memwal_remember_status reports the blob_id once the job lands", async (t) => {
+ const client = await clientFor(sessionWith({ pendingPolls: 0, blobId: "blob-late" }), t);
+ const result = await client.callTool({
+ name: "memwal_remember_status",
+ arguments: { job_id: "job-1" },
});
- const res = await callTool(session, registerRememberStatusTool, "memwal_remember_status", { job_id: "job-5" }, t);
-
- assert.equal(res.isError, true);
- assert.match(res.content[0].text, /Walrus Memory job failed/);
+ assert.equal(result.isError, undefined);
+ const text = textOf(result);
+ assert.match(text, /Saved to Walrus Memory/);
+ assert.match(text, /blob_id=blob-late/);
});
-test("memwal_remember_status says still writing while the job is running", async (t) => {
- const session = sessionWith({
- waitForRememberJob: async () => {
- throw jobStillRunning();
- },
+test("memwal_remember_status with waitMs=0 reads state without waiting", async (t) => {
+ const calls: string[] = [];
+ const client = await clientFor(sessionWith({ pendingPolls: 5 }, calls), t);
+ const result = await client.callTool({
+ name: "memwal_remember_status",
+ arguments: { job_id: "job-1", waitMs: 0 },
});
- const res = await callTool(session, registerRememberStatusTool, "memwal_remember_status", { job_id: "job-6", wait_seconds: 1 }, t);
-
- assert.equal(res.isError ?? false, false);
- assert.match(res.content[0].text, /Still writing/);
- assert.match(res.content[0].text, /job_id=job-6/);
+ assert.equal(result.isError, undefined);
+ assert.match(textOf(result), /STILL UPLOADING/);
+ // A zero budget must be a single GET — waitForRememberJob sleeps before
+ // its first poll, so routing it there would report "still running"
+ // without ever asking the relayer.
+ assert.deepEqual(calls, ["status:job-1"]);
});
-test("memwal_remember_status never asks the SDK for a deadline shorter than one poll", async (t) => {
- let seenTimeout: number | undefined;
- const session = sessionWith({
- waitForRememberJob: async (_id: string, opts: { timeoutMs: number }) => {
- seenTimeout = opts.timeoutMs;
- throw jobStillRunning();
- },
+test("memwal_remember_status surfaces a failed job as an error", async (t) => {
+ const client = await clientFor(
+ sessionWith({ pendingPolls: 0, failWith: "walrus upload rejected" }, []),
+ t
+ );
+ const result = await client.callTool({
+ name: "memwal_remember_status",
+ arguments: { job_id: "job-1", waitMs: 0 },
});
- await callTool(session, registerRememberStatusTool, "memwal_remember_status", { job_id: "job-7", wait_seconds: 0 }, t);
-
- assert.ok(
- seenTimeout !== undefined && seenTimeout >= 250,
- `wait_seconds=0 must still allow one status read, got ${seenTimeout}`
- );
+ assert.equal(result.isError, true);
+ assert.match(textOf(result), /Walrus Memory job failed/);
+ assert.match(textOf(result), /walrus upload rejected/);
});
diff --git a/services/server/scripts/mcp/__tests__/tool-annotations.test.ts b/services/server/scripts/mcp/__tests__/tool-annotations.test.ts
index bef7caab6..86c3db199 100644
--- a/services/server/scripts/mcp/__tests__/tool-annotations.test.ts
+++ b/services/server/scripts/mcp/__tests__/tool-annotations.test.ts
@@ -32,14 +32,14 @@ test("tools/list publishes safe titles and behavior annotations for every remote
title: "Remember a Fact",
annotations: { readOnlyHint: false, destructiveHint: false },
},
- memwal_remember_status: {
- title: "Check a Remember Job",
- annotations: { readOnlyHint: true, destructiveHint: false },
- },
memwal_remember_bulk: {
title: "Remember Multiple Facts",
annotations: { readOnlyHint: false, destructiveHint: false },
},
+ memwal_remember_status: {
+ title: "Check a Remember Job",
+ annotations: { readOnlyHint: true, destructiveHint: false },
+ },
memwal_analyze: {
title: "Analyze and Remember",
annotations: { readOnlyHint: false, destructiveHint: true },
diff --git a/services/server/scripts/mcp/__tests__/tool-scope.test.ts b/services/server/scripts/mcp/__tests__/tool-scope.test.ts
index 1251c4425..1cdc272eb 100644
--- a/services/server/scripts/mcp/__tests__/tool-scope.test.ts
+++ b/services/server/scripts/mcp/__tests__/tool-scope.test.ts
@@ -7,8 +7,8 @@ import { createMcpServer } from "../server.js";
const WRITE_TOOLS = [
"memwal_remember",
- "memwal_remember_status",
"memwal_remember_bulk",
+ "memwal_remember_status",
"memwal_analyze",
"memwal_restore",
];
diff --git a/services/server/scripts/mcp/server.ts b/services/server/scripts/mcp/server.ts
index ad1c1c7e3..094094088 100644
--- a/services/server/scripts/mcp/server.ts
+++ b/services/server/scripts/mcp/server.ts
@@ -50,6 +50,13 @@ const INSTRUCTIONS = [
"summary. Skip one-off tasks, the current file or bug, and small talk. Use",
"memwal_remember_bulk when several distinct facts arrived at once.",
"",
+ "A Walrus write is queued, not instant: memwal_remember normally returns a job_id with",
+ "the fact ACCEPTED but NOT YET SAVED, and storing it takes roughly another 30-60s. That",
+ "is the healthy path, not an error. Tell the user the fact is being saved rather than",
+ "that it is saved, and do not re-send it — that queues a duplicate. A job can still fail",
+ "after acceptance, so when it matters that a fact landed, resolve the job_id with",
+ "memwal_remember_status; only the blob_id it returns means the fact is stored.",
+ "",
"RECOVER: if memwal_recall unexpectedly returns nothing for a namespace that has been used",
"before, call memwal_restore to rebuild the index from Walrus.",
"",
diff --git a/services/server/scripts/mcp/tools/annotations.ts b/services/server/scripts/mcp/tools/annotations.ts
index eda6deefb..71465d66c 100644
--- a/services/server/scripts/mcp/tools/annotations.ts
+++ b/services/server/scripts/mcp/tools/annotations.ts
@@ -11,14 +11,15 @@ export const TOOL_METADATA = {
title: "Remember a Fact",
annotations: { readOnlyHint: false, destructiveHint: false },
},
- memwal_remember_status: {
- title: "Check a Remember Job",
- annotations: { readOnlyHint: true, destructiveHint: false },
- },
memwal_remember_bulk: {
title: "Remember Multiple Facts",
annotations: { readOnlyHint: false, destructiveHint: false },
},
+ memwal_remember_status: {
+ title: "Check a Remember Job",
+ // Reads the state of a write already in flight; starts no new work.
+ annotations: { readOnlyHint: true, destructiveHint: false },
+ },
memwal_analyze: {
title: "Analyze and Remember",
// Context recall may remove stale vector rows for blobs confirmed absent.
diff --git a/services/server/scripts/mcp/tools/remember-status.ts b/services/server/scripts/mcp/tools/remember-status.ts
index af38f2794..8202b15e4 100644
--- a/services/server/scripts/mcp/tools/remember-status.ts
+++ b/services/server/scripts/mcp/tools/remember-status.ts
@@ -3,54 +3,47 @@ import type { McpServer } from "@modelcontextprotocol/sdk/server/mcp.js";
import type { MemWalSession } from "../auth.js";
import { TOOL_METADATA } from "./annotations.js";
import { wrapTool, walruscanBlobUrl } from "./util.js";
+import {
+ REMEMBER_POLL_INTERVAL_MS,
+ isStillRunning,
+ nameJobError,
+} from "./remember-wait.js";
-/** `waitForRememberJob` rejects with 504 when its deadline passes with the job
- * still running, and 500 when the job itself failed. Neither is a distinct
- * error class in any shipped SDK line, so the status code is the only stable
- * discriminator. */
-const JOB_STILL_RUNNING_STATUS = 504;
-const JOB_FAILED_STATUS = 500;
-
-/** Default settle window. Short: this tool answers "did it land yet", and an
- * agent that wants to keep waiting can just call it again. */
-const DEFAULT_WAIT_SECONDS = 2;
-const MAX_WAIT_SECONDS = 30;
+/**
+ * Ceiling on a single status wait. Past this an MCP client is more likely to
+ * time out the call than the job is to finish, and the caller can simply ask
+ * again — the job_id stays valid.
+ */
+const MAX_STATUS_WAIT_MS = 60_000;
-const REMEMBER_STATUS_INPUT = {
+const STATUS_INPUT = {
job_id: z
.string()
.min(1)
- .describe(
- "The job_id returned by memwal_remember when the write was still in flight."
- ),
- wait_seconds: z
+ .describe("The job_id returned by memwal_remember when the write had not landed yet."),
+ waitMs: z
.number()
.int()
.min(0)
- .max(MAX_WAIT_SECONDS)
+ .max(MAX_STATUS_WAIT_MS)
.optional()
.describe(
- `How long to wait for the job to settle before answering (default ${DEFAULT_WAIT_SECONDS}s, max ${MAX_WAIT_SECONDS}s). Use 0 for an immediate answer.`
+ "How long to wait for the job to finish, in milliseconds (0-60000, default 10000). Pass 0 to read the current state without waiting."
),
} as const;
+/** Wait applied when the caller does not choose one. */
+const DEFAULT_STATUS_WAIT_MS = 10_000;
+
/**
- * memwal_remember_status — resolve a `memwal_remember` job that had not
- * finished when the tool returned.
- *
- * This is the other half of the fast-return path in `remember.ts`. That tool
- * stops waiting once the write is durably accepted, which means a job can
- * still fail afterwards — a Walrus upload or SEAL encrypt outage lands the job
- * in `failed`, long after the agent was told it was accepted. Without a way to
- * ask after the fact, returning early would turn a slow write into a silently
- * lost one.
+ * memwal_remember_status — resolve a remember job that `memwal_remember`
+ * handed back as still in flight.
*
- * `waitForRememberJob` already encodes the outcomes an agent has to tell apart,
- * so this tool leans on it rather than re-deriving them: it resolves at `done`,
- * and otherwise rejects with a plain Error carrying a `status` — 500 when the
- * job failed, 504 when only our wait ran out and the job is still going. The
- * SDK ships no dedicated error classes for these, so `status` is what we match
- * on; a constructor-name check silently never fires.
+ * This is the other half of the bounded wait: `memwal_remember` refuses to
+ * claim a fact is saved when it isn't, so something has to be able to say
+ * whether it landed. Three outcomes, kept distinct because an agent acts
+ * differently on each — saved (blob_id), still running (ask again), failed
+ * (the fact is NOT stored and must be re-sent).
*/
export function registerRememberStatusTool(
server: McpServer,
@@ -61,62 +54,86 @@ export function registerRememberStatusTool(
{
...TOOL_METADATA.memwal_remember_status,
description:
- "Check whether an in-flight memwal_remember job finished. Call this with the job_id from a memwal_remember result that came back still writing, when you need to confirm the fact was actually stored. Returns the blob_id once stored, an error if the write failed, or tells you it is still running.",
- inputSchema: REMEMBER_STATUS_INPUT,
+ "Check whether an in-flight memwal_remember write has landed. Call this with the job_id memwal_remember returned when it reported the fact was NOT saved yet. Returns the blob_id once stored, reports that it is still uploading (call again), or reports that it failed — in which case the fact was never stored and you should send it again with memwal_remember.",
+ inputSchema: STATUS_INPUT,
},
- wrapTool<{ job_id: string; wait_seconds?: number }>(
+ wrapTool<{ job_id: string; waitMs?: number }>(
session,
"memwal_remember_status",
- async ({ job_id, wait_seconds }) => {
- const waitSeconds = Math.min(
- wait_seconds ?? DEFAULT_WAIT_SECONDS,
- MAX_WAIT_SECONDS
- );
- try {
- const result = await session.memwal.waitForRememberJob(
- job_id,
- // Floor at one poll: a zero deadline would expire
- // before the first status read and report "still
- // running" without ever having asked.
- { timeoutMs: Math.max(250, waitSeconds * 1000) }
- );
- return {
- content: [
- {
- type: "text",
- text: `Stored. job_id=${job_id} blob_id=${result.blob_id} namespace=${result.namespace}\nExplorer: ${walruscanBlobUrl(result.blob_id)}`,
- },
- ],
- };
- } catch (err: any) {
- // Still running is the expected answer here, not a failure.
- if (err?.status === JOB_STILL_RUNNING_STATUS) {
- return {
- content: [
- {
- type: "text",
- text: `Still writing after ${waitSeconds}s. job_id=${job_id}. Check again shortly; do not tell the user it is saved yet.`,
- },
- ],
- };
+ async ({ job_id, waitMs }) => {
+ const budget = waitMs ?? DEFAULT_STATUS_WAIT_MS;
+
+ // A zero budget means "read the current state", which is a
+ // single GET. waitForRememberJob cannot express that: it
+ // sleeps before its first poll, so a 0ms deadline would
+ // return "still running" without ever asking the relayer.
+ if (budget === 0) {
+ const status = await session.memwal.getRememberStatus(job_id);
+ if (status.status === "done") {
+ return saved(status.blob_id ?? "", status.namespace);
}
- // A terminal failure is the case this tool exists for, so
- // it is named plainly rather than left to the generic
- // "Tool error" prefix: the fact was NOT stored.
- if (err?.status === JOB_FAILED_STATUS) {
- return {
- content: [
- {
- type: "text",
- text: `Walrus Memory job failed: job_id=${job_id} was not stored — ${err?.message ?? "unknown error"}. The fact is NOT in memory; save it again.`,
- },
- ],
- isError: true,
- };
+ if (status.status === "failed") {
+ throw nameJobError(
+ Object.assign(
+ new Error(
+ `remember job failed: ${status.error ?? "unknown error"}`
+ ),
+ { status: 500, jobId: job_id }
+ )
+ );
}
- throw err;
+ if (status.status === "not_found") {
+ throw nameJobError(
+ Object.assign(
+ new Error(`remember job not found: ${job_id}`),
+ { status: 404, jobId: job_id }
+ )
+ );
+ }
+ return stillRunning(job_id, status.status);
+ }
+
+ try {
+ const result = await session.memwal.waitForRememberJob(job_id, {
+ timeoutMs: budget,
+ pollIntervalMs: REMEMBER_POLL_INTERVAL_MS,
+ });
+ return saved(result.blob_id, result.namespace);
+ } catch (err) {
+ if (isStillRunning(err)) return stillRunning(job_id);
+ throw nameJobError(err);
}
}
)
);
}
+
+function saved(blobId: string, namespace?: string) {
+ return {
+ content: [
+ {
+ type: "text" as const,
+ text:
+ `Saved to Walrus Memory. blob_id=${blobId}` +
+ (namespace ? ` namespace=${namespace}` : "") +
+ `\nExplorer: ${walruscanBlobUrl(blobId)}`,
+ },
+ ],
+ };
+}
+
+function stillRunning(jobId: string, state?: string) {
+ return {
+ content: [
+ {
+ type: "text" as const,
+ text:
+ `STILL UPLOADING — not saved yet${state ? ` (state: ${state})` : ""}.\n` +
+ `job_id=${jobId}\n` +
+ `The job is still queued or uploading. Call memwal_remember_status again ` +
+ `with this job_id. Do not re-send the fact with memwal_remember — that ` +
+ `queues a duplicate behind this one.`,
+ },
+ ],
+ };
+}
diff --git a/services/server/scripts/mcp/tools/remember-wait.ts b/services/server/scripts/mcp/tools/remember-wait.ts
new file mode 100644
index 000000000..826399d85
--- /dev/null
+++ b/services/server/scripts/mcp/tools/remember-wait.ts
@@ -0,0 +1,138 @@
+/**
+ * Shared wait-budget and job-error handling for the two remember tools.
+ *
+ * `memwal_remember` used to block on `rememberAndWait` until the write reached
+ * `done`. Measured against the production relayer that is 30–75s for a single
+ * short fact, and the agent can do nothing with the wait: the job is durably
+ * accepted ~1s in, and everything after that is upload queue time
+ * (`WALRUS_UPLOAD_PER_WALLET_CONCURRENCY` defaults to 1, so a second write
+ * waits for the first).
+ *
+ * So the tool now waits a bounded budget and then hands the caller a job_id.
+ * It does NOT claim the fact is saved when it isn't — a job can still fail
+ * after acceptance (one observed failure: "Memory encryption backend is
+ * unavailable" 31.6s in, from the SEAL sidecar being unreachable).
+ */
+import { createLogger } from "../logger.js";
+
+const log = createLogger("mcp");
+
+/**
+ * Default wait before `memwal_remember` hands back a job_id.
+ *
+ * Zero — the tool returns at accept (~1.1s measured). A non-zero budget below
+ * the real completion time is the worst of both: against the measured 30–75s
+ * distribution a 10s wait still lands in the pending branch on nearly every
+ * call, so the caller pays the 10s AND gets no guarantee. Either wait long
+ * enough to actually mean it (set this to 90000) or don't wait at all.
+ *
+ * What makes returning at accept safe from disconnects: the job is a row in
+ * `remember_jobs` driven by the relayer (`spawn_persisted_remember_preparation`
+ * in services/server/src/routes/remember.rs), not work held in this process.
+ * Closing the client does not cancel it.
+ *
+ * What it is NOT safe from: a job that fails after acceptance. Nothing here
+ * can catch that — only a later `memwal_remember_status` call can.
+ */
+const DEFAULT_REMEMBER_WAIT_MS = 0;
+
+/**
+ * Hard ceiling on the wait budget. 90s matches the timeout the tool used
+ * while it still blocked to terminal, so an operator can restore the old
+ * always-block behaviour but cannot push the call past what MCP clients
+ * are willing to wait for.
+ */
+const MAX_REMEMBER_WAIT_MS = 90_000;
+
+/**
+ * How long `memwal_remember` waits for the write to land before returning a
+ * job_id instead. `0` returns at accept (~1s).
+ *
+ * Read once — it cannot change mid-process — but validated, because a typo'd
+ * value must not silently pick a wait nobody asked for. `Number.parseInt`
+ * alone accepts "10s" as 10 (a 10ms wait, effectively fire-and-forget) and
+ * yields NaN for "" or "abc", and every NaN comparison is false.
+ */
+export function parseWaitBudget(raw: string | undefined): number {
+ if (raw === undefined || raw.trim() === "") return DEFAULT_REMEMBER_WAIT_MS;
+ const parsed = Number(raw);
+ if (!Number.isFinite(parsed) || parsed < 0) {
+ log.warn("remember.wait_budget_invalid", {
+ value: raw,
+ usingMs: DEFAULT_REMEMBER_WAIT_MS,
+ });
+ return DEFAULT_REMEMBER_WAIT_MS;
+ }
+ if (parsed > MAX_REMEMBER_WAIT_MS) {
+ log.warn("remember.wait_budget_clamped", {
+ value: raw,
+ usingMs: MAX_REMEMBER_WAIT_MS,
+ });
+ return MAX_REMEMBER_WAIT_MS;
+ }
+ return Math.floor(parsed);
+}
+
+export const REMEMBER_WAIT_MS = parseWaitBudget(
+ process.env.MEMWAL_MCP_REMEMBER_WAIT_MS
+);
+
+/**
+ * Poll interval for a bounded wait.
+ *
+ * `waitForRememberJob` sleeps BEFORE its first poll and defaults to 1500ms
+ * with 1.5^attempt backoff, which spends a 10s budget on ~5 polls and can
+ * miss a write that landed at 1.2s. 400ms catches the fast case while
+ * staying far below the relayer's 60 weighted-requests/min delegate-key
+ * limit — the backoff reaches ~6 polls in 10s, not 25.
+ */
+export const REMEMBER_POLL_INTERVAL_MS = 400;
+
+/**
+ * `waitForRememberJob` signals outcome through `status` on a plain Error:
+ * 504 = still running at the deadline, 500 = the job failed, 404 = no such
+ * job. Only 504 is a non-error for us — the write is still in flight and the
+ * caller gets a job_id to resolve later.
+ */
+export function isStillRunning(err: unknown): boolean {
+ return (err as { status?: number } | null)?.status === 504;
+}
+
+/**
+ * Give a job error the name `wrapTool` routes on, so the agent can tell a
+ * failed write from a missing one without parsing the message. The SDK throws
+ * an unnamed Error with a status code; `wrapTool` cannot classify that.
+ */
+export function nameJobError(err: unknown): unknown {
+ if (!(err instanceof Error)) return err;
+ const status = (err as { status?: number }).status;
+ if (status === 500) err.name = "MemWalRememberJobFailed";
+ else if (status === 404) err.name = "MemWalRememberJobNotFound";
+ else if (status === 504) err.name = "MemWalRememberJobTimeout";
+ return err;
+}
+
+/**
+ * The line shown when a write is accepted but has not landed inside the wait
+ * budget. Worded so an agent cannot read it as success: the fact is NOT saved
+ * yet, and there is exactly one way to find out whether it lands.
+ */
+export function pendingMessage(jobId: string, waitedMs: number): string {
+ // The zero-budget path never waited, so saying "has not finished after
+ // 0.0s" would misdescribe it — and that is the default path, the one an
+ // agent reads on nearly every call.
+ const opening =
+ waitedMs === 0
+ ? "ACCEPTED, NOT YET SAVED — the relayer has durably queued this write."
+ : `NOT SAVED YET — the write was accepted but has not finished after ${(waitedMs / 1000).toFixed(1)}s.`;
+
+ return (
+ `${opening}\n` +
+ `job_id=${jobId}\n` +
+ `Walrus uploads queue, so storing typically takes another 30-60s. Do NOT tell the ` +
+ `user the fact is stored — say it is being saved. Call memwal_remember_status with ` +
+ `this job_id to get the blob_id once it lands, or to learn that it failed; a job ` +
+ `CAN fail after acceptance, and this is the only way to find out. Do not re-send ` +
+ `the same fact with memwal_remember — that queues a second copy behind this one.`
+ );
+}
diff --git a/services/server/scripts/mcp/tools/remember.ts b/services/server/scripts/mcp/tools/remember.ts
index 34632bddb..9e6ba952e 100644
--- a/services/server/scripts/mcp/tools/remember.ts
+++ b/services/server/scripts/mcp/tools/remember.ts
@@ -3,9 +3,13 @@ import type { McpServer } from "@modelcontextprotocol/sdk/server/mcp.js";
import type { MemWalSession } from "../auth.js";
import { TOOL_METADATA } from "./annotations.js";
import { wrapTool, walruscanBlobUrl } from "./util.js";
-import { createLogger } from "../logger.js";
-
-const log = createLogger("mcp");
+import {
+ REMEMBER_WAIT_MS,
+ REMEMBER_POLL_INTERVAL_MS,
+ isStillRunning,
+ nameJobError,
+ pendingMessage,
+} from "./remember-wait.js";
const REMEMBER_INPUT = {
text: z
@@ -22,59 +26,14 @@ const REMEMBER_INPUT = {
),
} as const;
-/**
- * How long `memwal_remember` waits for the write to finish before handing the
- * agent a job id instead.
- *
- * The write itself is a background job (`pending` → `running` → `uploaded` →
- * `done`). Blocking to `done` used to be the whole tool, which made a healthy
- * save cost as much as the slowest step in that pipeline — measured at 30-75s
- * against production, dominated by the Walrus upload phase. Almost none of
- * that time tells the agent anything it can act on: the job is already durably
- * accepted a second or so in.
- *
- * So this is a deadline for the *answer*, not for the write. The job keeps
- * running past it either way; the only question is whether the tool stays
- * parked. `memwal_remember_status` resolves the ones that run long.
- *
- * Set to `0` to always return as soon as the job is accepted. Set it to the
- * old `90000` to restore the previous always-block behaviour.
- */
-const DEFAULT_REMEMBER_WAIT_MS = 10_000;
-
-/** `waitForRememberJob` rejects with this HTTP status when the deadline passes
- * without the job settling. The job itself is still running; only the wait
- * ended. A genuinely failed job rejects with 500 instead. */
-const JOB_STILL_RUNNING_STATUS = 504;
-
-/** Hard ceiling — the relayer's own remember deadline. Waiting past it cannot
- * observe anything the job has not already settled. */
-const MAX_REMEMBER_WAIT_MS = 90_000;
-
-const REMEMBER_WAIT_MS = (() => {
- const raw = process.env.MEMWAL_MCP_REMEMBER_WAIT_MS;
- if (raw === undefined || raw.trim() === "") return DEFAULT_REMEMBER_WAIT_MS;
- const parsed = Number(raw);
- // Zero is meaningful here (return at accept), so it is allowed while every
- // other unusable value falls back rather than silently disabling the wait.
- if (!Number.isFinite(parsed) || parsed < 0) {
- log.warn("remember.wait_ms_invalid", {
- value: raw,
- usingMs: DEFAULT_REMEMBER_WAIT_MS,
- });
- return DEFAULT_REMEMBER_WAIT_MS;
- }
- return Math.min(parsed, MAX_REMEMBER_WAIT_MS);
-})();
-
/**
* memwal_remember — persist a durable fact to MemWal.
*
- * Returns once the write is durably accepted by the relayer, waiting up to
- * `MEMWAL_MCP_REMEMBER_WAIT_MS` for it to finish so the common fast case still
- * comes back with a `blob_id`. A job that outlives that window is reported as
- * still writing, with its `job_id`, and is resolved by
- * `memwal_remember_status` — the job is NOT abandoned.
+ * Returns as soon as the blob is written end-to-end (embed → SEAL encrypt →
+ * Walrus upload → on-chain) when that happens inside `REMEMBER_WAIT_MS`.
+ * Otherwise it returns the job_id and says plainly that the fact is not saved
+ * yet — see `remember-wait.ts` for why the old always-block behaviour cost
+ * 30–75s per call.
*
* Call this PROACTIVELY whenever the user reveals a durable fact about
* themselves or the project (preference, decision, constraint, correction,
@@ -91,41 +50,34 @@ export function registerRememberTool(
{
...TOOL_METADATA.memwal_remember,
description:
- "Save a durable fact about the user or project to their Walrus Memory. Call this PROACTIVELY whenever the user states a preference, decision, constraint, correction, identity detail, or recurring workflow — even if they did not say 'remember this'. Skip one-off tasks, the current file or bug, and small talk. Pass the full statement; do not summarize. To save several facts at once, use memwal_remember_bulk instead. If the result says the write is still in flight, it carries a job_id — confirm it later with memwal_remember_status rather than telling the user it is saved.",
+ "Save a durable fact about the user or project to their Walrus Memory. Call this PROACTIVELY whenever the user states a preference, decision, constraint, correction, identity detail, or recurring workflow — even if they did not say 'remember this'. Skip one-off tasks, the current file or bug, and small talk. Pass the full statement; do not summarize. To save several facts at once, use memwal_remember_bulk instead. Walrus writes queue, so this may return a job_id with the fact NOT yet saved — in that case say so rather than claiming it is stored, and resolve it with memwal_remember_status.",
inputSchema: REMEMBER_INPUT,
},
wrapTool<{ text: string; namespace?: string }>(session, "memwal_remember", async ({ text, namespace }) => {
+ // Two steps rather than `rememberAndWait`, because the accept and
+ // the wait need separate budgets: acceptance is the part that
+ // must succeed, the wait is a courtesy we cut short.
const accepted = await session.memwal.rememberAsync(text, namespace);
- const stillWriting = () => {
- log.info("remember.returned_in_flight", {
- jobId: accepted.job_id,
- waitedMs: REMEMBER_WAIT_MS,
- accountId: session.accountId ?? null,
- });
- return {
- content: [
- {
- type: "text" as const,
- text:
- `Accepted by Walrus Memory and still writing. job_id=${accepted.job_id}` +
- `${namespace ? ` namespace=${namespace}` : ""}\n` +
- (REMEMBER_WAIT_MS === 0
- ? "The tool did not wait for the write, so there is no blob_id yet. "
- : `The write did not finish within ${Math.round(REMEMBER_WAIT_MS / 1000)}s, so there is no blob_id yet. `) +
- `Do not tell the user it is saved — confirm with memwal_remember_status(job_id="${accepted.job_id}").`,
- },
- ],
- };
- };
+ const pending = (waitedMs: number) => ({
+ content: [
+ {
+ type: "text" as const,
+ text: pendingMessage(accepted.job_id, waitedMs),
+ },
+ ],
+ });
- if (REMEMBER_WAIT_MS === 0) return stillWriting();
+ // A zero budget is the documented fire-and-accept mode. Skip the
+ // wait entirely instead of entering a loop that cannot poll.
+ if (REMEMBER_WAIT_MS === 0) return pending(0);
+ const startedAt = Date.now();
try {
- const result = await session.memwal.waitForRememberJob(
- accepted.job_id,
- { timeoutMs: REMEMBER_WAIT_MS }
- );
+ const result = await session.memwal.waitForRememberJob(accepted.job_id, {
+ timeoutMs: REMEMBER_WAIT_MS,
+ pollIntervalMs: REMEMBER_POLL_INTERVAL_MS,
+ });
return {
content: [
{
@@ -134,17 +86,11 @@ export function registerRememberTool(
},
],
};
- } catch (err: any) {
- // Only our own wait expiring is a non-failure. The SDK signals
- // that as a plain Error carrying `status: 504` (a failed job
- // carries 500) — it has no dedicated error class, in either the
- // pinned 0.0.x or the current 0.1.x line, so matching on a
- // constructor name would never fire and every slow write would
- // surface as an error.
- if (err?.status === JOB_STILL_RUNNING_STATUS) {
- return stillWriting();
- }
- throw err;
+ } catch (err) {
+ // Still running at the deadline is the expected path, not a
+ // failure — the job is durably accepted and keeps going.
+ if (isStillRunning(err)) return pending(Date.now() - startedAt);
+ throw nameJobError(err);
}
})
);
diff --git a/services/server/scripts/mcp/tools/util.ts b/services/server/scripts/mcp/tools/util.ts
index 7068c7fa2..6622a1d66 100644
--- a/services/server/scripts/mcp/tools/util.ts
+++ b/services/server/scripts/mcp/tools/util.ts
@@ -165,13 +165,19 @@ export function wrapTool(
// Name the failure in the structured line too. Without this the log
// says a call failed and the operator still has to go find the
// separate console.error below to learn how.
+ // Prefer an explicitly set `name` over the constructor's. The SDK
+ // signals a job outcome with a status code on a plain Error, whose
+ // constructor is always `Error` — routing on that alone left the
+ // switch below unreachable for exactly the cases it names.
+ const name = err?.name && err.name !== "Error"
+ ? err.name
+ : err?.constructor?.name ?? "Error";
log.warn("tool.failed", {
...outcomeFields(),
- errName: err?.constructor?.name ?? "Error",
+ errName: name,
errMessage: err?.message ?? String(err),
causeCode: err?.cause?.code ?? null,
});
- const name = err?.constructor?.name ?? "Error";
const msg = err?.message ?? String(err);
const cause = err?.cause;
const causeStr = cause
diff --git a/services/server/scripts/package-lock.json b/services/server/scripts/package-lock.json
index f68a66a28..1df76c793 100644
--- a/services/server/scripts/package-lock.json
+++ b/services/server/scripts/package-lock.json
@@ -9,7 +9,7 @@
"version": "0.1.0",
"dependencies": {
"@modelcontextprotocol/sdk": "1.29.0",
- "@mysten-incubation/memwal": "0.0.3",
+ "@mysten-incubation/memwal": "0.0.4",
"@mysten/seal": "1.1.0",
"@mysten/sui": "2.17.0",
"@mysten/walrus": "1.1.7",
@@ -621,9 +621,9 @@
}
},
"node_modules/@mysten-incubation/memwal": {
- "version": "0.0.3",
- "resolved": "https://registry.npmjs.org/@mysten-incubation/memwal/-/memwal-0.0.3.tgz",
- "integrity": "sha512-TafWL5MPEOPXmQstap5YP/5j6tm6x+ZJXTnEG6KKoIfyE7krKZ8lIZCsa6EcS4hrzY1rDtr+El5RxjmKlGhCig==",
+ "version": "0.0.4",
+ "resolved": "https://registry.npmjs.org/@mysten-incubation/memwal/-/memwal-0.0.4.tgz",
+ "integrity": "sha512-HBvq2ipq/gJoAWrn7oNHVCaMypTE7/9AtFwk8YSqd9njwpYqssGZ/18B3S8orzZdDsQmRSbgRDcnEZ8xwozaKg==",
"license": "Apache-2.0",
"dependencies": {
"@noble/ed25519": "^2.3.0",
diff --git a/services/server/scripts/package.json b/services/server/scripts/package.json
index 8cb3f4980..9d4854f61 100644
--- a/services/server/scripts/package.json
+++ b/services/server/scripts/package.json
@@ -10,7 +10,7 @@
},
"dependencies": {
"@modelcontextprotocol/sdk": "1.29.0",
- "@mysten-incubation/memwal": "0.0.3",
+ "@mysten-incubation/memwal": "0.0.4",
"@mysten/seal": "1.1.0",
"@mysten/sui": "2.17.0",
"@mysten/walrus": "1.1.7",
From af92c910aeecc923dd63f46bcdb928722eb7ec0b Mon Sep 17 00:00:00 2001
From: Harry Phan
Date: Tue, 15 Sep 2026 21:33:59 +0700
Subject: [PATCH 004/132] perf(relayer): pick the least-loaded upload wallet,
not the next one
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit
Round-robin `next_index()` hands out the next wallet in sequence whether or
not it is mid-upload. The sidecar allows one upload per wallet
(WALRUS_UPLOAD_PER_WALLET_CONCURRENCY defaults to 1), so landing on a busy
wallet costs the full upload ahead of it while other wallets sit idle.
Observed in production as a job waiting 6.2s for its assigned wallet with the
global semaphore reporting available:2, queued:0 — the contention was never
global, only per-wallet:
[walrus/upload] limiter_acquired {"keyIndex":6,"waitMs":6221,
"limits":{"global":{"capacity":8,"available":2,"queued":0},
"perWalletCapacity":1,"wallet":{"capacity":1,"available":0,"queued":1}}}
`least_loaded_index()` picks the wallet with the fewest in-flight attempts and
falls back to round-robin ordering among equals, so an idle pool still spreads
evenly and a fully busy one does not pile onto one signer.
Marking a wallet busy is a `WalletAttemptGuard` rather than paired
increment/decrement calls: the upload path has many early returns, and every
one of them has to release the slot. A guard cannot forget.
Tests: 11 new unit tests covering selection, guard release on early return,
nested attempts, saturating release, and an out-of-range index. Lib suite goes
466 → 477 passing with the same 21 pre-existing failures (no database in the
sandbox). cargo check clean, rustfmt clean on the touched files.
---
services/server/src/engine/walrus_seal.rs | 12 +-
services/server/src/jobs.rs | 7 +
services/server/src/routes/analyze.rs | 2 +-
services/server/src/routes/remember.rs | 4 +-
services/server/src/types.rs | 220 ++++++++++++++++++++++
5 files changed, 239 insertions(+), 6 deletions(-)
diff --git a/services/server/src/engine/walrus_seal.rs b/services/server/src/engine/walrus_seal.rs
index 54c0e25a6..413e23f63 100644
--- a/services/server/src/engine/walrus_seal.rs
+++ b/services/server/src/engine/walrus_seal.rs
@@ -240,15 +240,21 @@ impl MemoryEngine for WalrusSealEngine {
importance: f32,
agent_public_key: Option<&str>,
) -> Result {
- // Pick the next Sui key slot (round-robin) so concurrent stores
- // don't serialise on one signer.
- let key_index = self.key_pool.next_index().ok_or_else(|| {
+ // Pick the least-loaded Sui key slot so concurrent stores don't
+ // serialise on one signer. Round-robin alone could land on a wallet
+ // that is mid-upload while another sits idle; the per-wallet limit is
+ // 1, so that costs the full upload of whatever is ahead.
+ let key_index = self.key_pool.least_loaded_index().ok_or_else(|| {
AppError::Internal(
"No Sui keys configured (set SERVER_SUI_PRIVATE_KEYS or SERVER_SUI_PRIVATE_KEY)"
.into(),
)
})?;
+ // Hold the slot for the duration of the upload, so a store running
+ // concurrently sees this wallet as busy and picks another.
+ let _wallet_slot = self.key_pool.begin_attempt(key_index);
+
// Upload the prepared ciphertext to Walrus via the relay sidecar
// (pool key pays gas). `defer_transfer = false` — the blob is
// transferred to `owner` immediately, same as the inlined
diff --git a/services/server/src/jobs.rs b/services/server/src/jobs.rs
index bdaacd94a..8920fd438 100644
--- a/services/server/src/jobs.rs
+++ b/services/server/src/jobs.rs
@@ -521,6 +521,13 @@ pub(crate) async fn execute_wallet_job(
.into_apalis_error());
}
};
+ // Mark this wallet busy for the rest of the attempt, so a
+ // concurrently-enqueued job picks an idle wallet instead of
+ // queueing behind this upload. Held by guard rather than paired
+ // calls because every return below — and there are many — has to
+ // release it.
+ let _wallet_slot = state.key_pool.begin_attempt(wallet_index);
+
if wallet_index != enqueued_wallet_index || attempt_info.current > 1 {
tracing::info!(
"[wallet-job:upload] selected wallet for attempt: enqueued={} executing={} attempt={}/{}",
diff --git a/services/server/src/routes/analyze.rs b/services/server/src/routes/analyze.rs
index d8aa1c8d9..b96ad82a3 100644
--- a/services/server/src/routes/analyze.rs
+++ b/services/server/src/routes/analyze.rs
@@ -865,7 +865,7 @@ pub async fn analyze(
}
// Pick next wallet slot (round-robin) and enqueue UploadAndTransfer
- let Some(wallet_index) = state.key_pool.next_index() else {
+ let Some(wallet_index) = state.key_pool.least_loaded_index() else {
rate_limit::release_storage_quota(&state, &all_ids[idx..]).await;
return Err(AppError::Internal("No Sui keys configured".into()));
};
diff --git a/services/server/src/routes/remember.rs b/services/server/src/routes/remember.rs
index 62a99dd31..3311e33e6 100644
--- a/services/server/src/routes/remember.rs
+++ b/services/server/src/routes/remember.rs
@@ -226,7 +226,7 @@ fn spawn_prepare_remember_job(
)
.await?;
- let wallet_index = state.key_pool.next_index().ok_or_else(|| {
+ let wallet_index = state.key_pool.least_loaded_index().ok_or_else(|| {
AppError::Internal(
"No Sui keys configured (set SERVER_SUI_PRIVATE_KEYS or SERVER_SUI_PRIVATE_KEY)"
.into(),
@@ -410,7 +410,7 @@ fn spawn_prepare_bulk_remember_job(
for (job_id, namespace, vector, encrypted) in prepared {
let wallet_index = state
.key_pool
- .next_index()
+ .least_loaded_index()
.ok_or_else(|| AppError::Internal("No Sui keys configured".into()))?;
let encrypted_b64 =
base64::engine::general_purpose::STANDARD.encode(&encrypted);
diff --git a/services/server/src/types.rs b/services/server/src/types.rs
index 61c113537..db31b3f31 100644
--- a/services/server/src/types.rs
+++ b/services/server/src/types.rs
@@ -290,13 +290,30 @@ pub struct AppState {
pub struct KeyPool {
keys: Vec,
cursor: AtomicUsize,
+ /// Wallet transactions currently executing, per key.
+ ///
+ /// The sidecar enforces one upload at a time per wallet (concurrent
+ /// transactions from one signer can equivocate its owned objects, which
+ /// then stay locked until the epoch boundary). Blind round-robin does not
+ /// know that, so it hands a job to a wallet that is mid-upload while other
+ /// wallets sit idle — observed in production as a job waiting 6.2s for its
+ /// assigned wallet while the global limiter still had two free slots and
+ /// an empty queue.
+ ///
+ /// Counted per *executing attempt*, incremented and decremented inside one
+ /// scope by `WalletAttemptGuard`. Nothing crosses the job-queue boundary,
+ /// so a counter cannot drift: a restart zeroes it, which is accurate, and
+ /// every early return still releases because `Drop` runs.
+ inflight: Vec,
}
impl KeyPool {
pub fn new(keys: Vec) -> Self {
+ let inflight = keys.iter().map(|_| AtomicUsize::new(0)).collect();
Self {
keys,
cursor: AtomicUsize::new(0),
+ inflight,
}
}
@@ -318,6 +335,73 @@ impl KeyPool {
}
}
+ /// Returns the least-loaded key index, breaking ties in round-robin order.
+ ///
+ /// Join-shortest-queue rather than `next_index()`'s blind modulo: a wallet
+ /// that is mid-upload is skipped in favour of an idle one, which is the
+ /// whole point — the per-wallet limit is 1, so landing on a busy wallet
+ /// means queueing behind it even when the pool has capacity.
+ ///
+ /// The tie-break matters as much as the minimum. On an idle pool every
+ /// counter is 0, and a plain `argmin` would return index 0 every time,
+ /// funnelling all traffic onto one wallet — strictly worse than the
+ /// round-robin this replaces. Starting the scan at the rotating cursor
+ /// keeps equally-loaded keys spreading exactly as before.
+ pub fn least_loaded_index(&self) -> Option {
+ let len = self.keys.len();
+ if len == 0 {
+ return None;
+ }
+ let start = self.cursor.fetch_add(1, Ordering::Relaxed) % len;
+ let mut best = start;
+ let mut best_load = self.inflight[start].load(Ordering::Relaxed);
+ for step in 1..len {
+ let idx = (start + step) % len;
+ let load = self.inflight[idx].load(Ordering::Relaxed);
+ // Strictly less, so the first key scanned — the cursor's own —
+ // wins any tie and the rotation is preserved.
+ if load < best_load {
+ best = idx;
+ best_load = load;
+ }
+ }
+ Some(best)
+ }
+
+ /// Marks `index` busy for as long as the returned guard lives.
+ ///
+ /// Call this around the wallet transaction itself, not around enqueueing
+ /// one: the counter is meant to answer "is this wallet signing right now",
+ /// which is what the per-wallet limit actually serialises on.
+ pub fn begin_attempt(self: &Arc, index: usize) -> WalletAttemptGuard {
+ if let Some(slot) = self.inflight.get(index) {
+ slot.fetch_add(1, Ordering::Relaxed);
+ }
+ WalletAttemptGuard {
+ pool: Arc::clone(self),
+ index,
+ }
+ }
+
+ /// In-flight count per key. Observability only.
+ pub fn inflight_snapshot(&self) -> Vec {
+ self.inflight
+ .iter()
+ .map(|slot| slot.load(Ordering::Relaxed))
+ .collect()
+ }
+
+ fn end_attempt(&self, index: usize) {
+ if let Some(slot) = self.inflight.get(index) {
+ // Saturating: an extra release must not wrap to usize::MAX and
+ // leave this key looking permanently busiest, which would exclude
+ // it from selection for the life of the process.
+ let _ = slot.fetch_update(Ordering::Relaxed, Ordering::Relaxed, |v| {
+ Some(v.saturating_sub(1))
+ });
+ }
+ }
+
#[allow(dead_code)]
pub fn is_empty(&self) -> bool {
self.keys.is_empty()
@@ -329,6 +413,22 @@ impl KeyPool {
}
}
+/// Releases a wallet's in-flight count when dropped.
+///
+/// A guard rather than paired calls because `execute_wallet_job` returns from
+/// many points; every one of them has to decrement, and `Drop` is the only way
+/// to get that for free — including while unwinding from a panic.
+pub struct WalletAttemptGuard {
+ pool: Arc,
+ index: usize,
+}
+
+impl Drop for WalletAttemptGuard {
+ fn drop(&mut self) {
+ self.pool.end_attempt(self.index);
+ }
+}
+
// ============================================================
// Config
// ============================================================
@@ -2223,6 +2323,126 @@ mod tests {
static WALRUS_STORAGE_EPOCHS_ENV_LOCK: Mutex<()> = Mutex::new(());
+ fn pool(n: usize) -> Arc {
+ Arc::new(KeyPool::new((0..n).map(|i| format!("key{i}")).collect()))
+ }
+
+ #[test]
+ fn an_idle_pool_still_spreads_round_robin() {
+ // The regression this guards: with every counter at 0 a plain argmin
+ // returns index 0 forever, funnelling the whole pool onto one wallet
+ // — worse than the round-robin it replaces.
+ let pool = pool(4);
+ let picked: Vec<_> = (0..8).map(|_| pool.least_loaded_index().unwrap()).collect();
+ assert_eq!(picked, vec![0, 1, 2, 3, 0, 1, 2, 3]);
+ }
+
+ #[test]
+ fn a_busy_wallet_is_skipped_for_an_idle_one() {
+ // The production case: a job waited 6.2s for its assigned wallet while
+ // other wallets sat idle, because round-robin could not see the load.
+ let pool = pool(4);
+ let _busy = pool.begin_attempt(1);
+ // Cursor lands on 1 next, but 1 is busy and 2 is not.
+ let _ = pool.least_loaded_index();
+ assert_eq!(pool.least_loaded_index().unwrap(), 2);
+ }
+
+ #[test]
+ fn selection_avoids_every_busy_wallet_until_only_busy_ones_remain() {
+ let pool = pool(3);
+ let _a = pool.begin_attempt(0);
+ let _b = pool.begin_attempt(1);
+ // Only wallet 2 is free, so every pick goes there regardless of cursor.
+ for _ in 0..6 {
+ assert_eq!(pool.least_loaded_index().unwrap(), 2);
+ }
+ }
+
+ #[test]
+ fn a_fully_busy_pool_falls_back_to_spreading_evenly() {
+ // Saturated is not a special case: equal load means the tie-break
+ // decides, so behaviour degrades exactly to round-robin.
+ let pool = pool(3);
+ let _a = pool.begin_attempt(0);
+ let _b = pool.begin_attempt(1);
+ let _c = pool.begin_attempt(2);
+ let picked: Vec<_> = (0..6).map(|_| pool.least_loaded_index().unwrap()).collect();
+ assert_eq!(picked, vec![0, 1, 2, 0, 1, 2]);
+ }
+
+ #[test]
+ fn dropping_the_guard_frees_the_wallet() {
+ let pool = pool(2);
+ {
+ let _busy = pool.begin_attempt(0);
+ assert_eq!(pool.inflight_snapshot(), vec![1, 0]);
+ }
+ assert_eq!(pool.inflight_snapshot(), vec![0, 0]);
+ }
+
+ #[test]
+ fn the_guard_releases_on_an_early_return() {
+ // `execute_wallet_job` returns from many points; the guard exists so
+ // none of them has to remember to decrement.
+ let pool = pool(2);
+ fn bail(pool: &Arc) -> Result<(), ()> {
+ let _slot = pool.begin_attempt(1);
+ Err(())
+ }
+ assert!(bail(&pool).is_err());
+ assert_eq!(pool.inflight_snapshot(), vec![0, 0]);
+ }
+
+ #[test]
+ fn nested_attempts_on_one_wallet_count_and_release_independently() {
+ let pool = pool(2);
+ let first = pool.begin_attempt(0);
+ let second = pool.begin_attempt(0);
+ assert_eq!(pool.inflight_snapshot(), vec![2, 0]);
+ drop(first);
+ assert_eq!(pool.inflight_snapshot(), vec![1, 0]);
+ drop(second);
+ assert_eq!(pool.inflight_snapshot(), vec![0, 0]);
+ }
+
+ #[test]
+ fn releasing_below_zero_saturates_instead_of_wrapping() {
+ // A wrapped counter would read as usize::MAX and exclude the wallet
+ // from selection for the life of the process.
+ let pool = pool(2);
+ pool.end_attempt(0);
+ pool.end_attempt(0);
+ assert_eq!(pool.inflight_snapshot(), vec![0, 0]);
+ let _busy = pool.begin_attempt(1);
+ assert_eq!(pool.least_loaded_index().unwrap(), 0);
+ }
+
+ #[test]
+ fn an_out_of_range_index_is_ignored_rather_than_panicking() {
+ let pool = pool(2);
+ let guard = pool.begin_attempt(99);
+ assert_eq!(pool.inflight_snapshot(), vec![0, 0]);
+ drop(guard);
+ assert_eq!(pool.inflight_snapshot(), vec![0, 0]);
+ }
+
+ #[test]
+ fn an_empty_pool_selects_nothing() {
+ let pool = Arc::new(KeyPool::new(vec![]));
+ assert_eq!(pool.least_loaded_index(), None);
+ assert_eq!(pool.next_index(), None);
+ }
+
+ #[test]
+ fn next_index_is_unchanged_for_the_retry_path() {
+ // Retries derive their wallet from the job's own start index, not the
+ // global cursor, so `next_index` has to keep its old semantics.
+ let pool = pool(3);
+ let picked: Vec<_> = (0..6).map(|_| pool.next_index().unwrap()).collect();
+ assert_eq!(picked, vec![0, 1, 2, 0, 1, 2]);
+ }
+
#[test]
fn balance_monitor_interval_has_a_safe_minimum() {
assert_eq!(normalized_balance_monitor_interval(1), 30);
From 956de95dc831ef44d8215ebc37ea80bbf6741e0a Mon Sep 17 00:00:00 2001
From: Harry Phan
Date: Tue, 15 Sep 2026 21:35:06 +0700
Subject: [PATCH 005/132] chore(mcp): move the sidecar off the stale 0.0.3-era
SDK pin
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit
The sidecar installs @mysten-incubation/memwal from npm (Dockerfile runs
`npm ci` in scripts/), so packages/sdk in this repo is not what production
runs — and the pin had drifted far behind it. 0.0.4 predates every release
that matters here; most importantly it has no idempotency support at all,
so `rememberAsync` cannot send an idempotency_key and the relayer treats
every retry of a write as a brand-new one. A replayed remember therefore
mints a SECOND paid Walrus blob for a write already in flight. The SDK
grew keys in 0.1.2 ("collapse retries onto one paid remember job") and
production never received it.
0.1.7 is the current published version. The MCP layer only calls
analyzeAndWait / health / recall / rememberAndWait / rememberBulkAndWait /
restore, and deliberately excludes the manual-mode methods that carry the
one breaking change in the range (0.1.5 reshaped rememberManual), so the
used surface is unchanged. Verified by `npm ci` against the regenerated
lockfile followed by a clean `tsc --noEmit` over scripts/.
Also picks up the widened zod peer (^3.23.0 || ^4.0.0), which the
sidecar's own zod ^3.25.0 already satisfied.
---
services/server/scripts/package-lock.json | 13 ++++++++-----
services/server/scripts/package.json | 4 ++--
2 files changed, 10 insertions(+), 7 deletions(-)
diff --git a/services/server/scripts/package-lock.json b/services/server/scripts/package-lock.json
index 1df76c793..306bde7f4 100644
--- a/services/server/scripts/package-lock.json
+++ b/services/server/scripts/package-lock.json
@@ -9,7 +9,7 @@
"version": "0.1.0",
"dependencies": {
"@modelcontextprotocol/sdk": "1.29.0",
- "@mysten-incubation/memwal": "0.0.4",
+ "@mysten-incubation/memwal": "0.1.7",
"@mysten/seal": "1.1.0",
"@mysten/sui": "2.17.0",
"@mysten/walrus": "1.1.7",
@@ -621,20 +621,23 @@
}
},
"node_modules/@mysten-incubation/memwal": {
- "version": "0.0.4",
- "resolved": "https://registry.npmjs.org/@mysten-incubation/memwal/-/memwal-0.0.4.tgz",
- "integrity": "sha512-HBvq2ipq/gJoAWrn7oNHVCaMypTE7/9AtFwk8YSqd9njwpYqssGZ/18B3S8orzZdDsQmRSbgRDcnEZ8xwozaKg==",
+ "version": "0.1.7",
+ "resolved": "https://registry.npmjs.org/@mysten-incubation/memwal/-/memwal-0.1.7.tgz",
+ "integrity": "sha512-bVkk0+Zsjl6A4g6j7gvVhyFjmtHwhXLimgm8CoQGhRoM1Pz1XrR8/R5Uio+vah1JB3s6ICTtt1ycCcOAamyTfg==",
"license": "Apache-2.0",
"dependencies": {
"@noble/ed25519": "^2.3.0",
"@noble/hashes": "^2.0.0"
},
+ "engines": {
+ "node": ">=20.0.0"
+ },
"peerDependencies": {
"@mysten/seal": ">=1.1.0",
"@mysten/sui": ">=2.5.0",
"@mysten/walrus": ">=1.0.3",
"ai": ">=4.0.0",
- "zod": "^3.23.0"
+ "zod": "^3.23.0 || ^4.0.0"
},
"peerDependenciesMeta": {
"@mysten/walrus": {
diff --git a/services/server/scripts/package.json b/services/server/scripts/package.json
index 9d4854f61..78384fafe 100644
--- a/services/server/scripts/package.json
+++ b/services/server/scripts/package.json
@@ -5,12 +5,12 @@
"type": "module",
"scripts": {
"sidecar": "tsx sidecar-server.ts",
- "typecheck": "tsc --noEmit",
+ "typecheck": "tsc --noEmit",
"test": "node --test --import tsx './mcp/__tests__/*.test.ts' './__tests__/*.test.ts'"
},
"dependencies": {
"@modelcontextprotocol/sdk": "1.29.0",
- "@mysten-incubation/memwal": "0.0.4",
+ "@mysten-incubation/memwal": "0.1.7",
"@mysten/seal": "1.1.0",
"@mysten/sui": "2.17.0",
"@mysten/walrus": "1.1.7",
From 83534a6b56da25ba8d1f7503edae95b09987db67 Mon Sep 17 00:00:00 2001
From: Harry Phan
Date: Tue, 15 Sep 2026 21:35:10 +0700
Subject: [PATCH 006/132] perf(mcp): stop bulk and analyze polling on the SDK's
10s ceiling
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit
memwal_remember no longer blocks to terminal, but memwal_remember_bulk and
memwal_analyze still do, and both were left on the SDK's default cadence.
That default backs off as min(10s, 1500ms * 1.5^attempt), so status checks
land roughly 1.5, 3.75, 7.1, 12.2, 19.8 and 29.8s apart. Writes finish in
the 15-35s band, which is exactly where those gaps are widest: a batch that
truly completed at 20.5s is not reported until 29.8s. None of that is work,
it is waiting to be told the work finished.
Reuses REMEMBER_POLL_INTERVAL_MS rather than picking a second number — the
rate-limit reasoning behind 400ms is the same one, and a bulk poll covers
every pending job in one request (/api/remember/bulk/status takes all the
ids), so the request budget matches a single remember's.
---
services/server/scripts/mcp/tools/analyze.ts | 6 ++++++
services/server/scripts/mcp/tools/remember-bulk.ts | 10 ++++++++++
2 files changed, 16 insertions(+)
diff --git a/services/server/scripts/mcp/tools/analyze.ts b/services/server/scripts/mcp/tools/analyze.ts
index 2ea48f8d3..d4cdaaf85 100644
--- a/services/server/scripts/mcp/tools/analyze.ts
+++ b/services/server/scripts/mcp/tools/analyze.ts
@@ -3,6 +3,7 @@ import type { McpServer } from "@modelcontextprotocol/sdk/server/mcp.js";
import type { MemWalSession } from "../auth.js";
import { TOOL_METADATA } from "./annotations.js";
import { wrapTool, explorerFooter } from "./util.js";
+import { REMEMBER_POLL_INTERVAL_MS } from "./remember-wait.js";
const ANALYZE_INPUT = {
text: z
@@ -39,6 +40,11 @@ export function registerAnalyzeTool(
wrapTool<{ text: string; namespace?: string }>(session, "memwal_analyze", async ({ text, namespace }) => {
const result = await session.memwal.analyzeAndWait(text, namespace, {
timeoutMs: 180_000,
+ // Same reasoning as `memwal_remember_bulk`: this tool blocks to
+ // terminal, so the SDK's 10s backoff ceiling is dead time added
+ // to every extraction. The budget here is the longest of the
+ // three, which is exactly where that ceiling hurts most.
+ pollIntervalMs: REMEMBER_POLL_INTERVAL_MS,
});
const lines = result.results.map(
(r, i) =>
diff --git a/services/server/scripts/mcp/tools/remember-bulk.ts b/services/server/scripts/mcp/tools/remember-bulk.ts
index 69ed965f2..80d9317a3 100644
--- a/services/server/scripts/mcp/tools/remember-bulk.ts
+++ b/services/server/scripts/mcp/tools/remember-bulk.ts
@@ -3,6 +3,7 @@ import type { McpServer } from "@modelcontextprotocol/sdk/server/mcp.js";
import type { MemWalSession } from "../auth.js";
import { TOOL_METADATA } from "./annotations.js";
import { wrapTool, explorerFooter } from "./util.js";
+import { REMEMBER_POLL_INTERVAL_MS } from "./remember-wait.js";
const REMEMBER_BULK_INPUT = {
facts: z
@@ -43,6 +44,15 @@ export function registerRememberBulkTool(
const items = facts.map((text) => ({ text, namespace }));
const result = await session.memwal.rememberBulkAndWait(items, {
timeoutMs: 120_000,
+ // This tool still blocks to terminal, so unlike `memwal_remember`
+ // the poll cadence IS the wait: at the SDK's 1500ms default the
+ // backoff tops out at its 10s ceiling and checks land ~1.5, 3.75,
+ // 7.1, 12.2, 19.8, 29.8s apart — a batch that finished at 20.5s is
+ // not reported until 29.8s. One poll covers every pending job in
+ // the batch (`/api/remember/bulk/status` takes all the ids), so
+ // the rate-limit budget behind this number is the same as a
+ // single remember's.
+ pollIntervalMs: REMEMBER_POLL_INTERVAL_MS,
});
const lines = result.results.map((r, i) => {
// Label each result with its source fact by index. The SDK
From 62fd954253df3501b4e67e21e401c8d6f9abacec Mon Sep 17 00:00:00 2001
From: Harry Phan
Date: Tue, 15 Sep 2026 21:35:14 +0700
Subject: [PATCH 007/132] fix(mcp): give each tool its own orphan deadline
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit
The sweeper applied one callTimeoutMs to every tracked request. That number
is DEFAULT_CALL_TIMEOUT_MS, sized for memwal_analyze — the slowest tool
there is — so a memwal_remember whose reply was lost (the relayer answered,
the stream dropped before it arrived) kept the agent blocked for 240s even
though that tool cannot still be working: it gives up on its own job long
before. Users read a four-minute silence as a hang and reload the client,
which is the reload-for-minutes symptom.
Each entry now carries the deadline its own tool enforces plus 30s of
transport headroom, fixed when the request is first tracked so a reconnect
replay keeps the original budget. memwal_remember lands at 120s. The
stalled-handshake shortcut still wins when it is tighter but can no longer
extend a call past its tool's ceiling.
Deliberately generous headroom: the sidecar answers at its deadline with a
result or an error envelope rather than going quiet, so a reply is one hop
behind it. Cutting a merely-late reply off early is the expensive mistake,
because the agent then retries a write that actually landed.
Behaviour is unchanged wherever MEMWAL_MCP_CALL_TIMEOUT_MS is set, which is
every existing expiry test — resolveDeadlineMs returns the override as-is,
and in the stalled path min(stalledHandshakeMs, callTimeoutMs) is already
stalledHandshakeMs. Also drops the duplicate local toolNameOf in favour of
the module-scope one this needs.
---
packages/mcp/src/bridge.ts | 77 +++++++++++++++++++++++++++++++++-----
1 file changed, 68 insertions(+), 9 deletions(-)
diff --git a/packages/mcp/src/bridge.ts b/packages/mcp/src/bridge.ts
index 91b6f2a4e..93b32c5b0 100644
--- a/packages/mcp/src/bridge.ts
+++ b/packages/mcp/src/bridge.ts
@@ -270,6 +270,57 @@ const DEFAULT_CALL_TIMEOUT_MS = SLOWEST_SERVER_TOOL_MS + 60_000;
/** An override below this is a mistake, not an intent. */
const MIN_CALL_TIMEOUT_MS = 1_000;
+/** Longest a given tool can legitimately take server-side, keyed by tool name.
+ *
+ * `DEFAULT_CALL_TIMEOUT_MS` is sized for `memwal_analyze`, the slowest tool
+ * there is. Applying that one number to every call means a request whose reply
+ * is lost — the relayer answered, the stream dropped before it arrived — keeps
+ * the agent blocked for 240s even when the tool could not still be working.
+ * Users read that as a hang and reload the client.
+ *
+ * Each entry is the ceiling the matching tool enforces on itself in
+ * `services/server/scripts/mcp/tools/`: `MAX_REMEMBER_WAIT_MS` for
+ * `memwal_remember` (its default wait is 0 — it returns at accept — but an
+ * operator can raise `MEMWAL_MCP_REMEMBER_WAIT_MS` up to that cap),
+ * `MAX_STATUS_WAIT_MS` for `memwal_remember_status`, and the fixed `timeoutMs`
+ * the bulk and analyze tools pass to the SDK. Keep them in lockstep: a value
+ * below a tool's real ceiling abandons healthy work. Unlisted tools keep the
+ * default. */
+const TOOL_DEADLINE_MS: Readonly> = {
+ memwal_remember: 90_000,
+ memwal_remember_status: 60_000,
+ memwal_remember_bulk: 120_000,
+ memwal_analyze: SLOWEST_SERVER_TOOL_MS,
+};
+
+/** Absorbs relayer + transport overhead on top of a tool's own ceiling. The
+ * sidecar answers at its deadline with a result or an error envelope rather
+ * than going quiet, so the reply is one network hop behind it; 30s is many
+ * times that. Cutting a merely-late reply off early is the expensive mistake —
+ * the agent would retry a write that actually landed. */
+const ORPHAN_HEADROOM_MS = 30_000;
+
+/** Tool name for a `tools/call`, or null for any other JSON-RPC method. */
+function toolNameOf(msg: RpcMessage): string | null {
+ if (msg.method !== "tools/call") return null;
+ const params = msg.params;
+ if (params == null || typeof params !== "object") return null;
+ const name = (params as { name?: unknown }).name;
+ return typeof name === "string" ? name : null;
+}
+
+/** Deadline for one tracked request. A tool with a known ceiling gets that
+ * plus headroom; everything else keeps the global default. An explicit
+ * `MEMWAL_MCP_CALL_TIMEOUT_MS` pins every call, so tests still drive expiry
+ * from one knob. */
+function resolveDeadlineMs(msg: RpcMessage): number {
+ const fallback = resolveCallTimeoutMs();
+ if (process.env.MEMWAL_MCP_CALL_TIMEOUT_MS) return fallback;
+ const tool = toolNameOf(msg);
+ const ceiling = tool === null ? undefined : TOOL_DEADLINE_MS[tool];
+ return ceiling === undefined ? fallback : ceiling + ORPHAN_HEADROOM_MS;
+}
+
/** Without a cap, a long deadline drifts by a third of itself. */
const MAX_ORPHAN_SWEEP_MS = 5_000;
@@ -407,6 +458,10 @@ interface InFlightEntry {
* mid-session outage — the ordinary case — a request that never left the
* process was indistinguishable from one already sent. */
sent?: boolean;
+ /** How long this call may go unanswered before the sweeper declares its
+ * reply lost. Fixed when the request is first tracked, so a reconnect
+ * replay keeps the original budget. */
+ deadlineMs: number;
}
/** The relayer rejected the saved delegate key (HTTP 401 on the handshake).
@@ -1987,7 +2042,11 @@ export async function runBridge(
msg.id !== undefined &&
msg.id !== null
) {
- inFlight.set(msg.id, { msg, startedAt: Date.now() });
+ inFlight.set(msg.id, {
+ msg,
+ startedAt: Date.now(),
+ deadlineMs: resolveDeadlineMs(msg),
+ });
}
// Relayer session not up yet, OR the post-connect flush is still
// draining — buffer so this request stays behind everything that
@@ -2243,13 +2302,6 @@ export async function runBridge(
"memwal_analyze",
]);
- /** Name of the tool a tracked request was calling, when it was one. */
- function toolNameOf(msg: RpcMessage): string | null {
- if (msg.method !== "tools/call") return null;
- const params = msg.params as { name?: unknown } | undefined;
- return typeof params?.name === "string" ? params.name : null;
- }
-
function expiredRequestReport(
neverSent: boolean,
now: number,
@@ -2355,8 +2407,15 @@ export async function runBridge(
// invites a duplicate write.
const handshakeIsStalled =
handshakeStalledMs !== null && handshakeStalledMs > stalledHandshakeMs;
+ // `entry.deadlineMs` is this tool's own ceiling plus headroom, not
+ // the global one sized for the slowest tool — so a `memwal_remember`
+ // whose reply is lost is answered at 120s instead of 240s. The
+ // stalled-handshake shortcut still wins when it is tighter, but can
+ // never extend a tool past its own deadline.
const deadlineMs =
- neverSent && handshakeIsStalled ? stalledHandshakeMs : callTimeoutMs;
+ neverSent && handshakeIsStalled
+ ? Math.min(stalledHandshakeMs, entry.deadlineMs)
+ : entry.deadlineMs;
if (elapsedMs <= deadlineMs) continue;
// Built only for what actually expired: this walks `pendingForward`
// and interpolates two user-facing strings, and the branch it
From 1540349e65e7d28de68bca855dd93ed8e31fe963 Mon Sep 17 00:00:00 2001
From: Harry Phan
Date: Tue, 15 Sep 2026 21:35:19 +0700
Subject: [PATCH 008/132] perf(sdk): cut poll dead time and collapse replayed
writes onto one job
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit
Three changes to what a caller waits for, none to what a write means.
The backoff ceiling drops from 10s to 2s and the base default from 1500ms
to 600ms. The ceiling is pure observation cost — a job that finished is not
reported until the next poll lands — and at 10s the checks fell at ~1.5,
3.75, 7.1, 12.2, 19.8 and 29.8s, straddling the 15-35s band where writes
actually complete. Polling is one indexed row read on remember_jobs, so the
extra checks are cheap; callers on a long budget still pass a larger base
because the relayer's per-delegate-key rate limit, not cost, is the real
constraint.
Both wait loops slept BEFORE their first check, so an idempotent replay of
a write the relayer had already finished paid a full interval for a result
that was ready on arrival. They now check first and sleep second.
Generated idempotency keys are derived from the content over a 30-minute
bucket instead of crypto.randomUUID(). pendingRememberKeys only dedupes
retries that reuse one client instance, and the MCP sidecar builds a fresh
MemWal per transport session — so a reconnect replay found an empty map and
a random key read as a brand-new write, minting a second paid Walrus blob
for one already in flight. The bucket bounds the collapse: remember_jobs
rows are never pruned, so an unbucketed key would dedupe against a job from
any point in history and re-saving a since-deleted fact would hand back the
old blob id instead of storing it again.
Callers passing an explicit idempotencyKey are unaffected, and distinct
text or namespaces still derive distinct keys.
---
...remember-poll-cadence-and-replay-dedupe.md | 13 ++
packages/sdk/src/memwal.ts | 64 ++++++++-
packages/sdk/test/remember-latency.test.mjs | 132 ++++++++++++++++++
3 files changed, 202 insertions(+), 7 deletions(-)
create mode 100644 .changeset/remember-poll-cadence-and-replay-dedupe.md
create mode 100644 packages/sdk/test/remember-latency.test.mjs
diff --git a/.changeset/remember-poll-cadence-and-replay-dedupe.md b/.changeset/remember-poll-cadence-and-replay-dedupe.md
new file mode 100644
index 000000000..34d7ef54b
--- /dev/null
+++ b/.changeset/remember-poll-cadence-and-replay-dedupe.md
@@ -0,0 +1,13 @@
+---
+"@mysten-incubation/memwal": patch
+---
+
+Cut the dead time a `remember` spends waiting to be told it finished, and stop a reconnect replay from minting a second paid write.
+
+Job polling backed off as `min(10s, base * 1.5^min(attempt, 6))` from a 1500ms base, so status checks landed at roughly 1.5/3.75/7.1/12.2/19.8/29.8s. Real writes finish in the 15–35s band — a Walrus sliver upload plus three sequential Sui transactions — which is exactly where those gaps are widest, so a write that truly completed at 20.5s was not reported to the caller until 29.8s. That lag is pure observation cost: the job was done, nobody had asked yet. The backoff now caps at 2s and the base default drops to 600ms, putting average dead time near 1s instead of ~5s. Polling is a single indexed row read, so a 30s write costs ~17 checks instead of ~6.
+
+`waitForRememberJob` and `waitForRememberJobs` also slept *before* their first status check, so an idempotent replay of a write the relayer had already finished still paid a full poll interval for a result that was ready on arrival. Both now check first and sleep second.
+
+Generated idempotency keys are derived from the content (`sha256` over a 30-minute time bucket plus namespace and text) instead of `crypto.randomUUID()`. `pendingRememberKeys` only ever dedupes retries that reuse one client instance, and the MCP sidecar builds a fresh `MemWal` per transport session — so when a stdio bridge reconnects a dropped stream and the agent re-issues the same `memwal_remember`, the map is empty and a random key reads as a brand-new write. The relayer would then mint a second paid Walrus blob for a write already in flight, doubling queue load exactly when the queue was already slow enough to have caused the drop. The bucket bounds the collapse: `remember_jobs` rows are never pruned, so an unbucketed key would dedupe against a job from any point in history and a re-save of a since-deleted fact would return the old blob id instead of storing it again.
+
+Callers passing an explicit `idempotencyKey` are unaffected, and distinct text or namespaces still get distinct keys.
diff --git a/packages/sdk/src/memwal.ts b/packages/sdk/src/memwal.ts
index c49c90481..1a32b2a53 100644
--- a/packages/sdk/src/memwal.ts
+++ b/packages/sdk/src/memwal.ts
@@ -121,13 +121,56 @@ function sleep(ms: number): Promise {
return new Promise((resolve) => setTimeout(resolve, ms));
}
+/** Ceiling on the gap between two status checks.
+ *
+ * This is pure *observation* cost: a job that finished is not reported until
+ * the next poll lands, so the cap is the worst-case dead time bolted onto
+ * every wait, and half of it is the average. At the 10s ceiling checks landed
+ * ~1.5/3.75/7.1/12.2/19.8/29.8s apart, so a write that truly completed at
+ * 20.5s was not seen until 29.8s — and writes finish in the 15–35s band
+ * (Walrus sliver upload plus three sequential Sui transactions), exactly where
+ * those gaps were widest.
+ *
+ * 2s holds average dead time near 1s. The extra requests are cheap — polling
+ * is one indexed row read on `remember_jobs` — but they are not free against
+ * the relayer's per-delegate-key rate limit, which is why callers on a long
+ * budget (`services/server/scripts/mcp/tools/remember-wait.ts`) pass a larger
+ * base rather than relying on this floor. */
+const POLL_MAX_DELAY_MS = 2_000;
+
function pollingDelayMs(baseMs: number, attempt: number): number {
const base = Math.max(100, baseMs);
- const capped = Math.min(10_000, base * 1.5 ** Math.min(attempt, 6));
+ const capped = Math.min(POLL_MAX_DELAY_MS, base * 1.5 ** Math.min(attempt, 6));
const jitter = 0.75 + Math.random() * 0.5;
return Math.floor(capped * jitter);
}
+/** Window over which the same (namespace, text) resolves to the same
+ * idempotency key.
+ *
+ * `pendingRememberKeys` only dedupes retries that reuse one client instance.
+ * The MCP sidecar builds a fresh `MemWal` per transport session, so when the
+ * bridge reconnects a dropped stream and the agent re-issues the same
+ * `memwal_remember`, that map is empty. A random key then reads as a brand-new
+ * write and the relayer mints a SECOND paid Walrus blob for a write already in
+ * flight — doubling queue load exactly when the queue is already slow enough
+ * to have caused the drop.
+ *
+ * Deriving the key from the content makes that replay land on the existing
+ * job. The bucket bounds how long the collapse lasts: `remember_jobs` rows are
+ * never pruned, so an unbucketed key would dedupe against a job from any point
+ * in history and re-saving a fact the user had since deleted would return the
+ * old row's blob id instead of storing it again. Retries happen seconds after
+ * the original, so 30 minutes covers them with ~0.1% chance of a replay
+ * straddling the boundary — and straddling only costs the old behaviour (a
+ * duplicate job), never a wrong result. */
+const IDEMPOTENCY_BUCKET_MS = 30 * 60 * 1000;
+
+async function derivedIdempotencyKey(requestIdentity: string): Promise {
+ const bucket = Math.floor(Date.now() / IDEMPOTENCY_BUCKET_MS);
+ return `r1-${await sha256hex(`${bucket}\0${requestIdentity}`)}`;
+}
+
function isTransientPollingStatus(status: number): boolean {
return status === 0 || status === 429 || status >= 500;
}
@@ -268,7 +311,7 @@ export class MemWal {
const generatedKey = options.idempotencyKey === undefined;
const idempotencyKey = options.idempotencyKey
?? this.pendingRememberKeys.get(requestIdentity)
- ?? crypto.randomUUID();
+ ?? (await derivedIdempotencyKey(requestIdentity));
if (generatedKey) this.pendingRememberKeys.set(requestIdentity, idempotencyKey);
const accepted = await this.signedRequest(
@@ -316,12 +359,17 @@ export class MemWal {
jobId: string,
opts: { pollIntervalMs?: number; timeoutMs?: number } = {},
): Promise {
- const { pollIntervalMs = 1500, timeoutMs = 60_000 } = opts;
+ const { pollIntervalMs = 600, timeoutMs = 60_000 } = opts;
const deadline = Date.now() + timeoutMs;
let attempt = 0;
while (Date.now() < deadline) {
- await sleep(pollingDelayMs(pollIntervalMs, attempt++));
+ // Check first, sleep second. An idempotent replay — the same key
+ // for a write that already finished — is `done` on the server
+ // before we ask, so the old sleep-first order billed it a full
+ // poll delay for a result that was ready on arrival.
+ if (attempt > 0) await sleep(pollingDelayMs(pollIntervalMs, attempt - 1));
+ attempt++;
let status: RememberStatusResponse;
@@ -385,7 +433,7 @@ export class MemWal {
const generatedKey = opts.idempotencyKey === undefined;
const idempotencyKey = opts.idempotencyKey
?? this.pendingRememberKeys.get(requestIdentity)
- ?? crypto.randomUUID();
+ ?? (await derivedIdempotencyKey(requestIdentity));
if (generatedKey) this.pendingRememberKeys.set(requestIdentity, idempotencyKey);
const accepted = await this.rememberAsync(text, resolvedNamespace, { idempotencyKey });
@@ -473,7 +521,7 @@ export class MemWal {
namespaces: string[] = [],
opts: RememberBulkOptions = {},
): Promise {
- const { pollIntervalMs = 1500, timeoutMs = 120_000 } = opts;
+ const { pollIntervalMs = 600, timeoutMs = 120_000 } = opts;
const deadline = Date.now() + timeoutMs;
const results: RememberBulkItemResult[] = jobIds.map((jobId, idx) => ({
id: jobId,
@@ -486,7 +534,9 @@ export class MemWal {
let attempt = 0;
while (pending.size > 0 && Date.now() < deadline) {
- await sleep(pollingDelayMs(pollIntervalMs, attempt++));
+ // Check first, sleep second — see `waitForRememberJob`.
+ if (attempt > 0) await sleep(pollingDelayMs(pollIntervalMs, attempt - 1));
+ attempt++;
const pendingIds = jobIds.filter((jobId) => pending.has(jobId));
if (pendingIds.length === 0) {
diff --git a/packages/sdk/test/remember-latency.test.mjs b/packages/sdk/test/remember-latency.test.mjs
new file mode 100644
index 000000000..fc51e2de9
--- /dev/null
+++ b/packages/sdk/test/remember-latency.test.mjs
@@ -0,0 +1,132 @@
+import assert from "node:assert/strict";
+import test from "node:test";
+
+import { MemWal } from "../dist/memwal.js";
+
+const originalFetch = globalThis.fetch;
+
+test.afterEach(() => {
+ globalThis.fetch = originalFetch;
+});
+
+/** Stub relayer. `onStatus(pollIndex)` decides what each status poll returns. */
+function stubRelayer({ onStatus, posted = [] }) {
+ let polls = 0;
+ globalThis.fetch = async (url, init = {}) => {
+ const path = new URL(url).pathname;
+ if (path === "/version") {
+ return Response.json({
+ apiVersion: "1.0.0",
+ relayerVersion: "1.0.0",
+ minSupportedSdk: { typescript: "0.0.4" },
+ });
+ }
+ if (path === "/api/config") {
+ return Response.json({ packageId: "0x1", network: "testnet" });
+ }
+ if (path === "/api/remember" && init.method === "POST") {
+ posted.push({ body: JSON.parse(init.body), at: Date.now() });
+ return Response.json({ job_id: "job-1", status: "pending" }, { status: 202 });
+ }
+ if (path === "/api/remember/job-1") {
+ return Response.json(onStatus(polls++));
+ }
+ throw new Error(`unexpected request ${path}`);
+ };
+ return { posted, pollCount: () => polls };
+}
+
+function newClient() {
+ const client = MemWal.create({
+ key: new Uint8Array(32).fill(1),
+ accountId: "0x1",
+ serverUrl: "https://relayer.example",
+ });
+ client.buildSealSession = async () => "test-session";
+ return client;
+}
+
+const DONE = {
+ job_id: "job-1",
+ status: "done",
+ blob_id: "blob-1",
+ owner: "0x1",
+ namespace: "default",
+};
+
+test("a job that is already done resolves without paying a poll delay first", async () => {
+ stubRelayer({ onStatus: () => DONE });
+
+ const started = Date.now();
+ const result = await newClient().rememberAndWait("already finished");
+ const elapsed = Date.now() - started;
+
+ assert.equal(result.blob_id, "blob-1");
+ // Sleep-first polling billed this a full base interval (~600ms, and ~1.5s
+ // before the interval was lowered) for a result the server had ready.
+ assert.ok(elapsed < 250, `expected an immediate first poll, took ${elapsed}ms`);
+});
+
+test("poll gaps stay bounded so a finished write is observed promptly", async () => {
+ const seenAt = [];
+ // Never terminal: let the loop run its full backoff ramp against the clock.
+ stubRelayer({
+ onStatus: () => {
+ seenAt.push(Date.now());
+ return { job_id: "job-1", status: "running" };
+ },
+ });
+
+ await assert.rejects(
+ newClient().rememberAndWait("slow write", undefined, { timeoutMs: 9_000 }),
+ /timed out/,
+ );
+
+ const gaps = seenAt.slice(1).map((t, i) => t - seenAt[i]);
+ const worst = Math.max(...gaps);
+ // The cap is 2s; jitter can stretch one gap to 2.5s. The old 10s cap put
+ // checks at 1.5/3.75/7.1/12.2/19.8/29.8s — a write finishing at 20.5s was
+ // not seen until 29.8s, which is most of what a user experienced as a slow
+ // remember.
+ assert.ok(worst < 2_600, `worst poll gap ${worst}ms exceeds the 2s cap + jitter`);
+ // And it must actually be polling, not spinning.
+ assert.ok(gaps.length >= 4, `expected a real ramp, saw ${gaps.length} gaps`);
+ assert.ok(Math.min(...gaps) > 50, "polling should not busy-loop");
+});
+
+test("the same fact reuses one idempotency key across client instances", async () => {
+ const posted = [];
+ stubRelayer({ onStatus: () => DONE, posted });
+
+ // Two instances = what the MCP sidecar builds when the stdio bridge
+ // reconnects a dropped SSE stream. A per-instance random key made the
+ // replay look like a new write, so the relayer minted a second paid
+ // Walrus blob for a write already in flight.
+ await newClient().rememberAndWait("a durable fact");
+ await newClient().rememberAndWait("a durable fact");
+
+ assert.equal(posted.length, 2);
+ assert.equal(posted[0].body.idempotency_key, posted[1].body.idempotency_key);
+});
+
+test("different text and different namespaces get different keys", async () => {
+ const posted = [];
+ stubRelayer({ onStatus: () => DONE, posted });
+
+ const client = newClient();
+ await client.rememberAndWait("fact one");
+ await client.rememberAndWait("fact two");
+ await client.rememberAndWait("fact one", "other-namespace");
+
+ const keys = posted.map((p) => p.body.idempotency_key);
+ assert.equal(new Set(keys).size, 3, "distinct writes must not collapse onto one job");
+});
+
+test("an explicit idempotency key still wins", async () => {
+ const posted = [];
+ stubRelayer({ onStatus: () => DONE, posted });
+
+ await newClient().rememberAndWait("a fact", undefined, { idempotencyKey: "caller-owned" });
+
+ assert.equal(posted[0].body.idempotency_key, "caller-owned");
+});
From d83b900798cd398f7fc1d568865b4af2e90f3d0a Mon Sep 17 00:00:00 2001
From: Harry Phan
Date: Tue, 15 Sep 2026 21:45:07 +0700
Subject: [PATCH 009/132] perf(mcp): return memwal_remember_bulk at accept,
like memwal_remember
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit
memwal_remember stopped blocking on the whole Walrus write; bulk did not,
and bulk is the path an agent actually takes. The server instructions send
it here whenever it learned more than one thing, so leaving this tool on a
fixed 120s block meant the common multi-fact turn still stalled — the fast
return only covered the case that mattered less.
Blocking is also worse per fact here, not better. A batch is N separate
Walrus writes contending for the same upload slots
(WALRUS_UPLOAD_PER_WALLET_CONCURRENCY defaults to 1), so they land one after
another. Against the measured 30-75s single-write spread a five-fact batch
could exhaust the whole budget and come back as nothing but timeouts, having
held the agent for two minutes first.
Same shape as memwal_remember: rememberBulkAsync to accept, then hand back
the job_ids. Safe on the same grounds — remember_bulk commits every
remember_jobs row before it spawns preparation or answers
(services/server/src/routes/remember.rs), so acceptance survives the client
going away. Each job_id is returned paired with its fact, because with a
batch "one of these failed" is only actionable if you can tell which.
A non-zero MEMWAL_MCP_REMEMBER_WAIT_MS still waits, and that path now
reports a mixed batch honestly: waitForRememberJobs does not throw on
expiry, it marks stragglers `timeout`, so those are shown as still uploading
with their job_ids rather than as failures.
memwal_remember_status takes job_ids to settle a whole batch in one call.
It deliberately does not throw on a failed job the way the single-id path
does: a batch comes back mixed, and throwing on the first failure would
discard the blob_ids of the writes that did land.
---
.../remember-bulk-fast-return.test.ts | 217 ++++++++++++++++++
.../server/scripts/mcp/tools/remember-bulk.ts | 99 ++++++--
.../scripts/mcp/tools/remember-status.ts | 117 +++++++++-
.../server/scripts/mcp/tools/remember-wait.ts | 44 ++++
4 files changed, 452 insertions(+), 25 deletions(-)
create mode 100644 services/server/scripts/mcp/__tests__/remember-bulk-fast-return.test.ts
diff --git a/services/server/scripts/mcp/__tests__/remember-bulk-fast-return.test.ts b/services/server/scripts/mcp/__tests__/remember-bulk-fast-return.test.ts
new file mode 100644
index 000000000..e0b956005
--- /dev/null
+++ b/services/server/scripts/mcp/__tests__/remember-bulk-fast-return.test.ts
@@ -0,0 +1,217 @@
+import assert from "node:assert/strict";
+import test, { type TestContext } from "node:test";
+import { Client } from "@modelcontextprotocol/sdk/client/index.js";
+import { InMemoryTransport } from "@modelcontextprotocol/sdk/inMemory.js";
+import type { MemWalSession } from "../auth.js";
+import { createMcpServer } from "../server.js";
+
+/**
+ * `memwal_remember_bulk` used to block until every job in the batch reached a
+ * terminal state, on a fixed 120s budget. That made it the slowest tool while
+ * the server instructions steer an agent to it for any multi-fact turn, and a
+ * batch is N separate Walrus writes contending for one upload slot per wallet
+ * — so a five-fact batch could burn the whole budget and return nothing but
+ * timeouts.
+ *
+ * It now mirrors `memwal_remember`: accept, then hand back the job_ids. These
+ * tests pin the two things that keeps honest — an accepted batch must never
+ * read as saved, and every job_id must come back paired with its fact so a
+ * later partial failure is actionable.
+ */
+
+interface BulkBehaviour {
+ /** Per-job terminal state, in input order. */
+ states: Array<"done" | "failed" | "timeout">;
+}
+
+function sessionWith(b: BulkBehaviour, calls: string[] = []): MemWalSession {
+ const jobIds = b.states.map((_, i) => `job-${i + 1}`);
+ return {
+ oauthScope: "memwal:read memwal:write",
+ namespace: "default",
+ memwal: {
+ async rememberBulkAsync(items: Array<{ text: string }>) {
+ calls.push(`bulkAsync:${items.map((i) => i.text).join("|")}`);
+ return { job_ids: jobIds, total: jobIds.length, status: "accepted" };
+ },
+ async rememberBulkAndWait() {
+ calls.push("bulkAndWait");
+ throw new Error("bulk must not block to terminal any more");
+ },
+ async waitForRememberJobs(ids: string[]) {
+ calls.push(`waitJobs:${ids.join(",")}`);
+ return {
+ results: b.states.map((status, i) => ({
+ id: jobIds[i],
+ blob_id: status === "done" ? `blob-${i + 1}` : "",
+ status,
+ namespace: "default",
+ error: status === "failed" ? "walrus upload rejected" : undefined,
+ })),
+ total: b.states.length,
+ succeeded: b.states.filter((s) => s === "done").length,
+ failed: b.states.filter((s) => s !== "done").length,
+ };
+ },
+ async getRememberBulkStatus(ids: string[]) {
+ calls.push(`bulkStatus:${ids.join(",")}`);
+ return {
+ results: ids.map((id, i) => ({
+ job_id: id,
+ status: b.states[i] === "timeout" ? "running" : b.states[i],
+ blob_id: b.states[i] === "done" ? `blob-${i + 1}` : undefined,
+ error: b.states[i] === "failed" ? "walrus upload rejected" : undefined,
+ })),
+ };
+ },
+ },
+ } as unknown as MemWalSession;
+}
+
+async function clientFor(session: MemWalSession, t: TestContext): Promise {
+ const [clientTransport, serverTransport] = InMemoryTransport.createLinkedPair();
+ const server = createMcpServer(session);
+ const client = new Client({ name: "remember-bulk-fast-return-test", version: "1.0.0" });
+ t.after(async () => {
+ await client.close();
+ await server.close();
+ });
+ await server.connect(serverTransport);
+ await client.connect(clientTransport);
+ return client;
+}
+
+function textOf(result: unknown): string {
+ return (result as { content: Array<{ text: string }> }).content
+ .map((c) => c.text)
+ .join("\n");
+}
+
+test("memwal_remember_bulk returns at accept and never reads as saved", async (t) => {
+ const calls: string[] = [];
+ const client = await clientFor(
+ sessionWith({ states: ["done", "done"] }, calls),
+ t,
+ );
+
+ const text = textOf(
+ await client.callTool({
+ name: "memwal_remember_bulk",
+ arguments: { facts: ["likes espresso", "works in Hanoi"] },
+ }),
+ );
+
+ // Accepted, not stored — and it must not have blocked on the batch.
+ assert.match(text, /ACCEPTED, NOT YET SAVED/);
+ assert.ok(calls.some((c) => c.startsWith("bulkAsync:")), "should accept via rememberBulkAsync");
+ assert.ok(!calls.includes("bulkAndWait"), "must not block to terminal");
+ assert.ok(
+ !calls.some((c) => c.startsWith("waitJobs:")),
+ "a zero budget must not enter the wait loop",
+ );
+ // No blob_id may appear — that is the token an agent reads as "stored".
+ assert.ok(!/blob_id=/.test(text), `accepted batch leaked a blob_id: ${text}`);
+});
+
+test("every job_id comes back paired with its fact", async (t) => {
+ const client = await clientFor(sessionWith({ states: ["done", "done", "done"] }), t);
+
+ const text = textOf(
+ await client.callTool({
+ name: "memwal_remember_bulk",
+ arguments: { facts: ["alpha fact", "beta fact", "gamma fact"] },
+ }),
+ );
+
+ // "one of these failed" is only actionable if the agent can tell which.
+ assert.match(text, /job_id=job-1 — alpha fact/);
+ assert.match(text, /job_id=job-2 — beta fact/);
+ assert.match(text, /job_id=job-3 — gamma fact/);
+});
+
+test("the accepted batch tells the agent how to settle it and not to re-send", async (t) => {
+ const client = await clientFor(sessionWith({ states: ["done"] }), t);
+
+ const text = textOf(
+ await client.callTool({
+ name: "memwal_remember_bulk",
+ arguments: { facts: ["a durable fact"] },
+ }),
+ );
+
+ assert.match(text, /memwal_remember_status/);
+ assert.match(text, /job_ids=/);
+ // Re-sending queues a second paid copy behind the first.
+ assert.match(text, /[Dd]o not re-send/);
+});
+
+test("memwal_remember_status settles a whole batch in one call", async (t) => {
+ const calls: string[] = [];
+ const client = await clientFor(
+ sessionWith({ states: ["done", "failed", "timeout"] }, calls),
+ t,
+ );
+
+ const text = textOf(
+ await client.callTool({
+ name: "memwal_remember_status",
+ arguments: { job_ids: ["job-1", "job-2", "job-3"], waitMs: 1000 },
+ }),
+ );
+
+ // A mixed batch must report every outcome rather than throwing on the
+ // first failure — otherwise the blob_ids that DID land are lost.
+ assert.match(text, /1\/3 saved/);
+ assert.match(text, /blob_id=blob-1/);
+ assert.match(text, /NOT STORED/);
+ assert.match(text, /still uploading/);
+ // And it must name which ids still need chasing, and which need re-sending.
+ assert.match(text, /job_ids=\[job-3\]/);
+ assert.match(text, /must be sent again/);
+ assert.ok(calls.some((c) => c.startsWith("waitJobs:")), "should use the batch wait");
+});
+
+test("memwal_remember_status with waitMs=0 reads a batch without waiting", async (t) => {
+ const calls: string[] = [];
+ const client = await clientFor(sessionWith({ states: ["done", "timeout"] }, calls), t);
+
+ const text = textOf(
+ await client.callTool({
+ name: "memwal_remember_status",
+ arguments: { job_ids: ["job-1", "job-2"], waitMs: 0 },
+ }),
+ );
+
+ assert.ok(
+ calls.some((c) => c.startsWith("bulkStatus:")),
+ "a zero budget must be a single batched read",
+ );
+ assert.ok(!calls.some((c) => c.startsWith("waitJobs:")), "must not enter the wait loop");
+ assert.match(text, /1\/2 saved/);
+});
+
+test("memwal_remember_status rejects job_id and job_ids together", async (t) => {
+ const client = await clientFor(sessionWith({ states: ["done"] }), t);
+
+ const result = await client.callTool({
+ name: "memwal_remember_status",
+ arguments: { job_id: "job-1", job_ids: ["job-2"] },
+ });
+
+ // They would describe different writes; guessing which one the caller
+ // meant would report the wrong fact's fate.
+ assert.equal((result as { isError?: boolean }).isError, true);
+ assert.match(textOf(result), /not both/);
+});
+
+test("memwal_remember_status requires one of job_id or job_ids", async (t) => {
+ const client = await clientFor(sessionWith({ states: ["done"] }), t);
+
+ const result = await client.callTool({
+ name: "memwal_remember_status",
+ arguments: { waitMs: 0 },
+ });
+
+ assert.equal((result as { isError?: boolean }).isError, true);
+ assert.match(textOf(result), /job_id.*job_ids/s);
+});
diff --git a/services/server/scripts/mcp/tools/remember-bulk.ts b/services/server/scripts/mcp/tools/remember-bulk.ts
index 80d9317a3..01c48a1a9 100644
--- a/services/server/scripts/mcp/tools/remember-bulk.ts
+++ b/services/server/scripts/mcp/tools/remember-bulk.ts
@@ -3,7 +3,11 @@ import type { McpServer } from "@modelcontextprotocol/sdk/server/mcp.js";
import type { MemWalSession } from "../auth.js";
import { TOOL_METADATA } from "./annotations.js";
import { wrapTool, explorerFooter } from "./util.js";
-import { REMEMBER_POLL_INTERVAL_MS } from "./remember-wait.js";
+import {
+ REMEMBER_POLL_INTERVAL_MS,
+ REMEMBER_WAIT_MS,
+ pendingBulkMessage,
+} from "./remember-wait.js";
const REMEMBER_BULK_INPUT = {
facts: z
@@ -22,11 +26,20 @@ const REMEMBER_BULK_INPUT = {
} as const;
/**
- * memwal_remember_bulk — persist several durable facts in one batched request
- * and return only once every job reaches a terminal state. Wraps the SDK's
- * `rememberBulkAndWait` (embed + SEAL-encrypt all items concurrently, upload
- * N blobs in parallel). Prefer this over N separate `memwal_remember` calls
- * when you learned multiple distinct facts at once.
+ * memwal_remember_bulk — persist several durable facts in one batched request.
+ *
+ * Mirrors `memwal_remember`: returns once every job reaches a terminal state
+ * if that happens inside `REMEMBER_WAIT_MS`, and otherwise hands back the
+ * job_ids saying plainly that the facts are not saved yet.
+ *
+ * Blocking here was worse than blocking on a single fact, not better. The
+ * server instructions steer an agent to this tool whenever it learned more
+ * than one thing, so it is the common path — and a batch is N separate Walrus
+ * writes contending for the same upload slots
+ * (`WALRUS_UPLOAD_PER_WALLET_CONCURRENCY` defaults to 1), so they land one
+ * after another rather than together. Against a 30–75s single-write spread a
+ * five-fact batch could exhaust the old fixed 120s budget outright and return
+ * nothing but timeouts, having blocked the agent for two minutes first.
*/
export function registerRememberBulkTool(
server: McpServer,
@@ -37,23 +50,58 @@ export function registerRememberBulkTool(
{
...TOOL_METADATA.memwal_remember_bulk,
description:
- "Save multiple durable facts in one call. Use when you learned several distinct facts at once (onboarding details, a list of preferences, decisions from a discussion). Pass an array of complete fact statements (max 20) — do not summarize. Prefer this over repeated memwal_remember calls.",
+ "Save multiple durable facts in one call. Use when you learned several distinct facts at once (onboarding details, a list of preferences, decisions from a discussion). Pass an array of complete fact statements (max 20) — do not summarize. Prefer this over repeated memwal_remember calls. Walrus writes queue, so this normally returns job_ids with the facts NOT yet saved — in that case say they are being saved rather than claiming they are stored, and resolve them with memwal_remember_status.",
inputSchema: REMEMBER_BULK_INPUT,
},
wrapTool<{ facts: string[]; namespace?: string }>(session, "memwal_remember_bulk", async ({ facts, namespace }) => {
const items = facts.map((text) => ({ text, namespace }));
- const result = await session.memwal.rememberBulkAndWait(items, {
- timeoutMs: 120_000,
- // This tool still blocks to terminal, so unlike `memwal_remember`
- // the poll cadence IS the wait: at the SDK's 1500ms default the
- // backoff tops out at its 10s ceiling and checks land ~1.5, 3.75,
- // 7.1, 12.2, 19.8, 29.8s apart — a batch that finished at 20.5s is
- // not reported until 29.8s. One poll covers every pending job in
- // the batch (`/api/remember/bulk/status` takes all the ids), so
- // the rate-limit budget behind this number is the same as a
- // single remember's.
- pollIntervalMs: REMEMBER_POLL_INTERVAL_MS,
+ // Two steps rather than `rememberBulkAndWait`, for the same reason
+ // `memwal_remember` splits them: acceptance is the part that must
+ // succeed, the wait is a courtesy we cut short.
+ const accepted = await session.memwal.rememberBulkAsync(items);
+
+ // Pair each job with its fact up front. Every later branch needs
+ // it, and the relayer returns job_ids in input order.
+ const entries = accepted.job_ids.map((jobId, i) => ({
+ jobId,
+ text: facts[i] ?? "",
+ }));
+
+ const pending = (waitedMs: number) => ({
+ content: [
+ {
+ type: "text" as const,
+ text: pendingBulkMessage(entries, waitedMs),
+ },
+ ],
});
+
+ if (REMEMBER_WAIT_MS === 0) return pending(0);
+
+ const startedAt = Date.now();
+ const namespaces = items.map(
+ (item) => item.namespace ?? session.namespace ?? "default"
+ );
+ // Unlike `waitForRememberJob`, this never throws on expiry — it
+ // reports the stragglers as `timeout` per item, so a batch can come
+ // back part landed and part still in flight.
+ const result = await session.memwal.waitForRememberJobs(
+ accepted.job_ids,
+ namespaces,
+ {
+ timeoutMs: REMEMBER_WAIT_MS,
+ pollIntervalMs: REMEMBER_POLL_INTERVAL_MS,
+ }
+ );
+ const waitedMs = Date.now() - startedAt;
+
+ const unfinished = result.results.flatMap((r, i) =>
+ r.status === "timeout" ? [{ jobId: r.id, text: facts[i] ?? "" }] : []
+ );
+ // Nothing landed inside the budget — the ordinary outcome when the
+ // queue is busy. Say so once rather than printing N timeout rows.
+ if (unfinished.length === result.results.length) return pending(waitedMs);
+
const lines = result.results.map((r, i) => {
// Label each result with its source fact by index. The SDK
// returns results in input order, but guard against a length /
@@ -61,18 +109,27 @@ export function registerRememberBulkTool(
const text = facts[i] ?? "";
const blob = r.blob_id ? ` blob_id=${r.blob_id}` : "";
const err = r.error ? ` error=${r.error}` : "";
- return `${i + 1}. [${r.status}]${blob}${err}${text ? ` — ${text}` : ""}`;
+ // `timeout` is not a failure — the write is still running and
+ // its job_id is how the caller settles it later.
+ const state = r.status === "timeout" ? `still uploading, job_id=${r.id}` : r.status;
+ return `${i + 1}. [${state}]${blob}${err}${text ? ` — ${text}` : ""}`;
});
const summary = `Saved ${result.succeeded}/${result.total} fact(s) to Walrus Memory (failed=${result.failed}).`;
const footer = result.succeeded > 0 ? `\n\n${explorerFooter()}` : "";
+ const tail = unfinished.length
+ ? `\n\n${unfinished.length} write(s) are STILL UPLOADING and are NOT saved yet. ` +
+ `Do not claim those facts are stored; resolve them with memwal_remember_status ` +
+ `using job_ids=[${unfinished.map((u) => u.jobId).join(", ")}]. Do not re-send ` +
+ `them — that queues duplicates behind the originals.`
+ : "";
return {
content: [
{
type: "text",
text:
- lines.length > 0
+ (lines.length > 0
? `${summary}\n\n${lines.join("\n")}${footer}`
- : `${summary}${footer}`,
+ : `${summary}${footer}`) + tail,
},
],
};
diff --git a/services/server/scripts/mcp/tools/remember-status.ts b/services/server/scripts/mcp/tools/remember-status.ts
index 8202b15e4..abefa6507 100644
--- a/services/server/scripts/mcp/tools/remember-status.ts
+++ b/services/server/scripts/mcp/tools/remember-status.ts
@@ -2,7 +2,7 @@ import { z } from "zod";
import type { McpServer } from "@modelcontextprotocol/sdk/server/mcp.js";
import type { MemWalSession } from "../auth.js";
import { TOOL_METADATA } from "./annotations.js";
-import { wrapTool, walruscanBlobUrl } from "./util.js";
+import { wrapTool, walruscanBlobUrl, explorerFooter } from "./util.js";
import {
REMEMBER_POLL_INTERVAL_MS,
isStillRunning,
@@ -16,11 +16,23 @@ import {
*/
const MAX_STATUS_WAIT_MS = 60_000;
+/** Matches the bulk write cap, so a whole batch settles in one call. */
+const MAX_STATUS_JOB_IDS = 20;
+
const STATUS_INPUT = {
job_id: z
.string()
.min(1)
+ .optional()
.describe("The job_id returned by memwal_remember when the write had not landed yet."),
+ job_ids: z
+ .array(z.string().min(1))
+ .min(1)
+ .max(MAX_STATUS_JOB_IDS)
+ .optional()
+ .describe(
+ "The job_ids returned by memwal_remember_bulk when the writes had not landed yet. Pass the whole batch in one call rather than polling each id separately. Supply either this or job_id."
+ ),
waitMs: z
.number()
.int()
@@ -54,15 +66,31 @@ export function registerRememberStatusTool(
{
...TOOL_METADATA.memwal_remember_status,
description:
- "Check whether an in-flight memwal_remember write has landed. Call this with the job_id memwal_remember returned when it reported the fact was NOT saved yet. Returns the blob_id once stored, reports that it is still uploading (call again), or reports that it failed — in which case the fact was never stored and you should send it again with memwal_remember.",
+ "Check whether in-flight Walrus Memory writes have landed. Call this with the job_id memwal_remember returned, or job_ids from memwal_remember_bulk, when the write was reported NOT saved yet. Returns the blob_id once stored, reports that it is still uploading (call again with the ids still listed), or reports that it failed — in which case the fact was never stored and you should send it again. A batch can come back mixed, so read every line before telling the user anything is saved.",
inputSchema: STATUS_INPUT,
},
- wrapTool<{ job_id: string; waitMs?: number }>(
+ wrapTool<{ job_id?: string; job_ids?: string[]; waitMs?: number }>(
session,
"memwal_remember_status",
- async ({ job_id, waitMs }) => {
+ async ({ job_id, job_ids, waitMs }) => {
const budget = waitMs ?? DEFAULT_STATUS_WAIT_MS;
+ // Exactly one of the two. Enforced here rather than in the
+ // schema because `registerTool` takes a raw shape, which has
+ // nowhere to hang a cross-field refinement.
+ const batch = job_ids ?? [];
+ if (batch.length > 0 && job_id) {
+ throw new Error(
+ "Pass job_id or job_ids, not both — they would describe different writes."
+ );
+ }
+ if (batch.length > 0) return await settleBatch(session, batch, budget);
+ if (!job_id) {
+ throw new Error(
+ "Pass job_id (from memwal_remember) or job_ids (from memwal_remember_bulk)."
+ );
+ }
+
// A zero budget means "read the current state", which is a
// single GET. waitForRememberJob cannot express that: it
// sleeps before its first poll, so a 0ms deadline would
@@ -108,6 +136,87 @@ export function registerRememberStatusTool(
);
}
+/**
+ * Settle a batch of job_ids in one report.
+ *
+ * Unlike the single-job path this never throws on a failed job: a batch
+ * routinely comes back mixed, and throwing on the first failure would hide the
+ * blob_ids of the writes that did land — the agent would have no way to tell
+ * which facts still need re-sending.
+ */
+async function settleBatch(
+ session: MemWalSession,
+ jobIds: string[],
+ budgetMs: number
+) {
+ // A zero budget is a single batched read, the same shortcut the one-job
+ // path takes: `waitForRememberJobs` sleeps before its first poll, so a 0ms
+ // deadline would report everything as still running without ever asking.
+ const rows =
+ budgetMs === 0
+ ? (await session.memwal.getRememberBulkStatus(jobIds)).results.map((r) => ({
+ id: r.job_id,
+ status: r.status,
+ blob_id: r.blob_id ?? "",
+ error: r.error,
+ }))
+ : (
+ await session.memwal.waitForRememberJobs(jobIds, [], {
+ timeoutMs: budgetMs,
+ pollIntervalMs: REMEMBER_POLL_INTERVAL_MS,
+ })
+ ).results.map((r) => ({
+ id: r.id,
+ status: r.status,
+ blob_id: r.blob_id,
+ error: r.error,
+ }));
+
+ // `timeout` (the waited path) and pending/running/uploaded (the immediate
+ // read) are the same thing to a caller: still in flight, ask again.
+ const inFlight = rows.filter(
+ (r) => r.status !== "done" && r.status !== "failed" && r.status !== "not_found"
+ );
+ const failed = rows.filter((r) => r.status === "failed" || r.status === "not_found");
+ const done = rows.filter((r) => r.status === "done");
+
+ const lines = rows.map((r, i) => {
+ const blob = r.blob_id ? ` blob_id=${r.blob_id}` : "";
+ const err = r.error ? ` error=${r.error}` : "";
+ const state =
+ r.status === "done"
+ ? "saved"
+ : r.status === "failed" || r.status === "not_found"
+ ? `NOT STORED (${r.status})`
+ : "still uploading";
+ return `${i + 1}. [${state}] job_id=${r.id}${blob}${err}`;
+ });
+
+ const parts = [
+ `${done.length}/${rows.length} saved` +
+ (inFlight.length ? `, ${inFlight.length} still uploading` : "") +
+ (failed.length ? `, ${failed.length} NOT stored` : "") +
+ ".",
+ lines.join("\n"),
+ ];
+ if (inFlight.length) {
+ parts.push(
+ `Still uploading — call memwal_remember_status again with job_ids=[${inFlight
+ .map((r) => r.id)
+ .join(", ")}]. Do not re-send those facts; that queues duplicates.`
+ );
+ }
+ if (failed.length) {
+ parts.push(
+ `NOT stored — these facts were never saved and must be sent again with ` +
+ `memwal_remember or memwal_remember_bulk.`
+ );
+ }
+ if (done.length) parts.push(explorerFooter());
+
+ return { content: [{ type: "text" as const, text: parts.join("\n\n") }] };
+}
+
function saved(blobId: string, namespace?: string) {
return {
content: [
diff --git a/services/server/scripts/mcp/tools/remember-wait.ts b/services/server/scripts/mcp/tools/remember-wait.ts
index 826399d85..f66f31a34 100644
--- a/services/server/scripts/mcp/tools/remember-wait.ts
+++ b/services/server/scripts/mcp/tools/remember-wait.ts
@@ -136,3 +136,47 @@ export function pendingMessage(jobId: string, waitedMs: number): string {
`the same fact with memwal_remember — that queues a second copy behind this one.`
);
}
+
+/** Longest fact echoed back in a pending listing. The line exists so the
+ * agent can tell which job_id belongs to which fact, not to reproduce the
+ * fact — and 20 of them at full length would crowd out the instructions
+ * underneath. */
+const PENDING_FACT_PREVIEW_CHARS = 80;
+
+function previewFact(text: string): string {
+ const flat = text.replace(/\s+/g, " ").trim();
+ return flat.length > PENDING_FACT_PREVIEW_CHARS
+ ? `${flat.slice(0, PENDING_FACT_PREVIEW_CHARS - 1)}…`
+ : flat;
+}
+
+/**
+ * The bulk counterpart of `pendingMessage`. Same contract — an agent must not
+ * read it as success — with the one addition bulk needs: each job_id is paired
+ * with the fact it carries, because "one of these five failed" is only
+ * actionable if the agent can tell which.
+ */
+export function pendingBulkMessage(
+ entries: Array<{ jobId: string; text: string }>,
+ waitedMs: number,
+): string {
+ const n = entries.length;
+ const opening =
+ waitedMs === 0
+ ? `ACCEPTED, NOT YET SAVED — the relayer has durably queued ${n} write(s).`
+ : `NOT SAVED YET — ${n} write(s) were accepted but had not finished after ${(waitedMs / 1000).toFixed(1)}s.`;
+
+ const lines = entries
+ .map((e, i) => `${i + 1}. job_id=${e.jobId} — ${previewFact(e.text)}`)
+ .join("\n");
+
+ return (
+ `${opening}\n${lines}\n` +
+ `Walrus uploads queue and are written one at a time per wallet, so a batch takes ` +
+ `longer than a single fact. Do NOT tell the user these facts are stored — say they ` +
+ `are being saved. Call memwal_remember_status with job_ids=[...] to get the blob_ids ` +
+ `once they land, or to learn that one failed; a job CAN fail after acceptance, and ` +
+ `this is the only way to find out. Do not re-send these facts with ` +
+ `memwal_remember_bulk — that queues a second copy behind them.`
+ );
+}
From 10ba9e1c5f2ab48bf0656e64ff760c16d787bfd6 Mon Sep 17 00:00:00 2001
From: Harry Phan
Date: Tue, 15 Sep 2026 21:50:36 +0700
Subject: [PATCH 010/132] fix(mcp): bound the relayer calls the SDK leaves
without a deadline
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit
memwal_remember was observed still running past 120s by an MCP client, on a
tool documented as capping at 90s. The cap is not a bound.
The SDK's signedRequest aborts a request only when the caller hands it a
signal, and of the memory methods only recall() does (15s). rememberAsync,
rememberBulkAsync and every job-status read call it with no signal, so the
underlying fetch has no deadline. waitForRememberJob then tests
`Date.now() < deadline` at the TOP of its poll loop, which bounds when the
next poll starts, not how long one takes — so a single stalled socket runs
as long as it stays open and sails straight past timeoutMs. With nothing
else in the way the stdio bridge's orphan sweeper is the first thing to fire,
minutes later, which is what a user sees as the client hanging.
Returning at accept does not fix this on its own: the accept POST is one of
the unbounded calls, so memwal_remember could hang indefinitely even at
MEMWAL_MCP_REMEMBER_WAIT_MS=0.
Every entry point now runs under a deadline — accepts at 15s (the only
deadline the SDK sets for itself; a healthy accept is ~1.1s), waits at their
own budget plus grace, since the budget cannot bound its own last poll.
The request is NOT cancelled: the SDK exposes no way to pass a signal, so
fetch keeps running until it settles. What this bounds is how long the agent
waits, which is the part a user experiences as a hang. An orphaned request
costs one socket and resolves into a promise nobody reads.
The timeout error is named MemWalRelayerUnresponsive rather than reusing a
job-failure name, because "we do not know whether it was queued" is not
"it failed" — and it says a retry is safe, since the SDK holds the same
idempotency key until an accept succeeds, so a retry attaches to the
existing job instead of queueing a second paid copy.
Also documents MEMWAL_MCP_REMEMBER_WAIT_MS, which this branch introduced
without an entry.
---
docs/reference/environment-variables.md | 2 +
.../mcp/__tests__/remember-deadline.test.ts | 149 ++++++++++++++++++
.../server/scripts/mcp/tools/remember-bulk.ts | 16 +-
.../scripts/mcp/tools/remember-status.ts | 36 +++--
.../server/scripts/mcp/tools/remember-wait.ts | 107 +++++++++++++
services/server/scripts/mcp/tools/remember.ts | 18 ++-
6 files changed, 307 insertions(+), 21 deletions(-)
create mode 100644 services/server/scripts/mcp/__tests__/remember-deadline.test.ts
diff --git a/docs/reference/environment-variables.md b/docs/reference/environment-variables.md
index 5ccc2af17..ce06284fc 100644
--- a/docs/reference/environment-variables.md
+++ b/docs/reference/environment-variables.md
@@ -155,6 +155,8 @@ These are not all enforced at boot, but most real deployments need them.
| `MCP_MAX_TOTAL_SESSIONS` | `1000` | Maximum active MCP sessions across SSE and Streamable HTTP transports |
| `MCP_MAX_SESSIONS_PER_IP` | `16` | Maximum active MCP sessions from one source IP |
| `MCP_MAX_NEW_SESSIONS_PER_IP_PER_MIN` | `30` | Maximum new MCP sessions opened by one source IP per minute |
+| `MEMWAL_MCP_REMEMBER_WAIT_MS` | `0` | How long `memwal_remember` / `memwal_remember_bulk` wait for a write to land before returning the job_id instead. `0` returns once the relayer has durably accepted the job (~1s). A value between the two is the worst of both — against a 30-75s completion spread it pays the wait and still returns pending — so either leave it at `0` or set it high enough to mean it. Clamped to `90000`; invalid values fall back to the default |
+| `MEMWAL_MCP_ACCEPT_DEADLINE_MS` | `15000` | How long a single relayer request may stall before the MCP tool gives up on it. The SDK passes an abort signal only on `recall`, so `rememberAsync` and the job-status reads have no deadline of their own and `MEMWAL_MCP_REMEMBER_WAIT_MS` bounds only when the next poll starts, not how long one takes — without this a stalled socket keeps a tool running indefinitely. Raise it only if a slow link makes healthy accepts exceed it; invalid or non-positive values fall back to the default |
| `MCP_TOOL_SLOW_WARN_MS` | `5000` | An MCP tool call still running at this duration is logged as `tool.slow` (`settled: false`) at `warn` — which is what makes a hang visible, since a hang never settles. A call that finishes at or above it is logged as `tool.slow` (`settled: true`) instead of `tool.done` |
| `TRUSTED_PROXY_HOPS` | `0` | Number of trusted reverse-proxy hops to walk from the right of `X-Forwarded-For`; `0` ignores XFF and uses the TCP peer |
| `WRITES_PAUSED` | `false` | When `1` / `true` / `yes` / `on`, write routes (`POST /api/remember`, `/api/remember/manual`, `/api/remember/bulk`, `/api/analyze`) return HTTP 503 `{"error":"writes are paused"}`. `GET /health` stays HTTP 200 with `status: "ok"` and `writes: "paused"`. Reads (`recall`, `restore`, health) stay available |
diff --git a/services/server/scripts/mcp/__tests__/remember-deadline.test.ts b/services/server/scripts/mcp/__tests__/remember-deadline.test.ts
new file mode 100644
index 000000000..c74d55030
--- /dev/null
+++ b/services/server/scripts/mcp/__tests__/remember-deadline.test.ts
@@ -0,0 +1,149 @@
+// Bound the SDK calls that have no deadline of their own. Set before the
+// module under test is imported — ACCEPT_DEADLINE_MS is read once at load.
+process.env.MEMWAL_MCP_ACCEPT_DEADLINE_MS = "150";
+
+import assert from "node:assert/strict";
+import test, { type TestContext } from "node:test";
+import { Client } from "@modelcontextprotocol/sdk/client/index.js";
+import { InMemoryTransport } from "@modelcontextprotocol/sdk/inMemory.js";
+import type { MemWalSession } from "../auth.js";
+
+// Dynamic, not static: ESM hoists every static import above the assignment
+// above, so the module would read the real 15s default before the line that
+// shortens it ever runs — and each tool test would then take 15 seconds.
+const { createMcpServer } = await import("../server.js");
+const { ACCEPT_DEADLINE_MS, withDeadline } = await import("../tools/remember-wait.js");
+
+/**
+ * The SDK's `signedRequest` aborts a request only when the caller passes a
+ * signal, and of the memory methods only `recall()` does. `rememberAsync`,
+ * `rememberBulkAsync` and every job-status poll call it with none, so `fetch`
+ * runs with no deadline. `timeoutMs` is checked at the top of the poll loop, so
+ * it bounds when the next poll STARTS, not how long one takes — which is how a
+ * tool documented as capping at 90s was observed still running past 120s.
+ *
+ * Returning at accept does not fix that by itself: the accept POST is one of
+ * the unbounded calls. These tests pin that every entry point is bounded, and
+ * that the resulting error never reads as a saved write.
+ */
+
+/** A promise that never settles — what a stalled socket looks like from here. */
+function hangs(): Promise {
+ return new Promise(() => {});
+}
+
+function sessionThatHangs(): MemWalSession {
+ return {
+ oauthScope: "memwal:read memwal:write",
+ namespace: "default",
+ memwal: {
+ rememberAsync: hangs,
+ rememberBulkAsync: hangs,
+ getRememberStatus: hangs,
+ getRememberBulkStatus: hangs,
+ waitForRememberJob: hangs,
+ waitForRememberJobs: hangs,
+ },
+ } as unknown as MemWalSession;
+}
+
+async function clientFor(session: MemWalSession, t: TestContext): Promise {
+ const [clientTransport, serverTransport] = InMemoryTransport.createLinkedPair();
+ const server = createMcpServer(session);
+ const client = new Client({ name: "remember-deadline-test", version: "1.0.0" });
+ t.after(async () => {
+ await client.close();
+ await server.close();
+ });
+ await server.connect(serverTransport);
+ await client.connect(clientTransport);
+ return client;
+}
+
+function textOf(result: unknown): string {
+ return (result as { content: Array<{ text: string }> }).content
+ .map((c) => c.text)
+ .join("\n");
+}
+
+test("the accept deadline is read from the environment and validated", () => {
+ assert.equal(ACCEPT_DEADLINE_MS, 150);
+});
+
+test("withDeadline passes a value through untouched when work finishes first", async () => {
+ assert.equal(await withDeadline(Promise.resolve("ok"), 1_000, "nope"), "ok");
+});
+
+test("withDeadline rejects with a named error once the deadline passes", async () => {
+ await assert.rejects(
+ withDeadline(hangs(), 20, "relayer went quiet"),
+ (err: Error) => {
+ // `wrapTool` routes on the name, so it has to be distinct from a
+ // job failure — this is "we do not know", not "it failed".
+ assert.equal(err.name, "MemWalRelayerUnresponsive");
+ assert.match(err.message, /relayer went quiet/);
+ return true;
+ },
+ );
+});
+
+test("withDeadline does not reject once the work has already resolved", async () => {
+ // A leftover timer firing after resolution would reject a promise nobody
+ // is racing any more, surfacing as an unhandled rejection.
+ const value = await withDeadline(Promise.resolve(1), 10, "nope");
+ assert.equal(value, 1);
+ await new Promise((r) => setTimeout(r, 40));
+});
+
+test("memwal_remember cannot hang forever on a stalled accept", async (t) => {
+ const client = await clientFor(sessionThatHangs(), t);
+
+ const result = await client.callTool({
+ name: "memwal_remember",
+ arguments: { text: "a durable fact" },
+ });
+
+ assert.equal((result as { isError?: boolean }).isError, true);
+ const text = textOf(result);
+ // Must not read as stored, and must say a retry is safe rather than
+ // leaving the agent to guess (a blind retry would risk a second paid copy).
+ assert.ok(!/^Saved to Walrus Memory/m.test(text), `read as saved: ${text}`);
+ assert.match(text, /did not accept/);
+ assert.match(text, /[Rr]etry/);
+});
+
+test("memwal_remember_bulk cannot hang forever on a stalled accept", async (t) => {
+ const client = await clientFor(sessionThatHangs(), t);
+
+ const result = await client.callTool({
+ name: "memwal_remember_bulk",
+ arguments: { facts: ["one", "two"] },
+ });
+
+ assert.equal((result as { isError?: boolean }).isError, true);
+ assert.match(textOf(result), /did not accept/);
+});
+
+test("memwal_remember_status cannot hang forever on a stalled read", async (t) => {
+ const client = await clientFor(sessionThatHangs(), t);
+
+ const result = await client.callTool({
+ name: "memwal_remember_status",
+ arguments: { job_id: "job-1", waitMs: 0 },
+ });
+
+ assert.equal((result as { isError?: boolean }).isError, true);
+ assert.match(textOf(result), /did not accept/);
+});
+
+test("a stalled batch status read is bounded too", async (t) => {
+ const client = await clientFor(sessionThatHangs(), t);
+
+ const result = await client.callTool({
+ name: "memwal_remember_status",
+ arguments: { job_ids: ["job-1", "job-2"], waitMs: 0 },
+ });
+
+ assert.equal((result as { isError?: boolean }).isError, true);
+ assert.match(textOf(result), /did not accept/);
+});
diff --git a/services/server/scripts/mcp/tools/remember-bulk.ts b/services/server/scripts/mcp/tools/remember-bulk.ts
index 01c48a1a9..3c115fd9e 100644
--- a/services/server/scripts/mcp/tools/remember-bulk.ts
+++ b/services/server/scripts/mcp/tools/remember-bulk.ts
@@ -7,6 +7,8 @@ import {
REMEMBER_POLL_INTERVAL_MS,
REMEMBER_WAIT_MS,
pendingBulkMessage,
+ withAcceptDeadline,
+ withWaitDeadline,
} from "./remember-wait.js";
const REMEMBER_BULK_INPUT = {
@@ -58,7 +60,10 @@ export function registerRememberBulkTool(
// Two steps rather than `rememberBulkAndWait`, for the same reason
// `memwal_remember` splits them: acceptance is the part that must
// succeed, the wait is a courtesy we cut short.
- const accepted = await session.memwal.rememberBulkAsync(items);
+ const accepted = await withAcceptDeadline(
+ session.memwal.rememberBulkAsync(items),
+ "memwal_remember_bulk batch",
+ );
// Pair each job with its fact up front. Every later branch needs
// it, and the relayer returns job_ids in input order.
@@ -85,13 +90,12 @@ export function registerRememberBulkTool(
// Unlike `waitForRememberJob`, this never throws on expiry — it
// reports the stragglers as `timeout` per item, so a batch can come
// back part landed and part still in flight.
- const result = await session.memwal.waitForRememberJobs(
- accepted.job_ids,
- namespaces,
- {
+ const result = await withWaitDeadline(
+ session.memwal.waitForRememberJobs(accepted.job_ids, namespaces, {
timeoutMs: REMEMBER_WAIT_MS,
pollIntervalMs: REMEMBER_POLL_INTERVAL_MS,
- }
+ }),
+ REMEMBER_WAIT_MS,
);
const waitedMs = Date.now() - startedAt;
diff --git a/services/server/scripts/mcp/tools/remember-status.ts b/services/server/scripts/mcp/tools/remember-status.ts
index abefa6507..97d4adcb8 100644
--- a/services/server/scripts/mcp/tools/remember-status.ts
+++ b/services/server/scripts/mcp/tools/remember-status.ts
@@ -7,6 +7,8 @@ import {
REMEMBER_POLL_INTERVAL_MS,
isStillRunning,
nameJobError,
+ withAcceptDeadline,
+ withWaitDeadline,
} from "./remember-wait.js";
/**
@@ -96,7 +98,10 @@ export function registerRememberStatusTool(
// sleeps before its first poll, so a 0ms deadline would
// return "still running" without ever asking the relayer.
if (budget === 0) {
- const status = await session.memwal.getRememberStatus(job_id);
+ const status = await withAcceptDeadline(
+ session.memwal.getRememberStatus(job_id),
+ "status read",
+ );
if (status.status === "done") {
return saved(status.blob_id ?? "", status.namespace);
}
@@ -122,10 +127,13 @@ export function registerRememberStatusTool(
}
try {
- const result = await session.memwal.waitForRememberJob(job_id, {
- timeoutMs: budget,
- pollIntervalMs: REMEMBER_POLL_INTERVAL_MS,
- });
+ const result = await withWaitDeadline(
+ session.memwal.waitForRememberJob(job_id, {
+ timeoutMs: budget,
+ pollIntervalMs: REMEMBER_POLL_INTERVAL_MS,
+ }),
+ budget,
+ );
return saved(result.blob_id, result.namespace);
} catch (err) {
if (isStillRunning(err)) return stillRunning(job_id);
@@ -154,17 +162,25 @@ async function settleBatch(
// deadline would report everything as still running without ever asking.
const rows =
budgetMs === 0
- ? (await session.memwal.getRememberBulkStatus(jobIds)).results.map((r) => ({
+ ? (
+ await withAcceptDeadline(
+ session.memwal.getRememberBulkStatus(jobIds),
+ "batch status read",
+ )
+ ).results.map((r) => ({
id: r.job_id,
status: r.status,
blob_id: r.blob_id ?? "",
error: r.error,
}))
: (
- await session.memwal.waitForRememberJobs(jobIds, [], {
- timeoutMs: budgetMs,
- pollIntervalMs: REMEMBER_POLL_INTERVAL_MS,
- })
+ await withWaitDeadline(
+ session.memwal.waitForRememberJobs(jobIds, [], {
+ timeoutMs: budgetMs,
+ pollIntervalMs: REMEMBER_POLL_INTERVAL_MS,
+ }),
+ budgetMs,
+ )
).results.map((r) => ({
id: r.id,
status: r.status,
diff --git a/services/server/scripts/mcp/tools/remember-wait.ts b/services/server/scripts/mcp/tools/remember-wait.ts
index f66f31a34..2f2fef4e6 100644
--- a/services/server/scripts/mcp/tools/remember-wait.ts
+++ b/services/server/scripts/mcp/tools/remember-wait.ts
@@ -180,3 +180,110 @@ export function pendingBulkMessage(
`memwal_remember_bulk — that queues a second copy behind them.`
);
}
+
+/**
+ * Longest we let the relayer take to ACCEPT a write before giving up on it.
+ *
+ * The SDK's `signedRequest` only aborts a request when the caller hands it a
+ * signal, and of the memory methods only `recall()` does (15s). `rememberAsync`
+ * and every job-status poll call it with no signal at all, so the underlying
+ * `fetch` has no deadline of its own. That makes `timeoutMs` a loop-entry
+ * check rather than a bound: `waitForRememberJob` tests `Date.now() < deadline`
+ * at the top of each iteration, so one stalled HTTP request runs as long as the
+ * socket stays open and a tool documented as capping at 90s is observed past
+ * 120s. Returning at accept does not fix that on its own — the accept POST is
+ * exactly one of the unbounded calls.
+ *
+ * 15s matches the only deadline the SDK sets for itself. A healthy accept is
+ * ~1.1s, so this fires only when something is genuinely wrong.
+ */
+const DEFAULT_ACCEPT_DEADLINE_MS = 15_000;
+
+/**
+ * Read once, validated the same way as the wait budget: a typo must not
+ * silently pick a deadline nobody asked for. Exposed as an env knob because an
+ * operator on a slow link is the one person who can tell a hung relayer from a
+ * merely distant one.
+ */
+export const ACCEPT_DEADLINE_MS = (() => {
+ const raw = process.env.MEMWAL_MCP_ACCEPT_DEADLINE_MS;
+ if (raw === undefined || raw.trim() === "") return DEFAULT_ACCEPT_DEADLINE_MS;
+ const parsed = Number(raw);
+ if (!Number.isFinite(parsed) || parsed <= 0) {
+ log.warn("remember.accept_deadline_invalid", {
+ value: raw,
+ usingMs: DEFAULT_ACCEPT_DEADLINE_MS,
+ });
+ return DEFAULT_ACCEPT_DEADLINE_MS;
+ }
+ return Math.floor(parsed);
+})();
+
+/**
+ * Grace added to a wait budget before we stop believing the SDK will return.
+ *
+ * The budget bounds when the SDK starts its last poll, not when that poll
+ * finishes, so a stalled request can overshoot by an unbounded amount. This
+ * caps the overshoot instead.
+ */
+const WAIT_OVERSHOOT_GRACE_MS = 10_000;
+
+class DeadlineExceededError extends Error {
+ constructor(message: string) {
+ super(message);
+ this.name = "MemWalRelayerUnresponsive";
+ }
+}
+
+/**
+ * Bound an SDK call that has no deadline of its own.
+ *
+ * The underlying request is NOT cancelled — the SDK gives us no way to pass a
+ * signal, so `fetch` keeps running until it settles or the socket dies. What
+ * this bounds is how long the agent waits on it, which is the part the user
+ * experiences as a hang. The orphaned request costs one socket and resolves
+ * into a promise nobody reads.
+ */
+export async function withDeadline(
+ work: Promise,
+ ms: number,
+ message: string,
+): Promise {
+ let timer: NodeJS.Timeout | undefined;
+ try {
+ return await Promise.race([
+ work,
+ new Promise((_, reject) => {
+ timer = setTimeout(() => reject(new DeadlineExceededError(message)), ms);
+ // Never hold the process open for a deadline nobody is waiting on.
+ timer.unref?.();
+ }),
+ ]);
+ } finally {
+ if (timer) clearTimeout(timer);
+ }
+}
+
+/** Bound the accept leg of a write. */
+export function withAcceptDeadline(work: Promise, what: string): Promise {
+ return withDeadline(
+ work,
+ ACCEPT_DEADLINE_MS,
+ `Walrus Memory did not accept the ${what} within ${ACCEPT_DEADLINE_MS / 1000}s — the ` +
+ `relayer is unreachable or not responding. The write may or may not have been ` +
+ `queued, so do NOT tell the user it was saved. Retrying ${what} in this session ` +
+ `is safe: the SDK reuses the same idempotency key until an accept succeeds, so a ` +
+ `retry attaches to the existing job instead of queueing a second paid copy.`,
+ );
+}
+
+/** Bound a status wait at its own budget plus the overshoot grace. */
+export function withWaitDeadline(work: Promise, budgetMs: number): Promise {
+ return withDeadline(
+ work,
+ budgetMs + WAIT_OVERSHOOT_GRACE_MS,
+ `Walrus Memory stopped responding while waiting for the write to land. The job is ` +
+ `still queued relayer-side — do NOT tell the user it was saved, and do NOT re-send ` +
+ `the fact. Call memwal_remember_status with the job_id to settle it.`,
+ );
+}
diff --git a/services/server/scripts/mcp/tools/remember.ts b/services/server/scripts/mcp/tools/remember.ts
index 9e6ba952e..74611cc23 100644
--- a/services/server/scripts/mcp/tools/remember.ts
+++ b/services/server/scripts/mcp/tools/remember.ts
@@ -9,6 +9,8 @@ import {
isStillRunning,
nameJobError,
pendingMessage,
+ withAcceptDeadline,
+ withWaitDeadline,
} from "./remember-wait.js";
const REMEMBER_INPUT = {
@@ -57,7 +59,10 @@ export function registerRememberTool(
// Two steps rather than `rememberAndWait`, because the accept and
// the wait need separate budgets: acceptance is the part that
// must succeed, the wait is a courtesy we cut short.
- const accepted = await session.memwal.rememberAsync(text, namespace);
+ const accepted = await withAcceptDeadline(
+ session.memwal.rememberAsync(text, namespace),
+ "memwal_remember write",
+ );
const pending = (waitedMs: number) => ({
content: [
@@ -74,10 +79,13 @@ export function registerRememberTool(
const startedAt = Date.now();
try {
- const result = await session.memwal.waitForRememberJob(accepted.job_id, {
- timeoutMs: REMEMBER_WAIT_MS,
- pollIntervalMs: REMEMBER_POLL_INTERVAL_MS,
- });
+ const result = await withWaitDeadline(
+ session.memwal.waitForRememberJob(accepted.job_id, {
+ timeoutMs: REMEMBER_WAIT_MS,
+ pollIntervalMs: REMEMBER_POLL_INTERVAL_MS,
+ }),
+ REMEMBER_WAIT_MS,
+ );
return {
content: [
{
From 9385153ed894a56d60d82ba3e841c7ae42c6c12b Mon Sep 17 00:00:00 2001
From: Harry Phan
Date: Tue, 15 Sep 2026 22:04:18 +0700
Subject: [PATCH 011/132] fix(sdk): give every relayer request a deadline
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit
`fetch` has no timeout of its own, and this SDK passed an abort signal on
exactly one method — `recall`, at 15s. The accept POST, every job-status
read, and the `/version` and `/config` handshake calls that run before any
of them could stay pending for as long as the socket stayed open.
That is not merely untidy. A poll loop tests its budget at the TOP of each
iteration, so it bounds when the next request starts, not how long one
takes: a single stalled read ran straight past `timeoutMs`. A
`memwal_remember` documented as capping at 90s was seen by an MCP client
still running after 120s, with the stdio bridge's orphan sweeper the first
thing to fire, minutes later. The MCP layer now wraps its own calls, but
that only covers one consumer — the hole is here.
Default 30s, matching the relayer's own outbound HTTP client: any call that
needs the relayer to reach the sidecar, Walrus or OpenAI has already failed
upstream by the time it fires. Settable via `requestTimeoutMs`; a
non-positive or non-finite value falls back to the default rather than
disabling the bound, since "no deadline" is the bug being fixed.
Two endpoints legitimately outrun it and say so: `restore` (60s — the route
bounds itself at 55s server-side and answers with an error rather than going
quiet, so a tighter client deadline would abandon a reply already on its
way) and `analyze` (60s — it runs the extractor LLM inline before it
accepts). `recall` keeps 15s, now a named constant instead of a hand-rolled
AbortController.
Each poll inside a wait loop is bounded by the client deadline clamped to
the remaining budget. Both directions carry weight: the remaining budget
stops a poll outliving the wait it belongs to, and the client deadline stops
ONE stalled poll swallowing the whole budget, which would leave the loop no
room to retry. Expiry raises `MemWalRequestTimeout` with `status: 504` —
already transient to `isTransientPollingStatus` — so a stalled poll is
retried against what is left instead of failing the wait.
An abort the caller asked for, and any other transport error, propagates
untouched: an operator debugging DNS or TLS needs the original error, not
"timed out".
---
.changeset/sdk-bound-every-relayer-request.md | 11 +
packages/sdk/src/memwal.ts | 218 ++++++++++++++++--
packages/sdk/src/types.ts | 12 +
packages/sdk/test/request-timeout.test.mjs | 174 ++++++++++++++
.../server/scripts/mcp/tools/remember-wait.ts | 5 +
5 files changed, 395 insertions(+), 25 deletions(-)
create mode 100644 .changeset/sdk-bound-every-relayer-request.md
create mode 100644 packages/sdk/test/request-timeout.test.mjs
diff --git a/.changeset/sdk-bound-every-relayer-request.md b/.changeset/sdk-bound-every-relayer-request.md
new file mode 100644
index 000000000..83b7aca27
--- /dev/null
+++ b/.changeset/sdk-bound-every-relayer-request.md
@@ -0,0 +1,11 @@
+---
+"@mysten-incubation/memwal": patch
+---
+
+Give every relayer request a deadline. `fetch` has none of its own, and the SDK passed an abort signal on exactly one method (`recall`, 15s) — so the accept POST, every job-status read, and the `/version` and `/config` handshake calls that run before any of them could stay pending for as long as the socket stayed open.
+
+That was not merely untidy. A poll loop checks its budget at the *top* of each iteration, which bounds when the next request starts, not how long one takes — so a single stalled read ran straight past `timeoutMs`. A `memwal_remember` documented as capping at 90s was observed by an MCP client still running after 120s, with the stdio bridge's orphan sweeper the first thing to fire, minutes later.
+
+Requests now default to a 30s deadline, matching the relayer's own outbound HTTP client: any call that depends on the relayer reaching the sidecar, Walrus or OpenAI has already failed upstream by the time it fires. Configure it with `requestTimeoutMs` on `MemWal.create`; a non-positive or non-finite value falls back to the default rather than disabling the bound. The two endpoints that legitimately run longer carry their own: `restore` (60s — the route bounds itself at 55s server-side and answers rather than going quiet) and `analyze` (60s — it runs the extractor LLM inline before accepting). `recall` keeps its 15s, now as a named constant rather than a hand-rolled `AbortController`.
+
+Inside the wait loops each poll is bounded by the client deadline clamped to the remaining budget. Both directions matter: the remaining budget stops a poll outliving the wait it belongs to, and the client deadline stops one stalled poll swallowing the whole budget, so the loop still gets to retry. An expired request raises `MemWalRequestTimeout` carrying `status: 504`, which `isTransientPollingStatus` already treats as retryable — so a stalled poll is retried against what is left rather than failing the wait outright. A caller's own abort, and any other transport error, propagates unchanged.
diff --git a/packages/sdk/src/memwal.ts b/packages/sdk/src/memwal.ts
index 1a32b2a53..f0b4e313d 100644
--- a/packages/sdk/src/memwal.ts
+++ b/packages/sdk/src/memwal.ts
@@ -115,6 +115,15 @@ const SEAL_SESSION_TTL_MIN = 5;
// a key server that sees it as expired.
const SEAL_SESSION_SAFETY_MARGIN_MS = 30_000;
+/** Per-call knobs for `signedRequest`. `timeoutMs` overrides the client-wide
+ * deadline for one endpoint; `signal` is the caller's own cancellation and is
+ * honoured alongside it, never replaced by it. */
+interface SignedRequestOptions {
+ includeDelegateKey?: boolean;
+ signal?: AbortSignal;
+ timeoutMs?: number;
+}
+
type RememberStatusResponse = RememberJobStatus | { error?: string };
function sleep(ms: number): Promise {
@@ -171,6 +180,89 @@ async function derivedIdempotencyKey(requestIdentity: string): Promise {
return `r1-${await sha256hex(`${bucket}\0${requestIdentity}`)}`;
}
+/**
+ * Deadline for a single relayer request when the call site names no other.
+ *
+ * `fetch` imposes no timeout, so before this every request here could hang for
+ * as long as the socket stayed open. That is not a theoretical gap: a poll loop
+ * checks its budget at the top of each iteration, which bounds when the next
+ * request STARTS, not how long one takes — so one stalled read blew straight
+ * past a documented 90s cap and left an MCP tool running past 120s.
+ *
+ * 30s mirrors the relayer's own outbound HTTP client, so any call that depends
+ * on the relayer talking to the sidecar, Walrus, or OpenAI has already failed
+ * upstream by the time this fires.
+ */
+const DEFAULT_REQUEST_TIMEOUT_MS = 30_000;
+
+/** `POST /api/restore` bounds itself at 55s server-side and answers with an
+ * error rather than going quiet, so the client must outlast that or it would
+ * abandon a response already on its way. */
+const RESTORE_REQUEST_TIMEOUT_MS = 60_000;
+
+/** `POST /api/analyze` runs the extractor LLM inline before it accepts, so it
+ * is the one write that legitimately outruns the default. */
+const ANALYZE_REQUEST_TIMEOUT_MS = 60_000;
+
+/** `POST /api/recall` has carried its own 15s deadline since before the rest
+ * had any; keeping it named makes that a decision rather than an accident. */
+const RECALL_REQUEST_TIMEOUT_MS = 15_000;
+
+/**
+ * Abort signal that fires after `ms`, or when `caller` aborts — whichever is
+ * first. Built by hand rather than with `AbortSignal.any`, which lands too
+ * recently to rely on across Node, browsers and Workers alike.
+ *
+ * Returns the signal plus a `dispose` the caller must run in a `finally`, so a
+ * pending timer never outlives its request.
+ */
+function deadlineSignal(
+ ms: number,
+ caller?: AbortSignal,
+): { signal: AbortSignal; timedOut: () => boolean; dispose: () => void } {
+ const controller = new AbortController();
+ let expired = false;
+
+ const onCallerAbort = () => controller.abort(caller?.reason);
+ if (caller) {
+ if (caller.aborted) controller.abort(caller.reason);
+ else caller.addEventListener("abort", onCallerAbort, { once: true });
+ }
+
+ const timer = setTimeout(() => {
+ expired = true;
+ controller.abort();
+ }, ms);
+ // Never hold a process open for a deadline nobody is waiting on.
+ (timer as unknown as { unref?: () => void }).unref?.();
+
+ return {
+ signal: controller.signal,
+ timedOut: () => expired,
+ dispose: () => {
+ clearTimeout(timer);
+ caller?.removeEventListener("abort", onCallerAbort);
+ },
+ };
+}
+
+/**
+ * The error a request deadline produces.
+ *
+ * `status: 504` is load-bearing, not decoration: `isTransientPollingStatus`
+ * treats it as retryable, so one stalled poll inside a wait loop is abandoned
+ * and retried against the remaining budget instead of failing the whole wait.
+ */
+function requestTimeoutError(method: string, path: string, ms: number): Error {
+ const err = new Error(
+ `Walrus Memory request timed out after ${ms}ms (${method} ${path}). The relayer ` +
+ `accepted the connection but did not answer in time.`,
+ );
+ err.name = "MemWalRequestTimeout";
+ (err as Error & { status?: number }).status = 504;
+ return err;
+}
+
function isTransientPollingStatus(status: number): boolean {
return status === 0 || status === 429 || status >= 500;
}
@@ -228,6 +320,8 @@ export class MemWal {
private serverUrl: string;
private namespace: string;
private accountId: string;
+ /** Deadline applied to any request that does not name its own. */
+ private requestTimeoutMs: number;
// ENG-1697 state — all internal, never surfaced to user code.
// The public API (`MemWal.create({ key, accountId })`) is unchanged.
@@ -262,6 +356,12 @@ export class MemWal {
// non-localhost host.
this.serverUrl = normalizeServerUrl(config.serverUrl ?? "https://relayer.memory.walrus.xyz");
this.namespace = config.namespace ?? "default";
+ // A non-positive or non-finite override would disable the backstop
+ // entirely, which is the bug this exists to prevent.
+ this.requestTimeoutMs =
+ Number.isFinite(config.requestTimeoutMs) && (config.requestTimeoutMs as number) > 0
+ ? (config.requestTimeoutMs as number)
+ : DEFAULT_REQUEST_TIMEOUT_MS;
}
/**
@@ -379,6 +479,17 @@ export class MemWal {
`/api/remember/${jobId}`,
{},
[200, 404],
+ // Bound each poll by the client deadline, and never past
+ // what is left of the budget. Without this the loop only
+ // checks the deadline between polls, so one stalled read
+ // runs past `timeoutMs` however small it was — the whole
+ // reason a 90s wait was seen still going at 120s. Taking
+ // the min of the two matters in both directions: the
+ // remaining budget keeps a poll from outliving the wait,
+ // and the client deadline keeps ONE stalled poll from
+ // swallowing the entire budget, so the loop still gets to
+ // retry. An expired poll surfaces as a transient 504.
+ { timeoutMs: this.pollDeadlineMs(deadline) },
);
} catch (err) {
const httpStatus = (err as { status?: number }).status ?? 0;
@@ -508,11 +619,16 @@ export class MemWal {
return accepted;
}
- async getRememberBulkStatus(jobIds: string[]): Promise {
+ async getRememberBulkStatus(
+ jobIds: string[],
+ opts: { timeoutMs?: number } = {},
+ ): Promise {
return this.signedRequest(
"POST",
"/api/remember/bulk/status",
{ job_ids: jobIds },
+ [200],
+ { timeoutMs: opts.timeoutMs },
);
}
@@ -546,7 +662,10 @@ export class MemWal {
let batchStatus: RememberBulkStatusResult;
try {
- batchStatus = await this.getRememberBulkStatus(pendingIds);
+ // Bounded the same way as the single-job poll.
+ batchStatus = await this.getRememberBulkStatus(pendingIds, {
+ timeoutMs: this.pollDeadlineMs(deadline),
+ });
} catch (err) {
const httpStatus = (err as { status?: number }).status ?? 0;
if (isTransientPollingStatus(httpStatus)) {
@@ -716,9 +835,7 @@ export class MemWal {
const limit = options.topK ?? options.limit ?? 10;
const resolvedNamespace = options.namespace ?? this.namespace;
- const ac = new AbortController();
- const tid = setTimeout(() => ac.abort(), 15000);
- try {
+ {
const result = await this.signedRequest("POST", "/api/recall", {
query,
limit,
@@ -732,7 +849,7 @@ export class MemWal {
// request byte-identical and the relayer applies its own
// "relevance" default.
sort: options.sort,
- }, { signal: ac.signal });
+ }, { timeoutMs: RECALL_REQUEST_TIMEOUT_MS });
let processed = result;
if (typeof options.maxDistance === "number") {
@@ -766,8 +883,6 @@ export class MemWal {
}
return processed;
- } finally {
- clearTimeout(tid);
}
}
@@ -893,7 +1008,9 @@ export class MemWal {
};
const wireOccurredAt = occurredAtToWire(options.occurredAt);
if (wireOccurredAt !== undefined) body.occurred_at = wireOccurredAt;
- return this.signedRequest("POST", "/api/analyze", body, [200, 202]);
+ return this.signedRequest("POST", "/api/analyze", body, [200, 202], {
+ timeoutMs: ANALYZE_REQUEST_TIMEOUT_MS,
+ });
}
/**
@@ -962,10 +1079,13 @@ export class MemWal {
* ```
*/
async restore(namespace: string, limit: number = 10): Promise {
- const result = await this.signedRequest("POST", "/api/restore", {
- namespace,
- limit,
- });
+ const result = await this.signedRequest(
+ "POST",
+ "/api/restore",
+ { namespace, limit },
+ [200],
+ { timeoutMs: RESTORE_REQUEST_TIMEOUT_MS },
+ );
// Relayers older than WALM-319 omit `truncated` entirely — treat
// "not present" as "not known to be truncated" rather than drop
// the field or require every relayer version to send it.
@@ -1065,7 +1185,7 @@ export class MemWal {
* Check server health. The endpoint is public and does not require request signing.
*/
async health(): Promise {
- const res = await fetch(`${this.serverUrl}/health`);
+ const res = await this.fetchWithDeadline(`${this.serverUrl}/health`);
if (!res.ok) {
throw new Error(`Health check failed: ${res.status}`);
}
@@ -1110,13 +1230,13 @@ export class MemWal {
}
private async fetchCompatibilityMetadata(): Promise {
- const versionRes = await fetch(`${this.serverUrl}/version`, { method: "GET" });
+ const versionRes = await this.fetchWithDeadline(`${this.serverUrl}/version`, { method: "GET" });
let body: Partial;
if (versionRes.ok) {
body = (await versionRes.json()) as Partial;
} else if (versionRes.status === 404 || versionRes.status === 405) {
- const healthRes = await fetch(`${this.serverUrl}/health`, { method: "GET" });
+ const healthRes = await this.fetchWithDeadline(`${this.serverUrl}/health`, { method: "GET" });
if (!healthRes.ok) {
throw new Error(
`Walrus Memory compatibility check failed: GET /version returned ` +
@@ -1159,7 +1279,7 @@ export class MemWal {
private async fetchServerConfig(): Promise {
if (this.serverConfig) return this.serverConfig;
- const res = await fetch(`${this.serverUrl}/config`, { method: "GET" });
+ const res = await this.fetchWithDeadline(`${this.serverUrl}/config`, { method: "GET" });
if (!res.ok) {
throw new Error(`GET /config returned ${res.status}`);
}
@@ -1372,12 +1492,45 @@ export class MemWal {
* @param acceptedStatuses - HTTP status codes to treat as success (default [200]).
* Pass [200, 202] for endpoints that return 202 Accepted.
*/
+ /**
+ * `fetch` with this client's deadline applied.
+ *
+ * For the handshake endpoints that skip request signing — `/health`,
+ * `/version`, `/config`. They are the worst place to leave unbounded: the
+ * compatibility probe and config fetch run before the first real call, so
+ * one stalled socket there hangs every method on the client, not just one.
+ */
+ /** Deadline for one status poll: the client deadline, clamped so it can
+ * never outlive the wait budget it belongs to. */
+ private pollDeadlineMs(deadline: number): number {
+ return Math.max(1, Math.min(this.requestTimeoutMs, deadline - Date.now()));
+ }
+
+ private async fetchWithDeadline(
+ url: string,
+ init: RequestInit = {},
+ timeoutMs?: number,
+ ): Promise {
+ const ms = timeoutMs ?? this.requestTimeoutMs;
+ const deadline = deadlineSignal(ms);
+ try {
+ return await fetch(url, { ...init, signal: deadline.signal });
+ } catch (err) {
+ if (deadline.timedOut()) {
+ throw requestTimeoutError(init.method ?? "GET", url, ms);
+ }
+ throw err;
+ } finally {
+ deadline.dispose();
+ }
+ }
+
private async signedRequest(
method: string,
path: string,
body: object,
- acceptedStatusesOrOptions: number[] | { includeDelegateKey?: boolean; signal?: AbortSignal } = [200],
- requestOptions: { includeDelegateKey?: boolean; signal?: AbortSignal } = {},
+ acceptedStatusesOrOptions: number[] | SignedRequestOptions = [200],
+ requestOptions: SignedRequestOptions = {},
): Promise {
const acceptedStatuses = Array.isArray(acceptedStatusesOrOptions)
? acceptedStatusesOrOptions
@@ -1424,12 +1577,27 @@ export class MemWal {
if (options.includeDelegateKey !== false) {
headers["x-seal-session"] = await this.buildSealSession();
}
- const res = await fetch(url, {
- method,
- headers,
- body: method === "GET" ? undefined : bodyStr,
- signal: options.signal,
- });
+ // Bound the request. `fetch` never times out on its own, so this is the
+ // only thing standing between a stalled socket and a call that hangs
+ // for as long as the connection stays open.
+ const deadlineMs = options.timeoutMs ?? this.requestTimeoutMs;
+ const deadline = deadlineSignal(deadlineMs, options.signal);
+ let res: Response;
+ try {
+ res = await fetch(url, {
+ method,
+ headers,
+ body: method === "GET" ? undefined : bodyStr,
+ signal: deadline.signal,
+ });
+ } catch (err) {
+ // Translate our own expiry into something a caller can classify.
+ // An abort the CALLER asked for is theirs and propagates untouched.
+ if (deadline.timedOut()) throw requestTimeoutError(method, path, deadlineMs);
+ throw err;
+ } finally {
+ deadline.dispose();
+ }
if (!acceptedStatuses.includes(res.status)) {
// LOW-26: sanitize server error bodies before surfacing to callers.
diff --git a/packages/sdk/src/types.ts b/packages/sdk/src/types.ts
index e6b374c50..e5824f8f2 100644
--- a/packages/sdk/src/types.ts
+++ b/packages/sdk/src/types.ts
@@ -21,6 +21,18 @@ export interface MemWalConfig {
serverUrl?: string;
/** Default namespace for memory isolation (default: "default") */
namespace?: string;
+ /**
+ * Deadline for a single relayer request, in milliseconds (default: 30000).
+ *
+ * `fetch` has no timeout of its own, so without this a stalled connection
+ * keeps a call pending for as long as the socket stays open — which is how
+ * a `remember` whose poll budget was 90s could still be running after two
+ * minutes. Endpoints that are legitimately slower (`restore`, `analyze`)
+ * carry their own larger deadline and ignore this.
+ *
+ * Raise it only for a genuinely slow link; it is a backstop, not a budget.
+ */
+ requestTimeoutMs?: number;
}
// ============================================================
diff --git a/packages/sdk/test/request-timeout.test.mjs b/packages/sdk/test/request-timeout.test.mjs
new file mode 100644
index 000000000..1ceaa700a
--- /dev/null
+++ b/packages/sdk/test/request-timeout.test.mjs
@@ -0,0 +1,174 @@
+import assert from "node:assert/strict";
+import test from "node:test";
+
+import { MemWal } from "../dist/memwal.js";
+
+const originalFetch = globalThis.fetch;
+
+test.afterEach(() => {
+ globalThis.fetch = originalFetch;
+});
+
+/**
+ * `fetch` has no timeout of its own, and the SDK used to pass a signal on
+ * exactly one method (`recall`). Everything else — the accept POST, every job
+ * status read, and the `/version` and `/config` handshake calls that run before
+ * any of them — could stay pending for as long as the socket stayed open.
+ *
+ * A poll loop only checks its budget between polls, so that was not merely
+ * untidy: one stalled read ran straight past `timeoutMs`, which is how a
+ * `memwal_remember` documented as capping at 90s was seen still running after
+ * 120s by an MCP client.
+ */
+
+/** Stands in for a stalled socket: never answers, but honours abort the way a
+ * real `fetch` does — which is also what proves the signal reaches it. */
+function hangUntilAborted(init = {}) {
+ return new Promise((_, reject) => {
+ const signal = init.signal;
+ if (!signal) return; // no signal reaching fetch => hangs forever => test times out
+ if (signal.aborted) return reject(abortError());
+ signal.addEventListener("abort", () => reject(abortError()), { once: true });
+ });
+}
+
+function abortError() {
+ const err = new Error("This operation was aborted");
+ err.name = "AbortError";
+ return err;
+}
+
+/** Stub relayer whose handshake always answers; `onApi` decides the rest. */
+function stubRelayer(onApi) {
+ globalThis.fetch = async (url, init = {}) => {
+ const path = new URL(url).pathname;
+ if (path === "/version") {
+ return Response.json({
+ apiVersion: "1.0.0",
+ relayerVersion: "1.0.0",
+ minSupportedSdk: { typescript: "0.0.4" },
+ });
+ }
+ if (path === "/config") {
+ return Response.json({ packageId: "0x1", network: "testnet" });
+ }
+ return onApi(path, init, url);
+ };
+}
+
+function clientWith(extra = {}) {
+ const client = MemWal.create({
+ key: new Uint8Array(32).fill(1),
+ accountId: "0x1",
+ serverUrl: "https://relayer.example",
+ ...extra,
+ });
+ client.buildSealSession = async () => "test-session";
+ return client;
+}
+
+test("a stalled accept is bounded instead of hanging forever", async () => {
+ stubRelayer((path, init) =>
+ path === "/api/remember" ? hangUntilAborted(init) : Promise.reject(new Error(path)),
+ );
+
+ const started = Date.now();
+ await assert.rejects(
+ clientWith({ requestTimeoutMs: 120 }).rememberAsync("a durable fact"),
+ (err) => {
+ assert.equal(err.name, "MemWalRequestTimeout");
+ // 504 is load-bearing: `isTransientPollingStatus` treats it as
+ // retryable, so a stalled poll inside a wait loop is retried
+ // against the remaining budget rather than failing the whole wait.
+ assert.equal(err.status, 504);
+ assert.match(err.message, /POST \/api\/remember/);
+ return true;
+ },
+ );
+ assert.ok(Date.now() - started < 2_000, "should give up at the deadline, not hang");
+});
+
+test("a stalled handshake is bounded too", async () => {
+ // `/version` runs before any memory call, so leaving it unbounded hangs
+ // every method on the client rather than one request.
+ globalThis.fetch = async (url, init = {}) => hangUntilAborted(init);
+
+ await assert.rejects(
+ clientWith({ requestTimeoutMs: 120 }).rememberAsync("a durable fact"),
+ (err) => {
+ assert.equal(err.name, "MemWalRequestTimeout");
+ return true;
+ },
+ );
+});
+
+test("waitForRememberJob stays inside its budget when every poll stalls", async () => {
+ stubRelayer((path, init) =>
+ path.startsWith("/api/remember/") ? hangUntilAborted(init) : Promise.reject(new Error(path)),
+ );
+
+ const started = Date.now();
+ await assert.rejects(
+ clientWith({ requestTimeoutMs: 10_000 }).waitForRememberJob("job-1", {
+ timeoutMs: 400,
+ pollIntervalMs: 50,
+ }),
+ /timed out/,
+ );
+ // The budget is the bound now. Previously the loop only checked it between
+ // polls, so a single stalled read outlived it by however long the socket
+ // stayed open — here that would have been the 10s client deadline.
+ assert.ok(
+ Date.now() - started < 3_000,
+ `wait overran its budget: ${Date.now() - started}ms`,
+ );
+});
+
+test("an expired poll is retried rather than failing the whole wait", async () => {
+ let call = 0;
+ stubRelayer((path, init) => {
+ if (!path.startsWith("/api/remember/")) return Promise.reject(new Error(path));
+ // First poll stalls; the retry answers.
+ if (call++ === 0) return hangUntilAborted(init);
+ return Response.json({
+ job_id: "job-1",
+ status: "done",
+ blob_id: "blob-1",
+ owner: "0x1",
+ namespace: "default",
+ });
+ });
+
+ const result = await clientWith({ requestTimeoutMs: 100 }).waitForRememberJob("job-1", {
+ timeoutMs: 5_000,
+ pollIntervalMs: 20,
+ });
+
+ assert.equal(result.blob_id, "blob-1");
+ assert.ok(call >= 2, "the stalled poll should have been retried, not fatal");
+});
+
+test("a transport failure that is not a deadline keeps its own identity", async () => {
+ // The deadline path must not swallow real network errors — an operator
+ // debugging DNS or TLS needs the original, not "timed out".
+ stubRelayer(() => Promise.reject(new TypeError("fetch failed")));
+
+ await assert.rejects(
+ clientWith({ requestTimeoutMs: 5_000 }).rememberAsync("a durable fact"),
+ (err) => {
+ assert.notEqual(err.name, "MemWalRequestTimeout");
+ assert.match(err.message, /fetch failed/);
+ return true;
+ },
+ );
+});
+
+test("an unusable requestTimeoutMs falls back to the default rather than disabling the bound", async () => {
+ // 0 / negative / NaN would mean "no deadline", which is the bug this
+ // exists to prevent — a typo must not silently restore it.
+ for (const bad of [0, -1, Number.NaN, undefined]) {
+ const client = clientWith({ requestTimeoutMs: bad });
+ assert.equal(client.requestTimeoutMs, 30_000, `bad value ${bad} disabled the bound`);
+ }
+ assert.equal(clientWith({ requestTimeoutMs: 1234 }).requestTimeoutMs, 1234);
+});
diff --git a/services/server/scripts/mcp/tools/remember-wait.ts b/services/server/scripts/mcp/tools/remember-wait.ts
index 2f2fef4e6..1a1ce2e7b 100644
--- a/services/server/scripts/mcp/tools/remember-wait.ts
+++ b/services/server/scripts/mcp/tools/remember-wait.ts
@@ -196,6 +196,11 @@ export function pendingBulkMessage(
*
* 15s matches the only deadline the SDK sets for itself. A healthy accept is
* ~1.1s, so this fires only when something is genuinely wrong.
+ *
+ * The SDK grows its own 30s per-request backstop in the release after the
+ * pinned 0.1.7, which does not make this redundant: that one is a floor for
+ * every consumer, this is the tighter bound an interactive agent needs, and
+ * whichever is smaller fires first.
*/
const DEFAULT_ACCEPT_DEADLINE_MS = 15_000;
From d6d18b8dd6f294098ae462a7a9b4e37f5925ddb0 Mon Sep 17 00:00:00 2001
From: Harry Phan
Date: Tue, 15 Sep 2026 22:13:41 +0700
Subject: [PATCH 012/132] fix(relayer): fail remember jobs whose preparation
never ran
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit
`remember` commits the job row, then spawns preparation — summarize, embed,
SEAL encrypt, enqueue — in a `tokio::spawn` inside the relayer process. If
the process stops in that window the row is left at `pending` with nothing
to resume it, and the sweeper never looked at `pending` at all. The state
machine in migration 005 does not even name a `pending → failed` edge: the
state had no exit.
Nothing surfaced it either. `memwal_remember_status` read the row and
reported "still uploading" indefinitely, so a write the user was told was
on its way was simply gone. Returning at accept made that worse rather than
causing it: the blocking call at least ended in a timeout the caller saw,
where now nobody is waiting to notice.
`pending` cannot be swept wholesale — a job that IS prepared waits at
`pending` until a wallet worker takes it, and with
WALRUS_UPLOAD_PER_WALLET_CONCURRENCY defaulting to 1 that queue is
legitimately minutes deep, so failing those would abandon paid work about to
run. `preparation_encrypted_b64 IS NULL` separates them: it is written by
the statement immediately before `enqueue_wallet_job`, so its absence means
the job never reached the queue.
Failing is the only available outcome, not a preference. The row holds the
SEAL ciphertext and never the plaintext, so a job that died before
encrypting has nothing to retry from — the error says the fact was never
stored and must be sent again, rather than implying a job merely died.
Clearing `prepare_claim_token` is what makes this safe against a preparation
that was slow rather than dead: that task's own UPDATE is fenced on the
token, so it now matches zero rows, logs the lost claim and returns before
`enqueue_wallet_job`. It cannot mint a paid blob for a job just declared
dead.
Quota needs nothing extra — `main` already runs
`release_reservations_for_terminal_jobs` immediately after this on the same
tick, ordered that way so rows this pass just failed are reconciled without
waiting another minute.
Composes with the existing idempotency recovery: a failed row with no
blob_id is exactly what the remember route resets and re-prepares, so a
client that does retry the same key gets a real second attempt instead of
collapsing onto a corpse.
Migration 005's comment still lists the old transitions. It is deliberately
left alone — sqlx checksums migration files, so editing a shipped one breaks
`migrate` on every deployed database; the transition is documented on the
sweeper instead.
---
services/server/src/storage/db.rs | 263 +++++++++++++++++++++++++++++-
1 file changed, 259 insertions(+), 4 deletions(-)
diff --git a/services/server/src/storage/db.rs b/services/server/src/storage/db.rs
index 90e8cd18c..a45d31ca5 100644
--- a/services/server/src/storage/db.rs
+++ b/services/server/src/storage/db.rs
@@ -2103,9 +2103,42 @@ impl VectorDb {
Ok(rows)
}
- /// Mark worker-claimed remember jobs as failed when no worker has updated
- /// them within the stale TTL. Pending rows are left alone because they may
- /// simply be waiting behind legitimate queue backlog.
+ /// Mark remember jobs as failed once nothing can still move them.
+ ///
+ /// Two shapes of stuck, which need different tests:
+ ///
+ /// * `running` / `uploaded` — a worker claimed the job and stopped
+ /// updating it.
+ /// * `pending` with no preparation payload — the row was committed by the
+ /// route, but the spawned preparation (summarize → embed + SEAL encrypt →
+ /// enqueue) never finished, so nothing was ever queued. Preparation runs
+ /// in a `tokio::spawn` inside the relayer process, so a restart in that
+ /// window leaves the row behind with no task to resume it.
+ ///
+ /// `pending` cannot be swept wholesale — a job that IS prepared sits at
+ /// `pending` until a wallet worker picks it up, which under upload backlog
+ /// is legitimately many minutes (`WALRUS_UPLOAD_PER_WALLET_CONCURRENCY`
+ /// defaults to 1). `preparation_encrypted_b64 IS NULL` is what separates
+ /// the two: it is written in the same statement that precedes
+ /// `enqueue_wallet_job`, so its absence means the job never reached the
+ /// queue.
+ ///
+ /// Failing is the only option, not a choice. The row stores the SEAL
+ /// ciphertext, never the plaintext, so a job that died before encrypting
+ /// has nothing left to retry from — the fact is gone and the owner has to
+ /// send it again. Saying so is strictly better than the alternative, which
+ /// was a row sitting at `pending` forever while `memwal_remember_status`
+ /// reported it as still uploading.
+ ///
+ /// Clearing `prepare_claim_token` is what makes this safe against a
+ /// preparation that is merely very slow rather than dead: that task's own
+ /// UPDATE is fenced on the token, so it now matches zero rows, logs the
+ /// lost claim and returns *before* `enqueue_wallet_job` — it cannot mint a
+ /// paid blob for a job just declared dead.
+ ///
+ /// Quota needs no special handling here: `main` runs
+ /// `release_reservations_for_terminal_jobs` immediately after this on the
+ /// same tick, which is what reclaims the bytes these rows had reserved.
pub async fn fail_stale_remember_jobs(
&self,
stale_after: std::time::Duration,
@@ -2128,7 +2161,44 @@ impl VectorDb {
if rows > 0 {
tracing::warn!("Marked {} stale remember jobs as failed", rows);
}
- Ok(rows)
+
+ // Second pass rather than one OR'd predicate: this branch clears the
+ // preparation claim and carries its own error text, and the two are
+ // different enough that folding them together would hide which case
+ // actually fired in the logs.
+ let orphaned = sqlx::query(
+ "UPDATE remember_jobs
+ SET status = 'failed',
+ error_msg = COALESCE(
+ error_msg,
+ 'preparation never completed — the relayer stopped before this write was encrypted, so the fact was never stored and must be sent again'
+ ),
+ prepare_claim_token = NULL,
+ prepare_claimed_at = NULL,
+ updated_at = NOW()
+ WHERE status = 'pending'
+ AND preparation_encrypted_b64 IS NULL
+ AND updated_at < NOW() - ($1 * INTERVAL '1 second')",
+ )
+ .bind(stale_after_secs)
+ .execute(&self.pool)
+ .await
+ .map_err(|e| {
+ AppError::Internal(format!("Failed to fail orphaned remember preparations: {}", e))
+ })?;
+
+ let orphaned_rows = orphaned.rows_affected();
+ if orphaned_rows > 0 {
+ // Distinct wording from the sweep above: this one means writes were
+ // accepted and silently lost, which is an availability signal about
+ // the relayer, not a Walrus or wallet problem.
+ tracing::warn!(
+ "Marked {} remember jobs as failed whose preparation never completed",
+ orphaned_rows
+ );
+ }
+
+ Ok(rows + orphaned_rows)
}
/// Rows whose expiry data has never been synced, or was synced more
@@ -3629,3 +3699,188 @@ mod quota_admission_tests {
cleanup(&db, &owner).await;
}
}
+
+#[cfg(test)]
+mod stale_sweep_tests {
+ use super::*;
+ use std::time::Duration;
+
+ fn test_database_url() -> String {
+ std::env::var("DATABASE_URL")
+ .unwrap_or_else(|_| "postgresql://memwal:memwal_secret@localhost:5432/memwal".into())
+ }
+
+ async fn test_db() -> VectorDb {
+ VectorDb::new(&test_database_url())
+ .await
+ .expect("test database must be reachable with pgvector installed")
+ }
+
+ /// Unique per test so concurrent runs cannot see each other's rows.
+ fn unique_owner(tag: &str) -> String {
+ format!("0xtest-{}-{}", tag, uuid::Uuid::new_v4())
+ }
+
+ /// Insert one remember job, aged by `age_secs`, optionally already prepared.
+ async fn seed_job(
+ db: &VectorDb,
+ owner: &str,
+ status: &str,
+ prepared: bool,
+ claim_token: Option<&str>,
+ age_secs: i64,
+ ) -> String {
+ let id = uuid::Uuid::new_v4().to_string();
+ sqlx::query(
+ "INSERT INTO remember_jobs
+ (id, owner, namespace, status, preparation_encrypted_b64,
+ prepare_claim_token, prepare_claimed_at, created_at, updated_at)
+ VALUES ($1, $2, 'default', $3, $4, $5,
+ NOW() - ($6 * INTERVAL '1 second'),
+ NOW() - ($6 * INTERVAL '1 second'),
+ NOW() - ($6 * INTERVAL '1 second'))",
+ )
+ .bind(&id)
+ .bind(owner)
+ .bind(status)
+ .bind(if prepared { Some("ZW5jcnlwdGVk") } else { None })
+ .bind(claim_token)
+ .bind(age_secs)
+ .execute(&db.pool)
+ .await
+ .expect("seed remember job");
+ id
+ }
+
+ async fn status_of(db: &VectorDb, id: &str) -> String {
+ sqlx::query_scalar("SELECT status FROM remember_jobs WHERE id = $1")
+ .bind(id)
+ .fetch_one(&db.pool)
+ .await
+ .expect("read status")
+ }
+
+ /// A row committed by the route whose preparation never ran has no task
+ /// left to resume it: preparation lives in a `tokio::spawn` inside the
+ /// relayer, so a restart in that window strands it. Before this it sat at
+ /// `pending` forever and `memwal_remember_status` reported it as still
+ /// uploading — a write silently lost while the user was told it was coming.
+ #[tokio::test]
+ async fn orphaned_preparation_is_failed() {
+ let db = test_db().await;
+ let owner = unique_owner("orphan");
+ let id = seed_job(&db, &owner, "pending", false, Some("claim-1"), 900).await;
+
+ db.fail_stale_remember_jobs(Duration::from_secs(600))
+ .await
+ .expect("sweep");
+
+ assert_eq!(status_of(&db, &id).await, "failed");
+ let msg: Option =
+ sqlx::query_scalar("SELECT error_msg FROM remember_jobs WHERE id = $1")
+ .bind(&id)
+ .fetch_one(&db.pool)
+ .await
+ .unwrap();
+ assert!(
+ msg.unwrap_or_default().contains("never stored"),
+ "the message has to say the fact is gone, not merely that a job died",
+ );
+ }
+
+ /// The reason `pending` cannot be swept wholesale. A prepared job waits at
+ /// `pending` until a wallet worker takes it, and with
+ /// `WALRUS_UPLOAD_PER_WALLET_CONCURRENCY` defaulting to 1 that queue is
+ /// legitimately minutes deep. Failing these would abandon paid work that
+ /// was about to run.
+ #[tokio::test]
+ async fn prepared_job_waiting_on_the_upload_queue_is_left_alone() {
+ let db = test_db().await;
+ let owner = unique_owner("queued");
+ let id = seed_job(&db, &owner, "pending", true, Some("claim-1"), 900).await;
+
+ db.fail_stale_remember_jobs(Duration::from_secs(600))
+ .await
+ .expect("sweep");
+
+ assert_eq!(status_of(&db, &id).await, "pending");
+ }
+
+ /// Preparation itself takes a moment (summarize, embed, SEAL encrypt), so
+ /// a young unprepared row is in-flight, not orphaned.
+ #[tokio::test]
+ async fn a_preparation_still_in_flight_is_left_alone() {
+ let db = test_db().await;
+ let owner = unique_owner("young");
+ let id = seed_job(&db, &owner, "pending", false, Some("claim-1"), 5).await;
+
+ db.fail_stale_remember_jobs(Duration::from_secs(600))
+ .await
+ .expect("sweep");
+
+ assert_eq!(status_of(&db, &id).await, "pending");
+ }
+
+ /// What makes the sweep safe against a preparation that was merely very
+ /// slow rather than dead. Its own UPDATE is fenced on the claim token, so
+ /// once the sweeper clears it that statement matches zero rows and the task
+ /// returns before `enqueue_wallet_job` — it cannot mint a paid blob for a
+ /// job just declared dead.
+ #[tokio::test]
+ async fn clearing_the_claim_fences_a_late_preparation() {
+ let db = test_db().await;
+ let owner = unique_owner("fence");
+ let id = seed_job(&db, &owner, "pending", false, Some("claim-1"), 900).await;
+
+ db.fail_stale_remember_jobs(Duration::from_secs(600))
+ .await
+ .expect("sweep");
+
+ // Exactly the statement `spawn_prepare_remember_job` runs when it
+ // finishes, token and all.
+ let late = sqlx::query(
+ "UPDATE remember_jobs SET preparation_encrypted_b64 = $1, updated_at = NOW()
+ WHERE id = $2 AND ($3::TEXT IS NULL OR prepare_claim_token = $3)",
+ )
+ .bind("ZW5jcnlwdGVk")
+ .bind(&id)
+ .bind("claim-1")
+ .execute(&db.pool)
+ .await
+ .expect("late preparation");
+
+ assert_eq!(
+ late.rows_affected(),
+ 0,
+ "a late preparation must be fenced out, or it would queue a paid write for a dead job",
+ );
+ }
+
+ /// The pre-existing sweep is unchanged.
+ #[tokio::test]
+ async fn a_stalled_worker_claim_still_fails() {
+ let db = test_db().await;
+ let owner = unique_owner("running");
+ let id = seed_job(&db, &owner, "running", true, None, 900).await;
+
+ db.fail_stale_remember_jobs(Duration::from_secs(600))
+ .await
+ .expect("sweep");
+
+ assert_eq!(status_of(&db, &id).await, "failed");
+ }
+
+ /// A finished write is terminal and the sweeper must never touch it.
+ #[tokio::test]
+ async fn a_done_job_is_never_swept() {
+ let db = test_db().await;
+ let owner = unique_owner("done");
+ let id = seed_job(&db, &owner, "done", true, None, 900).await;
+
+ db.fail_stale_remember_jobs(Duration::from_secs(600))
+ .await
+ .expect("sweep");
+
+ assert_eq!(status_of(&db, &id).await, "done");
+ }
+}
From c1b2561a3447cb9c7163cbbc2f808ee2703dc862 Mon Sep 17 00:00:00 2001
From: Nikola Le <91601109+nikola0x0@users.noreply.github.com>
Date: Wed, 16 Sep 2026 09:47:36 +0700
Subject: [PATCH 013/132] fix(server): recall sort precedence and the embedding
size check (WALM-470) (#911)
* fix(server): stop scoring_weights reordering an explicit recall sort (WALM-470)
A recall request carrying both sort and scoring_weights returned neither
order: select_hits_for_sort ordered and truncated the hits, then the
ranker reordered the survivors by composite score, so sort=recent stopped
meaning newest-first.
Per the decision on WALM-460, an explicit sort is the order and weights
apply only when sort is omitted. RecallRequest.sort becomes
Option so omitted and "relevance" differ, and
resolve_scoring_weights validates the weights, then suppresses them when
sort is set. The SDK docs and both API references state the rule.
* fix(server): narrow the embedding size check to the provider path (WALM-470)
The WALM-423 check ran before the embedder looked for an API key, so
deployments without one (local dev, CI, self-hosted) rejected remember
and recall text over 16 KiB, though the key-less fallback hashes locally
and has no context window. It now runs only when a provider key is set.
The same change mapped every provider 400 to "input exceeds the model
context limit". A 400 now becomes BadRequest only when its body names a
context-length problem; anything else, such as an unknown model id,
stays Internal so it is not blamed on the caller and still alerts.
The embed call moves into embed_text, which takes only the key, base URL
and text, so the tests can drive it against a local stand-in provider.
* docs(sdk,server): add the 0.1.7 changelog entry and trim WALM-470 comments
Review follow-up: record the recall sort precedence change under the
unreleased 0.1.7 in both SDK changelogs (no version bump), and drop
ticket ids, dates and design history from the new code comments.
---------
Co-authored-by: Le Tien Phat <91601109+Niko1444@users.noreply.github.com>
---
docs/relayer/api-reference.md | 4 +-
docs/sdk/api-reference.md | 3 +
docs/sdk/changelog.mdx | 5 +-
packages/sdk/CHANGELOG.md | 1 +
packages/sdk/src/types.ts | 5 +
services/server/src/routes/mod.rs | 139 ++++++++++
services/server/src/routes/recall.rs | 14 +-
services/server/src/services/embedder.rs | 318 ++++++++++++++++-------
services/server/src/types.rs | 21 +-
9 files changed, 404 insertions(+), 106 deletions(-)
diff --git a/docs/relayer/api-reference.md b/docs/relayer/api-reference.md
index 55a6446a7..603765558 100644
--- a/docs/relayer/api-reference.md
+++ b/docs/relayer/api-reference.md
@@ -269,6 +269,8 @@ Search for memories matching a natural language query. Returns decrypted plainte
`limit` defaults to `10`; the server caps it at `100`. `namespace` defaults to `"default"`. `scoring_weights` takes an optional object; omit it to keep the plain cosine-distance order.
+`sort` is optional: `"relevance"` (the cosine order, and the behaviour when omitted) or `"recent"` (the newest among the semantic matches). An explicit `sort`, `"relevance"` included, is the order, and the relayer ignores `scoring_weights` for that request.
+
#### Scoring weights
The optional `scoring_weights` object turns on composite ranking. The same object works on `/api/recall`, `/api/recall/manual`, and `/api/ask`.
@@ -297,7 +299,7 @@ The optional `scoring_weights` object turns on composite ranking. The same objec
}
```
-`score` only appears when `scoring_weights` sets a nonzero `recency` or `importance` weight. A request that sets only the `semantic` weight keeps the plain cosine order, and the relayer omits `score`. `dropped_count` only appears when at least one match dropped out because its blob download or decryption failed; the relayer omits those matches from `results`.
+`score` only appears when `scoring_weights` sets a nonzero `recency` or `importance` weight and `sort` is omitted. A request that sets only the `semantic` weight keeps the plain cosine order, and the relayer omits `score`. `dropped_count` only appears when at least one match dropped out because its blob download or decryption failed; the relayer omits those matches from `results`.
### `POST /api/remember/manual`
diff --git a/docs/sdk/api-reference.md b/docs/sdk/api-reference.md
index 3e3054c66..ec931610e 100644
--- a/docs/sdk/api-reference.md
+++ b/docs/sdk/api-reference.md
@@ -203,6 +203,9 @@ Two limits worth knowing:
Omitting `sort` leaves the request byte-identical to a plain cosine recall, so
existing callers see no change.
+An explicit `sort`, `"relevance"` included, is the order: the relayer ignores
+`scoringWeights` for that request. Weights re-rank only when `sort` is omitted.
+
#### `scoringWeights`
`scoringWeights` blends recency and importance into the relayer's ranking:
diff --git a/docs/sdk/changelog.mdx b/docs/sdk/changelog.mdx
index 27064137e..926cb73d8 100644
--- a/docs/sdk/changelog.mdx
+++ b/docs/sdk/changelog.mdx
@@ -28,12 +28,12 @@ questions:
- When was bulk remember added to the Walrus Memory SDK?
- What security improvements have been made to the MemWal SDK?
answer: >-
- The latest TypeScript SDK release is 0.1.7. Request bodies are hashed with `@noble/hashes` rather than WebCrypto-or-`node:crypto`, so the SDK no longer imports a Node builtin that Vite silently externalises into a runtime crash in the browser, and it declares a Node 20 floor. `restore()` results include `failed` (required like `truncated`; SDK defaults omitted to `0`) for permanent decrypt/UTF-8 failures instead of folding them into `skipped`. Empty-body 401s use the AUTH_REJECTED troubleshooting message instead of telling callers to run `memwal_login`. Account and manual PTBs use typed `tx.pure` helpers so they work with modern `@mysten/sui`. 0.1.6 added optional `created_at` on `recall()` results, plus `sort` and `scoringWeights` on `RecallOptions`, and reports HTTP 503 as a retryable upstream outage instead of a sign-in failure.
+ The latest TypeScript SDK release is 0.1.7. Request bodies are hashed with `@noble/hashes` rather than WebCrypto-or-`node:crypto`, so the SDK no longer imports a Node builtin that Vite silently externalises into a runtime crash in the browser, and it declares a Node 20 floor. `restore()` results include `failed` (required like `truncated`; SDK defaults omitted to `0`) for permanent decrypt/UTF-8 failures instead of folding them into `skipped`. Empty-body 401s use the AUTH_REJECTED troubleshooting message instead of telling callers to run `memwal_login`. Account and manual PTBs use typed `tx.pure` helpers so they work with modern `@mysten/sui`. An explicit `sort` on `recall()`, `"relevance"` included, now makes the relayer ignore `scoringWeights`; weights re-rank only when `sort` is omitted. 0.1.6 added optional `created_at` on `recall()` results, plus `sort` and `scoringWeights` on `RecallOptions`, and reports HTTP 503 as a retryable upstream outage instead of a sign-in failure.
---
## 0.1.7
-This release removes the Node `crypto` import that crashed bundled browser builds at runtime, adds `failed` on `restore()` results, declares a Node 20 floor, stops telling headless SDK clients to call `memwal_login` on empty-body 401s, and switches account and manual PTBs to typed `tx.pure` helpers.
+This release removes the Node `crypto` import that crashed bundled browser builds at runtime, adds `failed` on `restore()` results, declares a Node 20 floor, stops telling headless SDK clients to call `memwal_login` on empty-body 401s, switches account and manual PTBs to typed `tx.pure` helpers, and makes an explicit `sort` on `recall()` win over `scoringWeights`.
### Added
@@ -45,6 +45,7 @@ This release removes the Node `crypto` import that crashed bundled browser build
- Declare `engines.node >= 20.0.0`, matching `memwal-mcp` and `openclaw-memory-memwal`. The SDK was the only published package without a floor. (WALM-599)
- Empty-body 401s now use the same AUTH_REJECTED troubleshooting message as credential 401s instead of telling callers to run `memwal_login`. Headless SDK clients do not have that MCP tool.
- `account.ts` and `manual.ts` PTBs use typed `tx.pure` helpers instead of the legacy untyped moveCall argument syntax that fails under modern `@mysten/sui`.
+- An explicit `sort` on `recall()`, `"relevance"` included, is now the order: the relayer ignores `scoringWeights` for that request, and weights re-rank only when `sort` is omitted. Setting both used to return neither order, so `sort: "recent"` stopped meaning newest-first once `scoringWeights` carried a recency weight. (WALM-470)
## 0.1.6
diff --git a/packages/sdk/CHANGELOG.md b/packages/sdk/CHANGELOG.md
index 75fa71aa5..df83eb064 100644
--- a/packages/sdk/CHANGELOG.md
+++ b/packages/sdk/CHANGELOG.md
@@ -12,6 +12,7 @@
- Declare `engines.node >= 20.0.0`, matching `memwal-mcp` and `openclaw-memory-memwal`. The SDK was the only published package without a floor. (WALM-599)
- Empty-body 401s now use the same AUTH_REJECTED troubleshooting message as credential 401s instead of telling callers to run `memwal_login`. Headless SDK clients do not have that MCP tool.
- `account.ts` and `manual.ts` PTBs use typed `tx.pure` helpers instead of the legacy untyped moveCall argument syntax that fails under modern `@mysten/sui`.
+- An explicit `sort` on `recall()`, `"relevance"` included, is now the order: the relayer ignores `scoringWeights` for that request, and weights re-rank only when `sort` is omitted. Setting both used to return neither order, so `sort: "recent"` stopped meaning newest-first once `scoringWeights` carried a recency weight. (WALM-470)
## 0.1.6
diff --git a/packages/sdk/src/types.ts b/packages/sdk/src/types.ts
index e6b374c50..8a5b34e1c 100644
--- a/packages/sdk/src/types.ts
+++ b/packages/sdk/src/types.ts
@@ -165,6 +165,9 @@ export interface RecallOptions {
*
* For newest-wins, use `sort: "recent"` instead. It over-fetches
* candidates server-side before ordering them by write-time.
+ *
+ * Ignored when `sort` is set: an explicit `sort`, `"relevance"` included,
+ * decides the order, and weights apply only when `sort` is omitted.
*/
scoringWeights?: ScoringWeights;
/**
@@ -179,6 +182,8 @@ export interface RecallOptions {
* widens the candidate set; weights only re-rank the set that was already
* returned, so weights alone cannot surface a record that fell outside
* the window.
+ *
+ * Setting `sort` at all makes the relayer ignore `scoringWeights`.
*/
sort?: "relevance" | "recent";
}
diff --git a/services/server/src/routes/mod.rs b/services/server/src/routes/mod.rs
index abd037568..123894c6c 100644
--- a/services/server/src/routes/mod.rs
+++ b/services/server/src/routes/mod.rs
@@ -249,6 +249,21 @@ pub(super) fn select_hits_for_sort(
hits
}
+/// The scoring weights `/api/recall` hands to the ranker. An explicit `sort`
+/// must not be re-ranked, so it yields the default (no-op) weights; malformed
+/// weights are rejected either way.
+pub(super) fn resolve_scoring_weights(
+ sort: Option,
+ requested: Option,
+) -> Result {
+ let requested = requested.unwrap_or_default();
+ requested.validate()?;
+ if sort.is_some() {
+ return Ok(crate::types::ScoringWeights::default());
+ }
+ Ok(requested)
+}
+
/// Project the ranker's output onto the `/api/recall` and `/api/ask` wire
/// shape, preserving ranked order.
///
@@ -461,6 +476,130 @@ mod recall_sort_tests {
}
}
+/// Each test replays the recall handler's ordering (`select_hits_for_sort`,
+/// then the ranker with the resolved weights) over hand-scored rows.
+#[cfg(test)]
+mod recall_sort_precedence_tests {
+ use crate::engine::HydratedMemory;
+ use crate::services::ranker::{CompositeRanker, Ranker};
+ use crate::types::{AppError, RecallSort, ScoringWeights, SearchHit};
+
+ fn ts(rfc3339: &str) -> chrono::DateTime {
+ chrono::DateTime::parse_from_rfc3339(rfc3339)
+ .unwrap()
+ .with_timezone(&chrono::Utc)
+ }
+
+ fn now() -> chrono::DateTime {
+ ts("2026-07-06T00:00:00Z")
+ }
+
+ /// `distance` ascending is the order pgvector hands back.
+ fn search_hit(blob_id: &str, distance: f64, created_at: &str) -> SearchHit {
+ SearchHit {
+ blob_id: blob_id.to_string(),
+ distance,
+ created_at: ts(created_at),
+ importance: 0.5,
+ }
+ }
+
+ fn weights(recency: f64) -> ScoringWeights {
+ ScoringWeights {
+ semantic: 1.0,
+ recency,
+ recency_half_life_days: 30.0,
+ importance: 0.0,
+ }
+ }
+
+ fn recall_order(
+ hits: Vec,
+ sort: Option,
+ requested: ScoringWeights,
+ ) -> Vec {
+ let weights = super::resolve_scoring_weights(sort, Some(requested)).unwrap();
+ let limit = hits.len();
+ let hydrated: Vec =
+ super::select_hits_for_sort(hits, sort.unwrap_or_default(), limit)
+ .into_iter()
+ .map(|h| HydratedMemory {
+ blob_id: h.blob_id,
+ text: String::new(),
+ distance: h.distance,
+ created_at: Some(h.created_at),
+ importance: Some(h.importance),
+ })
+ .collect();
+ CompositeRanker
+ .rank(hydrated, &weights, now())
+ .into_iter()
+ .map(|r| r.memory.blob_id)
+ .collect()
+ }
+
+ /// Composite scores at semantic 1.0 / recency 0.3 / half-life 30d:
+ /// old-but-close = 0.90 + 0.3 * 2^(-30/30) = 1.05
+ /// newest-but-far = 0.10 + 0.3 * 2^(0/30) = 0.40
+ fn ticket_repro_hits() -> Vec {
+ vec![
+ search_hit("old-but-close", 0.10, "2026-06-06T00:00:00Z"),
+ search_hit("newest-but-far", 0.90, "2026-07-06T00:00:00Z"),
+ ]
+ }
+
+ /// Composite scores at semantic 1.0 / recency 0.8 / half-life 30d:
+ /// close-old = 0.90 + 0.8 * 2^(-60/30) = 1.10
+ /// far-new = 0.80 + 0.8 * 2^(0/30) = 1.60
+ /// so the ranker, when it runs, reverses the cosine order.
+ fn weights_reverse_cosine_hits() -> Vec {
+ vec![
+ search_hit("close-old", 0.10, "2026-05-07T00:00:00Z"),
+ search_hit("far-new", 0.20, "2026-07-06T00:00:00Z"),
+ ]
+ }
+
+ #[test]
+ fn explicit_recent_is_not_reordered_by_scoring_weights() {
+ assert_eq!(
+ recall_order(ticket_repro_hits(), Some(RecallSort::Recent), weights(0.3)),
+ ["newest-but-far", "old-but-close"]
+ );
+ }
+
+ #[test]
+ fn explicit_relevance_is_not_reordered_by_scoring_weights() {
+ assert_eq!(
+ recall_order(
+ weights_reverse_cosine_hits(),
+ Some(RecallSort::Relevance),
+ weights(0.8)
+ ),
+ ["close-old", "far-new"]
+ );
+ }
+
+ #[test]
+ fn omitted_sort_lets_scoring_weights_rerank() {
+ assert_eq!(
+ recall_order(weights_reverse_cosine_hits(), None, weights(0.8)),
+ ["far-new", "close-old"]
+ );
+ }
+
+ /// Malformed weights are a 400 even when an explicit `sort` would ignore
+ /// them — `sort` does not quietly excuse a bad request.
+ #[test]
+ fn malformed_weights_are_rejected_even_when_sort_is_explicit() {
+ let mut malformed = weights(0.3);
+ malformed.semantic = f64::NAN;
+ assert!(matches!(
+ super::resolve_scoring_weights(Some(RecallSort::Recent), Some(malformed)),
+ Err(AppError::BadRequest(_))
+ ));
+ }
+}
+
#[cfg(test)]
mod recall_result_mapping_tests {
use crate::engine::HydratedMemory;
diff --git a/services/server/src/routes/recall.rs b/services/server/src/routes/recall.rs
index 018c1faf9..d23615025 100644
--- a/services/server/src/routes/recall.rs
+++ b/services/server/src/routes/recall.rs
@@ -162,8 +162,9 @@ pub async fn recall(
// Validate scoring_weights up front — fail fast on malformed input
// (NaN, out-of-range, sub-floor half-life) BEFORE we spend an embed +
// vector search + Walrus + SEAL round-trip just to 400 at the end.
- let weights = body.scoring_weights.clone().unwrap_or_default();
- weights.validate()?;
+ // An explicit `sort` suppresses them; see `resolve_scoring_weights`.
+ let weights = super::resolve_scoring_weights(body.sort, body.scoring_weights.clone())?;
+ let sort = body.sort.unwrap_or_default();
// Owner is derived from delegate key via onchain verification (auth middleware)
let owner = &auth.owner;
@@ -173,6 +174,11 @@ pub async fn recall(
owner = %owner,
namespace = %namespace,
ranker_active = weights.is_ranker_active(),
+ scoring_weights_ignored = body.sort.is_some()
+ && body
+ .scoring_weights
+ .as_ref()
+ .is_some_and(ScoringWeights::is_ranker_active),
"recall request"
);
@@ -187,7 +193,7 @@ pub async fn recall(
// row is frequently a mediocre semantic match and would otherwise fall
// outside the cosine top-`limit` entirely. `Relevance` fetches exactly
// `limit`, so the default path issues the identical query it always has.
- let candidate_limit = body.sort.candidate_limit(limit);
+ let candidate_limit = sort.candidate_limit(limit);
let t1 = std::time::Instant::now();
let hits = state
.db
@@ -199,7 +205,7 @@ pub async fn recall(
// only distance + created_at, both already on the row, so the over-fetch
// costs one wider SQL query instead of 5x the Walrus downloads and SEAL
// decrypts.
- let hits = super::select_hits_for_sort(hits, body.sort, limit);
+ let hits = super::select_hits_for_sort(hits, sort, limit);
let hit_count = hits.len();
if hits.is_empty() {
diff --git a/services/server/src/services/embedder.rs b/services/server/src/services/embedder.rs
index 90252e6a3..f7956a51c 100644
--- a/services/server/src/services/embedder.rs
+++ b/services/server/src/services/embedder.rs
@@ -27,8 +27,26 @@ pub const EMBEDDING_MODEL: &str = "openai/text-embedding-3-small";
pub const EMBEDDING_DIMS: usize = 1536;
/// 16384 = 8192 tokens × 2 chars/token under cl100k; do not use admin 64KiB.
+///
+/// A limit of [`EMBEDDING_MODEL`]'s context window, so it applies only when a
+/// provider key is set. The key-less fallback hashes locally with no limit.
pub(crate) const MAX_EMBED_INPUT_BYTES: usize = 16384;
+/// Whether a provider 400 complains about input length; any other 400 is a
+/// server-side problem, not the caller's.
+fn is_context_length_error(body: &str) -> bool {
+ let body = body.to_ascii_lowercase();
+ [
+ "context_length_exceeded",
+ "maximum context length",
+ "reduce the length",
+ "too many tokens",
+ "string too long",
+ ]
+ .iter()
+ .any(|needle| body.contains(needle))
+}
+
fn reject_oversized_embed_input(text: &str) -> Result<(), AppError> {
if text.len() > MAX_EMBED_INPUT_BYTES {
return Err(AppError::BadRequest(format!(
@@ -67,114 +85,130 @@ impl OpenAiEmbedder {
impl Embedder for OpenAiEmbedder {
#[tracing::instrument(name = "embedder.embed", skip_all, fields(text_len = text.len()))]
async fn embed(&self, text: &str) -> Result, AppError> {
- reject_oversized_embed_input(text)?;
- match &self.config.openai_api_key {
- Some(api_key) => {
- // Real embedding via OpenRouter/OpenAI-compatible API
- let url = format!("{}/embeddings", self.config.openai_api_base);
-
- let started = std::time::Instant::now();
- let resp = self
- .http_client
- .post(&url)
- .header("Authorization", format!("Bearer {}", api_key))
- .header("Content-Type", "application/json")
- .json(&EmbeddingApiRequest {
- model: EMBEDDING_MODEL.to_string(),
- input: text.to_string(),
- })
- .send()
- .await
- .map_err(|e| {
- crate::observability::observe_external(
- "openai",
- "embeddings",
- "transport_error",
- started.elapsed(),
- );
- AppError::Internal(format!("Embedding API request failed: {}", e))
- })?;
- let status_label = resp.status().as_u16().to_string();
- crate::observability::observe_external(
- "openai",
- "embeddings",
- &status_label,
- started.elapsed(),
- );
+ embed_text(
+ &self.http_client,
+ self.config.openai_api_key.as_deref(),
+ &self.config.openai_api_base,
+ text,
+ )
+ .await
+ }
+}
- if !resp.status().is_success() {
- let status = resp.status();
- let body = resp.text().await.unwrap_or_default();
- if crate::services::extractor::is_upstream_status_transient(status) {
- return Err(AppError::UpstreamUnavailable(format!(
- "Embedding API upstream error ({}): {}",
- status, body
- )));
- }
- if status == reqwest::StatusCode::BAD_REQUEST {
- tracing::warn!(%status, body, "embedding API returned 400");
- return Err(AppError::BadRequest(
- "embedding input exceeds the model context limit".into(),
- ));
- }
- return Err(AppError::Internal(format!(
- "Embedding API error ({}): {}",
- status, body
- )));
- }
+/// The embedding call itself, taking only what it reads from `Config` so the
+/// tests can drive it against a local stand-in provider.
+async fn embed_text(
+ http_client: &reqwest::Client,
+ api_key: Option<&str>,
+ api_base: &str,
+ text: &str,
+) -> Result, AppError> {
+ match api_key {
+ Some(api_key) => {
+ reject_oversized_embed_input(text)?;
+ // Real embedding via OpenRouter/OpenAI-compatible API
+ let url = format!("{}/embeddings", api_base);
- // same pattern as the extractor — capture body
- // as text first so we can (1) treat transport-level
- // failures as transient, and (2) detect OpenRouter
- // error envelopes wrapped in HTTP 200. Both route to
- // `AppError::UpstreamUnavailable` (HTTP 503) so the
- // SDK / harness retry policy can recover. See
- // `extractor::parse_openrouter_error_envelope`.
- let body = resp.text().await.map_err(|e| {
- AppError::UpstreamUnavailable(format!(
- "Failed to read embedding response body: {}",
- e
- ))
+ let started = std::time::Instant::now();
+ let resp = http_client
+ .post(&url)
+ .header("Authorization", format!("Bearer {}", api_key))
+ .header("Content-Type", "application/json")
+ .json(&EmbeddingApiRequest {
+ model: EMBEDDING_MODEL.to_string(),
+ input: text.to_string(),
+ })
+ .send()
+ .await
+ .map_err(|e| {
+ crate::observability::observe_external(
+ "openai",
+ "embeddings",
+ "transport_error",
+ started.elapsed(),
+ );
+ AppError::Internal(format!("Embedding API request failed: {}", e))
})?;
+ let status_label = resp.status().as_u16().to_string();
+ crate::observability::observe_external(
+ "openai",
+ "embeddings",
+ &status_label,
+ started.elapsed(),
+ );
- if let Some(envelope) =
- crate::services::extractor::parse_openrouter_error_envelope(&body)
- {
+ if !resp.status().is_success() {
+ let status = resp.status();
+ let body = resp.text().await.unwrap_or_default();
+ if crate::services::extractor::is_upstream_status_transient(status) {
return Err(AppError::UpstreamUnavailable(format!(
- "OpenRouter upstream error (code={}): {}",
- envelope.code, envelope.message
+ "Embedding API upstream error ({}): {}",
+ status, body
)));
}
+ if status == reqwest::StatusCode::BAD_REQUEST && is_context_length_error(&body) {
+ tracing::warn!(%status, body, "embedding API rejected the input length");
+ return Err(AppError::BadRequest(
+ "embedding input exceeds the model context limit".into(),
+ ));
+ }
+ return Err(AppError::Internal(format!(
+ "Embedding API error ({}): {}",
+ status, body
+ )));
+ }
- let api_resp: EmbeddingApiResponse = serde_json::from_str(&body).map_err(|e| {
- AppError::Internal(format!("Failed to parse embedding response: {}", e))
- })?;
+ // same pattern as the extractor — capture body
+ // as text first so we can (1) treat transport-level
+ // failures as transient, and (2) detect OpenRouter
+ // error envelopes wrapped in HTTP 200. Both route to
+ // `AppError::UpstreamUnavailable` (HTTP 503) so the
+ // SDK / harness retry policy can recover. See
+ // `extractor::parse_openrouter_error_envelope`.
+ let body = resp.text().await.map_err(|e| {
+ AppError::UpstreamUnavailable(format!(
+ "Failed to read embedding response body: {}",
+ e
+ ))
+ })?;
- let vector = api_resp
- .data
- .into_iter()
- .next()
- .ok_or_else(|| AppError::Internal("Embedding API returned no data".into()))?
- .embedding;
- Ok(vector)
- }
- None => {
- // Mock embedding (deterministic hash-based) — for keyless dev
- tracing::warn!(" → Using MOCK embedding (no OPENAI_API_KEY set)");
- use sha2::Digest;
- let hash = sha2::Sha256::digest(text.as_bytes());
- let mock_vector: Vec = hash
- .iter()
- .cycle()
- .take(EMBEDDING_DIMS)
- .enumerate()
- .map(|(i, &b)| {
- let val = (b as f32 / 255.0) * 2.0 - 1.0;
- val * (1.0 + (i as f32 * 0.001).sin())
- })
- .collect();
- Ok(mock_vector)
+ if let Some(envelope) =
+ crate::services::extractor::parse_openrouter_error_envelope(&body)
+ {
+ return Err(AppError::UpstreamUnavailable(format!(
+ "OpenRouter upstream error (code={}): {}",
+ envelope.code, envelope.message
+ )));
}
+
+ let api_resp: EmbeddingApiResponse = serde_json::from_str(&body).map_err(|e| {
+ AppError::Internal(format!("Failed to parse embedding response: {}", e))
+ })?;
+
+ let vector = api_resp
+ .data
+ .into_iter()
+ .next()
+ .ok_or_else(|| AppError::Internal("Embedding API returned no data".into()))?
+ .embedding;
+ Ok(vector)
+ }
+ None => {
+ // Mock embedding (deterministic hash-based) — for keyless dev
+ tracing::warn!(" → Using MOCK embedding (no OPENAI_API_KEY set)");
+ use sha2::Digest;
+ let hash = sha2::Sha256::digest(text.as_bytes());
+ let mock_vector: Vec = hash
+ .iter()
+ .cycle()
+ .take(EMBEDDING_DIMS)
+ .enumerate()
+ .map(|(i, &b)| {
+ let val = (b as f32 / 255.0) * 2.0 - 1.0;
+ val * (1.0 + (i as f32 * 0.001).sin())
+ })
+ .collect();
+ Ok(mock_vector)
}
}
}
@@ -251,4 +285,92 @@ mod tests {
"input is over the embedding input limit",
);
}
+
+ // ── embed_text: size limit and upstream 400 mapping ──────────────
+
+ /// Stand-in embeddings provider answering every `POST /embeddings` with
+ /// `status` and `body`. Returns its base URL and a call counter.
+ async fn stub_provider(
+ status: u16,
+ body: &'static str,
+ ) -> (String, std::sync::Arc) {
+ use std::sync::atomic::{AtomicUsize, Ordering};
+ let calls = std::sync::Arc::new(AtomicUsize::new(0));
+ let counter = std::sync::Arc::clone(&calls);
+ let app = axum::Router::new().route(
+ "/embeddings",
+ axum::routing::post(move || {
+ let counter = std::sync::Arc::clone(&counter);
+ async move {
+ counter.fetch_add(1, Ordering::SeqCst);
+ (axum::http::StatusCode::from_u16(status).unwrap(), body)
+ }
+ }),
+ );
+ let listener = tokio::net::TcpListener::bind("127.0.0.1:0").await.unwrap();
+ let addr = listener.local_addr().unwrap();
+ tokio::spawn(async move { axum::serve(listener, app).await.unwrap() });
+ (format!("http://{addr}"), calls)
+ }
+
+ #[tokio::test]
+ async fn without_a_key_input_over_the_model_limit_still_embeds() {
+ let text = "a".repeat(100 * 1024);
+ let vector = super::embed_text(
+ &reqwest::Client::new(),
+ None,
+ "http://unused.invalid",
+ &text,
+ )
+ .await
+ .expect("key-less embedding has no input limit");
+ assert_eq!(vector.len(), super::EMBEDDING_DIMS);
+ }
+
+ #[tokio::test]
+ async fn with_a_key_oversized_input_is_rejected_before_calling_the_provider() {
+ let (base, calls) = stub_provider(200, r#"{"data":[{"embedding":[0.0]}]}"#).await;
+ let text = "q".repeat(super::MAX_EMBED_INPUT_BYTES + 1);
+ let result = super::embed_text(&reqwest::Client::new(), Some("key"), &base, &text).await;
+ assert!(
+ matches!(result, Err(crate::types::AppError::BadRequest(_))),
+ "got {result:?}"
+ );
+ assert_eq!(calls.load(std::sync::atomic::Ordering::SeqCst), 0);
+ }
+
+ #[tokio::test]
+ async fn a_provider_400_about_input_length_is_the_callers_bad_request() {
+ for body in [
+ r#"{"error":{"code":"context_length_exceeded","message":"..."}}"#,
+ r#"{"error":{"message":"This model's maximum context length is 8192 tokens"}}"#,
+ r#"{"error":{"message":"Please reduce the length of the messages."}}"#,
+ r#"{"error":{"message":"Too many tokens in input"}}"#,
+ r#"{"error":{"message":"String too long. Expected a string with maximum length 8192"}}"#,
+ ] {
+ let (base, _) = stub_provider(400, body).await;
+ let result = super::embed_text(&reqwest::Client::new(), Some("key"), &base, "q").await;
+ assert!(
+ matches!(result, Err(crate::types::AppError::BadRequest(_))),
+ "{body} → {result:?}"
+ );
+ }
+ }
+
+ #[tokio::test]
+ async fn any_other_provider_400_is_an_internal_error() {
+ for body in [
+ r#"{"error":{"message":"The model `text-embedding-9` does not exist"}}"#,
+ r#"{"error":{"code":"invalid_api_key","message":"Incorrect API key provided"}}"#,
+ r#"{"error":{"message":"Unrecognized request argument supplied: dimensions"}}"#,
+ "",
+ ] {
+ let (base, _) = stub_provider(400, body).await;
+ let result = super::embed_text(&reqwest::Client::new(), Some("key"), &base, "q").await;
+ assert!(
+ matches!(result, Err(crate::types::AppError::Internal(_))),
+ "{body} → {result:?}"
+ );
+ }
+ }
}
diff --git a/services/server/src/types.rs b/services/server/src/types.rs
index 61c113537..0da31d77d 100644
--- a/services/server/src/types.rs
+++ b/services/server/src/types.rs
@@ -1414,8 +1414,11 @@ pub struct RecallRequest {
pub scoring_weights: Option,
/// How to order results. Omitted → [`RecallSort::Relevance`], today's
/// behaviour. See [`RecallSort`].
+ ///
+ /// `Option` because an explicit `sort`, `relevance` included, suppresses
+ /// `scoring_weights`, so omitted and explicit must stay distinct.
#[serde(default)]
- pub sort: RecallSort,
+ pub sort: Option,
}
/// Result ordering mode for `/api/recall`.
@@ -3178,6 +3181,22 @@ mod tests {
.unwrap();
}
+ // ── RecallRequest.sort — omitted vs explicit ─────────────────────────
+
+ #[test]
+ fn recall_sort_keeps_omitted_apart_from_explicit_relevance() {
+ let parse = |body: &str| serde_json::from_str::(body).unwrap().sort;
+ assert_eq!(parse(r#"{"query":"q"}"#), None);
+ assert_eq!(
+ parse(r#"{"query":"q","sort":"relevance"}"#),
+ Some(RecallSort::Relevance)
+ );
+ assert_eq!(
+ parse(r#"{"query":"q","sort":"recent"}"#),
+ Some(RecallSort::Recent)
+ );
+ }
+
// ── ScoringWeights::is_ranker_active() — opt-in predicate ────────────
#[test]
From 7cbe95204ce7decf81e76bf54a103b93d00d4e35 Mon Sep 17 00:00:00 2001
From: Nikola Le <91601109+nikola0x0@users.noreply.github.com>
Date: Wed, 16 Sep 2026 09:47:49 +0700
Subject: [PATCH 014/132] fix(relayer): point the upload-queue saturation alert
at real counters (WALM-608) (#910)
* fix(relayer): point the upload-queue saturation alert at real counters (WALM-608)
The saturation monitor read queuedWalrusUploads, activeWalrusUploads and
walrusUploadLimits.globalCapacity from the sidecar's /health. Since
cef97282 (v1_new port, 31 Jul) /health is bare liveness and returns none
of them, so every read fell through to unwrap_or(0) and the alert could
never fire.
/ready has the counters but waits on Sui, Walrus and an uncached archival
GraphQL query, each with a 5s timeout, against the monitor's 2s budget.
Backlogs arrive with Sui RPC pressure, so polling it would go blind
exactly when the alert matters.
- sidecar: add GET /metrics/uploads, serving the in-memory counters and
limits with no auth, in both route modes
- relayer: poll it; move parsing and the consecutive-check state into
sidecar_saturation.rs; a missing or non-integer field, a non-JSON body
or a non-2xx status logs an error instead of reading as an empty queue
- docs: list the endpoint and the built-in alert in relayer observability
* docs(relayer): trim WALM-608 comments to what the code guarantees
Review follow-up: drop the incident history from the route, test and
parser comments; the PR body keeps it.
---------
Co-authored-by: Le Tien Phat <91601109+Niko1444@users.noreply.github.com>
---
docs/relayer/observability.md | 6 +-
.../__tests__/sidecar-upload-metrics.test.ts | 69 +++++++
services/server/scripts/sidecar-server.ts | 1 +
services/server/scripts/sidecar/app.ts | 4 +-
.../server/scripts/sidecar/routes/health.ts | 21 ++-
services/server/src/main.rs | 122 +++++++-----
services/server/src/sidecar_saturation.rs | 176 ++++++++++++++++++
7 files changed, 347 insertions(+), 52 deletions(-)
create mode 100644 services/server/scripts/__tests__/sidecar-upload-metrics.test.ts
create mode 100644 services/server/src/sidecar_saturation.rs
diff --git a/docs/relayer/observability.md b/docs/relayer/observability.md
index d465d2dd8..1563bb28d 100644
--- a/docs/relayer/observability.md
+++ b/docs/relayer/observability.md
@@ -75,12 +75,15 @@ The Rust relayer exposes Prometheus metrics at:
GET /metrics
```
-The TypeScript sidecar also exposes wallet-specific counters at:
+The TypeScript sidecar also exposes wallet-specific counters and upload-queue counters at:
```text
GET /metrics/wallet
+GET /metrics/uploads
```
+`/metrics/uploads` returns `activeWalrusUploads`, `queuedWalrusUploads`, and `walrusUploadLimits` from in-memory counters, with no Sui or Walrus calls.
+
Core relayer metrics:
| Metric | Labels | Notes |
@@ -133,6 +136,7 @@ Create panels for:
| DB saturation | PostgreSQL pool open connections near configured max, or idle connections stay at 0 |
| Wallet lock canary | Sidecar `walletLockErrorsTotal` is greater than 0 |
| Permanent wallet failures | Sidecar `walletPermanentFailuresTotal` increases |
+| Upload queue saturation | Built in: the relayer polls sidecar `/metrics/uploads` and alerts Slack when `queuedWalrusUploads` stays above `SIDECAR_QUEUE_SATURATION_THRESHOLD` (default 20) for `SIDECAR_QUEUE_SATURATION_CONSECUTIVE` checks (default 4) polled every `SIDECAR_QUEUE_SATURATION_INTERVAL_SECS` (default 30) |
## APM Integration
diff --git a/services/server/scripts/__tests__/sidecar-upload-metrics.test.ts b/services/server/scripts/__tests__/sidecar-upload-metrics.test.ts
new file mode 100644
index 000000000..2351c56a3
--- /dev/null
+++ b/services/server/scripts/__tests__/sidecar-upload-metrics.test.ts
@@ -0,0 +1,69 @@
+import test from "node:test";
+import assert from "node:assert/strict";
+import type { Server } from "node:http";
+
+const { Ed25519Keypair } = await import("@mysten/sui/keypairs/ed25519");
+process.env.SERVER_SUI_PRIVATE_KEYS = new Ed25519Keypair().getSecretKey();
+process.env.SIDECAR_AUTH_TOKEN = "upload-metrics-test-token";
+process.env.SUI_NETWORK = "testnet";
+process.env.SUI_GRPC_URL = "https://upload-metrics.testnet.example";
+process.env.WALRUS_PACKAGE_ID = `0x${"a".repeat(64)}`;
+process.env.WALRUS_UPLOAD_MAX_CONCURRENCY = "4";
+process.env.WALRUS_UPLOAD_PER_WALLET_CONCURRENCY = "1";
+process.env.WALRUS_UPLOAD_ACQUIRE_TIMEOUT_MS = "30000";
+
+const { getWalrusClient, suiClient } = await import("../sidecar/clients.js");
+let dependencyCalls = 0;
+(suiClient.ledgerService as any).getServiceInfo = async () => {
+ dependencyCalls += 1;
+ throw new Error("upload metrics must not call Sui");
+};
+(getWalrusClient() as any).getBlobType = async () => {
+ dependencyCalls += 1;
+ throw new Error("upload metrics must not call Walrus");
+};
+const { acquireWalrusUploadSlots } = await import("../sidecar/concurrency.js");
+const { createSidecarApp } = await import("../sidecar/app.js");
+
+async function listen(mode: "full" | "writer"): Promise<{ server: Server; baseUrl: string }> {
+ return new Promise((resolve) => {
+ const server = createSidecarApp(mode).listen(0, "127.0.0.1", () => {
+ const address = server.address();
+ assert.ok(address && typeof address !== "string");
+ resolve({ server, baseUrl: `http://127.0.0.1:${address.port}` });
+ });
+ });
+}
+
+async function close(server: Server): Promise {
+ await new Promise((resolve, reject) => {
+ server.close((error) => (error ? reject(error) : resolve()));
+ });
+}
+
+for (const mode of ["full", "writer"] as const) {
+ test(`${mode} mode: /metrics/uploads reports live queue counters without a token`, async () => {
+ const { server, baseUrl } = await listen(mode);
+ // Wallet 0 has one slot: the first upload runs, the second waits in the queue.
+ const releaseRunning = await acquireWalrusUploadSlots(0, "running-upload");
+ const queuedUpload = acquireWalrusUploadSlots(0, "queued-upload");
+ try {
+ const res = await fetch(`${baseUrl}/metrics/uploads`);
+ assert.equal(res.status, 200);
+ assert.deepEqual(await res.json(), {
+ activeWalrusUploads: 1,
+ queuedWalrusUploads: 1,
+ walrusUploadLimits: {
+ globalCapacity: 4,
+ perWalletCapacity: 1,
+ acquireTimeoutMs: 30000,
+ },
+ });
+ assert.equal(dependencyCalls, 0);
+ } finally {
+ releaseRunning();
+ (await queuedUpload)();
+ await close(server);
+ }
+ });
+}
diff --git a/services/server/scripts/sidecar-server.ts b/services/server/scripts/sidecar-server.ts
index a76986b6e..8ec5acef8 100644
--- a/services/server/scripts/sidecar-server.ts
+++ b/services/server/scripts/sidecar-server.ts
@@ -21,6 +21,7 @@
* GET /health → local liveness (no auth)
* GET /ready → Sui/Walrus execution identity + limits (no auth)
* GET /metrics/wallet → aggregate wallet-execution metrics (no auth)
+ * GET /metrics/uploads → upload-queue counters + limits, no I/O (no auth)
* GET /internal/wallet-balances → per-wallet balances (sidecar auth)
* /mcp/* → MCP session routes (own auth; see mcp/)
* POST /seal/encrypt → { data, owner, packageId, accountId } → { encryptedData }
diff --git a/services/server/scripts/sidecar/app.ts b/services/server/scripts/sidecar/app.ts
index b9b9eadb2..dc35a245e 100644
--- a/services/server/scripts/sidecar/app.ts
+++ b/services/server/scripts/sidecar/app.ts
@@ -3,7 +3,7 @@
*
* Registration order is load-bearing:
* 1. request-id + CORS-strip middleware run for every request.
- * 2. /health, /ready, /metrics/wallet, and full-mode MCP routes are mounted BEFORE
+ * 2. /health, /ready, /metrics/*, and full-mode MCP routes are mounted BEFORE
* the shared-secret middleware — they must stay reachable without the
* sidecar token (probes, scrapers, and MCP traffic that carries the
* end-user's own Bearer token instead).
@@ -22,6 +22,7 @@ import {
import {
registerHealthRoute,
registerInternalWalletBalancesRoute,
+ registerUploadMetricsRoute,
registerWalletMetricsRoute,
} from "./routes/health.js";
import { registerSealRoutes } from "./routes/seal.js";
@@ -72,6 +73,7 @@ export function createSidecarApp(mode: "full" | "writer" = SIDECAR_ROUTE_MODE):
// Wallet-execution metrics — placed before auth so operators / scrapers
// don't need a token.
registerWalletMetricsRoute(app);
+ registerUploadMetricsRoute(app);
app.use(sharedSecretAuthMiddleware);
diff --git a/services/server/scripts/sidecar/routes/health.ts b/services/server/scripts/sidecar/routes/health.ts
index 9b5edc9e6..82a118ca9 100644
--- a/services/server/scripts/sidecar/routes/health.ts
+++ b/services/server/scripts/sidecar/routes/health.ts
@@ -2,8 +2,9 @@
* Unauthenticated observability endpoints.
*
* All are registered BEFORE the shared-secret middleware (see app.ts):
- * /health is local liveness, /ready validates upload execution identity, and
- * /metrics/wallet exposes aggregate metrics to unauthenticated scrapers.
+ * /health is local liveness, /ready validates upload execution identity,
+ * /metrics/uploads serves the upload-queue counters, and /metrics/wallet
+ * exposes aggregate metrics to unauthenticated scrapers.
* Per-wallet addresses and balances are served separately behind sidecar auth.
*/
@@ -138,6 +139,22 @@ export function registerHealthRoute(app: Express, requireProvenance = true): voi
});
}
+// In-memory upload limiter counters for the relayer's saturation probe; no I/O.
+export function registerUploadMetricsRoute(app: Express): void {
+ app.get("/metrics/uploads", (_req: Request, res: ExpressResponse) => {
+ const uploads = getUploadCounts();
+ res.json({
+ activeWalrusUploads: uploads.active,
+ queuedWalrusUploads: uploads.queued,
+ walrusUploadLimits: {
+ globalCapacity: WALRUS_UPLOAD_MAX_CONCURRENCY,
+ perWalletCapacity: WALRUS_UPLOAD_PER_WALLET_CONCURRENCY,
+ acquireTimeoutMs: WALRUS_UPLOAD_ACQUIRE_TIMEOUT_MS,
+ },
+ });
+ });
+}
+
// Wallet-execution metrics (observability).
//
// `walletObjectLockEquivocationTotal` is the canary for concurrent uploads
diff --git a/services/server/src/main.rs b/services/server/src/main.rs
index 292bdd6c1..02bdb0e83 100644
--- a/services/server/src/main.rs
+++ b/services/server/src/main.rs
@@ -14,6 +14,7 @@ mod routes;
mod security_delete_auth;
mod security_delete_error;
mod services;
+mod sidecar_saturation;
mod storage;
mod sui;
mod types;
@@ -1431,10 +1432,10 @@ async fn main() {
// Sidecar upload-queue saturation monitor. The watchdog above only
// checks that /health answers; during the 2026-06-10 congestion incident
// it stayed green while 120 uploads queued and jobs burned their retry
- // budgets. This monitor reads the queue counters that /health already
- // exposes and alerts ops while there is still time to act (add wallets /
- // throttle the burst) — before queued requests outlive the sidecar's
- // 120s acquire timeout and start failing.
+ // budgets. This monitor reads the sidecar's upload-queue counters and
+ // alerts ops while there is still time to act (add wallets / throttle the
+ // burst) — before queued requests outlive the sidecar's 120s acquire
+ // timeout and start failing.
let saturation_threshold = parse_env_u64("SIDECAR_QUEUE_SATURATION_THRESHOLD", 20, 1, 10_000);
let saturation_consecutive = parse_env_u32("SIDECAR_QUEUE_SATURATION_CONSECUTIVE", 4, 1, 100);
let saturation_interval_secs =
@@ -1447,13 +1448,16 @@ async fn main() {
);
{
let monitor_client = state.http_client.clone();
- let monitor_url = health_url.clone();
+ let monitor_url = format!("{}{}", sidecar_url, sidecar_saturation::UPLOAD_METRICS_PATH);
let monitor_alerts = Arc::clone(&state.alerts);
let monitor_network = config.sui_network.clone();
tokio::spawn(async move {
let mut interval =
tokio::time::interval(std::time::Duration::from_secs(saturation_interval_secs));
- let mut consecutive_saturated = 0u32;
+ let mut tracker = sidecar_saturation::SaturationTracker::new(
+ saturation_threshold,
+ saturation_consecutive,
+ );
loop {
interval.tick().await;
let body = match monitor_client
@@ -1465,57 +1469,79 @@ async fn main() {
Ok(resp) if resp.status().is_success() => {
match resp.json::().await {
Ok(v) => v,
- Err(_) => continue,
+ Err(err) => {
+ tracing::error!(
+ " sidecar: upload metrics body is not JSON, saturation alert is blind: {}",
+ err
+ );
+ continue;
+ }
}
}
- // Liveness problems are the watchdog's job; only the
- // healthy-but-saturated case belongs here.
- _ => continue,
+ // The sidecar answered, so the watchdog stays green; a
+ // non-2xx here means the metrics route itself is broken.
+ Ok(resp) => {
+ tracing::error!(
+ " sidecar: {} returned status={}, saturation alert is blind",
+ sidecar_saturation::UPLOAD_METRICS_PATH,
+ resp.status()
+ );
+ continue;
+ }
+ // An unreachable sidecar is the watchdog's job.
+ Err(_) => continue,
};
- let queued = body["queuedWalrusUploads"].as_u64().unwrap_or(0);
- let active = body["activeWalrusUploads"].as_u64().unwrap_or(0);
- let global_capacity = body["walrusUploadLimits"]["globalCapacity"]
- .as_u64()
- .unwrap_or(0);
-
- if queued > saturation_threshold {
- consecutive_saturated = consecutive_saturated.saturating_add(1);
- tracing::warn!(
- " sidecar: upload queue saturated queued={} active={} capacity={} consecutive={}/{}",
- queued,
- active,
- global_capacity,
- consecutive_saturated,
- saturation_consecutive,
- );
- } else {
- if consecutive_saturated >= saturation_consecutive {
+ let sample = match sidecar_saturation::parse_upload_metrics(&body) {
+ Ok(sample) => sample,
+ Err(field) => {
+ tracing::error!(
+ " sidecar: upload metrics missing {}, saturation alert is blind",
+ field
+ );
+ continue;
+ }
+ };
+
+ match tracker.observe(sample.queued) {
+ sidecar_saturation::QueueCheck::Clear => {}
+ sidecar_saturation::QueueCheck::Drained => {
tracing::info!(
" sidecar: upload queue drained (queued={} <= threshold {})",
- queued,
+ sample.queued,
saturation_threshold,
);
}
- consecutive_saturated = 0;
- }
-
- // Alert once per crossing; the AlertManager dedup window
- // handles re-alerting if the backlog persists.
- if consecutive_saturated >= saturation_consecutive {
- let alert = crate::alerts::WalrusUploadQueueSaturatedAlert {
- sui_network: monitor_network.clone(),
- queued,
- active,
- global_capacity,
- threshold: saturation_threshold,
- consecutive_checks: consecutive_saturated,
- };
- if let Err(err) = monitor_alerts
- .notify_walrus_upload_queue_saturated(alert)
- .await
- {
- tracing::warn!(" sidecar: saturation alert delivery failed: {}", err);
+ sidecar_saturation::QueueCheck::Saturated { consecutive, alert } => {
+ tracing::warn!(
+ " sidecar: upload queue saturated queued={} active={} capacity={} consecutive={}/{}",
+ sample.queued,
+ sample.active,
+ sample.global_capacity,
+ consecutive,
+ saturation_consecutive,
+ );
+ // Alert on every saturated check past the streak; the
+ // AlertManager dedup window handles re-alerting.
+ if alert {
+ let alert = crate::alerts::WalrusUploadQueueSaturatedAlert {
+ sui_network: monitor_network.clone(),
+ queued: sample.queued,
+ active: sample.active,
+ global_capacity: sample.global_capacity,
+ threshold: saturation_threshold,
+ consecutive_checks: consecutive,
+ };
+ if let Err(err) = monitor_alerts
+ .notify_walrus_upload_queue_saturated(alert)
+ .await
+ {
+ tracing::warn!(
+ " sidecar: saturation alert delivery failed: {}",
+ err
+ );
+ }
+ }
}
}
}
diff --git a/services/server/src/sidecar_saturation.rs b/services/server/src/sidecar_saturation.rs
new file mode 100644
index 000000000..636bf1c45
--- /dev/null
+++ b/services/server/src/sidecar_saturation.rs
@@ -0,0 +1,176 @@
+//! Decision logic for the sidecar upload-queue saturation monitor.
+//!
+//! The polling loop lives in `main.rs`; parsing and the consecutive-check
+//! state live here so the alert decision can be tested without a sidecar.
+
+use serde_json::Value;
+
+/// Sidecar route serving the upload-queue counters (`registerUploadMetricsRoute`
+/// in `scripts/sidecar/routes/health.ts`).
+pub const UPLOAD_METRICS_PATH: &str = "/metrics/uploads";
+
+#[derive(Debug, Clone, Copy, PartialEq, Eq)]
+pub struct UploadQueueSample {
+ pub queued: u64,
+ pub active: u64,
+ pub global_capacity: u64,
+}
+
+/// Reads the counters from a `/metrics/uploads` body. A missing or
+/// non-integer field is `Err` with that field's JSON pointer; it must never be
+/// read as 0.
+pub fn parse_upload_metrics(body: &Value) -> Result {
+ let field =
+ |pointer: &'static str| body.pointer(pointer).and_then(Value::as_u64).ok_or(pointer);
+ Ok(UploadQueueSample {
+ queued: field("/queuedWalrusUploads")?,
+ active: field("/activeWalrusUploads")?,
+ global_capacity: field("/walrusUploadLimits/globalCapacity")?,
+ })
+}
+
+#[derive(Debug, PartialEq, Eq)]
+pub enum QueueCheck {
+ /// At or below the threshold, with no alerting streak to end.
+ Clear,
+ /// Back at or below the threshold after a streak that alerted.
+ Drained,
+ /// Above the threshold; `alert` once the streak is long enough.
+ Saturated { consecutive: u32, alert: bool },
+}
+
+pub struct SaturationTracker {
+ threshold: u64,
+ alert_after: u32,
+ consecutive: u32,
+}
+
+impl SaturationTracker {
+ pub fn new(threshold: u64, alert_after: u32) -> Self {
+ Self {
+ threshold,
+ alert_after,
+ consecutive: 0,
+ }
+ }
+
+ pub fn observe(&mut self, queued: u64) -> QueueCheck {
+ if queued > self.threshold {
+ self.consecutive = self.consecutive.saturating_add(1);
+ return QueueCheck::Saturated {
+ consecutive: self.consecutive,
+ alert: self.consecutive >= self.alert_after,
+ };
+ }
+ let alerted = self.consecutive >= self.alert_after;
+ self.consecutive = 0;
+ if alerted {
+ QueueCheck::Drained
+ } else {
+ QueueCheck::Clear
+ }
+ }
+}
+
+#[cfg(test)]
+mod tests {
+ use super::*;
+ use serde_json::json;
+
+ #[test]
+ fn parses_the_upload_metrics_body() {
+ let body = json!({
+ "activeWalrusUploads": 5,
+ "queuedWalrusUploads": 118,
+ "walrusUploadLimits": {
+ "globalCapacity": 5,
+ "perWalletCapacity": 1,
+ "acquireTimeoutMs": 120000
+ }
+ });
+ assert_eq!(
+ parse_upload_metrics(&body),
+ Ok(UploadQueueSample {
+ queued: 118,
+ active: 5,
+ global_capacity: 5,
+ })
+ );
+ }
+
+ #[test]
+ fn a_liveness_body_is_an_error_not_an_empty_queue() {
+ let body = json!({ "status": "ok", "uptimeMs": 86_400_000 });
+ assert_eq!(parse_upload_metrics(&body), Err("/queuedWalrusUploads"));
+ }
+
+ #[test]
+ fn a_missing_nested_capacity_is_an_error() {
+ let body = json!({ "activeWalrusUploads": 5, "queuedWalrusUploads": 118 });
+ assert_eq!(
+ parse_upload_metrics(&body),
+ Err("/walrusUploadLimits/globalCapacity")
+ );
+ }
+
+ #[test]
+ fn a_non_integer_counter_is_an_error() {
+ let body = json!({
+ "activeWalrusUploads": 5,
+ "queuedWalrusUploads": "118",
+ "walrusUploadLimits": { "globalCapacity": 5 }
+ });
+ assert_eq!(parse_upload_metrics(&body), Err("/queuedWalrusUploads"));
+ }
+
+ #[test]
+ fn alerts_once_the_queue_stays_above_threshold_for_the_configured_checks() {
+ let mut tracker = SaturationTracker::new(20, 3);
+ let checks: Vec = [21, 40, 120, 118].map(|q| tracker.observe(q)).into();
+ assert_eq!(
+ checks,
+ vec![
+ QueueCheck::Saturated {
+ consecutive: 1,
+ alert: false
+ },
+ QueueCheck::Saturated {
+ consecutive: 2,
+ alert: false
+ },
+ QueueCheck::Saturated {
+ consecutive: 3,
+ alert: true
+ },
+ QueueCheck::Saturated {
+ consecutive: 4,
+ alert: true
+ },
+ ]
+ );
+ }
+
+ #[test]
+ fn a_queue_at_the_threshold_resets_the_streak() {
+ let mut tracker = SaturationTracker::new(20, 3);
+ tracker.observe(21);
+ tracker.observe(21);
+ assert_eq!(tracker.observe(20), QueueCheck::Clear);
+ assert_eq!(
+ tracker.observe(21),
+ QueueCheck::Saturated {
+ consecutive: 1,
+ alert: false
+ }
+ );
+ }
+
+ #[test]
+ fn draining_is_reported_only_after_a_streak_that_alerted() {
+ let mut tracker = SaturationTracker::new(20, 2);
+ tracker.observe(50);
+ tracker.observe(50);
+ assert_eq!(tracker.observe(0), QueueCheck::Drained);
+ assert_eq!(tracker.observe(0), QueueCheck::Clear);
+ }
+}
From 0c765a03f7e8a9a2305c953807765a41f6f45ba2 Mon Sep 17 00:00:00 2001
From: Nikola Le <91601109+nikola0x0@users.noreply.github.com>
Date: Wed, 16 Sep 2026 10:05:30 +0700
Subject: [PATCH 015/132] fix(mcp,server): stop losing the delegate key when a
login is interrupted [WALM-332] (#793)
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit
* feat(server): add GET /api/whoami so a delegate key can resolve its own account
A client that holds a delegate key but lost the surrounding metadata has no
way to rebuild credentials: `account_id` is required locally, and nothing
exposed it. `find_account_by_delegate_key` is internal to the auth middleware,
`account_exists` takes an owner address and returns only `{exists}`, and
`StatsResponse` carries `owner` rather than the account id.
The middleware already resolves exactly what is needed while authenticating,
so this hands back what it computed instead of doing new work: `account_id`
and `owner` from the registry scan, `package_id` from config.
Returning `account_id` is safe here precisely because the route is
authenticated — the caller proved possession of a key registered against that
account, so it only ever learns about itself. The public existence-check route
deliberately withholds it, and that reasoning is unchanged.
The field mapping is factored into `whoami_response` so it can be tested
without a live AppState (this codebase has no axum-handler harness).
`account_id` and `owner` are both 0x-prefixed 32-byte hex, so transposing them
would compile and silently hand back the wrong identity — pinned by a test.
Motivated by WALM-332.
* fix(mcp): stop losing the delegate key when a login is interrupted
The browser registers our delegate public key on-chain — a paid, irreversible
action — and only afterwards POSTs the callback that makes us save the private
half. Until now that private half existed solely in memory (`login.ts` created
it, `saveCreds` persisted it only on success), and the callback listener lived
in the same process.
So any death in that window destroyed the only copy of a key that had already
been paid for and committed on-chain, leaving an orphaned registration nobody
could use. Nothing reported it: the browser's POST hit a closed port, and the
process was gone so it logged nothing. `handleLocalLogin` had already returned
the URL and told the client the call succeeded.
Reproduced against the real binary over stdio: preflight succeeds while alive
(200), the process is killed, and the callback POST gets ECONNREFUSED. Ports
are never reused across restarts, so the stale tab cannot reach a new listener
either — it fails at /preflight, never reaching the state check.
Note this is NOT the state-nonce mismatch WALM-332 originally described. That
path is unreachable: the callback handler gates `preflightVerified` before it
compares state, so a bad-state 403 requires the same live process to have
accepted a preflight carrying its own nonce.
The fix is write-ahead. Persist the keypair before anything can hand the public
key to a browser, and clear it once the key is safely in credentials.json. On
the next start a stranded record is reclaimed via the new authenticated
`GET /api/whoami`, which supplies the account metadata the lost callback would
have carried.
Two deliberate constraints:
- A rejected key is never deleted. A 401 means "not registered" on mainnet,
but on testnet the registry scan is disabled outright and a genuinely
registered key is rejected for want of an x-account-id hint. Deleting would
destroy a paid key in exactly that environment, so the record waits out its
24h TTL instead.
- Recovery never rolls back a newer sign-in. If the user gave up and signed
in again, adopting the older stranded key would silently downgrade them.
Also fixes the swallowed failure in `startOrReuseLoginFlow`: a login that fails
while the process is still alive (timeout, listener error) was eaten into a
`warn` and never reached the client. It now logs at error, writes to stderr,
and emits an MCP notification so the agent stops waiting on a dead flow.
The canonical signature string is duplicated across Rust and TypeScript because
they cannot share code. A silent drift there would fail only in production, as
an opaque 401, so the exact literal is pinned by a test on each side with a
comment pointing at the other.
WALM-332.
* docs(relayer): drop em dashes from the whoami section
The Sui docs style guide disallows em dashes in prose. Split the first
aside into its own sentence and parenthesized the second; wording is
otherwise unchanged.
* fix(mcp): make the WALM-332 recovery path actually able to reclaim a key
Review follow-ups on WALM-332. The write-ahead record was in place, but nothing
downstream of it worked.
- `whoami` signed `String(Date.now())`. The relayer freshness-checks
`x-timestamp` against `Utc::now().timestamp()` — seconds — so a millisecond
value was always outside the drift window and every recovery attempt 401'd
with ERR_TIMESTAMP_OUT_OF_BOUNDS. Recovery could not have succeeded once.
- Every non-200 was `rejected`, which tells the user to sign in again and revoke
the key. That is the wrong action for a 503 `AUTH_UPSTREAM_UNAVAILABLE`, a
429, or a 404 from a relayer too old to serve the route — the key is fine, and
re-registering costs gas for nothing. `rejected` is now 401/403 without the
upstream-unavailable marker; everything else is `unavailable` and retried.
- Every `loginFlow` minted a keypair and overwrote `login-pending.json`.
Recovery only runs at process start and is skipped for `--login`, so a
timed-out login followed by `memwal_login` in the same process replaced the
only copy of a key the browser may already have paid to register. A still
valid record for the same relayer is now reused, `createdAt` included so the
TTL keeps measuring from the attempt that may have registered it.
- Logout cleared `credentials.json` and left the pending record, so the next
start recovered from it and signed the user back in. Both logout paths now
clear it. Deliberately not folded into `clearCreds`, which also runs on 401
session teardown where a newer stranded key is what recovery still needs.
- `savePendingLogin` swallowed write errors, restoring the original loss with no
log line. It now logs and throws: the invariant is that the key is durable
before its public half can reach a browser, and a directory that cannot take
this file cannot take `credentials.json` either — that login was going to fail
at the callback anyway, one paid `add_delegate_key` later.
On the reviewer's suggestion to exempt `GET /api/whoami` from the testnet
`x-account-id` requirement: that gate is not policy. The registry scan behind it
runs over Sui JSON-RPC, which testnet no longer serves (auth.rs "Strategy 3"),
so exempting the route would only route it to a retired endpoint. Documented as
mainnet-only in the API reference instead.
Also reattached the orphaned `whoami` JSDoc and skipped the pending-record mode
assertion on Windows, where the bits are not enforced.
Co-Authored-By: Claude Opus 5 (1M context)
Claude-Session: https://claude.ai/code/session_01HsS2mBzMfpzy3QE8EMiKvS
* docs(relayer): drop the em dash from the whoami network note
Style-guide audit: no em dashes in prose.
Co-Authored-By: Claude Opus 5 (1M context)
Claude-Session: https://claude.ai/code/session_01HsS2mBzMfpzy3QE8EMiKvS
* fix(mcp): stop pointing a failed sign-in at revoking the key it can reclaim
Review follow-ups.
The logout comment justified keeping `clearPendingLogin` out of `clearCreds`
by saying `clearCreds` also runs on 401 session teardown. It does not: this
tree deliberately refuses to wipe credentials on a relayer 401 (creds-wipe DoS,
bridge.ts). The two logout paths are its only callers. Reworded to the reason
that is actually true — discarding a reclaimable key is a decision only an
explicit sign-out gets to make, and `clearCreds` is exported. Renamed the test
that repeated the same false claim.
The login-timeout message told the user to remove unused keys from the
dashboard. Write-ahead exists precisely so that key survives, and revoking it
destroys what the next start would reclaim. It now points at the two paths that
work: run login again and the same key is reused, or restart and it is
reclaimed.
The `rejected` stranded notice had the same defect in the other order, telling
the user to sign in again and then revoke. With keypair reuse that is
self-defeating: the sign-in adopts that very key. Reworded. `superseded` keeps
its revoke advice, which is correct there — that record is cleared, so the key
really is dangling.
Co-Authored-By: Claude Opus 5 (1M context)
Claude-Session: https://claude.ai/code/session_01HsS2mBzMfpzy3QE8EMiKvS
* docs(mcp): move the WALM-332 note into the unreleased 0.0.13 section
Review asked for a 0.0.12 -> 0.0.13 bump across package.json, the verify
script and the six plugin/marketplace manifests. origin/dev has since done
exactly that itself, for WALM-480: every manifest, the verify script and both
changelogs are already on 0.0.13, and npm still has `latest` at 0.0.12 with
only a `0.0.13-dev.0` prerelease published. So 0.0.13 is open, not released,
and this change belongs in it rather than in a further bump.
Merged dev and moved the #793 note out of the shipped 0.0.12 section into
0.0.13. docs/mcp/changelog.mdx gets the same entry, plus the release summary
and the `answer` frontmatter the reviewer flagged as missing.
Also capitalizes Testnet in the whoami note, the one style-guide violation
still outstanding from the docs audit.
* docs(mcp): apply the style-guide wording, and fix three stale comments
Review nits.
The audit re-ran on 71981f4 and still flagged the 0.0.13 bullet: `on-chain`
-> `onchain` (docs run 155 to 19 that way), and two passive constructions.
Applied to `packages/mcp/CHANGELOG.md` and `docs/mcp/changelog.mdx` together so
the two stay byte-identical, and to the `answer` frontmatter, which carried the
same hyphenation the audit does not scan.
Three comments still described advice ab1636b replaced:
- `recovery.ts` — the denied/unavailable split justified `rejected` by advice
("sign in again, then revoke the key") that the branch no longer gives. The
reason still holds under the new copy, so it now states that one.
- `recovery.ts` — `formatStrandedLoginNotice`'s JSDoc called revocation *the*
actionable step. It is now the abandon path only; naming the key serves both.
- `bridge.ts` — the logout comment pointed at "the relayer-401 handling below".
It is above: the module doc, and the SSE 401 path.
`superseded` keeps its revoke advice, which is correct there. No user-facing
string changed, so the tests pinning the unavailable copy are untouched.
* fix(mcp): send an approved stranded key to a restart, not a revoke or a retry
loginFailureNotice and the troubleshooting page still told the user to
remove the key from the dashboard and sign in again. That revokes the
only copy a restart would reclaim.
"Sign in again, the same key is reused" is not a fix for an approved key
either. ConnectMcp.tsx always sends add_delegate_key, and the contract
aborts on a key that is already registered. So the notice, the timeout
reason, the bridge warning, the rejected-recovery notice and the
troubleshooting page now split on whether the wallet step was approved:
approved means restart to reclaim, not approved means sign in again, and
the dashboard is only for abandoning the key. On Testnet, where reclaim
cannot confirm the key, remove it first and then sign in.
The notice and the rejected-recovery notice are pinned by tests.
* fix(mcp,app): stop the failure surfaces retrying a key only a restart reclaims
Two surfaces still contradicted the notice they sit next to.
auth-required appended the generic LOGIN_INSTRUCTION after a failed
sign-in, so the same blob said "restart the MCP client and it is
reclaimed" and "no terminal command, no client restart", and led with
memwal_login for a key that cannot be registered twice. A failure now
gets a retry instruction scoped to the case it fixes: the wallet step was
never approved. The concatenated blob is pinned so it cannot say both.
The dashboard's callback-failed card told the user to sign in again and
remove the unused key, which throws away the registration a restart would
reclaim. It now points at the restart, and at the dashboard only for
abandoning the key, matching the troubleshooting page it links to.
---------
Co-authored-by: Le Tien Phat <91601109+Niko1444@users.noreply.github.com>
Co-authored-by: Claude Opus 5 (1M context)
---
apps/app/src/pages/ConnectMcp.test.tsx | 12 +-
apps/app/src/pages/ConnectMcp.tsx | 9 +-
docs/mcp/changelog.mdx | 5 +-
docs/relayer/api-reference.md | 20 ++
docs/troubleshooting/overview.md | 8 +-
packages/mcp/CHANGELOG.md | 1 +
packages/mcp/src/auth-required.ts | 19 +-
packages/mcp/src/auth.ts | 154 ++++++++
packages/mcp/src/bridge.ts | 26 +-
packages/mcp/src/crypto.ts | 15 +-
packages/mcp/src/index.ts | 31 +-
packages/mcp/src/login.ts | 94 ++++-
packages/mcp/src/messages.ts | 15 +-
packages/mcp/src/recovery.ts | 289 +++++++++++++++
.../test/credential-file-permissions.test.mjs | 30 ++
.../mcp/test/login-failure-notice.test.mjs | 29 +-
.../mcp/test/login-recovery-signing.test.mjs | 95 +++++
packages/mcp/test/login-recovery.test.mjs | 338 ++++++++++++++++++
packages/mcp/test/login-write-ahead.test.mjs | 283 +++++++++++++++
.../mcp/test/logout-invalidation.test.mjs | 54 +++
services/server/src/main.rs | 5 +
services/server/src/routes/accounts.rs | 120 ++++++-
services/server/src/routes/mod.rs | 2 +-
services/server/src/types.rs | 18 +
24 files changed, 1639 insertions(+), 33 deletions(-)
create mode 100644 packages/mcp/src/recovery.ts
create mode 100644 packages/mcp/test/login-recovery-signing.test.mjs
create mode 100644 packages/mcp/test/login-recovery.test.mjs
create mode 100644 packages/mcp/test/login-write-ahead.test.mjs
diff --git a/apps/app/src/pages/ConnectMcp.test.tsx b/apps/app/src/pages/ConnectMcp.test.tsx
index ad569a68f..294631260 100644
--- a/apps/app/src/pages/ConnectMcp.test.tsx
+++ b/apps/app/src/pages/ConnectMcp.test.tsx
@@ -116,10 +116,16 @@ describe('MCP sign-in hand-off', () => {
expect(screen.getByText(/Nothing answered on your computer/i)).toBeInTheDocument()
expect(screen.getByText(/stopped waiting after this tab was already open/i)).toBeInTheDocument()
expect(screen.queryByText(/opened too late/i)).not.toBeInTheDocument()
- expect(screen.getByText(/Sign in again and open the new link straight away/i)).toBeInTheDocument()
- expect(screen.getByText(/left running through the wallet prompt/i)).toBeInTheDocument()
- expect(screen.getByText(/unused key from this attempt is already on your account/i)).toBeInTheDocument()
expect(screen.queryByText(/usually works/i)).not.toBeInTheDocument()
+ // The key is registered and saved by the client that started this
+ // sign-in, so a restart reclaims it. Signing in again cannot register
+ // it twice, and removing it from the dashboard throws it away.
+ expect(screen.getByText(/Restart your MCP client within 24 hours/i)).toBeInTheDocument()
+ expect(screen.getByText(/only if you mean to abandon it/i)).toBeInTheDocument()
+ expect(
+ screen.queryByText(/Remove it from the dashboard if you are not using it/i),
+ ).not.toBeInTheDocument()
+ expect(screen.queryByText(/Sign in again and open the new link straight away/i)).not.toBeInTheDocument()
})
it('explains an expired link when preflight cannot reach the listener', async () => {
diff --git a/apps/app/src/pages/ConnectMcp.tsx b/apps/app/src/pages/ConnectMcp.tsx
index 99ddf63e4..9e6edc4b4 100644
--- a/apps/app/src/pages/ConnectMcp.tsx
+++ b/apps/app/src/pages/ConnectMcp.tsx
@@ -620,10 +620,11 @@ function SuccessCard({
: 'The app on your computer rejected this hand-off. That usually means this tab is leftover from a sign-in that already finished, or the request did not match what the app expected.'}
- Sign in again and open the new link straight away. A
- retry only helps once the MCP client is left running through the wallet
- prompt. The unused key from this attempt is already on your account.
- Remove it from the dashboard if you are not using it.{' '}
+ Restart your MCP client within 24 hours. On start it
+ finds the key from this attempt and signs you in, with no second wallet
+ prompt. Do not sign in again first: this key is already registered to
+ your account, and it cannot be registered twice. Remove it from the
+ dashboard only if you mean to abandon it.{' '}
{config.docsUrl && (
-
- The latest MCP package release is 0.0.13. A `memwal_remember`, `memwal_remember_bulk` or `memwal_analyze` whose reply never arrives is no longer reported as safe to retry: the relayer accepts those with HTTP 202 and finishes them in a durable queue, so the write may already have landed, and `/api/remember/bulk` has no idempotency key — repeating it stores a second paid copy. The tool now says the write may have completed and points at `memwal_recall` to check before re-saving, while a lost read still says plainly that retrying is safe. When the relayer rejects the saved delegate key, tool calls now get an auth error pointing at `memwal_login` instead of waiting minutes for a retry hint that cannot work. It writes the credentials file by creating a new 0600 file and renaming it into place, so a sign-in never puts the delegate private key into a credentials.json that a manual chmod or a restored backup left world-readable. A completed sign-in is now confirmed with a notification and a one-shot banner naming the account and the resolved credentials path, and the bridge keeps reading stdin after an in-session login instead of going deaf. Unrecognised command-line options now warn instead of being silently ignored, `--help` lists the network presets and the URLs each resolves to, `memwal_health` names the relayer the client dialled, and `memwal_restore` reports `failed` and retries the same page when truncation is a download or embed blip, instead of always telling the agent to raise `limit`. Plugin launch configs (`.mcp.json`, Cursor/Codex copies, and the Codex fallback installer) now pin `@mysten-incubation/memwal-mcp@0.0.13` so npx cannot keep a cached 0.0.5. Version 0.0.12 forwards the MCP client's initialize.clientInfo to the relayer so sidecar logs can name the coding agent (Claude Code, Codex, Cursor, and others) on each session and tool call, and it resolves the credential directory on every access so MEMWAL_CREDS_DIR can override it.
+ The latest MCP package release is 0.0.13. A `memwal_remember`, `memwal_remember_bulk` or `memwal_analyze` whose reply never arrives is no longer reported as safe to retry: the relayer accepts those with HTTP 202 and finishes them in a durable queue, so the write may already have landed, and `/api/remember/bulk` has no idempotency key — repeating it stores a second paid copy. The tool now says the write may have completed and points at `memwal_recall` to check before re-saving, while a lost read still says plainly that retrying is safe. When the relayer rejects the saved delegate key, tool calls now get an auth error pointing at `memwal_login` instead of waiting minutes for a retry hint that cannot work. It saves the delegate keypair before the sign-in URL reaches the browser and reclaims it on the next start, so a login interrupted after the onchain registration no longer loses a key the user already paid for. It writes the credentials file by creating a new 0600 file and renaming it into place, so a sign-in never puts the delegate private key into a credentials.json that a manual chmod or a restored backup left world-readable. A completed sign-in is now confirmed with a notification and a one-shot banner naming the account and the resolved credentials path, and the bridge keeps reading stdin after an in-session login instead of going deaf. Unrecognised command-line options now warn instead of being silently ignored, `--help` lists the network presets and the URLs each resolves to, `memwal_health` names the relayer the client dialled, and `memwal_restore` reports `failed` and retries the same page when truncation is a download or embed blip, instead of always telling the agent to raise `limit`. Plugin launch configs (`.mcp.json`, Cursor/Codex copies, and the Codex fallback installer) now pin `@mysten-incubation/memwal-mcp@0.0.13` so npx cannot keep a cached 0.0.5. Version 0.0.12 forwards the MCP client's initialize.clientInfo to the relayer so sidecar logs can name the coding agent (Claude Code, Codex, Cursor, and others) on each session and tool call, and it resolves the credential directory on every access so MEMWAL_CREDS_DIR can override it.
---
## 0.0.13
-This release stops a write whose reply was lost from being reported as safe to retry — repeating one can store a second paid copy — answers tool calls with an auth error pointing at `memwal_login` when the relayer rejects the saved delegate key, writes the credentials file through a fresh `0600` file that it renames into place, confirms a completed sign-in and keeps the bridge reading stdin afterwards, warns on unrecognised command-line options instead of ignoring them, documents the network presets in `--help`, names the relayer in `memwal_health`, reports restore `failed` counts when truncation is a transient download or embed blip, and pins plugin launch configs (`.mcp.json`, Cursor/Codex copies, and the Codex fallback installer) so npx cannot keep a cached 0.0.5.
+This release stops a write whose reply was lost from being reported as safe to retry — repeating one can store a second paid copy — answers tool calls with an auth error pointing at `memwal_login` when the relayer rejects the saved delegate key, saves the delegate keypair before sign-in hands a URL to the browser and reclaims it on the next start, writes the credentials file through a fresh `0600` file that it renames into place, confirms a completed sign-in and keeps the bridge reading stdin afterwards, warns on unrecognised command-line options instead of ignoring them, documents the network presets in `--help`, names the relayer in `memwal_health`, reports restore `failed` counts when truncation is a transient download or embed blip, and pins plugin launch configs (`.mcp.json`, Cursor/Codex copies, and the Codex fallback installer) so npx cannot keep a cached 0.0.5.
### Fixed
@@ -41,6 +41,7 @@ This release stops a write whose reply was lost from being reported as safe to r
- Answer tool calls with an auth error when the relayer rejects the saved delegate key, instead of parking them until the call deadline. A 401 on the SSE handshake was treated like any other connect failure, so the bridge retried a key that could never be accepted while the queued `memwal_recall` waited out the orphan sweeper — up to four minutes — and then came back as "the connection to the relayer dropped, please retry", advice that cannot work. The bridge now names the rejection and points at `memwal_login`, whether the key is rejected at startup or revoked mid-session, and refuses later requests immediately while it stays rejected. Any accepted handshake resumes normal buffering, so both a re-login and a transient WAF or rate-limit 401 recover on their own. Credentials are still never wiped automatically. (#365, WALM-602)
- Back off when the relayer refuses the SSE handshake with HTTP 429 instead of retrying ~500ms later. The bridge now honours a `Retry-After` (clamped to 60s) and falls back to a 5s floor when the header is absent — the `ip_active_cap` shape — and carries the deadline across reconnect attempts. It also prints one stderr line saying this is a rate limit rather than a bad config or bad credentials, so a throttled bridge no longer reads as a broken one. (WALM-386)
- Answer a request that is still buffered while the handshake keeps failing, instead of holding it for the full call timeout and then blaming a dropped reply. A request that was never sent cannot have executed, so after 90s of consecutive handshake failures the bridge fails it with the reason the handshake actually gave — rather than parking it four minutes and returning "the connection to the relayer dropped, please retry", which named the wrong layer and invited a retry of a `remember` that had never left the process. The full deadline still applies while the handshake is healthy. Override with `MEMWAL_MCP_STALLED_HANDSHAKE_MS`. (WALM-618)
+- Persist the delegate keypair before sign-in hands the URL to the browser, and reclaim it on the next start. The browser's onchain `add_delegate_key` costs gas and is irreversible, and it happens before the callback that saves the private half, so a client that died in that window destroyed the only copy of a key the user had already paid for and left an orphaned registration nobody could use. (#793) Signing out discards the pending record along with the credentials, a second sign-in against the same relayer reuses the stranded key rather than minting over it, and a sign-in that cannot write the record fails instead of publishing a URL it cannot back. Reclaiming works on Mainnet; Testnet requires an account-id hint the recovering client does not have.
- `memwal_restore` reports `failed` and retries the same page when `truncated` is a download/embed blip (`restored=0` and `skipped+failed < total`), instead of always telling the agent to raise `limit` (WALM-480).
- Unrecognised options now warn on stderr and in the structured log (`cli.unrecognised_arg`) instead of being dropped in silence, so a typo'd `--namesapce work` no longer writes to the default namespace with nothing to say it had. The warning names the option key only, keeping a mistyped value-taking flag (`--tokenn=hunter2`) from putting the secret on stderr, and it warns rather than exits so an option from a newer config cannot brick the server. (#630)
- `--help` now lists the network presets (`--prod`, `--dev`, `--staging`, `--local`) with the relayer and web URLs each resolves to, rendered from the preset table rather than retyped so a new preset cannot ship undocumented the way `--prod` did. The `--label` default is corrected to "MCP Client", which is what the code actually falls back to. (#630)
diff --git a/docs/relayer/api-reference.md b/docs/relayer/api-reference.md
index 603765558..0ffd9a42d 100644
--- a/docs/relayer/api-reference.md
+++ b/docs/relayer/api-reference.md
@@ -159,6 +159,26 @@ Proxy to the sidecar's `/sponsor/execute` endpoint. `sender` must match the shor
Every route below requires the signed headers described in [Authentication](#authentication).
+### `GET /api/whoami`
+
+Return the account identity the caller's delegate key resolves to. Takes no request body.
+
+Authentication already resolves the account before any handler runs, so this route just hands back what the middleware computed. Returning `account_id` is safe here precisely because the route is authenticated. The caller has proven it holds a delegate key registered against this account, so it only ever learns about itself. The public `GET /api/accounts/:owner/exists` route deliberately withholds it.
+
+The motivating use is rebuilding local credentials: a client that holds a working delegate key but has lost the surrounding metadata (an interrupted sign-in, a wiped config file) needs `account_id`, `owner`, and `package_id` to write a usable credentials file, and the key alone proves entitlement to all three.
+
+**Response:**
+
+```json
+{
+ "account_id": "0x...",
+ "owner": "0x...",
+ "package_id": "0x..."
+}
+```
+
+**Mainnet only, when the caller cannot send `x-account-id`.** Recovering a lost account id is the one case where the client has no id to send, so authentication has to find it by scanning the `AccountRegistry` for the delegate key. That scan runs over Sui JSON-RPC, which Testnet no longer serves, so Testnet requires the `x-account-id` hint for delegate-key authentication and rejects the request with `401` when it is absent, including this one. A caller that already knows its account id can use this route on either network; a caller recovering one cannot use it on Testnet.
+
### `POST /api/remember`
Submit text as an encrypted memory job. The relayer returns after creating a background job; embedding, Seal encryption, Walrus upload, and vector indexing continue asynchronously.
diff --git a/docs/troubleshooting/overview.md b/docs/troubleshooting/overview.md
index f97660bd1..eaa9fc1a5 100644
--- a/docs/troubleshooting/overview.md
+++ b/docs/troubleshooting/overview.md
@@ -88,13 +88,17 @@ This section covers problems that appear before the memory tools work.
**Symptom:** The sign-in page confirms that your delegate key was registered, but says it could not hand the credentials back to your computer. The agent stays logged out and `~/.memwal/credentials.json` does not appear.
-**Cause:** Signing in has two halves. Your browser registers a delegate key onchain, then sends that key back to a short-lived listener the MCP package runs on `127.0.0.1`. The unused key from this attempt is already on your account and should be revoked. Signing in again is a full new attempt, including the wallet step. The usual reasons:
+**Cause:** Signing in has two halves. Your browser registers a delegate key onchain, then sends that key back to a short-lived listener the MCP package runs on `127.0.0.1`. The key from this attempt is already on your account, and the MCP package saved its private half before it opened the sign-in page, so you can still use it. The usual reasons the hand-off fails:
- The MCP client restarted, or the login command was cancelled, while the browser tab was still open.
- This tab is leftover from a sign-in that already finished, or the hand-off did not match what the app expected.
- Local software such as a firewall, a VPN client, or a browser extension blocks requests from a website to `127.0.0.1`.
-**Fix:** Call `memwal_login` again and open the new URL promptly. A retry only helps once the MCP client is left running through the wallet prompt. Remove the unused key from the Delegate keys panel in the dashboard; it is already on your account.
+**Fix:** Restart your MCP client within 24 hours. On start, the MCP package finds the saved key, confirms it with the relayer, and signs you in with it, with no second wallet step. Do not call `memwal_login` first. It reuses the same key, and the wallet step fails because that key is already registered. Do not remove the key from the dashboard either, unless you mean to abandon it.
+
+On Testnet the relayer cannot confirm the key at start, so the MCP package prints a notice instead of signing you in. Remove the key it names from the Delegate keys panel in the dashboard, then call `memwal_login` and open the new URL promptly.
+
+MCP package versions before 0.0.13 do not save the key before the browser step. On those versions, remove the unused key from the dashboard and call `memwal_login` again.
If it keeps failing, confirm that nothing blocks localhost traffic, then run `npx -y @mysten-incubation/memwal-mcp login --prod` directly in a terminal. A terminal sign-in prints the failure reason instead of leaving it in the MCP client's logs.
diff --git a/packages/mcp/CHANGELOG.md b/packages/mcp/CHANGELOG.md
index 309ffa8b0..4b5b5d2f7 100644
--- a/packages/mcp/CHANGELOG.md
+++ b/packages/mcp/CHANGELOG.md
@@ -8,6 +8,7 @@
- Answer tool calls with an auth error when the relayer rejects the saved delegate key, instead of parking them until the call deadline. A 401 on the SSE handshake was treated like any other connect failure, so the bridge retried a key that could never be accepted while the queued `memwal_recall` waited out the orphan sweeper — up to four minutes — and then came back as "the connection to the relayer dropped, please retry", advice that cannot work. The bridge now names the rejection and points at `memwal_login`, whether the key is rejected at startup or revoked mid-session, and refuses later requests immediately while it stays rejected. Any accepted handshake resumes normal buffering, so both a re-login and a transient WAF or rate-limit 401 recover on their own. Credentials are still never wiped automatically. (#365, WALM-602)
- Back off when the relayer refuses the SSE handshake with HTTP 429 instead of retrying ~500ms later. The bridge now honours a `Retry-After` (clamped to 60s) and falls back to a 5s floor when the header is absent — the `ip_active_cap` shape — and carries the deadline across reconnect attempts. It also prints one stderr line saying this is a rate limit rather than a bad config or bad credentials, so a throttled bridge no longer reads as a broken one. (WALM-386)
- Answer a request that is still buffered while the handshake keeps failing, instead of holding it for the full call timeout and then blaming a dropped reply. A request that was never sent cannot have executed, so after 90s of consecutive handshake failures the bridge fails it with the reason the handshake actually gave — rather than parking it four minutes and returning "the connection to the relayer dropped, please retry", which named the wrong layer and invited a retry of a `remember` that had never left the process. The full deadline still applies while the handshake is healthy. Override with `MEMWAL_MCP_STALLED_HANDSHAKE_MS`. (WALM-618)
+- Persist the delegate keypair before sign-in hands the URL to the browser, and reclaim it on the next start. The browser's onchain `add_delegate_key` costs gas and is irreversible, and it happens before the callback that saves the private half, so a client that died in that window destroyed the only copy of a key the user had already paid for and left an orphaned registration nobody could use. (#793) Signing out discards the pending record along with the credentials, a second sign-in against the same relayer reuses the stranded key rather than minting over it, and a sign-in that cannot write the record fails instead of publishing a URL it cannot back. Reclaiming works on Mainnet; Testnet requires an account-id hint the recovering client does not have.
- `memwal_restore` reports `failed` and retries the same page when `truncated` is a download/embed blip (`restored=0` and `skipped+failed < total`), instead of always telling the agent to raise `limit` (WALM-480).
- Unrecognised options now warn on stderr and in the structured log (`cli.unrecognised_arg`) instead of being dropped in silence, so a typo'd `--namesapce work` no longer writes to the default namespace with nothing to say it had. The warning names the option key only, keeping a mistyped value-taking flag (`--tokenn=hunter2`) from putting the secret on stderr, and it warns rather than exits so an option from a newer config cannot brick the server. (#630)
- `--help` now lists the network presets (`--prod`, `--dev`, `--staging`, `--local`) with the relayer and web URLs each resolves to, rendered from the preset table rather than retyped so a new preset cannot ship undocumented the way `--prod` did. The `--label` default is corrected to "MCP Client", which is what the code actually falls back to. (#630)
diff --git a/packages/mcp/src/auth-required.ts b/packages/mcp/src/auth-required.ts
index ed6bdfda2..0a8f808d6 100644
--- a/packages/mcp/src/auth-required.ts
+++ b/packages/mcp/src/auth-required.ts
@@ -200,6 +200,18 @@ const LOGIN_INSTRUCTION = [
"`add_delegate_key` transaction. Credentials land at `~/.memwal/credentials.json`.",
].join("\n");
+/** Replaces {@link LOGIN_INSTRUCTION} after a failed attempt, which
+ * {@link loginFailureNotice} has just described. The generic copy promises "no
+ * client restart" and leads with `memwal_login`; for a key the user already
+ * approved, a restart is the only thing that recovers it and signing in again
+ * cannot register it twice. So the retry is offered only for the case it
+ * actually fixes. */
+const LOGIN_RETRY_INSTRUCTION = [
+ "If you did not approve the wallet step, start a new sign-in: call the `memwal_login`",
+ "tool from this client, or run `npx -y @mysten-incubation/memwal-mcp login`. Open the",
+ "new link straight away and leave this client running through the wallet prompt.",
+].join("\n");
+
/** Set when a background `memwal_login` ends without credentials. The tool call
* already returned the URL by then, so this is the only place left to say so. */
let lastLoginFailure: string | null = null;
@@ -473,7 +485,12 @@ function handleAuthLine(
id,
result: {
content: [
- { type: "text", text: `${loginFailureNotice(lastLoginFailure)}${LOGIN_INSTRUCTION}` },
+ {
+ type: "text",
+ text: lastLoginFailure
+ ? `${loginFailureNotice(lastLoginFailure)}${LOGIN_RETRY_INSTRUCTION}`
+ : LOGIN_INSTRUCTION,
+ },
],
isError: true,
},
diff --git a/packages/mcp/src/auth.ts b/packages/mcp/src/auth.ts
index be9d3bf73..828570d3b 100644
--- a/packages/mcp/src/auth.ts
+++ b/packages/mcp/src/auth.ts
@@ -20,6 +20,7 @@ import {
unlinkSync,
existsSync,
} from "node:fs";
+import { log } from "./logger.js";
export interface MemWalCredentials {
/** 64-hex Ed25519 private key seed (32 bytes). NEVER log this. */
@@ -405,3 +406,156 @@ function isValid(obj: unknown): obj is MemWalCredentials {
c.version === 1
);
}
+
+/* ------------------------------------------------------------------------- *
+ * Pending login — write-ahead for the delegate keypair (WALM-332).
+ *
+ * The browser registers our delegate public key on-chain, which costs gas and
+ * cannot be undone, and only afterwards POSTs the callback that makes us save
+ * the matching private key. Losing this process in that window used to destroy
+ * the only copy of the key, stranding a paid registration nobody could use.
+ *
+ * So the keypair is written here BEFORE the browser is given the connect URL,
+ * and cleared once `saveCreds` has the key safely in `credentials.json`. A
+ * record that outlives its flow is recovered on next start.
+ * ------------------------------------------------------------------------- */
+
+const PENDING_FILE = "login-pending.json";
+
+/**
+ * How long a stranded record stays recoverable.
+ *
+ * Deliberately far longer than the 5-minute login timeout: the whole point is
+ * to survive a client restart, and a user who quits for the evening and comes
+ * back tomorrow is exactly the case worth covering. The cost of holding it is
+ * an unregistered key on disk, which grants nothing.
+ */
+export const PENDING_LOGIN_TTL_MS = 24 * 60 * 60_000;
+
+export interface PendingLogin {
+ /** 64-hex Ed25519 private key seed. NEVER log this. */
+ delegatePrivateKey: string;
+ delegatePublicKeyHex: string;
+ delegateAddress: string;
+ /** Relayer the flow was started against — recovery must not repoint. */
+ relayerUrl: string;
+ label?: string;
+ /** ISO timestamp, for TTL expiry. */
+ createdAt: string;
+ version: 1;
+}
+
+/** Sits beside whichever credentials file `credsPath()` resolves to, so a
+ * project-local sign-in recovers into that same project. */
+export function pendingLoginPath(): string {
+ return join(dirname(credsPath()), PENDING_FILE);
+}
+
+/**
+ * Persist the pending keypair. Throws if it cannot.
+ *
+ * Deliberately NOT best-effort. The invariant this record exists to hold is
+ * that the delegate private key is on disk before its public half can reach a
+ * browser that will pay gas to register it. Swallowing the error would publish
+ * the connect URL while claiming a durability that does not exist — the
+ * original WALM-332 loss, now silent.
+ *
+ * Failing the login costs the user nothing: this file sits beside
+ * `credentials.json`, so a directory that cannot take it cannot take the
+ * credentials either. The same login would have failed at the callback anyway,
+ * one on-chain `add_delegate_key` later.
+ */
+export function savePendingLogin(pending: PendingLogin): void {
+ const path = pendingLoginPath();
+ try {
+ // The record holds the same plaintext private key as `credentials.json`,
+ // so it gets the same fresh-inode write.
+ writeSecretFile(path, JSON.stringify(pending, null, 2));
+ } catch (err) {
+ const msg = err instanceof Error ? err.message : String(err);
+ log.error("login.pending.write_failed", { path, msg });
+ throw new Error(
+ `Could not write the login write-ahead record at ${path}: ${msg}. ` +
+ `Refusing to start a sign-in that could register a delegate key on-chain ` +
+ `without being able to save it.`,
+ );
+ }
+}
+
+/**
+ * A pending record this login can adopt instead of minting a new keypair.
+ *
+ * `loginFlow` used to generate a fresh keypair every call and overwrite the
+ * record unconditionally. Recovery only runs at process start and is skipped
+ * for `--login` / `forceLogin`, so a timed-out login followed by `memwal_login`
+ * in the same process replaced the only copy of a key the browser may already
+ * have paid to register. Reusing the record keeps that key reclaimable.
+ *
+ * Scoped to the same relayer: a key registered against one relayer's account
+ * proves nothing to another, and `recovery` must never repoint. TTL and shape
+ * are already enforced by {@link loadPendingLogin}.
+ */
+export function reusablePendingLogin(relayerUrl: string): PendingLogin | null {
+ const pending = loadPendingLogin();
+ if (!pending) return null;
+ if (pending.relayerUrl !== relayerUrl) {
+ log.warn("login.pending.relayer_changed", {
+ publicKey: pending.delegatePublicKeyHex,
+ from: pending.relayerUrl,
+ to: relayerUrl,
+ });
+ return null;
+ }
+ return pending;
+}
+
+/**
+ * Load a pending record, or null if there is none, it is malformed, or it has
+ * aged out. An expired record is deleted on read rather than left to linger.
+ */
+export function loadPendingLogin(): PendingLogin | null {
+ const path = pendingLoginPath();
+ if (!existsSync(path)) return null;
+ let parsed: unknown;
+ try {
+ parsed = JSON.parse(readFileSync(path, "utf8"));
+ } catch {
+ clearPendingLogin();
+ return null;
+ }
+ if (!isValidPending(parsed)) {
+ clearPendingLogin();
+ return null;
+ }
+ const age = Date.now() - Date.parse(parsed.createdAt);
+ if (!Number.isFinite(age) || age > PENDING_LOGIN_TTL_MS) {
+ clearPendingLogin();
+ return null;
+ }
+ return parsed;
+}
+
+/** Remove the pending record. Safe to call when there isn't one. */
+export function clearPendingLogin(): void {
+ try {
+ const path = pendingLoginPath();
+ if (existsSync(path)) unlinkSync(path);
+ } catch {
+ /* best effort */
+ }
+}
+
+function isValidPending(obj: unknown): obj is PendingLogin {
+ if (!obj || typeof obj !== "object") return false;
+ const p = obj as Record;
+ return (
+ typeof p.delegatePrivateKey === "string" &&
+ /^[0-9a-fA-F]{64}$/.test(p.delegatePrivateKey) &&
+ typeof p.delegatePublicKeyHex === "string" &&
+ /^[0-9a-fA-F]{64}$/.test(p.delegatePublicKeyHex) &&
+ typeof p.delegateAddress === "string" &&
+ typeof p.relayerUrl === "string" &&
+ typeof p.createdAt === "string" &&
+ p.version === 1
+ );
+}
diff --git a/packages/mcp/src/bridge.ts b/packages/mcp/src/bridge.ts
index 6b9319ee3..1b0e0df02 100644
--- a/packages/mcp/src/bridge.ts
+++ b/packages/mcp/src/bridge.ts
@@ -15,7 +15,7 @@
* Re-auth requires an explicit `memwal-mcp login` from the user.
*/
import type { MemWalCredentials } from "./auth.js";
-import { clearCreds, credsPath, loadCreds } from "./auth.js";
+import { clearCreds, clearPendingLogin, credsPath, loadCreds } from "./auth.js";
import { TOOL_DEFINITIONS } from "./auth-required.js";
import {
clientInfoHeaders,
@@ -810,6 +810,11 @@ async function handleLocalLogin(
},
});
},
+ // This tool call has already returned "here is your URL, go sign in",
+ // so a later failure has no response left to ride home on. Without an
+ // out-of-band notification the agent sits waiting on a flow that is
+ // already dead. MCP logging notifications are fire-and-forget and safe
+ // to emit at any point in the session.
(err) => {
const msg = err instanceof Error ? err.message : String(err);
log.warn("memwal_login.bridge.failed", { msg });
@@ -819,7 +824,15 @@ async function handleLocalLogin(
params: {
level: "warning",
logger: "memwal-mcp",
- data: `Walrus Memory sign-in did not complete: ${msg}. Existing credentials are unchanged; call memwal_login again to retry.`,
+ // The reclaim is only possible because of the write-ahead
+ // record (WALM-332): a key the browser already paid to
+ // register is no longer lost with the process. A retry
+ // cannot help that key, since it reuses it and the
+ // dashboard cannot register it twice.
+ data:
+ `Walrus Memory sign-in did not complete: ${msg}. Existing credentials are ` +
+ `unchanged. If you approved the wallet step, the next start reclaims that ` +
+ `key; otherwise call memwal_login again to retry.`,
},
});
},
@@ -861,6 +874,15 @@ async function handleLocalLogin(
function handleLocalLogout(): { text: string; isError: boolean } {
try {
const cleared = clearCreds();
+ // Explicit sign-out discards the write-ahead record too. Without this
+ // an interrupted re-login leaves `login-pending.json` behind, and the
+ // next start's `recoverPendingLogin` signs the user straight back in.
+ //
+ // Kept out of `clearCreds()` so only a deliberate sign-out discards a
+ // key that may still be reclaimable. `clearCreds` is exported, and a
+ // 401 deliberately does NOT wipe credentials (see the relayer-401
+ // handling above), so the two are not the same decision.
+ clearPendingLogin();
log.info("memwal_logout.bridge.success", {
removedPath: cleared.removedPath ?? null,
fallbackPath: cleared.fallbackPath ?? null,
diff --git a/packages/mcp/src/crypto.ts b/packages/mcp/src/crypto.ts
index 2d138a96a..fbc3aa53e 100644
--- a/packages/mcp/src/crypto.ts
+++ b/packages/mcp/src/crypto.ts
@@ -1,7 +1,7 @@
/**
* Ed25519 helpers — pure-JS via @noble/ed25519.
*/
-import { getPublicKeyAsync, utils } from "@noble/ed25519";
+import { getPublicKeyAsync, signAsync, utils } from "@noble/ed25519";
import { blake2b } from "@noble/hashes/blake2.js";
function hex(bytes: Uint8Array): string {
@@ -48,4 +48,17 @@ export function deriveSuiAddress(pubKey: Uint8Array): string {
return "0x" + hex(digest);
}
+/**
+ * Sign a canonical request message with a delegate private key.
+ *
+ * The relayer authenticates `/api/*` by Ed25519 signature over
+ * `{timestamp}.{method}.{path_and_query}.{body_sha256}.{nonce}.{account_id}`
+ * (`services/server/src/auth.rs`, which calls itself the single source of
+ * truth for that format). Keep the two in lockstep.
+ */
+export async function signMessage(privateKeyHex: string, message: string): Promise {
+ const sig = await signAsync(new TextEncoder().encode(message), fromHex(privateKeyHex));
+ return hex(sig);
+}
+
export { hex as bytesToHex, fromHex as hexToBytes };
diff --git a/packages/mcp/src/index.ts b/packages/mcp/src/index.ts
index 02b9b3fff..9c0ded0dd 100644
--- a/packages/mcp/src/index.ts
+++ b/packages/mcp/src/index.ts
@@ -9,7 +9,8 @@
* 5. On 401 (revoked key), the bridge wipes credentials before throwing
* — the next process spawn will re-trigger login.
*/
-import { clearCreds, credsPath, loadCreds } from "./auth.js";
+import { clearCreds, clearPendingLogin, credsPath, loadCreds } from "./auth.js";
+import { recoverPendingLogin, formatStrandedLoginNotice } from "./recovery.js";
import { runAuthRequiredServer } from "./auth-required.js";
import { notePendingLoginSuccess, runBridge } from "./bridge.js";
import { loginFlow } from "./login.js";
@@ -149,6 +150,15 @@ export async function main(argv: string[] = process.argv.slice(2)): Promise {
if (!creds) {
error = new Error(
- `Login timed out after ${cfg.timeoutMs}ms. If you already approved the wallet transaction, a delegate key may exist on-chain without local credentials. Remove unused keys from the dashboard, then run login again.`,
+ `Login timed out after ${cfg.timeoutMs}ms. If you already approved the wallet transaction, that delegate key is saved on disk: the next start reclaims it, and running login again cannot register it a second time. Only remove it from the dashboard if you mean to abandon it.`,
);
server.close();
resolve();
@@ -431,6 +478,9 @@ export async function loginFlow(opts: LoginOptions = {}): Promise Promise | void,
+ /** Called when the flow fails after the caller has already been handed the
+ * URL and returned. Without this the failure is unobservable — see the
+ * catch below. */
onFailure?: (err: unknown) => void,
): InflightLogin {
if (inflightLogin) return inflightLogin;
@@ -513,24 +566,45 @@ export function startOrReuseLoginFlow(
},
});
+ // The one place a failed flow is reported. It has to be this catch rather
+ // than the `onSuccess` chain below: a caller that only passes `onFailure`
+ // still needs to hear about it, and reporting from both would emit the
+ // failure twice for one flow.
result.catch((err) => {
rejectUrl(err);
+ const msg = err instanceof Error ? err.message : String(err);
+ // This used to be a `warn` and nothing else, which made a background
+ // login failure invisible: `handleLocalLogin` has already returned the
+ // URL and told the client it succeeded, so without a signal here the
+ // agent waits forever on a flow that is already dead (WALM-332). Error
+ // level so it surfaces in client log views, `note` so it reads as a
+ // sentence on stderr, and `onFailure` so the caller can put it in
+ // front of the agent in-band. The pending write-ahead record is
+ // deliberately left in place — the next start reports the stranded key.
+ log.error("login.inflight.failed", { msg });
+ note(`Walrus Memory sign-in failed: ${msg}`);
try {
onFailure?.(err);
} catch {
- /* caller errors don't break the flow */
+ /* a reporting failure must not mask the original one */
}
});
if (onSuccess) {
- result
+ void result
.then(async (creds) => {
- await onSuccess(creds);
+ try {
+ await onSuccess(creds);
+ } catch (err) {
+ // A credential handoff that throws is its own failure, and
+ // distinct from the flow failing — the sign-in did work.
+ log.error("login.on_success_failed", {
+ msg: err instanceof Error ? err.message : String(err),
+ });
+ }
})
- .catch((err) => {
- log.warn("login.inflight.failed", {
- msg: err instanceof Error ? err.message : String(err),
- });
+ .catch(() => {
+ /* the flow's own rejection is reported in the catch above */
});
}
diff --git a/packages/mcp/src/messages.ts b/packages/mcp/src/messages.ts
index 11e7d6176..c9143a402 100644
--- a/packages/mcp/src/messages.ts
+++ b/packages/mcp/src/messages.ts
@@ -117,6 +117,11 @@ export function loginSuccessNotification(info: LoginSuccessInfo): string {
*
* `reason` null — no attempt on record — yields the empty string, so callers
* can prefix unconditionally.
+ *
+ * The advice splits on whether the user approved the wallet step. If they did,
+ * the key is on-chain and in the write-ahead record (WALM-332), so a restart
+ * reclaims it. Signing in again would reuse that same key, and the dashboard's
+ * `add_delegate_key` aborts on a key that is already registered.
*/
export function loginFailureNotice(reason: string | null): string {
if (!reason) return "";
@@ -125,10 +130,12 @@ export function loginFailureNotice(reason: string | null): string {
"",
`Reason: ${reason}`,
"",
- "The unused key from this attempt may already be registered on your account. Remove it",
- "from the dashboard if you are not using it. Sign in again and open the new link",
- "straight away. A retry only helps once the MCP client is left running through the",
- "wallet prompt.",
+ "If you approved the wallet step, that key is registered and saved on this machine.",
+ "Restart the MCP client within 24 hours and it is reclaimed. Signing in again cannot",
+ "register the same key twice, and removing it from the dashboard abandons it.",
+ "",
+ "If you did not approve it, sign in again and open the new link straight away. A",
+ "retry only helps once the MCP client is left running through the wallet prompt.",
"",
"---",
"",
diff --git a/packages/mcp/src/recovery.ts b/packages/mcp/src/recovery.ts
new file mode 100644
index 000000000..468caf0bd
--- /dev/null
+++ b/packages/mcp/src/recovery.ts
@@ -0,0 +1,289 @@
+/**
+ * Recovery for a login that was interrupted after the browser registered our
+ * delegate key on-chain but before the callback could save it (WALM-332).
+ *
+ * `loginFlow` write-aheads the keypair to `login-pending.json` before the
+ * browser can act, so the key itself survives losing the process. What does
+ * not survive is the metadata the callback would have carried — `accountId`,
+ * `walletAddress`, `packageId` — and `credentials.json` is not loadable
+ * without them. `GET /api/whoami` closes that gap: the relayer resolves the
+ * account from the delegate key during authentication anyway, so it can hand
+ * back the identity the key already proves.
+ */
+import { randomUUID, createHash } from "node:crypto";
+
+import type { MemWalCredentials } from "./auth.js";
+import { loadCreds, saveCreds, loadPendingLogin, clearPendingLogin } from "./auth.js";
+import { signMessage } from "./crypto.js";
+import { log } from "./logger.js";
+
+export type RecoveryOutcome =
+ /** Nothing was pending. The overwhelmingly common case. */
+ | "no-pending"
+ /** Key was registered; `credentials.json` has been rebuilt from it. */
+ | "recovered"
+ /** A newer sign-in already happened — the pending key is stale. */
+ | "superseded"
+ /** Relayer would not authenticate the key. Deliberately non-destructive. */
+ | "rejected"
+ /** Relayer unreachable or erroring. Record kept for a later attempt. */
+ | "unavailable";
+
+export interface RecoveryResult {
+ outcome: RecoveryOutcome;
+ /** Set when a key may be registered on-chain but is not usable locally. */
+ strandedPublicKey?: string;
+ credentials?: MemWalCredentials;
+}
+
+/** Same wall-clock budget as a normal cold-start probe: recovery must never
+ * be the reason a client hangs at startup. */
+const WHOAMI_TIMEOUT_MS = 10_000;
+
+/** `x-auth-error` value the relayer sets when it could not reach Sui to check
+ * the key at all. Named in `services/server/src/auth.rs`. */
+const AUTH_UPSTREAM_UNAVAILABLE = "AUTH_UPSTREAM_UNAVAILABLE";
+
+interface WhoamiResponse {
+ account_id: string;
+ owner: string;
+ package_id: string;
+}
+
+function isWhoami(o: unknown): o is WhoamiResponse {
+ if (!o || typeof o !== "object") return false;
+ const w = o as Record;
+ return (
+ typeof w.account_id === "string" &&
+ /^0x[0-9a-fA-F]{64}$/.test(w.account_id) &&
+ typeof w.owner === "string" &&
+ typeof w.package_id === "string"
+ );
+}
+
+/**
+ * Build the exact string the relayer will rebuild and verify against.
+ *
+ * `services/server/src/auth.rs` calls itself the single source of truth for
+ * this format, and it is reproduced here rather than imported because the two
+ * live in different languages. That duplication is the risk: get it subtly
+ * wrong — a trimmed trailing separator, a missing empty field — and every
+ * recovery attempt fails with an opaque 401 that no type checker would have
+ * caught. Exported so a test can pin it against the identical literal asserted
+ * in `routes::accounts::tests::whoami_recovery_request_canonical_message_is_stable`.
+ *
+ * `accountId` is empty for recovery: not knowing it is the reason we are here,
+ * and the server defaults its hint to `""` when the header is absent.
+ */
+export function canonicalRequestMessage(parts: {
+ timestamp: string;
+ method: string;
+ path: string;
+ bodyHash: string;
+ nonce: string;
+ accountId?: string;
+}): string {
+ const { timestamp, method, path, bodyHash, nonce, accountId = "" } = parts;
+ return `${timestamp}.${method}.${path}.${bodyHash}.${nonce}.${accountId}`;
+}
+
+/** sha256 of an empty body. A GET sends none; the server hashes it anyway. */
+export const EMPTY_BODY_SHA256 = createHash("sha256").update("").digest("hex");
+
+/**
+ * Ask the relayer who this delegate key belongs to.
+ *
+ * The account id is signed as an empty string and its header omitted, because
+ * not knowing it is the entire reason we are here. The server defaults the
+ * hint to `""` when the header is absent, so both sides build the same
+ * canonical message.
+ *
+ * Returns null only when the request never produced a response at all.
+ */
+async function whoami(
+ relayerUrl: string,
+ privateKeyHex: string,
+ publicKeyHex: string,
+): Promise<{ status: number; body: unknown; authError: string | null } | null> {
+ const path = "/api/whoami";
+ // SECONDS. `services/server/src/auth.rs` freshness-checks `x-timestamp`
+ // against `chrono::Utc::now().timestamp()` within a drift window of a few
+ // minutes, so a millisecond value (~10^12) is always outside it and every
+ // request 401s with ERR_TIMESTAMP_OUT_OF_BOUNDS.
+ const timestamp = Math.floor(Date.now() / 1000).toString();
+ const nonce = randomUUID();
+ const message = canonicalRequestMessage({
+ timestamp,
+ method: "GET",
+ path,
+ bodyHash: EMPTY_BODY_SHA256,
+ nonce,
+ });
+ const signature = await signMessage(privateKeyHex, message);
+
+ const controller = new AbortController();
+ const timer = setTimeout(() => controller.abort(), WHOAMI_TIMEOUT_MS);
+ timer.unref?.();
+ try {
+ const resp = await fetch(`${relayerUrl.replace(/\/+$/, "")}${path}`, {
+ method: "GET",
+ headers: {
+ "x-public-key": publicKeyHex,
+ "x-signature": signature,
+ "x-timestamp": timestamp,
+ "x-nonce": nonce,
+ },
+ signal: controller.signal,
+ });
+ const text = await resp.text();
+ let body: unknown = null;
+ try {
+ body = JSON.parse(text);
+ } catch {
+ /* non-JSON error page — status is what matters */
+ }
+ return {
+ status: resp.status,
+ body,
+ authError: resp.headers.get("x-auth-error"),
+ };
+ } catch {
+ return null;
+ } finally {
+ clearTimeout(timer);
+ }
+}
+
+/**
+ * Attempt to turn a stranded pending login into usable credentials.
+ *
+ * Never throws, and never deletes a record that might still be recoverable —
+ * a stranded key is the user's paid-for property, and the cost of keeping it
+ * around until its TTL is a file that grants nothing.
+ */
+export async function recoverPendingLogin(): Promise {
+ const pending = loadPendingLogin();
+ if (!pending) return { outcome: "no-pending" };
+
+ // Ordering guard. If a later sign-in already succeeded, its credentials
+ // are the user's current intent and must not be rolled back to an older
+ // stranded key. Note this compares against the *pending* record's start
+ // time, so a login begun after the last successful one still wins.
+ const existing = loadCreds();
+ if (existing && Date.parse(existing.createdAt) >= Date.parse(pending.createdAt)) {
+ log.warn("login.pending.superseded", {
+ publicKey: pending.delegatePublicKeyHex,
+ });
+ clearPendingLogin();
+ return { outcome: "superseded", strandedPublicKey: pending.delegatePublicKeyHex };
+ }
+
+ const res = await whoami(
+ pending.relayerUrl,
+ pending.delegatePrivateKey,
+ pending.delegatePublicKeyHex,
+ );
+
+ if (res === null) {
+ log.warn("login.pending.relayer_unreachable", {
+ publicKey: pending.delegatePublicKeyHex,
+ });
+ return { outcome: "unavailable", strandedPublicKey: pending.delegatePublicKeyHex };
+ }
+
+ if (res.status !== 200 || !isWhoami(res.body)) {
+ // `rejected` is reserved for the relayer actually denying this
+ // identity, because that is the only outcome whose advice — sign in
+ // again, after removing the key from the dashboard if it was already
+ // registered — is worth giving. During a transient upstream
+ // failure that advice is worse than silence: the key is still good, and
+ // `unavailable` correctly says the next start retries it with no action
+ // from the user.
+ //
+ // So only 401/403 is a denial. A 503 carrying
+ // `x-auth-error: AUTH_UPSTREAM_UNAVAILABLE` is Sui RPC being down, and
+ // 429 / 5xx / a 404 from a relayer too old to serve this route are all
+ // "ask again later" — as is a 200 whose body is not a whoami, which
+ // means we are not talking to the endpoint we think we are.
+ const denied =
+ (res.status === 401 || res.status === 403) &&
+ res.authError !== AUTH_UPSTREAM_UNAVAILABLE;
+ const outcome: RecoveryOutcome = denied ? "rejected" : "unavailable";
+ // Non-destructive either way. A 401 is ambiguous even when it IS a
+ // denial: on testnet the registry scan is disabled outright and a
+ // genuinely registered key is refused for want of an x-account-id hint
+ // (services/server/src/auth.rs — "x-account-id is required for
+ // delegate-key authentication on testnet"). Clearing here would destroy
+ // a recoverable key in exactly that environment, so the record is
+ // always left for its TTL to retire.
+ log.warn(`login.pending.${outcome}`, {
+ publicKey: pending.delegatePublicKeyHex,
+ status: res.status,
+ authError: res.authError,
+ });
+ return { outcome, strandedPublicKey: pending.delegatePublicKeyHex };
+ }
+
+ const creds: MemWalCredentials = {
+ delegatePrivateKey: pending.delegatePrivateKey,
+ delegatePublicKeyHex: pending.delegatePublicKeyHex,
+ delegateAddress: pending.delegateAddress,
+ walletAddress: res.body.owner,
+ accountId: res.body.account_id,
+ packageId: res.body.package_id,
+ relayerUrl: pending.relayerUrl,
+ label: pending.label,
+ createdAt: new Date().toISOString(),
+ version: 1,
+ };
+ saveCreds(creds);
+ clearPendingLogin();
+ log.info("login.pending.recovered", {
+ accountId: creds.accountId,
+ delegateAddress: creds.delegateAddress,
+ });
+ return { outcome: "recovered", credentials: creds };
+}
+
+/**
+ * The line to show the user when a stranded key could not be reclaimed.
+ *
+ * Names the key so the user can find the registration they paid for in the
+ * dashboard. Signing in again reuses this key, and the dashboard's
+ * `add_delegate_key` aborts on one that is already registered, so a key the
+ * user approved has to be removed there before a new sign-in can finish. A
+ * user told only "login failed" can do neither.
+ */
+export function formatStrandedLoginNotice(result: RecoveryResult): string | null {
+ if (!result.strandedPublicKey) return null;
+ if (result.outcome === "recovered" || result.outcome === "no-pending") return null;
+
+ const key = result.strandedPublicKey;
+ const lines = [
+ `⚠️ Your last Walrus Memory sign-in did not finish.`,
+ ``,
+ `A delegate key may have been registered on-chain without being saved locally:`,
+ ` ${key}`,
+ ``,
+ ];
+ if (result.outcome === "superseded") {
+ lines.push(
+ `You have since signed in again, so your current credentials are fine.`,
+ `Revoke the key above from the dashboard if you don't recognise it.`,
+ );
+ } else if (result.outcome === "unavailable") {
+ lines.push(
+ `The relayer could not be reached to check. This will be retried on the`,
+ `next start — no action needed yet.`,
+ );
+ } else {
+ lines.push(
+ `The relayer did not accept it. If you never approved the wallet step, run`,
+ `\`memwal_login\`: it reuses this key. If you did, remove the key above from`,
+ `the dashboard first and then run \`memwal_login\`, because the wallet step`,
+ `cannot register a key that is already there. This is expected on Testnet,`,
+ `where the relayer cannot confirm a registered key at start.`,
+ );
+ }
+ return lines.join("\n");
+}
diff --git a/packages/mcp/test/credential-file-permissions.test.mjs b/packages/mcp/test/credential-file-permissions.test.mjs
index f85f583e2..e0e0b55e6 100644
--- a/packages/mcp/test/credential-file-permissions.test.mjs
+++ b/packages/mcp/test/credential-file-permissions.test.mjs
@@ -172,6 +172,36 @@ test("the backup of a displaced account is written at 0600", async (t) => {
assert.equal(JSON.parse(readFileSync(saved.backedUpTo, "utf8")).delegatePrivateKey, OLD_KEY);
});
+// The login write-ahead record (WALM-332) holds the same plaintext key before
+// the browser ever sees its public half, so it needs the same property.
+test("savePendingLogin never writes the key through a pre-existing permissive file", POSIX_ONLY, async (t) => {
+ const { auth } = await sandbox(t);
+ const path = auth.pendingLoginPath();
+ const makePending = (delegatePrivateKey) => ({
+ delegatePrivateKey,
+ delegatePublicKeyHex: "d".repeat(64),
+ delegateAddress: "0x" + "e".repeat(64),
+ relayerUrl: "https://relayer.example",
+ label: "Test",
+ createdAt: new Date().toISOString(),
+ version: 1,
+ });
+ mkdirSync(dirname(path), { recursive: true, mode: 0o700 });
+ writeFileSync(path, JSON.stringify(makePending(OLD_KEY)), { mode: 0o644 });
+
+ const attackerFd = openSync(path, "r");
+ t.after(() => closeSync(attackerFd));
+
+ auth.savePendingLogin(makePending(NEW_KEY));
+
+ assert.ok(
+ !readThroughOpenFd(attackerFd).includes(NEW_KEY),
+ "the pending private key must never be readable through the pre-existing 0644 inode",
+ );
+ assert.equal(JSON.parse(readFileSync(path, "utf8")).delegatePrivateKey, NEW_KEY);
+ assert.equal(modeOf(path), 0o600);
+});
+
test("saveCreds leaves no temporary file behind", async (t) => {
const { auth, home } = await sandbox(t, { existingFileMode: 0o644 });
diff --git a/packages/mcp/test/login-failure-notice.test.mjs b/packages/mcp/test/login-failure-notice.test.mjs
index 05c6cc8b9..07d1abcde 100644
--- a/packages/mcp/test/login-failure-notice.test.mjs
+++ b/packages/mcp/test/login-failure-notice.test.mjs
@@ -208,9 +208,36 @@ test("a sign-in that never completes is reported on the next tool call", async (
assert.match(text, /never completed/);
assert.match(text, /left running through the/);
assert.doesNotMatch(text, /usually works/);
- assert.match(text, /already be registered on your account/);
+ assertKeepsTheStrandedKey(text);
// Still tells them how to sign in, rather than replacing the instruction.
assert.match(text, /memwal_login/);
+ // ...but not by promising the opposite of the notice above it: an approved
+ // key is reclaimed by a restart, so the blob cannot also sell "no restart".
+ assert.doesNotMatch(text, /no client restart/i);
+});
+
+/**
+ * A failed attempt leaves its key in the write-ahead record (WALM-332), so the
+ * notice must send an approved key to a restart, which reclaims it. Signing in
+ * again reuses that key and the dashboard cannot register it twice, and
+ * removing it from the dashboard throws away the registration the user paid
+ * for. Both the wrapper and the timeout reason inside it are checked.
+ */
+function assertKeepsTheStrandedKey(text) {
+ assert.match(text, /Restart the MCP client/);
+ assert.match(text, /reclaim/);
+ assert.doesNotMatch(text, /revoke/i);
+ for (const sentence of text.split(/(?<=\.)\s+/)) {
+ if (/dashboard/.test(sentence)) {
+ assert.match(sentence, /abandon/, `dashboard advice must be limited to abandoning: "${sentence}"`);
+ }
+ }
+}
+
+test("the failure notice keeps a stranded key reclaimable", async () => {
+ const { loginFailureNotice } = await import("../dist/messages.js");
+ assert.equal(loginFailureNotice(null), "");
+ assertKeepsTheStrandedKey(loginFailureNotice("Login timed out after 1ms."));
});
test("a signed-in memwal_login timeout warns through the bridge", async (t) => {
diff --git a/packages/mcp/test/login-recovery-signing.test.mjs b/packages/mcp/test/login-recovery-signing.test.mjs
new file mode 100644
index 000000000..8234c7974
--- /dev/null
+++ b/packages/mcp/test/login-recovery-signing.test.mjs
@@ -0,0 +1,95 @@
+/**
+ * WALM-332 — the client's signed request must match the relayer byte for byte.
+ *
+ * Recovery authenticates with a signature over a canonical message defined in
+ * `services/server/src/auth.rs`. The two implementations are in different
+ * languages and cannot share code, so the format is duplicated — and a subtle
+ * mismatch (a trimmed trailing separator, a dropped empty field) would compile,
+ * pass every other test, and fail in production as an opaque 401.
+ *
+ * The literal below is asserted verbatim in
+ * `routes::accounts::tests::whoami_recovery_request_canonical_message_is_stable`.
+ * Change one and this fails; change the format in auth.rs and both fail.
+ */
+import { test } from "node:test";
+import assert from "node:assert/strict";
+import { createHash } from "node:crypto";
+import { verifyAsync } from "@noble/ed25519";
+
+const { canonicalRequestMessage, EMPTY_BODY_SHA256 } = await import("../dist/recovery.js");
+const { signMessage, hexToBytes } = await import("../dist/crypto.js");
+const { generateKeypair } = await import("../dist/crypto.js");
+
+/** Must equal the Rust fixture exactly. */
+const PINNED =
+ "1700000000.GET./api/whoami." +
+ "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855." +
+ "550e8400-e29b-41d4-a716-446655440000.";
+
+test("the client builds the canonical message the relayer expects", () => {
+ const message = canonicalRequestMessage({
+ timestamp: "1700000000",
+ method: "GET",
+ path: "/api/whoami",
+ bodyHash: EMPTY_BODY_SHA256,
+ nonce: "550e8400-e29b-41d4-a716-446655440000",
+ });
+
+ assert.equal(message, PINNED, "must match services/server/src/auth.rs byte for byte");
+ assert.ok(message.endsWith("."), "the empty account id keeps a trailing separator");
+ assert.equal(message.split(".").length - 1, 5, "six fields, five separators");
+});
+
+test("the empty-body hash is the real sha256 of nothing", () => {
+ assert.equal(EMPTY_BODY_SHA256, createHash("sha256").update("").digest("hex"));
+ assert.equal(
+ EMPTY_BODY_SHA256,
+ "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
+ "the well-known constant the server will compute for a bodyless GET",
+ );
+});
+
+test("an omitted account id is empty, never the string 'undefined'", () => {
+ // A plain template interpolation of an absent value would produce
+ // "...undefined" here, which signs cleanly and is rejected by the server
+ // with no clue why.
+ const message = canonicalRequestMessage({
+ timestamp: "1",
+ method: "GET",
+ path: "/p",
+ bodyHash: "h",
+ nonce: "n",
+ });
+ assert.equal(message, "1.GET./p.h.n.");
+ assert.doesNotMatch(message, /undefined|null/);
+});
+
+test("the signature the client produces verifies against its public key", async () => {
+ const kp = await generateKeypair();
+ const message = canonicalRequestMessage({
+ timestamp: "1700000000",
+ method: "GET",
+ path: "/api/whoami",
+ bodyHash: EMPTY_BODY_SHA256,
+ nonce: "550e8400-e29b-41d4-a716-446655440000",
+ });
+
+ const sigHex = await signMessage(kp.privateKeyHex, message);
+ assert.match(sigHex, /^[0-9a-f]{128}$/, "Ed25519 signatures are 64 bytes");
+
+ const ok = await verifyAsync(
+ hexToBytes(sigHex),
+ new TextEncoder().encode(message),
+ hexToBytes(kp.publicKeyHex),
+ );
+ assert.equal(ok, true, "the relayer must be able to verify what we signed");
+
+ // And it must not verify a tampered message — otherwise the assertion above
+ // proves nothing.
+ const tampered = await verifyAsync(
+ hexToBytes(sigHex),
+ new TextEncoder().encode(message.replace("/api/whoami", "/api/stats")),
+ hexToBytes(kp.publicKeyHex),
+ );
+ assert.equal(tampered, false, "a different path must not verify");
+});
diff --git a/packages/mcp/test/login-recovery.test.mjs b/packages/mcp/test/login-recovery.test.mjs
new file mode 100644
index 000000000..7e68d8762
--- /dev/null
+++ b/packages/mcp/test/login-recovery.test.mjs
@@ -0,0 +1,338 @@
+/**
+ * WALM-332 — reclaiming a delegate key stranded by an interrupted login.
+ *
+ * The write-ahead record keeps the key alive; these cover turning it back
+ * into usable credentials, and the cases where we must NOT.
+ */
+import { test } from "node:test";
+import assert from "node:assert/strict";
+import http from "node:http";
+import { existsSync, mkdtempSync, mkdirSync, writeFileSync, readFileSync, rmSync } from "node:fs";
+import { tmpdir } from "node:os";
+import { join } from "node:path";
+
+const ACCOUNT = `0x${"a".repeat(64)}`;
+const OWNER = `0x${"b".repeat(64)}`;
+const PACKAGE = `0x${"c".repeat(64)}`;
+
+function freshHome() {
+ const home = mkdtempSync(join(tmpdir(), "memwal-recovery-"));
+ // HOME alone is not a sandbox. os.homedir() reads USERPROFILE on Windows
+ // and ignores HOME, and credsPath() checks for a project-local .memwal
+ // above the working directory before it ever consults the home directory —
+ // which here is the real checkout. MEMWAL_CREDS_DIR overrides both, and
+ // pointing it at the sandbox's .memwal keeps the paths below unchanged.
+ process.env.HOME = home;
+ process.env.USERPROFILE = home;
+ process.env.MEMWAL_CREDS_DIR = join(home, ".memwal");
+ mkdirSync(join(home, ".memwal"), { recursive: true });
+ return home;
+}
+
+const pendingPath = (h) => join(h, ".memwal", "login-pending.json");
+const credsPath = (h) => join(h, ".memwal", "credentials.json");
+
+/** A relayer that answers /api/whoami however the test wants. */
+function startWhoami(handler) {
+ const server = http.createServer((req, res) => {
+ const url = new URL(req.url, "http://127.0.0.1");
+ if (url.pathname !== "/api/whoami") {
+ res.writeHead(404).end();
+ return;
+ }
+ handler(req, res);
+ });
+ return new Promise((r) =>
+ server.listen(0, "127.0.0.1", () =>
+ r({ server, url: `http://127.0.0.1:${server.address().port}` }),
+ ),
+ );
+}
+
+const okWhoami = (req, res) => {
+ // Assert the client proved possession rather than just asking nicely.
+ for (const h of ["x-public-key", "x-signature", "x-timestamp", "x-nonce"]) {
+ if (!req.headers[h]) {
+ res.writeHead(400).end(JSON.stringify({ missing: h }));
+ return;
+ }
+ }
+ res.writeHead(200, { "content-type": "application/json" });
+ res.end(JSON.stringify({ account_id: ACCOUNT, owner: OWNER, package_id: PACKAGE }));
+};
+
+function writePending(home, relayerUrl, overrides = {}) {
+ const pending = {
+ delegatePrivateKey: "11".repeat(32),
+ delegatePublicKeyHex: "22".repeat(32),
+ delegateAddress: `0x${"3".repeat(64)}`,
+ relayerUrl,
+ label: "Recovery test",
+ createdAt: new Date().toISOString(),
+ version: 1,
+ ...overrides,
+ };
+ writeFileSync(pendingPath(home), JSON.stringify(pending), { mode: 0o600 });
+ return pending;
+}
+
+const importRecovery = () => import(`../dist/recovery.js?t=${Date.now()}${Math.random()}`);
+
+test("a stranded key is reclaimed into usable credentials", async (t) => {
+ const home = freshHome();
+ const { server, url } = await startWhoami(okWhoami);
+ t.after(() => {
+ server.close();
+ rmSync(home, { recursive: true, force: true });
+ });
+
+ const pending = writePending(home, url);
+ const { recoverPendingLogin } = await importRecovery();
+ const result = await recoverPendingLogin();
+
+ assert.equal(result.outcome, "recovered");
+
+ const creds = JSON.parse(readFileSync(credsPath(home), "utf8"));
+ assert.equal(creds.accountId, ACCOUNT, "accountId comes from the relayer");
+ assert.equal(creds.walletAddress, OWNER);
+ assert.equal(creds.packageId, PACKAGE);
+ assert.equal(
+ creds.delegatePrivateKey,
+ pending.delegatePrivateKey,
+ "the reclaimed key must be the one that was registered",
+ );
+ assert.equal(
+ existsSync(pendingPath(home)),
+ false,
+ "pending record cleared once the key is safe",
+ );
+});
+
+test("recovery never rolls back a newer sign-in", async (t) => {
+ const home = freshHome();
+ const { server, url } = await startWhoami(okWhoami);
+ t.after(() => {
+ server.close();
+ rmSync(home, { recursive: true, force: true });
+ });
+
+ // Pending login started BEFORE the credentials currently on disk: the user
+ // gave up on it and signed in again. Adopting it would silently downgrade
+ // them to a key they already abandoned.
+ writePending(home, url, { createdAt: new Date(Date.now() - 60_000).toISOString() });
+ const current = {
+ delegatePrivateKey: "99".repeat(32),
+ delegatePublicKeyHex: "88".repeat(32),
+ delegateAddress: `0x${"7".repeat(64)}`,
+ walletAddress: OWNER,
+ accountId: `0x${"d".repeat(64)}`,
+ packageId: PACKAGE,
+ relayerUrl: url,
+ createdAt: new Date().toISOString(),
+ version: 1,
+ };
+ writeFileSync(credsPath(home), JSON.stringify(current), { mode: 0o600 });
+
+ const { recoverPendingLogin } = await importRecovery();
+ const result = await recoverPendingLogin();
+
+ assert.equal(result.outcome, "superseded");
+ const after = JSON.parse(readFileSync(credsPath(home), "utf8"));
+ assert.deepEqual(after, current, "existing credentials must be untouched");
+ assert.ok(result.strandedPublicKey, "the abandoned key is still reported so it can be revoked");
+});
+
+test("a rejected key is reported but never deleted", async (t) => {
+ const home = freshHome();
+ // 401 is ambiguous — on testnet even a valid registered key is rejected
+ // for want of an account hint. Deleting here would destroy a paid key.
+ const { server, url } = await startWhoami((_req, res) => {
+ res.writeHead(401).end("{}");
+ });
+ t.after(() => {
+ server.close();
+ rmSync(home, { recursive: true, force: true });
+ });
+
+ writePending(home, url);
+ const { recoverPendingLogin, formatStrandedLoginNotice } = await importRecovery();
+ const result = await recoverPendingLogin();
+
+ assert.equal(result.outcome, "rejected");
+ assert.equal(
+ existsSync(pendingPath(home)),
+ true,
+ "the record must survive an ambiguous rejection",
+ );
+ assert.equal(existsSync(credsPath(home)), false, "no credentials written");
+
+ const notice = formatStrandedLoginNotice(result);
+ assert.match(notice, /22{10}/, "the notice names the key so it can be revoked");
+ // Signing in again reuses this key, and the dashboard's add_delegate_key
+ // aborts on one already registered, so that alone cannot recover an
+ // approved key.
+ assert.match(
+ notice,
+ /cannot\s+register a key that is already there/,
+ "must not promise that signing in again recovers a key the user approved",
+ );
+});
+
+test("an unreachable relayer keeps the record for a later attempt", async (t) => {
+ const home = freshHome();
+ t.after(() => rmSync(home, { recursive: true, force: true }));
+
+ // Nothing is listening on this port.
+ writePending(home, "http://127.0.0.1:1");
+ const { recoverPendingLogin } = await importRecovery();
+ const result = await recoverPendingLogin();
+
+ assert.equal(result.outcome, "unavailable");
+ assert.equal(existsSync(pendingPath(home)), true);
+});
+
+test("an expired pending record is discarded rather than recovered", async (t) => {
+ const home = freshHome();
+ const { server, url } = await startWhoami(okWhoami);
+ t.after(() => {
+ server.close();
+ rmSync(home, { recursive: true, force: true });
+ });
+
+ writePending(home, url, {
+ createdAt: new Date(Date.now() - 25 * 60 * 60_000).toISOString(),
+ });
+ const { recoverPendingLogin } = await importRecovery();
+ const result = await recoverPendingLogin();
+
+ assert.equal(result.outcome, "no-pending");
+ assert.equal(existsSync(pendingPath(home)), false, "expired record is cleaned up");
+ assert.equal(existsSync(credsPath(home)), false);
+});
+
+test("no pending record is a silent no-op", async (t) => {
+ const home = freshHome();
+ t.after(() => rmSync(home, { recursive: true, force: true }));
+
+ const { recoverPendingLogin, formatStrandedLoginNotice } = await importRecovery();
+ const result = await recoverPendingLogin();
+
+ assert.equal(result.outcome, "no-pending");
+ assert.equal(formatStrandedLoginNotice(result), null);
+});
+
+/**
+ * The relayer freshness-checks `x-timestamp` against `Utc::now().timestamp()`
+ * — SECONDS. `String(Date.now())` is milliseconds, ~10^12, which is outside
+ * every drift window there will ever be, so whoami 401'd on every attempt and
+ * recovery could not have worked at all.
+ */
+test("whoami signs a Unix timestamp in seconds, not milliseconds", async (t) => {
+ const home = freshHome();
+ let seen = null;
+ const { server, url } = await startWhoami((req, res) => {
+ seen = req.headers["x-timestamp"];
+ okWhoami(req, res);
+ });
+ t.after(() => {
+ server.close();
+ rmSync(home, { recursive: true, force: true });
+ });
+
+ writePending(home, url);
+ const { recoverPendingLogin } = await importRecovery();
+ await recoverPendingLogin();
+
+ assert.match(seen ?? "", /^\d{10}$/, `expected 10-digit seconds, got ${seen}`);
+ const skew = Math.abs(Number(seen) - Math.floor(Date.now() / 1000));
+ assert.ok(skew < 300, `timestamp is ${skew}s from now — outside the relayer's window`);
+});
+
+/**
+ * `rejected` tells the user to sign in again and revoke the key. That advice is
+ * actively harmful when the relayer merely could not reach Sui: the key is
+ * fine, and re-registering costs gas for nothing.
+ */
+for (const [label, status, headers] of [
+ ["a 503 with AUTH_UPSTREAM_UNAVAILABLE", 503, { "x-auth-error": "AUTH_UPSTREAM_UNAVAILABLE" }],
+ ["a bare 500", 500, {}],
+ ["a 429", 429, {}],
+ ["a 404 from a relayer without the route", 404, {}],
+]) {
+ test(`${label} is retryable, not a rejection`, async (t) => {
+ const home = freshHome();
+ const { server, url } = await startWhoami((_req, res) => {
+ res.writeHead(status, headers).end("{}");
+ });
+ t.after(() => {
+ server.close();
+ rmSync(home, { recursive: true, force: true });
+ });
+
+ writePending(home, url);
+ const { recoverPendingLogin, formatStrandedLoginNotice } = await importRecovery();
+ const result = await recoverPendingLogin();
+
+ assert.equal(result.outcome, "unavailable", `status ${status} should not read as a denial`);
+ assert.equal(existsSync(pendingPath(home)), true, "the record must survive");
+
+ const notice = formatStrandedLoginNotice(result);
+ assert.doesNotMatch(
+ notice,
+ /revoke/i,
+ "must not send the user to revoke a key that may be perfectly good",
+ );
+ assert.match(notice, /retried/i, "should say it will be retried");
+ });
+}
+
+test("a 401 carrying AUTH_UPSTREAM_UNAVAILABLE is still retryable", async (t) => {
+ // The status alone is not enough: the header is what distinguishes
+ // "we could not check" from "we checked and said no".
+ const home = freshHome();
+ const { server, url } = await startWhoami((_req, res) => {
+ res.writeHead(401, { "x-auth-error": "AUTH_UPSTREAM_UNAVAILABLE" }).end("{}");
+ });
+ t.after(() => {
+ server.close();
+ rmSync(home, { recursive: true, force: true });
+ });
+
+ writePending(home, url);
+ const { recoverPendingLogin } = await importRecovery();
+ assert.equal((await recoverPendingLogin()).outcome, "unavailable");
+});
+
+test("a 200 that is not a whoami body is retryable, not a rejection", async (t) => {
+ // Means we are not talking to the endpoint we think we are — nothing has
+ // denied this key.
+ const home = freshHome();
+ const { server, url } = await startWhoami((_req, res) => {
+ res.writeHead(200, { "content-type": "application/json" }).end('{"hello":"world"}');
+ });
+ t.after(() => {
+ server.close();
+ rmSync(home, { recursive: true, force: true });
+ });
+
+ writePending(home, url);
+ const { recoverPendingLogin } = await importRecovery();
+ assert.equal((await recoverPendingLogin()).outcome, "unavailable");
+});
+
+test("a plain 401 is still a rejection", async (t) => {
+ // Regression guard on the split above: widening `unavailable` must not
+ // swallow the one case where the relayer really did deny the identity.
+ const home = freshHome();
+ const { server, url } = await startWhoami((_req, res) => {
+ res.writeHead(401).end("{}");
+ });
+ t.after(() => {
+ server.close();
+ rmSync(home, { recursive: true, force: true });
+ });
+
+ writePending(home, url);
+ const { recoverPendingLogin } = await importRecovery();
+ assert.equal((await recoverPendingLogin()).outcome, "rejected");
+});
diff --git a/packages/mcp/test/login-write-ahead.test.mjs b/packages/mcp/test/login-write-ahead.test.mjs
new file mode 100644
index 000000000..fa4667d51
--- /dev/null
+++ b/packages/mcp/test/login-write-ahead.test.mjs
@@ -0,0 +1,283 @@
+/**
+ * WALM-332 — the login flow must not lose the delegate private key.
+ *
+ * The browser registers the delegate key on-chain (a paid, irreversible
+ * action) and only then POSTs the callback that causes us to save it. If this
+ * process dies in that window, the in-memory keypair is destroyed and the
+ * user is left with an on-chain registration nobody holds the key to.
+ *
+ * The fix is write-ahead: persist the pending keypair BEFORE the browser is
+ * able to act on it, and clear it once credentials are safely saved.
+ */
+import { test } from "node:test";
+import assert from "node:assert/strict";
+import { existsSync, mkdtempSync, mkdirSync, readFileSync, writeFileSync, statSync, rmSync, chmodSync } from "node:fs";
+import { tmpdir } from "node:os";
+import { join } from "node:path";
+
+const WEB = "https://memory.example";
+const RELAYER = "https://relayer.example";
+
+function freshHome() {
+ const home = mkdtempSync(join(tmpdir(), "memwal-writeahead-"));
+ // HOME alone is not a sandbox. os.homedir() reads USERPROFILE on Windows
+ // and ignores HOME, and credsPath() checks for a project-local .memwal
+ // above the working directory before it ever consults the home directory —
+ // which here is the real checkout. MEMWAL_CREDS_DIR overrides both, and
+ // pointing it at the sandbox's .memwal keeps the paths below unchanged.
+ process.env.HOME = home;
+ process.env.USERPROFILE = home;
+ process.env.MEMWAL_CREDS_DIR = join(home, ".memwal");
+ return home;
+}
+
+const pendingPath = (home) => join(home, ".memwal", "login-pending.json");
+const credsPath = (home) => join(home, ".memwal", "credentials.json");
+
+/** Start a login flow and resolve once the connect URL has been published. */
+async function startLogin(overrides = {}) {
+ const { loginFlow } = await import(`../dist/login.js?t=${Date.now()}${Math.random()}`);
+ let publishUrl;
+ const urlReady = new Promise((resolve) => {
+ publishUrl = resolve;
+ });
+ const flow = loginFlow({
+ webUrl: WEB,
+ relayerUrl: RELAYER,
+ label: "Write-ahead test",
+ timeoutMs: 4_000,
+ openBrowser: false,
+ onUrl: publishUrl,
+ ...overrides,
+ });
+ // The flow rejects on timeout; nobody is going to complete it in these
+ // tests, so absorb it rather than tripping an unhandled rejection.
+ flow.catch(() => {});
+ // A flow that fails BEFORE publishing — the write-ahead record cannot be
+ // written, say — must surface here as a rejection. Awaiting `urlReady`
+ // alone would hang forever on a URL that is never coming.
+ const failedEarly = flow.then(() => {
+ throw new Error("login resolved without ever publishing a URL");
+ });
+ failedEarly.catch(() => {});
+ return { flow, url: new URL(await Promise.race([urlReady, failedEarly])) };
+}
+
+test("the delegate keypair is on disk before the browser is given the connect URL", async (t) => {
+ const home = freshHome();
+ t.after(() => rmSync(home, { recursive: true, force: true }));
+
+ const { url } = await startLogin();
+
+ // The URL is what the user clicks; by the time it exists, the browser can
+ // register this public key on-chain. The private half must already be safe.
+ assert.ok(
+ existsSync(pendingPath(home)),
+ "login-pending.json must exist by the time the connect URL is published",
+ );
+
+ const pending = JSON.parse(readFileSync(pendingPath(home), "utf8"));
+ const publicKeyInUrl = url.searchParams.get("publicKey");
+
+ assert.equal(
+ pending.delegatePublicKeyHex?.toLowerCase(),
+ publicKeyInUrl?.toLowerCase(),
+ "the persisted record must be for the exact key the browser was sent",
+ );
+ assert.match(
+ pending.delegatePrivateKey ?? "",
+ /^(0x)?[0-9a-f]{64}$/i,
+ "the private key must be recoverable from the record",
+ );
+ assert.equal(pending.relayerUrl, RELAYER);
+ assert.ok(pending.createdAt, "record needs a timestamp so it can expire");
+
+ // Same handling as credentials.json — owner-only. Windows does not enforce
+ // POSIX mode bits, and `savePendingLogin` treats `chmodSync` as best-effort
+ // there, so asserting them would test the platform rather than the code.
+ if (process.platform !== "win32") {
+ assert.equal(
+ statSync(pendingPath(home)).mode & 0o777,
+ 0o600,
+ "pending login must be owner-only, like credentials.json",
+ );
+ }
+
+ // Nothing has completed, so no credentials yet.
+ assert.equal(existsSync(credsPath(home)), false);
+});
+
+test("a completed login clears the pending record", async (t) => {
+ const home = freshHome();
+ t.after(() => rmSync(home, { recursive: true, force: true }));
+
+ const { flow, url } = await startLogin({ timeoutMs: 15_000 });
+ const port = url.searchParams.get("port");
+ const state = url.searchParams.get("connectState");
+ const publicKey = url.searchParams.get("publicKey");
+
+ assert.ok(existsSync(pendingPath(home)), "precondition: pending record written");
+
+ const post = (path, body) =>
+ fetch(`http://127.0.0.1:${port}${path}`, {
+ method: "POST",
+ headers: { "content-type": "application/json", origin: WEB },
+ body: JSON.stringify(body),
+ });
+
+ await post("/preflight", { state, publicKey, relayer: RELAYER });
+ await post("/callback", {
+ state,
+ accountId: `0x${"1".repeat(64)}`,
+ walletAddress: `0x${"2".repeat(64)}`,
+ packageId: `0x${"3".repeat(64)}`,
+ });
+
+ await flow;
+
+ assert.equal(existsSync(credsPath(home)), true, "credentials should be saved");
+ assert.equal(
+ existsSync(pendingPath(home)),
+ false,
+ "pending record must be cleared once the key is safely in credentials.json",
+ );
+});
+
+/**
+ * Recovery only runs at process start, and is skipped for `--login` /
+ * `forceLogin`. So a login that times out, followed by `memwal_login` in the
+ * same process, used to mint a fresh keypair and overwrite the record — and if
+ * the browser had already paid for `add_delegate_key` on the first key, the
+ * private half went with it.
+ */
+test("a second login for the same relayer reuses the stranded keypair", async (t) => {
+ const home = freshHome();
+ t.after(() => rmSync(home, { recursive: true, force: true }));
+
+ const first = await startLogin();
+ const stranded = JSON.parse(readFileSync(pendingPath(home), "utf8"));
+ first.flow.catch(() => {});
+
+ const second = await startLogin();
+ const after = JSON.parse(readFileSync(pendingPath(home), "utf8"));
+ second.flow.catch(() => {});
+
+ assert.equal(
+ after.delegatePrivateKey,
+ stranded.delegatePrivateKey,
+ "the paid-for key must not be replaced by a second attempt",
+ );
+ assert.equal(
+ second.url.searchParams.get("publicKey")?.toLowerCase(),
+ stranded.delegatePublicKeyHex.toLowerCase(),
+ "the browser should be sent the key that may already be registered",
+ );
+ assert.equal(
+ after.createdAt,
+ stranded.createdAt,
+ "reusing must not extend the TTL past the attempt that may have registered it",
+ );
+});
+
+test("a login against a different relayer does not reuse the record", async (t) => {
+ // A key registered against one relayer's account proves nothing to
+ // another, and recovery must never repoint a record at a new relayer.
+ const home = freshHome();
+ t.after(() => rmSync(home, { recursive: true, force: true }));
+
+ const first = await startLogin();
+ const stranded = JSON.parse(readFileSync(pendingPath(home), "utf8"));
+ first.flow.catch(() => {});
+
+ const second = await startLogin({ relayerUrl: "https://other-relayer.example" });
+ const after = JSON.parse(readFileSync(pendingPath(home), "utf8"));
+ second.flow.catch(() => {});
+
+ assert.notEqual(after.delegatePrivateKey, stranded.delegatePrivateKey);
+ assert.equal(after.relayerUrl, "https://other-relayer.example");
+});
+
+test("a login refuses to start when the write-ahead record cannot be persisted", async (t) => {
+ // The whole invariant is that the key is durable before its public half can
+ // reach a browser that will pay to register it. Continuing anyway would
+ // publish the URL while only pretending to hold that.
+ const home = freshHome();
+ const dir = join(home, ".memwal");
+ mkdirSync(dir, { recursive: true });
+ t.after(() => {
+ try {
+ chmodSync(dir, 0o700);
+ } catch {
+ /* nothing to restore */
+ }
+ rmSync(home, { recursive: true, force: true });
+ });
+
+ // Read-only directory. Root ignores mode bits, and Windows does not
+ // enforce them at all, so only assert where the setup actually bites.
+ chmodSync(dir, 0o500);
+ let writable = true;
+ try {
+ writeFileSync(join(dir, ".probe"), "x");
+ } catch {
+ writable = false;
+ }
+ t.diagnostic(`credentials dir writable after chmod 0500: ${writable}`);
+ if (writable) {
+ t.skip("the sandbox directory is still writable — cannot provoke the failure here");
+ return;
+ }
+
+ await assert.rejects(
+ () => startLogin(),
+ /write-ahead/i,
+ "the login must fail loudly rather than publish a URL it cannot back",
+ );
+ assert.equal(existsSync(pendingPath(home)), false, "nothing should have been written");
+});
+
+/**
+ * `clearPendingLogin()` used to run only after a successful callback. CLI
+ * `--logout` and the `memwal_logout` tool both cleared `credentials.json`
+ * alone, so an interrupted re-login left the pending key behind and the next
+ * start's `recoverPendingLogin` signed the user straight back in — a logout
+ * that undid itself.
+ */
+test("logging out discards the pending record, not just the credentials", async (t) => {
+ const home = freshHome();
+ t.after(() => rmSync(home, { recursive: true, force: true }));
+
+ const { flow } = await startLogin();
+ flow.catch(() => {});
+ assert.ok(existsSync(pendingPath(home)), "precondition: a pending record exists");
+
+ const { main } = await import(`../dist/index.js?t=${Date.now()}${Math.random()}`);
+ await main(["--logout"]);
+
+ assert.equal(
+ existsSync(pendingPath(home)),
+ false,
+ "an explicit logout must not leave a key that signs the user back in",
+ );
+});
+
+test("clearing credentials on its own keeps the pending record", async (t) => {
+ // Discarding a key that may still be reclaimable is a decision only an
+ // explicit sign-out gets to make, which is why the pending clear lives in
+ // the logout paths rather than inside `clearCreds` (which is exported, and
+ // which a relayer 401 deliberately does not call).
+ const home = freshHome();
+ t.after(() => rmSync(home, { recursive: true, force: true }));
+
+ const { flow } = await startLogin();
+ flow.catch(() => {});
+ assert.ok(existsSync(pendingPath(home)), "precondition: a pending record exists");
+
+ const { clearCreds } = await import(`../dist/auth.js?t=${Date.now()}${Math.random()}`);
+ clearCreds();
+
+ assert.ok(
+ existsSync(pendingPath(home)),
+ "clearCreds must not discard a key that may still be reclaimable",
+ );
+});
diff --git a/packages/mcp/test/logout-invalidation.test.mjs b/packages/mcp/test/logout-invalidation.test.mjs
index e8b7020f5..3adefb2c0 100644
--- a/packages/mcp/test/logout-invalidation.test.mjs
+++ b/packages/mcp/test/logout-invalidation.test.mjs
@@ -675,3 +675,57 @@ test("a non-tool request after logout is answered locally instead of hanging", a
"nothing after logout should have reached the relayer",
);
});
+
+/**
+ * The tool half of the same fix the CLI gets in login-write-ahead: an explicit
+ * `memwal_logout` must discard `login-pending.json` too. Left behind, the next
+ * start's `recoverPendingLogin` rebuilds credentials from it and signs the user
+ * back in — a logout that undoes itself.
+ */
+test("memwal_logout discards a stranded pending login as well as the credentials", async (t) => {
+ const mock = await startMockRelayer();
+ const home = mkdtempSync(join(tmpdir(), "memwal-logout-pending-"));
+ const credsPath = join(home, ".memwal", "credentials.json");
+ const pendingPath = join(home, ".memwal", "login-pending.json");
+ mkdirSync(dirname(credsPath), { recursive: true });
+ writeFileSync(credsPath, JSON.stringify(makeCreds(mock.base)), { mode: 0o600 });
+ writeFileSync(
+ pendingPath,
+ JSON.stringify({
+ delegatePrivateKey: "11".repeat(32),
+ delegatePublicKeyHex: "22".repeat(32),
+ delegateAddress: `0x${"3".repeat(64)}`,
+ relayerUrl: mock.base,
+ label: "Interrupted re-login",
+ createdAt: new Date().toISOString(),
+ version: 1,
+ }),
+ { mode: 0o600 },
+ );
+
+ t.after(() => {
+ mock.server.close();
+ rmSync(home, { recursive: true, force: true });
+ });
+
+ const { send, waitFor } = startBridge(t, mock, home);
+
+ send({ jsonrpc: "2.0", id: 1, method: "initialize", params: {} });
+ await waitFor((m) => m.id === 1 && m.result, 10_000);
+
+ send({
+ jsonrpc: "2.0",
+ id: 2,
+ method: "tools/call",
+ params: { name: "memwal_logout", arguments: {} },
+ });
+ const out = await waitFor((m) => m.id === 2, 10_000);
+ assert.notEqual(out.result?.isError, true, "logout should succeed");
+
+ assert.equal(existsSync(credsPath), false, "credentials should be gone");
+ assert.equal(
+ existsSync(pendingPath),
+ false,
+ "the pending record must go too, or the next start signs the user back in",
+ );
+});
diff --git a/services/server/src/main.rs b/services/server/src/main.rs
index 02bdb0e83..88c6ecced 100644
--- a/services/server/src/main.rs
+++ b/services/server/src/main.rs
@@ -2007,6 +2007,11 @@ async fn main() {
// Mode-blind; owner-scoped via AuthInfo.
.route("/api/forget", post(routes::forget))
.route("/api/stats", post(routes::stats))
+ // Identity echo — tells an authenticated caller which account its
+ // delegate key resolves to. Must stay inside this authed group; the
+ // account_id it returns is deliberately withheld from the public
+ // /api/accounts/{owner}/exists route. See routes::whoami.
+ .route("/api/whoami", get(routes::whoami))
// Router::layer runs middleware bottom-to-top (last added runs first).
// Keep auth outer so AuthInfo is in request extensions before rate limiting reads it.
.layer(middleware::from_fn_with_state(
diff --git a/services/server/src/routes/accounts.rs b/services/server/src/routes/accounts.rs
index 4c0af8b5f..4daad78ad 100644
--- a/services/server/src/routes/accounts.rs
+++ b/services/server/src/routes/accounts.rs
@@ -6,7 +6,7 @@
//! Console to hold a delegate key or run its own onchain scan.
use axum::extract::{Path, State};
-use axum::Json;
+use axum::{Extension, Json};
use std::sync::Arc;
use crate::routes::sponsor::validate_sui_address;
@@ -74,6 +74,46 @@ pub async fn account_exists(
Ok(Json(AccountExistsResponse { exists }))
}
+/// GET /api/whoami
+///
+/// Returns the account identity the caller's delegate key resolves to.
+///
+/// Authenticated, and that is what makes returning `account_id` here
+/// acceptable where `account_exists` deliberately withholds it: the auth
+/// middleware has already proven the caller holds a delegate key registered
+/// against this account, so this only ever tells a caller about itself. Do
+/// not move this into the unauthenticated router group.
+///
+/// Motivating case (WALM-332): a login interrupted between the browser's
+/// on-chain `add_delegate_key` and the localhost callback leaves the client
+/// holding a valid delegate key but none of the surrounding metadata, so it
+/// cannot write a usable `credentials.json`. Everything needed to rebuild one
+/// is already resolved during authentication — `account_id` and `owner` from
+/// the registry scan, `package_id` from config — so this endpoint hands back
+/// what the middleware already computed rather than doing new work.
+pub async fn whoami(
+ State(state): State>,
+ Extension(auth): Extension,
+) -> Result, AppError> {
+ Ok(Json(whoami_response(auth, state.config.package_id.clone())))
+}
+
+/// The field mapping, factored out of the handler so it is testable without a
+/// live `AppState`/DB (same reason as `account_exists` above — this codebase
+/// has no axum-handler test harness).
+///
+/// Worth isolating rather than inlining: `account_id` and `owner` are both
+/// 0x-prefixed 32-byte hex, so transposing them is invisible to the type
+/// checker and would produce credentials that authenticate as the wrong
+/// identity. The test below pins the mapping.
+fn whoami_response(auth: AuthInfo, package_id: String) -> WhoamiResponse {
+ WhoamiResponse {
+ account_id: auth.account_id,
+ owner: auth.owner,
+ package_id,
+ }
+}
+
#[cfg(test)]
mod tests {
use super::*;
@@ -103,4 +143,82 @@ mod tests {
// Already-lowercase input must be unaffected (idempotent).
assert_eq!(normalized.to_ascii_lowercase(), normalized);
}
+
+ // ── GET /api/whoami (WALM-332 recovery) ──────────────────────
+
+ fn auth_fixture() -> AuthInfo {
+ AuthInfo {
+ public_key: "aa".repeat(32),
+ owner: format!("0x{}", "b".repeat(64)),
+ account_id: format!("0x{}", "a".repeat(64)),
+ delegate_key: None,
+ seal_session: None,
+ }
+ }
+
+ /// `account_id` and `owner` are indistinguishable by type — both
+ /// 0x-prefixed 32-byte hex — so a transposition here would compile, pass
+ /// every other test, and hand a recovering client credentials that
+ /// authenticate as the wrong identity. Pin the mapping explicitly.
+ #[test]
+ fn whoami_maps_account_and_owner_without_transposing_them() {
+ let auth = auth_fixture();
+ let package_id = format!("0x{}", "c".repeat(64));
+
+ let resp = whoami_response(auth, package_id.clone());
+
+ assert_eq!(
+ resp.account_id,
+ format!("0x{}", "a".repeat(64)),
+ "account_id must come from AuthInfo::account_id, not owner"
+ );
+ assert_eq!(
+ resp.owner,
+ format!("0x{}", "b".repeat(64)),
+ "owner must come from AuthInfo::owner, not account_id"
+ );
+ assert_eq!(resp.package_id, package_id, "package_id comes from config");
+ assert_ne!(resp.account_id, resp.owner, "fixture must distinguish them");
+ }
+
+ /// The recovering MCP client cannot send `x-account-id` — not knowing it
+ /// is the whole reason it is calling this endpoint — so it signs the
+ /// canonical message with an empty account field and omits the header.
+ /// The server defaults the hint to `""` for exactly this case.
+ ///
+ /// This pins the literal string both sides must build. It is duplicated
+ /// verbatim in `packages/mcp/test/login-recovery-signing.test.mjs`; if the
+ /// canonical format in `auth.rs` ever changes, both fail together rather
+ /// than recovery silently breaking in production.
+ #[test]
+ fn whoami_recovery_request_canonical_message_is_stable() {
+ let timestamp = "1700000000";
+ let method = "GET";
+ let path = "/api/whoami";
+ // sha256 of an empty body — a GET carries none, but the server hashes
+ // the empty body all the same.
+ let body_hash = "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855";
+ let nonce = "550e8400-e29b-41d4-a716-446655440000";
+ let account_id_for_sig = String::new();
+
+ let message = format!(
+ "{}.{}.{}.{}.{}.{}",
+ timestamp, method, path, body_hash, nonce, account_id_for_sig
+ );
+
+ let expected = concat!(
+ "1700000000.GET./api/whoami.",
+ "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855.",
+ "550e8400-e29b-41d4-a716-446655440000."
+ );
+ assert_eq!(message, expected);
+ // Six fields → five separators. Nothing else in the message contains a
+ // dot: the nonce is hyphen-separated and the body hash is bare hex.
+ assert_eq!(message.matches('.').count(), 5);
+ assert!(
+ message.ends_with('.'),
+ "the empty account id leaves a trailing separator — the client must \
+ reproduce this exactly, not trim it"
+ );
+ }
}
diff --git a/services/server/src/routes/mod.rs b/services/server/src/routes/mod.rs
index 123894c6c..c9f17f349 100644
--- a/services/server/src/routes/mod.rs
+++ b/services/server/src/routes/mod.rs
@@ -38,7 +38,7 @@ mod sponsor;
// Re-export every handler so `main.rs` keeps using `routes::`
// without having to know which submodule each handler lives in.
-pub use accounts::account_exists;
+pub use accounts::{account_exists, whoami};
pub use admin::{ask, embed, forget, get_config, health, restore, stats, version};
pub use analyze::analyze;
pub use memory_read::{list_owner_agents, list_owner_memories, list_owner_namespaces};
diff --git a/services/server/src/types.rs b/services/server/src/types.rs
index 0da31d77d..76b2e0376 100644
--- a/services/server/src/types.rs
+++ b/services/server/src/types.rs
@@ -1886,6 +1886,24 @@ pub struct AccountExistsResponse {
pub exists: bool,
}
+/// GET /api/whoami — the identity the caller's delegate key resolves to.
+///
+/// Unlike `AccountExistsResponse` this *does* carry `account_id`, which is
+/// safe here precisely because the route is authenticated: the caller proved
+/// possession of a delegate key already registered against this account, so
+/// it is being told its own identity, not anyone else's.
+///
+/// Exists so a client that holds a delegate key but lost the surrounding
+/// metadata can rebuild `credentials.json` (WALM-332). All three fields are
+/// required for that: `account_id` and `owner` come from the registry scan,
+/// `package_id` from server config.
+#[derive(Debug, Serialize)]
+pub struct WhoamiResponse {
+ pub account_id: String,
+ pub owner: String,
+ pub package_id: String,
+}
+
/// POST /api/stats — count + stored bytes for a namespace.
/// Used by the benchmark harness for verification. Mode-blind.
#[derive(Debug, Deserialize)]
From efd82a929016da58a3925cf6c39679886cc5811c Mon Sep 17 00:00:00 2001
From: Harry Phan
Date: Wed, 16 Sep 2026 11:42:45 +0700
Subject: [PATCH 016/132] docs(mcp): tell agents that bulk returns job_ids too
The session instructions are what actually steer an agent, and they still
described only memwal_remember as returning ACCEPTED-but-not-saved. Now that
memwal_remember_bulk behaves the same way, an agent reading this would take
a pending batch for a stored one and tell the user their facts were saved.
Also names the two things that only apply to a batch: its writes are stored
one at a time rather than together, so it takes longer than a single fact,
and memwal_remember_status accepts job_ids to settle the whole batch in one
call rather than one id at a time.
---
services/server/scripts/mcp/server.ts | 14 ++++++++------
1 file changed, 8 insertions(+), 6 deletions(-)
diff --git a/services/server/scripts/mcp/server.ts b/services/server/scripts/mcp/server.ts
index 094094088..561d8bf55 100644
--- a/services/server/scripts/mcp/server.ts
+++ b/services/server/scripts/mcp/server.ts
@@ -50,12 +50,14 @@ const INSTRUCTIONS = [
"summary. Skip one-off tasks, the current file or bug, and small talk. Use",
"memwal_remember_bulk when several distinct facts arrived at once.",
"",
- "A Walrus write is queued, not instant: memwal_remember normally returns a job_id with",
- "the fact ACCEPTED but NOT YET SAVED, and storing it takes roughly another 30-60s. That",
- "is the healthy path, not an error. Tell the user the fact is being saved rather than",
- "that it is saved, and do not re-send it — that queues a duplicate. A job can still fail",
- "after acceptance, so when it matters that a fact landed, resolve the job_id with",
- "memwal_remember_status; only the blob_id it returns means the fact is stored.",
+ "A Walrus write is queued, not instant: memwal_remember and memwal_remember_bulk normally",
+ "return job_ids with the facts ACCEPTED but NOT YET SAVED, and storing each takes roughly",
+ "another 30-60s. A batch is slower still — its writes are stored one at a time, not",
+ "together. That is the healthy path, not an error. Tell the user the facts are being saved",
+ "rather than that they are saved, and do not re-send them — that queues duplicates. A job",
+ "can still fail after acceptance, so when it matters that a fact landed, resolve the ids",
+ "with memwal_remember_status (it takes job_id, or job_ids for a whole batch); only a",
+ "blob_id means that fact is stored.",
"",
"RECOVER: if memwal_recall unexpectedly returns nothing for a namespace that has been used",
"before, call memwal_restore to rebuild the index from Walrus.",
From a3531141607bf5592d7dfc363103e11a5ac2832e Mon Sep 17 00:00:00 2001
From: Harry Phan
Date: Wed, 16 Sep 2026 11:53:15 +0700
Subject: [PATCH 017/132] fix(relayer): space out upload retries instead of
burning them in a second
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit
Apalis attaches no retry/backoff layer, so a retriable upload error is
re-polled almost immediately. Production shows what that costs: one job took
attempts 2, 3, 4 and 5 against Walrus `503 Too Many Requests` inside a single
second, rotated through four wallets, and died as "exhausted retries" about a
second after its first failure. The upstream limit is time-based, so rotating
wallets cannot help — only waiting can, and nothing was waiting.
Reuse the existing `backoff_duration` schedule (2s, 4s, 8s, 16s) between
upload attempts, the way the lock-defer path already does. The wallet slot is
held across the sleep on purpose: a backing-off job is still that wallet's
turn, and releasing it would invite another job onto a wallet about to retry.
`upload_retry_backoff` returns None for an aborting error and for the final
attempt, so nothing sleeps when no retry is coming.
19 of the 24 hours of upload failures sampled were this one rate-limit, and
roughly 23% of jobs reached a second attempt.
---
services/server/src/jobs.rs | 82 ++++++++++++++++++++++++++++++++++++-
1 file changed, 81 insertions(+), 1 deletion(-)
diff --git a/services/server/src/jobs.rs b/services/server/src/jobs.rs
index 8920fd438..a0073709a 100644
--- a/services/server/src/jobs.rs
+++ b/services/server/src/jobs.rs
@@ -400,6 +400,31 @@ pub fn backoff_duration(attempt: u32) -> std::time::Duration {
std::time::Duration::from_secs(2u64.pow(attempt))
}
+/// How long a failed upload attempt should pause before Apalis re-queues it,
+/// or `None` when no pause is warranted.
+///
+/// Apalis attaches no retry/backoff layer (see `WalletJobError`'s doc
+/// comment), so a retriable error is re-polled almost immediately and the
+/// whole attempt budget burns inside one upstream rate-limit window. Observed
+/// in production: a single job took attempts 2, 3, 4 and 5 against Walrus
+/// `503 Too Many Requests` within one second, rotating through four wallets
+/// that never had a chance to land, and died as "exhausted retries" about a
+/// second after its first failure. The upstream limit is time-based, so
+/// rotating wallets cannot help — only waiting can.
+///
+/// Returns `None` for an aborting error (retrying it is pointless) and for the
+/// final attempt (nothing is coming, so the sleep would only delay the
+/// failure the caller is already reporting).
+fn upload_retry_backoff(
+ classified: &WalletJobError,
+ attempt_info: WalletJobAttemptInfo,
+) -> Option {
+ if classified.aborts_retries() || attempt_info.current >= attempt_info.max {
+ return None;
+ }
+ Some(backoff_duration(attempt_info.current as u32))
+}
+
pub(crate) fn wallet_job_request(
job: WalletJob,
) -> Request {
@@ -2013,6 +2038,20 @@ async fn execute_upload_and_transfer_locked(
classified.kind(),
!classified.aborts_retries()
);
+ // The wallet slot stays held across this sleep. That is
+ // deliberate: a backing-off job is still this wallet's turn, and
+ // releasing it would invite another job onto a wallet that is
+ // about to retry anyway.
+ if let Some(delay) = upload_retry_backoff(&classified, attempt_info) {
+ tracing::info!(
+ "[wallet-job:upload] job_id={} backing off {:?} before attempt {}/{}",
+ remember_job_id.as_deref().unwrap_or("-"),
+ delay,
+ attempt_info.current + 1,
+ attempt_info.max,
+ );
+ tokio::time::sleep(delay).await;
+ }
return Err(classified);
}
};
@@ -2870,7 +2909,7 @@ mod tests {
is_walrus_package_version_mismatch, load_upload_journal, lock_outcome,
mark_remember_job_failed, parse_locked_object_info, parse_wal_balance_alert_info,
persist_upload_journal, persist_uploaded_state, recovery_seal_persistence,
- update_remember_job_after_wallet_error, upload_resume_disposition,
+ update_remember_job_after_wallet_error, upload_resume_disposition, upload_retry_backoff,
wallet_index_for_upload_attempt, wallet_job_request, JobUploadLock, LockOutcome,
UploadResume, WalletJob, WalletJobAttemptInfo, WalletJobError, WalletOperation,
MAX_ATTEMPTS, MAX_CONGESTION_REQUEUES,
@@ -3040,6 +3079,47 @@ different transaction: TransactionDigest(8bjFgRyXRRYwrzQapgEjpHnGhdfNDY7d6xA82Bt
assert_eq!(backoff_duration(5), std::time::Duration::from_secs(32));
}
+ #[test]
+ fn upload_retry_backs_off_between_attempts() {
+ // The production failure this exists for: attempts 2..5 against Walrus
+ // 503 "Too Many Requests" inside one second, four wallet rotations
+ // that never had a chance to land. Each attempt must now be spaced.
+ let rate_limited = WalletJobError::Transient(
+ "durable Walrus upload failed (503 Service Unavailable): {\"error\":\"Too Many Requests\"}"
+ .into(),
+ );
+ for attempt in 1..5usize {
+ assert_eq!(
+ upload_retry_backoff(
+ &rate_limited,
+ WalletJobAttemptInfo {
+ current: attempt,
+ max: 5
+ }
+ ),
+ Some(backoff_duration(attempt as u32)),
+ "attempt {attempt} should wait before the next one",
+ );
+ }
+ }
+
+ #[test]
+ fn upload_retry_does_not_back_off_when_nothing_follows() {
+ let transient = WalletJobError::Transient("upstream hiccup".into());
+ // Final attempt: sleeping only delays the failure already being reported.
+ assert_eq!(
+ upload_retry_backoff(&transient, WalletJobAttemptInfo { current: 5, max: 5 }),
+ None
+ );
+ // Aborting errors never retry, so spacing them buys nothing.
+ let aborting = WalletJobError::GasPoolExhausted("balance::split ENotEnough".into());
+ assert!(aborting.aborts_retries());
+ assert_eq!(
+ upload_retry_backoff(&aborting, WalletJobAttemptInfo { current: 1, max: 5 }),
+ None
+ );
+ }
+
#[test]
fn congestion_backoff_is_minutes_scale_and_capped() {
assert_eq!(congestion_backoff_secs(0), 30);
From 524c3e8066e4879c142dc6d7dd704f38f9a430ac Mon Sep 17 00:00:00 2001
From: Harry Phan
Date: Wed, 16 Sep 2026 11:53:21 +0700
Subject: [PATCH 018/132] fix(mcp): report accepted-then-failed writes on the
next recall
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit
`memwal_remember` now returns as soon as the relayer accepts the job, which
leaves a window where the write still dies — a SEAL encrypt outage, an
exhausted upload budget — with nobody listening. `memwal_remember_status`
answers for one job, but nothing obliges an agent to ask, and storing a memory
is typically the last thing it does in a turn. An unasked question is the same
as a silent loss, and for a product whose promise is durable memory, silent
loss is worse than slow.
Recall is the call an agent always makes, so the bad news rides along there.
`/api/recall` now carries `failed_writes` — this owner's writes that reached
`failed` in the last 24h, capped at 5 — and the MCP tool renders them as a
warning naming the job, the namespace and the relayer's own error text, so the
caller can judge whether re-sending will work.
Scoped to `AuthInfo.owner`, never request input. Bounded by window and limit
because recall is the hottest authed route;
`remember_jobs (owner, status, updated_at DESC)` from migration 006 already
covers the predicate and ordering. A lookup failure degrades to "nothing to
report" rather than failing the recall it is attached to.
Reports repeat until they age out. Suppressing after one sighting would put
the burden back on the caller remembering to act, which is the failure mode
this removes. `failed_writes` is skipped when empty, so an older client and an
older relayer both see exactly today's response.
---
.../__tests__/recall-failed-writes.test.ts | 88 +++++++++++++++++++
services/server/scripts/mcp/tools/recall.ts | 50 ++++++++++-
services/server/src/routes/recall.rs | 40 +++++++++
services/server/src/storage/db.rs | 74 ++++++++++++++++
services/server/src/types.rs | 32 +++++++
5 files changed, 282 insertions(+), 2 deletions(-)
create mode 100644 services/server/scripts/mcp/__tests__/recall-failed-writes.test.ts
diff --git a/services/server/scripts/mcp/__tests__/recall-failed-writes.test.ts b/services/server/scripts/mcp/__tests__/recall-failed-writes.test.ts
new file mode 100644
index 000000000..361755e96
--- /dev/null
+++ b/services/server/scripts/mcp/__tests__/recall-failed-writes.test.ts
@@ -0,0 +1,88 @@
+/**
+ * `memwal_recall` reports writes that were accepted and then failed.
+ *
+ * `memwal_remember` returns as soon as the relayer accepts the job, which
+ * leaves a window where the write still dies — a SEAL outage, an exhausted
+ * upload budget — with nobody listening. `memwal_remember_status` can settle
+ * one job, but nothing obliges an agent to call it, and storing a memory is
+ * typically the last thing it does in a turn. An unasked question is the same
+ * as a silent loss, and silent loss is the worst outcome for a product whose
+ * whole promise is durable memory.
+ *
+ * Recall is the call an agent always makes, so the report rides along there.
+ * These tests pin the rendering, and in particular that it never claims
+ * anything when the relayer said nothing.
+ */
+import test from "node:test";
+import assert from "node:assert/strict";
+
+import { formatFailedWrites } from "../tools/recall.js";
+
+test("says nothing when the relayer reported no failures", () => {
+ assert.equal(formatFailedWrites({ results: [], total: 0 }), "");
+ assert.equal(formatFailedWrites({ failed_writes: [] }), "");
+});
+
+test("says nothing when the relayer predates the field", () => {
+ // An older deployment omits `failed_writes` entirely. That must read as
+ // "nothing to report", never as an error or an empty warning banner.
+ assert.equal(formatFailedWrites({ results: [] }), "");
+ assert.equal(formatFailedWrites(null), "");
+ assert.equal(formatFailedWrites(undefined), "");
+ assert.equal(formatFailedWrites({ failed_writes: "not-an-array" }), "");
+});
+
+test("names the job and states plainly that the fact is not stored", () => {
+ const text = formatFailedWrites({
+ failed_writes: [
+ {
+ job_id: "9d304948-c693-408c-aac7-d7ef9ab98f5d",
+ namespace: "default",
+ error: "Memory encryption backend is unavailable",
+ failed_at: "2026-09-15T15:10:11Z",
+ },
+ ],
+ });
+
+ assert.match(text, /NOT stored/);
+ assert.match(text, /9d304948-c693-408c-aac7-d7ef9ab98f5d/);
+ assert.match(text, /ns=default/);
+ // The relayer's own message is passed through: the difference between a
+ // SEAL outage and a spent retry budget is what tells the caller whether
+ // re-sending is likely to work.
+ assert.match(text, /Memory encryption backend is unavailable/);
+ assert.match(text, /memwal_remember/);
+});
+
+test("reads as singular for one failure and plural for several", () => {
+ const one = formatFailedWrites({
+ failed_writes: [{ job_id: "a", namespace: "default" }],
+ });
+ assert.match(one, /1 earlier write was accepted/);
+ assert.match(one, /that fact is\b/);
+
+ const many = formatFailedWrites({
+ failed_writes: [
+ { job_id: "a", namespace: "default" },
+ { job_id: "b", namespace: "default" },
+ ],
+ });
+ assert.match(many, /2 earlier writes were accepted/);
+ assert.match(many, /those facts are\b/);
+});
+
+test("survives a malformed entry rather than dropping the whole report", () => {
+ // The warning matters more than its formatting: a row missing fields must
+ // still be surfaced, because the alternative is the silent loss this
+ // exists to prevent.
+ const text = formatFailedWrites({
+ failed_writes: [{}, { job_id: "b6da0c96", error: " " }],
+ });
+
+ assert.match(text, /2 earlier writes were accepted/);
+ assert.match(text, /\(unknown job\)/);
+ assert.match(text, /b6da0c96/);
+ // A blank error string adds no information, so it is left off entirely
+ // rather than rendered as a dangling dash.
+ assert.doesNotMatch(text, /b6da0c96.*—/);
+});
diff --git a/services/server/scripts/mcp/tools/recall.ts b/services/server/scripts/mcp/tools/recall.ts
index c48ff4ab5..18b6f364a 100644
--- a/services/server/scripts/mcp/tools/recall.ts
+++ b/services/server/scripts/mcp/tools/recall.ts
@@ -55,6 +55,51 @@ function dedupeKey(text: string): string {
* budget on one fact and crowds out everything else it asked for, which is a
* read-side problem worth fixing on the read side.
*/
+/** One accepted-then-failed write, as the relayer reports it on recall. */
+interface FailedWrite {
+ job_id?: unknown;
+ namespace?: unknown;
+ error?: unknown;
+ failed_at?: unknown;
+}
+
+/**
+ * Render the relayer's report of writes that were accepted and then failed.
+ *
+ * `memwal_remember` returns as soon as the write is accepted, so a job that
+ * dies after that has nobody listening. `memwal_remember_status` can answer
+ * for one, but nothing obliges an agent to ask, and storing a memory is
+ * usually the last thing it does in a turn — so an unasked question is the
+ * same as a silent loss. Recall is the call an agent always makes, which is
+ * why the report rides along here.
+ *
+ * Read defensively off the response rather than through the SDK's typed
+ * result: the field is newer than the pinned SDK's types, and a relayer that
+ * does not send it at all (older deployment) must read as "nothing to
+ * report", not as an error.
+ */
+export function formatFailedWrites(result: unknown): string {
+ const raw = (result as { failed_writes?: unknown } | null)?.failed_writes;
+ if (!Array.isArray(raw) || raw.length === 0) return "";
+
+ const lines = raw.map((entry: FailedWrite) => {
+ const jobId = typeof entry?.job_id === "string" ? entry.job_id : "(unknown job)";
+ const ns = typeof entry?.namespace === "string" ? ` ns=${entry.namespace}` : "";
+ const why = typeof entry?.error === "string" && entry.error.trim() !== ""
+ ? ` — ${entry.error}`
+ : "";
+ return ` job_id=${jobId}${ns}${why}`;
+ });
+
+ const n = raw.length;
+ return (
+ `\n\n⚠ ${n} earlier ${n === 1 ? "write was" : "writes were"} accepted but then FAILED, ` +
+ `so ${n === 1 ? "that fact is" : "those facts are"} NOT stored:\n` +
+ lines.join("\n") +
+ `\nSend ${n === 1 ? "it" : "them"} again with memwal_remember if still wanted.`
+ );
+}
+
export function collapseDuplicates(
results: T[],
): { unique: T[]; collapsed: number } {
@@ -171,13 +216,14 @@ export function registerRecallTool(
const result = await session.memwal.recall(query, limit, namespace);
const droppedRaw = (result as { dropped_count?: unknown }).dropped_count;
const dropped = typeof droppedRaw === "number" ? droppedRaw : 0;
+ const failureReport = formatFailedWrites(result);
const filtered = filterByMaxDistance(result.results, maxDistance);
if (filtered.length === 0) {
return {
content: [
{
type: "text",
- text: emptyRecallText(result.results.length, dropped),
+ text: emptyRecallText(result.results.length, dropped) + failureReport,
},
],
};
@@ -201,7 +247,7 @@ export function registerRecallTool(
content: [
{
type: "text",
- text: lines.join("\n"),
+ text: lines.join("\n") + failureReport,
},
],
};
diff --git a/services/server/src/routes/recall.rs b/services/server/src/routes/recall.rs
index 018c1faf9..8f5a86627 100644
--- a/services/server/src/routes/recall.rs
+++ b/services/server/src/routes/recall.rs
@@ -15,6 +15,42 @@ use std::sync::Arc;
use crate::types::*;
+/// How far back the failure report on a recall response looks.
+///
+/// A day covers the gap between one working session and the next, which is
+/// when an agent would otherwise never learn that yesterday's last write died
+/// after it was accepted.
+const FAILED_WRITE_REPORT_WINDOW: std::time::Duration =
+ std::time::Duration::from_secs(24 * 60 * 60);
+
+/// Most failures reported on one recall. A caller acting on this re-sends the
+/// facts; a wall of them would crowd out the memories it actually asked for.
+const FAILED_WRITE_REPORT_LIMIT: i64 = 5;
+
+/// Recent writes that were accepted and then failed, for `owner`.
+///
+/// Never fails the recall it is attached to. This is a courtesy report on a
+/// read path — losing it costs the caller a warning, while propagating the
+/// error would cost them the memories they actually asked for, so a failed
+/// lookup degrades to "nothing to report" and says so in the log.
+async fn failed_writes_for(state: &AppState, owner: &str) -> Vec {
+ match state
+ .db
+ .recent_failed_remember_jobs(owner, FAILED_WRITE_REPORT_WINDOW, FAILED_WRITE_REPORT_LIMIT)
+ .await
+ {
+ Ok(failed) => failed,
+ Err(e) => {
+ tracing::warn!(
+ "recall: failed-write report unavailable for owner={}: {}",
+ owner,
+ e
+ );
+ Vec::new()
+ }
+ }
+}
+
// ============================================================
// Recall query-embedding cache (Redis) — wraps the Embedder service
// ============================================================
@@ -211,6 +247,9 @@ pub async fn recall(
results: vec![],
total: 0,
dropped_count: 0,
+ // Reported even with no hits: an empty recall is exactly when a
+ // caller is most likely to be looking for the fact that failed.
+ failed_writes: failed_writes_for(&state, owner).await,
}));
}
@@ -297,6 +336,7 @@ pub async fn recall(
results,
total,
dropped_count,
+ failed_writes: failed_writes_for(&state, owner).await,
}))
}
diff --git a/services/server/src/storage/db.rs b/services/server/src/storage/db.rs
index a45d31ca5..ae4f736ad 100644
--- a/services/server/src/storage/db.rs
+++ b/services/server/src/storage/db.rs
@@ -1837,6 +1837,80 @@ impl VectorDb {
Ok(row)
}
+ /// Writes this owner started that ended in `failed` within `window`.
+ ///
+ /// Serves the failure report attached to recall responses. Owner-scoped
+ /// from `AuthInfo`, never from request input, so one account cannot read
+ /// another's failures.
+ ///
+ /// Bounded by both a time window and `limit` because this runs on the
+ /// read path: recall is the hottest authed route, and an account with a
+ /// long tail of old failures must not turn every recall into a large
+ /// scan. `remember_jobs (owner, status, updated_at DESC)` (migration 006)
+ /// covers the predicate and the ordering, so this is an index range scan
+ /// of at most `limit` rows.
+ ///
+ /// Failures repeat across calls until they age out of the window. That is
+ /// deliberate — suppressing a report after one sighting would put the
+ /// notice back on the caller remembering to act on it, which is the
+ /// failure mode this whole path exists to remove.
+ pub async fn recent_failed_remember_jobs(
+ &self,
+ owner: &str,
+ window: std::time::Duration,
+ limit: i64,
+ ) -> Result, AppError> {
+ let started = std::time::Instant::now();
+ let rows = sqlx::query_as::<
+ _,
+ (
+ String,
+ String,
+ Option,
+ chrono::DateTime,
+ ),
+ >(
+ "SELECT id, namespace, error_msg, updated_at
+ FROM remember_jobs
+ WHERE owner = $1 AND status = 'failed' AND updated_at >= $2
+ ORDER BY updated_at DESC
+ LIMIT $3",
+ )
+ .bind(owner)
+ .bind(chrono::Utc::now() - chrono::Duration::from_std(window).unwrap_or_default())
+ .bind(limit)
+ .fetch_all(&self.pool)
+ .await;
+
+ let rows = match rows {
+ Ok(rows) => rows,
+ Err(e) => {
+ crate::observability::observe_db(
+ "remember_jobs.recent_failed",
+ "error",
+ started.elapsed(),
+ );
+ return Err(AppError::Internal(format!(
+ "Failed to list failed remember jobs: {}",
+ e
+ )));
+ }
+ };
+ crate::observability::observe_db("remember_jobs.recent_failed", "ok", started.elapsed());
+
+ Ok(rows
+ .into_iter()
+ .map(
+ |(job_id, namespace, error, failed_at)| crate::types::FailedWrite {
+ job_id,
+ namespace,
+ error,
+ failed_at: failed_at.to_rfc3339(),
+ },
+ )
+ .collect())
+ }
+
/// Hard-delete all vector index rows for a given owner + namespace.
/// (Walrus blobs themselves persist — Walrus has no delete; this only
/// removes the local `vector_entries` rows, so the memories stop being
diff --git a/services/server/src/types.rs b/services/server/src/types.rs
index db31b3f31..151e9a879 100644
--- a/services/server/src/types.rs
+++ b/services/server/src/types.rs
@@ -1569,6 +1569,38 @@ pub struct RecallResponse {
/// failed and were silently omitted from `results`. Zero on the happy path.
#[serde(default, skip_serializing_if = "is_zero_usize")]
pub dropped_count: usize,
+ /// Writes this owner started recently that ended in `failed` — facts the
+ /// caller was told were accepted but that were never stored.
+ ///
+ /// Carried on the *read* path on purpose. A write now returns as soon as
+ /// the relayer accepts the job, so a failure after that point has no
+ /// caller left listening: `memwal_remember_status` answers it, but nothing
+ /// obliges an agent to ask, and saving a memory is typically the last
+ /// thing it does in a turn. Recall is the call an agent always makes, so
+ /// attaching the bad news here is what turns a silent loss into a visible
+ /// one.
+ ///
+ /// Empty on the happy path and omitted from the wire, so an older client
+ /// that ignores the field sees exactly today's response.
+ #[serde(default, skip_serializing_if = "Vec::is_empty")]
+ pub failed_writes: Vec,
+}
+
+/// One write that was accepted and then failed, as reported back on recall.
+#[derive(Debug, Serialize)]
+pub struct FailedWrite {
+ /// The `job_id` the write returned when it was accepted, so a caller can
+ /// match this against what it was told at the time.
+ pub job_id: String,
+ pub namespace: String,
+ /// The relayer's own failure message, verbatim. Passed through rather than
+ /// summarised: the distinction between (say) a SEAL outage and an
+ /// exhausted upload budget is what tells a caller whether re-sending the
+ /// fact is likely to work.
+ #[serde(skip_serializing_if = "Option::is_none")]
+ pub error: Option,
+ /// When the job reached `failed`, RFC 3339.
+ pub failed_at: String,
}
fn is_zero_usize(n: &usize) -> bool {
From 719feb6ca81874af17764bb9412c1143ce4ac9e4 Mon Sep 17 00:00:00 2001
From: Harry Phan
Date: Wed, 16 Sep 2026 12:09:10 +0700
Subject: [PATCH 019/132] fix(mcp): keep the status wait under the MCP client's
own deadline
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit
`memwal_remember_status` advertised `waitMs` up to 60000. The MCP SDK times a
request out at `DEFAULT_REQUEST_TIMEOUT_MSEC` — also 60000 — unless the caller
overrides it, so a caller using the documented maximum always lost the race:
the client gave up first and the agent saw `MCP error -32001: Request timed
out` with no way to tell a slow write from a broken tool.
Found by driving the real tools against the production relayer rather than a
mock: `waitMs: 60000` on a three-job batch reproduced it every time.
The ceiling is now 45s, which leaves headroom for the round trip and still
covers the median write, and the tool description interpolates the constants
instead of restating them, so the advertised range cannot drift from the
schema again. A job outliving the wait is not lost — the job_id stays valid
and the caller asks again, which is why this tool is separate from the write.
Also start the recall failure report concurrently instead of awaiting it after
hydration. The published SDK aborts a recall after a hard 15s that no caller
can raise, and a live recall was measured landing on exactly 15.0s, so the
report has to cost the critical path nothing.
Adds a regression test asserting the tool's advertised maximum keeps at least
10s of headroom under the SDK's request deadline, and a live end-to-end script
that walks remember → status and bulk → job_ids → blob_ids against a real
relayer, since the mocked tests cannot catch a client/tool deadline collision.
---
.../server/scripts/mcp/__tests__/live-e2e.mjs | 162 ++++++++++++++++++
.../mcp/__tests__/remember-deadline.test.ts | 45 +++++
.../scripts/mcp/tools/remember-status.ts | 28 ++-
services/server/src/routes/recall.rs | 18 +-
4 files changed, 243 insertions(+), 10 deletions(-)
create mode 100644 services/server/scripts/mcp/__tests__/live-e2e.mjs
diff --git a/services/server/scripts/mcp/__tests__/live-e2e.mjs b/services/server/scripts/mcp/__tests__/live-e2e.mjs
new file mode 100644
index 000000000..18357ece7
--- /dev/null
+++ b/services/server/scripts/mcp/__tests__/live-e2e.mjs
@@ -0,0 +1,162 @@
+/**
+ * End-to-end exercise of the PR's MCP tools against a LIVE relayer.
+ *
+ * The unit tests drive these tools through a mocked session. That pins the
+ * wording and the branching, but it cannot answer the question that actually
+ * matters before merge: does `memwal_remember_bulk` hand back job_ids the
+ * real `memwal_remember_status` can then resolve into real blob_ids?
+ *
+ * So this wires the same in-process MCP server the unit tests use to a real
+ * MemWal session, and walks the flow a user's agent would walk. It is not a
+ * unit test — it writes paid blobs — which is why it lives here as a script
+ * rather than under `node --test`.
+ *
+ * NODE_USE_ENV_PROXY=1 MEMWAL_CREDS_DIR= \
+ * node --import tsx mcp/__tests__/live-e2e.mjs
+ */
+import { readFileSync } from "node:fs";
+import assert from "node:assert/strict";
+import { Client } from "@modelcontextprotocol/sdk/client/index.js";
+import { InMemoryTransport } from "@modelcontextprotocol/sdk/inMemory.js";
+import { MemWal } from "@mysten-incubation/memwal";
+import { createMcpServer } from "../server.js";
+
+const credsDir = process.env.MEMWAL_CREDS_DIR;
+if (!credsDir) throw new Error("set MEMWAL_CREDS_DIR");
+const creds = JSON.parse(readFileSync(`${credsDir}/credentials.json`, "utf8"));
+const NS = "memwal-probe";
+
+const memwal = new MemWal({
+ key: creds.delegatePrivateKey,
+ accountId: creds.accountId,
+ serverUrl: creds.relayerUrl,
+ namespace: NS,
+});
+
+const session = {
+ accountId: creds.accountId,
+ delegateKeyHex: creds.delegatePrivateKey,
+ delegatePubKeyHex: creds.delegatePublicKeyHex,
+ namespace: NS,
+ memwal,
+ relayerUrl: creds.relayerUrl,
+ authMethod: "delegate-key",
+ oauthScope: "memwal:read memwal:write",
+ agentClient: "other",
+};
+
+const server = createMcpServer(session);
+const client = new Client({ name: "live-e2e", version: "1.0.0" }, { capabilities: {} });
+const [clientT, serverT] = InMemoryTransport.createLinkedPair();
+await Promise.all([client.connect(clientT), server.connect(serverT)]);
+
+const ms = () => performance.now();
+const fmt = (n) => (n < 1000 ? `${Math.round(n)}ms` : `${(n / 1000).toFixed(1)}s`);
+const textOf = (r) => r.content?.[0]?.text ?? "";
+
+async function call(name, args) {
+ const t0 = ms();
+ const r = await client.callTool({ name, arguments: args });
+ return { ms: ms() - t0, text: textOf(r), isError: r.isError === true };
+}
+
+function show(label, r, extra = "") {
+ console.log(
+ ` ${label.padEnd(30)} ${fmt(r.ms).padStart(8)} ${r.isError ? "ERR " : "ok "} ${extra}`,
+ );
+}
+
+const stamp = new Date().toISOString();
+const failures = [];
+function check(name, fn) {
+ try {
+ fn();
+ console.log(` ✔ ${name}`);
+ } catch (e) {
+ failures.push(name);
+ console.log(` ✖ ${name}\n ${e.message.split("\n")[0]}`);
+ }
+}
+
+console.log(`\nlive e2e — relayer ${creds.relayerUrl} ns=${NS}\n`);
+
+// ── single write ────────────────────────────────────────────────
+console.log("single remember → status");
+const single = await call("memwal_remember", {
+ text: `live-e2e single ${stamp}`,
+ namespace: NS,
+});
+show("memwal_remember", single);
+const singleJob = /job_id=([0-9a-f-]+)/.exec(single.text)?.[1];
+
+check("remember returns fast", () => assert.ok(single.ms < 5000, `${fmt(single.ms)}`));
+check("remember hands back a job_id", () => assert.ok(singleJob, single.text.slice(0, 120)));
+check("remember does not claim it is saved", () =>
+ assert.doesNotMatch(single.text, /Saved to Walrus Memory/));
+
+const statusNoWait = await call("memwal_remember_status", { job_id: singleJob, waitMs: 0 });
+show("status waitMs=0", statusNoWait);
+check("waitMs=0 answers immediately", () =>
+ assert.ok(statusNoWait.ms < 3000, `${fmt(statusNoWait.ms)}`));
+
+// ── bulk write → job_ids → batch status ─────────────────────────
+console.log("\nbulk remember → status with job_ids");
+const bulk = await call("memwal_remember_bulk", {
+ namespace: NS,
+ facts: [1, 2, 3].map((i) => `live-e2e bulk ${stamp} item ${i}`),
+});
+show("memwal_remember_bulk", bulk);
+const bulkJobs = [...bulk.text.matchAll(/job_id=([0-9a-f-]+)/g)].map((m) => m[1]);
+
+check("bulk returns fast", () => assert.ok(bulk.ms < 10_000, `${fmt(bulk.ms)}`));
+check("bulk hands back one job_id per fact", () =>
+ assert.equal(bulkJobs.length, 3, `got ${bulkJobs.length}: ${bulk.text.slice(0, 200)}`));
+check("bulk pairs each job_id with its fact", () =>
+ assert.match(bulk.text, /item 1/));
+
+// THE question: does job_ids actually resolve to data?
+// 45s is the tool's ceiling — deliberately under the MCP client's own 60s
+// request deadline, so the tool answers rather than the client giving up.
+const batch = await call("memwal_remember_status", { job_ids: bulkJobs, waitMs: 45_000 });
+show("status job_ids (45s budget)", batch);
+check("job_ids returns a line per job", () => {
+ // Guard the loop: with no ids collected it would pass by doing nothing,
+ // which is exactly the case this check exists to catch.
+ assert.ok(bulkJobs.length > 0, "no job_ids to resolve");
+ for (const id of bulkJobs) assert.match(batch.text, new RegExp(id.slice(0, 8)));
+});
+check("job_ids resolves to real blob_ids", () => {
+ const blobs = [...batch.text.matchAll(/blob_id=([A-Za-z0-9_-]{20,})/g)];
+ assert.ok(blobs.length > 0, `no blob_id in: ${batch.text.slice(0, 300)}`);
+});
+
+// ── edge cases ──────────────────────────────────────────────────
+console.log("\nedge cases");
+const both = await call("memwal_remember_status", { job_id: "a", job_ids: ["b"] });
+show("job_id + job_ids together", both);
+check("rejects both ids at once", () => assert.ok(both.isError || /not both/i.test(both.text)));
+
+const neither = await call("memwal_remember_status", {});
+show("neither id", neither);
+check("rejects an empty call", () => assert.ok(neither.isError || /Pass job_id/i.test(neither.text)));
+
+const unknown = await call("memwal_remember_status", {
+ job_id: "00000000-0000-0000-0000-000000000000",
+ waitMs: 0,
+});
+show("unknown job_id", unknown);
+check("an unknown job is an error, not a silent ok", () =>
+ assert.ok(unknown.isError || /not found/i.test(unknown.text)));
+
+// ── recall, including the failed-write report ───────────────────
+console.log("\nrecall");
+const recall = await call("memwal_recall", { query: "live-e2e", limit: 5 });
+show("memwal_recall", recall, `${recall.text.split("\n").length} lines`);
+check("recall answers under the SDK's 15s abort", () =>
+ assert.ok(recall.ms < 15_000, `${fmt(recall.ms)}`));
+check("recall never invents a failure report", () => {
+ if (/FAILED/.test(recall.text)) assert.match(recall.text, /NOT stored/);
+});
+
+console.log(`\n${failures.length === 0 ? "all checks passed" : `${failures.length} FAILED: ${failures.join(", ")}`}\n`);
+process.exit(failures.length === 0 ? 0 : 1);
diff --git a/services/server/scripts/mcp/__tests__/remember-deadline.test.ts b/services/server/scripts/mcp/__tests__/remember-deadline.test.ts
index c74d55030..8adca54a4 100644
--- a/services/server/scripts/mcp/__tests__/remember-deadline.test.ts
+++ b/services/server/scripts/mcp/__tests__/remember-deadline.test.ts
@@ -147,3 +147,48 @@ test("a stalled batch status read is bounded too", async (t) => {
assert.equal((result as { isError?: boolean }).isError, true);
assert.match(textOf(result), /did not accept/);
});
+
+/**
+ * The status tool must answer before the MCP client gives up on it.
+ *
+ * `@modelcontextprotocol/sdk` times a request out at
+ * `DEFAULT_REQUEST_TIMEOUT_MSEC` (60s) unless the caller overrides it. A tool
+ * whose advertised maximum equals that deadline loses every race it enters:
+ * the client reports `MCP error -32001: Request timed out` and the agent
+ * cannot tell a slow write from a broken tool. Caught live against the
+ * production relayer with `waitMs: 60000` on a three-job batch.
+ */
+test("the status wait ceiling stays under the MCP client's own deadline", async (t: TestContext) => {
+ const { DEFAULT_REQUEST_TIMEOUT_MSEC } = await import(
+ "@modelcontextprotocol/sdk/shared/protocol.js"
+ );
+ const [clientTransport, serverTransport] = InMemoryTransport.createLinkedPair();
+ const server = createMcpServer({
+ oauthScope: "memwal:read memwal:write",
+ } as MemWalSession);
+ const client = new Client({ name: "status-deadline-test", version: "1.0.0" });
+ t.after(async () => {
+ await client.close();
+ await server.close();
+ });
+ await server.connect(serverTransport);
+ await client.connect(clientTransport);
+
+ const { tools } = await client.listTools();
+ const status = tools.find((tool) => tool.name === "memwal_remember_status");
+ assert.ok(status, "memwal_remember_status is registered");
+
+ const max = (
+ status.inputSchema as { properties?: { waitMs?: { maximum?: number } } }
+ ).properties?.waitMs?.maximum;
+ assert.equal(typeof max, "number", "waitMs advertises a maximum");
+ assert.ok(
+ max < DEFAULT_REQUEST_TIMEOUT_MSEC,
+ `waitMs max ${max}ms must stay under the client deadline ${DEFAULT_REQUEST_TIMEOUT_MSEC}ms`,
+ );
+ // Headroom for the round trip, not just a strict inequality.
+ assert.ok(
+ DEFAULT_REQUEST_TIMEOUT_MSEC - max >= 10_000,
+ `only ${DEFAULT_REQUEST_TIMEOUT_MSEC - max}ms of headroom before the client gives up`,
+ );
+});
diff --git a/services/server/scripts/mcp/tools/remember-status.ts b/services/server/scripts/mcp/tools/remember-status.ts
index 97d4adcb8..74c336fea 100644
--- a/services/server/scripts/mcp/tools/remember-status.ts
+++ b/services/server/scripts/mcp/tools/remember-status.ts
@@ -12,11 +12,26 @@ import {
} from "./remember-wait.js";
/**
- * Ceiling on a single status wait. Past this an MCP client is more likely to
- * time out the call than the job is to finish, and the caller can simply ask
- * again — the job_id stays valid.
+ * Ceiling on a single status wait, held under the MCP client's own deadline.
+ *
+ * `@modelcontextprotocol/sdk` times a request out after
+ * `DEFAULT_REQUEST_TIMEOUT_MSEC` = 60s unless the caller overrides it. A tool
+ * that waits the full 60s therefore loses every race it enters: the client
+ * gives up first and the agent sees `MCP error -32001: Request timed out`
+ * instead of the answer the tool was about to return. Confirmed live against
+ * the production relayer — `waitMs: 60000` on a three-job batch returned
+ * exactly that, with no way for the caller to tell a slow write from a broken
+ * tool.
+ *
+ * 45s leaves room for the round trip and still covers the median write. A job
+ * that outlives it is not lost: the job_id stays valid and the caller asks
+ * again, which is the whole point of this tool being separate from the write.
*/
-const MAX_STATUS_WAIT_MS = 60_000;
+const MAX_STATUS_WAIT_MS = 45_000;
+
+/** Wait applied when the caller does not choose one. Declared above
+ * `STATUS_INPUT` because the tool description interpolates it. */
+const DEFAULT_STATUS_WAIT_MS = 10_000;
/** Matches the bulk write cap, so a whole batch settles in one call. */
const MAX_STATUS_JOB_IDS = 20;
@@ -42,13 +57,10 @@ const STATUS_INPUT = {
.max(MAX_STATUS_WAIT_MS)
.optional()
.describe(
- "How long to wait for the job to finish, in milliseconds (0-60000, default 10000). Pass 0 to read the current state without waiting."
+ `How long to wait for the job to finish, in milliseconds (0-${MAX_STATUS_WAIT_MS}, default ${DEFAULT_STATUS_WAIT_MS}). Pass 0 to read the current state without waiting.`
),
} as const;
-/** Wait applied when the caller does not choose one. */
-const DEFAULT_STATUS_WAIT_MS = 10_000;
-
/**
* memwal_remember_status — resolve a remember job that `memwal_remember`
* handed back as still in flight.
diff --git a/services/server/src/routes/recall.rs b/services/server/src/routes/recall.rs
index 8f5a86627..6c6316014 100644
--- a/services/server/src/routes/recall.rs
+++ b/services/server/src/routes/recall.rs
@@ -204,6 +204,18 @@ pub async fn recall(
// Owner is derived from delegate key via onchain verification (auth middleware)
let owner = &auth.owner;
let namespace = &body.namespace;
+
+ // Started here rather than awaited at the end, so it overlaps the embed,
+ // search, Walrus download and SEAL decrypt that follow instead of adding
+ // to them. The published SDK aborts a recall after a hard 15s that no
+ // caller can raise, and recall has been measured landing on exactly that
+ // — so this report has to cost the critical path nothing.
+ let failed_writes = {
+ let state = state.clone();
+ let owner = owner.clone();
+ tokio::spawn(async move { failed_writes_for(&state, &owner).await })
+ };
+
tracing::info!(
query_len = body.query.len(),
owner = %owner,
@@ -249,7 +261,7 @@ pub async fn recall(
dropped_count: 0,
// Reported even with no hits: an empty recall is exactly when a
// caller is most likely to be looking for the fact that failed.
- failed_writes: failed_writes_for(&state, owner).await,
+ failed_writes: failed_writes.await.unwrap_or_default(),
}));
}
@@ -336,7 +348,9 @@ pub async fn recall(
results,
total,
dropped_count,
- failed_writes: failed_writes_for(&state, owner).await,
+ // A panic in the report task must not take the recall with it; the
+ // caller loses a warning, not their memories.
+ failed_writes: failed_writes.await.unwrap_or_default(),
}))
}
From df314c3eb2e209274bcba8881c4d39ecba06c26f Mon Sep 17 00:00:00 2001
From: Harry Phan
Date: Wed, 16 Sep 2026 12:11:01 +0700
Subject: [PATCH 020/132] test(mcp): assert the batch status accounts for jobs,
not that they landed
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit
The live check demanded a blob_id for every job in a 45s window. Measured
write latency is p50 ~34s and p90 ~65s, so that window legitimately expires
with writes still in flight — the check was asserting the relayer be fast
rather than the tool be correct.
What must hold is that every job comes back with a definite state, that the
count of blob_ids matches the count reported saved, and that a report with
nothing saved says so rather than reading as success.
---
services/server/scripts/mcp/__tests__/live-e2e.mjs | 11 +++++++++--
1 file changed, 9 insertions(+), 2 deletions(-)
diff --git a/services/server/scripts/mcp/__tests__/live-e2e.mjs b/services/server/scripts/mcp/__tests__/live-e2e.mjs
index 18357ece7..9420d4ecd 100644
--- a/services/server/scripts/mcp/__tests__/live-e2e.mjs
+++ b/services/server/scripts/mcp/__tests__/live-e2e.mjs
@@ -125,9 +125,16 @@ check("job_ids returns a line per job", () => {
assert.ok(bulkJobs.length > 0, "no job_ids to resolve");
for (const id of bulkJobs) assert.match(batch.text, new RegExp(id.slice(0, 8)));
});
-check("job_ids resolves to real blob_ids", () => {
+check("job_ids accounts for every job, saved or not", () => {
+ // Not "must have blob_ids": measured p50 is ~34s and p90 ~65s, so a 45s
+ // budget legitimately expires with writes still in flight. What must hold
+ // is that every job comes back with a definite state and the report never
+ // reads as success when nothing landed.
+ assert.match(batch.text, /\d+\/\d+ saved/, batch.text.slice(0, 200));
const blobs = [...batch.text.matchAll(/blob_id=([A-Za-z0-9_-]{20,})/g)];
- assert.ok(blobs.length > 0, `no blob_id in: ${batch.text.slice(0, 300)}`);
+ const saved = Number(/(\d+)\/\d+ saved/.exec(batch.text)?.[1] ?? "0");
+ assert.equal(blobs.length, saved, "a blob_id for each job reported saved, and no more");
+ if (saved === 0) assert.match(batch.text, /still uploading/);
});
// ── edge cases ──────────────────────────────────────────────────
From 1946ff18492e0da82f1d5fd0223036d56e212ba2 Mon Sep 17 00:00:00 2001
From: Harry Phan
Date: Wed, 16 Sep 2026 12:13:08 +0700
Subject: [PATCH 021/132] fix(mcp): let the bridge's cold-start tools match
what the sidecar accepts
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit
The bridge carries its own copy of the tool list to answer tools/list before
the relayer session is up. Two entries had drifted from the sidecar and both
break a real flow during that window.
memwal_remember_status advertised only `job_id`, marked it required, and set
additionalProperties:false — while the sidecar takes `job_id` OR `job_ids`,
and the pending body memwal_remember_bulk returns tells the agent to come
back with `job_ids=[...]`. A schema-validating client refuses that call, so
the instruction the tool itself gives is unfollowable and a batch cannot be
settled at all until tools/list_changed lands.
memwal_remember_bulk still carried its pre-fast-return description, with no
hint that a result can come back ACCEPTED-but-not-saved. An agent reading it
reports a queued batch to the user as stored.
Neither id field is marked required: the sidecar rejects both-at-once and
neither-at-all, which JSON Schema cannot express here without a `oneOf` some
clients mishandle, so that stays a handler check.
tool-definitions.test.mjs only ever compared the bridge against literals,
never against the sidecar — which is why the drift was invisible. It now
pins the parts an agent acts on: that a batch can be settled, and that both
write tools admit a result may not be saved yet. Verified against the old
definitions, where both new tests fail.
---
packages/mcp/src/auth-required.ts | 19 +++++++--
packages/mcp/test/tool-definitions.test.mjs | 44 +++++++++++++++++++++
2 files changed, 60 insertions(+), 3 deletions(-)
diff --git a/packages/mcp/src/auth-required.ts b/packages/mcp/src/auth-required.ts
index b099d0d56..eb7a9f919 100644
--- a/packages/mcp/src/auth-required.ts
+++ b/packages/mcp/src/auth-required.ts
@@ -72,7 +72,7 @@ function buildToolDefinitions(proactive: boolean) {
title: "Remember Multiple Facts",
annotations: { readOnlyHint: false, destructiveHint: false },
description:
- "Save multiple durable facts in one call. Use when you learned several distinct facts at once (onboarding details, a list of preferences, decisions from a discussion). Pass an array of complete fact statements (max 20) — do not summarize. Prefer this over repeated memwal_remember calls.",
+ "Save multiple durable facts in one call. Use when you learned several distinct facts at once (onboarding details, a list of preferences, decisions from a discussion). Pass an array of complete fact statements (max 20) — do not summarize. Prefer this over repeated memwal_remember calls. Walrus writes queue, so this normally returns job_ids with the facts NOT yet saved — in that case say they are being saved rather than claiming they are stored, and resolve them with memwal_remember_status.",
inputSchema: {
type: "object",
properties: {
@@ -93,14 +93,27 @@ function buildToolDefinitions(proactive: boolean) {
title: "Check a Remember Job",
annotations: { readOnlyHint: true, destructiveHint: false },
description:
- "Check whether an in-flight memwal_remember write has landed. Call this with the job_id memwal_remember returned when it reported the fact was NOT saved yet. Returns the blob_id once stored, reports that it is still uploading (call again), or reports that it failed \u2014 in which case the fact was never stored and you should send it again with memwal_remember.",
+ "Check whether in-flight Walrus Memory writes have landed. Call this with the job_id memwal_remember returned, or job_ids from memwal_remember_bulk, when the write was reported NOT saved yet. Returns the blob_id once stored, reports that it is still uploading (call again with the ids still listed), or reports that it failed \u2014 in which case the fact was never stored and you should send it again. A batch can come back mixed, so read every line before telling the user anything is saved.",
inputSchema: {
type: "object",
properties: {
job_id: { type: "string", minLength: 1 },
+ // The sidecar takes either one id or a whole batch, and the
+ // pending body `memwal_remember_bulk` returns tells the agent
+ // to come back with `job_ids`. Advertising only `job_id` —
+ // required, under `additionalProperties: false` — made that
+ // instruction unfollowable for the whole cold-start window.
+ job_ids: {
+ type: "array",
+ items: { type: "string", minLength: 1 },
+ minItems: 1,
+ maxItems: 20,
+ },
waitMs: { type: "integer", minimum: 0, maximum: 60000, default: 10000 },
},
- required: ["job_id"],
+ // Neither is required on its own; the sidecar rejects passing both
+ // and rejects passing neither, which JSON Schema cannot express
+ // here without a `oneOf` that some clients mishandle.
additionalProperties: false,
},
},
diff --git a/packages/mcp/test/tool-definitions.test.mjs b/packages/mcp/test/tool-definitions.test.mjs
index 6bbdbdb6c..dc60266c2 100644
--- a/packages/mcp/test/tool-definitions.test.mjs
+++ b/packages/mcp/test/tool-definitions.test.mjs
@@ -65,3 +65,47 @@ test("memwal_recall is advertised as a read-only search", () => {
destructiveHint: false,
});
});
+
+/**
+ * The bridge carries its own copy of the tool list for the cold-start window,
+ * and nothing compares it to the sidecar's — the tests above only pin it
+ * against literals. That is how the two drifted: the sidecar grew batch
+ * settling and a pending-result contract, the bridge's copy did not, and the
+ * mismatch is invisible until a real session hits it.
+ *
+ * These pin the parts an agent acts on, so the next divergence fails here
+ * instead of in someone's first save of the session.
+ */
+
+test("cold-start memwal_remember_status accepts a whole batch", () => {
+ for (const list of [TOOL_DEFINITIONS, SIGNED_OUT_TOOL_DEFINITIONS]) {
+ const tool = list.find((t) => t.name === "memwal_remember_status");
+ assert.ok(tool, "missing memwal_remember_status");
+
+ // `memwal_remember_bulk` hands back a pending body telling the agent to
+ // call this with job_ids. Under additionalProperties:false an absent
+ // property makes that instruction unfollowable.
+ const ids = tool.inputSchema.properties.job_ids;
+ assert.ok(ids, "job_ids must be advertised, or a batch cannot be settled");
+ assert.equal(ids.type, "array");
+ assert.equal(ids.items.type, "string");
+ // Matches MAX_BULK_ITEMS, so a full batch settles in one call.
+ assert.equal(ids.maxItems, 20);
+
+ // Requiring job_id would reject the batch form outright.
+ assert.ok(
+ !(tool.inputSchema.required ?? []).includes("job_id"),
+ "job_id must not be required — the batch form passes job_ids instead",
+ );
+ }
+});
+
+test("cold-start write tools warn that a result may not be saved yet", () => {
+ // Both write tools return at accept now. An agent that was never told a
+ // pending result is normal reports it to the user as stored.
+ for (const name of ["memwal_remember", "memwal_remember_bulk"]) {
+ const d = desc(TOOL_DEFINITIONS, name);
+ assert.match(d, /NOT yet saved|NOT saved/i, `${name} omits the pending warning`);
+ assert.match(d, /memwal_remember_status/, `${name} does not say how to settle it`);
+ }
+});
From 1f1210a3718305533d571d5dfd109369f268b68d Mon Sep 17 00:00:00 2001
From: Harry Phan
Date: Wed, 16 Sep 2026 12:16:19 +0700
Subject: [PATCH 022/132] fix(mcp): stop a still-uploading row reading as a
failed one
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit
Observed against the live relayer while timing the flow: settling a batch
printed
4. [still uploading] job_id=2868f48f… error=polling timed out after 45000ms
`waitForRememberJobs` stamps that message on every row that had not landed
when the budget ran out. It is our clock expiring, not the job failing — the
write is still on its way — but rendering it as `error=` next to "still
uploading" tells the agent the opposite, and the agent tells the user. Only a
terminal row (failed / not_found) explains itself now.
Also re-syncs the bridge's advertised waitMs ceiling, which drifted again in
the other direction when the sidecar lowered MAX_STATUS_WAIT_MS to 45s. The
bridge still advertised 60000, and a caller taking it at its word got
MCP error -32602: Number must be less than or equal to 45000
That is the same failure shape as the job_ids drift, so it gets the same
treatment: a test pinning the bound rather than a one-off correction.
---
packages/mcp/src/auth-required.ts | 2 +-
packages/mcp/test/tool-definitions.test.mjs | 11 +++++++++++
services/server/scripts/mcp/tools/remember-status.ts | 8 +++++++-
3 files changed, 19 insertions(+), 2 deletions(-)
diff --git a/packages/mcp/src/auth-required.ts b/packages/mcp/src/auth-required.ts
index eb7a9f919..cfc569ecc 100644
--- a/packages/mcp/src/auth-required.ts
+++ b/packages/mcp/src/auth-required.ts
@@ -109,7 +109,7 @@ function buildToolDefinitions(proactive: boolean) {
minItems: 1,
maxItems: 20,
},
- waitMs: { type: "integer", minimum: 0, maximum: 60000, default: 10000 },
+ waitMs: { type: "integer", minimum: 0, maximum: 45000, default: 10000 },
},
// Neither is required on its own; the sidecar rejects passing both
// and rejects passing neither, which JSON Schema cannot express
diff --git a/packages/mcp/test/tool-definitions.test.mjs b/packages/mcp/test/tool-definitions.test.mjs
index dc60266c2..1386921f5 100644
--- a/packages/mcp/test/tool-definitions.test.mjs
+++ b/packages/mcp/test/tool-definitions.test.mjs
@@ -100,6 +100,17 @@ test("cold-start memwal_remember_status accepts a whole batch", () => {
}
});
+test("cold-start waitMs bound matches the sidecar's ceiling", () => {
+ // The sidecar validates waitMs with zod and rejects anything above its own
+ // cap. Advertising a larger maximum invites the agent to send a value that
+ // comes straight back as an MCP validation error — observed live at 60000
+ // once the sidecar lowered its ceiling to 45000.
+ for (const list of [TOOL_DEFINITIONS, SIGNED_OUT_TOOL_DEFINITIONS]) {
+ const tool = list.find((t) => t.name === "memwal_remember_status");
+ assert.equal(tool.inputSchema.properties.waitMs.maximum, 45000);
+ }
+});
+
test("cold-start write tools warn that a result may not be saved yet", () => {
// Both write tools return at accept now. An agent that was never told a
// pending result is normal reports it to the user as stored.
diff --git a/services/server/scripts/mcp/tools/remember-status.ts b/services/server/scripts/mcp/tools/remember-status.ts
index 74c336fea..abfc6a6bf 100644
--- a/services/server/scripts/mcp/tools/remember-status.ts
+++ b/services/server/scripts/mcp/tools/remember-status.ts
@@ -210,7 +210,13 @@ async function settleBatch(
const lines = rows.map((r, i) => {
const blob = r.blob_id ? ` blob_id=${r.blob_id}` : "";
- const err = r.error ? ` error=${r.error}` : "";
+ // `waitForRememberJobs` stamps "polling timed out after Nms" on rows
+ // that simply had not landed when the budget ran out. That is our
+ // clock expiring, not the job failing, so showing it as `error=` next
+ // to "still uploading" reads like the write broke when it is still on
+ // its way. Only a terminal row gets to explain itself.
+ const terminal = r.status === "failed" || r.status === "not_found";
+ const err = terminal && r.error ? ` error=${r.error}` : "";
const state =
r.status === "done"
? "saved"
From fa918d3f70b5b1995f4d869684fd236abf221ad7 Mon Sep 17 00:00:00 2001
From: Harry Phan
Date: Wed, 16 Sep 2026 13:26:54 +0700
Subject: [PATCH 023/132] fix(relayer): stop the orphan sweep from failing
healthy bulk and analyze writes
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit
The sweep I added keyed only on `preparation_encrypted_b64 IS NULL`, on the
assumption that column marks "preparation never finished". It does not. It is
written in exactly one place — spawn_prepare_remember_job, the SINGLE remember
path. `/api/remember/bulk` and `/api/analyze` insert their rows directly and
never write it, so for those two the column is ALWAYS NULL, healthy or not.
The predicate therefore matched every bulk and analyze job that sat in
`pending` past the 10 minute TTL — which is ordinary, not pathological, since
WALRUS_UPLOAD_PER_WALLET_CONCURRENCY defaults to 1 and the queue is minutes
deep under load. It would have marked live paid writes `failed` while their
upload was still queued, and the recall failure report would then have told
the user to send them again. Worse than the gap it was meant to close.
`prepare_claimed_at IS NOT NULL` is the missing half: only
claim_remember_preparation sets it, and only the single path calls it (analyze
passes prepare_claim_token: None). Together the two columns mean "this row
claimed a preparation slot and never redeemed it", which is the state that
actually has nothing left to resume it.
Orphaned bulk and analyze preparations stay unswept. That is the pre-existing
behaviour, left alone deliberately rather than guessed at: neither path
persists anything that separates stranded from queued, so sweeping them needs
a durable marker they do not have yet.
Two tests seed exactly the row shape those endpoints write — pending, no
claim, no preparation, well past the TTL — and assert it survives.
Also repairs the build: RecallResponse gained `failed_writes` but two test
initializers in routes/recall.rs were not updated, so `cargo test` did not
compile on this branch at all (3 errors, present before this commit).
---
services/server/src/routes/recall.rs | 2 +
services/server/src/storage/db.rs | 64 ++++++++++++++++++++++++++--
2 files changed, 62 insertions(+), 4 deletions(-)
diff --git a/services/server/src/routes/recall.rs b/services/server/src/routes/recall.rs
index 6c6316014..2d0d141b3 100644
--- a/services/server/src/routes/recall.rs
+++ b/services/server/src/routes/recall.rs
@@ -464,6 +464,7 @@ mod tests {
results: vec![],
total: 0,
dropped_count: 3,
+ failed_writes: vec![],
};
let json = serde_json::to_value(&resp).unwrap();
assert_eq!(json["dropped_count"], 3);
@@ -475,6 +476,7 @@ mod tests {
results: vec![],
total: 0,
dropped_count: 0,
+ failed_writes: vec![],
};
let json = serde_json::to_value(&resp).unwrap();
// skip_serializing_if = "is_zero_usize" → field absent
diff --git a/services/server/src/storage/db.rs b/services/server/src/storage/db.rs
index ae4f736ad..16ae65c3b 100644
--- a/services/server/src/storage/db.rs
+++ b/services/server/src/storage/db.rs
@@ -2192,10 +2192,27 @@ impl VectorDb {
/// `pending` cannot be swept wholesale — a job that IS prepared sits at
/// `pending` until a wallet worker picks it up, which under upload backlog
/// is legitimately many minutes (`WALRUS_UPLOAD_PER_WALLET_CONCURRENCY`
- /// defaults to 1). `preparation_encrypted_b64 IS NULL` is what separates
- /// the two: it is written in the same statement that precedes
- /// `enqueue_wallet_job`, so its absence means the job never reached the
- /// queue.
+ /// defaults to 1). Two columns together say "this one is never coming":
+ ///
+ /// * `prepare_claimed_at IS NOT NULL` — the row belongs to the single
+ /// `remember` path, the only one that claims a preparation slot
+ /// (`claim_remember_preparation`). This is load-bearing, not decoration:
+ /// `/api/remember/bulk` and `/api/analyze` insert their rows directly and
+ /// never claim, so `preparation_encrypted_b64` is ALWAYS NULL for them,
+ /// healthy or not. Without this clause the sweep fails every bulk and
+ /// analyze write that waits out the TTL in a normal upload backlog —
+ /// killing paid work that was about to run and inviting the caller to
+ /// re-send it.
+ /// * `preparation_encrypted_b64 IS NULL` — that claim was never redeemed.
+ /// The column is written by the statement immediately before
+ /// `enqueue_wallet_job`, so its absence means the job never reached the
+ /// queue.
+ ///
+ /// Orphaned bulk and analyze preparations are therefore still not swept.
+ /// That is the pre-existing behaviour, deliberately left alone rather than
+ /// guessed at: neither path persists anything that distinguishes "stranded"
+ /// from "queued", so sweeping them needs a durable marker they do not yet
+ /// have.
///
/// Failing is the only option, not a choice. The row stores the SEAL
/// ciphertext, never the plaintext, so a job that died before encrypting
@@ -2251,6 +2268,7 @@ impl VectorDb {
prepare_claimed_at = NULL,
updated_at = NOW()
WHERE status = 'pending'
+ AND prepare_claimed_at IS NOT NULL
AND preparation_encrypted_b64 IS NULL
AND updated_at < NOW() - ($1 * INTERVAL '1 second')",
)
@@ -3944,6 +3962,44 @@ mod stale_sweep_tests {
assert_eq!(status_of(&db, &id).await, "failed");
}
+ /// `/api/remember/bulk` inserts its rows directly and never claims a
+ /// preparation slot, so `preparation_encrypted_b64` is ALWAYS NULL for a
+ /// bulk job — healthy or not. Keying the sweep on that column alone would
+ /// fail every bulk write that waits out the TTL behind a normal upload
+ /// backlog, destroying paid work that was about to run.
+ #[tokio::test]
+ async fn a_queued_bulk_job_is_never_swept() {
+ let db = test_db().await;
+ let owner = unique_owner("bulk");
+ // Exactly what remember_bulk writes: pending, no claim, no preparation.
+ let id = seed_job(&db, &owner, "pending", false, None, 900).await;
+
+ db.fail_stale_remember_jobs(Duration::from_secs(600))
+ .await
+ .expect("sweep");
+
+ assert_eq!(
+ status_of(&db, &id).await,
+ "pending",
+ "a bulk job waiting on the upload queue must survive the sweep",
+ );
+ }
+
+ /// `/api/analyze` inserts the same shape (`prepare_claim_token: None`), so
+ /// it needs the same protection.
+ #[tokio::test]
+ async fn a_queued_analyze_job_is_never_swept() {
+ let db = test_db().await;
+ let owner = unique_owner("analyze");
+ let id = seed_job(&db, &owner, "pending", false, None, 3600).await;
+
+ db.fail_stale_remember_jobs(Duration::from_secs(600))
+ .await
+ .expect("sweep");
+
+ assert_eq!(status_of(&db, &id).await, "pending");
+ }
+
/// A finished write is terminal and the sweeper must never touch it.
#[tokio::test]
async fn a_done_job_is_never_swept() {
From f7ee4ccf1a4410a8a7282f0040988b709cd03f37 Mon Sep 17 00:00:00 2001
From: Harry Phan
Date: Wed, 16 Sep 2026 13:41:30 +0700
Subject: [PATCH 024/132] fix: sanitize the recall failure report, and stop
telling bulk a retry is free
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit
Two defects with the same shape: a message written for the single-remember
path was reused where its guarantees do not hold.
SECURITY — the accepted-then-failed report attached to every recall read
`remember_jobs.error_msg` straight out of the table. Every other client-facing
view of that column (`GET /api/remember/:job_id`,
`POST /api/remember/bulk/status`) runs it through
`sanitize_job_error_for_client` first, which exists to do two things: replace
an infrastructure-funding failure with INFRA_JOB_ERROR_MESSAGE, and redact
long hex runs. Skipping it meant a WAL shortfall published the relayer's own
hot wallet and its exact balance — "Insufficient balance of 0x356a26…::wal::WAL
for owner 0x8d3c1f0a…c0d. Required: 64367730, Available: 10708877" — to every
tenant whose write landed on that wallet, on every recall, for 24 hours. It
also reads to an agent as "top that address up", which is the precise scam
confusion INFRA_JOB_ERROR_MESSAGE was written to prevent. The repo already
asserts this cannot happen (infra_wal_balance_failure_hides_relayer_wallet_address);
that assertion was simply never extended to this path. Sanitized at the recall
boundary rather than in storage, since `routes` is not reachable from the lib.
CORRECTNESS — `withAcceptDeadline` emitted one message for every caller,
ending "the SDK reuses the same idempotency key, so a retry attaches to the
existing job instead of queueing a second paid copy". True for
`POST /api/remember`, which carries a content-derived key. False for
`POST /api/remember/bulk`, which carries none: the handler mints a fresh uuid
per item and inserts with no conflict clause, so a retry is N more paid Walrus
blobs for the same N facts. The deadline makes that near-certain rather than
merely possible — `withDeadline` deliberately does not cancel the request, so
when it fires the relayer has usually accepted already. The advice is now
chosen per path, and bulk is told to check `memwal_recall` before re-sending
anything.
Tests pin both directions, because collapsing the two messages into one is how
this happened: bulk must never claim idempotency it does not have, and the
single path must keep saying a retry is safe.
---
.../mcp/__tests__/remember-deadline.test.ts | 26 +++++++++-
.../server/scripts/mcp/tools/remember-bulk.ts | 1 +
.../scripts/mcp/tools/remember-status.ts | 2 +
.../server/scripts/mcp/tools/remember-wait.ts | 38 ++++++++++++---
services/server/scripts/mcp/tools/remember.ts | 1 +
services/server/src/routes/recall.rs | 47 ++++++++++++++++++-
services/server/src/routes/remember.rs | 5 +-
services/server/src/storage/db.rs | 4 ++
services/server/src/types.rs | 15 ++++--
9 files changed, 125 insertions(+), 14 deletions(-)
diff --git a/services/server/scripts/mcp/__tests__/remember-deadline.test.ts b/services/server/scripts/mcp/__tests__/remember-deadline.test.ts
index 8adca54a4..5e3499115 100644
--- a/services/server/scripts/mcp/__tests__/remember-deadline.test.ts
+++ b/services/server/scripts/mcp/__tests__/remember-deadline.test.ts
@@ -121,7 +121,31 @@ test("memwal_remember_bulk cannot hang forever on a stalled accept", async (t) =
});
assert.equal((result as { isError?: boolean }).isError, true);
- assert.match(textOf(result), /did not accept/);
+ const text = textOf(result);
+ assert.match(text, /did not accept/);
+ // /api/remember/bulk carries NO idempotency key — the handler mints a
+ // fresh uuid per item — so the single path's "retrying is safe" line is a
+ // lie here, and an expensive one: withDeadline does not cancel the request,
+ // so the relayer has usually accepted by the time this fires.
+ assert.match(text, /Do NOT retry blindly/);
+ assert.match(text, /second time at full cost|SECOND time at full cost/i);
+ assert.doesNotMatch(
+ text,
+ /Retrying in this session is safe/,
+ "bulk must never claim idempotency it does not have",
+ );
+});
+
+test("memwal_remember's accept timeout still says a retry is safe", async (t) => {
+ // The single path DOES carry a content-derived idempotency key, so the
+ // opposite advice is correct there — and worth pinning, because collapsing
+ // both messages into one is exactly how the bulk bug happened.
+ const client = await clientFor(sessionThatHangs(), t);
+ const result = await client.callTool({
+ name: "memwal_remember",
+ arguments: { text: "a durable fact" },
+ });
+ assert.match(textOf(result), /Retrying in this session is safe/);
});
test("memwal_remember_status cannot hang forever on a stalled read", async (t) => {
diff --git a/services/server/scripts/mcp/tools/remember-bulk.ts b/services/server/scripts/mcp/tools/remember-bulk.ts
index 3c115fd9e..42c61b687 100644
--- a/services/server/scripts/mcp/tools/remember-bulk.ts
+++ b/services/server/scripts/mcp/tools/remember-bulk.ts
@@ -63,6 +63,7 @@ export function registerRememberBulkTool(
const accepted = await withAcceptDeadline(
session.memwal.rememberBulkAsync(items),
"memwal_remember_bulk batch",
+ { idempotent: false },
);
// Pair each job with its fact up front. Every later branch needs
diff --git a/services/server/scripts/mcp/tools/remember-status.ts b/services/server/scripts/mcp/tools/remember-status.ts
index abfc6a6bf..15c19b65a 100644
--- a/services/server/scripts/mcp/tools/remember-status.ts
+++ b/services/server/scripts/mcp/tools/remember-status.ts
@@ -113,6 +113,7 @@ export function registerRememberStatusTool(
const status = await withAcceptDeadline(
session.memwal.getRememberStatus(job_id),
"status read",
+ { idempotent: true },
);
if (status.status === "done") {
return saved(status.blob_id ?? "", status.namespace);
@@ -178,6 +179,7 @@ async function settleBatch(
await withAcceptDeadline(
session.memwal.getRememberBulkStatus(jobIds),
"batch status read",
+ { idempotent: true },
)
).results.map((r) => ({
id: r.job_id,
diff --git a/services/server/scripts/mcp/tools/remember-wait.ts b/services/server/scripts/mcp/tools/remember-wait.ts
index 1a1ce2e7b..01259c66c 100644
--- a/services/server/scripts/mcp/tools/remember-wait.ts
+++ b/services/server/scripts/mcp/tools/remember-wait.ts
@@ -269,16 +269,40 @@ export async function withDeadline(
}
}
-/** Bound the accept leg of a write. */
-export function withAcceptDeadline(work: Promise, what: string): Promise {
+/** Bound the accept leg of a write.
+ *
+ * `idempotent` is not cosmetic. `POST /api/remember` carries a content-derived
+ * idempotency_key, so a retry collapses onto the job already in flight.
+ * `POST /api/remember/bulk` carries none at all — the handler mints a fresh
+ * uuid per item and inserts with no conflict clause — so a retry there is N
+ * more paid Walrus blobs for the same N facts.
+ *
+ * That distinction decides what we may tell the agent, and the deadline makes
+ * it urgent rather than theoretical: `withDeadline` does not cancel the
+ * underlying request, so when it fires the relayer has usually accepted
+ * already. Inviting a blind retry on the bulk path is close to guaranteeing
+ * the duplicate.
+ */
+export function withAcceptDeadline(
+ work: Promise,
+ what: string,
+ opts: { idempotent: boolean },
+): Promise {
+ const shared =
+ `Walrus Memory did not accept the ${what} within ${ACCEPT_DEADLINE_MS / 1000}s — the ` +
+ `relayer is unreachable or not responding. The write may or may not have been queued, ` +
+ `so do NOT tell the user it was saved.`;
+
return withDeadline(
work,
ACCEPT_DEADLINE_MS,
- `Walrus Memory did not accept the ${what} within ${ACCEPT_DEADLINE_MS / 1000}s — the ` +
- `relayer is unreachable or not responding. The write may or may not have been ` +
- `queued, so do NOT tell the user it was saved. Retrying ${what} in this session ` +
- `is safe: the SDK reuses the same idempotency key until an accept succeeds, so a ` +
- `retry attaches to the existing job instead of queueing a second paid copy.`,
+ opts.idempotent
+ ? `${shared} Retrying in this session is safe: the write carries a content-derived ` +
+ `idempotency key until an accept succeeds, so a retry attaches to the existing job ` +
+ `instead of queueing a second paid copy.`
+ : `${shared} Do NOT retry blindly — this endpoint carries no idempotency key, so a ` +
+ `re-send stores every fact a SECOND time at full cost. Check with memwal_recall ` +
+ `first, and only re-send what is genuinely missing.`,
);
}
diff --git a/services/server/scripts/mcp/tools/remember.ts b/services/server/scripts/mcp/tools/remember.ts
index 74611cc23..5500ad629 100644
--- a/services/server/scripts/mcp/tools/remember.ts
+++ b/services/server/scripts/mcp/tools/remember.ts
@@ -62,6 +62,7 @@ export function registerRememberTool(
const accepted = await withAcceptDeadline(
session.memwal.rememberAsync(text, namespace),
"memwal_remember write",
+ { idempotent: true },
);
const pending = (waitedMs: number) => ({
diff --git a/services/server/src/routes/recall.rs b/services/server/src/routes/recall.rs
index 2d0d141b3..c706eaae3 100644
--- a/services/server/src/routes/recall.rs
+++ b/services/server/src/routes/recall.rs
@@ -39,7 +39,25 @@ async fn failed_writes_for(state: &AppState, owner: &str) -> Vec {
.recent_failed_remember_jobs(owner, FAILED_WRITE_REPORT_WINDOW, FAILED_WRITE_REPORT_LIMIT)
.await
{
- Ok(failed) => failed,
+ Ok(failed) => failed
+ .into_iter()
+ .map(|mut w| {
+ // Every other client-facing view of `remember_jobs.error_msg`
+ // runs it through this first — `GET /api/remember/:job_id` and
+ // `POST /api/remember/bulk/status` both do. Reading the column
+ // straight into a recall response skipped both of the
+ // sanitizer's jobs: swapping an infrastructure-funding failure
+ // for INFRA_JOB_ERROR_MESSAGE (whose text exists to stop a user
+ // reading "Insufficient balance ... for owner 0x…" as an
+ // instruction to top that address up), and redacting long hex
+ // runs so the relayer's own wallet never reaches a tenant.
+ //
+ // These rows are `status = 'failed'` by construction — the
+ // query selects on it — so the status argument is fixed.
+ w.error = super::remember::sanitize_job_error_for_client("failed", w.error);
+ w
+ })
+ .collect(),
Err(e) => {
tracing::warn!(
"recall: failed-write report unavailable for owner={}: {}",
@@ -458,6 +476,33 @@ mod tests {
// ── RecallResponse dropped_count serialization ───────────────
+ /// The failure report added for accepted-then-failed writes reads the same
+ /// `remember_jobs.error_msg` the job-status endpoints read, and reaches the
+ /// same untrusted caller — so it has to be sanitized the same way. It was
+ /// not, which put the relayer's own wallet address and balance shortfall
+ /// into every recall response for 24 hours after an infra failure.
+ ///
+ /// `infra_wal_balance_failure_hides_relayer_wallet_address` in
+ /// routes::remember pins this for `GET /api/remember/:job_id`; this pins
+ /// the same guarantee for the recall path.
+ #[test]
+ fn failed_write_report_hides_relayer_wallet_address() {
+ let raw = "walrus upload failed: Insufficient balance of \
+0x356a26eb9e012a68958082340d4c4116e7f55615ef27affcff209cf0ae544f59::wal::WAL for owner \
+0x8d3c1f0a9b2e4d6c7a5f8e1b0d4c9a2f3e6b7d8c1a0f9e2b3c4d5a6f7e8b9c0d. Required: 64367730, \
+Available: 10708877";
+
+ let out = crate::routes::remember::sanitize_job_error_for_client("failed", Some(raw.to_string()))
+ .expect("a failed job keeps an error");
+
+ // The operator's hot wallet and its shortfall are not the tenant's
+ // business, and reading them as "top this address up" is the exact
+ // confusion INFRA_JOB_ERROR_MESSAGE exists to prevent.
+ assert!(!out.contains("0x8d3c1f0a"), "wallet address leaked: {}", out);
+ assert!(!out.contains("Available"), "balance leaked: {}", out);
+ assert!(!out.contains("10708877"), "shortfall leaked: {}", out);
+ }
+
#[test]
fn recall_response_includes_dropped_count_when_nonzero() {
let resp = crate::types::RecallResponse {
diff --git a/services/server/src/routes/remember.rs b/services/server/src/routes/remember.rs
index 3311e33e6..d7f261a30 100644
--- a/services/server/src/routes/remember.rs
+++ b/services/server/src/routes/remember.rs
@@ -115,7 +115,10 @@ fn redact_hex_addresses(msg: &str) -> String {
/// current `status`. Infrastructure failures collapse to fixed copy, chosen by
/// whether the job has stopped retrying. Everything else keeps its text with
/// addresses redacted. The DB row is untouched.
-fn sanitize_job_error_for_client(status: &str, error_msg: Option) -> Option {
+pub(crate) fn sanitize_job_error_for_client(
+ status: &str,
+ error_msg: Option,
+) -> Option {
let msg = error_msg?;
if crate::jobs::WalletJobError::is_infrastructure_funding_error(&msg) {
return Some(if status == "failed" {
diff --git a/services/server/src/storage/db.rs b/services/server/src/storage/db.rs
index 16ae65c3b..9cf7d718d 100644
--- a/services/server/src/storage/db.rs
+++ b/services/server/src/storage/db.rs
@@ -1904,6 +1904,10 @@ impl VectorDb {
|(job_id, namespace, error, failed_at)| crate::types::FailedWrite {
job_id,
namespace,
+ // Raw here on purpose: `routes` is not reachable from the
+ // lib crate, and this is the storage layer. The caller
+ // (`routes::recall::failed_writes_for`) sanitizes before
+ // any of it reaches a client.
error,
failed_at: failed_at.to_rfc3339(),
},
diff --git a/services/server/src/types.rs b/services/server/src/types.rs
index 151e9a879..2e81dc368 100644
--- a/services/server/src/types.rs
+++ b/services/server/src/types.rs
@@ -1593,10 +1593,17 @@ pub struct FailedWrite {
/// match this against what it was told at the time.
pub job_id: String,
pub namespace: String,
- /// The relayer's own failure message, verbatim. Passed through rather than
- /// summarised: the distinction between (say) a SEAL outage and an
- /// exhausted upload budget is what tells a caller whether re-sending the
- /// fact is likely to work.
+ /// The failure message, after `sanitize_job_error_for_client` — the same
+ /// treatment `GET /api/remember/:job_id` and the bulk status endpoint give
+ /// it, and for the same two reasons. An infrastructure-funding failure is
+ /// replaced wholesale (its raw text names the relayer's own wallet and its
+ /// balance shortfall, which is neither the tenant's business nor safe to
+ /// show them: it reads as "top this address up"). Everything else keeps its
+ /// wording with long hex runs redacted.
+ ///
+ /// What survives is the part a caller can act on: whether this looks
+ /// transient or permanent, and so whether re-sending the fact is likely to
+ /// work.
#[serde(skip_serializing_if = "Option::is_none")]
pub error: Option,
/// When the job reached `failed`, RFC 3339.
From a4a94e410f41b7662acadaa0d34e31804107970d Mon Sep 17 00:00:00 2001
From: Harry Phan
Date: Wed, 16 Sep 2026 13:47:19 +0700
Subject: [PATCH 025/132] fix(sdk): keep the request deadline ref'd so it can
actually fire
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit
`deadlineSignal` unref'd its timer, which reads as tidy and is exactly
wrong for this timer: the deadline is the one thing a caller IS waiting
on. Unref'd, it stops the clock the moment nothing else holds the event
loop open, and the stalled request it exists to bound then hangs forever
— the bug the deadline was added to prevent.
CI caught it as six cancelled tests in request-timeout.test.mjs
("Promise resolution is still pending but the event loop has already
resolved"): the stub fetch is a bare promise with no socket behind it,
so the loop drained and the deadline never fired. A real socket normally
keeps the loop alive, which is why this survived manual testing — but
"normally" is not a bound, and any caller whose transport does not ref
the loop inherits the unbounded hang.
Both call sites already run `dispose()` in a `finally`, so a ref'd timer
cannot outlive its request either.
---
packages/sdk/src/memwal.ts | 7 +++++--
1 file changed, 5 insertions(+), 2 deletions(-)
diff --git a/packages/sdk/src/memwal.ts b/packages/sdk/src/memwal.ts
index f0b4e313d..7e7181f67 100644
--- a/packages/sdk/src/memwal.ts
+++ b/packages/sdk/src/memwal.ts
@@ -233,8 +233,11 @@ function deadlineSignal(
expired = true;
controller.abort();
}, ms);
- // Never hold a process open for a deadline nobody is waiting on.
- (timer as unknown as { unref?: () => void }).unref?.();
+ // Deliberately NOT unref'd. A deadline is the one timer somebody IS
+ // waiting on: unref'd, it stops firing the moment nothing else holds the
+ // loop open, and the stalled request it was meant to bound hangs forever
+ // instead. `dispose()` runs in the caller's `finally`, so the timer cannot
+ // outlive its request either way.
return {
signal: controller.signal,
From f205b6d084f52be417f61527830178290ca8ee3f Mon Sep 17 00:00:00 2001
From: Harry Phan
Date: Wed, 16 Sep 2026 13:47:24 +0700
Subject: [PATCH 026/132] test(server): seed a bulk job the way the routes
actually write it
`a_queued_bulk_job_is_never_swept` and `a_queued_analyze_job_is_never_swept`
failed against a sweep that is correct. The helper was the problem: it
stamped `prepare_claimed_at` on every seeded row, claim token or not, so
a row meant to stand in for a queued bulk write carried the one column
the orphan pass keys on and was duly failed.
`/api/remember/bulk` and `/api/analyze` insert neither the token nor the
timestamp; only the single-write path claims, and it writes both at once.
Stamping the timestamp alone is a shape the database never holds, so the
two tests were asserting against a fiction while the sweep they guard
went unexercised.
Seed the timestamp only alongside a token. Every row that must be swept
already passes one, so the orphan-preparation tests are unaffected.
Verified by `cargo check --tests`; the DB tests themselves need a
pgvector Postgres, which CI has and this machine does not.
---
services/server/src/storage/db.rs | 10 +++++++++-
1 file changed, 9 insertions(+), 1 deletion(-)
diff --git a/services/server/src/storage/db.rs b/services/server/src/storage/db.rs
index 9cf7d718d..11d2bb9e2 100644
--- a/services/server/src/storage/db.rs
+++ b/services/server/src/storage/db.rs
@@ -3818,6 +3818,13 @@ mod stale_sweep_tests {
}
/// Insert one remember job, aged by `age_secs`, optionally already prepared.
+ ///
+ /// `prepare_claimed_at` is stamped only alongside a claim token, because
+ /// that is the only way a row can reach the database: the single-write
+ /// path claims and stamps together, while `/api/remember/bulk` and
+ /// `/api/analyze` insert neither. Stamping it unconditionally would hand
+ /// every seeded row the one column the orphan sweep keys on, so a helper
+ /// detail — not the sweep — would decide what the tests below prove.
async fn seed_job(
db: &VectorDb,
owner: &str,
@@ -3832,7 +3839,8 @@ mod stale_sweep_tests {
(id, owner, namespace, status, preparation_encrypted_b64,
prepare_claim_token, prepare_claimed_at, created_at, updated_at)
VALUES ($1, $2, 'default', $3, $4, $5,
- NOW() - ($6 * INTERVAL '1 second'),
+ CASE WHEN $5::text IS NULL THEN NULL
+ ELSE NOW() - ($6 * INTERVAL '1 second') END,
NOW() - ($6 * INTERVAL '1 second'),
NOW() - ($6 * INTERVAL '1 second'))",
)
From 99a9bb3b79029f3ccd2ee6af8e211b415e0cd15f Mon Sep 17 00:00:00 2001
From: Harry Phan
Date: Wed, 16 Sep 2026 13:47:24 +0700
Subject: [PATCH 027/132] test(mcp): expect memwal_remember_status in
cold-start discovery
The cold-start tool list gained `memwal_remember_status` so a client that
keeps the first `tools/list` can resolve the job ids `memwal_remember`
and `memwal_remember_bulk` now return. The login-handoff test pins that
list exactly, and was never updated, so it failed on the tool being
present rather than on anything being wrong.
Title and annotations are copied from the server definition
(read-only, non-destructive), so the assertion keeps pinning the
metadata clients receive rather than just the name.
---
packages/mcp/test/login-handoff.test.mjs | 4 ++++
1 file changed, 4 insertions(+)
diff --git a/packages/mcp/test/login-handoff.test.mjs b/packages/mcp/test/login-handoff.test.mjs
index d5a3a200f..18340a0c0 100644
--- a/packages/mcp/test/login-handoff.test.mjs
+++ b/packages/mcp/test/login-handoff.test.mjs
@@ -183,6 +183,10 @@ test("auth-required mode picks up credentials mid-session without a restart", as
title: "Remember Multiple Facts",
annotations: { readOnlyHint: false, destructiveHint: false },
},
+ memwal_remember_status: {
+ title: "Check a Remember Job",
+ annotations: { readOnlyHint: true, destructiveHint: false },
+ },
memwal_recall: {
title: "Recall Memories",
annotations: { readOnlyHint: true, destructiveHint: false },
From e903f43c7b8935bc118eaf3082573f925ff68d73 Mon Sep 17 00:00:00 2001
From: Harry Phan
Date: Wed, 16 Sep 2026 13:58:46 +0700
Subject: [PATCH 028/132] fix(mcp): honour the relayer's retry_after instead of
dropping the fact
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit
Observed live against production while benchmarking: once the per-delegate-key
budget (60 weighted requests/minute) was spent, memwal_remember failed four
times in a row with
Tool error: Walrus Memory server error (429): {"error":"Rate limit exceeded",
"layer":"delegate_key","limit":"60 weighted-requests/min","retry_after_seconds":60}
and the facts were never written. Nothing retried, nothing honoured the
advertised cooldown, and nothing told the user a memory had been dropped. That
is the quietest failure this system has.
Fast-return makes it likelier rather than rarer: settling a batch adds requests
on top of the write, so an agent saving several facts in one turn spends the
budget faster than one that blocked.
A short cooldown is now absorbed (503 AUTH_UPSTREAM_UNAVAILABLE advises ~5s),
and a long one is reported. Sleeping out a 60s cooldown inside a tool call
would just be the hang this branch exists to remove, and the MCP client would
time out first — so the message names the wait, states plainly that the fact
was NOT saved, and points at the cheaper shape: one memwal_remember_bulk
instead of N memwal_remember calls, one memwal_remember_status(job_ids) instead
of N status calls.
Only rejections that provably never reached the handler are retried — 429 from
the limiter, AUTH_UPSTREAM_UNAVAILABLE from the delegate-key lookup. A 500 is
left alone: it could have been thrown after a write started, and
/api/remember/bulk has no idempotency key, so retrying it would store every
fact twice.
The absorb budget lives inside the accept deadline (8s of 15s) because
withAcceptDeadline wraps this; a test pins that ordering, since growing the
budget past the deadline would turn every absorbed retry into a spurious
"did not accept".
---
.../mcp/__tests__/remember-deadline.test.ts | 100 +++++++++++++++++-
.../server/scripts/mcp/tools/remember-bulk.ts | 9 +-
.../server/scripts/mcp/tools/remember-wait.ts | 93 +++++++++++++++-
services/server/scripts/mcp/tools/remember.ts | 6 +-
4 files changed, 204 insertions(+), 4 deletions(-)
diff --git a/services/server/scripts/mcp/__tests__/remember-deadline.test.ts b/services/server/scripts/mcp/__tests__/remember-deadline.test.ts
index 5e3499115..0907ea03f 100644
--- a/services/server/scripts/mcp/__tests__/remember-deadline.test.ts
+++ b/services/server/scripts/mcp/__tests__/remember-deadline.test.ts
@@ -12,7 +12,8 @@ import type { MemWalSession } from "../auth.js";
// above, so the module would read the real 15s default before the line that
// shortens it ever runs — and each tool test would then take 15 seconds.
const { createMcpServer } = await import("../server.js");
-const { ACCEPT_DEADLINE_MS, withDeadline } = await import("../tools/remember-wait.js");
+const { ACCEPT_DEADLINE_MS, withDeadline, MAX_ABSORBED_COOLDOWN_MS, DEFAULT_ACCEPT_DEADLINE_MS } =
+ await import("../tools/remember-wait.js");
/**
* The SDK's `signedRequest` aborts a request only when the caller passes a
@@ -216,3 +217,100 @@ test("the status wait ceiling stays under the MCP client's own deadline", async
`only ${DEFAULT_REQUEST_TIMEOUT_MSEC - max}ms of headroom before the client gives up`,
);
});
+
+/**
+ * Rate limiting is the quietest way this system loses a memory: the relayer
+ * answers 429 with `retry_after_seconds`, and before this the tool surfaced it
+ * as a bare `Tool error` while the fact was simply never written. Observed
+ * live against production during benchmarking, four calls in a row.
+ */
+
+function rejectsWith(status, serverCode, retryAfterSeconds, thenValue) {
+ let n = 0;
+ return async () => {
+ if (n++ === 0) {
+ const e = new Error(`stub ${status}`);
+ Object.assign(e, { status, serverCode, retryAfterSeconds });
+ throw e;
+ }
+ return thenValue;
+ };
+}
+
+function sessionRejecting(fn) {
+ return {
+ oauthScope: "memwal:read memwal:write",
+ namespace: "default",
+ memwal: { rememberAsync: fn, rememberBulkAsync: fn },
+ } as unknown as MemWalSession;
+}
+
+test("a short cooldown is absorbed and the write still lands", async (t) => {
+ // 503 AUTH_UPSTREAM_UNAVAILABLE advises ~5s; inside the absorb budget, so
+ // the agent should never see it.
+ //
+ // A sub-second cooldown here only because this file shortens
+ // ACCEPT_DEADLINE_MS to 150ms: the retry sleeps INSIDE the accept deadline,
+ // so the wait has to fit within it. In production that is 8s of absorb
+ // inside a 15s deadline — see the invariant pinned below.
+ const client = await clientFor(
+ sessionRejecting(rejectsWith(503, "AUTH_UPSTREAM_UNAVAILABLE", 0.05, { job_id: "job-1", status: "pending" })),
+ t,
+ );
+ const res = await client.callTool({ name: "memwal_remember", arguments: { text: "a fact" } });
+ assert.notEqual((res as { isError?: boolean }).isError, true);
+ assert.match(textOf(res), /job_id=job-1/);
+});
+
+test("a 60s rate-limit cooldown is reported, not slept through", async (t) => {
+ // Sleeping 60s inside a tool call is the hang this work exists to remove,
+ // and the MCP client would time out first.
+ const client = await clientFor(
+ sessionRejecting(rejectsWith(429, "Rate limit exceeded", 60, { job_id: "nope", status: "pending" })),
+ t,
+ );
+ const started = Date.now();
+ const res = await client.callTool({ name: "memwal_remember", arguments: { text: "a fact" } });
+ assert.ok(Date.now() - started < 5_000, "must not sit on a 60s cooldown");
+ assert.equal((res as { isError?: boolean }).isError, true);
+ const text = textOf(res);
+ // The agent must not tell the user this is being saved.
+ assert.match(text, /NOT\s+SAVED/i);
+ assert.match(text, /60s|about 60/);
+ // And it should say how to stop burning the budget.
+ assert.match(text, /memwal_remember_bulk/);
+});
+
+test("a rate-limited batch says the facts were not saved", async (t) => {
+ const client = await clientFor(
+ sessionRejecting(rejectsWith(429, "Rate limit exceeded", 60, { job_ids: [], total: 0, status: "x" })),
+ t,
+ );
+ const res = await client.callTool({ name: "memwal_remember_bulk", arguments: { facts: ["a", "b"] } });
+ assert.equal((res as { isError?: boolean }).isError, true);
+ assert.match(textOf(res), /NOT\s+SAVED/i);
+});
+
+test("a non-retryable error is not retried", async (t) => {
+ // A 500 could have been thrown after a write started; retrying bulk there
+ // would store every fact twice.
+ let calls = 0;
+ const client = await clientFor(
+ sessionRejecting(async () => { calls++; const e = new Error("boom"); Object.assign(e, { status: 500 }); throw e; }),
+ t,
+ );
+ const res = await client.callTool({ name: "memwal_remember_bulk", arguments: { facts: ["a"] } });
+ assert.equal((res as { isError?: boolean }).isError, true);
+ assert.equal(calls, 1, "a 500 must not be retried");
+});
+
+test("the absorb budget fits inside the accept deadline", () => {
+ // withAcceptDeadline wraps withRelayerRetry, so a cooldown we choose to sit
+ // on is spent against the accept deadline. If the absorb budget ever grew
+ // past it, every absorbed retry would be cut off mid-wait and surface as
+ // "did not accept" instead of succeeding.
+ assert.ok(
+ MAX_ABSORBED_COOLDOWN_MS < DEFAULT_ACCEPT_DEADLINE_MS,
+ `absorb ${MAX_ABSORBED_COOLDOWN_MS}ms must stay under the ${DEFAULT_ACCEPT_DEADLINE_MS}ms accept deadline`,
+ );
+});
diff --git a/services/server/scripts/mcp/tools/remember-bulk.ts b/services/server/scripts/mcp/tools/remember-bulk.ts
index 42c61b687..a758e9150 100644
--- a/services/server/scripts/mcp/tools/remember-bulk.ts
+++ b/services/server/scripts/mcp/tools/remember-bulk.ts
@@ -9,6 +9,7 @@ import {
pendingBulkMessage,
withAcceptDeadline,
withWaitDeadline,
+ withRelayerRetry,
} from "./remember-wait.js";
const REMEMBER_BULK_INPUT = {
@@ -61,7 +62,13 @@ export function registerRememberBulkTool(
// `memwal_remember` splits them: acceptance is the part that must
// succeed, the wait is a courtesy we cut short.
const accepted = await withAcceptDeadline(
- session.memwal.rememberBulkAsync(items),
+ // Safe to wrap despite bulk having no idempotency key: the
+ // retry only fires on rejections that never reached the
+ // handler, so no job row can exist to duplicate.
+ withRelayerRetry(
+ () => session.memwal.rememberBulkAsync(items),
+ "save these facts",
+ ),
"memwal_remember_bulk batch",
{ idempotent: false },
);
diff --git a/services/server/scripts/mcp/tools/remember-wait.ts b/services/server/scripts/mcp/tools/remember-wait.ts
index 01259c66c..90bb64b7e 100644
--- a/services/server/scripts/mcp/tools/remember-wait.ts
+++ b/services/server/scripts/mcp/tools/remember-wait.ts
@@ -202,7 +202,7 @@ export function pendingBulkMessage(
* every consumer, this is the tighter bound an interactive agent needs, and
* whichever is smaller fires first.
*/
-const DEFAULT_ACCEPT_DEADLINE_MS = 15_000;
+export const DEFAULT_ACCEPT_DEADLINE_MS = 15_000;
/**
* Read once, validated the same way as the wait budget: a typo must not
@@ -316,3 +316,94 @@ export function withWaitDeadline(work: Promise, budgetMs: number): Promise
`the fact. Call memwal_remember_status with the job_id to settle it.`,
);
}
+
+/**
+ * Longest we will sit on a relayer-advised cooldown before handing the problem
+ * back to the agent.
+ *
+ * The relayer answers a spent rate-limit budget with `retry_after_seconds: 60`.
+ * Sleeping that out inside a tool call is not a fix — it is the 60s hang this
+ * whole change set exists to remove, and the MCP client would time out first.
+ * So a short cooldown is absorbed and a long one is reported, with the wait
+ * named so the agent can come back rather than guess.
+ */
+export const MAX_ABSORBED_COOLDOWN_MS = 8_000;
+
+/** Attempts, including the first. Two retries is enough for a transient blip;
+ * more just delays an answer the agent could act on. */
+const RELAYER_RETRY_ATTEMPTS = 3;
+
+/**
+ * Errors where the request provably did NOT reach the handler, so re-sending
+ * cannot duplicate work.
+ *
+ * This matters most for `/api/remember/bulk`, which carries no idempotency key
+ * — a blind retry there would store every fact twice. Both cases below are
+ * rejections BEFORE any job row exists: 429 comes from the rate limiter, and
+ * AUTH_UPSTREAM_UNAVAILABLE from the delegate-key lookup failing open. Any
+ * other 5xx could have been thrown after a write started, so it is not retried.
+ */
+function isSafelyRetryable(err: unknown): boolean {
+ const e = err as { status?: number; serverCode?: string } | null;
+ if (e?.status === 429) return true;
+ return e?.status === 503 && e?.serverCode === "AUTH_UPSTREAM_UNAVAILABLE";
+}
+
+function advisedCooldownMs(err: unknown): number {
+ const secs = (err as { retryAfterSeconds?: number } | null)?.retryAfterSeconds;
+ return typeof secs === "number" && secs > 0 ? secs * 1000 : 1_000;
+}
+
+/**
+ * Honour the relayer's own `retry_after` instead of surfacing a raw 429.
+ *
+ * Observed against production: once the per-delegate-key budget (60 weighted
+ * requests/minute) is spent, `memwal_remember` fails with
+ * `Tool error: ... 429 ... retry_after_seconds: 60` and the fact is simply
+ * never written. Nothing retried, and nothing told the user their memory had
+ * been dropped — the worst failure this system has, because it is silent.
+ *
+ * Fast-return makes it likelier, not rarer: settling a batch adds requests on
+ * top of the write itself, so an agent saving several facts in one turn spends
+ * the budget faster than one that blocked.
+ */
+export async function withRelayerRetry(work: () => Promise, what: string): Promise {
+ let last: unknown;
+ for (let attempt = 1; attempt <= RELAYER_RETRY_ATTEMPTS; attempt++) {
+ try {
+ return await work();
+ } catch (err) {
+ last = err;
+ if (!isSafelyRetryable(err)) throw err;
+
+ const cooldown = advisedCooldownMs(err);
+ if (attempt === RELAYER_RETRY_ATTEMPTS || cooldown > MAX_ABSORBED_COOLDOWN_MS) {
+ const secs = Math.ceil(cooldown / 1000);
+ const limited = (err as { status?: number }).status === 429;
+ const e = new Error(
+ limited
+ ? `Walrus Memory rate limit reached while trying to ${what}. THE FACT WAS NOT ` +
+ `SAVED — tell the user it could not be stored rather than that it is being ` +
+ `saved. The limit is per delegate key and resets in about ${secs}s; retry ` +
+ `after that. To spend less of the budget, save several facts with one ` +
+ `memwal_remember_bulk call instead of repeated memwal_remember calls, and ` +
+ `settle a batch with a single memwal_remember_status(job_ids=[...]).`
+ : `Walrus Memory could not ${what}: the relayer's credential check is ` +
+ `temporarily unavailable. THE FACT WAS NOT SAVED. Retry in about ${secs}s.`,
+ );
+ e.name = "MemWalRelayerUnavailable";
+ (e as Error & { status?: number }).status = (err as { status?: number }).status;
+ throw e;
+ }
+
+ log.warn("remember.relayer_retry", {
+ what,
+ attempt,
+ status: (err as { status?: number }).status,
+ cooldownMs: cooldown,
+ });
+ await new Promise((r) => setTimeout(r, cooldown));
+ }
+ }
+ throw last;
+}
diff --git a/services/server/scripts/mcp/tools/remember.ts b/services/server/scripts/mcp/tools/remember.ts
index 5500ad629..23e5bd843 100644
--- a/services/server/scripts/mcp/tools/remember.ts
+++ b/services/server/scripts/mcp/tools/remember.ts
@@ -11,6 +11,7 @@ import {
pendingMessage,
withAcceptDeadline,
withWaitDeadline,
+ withRelayerRetry,
} from "./remember-wait.js";
const REMEMBER_INPUT = {
@@ -60,7 +61,10 @@ export function registerRememberTool(
// the wait need separate budgets: acceptance is the part that
// must succeed, the wait is a courtesy we cut short.
const accepted = await withAcceptDeadline(
- session.memwal.rememberAsync(text, namespace),
+ withRelayerRetry(
+ () => session.memwal.rememberAsync(text, namespace),
+ "save this fact",
+ ),
"memwal_remember write",
{ idempotent: true },
);
From 1d17fbfd714b4c9348d6e0a714d472a3b158d091 Mon Sep 17 00:00:00 2001
From: Harry Phan
Date: Wed, 16 Sep 2026 14:19:34 +0700
Subject: [PATCH 029/132] fix(relayer): let a failed job be re-prepared without
waiting out the claim TTL
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit
`memwal_remember` tells an agent to send a failed fact again. The derived
idempotency key collapses that retry onto the failed row, and
`claim_remember_preparation` refused the claim for 60s — while the handler
answered 202 ACCEPTED regardless. The caller was told the write was durably
queued while nothing at all was running, which is the one thing this branch's
whole pending-vs-saved contract exists to prevent.
The TTL protects a preparation that is still RUNNING. A job that reached
`failed` has none: its preparation either errored on its own or the stale
sweeper failed it and cleared its token. So `status = 'failed'` now bypasses
the age check.
That is safe because fencing is done by the token, not the clock. A claim
rotates `prepare_claim_token`, and a straggler's own write is
`WHERE ... prepare_claim_token = `, so it matches zero rows and returns
before `enqueue_wallet_job` — it cannot mint a paid blob for a job someone
else has re-prepared. The test drives exactly that: re-claim a just-failed job,
then watch the previous token's UPDATE affect nothing.
Second half: stop answering "pending" when no claim was taken. Losing the race
now means a concurrent retry won it, so the handler re-reads and reports what
that winner actually left behind instead of asserting a state it never reached.
A companion test pins that the TTL still does its real job — a `pending` row
claimed a moment ago keeps its claim.
---
services/server/src/routes/remember.rs | 115 ++++++++++++++++++++++++-
1 file changed, 113 insertions(+), 2 deletions(-)
diff --git a/services/server/src/routes/remember.rs b/services/server/src/routes/remember.rs
index d7f261a30..57c536a2a 100644
--- a/services/server/src/routes/remember.rs
+++ b/services/server/src/routes/remember.rs
@@ -821,12 +821,35 @@ pub async fn remember(
namespace_owned,
auth.public_key.clone(),
);
+ return Ok((
+ StatusCode::ACCEPTED,
+ Json(RememberAcceptedResponse {
+ job_id: existing_id,
+ status: "pending".to_string(),
+ }),
+ ));
}
+
+ // Losing the claim now means a concurrent retry took it, not
+ // that the TTL blocked us — `failed` rows are re-claimable
+ // immediately. Report whatever that winner left behind rather
+ // than asserting "pending" on its behalf: answering with a
+ // state we did not reach is what told callers a dead job was
+ // queued.
+ let actual: String = sqlx::query_scalar(
+ "SELECT status FROM remember_jobs WHERE id = $1",
+ )
+ .bind(&existing_id)
+ .fetch_optional(state.db.pool())
+ .await
+ .map_err(|e| AppError::Internal(format!("Failed to re-read job status: {}", e)))?
+ .unwrap_or_else(|| existing_status.clone());
+
return Ok((
StatusCode::ACCEPTED,
Json(RememberAcceptedResponse {
job_id: existing_id,
- status: "pending".to_string(),
+ status: actual,
}),
));
}
@@ -1023,6 +1046,25 @@ fn should_spawn_after_reset(rows_affected: u64) -> bool {
rows_affected == 1
}
+/// How long a preparation claim fences other claimants.
+///
+/// The TTL exists to stop a second request stealing a claim from a preparation
+/// that is still running. It must NOT apply to a job that already reached
+/// `failed`: that preparation is over — it either errored on its own or the
+/// stale sweeper failed it and cleared its token — so there is no live task to
+/// protect, and waiting out the TTL only blocks the retry the caller was just
+/// told to make.
+///
+/// That was not theoretical. `memwal_remember` tells an agent to send a failed
+/// fact again; the derived idempotency key collapses the retry onto the failed
+/// row; the claim was refused because it was less than 60s old; and the route
+/// answered 202 ACCEPTED anyway. The caller was told the write was queued while
+/// nothing whatsoever was running.
+///
+/// Letting a `failed` row be re-claimed immediately is safe because fencing is
+/// done by the TOKEN, not the clock: a new claim rotates `prepare_claim_token`,
+/// and any straggler's own UPDATE is `WHERE ... prepare_claim_token = `,
+/// so it matches zero rows and returns before `enqueue_wallet_job`.
const PREPARE_CLAIM_TTL_SECS: i64 = 60;
async fn claim_remember_preparation(
@@ -1031,7 +1073,7 @@ async fn claim_remember_preparation(
) -> Result, AppError> {
let token = uuid::Uuid::new_v4().to_string();
let claimed: Option = sqlx::query_scalar(
- "UPDATE remember_jobs SET prepare_claimed_at = NOW(), prepare_claim_token = $3, status = CASE WHEN status = 'failed' AND blob_id IS NULL THEN 'pending' ELSE status END, error_msg = CASE WHEN blob_id IS NULL THEN NULL ELSE error_msg END, updated_at = NOW() WHERE id = $1 AND blob_id IS NULL AND status IN ('pending', 'failed') AND (prepare_claimed_at IS NULL OR prepare_claimed_at < NOW() - make_interval(secs => $2)) RETURNING prepare_claim_token",
+ "UPDATE remember_jobs SET prepare_claimed_at = NOW(), prepare_claim_token = $3, status = CASE WHEN status = 'failed' AND blob_id IS NULL THEN 'pending' ELSE status END, error_msg = CASE WHEN blob_id IS NULL THEN NULL ELSE error_msg END, updated_at = NOW() WHERE id = $1 AND blob_id IS NULL AND status IN ('pending', 'failed') AND (prepare_claimed_at IS NULL OR prepare_claimed_at < NOW() - make_interval(secs => $2) OR status = 'failed') RETURNING prepare_claim_token",
)
.bind(job_id)
.bind(PREPARE_CLAIM_TTL_SECS)
@@ -1584,6 +1626,75 @@ mod tests {
assert_eq!(status, "done");
}
+ /// `memwal_remember` tells an agent to send a failed fact again. The
+ /// derived idempotency key collapses that retry onto the failed row, and
+ /// the claim used to be refused for 60s — while the route answered 202
+ /// ACCEPTED regardless, so the caller was told a write was queued when
+ /// nothing was running. A job that has finished failing has no live
+ /// preparation to fence.
+ #[tokio::test]
+ async fn a_failed_job_is_reclaimable_immediately() {
+ let pool = idem_test_pool().await;
+ let job_id = uuid::Uuid::new_v4().to_string();
+ sqlx::query(
+ "INSERT INTO remember_jobs (id, owner, namespace, status, prepare_claimed_at, prepare_claim_token)
+ VALUES ($1, '0xowner', 'ns', 'failed', NOW(), 'stale-token')",
+ )
+ .bind(&job_id)
+ .execute(&pool)
+ .await
+ .unwrap();
+
+ // Claimed one second ago — well inside PREPARE_CLAIM_TTL_SECS.
+ let claimed = claim_remember_preparation(&pool, &job_id).await.unwrap();
+ assert!(
+ claimed.is_some(),
+ "a failed job must be re-claimable without waiting out the TTL",
+ );
+
+ // And the retry is actually live, not merely reported as such.
+ let status: String = sqlx::query_scalar("SELECT status FROM remember_jobs WHERE id = $1")
+ .bind(&job_id)
+ .fetch_one(&pool)
+ .await
+ .unwrap();
+ assert_eq!(status, "pending");
+
+ // Rotating the token is what fences the previous attempt, so the old
+ // one can no longer write its preparation or reach the wallet queue.
+ let straggler = sqlx::query(
+ "UPDATE remember_jobs SET preparation_encrypted_b64 = 'late'
+ WHERE id = $1 AND prepare_claim_token = $2",
+ )
+ .bind(&job_id)
+ .bind("stale-token")
+ .execute(&pool)
+ .await
+ .unwrap();
+ assert_eq!(straggler.rows_affected(), 0, "the old claim must be fenced out");
+ }
+
+ /// The TTL still does its real job: a claim on a job that is genuinely
+ /// mid-preparation (`pending`, claimed just now) must not be stolen.
+ #[tokio::test]
+ async fn a_live_pending_claim_is_still_fenced() {
+ let pool = idem_test_pool().await;
+ let job_id = uuid::Uuid::new_v4().to_string();
+ sqlx::query(
+ "INSERT INTO remember_jobs (id, owner, namespace, status, prepare_claimed_at, prepare_claim_token)
+ VALUES ($1, '0xowner', 'ns', 'pending', NOW(), 'live-token')",
+ )
+ .bind(&job_id)
+ .execute(&pool)
+ .await
+ .unwrap();
+
+ assert!(
+ claim_remember_preparation(&pool, &job_id).await.unwrap().is_none(),
+ "a preparation still running must keep its claim",
+ );
+ }
+
#[tokio::test]
async fn initial_preparation_claim_is_fenced_after_reclaim() {
let pool = idem_test_pool().await;
From b793b3fe774f0cb9c431ffee99eb1b6800ddf0b9 Mon Sep 17 00:00:00 2001
From: Harry Phan
Date: Wed, 16 Sep 2026 14:23:15 +0700
Subject: [PATCH 030/132] perf(mcp): return memwal_analyze at accept, like the
remember tools
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit
analyze was the last tool still blocking to terminal, which left it the
slowest in the set by a wide margin: 37.0s measured against dev in the same
session where memwal_remember had dropped to 0.2s there. Its wait has the same
shape as bulk's — N Walrus writes, one upload per wallet — so there was no
reason for the answer to be shaped differently.
Extraction is still waited for. `analyze()` resolves once the LLM has run, so
the facts it found lead the reply, which is the half an agent can act on
straight away. Only the upload of those facts is handed back as job_ids, each
paired with its fact so a later partial failure is actionable.
Text that yields nothing says "Extracted 0 facts — nothing was saved" rather
than handing back an empty batch and a status tool to call about it.
6 new tests. Suite: 302 run, 269 pass, 33 fail — the same 33 sandbox
port-binding failures present on dev. tsc --noEmit clean.
---
.../mcp/__tests__/analyze-fast-return.test.ts | 146 ++++++++++++++++++
services/server/scripts/mcp/tools/analyze.ts | 125 ++++++++++++---
2 files changed, 252 insertions(+), 19 deletions(-)
create mode 100644 services/server/scripts/mcp/__tests__/analyze-fast-return.test.ts
diff --git a/services/server/scripts/mcp/__tests__/analyze-fast-return.test.ts b/services/server/scripts/mcp/__tests__/analyze-fast-return.test.ts
new file mode 100644
index 000000000..27ee0d118
--- /dev/null
+++ b/services/server/scripts/mcp/__tests__/analyze-fast-return.test.ts
@@ -0,0 +1,146 @@
+/**
+ * `memwal_analyze` returns once the facts are extracted and queued.
+ *
+ * It was the last tool still blocking to terminal after the two remember tools
+ * moved to a bounded wait, which left it the slowest in the set by a wide
+ * margin — 37.0s measured against dev in the same session where
+ * `memwal_remember` came back in 0.2s. The wait has the same shape as bulk's
+ * (N Walrus writes, one upload per wallet), so there was no reason for the
+ * answer to be shaped differently.
+ *
+ * What must stay true, and is what these tests pin: extraction is still waited
+ * for, because the facts are the part an agent can act on; the reply never
+ * reads as saved when the writes are still in flight; and every job_id comes
+ * back paired with the fact it carries, so a later partial failure is
+ * actionable.
+ */
+// A small non-zero wait so the bounded-wait branch is reachable at all. The
+// shipped default is 0 — the tool returns at accept and never polls — which
+// would make every assertion below about a partly landed batch unreachable.
+// Set before the dynamic import, because the budget is read once at load.
+process.env.MEMWAL_MCP_REMEMBER_WAIT_MS = "500";
+
+import assert from "node:assert/strict";
+import test, { type TestContext } from "node:test";
+import { Client } from "@modelcontextprotocol/sdk/client/index.js";
+import { InMemoryTransport } from "@modelcontextprotocol/sdk/inMemory.js";
+import type { MemWalSession } from "../auth.js";
+
+// Dynamic, not static: ESM hoists static imports above the assignment above,
+// so the module would read the real default before the line that changes it.
+const { createMcpServer } = await import("../server.js");
+
+const FACTS = ["User drinks oat milk", "User lives in Ho Chi Minh City"];
+
+function sessionWith(
+ opts: { states?: Array<"done" | "failed" | "timeout">; facts?: string[] } = {},
+ calls: string[] = [],
+): MemWalSession {
+ const facts = opts.facts ?? FACTS;
+ const jobIds = facts.map((_, i) => `analyze-job-${i + 1}`);
+ return {
+ oauthScope: "memwal:read memwal:write",
+ namespace: "default",
+ memwal: {
+ async analyze(text: string) {
+ calls.push(`analyze:${text.slice(0, 20)}`);
+ return {
+ job_ids: jobIds,
+ facts: facts.map((t) => ({ text: t })),
+ fact_count: facts.length,
+ status: "accepted",
+ owner: "0xowner",
+ };
+ },
+ async analyzeAndWait() {
+ calls.push("analyzeAndWait");
+ throw new Error("analyze must not block to terminal any more");
+ },
+ async waitForRememberJobs(ids: string[]) {
+ calls.push(`waitJobs:${ids.join(",")}`);
+ const states = opts.states ?? facts.map(() => "done" as const);
+ return {
+ results: states.map((status, i) => ({
+ id: jobIds[i],
+ blob_id: status === "done" ? `blob-${i + 1}` : "",
+ status,
+ namespace: "default",
+ })),
+ total: states.length,
+ succeeded: states.filter((s) => s === "done").length,
+ failed: states.filter((s) => s === "failed").length,
+ };
+ },
+ },
+ } as unknown as MemWalSession;
+}
+
+async function callAnalyze(
+ session: MemWalSession,
+ t: TestContext,
+ text = "a long note about the user",
+): Promise {
+ const [clientTransport, serverTransport] = InMemoryTransport.createLinkedPair();
+ const server = createMcpServer(session);
+ const client = new Client({ name: "analyze-test", version: "1.0.0" });
+ t.after(async () => {
+ await client.close();
+ await server.close();
+ });
+ await server.connect(serverTransport);
+ await client.connect(clientTransport);
+
+ const result = await client.callTool({
+ name: "memwal_analyze",
+ arguments: { text },
+ });
+ return (result as { content: Array<{ text: string }> }).content
+ .map((c) => c.text)
+ .join("\n");
+}
+
+test("analyze never blocks to terminal", async (t) => {
+ const calls: string[] = [];
+ await callAnalyze(sessionWith({}, calls), t);
+ assert.ok(
+ !calls.includes("analyzeAndWait"),
+ `took the blocking path: ${calls.join(", ")}`,
+ );
+ assert.ok(calls.some((c) => c.startsWith("analyze:")), calls.join(", "));
+});
+
+test("the extracted facts come back even though the writes have not landed", async (t) => {
+ // Extraction is the half an agent can use straight away. Handing back only
+ // job_ids would make the tool useless until a second call.
+ const text = await callAnalyze(sessionWith(), t);
+ for (const fact of FACTS) assert.match(text, new RegExp(fact));
+});
+
+test("an in-flight analyze does not read as saved", async (t) => {
+ const text = await callAnalyze(sessionWith({ states: ["timeout", "timeout"] }), t);
+ assert.match(text, /NOT SAVED YET|ACCEPTED, NOT YET SAVED/);
+ assert.match(text, /memwal_remember_status/);
+ assert.doesNotMatch(text, /^Saved to Walrus Memory/m);
+});
+
+test("every job_id is paired with the fact it carries", async (t) => {
+ // "one of these failed" is only actionable if the agent can tell which.
+ const text = await callAnalyze(sessionWith({ states: ["timeout", "timeout"] }), t);
+ assert.match(text, /analyze-job-1/);
+ assert.match(text, /analyze-job-2/);
+ assert.match(text, /analyze-job-1 — User drinks oat milk/);
+});
+
+test("a partly landed batch reports both halves", async (t) => {
+ const text = await callAnalyze(sessionWith({ states: ["done", "timeout"] }), t);
+ assert.match(text, /blob-1/, "the landed write shows its blob_id");
+ assert.match(text, /analyze-job-2/, "the straggler shows its job_id");
+ assert.match(text, /memwal_remember_status/, "and how to settle it");
+});
+
+test("text with nothing worth saving says so instead of handing back an empty batch", async (t) => {
+ const text = await callAnalyze(sessionWith({ facts: [] }), t);
+ assert.match(text, /Extracted 0 facts/);
+ assert.match(text, /nothing was saved/);
+ assert.doesNotMatch(text, /memwal_remember_status/);
+});
diff --git a/services/server/scripts/mcp/tools/analyze.ts b/services/server/scripts/mcp/tools/analyze.ts
index d4cdaaf85..c85f3530f 100644
--- a/services/server/scripts/mcp/tools/analyze.ts
+++ b/services/server/scripts/mcp/tools/analyze.ts
@@ -3,7 +3,14 @@ import type { McpServer } from "@modelcontextprotocol/sdk/server/mcp.js";
import type { MemWalSession } from "../auth.js";
import { TOOL_METADATA } from "./annotations.js";
import { wrapTool, explorerFooter } from "./util.js";
-import { REMEMBER_POLL_INTERVAL_MS } from "./remember-wait.js";
+import {
+ REMEMBER_POLL_INTERVAL_MS,
+ REMEMBER_WAIT_MS,
+ pendingBulkMessage,
+ withAcceptDeadline,
+ withRelayerRetry,
+ withWaitDeadline,
+} from "./remember-wait.js";
const ANALYZE_INPUT = {
text: z
@@ -21,9 +28,22 @@ const ANALYZE_INPUT = {
} as const;
/**
- * memwal_analyze — let Walrus Memory's LLM extract distinct facts from a piece of
- * text and persist each as its own memory. Resolves only after all extracted
- * facts have been written end-to-end (or the call times out).
+ * memwal_analyze — let Walrus Memory's LLM extract distinct facts from a piece
+ * of text and persist each as its own memory.
+ *
+ * Returns once the relayer has extracted the facts and durably queued a write
+ * for each, the same contract as `memwal_remember_bulk`.
+ *
+ * Blocking to terminal was left in place while the two remember tools moved to
+ * a bounded wait, which made this the slowest tool in the set by a wide
+ * margin: measured at 37.0s against dev after `memwal_remember` had dropped to
+ * 0.2s there. The shape of the wait is the same as bulk's — N Walrus writes,
+ * one upload per wallet — so there was no reason for the answer to be shaped
+ * differently.
+ *
+ * Extraction itself is worth waiting for, and this still does: `analyze()`
+ * resolves after the LLM has run, so the facts it found are in the reply.
+ * Only the upload of those facts is handed back as job_ids.
*/
export function registerAnalyzeTool(
server: McpServer,
@@ -34,34 +54,101 @@ export function registerAnalyzeTool(
{
...TOOL_METADATA.memwal_analyze,
description:
- "Extract memorable facts from a longer passage of text (preferences, habits, biographical info, constraints) and save each as a separate Walrus Memory memory. Use this when you want MemWal's LLM to split the facts out of a transcript or notes for you; if you already know the exact facts, use memwal_remember or memwal_remember_bulk instead.",
+ "Extract memorable facts from a longer passage of text (preferences, habits, biographical info, constraints) and save each as a separate Walrus Memory memory. Use this when you want MemWal's LLM to split the facts out of a transcript or notes for you; if you already know the exact facts, use memwal_remember or memwal_remember_bulk instead. The extracted facts come back immediately; if the result says the writes are still in flight it carries job_ids — confirm them with memwal_remember_status rather than telling the user they are saved.",
inputSchema: ANALYZE_INPUT,
},
wrapTool<{ text: string; namespace?: string }>(session, "memwal_analyze", async ({ text, namespace }) => {
- const result = await session.memwal.analyzeAndWait(text, namespace, {
- timeoutMs: 180_000,
- // Same reasoning as `memwal_remember_bulk`: this tool blocks to
- // terminal, so the SDK's 10s backoff ceiling is dead time added
- // to every extraction. The budget here is the longest of the
- // three, which is exactly where that ceiling hurts most.
- pollIntervalMs: REMEMBER_POLL_INTERVAL_MS,
+ // `analyze` (not `analyzeAndWait`) returns once extraction is done
+ // and every fact has a queued job, which is the point this tool can
+ // usefully answer at.
+ const accepted = await withAcceptDeadline(
+ // Same reasoning as bulk: the retry only fires on rejections
+ // that never reached the handler, so no job row can exist yet
+ // to duplicate.
+ withRelayerRetry(
+ () => session.memwal.analyze(text, namespace),
+ "analyze this text",
+ ),
+ "memwal_analyze extraction",
+ { idempotent: false },
+ );
+
+ const facts = accepted.facts ?? [];
+ // Nothing to wait on, and nothing to confirm later. Say so plainly
+ // rather than handing back an empty job list.
+ if (accepted.job_ids.length === 0) {
+ return {
+ content: [
+ {
+ type: "text" as const,
+ text: `Extracted 0 facts from that text — nothing was saved.`,
+ },
+ ],
+ };
+ }
+
+ const entries = accepted.job_ids.map((jobId, i) => ({
+ jobId,
+ text: facts[i]?.text ?? "",
+ }));
+
+ // The extraction result is the part of this call an agent can act
+ // on immediately, so it leads — the write status follows it.
+ const extracted = `Extracted ${facts.length} fact(s):\n${entries
+ .map((e, i) => `${i + 1}. ${e.text || "(unknown fact)"}`)
+ .join("\n")}`;
+
+ const pending = (waitedMs: number) => ({
+ content: [
+ {
+ type: "text" as const,
+ text: `${extracted}\n\n${pendingBulkMessage(entries, waitedMs)}`,
+ },
+ ],
});
+
+ if (REMEMBER_WAIT_MS === 0) return pending(0);
+
+ const startedAt = Date.now();
+ const namespaces = entries.map(
+ () => namespace ?? session.namespace ?? "default"
+ );
+ const result = await withWaitDeadline(
+ session.memwal.waitForRememberJobs(accepted.job_ids, namespaces, {
+ timeoutMs: REMEMBER_WAIT_MS,
+ pollIntervalMs: REMEMBER_POLL_INTERVAL_MS,
+ }),
+ REMEMBER_WAIT_MS,
+ );
+ const waitedMs = Date.now() - startedAt;
+
+ const unfinished = result.results.filter((r) => r.status === "timeout");
+ if (unfinished.length === result.results.length) return pending(waitedMs);
+
const lines = result.results.map(
(r, i) =>
`${i + 1}. [${r.status}]${r.blob_id ? ` blob_id=${r.blob_id}` : ""} ${
- result.facts[i]?.text ?? "(unknown fact)"
+ entries[i]?.text || "(unknown fact)"
}`
);
- const summary = `Extracted ${result.facts.length} fact(s) — succeeded=${result.succeeded} failed=${result.failed}`;
+ const summary = `Extracted ${facts.length} fact(s) — succeeded=${result.succeeded} failed=${result.failed}`;
+ const stragglers =
+ unfinished.length > 0
+ ? `\n\n${pendingBulkMessage(
+ result.results.flatMap((r, i) =>
+ r.status === "timeout"
+ ? [{ jobId: r.id, text: entries[i]?.text ?? "" }]
+ : [],
+ ),
+ waitedMs,
+ )}`
+ : "";
const footer = result.succeeded > 0 ? `\n\n${explorerFooter()}` : "";
return {
content: [
{
- type: "text",
- text:
- lines.length > 0
- ? `${summary}\n\n${lines.join("\n")}${footer}`
- : `${summary}${footer}`,
+ type: "text" as const,
+ text: `${summary}\n\n${lines.join("\n")}${stragglers}${footer}`,
},
],
};
From 8f923d0d121db1e2a94f7cc187e09e8022c38158 Mon Sep 17 00:00:00 2001
From: Harry Phan
Date: Wed, 16 Sep 2026 14:26:33 +0700
Subject: [PATCH 031/132] fix(relayer): report a failed write once, and let the
transfer arm mark its wallet busy
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit
Two findings from the completeness pass over this branch.
REPORT ONCE. The accepted-then-failed report attached to recall selected on
owner + status + a 24h window with nothing recording that a row had been
shown. So the same failures rode along on every recall for a full day while
the message told the agent to send those facts again — a compliant agent
re-sent, the original row stayed `failed` and in-window, and the next recall
asked for the same thing. Every pass was another paid Walrus write, and
`collapseDuplicates` on the read side hid the pile-up from the user.
Rows are now stamped in the same statement that returns them, so the claim and
the report cannot separate; two concurrent recalls cannot both surface the same
failure (`FOR UPDATE SKIP LOCKED`). Migration 021 adds the column plus a
partial index matching the query — failed and unreported only, so it stays
small on a table that is never pruned.
MARK THE WALLET BUSY. `least_loaded_index` answers "is this key signing right
now", and only `UploadAndTransfer` was telling it. `SetMetadataAndTransfer`
signs on the very same wallet and held no guard, so a key mid-transfer read as
idle — and join-shortest-queue then steered new uploads onto it *because* it
looked free, queueing them behind the transaction. That is precisely the
failure the least-loaded change was written to remove, so the arm that
regressed it is the arm that had to be fixed. It guards
`enqueued_wallet_index`, not a fresh pick, because the operation must run on
the key that already owns the blob object.
`FinalizeUploadedBlob` deliberately stays unguarded: it only inserts the vector
row, signs nothing, and holding a wallet slot through a DB write would make an
idle key look busy.
No new unit test for the guard: the pool's selection logic is already covered
(a_busy_wallet_is_skipped_for_an_idle_one), and what broke was a missing call
site, which only an integration test over the job arms could catch.
---
.../021_failed_write_report_ack.sql | 21 ++++++++++++
services/server/src/jobs.rs | 13 +++++++
services/server/src/storage/db.rs | 34 ++++++++++++++++---
3 files changed, 63 insertions(+), 5 deletions(-)
create mode 100644 services/server/migrations/021_failed_write_report_ack.sql
diff --git a/services/server/migrations/021_failed_write_report_ack.sql b/services/server/migrations/021_failed_write_report_ack.sql
new file mode 100644
index 000000000..2f91ef1c4
--- /dev/null
+++ b/services/server/migrations/021_failed_write_report_ack.sql
@@ -0,0 +1,21 @@
+-- Acknowledge an accepted-then-failed write once it has been reported.
+--
+-- `POST /api/recall` attaches recently-failed writes so a user learns a fact
+-- was silently lost. Without a marker the same rows ride along on EVERY recall
+-- for the whole 24h window, and the text tells the agent to send them again —
+-- so an agent that complies re-sends, the original row is still `failed` and
+-- still in-window, and the next recall asks for the same thing. Each pass is a
+-- fresh paid Walrus write.
+--
+-- Stamped when a report goes out, then filtered on, so each failure is
+-- surfaced once. NULL means "not yet reported", which is what every existing
+-- row should be.
+ALTER TABLE remember_jobs
+ ADD COLUMN IF NOT EXISTS failure_reported_at TIMESTAMPTZ;
+
+-- The report query is on the hot recall path: owner + status + window, newest
+-- first, unreported only. Partial so the index stays small — it only ever
+-- serves rows that are failed and still unreported.
+CREATE INDEX IF NOT EXISTS remember_jobs_unreported_failures_idx
+ ON remember_jobs (owner, updated_at DESC)
+ WHERE status = 'failed' AND failure_reported_at IS NULL;
diff --git a/services/server/src/jobs.rs b/services/server/src/jobs.rs
index a0073709a..687df6f89 100644
--- a/services/server/src/jobs.rs
+++ b/services/server/src/jobs.rs
@@ -622,6 +622,19 @@ pub(crate) async fn execute_wallet_job(
policy_package_id,
end_epoch,
} => {
+ // Mark the wallet busy for this transaction too. `least_loaded_index`
+ // answers "is this key signing right now", and only the upload arm
+ // was telling it — so a metadata+transfer, which signs on the very
+ // same wallet, read as idle. A concurrently-enqueued upload would
+ // then pick that key precisely because it looked free, and queue
+ // behind the transaction anyway. That is the failure join-shortest-
+ // queue exists to avoid, and it showed up as the pool converging on
+ // whichever key was mid-transfer.
+ //
+ // `enqueued_wallet_index` rather than a fresh pick: this operation
+ // must run on the key that already owns the blob object.
+ let _wallet_slot = state.key_pool.begin_attempt(enqueued_wallet_index);
+
let result = execute_set_metadata_and_transfer(
state,
enqueued_wallet_index,
diff --git a/services/server/src/storage/db.rs b/services/server/src/storage/db.rs
index 11d2bb9e2..726bebc15 100644
--- a/services/server/src/storage/db.rs
+++ b/services/server/src/storage/db.rs
@@ -1854,6 +1854,18 @@ impl VectorDb {
/// deliberate — suppressing a report after one sighting would put the
/// notice back on the caller remembering to act on it, which is the
/// failure mode this whole path exists to remove.
+ /// Accepted-then-failed writes this owner has not been told about yet.
+ ///
+ /// Reporting is one-shot by construction. The rows are stamped in the same
+ /// statement that returns them, so the next recall sees them acknowledged
+ /// and stays quiet. Without that the same failures rode along on every
+ /// recall for the full window while the message told the agent to send the
+ /// facts again — so a compliant agent re-sent, the original row stayed
+ /// `failed` and in-window, and the next recall asked for the same thing.
+ /// Every pass was another paid Walrus write.
+ ///
+ /// `FOR UPDATE SKIP LOCKED` keeps two concurrent recalls from claiming the
+ /// same row and both reporting it.
pub async fn recent_failed_remember_jobs(
&self,
owner: &str,
@@ -1870,11 +1882,23 @@ impl VectorDb {
chrono::DateTime,
),
>(
- "SELECT id, namespace, error_msg, updated_at
- FROM remember_jobs
- WHERE owner = $1 AND status = 'failed' AND updated_at >= $2
- ORDER BY updated_at DESC
- LIMIT $3",
+ // Claim-and-return in one statement: the UPDATE stamps the rows
+ // it is about to hand back, so a second recall — or a concurrent
+ // one in another session — cannot report the same failure again.
+ // Doing it as two statements would leave a window where both see
+ // the row unreported and both tell the agent to re-send it.
+ "UPDATE remember_jobs SET failure_reported_at = NOW()
+ WHERE id IN (
+ SELECT id FROM remember_jobs
+ WHERE owner = $1
+ AND status = 'failed'
+ AND failure_reported_at IS NULL
+ AND updated_at >= $2
+ ORDER BY updated_at DESC
+ LIMIT $3
+ FOR UPDATE SKIP LOCKED
+ )
+ RETURNING id, namespace, error_msg, updated_at",
)
.bind(owner)
.bind(chrono::Utc::now() - chrono::Duration::from_std(window).unwrap_or_default())
From eb8d35dfcc89f59b22e9a9ecee251c5928a3bf71 Mon Sep 17 00:00:00 2001
From: Harry Phan
Date: Wed, 16 Sep 2026 14:37:32 +0700
Subject: [PATCH 032/132] fix(relayer): back off on the durable upload exit
too, not just the legacy one
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit
The backoff added earlier covered one of the upload job's three retryable
exits. Dev proved it was the wrong one: a burst of 12 concurrent writes
produced
14:33:47 job_id=423e8f6f… durable Walrus upload request failed:
error sending request for url (localhost:9000/walrus/upload-step-v3)
classification=transient retryable=true
14:33:47 selected wallet for attempt: attempt=2/5
— the retry starting in the same second, with no `backing off` line anywhere
in 24h of dev logs. The durable upload path returns through
`execute_upload_and_transfer_locked`, which has its own exit and so needed its
own spacing; only the legacy sidecar exit had been given one.
The contrast in the same trace is the point: the lock-defer path, which has
had a backoff all along, spaced its retry correctly —
14:33:54 deferring upload (attempt 1/5)
14:33:56 (retry) ← 2.0s, matching backoff_duration(1)
Still not covered: `insert_vector_and_mark_remember_done` takes no
`attempt_info`, so its exit cannot space its own retry without threading the
attempt through. Left alone rather than widened blindly — it fires after the
blob is already minted, which is a different failure shape.
---
services/server/src/jobs.rs | 16 ++++++++++++++++
1 file changed, 16 insertions(+)
diff --git a/services/server/src/jobs.rs b/services/server/src/jobs.rs
index 687df6f89..d60999d87 100644
--- a/services/server/src/jobs.rs
+++ b/services/server/src/jobs.rs
@@ -1804,6 +1804,22 @@ async fn execute_upload_and_transfer_locked(
err.kind(),
!err.aborts_retries()
);
+ // The durable upload has its own exit, so it needs its own
+ // spacing — this is the path a real retry actually took.
+ // Observed on dev: `durable Walrus upload request failed`
+ // classified transient at attempt 1/5, with the next attempt
+ // starting in the same second because only the legacy exit
+ // below had been given a backoff.
+ if let Some(delay) = upload_retry_backoff(&err, attempt_info) {
+ tracing::info!(
+ "[wallet-job:upload] job_id={} backing off {:?} before attempt {}/{}",
+ jid,
+ delay,
+ attempt_info.current + 1,
+ attempt_info.max,
+ );
+ tokio::time::sleep(delay).await;
+ }
Err(err)
}
};
From b551a9fc835f3a794993fd0bfdfdb6c02f1cc3a1 Mon Sep 17 00:00:00 2001
From: Harry Phan
Date: Wed, 16 Sep 2026 14:39:15 +0700
Subject: [PATCH 033/132] fix(mcp): keep the sidecar's deadline timer ref'd too
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit
a4a94e41 corrected this in the SDK; the sidecar's `withDeadline` had the same
line and was missed. It matters more here, not less: the sidecar ships inside
the relayer image, while the SDK copy sits behind a pinned published version,
so this is the deadline actually running in production.
The reasoning was wrong in the same way. `unref()` is right for a background
sweeper nobody awaits — which is where it was copied from — but a request
deadline is the one timer somebody IS waiting on. Unref'd it stops firing as
soon as nothing else holds the loop open, so the stalled request it exists to
bound hangs forever, which is precisely the failure it was added to prevent.
`dispose()` still runs in the caller's `finally`, so the timer cannot outlive
its request either way.
---
services/server/scripts/mcp/tools/remember-wait.ts | 8 ++++++--
1 file changed, 6 insertions(+), 2 deletions(-)
diff --git a/services/server/scripts/mcp/tools/remember-wait.ts b/services/server/scripts/mcp/tools/remember-wait.ts
index 90bb64b7e..b67a55388 100644
--- a/services/server/scripts/mcp/tools/remember-wait.ts
+++ b/services/server/scripts/mcp/tools/remember-wait.ts
@@ -260,8 +260,12 @@ export async function withDeadline(
work,
new Promise((_, reject) => {
timer = setTimeout(() => reject(new DeadlineExceededError(message)), ms);
- // Never hold the process open for a deadline nobody is waiting on.
- timer.unref?.();
+ // Deliberately NOT unref'd — the same correction a4a94e41 made
+ // in the SDK, which this had copied. A deadline is the one
+ // timer somebody IS waiting on: unref'd it stops firing the
+ // moment nothing else holds the loop open, so the stalled
+ // request it exists to bound hangs forever instead. The
+ // `finally` below clears it, so it cannot outlive its work.
}),
]);
} finally {
From da91fed7b07cf9bd66a485511a29409499fe6de1 Mon Sep 17 00:00:00 2001
From: Harry Phan
Date: Wed, 16 Sep 2026 14:43:38 +0700
Subject: [PATCH 034/132] fix(mcp,sdk): restore the blocking default, and ship
the version bumps by hand
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit
Review items on this head.
KEEP THE WAIT. `DEFAULT_REMEMBER_WAIT_MS` was 0, so `memwal_remember` returned
at accept with no blob_id. D1 already decided the opposite: a result means the
fact landed. Returning at accept is a product change on its own ticket, and
the poll-cadence work it was bundled with belongs to #902. The default is the
full 90s ceiling again; accept-and-continue stays reachable at
`MEMWAL_MCP_REMEMBER_WAIT_MS=0` for an operator who chooses it, and the two
suites that cover that path now opt in explicitly instead of inheriting it.
A budget between zero and the real completion time stays the setting to avoid
— against a 30-75s spread it pays the wait and still returns pending — so the
default is the ceiling rather than something in between.
MAKE THE IDEMPOTENCY CLAIM TRUE. The accept-timeout message told callers a
retry was safe "because the write carries a content-derived idempotency key".
That key exists in packages/sdk, which is NOT what runs here: the sidecar
installs published 0.1.7, whose rememberAsync mints a crypto.randomUUID() and
caches it per client instance — and a fresh client is built per transport
session, so a retry after a reconnect stored the fact a second time at full
cost. Rather than soften the wording, the tool now computes the key itself and
passes it, which 0.1.7 accepts. The promise is true in the version actually
deployed. Bulk still says the opposite, because /api/remember/bulk takes no
key at all.
DO NOT ASK FOR TEXT WE DO NOT HAVE. The recall failure report told the agent to
send the fact again, but the relayer stores only SEAL ciphertext, so a failed
write's wording is gone. It now says so and asks the user to restate it, rather
than inviting the agent to invent it. (The repeat-every-recall half was fixed
in 8f923d0d — the report is one-shot.)
RELEASE BY HAND. The MCP contract, tools and transport changed under published
0.0.13, and the SDK's poll, timeout and idempotency behaviour under 0.1.7. Both
bumped manually: 0.0.14 and 0.1.8, dual changelogs, every manifest the release
verifier checks, and the verifier's own pins. The two changesets are deleted —
this repo releases these by hand, so leaving them would have double-bumped.
`node scripts/verify-manual-sdk-release.mjs` passes for all four packages,
including the plugin npx pins it caught me missing in .mcp.json,
.cursor-mcp.json and .codex-mcp.json.
---
...remember-poll-cadence-and-replay-dedupe.md | 13 ----
.changeset/sdk-bound-every-relayer-request.md | 11 ---
.claude-plugin/marketplace.json | 2 +-
.cursor-plugin/marketplace.json | 2 +-
docs/mcp/changelog.mdx | 15 ++++
docs/sdk/changelog.mdx | 8 ++
packages/mcp/CHANGELOG.md | 15 ++++
packages/mcp/package.json | 4 +-
.../mcp/plugin/.claude-plugin/plugin.json | 26 ++++---
packages/mcp/plugin/.codex-mcp.json | 2 +-
packages/mcp/plugin/.codex-plugin/plugin.json | 63 +++++++++-------
packages/mcp/plugin/.cursor-mcp.json | 2 +-
.../mcp/plugin/.cursor-plugin/plugin.json | 27 ++++---
packages/mcp/plugin/.mcp.json | 2 +-
packages/mcp/plugin/plugin.json | 28 ++++---
packages/sdk/CHANGELOG.md | 8 ++
packages/sdk/package.json | 4 +-
scripts/verify-manual-sdk-release.mjs | 4 +-
.../remember-bulk-fast-return.test.ts | 7 +-
.../mcp/__tests__/remember-deadline.test.ts | 5 +-
.../__tests__/remember-fast-return.test.ts | 49 +++++++-----
services/server/scripts/mcp/tools/recall.ts | 5 +-
.../server/scripts/mcp/tools/remember-wait.ts | 74 ++++++++++++++-----
services/server/scripts/mcp/tools/remember.ts | 9 ++-
24 files changed, 254 insertions(+), 131 deletions(-)
delete mode 100644 .changeset/remember-poll-cadence-and-replay-dedupe.md
delete mode 100644 .changeset/sdk-bound-every-relayer-request.md
diff --git a/.changeset/remember-poll-cadence-and-replay-dedupe.md b/.changeset/remember-poll-cadence-and-replay-dedupe.md
deleted file mode 100644
index 34d7ef54b..000000000
--- a/.changeset/remember-poll-cadence-and-replay-dedupe.md
+++ /dev/null
@@ -1,13 +0,0 @@
----
-"@mysten-incubation/memwal": patch
----
-
-Cut the dead time a `remember` spends waiting to be told it finished, and stop a reconnect replay from minting a second paid write.
-
-Job polling backed off as `min(10s, base * 1.5^min(attempt, 6))` from a 1500ms base, so status checks landed at roughly 1.5/3.75/7.1/12.2/19.8/29.8s. Real writes finish in the 15–35s band — a Walrus sliver upload plus three sequential Sui transactions — which is exactly where those gaps are widest, so a write that truly completed at 20.5s was not reported to the caller until 29.8s. That lag is pure observation cost: the job was done, nobody had asked yet. The backoff now caps at 2s and the base default drops to 600ms, putting average dead time near 1s instead of ~5s. Polling is a single indexed row read, so a 30s write costs ~17 checks instead of ~6.
-
-`waitForRememberJob` and `waitForRememberJobs` also slept *before* their first status check, so an idempotent replay of a write the relayer had already finished still paid a full poll interval for a result that was ready on arrival. Both now check first and sleep second.
-
-Generated idempotency keys are derived from the content (`sha256` over a 30-minute time bucket plus namespace and text) instead of `crypto.randomUUID()`. `pendingRememberKeys` only ever dedupes retries that reuse one client instance, and the MCP sidecar builds a fresh `MemWal` per transport session — so when a stdio bridge reconnects a dropped stream and the agent re-issues the same `memwal_remember`, the map is empty and a random key reads as a brand-new write. The relayer would then mint a second paid Walrus blob for a write already in flight, doubling queue load exactly when the queue was already slow enough to have caused the drop. The bucket bounds the collapse: `remember_jobs` rows are never pruned, so an unbucketed key would dedupe against a job from any point in history and a re-save of a since-deleted fact would return the old blob id instead of storing it again.
-
-Callers passing an explicit `idempotencyKey` are unaffected, and distinct text or namespaces still get distinct keys.
diff --git a/.changeset/sdk-bound-every-relayer-request.md b/.changeset/sdk-bound-every-relayer-request.md
deleted file mode 100644
index 83b7aca27..000000000
--- a/.changeset/sdk-bound-every-relayer-request.md
+++ /dev/null
@@ -1,11 +0,0 @@
----
-"@mysten-incubation/memwal": patch
----
-
-Give every relayer request a deadline. `fetch` has none of its own, and the SDK passed an abort signal on exactly one method (`recall`, 15s) — so the accept POST, every job-status read, and the `/version` and `/config` handshake calls that run before any of them could stay pending for as long as the socket stayed open.
-
-That was not merely untidy. A poll loop checks its budget at the *top* of each iteration, which bounds when the next request starts, not how long one takes — so a single stalled read ran straight past `timeoutMs`. A `memwal_remember` documented as capping at 90s was observed by an MCP client still running after 120s, with the stdio bridge's orphan sweeper the first thing to fire, minutes later.
-
-Requests now default to a 30s deadline, matching the relayer's own outbound HTTP client: any call that depends on the relayer reaching the sidecar, Walrus or OpenAI has already failed upstream by the time it fires. Configure it with `requestTimeoutMs` on `MemWal.create`; a non-positive or non-finite value falls back to the default rather than disabling the bound. The two endpoints that legitimately run longer carry their own: `restore` (60s — the route bounds itself at 55s server-side and answers rather than going quiet) and `analyze` (60s — it runs the extractor LLM inline before accepting). `recall` keeps its 15s, now as a named constant rather than a hand-rolled `AbortController`.
-
-Inside the wait loops each poll is bounded by the client deadline clamped to the remaining budget. Both directions matter: the remaining budget stops a poll outliving the wait it belongs to, and the client deadline stops one stalled poll swallowing the whole budget, so the loop still gets to retry. An expired request raises `MemWalRequestTimeout` carrying `status: 504`, which `isTransientPollingStatus` already treats as retryable — so a stalled poll is retried against what is left rather than failing the wait outright. A caller's own abort, and any other transport error, propagates unchanged.
diff --git a/.claude-plugin/marketplace.json b/.claude-plugin/marketplace.json
index 1070950e9..11b63198a 100644
--- a/.claude-plugin/marketplace.json
+++ b/.claude-plugin/marketplace.json
@@ -11,7 +11,7 @@
"name": "memwal",
"source": "./packages/mcp/plugin",
"description": "Automatic Walrus Memory — proactive recall and durable-fact saving via the MemWal MCP + lifecycle hooks.",
- "version": "0.0.13"
+ "version": "0.0.14"
}
]
}
diff --git a/.cursor-plugin/marketplace.json b/.cursor-plugin/marketplace.json
index a4b700e2b..35ec36556 100644
--- a/.cursor-plugin/marketplace.json
+++ b/.cursor-plugin/marketplace.json
@@ -11,7 +11,7 @@
"name": "memwal",
"source": "./packages/mcp/plugin",
"description": "Automatic Walrus Memory — proactive recall and durable-fact saving via the MemWal MCP + lifecycle hooks.",
- "version": "0.0.13"
+ "version": "0.0.14"
}
]
}
diff --git a/docs/mcp/changelog.mdx b/docs/mcp/changelog.mdx
index cd8cc258a..9d7d7c2a2 100644
--- a/docs/mcp/changelog.mdx
+++ b/docs/mcp/changelog.mdx
@@ -31,6 +31,21 @@ answer: >-
The latest MCP package release is 0.0.13. A `memwal_remember`, `memwal_remember_bulk` or `memwal_analyze` whose reply never arrives is no longer reported as safe to retry: the relayer accepts those with HTTP 202 and finishes them in a durable queue, so the write may already have landed, and `/api/remember/bulk` has no idempotency key — repeating it stores a second paid copy. The tool now says the write may have completed and points at `memwal_recall` to check before re-saving, while a lost read still says plainly that retrying is safe. When the relayer rejects the saved delegate key, tool calls now get an auth error pointing at `memwal_login` instead of waiting minutes for a retry hint that cannot work. It writes the credentials file by creating a new 0600 file and renaming it into place, so a sign-in never puts the delegate private key into a credentials.json that a manual chmod or a restored backup left world-readable. A completed sign-in is now confirmed with a notification and a one-shot banner naming the account and the resolved credentials path, and the bridge keeps reading stdin after an in-session login instead of going deaf. Unrecognised command-line options now warn instead of being silently ignored, `--help` lists the network presets and the URLs each resolves to, `memwal_health` names the relayer the client dialled, and `memwal_restore` reports `failed` and retries the same page when truncation is a download or embed blip, instead of always telling the agent to raise `limit`. Plugin launch configs (`.mcp.json`, Cursor/Codex copies, and the Codex fallback installer) now pin `@mysten-incubation/memwal-mcp@0.0.13` so npx cannot keep a cached 0.0.5. Version 0.0.12 forwards the MCP client's initialize.clientInfo to the relayer so sidecar logs can name the coding agent (Claude Code, Codex, Cursor, and others) on each session and tool call, and it resolves the credential directory on every access so MEMWAL_CREDS_DIR can override it.
---
+## 0.0.14
+
+### Fixed
+
+- Bound every relayer call the tools make. The pinned SDK aborts a request only when the caller passes a signal, which none of the memory methods do, so a stalled socket kept a tool running with no ceiling — `memwal_remember` was observed still going past 120s against a 90s budget. Accepts are bounded at 15s (`MEMWAL_MCP_ACCEPT_DEADLINE_MS`), waits at their own budget plus grace. The request is not cancelled — the SDK exposes no way to pass a signal — but the agent is no longer held by it.
+- Honour the relayer's `retry_after` instead of dropping the write. Once the per-delegate-key budget (60 weighted requests/minute) is spent the relayer answers 429 with a cooldown, and nothing backed off: the fact was never written and the agent saw only an opaque tool error. A short cooldown is now absorbed; a long one is reported with the wait named, stating plainly that the fact was NOT saved and pointing at the cheaper shape — one `memwal_remember_bulk` rather than N single calls, one `memwal_remember_status(job_ids)` rather than N status calls. Only rejections that provably never reached the handler retry, so `/api/remember/bulk`, which carries no idempotency key, cannot be duplicated by a retry.
+- `memwal_remember` sends a content-derived idempotency key, so the retry its own timeout message invites really does attach to the job already in flight instead of storing a second paid copy. The key is computed by the tool rather than relied on from the SDK, whose published build mints a random UUID per client instance.
+- `memwal_remember_status` accepts `job_ids` to settle a whole batch in one call, and reports a mixed batch honestly — a still-uploading row no longer renders the poll timeout as `error=`, which read as a failed write. Settling in one request also matters against the rate limit: 20 ids cost one request, not twenty.
+- The bridge's cold-start tool list no longer disagrees with the sidecar's. `memwal_remember_status` advertised only `job_id`, required, under `additionalProperties: false`, so the batch call the tools themselves instruct was rejected until `tools/list_changed` arrived; the `waitMs` ceiling advertised 60000 after the sidecar lowered it to 45000, which came back as an MCP validation error; and `memwal_remember_bulk` still carried its pre-queue description. Tests now pin the parts an agent acts on.
+- The accepted-then-failed report attached to `memwal_recall` no longer tells the agent to re-send text it does not have. The relayer stores only the SEAL ciphertext, so a failed write's wording cannot be recovered — the report now says so and asks the user to restate the fact rather than inviting the agent to guess.
+
+### Changed
+
+- `memwal_remember` keeps blocking until the write reaches `done`, so a result carries a real `blob_id`. Returning at accept is available behind `MEMWAL_MCP_REMEMBER_WAIT_MS=0` for an operator who wants it, and remains its own product decision rather than a side effect of the latency work here.
+
## 0.0.13
This release stops a write whose reply was lost from being reported as safe to retry — repeating one can store a second paid copy — answers tool calls with an auth error pointing at `memwal_login` when the relayer rejects the saved delegate key, writes the credentials file through a fresh `0600` file that it renames into place, confirms a completed sign-in and keeps the bridge reading stdin afterwards, warns on unrecognised command-line options instead of ignoring them, documents the network presets in `--help`, names the relayer in `memwal_health`, reports restore `failed` counts when truncation is a transient download or embed blip, and pins plugin launch configs (`.mcp.json`, Cursor/Codex copies, and the Codex fallback installer) so npx cannot keep a cached 0.0.5.
diff --git a/docs/sdk/changelog.mdx b/docs/sdk/changelog.mdx
index 27064137e..d7f029562 100644
--- a/docs/sdk/changelog.mdx
+++ b/docs/sdk/changelog.mdx
@@ -31,6 +31,14 @@ answer: >-
The latest TypeScript SDK release is 0.1.7. Request bodies are hashed with `@noble/hashes` rather than WebCrypto-or-`node:crypto`, so the SDK no longer imports a Node builtin that Vite silently externalises into a runtime crash in the browser, and it declares a Node 20 floor. `restore()` results include `failed` (required like `truncated`; SDK defaults omitted to `0`) for permanent decrypt/UTF-8 failures instead of folding them into `skipped`. Empty-body 401s use the AUTH_REJECTED troubleshooting message instead of telling callers to run `memwal_login`. Account and manual PTBs use typed `tx.pure` helpers so they work with modern `@mysten/sui`. 0.1.6 added optional `created_at` on `recall()` results, plus `sort` and `scoringWeights` on `RecallOptions`, and reports HTTP 503 as a retryable upstream outage instead of a sign-in failure.
---
+## 0.1.8
+
+### Fixed
+
+- Every relayer request now carries a deadline. `fetch` has none of its own and the SDK passed an abort signal on exactly one method (`recall`, 15s), so the accept POST, every job-status read, and the `/version` and `/config` handshake calls could stay pending for as long as the socket stayed open. A poll loop checks its budget at the *top* of each iteration, which bounds when the next request starts rather than how long one takes — so a single stalled read ran straight past `timeoutMs`, and a `memwal_remember` documented as capping at 90s was observed by an MCP client still running after 120s. Requests default to 30s, matching the relayer's own outbound client; set it with `requestTimeoutMs` on `MemWal.create`, where a non-positive or non-finite value falls back to the default rather than disabling the bound. `restore` (60s) and `analyze` (60s) carry their own, since the route self-bounds at 55s and the extractor LLM runs inline respectively. Inside the wait loops each poll is bounded by the client deadline clamped to the remaining budget, so one stalled poll can neither outlive the wait nor swallow it. An expired request raises `MemWalRequestTimeout` with `status: 504`, which the existing transient-poll handling already retries; a caller's own abort and every other transport error propagate unchanged.
+- Cut the dead time a job wait spends before it is told the job finished. Polling backed off as `min(10s, base * 1.5^attempt)` from a 1500ms base, putting status checks roughly 1.5/3.75/7.1/12.2/19.8/29.8s apart — and real writes complete in the 15–35s band, exactly where those gaps are widest, so a write that finished at 20.5s was not reported until 29.8s. The ceiling is now 2s and the base default 600ms. Both wait loops also slept *before* their first check, so a replay of a write the relayer had already finished paid a full interval for a result that was ready on arrival; they now check first and sleep second.
+- Generated idempotency keys are derived from the content (a 30-minute bucket plus namespace and text) instead of `crypto.randomUUID()`. The per-instance key map only ever deduped retries that reused one client, and callers such as the MCP sidecar build a fresh client per session — so a replay after a reconnect read as a brand-new write and the relayer minted a second paid Walrus blob for one already in flight. The bucket bounds the collapse, since `remember_jobs` rows are never pruned. Callers passing an explicit `idempotencyKey` are unaffected, and distinct text or namespaces still derive distinct keys.
+
## 0.1.7
This release removes the Node `crypto` import that crashed bundled browser builds at runtime, adds `failed` on `restore()` results, declares a Node 20 floor, stops telling headless SDK clients to call `memwal_login` on empty-body 401s, and switches account and manual PTBs to typed `tx.pure` helpers.
diff --git a/packages/mcp/CHANGELOG.md b/packages/mcp/CHANGELOG.md
index 309ffa8b0..8ce426bd4 100644
--- a/packages/mcp/CHANGELOG.md
+++ b/packages/mcp/CHANGELOG.md
@@ -1,5 +1,20 @@
# @mysten-incubation/memwal-mcp
+## 0.0.14
+
+### Fixed
+
+- Bound every relayer call the tools make. The pinned SDK aborts a request only when the caller passes a signal, which none of the memory methods do, so a stalled socket kept a tool running with no ceiling — `memwal_remember` was observed still going past 120s against a 90s budget. Accepts are bounded at 15s (`MEMWAL_MCP_ACCEPT_DEADLINE_MS`), waits at their own budget plus grace. The request is not cancelled — the SDK exposes no way to pass a signal — but the agent is no longer held by it.
+- Honour the relayer's `retry_after` instead of dropping the write. Once the per-delegate-key budget (60 weighted requests/minute) is spent the relayer answers 429 with a cooldown, and nothing backed off: the fact was never written and the agent saw only an opaque tool error. A short cooldown is now absorbed; a long one is reported with the wait named, stating plainly that the fact was NOT saved and pointing at the cheaper shape — one `memwal_remember_bulk` rather than N single calls, one `memwal_remember_status(job_ids)` rather than N status calls. Only rejections that provably never reached the handler retry, so `/api/remember/bulk`, which carries no idempotency key, cannot be duplicated by a retry.
+- `memwal_remember` sends a content-derived idempotency key, so the retry its own timeout message invites really does attach to the job already in flight instead of storing a second paid copy. The key is computed by the tool rather than relied on from the SDK, whose published build mints a random UUID per client instance.
+- `memwal_remember_status` accepts `job_ids` to settle a whole batch in one call, and reports a mixed batch honestly — a still-uploading row no longer renders the poll timeout as `error=`, which read as a failed write. Settling in one request also matters against the rate limit: 20 ids cost one request, not twenty.
+- The bridge's cold-start tool list no longer disagrees with the sidecar's. `memwal_remember_status` advertised only `job_id`, required, under `additionalProperties: false`, so the batch call the tools themselves instruct was rejected until `tools/list_changed` arrived; the `waitMs` ceiling advertised 60000 after the sidecar lowered it to 45000, which came back as an MCP validation error; and `memwal_remember_bulk` still carried its pre-queue description. Tests now pin the parts an agent acts on.
+- The accepted-then-failed report attached to `memwal_recall` no longer tells the agent to re-send text it does not have. The relayer stores only the SEAL ciphertext, so a failed write's wording cannot be recovered — the report now says so and asks the user to restate the fact rather than inviting the agent to guess.
+
+### Changed
+
+- `memwal_remember` keeps blocking until the write reaches `done`, so a result carries a real `blob_id`. Returning at accept is available behind `MEMWAL_MCP_REMEMBER_WAIT_MS=0` for an operator who wants it, and remains its own product decision rather than a side effect of the latency work here.
+
## 0.0.13
### Fixed
diff --git a/packages/mcp/package.json b/packages/mcp/package.json
index 44738b58d..69ed21ca3 100644
--- a/packages/mcp/package.json
+++ b/packages/mcp/package.json
@@ -1,7 +1,7 @@
{
"name": "@mysten-incubation/memwal-mcp",
- "version": "0.0.13",
- "description": "Walrus Memory MCP client — single-binary stdio MCP server that bridges Cursor / Claude Desktop / Antigravity / Claude Code to the Walrus Memory relayer. Handles browser-based wallet login on first run.",
+ "version": "0.0.14",
+ "description": "Walrus Memory MCP client \u2014 single-binary stdio MCP server that bridges Cursor / Claude Desktop / Antigravity / Claude Code to the Walrus Memory relayer. Handles browser-based wallet login on first run.",
"type": "module",
"engines": {
"node": ">=20.0.0"
diff --git a/packages/mcp/plugin/.claude-plugin/plugin.json b/packages/mcp/plugin/.claude-plugin/plugin.json
index 66925ea5d..f291f6394 100644
--- a/packages/mcp/plugin/.claude-plugin/plugin.json
+++ b/packages/mcp/plugin/.claude-plugin/plugin.json
@@ -1,12 +1,18 @@
{
- "name": "memwal",
- "version": "0.0.13",
- "description": "Automatic Walrus Memory for Claude Code — proactive recall and durable-fact saving via the MemWal MCP + lifecycle hooks.",
- "author": {
- "name": "Mysten Labs"
- },
- "homepage": "https://memory.walrus.xyz",
- "repository": "https://github.com/MystenLabs/MemWal",
- "license": "Apache-2.0",
- "keywords": ["memory", "mcp", "walrus", "sui", "semantic-search"]
+ "name": "memwal",
+ "version": "0.0.14",
+ "description": "Automatic Walrus Memory for Claude Code \u2014 proactive recall and durable-fact saving via the MemWal MCP + lifecycle hooks.",
+ "author": {
+ "name": "Mysten Labs"
+ },
+ "homepage": "https://memory.walrus.xyz",
+ "repository": "https://github.com/MystenLabs/MemWal",
+ "license": "Apache-2.0",
+ "keywords": [
+ "memory",
+ "mcp",
+ "walrus",
+ "sui",
+ "semantic-search"
+ ]
}
diff --git a/packages/mcp/plugin/.codex-mcp.json b/packages/mcp/plugin/.codex-mcp.json
index 345b9c487..b833c2d77 100644
--- a/packages/mcp/plugin/.codex-mcp.json
+++ b/packages/mcp/plugin/.codex-mcp.json
@@ -2,7 +2,7 @@
"mcpServers": {
"memwal": {
"command": "npx",
- "args": ["-y", "@mysten-incubation/memwal-mcp@0.0.13"]
+ "args": ["-y", "@mysten-incubation/memwal-mcp@0.0.14"]
}
}
}
diff --git a/packages/mcp/plugin/.codex-plugin/plugin.json b/packages/mcp/plugin/.codex-plugin/plugin.json
index bdbfa5d91..3b70638ca 100644
--- a/packages/mcp/plugin/.codex-plugin/plugin.json
+++ b/packages/mcp/plugin/.codex-plugin/plugin.json
@@ -1,29 +1,38 @@
{
- "name": "memwal",
- "version": "0.0.13",
- "description": "Persistent Walrus Memory for Codex. Remembers decisions, preferences, and project context across sessions.",
- "author": {
- "name": "Mysten Labs",
- "url": "https://memory.walrus.xyz"
- },
- "homepage": "https://memory.walrus.xyz",
- "repository": "https://github.com/MystenLabs/MemWal",
- "license": "Apache-2.0",
- "keywords": ["memory", "personalization", "mcp", "walrus", "semantic-search"],
- "mcpServers": "./.codex-mcp.json",
- "hooks": "./hooks/codex-hooks.json",
- "interface": {
- "displayName": "MemWal",
- "shortDescription": "Portable, encrypted memory layer for AI coding workflows",
- "longDescription": "MemWal adds long-term, user-owned memory to Codex. Store decisions, preferences, and session context; memories are encrypted with SEAL and stored on Walrus, and retrieved via semantic search so Codex always has the right context.",
- "developerName": "Mysten Labs",
- "category": "Productivity",
- "capabilities": ["Read", "Write"],
- "websiteURL": "https://memory.walrus.xyz",
- "defaultPrompt": [
- "Search my memories for recent project decisions",
- "Remember that I prefer pnpm and TypeScript strict mode",
- "What do you know about my coding preferences?"
- ]
- }
+ "name": "memwal",
+ "version": "0.0.14",
+ "description": "Persistent Walrus Memory for Codex. Remembers decisions, preferences, and project context across sessions.",
+ "author": {
+ "name": "Mysten Labs",
+ "url": "https://memory.walrus.xyz"
+ },
+ "homepage": "https://memory.walrus.xyz",
+ "repository": "https://github.com/MystenLabs/MemWal",
+ "license": "Apache-2.0",
+ "keywords": [
+ "memory",
+ "personalization",
+ "mcp",
+ "walrus",
+ "semantic-search"
+ ],
+ "mcpServers": "./.codex-mcp.json",
+ "hooks": "./hooks/codex-hooks.json",
+ "interface": {
+ "displayName": "MemWal",
+ "shortDescription": "Portable, encrypted memory layer for AI coding workflows",
+ "longDescription": "MemWal adds long-term, user-owned memory to Codex. Store decisions, preferences, and session context; memories are encrypted with SEAL and stored on Walrus, and retrieved via semantic search so Codex always has the right context.",
+ "developerName": "Mysten Labs",
+ "category": "Productivity",
+ "capabilities": [
+ "Read",
+ "Write"
+ ],
+ "websiteURL": "https://memory.walrus.xyz",
+ "defaultPrompt": [
+ "Search my memories for recent project decisions",
+ "Remember that I prefer pnpm and TypeScript strict mode",
+ "What do you know about my coding preferences?"
+ ]
+ }
}
diff --git a/packages/mcp/plugin/.cursor-mcp.json b/packages/mcp/plugin/.cursor-mcp.json
index 345b9c487..b833c2d77 100644
--- a/packages/mcp/plugin/.cursor-mcp.json
+++ b/packages/mcp/plugin/.cursor-mcp.json
@@ -2,7 +2,7 @@
"mcpServers": {
"memwal": {
"command": "npx",
- "args": ["-y", "@mysten-incubation/memwal-mcp@0.0.13"]
+ "args": ["-y", "@mysten-incubation/memwal-mcp@0.0.14"]
}
}
}
diff --git a/packages/mcp/plugin/.cursor-plugin/plugin.json b/packages/mcp/plugin/.cursor-plugin/plugin.json
index 68bfe76ff..d7eb97856 100644
--- a/packages/mcp/plugin/.cursor-plugin/plugin.json
+++ b/packages/mcp/plugin/.cursor-plugin/plugin.json
@@ -1,12 +1,19 @@
{
- "name": "memwal",
- "version": "0.0.13",
- "description": "Automatic Walrus Memory for Cursor — proactive recall and durable-fact saving via the MemWal MCP + lifecycle hooks.",
- "author": { "name": "Mysten Labs" },
- "homepage": "https://memory.walrus.xyz",
- "repository": "https://github.com/MystenLabs/MemWal",
- "license": "Apache-2.0",
- "keywords": ["memory", "mcp", "walrus", "semantic-search"],
- "hooks": "./hooks/cursor-hooks.json",
- "mcpServers": ".cursor-mcp.json"
+ "name": "memwal",
+ "version": "0.0.14",
+ "description": "Automatic Walrus Memory for Cursor \u2014 proactive recall and durable-fact saving via the MemWal MCP + lifecycle hooks.",
+ "author": {
+ "name": "Mysten Labs"
+ },
+ "homepage": "https://memory.walrus.xyz",
+ "repository": "https://github.com/MystenLabs/MemWal",
+ "license": "Apache-2.0",
+ "keywords": [
+ "memory",
+ "mcp",
+ "walrus",
+ "semantic-search"
+ ],
+ "hooks": "./hooks/cursor-hooks.json",
+ "mcpServers": ".cursor-mcp.json"
}
diff --git a/packages/mcp/plugin/.mcp.json b/packages/mcp/plugin/.mcp.json
index 345b9c487..b833c2d77 100644
--- a/packages/mcp/plugin/.mcp.json
+++ b/packages/mcp/plugin/.mcp.json
@@ -2,7 +2,7 @@
"mcpServers": {
"memwal": {
"command": "npx",
- "args": ["-y", "@mysten-incubation/memwal-mcp@0.0.13"]
+ "args": ["-y", "@mysten-incubation/memwal-mcp@0.0.14"]
}
}
}
diff --git a/packages/mcp/plugin/plugin.json b/packages/mcp/plugin/plugin.json
index 17b1993e0..c9d660ffd 100644
--- a/packages/mcp/plugin/plugin.json
+++ b/packages/mcp/plugin/plugin.json
@@ -1,12 +1,20 @@
{
- "id": "memwal",
- "name": "memwal",
- "version": "0.0.13",
- "description": "Automatic Walrus Memory for Antigravity — proactive recall and durable-fact saving via the MemWal MCP + lifecycle hooks.",
- "author": { "name": "Mysten Labs" },
- "homepage": "https://memory.walrus.xyz",
- "repository": "https://github.com/MystenLabs/MemWal",
- "license": "Apache-2.0",
- "keywords": ["memory", "mcp", "walrus", "sui", "semantic-search"],
- "contextFileName": "AGENTS.md"
+ "id": "memwal",
+ "name": "memwal",
+ "version": "0.0.14",
+ "description": "Automatic Walrus Memory for Antigravity \u2014 proactive recall and durable-fact saving via the MemWal MCP + lifecycle hooks.",
+ "author": {
+ "name": "Mysten Labs"
+ },
+ "homepage": "https://memory.walrus.xyz",
+ "repository": "https://github.com/MystenLabs/MemWal",
+ "license": "Apache-2.0",
+ "keywords": [
+ "memory",
+ "mcp",
+ "walrus",
+ "sui",
+ "semantic-search"
+ ],
+ "contextFileName": "AGENTS.md"
}
diff --git a/packages/sdk/CHANGELOG.md b/packages/sdk/CHANGELOG.md
index 75fa71aa5..c737ed689 100644
--- a/packages/sdk/CHANGELOG.md
+++ b/packages/sdk/CHANGELOG.md
@@ -1,5 +1,13 @@
# @mysten-incubation/memwal
+## 0.1.8
+
+### Fixed
+
+- Every relayer request now carries a deadline. `fetch` has none of its own and the SDK passed an abort signal on exactly one method (`recall`, 15s), so the accept POST, every job-status read, and the `/version` and `/config` handshake calls could stay pending for as long as the socket stayed open. A poll loop checks its budget at the *top* of each iteration, which bounds when the next request starts rather than how long one takes — so a single stalled read ran straight past `timeoutMs`, and a `memwal_remember` documented as capping at 90s was observed by an MCP client still running after 120s. Requests default to 30s, matching the relayer's own outbound client; set it with `requestTimeoutMs` on `MemWal.create`, where a non-positive or non-finite value falls back to the default rather than disabling the bound. `restore` (60s) and `analyze` (60s) carry their own, since the route self-bounds at 55s and the extractor LLM runs inline respectively. Inside the wait loops each poll is bounded by the client deadline clamped to the remaining budget, so one stalled poll can neither outlive the wait nor swallow it. An expired request raises `MemWalRequestTimeout` with `status: 504`, which the existing transient-poll handling already retries; a caller's own abort and every other transport error propagate unchanged.
+- Cut the dead time a job wait spends before it is told the job finished. Polling backed off as `min(10s, base * 1.5^attempt)` from a 1500ms base, putting status checks roughly 1.5/3.75/7.1/12.2/19.8/29.8s apart — and real writes complete in the 15–35s band, exactly where those gaps are widest, so a write that finished at 20.5s was not reported until 29.8s. The ceiling is now 2s and the base default 600ms. Both wait loops also slept *before* their first check, so a replay of a write the relayer had already finished paid a full interval for a result that was ready on arrival; they now check first and sleep second.
+- Generated idempotency keys are derived from the content (a 30-minute bucket plus namespace and text) instead of `crypto.randomUUID()`. The per-instance key map only ever deduped retries that reused one client, and callers such as the MCP sidecar build a fresh client per session — so a replay after a reconnect read as a brand-new write and the relayer minted a second paid Walrus blob for one already in flight. The bucket bounds the collapse, since `remember_jobs` rows are never pruned. Callers passing an explicit `idempotencyKey` are unaffected, and distinct text or namespaces still derive distinct keys.
+
## 0.1.7
### Added
diff --git a/packages/sdk/package.json b/packages/sdk/package.json
index ca57884f5..e2a6472bb 100644
--- a/packages/sdk/package.json
+++ b/packages/sdk/package.json
@@ -1,7 +1,7 @@
{
"name": "@mysten-incubation/memwal",
- "version": "0.1.7",
- "description": "Walrus Memory — Privacy-first AI memory SDK with Ed25519 delegate key auth",
+ "version": "0.1.8",
+ "description": "Walrus Memory \u2014 Privacy-first AI memory SDK with Ed25519 delegate key auth",
"type": "module",
"engines": {
"node": ">=20.0.0"
diff --git a/scripts/verify-manual-sdk-release.mjs b/scripts/verify-manual-sdk-release.mjs
index 314758341..66f89a563 100644
--- a/scripts/verify-manual-sdk-release.mjs
+++ b/scripts/verify-manual-sdk-release.mjs
@@ -5,7 +5,7 @@ import { readFileSync } from "node:fs";
const releases = [
{
name: "TypeScript SDK",
- version: "0.1.7",
+ version: "0.1.8",
manifests: [["packages/sdk/package.json", "version"]],
changelogs: ["packages/sdk/CHANGELOG.md", "docs/sdk/changelog.mdx"],
},
@@ -23,7 +23,7 @@ const releases = [
},
{
name: "MCP package",
- version: "0.0.13",
+ version: "0.0.14",
manifests: [
["packages/mcp/package.json", "version"],
[".claude-plugin/marketplace.json", "plugin-version"],
diff --git a/services/server/scripts/mcp/__tests__/remember-bulk-fast-return.test.ts b/services/server/scripts/mcp/__tests__/remember-bulk-fast-return.test.ts
index e0b956005..f94be7de7 100644
--- a/services/server/scripts/mcp/__tests__/remember-bulk-fast-return.test.ts
+++ b/services/server/scripts/mcp/__tests__/remember-bulk-fast-return.test.ts
@@ -1,9 +1,14 @@
+// Same opt-in as remember-fast-return: the wait is the default, this file
+// covers the accept-and-continue path behind the knob.
+process.env.MEMWAL_MCP_REMEMBER_WAIT_MS = "0";
+
import assert from "node:assert/strict";
import test, { type TestContext } from "node:test";
import { Client } from "@modelcontextprotocol/sdk/client/index.js";
import { InMemoryTransport } from "@modelcontextprotocol/sdk/inMemory.js";
import type { MemWalSession } from "../auth.js";
-import { createMcpServer } from "../server.js";
+
+const { createMcpServer } = await import("../server.js");
/**
* `memwal_remember_bulk` used to block until every job in the batch reached a
diff --git a/services/server/scripts/mcp/__tests__/remember-deadline.test.ts b/services/server/scripts/mcp/__tests__/remember-deadline.test.ts
index 0907ea03f..87958265d 100644
--- a/services/server/scripts/mcp/__tests__/remember-deadline.test.ts
+++ b/services/server/scripts/mcp/__tests__/remember-deadline.test.ts
@@ -1,6 +1,9 @@
// Bound the SDK calls that have no deadline of their own. Set before the
// module under test is imported — ACCEPT_DEADLINE_MS is read once at load.
process.env.MEMWAL_MCP_ACCEPT_DEADLINE_MS = "150";
+// This file is about the ACCEPT leg, so skip the terminal wait that is now the
+// default — otherwise every case here would also drive the poll loop.
+process.env.MEMWAL_MCP_REMEMBER_WAIT_MS = "0";
import assert from "node:assert/strict";
import test, { type TestContext } from "node:test";
@@ -146,7 +149,7 @@ test("memwal_remember's accept timeout still says a retry is safe", async (t) =>
name: "memwal_remember",
arguments: { text: "a durable fact" },
});
- assert.match(textOf(result), /Retrying in this session is safe/);
+ assert.match(textOf(result), /Retrying is safe/);
});
test("memwal_remember_status cannot hang forever on a stalled read", async (t) => {
diff --git a/services/server/scripts/mcp/__tests__/remember-fast-return.test.ts b/services/server/scripts/mcp/__tests__/remember-fast-return.test.ts
index 53ec0f8a8..b8381a298 100644
--- a/services/server/scripts/mcp/__tests__/remember-fast-return.test.ts
+++ b/services/server/scripts/mcp/__tests__/remember-fast-return.test.ts
@@ -1,21 +1,31 @@
+// Opt into accept-and-continue for this file. It is NOT the default — D1 kept
+// the wait, so a plain call blocks and returns a blob_id — but the path stays
+// reachable via this knob, and it is the path these tests cover.
+process.env.MEMWAL_MCP_REMEMBER_WAIT_MS = "0";
+
import assert from "node:assert/strict";
import test, { type TestContext } from "node:test";
import { Client } from "@modelcontextprotocol/sdk/client/index.js";
import { InMemoryTransport } from "@modelcontextprotocol/sdk/inMemory.js";
import type { MemWalSession } from "../auth.js";
-import { createMcpServer } from "../server.js";
-import { REMEMBER_WAIT_MS, parseWaitBudget } from "../tools/remember-wait.js";
+
+// Dynamic: static imports hoist above the assignment above, so the module
+// would read the real default before the line that overrides it ever runs.
+const { createMcpServer } = await import("../server.js");
+const { REMEMBER_WAIT_MS, parseWaitBudget, MAX_REMEMBER_WAIT_MS } =
+ await import("../tools/remember-wait.js");
/**
- * `memwal_remember` no longer blocks until a Walrus write reaches `done` —
- * that cost 30–75s per call against production. It returns at accept (~1.1s)
- * and hands back a job_id.
+ * `memwal_remember` blocks to terminal by default, so a result carries a real
+ * blob_id — D1 settled that, and returning at accept is its own product
+ * ticket. What this file covers is the accept-and-continue path an operator
+ * opts into with `MEMWAL_MCP_REMEMBER_WAIT_MS=0`.
*
- * The risk that buys is an agent reading "accepted" as "saved", so these tests
- * pin what keeps it honest: an accepted result must never read as success or
- * carry a blob_id, and `memwal_remember_status` — now the only thing that can
- * observe a job failing after acceptance — must report that failure as an
- * error rather than as a write still in flight.
+ * The risk that path buys is an agent reading "accepted" as "saved", so these
+ * tests pin what keeps it honest: an accepted result must never read as
+ * success or carry a blob_id, and `memwal_remember_status` — the only thing
+ * that can observe a job failing after acceptance — must report that failure
+ * as an error rather than as a write still in flight.
*/
interface FakeJob {
@@ -94,20 +104,23 @@ function textOf(result: unknown): string {
.join("\n");
}
-test("the default budget is zero — memwal_remember returns at accept", () => {
- // A non-zero budget below the real completion time is the worst case:
- // the caller pays the wait and still gets no guarantee.
- assert.equal(parseWaitBudget(undefined), 0);
- assert.equal(parseWaitBudget(""), 0);
+test("the default is the full wait, and zero is opt-in", () => {
+ // D1: a result means the fact landed, so an unset budget blocks to
+ // terminal. The in-between is the setting to avoid — a budget under the
+ // real completion time pays the wait AND still returns pending.
+ assert.equal(parseWaitBudget(undefined), MAX_REMEMBER_WAIT_MS);
+ assert.equal(parseWaitBudget(""), MAX_REMEMBER_WAIT_MS);
+ // ...and this file asked for the opt-in path at the top.
assert.equal(REMEMBER_WAIT_MS, 0);
+ assert.equal(parseWaitBudget("0"), 0);
});
test("a typo'd budget falls back to the default instead of picking one nobody asked for", () => {
// Number("10s") is NaN, and every NaN comparison is false — an unvalidated
// parse would sail past a range check.
- assert.equal(parseWaitBudget("10s"), 0);
- assert.equal(parseWaitBudget("abc"), 0);
- assert.equal(parseWaitBudget("-1"), 0);
+ assert.equal(parseWaitBudget("10s"), MAX_REMEMBER_WAIT_MS);
+ assert.equal(parseWaitBudget("abc"), MAX_REMEMBER_WAIT_MS);
+ assert.equal(parseWaitBudget("-1"), MAX_REMEMBER_WAIT_MS);
});
test("a budget past the ceiling is clamped, not honoured", () => {
diff --git a/services/server/scripts/mcp/tools/recall.ts b/services/server/scripts/mcp/tools/recall.ts
index 18b6f364a..47a91a1f9 100644
--- a/services/server/scripts/mcp/tools/recall.ts
+++ b/services/server/scripts/mcp/tools/recall.ts
@@ -96,7 +96,10 @@ export function formatFailedWrites(result: unknown): string {
`\n\n⚠ ${n} earlier ${n === 1 ? "write was" : "writes were"} accepted but then FAILED, ` +
`so ${n === 1 ? "that fact is" : "those facts are"} NOT stored:\n` +
lines.join("\n") +
- `\nSend ${n === 1 ? "it" : "them"} again with memwal_remember if still wanted.`
+ `\nThe text is not recoverable — the relayer stores only the SEAL ciphertext, and these `
+ + `writes failed before it was readable. Do not guess at what ${n === 1 ? "it" : "they"} said. `
+ + `If the fact still matters, ask the user to state it again, then save it with `
+ + `memwal_remember.`
);
}
diff --git a/services/server/scripts/mcp/tools/remember-wait.ts b/services/server/scripts/mcp/tools/remember-wait.ts
index b67a55388..25aae7755 100644
--- a/services/server/scripts/mcp/tools/remember-wait.ts
+++ b/services/server/scripts/mcp/tools/remember-wait.ts
@@ -13,28 +13,42 @@
* after acceptance (one observed failure: "Memory encryption backend is
* unavailable" 31.6s in, from the SEAL sidecar being unreachable).
*/
+import { createHash } from "node:crypto";
+
import { createLogger } from "../logger.js";
const log = createLogger("mcp");
+/** Window over which the same (namespace, text) resolves to one key. */
+const IDEMPOTENCY_BUCKET_MS = 30 * 60 * 1000;
+
/**
- * Default wait before `memwal_remember` hands back a job_id.
+ * Content-derived idempotency key for a single `remember`.
*
- * Zero — the tool returns at accept (~1.1s measured). A non-zero budget below
- * the real completion time is the worst of both: against the measured 30–75s
- * distribution a 10s wait still lands in the pending branch on nearly every
- * call, so the caller pays the 10s AND gets no guarantee. Either wait long
- * enough to actually mean it (set this to 90000) or don't wait at all.
+ * Computed HERE rather than relied on from the SDK. The sidecar installs the
+ * published `@mysten-incubation/memwal` (0.1.7), whose `rememberAsync` mints a
+ * `crypto.randomUUID()` and caches it per client instance — and the sidecar
+ * builds a fresh client per transport session, so a retry after a reconnect
+ * gets a brand-new key and the relayer stores the fact a second time at full
+ * cost. The accept-timeout message promises the caller a retry is safe, so the
+ * key that makes it safe has to exist in the version actually running, not in
+ * an unreleased source tree.
*
- * What makes returning at accept safe from disconnects: the job is a row in
- * `remember_jobs` driven by the relayer (`spawn_persisted_remember_preparation`
- * in services/server/src/routes/remember.rs), not work held in this process.
- * Closing the client does not cancel it.
+ * The bucket bounds the collapse: `remember_jobs` rows are never pruned, so an
+ * unbucketed key would dedupe against a job from any point in history and a
+ * deliberate re-save of a since-deleted fact would hand back the old blob id.
+ * Retries happen seconds after the original, so 30 minutes covers them.
*
- * What it is NOT safe from: a job that fails after acceptance. Nothing here
- * can catch that — only a later `memwal_remember_status` call can.
+ * `/api/remember/bulk` takes no key at all, which is why the bulk tool stays
+ * `idempotent: false` instead of calling this.
*/
-const DEFAULT_REMEMBER_WAIT_MS = 0;
+export function derivedIdempotencyKey(namespace: string | undefined, text: string): string {
+ const bucket = Math.floor(Date.now() / IDEMPOTENCY_BUCKET_MS);
+ const digest = createHash("sha256")
+ .update(`${bucket}\0${namespace ?? ""}\0${text}`)
+ .digest("hex");
+ return `r1-${digest}`;
+}
/**
* Hard ceiling on the wait budget. 90s matches the timeout the tool used
@@ -42,7 +56,33 @@ const DEFAULT_REMEMBER_WAIT_MS = 0;
* always-block behaviour but cannot push the call past what MCP clients
* are willing to wait for.
*/
-const MAX_REMEMBER_WAIT_MS = 90_000;
+export const MAX_REMEMBER_WAIT_MS = 90_000;
+
+/**
+ * Default wait before `memwal_remember` hands back a job_id.
+ *
+ * Blocks to terminal, so a successful call returns a real `blob_id` and the
+ * agent can say the fact is stored. D1 settled this: returning at accept is a
+ * product change on its own ticket, not a side effect of a latency fix, and
+ * the contract callers have today is "a result means it landed".
+ *
+ * That leaves the poll cadence as the part this file may legitimately shorten,
+ * and the cadence work belongs to #902 — the wait itself stays.
+ *
+ * A budget BETWEEN zero and the real completion time is the one setting to
+ * avoid. Against a measured 30–75s spread a 10s wait pays the 10s and still
+ * lands in the pending branch on nearly every call: the cost of blocking with
+ * none of the guarantee. So this is the full ceiling, and an operator who
+ * genuinely wants accept-and-continue sets `MEMWAL_MCP_REMEMBER_WAIT_MS=0`
+ * knowingly rather than inheriting it.
+ *
+ * The pending branch is still reachable and still correct — a write slower
+ * than the ceiling returns a job_id and says plainly it is not saved yet —
+ * it is simply no longer the default path.
+ */
+const DEFAULT_REMEMBER_WAIT_MS = MAX_REMEMBER_WAIT_MS;
+
+
/**
* How long `memwal_remember` waits for the write to land before returning a
@@ -301,9 +341,9 @@ export function withAcceptDeadline(
work,
ACCEPT_DEADLINE_MS,
opts.idempotent
- ? `${shared} Retrying in this session is safe: the write carries a content-derived ` +
- `idempotency key until an accept succeeds, so a retry attaches to the existing job ` +
- `instead of queueing a second paid copy.`
+ ? `${shared} Retrying is safe: this write carries a content-derived idempotency ` +
+ `key, so a retry attaches to the job already in flight instead of queueing a ` +
+ `second paid copy.`
: `${shared} Do NOT retry blindly — this endpoint carries no idempotency key, so a ` +
`re-send stores every fact a SECOND time at full cost. Check with memwal_recall ` +
`first, and only re-send what is genuinely missing.`,
diff --git a/services/server/scripts/mcp/tools/remember.ts b/services/server/scripts/mcp/tools/remember.ts
index 23e5bd843..0e7c0dfa6 100644
--- a/services/server/scripts/mcp/tools/remember.ts
+++ b/services/server/scripts/mcp/tools/remember.ts
@@ -12,6 +12,7 @@ import {
withAcceptDeadline,
withWaitDeadline,
withRelayerRetry,
+ derivedIdempotencyKey,
} from "./remember-wait.js";
const REMEMBER_INPUT = {
@@ -62,7 +63,13 @@ export function registerRememberTool(
// must succeed, the wait is a courtesy we cut short.
const accepted = await withAcceptDeadline(
withRelayerRetry(
- () => session.memwal.rememberAsync(text, namespace),
+ () =>
+ session.memwal.rememberAsync(text, namespace, {
+ // Ours, not the SDK's random one — see
+ // derivedIdempotencyKey. This is what makes the
+ // accept-timeout message's retry promise true.
+ idempotencyKey: derivedIdempotencyKey(namespace, text),
+ }),
"save this fact",
),
"memwal_remember write",
From fbc1d310ecc7555370844d3f0f2e2df0bf95199a Mon Sep 17 00:00:00 2001
From: Harry Phan
Date: Wed, 16 Sep 2026 14:46:22 +0700
Subject: [PATCH 035/132] revert(sdk): hand the poll cadence back to #902
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit
I had changed the backoff here as part of the latency work. #902 (WALM-623)
already owns it, does it better, and corrects a fact I had wrong.
Its version returns 0 for attempt 0 — the immediate first check I had built by
restructuring both wait loops — inside `pollingDelayMs` itself, which is the
right place. Its ceiling is 3s against a documented 30 weighted-requests/min
quota for status GETs. I had read 60/min off a 429 body and picked 2s, not
accounting for status reads being weighted 2, so my cap sat closer to the
quota than the one written by someone who knew the weighting.
It also moves the function to its own module and applies the same fix to the
Python SDK. Keeping my copy would have meant two implementations of one
policy, and a conflict on merge: #902 deletes the function my change edited.
Reverted here: the 2s ceiling, the 600ms base defaults, and the check-first
restructuring of both loops. The latency tests that pinned that behaviour go
with it — #902 pins the same thing in test/polling-delay.test.mjs, and
asserting it twice would leave one copy silently wrong the next time the cap
moves. What stays in that file is what this PR actually owns: the derived
idempotency key. The 0.1.8 changelog drops its poll bullet to match.
What this PR keeps in the SDK is unrelated to cadence: per-request deadlines
(#918's own finding — `fetch` had none, so a stalled socket outran `timeoutMs`
entirely) and content-derived idempotency keys.
Verified against #902: the two branches now auto-merge, with no conflicting
hunks in memwal.ts or either changelog.
---
docs/sdk/changelog.mdx | 1 -
packages/sdk/CHANGELOG.md | 1 -
packages/sdk/src/memwal.ts | 34 ++-----------
...test.mjs => remember-derived-key.test.mjs} | 48 ++++---------------
4 files changed, 13 insertions(+), 71 deletions(-)
rename packages/sdk/test/{remember-latency.test.mjs => remember-derived-key.test.mjs} (64%)
diff --git a/docs/sdk/changelog.mdx b/docs/sdk/changelog.mdx
index d7f029562..38dfa3f0a 100644
--- a/docs/sdk/changelog.mdx
+++ b/docs/sdk/changelog.mdx
@@ -36,7 +36,6 @@ answer: >-
### Fixed
- Every relayer request now carries a deadline. `fetch` has none of its own and the SDK passed an abort signal on exactly one method (`recall`, 15s), so the accept POST, every job-status read, and the `/version` and `/config` handshake calls could stay pending for as long as the socket stayed open. A poll loop checks its budget at the *top* of each iteration, which bounds when the next request starts rather than how long one takes — so a single stalled read ran straight past `timeoutMs`, and a `memwal_remember` documented as capping at 90s was observed by an MCP client still running after 120s. Requests default to 30s, matching the relayer's own outbound client; set it with `requestTimeoutMs` on `MemWal.create`, where a non-positive or non-finite value falls back to the default rather than disabling the bound. `restore` (60s) and `analyze` (60s) carry their own, since the route self-bounds at 55s and the extractor LLM runs inline respectively. Inside the wait loops each poll is bounded by the client deadline clamped to the remaining budget, so one stalled poll can neither outlive the wait nor swallow it. An expired request raises `MemWalRequestTimeout` with `status: 504`, which the existing transient-poll handling already retries; a caller's own abort and every other transport error propagate unchanged.
-- Cut the dead time a job wait spends before it is told the job finished. Polling backed off as `min(10s, base * 1.5^attempt)` from a 1500ms base, putting status checks roughly 1.5/3.75/7.1/12.2/19.8/29.8s apart — and real writes complete in the 15–35s band, exactly where those gaps are widest, so a write that finished at 20.5s was not reported until 29.8s. The ceiling is now 2s and the base default 600ms. Both wait loops also slept *before* their first check, so a replay of a write the relayer had already finished paid a full interval for a result that was ready on arrival; they now check first and sleep second.
- Generated idempotency keys are derived from the content (a 30-minute bucket plus namespace and text) instead of `crypto.randomUUID()`. The per-instance key map only ever deduped retries that reused one client, and callers such as the MCP sidecar build a fresh client per session — so a replay after a reconnect read as a brand-new write and the relayer minted a second paid Walrus blob for one already in flight. The bucket bounds the collapse, since `remember_jobs` rows are never pruned. Callers passing an explicit `idempotencyKey` are unaffected, and distinct text or namespaces still derive distinct keys.
## 0.1.7
diff --git a/packages/sdk/CHANGELOG.md b/packages/sdk/CHANGELOG.md
index c737ed689..8c44c0b9f 100644
--- a/packages/sdk/CHANGELOG.md
+++ b/packages/sdk/CHANGELOG.md
@@ -5,7 +5,6 @@
### Fixed
- Every relayer request now carries a deadline. `fetch` has none of its own and the SDK passed an abort signal on exactly one method (`recall`, 15s), so the accept POST, every job-status read, and the `/version` and `/config` handshake calls could stay pending for as long as the socket stayed open. A poll loop checks its budget at the *top* of each iteration, which bounds when the next request starts rather than how long one takes — so a single stalled read ran straight past `timeoutMs`, and a `memwal_remember` documented as capping at 90s was observed by an MCP client still running after 120s. Requests default to 30s, matching the relayer's own outbound client; set it with `requestTimeoutMs` on `MemWal.create`, where a non-positive or non-finite value falls back to the default rather than disabling the bound. `restore` (60s) and `analyze` (60s) carry their own, since the route self-bounds at 55s and the extractor LLM runs inline respectively. Inside the wait loops each poll is bounded by the client deadline clamped to the remaining budget, so one stalled poll can neither outlive the wait nor swallow it. An expired request raises `MemWalRequestTimeout` with `status: 504`, which the existing transient-poll handling already retries; a caller's own abort and every other transport error propagate unchanged.
-- Cut the dead time a job wait spends before it is told the job finished. Polling backed off as `min(10s, base * 1.5^attempt)` from a 1500ms base, putting status checks roughly 1.5/3.75/7.1/12.2/19.8/29.8s apart — and real writes complete in the 15–35s band, exactly where those gaps are widest, so a write that finished at 20.5s was not reported until 29.8s. The ceiling is now 2s and the base default 600ms. Both wait loops also slept *before* their first check, so a replay of a write the relayer had already finished paid a full interval for a result that was ready on arrival; they now check first and sleep second.
- Generated idempotency keys are derived from the content (a 30-minute bucket plus namespace and text) instead of `crypto.randomUUID()`. The per-instance key map only ever deduped retries that reused one client, and callers such as the MCP sidecar build a fresh client per session — so a replay after a reconnect read as a brand-new write and the relayer minted a second paid Walrus blob for one already in flight. The bucket bounds the collapse, since `remember_jobs` rows are never pruned. Callers passing an explicit `idempotencyKey` are unaffected, and distinct text or namespaces still derive distinct keys.
## 0.1.7
diff --git a/packages/sdk/src/memwal.ts b/packages/sdk/src/memwal.ts
index 7e7181f67..0c3d264e9 100644
--- a/packages/sdk/src/memwal.ts
+++ b/packages/sdk/src/memwal.ts
@@ -130,26 +130,9 @@ function sleep(ms: number): Promise {
return new Promise((resolve) => setTimeout(resolve, ms));
}
-/** Ceiling on the gap between two status checks.
- *
- * This is pure *observation* cost: a job that finished is not reported until
- * the next poll lands, so the cap is the worst-case dead time bolted onto
- * every wait, and half of it is the average. At the 10s ceiling checks landed
- * ~1.5/3.75/7.1/12.2/19.8/29.8s apart, so a write that truly completed at
- * 20.5s was not seen until 29.8s — and writes finish in the 15–35s band
- * (Walrus sliver upload plus three sequential Sui transactions), exactly where
- * those gaps were widest.
- *
- * 2s holds average dead time near 1s. The extra requests are cheap — polling
- * is one indexed row read on `remember_jobs` — but they are not free against
- * the relayer's per-delegate-key rate limit, which is why callers on a long
- * budget (`services/server/scripts/mcp/tools/remember-wait.ts`) pass a larger
- * base rather than relying on this floor. */
-const POLL_MAX_DELAY_MS = 2_000;
-
function pollingDelayMs(baseMs: number, attempt: number): number {
const base = Math.max(100, baseMs);
- const capped = Math.min(POLL_MAX_DELAY_MS, base * 1.5 ** Math.min(attempt, 6));
+ const capped = Math.min(10_000, base * 1.5 ** Math.min(attempt, 6));
const jitter = 0.75 + Math.random() * 0.5;
return Math.floor(capped * jitter);
}
@@ -462,17 +445,12 @@ export class MemWal {
jobId: string,
opts: { pollIntervalMs?: number; timeoutMs?: number } = {},
): Promise {
- const { pollIntervalMs = 600, timeoutMs = 60_000 } = opts;
+ const { pollIntervalMs = 1500, timeoutMs = 60_000 } = opts;
const deadline = Date.now() + timeoutMs;
let attempt = 0;
while (Date.now() < deadline) {
- // Check first, sleep second. An idempotent replay — the same key
- // for a write that already finished — is `done` on the server
- // before we ask, so the old sleep-first order billed it a full
- // poll delay for a result that was ready on arrival.
- if (attempt > 0) await sleep(pollingDelayMs(pollIntervalMs, attempt - 1));
- attempt++;
+ await sleep(pollingDelayMs(pollIntervalMs, attempt++));
let status: RememberStatusResponse;
@@ -640,7 +618,7 @@ export class MemWal {
namespaces: string[] = [],
opts: RememberBulkOptions = {},
): Promise {
- const { pollIntervalMs = 600, timeoutMs = 120_000 } = opts;
+ const { pollIntervalMs = 1500, timeoutMs = 120_000 } = opts;
const deadline = Date.now() + timeoutMs;
const results: RememberBulkItemResult[] = jobIds.map((jobId, idx) => ({
id: jobId,
@@ -653,9 +631,7 @@ export class MemWal {
let attempt = 0;
while (pending.size > 0 && Date.now() < deadline) {
- // Check first, sleep second — see `waitForRememberJob`.
- if (attempt > 0) await sleep(pollingDelayMs(pollIntervalMs, attempt - 1));
- attempt++;
+ await sleep(pollingDelayMs(pollIntervalMs, attempt++));
const pendingIds = jobIds.filter((jobId) => pending.has(jobId));
if (pendingIds.length === 0) {
diff --git a/packages/sdk/test/remember-latency.test.mjs b/packages/sdk/test/remember-derived-key.test.mjs
similarity index 64%
rename from packages/sdk/test/remember-latency.test.mjs
rename to packages/sdk/test/remember-derived-key.test.mjs
index fc51e2de9..41aaa5eda 100644
--- a/packages/sdk/test/remember-latency.test.mjs
+++ b/packages/sdk/test/remember-derived-key.test.mjs
@@ -1,3 +1,11 @@
+/**
+ * Idempotency keys for `remember`, derived from content rather than random.
+ *
+ * Poll cadence is deliberately NOT tested here — WALM-623 (#902) owns the
+ * backoff, including the immediate first attempt, and pins it in
+ * test/polling-delay.test.mjs. Asserting it from two places would leave one
+ * copy silently wrong the next time the cap moves.
+ */
import assert from "node:assert/strict";
import test from "node:test";
@@ -54,46 +62,6 @@ const DONE = {
namespace: "default",
};
-test("a job that is already done resolves without paying a poll delay first", async () => {
- stubRelayer({ onStatus: () => DONE });
-
- const started = Date.now();
- const result = await newClient().rememberAndWait("already finished");
- const elapsed = Date.now() - started;
-
- assert.equal(result.blob_id, "blob-1");
- // Sleep-first polling billed this a full base interval (~600ms, and ~1.5s
- // before the interval was lowered) for a result the server had ready.
- assert.ok(elapsed < 250, `expected an immediate first poll, took ${elapsed}ms`);
-});
-
-test("poll gaps stay bounded so a finished write is observed promptly", async () => {
- const seenAt = [];
- // Never terminal: let the loop run its full backoff ramp against the clock.
- stubRelayer({
- onStatus: () => {
- seenAt.push(Date.now());
- return { job_id: "job-1", status: "running" };
- },
- });
-
- await assert.rejects(
- newClient().rememberAndWait("slow write", undefined, { timeoutMs: 9_000 }),
- /timed out/,
- );
-
- const gaps = seenAt.slice(1).map((t, i) => t - seenAt[i]);
- const worst = Math.max(...gaps);
- // The cap is 2s; jitter can stretch one gap to 2.5s. The old 10s cap put
- // checks at 1.5/3.75/7.1/12.2/19.8/29.8s — a write finishing at 20.5s was
- // not seen until 29.8s, which is most of what a user experienced as a slow
- // remember.
- assert.ok(worst < 2_600, `worst poll gap ${worst}ms exceeds the 2s cap + jitter`);
- // And it must actually be polling, not spinning.
- assert.ok(gaps.length >= 4, `expected a real ramp, saw ${gaps.length} gaps`);
- assert.ok(Math.min(...gaps) > 50, "polling should not busy-loop");
-});
-
test("the same fact reuses one idempotency key across client instances", async () => {
const posted = [];
stubRelayer({ onStatus: () => DONE, posted });
From 68fd4c474ced95346dc69af8db27271638d775b7 Mon Sep 17 00:00:00 2001
From: Harry Phan
Date: Wed, 16 Sep 2026 16:05:43 +0700
Subject: [PATCH 036/132] fix(mcp): close the three gaps D1's default left
behind
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit
Restoring the wait changed which outcome is normal, and three things had not
caught up with it.
COPY. The session instructions — in both the sidecar and the bridge's own copy
— still told agents that `memwal_remember` "normally returns a job_id with the
fact ACCEPTED but NOT YET SAVED" and that this "is the healthy path". Since
the default went back to blocking, the healthy path is a blob_id and the
pending result is the exception. An agent reading the old text would report a
stored fact as merely queued, which is the mirror of the failure the pending
wording was written to prevent. Both instruction copies and all four tool
descriptions (sidecar and the bridge's cold-start set) now lead with the
blob_id and describe the job_id as what happens when a write outruns the
budget.
ANALYZE. `memwal_analyze` ran its extraction under the 15s ACCEPT deadline.
That number is sized for an accept — measured ~1.1s — but `/api/analyze` runs
the extractor LLM inline before it answers, which is exactly why the SDK
allows that call 60s where it allows a remember 30s. Any transcript long
enough to be worth extracting from would have been cut off mid-LLM and
reported as "did not accept". `withAcceptDeadline` now takes a ceiling and
analyze names 60s.
DROPPED JOB_ID. `withWaitDeadline` raises MemWalRelayerUnresponsive, which is
not status 504, so `isStillRunning` was false and the error fell through to
`throw` — discarding the job_id of a write that had been accepted and was
still running. That id is the only way to settle the job, and carrying it is
the entire reason the pending result exists. Our deadline firing means the
relayer went quiet on US, not that the job stopped, so it now reads as still
running. A test drives an accept that succeeds followed by a poll that never
answers, and asserts the id survives.
---
packages/mcp/src/auth-required.ts | 4 +--
packages/mcp/src/instructions.ts | 14 ++++----
.../mcp/__tests__/remember-deadline.test.ts | 36 +++++++++++++++++++
services/server/scripts/mcp/server.ts | 16 ++++-----
services/server/scripts/mcp/tools/analyze.ts | 7 +++-
.../server/scripts/mcp/tools/remember-bulk.ts | 2 +-
.../server/scripts/mcp/tools/remember-wait.ts | 15 +++++---
services/server/scripts/mcp/tools/remember.ts | 2 +-
8 files changed, 73 insertions(+), 23 deletions(-)
diff --git a/packages/mcp/src/auth-required.ts b/packages/mcp/src/auth-required.ts
index cfc569ecc..96343f105 100644
--- a/packages/mcp/src/auth-required.ts
+++ b/packages/mcp/src/auth-required.ts
@@ -41,7 +41,7 @@ interface RpcMessage {
const SIGNED_OUT_REMEMBER =
"Save a fact to the user's Walrus Memory personal memory. Call ONLY when the user explicitly asks to remember/save something. Pass the full, detailed text — never summarize.";
const SIGNED_IN_REMEMBER =
- "Save a durable fact about the user or project to their Walrus Memory. Call this PROACTIVELY whenever the user states a preference, decision, constraint, correction, identity detail, or recurring workflow — even if they did not say 'remember this'. Skip one-off tasks, the current file or bug, and small talk. Pass the full statement; do not summarize. To save several facts at once, use memwal_remember_bulk instead. Walrus writes queue, so this may return a job_id with the fact NOT yet saved — in that case say so rather than claiming it is stored, and resolve it with memwal_remember_status.";
+ "Save a durable fact about the user or project to their Walrus Memory. Call this PROACTIVELY whenever the user states a preference, decision, constraint, correction, identity detail, or recurring workflow — even if they did not say 'remember this'. Skip one-off tasks, the current file or bug, and small talk. Pass the full statement; do not summarize. To save several facts at once, use memwal_remember_bulk instead. A Walrus write takes 30-60s and this call waits for it, so a success carries a blob_id and means the fact is stored. If it outruns that budget you get a job_id and the fact is NOT yet saved — say so rather than claiming it is stored, and resolve it with memwal_remember_status.";
const SIGNED_OUT_RECALL =
"Search the user's Walrus Memory for facts relevant to a query. Returns matching memories ranked by relevance.";
const SIGNED_IN_RECALL =
@@ -72,7 +72,7 @@ function buildToolDefinitions(proactive: boolean) {
title: "Remember Multiple Facts",
annotations: { readOnlyHint: false, destructiveHint: false },
description:
- "Save multiple durable facts in one call. Use when you learned several distinct facts at once (onboarding details, a list of preferences, decisions from a discussion). Pass an array of complete fact statements (max 20) — do not summarize. Prefer this over repeated memwal_remember calls. Walrus writes queue, so this normally returns job_ids with the facts NOT yet saved — in that case say they are being saved rather than claiming they are stored, and resolve them with memwal_remember_status.",
+ "Save multiple durable facts in one call. Use when you learned several distinct facts at once (onboarding details, a list of preferences, decisions from a discussion). Pass an array of complete fact statements (max 20) — do not summarize. Prefer this over repeated memwal_remember calls. A Walrus write takes 30-60s and this call waits for them, so a success carries blob_ids and means the facts are stored. If they outrun that budget you get job_ids and the facts are NOT yet saved — say they are being saved rather than stored, and resolve them with memwal_remember_status.",
inputSchema: {
type: "object",
properties: {
diff --git a/packages/mcp/src/instructions.ts b/packages/mcp/src/instructions.ts
index e964cbfc0..057301bae 100644
--- a/packages/mcp/src/instructions.ts
+++ b/packages/mcp/src/instructions.ts
@@ -39,12 +39,14 @@ export const PROACTIVE_INSTRUCTIONS = [
"summary. Skip one-off tasks, the current file or bug, and small talk. Use",
"memwal_remember_bulk when several distinct facts arrived at once.",
"",
- "A Walrus write is queued, not instant: memwal_remember normally returns a job_id with",
- "the fact ACCEPTED but NOT YET SAVED, and storing it takes roughly another 30-60s. That",
- "is the healthy path, not an error. Tell the user the fact is being saved rather than",
- "that it is saved, and do not re-send it — that queues a duplicate. A job can still fail",
- "after acceptance, so when it matters that a fact landed, resolve the job_id with",
- "memwal_remember_status; only the blob_id it returns means the fact is stored.",
+ "A Walrus write takes roughly 30-60s, and memwal_remember and memwal_remember_bulk wait",
+ "for it: a successful call comes back with a blob_id, and that means the fact is stored.",
+ "",
+ "If a write outruns that budget the call returns job_ids instead, saying the facts are",
+ "ACCEPTED but NOT YET SAVED. That is the exception, not the normal result. When it",
+ "happens, say the facts are being saved rather than that they are saved, do NOT re-send",
+ "them — that queues duplicates — and resolve the ids with memwal_remember_status (it takes",
+ "job_id, or job_ids for a whole batch). Only a blob_id means a fact is stored.",
"",
"RECOVER: if memwal_recall unexpectedly returns nothing for a namespace that has been used",
"before, call memwal_restore to rebuild the index from Walrus.",
diff --git a/services/server/scripts/mcp/__tests__/remember-deadline.test.ts b/services/server/scripts/mcp/__tests__/remember-deadline.test.ts
index 87958265d..8612007b7 100644
--- a/services/server/scripts/mcp/__tests__/remember-deadline.test.ts
+++ b/services/server/scripts/mcp/__tests__/remember-deadline.test.ts
@@ -317,3 +317,39 @@ test("the absorb budget fits inside the accept deadline", () => {
`absorb ${MAX_ABSORBED_COOLDOWN_MS}ms must stay under the ${DEFAULT_ACCEPT_DEADLINE_MS}ms accept deadline`,
);
});
+
+test("a relayer that goes quiet mid-wait still hands back the job_id", async (t) => {
+ // The job was accepted and is a row in remember_jobs. Our wait deadline
+ // firing means the relayer stopped answering US, not that the write
+ // stopped — so the caller must still get the id needed to settle it.
+ // Before this, the deadline error fell through to `throw` and the id was
+ // lost, which is the one thing the pending result exists to carry.
+ const session = {
+ oauthScope: "memwal:read memwal:write",
+ namespace: "default",
+ memwal: {
+ async rememberAsync() {
+ return { job_id: "job-live", status: "pending" };
+ },
+ // Accepted fine, then the relayer stops answering the poll.
+ waitForRememberJob: () => new Promise(() => {}),
+ },
+ } as unknown as MemWalSession;
+
+ const [ct, st] = InMemoryTransport.createLinkedPair();
+ const server = createMcpServer(session);
+ const client = new Client({ name: "quiet-relayer", version: "1.0.0" });
+ t.after(async () => { await client.close(); await server.close(); });
+ await server.connect(st);
+ await client.connect(ct);
+
+ const res = await client.callTool({
+ name: "memwal_remember",
+ arguments: { text: "a durable fact" },
+ });
+ const text = textOf(res);
+
+ assert.match(text, /job_id=job-live/, `job_id was dropped: ${text}`);
+ assert.match(text, /NOT YET SAVED|NOT SAVED/i);
+ assert.doesNotMatch(text, /^Saved to Walrus Memory/m);
+});
diff --git a/services/server/scripts/mcp/server.ts b/services/server/scripts/mcp/server.ts
index 561d8bf55..78de98884 100644
--- a/services/server/scripts/mcp/server.ts
+++ b/services/server/scripts/mcp/server.ts
@@ -50,14 +50,14 @@ const INSTRUCTIONS = [
"summary. Skip one-off tasks, the current file or bug, and small talk. Use",
"memwal_remember_bulk when several distinct facts arrived at once.",
"",
- "A Walrus write is queued, not instant: memwal_remember and memwal_remember_bulk normally",
- "return job_ids with the facts ACCEPTED but NOT YET SAVED, and storing each takes roughly",
- "another 30-60s. A batch is slower still — its writes are stored one at a time, not",
- "together. That is the healthy path, not an error. Tell the user the facts are being saved",
- "rather than that they are saved, and do not re-send them — that queues duplicates. A job",
- "can still fail after acceptance, so when it matters that a fact landed, resolve the ids",
- "with memwal_remember_status (it takes job_id, or job_ids for a whole batch); only a",
- "blob_id means that fact is stored.",
+ "A Walrus write takes roughly 30-60s, and memwal_remember and memwal_remember_bulk wait",
+ "for it: a successful call comes back with a blob_id, and that means the fact is stored.",
+ "",
+ "If a write outruns that budget the call returns job_ids instead, saying the facts are",
+ "ACCEPTED but NOT YET SAVED. That is the exception, not the normal result. When it",
+ "happens, say the facts are being saved rather than that they are saved, do NOT re-send",
+ "them — that queues duplicates — and resolve the ids with memwal_remember_status (it takes",
+ "job_id, or job_ids for a whole batch). Only a blob_id means a fact is stored.",
"",
"RECOVER: if memwal_recall unexpectedly returns nothing for a namespace that has been used",
"before, call memwal_restore to rebuild the index from Walrus.",
diff --git a/services/server/scripts/mcp/tools/analyze.ts b/services/server/scripts/mcp/tools/analyze.ts
index c85f3530f..46cde62fc 100644
--- a/services/server/scripts/mcp/tools/analyze.ts
+++ b/services/server/scripts/mcp/tools/analyze.ts
@@ -70,7 +70,12 @@ export function registerAnalyzeTool(
"analyze this text",
),
"memwal_analyze extraction",
- { idempotent: false },
+ // Not an accept. `/api/analyze` runs the extractor LLM inline
+ // before it answers — which is why the SDK allows this call 60s
+ // where it allows a remember 30s. The 15s accept ceiling would
+ // have cut off healthy extraction on any transcript long enough
+ // to be worth extracting from.
+ { idempotent: false, deadlineMs: 60_000 },
);
const facts = accepted.facts ?? [];
diff --git a/services/server/scripts/mcp/tools/remember-bulk.ts b/services/server/scripts/mcp/tools/remember-bulk.ts
index a758e9150..605171baf 100644
--- a/services/server/scripts/mcp/tools/remember-bulk.ts
+++ b/services/server/scripts/mcp/tools/remember-bulk.ts
@@ -53,7 +53,7 @@ export function registerRememberBulkTool(
{
...TOOL_METADATA.memwal_remember_bulk,
description:
- "Save multiple durable facts in one call. Use when you learned several distinct facts at once (onboarding details, a list of preferences, decisions from a discussion). Pass an array of complete fact statements (max 20) — do not summarize. Prefer this over repeated memwal_remember calls. Walrus writes queue, so this normally returns job_ids with the facts NOT yet saved — in that case say they are being saved rather than claiming they are stored, and resolve them with memwal_remember_status.",
+ "Save multiple durable facts in one call. Use when you learned several distinct facts at once (onboarding details, a list of preferences, decisions from a discussion). Pass an array of complete fact statements (max 20) — do not summarize. Prefer this over repeated memwal_remember calls. A Walrus write takes 30-60s and this call waits for them, so a success carries blob_ids and means the facts are stored. If they outrun that budget you get job_ids and the facts are NOT yet saved — say they are being saved rather than stored, and resolve them with memwal_remember_status.",
inputSchema: REMEMBER_BULK_INPUT,
},
wrapTool<{ facts: string[]; namespace?: string }>(session, "memwal_remember_bulk", async ({ facts, namespace }) => {
diff --git a/services/server/scripts/mcp/tools/remember-wait.ts b/services/server/scripts/mcp/tools/remember-wait.ts
index 25aae7755..901963d22 100644
--- a/services/server/scripts/mcp/tools/remember-wait.ts
+++ b/services/server/scripts/mcp/tools/remember-wait.ts
@@ -135,7 +135,13 @@ export const REMEMBER_POLL_INTERVAL_MS = 400;
* caller gets a job_id to resolve later.
*/
export function isStillRunning(err: unknown): boolean {
- return (err as { status?: number } | null)?.status === 504;
+ // 504: `waitForRememberJob` hit its own budget — the job is still going.
+ if ((err as { status?: number } | null)?.status === 504) return true;
+ // Our deadline firing means the RELAYER stopped answering us, not that the
+ // job stopped. It was accepted, it is a row in `remember_jobs`, and the
+ // caller needs the job_id to settle it. Treating this as a plain error
+ // threw that id away — the one thing the pending result exists to return.
+ return (err as { name?: string } | null)?.name === "MemWalRelayerUnresponsive";
}
/**
@@ -330,16 +336,17 @@ export async function withDeadline(
export function withAcceptDeadline(
work: Promise,
what: string,
- opts: { idempotent: boolean },
+ opts: { idempotent: boolean; deadlineMs?: number },
): Promise {
+ const ms = opts.deadlineMs ?? ACCEPT_DEADLINE_MS;
const shared =
- `Walrus Memory did not accept the ${what} within ${ACCEPT_DEADLINE_MS / 1000}s — the ` +
+ `Walrus Memory did not accept the ${what} within ${ms / 1000}s — the ` +
`relayer is unreachable or not responding. The write may or may not have been queued, ` +
`so do NOT tell the user it was saved.`;
return withDeadline(
work,
- ACCEPT_DEADLINE_MS,
+ ms,
opts.idempotent
? `${shared} Retrying is safe: this write carries a content-derived idempotency ` +
`key, so a retry attaches to the job already in flight instead of queueing a ` +
diff --git a/services/server/scripts/mcp/tools/remember.ts b/services/server/scripts/mcp/tools/remember.ts
index 0e7c0dfa6..1c31bb6a5 100644
--- a/services/server/scripts/mcp/tools/remember.ts
+++ b/services/server/scripts/mcp/tools/remember.ts
@@ -54,7 +54,7 @@ export function registerRememberTool(
{
...TOOL_METADATA.memwal_remember,
description:
- "Save a durable fact about the user or project to their Walrus Memory. Call this PROACTIVELY whenever the user states a preference, decision, constraint, correction, identity detail, or recurring workflow — even if they did not say 'remember this'. Skip one-off tasks, the current file or bug, and small talk. Pass the full statement; do not summarize. To save several facts at once, use memwal_remember_bulk instead. Walrus writes queue, so this may return a job_id with the fact NOT yet saved — in that case say so rather than claiming it is stored, and resolve it with memwal_remember_status.",
+ "Save a durable fact about the user or project to their Walrus Memory. Call this PROACTIVELY whenever the user states a preference, decision, constraint, correction, identity detail, or recurring workflow — even if they did not say 'remember this'. Skip one-off tasks, the current file or bug, and small talk. Pass the full statement; do not summarize. To save several facts at once, use memwal_remember_bulk instead. A Walrus write takes 30-60s and this call waits for it, so a success carries a blob_id and means the fact is stored. If it outruns that budget you get a job_id and the fact is NOT yet saved — say so rather than claiming it is stored, and resolve it with memwal_remember_status.",
inputSchema: REMEMBER_INPUT,
},
wrapTool<{ text: string; namespace?: string }>(session, "memwal_remember", async ({ text, namespace }) => {
From c3882e21cf456ab5d50caa1ebba767bfda7f534f Mon Sep 17 00:00:00 2001
From: Harry Phan
Date: Wed, 16 Sep 2026 16:08:20 +0700
Subject: [PATCH 037/132] test(mcp): make the dropped-job_id test actually
reach the wait path
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit
The test I added in 68fd4c47 proved nothing. It lived in
remember-deadline.test.ts, which sets MEMWAL_MCP_REMEMBER_WAIT_MS=0 at the top
so the whole file can exercise the accept leg — and with a zero budget
`memwal_remember` returns at accept and never enters the wait at all. It
passed against the bug it was written for.
Caught by reverting the fix and watching the test stay green, which is the
check I should have run before claiming it covered anything.
Moved to its own file that keeps a real budget, and driven through the error
`withWaitDeadline` actually raises rather than through a hang, so it runs in
milliseconds instead of waiting out budget-plus-grace. Verified both ways this
time: it fails with the fix reverted and passes with it restored.
Two guards against the same mistake recurring. The file asserts its own budget
is non-zero, because every case in it is vacuous otherwise. And a companion
test pins the opposite direction — a job that genuinely failed (500) must stay
an error rather than being laundered into "still uploading" by the branch that
now forgives our own deadline.
---
.../mcp/__tests__/remember-deadline.test.ts | 36 -------
.../mcp/__tests__/remember-wait-hang.test.ts | 95 +++++++++++++++++++
2 files changed, 95 insertions(+), 36 deletions(-)
create mode 100644 services/server/scripts/mcp/__tests__/remember-wait-hang.test.ts
diff --git a/services/server/scripts/mcp/__tests__/remember-deadline.test.ts b/services/server/scripts/mcp/__tests__/remember-deadline.test.ts
index 8612007b7..87958265d 100644
--- a/services/server/scripts/mcp/__tests__/remember-deadline.test.ts
+++ b/services/server/scripts/mcp/__tests__/remember-deadline.test.ts
@@ -317,39 +317,3 @@ test("the absorb budget fits inside the accept deadline", () => {
`absorb ${MAX_ABSORBED_COOLDOWN_MS}ms must stay under the ${DEFAULT_ACCEPT_DEADLINE_MS}ms accept deadline`,
);
});
-
-test("a relayer that goes quiet mid-wait still hands back the job_id", async (t) => {
- // The job was accepted and is a row in remember_jobs. Our wait deadline
- // firing means the relayer stopped answering US, not that the write
- // stopped — so the caller must still get the id needed to settle it.
- // Before this, the deadline error fell through to `throw` and the id was
- // lost, which is the one thing the pending result exists to carry.
- const session = {
- oauthScope: "memwal:read memwal:write",
- namespace: "default",
- memwal: {
- async rememberAsync() {
- return { job_id: "job-live", status: "pending" };
- },
- // Accepted fine, then the relayer stops answering the poll.
- waitForRememberJob: () => new Promise(() => {}),
- },
- } as unknown as MemWalSession;
-
- const [ct, st] = InMemoryTransport.createLinkedPair();
- const server = createMcpServer(session);
- const client = new Client({ name: "quiet-relayer", version: "1.0.0" });
- t.after(async () => { await client.close(); await server.close(); });
- await server.connect(st);
- await client.connect(ct);
-
- const res = await client.callTool({
- name: "memwal_remember",
- arguments: { text: "a durable fact" },
- });
- const text = textOf(res);
-
- assert.match(text, /job_id=job-live/, `job_id was dropped: ${text}`);
- assert.match(text, /NOT YET SAVED|NOT SAVED/i);
- assert.doesNotMatch(text, /^Saved to Walrus Memory/m);
-});
diff --git a/services/server/scripts/mcp/__tests__/remember-wait-hang.test.ts b/services/server/scripts/mcp/__tests__/remember-wait-hang.test.ts
new file mode 100644
index 000000000..3bedb9448
--- /dev/null
+++ b/services/server/scripts/mcp/__tests__/remember-wait-hang.test.ts
@@ -0,0 +1,95 @@
+// The wait path only exists when there IS a wait, so this file keeps the
+// default budget rather than the accept-and-continue knob the deadline tests
+// use. Without that, `memwal_remember` returns at accept and never reaches the
+// code under test — which is exactly how the first version of this test came
+// to pass against the bug it was written for.
+process.env.MEMWAL_MCP_REMEMBER_WAIT_MS = "5000";
+
+import assert from "node:assert/strict";
+import test, { type TestContext } from "node:test";
+import { Client } from "@modelcontextprotocol/sdk/client/index.js";
+import { InMemoryTransport } from "@modelcontextprotocol/sdk/inMemory.js";
+import type { MemWalSession } from "../auth.js";
+
+const { createMcpServer } = await import("../server.js");
+const { REMEMBER_WAIT_MS } = await import("../tools/remember-wait.js");
+
+function textOf(result: unknown): string {
+ return (result as { content: Array<{ text: string }> }).content
+ .map((c) => c.text)
+ .join("\n");
+}
+
+async function clientFor(session: MemWalSession, t: TestContext): Promise {
+ const [ct, st] = InMemoryTransport.createLinkedPair();
+ const server = createMcpServer(session);
+ const client = new Client({ name: "wait-hang", version: "1.0.0" });
+ t.after(async () => { await client.close(); await server.close(); });
+ await server.connect(st);
+ await client.connect(ct);
+ return client;
+}
+
+test("this file actually exercises the wait path", () => {
+ // Guards the mistake above: if the budget is ever 0 here, every test below
+ // passes without running the code it targets.
+ assert.notEqual(REMEMBER_WAIT_MS, 0);
+});
+
+test("a relayer that goes quiet mid-wait still hands back the job_id", async (t) => {
+ // The write was accepted — it is a row in remember_jobs and still running.
+ // Our wait deadline firing means the relayer stopped answering US. The
+ // job_id is the only way to settle it, and carrying it is the entire
+ // reason the pending result exists; the deadline error used to fall
+ // through to `throw` and discard it.
+ const client = await clientFor({
+ oauthScope: "memwal:read memwal:write",
+ namespace: "default",
+ memwal: {
+ async rememberAsync() {
+ return { job_id: "job-live", status: "pending" };
+ },
+ async waitForRememberJob() {
+ // What withWaitDeadline raises once the relayer goes silent.
+ const err = new Error("Walrus Memory stopped responding while waiting");
+ err.name = "MemWalRelayerUnresponsive";
+ throw err;
+ },
+ },
+ } as unknown as MemWalSession, t);
+
+ const res = await client.callTool({
+ name: "memwal_remember",
+ arguments: { text: "a durable fact" },
+ });
+ const text = textOf(res);
+
+ assert.match(text, /job_id=job-live/, `job_id was dropped: ${text}`);
+ assert.doesNotMatch(text, /^Saved to Walrus Memory/m, "must not read as stored");
+});
+
+test("a job that genuinely failed is still an error, not a pending result", async (t) => {
+ // The counterpart: a 500 means the write is dead, so it must NOT be
+ // laundered into "still uploading" by the same branch.
+ const client = await clientFor({
+ oauthScope: "memwal:read memwal:write",
+ namespace: "default",
+ memwal: {
+ async rememberAsync() {
+ return { job_id: "job-dead", status: "pending" };
+ },
+ async waitForRememberJob() {
+ const err = new Error("remember job failed: walrus upload rejected");
+ Object.assign(err, { status: 500 });
+ throw err;
+ },
+ },
+ } as unknown as MemWalSession, t);
+
+ const res = await client.callTool({
+ name: "memwal_remember",
+ arguments: { text: "a durable fact" },
+ });
+ assert.equal((res as { isError?: boolean }).isError, true);
+ assert.match(textOf(res), /failed/i);
+});
From 0f4afd8966637693978dbde667ce8cd9e695435c Mon Sep 17 00:00:00 2001
From: Harry Phan
Date: Wed, 16 Sep 2026 16:59:49 +0700
Subject: [PATCH 038/132] fix(mcp): stop counting in-flight writes as failed,
and correct the transport doc
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit
Three of the review's suggestions. The two bugs among them:
BULK SUMMARY. `result.failed` is total-minus-succeeded, so a write still
uploading was counted as failed — while the tail of the same message said that
job was on its way. An agent reading `failed=` re-sends an in-flight write,
which is the duplicate this branch exists to avoid. The summary now counts
only terminal failures and names the in-flight ones separately.
TRANSPORT DOC. `MEMWAL_MCP_TRANSPORT=http` was documented as having "no
replay-on-reconnect". It does not: `openRelaySession` hands the Streamable
session to the same `runBridge`, which replays its in-flight map on either
transport — and on Streamable a disconnect mid-send can look sent, so a
replayed write may duplicate. Making replay transport-aware is a real change
to the reconnect path and does not belong in this PR, so the doc now says what
the code does and warns what opting in costs.
Also trims the comments the review called out — the sweeper block, FailedWrite,
and the SDK poll/idempotency essays. Kept the load-bearing constraints
(`prepare_claimed_at IS NOT NULL` so bulk and analyze rows are not swept; why
the idempotency bucket exists; that a poll loop only checks its budget between
polls) and dropped the narration.
Not addressed, because it is not mine to do: the suggestion to split this into
four PRs. It is the right call — the remember-contract piece is the one D1
deferred, and the transport rewrite is independently risky — but resequencing
someone else's PR needs the author.
---
docs/reference/environment-variables.md | 2 +-
packages/sdk/src/memwal.ts | 40 +++---------
.../server/scripts/mcp/tools/remember-bulk.ts | 14 ++++-
.../server/scripts/mcp/tools/remember-wait.ts | 17 ++---
services/server/src/storage/db.rs | 63 ++++---------------
services/server/src/types.rs | 4 ++
6 files changed, 46 insertions(+), 94 deletions(-)
diff --git a/docs/reference/environment-variables.md b/docs/reference/environment-variables.md
index ce06284fc..db603f31a 100644
--- a/docs/reference/environment-variables.md
+++ b/docs/reference/environment-variables.md
@@ -70,7 +70,7 @@ The stdio MCP package reads these environment variables directly. A CLI flag tak
| `MEMWAL_CLIENT_LABEL` | `--label ` | `MCP Client` / `Walrus Memory MCP` | Friendly delegate-key label shown in the dashboard |
| `MEMWAL_MCP_DEBUG` | none | `0` | Set to `1` for verbose stderr logging |
| `MEMWAL_CREDS_DIR` | none | `~/.memwal` | Directory holding `credentials.json`. Overrides both project-local and `~/.memwal` credentials, re-read on every access so a test can redirect it after import. Mainly for tests, which must not write into the real credential directory |
-| `MEMWAL_MCP_TRANSPORT` | none | `sse` | Which relayer transport the stdio bridge dials. `sse` uses the legacy split (`POST /api/mcp/messages` + `GET /api/mcp/sse`). `http` (aliases `streamable`, `streamable-http`) uses the Streamable HTTP endpoint `/api/mcp`, where a call is answered on the same request — no idle watchdog and no replay-on-reconnect. Unrecognised values fall back to `sse` |
+| `MEMWAL_MCP_TRANSPORT` | none | `sse` | Which relayer transport the stdio bridge dials. `sse` uses the legacy split (`POST /api/mcp/messages` + `GET /api/mcp/sse`). `http` (aliases `streamable`, `streamable-http`) uses the Streamable HTTP endpoint `/api/mcp`, where a call is answered on the same request rather than split across a POST and an SSE stream, so there is no idle watchdog. Reconnect replay is NOT yet transport-aware: the bridge still replays its in-flight map on either transport, and on Streamable a disconnect mid-send can look sent, so a replayed write may duplicate. Opt in knowing that. Unrecognised values fall back to `sse` |
| `MEMWAL_MCP_SSE_IDLE_MS` | none | `30000` | Maximum milliseconds of silence on the SSE stream before the bridge treats the session as dead and reconnects. Values below `500` are ignored and fall back to the default. Mainly for tests |
| `MEMWAL_MCP_CALL_TIMEOUT_MS` | none | `240000` | Maximum milliseconds a single request might wait for its response before the bridge answers with a retryable error. Covers a reply lost while the stream itself stays healthy, which `MEMWAL_MCP_SSE_IDLE_MS` cannot detect. The default is derived in code from the slowest server-side tool deadline plus headroom, so it moves with that tool rather than being pinned here. Values below `1000` are ignored and fall back to the default |
| `MEMWAL_MCP_THROTTLE_FLOOR_MS` | none | `5000` | Minimum milliseconds the bridge waits before retrying an SSE handshake the relayer refused with HTTP 429 and no `Retry-After`. A `Retry-After` on the response wins instead. Either way the wait is capped at `60000`. Non-numeric or negative values are ignored and fall back to the default. Mainly for tests |
diff --git a/packages/sdk/src/memwal.ts b/packages/sdk/src/memwal.ts
index 0c3d264e9..569f33fcc 100644
--- a/packages/sdk/src/memwal.ts
+++ b/packages/sdk/src/memwal.ts
@@ -137,25 +137,13 @@ function pollingDelayMs(baseMs: number, attempt: number): number {
return Math.floor(capped * jitter);
}
-/** Window over which the same (namespace, text) resolves to the same
- * idempotency key.
+/** Window over which the same (namespace, text) resolves to one key.
*
- * `pendingRememberKeys` only dedupes retries that reuse one client instance.
- * The MCP sidecar builds a fresh `MemWal` per transport session, so when the
- * bridge reconnects a dropped stream and the agent re-issues the same
- * `memwal_remember`, that map is empty. A random key then reads as a brand-new
- * write and the relayer mints a SECOND paid Walrus blob for a write already in
- * flight — doubling queue load exactly when the queue is already slow enough
- * to have caused the drop.
- *
- * Deriving the key from the content makes that replay land on the existing
- * job. The bucket bounds how long the collapse lasts: `remember_jobs` rows are
- * never pruned, so an unbucketed key would dedupe against a job from any point
- * in history and re-saving a fact the user had since deleted would return the
- * old row's blob id instead of storing it again. Retries happen seconds after
- * the original, so 30 minutes covers them with ~0.1% chance of a replay
- * straddling the boundary — and straddling only costs the old behaviour (a
- * duplicate job), never a wrong result. */
+ * `pendingRememberKeys` only dedupes retries that reuse one client instance,
+ * and callers such as the MCP sidecar build a fresh client per session — so a
+ * random key let a reconnect replay mint a second paid blob. The bucket bounds
+ * the collapse: `remember_jobs` rows are never pruned, so an unbucketed key
+ * would dedupe against a job from any point in history. */
const IDEMPOTENCY_BUCKET_MS = 30 * 60 * 1000;
async function derivedIdempotencyKey(requestIdentity: string): Promise {
@@ -163,19 +151,11 @@ async function derivedIdempotencyKey(requestIdentity: string): Promise {
return `r1-${await sha256hex(`${bucket}\0${requestIdentity}`)}`;
}
-/**
- * Deadline for a single relayer request when the call site names no other.
+/** Deadline for a request that names no other.
*
- * `fetch` imposes no timeout, so before this every request here could hang for
- * as long as the socket stayed open. That is not a theoretical gap: a poll loop
- * checks its budget at the top of each iteration, which bounds when the next
- * request STARTS, not how long one takes — so one stalled read blew straight
- * past a documented 90s cap and left an MCP tool running past 120s.
- *
- * 30s mirrors the relayer's own outbound HTTP client, so any call that depends
- * on the relayer talking to the sidecar, Walrus, or OpenAI has already failed
- * upstream by the time this fires.
- */
+ * `fetch` imposes none, and a poll loop checks its budget only between polls —
+ * so one stalled read ran past `timeoutMs` entirely. 30s mirrors the relayer's
+ * own outbound client. */
const DEFAULT_REQUEST_TIMEOUT_MS = 30_000;
/** `POST /api/restore` bounds itself at 55s server-side and answers with an
diff --git a/services/server/scripts/mcp/tools/remember-bulk.ts b/services/server/scripts/mcp/tools/remember-bulk.ts
index 605171baf..fd4a47c40 100644
--- a/services/server/scripts/mcp/tools/remember-bulk.ts
+++ b/services/server/scripts/mcp/tools/remember-bulk.ts
@@ -126,7 +126,19 @@ export function registerRememberBulkTool(
const state = r.status === "timeout" ? `still uploading, job_id=${r.id}` : r.status;
return `${i + 1}. [${state}]${blob}${err}${text ? ` — ${text}` : ""}`;
});
- const summary = `Saved ${result.succeeded}/${result.total} fact(s) to Walrus Memory (failed=${result.failed}).`;
+ // `result.failed` is total-minus-succeeded, so it counts a
+ // still-uploading write as failed — while the tail below says that
+ // same job is on its way. An agent reading `failed=` re-sends an
+ // in-flight write, which is the duplicate this branch exists to
+ // avoid. Count only what actually reached a terminal failure.
+ const reallyFailed = result.results.filter(
+ (r) => r.status !== "done" && r.status !== "timeout",
+ ).length;
+ const summary =
+ `Saved ${result.succeeded}/${result.total} fact(s) to Walrus Memory` +
+ (reallyFailed ? ` (failed=${reallyFailed})` : "") +
+ (unfinished.length ? ` (${unfinished.length} still uploading)` : "") +
+ ".";
const footer = result.succeeded > 0 ? `\n\n${explorerFooter()}` : "";
const tail = unfinished.length
? `\n\n${unfinished.length} write(s) are STILL UPLOADING and are NOT saved yet. ` +
diff --git a/services/server/scripts/mcp/tools/remember-wait.ts b/services/server/scripts/mcp/tools/remember-wait.ts
index 901963d22..70ca5fee1 100644
--- a/services/server/scripts/mcp/tools/remember-wait.ts
+++ b/services/server/scripts/mcp/tools/remember-wait.ts
@@ -405,19 +405,12 @@ function advisedCooldownMs(err: unknown): number {
return typeof secs === "number" && secs > 0 ? secs * 1000 : 1_000;
}
-/**
- * Honour the relayer's own `retry_after` instead of surfacing a raw 429.
- *
- * Observed against production: once the per-delegate-key budget (60 weighted
- * requests/minute) is spent, `memwal_remember` fails with
- * `Tool error: ... 429 ... retry_after_seconds: 60` and the fact is simply
- * never written. Nothing retried, and nothing told the user their memory had
- * been dropped — the worst failure this system has, because it is silent.
+/** Honour the relayer's `retry_after` instead of surfacing a raw 429.
*
- * Fast-return makes it likelier, not rarer: settling a batch adds requests on
- * top of the write itself, so an agent saving several facts in one turn spends
- * the budget faster than one that blocked.
- */
+ * Once the per-delegate-key budget is spent the write is simply never made,
+ * and nothing retried — the quietest failure here. A short cooldown is
+ * absorbed; a long one is reported, because sleeping it out inside a tool call
+ * is the hang this file exists to remove. */
export async function withRelayerRetry(work: () => Promise, what: string): Promise {
let last: unknown;
for (let attempt = 1; attempt <= RELAYER_RETRY_ATTEMPTS; attempt++) {
diff --git a/services/server/src/storage/db.rs b/services/server/src/storage/db.rs
index 726bebc15..9a71637bd 100644
--- a/services/server/src/storage/db.rs
+++ b/services/server/src/storage/db.rs
@@ -2205,59 +2205,22 @@ impl VectorDb {
Ok(rows)
}
- /// Mark remember jobs as failed once nothing can still move them.
+ /// Mark remember jobs as failed once nothing can still move them:
+ /// `running`/`uploaded` whose worker stopped updating them, and `pending`
+ /// rows whose preparation never finished.
///
- /// Two shapes of stuck, which need different tests:
+ /// `prepare_claimed_at IS NOT NULL` is load-bearing — only the single
+ /// `remember` path claims a preparation slot, so without it the sweep also
+ /// matches healthy `/api/remember/bulk` and `/api/analyze` rows, which
+ /// never set `preparation_encrypted_b64` at all.
///
- /// * `running` / `uploaded` — a worker claimed the job and stopped
- /// updating it.
- /// * `pending` with no preparation payload — the row was committed by the
- /// route, but the spawned preparation (summarize → embed + SEAL encrypt →
- /// enqueue) never finished, so nothing was ever queued. Preparation runs
- /// in a `tokio::spawn` inside the relayer process, so a restart in that
- /// window leaves the row behind with no task to resume it.
+ /// Clearing `prepare_claim_token` fences a slow preparation: its own
+ /// UPDATE is keyed on that token, so it can no longer reach
+ /// `enqueue_wallet_job`. Failing is the only option — the row stores
+ /// ciphertext, never plaintext, so nothing can be retried from.
///
- /// `pending` cannot be swept wholesale — a job that IS prepared sits at
- /// `pending` until a wallet worker picks it up, which under upload backlog
- /// is legitimately many minutes (`WALRUS_UPLOAD_PER_WALLET_CONCURRENCY`
- /// defaults to 1). Two columns together say "this one is never coming":
- ///
- /// * `prepare_claimed_at IS NOT NULL` — the row belongs to the single
- /// `remember` path, the only one that claims a preparation slot
- /// (`claim_remember_preparation`). This is load-bearing, not decoration:
- /// `/api/remember/bulk` and `/api/analyze` insert their rows directly and
- /// never claim, so `preparation_encrypted_b64` is ALWAYS NULL for them,
- /// healthy or not. Without this clause the sweep fails every bulk and
- /// analyze write that waits out the TTL in a normal upload backlog —
- /// killing paid work that was about to run and inviting the caller to
- /// re-send it.
- /// * `preparation_encrypted_b64 IS NULL` — that claim was never redeemed.
- /// The column is written by the statement immediately before
- /// `enqueue_wallet_job`, so its absence means the job never reached the
- /// queue.
- ///
- /// Orphaned bulk and analyze preparations are therefore still not swept.
- /// That is the pre-existing behaviour, deliberately left alone rather than
- /// guessed at: neither path persists anything that distinguishes "stranded"
- /// from "queued", so sweeping them needs a durable marker they do not yet
- /// have.
- ///
- /// Failing is the only option, not a choice. The row stores the SEAL
- /// ciphertext, never the plaintext, so a job that died before encrypting
- /// has nothing left to retry from — the fact is gone and the owner has to
- /// send it again. Saying so is strictly better than the alternative, which
- /// was a row sitting at `pending` forever while `memwal_remember_status`
- /// reported it as still uploading.
- ///
- /// Clearing `prepare_claim_token` is what makes this safe against a
- /// preparation that is merely very slow rather than dead: that task's own
- /// UPDATE is fenced on the token, so it now matches zero rows, logs the
- /// lost claim and returns *before* `enqueue_wallet_job` — it cannot mint a
- /// paid blob for a job just declared dead.
- ///
- /// Quota needs no special handling here: `main` runs
- /// `release_reservations_for_terminal_jobs` immediately after this on the
- /// same tick, which is what reclaims the bytes these rows had reserved.
+ /// Quota is reclaimed by `release_reservations_for_terminal_jobs`, which
+ /// `main` runs immediately after this on the same tick.
pub async fn fail_stale_remember_jobs(
&self,
stale_after: std::time::Duration,
diff --git a/services/server/src/types.rs b/services/server/src/types.rs
index 2e81dc368..16979037e 100644
--- a/services/server/src/types.rs
+++ b/services/server/src/types.rs
@@ -1,3 +1,7 @@
+ /// Sanitized by `sanitize_job_error_for_client`, as on every other
+ /// client-facing job-status path: an infrastructure-funding failure is
+ /// replaced wholesale (its raw text names the relayer's own wallet and
+ /// balance), and long hex runs are redacted.
use base64::Engine as _;
use serde::{Deserialize, Serialize};
use std::sync::Arc;
From 877a48f9e140a6b7c93222188ee2e47b366b36c6 Mon Sep 17 00:00:00 2001
From: Harry Phan
Date: Wed, 16 Sep 2026 17:05:07 +0700
Subject: [PATCH 039/132] fix(relayer): let a re-tried write be reported if it
fails again
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit
Two fixes from this branch are each correct alone and wrong together.
`claim_remember_preparation` resets a `failed` row for a fresh attempt: status
back to `pending`, `error_msg` cleared. The recall failure report fires once
per ROW, keyed on `failure_reported_at IS NULL`. The reset did not clear that
column, and the row is reused across attempts — so:
write fails -> recall reports it -> user re-sends -> row re-claimed and
reset -> retry fails again -> next recall skips it, because the row was
already marked reported.
The second failure is never surfaced. That is silent loss, which is the exact
thing the report was added to prevent, reached by way of the retry the report
itself asks for.
A re-claim now clears `failure_reported_at` alongside `error_msg`, under the
same `blob_id IS NULL` guard: a row being re-prepared is a fresh attempt, so
its report state belongs to the previous one.
Found by asking what happens when the claim-TTL change and the one-shot report
touch the same row, rather than by reading either in isolation — neither
file's own review would show it.
---
services/server/src/routes/remember.rs | 41 +++++++++++++++++++++++++-
1 file changed, 40 insertions(+), 1 deletion(-)
diff --git a/services/server/src/routes/remember.rs b/services/server/src/routes/remember.rs
index 57c536a2a..abf2b9540 100644
--- a/services/server/src/routes/remember.rs
+++ b/services/server/src/routes/remember.rs
@@ -1067,13 +1067,17 @@ fn should_spawn_after_reset(rows_affected: u64) -> bool {
/// so it matches zero rows and returns before `enqueue_wallet_job`.
const PREPARE_CLAIM_TTL_SECS: i64 = 60;
+/// Re-claiming clears `failure_reported_at` along with `error_msg`: the row is
+/// being reused for a fresh attempt, and the recall report only surfaces a
+/// failure once per row. Left set, a retry that failed AGAIN would never be
+/// reported — silent loss, which is the exact thing that report exists to stop.
async fn claim_remember_preparation(
pool: &sqlx::PgPool,
job_id: &str,
) -> Result, AppError> {
let token = uuid::Uuid::new_v4().to_string();
let claimed: Option = sqlx::query_scalar(
- "UPDATE remember_jobs SET prepare_claimed_at = NOW(), prepare_claim_token = $3, status = CASE WHEN status = 'failed' AND blob_id IS NULL THEN 'pending' ELSE status END, error_msg = CASE WHEN blob_id IS NULL THEN NULL ELSE error_msg END, updated_at = NOW() WHERE id = $1 AND blob_id IS NULL AND status IN ('pending', 'failed') AND (prepare_claimed_at IS NULL OR prepare_claimed_at < NOW() - make_interval(secs => $2) OR status = 'failed') RETURNING prepare_claim_token",
+ "UPDATE remember_jobs SET prepare_claimed_at = NOW(), prepare_claim_token = $3, status = CASE WHEN status = 'failed' AND blob_id IS NULL THEN 'pending' ELSE status END, error_msg = CASE WHEN blob_id IS NULL THEN NULL ELSE error_msg END, failure_reported_at = CASE WHEN blob_id IS NULL THEN NULL ELSE failure_reported_at END, updated_at = NOW() WHERE id = $1 AND blob_id IS NULL AND status IN ('pending', 'failed') AND (prepare_claimed_at IS NULL OR prepare_claimed_at < NOW() - make_interval(secs => $2) OR status = 'failed') RETURNING prepare_claim_token",
)
.bind(job_id)
.bind(PREPARE_CLAIM_TTL_SECS)
@@ -1632,6 +1636,41 @@ mod tests {
/// ACCEPTED regardless, so the caller was told a write was queued when
/// nothing was running. A job that has finished failing has no live
/// preparation to fence.
+ /// The recall failure report fires once per ROW, and a re-claim reuses the
+ /// row for a fresh attempt. If the flag survived that reset, a retry that
+ /// failed again would never be reported — silent loss, which is precisely
+ /// what the report exists to prevent. Two fixes that are each correct
+ /// alone and wrong together.
+ #[tokio::test]
+ async fn reclaiming_a_reported_failure_lets_it_be_reported_again() {
+ let pool = idem_test_pool().await;
+ let job_id = uuid::Uuid::new_v4().to_string();
+ sqlx::query(
+ "INSERT INTO remember_jobs (id, owner, namespace, status, error_msg, failure_reported_at)
+ VALUES ($1, '0xowner', 'ns', 'failed', 'first failure', NOW())",
+ )
+ .bind(&job_id)
+ .execute(&pool)
+ .await
+ .unwrap();
+
+ claim_remember_preparation(&pool, &job_id)
+ .await
+ .unwrap()
+ .expect("a failed job is re-claimable");
+
+ let reported: Option> =
+ sqlx::query_scalar("SELECT failure_reported_at FROM remember_jobs WHERE id = $1")
+ .bind(&job_id)
+ .fetch_one(&pool)
+ .await
+ .unwrap();
+ assert!(
+ reported.is_none(),
+ "a re-claimed row must be reportable again if it fails a second time",
+ );
+ }
+
#[tokio::test]
async fn a_failed_job_is_reclaimable_immediately() {
let pool = idem_test_pool().await;
From d71efdbb2fd0d7ed3ee626869829886228ecfadf Mon Sep 17 00:00:00 2001
From: Harry Phan
Date: Wed, 16 Sep 2026 17:29:39 +0700
Subject: [PATCH 040/132] fix(relayer): apply migration 021 so the failed-write
report has its column
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit
`021_failed_write_report_ack.sql` adds `remember_jobs.failure_reported_at`,
but nothing ran it. Production schema comes from the hand-written
`include_str!` chain in `VectorDb::new()`, which ended at 020 — `sqlx::migrate!`
covers only `migrations_legacy`, and the Dockerfile copies `migrations/` for
reference. The column therefore did not exist on any deploy, while two
statements shipped in the same change read and write it:
`recent_failed_remember_jobs` on every recall, and `claim_remember_preparation`
on every re-claim.
The recall side is silent — `failed_writes_for` swallows the error and returns
an empty report, so a user whose write failed is simply never told. The claim
side is not: it propagates with `?`, so a re-claim 500s.
No test covers `recent_failed_remember_jobs`, which is why CI stayed green.
---
services/server/src/storage/db.rs | 106 ++++++++++++++++++++++--------
1 file changed, 80 insertions(+), 26 deletions(-)
diff --git a/services/server/src/storage/db.rs b/services/server/src/storage/db.rs
index 9a71637bd..efd549929 100644
--- a/services/server/src/storage/db.rs
+++ b/services/server/src/storage/db.rs
@@ -1515,6 +1515,17 @@ impl VectorDb {
.await
.map_err(|e| AppError::Internal(format!("Failed to run migration 020: {}", e)))?;
+ // 021 adds `remember_jobs.failure_reported_at`, which
+ // `recent_failed_remember_jobs` writes on every recall and
+ // `claim_remember_preparation` clears on re-claim. Both are plain
+ // queries against a column that only exists if this runs, so a
+ // deploy that skips it fails those statements with 42703.
+ let migration_021 = include_str!("../../migrations/021_failed_write_report_ack.sql");
+ sqlx::raw_sql(migration_021)
+ .execute(&pool)
+ .await
+ .map_err(|e| AppError::Internal(format!("Failed to run migration 021: {}", e)))?;
+
tracing::info!("database connected and migrations applied");
Ok(Self {
@@ -1856,16 +1867,17 @@ impl VectorDb {
/// failure mode this whole path exists to remove.
/// Accepted-then-failed writes this owner has not been told about yet.
///
- /// Reporting is one-shot by construction. The rows are stamped in the same
- /// statement that returns them, so the next recall sees them acknowledged
- /// and stays quiet. Without that the same failures rode along on every
- /// recall for the full window while the message told the agent to send the
- /// facts again — so a compliant agent re-sent, the original row stayed
- /// `failed` and in-window, and the next recall asked for the same thing.
- /// Every pass was another paid Walrus write.
+ /// Read-only. Reporting is still one-shot, but the stamp is taken by
+ /// `claim_failed_write_reports` at the moment the response is built, not
+ /// here — this runs concurrently with the embed/search/decrypt that
+ /// follow, and any of those can fail or be abandoned. Stamping here meant
+ /// a recall that 500d on the embedding provider, or that the SDK aborted
+ /// at its hard 15s, still marked the rows reported: the user was never
+ /// told, on that recall or any later one, that their write had failed.
///
- /// `FOR UPDATE SKIP LOCKED` keeps two concurrent recalls from claiming the
- /// same row and both reporting it.
+ /// One-shot is preserved because the claim is a single conditional UPDATE
+ /// over these ids — a concurrent recall that got there first claims them
+ /// and this one is handed back nothing to report.
pub async fn recent_failed_remember_jobs(
&self,
owner: &str,
@@ -1882,23 +1894,13 @@ impl VectorDb {
chrono::DateTime,
),
>(
- // Claim-and-return in one statement: the UPDATE stamps the rows
- // it is about to hand back, so a second recall — or a concurrent
- // one in another session — cannot report the same failure again.
- // Doing it as two statements would leave a window where both see
- // the row unreported and both tell the agent to re-send it.
- "UPDATE remember_jobs SET failure_reported_at = NOW()
- WHERE id IN (
- SELECT id FROM remember_jobs
- WHERE owner = $1
- AND status = 'failed'
- AND failure_reported_at IS NULL
- AND updated_at >= $2
- ORDER BY updated_at DESC
- LIMIT $3
- FOR UPDATE SKIP LOCKED
- )
- RETURNING id, namespace, error_msg, updated_at",
+ "SELECT id, namespace, error_msg, updated_at FROM remember_jobs
+ WHERE owner = $1
+ AND status = 'failed'
+ AND failure_reported_at IS NULL
+ AND updated_at >= $2
+ ORDER BY updated_at DESC
+ LIMIT $3",
)
.bind(owner)
.bind(chrono::Utc::now() - chrono::Duration::from_std(window).unwrap_or_default())
@@ -1939,6 +1941,58 @@ impl VectorDb {
.collect())
}
+ /// Stamp `failure_reported_at` on the subset of `ids` not already
+ /// reported, and return the ids actually claimed.
+ ///
+ /// The conditional `failure_reported_at IS NULL` is what keeps the report
+ /// one-shot: two recalls that both read the same unreported row race here
+ /// and exactly one UPDATE matches, so only that one reports it. The other
+ /// gets an empty set and stays quiet — which matters because the report
+ /// text asks the agent to send the fact again, and a double report is a
+ /// duplicate paid Walrus write.
+ ///
+ /// Called at response construction, so a recall that never reaches the
+ /// client does not consume the report.
+ pub async fn claim_failed_write_reports(
+ &self,
+ ids: &[String],
+ ) -> Result, AppError> {
+ if ids.is_empty() {
+ return Ok(Vec::new());
+ }
+ let started = std::time::Instant::now();
+ let rows = sqlx::query_scalar::<_, String>(
+ "UPDATE remember_jobs SET failure_reported_at = NOW()
+ WHERE id = ANY($1) AND failure_reported_at IS NULL
+ RETURNING id",
+ )
+ .bind(ids)
+ .fetch_all(&self.pool)
+ .await;
+
+ match rows {
+ Ok(rows) => {
+ crate::observability::observe_db(
+ "remember_jobs.claim_failure_report",
+ "ok",
+ started.elapsed(),
+ );
+ Ok(rows)
+ }
+ Err(e) => {
+ crate::observability::observe_db(
+ "remember_jobs.claim_failure_report",
+ "error",
+ started.elapsed(),
+ );
+ Err(AppError::Internal(format!(
+ "Failed to claim failed-write reports: {}",
+ e
+ )))
+ }
+ }
+ }
+
/// Hard-delete all vector index rows for a given owner + namespace.
/// (Walrus blobs themselves persist — Walrus has no delete; this only
/// removes the local `vector_entries` rows, so the memories stop being
From 52b7e1a5edda385017daf7988ccfe1de87fe9edd Mon Sep 17 00:00:00 2001
From: Harry Phan
Date: Wed, 16 Sep 2026 17:29:39 +0700
Subject: [PATCH 041/132] fix(relayer): claim the failed-write report at
delivery, not before the work
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit
The report was stamped in the same statement that returned the rows, inside a
task spawned at the top of the recall handler so it overlaps the embed, search
and decrypt. A `tokio::spawn`ed task is not cancelled when its handle drops, so
the stamp committed whether or not the recall ever reached the client — and
three `?` early-returns sit after the spawn, as does the SDK's hard 15s abort
that recall has been measured landing on.
So an OpenAI 429 on the embedding, or a client that gave up, marked the rows
reported and returned a response carrying no report. The failure — whose whole
point is that the user does not otherwise know about it — was then never
surfaced on that recall or any later one.
Split into a read and a claim. The read runs early and keeps the overlap; the
claim is a single conditional UPDATE over those ids at response construction.
One-shot is preserved by `failure_reported_at IS NULL` in the UPDATE rather
than by doing it early: two concurrent recalls race there and exactly one wins,
so the report still cannot go out twice — which matters, because its text asks
the agent to re-send the fact and a double report is a duplicate paid write.
A claim error reports nothing rather than reporting unstamped rows, since an
unstamped report repeats on every recall for the whole window.
---
services/server/src/routes/recall.rs | 41 ++++++++++++++++++++++++++--
1 file changed, 39 insertions(+), 2 deletions(-)
diff --git a/services/server/src/routes/recall.rs b/services/server/src/routes/recall.rs
index a78fc3596..3a0c3d5b0 100644
--- a/services/server/src/routes/recall.rs
+++ b/services/server/src/routes/recall.rs
@@ -69,6 +69,42 @@ async fn failed_writes_for(state: &AppState, owner: &str) -> Vec {
}
}
+/// Claim the prefetched report at the point of delivery, and return only the
+/// rows this recall actually won.
+///
+/// `failed_writes_for` runs concurrently with the embed, search and decrypt,
+/// any of which can fail with `?` or be abandoned when the SDK hits its hard
+/// 15s abort. A `tokio::spawn`ed task is not cancelled when its handle drops,
+/// so stamping inside that task committed the acknowledgement for recalls that
+/// never returned a report — and the failure, whose whole point is that the
+/// user does not otherwise know about it, was then never surfaced again.
+///
+/// A claim error reports nothing rather than reporting unstamped rows: an
+/// unstamped report repeats on every recall for the full window, and its text
+/// asks the agent to re-send the fact, so each repeat is another paid write.
+async fn claim_for_delivery(state: &AppState, found: Vec) -> Vec {
+ if found.is_empty() {
+ return Vec::new();
+ }
+ let ids: Vec = found.iter().map(|w| w.job_id.clone()).collect();
+ match state.db.claim_failed_write_reports(&ids).await {
+ Ok(claimed) => {
+ let claimed: std::collections::HashSet = claimed.into_iter().collect();
+ found
+ .into_iter()
+ .filter(|w| claimed.contains(&w.job_id))
+ .collect()
+ }
+ Err(e) => {
+ tracing::warn!(
+ "recall: failed-write report could not be claimed, staying quiet: {}",
+ e
+ );
+ Vec::new()
+ }
+ }
+}
+
// ============================================================
// Recall query-embedding cache (Redis) — wraps the Embedder service
// ============================================================
@@ -285,7 +321,8 @@ pub async fn recall(
dropped_count: 0,
// Reported even with no hits: an empty recall is exactly when a
// caller is most likely to be looking for the fact that failed.
- failed_writes: failed_writes.await.unwrap_or_default(),
+ failed_writes: claim_for_delivery(&state, failed_writes.await.unwrap_or_default())
+ .await,
}));
}
@@ -374,7 +411,7 @@ pub async fn recall(
dropped_count,
// A panic in the report task must not take the recall with it; the
// caller loses a warning, not their memories.
- failed_writes: failed_writes.await.unwrap_or_default(),
+ failed_writes: claim_for_delivery(&state, failed_writes.await.unwrap_or_default()).await,
}))
}
From 0559ad257d74425a5eb5021e25a102fc3c87eeb6 Mon Sep 17 00:00:00 2001
From: Harry Phan
Date: Wed, 16 Sep 2026 17:29:44 +0700
Subject: [PATCH 042/132] fix(mcp): stop analyze counting in-flight writes as
failed, and keep its job_ids
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit
Two problems in `memwal_analyze`, both already solved in `memwal_remember_bulk`
and `memwal_remember`; this file was written from the same template one commit
earlier and missed the corrections.
`failed=` came straight from `result.failed`, which the SDK computes as
total-minus-succeeded — so a still-uploading write counted as a failure, while
the stragglers block directly below said those same jobs were on their way and
must not be re-sent. The normal case makes it contradict itself: one upload per
wallet at 30-60s each means a 4-fact transcript routinely lands 1 and leaves 3
queued, printing `failed=3`. An agent that believes the count either re-sends —
three duplicate paid Walrus blobs queued behind the originals — or reports data
loss that did not happen. Count only terminal failures, report the rest as
still uploading, and render them as bulk does instead of a raw `[timeout]`.
The wait also had no try/catch. `withWaitDeadline` raises
MemWalRelayerUnresponsive when the relayer goes quiet mid-poll, and letting it
propagate discarded both the job_ids and the extracted facts: nothing left to
settle the running writes with, and no way to get the extraction back without
paying the LLM for it again. Degrade to the pending branch, as
`memwal_remember` already does.
The existing test could not catch the count bug — its stub computed `failed` by
filtering on "failed", a value the real SDK never returns. Corrected, with the
two cases it should have pinned. Also drops a stale comment that told a reader
the shipped wait budget is 0; D1 left it at the 90s ceiling.
---
.../mcp/__tests__/analyze-fast-return.test.ts | 33 ++++++++--
services/server/scripts/mcp/tools/analyze.ts | 60 ++++++++++++++-----
2 files changed, 75 insertions(+), 18 deletions(-)
diff --git a/services/server/scripts/mcp/__tests__/analyze-fast-return.test.ts b/services/server/scripts/mcp/__tests__/analyze-fast-return.test.ts
index 27ee0d118..0270013b6 100644
--- a/services/server/scripts/mcp/__tests__/analyze-fast-return.test.ts
+++ b/services/server/scripts/mcp/__tests__/analyze-fast-return.test.ts
@@ -14,9 +14,10 @@
* back paired with the fact it carries, so a later partial failure is
* actionable.
*/
-// A small non-zero wait so the bounded-wait branch is reachable at all. The
-// shipped default is 0 — the tool returns at accept and never polls — which
-// would make every assertion below about a partly landed batch unreachable.
+// A small non-zero wait so the bounded-wait branch is reachable quickly. The
+// shipped default is the full 90s ceiling (D1 kept the block-to-terminal
+// contract), so leaving it unset would make every assertion below about a
+// partly landed batch wait out that ceiling instead of returning.
// Set before the dynamic import, because the budget is read once at load.
process.env.MEMWAL_MCP_REMEMBER_WAIT_MS = "500";
@@ -68,7 +69,13 @@ function sessionWith(
})),
total: states.length,
succeeded: states.filter((s) => s === "done").length,
- failed: states.filter((s) => s === "failed").length,
+ // Deliberately total-minus-succeeded, because that is what
+ // the real `waitForRememberJobs` returns — a `timeout`
+ // counts here. A stub that filtered on "failed" instead
+ // would model a value the SDK never produces, and the
+ // still-uploading-counted-as-failed case below could not
+ // fail no matter what the tool printed.
+ failed: states.length - states.filter((s) => s === "done").length,
};
},
},
@@ -138,6 +145,24 @@ test("a partly landed batch reports both halves", async (t) => {
assert.match(text, /memwal_remember_status/, "and how to settle it");
});
+test("a still-uploading write is not counted or labelled as failed", async (t) => {
+ // The straggler block already tells the agent this job is on its way and
+ // must not be re-sent. Printing `failed=1` next to it contradicts that,
+ // and an agent that believes the count re-sends — a duplicate paid Walrus
+ // write queued behind the original.
+ const text = await callAnalyze(sessionWith({ states: ["done", "timeout"] }), t);
+ assert.doesNotMatch(text, /failed=/, "a timeout is in flight, not failed");
+ assert.match(text, /1 still uploading/, "it is counted as in flight instead");
+ assert.doesNotMatch(text, /\[timeout\]/, "and not labelled with the raw status");
+ assert.match(text, /still uploading, job_id=analyze-job-2/);
+});
+
+test("a genuinely failed write is still counted as failed", async (t) => {
+ const text = await callAnalyze(sessionWith({ states: ["done", "failed"] }), t);
+ assert.match(text, /failed=1/, "a terminal failure must still be reported");
+ assert.doesNotMatch(text, /still uploading/);
+});
+
test("text with nothing worth saving says so instead of handing back an empty batch", async (t) => {
const text = await callAnalyze(sessionWith({ facts: [] }), t);
assert.match(text, /Extracted 0 facts/);
diff --git a/services/server/scripts/mcp/tools/analyze.ts b/services/server/scripts/mcp/tools/analyze.ts
index 46cde62fc..2a92b5f36 100644
--- a/services/server/scripts/mcp/tools/analyze.ts
+++ b/services/server/scripts/mcp/tools/analyze.ts
@@ -10,6 +10,7 @@ import {
withAcceptDeadline,
withRelayerRetry,
withWaitDeadline,
+ isStillRunning,
} from "./remember-wait.js";
const ANALYZE_INPUT = {
@@ -118,25 +119,56 @@ export function registerAnalyzeTool(
const namespaces = entries.map(
() => namespace ?? session.namespace ?? "default"
);
- const result = await withWaitDeadline(
- session.memwal.waitForRememberJobs(accepted.job_ids, namespaces, {
- timeoutMs: REMEMBER_WAIT_MS,
- pollIntervalMs: REMEMBER_POLL_INTERVAL_MS,
- }),
- REMEMBER_WAIT_MS,
- );
+ // The wait is a courtesy; the accept above is the part that had to
+ // succeed. If the relayer goes quiet mid-poll `withWaitDeadline`
+ // raises MemWalRelayerUnresponsive, and letting that propagate
+ // would discard both the job_ids and the extracted facts — the
+ // caller would have no way to settle writes that are still running
+ // and no way to get the extraction back without paying for it
+ // again. `memwal_remember` already degrades this way; so does this.
+ let result;
+ try {
+ result = await withWaitDeadline(
+ session.memwal.waitForRememberJobs(accepted.job_ids, namespaces, {
+ timeoutMs: REMEMBER_WAIT_MS,
+ pollIntervalMs: REMEMBER_POLL_INTERVAL_MS,
+ }),
+ REMEMBER_WAIT_MS,
+ );
+ } catch (err) {
+ if (!isStillRunning(err)) throw err;
+ return pending(Date.now() - startedAt);
+ }
const waitedMs = Date.now() - startedAt;
const unfinished = result.results.filter((r) => r.status === "timeout");
if (unfinished.length === result.results.length) return pending(waitedMs);
- const lines = result.results.map(
- (r, i) =>
- `${i + 1}. [${r.status}]${r.blob_id ? ` blob_id=${r.blob_id}` : ""} ${
- entries[i]?.text || "(unknown fact)"
- }`
- );
- const summary = `Extracted ${facts.length} fact(s) — succeeded=${result.succeeded} failed=${result.failed}`;
+ const lines = result.results.map((r, i) => {
+ // `timeout` is not a failure — the write is still running and
+ // its job_id is how the caller settles it later. Rendered the
+ // same way memwal_remember_bulk renders it.
+ const state =
+ r.status === "timeout" ? `still uploading, job_id=${r.id}` : r.status;
+ return `${i + 1}. [${state}]${r.blob_id ? ` blob_id=${r.blob_id}` : ""} ${
+ entries[i]?.text || "(unknown fact)"
+ }`;
+ });
+ // `result.failed` is total-minus-succeeded, so it counts a
+ // still-uploading write as failed — while the stragglers block
+ // below tells the agent those same jobs are on their way and must
+ // not be re-sent. An agent reading `failed=` re-sends an in-flight
+ // write, which is a duplicate paid Walrus blob queued behind the
+ // original. Count only what actually reached a terminal failure.
+ // Same correction as remember-bulk.ts; this file was written from
+ // the same template one commit earlier and missed it.
+ const reallyFailed = result.results.filter(
+ (r) => r.status !== "done" && r.status !== "timeout",
+ ).length;
+ const summary =
+ `Extracted ${facts.length} fact(s) — succeeded=${result.succeeded}` +
+ (reallyFailed ? ` failed=${reallyFailed}` : "") +
+ (unfinished.length ? ` (${unfinished.length} still uploading)` : "");
const stragglers =
unfinished.length > 0
? `\n\n${pendingBulkMessage(
From a04186ed4c1d2f3f091768de7a24164ba9e997bd Mon Sep 17 00:00:00 2001
From: Harry Phan
Date: Wed, 16 Sep 2026 17:31:12 +0700
Subject: [PATCH 043/132] fix(mcp): keep bulk's job_ids when the relayer goes
quiet, and correct the wait doc
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit
`memwal_remember_bulk` had the same unguarded wait as analyze. Its comment is
right that `waitForRememberJobs` reports stragglers per item rather than
throwing — but the throw comes from `withWaitDeadline` wrapped around it, before
any per-item result exists, so a relayer that goes silent mid-poll discarded
every job_id in the batch at once. `memwal_remember` already degrades to its
pending branch here.
Tested in remember-wait-hang.test.ts rather than the bulk file, because that one
pins MEMWAL_MCP_REMEMBER_WAIT_MS=0 and so cannot reach the wait path at all —
the same shape that let the first version of the single-write test pass against
the bug it was written for. Both new cases were confirmed to fail with the
guards removed.
The env-var reference also documented MEMWAL_MCP_REMEMBER_WAIT_MS as defaulting
to `0` and advised leaving it there, while `parseWaitBudget(undefined)` returns
90000. An operator reading it would expect ~1s calls and see the tool block to
terminal. Corrected to the real default, and analyze added to the list of tools
it governs.
---
docs/reference/environment-variables.md | 2 +-
.../mcp/__tests__/remember-wait-hang.test.ts | 65 +++++++++++++++++++
.../server/scripts/mcp/tools/remember-bulk.ts | 32 ++++++---
3 files changed, 88 insertions(+), 11 deletions(-)
diff --git a/docs/reference/environment-variables.md b/docs/reference/environment-variables.md
index db603f31a..e72c2a90e 100644
--- a/docs/reference/environment-variables.md
+++ b/docs/reference/environment-variables.md
@@ -155,7 +155,7 @@ These are not all enforced at boot, but most real deployments need them.
| `MCP_MAX_TOTAL_SESSIONS` | `1000` | Maximum active MCP sessions across SSE and Streamable HTTP transports |
| `MCP_MAX_SESSIONS_PER_IP` | `16` | Maximum active MCP sessions from one source IP |
| `MCP_MAX_NEW_SESSIONS_PER_IP_PER_MIN` | `30` | Maximum new MCP sessions opened by one source IP per minute |
-| `MEMWAL_MCP_REMEMBER_WAIT_MS` | `0` | How long `memwal_remember` / `memwal_remember_bulk` wait for a write to land before returning the job_id instead. `0` returns once the relayer has durably accepted the job (~1s). A value between the two is the worst of both — against a 30-75s completion spread it pays the wait and still returns pending — so either leave it at `0` or set it high enough to mean it. Clamped to `90000`; invalid values fall back to the default |
+| `MEMWAL_MCP_REMEMBER_WAIT_MS` | `90000` | How long `memwal_remember` / `memwal_remember_bulk` / `memwal_analyze` wait for a write to land before returning the job_id instead. The default is the full ceiling, so a successful call carries a real `blob_id` and the agent can say the fact is stored; a write slower than that returns a job_id and says plainly it is not saved yet. `0` returns as soon as the relayer has durably accepted the job (~1s) — accept-and-continue, chosen knowingly. A value between the two is the worst of both: against a 30-75s completion spread it pays the wait and still returns pending. Clamped to `90000`; invalid values fall back to the default |
| `MEMWAL_MCP_ACCEPT_DEADLINE_MS` | `15000` | How long a single relayer request may stall before the MCP tool gives up on it. The SDK passes an abort signal only on `recall`, so `rememberAsync` and the job-status reads have no deadline of their own and `MEMWAL_MCP_REMEMBER_WAIT_MS` bounds only when the next poll starts, not how long one takes — without this a stalled socket keeps a tool running indefinitely. Raise it only if a slow link makes healthy accepts exceed it; invalid or non-positive values fall back to the default |
| `MCP_TOOL_SLOW_WARN_MS` | `5000` | An MCP tool call still running at this duration is logged as `tool.slow` (`settled: false`) at `warn` — which is what makes a hang visible, since a hang never settles. A call that finishes at or above it is logged as `tool.slow` (`settled: true`) instead of `tool.done` |
| `TRUSTED_PROXY_HOPS` | `0` | Number of trusted reverse-proxy hops to walk from the right of `X-Forwarded-For`; `0` ignores XFF and uses the TCP peer |
diff --git a/services/server/scripts/mcp/__tests__/remember-wait-hang.test.ts b/services/server/scripts/mcp/__tests__/remember-wait-hang.test.ts
index 3bedb9448..82ca4ccb1 100644
--- a/services/server/scripts/mcp/__tests__/remember-wait-hang.test.ts
+++ b/services/server/scripts/mcp/__tests__/remember-wait-hang.test.ts
@@ -93,3 +93,68 @@ test("a job that genuinely failed is still an error, not a pending result", asyn
assert.equal((res as { isError?: boolean }).isError, true);
assert.match(textOf(res), /failed/i);
});
+
+test("a relayer that goes quiet mid-wait still hands back every bulk job_id", async (t) => {
+ // Same hole as the single-write case, on the path where it costs most: a
+ // batch loses N job_ids at once, and `waitForRememberJobs` reporting
+ // stragglers per item does not help — the throw comes from the deadline
+ // wrapper around it, before any per-item result exists.
+ const client = await clientFor({
+ oauthScope: "memwal:read memwal:write",
+ namespace: "default",
+ memwal: {
+ async rememberBulkAsync() {
+ return { job_ids: ["bulk-1", "bulk-2"], total: 2, status: "accepted" };
+ },
+ async waitForRememberJobs() {
+ const err = new Error("Walrus Memory stopped responding while waiting");
+ err.name = "MemWalRelayerUnresponsive";
+ throw err;
+ },
+ },
+ } as unknown as MemWalSession, t);
+
+ const text = textOf(
+ await client.callTool({
+ name: "memwal_remember_bulk",
+ arguments: { facts: ["first fact", "second fact"] },
+ }),
+ );
+ assert.match(text, /bulk-1/, `job_ids were dropped: ${text}`);
+ assert.match(text, /bulk-2/, `job_ids were dropped: ${text}`);
+ assert.doesNotMatch(text, /^Saved \d+\/\d+/m, "must not read as stored");
+});
+
+test("a relayer that goes quiet mid-wait keeps analyze's job_ids AND its facts", async (t) => {
+ // Analyze pays an LLM extraction before the writes are queued. Throwing
+ // away the wait discarded both the job_ids and that extraction, so the
+ // caller could neither settle the running writes nor recover the facts
+ // without paying for them again.
+ const client = await clientFor({
+ oauthScope: "memwal:read memwal:write",
+ namespace: "default",
+ memwal: {
+ async analyze() {
+ return {
+ job_ids: ["an-1", "an-2"],
+ facts: [{ text: "drinks oat milk" }, { text: "ships on Fridays" }],
+ status: "accepted",
+ };
+ },
+ async waitForRememberJobs() {
+ const err = new Error("Walrus Memory stopped responding while waiting");
+ err.name = "MemWalRelayerUnresponsive";
+ throw err;
+ },
+ },
+ } as unknown as MemWalSession, t);
+
+ const text = textOf(
+ await client.callTool({
+ name: "memwal_analyze",
+ arguments: { text: "a transcript worth extracting from" },
+ }),
+ );
+ assert.match(text, /an-1/, `job_ids were dropped: ${text}`);
+ assert.match(text, /drinks oat milk/, `extracted facts were dropped: ${text}`);
+});
diff --git a/services/server/scripts/mcp/tools/remember-bulk.ts b/services/server/scripts/mcp/tools/remember-bulk.ts
index fd4a47c40..486f20ce9 100644
--- a/services/server/scripts/mcp/tools/remember-bulk.ts
+++ b/services/server/scripts/mcp/tools/remember-bulk.ts
@@ -10,6 +10,7 @@ import {
withAcceptDeadline,
withWaitDeadline,
withRelayerRetry,
+ isStillRunning,
} from "./remember-wait.js";
const REMEMBER_BULK_INPUT = {
@@ -95,16 +96,27 @@ export function registerRememberBulkTool(
const namespaces = items.map(
(item) => item.namespace ?? session.namespace ?? "default"
);
- // Unlike `waitForRememberJob`, this never throws on expiry — it
- // reports the stragglers as `timeout` per item, so a batch can come
- // back part landed and part still in flight.
- const result = await withWaitDeadline(
- session.memwal.waitForRememberJobs(accepted.job_ids, namespaces, {
- timeoutMs: REMEMBER_WAIT_MS,
- pollIntervalMs: REMEMBER_POLL_INTERVAL_MS,
- }),
- REMEMBER_WAIT_MS,
- );
+ // `waitForRememberJobs` never throws on expiry — it reports the
+ // stragglers as `timeout` per item, so a batch can come back part
+ // landed and part still in flight. `withWaitDeadline` around it
+ // does throw, though: a relayer that goes quiet mid-poll raises
+ // MemWalRelayerUnresponsive, and letting that propagate discards
+ // every job_id in the batch, leaving the caller nothing to settle
+ // accepted writes with. `memwal_remember` already degrades to its
+ // pending branch here; so does this.
+ let result;
+ try {
+ result = await withWaitDeadline(
+ session.memwal.waitForRememberJobs(accepted.job_ids, namespaces, {
+ timeoutMs: REMEMBER_WAIT_MS,
+ pollIntervalMs: REMEMBER_POLL_INTERVAL_MS,
+ }),
+ REMEMBER_WAIT_MS,
+ );
+ } catch (err) {
+ if (!isStillRunning(err)) throw err;
+ return pending(Date.now() - startedAt);
+ }
const waitedMs = Date.now() - startedAt;
const unfinished = result.results.flatMap((r, i) =>
From bc7bf462b414569005356e55e28af07d15d882fd Mon Sep 17 00:00:00 2001
From: Harry Phan
Date: Wed, 16 Sep 2026 17:55:14 +0700
Subject: [PATCH 044/132] docs(mcp): record the 60s client-timeout constraint
next to the wait budget
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit
The MCP TypeScript SDK defaults a tools/call to DEFAULT_REQUEST_TIMEOUT_MSEC =
60_000 (shared/protocol.js:8, applied at :712). Our default wait is 90_000, so
on a host that sets no timeout of its own the pending branch — whose whole
purpose is to hand back a job_id — arrives 30s after the client gave up.
docs/troubleshooting/overview.md already records the symptom, including that
retrying duplicates the memory; the code said nothing about it.
No behaviour change. Four independent fixes were designed for this and every one
was refuted, so this records what is now verified rather than guessing at the
remedy:
- The legs are sequential, so the ceiling is ACCEPT_DEADLINE_MS (15_000) + the
budget + WAIT_OVERSHOOT_GRACE_MS (10_000). Fitting under 60_000 needs a budget
of 35_000 or less, against a measured 26.7-47.9s time-to-saved — which is the
D1 trade, not a free win.
- memwal_analyze cannot be fixed by this constant at all: its 60_000ms
extraction deadline fully precedes the wait, so it exceeds 60s for any value
here including 0.
- Hosts differ; Claude Code sets its own far larger ceiling, so this does not
bite there.
Lowering the default is a D1-level product decision and belongs with whoever
owns that, not in a follow-up fix branch.
---
.../server/scripts/mcp/tools/remember-wait.ts | 21 +++++++++++++++++++
1 file changed, 21 insertions(+)
diff --git a/services/server/scripts/mcp/tools/remember-wait.ts b/services/server/scripts/mcp/tools/remember-wait.ts
index 70ca5fee1..fefc19875 100644
--- a/services/server/scripts/mcp/tools/remember-wait.ts
+++ b/services/server/scripts/mcp/tools/remember-wait.ts
@@ -79,6 +79,27 @@ export const MAX_REMEMBER_WAIT_MS = 90_000;
* The pending branch is still reachable and still correct — a write slower
* than the ceiling returns a job_id and says plainly it is not saved yet —
* it is simply no longer the default path.
+ *
+ * KNOWN CONSTRAINT, not yet acted on. The MCP TypeScript SDK defaults a
+ * tools/call to `DEFAULT_REQUEST_TIMEOUT_MSEC = 60_000`
+ * (@modelcontextprotocol/sdk, shared/protocol.js:8, applied at :712 as
+ * `options?.timeout ?? DEFAULT_REQUEST_TIMEOUT_MSEC`). A host that sets no
+ * timeout of its own therefore aborts at 60s, and the pending branch — whose
+ * entire purpose is to hand back a job_id — lands 30s after it has already
+ * given up. docs/troubleshooting/overview.md records the symptom already
+ * ("can exceed the MCP host's tool-call timeout"), including that a retry
+ * duplicates the memory. Hosts differ: Claude Code sets its own, far larger
+ * ceiling, so this does not bite there.
+ *
+ * The arithmetic for anyone changing this: the legs are sequential, so the
+ * ceiling is ACCEPT_DEADLINE_MS (15_000) + this budget +
+ * WAIT_OVERSHOOT_GRACE_MS (10_000). Fitting under 60_000 needs a budget of
+ * 35_000 or less — against a measured 26.7-47.9s time-to-saved, which is the
+ * D1 trade, not a free win.
+ *
+ * This constant cannot fix `memwal_analyze` either way: its extraction leg is
+ * a 60_000ms deadline (analyze.ts) that fully precedes the wait, so analyze
+ * exceeds 60s for ANY value here, including 0. That needs its own decision.
*/
const DEFAULT_REMEMBER_WAIT_MS = MAX_REMEMBER_WAIT_MS;
From 7a033c9fb6f97eea992ba501d68ec5d28a3a22eb Mon Sep 17 00:00:00 2001
From: Harry Phan
Date: Wed, 16 Sep 2026 17:59:18 +0700
Subject: [PATCH 045/132] fix(relayer): give the remember test pool migration
021, and cover the report split
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit
Wiring 021 into VectorDb::new() fixed production but not the four red tests:
`idem_test_pool` (routes/remember.rs) builds its own minimal schema from
migrations 005, 012 and 013 by include_str! and never calls VectorDb::new(), so
a migration added to that chain does not reach it. Hence they kept failing with
42703 `column "failure_reported_at" does not exist` after the previous commit.
Both are needed: the chain is what a real deploy runs, this helper is what these
tests run. `jobs.rs::test_pool` has the same shape but exercises no path that
touches the column, and its tests pass — left alone rather than changed on
speculation.
Also adds the first test for either half of the failed-write report.
`recent_failed_remember_jobs` and `claim_failed_write_reports` had no coverage at
all, which is why splitting them could be got wrong silently. The new case pins
the two properties the split depends on: the read is non-consuming, so a recall
that dies on an embedding 429 or is abandoned at the SDK's hard abort leaves the
failure still reportable; and the claim is exclusive, so two concurrent recalls
cannot both tell the agent to re-send the same fact.
Needs Postgres, so it runs in CI rather than locally.
---
services/server/src/routes/remember.rs | 11 ++++
services/server/src/storage/db.rs | 75 ++++++++++++++++++++++++++
2 files changed, 86 insertions(+)
diff --git a/services/server/src/routes/remember.rs b/services/server/src/routes/remember.rs
index abf2b9540..a27bfa6b1 100644
--- a/services/server/src/routes/remember.rs
+++ b/services/server/src/routes/remember.rs
@@ -1937,6 +1937,17 @@ mod tests {
.execute(&pool)
.await
.unwrap();
+ // 021 adds `failure_reported_at`, which `claim_remember_preparation`
+ // clears on re-claim. This helper builds its own minimal schema rather
+ // than going through `VectorDb::new()`, so a migration wired into that
+ // chain does not reach it — every column a test in this module touches
+ // has to be listed here explicitly.
+ sqlx::raw_sql(include_str!(
+ "../../migrations/021_failed_write_report_ack.sql"
+ ))
+ .execute(&pool)
+ .await
+ .unwrap();
pool
}
diff --git a/services/server/src/storage/db.rs b/services/server/src/storage/db.rs
index efd549929..9741bcc1d 100644
--- a/services/server/src/storage/db.rs
+++ b/services/server/src/storage/db.rs
@@ -3434,6 +3434,81 @@ mod quota_admission_tests {
.expect("test database must be reachable with pgvector installed")
}
+ /// The failed-write report must survive a recall that never lands, and
+ /// must still go out only once.
+ ///
+ /// Reading and claiming used to be one statement, stamped inside a task
+ /// spawned before the embed/search/decrypt that can each fail with `?` or
+ /// be abandoned when the SDK hits its hard abort. That burned the
+ /// acknowledgement for recalls that returned no report at all, and the
+ /// failure — which the user has no other way to learn about — was never
+ /// surfaced again. Splitting them is only safe if the read is genuinely
+ /// non-consuming and the claim is genuinely exclusive; both are asserted
+ /// here because nothing else covers either function.
+ #[tokio::test]
+ async fn failed_write_report_reads_freely_but_claims_once() {
+ let db = test_db().await;
+ let owner = unique_owner("failed-write-report");
+ let job_id = uuid::Uuid::new_v4().to_string();
+ sqlx::query(
+ "INSERT INTO remember_jobs (id, owner, namespace, status, error_msg)
+ VALUES ($1, $2, 'ns', 'failed', 'walrus upload rejected')",
+ )
+ .bind(&job_id)
+ .bind(&owner)
+ .execute(db.pool())
+ .await
+ .unwrap();
+
+ let window = std::time::Duration::from_secs(24 * 60 * 60);
+
+ // Read twice. A recall that dies after this point must leave the row
+ // reportable, so neither read may consume it.
+ for attempt in 0..2 {
+ let found = db
+ .recent_failed_remember_jobs(&owner, window, 5)
+ .await
+ .unwrap();
+ assert_eq!(
+ found.len(),
+ 1,
+ "read {} consumed the report; an abandoned recall would lose it",
+ attempt,
+ );
+ assert_eq!(found[0].job_id, job_id);
+ }
+
+ // Claiming is what acknowledges it, and only the first claim wins —
+ // the report text asks the agent to re-send the fact, so a second
+ // report is a duplicate paid Walrus write.
+ let first = db
+ .claim_failed_write_reports(&[job_id.clone()])
+ .await
+ .unwrap();
+ assert_eq!(first, vec![job_id.clone()], "the first claim must win");
+
+ let second = db
+ .claim_failed_write_reports(&[job_id.clone()])
+ .await
+ .unwrap();
+ assert!(
+ second.is_empty(),
+ "a second claim must report nothing, got {:?}",
+ second,
+ );
+
+ // And the row is now invisible to the read, so later recalls stay quiet.
+ let after = db
+ .recent_failed_remember_jobs(&owner, window, 5)
+ .await
+ .unwrap();
+ assert!(
+ after.is_empty(),
+ "a claimed failure must not be read again, got {:?}",
+ after,
+ );
+ }
+
/// Unique per test so concurrent runs cannot see each other's rows, and so
/// a failed run leaves no state that poisons the next one.
fn unique_owner(tag: &str) -> String {
From f2e461d631f7bc67cef20c1a7c767c2dbf399514 Mon Sep 17 00:00:00 2001
From: Harry Phan
Date: Wed, 16 Sep 2026 18:03:17 +0700
Subject: [PATCH 046/132] fix(relayer): serialise VectorDb::new() in tests so
the migration chain stops racing
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit
Two stale-sweep tests failed on an unreachable database, not on anything they
assert:
Failed to run migration 011: duplicate key value violates unique
constraint "pg_type_typname_nsp_index"
DETAIL: Key (typname, typnamespace)=(mcp_oauth_clients, 2200) already exists
`CREATE TABLE IF NOT EXISTS` is not atomic in Postgres: two tests calling
`VectorDb::new()` at the same moment both pass the existence check and one loses
on the `pg_type` insert. Every statement in the chain is idempotent, but the
chain is not concurrency-safe.
The race is pre-existing — `jobs::tests` already serialises its pool behind a
DB_SETUP_LOCK for this reason, while both `test_db()` helpers in this file did
not. It stayed latent because it needs two builds to overlap; adding one test to
the first module was enough to make it fire. Same guard, declared once so both
modules share it, held only across `VectorDb::new()` rather than the test body.
Confirms the previous commit: the four `failure_reported_at` failures are gone
(783 passed before, 786 after) and these two are a different fault.
---
services/server/src/storage/db.rs | 21 +++++++++++++++++++++
1 file changed, 21 insertions(+)
diff --git a/services/server/src/storage/db.rs b/services/server/src/storage/db.rs
index 9741bcc1d..d120e684b 100644
--- a/services/server/src/storage/db.rs
+++ b/services/server/src/storage/db.rs
@@ -16,6 +16,19 @@ pub struct VectorDb {
storage_alerts: Option<(Arc, String)>,
}
+/// Serialises `VectorDb::new()` across the test binary.
+///
+/// The migration chain is idempotent per statement but NOT safe to run
+/// concurrently: `CREATE TABLE IF NOT EXISTS` is not atomic in Postgres, so two
+/// tests that build a db at the same moment race inside migration 011 and one
+/// loses with `duplicate key value violates unique constraint
+/// "pg_type_typname_nsp_index"` on `mcp_oauth_clients`. Tests then fail on an
+/// unreachable database rather than on anything they assert. `jobs::tests`
+/// already guards its pool this way; these modules did not, which left the race
+/// latent until a test was added.
+#[cfg(test)]
+static DB_SETUP_LOCK: std::sync::OnceLock> = std::sync::OnceLock::new();
+
impl VectorDb {
pub fn with_storage_alerts(self, alerts: Arc, sui_network: String) -> Self {
Self {
@@ -3429,6 +3442,10 @@ mod quota_admission_tests {
}
async fn test_db() -> VectorDb {
+ let _guard = super::DB_SETUP_LOCK
+ .get_or_init(|| tokio::sync::Mutex::new(()))
+ .lock()
+ .await;
VectorDb::new(&test_database_url())
.await
.expect("test database must be reachable with pgvector installed")
@@ -3923,6 +3940,10 @@ mod stale_sweep_tests {
}
async fn test_db() -> VectorDb {
+ let _guard = super::DB_SETUP_LOCK
+ .get_or_init(|| tokio::sync::Mutex::new(()))
+ .lock()
+ .await;
VectorDb::new(&test_database_url())
.await
.expect("test database must be reachable with pgvector installed")
From 8fe4fee427e562ba4635e917e52393987cd307d8 Mon Sep 17 00:00:00 2001
From: Harry Phan
Date: Thu, 17 Sep 2026 11:08:06 +0700
Subject: [PATCH 047/132] refactor(server): make the migration pipeline a
checked list
021_failed_write_report_ack.sql reached dev without being wired into
VectorDb::new, so the column it creates never existed: four
routes::remember tests failed with 42703, and any deploy of that build
would have failed the same statements in production. The wiring was 24
hand-written blocks, one per migration, with nothing checking that list
against the directory.
Replace the blocks with three ordered tables and a run_migrations helper.
The order is unchanged item for item, including the two Rust steps
(backfill_updated_at, recover_invalid_pagination_index) and
014_storage_reservations running ahead of 014_memory_read_api_columns.
The pipeline stays hand-ordered rather than a sqlx::migrate! directory
scan: that would reject the duplicated 014 version outright, and cannot
interleave the two Rust steps. What is no longer manual is completeness
-- every_migration_file_is_wired_into_the_pipeline reads the migrations
directory and fails if a .sql file is missing from the tables. It needs
no database, so it runs in the ordinary Clippy + Unit tests job.
Test helpers now resolve migrations by name through the same tables
instead of repeating include_str! lists, so a rename fails loudly by
name rather than later as a missing column.
---
services/server/src/storage/db.rs | 415 +++++++++++++++---------------
1 file changed, 207 insertions(+), 208 deletions(-)
diff --git a/services/server/src/storage/db.rs b/services/server/src/storage/db.rs
index d120e684b..4745f7fd7 100644
--- a/services/server/src/storage/db.rs
+++ b/services/server/src/storage/db.rs
@@ -11,6 +11,128 @@ use crate::types::{AppError, SearchHit};
/// background sweep. Keep a single constant so the two cannot drift.
pub const TOMBSTONE_RETENTION: chrono::Duration = chrono::Duration::days(30);
+/// One migration file, paired with the name the pipeline reports it by.
+type Migration = (&'static str, &'static str);
+
+/// Pairs a migration's file name with its embedded contents so the two
+/// cannot drift apart.
+macro_rules! migration {
+ ($file:literal) => {
+ ($file, include_str!(concat!("../../migrations/", $file)))
+ };
+}
+
+/// The migration pipeline, split at the two points where `VectorDb::new`
+/// has to run Rust in between files.
+///
+/// This stays an explicit, hand-ordered list rather than a
+/// `sqlx::migrate!` directory scan, for three reasons that all still
+/// hold:
+///
+/// 1. `014_storage_reservations.sql` must run *before*
+/// `014_memory_read_api_columns.sql` — the reverse of their
+/// alphabetical order.
+/// 2. Two files share the version number `014`, which `sqlx::migrate!`
+/// rejects outright.
+/// 3. `backfill_updated_at` and `recover_invalid_pagination_index` are
+/// Rust steps that have to land between specific files.
+///
+/// What is no longer manual is *completeness*: every `.sql` file in
+/// `services/server/migrations` must appear in one of these three
+/// slices, and `every_migration_file_is_wired_into_the_pipeline` fails
+/// the test suite if one does not. Migration 021 reached dev unwired
+/// precisely because nothing checked that.
+const MIGRATIONS_BEFORE_BACKFILL: &[Migration] = &[
+ migration!("001_init.sql"),
+ migration!("002_add_namespace.sql"),
+ migration!("003_rate_limiter.sql"),
+ migration!("004_delegate_key_cache_expires.sql"),
+ migration!("005_remember_jobs.sql"),
+ // composite index on (owner, status, updated_at DESC) for bulk poll
+ migration!("006_bulk_remember.sql"),
+ // collapse per-wallet Apalis queues to a single `wallet_jobs` queue.
+ // Equivocation locks are no longer a practical concern on Sui (per
+ // Will Bradley, Mysten, 2026-05-12); concurrent workers on one wallet
+ // + retry handling is sufficient.
+ migration!("007_collapse_wallet_queues.sql"),
+ // nullable `plaintext` column for benchmark-mode storage
+ // (PlaintextEngine). NULL for all production rows — additive.
+ // Renumbered from 007 -> 008 during rebase onto dev to avoid collision
+ // with the wallet-queue collapse migration.
+ migration!("008_benchmark_plaintext.sql"),
+ // importance signal column on vector_entries.
+ migration!("009_importance_signal.sql"),
+ // Permanent restore-failure negative cache (GH #501 / WALM-299).
+ migration!("010_restore_failed_blobs.sql"),
+ // MCP OAuth 2.1 (Claude custom connectors): client registry,
+ // server-custodied delegate keys, and authorization state.
+ migration!("011_mcp_oauth.sql"),
+ // Durable idempotency, preparation, and paid-upload recovery state.
+ migration!("012_remember_write_idempotency.sql"),
+ // Build the owner/key uniqueness constraint without blocking writes.
+ migration!("013_remember_write_idempotency_index.sql"),
+ // per-owner storage quota reservations. Makes quota admission atomic
+ // with the eventual insert (GH #532 / WALM-359).
+ migration!("014_storage_reservations.sql"),
+ // owner-scoped read API: updated_at cursor column + agent_id/package_id.
+ // Split across 014-019 (see each file's header, and
+ // backfill_updated_at's / recover_invalid_pagination_index's doc
+ // comments below) to avoid holding ACCESS EXCLUSIVE across the
+ // full-table backfill or the index build.
+ migration!("014_memory_read_api_columns.sql"),
+];
+
+/// Applied after `backfill_updated_at`: 015 validates NOT NULL and will
+/// error if any `updated_at` row is still NULL.
+const MIGRATIONS_AFTER_BACKFILL: &[Migration] =
+ &[migration!("015_memory_read_api_updated_at_not_null.sql")];
+
+/// Applied after `recover_invalid_pagination_index`, which must precede
+/// 016's `CREATE INDEX CONCURRENTLY IF NOT EXISTS` — that would
+/// otherwise silently no-op forever against a permanently INVALID index
+/// left behind by an interrupted build.
+const MIGRATIONS_AFTER_INDEX_RECOVERY: &[Migration] = &[
+ // keyset-pagination index for the memories listing endpoint.
+ // Must stay in its own file/transaction — see 016's header comment.
+ migration!("016_memory_read_api_index.sql"),
+ // per-memory expiry columns.
+ migration!("017_memory_expiry_columns.sql"),
+ // index on expiry_synced_at so the periodic expiry refresh sweep
+ // doesn't full-scan vector_entries every tick. Must stay in its own
+ // file/transaction — see 018's header comment.
+ migration!("018_memory_expiry_synced_at_index.sql"),
+ // Finalizes updated_at NOT NULL cheaply using the validated CHECK
+ // constraint 015 set up — see 019's header.
+ migration!("019_memory_read_api_updated_at_set_not_null.sql"),
+ migration!("020_read_api_followups.sql"),
+ // 021 adds `remember_jobs.failure_reported_at`, which
+ // `recent_failed_remember_jobs` writes on every recall and
+ // `claim_remember_preparation` clears on re-claim. Both are plain
+ // queries against a column that only exists if this runs, so a
+ // deploy that skips it fails those statements with 42703.
+ migration!("021_failed_write_report_ack.sql"),
+];
+
+/// Every migration the pipeline applies, in the order it applies them.
+#[cfg(test)]
+fn all_migrations() -> impl Iterator- {
+ MIGRATIONS_BEFORE_BACKFILL
+ .iter()
+ .chain(MIGRATIONS_AFTER_BACKFILL)
+ .chain(MIGRATIONS_AFTER_INDEX_RECOVERY)
+}
+
+/// Applies one slice of the pipeline, naming the file that failed.
+async fn run_migrations(pool: &PgPool, migrations: &[Migration]) -> Result<(), AppError> {
+ for (name, sql) in migrations {
+ sqlx::raw_sql(sql)
+ .execute(pool)
+ .await
+ .map_err(|e| AppError::Internal(format!("Failed to run migration {}: {}", name, e)))?;
+ }
+ Ok(())
+}
+
pub struct VectorDb {
pool: PgPool,
storage_alerts: Option<(Arc
, String)>,
@@ -66,6 +188,65 @@ mod tests {
static VECTOR_SCHEMA_SETUP_LOCK: OnceLock> = OnceLock::new();
+ /// Looks a migration up in the pipeline by file name.
+ ///
+ /// Test helpers below build deliberately partial schemas — only the
+ /// tables a given module touches — so they cannot just replay the
+ /// whole pipeline. Going through this lookup still keeps them from
+ /// drifting: a renamed or deleted migration panics here by name
+ /// instead of failing later as a missing column.
+ fn migration_sql(name: &str) -> &'static str {
+ super::all_migrations()
+ .find(|(n, _)| *n == name)
+ .map(|(_, sql)| *sql)
+ .unwrap_or_else(|| panic!("migration {name} is not wired into the pipeline"))
+ }
+
+ /// Every `.sql` file in `services/server/migrations` must be wired
+ /// into the pipeline.
+ ///
+ /// Regression test for migration 021: the file was added and merged
+ /// to dev, but never listed in `VectorDb::new`, so the column it
+ /// creates never existed. Four `routes::remember` tests failed with
+ /// `column "failure_reported_at" does not exist` (42703), and any
+ /// deploy of that build would have failed the same statements in
+ /// production. Needs no database.
+ #[test]
+ fn every_migration_file_is_wired_into_the_pipeline() {
+ use std::collections::BTreeSet;
+
+ let dir = concat!(env!("CARGO_MANIFEST_DIR"), "/migrations");
+ let on_disk: BTreeSet = std::fs::read_dir(dir)
+ .expect("migrations directory should be readable")
+ .map(|entry| {
+ entry
+ .expect("migrations directory entry should be readable")
+ .file_name()
+ .to_string_lossy()
+ .into_owned()
+ })
+ .filter(|name| name.ends_with(".sql"))
+ .collect();
+
+ let wired: BTreeSet = super::all_migrations()
+ .map(|(name, _)| (*name).to_owned())
+ .collect();
+
+ let unwired: Vec<&String> = on_disk.difference(&wired).collect();
+ assert!(
+ unwired.is_empty(),
+ "migration file(s) exist on disk but are not applied by VectorDb::new: {unwired:?}. \
+ Add them to MIGRATIONS_BEFORE_BACKFILL / _AFTER_BACKFILL / \
+ _AFTER_INDEX_RECOVERY in the position the pipeline needs."
+ );
+
+ let missing: Vec<&String> = wired.difference(&on_disk).collect();
+ assert!(
+ missing.is_empty(),
+ "pipeline references migration file(s) that no longer exist: {missing:?}"
+ );
+ }
+
fn test_database_url() -> Option {
std::env::var("DATABASE_URL").ok()
}
@@ -87,13 +268,13 @@ mod tests {
.lock()
.await;
for migration in [
- include_str!("../../migrations/001_init.sql"),
- include_str!("../../migrations/002_add_namespace.sql"),
- include_str!("../../migrations/003_rate_limiter.sql"),
- include_str!("../../migrations/008_benchmark_plaintext.sql"),
- include_str!("../../migrations/009_importance_signal.sql"),
- include_str!("../../migrations/010_restore_failed_blobs.sql"),
- include_str!("../../migrations/014_memory_read_api_columns.sql"),
+ migration_sql("001_init.sql"),
+ migration_sql("002_add_namespace.sql"),
+ migration_sql("003_rate_limiter.sql"),
+ migration_sql("008_benchmark_plaintext.sql"),
+ migration_sql("009_importance_signal.sql"),
+ migration_sql("010_restore_failed_blobs.sql"),
+ migration_sql("014_memory_read_api_columns.sql"),
] {
sqlx::raw_sql(migration).execute(&pool).await.unwrap();
}
@@ -104,23 +285,21 @@ mod tests {
// CONCURRENTLY IF NOT EXISTS.
super::backfill_updated_at(&pool).await.unwrap();
- sqlx::raw_sql(include_str!(
- "../../migrations/015_memory_read_api_updated_at_not_null.sql"
- ))
- .execute(&pool)
- .await
- .unwrap();
+ sqlx::raw_sql(migration_sql("015_memory_read_api_updated_at_not_null.sql"))
+ .execute(&pool)
+ .await
+ .unwrap();
super::recover_invalid_pagination_index(&pool)
.await
.unwrap();
for migration in [
- include_str!("../../migrations/016_memory_read_api_index.sql"),
- include_str!("../../migrations/017_memory_expiry_columns.sql"),
- include_str!("../../migrations/018_memory_expiry_synced_at_index.sql"),
- include_str!("../../migrations/019_memory_read_api_updated_at_set_not_null.sql"),
- include_str!("../../migrations/020_read_api_followups.sql"),
+ migration_sql("016_memory_read_api_index.sql"),
+ migration_sql("017_memory_expiry_columns.sql"),
+ migration_sql("018_memory_expiry_synced_at_index.sql"),
+ migration_sql("019_memory_read_api_updated_at_set_not_null.sql"),
+ migration_sql("020_read_api_followups.sql"),
] {
sqlx::raw_sql(migration).execute(&pool).await.unwrap();
}
@@ -199,7 +378,7 @@ mod tests {
async fn oauth_test_db() -> Option {
let db = test_db().await?;
- sqlx::raw_sql(include_str!("../../migrations/011_mcp_oauth.sql"))
+ sqlx::raw_sql(migration_sql("011_mcp_oauth.sql"))
.execute(db.pool())
.await
.expect("OAuth migration must create tables on a fresh test database");
@@ -943,9 +1122,9 @@ mod tests {
async fn remember_jobs_test_db() -> Option {
let db = test_db().await?;
for migration in [
- include_str!("../../migrations/005_remember_jobs.sql"),
- include_str!("../../migrations/012_remember_write_idempotency.sql"),
- include_str!("../../migrations/013_remember_write_idempotency_index.sql"),
+ migration_sql("005_remember_jobs.sql"),
+ migration_sql("012_remember_write_idempotency.sql"),
+ migration_sql("013_remember_write_idempotency_index.sql"),
] {
sqlx::raw_sql(migration).execute(db.pool()).await.unwrap();
}
@@ -1343,201 +1522,21 @@ impl VectorDb {
.await
.map_err(|e| AppError::Internal(format!("Failed to connect to database: {}", e)))?;
- // Run migrations
- let migration_001 = include_str!("../../migrations/001_init.sql");
- sqlx::raw_sql(migration_001)
- .execute(&pool)
- .await
- .map_err(|e| AppError::Internal(format!("Failed to run migration 001: {}", e)))?;
-
- let migration_002 = include_str!("../../migrations/002_add_namespace.sql");
- sqlx::raw_sql(migration_002)
- .execute(&pool)
- .await
- .map_err(|e| AppError::Internal(format!("Failed to run migration 002: {}", e)))?;
-
- let migration_003 = include_str!("../../migrations/003_rate_limiter.sql");
- sqlx::raw_sql(migration_003)
- .execute(&pool)
- .await
- .map_err(|e| AppError::Internal(format!("Failed to run migration 003: {}", e)))?;
-
- let migration_004 = include_str!("../../migrations/004_delegate_key_cache_expires.sql");
- sqlx::raw_sql(migration_004)
- .execute(&pool)
- .await
- .map_err(|e| AppError::Internal(format!("Failed to run migration 004: {}", e)))?;
-
- let migration_005 = include_str!("../../migrations/005_remember_jobs.sql");
- sqlx::raw_sql(migration_005)
- .execute(&pool)
- .await
- .map_err(|e| AppError::Internal(format!("Failed to run migration 005: {}", e)))?;
-
- // composite index on (owner, status, updated_at DESC) for bulk poll
- let migration_006 = include_str!("../../migrations/006_bulk_remember.sql");
- sqlx::raw_sql(migration_006)
- .execute(&pool)
- .await
- .map_err(|e| AppError::Internal(format!("Failed to run migration 006: {}", e)))?;
-
- // collapse per-wallet Apalis queues to a single `wallet_jobs`
- // queue. Equivocation locks are no longer a practical concern on Sui
- // (per Will Bradley, Mysten, 2026-05-12); concurrent workers on one
- // wallet + retry handling is sufficient.
- let migration_007 = include_str!("../../migrations/007_collapse_wallet_queues.sql");
- sqlx::raw_sql(migration_007)
- .execute(&pool)
- .await
- .map_err(|e| AppError::Internal(format!("Failed to run migration 007: {}", e)))?;
-
- // nullable `plaintext` column for benchmark-mode storage
- // (PlaintextEngine). NULL for all production rows — additive.
- // Renumbered from 007 → 008 during rebase onto dev to avoid collision
- // with the wallet-queue collapse migration.
- let migration_008 = include_str!("../../migrations/008_benchmark_plaintext.sql");
- sqlx::raw_sql(migration_008)
- .execute(&pool)
- .await
- .map_err(|e| AppError::Internal(format!("Failed to run migration 008: {}", e)))?;
-
- // importance signal column on vector_entries.
- let migration_009 = include_str!("../../migrations/009_importance_signal.sql");
- sqlx::raw_sql(migration_009)
- .execute(&pool)
- .await
- .map_err(|e| AppError::Internal(format!("Failed to run migration 009: {}", e)))?;
-
- // Permanent restore-failure negative cache (GH #501 / WALM-299).
- let migration_010 = include_str!("../../migrations/010_restore_failed_blobs.sql");
- sqlx::raw_sql(migration_010)
- .execute(&pool)
- .await
- .map_err(|e| AppError::Internal(format!("Failed to run migration 010: {}", e)))?;
-
- // MCP OAuth 2.1 (Claude custom connectors): client registry,
- // server-custodied delegate keys, and authorization state.
- let migration_011 = include_str!("../../migrations/011_mcp_oauth.sql");
- sqlx::raw_sql(migration_011)
- .execute(&pool)
- .await
- .map_err(|e| AppError::Internal(format!("Failed to run migration 011: {}", e)))?;
-
- // Durable idempotency, preparation, and paid-upload recovery state.
- let migration_012 = include_str!("../../migrations/012_remember_write_idempotency.sql");
- sqlx::raw_sql(migration_012)
- .execute(&pool)
- .await
- .map_err(|e| AppError::Internal(format!("Failed to run migration 012: {}", e)))?;
-
- // Build the owner/key uniqueness constraint without blocking writes.
- let migration_013 =
- include_str!("../../migrations/013_remember_write_idempotency_index.sql");
- sqlx::raw_sql(migration_013)
- .execute(&pool)
- .await
- .map_err(|e| AppError::Internal(format!("Failed to run migration 013: {}", e)))?;
-
- // per-owner storage quota reservations. Makes quota admission atomic
- // with the eventual insert (GH #532 / WALM-359).
- let migration_014_reservations =
- include_str!("../../migrations/014_storage_reservations.sql");
- sqlx::raw_sql(migration_014_reservations)
- .execute(&pool)
- .await
- .map_err(|e| {
- AppError::Internal(format!(
- "Failed to run migration 014 (storage reservations): {}",
- e
- ))
- })?;
-
- // owner-scoped read API: updated_at cursor column + agent_id/package_id.
- // Split across 014-019 (see each file's header, and
- // backfill_updated_at's / recover_invalid_pagination_index's doc
- // comments above) to avoid holding ACCESS EXCLUSIVE across the
- // full-table backfill or index build.
- let migration_014_read_api =
- include_str!("../../migrations/014_memory_read_api_columns.sql");
- sqlx::raw_sql(migration_014_read_api)
- .execute(&pool)
- .await
- .map_err(|e| {
- AppError::Internal(format!(
- "Failed to run migration 014 (read API columns): {}",
- e
- ))
- })?;
+ // Run migrations. The ordering, and the two Rust steps woven
+ // between these slices, are load-bearing — see
+ // MIGRATIONS_BEFORE_BACKFILL's comment.
+ run_migrations(&pool, MIGRATIONS_BEFORE_BACKFILL).await?;
// Backfill runs as batched Rust code, not a migration file, since
// Postgres can't COMMIT mid-loop inside a plain migration
// statement — see backfill_updated_at()'s doc comment.
backfill_updated_at(&pool).await?;
- // Requires the backfill above to have already completed — this
- // validates NOT NULL and will error if any updated_at row is
- // still NULL.
- let migration_015 =
- include_str!("../../migrations/015_memory_read_api_updated_at_not_null.sql");
- sqlx::raw_sql(migration_015)
- .execute(&pool)
- .await
- .map_err(|e| AppError::Internal(format!("Failed to run migration 015: {}", e)))?;
+ run_migrations(&pool, MIGRATIONS_AFTER_BACKFILL).await?;
- // Must run before migration 016's CREATE INDEX CONCURRENTLY IF NOT
- // EXISTS, which would otherwise silently no-op forever against a
- // permanently INVALID index from an interrupted build.
recover_invalid_pagination_index(&pool).await?;
- // keyset-pagination index for the memories listing endpoint.
- // Must stay in its own file/transaction — see 016's header comment.
- let migration_016 = include_str!("../../migrations/016_memory_read_api_index.sql");
- sqlx::raw_sql(migration_016)
- .execute(&pool)
- .await
- .map_err(|e| AppError::Internal(format!("Failed to run migration 016: {}", e)))?;
-
- // per-memory expiry columns.
- let migration_017 = include_str!("../../migrations/017_memory_expiry_columns.sql");
- sqlx::raw_sql(migration_017)
- .execute(&pool)
- .await
- .map_err(|e| AppError::Internal(format!("Failed to run migration 017: {}", e)))?;
-
- // index on expiry_synced_at so the periodic expiry refresh sweep
- // doesn't full-scan vector_entries every tick. Must stay
- // in its own file/transaction — see 018's header comment.
- let migration_018 = include_str!("../../migrations/018_memory_expiry_synced_at_index.sql");
- sqlx::raw_sql(migration_018)
- .execute(&pool)
- .await
- .map_err(|e| AppError::Internal(format!("Failed to run migration 018: {}", e)))?;
-
- // Finalizes updated_at NOT NULL cheaply using the validated CHECK
- // constraint 015 set up — see 019's header.
- let migration_019 =
- include_str!("../../migrations/019_memory_read_api_updated_at_set_not_null.sql");
- sqlx::raw_sql(migration_019)
- .execute(&pool)
- .await
- .map_err(|e| AppError::Internal(format!("Failed to run migration 019: {}", e)))?;
-
- let migration_020 = include_str!("../../migrations/020_read_api_followups.sql");
- sqlx::raw_sql(migration_020)
- .execute(&pool)
- .await
- .map_err(|e| AppError::Internal(format!("Failed to run migration 020: {}", e)))?;
-
- // 021 adds `remember_jobs.failure_reported_at`, which
- // `recent_failed_remember_jobs` writes on every recall and
- // `claim_remember_preparation` clears on re-claim. Both are plain
- // queries against a column that only exists if this runs, so a
- // deploy that skips it fails those statements with 42703.
- let migration_021 = include_str!("../../migrations/021_failed_write_report_ack.sql");
- sqlx::raw_sql(migration_021)
- .execute(&pool)
- .await
- .map_err(|e| AppError::Internal(format!("Failed to run migration 021: {}", e)))?;
+ run_migrations(&pool, MIGRATIONS_AFTER_INDEX_RECOVERY).await?;
tracing::info!("database connected and migrations applied");
From d1df49f974b2d9a9a8f671738be666f777c7cc5a Mon Sep 17 00:00:00 2001
From: Harry Phan
Date: Thu, 17 Sep 2026 10:57:35 +0700
Subject: [PATCH 048/132] fix(mcp): stop cold start advertising tools the
relayer may not serve
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit
The bridge ships on npm and updates itself; a relayer ships per
environment and does not. So 0.0.14-dev.0 talked to prod and staging
still on 0.0.13, and its static cold-start list named
`memwal_remember_status` — which neither registers. The pending-write
wording then told the agent to go call it. One live run spent 90.67s on
a tool that does not exist there before erroring: the call was forwarded
into a session that would never answer it and sat in `inFlight` until
the orphan sweeper's deadline (its 60s ceiling plus 30s headroom).
Cold start cannot know what the relayer serves, so treat it as a floor
rather than a forecast. `BASELINE_RELAYER_TOOLS` is what the oldest
supported relayer registers; the cold-start lists carry only that, and
their descriptions name nothing outside it. Newer tools are not lost —
they arrive a beat later on the relayer's own `tools/list`, which
already triggers `notifications/tools/list_changed`. A name belongs in
the baseline once a prod release carries it, not when it lands on dev.
Skew can still arrive by other routes (an agent holding a list from
before a redeploy), so the bridge now records what the connected relayer
advertised and answers a call for anything outside that set itself,
naming the tools that do exist and saying plainly that nothing ran. A
stale tool list is not a transport fault and should not be priced like
one.
Tests: the cold-start set must stay inside what a relayer serves and
cover the baseline (the old assertion demanded an exact match with the
current sidecar, which is what encoded the bug); no cold-start
description may name an unadvertised tool; and an end-to-end test drives
a 0.0.13-shaped relayer to prove the refusal is local and immediate
rather than orphan-swept.
Closes #928
---
packages/mcp/CHANGELOG.md | 1 +
packages/mcp/src/auth-required.ts | 48 ++-
packages/mcp/src/bridge.ts | 55 ++++
packages/mcp/test/coldstart-init.test.mjs | 56 ++--
packages/mcp/test/tool-definitions.test.mjs | 62 +++-
packages/mcp/test/tool-not-served.test.mjs | 342 ++++++++++++++++++++
6 files changed, 537 insertions(+), 27 deletions(-)
create mode 100644 packages/mcp/test/tool-not-served.test.mjs
diff --git a/packages/mcp/CHANGELOG.md b/packages/mcp/CHANGELOG.md
index 0c06b4def..b47ac5fc1 100644
--- a/packages/mcp/CHANGELOG.md
+++ b/packages/mcp/CHANGELOG.md
@@ -4,6 +4,7 @@
### Fixed
+- The cold-start tool list no longer advertises a tool the relayer may not serve. The bridge ships on npm and updates itself while a relayer ships per environment, so 0.0.14-dev.0 dialled prod and staging still on 0.0.13: cold start named `memwal_remember_status`, which neither registers, and the pending-write wording sent the agent to go call it — one live run spent 90.67s there before erroring. Cold start is now a floor rather than a forecast (`BASELINE_RELAYER_TOOLS`): it carries only what the oldest supported relayer serves and its descriptions name nothing outside it, while newer tools still reach the client a beat later on the relayer's own `tools/list`. And a call for a tool the connected relayer does not serve is now answered locally and at once — naming the tools that do exist and saying plainly that nothing ran — instead of being forwarded into a wait that only ends at the orphan deadline. (#928)
- Bound every relayer call the tools make. The pinned SDK aborts a request only when the caller passes a signal, which none of the memory methods do, so a stalled socket kept a tool running with no ceiling — `memwal_remember` was observed still going past 120s against a 90s budget. Accepts are bounded at 15s (`MEMWAL_MCP_ACCEPT_DEADLINE_MS`), waits at their own budget plus grace. The request is not cancelled — the SDK exposes no way to pass a signal — but the agent is no longer held by it.
- Honour the relayer's `retry_after` instead of dropping the write. Once the per-delegate-key budget (60 weighted requests/minute) is spent the relayer answers 429 with a cooldown, and nothing backed off: the fact was never written and the agent saw only an opaque tool error. A short cooldown is now absorbed; a long one is reported with the wait named, stating plainly that the fact was NOT saved and pointing at the cheaper shape — one `memwal_remember_bulk` rather than N single calls, one `memwal_remember_status(job_ids)` rather than N status calls. Only rejections that provably never reached the handler retry, so `/api/remember/bulk`, which carries no idempotency key, cannot be duplicated by a retry.
- `memwal_remember` sends a content-derived idempotency key, so the retry its own timeout message invites really does attach to the job already in flight instead of storing a second paid copy. The key is computed by the tool rather than relied on from the SDK, whose published build mints a random UUID per client instance.
diff --git a/packages/mcp/src/auth-required.ts b/packages/mcp/src/auth-required.ts
index 6d2ef257f..f396be359 100644
--- a/packages/mcp/src/auth-required.ts
+++ b/packages/mcp/src/auth-required.ts
@@ -41,7 +41,7 @@ interface RpcMessage {
const SIGNED_OUT_REMEMBER =
"Save a fact to the user's Walrus Memory personal memory. Call ONLY when the user explicitly asks to remember/save something. Pass the full, detailed text — never summarize.";
const SIGNED_IN_REMEMBER =
- "Save a durable fact about the user or project to their Walrus Memory. Call this PROACTIVELY whenever the user states a preference, decision, constraint, correction, identity detail, or recurring workflow — even if they did not say 'remember this'. Skip one-off tasks, the current file or bug, and small talk. Pass the full statement; do not summarize. To save several facts at once, use memwal_remember_bulk instead. A Walrus write takes 30-60s and this call waits for it, so a success carries a blob_id and means the fact is stored. If it outruns that budget you get a job_id and the fact is NOT yet saved — say so rather than claiming it is stored, and resolve it with memwal_remember_status.";
+ "Save a durable fact about the user or project to their Walrus Memory. Call this PROACTIVELY whenever the user states a preference, decision, constraint, correction, identity detail, or recurring workflow — even if they did not say 'remember this'. Skip one-off tasks, the current file or bug, and small talk. Pass the full statement; do not summarize. To save several facts at once, use memwal_remember_bulk instead. A Walrus write takes 30-60s and this call waits for it, so a success carries a blob_id and means the fact is stored. If it outruns that budget you get a job_id and the fact is NOT yet saved — say so rather than claiming it is stored, and settle it with the job-status tool this server advertises (re-list tools if you do not see one).";
const SIGNED_OUT_RECALL =
"Search the user's Walrus Memory for facts relevant to a query. Returns matching memories ranked by relevance.";
const SIGNED_IN_RECALL =
@@ -72,7 +72,7 @@ function buildToolDefinitions(proactive: boolean) {
title: "Remember Multiple Facts",
annotations: { readOnlyHint: false, destructiveHint: false },
description:
- "Save multiple durable facts in one call. Use when you learned several distinct facts at once (onboarding details, a list of preferences, decisions from a discussion). Pass an array of complete fact statements (max 20) — do not summarize. Prefer this over repeated memwal_remember calls. A Walrus write takes 30-60s and this call waits for them, so a success carries blob_ids and means the facts are stored. If they outrun that budget you get job_ids and the facts are NOT yet saved — say they are being saved rather than stored, and resolve them with memwal_remember_status.",
+ "Save multiple durable facts in one call. Use when you learned several distinct facts at once (onboarding details, a list of preferences, decisions from a discussion). Pass an array of complete fact statements (max 20) — do not summarize. Prefer this over repeated memwal_remember calls. A Walrus write takes 30-60s and this call waits for them, so a success carries blob_ids and means the facts are stored. If they outrun that budget you get job_ids and the facts are NOT yet saved — say they are being saved rather than stored, and settle them with the job-status tool this server advertises (re-list tools if you do not see one).",
inputSchema: {
type: "object",
properties: {
@@ -198,15 +198,55 @@ function buildToolDefinitions(proactive: boolean) {
];
}
+/** Tool names the OLDEST relayer this bridge still talks to serves.
+ *
+ * The bridge ships on npm and updates itself; a relayer ships per environment
+ * and does not, so the bridge is routinely NEWER than the relayer it dials —
+ * 0.0.14-dev.0 against prod and staging on 0.0.13, which is GH #928. The
+ * cold-start list is served before any relayer capability is known, so every
+ * name in it that the dialled relayer does not serve is a tool the agent can
+ * be told about and then cannot call.
+ *
+ * Cold start is therefore a FLOOR, not a forecast: add a name here only once
+ * the tool has shipped to prod, never when it lands on dev. Newer tools reach
+ * the client a beat later anyway — the relayer's own `tools/list` replaces
+ * this one and `notifications/tools/list_changed` tells the client to re-read
+ * it. `memwal_remember_status` is the worked example: it exists on dev, not on
+ * prod/staging, and belongs here only after a prod release carries it. */
+export const BASELINE_RELAYER_TOOLS: ReadonlySet = new Set([
+ "memwal_remember",
+ "memwal_remember_bulk",
+ "memwal_recall",
+ "memwal_analyze",
+ "memwal_restore",
+ "memwal_health",
+]);
+
+/** Served by this process, so no relayer has to know about it. Advertising it
+ * at cold start is always safe. */
+const LOCALLY_SERVED_TOOLS: ReadonlySet = new Set(["memwal_login"]);
+
+/** Every tool this bridge can describe, baseline or not. Only the baseline
+ * subset is advertised at cold start; this is what the schema tests pin, so a
+ * newer tool's shape stays reviewed while it waits for a prod release. */
+export const ALL_TOOL_DEFINITIONS = buildToolDefinitions(true);
+
+/** Drop anything the oldest supported relayer would not serve. */
+function coldStartTools(proactive: boolean) {
+ return buildToolDefinitions(proactive).filter(
+ (t) => BASELINE_RELAYER_TOOLS.has(t.name) || LOCALLY_SERVED_TOOLS.has(t.name),
+ );
+}
+
/** Signed-in cold-start list (bridge). Credentials exist; the relayer session
* is not up yet. Proactive wording so clients that keep the first tools/list
* still save/recall without being asked. */
-export const TOOL_DEFINITIONS = buildToolDefinitions(true);
+export const TOOL_DEFINITIONS = coldStartTools(true);
/** Signed-out list (auth-required). No credentials, so every memory call
* fails: keep conservative wording or the model will spam remember and get
* a stream of auth errors. */
-export const SIGNED_OUT_TOOL_DEFINITIONS = buildToolDefinitions(false);
+export const SIGNED_OUT_TOOL_DEFINITIONS = coldStartTools(false);
/** How long to wait for the local listener to bind + emit its URL before we
* give up and return an error. Should be near-instant; 5s is paranoia. */
diff --git a/packages/mcp/src/bridge.ts b/packages/mcp/src/bridge.ts
index 91485c115..7c5f69010 100644
--- a/packages/mcp/src/bridge.ts
+++ b/packages/mcp/src/bridge.ts
@@ -214,6 +214,18 @@ const SIGNED_OUT_FAILURE = {
const UNAUTHORIZED_TEXT =
"❌ Walrus Memory rejected the saved credentials (HTTP 401). The delegate key may have been revoked or is no longer registered on this account. Call `memwal_login` to sign in again — saved credentials were NOT modified.";
+/** Reply for a `tools/call` naming a tool the connected relayer does not
+ * serve. Names what IS on offer, because the agent's next move is to pick one
+ * of those — "unknown tool" alone leaves it guessing or retrying. */
+export function unknownToolText(name: string, available: string[]): string {
+ return (
+ `❌ \`${name}\` is not a tool this Walrus Memory server offers. ` +
+ `Available: ${available.join(", ")}. ` +
+ `Your tool list is stale — re-read \`tools/list\` and use one of those instead. ` +
+ `Nothing ran, so nothing was saved or changed.`
+ );
+}
+
/** `failRequest` options for every credentials-rejected refusal, so one refused
* at handshake time and one refused on arrival afterwards read identically. */
const UNAUTHORIZED_FAILURE = {
@@ -1348,6 +1360,12 @@ export async function runBridge(
* client surfaces them in its tool palette. */
const pendingListIds = new Set();
+ /** Tool names the CONNECTED relayer advertised on its last `tools/list`,
+ * minus the ones we serve locally. Empty until the client has listed tools
+ * at least once over a live session — until then we know nothing about the
+ * relayer's capabilities and gate nothing. */
+ const upstreamToolNames = new Set();
+
/** IDs of forwarded `memwal_health` calls, each against the relayer URL the
* call went out on. Captured at send time rather than read at reply time so
* a reconnect that swapped credentials mid-flight cannot label the answer
@@ -1857,6 +1875,16 @@ export async function runBridge(
(t) => !LOCAL_TOOL_NAMES.has(t.name ?? ""),
);
result.tools = [...upstream, ...LOCAL_TOOL_DEFINITIONS];
+ // Record what this relayer actually serves. A
+ // later call for a name absent here is answered
+ // locally instead of being forwarded into a wait
+ // no reply will ever end.
+ upstreamToolNames.clear();
+ for (const t of upstream) {
+ if (typeof t.name === "string" && t.name !== "") {
+ upstreamToolNames.add(t.name);
+ }
+ }
}
}
if (
@@ -2035,6 +2063,33 @@ export async function runBridge(
return;
}
+ // Version skew: this bridge ships on npm and updates itself,
+ // while a relayer ships per environment and does not, so the
+ // bridge is routinely newer than the server it dials. A tool it
+ // advertised at cold start can be missing from the session that
+ // actually came up — `memwal_remember_status` against a prod
+ // relayer, which is GH #928. Forwarding that call parks it in
+ // `inFlight` until the orphan sweeper's deadline (60s + 30s
+ // headroom for that tool), and the user reads the 90s as a hang.
+ // A stale tool list is not a transport fault: say so now, while
+ // the agent can still act on it.
+ if (msg.method === "tools/call" && msg.id != null && upstreamToolNames.size > 0) {
+ const called = (msg.params as { name?: string } | undefined)?.name;
+ if (
+ typeof called === "string" &&
+ !upstreamToolNames.has(called) &&
+ !LOCAL_TOOL_NAMES.has(called)
+ ) {
+ const available = [...upstreamToolNames, ...LOCAL_TOOL_NAMES].sort();
+ log.warn("bridge.tool_not_served", { tool: called });
+ failRequest(msg, "tool not served", {
+ toolText: unknownToolText(called, available),
+ errorMessage: `${called} is not served by this Walrus Memory relayer`,
+ });
+ return;
+ }
+ }
+
// Fill in the configured default namespace for memory tool
// calls that didn't pass one. Mutates msg in place so the
// forwarded — and any replayed-on-reconnect — copy carries it.
diff --git a/packages/mcp/test/coldstart-init.test.mjs b/packages/mcp/test/coldstart-init.test.mjs
index 46303cb23..da894a1fe 100644
--- a/packages/mcp/test/coldstart-init.test.mjs
+++ b/packages/mcp/test/coldstart-init.test.mjs
@@ -27,6 +27,8 @@ import { tmpdir } from "node:os";
import { join, dirname, resolve } from "node:path";
import { fileURLToPath } from "node:url";
+import { BASELINE_RELAYER_TOOLS } from "../dist/auth-required.js";
+
const __dirname = dirname(fileURLToPath(import.meta.url));
const BIN = resolve(__dirname, "../dist/bin/memwal-mcp.js");
const EXPECTED_BEARER = "a".repeat(64);
@@ -37,10 +39,12 @@ const EXPECTED_ACCOUNT_ID = "0x" + "3".repeat(64);
* `initialize` would blow the assertion deadlines below. */
const SSE_DELAY_MS = 3_000;
-/** The tools the real relayer sidecar registers
- * (services/server/scripts/mcp/tools/index.ts). The cold-start static list must
- * cover exactly these (plus the locally-served login/logout), so the
- * static→refreshed transition doesn't change the tool set under the client. */
+/** The tools the CURRENT relayer sidecar registers
+ * (services/server/scripts/mcp/tools/index.ts) — i.e. a relayer as new as this
+ * bridge. The cold-start static list must be a SUBSET of this: it may lag the
+ * sidecar (a newer tool simply shows up on the post-connect re-list), but it
+ * must never advertise a name the relayer does not serve. A bridge is routinely
+ * newer than the relayer it dials, and over-advertising is GH #928. */
const UPSTREAM_TOOL_NAMES = [
"memwal_remember",
"memwal_remember_bulk",
@@ -325,13 +329,13 @@ test("initialize is answered locally during a slow relayer cold start; tools/cal
`initialize took ${initElapsed}ms — expected it answered locally, well before the ${SSE_DELAY_MS}ms relayer connect`,
);
- // 2) tools/list at cold start is served locally and instantly with the
- // static list, which must be EXACTLY the upstream tool set plus the
- // locally-served login/logout — each name once.
+ // 2) tools/list at cold start is served locally and instantly from the
+ // static list. It must fit INSIDE the upstream tool set (plus the
+ // locally-served login/logout) and cover the baseline — each name once.
send({ jsonrpc: "2.0", id: 2, method: "tools/list", params: {} });
const list = await waitFor((m) => m.id === 2 && m.result, SSE_DELAY_MS);
const coldNames = list.result.tools.map((t) => t.name);
- const expectedNames = new Set([...UPSTREAM_TOOL_NAMES, "memwal_login", "memwal_logout"]);
+ const servableNames = new Set([...UPSTREAM_TOOL_NAMES, "memwal_login", "memwal_logout"]);
// Unique names (TOOL_DEFINITIONS bundles its own memwal_login; a blind concat
// with the local login/logout defs would list it twice).
assert.equal(
@@ -339,12 +343,21 @@ test("initialize is answered locally during a slow relayer cold start; tools/cal
coldNames.length,
`cold tools/list has duplicate tool names: ${coldNames}`,
);
- // Exact set match — guards against the cold list drifting from the real
- // upstream registration (e.g. missing memwal_remember_bulk / memwal_health).
+ // Over-advertising is the GH #928 failure: the agent is handed a name the
+ // relayer cannot answer, and the call waits out the orphan deadline.
+ const overAdvertised = coldNames.filter((n) => !servableNames.has(n));
+ assert.deepEqual(
+ overAdvertised,
+ [],
+ `cold tools/list advertises tools no relayer serves: ${overAdvertised}`,
+ );
+ // Under-advertising below the baseline is the opposite drift: a tool every
+ // supported relayer has, missing for the whole cold-start window.
+ const missingBaseline = [...BASELINE_RELAYER_TOOLS].filter((n) => !coldNames.includes(n));
assert.deepEqual(
- new Set(coldNames),
- expectedNames,
- `cold tools/list set mismatch. got ${[...coldNames].sort()}, expected ${[...expectedNames].sort()}`,
+ missingBaseline,
+ [],
+ `cold tools/list omits baseline tools: ${missingBaseline}`,
);
// 3) tools/call sent BEFORE the stream is up must be buffered and served
@@ -375,9 +388,10 @@ test("initialize is answered locally during a slow relayer cold start; tools/cal
assert.equal(initReplies[0].msg.result.serverInfo.name, "memwal");
// 6) After connect, a re-list is forwarded upstream and spliced with
- // login/logout. That authoritative set must EQUAL the cold static set —
- // the static→refreshed transition must not change the tool set (each
- // name once, no dup even if upstream ever served login).
+ // login/logout. That authoritative set is what the client acts on, and
+ // every cold-start name must still be in it — the transition may ADD
+ // tools (a relayer newer than the baseline) but must never take one
+ // away under a client that already read the cold list.
send({ jsonrpc: "2.0", id: 4, method: "tools/list", params: {} });
const relist = await waitFor((m) => m.id === 4 && m.result, 10_000);
const splicedNames = relist.result.tools.map((t) => t.name);
@@ -386,10 +400,16 @@ test("initialize is answered locally during a slow relayer cold start; tools/cal
splicedNames.length,
`post-connect tools/list has duplicate tool names: ${splicedNames}`,
);
+ const withdrawn = coldNames.filter((n) => !splicedNames.includes(n));
+ assert.deepEqual(
+ withdrawn,
+ [],
+ `post-connect tools/list withdrew cold-start tools: ${withdrawn}. cold=${[...coldNames].sort()} spliced=${[...splicedNames].sort()}`,
+ );
assert.deepEqual(
new Set(splicedNames),
- new Set(coldNames),
- `cold and post-connect tool sets differ. cold=${[...coldNames].sort()} spliced=${[...splicedNames].sort()}`,
+ servableNames,
+ `post-connect tools/list must mirror the relayer. got ${[...splicedNames].sort()}, expected ${[...servableNames].sort()}`,
);
assert.ok(mock.getSseGetCount() >= 1, "expected at least one SSE handshake");
diff --git a/packages/mcp/test/tool-definitions.test.mjs b/packages/mcp/test/tool-definitions.test.mjs
index 1386921f5..51c0ef9c9 100644
--- a/packages/mcp/test/tool-definitions.test.mjs
+++ b/packages/mcp/test/tool-definitions.test.mjs
@@ -8,6 +8,8 @@ import assert from "node:assert/strict";
import {
TOOL_DEFINITIONS,
SIGNED_OUT_TOOL_DEFINITIONS,
+ ALL_TOOL_DEFINITIONS,
+ BASELINE_RELAYER_TOOLS,
} from "../dist/auth-required.js";
function desc(list, name) {
@@ -77,8 +79,11 @@ test("memwal_recall is advertised as a read-only search", () => {
* instead of in someone's first save of the session.
*/
-test("cold-start memwal_remember_status accepts a whole batch", () => {
- for (const list of [TOOL_DEFINITIONS, SIGNED_OUT_TOOL_DEFINITIONS]) {
+test("memwal_remember_status accepts a whole batch", () => {
+ // Not advertised at cold start (see the baseline tests below) — the shape
+ // is still pinned here, so it stays reviewed while it waits for the prod
+ // release that lets it into the cold-start list.
+ for (const list of [ALL_TOOL_DEFINITIONS]) {
const tool = list.find((t) => t.name === "memwal_remember_status");
assert.ok(tool, "missing memwal_remember_status");
@@ -100,12 +105,12 @@ test("cold-start memwal_remember_status accepts a whole batch", () => {
}
});
-test("cold-start waitMs bound matches the sidecar's ceiling", () => {
+test("memwal_remember_status waitMs bound matches the sidecar's ceiling", () => {
// The sidecar validates waitMs with zod and rejects anything above its own
// cap. Advertising a larger maximum invites the agent to send a value that
// comes straight back as an MCP validation error — observed live at 60000
// once the sidecar lowered its ceiling to 45000.
- for (const list of [TOOL_DEFINITIONS, SIGNED_OUT_TOOL_DEFINITIONS]) {
+ for (const list of [ALL_TOOL_DEFINITIONS]) {
const tool = list.find((t) => t.name === "memwal_remember_status");
assert.equal(tool.inputSchema.properties.waitMs.maximum, 45000);
}
@@ -117,6 +122,53 @@ test("cold-start write tools warn that a result may not be saved yet", () => {
for (const name of ["memwal_remember", "memwal_remember_bulk"]) {
const d = desc(TOOL_DEFINITIONS, name);
assert.match(d, /NOT yet saved|NOT saved/i, `${name} omits the pending warning`);
- assert.match(d, /memwal_remember_status/, `${name} does not say how to settle it`);
+ assert.match(d, /settle it|settle them/, `${name} does not say to settle the job`);
+ }
+});
+
+/**
+ * GH #928. The bridge ships on npm and updates itself; a relayer ships per
+ * environment and does not, so 0.0.14-dev.0 dialled prod and staging still on
+ * 0.0.13. Its cold-start list named `memwal_remember_status`, which neither
+ * serves, and the description told the agent to go call it — one live run
+ * spent 90.67s on a tool that does not exist there before erroring.
+ *
+ * The cold-start list is served before any relayer capability is known, so it
+ * has to be a floor: only tools the oldest supported relayer serves, and no
+ * description pointing at anything outside it.
+ */
+
+test("cold-start lists advertise nothing beyond the baseline relayer", () => {
+ for (const list of [TOOL_DEFINITIONS, SIGNED_OUT_TOOL_DEFINITIONS]) {
+ const names = list.map((t) => t.name);
+ const beyond = names.filter(
+ (n) => !BASELINE_RELAYER_TOOLS.has(n) && n !== "memwal_login",
+ );
+ assert.deepEqual(
+ beyond,
+ [],
+ `cold start advertises tools the oldest supported relayer cannot serve: ${beyond}`,
+ );
+ for (const baseline of BASELINE_RELAYER_TOOLS) {
+ assert.ok(names.includes(baseline), `cold start omits ${baseline}`);
+ }
+ }
+});
+
+test("no cold-start description names a tool cold start does not advertise", () => {
+ // The tool list and the prose have to agree. A description is an
+ // instruction the agent follows, so naming an unadvertised tool is the
+ // same defect as listing it — it just fails one step later.
+ for (const list of [TOOL_DEFINITIONS, SIGNED_OUT_TOOL_DEFINITIONS]) {
+ const advertised = new Set(list.map((t) => t.name));
+ for (const tool of list) {
+ const named = tool.description.match(/memwal_[a-z_]+/g) ?? [];
+ const dangling = [...new Set(named)].filter((n) => !advertised.has(n));
+ assert.deepEqual(
+ dangling,
+ [],
+ `${tool.name}'s description sends the agent to unadvertised tools: ${dangling}`,
+ );
+ }
}
});
diff --git a/packages/mcp/test/tool-not-served.test.mjs b/packages/mcp/test/tool-not-served.test.mjs
new file mode 100644
index 000000000..f0eecac31
--- /dev/null
+++ b/packages/mcp/test/tool-not-served.test.mjs
@@ -0,0 +1,342 @@
+/**
+ * Regression test for GH #928 — a `tools/call` for a tool the connected relayer
+ * does not serve must be refused locally and immediately.
+ *
+ * What happened live: the bridge ships on npm and updates itself, a relayer
+ * ships per environment and does not, so 0.0.14-dev.0 talked to prod and
+ * staging still on 0.0.13. Its cold-start list advertised
+ * `memwal_remember_status`, and the pending-write wording told the agent to go
+ * call it. Neither relayer registers that tool, so the call was forwarded into
+ * a session that would never answer it and sat in `inFlight` until the orphan
+ * sweeper's deadline: one run spent 90.67s before erroring.
+ *
+ * The relayer's own `tools/list` is the authority on what it serves. Once the
+ * bridge has seen it, a call for anything outside that set (plus the two tools
+ * the bridge serves itself) is answered here — a stale tool list is not a
+ * transport fault, and the agent can only act on it if it is told now.
+ */
+import { test } from "node:test";
+import assert from "node:assert/strict";
+import http from "node:http";
+import { spawn } from "node:child_process";
+import { mkdtempSync, mkdirSync, writeFileSync, rmSync } from "node:fs";
+import { tmpdir } from "node:os";
+import { join, dirname, resolve } from "node:path";
+import { fileURLToPath } from "node:url";
+
+import { unknownToolText } from "../dist/bridge.js";
+
+const __dirname = dirname(fileURLToPath(import.meta.url));
+const BIN = resolve(__dirname, "../dist/bin/memwal-mcp.js");
+const EXPECTED_BEARER = "a".repeat(64);
+const EXPECTED_ACCOUNT_ID = "0x" + "3".repeat(64);
+
+/** What a 0.0.13 relayer registers: no `memwal_remember_status`. */
+const OLD_RELAYER_TOOLS = [
+ "memwal_remember",
+ "memwal_remember_bulk",
+ "memwal_recall",
+ "memwal_analyze",
+ "memwal_restore",
+ "memwal_health",
+];
+
+/** The tool the newer bridge knows about and this relayer has never heard of. */
+const MISSING_TOOL = "memwal_remember_status";
+
+test("the refusal tells the agent what exists and that nothing ran", () => {
+ const text = unknownToolText(MISSING_TOOL, [...OLD_RELAYER_TOOLS, "memwal_login"].sort());
+
+ // Which tool was refused, or the agent cannot tell which of several calls
+ // this answers.
+ assert.match(text, new RegExp(MISSING_TOOL));
+ // What to do instead: the list it holds is stale, and here is the real one.
+ assert.match(text, /tools\/list/);
+ for (const name of OLD_RELAYER_TOOLS) assert.match(text, new RegExp(name));
+ // A write tool that errors is ambiguous about whether the write happened.
+ // Say it plainly: an agent that guesses here tells the user a fact is
+ // saved when it never left the process.
+ assert.match(text, /nothing was saved/i);
+});
+
+function hasBridgeAuth(req) {
+ return (
+ req.headers.authorization === `Bearer ${EXPECTED_BEARER}` &&
+ req.headers["x-memwal-account-id"] === EXPECTED_ACCOUNT_ID
+ );
+}
+
+/** Relayer on the OLD tool set. It answers what it knows and stays silent on
+ * anything else — which is what a call for an unregistered tool looks like from
+ * the bridge's side, and what made the old behaviour a 90s wait. */
+function startOldRelayer() {
+ const sessions = new Map();
+ const callsSeen = [];
+ const server = http.createServer((req, res) => {
+ const url = new URL(req.url, "http://127.0.0.1");
+ if (req.method === "GET" && url.pathname === "/version") {
+ res.writeHead(200, { "content-type": "application/json" });
+ res.end(
+ JSON.stringify({
+ apiVersion: "1.0.0",
+ relayerVersion: "0.0.13",
+ minSupportedSdk: { mcp: "0.0.1" },
+ }),
+ );
+ return;
+ }
+ if (req.method === "GET" && url.pathname === "/api/mcp/sse") {
+ if (!hasBridgeAuth(req)) {
+ res.writeHead(401);
+ res.end();
+ return;
+ }
+ const sessionId = `session-${sessions.size + 1}`;
+ res.writeHead(200, {
+ "content-type": "text/event-stream",
+ "cache-control": "no-cache",
+ connection: "keep-alive",
+ });
+ res.write(`event: endpoint\ndata: /api/mcp/messages?sessionId=${sessionId}\n\n`);
+ sessions.set(sessionId, { res });
+ const hb = setInterval(() => {
+ if (res.writableEnded) {
+ clearInterval(hb);
+ return;
+ }
+ res.write(":\n\n");
+ }, 200);
+ hb.unref?.();
+ res.on("close", () => clearInterval(hb));
+ return;
+ }
+ if (req.method === "POST" && url.pathname === "/api/mcp/messages") {
+ if (!hasBridgeAuth(req)) {
+ res.writeHead(401);
+ res.end();
+ return;
+ }
+ const session = sessions.get(url.searchParams.get("sessionId"));
+ let body = "";
+ req.on("data", (c) => (body += c));
+ req.on("end", () => {
+ if (!session) {
+ res.writeHead(404);
+ res.end();
+ return;
+ }
+ res.writeHead(202);
+ res.end();
+ let msg;
+ try {
+ msg = JSON.parse(body);
+ } catch {
+ return;
+ }
+ const reply = (result) =>
+ session.res.write(
+ `event: message\ndata: ${JSON.stringify({ jsonrpc: "2.0", id: msg.id, result })}\n\n`,
+ );
+ if (msg.method === "initialize") {
+ reply({
+ protocolVersion: "2024-11-05",
+ capabilities: { tools: { listChanged: true } },
+ serverInfo: { name: "memwal-upstream", version: "0.0.13" },
+ });
+ return;
+ }
+ if (msg.method === "tools/list") {
+ reply({
+ tools: OLD_RELAYER_TOOLS.map((name) => ({
+ name,
+ description: `upstream ${name}`,
+ inputSchema: { type: "object" },
+ })),
+ });
+ return;
+ }
+ if (msg.method === "tools/call") {
+ callsSeen.push(msg.params?.name);
+ if (msg.params?.name === "memwal_health") {
+ reply({
+ content: [{ type: "text", text: "status=ok version=0.0.13" }],
+ isError: false,
+ });
+ }
+ // Anything else: silence, exactly as an unregistered tool
+ // would behave if the call ever got this far.
+ return;
+ }
+ });
+ return;
+ }
+ res.writeHead(404);
+ res.end();
+ });
+ return new Promise((res) => {
+ server.listen(0, "127.0.0.1", () => {
+ const { port } = server.address();
+ res({ server, base: `http://127.0.0.1:${port}`, callsSeen });
+ });
+ });
+}
+
+function makeCreds(relayerUrl) {
+ return {
+ delegatePrivateKey: EXPECTED_BEARER,
+ delegatePublicKeyHex: "b".repeat(64),
+ delegateAddress: "0x" + "1".repeat(64),
+ walletAddress: "0x" + "2".repeat(64),
+ accountId: EXPECTED_ACCOUNT_ID,
+ packageId: "0x" + "4".repeat(64),
+ relayerUrl,
+ label: "Tool-not-served Test",
+ createdAt: new Date(0).toISOString(),
+ version: 1,
+ };
+}
+
+test("a tool the relayer does not serve is refused locally, not waited out", async (t) => {
+ const mock = await startOldRelayer();
+ const home = mkdtempSync(join(tmpdir(), "memwal-toolskew-test-"));
+ const credsPath = join(home, ".memwal", "credentials.json");
+ mkdirSync(dirname(credsPath), { recursive: true });
+ writeFileSync(credsPath, JSON.stringify(makeCreds(mock.base)), { mode: 0o600 });
+
+ const child = spawn(process.execPath, [BIN, "--relayer", mock.base, "--web-url", mock.base], {
+ env: { ...process.env, HOME: home, USERPROFILE: home },
+ stdio: ["pipe", "pipe", "pipe"],
+ });
+
+ const received = [];
+ const listeners = new Set();
+ let buf = "";
+ child.stdout.on("data", (d) => {
+ buf += d.toString();
+ let nl;
+ while ((nl = buf.indexOf("\n")) >= 0) {
+ const line = buf.slice(0, nl);
+ buf = buf.slice(nl + 1);
+ if (!line.trim()) continue;
+ let msg;
+ try {
+ msg = JSON.parse(line);
+ } catch {
+ continue;
+ }
+ received.push({ msg, at: Date.now() });
+ for (const l of [...listeners]) l(msg);
+ }
+ });
+ let stderrBuf = "";
+ child.stderr.on("data", (d) => (stderrBuf += d.toString()));
+
+ const send = (obj) => child.stdin.write(JSON.stringify(obj) + "\n");
+ const waitFor = (pred, ms = 15000) => {
+ const hit = received.find((r) => pred(r.msg));
+ if (hit) return Promise.resolve(hit.msg);
+ return new Promise((res, rej) => {
+ const timer = setTimeout(() => {
+ listeners.delete(l);
+ rej(
+ new Error(
+ `timed out waiting for message\n--- stderr ---\n${stderrBuf}\n--- received ---\n${received.map((r) => JSON.stringify(r.msg)).join("\n")}`,
+ ),
+ );
+ }, ms);
+ const l = (m) => {
+ if (pred(m)) {
+ clearTimeout(timer);
+ listeners.delete(l);
+ res(m);
+ }
+ };
+ listeners.add(l);
+ });
+ };
+
+ t.after(() => {
+ child.kill("SIGKILL");
+ mock.server.close();
+ rmSync(home, { recursive: true, force: true });
+ });
+
+ send({
+ jsonrpc: "2.0",
+ id: 1,
+ method: "initialize",
+ params: {
+ protocolVersion: "2025-06-18",
+ capabilities: {},
+ clientInfo: { name: "toolskew-test", version: "1.0.0" },
+ },
+ });
+ await waitFor((m) => m.id === 1 && m.result, 10_000);
+
+ // The cold-start list is a floor, so it must not name the tool this relayer
+ // lacks even before anything upstream is known.
+ send({ jsonrpc: "2.0", id: 2, method: "tools/list", params: {} });
+ const cold = await waitFor((m) => m.id === 2 && m.result, 10_000);
+ assert.ok(
+ !cold.result.tools.some((tool) => tool.name === MISSING_TOOL),
+ `cold-start tools/list advertises ${MISSING_TOOL}, which no released relayer serves`,
+ );
+
+ // Re-list once connected so the bridge learns what this relayer actually
+ // registers. That reply is the authority the refusal below is based on.
+ await waitFor((m) => m.method === "notifications/tools/list_changed", 10_000);
+ send({ jsonrpc: "2.0", id: 3, method: "tools/list", params: {} });
+ const upstream = await waitFor((m) => m.id === 3 && m.result, 10_000);
+ const upstreamNames = upstream.result.tools.map((tool) => tool.name);
+ assert.ok(
+ upstreamNames.includes("memwal_recall"),
+ `expected the relayer's own tool list, got ${upstreamNames}`,
+ );
+ assert.ok(!upstreamNames.includes(MISSING_TOOL));
+
+ // The call the live run made. Before the fix it was forwarded and left to
+ // the orphan sweeper: 60s tool ceiling + 30s headroom.
+ const startedAt = Date.now();
+ send({
+ jsonrpc: "2.0",
+ id: 4,
+ method: "tools/call",
+ params: { name: MISSING_TOOL, arguments: { job_id: "job-1" } },
+ });
+ const refusal = await waitFor((m) => m.id === 4, 10_000);
+ const elapsed = Date.now() - startedAt;
+
+ assert.equal(refusal.result?.isError, true, `expected an error result, got ${JSON.stringify(refusal)}`);
+ const text = refusal.result.content[0].text;
+ assert.match(text, new RegExp(MISSING_TOOL));
+ // The agent's way out has to be in the message: which tools exist, and that
+ // its own list is what is wrong.
+ assert.match(text, /tools\/list/);
+ assert.match(text, /memwal_recall/);
+ // And it must be unambiguous that the write did not happen, or the agent
+ // reports a fact as saved on the strength of an error.
+ assert.match(text, /nothing was saved/i);
+ assert.ok(
+ elapsed < 5_000,
+ `refusal took ${elapsed}ms — expected it answered locally, not at the orphan deadline`,
+ );
+
+ // Proof it never left the process: the relayer saw the health call we made
+ // nothing of, and never the missing tool.
+ assert.ok(
+ !mock.callsSeen.includes(MISSING_TOOL),
+ `bridge forwarded ${MISSING_TOOL} upstream: ${mock.callsSeen}`,
+ );
+
+ // A tool the relayer DOES serve still goes through, so the gate is a filter
+ // and not a wall.
+ send({
+ jsonrpc: "2.0",
+ id: 5,
+ method: "tools/call",
+ params: { name: "memwal_health", arguments: {} },
+ });
+ const health = await waitFor((m) => m.id === 5 && m.result, 10_000);
+ assert.notEqual(health.result?.isError, true);
+ assert.match(JSON.stringify(health.result), /status=ok/);
+});
From 1857c6b7c18c5f6f74fc1b41be6a54ec58b2c7db Mon Sep 17 00:00:00 2001
From: Harry Phan
Date: Thu, 17 Sep 2026 11:15:21 +0700
Subject: [PATCH 049/132] fix(relayer): stop register-shape rejections burning
the whole retry budget
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit
A malformed register transaction is deterministic for the same journal, but
classify_sidecar_error had no arm for it, so it fell through to the Transient
catch-all and Apalis retried it until the budget ran out. On dev this turned a
5-second failure into a 6-minute hang before the job died anyway: every remember
between 03:48Z and 04:08Z on 2026-09-17 failed with
registerTransaction must use a ValidDuring address-balance expiration
and zero blobs were certified over the window.
Classify the sidecar's register-transaction journal assertions Permanent so the
job dies on the first attempt and the caller is told to send the fact again.
Anchored on the assertion text, not the transport's NO_SIDE_EFFECT code:
retry/rpc.ts returns that code for any pre-submission failure, including
shared-infra blips carrying causeCode SHARED_SERVICE_UNAVAILABLE, which must
stay retryable. A test pins that case.
This makes the failure fast and legible. It does not make the write succeed —
the shape bug itself is still open, and is neither low WAL balance (the
security-delete reconciler reports low_balance: false on every tick) nor a
duplicated SDK (one @mysten/sui 2.17.0, one @mysten/walrus 1.1.7 in the lock).
---
services/server/src/jobs.rs | 106 ++++++++++++++++++++++++++++++++++++
1 file changed, 106 insertions(+)
diff --git a/services/server/src/jobs.rs b/services/server/src/jobs.rs
index d60999d87..554594a1c 100644
--- a/services/server/src/jobs.rs
+++ b/services/server/src/jobs.rs
@@ -2593,6 +2593,40 @@ impl WalletJobError {
lower.contains("timed out waiting for") && lower.contains("upload slot")
}
+ /// True if `msg` is one of the sidecar's register-transaction journal
+ /// assertions (`validatePreparedRegisterTransaction` and friends in
+ /// scripts/sidecar/routes/walrus-upload-journal.ts).
+ ///
+ /// These describe the SHAPE of a transaction the sidecar already built —
+ /// wrong gas source, missing WAL withdrawal, sender/gas-owner mismatch,
+ /// non-canonical bytes, digest mismatch. Replaying the journal rebuilds
+ /// the same shape, so every retry re-fails identically; the job burns its
+ /// whole retry budget before dying. Classify Permanent so it dies on the
+ /// first attempt and the caller is told to send the fact again.
+ ///
+ /// Anchored on the assertion text rather than the transport's
+ /// `NO_SIDE_EFFECT` code on purpose: that code means only "nothing reached
+ /// the chain" and is also returned for pre-submission infra blips
+ /// (`causeCode: SHARED_SERVICE_UNAVAILABLE` in retry/rpc.ts), which must
+ /// stay retryable.
+ pub fn is_register_transaction_shape_error(msg: &str) -> bool {
+ let lower = msg.to_ascii_lowercase();
+ if !lower.contains("registertransaction") {
+ return false;
+ }
+ lower.contains("must use a validduring address-balance expiration")
+ || lower.contains("must pay gas from the address balance")
+ || lower.contains("has no wal address-balance withdrawal")
+ || lower.contains("resolved wal from an owned coin")
+ || lower.contains("resolved the relay tip from an owned sui coin")
+ || lower.contains("must use a distinct gas owner")
+ || lower.contains("sender does not match")
+ || lower.contains("gas owner does not match")
+ || lower.contains("is not canonical base64")
+ || lower.contains("digest mismatch")
+ || lower.contains("contains invalid transactiondata")
+ }
+
/// True if `msg` is a pool-wallet WAL shortfall. Deliberately the
/// substring half of `parse_wal_balance_alert_info` without its
/// `available < WAL_BALANCE_LOW_THRESHOLD_MIST` gate: that threshold
@@ -2627,6 +2661,13 @@ impl WalletJobError {
if parse_wal_balance_alert_info(msg).is_some() {
return WalletJobError::WalrusBalanceLow(msg.to_string());
}
+ // Register-transaction shape assertions are deterministic for the same
+ // journal — see is_register_transaction_shape_error. Checked early so a
+ // shape rejection cannot fall through to the Transient catch-all at the
+ // end and spend the job's retry budget re-failing identically.
+ if Self::is_register_transaction_shape_error(msg) {
+ return WalletJobError::Permanent(msg.to_string());
+ }
// Sidecar upload limiter saturated — see UploadSlotCongestion docs.
if Self::is_upload_slot_congestion_error(msg) {
return WalletJobError::UploadSlotCongestion(msg.to_string());
@@ -2957,8 +2998,73 @@ SequenceNumber(884613305), o#B61aVqEgDskxru255FTdzua2RxbbnhDMFxmQ8SCxvj3n) alrea
different transaction: TransactionDigest(8bjFgRyXRRYwrzQapgEjpHnGhdfNDY7d6xA82BtHrp3F) \
{ k#80127c70.., k#81626d03.. } with 6842 stake].";
+ /// The exact production error string from the dev-relayer write outage of
+ /// 2026-09-17 (job d67d1fc2…, trace 12b3e920…). Every remember on dev failed
+ /// with this shape and, classified Transient, burned its retry budget before
+ /// dying — 0 blobs certified over the whole window.
+ const PROD_REGISTER_SHAPE_ERROR: &str =
+ "durable Walrus upload failed (503 Service Unavailable): \
+{\"error\":\"registerTransaction must use a ValidDuring address-balance expiration\",\
+\"code\":\"NO_SIDE_EFFECT\",\"traceId\":\"12b3e920-b94b-4100-bb35-fc0f0a1804e1\"}";
+
static DB_SETUP_LOCK: OnceLock> = OnceLock::new();
+ #[test]
+ fn register_shape_rejection_is_permanent() {
+ assert!(matches!(
+ WalletJobError::classify_sidecar_error(PROD_REGISTER_SHAPE_ERROR),
+ WalletJobError::Permanent(_)
+ ));
+ }
+
+ #[test]
+ fn every_register_shape_assertion_is_permanent() {
+ for msg in [
+ "registerTransaction must pay gas from the address balance",
+ "registerTransaction has no WAL address-balance withdrawal",
+ "registerTransaction resolved WAL from an owned coin",
+ "registerTransaction resolved the relay tip from an owned SUI coin",
+ "sponsored registerTransaction must use a distinct gas owner",
+ "sponsored registerTransaction sender does not match the wallet",
+ "registerTransaction gas owner does not match the journaled wallet",
+ "registerTransaction.transactionBytes is not canonical base64",
+ "registerTransaction digest mismatch: expected abc, got def",
+ "registerTransaction contains invalid TransactionData: bad bytes",
+ ] {
+ assert!(
+ matches!(
+ WalletJobError::classify_sidecar_error(msg),
+ WalletJobError::Permanent(_)
+ ),
+ "expected Permanent for {msg}"
+ );
+ }
+ }
+
+ /// `NO_SIDE_EFFECT` alone must stay retryable: retry/rpc.ts returns it for
+ /// any pre-submission failure, including shared-infra blips that the next
+ /// attempt succeeds through.
+ #[test]
+ fn no_side_effect_without_a_shape_assertion_stays_transient() {
+ assert!(matches!(
+ WalletJobError::classify_sidecar_error(
+ "durable Walrus upload failed (503 Service Unavailable): \
+{\"error\":\"fetch failed\",\"code\":\"NO_SIDE_EFFECT\",\
+\"causeCode\":\"SHARED_SERVICE_UNAVAILABLE\"}"
+ ),
+ WalletJobError::Transient(_)
+ ));
+ }
+
+ /// The shape check keys on the assertion text, so an unrelated message that
+ /// merely mentions a sender mismatch must not be swallowed by it.
+ #[test]
+ fn unrelated_sender_mismatch_is_not_a_shape_rejection() {
+ assert!(!WalletJobError::is_register_transaction_shape_error(
+ "sponsor failed: sender does not match the wallet"
+ ));
+ }
+
fn test_database_url() -> String {
std::env::var("DATABASE_URL")
.unwrap_or_else(|_| "postgresql://memwal:memwal_secret@localhost:5432/memwal".into())
From 1a4ccae2bb7704f983a5b6bf55fd6d101ec494a4 Mon Sep 17 00:00:00 2001
From: Harry Phan
Date: Thu, 17 Sep 2026 11:17:08 +0700
Subject: [PATCH 050/132] fix(sidecar): make a rejected register transaction
say what it saw
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit
`assertAddressBalanceRegisterTransaction` rejects on three conditions joined
by `||` — wrong expiration variant, minTimestamp set, maxTimestamp set — and
threw one sentence naming the invariant rather than the value that broke it.
On testnet (2026-09-17) every upload job failed here, across two consecutive
deployments: job 01b2613f burned attempts 1-5 over wallet keys 0/1/3/4, and
the relayer logs could not say which of the three conditions fired, because
the only evidence they carried was the sentence itself.
Append the observed shape: the variant, both epoch bounds, and both timestamp
bounds, rendered so `null` and `undefined` stay distinguishable — the guard
treats them differently but a template string does not. Same for the gas-payment
guard, which now reports the payment length it rejected.
The leading sentence is unchanged, so the Rust classifier's substring match and
the existing test regexes still hold.
---
.../__tests__/sidecar-query-helpers.test.ts | 36 ++++++++++++++++++
.../sidecar/routes/walrus-upload-journal.ts | 37 ++++++++++++++++++-
2 files changed, 71 insertions(+), 2 deletions(-)
diff --git a/services/server/scripts/__tests__/sidecar-query-helpers.test.ts b/services/server/scripts/__tests__/sidecar-query-helpers.test.ts
index 1391c8b75..5801f6e02 100644
--- a/services/server/scripts/__tests__/sidecar-query-helpers.test.ts
+++ b/services/server/scripts/__tests__/sidecar-query-helpers.test.ts
@@ -1024,3 +1024,39 @@ test("normal registration checkpoints the exact Blob from transaction effects",
})
);
});
+
+test("a rejected register expiration names the bound that broke the guard", () => {
+ // The guard rejects on three conditions joined by `||`. When every upload
+ // job on testnet hit this (2026-09-17) the logged sentence named the
+ // invariant but not the value, so the relayer logs could not say which
+ // condition fired. The message has to carry the shape it saw.
+ const signer = new Ed25519Keypair();
+ const transaction = new Transaction();
+ transaction.setSender(signer.toSuiAddress());
+ transaction.setGasOwner(signer.toSuiAddress());
+ transaction.setGasBudget(1_000n);
+ transaction.setGasPrice(1n);
+ transaction.setGasPayment([]);
+ transaction.setExpiration({
+ ValidDuring: {
+ minEpoch: "1",
+ maxEpoch: "2",
+ minTimestamp: "5",
+ maxTimestamp: null,
+ chain: "69WiPg3DAQiwdxfncX6wYQ2siKwAe6L9BZthQea3JNMD",
+ nonce: 1,
+ },
+ });
+
+ const resolved = TransactionDataBuilder.restore(transaction.getData() as never);
+ assert.throws(
+ () => assertAddressBalanceRegisterTransaction(resolved),
+ (error: Error) => {
+ assert.match(error.message, /must use a ValidDuring address-balance expiration/);
+ assert.match(error.message, /expiration=ValidDuring/);
+ assert.match(error.message, /minTimestamp="5"/);
+ assert.match(error.message, /maxTimestamp=null/);
+ return true;
+ },
+ );
+});
diff --git a/services/server/scripts/sidecar/routes/walrus-upload-journal.ts b/services/server/scripts/sidecar/routes/walrus-upload-journal.ts
index b2a7aa4ff..e517fb017 100644
--- a/services/server/scripts/sidecar/routes/walrus-upload-journal.ts
+++ b/services/server/scripts/sidecar/routes/walrus-upload-journal.ts
@@ -368,11 +368,41 @@ export function assertSponsoredRegisterTransaction(
assertRegisterTransactionUsesAddressBalanceWal(transactionData);
}
+/** Render a bound the way the guard tests it, so `null` and `undefined` — which
+ * the guard treats differently but a template string renders identically — stay
+ * distinguishable in a log line. */
+function describeExpirationBound(value: unknown): string {
+ if (value === null) return "null";
+ if (value === undefined) return "undefined";
+ return JSON.stringify(value);
+}
+
+/** Describe an expiration precisely enough to act on it from a production log.
+ *
+ * The guard below rejects on three separate conditions joined by `||`, so the
+ * bare sentence it used to throw could not say which one fired. A run of these
+ * failures on testnet (every upload job, both deployments, 2026-09-17) could not
+ * be diagnosed from the relayer logs at all: the classifier reported the string,
+ * and the string named the invariant rather than the value that broke it. */
+function describeExpiration(expiration: TransactionDataBuilder["expiration"]): string {
+ if (!expiration) return "expiration=none";
+ if (expiration.$kind !== "ValidDuring") return `expiration=${expiration.$kind}`;
+ const { minEpoch, maxEpoch, minTimestamp, maxTimestamp } = expiration.ValidDuring;
+ return "expiration=ValidDuring"
+ + ` minEpoch=${describeExpirationBound(minEpoch)}`
+ + ` maxEpoch=${describeExpirationBound(maxEpoch)}`
+ + ` minTimestamp=${describeExpirationBound(minTimestamp)}`
+ + ` maxTimestamp=${describeExpirationBound(maxTimestamp)}`;
+}
+
export function assertAddressBalanceRegisterTransaction(
transactionData: TransactionDataBuilder,
): bigint {
if (transactionData.gasData.payment?.length !== 0) {
- throw new Error("registerTransaction must pay gas from the address balance");
+ throw new Error(
+ "registerTransaction must pay gas from the address balance"
+ + ` (gasData.payment.length=${String(transactionData.gasData.payment?.length ?? "undefined")})`,
+ );
}
const expiration = transactionData.expiration;
@@ -381,7 +411,10 @@ export function assertAddressBalanceRegisterTransaction(
|| expiration.ValidDuring.minTimestamp !== null
|| expiration.ValidDuring.maxTimestamp !== null
) {
- throw new Error("registerTransaction must use a ValidDuring address-balance expiration");
+ throw new Error(
+ "registerTransaction must use a ValidDuring address-balance expiration"
+ + ` (${describeExpiration(expiration)})`,
+ );
}
assertRegisterTransactionUsesAddressBalanceWal(transactionData);
From 0b2a2bb373b2453d643b7ec4541ba14738e40001 Mon Sep 17 00:00:00 2001
From: Harry Phan
Date: Thu, 17 Sep 2026 11:24:41 +0700
Subject: [PATCH 051/132] fix(sidecar): give the direct-signed register its
ValidDuring window
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit
`flow.register()` hands back a transaction with no expiration, and
`prepareRegisterTransaction` then pays its gas from the address balance,
which puts a `FundsWithdrawal` input in it. Sui admits that withdrawal only
inside a `ValidDuring` window, and `assertAddressBalanceRegisterTransaction`
re-checks the same rule one line later, so every direct-signed register
rejected itself before it could be signed:
durable Walrus upload failed (503 Service Unavailable):
{"error":"registerTransaction must use a ValidDuring address-balance
expiration","code":"NO_SIDE_EFFECT"}
On dev that was every remember, on every owner and namespace — 0xffebf969,
0x6a8d47f4, sdk-e2e-*, default — with zero blobs certified over the window.
Sponsored journals are unaffected: Enoki builds its own gas data, and
DURABLE_ENOKI_REGISTER_ENABLED defaults to false, so this is the path dev
actually runs.
One epoch wide rather than a range: `validatePreparedRegisterTransaction`
hands `maxEpoch` back as the journal's expiry guard, so a window outliving
the reservation would keep replaying an entry Sui has already retired. The
nonce is derived from the transaction kind rather than drawn at random, so
re-preparing the same register rebuilds byte-identical and the journal stays
idempotent, while two different registers still reserve under different
nonces.
Every existing register fixture hand-sets ValidDuring, which is why the suite
stayed green while this path was dead. The new tests build the transaction
the way production does instead — and a withdrawal turning out to be
unresolvable offline without a window is its own evidence that it is not
optional here.
---
.../__tests__/sidecar-query-helpers.test.ts | 80 +++++++++++++++++++
.../sidecar/routes/walrus-upload-journal.ts | 48 +++++++++++
2 files changed, 128 insertions(+)
diff --git a/services/server/scripts/__tests__/sidecar-query-helpers.test.ts b/services/server/scripts/__tests__/sidecar-query-helpers.test.ts
index 5801f6e02..b9fa162e1 100644
--- a/services/server/scripts/__tests__/sidecar-query-helpers.test.ts
+++ b/services/server/scripts/__tests__/sidecar-query-helpers.test.ts
@@ -27,6 +27,7 @@ import {
WALRUS_PACKAGE_ID,
} from "../sidecar/config.js";
import {
+ addressBalanceExpiration,
assertAddressBalanceRegisterTransaction,
assertSponsoredRegisterTransactionKind,
createdBlobObjectIdFromTransaction,
@@ -1060,3 +1061,82 @@ test("a rejected register expiration names the bound that broke the guard", () =
},
);
});
+
+
+/** Build a register the way `prepareRegisterTransaction` does: gas from the
+ * address balance, a WAL withdrawal, and the expiration the production path now
+ * applies. Every other register fixture in this file hand-sets `ValidDuring`,
+ * which is exactly why the direct-signed path could reject every write in
+ * production while the suite stayed green. */
+async function directSignedRegister(): Promise {
+ const signer = new Ed25519Keypair();
+ const transaction = new Transaction();
+ transaction.setSender(signer.toSuiAddress());
+ transaction.setGasOwner(signer.toSuiAddress());
+ transaction.setGasBudget(1_000n);
+ transaction.setGasPrice(1n);
+ transaction.setGasPayment([]);
+ const walType = `0x${"2".repeat(64)}::wal::WAL`;
+ const withdrawal = transaction.withdrawal({ amount: 1n, type: walType });
+ const wal = transaction.moveCall({
+ target: "0x2::coin::redeem_funds",
+ typeArguments: [walType],
+ arguments: [withdrawal],
+ });
+ transaction.moveCall({
+ target: "0x2::coin::destroy_zero",
+ typeArguments: [walType],
+ arguments: [wal],
+ });
+ transaction.setExpiration(
+ addressBalanceExpiration(7n, await transaction.build({ onlyTransactionKind: true })),
+ );
+ return TransactionDataBuilder.fromBytes(await transaction.build());
+}
+
+test("the expiration prepareRegisterTransaction applies is the one the guard demands", async () => {
+ // The positive half of the dev outage: `flow.register()` hands back a
+ // transaction with no expiration, and the guard one line later requires a
+ // ValidDuring window, so every direct-signed register failed its own
+ // assertion and no blob was certified. A register carrying what
+ // addressBalanceExpiration builds passes, and reports the epoch the journal
+ // then uses as its expiry guard.
+ assert.equal(assertAddressBalanceRegisterTransaction(await directSignedRegister()), 7n);
+});
+
+test("a register whose expiration is not ValidDuring is rejected", async () => {
+ // The `$kind` arm of the guard, which the timestamp case above does not
+ // reach. Built with the real expiration first: a withdrawal cannot be
+ // resolved offline without one, which is its own evidence that the window
+ // is not optional here.
+ const data = await directSignedRegister();
+ data.expiration = { $kind: "Epoch", Epoch: "7" } as typeof data.expiration;
+ assert.throws(
+ () => assertAddressBalanceRegisterTransaction(data),
+ /must use a ValidDuring address-balance expiration/,
+ );
+});
+
+test("the address-balance nonce is derived from the transaction, not drawn at random", () => {
+ const kind = new Uint8Array([1, 2, 3]);
+ const otherKind = new Uint8Array([1, 2, 4]);
+
+ // Re-preparing the same register must rebuild byte-identical, or the
+ // journal stops being idempotent and a replay pays for a second blob.
+ assert.equal(
+ addressBalanceExpiration(9n, kind).ValidDuring.nonce,
+ addressBalanceExpiration(9n, kind).ValidDuring.nonce,
+ );
+ // Two different registers must still reserve under different nonces.
+ assert.notEqual(
+ addressBalanceExpiration(9n, kind).ValidDuring.nonce,
+ addressBalanceExpiration(9n, otherKind).ValidDuring.nonce,
+ );
+
+ const { minEpoch, maxEpoch, minTimestamp, maxTimestamp } =
+ addressBalanceExpiration(9n, kind).ValidDuring;
+ assert.equal(minEpoch, "9");
+ assert.equal(maxEpoch, "9");
+ assert.equal(minTimestamp, null);
+ assert.equal(maxTimestamp, null);
+});
diff --git a/services/server/scripts/sidecar/routes/walrus-upload-journal.ts b/services/server/scripts/sidecar/routes/walrus-upload-journal.ts
index e517fb017..cdef6a03d 100644
--- a/services/server/scripts/sidecar/routes/walrus-upload-journal.ts
+++ b/services/server/scripts/sidecar/routes/walrus-upload-journal.ts
@@ -4,6 +4,7 @@
* advance and return one checkpointable WriteBlobStep.
*/
+import { createHash } from "node:crypto";
import express, { type Express } from "express";
import type {
WriteBlobStep,
@@ -26,6 +27,7 @@ import {
JSON_LIMIT_WALRUS_UPLOAD,
MAX_WALRUS_EPOCHS,
SERVER_SUI_PRIVATE_KEYS,
+ SUI_CHAIN_IDENTIFIER,
SUI_NETWORK,
SUI_TYPE,
WALRUS_PACKAGE_ID,
@@ -256,6 +258,51 @@ function parsePreparedRegisterTransaction(raw: unknown): PreparedRegisterTransac
};
}
+/**
+ * The expiration a direct-signed register needs: one epoch of address-balance
+ * withdrawal.
+ *
+ * Paying gas from the address balance puts a `FundsWithdrawal` input in the
+ * transaction, and Sui admits that withdrawal only inside a `ValidDuring`
+ * window — the same rule `assertAddressBalanceRegisterTransaction` re-checks
+ * one line later. Neither `flow.register()` nor `Transaction.build()` sets one,
+ * so every direct-signed register failed its own assertion with
+ * "registerTransaction must use a ValidDuring address-balance expiration" and
+ * no blob was ever certified on this path.
+ *
+ * One epoch wide rather than a range: `validatePreparedRegisterTransaction`
+ * hands `maxEpoch` back as the journal's expiry guard, so a window outliving
+ * the reservation would keep replaying an entry Sui has already retired.
+ *
+ * The nonce is derived from the transaction kind rather than drawn at random,
+ * so re-preparing the same register — same blob, same epochs, same attributes —
+ * rebuilds byte-identical and the journal stays idempotent, while two different
+ * registers still reserve under different nonces.
+ */
+export function addressBalanceExpiration(epoch: bigint, transactionKind: Uint8Array) {
+ const bounded = String(epoch);
+ return {
+ ValidDuring: {
+ minEpoch: bounded,
+ maxEpoch: bounded,
+ minTimestamp: null,
+ maxTimestamp: null,
+ chain: SUI_CHAIN_IDENTIFIER,
+ nonce: createHash("sha256").update(transactionKind).digest().readUInt32BE(0),
+ },
+ } as const;
+}
+
+async function bindAddressBalanceExpiration(transaction: Transaction): Promise {
+ const transactionKind = await transaction.build({
+ client: suiClient as any,
+ onlyTransactionKind: true,
+ });
+ transaction.setExpiration(
+ addressBalanceExpiration(await currentSuiEpoch(), transactionKind),
+ );
+}
+
export async function prepareRegisterTransaction(
transaction: Transaction,
signer: Ed25519Keypair,
@@ -301,6 +348,7 @@ export async function prepareRegisterTransaction(
// Fail-closed sponsorship already returned above. Remaining path is the
// explicit phase-1 / unconfigured-Enoki direct sign.
transaction.setGasPayment([]);
+ await bindAddressBalanceExpiration(transaction);
const bytes = await transaction.build({ client: suiClient });
assertAddressBalanceRegisterTransaction(TransactionDataBuilder.fromBytes(bytes));
const signed = await signer.signTransaction(bytes);
From 723baeceed9d549724b0a956ca4dda8d759c1c8b Mon Sep 17 00:00:00 2001
From: Harry Phan
Date: Thu, 17 Sep 2026 11:24:44 +0700
Subject: [PATCH 052/132] fix(sdk): honour retry-after when a status poll is
rate-limited
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit
`waitForRememberJob` and `waitForRememberJobs` treat 429 as transient and
`continue`, which re-polls on the 1.5s->10s curve. The relayer's limiter
counts those polls against the delegate key — 60 weighted-requests/min — so a
`retry_after_seconds: 60` answered 10s later re-trips it on every attempt and
the loop starves itself. The caller then sees "still uploading" for as long as
it is willing to wait, including long after the job has failed: measured on
dev at 12 reads over 250s, where the read path 429'd while the job itself was
already dead. That is how a lost fact reads as a slow one.
Take the backoff from `Retry-After`, falling back to `retry_after_seconds` in
the body for proxies that strip the header, and clamp it to what is left of
the caller's own budget so a long backoff spends that remainder on one final
read rather than extending the wait.
---
packages/sdk/src/memwal.ts | 52 +++++-
.../sdk/test/remember-retry-after.test.mjs | 149 ++++++++++++++++++
2 files changed, 199 insertions(+), 2 deletions(-)
create mode 100644 packages/sdk/test/remember-retry-after.test.mjs
diff --git a/packages/sdk/src/memwal.ts b/packages/sdk/src/memwal.ts
index 569f33fcc..d49fbffdf 100644
--- a/packages/sdk/src/memwal.ts
+++ b/packages/sdk/src/memwal.ts
@@ -233,6 +233,44 @@ function isTransientPollingStatus(status: number): boolean {
return status === 0 || status === 429 || status >= 500;
}
+/**
+ * How long a transient polling rejection asked us to wait, in ms.
+ *
+ * A 429 from the relayer's rate limiter is not an invitation to poll again in
+ * a second. The limiter counts our own status reads, so answering a
+ * `retry_after_seconds: 60` with the 10s backoff cap re-trips it on every
+ * attempt and the loop starves itself: the job then reads as "still uploading"
+ * for as long as the caller is willing to wait — including long after it has
+ * actually failed, which is the state that costs a user their fact.
+ *
+ * Clamped to what is left of the caller's own budget, so a 10-minute backoff
+ * costs at most the wait the caller already asked for. The loop then spends
+ * that remainder on one final read at the boundary rather than on a backoff
+ * curve nobody asked for.
+ */
+function retryAfterDelayMs(err: unknown, deadline: number): number {
+ const seconds =
+ (err as { retryAfterSeconds?: number }).retryAfterSeconds
+ ?? retryAfterSecondsFromBody((err as { cause?: unknown }).cause);
+ if (seconds === undefined || !Number.isFinite(seconds) || seconds <= 0) return 0;
+ return Math.max(0, Math.min(seconds * 1000, deadline - Date.now()));
+}
+
+/**
+ * The relayer states the backoff in the 429 body as well as in `Retry-After`.
+ * A proxy that strips the header must not cost us the hint.
+ */
+function retryAfterSecondsFromBody(cause: unknown): number | undefined {
+ if (typeof cause !== "string") return undefined;
+ try {
+ const parsed = JSON.parse(cause) as { retry_after_seconds?: unknown };
+ const seconds = Number(parsed?.retry_after_seconds);
+ return Number.isFinite(seconds) ? seconds : undefined;
+ } catch {
+ return undefined;
+ }
+}
+
/**
* Normalise the legacy `(text, namespace)` and new `(text, options)`
* overloads of `analyze()` / `analyzeAndWait()` into a single
@@ -428,9 +466,13 @@ export class MemWal {
const { pollIntervalMs = 1500, timeoutMs = 60_000 } = opts;
const deadline = Date.now() + timeoutMs;
let attempt = 0;
+ let retryAfterMs = 0;
while (Date.now() < deadline) {
- await sleep(pollingDelayMs(pollIntervalMs, attempt++));
+ // A retry-after the server just gave us wins over our own curve;
+ // the backoff resumes from where it was on the next normal poll.
+ await sleep(retryAfterMs > 0 ? retryAfterMs : pollingDelayMs(pollIntervalMs, attempt++));
+ retryAfterMs = 0;
let status: RememberStatusResponse;
@@ -455,6 +497,7 @@ export class MemWal {
} catch (err) {
const httpStatus = (err as { status?: number }).status ?? 0;
if (isTransientPollingStatus(httpStatus)) {
+ retryAfterMs = retryAfterDelayMs(err, deadline);
continue;
}
throw err;
@@ -609,9 +652,13 @@ export class MemWal {
}));
const pending = new Set(jobIds);
let attempt = 0;
+ let retryAfterMs = 0;
while (pending.size > 0 && Date.now() < deadline) {
- await sleep(pollingDelayMs(pollIntervalMs, attempt++));
+ // A retry-after the server just gave us wins over our own curve;
+ // the backoff resumes from where it was on the next normal poll.
+ await sleep(retryAfterMs > 0 ? retryAfterMs : pollingDelayMs(pollIntervalMs, attempt++));
+ retryAfterMs = 0;
const pendingIds = jobIds.filter((jobId) => pending.has(jobId));
if (pendingIds.length === 0) {
@@ -628,6 +675,7 @@ export class MemWal {
} catch (err) {
const httpStatus = (err as { status?: number }).status ?? 0;
if (isTransientPollingStatus(httpStatus)) {
+ retryAfterMs = retryAfterDelayMs(err, deadline);
continue;
}
throw err;
diff --git a/packages/sdk/test/remember-retry-after.test.mjs b/packages/sdk/test/remember-retry-after.test.mjs
new file mode 100644
index 000000000..8cc1342b7
--- /dev/null
+++ b/packages/sdk/test/remember-retry-after.test.mjs
@@ -0,0 +1,149 @@
+import assert from "node:assert/strict";
+import test from "node:test";
+
+import { MemWal } from "../dist/memwal.js";
+
+const originalFetch = globalThis.fetch;
+
+test.afterEach(() => {
+ globalThis.fetch = originalFetch;
+});
+
+/** Stub a job whose first status read is rate-limited, and record when each
+ * read arrived so the test can measure the gap the SDK actually waited. */
+function rateLimitedJob({ header, body }) {
+ const polledAt = [];
+ globalThis.fetch = async (url, init = {}) => {
+ const path = new URL(url).pathname;
+ if (path === "/version") {
+ return Response.json({
+ apiVersion: "1.0.0",
+ relayerVersion: "1.0.0",
+ minSupportedSdk: { typescript: "0.0.4" },
+ });
+ }
+ if (path === "/api/config") {
+ return Response.json({ packageId: "0x1", network: "testnet" });
+ }
+ if (path === "/api/remember" && init.method === "POST") {
+ return Response.json({ job_id: "limited-job", status: "pending" }, { status: 202 });
+ }
+ if (path === "/api/remember/limited-job") {
+ polledAt.push(Date.now());
+ if (polledAt.length === 1) {
+ return new Response(JSON.stringify(body), {
+ status: 429,
+ headers: {
+ "content-type": "application/json",
+ ...(header ? { "retry-after": header } : {}),
+ },
+ });
+ }
+ return Response.json({
+ job_id: "limited-job",
+ status: "done",
+ blob_id: "blob-1",
+ owner: "0x1",
+ namespace: "default",
+ });
+ }
+ throw new Error(`unexpected request ${path}`);
+ };
+ return polledAt;
+}
+
+function client() {
+ const memwal = MemWal.create({
+ key: new Uint8Array(32).fill(1),
+ accountId: "0x1",
+ serverUrl: "https://relayer.example",
+ });
+ memwal.buildSealSession = async () => "test-session";
+ return memwal;
+}
+
+// The rate limiter counts our own status reads, so re-polling on the 1.5s→10s
+// curve after a `retry_after_seconds: 60` re-trips it every attempt and the
+// loop starves itself — the job reads as "still uploading" for as long as the
+// caller waits, including long after it has failed. `pollIntervalMs: 1` makes
+// the unfixed behaviour ~1ms, so the wait being ≥ the stated backoff is
+// unambiguous.
+test("a rate-limited status poll waits the retry-after from the body", async () => {
+ const polledAt = rateLimitedJob({
+ body: { error: "Rate limit exceeded", retry_after_seconds: 0.5 },
+ });
+
+ const result = await client().rememberAndWait("fact", undefined, {
+ pollIntervalMs: 1,
+ timeoutMs: 10_000,
+ });
+
+ assert.equal(result.blob_id, "blob-1");
+ assert.equal(polledAt.length, 2);
+ assert.ok(
+ polledAt[1] - polledAt[0] >= 450,
+ `expected to honour the 500ms backoff, waited ${polledAt[1] - polledAt[0]}ms`,
+ );
+});
+
+test("a rate-limited status poll waits the Retry-After header", async () => {
+ const polledAt = rateLimitedJob({
+ header: "1",
+ body: { error: "Rate limit exceeded" },
+ });
+
+ const result = await client().rememberAndWait("fact", undefined, {
+ pollIntervalMs: 1,
+ timeoutMs: 10_000,
+ });
+
+ assert.equal(result.blob_id, "blob-1");
+ assert.ok(
+ polledAt[1] - polledAt[0] >= 950,
+ `expected to honour the 1s backoff, waited ${polledAt[1] - polledAt[0]}ms`,
+ );
+});
+
+test("a retry-after longer than the remaining budget is clamped to it", async () => {
+ // Honouring a 10-minute backoff must not turn a 1s wait into a 10-minute
+ // one. The budget still buys a final read at its own boundary — that read
+ // is free and may be the answer — but nothing beyond it.
+ const polledAt = [];
+ globalThis.fetch = async (url, init = {}) => {
+ const path = new URL(url).pathname;
+ if (path === "/version") {
+ return Response.json({
+ apiVersion: "1.0.0",
+ relayerVersion: "1.0.0",
+ minSupportedSdk: { typescript: "0.0.4" },
+ });
+ }
+ if (path === "/api/config") {
+ return Response.json({ packageId: "0x1", network: "testnet" });
+ }
+ if (path === "/api/remember" && init.method === "POST") {
+ return Response.json({ job_id: "slow-job", status: "pending" }, { status: 202 });
+ }
+ if (path === "/api/remember/slow-job") {
+ polledAt.push(Date.now());
+ if (polledAt.length === 1) {
+ return new Response(
+ JSON.stringify({ error: "Rate limit exceeded", retry_after_seconds: 600 }),
+ { status: 429, headers: { "content-type": "application/json" } },
+ );
+ }
+ return Response.json({ job_id: "slow-job", status: "pending" });
+ }
+ throw new Error(`unexpected request ${path}`);
+ };
+
+ const startedAt = Date.now();
+ await assert.rejects(
+ client().rememberAndWait("fact", undefined, { pollIntervalMs: 1, timeoutMs: 800 }),
+ /timed out/,
+ );
+ assert.ok(
+ Date.now() - startedAt < 3_000,
+ `a 600s retry-after must be clamped to the caller's own timeout, took ${Date.now() - startedAt}ms`,
+ );
+});
From 3cb55624c995c74bf9efc236f15be38c022fe9cc Mon Sep 17 00:00:00 2001
From: Harry Phan
Date: Thu, 17 Sep 2026 11:21:57 +0700
Subject: [PATCH 053/132] fix(mcp): return memwal_remember at accept so the
agent can keep working
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit
DEFAULT_REMEMBER_WAIT_MS was the full 90s ceiling, which reverses here to 0.
This overturns D1. D1 kept the ceiling so that "a result means it landed"
stayed true, but the blocking default could not deliver that anyway. The legs
are sequential, so a full-ceiling call runs ACCEPT_DEADLINE_MS (15_000) +
90_000 + WAIT_OVERSHOOT_GRACE_MS (10_000) = 115s, against an MCP TypeScript
SDK whose tools/call default is DEFAULT_REQUEST_TIMEOUT_MSEC = 60_000
(shared/protocol.js:8, applied at :712). On any host that does not raise its
own timeout the call aborts mid-wait and the retry writes the fact a second
time — the duplicate already recorded in docs/troubleshooting/overview.md.
Claude Code raises its ceiling and never saw this; other hosts did.
Nor is there a budget that both blocks meaningfully and fits: anything under
the 60s client timeout is <=35_000, which against a measured 30-75s spread
lands in the pending branch on nearly every call. It would pay the wait and
still return a job_id. So: do not wait at all.
Returning at accept is safe from disconnects — the job is a row in
remember_jobs driven by spawn_persisted_remember_preparation, not work held in
the MCP process. It is NOT safe from a job that fails after acceptance, which
is exactly what dev did all of 2026-09-17; that is why the pending message
points at memwal_remember_status and never claims the fact is stored.
Measured against relayer.dev.memwal.ai with this default: memwal_remember
returns in 1.4-2.6s, versus 92-93s observed on the blocking default.
memwal_analyze is unaffected and still exceeds 60s: its extraction leg is a
separate 60_000ms deadline that fully precedes the wait, so no value of this
constant helps it. That needs its own decision.
---
.../__tests__/remember-fast-return.test.ts | 23 ++++---
.../server/scripts/mcp/tools/remember-wait.ts | 68 ++++++++++---------
2 files changed, 48 insertions(+), 43 deletions(-)
diff --git a/services/server/scripts/mcp/__tests__/remember-fast-return.test.ts b/services/server/scripts/mcp/__tests__/remember-fast-return.test.ts
index b8381a298..8695f81b9 100644
--- a/services/server/scripts/mcp/__tests__/remember-fast-return.test.ts
+++ b/services/server/scripts/mcp/__tests__/remember-fast-return.test.ts
@@ -104,23 +104,26 @@ function textOf(result: unknown): string {
.join("\n");
}
-test("the default is the full wait, and zero is opt-in", () => {
- // D1: a result means the fact landed, so an unset budget blocks to
- // terminal. The in-between is the setting to avoid — a budget under the
- // real completion time pays the wait AND still returns pending.
- assert.equal(parseWaitBudget(undefined), MAX_REMEMBER_WAIT_MS);
- assert.equal(parseWaitBudget(""), MAX_REMEMBER_WAIT_MS);
- // ...and this file asked for the opt-in path at the top.
+test("the default returns at accept, and blocking is opt-in", () => {
+ // Reverses D1. A full-ceiling call runs ACCEPT_DEADLINE_MS + 90_000 +
+ // WAIT_OVERSHOOT_GRACE_MS = 115s against the MCP SDK's 60s default
+ // tools/call timeout, so it could not honour "a result means it landed"
+ // anyway — it aborted mid-wait and the retry wrote the fact twice. Every
+ // budget that fits under 60s (<=35_000) is the in-between to avoid.
+ assert.equal(parseWaitBudget(undefined), 0);
+ assert.equal(parseWaitBudget(""), 0);
assert.equal(REMEMBER_WAIT_MS, 0);
assert.equal(parseWaitBudget("0"), 0);
+ // An operator who wants the old behaviour still has it.
+ assert.equal(parseWaitBudget("90000"), MAX_REMEMBER_WAIT_MS);
});
test("a typo'd budget falls back to the default instead of picking one nobody asked for", () => {
// Number("10s") is NaN, and every NaN comparison is false — an unvalidated
// parse would sail past a range check.
- assert.equal(parseWaitBudget("10s"), MAX_REMEMBER_WAIT_MS);
- assert.equal(parseWaitBudget("abc"), MAX_REMEMBER_WAIT_MS);
- assert.equal(parseWaitBudget("-1"), MAX_REMEMBER_WAIT_MS);
+ assert.equal(parseWaitBudget("10s"), 0);
+ assert.equal(parseWaitBudget("abc"), 0);
+ assert.equal(parseWaitBudget("-1"), 0);
});
test("a budget past the ceiling is clamped, not honoured", () => {
diff --git a/services/server/scripts/mcp/tools/remember-wait.ts b/services/server/scripts/mcp/tools/remember-wait.ts
index fefc19875..524ba2ff4 100644
--- a/services/server/scripts/mcp/tools/remember-wait.ts
+++ b/services/server/scripts/mcp/tools/remember-wait.ts
@@ -61,47 +61,49 @@ export const MAX_REMEMBER_WAIT_MS = 90_000;
/**
* Default wait before `memwal_remember` hands back a job_id.
*
- * Blocks to terminal, so a successful call returns a real `blob_id` and the
- * agent can say the fact is stored. D1 settled this: returning at accept is a
- * product change on its own ticket, not a side effect of a latency fix, and
- * the contract callers have today is "a result means it landed".
+ * Zero — the tool returns at accept (~1.1s measured) and the agent continues
+ * with other work while the write finishes. This REVERSES D1, which had kept
+ * the full ceiling on the grounds that "a result means it landed" is the
+ * contract callers have. Two things decided it the other way:
*
- * That leaves the poll cadence as the part this file may legitimately shorten,
- * and the cadence work belongs to #902 — the wait itself stays.
+ * 1. The blocking default could not honour that contract anyway. The MCP
+ * TypeScript SDK defaults a tools/call to
+ * `DEFAULT_REQUEST_TIMEOUT_MSEC = 60_000` (@modelcontextprotocol/sdk,
+ * shared/protocol.js:8, applied at :712 as
+ * `options?.timeout ?? DEFAULT_REQUEST_TIMEOUT_MSEC`). The legs are
+ * sequential, so a full-ceiling call reaches ACCEPT_DEADLINE_MS (15_000)
+ * + 90_000 + WAIT_OVERSHOOT_GRACE_MS (10_000) = 115s against a client that
+ * gives up at 60. On any host that does not raise its own timeout the call
+ * aborts mid-wait and the retry writes the fact a second time — see
+ * docs/troubleshooting/overview.md. Claude Code raises its ceiling and so
+ * never saw this; other hosts did.
*
- * A budget BETWEEN zero and the real completion time is the one setting to
- * avoid. Against a measured 30–75s spread a 10s wait pays the 10s and still
- * lands in the pending branch on nearly every call: the cost of blocking with
- * none of the guarantee. So this is the full ceiling, and an operator who
- * genuinely wants accept-and-continue sets `MEMWAL_MCP_REMEMBER_WAIT_MS=0`
- * knowingly rather than inheriting it.
+ * 2. A budget between zero and the real completion time is the worst setting,
+ * and against a measured 30–75s spread every budget that fits under the
+ * 60s client timeout (≤35_000) is exactly that: it pays the wait and still
+ * lands in the pending branch on nearly every call. There is no value that
+ * both blocks meaningfully and fits. So: do not wait at all.
*
- * The pending branch is still reachable and still correct — a write slower
- * than the ceiling returns a job_id and says plainly it is not saved yet —
- * it is simply no longer the default path.
+ * Returning at accept is safe from disconnects because the job is a row in
+ * `remember_jobs` driven by the relayer
+ * (`spawn_persisted_remember_preparation` in
+ * services/server/src/routes/remember.rs), not work held in this process.
+ * Closing the client does not cancel it.
*
- * KNOWN CONSTRAINT, not yet acted on. The MCP TypeScript SDK defaults a
- * tools/call to `DEFAULT_REQUEST_TIMEOUT_MSEC = 60_000`
- * (@modelcontextprotocol/sdk, shared/protocol.js:8, applied at :712 as
- * `options?.timeout ?? DEFAULT_REQUEST_TIMEOUT_MSEC`). A host that sets no
- * timeout of its own therefore aborts at 60s, and the pending branch — whose
- * entire purpose is to hand back a job_id — lands 30s after it has already
- * given up. docs/troubleshooting/overview.md records the symptom already
- * ("can exceed the MCP host's tool-call timeout"), including that a retry
- * duplicates the memory. Hosts differ: Claude Code sets its own, far larger
- * ceiling, so this does not bite there.
+ * It is NOT safe from a job that fails after acceptance — that is real, and
+ * dev proved it on 2026-09-17 when every register transaction was rejected
+ * post-accept. Only a later `memwal_remember_status` call surfaces that, which
+ * is why the pending message points at it and refuses to claim the fact is
+ * stored.
*
- * The arithmetic for anyone changing this: the legs are sequential, so the
- * ceiling is ACCEPT_DEADLINE_MS (15_000) + this budget +
- * WAIT_OVERSHOOT_GRACE_MS (10_000). Fitting under 60_000 needs a budget of
- * 35_000 or less — against a measured 26.7-47.9s time-to-saved, which is the
- * D1 trade, not a free win.
+ * An operator who wants the old always-block behaviour sets
+ * `MEMWAL_MCP_REMEMBER_WAIT_MS=90000` knowingly.
*
- * This constant cannot fix `memwal_analyze` either way: its extraction leg is
- * a 60_000ms deadline (analyze.ts) that fully precedes the wait, so analyze
+ * This constant does not fix `memwal_analyze`: its extraction leg is a
+ * 60_000ms deadline (analyze.ts) that fully precedes the wait, so analyze
* exceeds 60s for ANY value here, including 0. That needs its own decision.
*/
-const DEFAULT_REMEMBER_WAIT_MS = MAX_REMEMBER_WAIT_MS;
+const DEFAULT_REMEMBER_WAIT_MS = 0;
From 87573c727dd2d97f4ddc11eed8bd4bfca0e0da44 Mon Sep 17 00:00:00 2001
From: Harry Phan
Date: Thu, 17 Sep 2026 12:34:15 +0700
Subject: [PATCH 054/132] fix(mcp): stop misreporting the rate limit, and keep
analyze inside the client ceiling
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit
Two ways a tool call told the agent something that was not true.
1. The 429 message asserted "the limit is per delegate key and resets in about
Ns; retry after that". The relayer runs three limiter layers with different
windows (services/server/src/rate_limit.rs): delegate_key and account_burst
per minute, account_sustained per HOUR. On the hourly layer both halves were
wrong — it is the account's budget, not the key's, and retry_after_seconds
is a token-bucket refill hint, not a reset. An agent that believed the
message waited the advised window and retried into the same denial.
Measured on dev 2026-09-17: a 429 carrying account_sustained /
1000 weighted-requests/hour / retry_after_seconds 300, retried after 300s
and again after 420s, denied both times.
The message now reads `layer` and `limit` out of the body the relayer
already sends, names whose budget it is, and on the hourly layer says
plainly that waiting the advised number out is not a reset. A 429 whose body
will not parse still reports the limit rather than inventing a layer.
2. memwal_analyze budgeted its extraction leg at 60_000 — exactly the MCP SDK's
DEFAULT_REQUEST_TIMEOUT_MSEC. Extraction runs the extractor LLM inline, so a
slow one raced the host's abort and usually lost: no message the agent could
act on, on the one endpoint with no idempotency key, where the caller's
retry re-extracts and re-stores every fact. It is now derived from that
ceiling with 15s of headroom, so the tool raises its own timeout first.
4 tests. tsc clean. MCP suite 129 passed / 14 failed, against a 125/14 baseline
on this branch — same 14, which need ports the sandbox denies.
---
.../mcp/__tests__/analyze-fast-return.test.ts | 8 +-
.../mcp/__tests__/remember-deadline.test.ts | 104 ++++++++++++++++++
services/server/scripts/mcp/tools/analyze.ts | 8 +-
.../server/scripts/mcp/tools/remember-wait.ts | 102 ++++++++++++++++-
4 files changed, 213 insertions(+), 9 deletions(-)
diff --git a/services/server/scripts/mcp/__tests__/analyze-fast-return.test.ts b/services/server/scripts/mcp/__tests__/analyze-fast-return.test.ts
index 0270013b6..5b8619377 100644
--- a/services/server/scripts/mcp/__tests__/analyze-fast-return.test.ts
+++ b/services/server/scripts/mcp/__tests__/analyze-fast-return.test.ts
@@ -14,10 +14,10 @@
* back paired with the fact it carries, so a later partial failure is
* actionable.
*/
-// A small non-zero wait so the bounded-wait branch is reachable quickly. The
-// shipped default is the full 90s ceiling (D1 kept the block-to-terminal
-// contract), so leaving it unset would make every assertion below about a
-// partly landed batch wait out that ceiling instead of returning.
+// A small non-zero wait so the bounded-wait branch is reachable at all. The
+// shipped default is now 0 — the tools return at accept — so leaving it unset
+// would skip the wait entirely and every assertion below about a partly landed
+// batch would have nothing to observe.
// Set before the dynamic import, because the budget is read once at load.
process.env.MEMWAL_MCP_REMEMBER_WAIT_MS = "500";
diff --git a/services/server/scripts/mcp/__tests__/remember-deadline.test.ts b/services/server/scripts/mcp/__tests__/remember-deadline.test.ts
index 87958265d..bf1a0ceb4 100644
--- a/services/server/scripts/mcp/__tests__/remember-deadline.test.ts
+++ b/services/server/scripts/mcp/__tests__/remember-deadline.test.ts
@@ -294,6 +294,110 @@ test("a rate-limited batch says the facts were not saved", async (t) => {
assert.match(textOf(res), /NOT\s+SAVED/i);
});
+/**
+ * The relayer runs three limiter layers with different windows —
+ * `delegate_key` and `account_burst` per minute, `account_sustained` per HOUR
+ * (services/server/src/rate_limit.rs). The message used to assert "the limit is
+ * per delegate key and resets in about Ns; retry after that" for all of them.
+ *
+ * Both halves were wrong on the hourly layer, and an agent that believed them
+ * waited out the advised window and retried into the same denial. Measured on
+ * dev 2026-09-17: a 429 carrying `account_sustained` / `1000
+ * weighted-requests/hour` / `retry_after_seconds: 300`, retried after 300s and
+ * again after 420s, denied both times.
+ */
+function rejectsWithRateLimitBody(layer: string, limit: string, retryAfterSeconds: number) {
+ return async () => {
+ const e = new Error(
+ `Walrus Memory server error (429): ` +
+ JSON.stringify({
+ error: "Rate limit exceeded",
+ layer,
+ limit,
+ retry_after_seconds: retryAfterSeconds,
+ }),
+ );
+ Object.assign(e, { status: 429, serverCode: "Rate limit exceeded", retryAfterSeconds });
+ throw e;
+ };
+}
+
+test("a rate-limit message names the layer the relayer actually denied on", async (t) => {
+ const client = await clientFor(
+ sessionRejecting(
+ rejectsWithRateLimitBody("account_sustained", "1000 weighted-requests/hour", 300),
+ ),
+ t,
+ );
+ const res = await client.callTool({ name: "memwal_remember", arguments: { text: "a fact" } });
+ assert.equal((res as { isError?: boolean }).isError, true);
+ const text = textOf(res);
+
+ assert.match(text, /NOT\s+SAVED/i);
+ assert.match(text, /account_sustained/);
+ assert.match(text, /1000 weighted-requests\/hour/);
+ // It is the account's budget, not the delegate key's.
+ assert.match(text, /per account/);
+ assert.doesNotMatch(text, /per delegate key/);
+ // And the advised number must not be sold as a reset.
+ assert.match(text, /not a reset/i);
+ assert.doesNotMatch(text, /resets in/i);
+});
+
+test("a per-minute layer keeps the plain retry advice", async (t) => {
+ const client = await clientFor(
+ sessionRejecting(rejectsWithRateLimitBody("delegate_key", "60 weighted-requests/min", 60)),
+ t,
+ );
+ const res = await client.callTool({ name: "memwal_remember", arguments: { text: "a fact" } });
+ const text = textOf(res);
+ assert.match(text, /delegate_key/);
+ assert.match(text, /per delegate key/);
+ assert.match(text, /Retry after ~60s/);
+ // The hourly caveat belongs only to the hourly layer.
+ assert.doesNotMatch(text, /not a reset/i);
+});
+
+test("a 429 with no parsable body still reports the limit", async (t) => {
+ const client = await clientFor(
+ sessionRejecting(rejectsWith(429, "Rate limit exceeded", 60, { job_id: "nope", status: "pending" })),
+ t,
+ );
+ const res = await client.callTool({ name: "memwal_remember", arguments: { text: "a fact" } });
+ const text = textOf(res);
+ assert.match(text, /NOT\s+SAVED/i);
+ assert.match(text, /Retry after ~60s/);
+ // No body means no layer to name — it must not invent one.
+ assert.doesNotMatch(text, /Limit hit:/);
+});
+
+/**
+ * `memwal_analyze`'s extraction leg runs the extractor LLM inline, so it is the
+ * one leg long enough to race the MCP client's own ceiling. Budgeted level with
+ * that ceiling it lost the race, and `analyze` carries no idempotency key, so
+ * the caller's retry re-extracted and re-stored every fact.
+ */
+test("the analyze extraction deadline fits under the client ceiling", async () => {
+ const { DEFAULT_REQUEST_TIMEOUT_MSEC } = await import(
+ "@modelcontextprotocol/sdk/shared/protocol.js"
+ );
+ const { ANALYZE_EXTRACTION_DEADLINE_MS, MCP_CLIENT_DEFAULT_TIMEOUT_MS } = await import(
+ "../tools/remember-wait.js"
+ );
+
+ // The constant we derive from must be the one the SDK actually applies.
+ assert.equal(MCP_CLIENT_DEFAULT_TIMEOUT_MS, DEFAULT_REQUEST_TIMEOUT_MSEC);
+ assert.ok(
+ ANALYZE_EXTRACTION_DEADLINE_MS < DEFAULT_REQUEST_TIMEOUT_MSEC,
+ `extraction ${ANALYZE_EXTRACTION_DEADLINE_MS}ms must finish before the client gives up ` +
+ `at ${DEFAULT_REQUEST_TIMEOUT_MSEC}ms`,
+ );
+ assert.ok(
+ DEFAULT_REQUEST_TIMEOUT_MSEC - ANALYZE_EXTRACTION_DEADLINE_MS >= 10_000,
+ `only ${DEFAULT_REQUEST_TIMEOUT_MSEC - ANALYZE_EXTRACTION_DEADLINE_MS}ms of headroom`,
+ );
+});
+
test("a non-retryable error is not retried", async (t) => {
// A 500 could have been thrown after a write started; retrying bulk there
// would store every fact twice.
diff --git a/services/server/scripts/mcp/tools/analyze.ts b/services/server/scripts/mcp/tools/analyze.ts
index 2a92b5f36..b2d08698d 100644
--- a/services/server/scripts/mcp/tools/analyze.ts
+++ b/services/server/scripts/mcp/tools/analyze.ts
@@ -4,6 +4,7 @@ import type { MemWalSession } from "../auth.js";
import { TOOL_METADATA } from "./annotations.js";
import { wrapTool, explorerFooter } from "./util.js";
import {
+ ANALYZE_EXTRACTION_DEADLINE_MS,
REMEMBER_POLL_INTERVAL_MS,
REMEMBER_WAIT_MS,
pendingBulkMessage,
@@ -76,7 +77,12 @@ export function registerAnalyzeTool(
// where it allows a remember 30s. The 15s accept ceiling would
// have cut off healthy extraction on any transcript long enough
// to be worth extracting from.
- { idempotent: false, deadlineMs: 60_000 },
+ //
+ // Budgeted just under the MCP client's own ceiling rather than
+ // level with it: at 60_000 this leg raced the host's abort and
+ // the caller lost the error message, on the one endpoint whose
+ // retry duplicates every extracted fact.
+ { idempotent: false, deadlineMs: ANALYZE_EXTRACTION_DEADLINE_MS },
);
const facts = accepted.facts ?? [];
diff --git a/services/server/scripts/mcp/tools/remember-wait.ts b/services/server/scripts/mcp/tools/remember-wait.ts
index 524ba2ff4..221901e9c 100644
--- a/services/server/scripts/mcp/tools/remember-wait.ts
+++ b/services/server/scripts/mcp/tools/remember-wait.ts
@@ -380,6 +380,43 @@ export function withAcceptDeadline(
);
}
+/**
+ * The MCP TypeScript SDK's default `tools/call` timeout —
+ * `DEFAULT_REQUEST_TIMEOUT_MSEC` in @modelcontextprotocol/sdk
+ * (shared/protocol.js:8, applied at :712 as
+ * `options?.timeout ?? DEFAULT_REQUEST_TIMEOUT_MSEC`).
+ *
+ * A host that does not raise its own ceiling aborts the call at this mark. An
+ * abort is strictly worse than a timeout we raise ourselves: the caller gets no
+ * message it can act on, and on a non-idempotent endpoint its retry writes
+ * everything twice. So every leg this file bounds must finish inside it.
+ */
+export const MCP_CLIENT_DEFAULT_TIMEOUT_MS = 60_000;
+
+/**
+ * Headroom left under the client ceiling for JSON-RPC framing, transport and
+ * the relayer's own response write. A leg budgeted at exactly the ceiling loses
+ * the race it was meant to win.
+ */
+const CLIENT_TIMEOUT_HEADROOM_MS = 15_000;
+
+/**
+ * Deadline for `memwal_analyze`'s extraction leg.
+ *
+ * `/api/analyze` runs the extractor LLM inline before it answers, so this leg
+ * is real work, not an accept — 15s would cut off healthy extraction on any
+ * transcript worth extracting from. But it was budgeted at the full 60_000,
+ * exactly the client ceiling, so a slow extraction raced the host's abort and
+ * usually lost: the agent saw a dead call instead of a message, and `analyze`
+ * is the one endpoint with no idempotency key, so retrying it stores every
+ * extracted fact a second time.
+ *
+ * Derived from the ceiling rather than written as a literal so the two cannot
+ * drift apart.
+ */
+export const ANALYZE_EXTRACTION_DEADLINE_MS =
+ MCP_CLIENT_DEFAULT_TIMEOUT_MS - CLIENT_TIMEOUT_HEADROOM_MS;
+
/** Bound a status wait at its own budget plus the overshoot grace. */
export function withWaitDeadline(work: Promise, budgetMs: number): Promise {
return withDeadline(
@@ -428,6 +465,58 @@ function advisedCooldownMs(err: unknown): number {
return typeof secs === "number" && secs > 0 ? secs * 1000 : 1_000;
}
+/**
+ * What the relayer actually said it denied on.
+ *
+ * `rate_limit_response` (services/server/src/rate_limit.rs) answers a 429 with
+ * `{error, layer, limit, retry_after_seconds}`, and the SDK embeds that body
+ * verbatim in the error message. Read it rather than asserting a scope: there
+ * are three layers with different windows — `delegate_key` and `account_burst`
+ * per minute, `account_sustained` per HOUR — and naming the wrong one sends
+ * the caller to wait out a window that was never the one it hit.
+ */
+function rateLimitFacts(err: unknown): { layer?: string; limit?: string } {
+ const text = String((err as { message?: string } | null)?.message ?? "");
+ const open = text.indexOf("{");
+ const close = text.lastIndexOf("}");
+ if (open < 0 || close <= open) return {};
+ try {
+ const body = JSON.parse(text.slice(open, close + 1)) as Record;
+ return {
+ layer: typeof body.layer === "string" ? body.layer : undefined,
+ limit: typeof body.limit === "string" ? body.limit : undefined,
+ };
+ } catch {
+ // A body we cannot parse is not a reason to lose the 429 itself.
+ return {};
+ }
+}
+
+/** Whose budget the named layer belongs to. */
+function scopeOfLayer(layer: string | undefined): string {
+ if (layer === "delegate_key") return "per delegate key";
+ if (layer === "account_burst" || layer === "account_sustained") return "per account";
+ return "on this account";
+}
+
+/**
+ * The retry advice.
+ *
+ * `retry_after_seconds` is a refill hint from a token bucket, NOT a reset. On
+ * the hourly `account_sustained` layer the relayer still advises ~300s, which
+ * buys back only a slice of a 1000/hour budget — waiting it out and retrying
+ * re-trips the limit whenever the budget is genuinely spent. Saying "resets in
+ * Ns; retry after that" turned that into a retry loop that never converges, so
+ * the message now says what the number is worth.
+ */
+function retryAdvice(secs: number, layer: string | undefined): string {
+ return layer === "account_sustained"
+ ? `The relayer advises retrying in ~${secs}s, but that is a partial refill of an ` +
+ `hourly budget, not a reset — if the budget is spent, a retry then fails again. ` +
+ `Report the failure rather than waiting and retrying in a loop.`
+ : `Retry after ~${secs}s.`;
+}
+
/** Honour the relayer's `retry_after` instead of surfacing a raw 429.
*
* Once the per-delegate-key budget is spent the write is simply never made,
@@ -447,14 +536,19 @@ export async function withRelayerRetry(work: () => Promise, what: string):
if (attempt === RELAYER_RETRY_ATTEMPTS || cooldown > MAX_ABSORBED_COOLDOWN_MS) {
const secs = Math.ceil(cooldown / 1000);
const limited = (err as { status?: number }).status === 429;
+ const { layer, limit } = limited ? rateLimitFacts(err) : {};
+ const hit = layer
+ ? `Limit hit: ${layer} ${scopeOfLayer(layer)}` +
+ (limit ? ` (${limit})` : "") + ". "
+ : "";
const e = new Error(
limited
? `Walrus Memory rate limit reached while trying to ${what}. THE FACT WAS NOT ` +
`SAVED — tell the user it could not be stored rather than that it is being ` +
- `saved. The limit is per delegate key and resets in about ${secs}s; retry ` +
- `after that. To spend less of the budget, save several facts with one ` +
- `memwal_remember_bulk call instead of repeated memwal_remember calls, and ` +
- `settle a batch with a single memwal_remember_status(job_ids=[...]).`
+ `saved. ${hit}${retryAdvice(secs, layer)} To spend less of the budget, save ` +
+ `several facts with one memwal_remember_bulk call instead of repeated ` +
+ `memwal_remember calls, and settle a batch with a single ` +
+ `memwal_remember_status(job_ids=[...]).`
: `Walrus Memory could not ${what}: the relayer's credential check is ` +
`temporarily unavailable. THE FACT WAS NOT SAVED. Retry in about ${secs}s.`,
);
From 9e021942bfb5a5091a5e1198a096b4ac11f1e064 Mon Sep 17 00:00:00 2001
From: Harry Phan
Date: Thu, 17 Sep 2026 13:49:25 +0700
Subject: [PATCH 055/132] fix(mcp): stop the handoff test pinning a tool cold
start no longer serves
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit
d1df49f9 made the cold-start list a floor: only what the oldest supported
relayer serves, plus the locally served memwal_login. memwal_remember_status
is on dev but not prod/staging, so it left the pre-login tools/list — while
this test still pinned it, and MCP / Integration (login handoff) has failed
on every commit since:
not ok 64 - auth-required mode picks up credentials mid-session without a restart
- memwal_remember_status: { annotations: {...}, title: 'Check a Remember Job' }
The assertion runs before login, so it reads the floor, not the list the
relayer serves after handoff. Drop the entry and say so in the comment,
which still claimed cold start matched the post-handoff list — the claim
that put the entry there and would put it back.
---
packages/mcp/test/login-handoff.test.mjs | 12 ++++++------
1 file changed, 6 insertions(+), 6 deletions(-)
diff --git a/packages/mcp/test/login-handoff.test.mjs b/packages/mcp/test/login-handoff.test.mjs
index 18340a0c0..2cfb22627 100644
--- a/packages/mcp/test/login-handoff.test.mjs
+++ b/packages/mcp/test/login-handoff.test.mjs
@@ -167,8 +167,12 @@ test("auth-required mode picks up credentials mid-session without a restart", as
const init = await waitFor((m) => m.id === 1 && m.result);
assert.equal(init.result.serverInfo.name, "memwal");
- // Pre-login discovery must expose the same safety metadata clients will
- // receive after the bridge hands off to the remote relayer.
+ // Pre-login discovery is the cold-start FLOOR, not the post-handoff list:
+ // it carries only what the oldest supported relayer serves, plus the
+ // locally served memwal_login (see BASELINE_RELAYER_TOOLS). The relayer's
+ // own tools/list replaces it once the session is up, so a tool that has
+ // reached dev but not prod — memwal_remember_status today — is absent
+ // here on purpose. Assert the safety metadata of the floor itself.
send({ jsonrpc: "2.0", id: 10, method: "tools/list", params: {} });
const listed = await waitFor((m) => m.id === 10 && m.result);
const metadata = Object.fromEntries(
@@ -183,10 +187,6 @@ test("auth-required mode picks up credentials mid-session without a restart", as
title: "Remember Multiple Facts",
annotations: { readOnlyHint: false, destructiveHint: false },
},
- memwal_remember_status: {
- title: "Check a Remember Job",
- annotations: { readOnlyHint: true, destructiveHint: false },
- },
memwal_recall: {
title: "Recall Memories",
annotations: { readOnlyHint: true, destructiveHint: false },
From 8e50a6af1ba52bbd0b3627fb2f0b897f6ac87860 Mon Sep 17 00:00:00 2001
From: Harry Phan
Date: Thu, 17 Sep 2026 14:03:24 +0700
Subject: [PATCH 056/132] fix(sidecar): give direct storage-node uploads a
timeout knob
createWalrusClient sets an explicit 120s timeout on the upload-relay
branch and set none on the direct branch, so direct uploads silently ran
on StorageNodeClient's own default of 30s -- four times tighter, on the
path a deployment takes precisely when it has no relay to fan out for it,
and with no env to tune it. WALRUS_RELAY_TIMEOUT_MS looks like the knob
and is read only inside the relay branch, so on a direct deployment it
does nothing.
dev is the only environment on that path (WALRUS_DIRECT_UPLOAD=true, left
over from the v1 migration). Its uploads measured 51s end to end on 09-14
when the suite was green, so 30s per node request was already close, and
a loaded testnet now spends every remember there:
phase="registered" error="The operation was aborted due to timeout"
code=SHARED_SERVICE_UNAVAILABLE
register succeeds, the blob is on chain, and the sliver write times out.
The default stays 30s, so no environment changes behaviour on this
commit; what it gains is the knob, via WALRUS_STORAGE_NODE_TIMEOUT_MS.
Raising it is a deliberate per-environment call, because the value is per
NODE request: a larger one also lengthens how long a single unresponsive
node stalls a shard, which is why it is bounded under
WALRUS_UPLOAD_ACQUIRE_TIMEOUT_MS.
---
services/server/scripts/sidecar/clients.ts | 4 ++++
services/server/scripts/sidecar/config.ts | 24 ++++++++++++++++++++++
2 files changed, 28 insertions(+)
diff --git a/services/server/scripts/sidecar/clients.ts b/services/server/scripts/sidecar/clients.ts
index 40aada973..79c2ae861 100644
--- a/services/server/scripts/sidecar/clients.ts
+++ b/services/server/scripts/sidecar/clients.ts
@@ -23,6 +23,7 @@ import {
UPLOAD_RELAY_TIP_TIMEOUT_MS,
WALRUS_CLIENT_MAX_AGE_MS,
WALRUS_DIRECT_UPLOAD,
+ WALRUS_STORAGE_NODE_TIMEOUT_MS,
WALRUS_PACKAGE_ID,
WALRUS_STAKING_POOL_ID,
WALRUS_SYSTEM_OBJECT_ID,
@@ -70,6 +71,9 @@ function createWalrusClient(): WalrusClient {
!WALRUS_DIRECT_UPLOAD && !!WALRUS_UPLOAD_RELAY_URL && WALRUS_UPLOAD_RELAY_URL !== "none";
const baseConfig = {
suiClient: suiClient as any,
+ // Applies to both branches: the relay still reads slivers back from
+ // storage nodes, and the direct branch has nothing else to set it.
+ storageNodeClientOptions: { timeout: WALRUS_STORAGE_NODE_TIMEOUT_MS },
...(useRelay
? {
uploadRelay: {
diff --git a/services/server/scripts/sidecar/config.ts b/services/server/scripts/sidecar/config.ts
index a755c7097..d224d9c9f 100644
--- a/services/server/scripts/sidecar/config.ts
+++ b/services/server/scripts/sidecar/config.ts
@@ -234,6 +234,30 @@ export const WALRUS_UPLOAD_ACQUIRE_TIMEOUT_MS = parsePositiveIntEnv(
1_000,
180_000
);
+/** Per-request timeout for talking to a Walrus storage node.
+ *
+ * `createWalrusClient` gives the upload-relay branch an explicit 120s
+ * timeout and gave the direct branch none, so direct uploads silently took
+ * StorageNodeClient's own 30s default -- four times tighter, on the path a
+ * deployment runs precisely when it has no relay fanning out for it. dev
+ * uploads measured 51s end to end on a healthy day, so 30s per node left
+ * almost no headroom, and a loaded testnet turned every remember into
+ * "The operation was aborted due to timeout" after the blob was already
+ * registered on chain.
+ *
+ * The default keeps today's behaviour so no environment shifts silently;
+ * what changes is that the direct path now HAS a knob. Note this is per
+ * NODE request, not per write: raising it also lengthens how long a single
+ * unresponsive node stalls a shard, so keep it under
+ * WALRUS_UPLOAD_ACQUIRE_TIMEOUT_MS.
+ */
+export const WALRUS_STORAGE_NODE_TIMEOUT_MS = parsePositiveIntEnv(
+ "WALRUS_STORAGE_NODE_TIMEOUT_MS",
+ 30_000,
+ 1_000,
+ 180_000
+);
+
export const WALRUS_UPLOAD_EFFECTS_RETRY_DELAYS_MS = [2_000, 5_000, 10_000, 20_000, 40_000] as const;
export const DURABLE_UPLOAD_PROTOCOL_VERSION = 3;
From 5377867cf6561a72e9dab96c9139a50cd68067d3 Mon Sep 17 00:00:00 2001
From: Harry Phan
Date: Thu, 17 Sep 2026 15:28:00 +0700
Subject: [PATCH 057/132] fix(relayer,mcp): say when accepted writes are not
landing
/health reported a healthy write path throughout a total Walrus outage.
On dev 2026-09-17 a remember failed on an upload timeout and /health
answered write_ready=true 234ms later, while the public testnet publisher
itself returned nothing for 120s and six of eight sampled storage nodes
sat frozen on the same two checkpoints for over half an hour. Every write
was accepted and every write died minutes later, and nothing the client
could read said so.
write_ready was not lying: it means the relayer accepts and durably
queues a write, which it did -- acceptance is decoupled from completion
by design. It also cannot be repurposed. CI's wait-for-relayer gate
blocks on `write_ready is True`, so flipping it during a downstream
outage would hang every deploy behind the outage it is reporting.
So the third state goes on `writes`, which already carries write-path
state and which nothing gates on: "ok", "degraded", "paused". Degraded
means recent durable writes failed and none landed. Silence is not
failure -- a window that finished nothing stays "ok", so a quiet
deployment never reports itself broken -- and neither is one lost write:
it takes DURABLE_WRITE_FAILURE_THRESHOLD failures with zero successes,
which is the shape of an outage and not of bad luck. An operator pause
outranks it.
The probe fails open, like the Postgres one: a database that cannot
answer must not invent an outage. It caches for 30s rather than
write_ready's 2s because it aggregates over remember_jobs, which is
indexed on owner and on status but not on updated_at alone.
memwal_health now prints writes=degraded, so an agent stops queueing into
a hole instead of discovering it one failed job at a time.
---
services/server/scripts/mcp/tools/health.ts | 14 +++-
services/server/src/routes/admin.rs | 80 ++++++++++++++++++++-
services/server/src/storage/db.rs | 53 ++++++++++++++
services/server/src/types.rs | 26 +++++--
4 files changed, 164 insertions(+), 9 deletions(-)
diff --git a/services/server/scripts/mcp/tools/health.ts b/services/server/scripts/mcp/tools/health.ts
index b1ca2a8d0..dd403170f 100644
--- a/services/server/scripts/mcp/tools/health.ts
+++ b/services/server/scripts/mcp/tools/health.ts
@@ -40,8 +40,18 @@ export function registerHealthTool(
const relayerNote = session.publicRelayerUrl
? ` relayer=${session.publicRelayerUrl}`
: "";
- const pausedNote = extra.writes === "paused" ? " writes=paused" : "";
- const writeNote = `${readyNote}${pausedNote}`;
+ // "degraded" is the case write_ready cannot express: the relayer
+ // still accepts and durably queues a write, so write_ready stays
+ // true, but recent durable writes are failing and none are
+ // landing. Say so, or an agent keeps queueing writes that fail
+ // minutes later.
+ const writesStateNote =
+ extra.writes === "paused"
+ ? " writes=paused"
+ : extra.writes === "degraded"
+ ? " writes=degraded (accepted, but recent writes are failing downstream)"
+ : "";
+ const writeNote = `${readyNote}${writesStateNote}`;
return {
content: [
{
diff --git a/services/server/src/routes/admin.rs b/services/server/src/routes/admin.rs
index 501e7b03e..14229c076 100644
--- a/services/server/src/routes/admin.rs
+++ b/services/server/src/routes/admin.rs
@@ -150,11 +150,25 @@ pub async fn health(State(state): State>) -> Json
ask: ASK_SYSTEM_PROMPT_VERSION.to_string(),
},
write_ready: write_ready(&state).await,
- writes: writes_health_status(state.config.writes_paused),
+ writes: writes_health_status(
+ state.config.writes_paused,
+ durable_writes_degraded_probe(&state).await,
+ ),
})
}
const WRITE_READY_CACHE_TTL: std::time::Duration = std::time::Duration::from_secs(2);
+/// How far back `/health` looks when judging whether durable writes land.
+/// Long enough to span a few upload attempts with their backoff, short
+/// enough that recovery shows up without an operator waiting.
+const DURABLE_WRITE_WINDOW: std::time::Duration = std::time::Duration::from_secs(15 * 60);
+/// Failures inside the window before the write path is called degraded.
+/// Writes fail individually all the time; one or two is not an outage.
+const DURABLE_WRITE_FAILURE_THRESHOLD: i64 = 3;
+/// Cached far longer than `write_ready`: this one aggregates over
+/// `remember_jobs`, which has no index on `updated_at` alone, so it must
+/// not run on every load-balancer tick.
+const DURABLE_WRITE_CACHE_TTL: std::time::Duration = std::time::Duration::from_secs(30);
const WRITE_READY_PROBE_TIMEOUT: std::time::Duration = std::time::Duration::from_secs(1);
/// Neon refuses `smgrextend` once cluster size is at the cap; treat less
/// than 1MB remaining as not writable so `/health` trips before the next
@@ -316,6 +330,55 @@ fn postgres_can_accept_writes(used_bytes: i64, max_bytes: i64) -> bool {
static WRITE_READY_CACHE: std::sync::Mutex> =
std::sync::Mutex::new(None);
+static DURABLE_WRITE_CACHE: std::sync::Mutex > =
+ std::sync::Mutex::new(None);
+
+/// Whether recent durable writes say the write path is degraded.
+///
+/// Silence is not failure: a window that finished no writes leaves this
+/// false, so a quiet deployment never reports itself broken. Nor is a
+/// single loss -- only a window that failed at least
+/// `DURABLE_WRITE_FAILURE_THRESHOLD` times AND landed nothing at all
+/// counts, which is what a downstream outage looks like and what a run of
+/// unlucky individual writes does not.
+fn durable_writes_degraded(failed: i64, succeeded: i64) -> bool {
+ if succeeded > 0 {
+ return false;
+ }
+ failed >= DURABLE_WRITE_FAILURE_THRESHOLD
+}
+
+/// Reads recent write outcomes behind `DURABLE_WRITE_CACHE_TTL`.
+///
+/// Fails open, like the Postgres probe: a database that cannot answer this
+/// is not evidence that Walrus is down, and `/health` must not invent an
+/// outage out of its own query failing.
+async fn durable_writes_degraded_probe(state: &std::sync::Arc) -> bool {
+ {
+ let cache = DURABLE_WRITE_CACHE.lock().unwrap_or_else(|e| e.into_inner());
+ if let Some((at, degraded)) = *cache {
+ if at.elapsed() < DURABLE_WRITE_CACHE_TTL {
+ return degraded;
+ }
+ }
+ }
+
+ let degraded = match state.db.recent_write_outcomes(DURABLE_WRITE_WINDOW).await {
+ Ok((failed, succeeded)) => durable_writes_degraded(failed, succeeded),
+ Err(err) => {
+ tracing::warn!(
+ error = %err,
+ "durable-write outcome probe failed; leaving the write path reported healthy"
+ );
+ false
+ }
+ };
+ if let Ok(mut cache) = DURABLE_WRITE_CACHE.lock() {
+ *cache = Some((std::time::Instant::now(), degraded));
+ }
+ degraded
+}
+
/// GET /version
pub async fn version() -> Json {
Json(crate::compatibility::version_response())
@@ -1306,6 +1369,21 @@ mod tests {
);
}
+ #[test]
+ fn durable_writes_degraded_only_when_nothing_lands() {
+ let threshold = super::DURABLE_WRITE_FAILURE_THRESHOLD;
+ // No finished writes at all: a quiet window, not an outage.
+ assert!(!super::durable_writes_degraded(0, 0));
+ // Below the threshold: individual writes do fail.
+ assert!(!super::durable_writes_degraded(threshold - 1, 0));
+ // At and past it, with nothing landing: this is the outage shape.
+ assert!(super::durable_writes_degraded(threshold, 0));
+ assert!(super::durable_writes_degraded(threshold * 100, 0));
+ // A single success means the path works, however many failed
+ // alongside it -- Walrus is storing blobs, these writes lost.
+ assert!(!super::durable_writes_degraded(threshold * 100, 1));
+ }
+
#[test]
fn postgres_can_accept_writes_false_at_or_within_1mb_of_cap() {
let max = 3072 * 1024 * 1024;
diff --git a/services/server/src/storage/db.rs b/services/server/src/storage/db.rs
index 4745f7fd7..4b1364215 100644
--- a/services/server/src/storage/db.rs
+++ b/services/server/src/storage/db.rs
@@ -1890,6 +1890,59 @@ impl VectorDb {
/// One-shot is preserved because the claim is a single conditional UPDATE
/// over these ids — a concurrent recall that got there first claims them
/// and this one is handed back nothing to report.
+ /// How durable writes that finished inside `window` turned out,
+ /// across every owner: `(failed, succeeded)`.
+ ///
+ /// `write_ready` is a sidecar probe AND a Postgres size check. Neither
+ /// can see Walrus refusing every upload, so a total Walrus outage left
+ /// `/health` reporting a healthy write path while every remember
+ /// failed minutes after being accepted. This is the missing term.
+ ///
+ /// Counted rather than listed, and read behind a cache measured in
+ /// tens of seconds, because `remember_jobs` is indexed on `owner` and
+ /// on `status` but not on `updated_at` alone.
+ pub async fn recent_write_outcomes(
+ &self,
+ window: std::time::Duration,
+ ) -> Result<(i64, i64), AppError> {
+ let started = std::time::Instant::now();
+ let since =
+ chrono::Utc::now() - chrono::Duration::from_std(window).unwrap_or_default();
+ let outcome = sqlx::query_as::<_, (i64, i64)>(
+ "SELECT
+ count(*) FILTER (WHERE status = 'failed'),
+ count(*) FILTER (WHERE status IN ('done', 'uploaded'))
+ FROM remember_jobs
+ WHERE status IN ('failed', 'done', 'uploaded')
+ AND updated_at >= $1",
+ )
+ .bind(since)
+ .fetch_one(&self.pool)
+ .await;
+
+ match outcome {
+ Ok(counts) => {
+ crate::observability::observe_db(
+ "remember_jobs.recent_outcomes",
+ "ok",
+ started.elapsed(),
+ );
+ Ok(counts)
+ }
+ Err(e) => {
+ crate::observability::observe_db(
+ "remember_jobs.recent_outcomes",
+ "error",
+ started.elapsed(),
+ );
+ Err(AppError::Internal(format!(
+ "Failed to count recent remember outcomes: {}",
+ e
+ )))
+ }
+ }
+ }
+
pub async fn recent_failed_remember_jobs(
&self,
owner: &str,
diff --git a/services/server/src/types.rs b/services/server/src/types.rs
index 06dc9da80..eba0ec0d7 100644
--- a/services/server/src/types.rs
+++ b/services/server/src/types.rs
@@ -1033,9 +1033,11 @@ fn env_bool(name: &str) -> bool {
}
/// `/health` `writes` wire value: `"paused"` when `WRITES_PAUSED` is set.
-pub(crate) fn writes_health_status(paused: bool) -> String {
+pub(crate) fn writes_health_status(paused: bool, degraded: bool) -> String {
if paused {
"paused".to_string()
+ } else if degraded {
+ "degraded".to_string()
} else {
"ok".to_string()
}
@@ -2089,9 +2091,18 @@ pub struct HealthResponse {
/// fail open so CI `wait-for-relayer` does not hang. `status` stays
/// `"ok"` while the relayer process is up.
pub write_ready: bool,
- /// Write-path admission: `"ok"` or `"paused"`. `"paused"` when
- /// `WRITES_PAUSED` is set; write routes then return HTTP 503.
- /// Distinct from `write_ready`. `/health` stays HTTP 200.
+ /// Write-path state: `"ok"`, `"degraded"`, or `"paused"`.
+ ///
+ /// `"paused"` when `WRITES_PAUSED` is set; write routes then return
+ /// HTTP 503. `"degraded"` when recent durable writes have been failing
+ /// and none have landed -- the relayer still accepts and durably
+ /// queues a write, but Walrus is not storing it, so a caller should
+ /// expect the job to fail minutes later rather than queue more.
+ ///
+ /// Deliberately separate from `write_ready`, which stays true through
+ /// a downstream outage: CI's wait-for-relayer gate blocks on
+ /// `write_ready is True`, so folding this into it would make a Walrus
+ /// outage hang every deploy. `/health` stays HTTP 200 throughout.
pub writes: String,
}
@@ -3566,8 +3577,11 @@ mod tests {
#[tokio::test]
async fn writes_paused_maps_to_503_with_stable_message() {
- assert_eq!(writes_health_status(false), "ok");
- assert_eq!(writes_health_status(true), "paused");
+ assert_eq!(writes_health_status(false, false), "ok");
+ assert_eq!(writes_health_status(false, true), "degraded");
+ assert_eq!(writes_health_status(true, false), "paused");
+ // An operator pause is the stronger statement and wins.
+ assert_eq!(writes_health_status(true, true), "paused");
assert!(reject_if_writes_paused(false).is_ok());
let err = reject_if_writes_paused(true).expect_err("paused writes");
assert_eq!(err.kind(), "writes_paused");
From d5669d8b6199a0ef0472e126086edcd35d0c212e Mon Sep 17 00:00:00 2001
From: Harry Phan
Date: Thu, 17 Sep 2026 15:32:19 +0700
Subject: [PATCH 058/132] fix(mcp): stop a rate-limited poll reading as "still
uploading"
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit
memwal_remember_status answered `still uploading` for writes whose status
polls were being REFUSED. On dev 2026-09-17 that produced ~10 minutes of
"0/5 saved, 5 still uploading" while every POST /api/remember/bulk/status
was denied 429 — and because the answer read as progress, the only sane
response was to poll again, which spent more of the budget that was
already exhausted.
The masking is structural, not a wording slip. `waitForRememberJobs` runs
its own poll loop and swallows each poll's error, stamping every row
`timeout` when the budget expires. A batch whose polls were all refused is
therefore byte-identical to one that is genuinely still in flight. The
zero-budget path never had this problem: it is a single read, so the 429
surfaces.
So when the waited path reports that NOTHING moved — every row `timeout` —
confirm it with one direct read before calling it progress. A 429 there is
reported as the rate limit it is, naming the layer actually hit, which
matters because the three layers have different windows and waiting out
the wrong one converges on nothing. Any row that settled proves the polls
were getting through, so no probe is spent. A probe that fails for any
other reason changes nothing: a failed read is not evidence about a job,
and must not be rendered as a failed write.
The 429 message is the one withRelayerRetry already builds, extracted so
both paths say the same thing rather than drifting.
4 tests. tsc clean. MCP suite 133 passed / 14 failed, against a 129/14
baseline on this branch — same 14, which need ports the sandbox denies.
---
.../mcp/__tests__/status-rate-limit.test.ts | 188 ++++++++++++++++++
.../scripts/mcp/tools/remember-status.ts | 64 +++++-
.../server/scripts/mcp/tools/remember-wait.ts | 30 +++
3 files changed, 280 insertions(+), 2 deletions(-)
create mode 100644 services/server/scripts/mcp/__tests__/status-rate-limit.test.ts
diff --git a/services/server/scripts/mcp/__tests__/status-rate-limit.test.ts b/services/server/scripts/mcp/__tests__/status-rate-limit.test.ts
new file mode 100644
index 000000000..ce88d1aa0
--- /dev/null
+++ b/services/server/scripts/mcp/__tests__/status-rate-limit.test.ts
@@ -0,0 +1,188 @@
+import assert from "node:assert/strict";
+import test, { type TestContext } from "node:test";
+import { Client } from "@modelcontextprotocol/sdk/client/index.js";
+import { InMemoryTransport } from "@modelcontextprotocol/sdk/inMemory.js";
+import type { MemWalSession } from "../auth.js";
+
+const { createMcpServer } = await import("../server.js");
+
+/**
+ * A rate-limited poll must not read as "still uploading".
+ *
+ * `waitForRememberJobs` runs its own poll loop and swallows each poll's error,
+ * stamping every row `timeout` when the budget runs out. So a batch whose
+ * polls were all REFUSED with 429 came back looking exactly like a batch that
+ * was genuinely still uploading — and the agent, told it was progressing,
+ * polled again on a budget it had already spent.
+ *
+ * Observed on dev 2026-09-17: ~10 minutes of "0/N saved, N still uploading"
+ * while every `/api/remember/bulk/status` was being denied.
+ */
+
+interface Behaviour {
+ /** What the SDK's internal poll loop reports when its budget expires. */
+ waitStates: Array<"done" | "failed" | "timeout">;
+ /** How the confirming direct read behaves. */
+ probe: "rate-limited" | "still-running" | "done" | "throws";
+}
+
+function rateLimit429(): Error {
+ const e = new Error(
+ 'Walrus Memory server error (429): {"error":"Rate limit exceeded",' +
+ '"layer":"account_sustained","limit":"1000 weighted-requests/hour",' +
+ '"retry_after_seconds":300}',
+ );
+ (e as Error & { status?: number; retryAfterSeconds?: number }).status = 429;
+ (e as Error & { retryAfterSeconds?: number }).retryAfterSeconds = 300;
+ return e;
+}
+
+function sessionWith(b: Behaviour, calls: string[] = []): MemWalSession {
+ const jobIds = b.waitStates.map((_, i) => `job-${i + 1}`);
+ return {
+ oauthScope: "memwal:read memwal:write",
+ namespace: "default",
+ memwal: {
+ async waitForRememberJobs(ids: string[]) {
+ calls.push(`waitJobs:${ids.join(",")}`);
+ return {
+ results: b.waitStates.map((status, i) => ({
+ id: jobIds[i],
+ blob_id: status === "done" ? `blob-${i + 1}` : "",
+ status,
+ namespace: "default",
+ error:
+ status === "timeout"
+ ? "polling timed out after 30000ms"
+ : status === "failed"
+ ? "walrus upload rejected"
+ : undefined,
+ })),
+ total: b.waitStates.length,
+ succeeded: b.waitStates.filter((s) => s === "done").length,
+ failed: b.waitStates.filter((s) => s !== "done").length,
+ };
+ },
+ async getRememberBulkStatus(ids: string[]) {
+ calls.push(`bulkStatus:${ids.join(",")}`);
+ if (b.probe === "rate-limited") throw rateLimit429();
+ if (b.probe === "throws") throw new Error("relayer unreachable");
+ return {
+ results: ids.map((id, i) => ({
+ job_id: id,
+ status: b.probe === "done" ? "done" : "running",
+ blob_id: b.probe === "done" ? `blob-${i + 1}` : undefined,
+ error: undefined,
+ })),
+ };
+ },
+ },
+ } as unknown as MemWalSession;
+}
+
+async function clientFor(session: MemWalSession, t: TestContext): Promise {
+ const [clientTransport, serverTransport] = InMemoryTransport.createLinkedPair();
+ const server = createMcpServer(session);
+ const client = new Client({ name: "status-rate-limit-test", version: "1.0.0" });
+ t.after(async () => {
+ await client.close();
+ await server.close();
+ });
+ await server.connect(serverTransport);
+ await client.connect(clientTransport);
+ return client;
+}
+
+function textOf(result: unknown): string {
+ return (result as { content: Array<{ text: string }> }).content
+ .map((c) => c.text)
+ .join("\n");
+}
+
+test("a batch whose polls were all rate-limited reports the limit, not progress", async (t) => {
+ const calls: string[] = [];
+ const client = await clientFor(
+ sessionWith({ waitStates: ["timeout", "timeout"], probe: "rate-limited" }, calls),
+ t,
+ );
+
+ const result = await client.callTool({
+ name: "memwal_remember_status",
+ arguments: { job_ids: ["job-1", "job-2"], waitMs: 30000 },
+ });
+ const text = textOf(result);
+
+ // The whole point: the words that sent the agent back to poll must be gone.
+ assert.ok(
+ !/still uploading/i.test(text),
+ `a refused poll still read as progress: ${text}`,
+ );
+ assert.match(text, /rate limit/i);
+ // And it must name the layer it actually hit — the hourly one is not the
+ // per-minute one, and waiting out the wrong window converges on nothing.
+ assert.match(text, /account_sustained/);
+ assert.ok(
+ calls.some((c) => c.startsWith("bulkStatus:")),
+ "nothing moving must be confirmed with one direct read",
+ );
+});
+
+test("a batch that is genuinely still uploading is left alone", async (t) => {
+ const calls: string[] = [];
+ const client = await clientFor(
+ sessionWith({ waitStates: ["timeout", "timeout"], probe: "still-running" }, calls),
+ t,
+ );
+
+ const text = textOf(
+ await client.callTool({
+ name: "memwal_remember_status",
+ arguments: { job_ids: ["job-1", "job-2"], waitMs: 30000 },
+ }),
+ );
+
+ assert.match(text, /still uploading/i);
+ assert.ok(!/rate limit/i.test(text), `invented a rate limit: ${text}`);
+});
+
+test("the probe is skipped when any row already settled", async (t) => {
+ const calls: string[] = [];
+ const client = await clientFor(
+ sessionWith({ waitStates: ["done", "timeout"], probe: "rate-limited" }, calls),
+ t,
+ );
+
+ const text = textOf(
+ await client.callTool({
+ name: "memwal_remember_status",
+ arguments: { job_ids: ["job-1", "job-2"], waitMs: 30000 },
+ }),
+ );
+
+ // A settled row proves the polls were getting through, so spending another
+ // request to confirm would be the opposite of the point.
+ assert.ok(
+ !calls.some((c) => c.startsWith("bulkStatus:")),
+ "probed despite evidence the polls were working",
+ );
+ assert.match(text, /1\/2 saved/);
+});
+
+test("a probe that fails for any other reason does not invent an outcome", async (t) => {
+ const client = await clientFor(
+ sessionWith({ waitStates: ["timeout", "timeout"], probe: "throws" }),
+ t,
+ );
+
+ const text = textOf(
+ await client.callTool({
+ name: "memwal_remember_status",
+ arguments: { job_ids: ["job-1", "job-2"], waitMs: 30000 },
+ }),
+ );
+
+ // Falls back to what the wait reported rather than reading a failed probe
+ // as a failed write.
+ assert.match(text, /still uploading/i);
+ assert.ok(!/NOT stored/.test(text), `a failed probe was read as a failed write: ${text}`);
+});
diff --git a/services/server/scripts/mcp/tools/remember-status.ts b/services/server/scripts/mcp/tools/remember-status.ts
index 15c19b65a..9ed1656ee 100644
--- a/services/server/scripts/mcp/tools/remember-status.ts
+++ b/services/server/scripts/mcp/tools/remember-status.ts
@@ -6,7 +6,9 @@ import { wrapTool, walruscanBlobUrl, explorerFooter } from "./util.js";
import {
REMEMBER_POLL_INTERVAL_MS,
isStillRunning,
+ isRateLimited,
nameJobError,
+ rateLimitedError,
withAcceptDeadline,
withWaitDeadline,
} from "./remember-wait.js";
@@ -149,7 +151,29 @@ export function registerRememberStatusTool(
);
return saved(result.blob_id, result.namespace);
} catch (err) {
- if (isStillRunning(err)) return stillRunning(job_id);
+ if (isStillRunning(err)) {
+ // Same masking as the batch path: the SDK's poll loop
+ // hides a refused poll behind its own timeout, so
+ // confirm with one direct read before calling it
+ // progress.
+ try {
+ const status = await session.memwal.getRememberStatus(job_id);
+ if (status.status === "done") {
+ return saved(status.blob_id ?? "", status.namespace);
+ }
+ if (status.status !== "failed" && status.status !== "not_found") {
+ return stillRunning(job_id, status.status);
+ }
+ } catch (probe) {
+ if (isRateLimited(probe)) {
+ throw rateLimitedError(
+ probe,
+ "check whether the write landed"
+ );
+ }
+ }
+ return stillRunning(job_id);
+ }
throw nameJobError(err);
}
}
@@ -173,7 +197,7 @@ async function settleBatch(
// A zero budget is a single batched read, the same shortcut the one-job
// path takes: `waitForRememberJobs` sleeps before its first poll, so a 0ms
// deadline would report everything as still running without ever asking.
- const rows =
+ let rows =
budgetMs === 0
? (
await withAcceptDeadline(
@@ -202,6 +226,42 @@ async function settleBatch(
error: r.error,
}));
+ // `waitForRememberJobs` polls internally and swallows each poll's error,
+ // stamping every row "polling timed out" when the budget runs out. A batch
+ // whose polls were all REFUSED — a 429 on the status endpoint — is
+ // therefore indistinguishable from one that is genuinely still uploading,
+ // and reporting the refusal as progress is what sends an agent back to
+ // poll again on a budget it has already spent.
+ //
+ // Nothing moving at all is the shape that refusal takes, so confirm it
+ // with one direct read. Only then: if any row settled, the polls were
+ // clearly getting through and no probe is warranted.
+ const nothingMoved =
+ budgetMs > 0 && rows.length > 0 && rows.every((r) => r.status === "timeout");
+ if (nothingMoved) {
+ try {
+ rows = (
+ await withAcceptDeadline(
+ session.memwal.getRememberBulkStatus(jobIds),
+ "batch status read",
+ { idempotent: true },
+ )
+ ).results.map((r) => ({
+ id: r.job_id,
+ status: r.status,
+ blob_id: r.blob_id ?? "",
+ error: r.error,
+ }));
+ } catch (err) {
+ if (isRateLimited(err)) {
+ throw rateLimitedError(err, "check whether the writes landed");
+ }
+ // Any other probe failure is not evidence about the jobs; fall
+ // back to what the wait already reported rather than inventing an
+ // outcome from a failed read.
+ }
+ }
+
// `timeout` (the waited path) and pending/running/uploaded (the immediate
// read) are the same thing to a caller: still in flight, ask again.
const inFlight = rows.filter(
diff --git a/services/server/scripts/mcp/tools/remember-wait.ts b/services/server/scripts/mcp/tools/remember-wait.ts
index 221901e9c..e9c784465 100644
--- a/services/server/scripts/mcp/tools/remember-wait.ts
+++ b/services/server/scripts/mcp/tools/remember-wait.ts
@@ -517,6 +517,36 @@ function retryAdvice(secs: number, layer: string | undefined): string {
: `Retry after ~${secs}s.`;
}
+/** True if the relayer refused this call with a rate limit. */
+export function isRateLimited(err: unknown): boolean {
+ return (err as { status?: number } | null)?.status === 429;
+}
+
+/**
+ * The rate-limit error an agent can act on, built from what the relayer said.
+ *
+ * Shared with `memwal_remember_status` so a throttled poll cannot be reported
+ * as "still uploading": a denied poll tells us nothing about the job, and
+ * rendering it as progress is what turns a rate limit into an agent that keeps
+ * polling and spends more of the budget it has already exhausted.
+ */
+export function rateLimitedError(err: unknown, what: string): Error {
+ const secs = Math.ceil(advisedCooldownMs(err) / 1000);
+ const { layer, limit } = rateLimitFacts(err);
+ const hit = layer
+ ? `Limit hit: ${layer} ${scopeOfLayer(layer)}` + (limit ? ` (${limit})` : "") + ". "
+ : "";
+ const e = new Error(
+ `Walrus Memory rate limit reached while trying to ${what}. ${hit}` +
+ `${retryAdvice(secs, layer)} To spend less of the budget, save several facts ` +
+ `with one memwal_remember_bulk call instead of repeated memwal_remember calls, ` +
+ `and settle a batch with a single memwal_remember_status(job_ids=[...]).`,
+ );
+ e.name = "MemWalRelayerUnavailable";
+ (e as Error & { status?: number }).status = 429;
+ return e;
+}
+
/** Honour the relayer's `retry_after` instead of surfacing a raw 429.
*
* Once the per-delegate-key budget is spent the write is simply never made,
From 7d370dcba1955e22545ad0391772d03be209ef94 Mon Sep 17 00:00:00 2001
From: Harry Phan
Date: Thu, 17 Sep 2026 16:33:01 +0700
Subject: [PATCH 059/132] fix(sidecar): make upload-slot release idempotent and
observable
Double-calling the acquireWalrusUploadSlots release callback could free a
successor's reservation because AsyncSemaphore.release is not one-shot.
Route finally + error-path cleanup hit that path. Gate the returned
callback so a second call is a no-op, and log release / acquire-failure
with wait/hold durations and limiter counts so wallet vs global
bottlenecks are visible without equating job status to slot occupancy.
Also document WALRUS_DIRECT_UPLOAD, storage-node/relay timeouts, and the
upload concurrency knobs that the upload-slot investigation relied on.
---
docs/reference/environment-variables.md | 6 +++
.../__tests__/sidecar-slot-lifecycle.test.ts | 49 +++++++++++++++++++
.../server/scripts/sidecar/concurrency.ts | 25 +++++++++-
3 files changed, 79 insertions(+), 1 deletion(-)
create mode 100644 services/server/scripts/__tests__/sidecar-slot-lifecycle.test.ts
diff --git a/docs/reference/environment-variables.md b/docs/reference/environment-variables.md
index e72c2a90e..16c9c09c0 100644
--- a/docs/reference/environment-variables.md
+++ b/docs/reference/environment-variables.md
@@ -131,6 +131,12 @@ These are not all enforced at boot, but most real deployments need them.
| `MEMWAL_ACCOUNT_ID` | none | Optional account ID in server config |
| `WALRUS_PACKAGE_ID` | network default | Override the Walrus on-chain package used by the sidecar |
| `WALRUS_UPLOAD_RELAY_URL` | network default | Override the Walrus upload relay used by the sidecar |
+| `WALRUS_DIRECT_UPLOAD` | `false` | When `true`, skip the upload relay and write slivers to storage nodes directly. Prefer the relay when one is available; direct is the fallback path |
+| `WALRUS_RELAY_TIMEOUT_MS` | `120000` | Timeout for one upload-relay request. Only read when the relay path is active (`WALRUS_DIRECT_UPLOAD` unset/false and `WALRUS_UPLOAD_RELAY_URL` set) |
+| `WALRUS_STORAGE_NODE_TIMEOUT_MS` | `30000` | Timeout for one storage-node HTTP request on both the direct and relay paths (relay still reads slivers back from nodes). Per-node, not per write — raising it also lengthens how long one unresponsive node can stall a shard. Allowed range `1000`–`180000`; keep under `WALRUS_UPLOAD_ACQUIRE_TIMEOUT_MS` |
+| `WALRUS_UPLOAD_MAX_CONCURRENCY` | size of `SERVER_SUI_PRIVATE_KEYS` (min 1) | Sidecar-global cap on concurrent Walrus uploads |
+| `WALRUS_UPLOAD_PER_WALLET_CONCURRENCY` | `1` | Per-uploader-wallet cap on concurrent Walrus uploads |
+| `WALRUS_UPLOAD_ACQUIRE_TIMEOUT_MS` | `120000` | How long an upload waits for a free global/wallet slot before failing acquisition |
| `SEAL_SERVER_CONFIGS` | network default | Optional JSON SEAL server config override for independent or committee servers |
| `SEAL_KEY_SERVERS` | network default | Legacy comma-separated independent SEAL key server override. Used only when `SEAL_SERVER_CONFIGS` is unset. Deprecated but supported through relayer API `1.x` |
| `SEAL_THRESHOLD` | `min(2, total configured weight)` | Required configured server weight for SEAL encrypt/decrypt |
diff --git a/services/server/scripts/__tests__/sidecar-slot-lifecycle.test.ts b/services/server/scripts/__tests__/sidecar-slot-lifecycle.test.ts
new file mode 100644
index 000000000..5e56e43f5
--- /dev/null
+++ b/services/server/scripts/__tests__/sidecar-slot-lifecycle.test.ts
@@ -0,0 +1,49 @@
+import test from "node:test";
+import assert from "node:assert/strict";
+
+process.env.WALRUS_UPLOAD_MAX_CONCURRENCY = "1";
+process.env.WALRUS_UPLOAD_PER_WALLET_CONCURRENCY = "1";
+process.env.WALRUS_UPLOAD_ACQUIRE_TIMEOUT_MS = "1000";
+
+const { acquireWalrusUploadSlots, getUploadCounts, walrusUploadLimitSnapshot } =
+ await import("../sidecar/concurrency.js");
+
+test("global acquire timeout releases the wallet reservation and queue count", async () => {
+ const release = await acquireWalrusUploadSlots(0, "held-global");
+ try {
+ await assert.rejects(
+ acquireWalrusUploadSlots(1, "queued-global"),
+ /global upload slot/
+ );
+ assert.deepEqual(getUploadCounts(), { active: 1, queued: 0 });
+ assert.deepEqual(walrusUploadLimitSnapshot(1).wallet, {
+ capacity: 1,
+ available: 1,
+ queued: 0,
+ });
+ } finally {
+ release();
+ }
+ const releaseNext = await acquireWalrusUploadSlots(1, "after-timeout");
+ releaseNext();
+ assert.deepEqual(getUploadCounts(), { active: 0, queued: 0 });
+});
+
+test("releasing an old holder twice cannot free a successor's slot", async () => {
+ const releaseFirst = await acquireWalrusUploadSlots(0, "first");
+ const second = acquireWalrusUploadSlots(0, "second");
+ releaseFirst();
+ const releaseSecond = await second;
+ try {
+ releaseFirst();
+ assert.deepEqual(getUploadCounts(), { active: 1, queued: 0 });
+ assert.deepEqual(walrusUploadLimitSnapshot(0).wallet, {
+ capacity: 1,
+ available: 0,
+ queued: 0,
+ });
+ } finally {
+ releaseSecond();
+ }
+ assert.deepEqual(getUploadCounts(), { active: 0, queued: 0 });
+});
diff --git a/services/server/scripts/sidecar/concurrency.ts b/services/server/scripts/sidecar/concurrency.ts
index a2a2b7289..9355a70b2 100644
--- a/services/server/scripts/sidecar/concurrency.ts
+++ b/services/server/scripts/sidecar/concurrency.ts
@@ -126,7 +126,8 @@ export async function acquireWalrusUploadSlots(
queuedWalrusUploads = Math.max(0, queuedWalrusUploads - 1);
activeWalrusUploads += 1;
- const waitMs = Date.now() - startedAt;
+ const acquiredAt = Date.now();
+ const waitMs = acquiredAt - startedAt;
if (waitMs >= 1_000) {
console.warn(`[walrus/upload] [${traceId}] limiter_acquired ${JSON.stringify({
jobId,
@@ -136,15 +137,37 @@ export async function acquireWalrusUploadSlots(
})}`);
}
+ // The underlying AsyncSemaphore release is not one-shot: a second call
+ // can free capacity a successor still holds. Keep the returned callback
+ // idempotent so route `finally` + error-path cleanup cannot over-release.
+ let released = false;
return () => {
+ if (released) return;
+ released = true;
activeWalrusUploads = Math.max(0, activeWalrusUploads - 1);
releaseGlobal?.();
releaseWallet?.();
+ console.log(`[walrus/upload] [${traceId}] limiter_released ${JSON.stringify({
+ jobId,
+ keyIndex,
+ waitMs,
+ heldMs: Date.now() - acquiredAt,
+ counts: getUploadCounts(),
+ limits: walrusUploadLimitSnapshot(keyIndex),
+ })}`);
};
} catch (err) {
queuedWalrusUploads = Math.max(0, queuedWalrusUploads - 1);
releaseGlobal?.();
releaseWallet?.();
+ console.warn(`[walrus/upload] [${traceId}] limiter_acquire_failed ${JSON.stringify({
+ jobId,
+ keyIndex,
+ waitMs: Date.now() - startedAt,
+ error: err instanceof Error ? err.message : String(err),
+ counts: getUploadCounts(),
+ limits: walrusUploadLimitSnapshot(keyIndex),
+ })}`);
throw err;
}
}
From 22ef36d47459182811729d9aaae87cfb0fd717f6 Mon Sep 17 00:00:00 2001
From: Harry Phan
Date: Thu, 17 Sep 2026 16:51:04 +0700
Subject: [PATCH 060/132] fix(relayer): mark exhausted Walrus uploads failed
immediately
A retryable upload error on the final Apalis attempt (5/5) still wrote
status=running, so GET /api/remember/:job_id kept saying the write was in
flight until the 10-minute stale sweeper force-failed it. That is what
produced minutes of "still uploading" after every wallet attempt had
already died on the 2026-09-17 upload-slot investigation.
Pass attempt info into the remember_jobs error updater so an exhausted
transient failure is terminal (and releases the storage reservation)
instead of waiting on the sweeper. Lock-contention Defer and congestion
requeue paths still pass no attempt info, so a loser cannot mark failed
while a winner is working. The durable upload path also fires the
exhausted Slack alert it previously skipped.
---
services/server/src/jobs.rs | 121 +++++++++++++++++++++++++++++++++---
1 file changed, 111 insertions(+), 10 deletions(-)
diff --git a/services/server/src/jobs.rs b/services/server/src/jobs.rs
index 554594a1c..4f2214570 100644
--- a/services/server/src/jobs.rs
+++ b/services/server/src/jobs.rs
@@ -202,6 +202,7 @@ async fn update_remember_job_after_wallet_error(
remember_job_id: Option<&str>,
error: &WalletJobError,
msg: &str,
+ attempt_info: Option,
) {
let Some(jid) = remember_job_id else {
return;
@@ -210,13 +211,17 @@ async fn update_remember_job_after_wallet_error(
// Aborting errors (Permanent or ObjectLockedUntilEpoch) get no further
// retries, so the row is terminal — mark it failed rather than leaving it
// stuck on `running` forever. Retryable errors stay `running` for the next
- // attempt. The error_msg carries the lock detail; the object-lock case
- // also fires its own distinct Slack alert.
- let status = if error.aborts_retries() {
- "failed"
- } else {
- "running"
- };
+ // attempt — UNLESS this was already the final attempt. Exhausted retries
+ // used to stay `running` with an error_msg until the 10-minute stale
+ // sweeper force-failed them, so clients polling GET /api/remember/:job_id
+ // reported "still uploading" for minutes after every wallet attempt had
+ // already died (dev 2026-09-17 upload-slot investigation). Pass
+ // `attempt_info` only from the attempt that actually failed the upload;
+ // lock-contention Defer callers pass None so a loser cannot mark failed
+ // while a winner is still working.
+ let exhausted = attempt_info.is_some_and(|info| info.exhausted_by(error));
+ let terminal = error.aborts_retries() || exhausted;
+ let status = if terminal { "failed" } else { "running" };
// Terminal means no later attempt will ever insert the row, so the bytes
// this job reserved at admission must go back to the owner now rather than
@@ -226,7 +231,7 @@ async fn update_remember_job_after_wallet_error(
//
// Safe to run even when a concurrent attempt already won and released:
// release is a delete by id, so a second call is a no-op.
- if error.aborts_retries() {
+ if terminal {
crate::storage::db::release_storage_reservations_with_pool(pool, &[jid.to_string()]).await;
}
@@ -736,6 +741,7 @@ pub(crate) async fn execute_wallet_job(
remember_job_id.as_deref(),
&err,
&msg,
+ Some(attempt_info),
)
.await;
tracing::error!(
@@ -919,8 +925,14 @@ async fn insert_vector_and_mark_remember_done(
{
let msg = format!("insert_vector failed: {}", e);
let classified = WalletJobError::classify_sidecar_error(&msg);
- update_remember_job_after_wallet_error(state.db.pool(), remember_job_id, &classified, &msg)
- .await;
+ update_remember_job_after_wallet_error(
+ state.db.pool(),
+ remember_job_id,
+ &classified,
+ &msg,
+ None,
+ )
+ .await;
tracing::error!(
"[wallet-job:upload] job_id={} {} classification={} retryable={}",
remember_job_id.unwrap_or("-"),
@@ -1539,11 +1551,14 @@ async fn execute_upload_and_transfer(
attempt_info.max,
err,
);
+ // No attempt_info: a Defer loser must not mark the row failed
+ // on the final attempt while the lock holder is still working.
update_remember_job_after_wallet_error(
state.db.pool(),
Some(jid.as_str()),
&err,
err.message(),
+ None,
)
.await;
tokio::time::sleep(backoff_duration(attempt_info.current as u32)).await;
@@ -1744,6 +1759,7 @@ async fn execute_upload_and_transfer_locked(
remember_job_id.as_deref(),
&classified,
&msg,
+ None,
)
.await;
tracing::error!(
@@ -1790,11 +1806,23 @@ async fn execute_upload_and_transfer_locked(
// reclassifying its display text would incorrectly make it
// retryable and leave the polling row running.
let msg = err.message().to_string();
+ maybe_alert_walrus_upload_exhausted(
+ state,
+ &err,
+ attempt_info,
+ Some(jid.as_str()),
+ &owner,
+ &namespace,
+ wallet_index,
+ &msg,
+ )
+ .await;
update_remember_job_after_wallet_error(
state.db.pool(),
Some(jid.as_str()),
&err,
&msg,
+ Some(attempt_info),
)
.await;
tracing::error!(
@@ -1944,6 +1972,7 @@ async fn execute_upload_and_transfer_locked(
remember_job_id.as_deref(),
&classified,
&msg,
+ None,
)
.await;
@@ -2058,6 +2087,7 @@ async fn execute_upload_and_transfer_locked(
remember_job_id.as_deref(),
&classified,
&msg,
+ Some(attempt_info),
)
.await;
tracing::error!(
@@ -4175,6 +4205,7 @@ different transaction: TransactionDigest(8bjFgRyXRRYwrzQapgEjpHnGhdfNDY7d6xA82Bt
Some(job_id.as_str()),
&WalletJobError::Transient("another attempt of upload job is in progress".into()),
"another attempt of upload job is in progress",
+ None,
)
.await;
@@ -4197,6 +4228,76 @@ different transaction: TransactionDigest(8bjFgRyXRRYwrzQapgEjpHnGhdfNDY7d6xA82Bt
.await;
}
+ #[tokio::test]
+ async fn exhausted_transient_upload_marks_the_row_failed_immediately() {
+ // Dev 2026-09-17: after attempt 5/5 of a Walrus 503 timeout the row
+ // stayed `running` until the 10-minute stale sweeper. Clients polling
+ // the job then reported "still uploading" long after every wallet
+ // attempt was spent. The final attempt must mark failed itself.
+ let pool = test_pool().await;
+ let job_id = format!("remember-job-{}", uuid::Uuid::new_v4());
+ insert_job_with_status(&pool, &job_id, "running", None).await;
+
+ let err = WalletJobError::Transient(
+ "Internal Error: durable Walrus upload failed (503 Service Unavailable)".into(),
+ );
+ update_remember_job_after_wallet_error(
+ &pool,
+ Some(job_id.as_str()),
+ &err,
+ err.message(),
+ Some(WalletJobAttemptInfo {
+ current: MAX_ATTEMPTS as usize,
+ max: MAX_ATTEMPTS as usize,
+ }),
+ )
+ .await;
+
+ let row: (String, Option) =
+ sqlx::query_as("SELECT status, error_msg FROM remember_jobs WHERE id = $1")
+ .bind(&job_id)
+ .fetch_one(&pool)
+ .await
+ .unwrap();
+ assert_eq!(row.0, "failed");
+ assert!(
+ row.1.as_deref().unwrap_or("").contains("503"),
+ "error_msg must keep the upload failure: {:?}",
+ row.1
+ );
+
+ let _ = sqlx::query("DELETE FROM remember_jobs WHERE id = $1")
+ .bind(&job_id)
+ .execute(&pool)
+ .await;
+
+ // An earlier attempt must still leave the row running for the next try.
+ let mid_id = format!("remember-job-{}", uuid::Uuid::new_v4());
+ insert_job_with_status(&pool, &mid_id, "running", None).await;
+ update_remember_job_after_wallet_error(
+ &pool,
+ Some(mid_id.as_str()),
+ &err,
+ err.message(),
+ Some(WalletJobAttemptInfo {
+ current: (MAX_ATTEMPTS as usize) - 1,
+ max: MAX_ATTEMPTS as usize,
+ }),
+ )
+ .await;
+ let mid: (String,) = sqlx::query_as("SELECT status FROM remember_jobs WHERE id = $1")
+ .bind(&mid_id)
+ .fetch_one(&pool)
+ .await
+ .unwrap();
+ assert_eq!(mid.0, "running");
+
+ let _ = sqlx::query("DELETE FROM remember_jobs WHERE id = $1")
+ .bind(&mid_id)
+ .execute(&pool)
+ .await;
+ }
+
// RC-4: an uploaded-but-pending resume must route to the TRANSFER recovery op
// (carrying the stored object id), NOT to an index/done finalize — otherwise a
// never-transferred blob is prematurely marked done.
From b46013509ac544e7b0e54c28a1d73f79d6d43e17 Mon Sep 17 00:00:00 2001
From: Harry Phan
Date: Thu, 17 Sep 2026 18:19:07 +0700
Subject: [PATCH 061/132] fix(mcp): pin plugin launchers to published
0.0.14-dev.0
plugin.json / package.json already say 0.0.14, and .mcp.json pinned
npx to @mysten-incubation/memwal-mcp@0.0.14, but that version is not on
npm (only latest=0.0.13 and dist-tag dev=0.0.14-dev.0). Fresh installs
therefore fail to resolve the MCP server entirely.
Point the Claude/Cursor/Codex launcher pins at the published
0.0.14-dev.0 build, and make the Codex hooks installer read the pin
from .mcp.json so it cannot drift back to the unpublished plugin.json
version.
---
packages/mcp/plugin/.codex-mcp.json | 2 +-
packages/mcp/plugin/.cursor-mcp.json | 2 +-
packages/mcp/plugin/.mcp.json | 2 +-
.../mcp/plugin/scripts/install_codex_hooks.mjs | 16 ++++++++++++++++
4 files changed, 19 insertions(+), 3 deletions(-)
diff --git a/packages/mcp/plugin/.codex-mcp.json b/packages/mcp/plugin/.codex-mcp.json
index b833c2d77..daec5c8fd 100644
--- a/packages/mcp/plugin/.codex-mcp.json
+++ b/packages/mcp/plugin/.codex-mcp.json
@@ -2,7 +2,7 @@
"mcpServers": {
"memwal": {
"command": "npx",
- "args": ["-y", "@mysten-incubation/memwal-mcp@0.0.14"]
+ "args": ["-y", "@mysten-incubation/memwal-mcp@0.0.14-dev.0"]
}
}
}
diff --git a/packages/mcp/plugin/.cursor-mcp.json b/packages/mcp/plugin/.cursor-mcp.json
index b833c2d77..daec5c8fd 100644
--- a/packages/mcp/plugin/.cursor-mcp.json
+++ b/packages/mcp/plugin/.cursor-mcp.json
@@ -2,7 +2,7 @@
"mcpServers": {
"memwal": {
"command": "npx",
- "args": ["-y", "@mysten-incubation/memwal-mcp@0.0.14"]
+ "args": ["-y", "@mysten-incubation/memwal-mcp@0.0.14-dev.0"]
}
}
}
diff --git a/packages/mcp/plugin/.mcp.json b/packages/mcp/plugin/.mcp.json
index b833c2d77..daec5c8fd 100644
--- a/packages/mcp/plugin/.mcp.json
+++ b/packages/mcp/plugin/.mcp.json
@@ -2,7 +2,7 @@
"mcpServers": {
"memwal": {
"command": "npx",
- "args": ["-y", "@mysten-incubation/memwal-mcp@0.0.14"]
+ "args": ["-y", "@mysten-incubation/memwal-mcp@0.0.14-dev.0"]
}
}
}
diff --git a/packages/mcp/plugin/scripts/install_codex_hooks.mjs b/packages/mcp/plugin/scripts/install_codex_hooks.mjs
index f4d2a71fe..61ebaaeb9 100644
--- a/packages/mcp/plugin/scripts/install_codex_hooks.mjs
+++ b/packages/mcp/plugin/scripts/install_codex_hooks.mjs
@@ -33,6 +33,22 @@ const SCRIPT_DIR = dirname(fileURLToPath(import.meta.url));
const PLUGIN_ROOT = dirname(SCRIPT_DIR);
function resolveMcpVersion() {
+ // Prefer the pin in .mcp.json so the Codex installer cannot launch an
+ // unpublished package version when plugin.json has already been bumped
+ // for the next release (2026-09-17: plugin.json=0.0.14 but npm had no
+ // 0.0.14 — only 0.0.13 / 0.0.14-dev.0).
+ try {
+ const mcp = JSON.parse(readFileSync(join(PLUGIN_ROOT, ".mcp.json"), "utf8"));
+ const args = mcp?.mcpServers?.memwal?.args;
+ if (Array.isArray(args)) {
+ const pinned = args.find(
+ (a) => typeof a === "string" && a.startsWith("@mysten-incubation/memwal-mcp@"),
+ );
+ if (pinned) return pinned.slice("@mysten-incubation/memwal-mcp@".length);
+ }
+ } catch {
+ // fall through
+ }
return JSON.parse(readFileSync(join(PLUGIN_ROOT, "plugin.json"), "utf8")).version;
}
From d8e2ad0a1a7d2cb2fa016bddff18dff9f112854d Mon Sep 17 00:00:00 2001
From: Harry Phan
Date: Thu, 17 Sep 2026 22:12:08 +0700
Subject: [PATCH 062/132] fix(mcp): create hook state markers exclusively in a
private directory
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit
The prompt hook checked whether a marker file existed and then wrote to
that path. A dangling symlink planted at the marker name passed the
existsSync check and absorbed the write, creating the symlink's target
with the contents `1` — a fixed-content new-file write outside the state
directory.
The state directory is now created at 0700 and verified with lstat on
every use: a real directory, owned by the current user, not group- or
world-accessible (a loose mode is tightened, anything else is refused).
Markers are created with O_CREAT|O_EXCL|O_NOFOLLOW, so an occupied path —
a symlink included — is refused instead of followed, and "the exclusive
create succeeded" is itself the first-time answer, with no separate
existence check to race. bumpCounter lstats before reading and opens with
O_NOFOLLOW for both the read and the write. When no trustworthy directory
can be established the helpers degrade to in-process state, so a hook
still answers, still exits 0, and never blocks the session.
WALM-644
---
packages/mcp/plugin/scripts/lib/hook-io.mjs | 169 +++++++++++++++--
packages/mcp/test/hook-state-symlink.test.mjs | 171 ++++++++++++++++++
2 files changed, 323 insertions(+), 17 deletions(-)
create mode 100644 packages/mcp/test/hook-state-symlink.test.mjs
diff --git a/packages/mcp/plugin/scripts/lib/hook-io.mjs b/packages/mcp/plugin/scripts/lib/hook-io.mjs
index 14fb6aa68..aa4b31377 100644
--- a/packages/mcp/plugin/scripts/lib/hook-io.mjs
+++ b/packages/mcp/plugin/scripts/lib/hook-io.mjs
@@ -5,7 +5,19 @@
* from stdin, optionally emits a `hookSpecificOutput` directive on stdout,
* and always exits 0 — a hook must never block the session.
*/
-import { readFileSync, existsSync, writeFileSync, mkdirSync } from "node:fs";
+import {
+ readFileSync,
+ mkdirSync,
+ chmodSync,
+ lstatSync,
+ fstatSync,
+ openSync,
+ readSync,
+ writeSync,
+ ftruncateSync,
+ closeSync,
+ constants,
+} from "node:fs";
import { tmpdir } from "node:os";
import { join } from "node:path";
@@ -29,13 +41,44 @@ export function emitContext(hookEventName, additionalContext) {
);
}
-const STATE_DIR = join(process.env.TMPDIR || tmpdir(), "memwal-hooks");
+// Hook state is throttle bookkeeping, not data — but it lives in a temp dir
+// that may be shared, so it is handled as hostile ground (WALM-644): the
+// directory is private and verified before use, and every marker is created
+// exclusively and without following symlinks. Checking a path and then writing
+// to it is exactly the pattern a planted symlink turns into a write elsewhere.
+const DIR_MODE = 0o700;
+const FILE_MODE = 0o600;
+// Undefined on Windows, where the flag does not exist; 0 leaves the mask alone.
+const O_NOFOLLOW = constants.O_NOFOLLOW ?? 0;
-function ensureDir() {
+// Used when no trustworthy directory exists or the filesystem refuses us.
+// Hooks are one-shot processes, so this only spans the current invocation —
+// the point is to degrade quietly rather than throw or write somewhere unsafe.
+const memoryState = new Map();
+
+/**
+ * Absolute path of the private hook-state directory, or null when no safe
+ * directory could be established. Resolved per call rather than frozen at
+ * import, so a changed TMPDIR is honoured.
+ */
+export function stateDir() {
+ const base = join(process.env.TMPDIR || tmpdir(), "memwal-hooks");
try {
- mkdirSync(STATE_DIR, { recursive: true });
+ mkdirSync(base, { recursive: true, mode: DIR_MODE });
} catch {
- /* best effort */
+ // Already there, most likely; the checks below decide if it is usable.
+ }
+ try {
+ // lstat, not stat: a symlink parked here must be rejected, not walked.
+ const st = lstatSync(base);
+ if (!st.isDirectory()) return null;
+ if (typeof process.getuid === "function") {
+ if (st.uid !== process.getuid()) return null; // someone else's dir
+ if (st.mode & 0o077) chmodSync(base, DIR_MODE); // shared temp dir
+ }
+ return base;
+ } catch {
+ return null;
}
}
@@ -50,32 +93,124 @@ function safe(s) {
* false thereafter — used to inject a rubric or banner only once per session.
*/
export function firstTime(name, sessionId) {
- ensureDir();
- const f = join(STATE_DIR, `${safe(name)}_${safe(sessionId)}`);
- if (existsSync(f)) return false;
+ const key = `${safe(name)}_${safe(sessionId)}`;
+ const dir = stateDir();
+ if (!dir) return memoryFirstTime(key);
+
+ let fd;
try {
- writeFileSync(f, "1");
+ // O_CREAT|O_EXCL fails with EEXIST when anything already occupies the
+ // path — a dangling symlink included — so the marker is either a fresh
+ // file inside the private dir or nothing at all. Success is itself the
+ // "first time" answer; there is no separate existence check to race.
+ fd = openSync(
+ join(dir, key),
+ constants.O_CREAT | constants.O_EXCL | constants.O_WRONLY | O_NOFOLLOW,
+ FILE_MODE
+ );
+ } catch (err) {
+ // Path taken => seen before. Anything else, fall back to memory.
+ if (err?.code === "EEXIST" || err?.code === "ELOOP") return false;
+ return memoryFirstTime(key);
+ }
+
+ try {
+ writeSync(fd, "1");
} catch {
- /* best effort */
+ /* best effort: the marker existing is what matters, not its contents */
+ } finally {
+ closeQuietly(fd);
}
return true;
}
/** Increment and return a per-(name, session) counter. */
export function bumpCounter(name, sessionId) {
- ensureDir();
- const f = join(STATE_DIR, `count_${safe(name)}_${safe(sessionId)}`);
- let n = 0;
+ const key = `count_${safe(name)}_${safe(sessionId)}`;
+ const dir = stateDir();
+ if (!dir) return memoryBump(key);
+ const f = join(dir, key);
+
+ // Refuse anything that is not a plain file: a symlink here would redirect
+ // both the read and the write.
try {
- n = parseInt(readFileSync(f, "utf8"), 10) || 0;
+ if (!lstatSync(f).isFile()) return memoryBump(key);
+ } catch (err) {
+ if (err?.code !== "ENOENT") return memoryBump(key);
+ }
+
+ let fd;
+ let created = false;
+ try {
+ fd = openSync(
+ f,
+ constants.O_CREAT | constants.O_EXCL | constants.O_RDWR | O_NOFOLLOW,
+ FILE_MODE
+ );
+ created = true;
+ } catch (err) {
+ if (err?.code !== "EEXIST") return memoryBump(key);
+ try {
+ // No O_CREAT: with O_NOFOLLOW this fails on a symlink instead of
+ // opening whatever it points at.
+ fd = openSync(f, constants.O_RDWR | O_NOFOLLOW);
+ } catch {
+ return memoryBump(key);
+ }
+ }
+
+ try {
+ let n = 0;
+ if (!created) {
+ // The open above already refused symlinks; confirm on the fd that
+ // we are not talking to a device or a fifo that would block.
+ if (!fstatSync(fd).isFile()) return memoryBump(key);
+ n = parseInt(readAll(fd), 10) || 0;
+ }
+ n += 1;
+ try {
+ const buf = Buffer.from(String(n));
+ ftruncateSync(fd, 0);
+ writeSync(fd, buf, 0, buf.length, 0);
+ } catch {
+ /* best effort: the caller still gets a monotonic-enough count */
+ }
+ return n;
} catch {
- /* missing -> 0 */
+ return memoryBump(key);
+ } finally {
+ closeQuietly(fd);
+ }
+}
+
+function readAll(fd) {
+ const chunks = [];
+ const buf = Buffer.alloc(64);
+ let bytes;
+ while ((bytes = readSync(fd, buf, 0, buf.length, null)) > 0) {
+ chunks.push(Buffer.from(buf.subarray(0, bytes)));
}
- n += 1;
+ return Buffer.concat(chunks).toString("utf8");
+}
+
+function closeQuietly(fd) {
try {
- writeFileSync(f, String(n));
+ closeSync(fd);
} catch {
/* best effort */
}
+}
+
+function memoryFirstTime(key) {
+ const k = `once:${key}`;
+ if (memoryState.has(k)) return false;
+ memoryState.set(k, 1);
+ return true;
+}
+
+function memoryBump(key) {
+ const k = `count:${key}`;
+ const n = (memoryState.get(k) || 0) + 1;
+ memoryState.set(k, n);
return n;
}
diff --git a/packages/mcp/test/hook-state-symlink.test.mjs b/packages/mcp/test/hook-state-symlink.test.mjs
new file mode 100644
index 000000000..123d715ad
--- /dev/null
+++ b/packages/mcp/test/hook-state-symlink.test.mjs
@@ -0,0 +1,171 @@
+/**
+ * Hook state must not be redirectable through a symlink (WALM-644).
+ *
+ * The old `existsSync(marker) ? skip : writeFileSync(marker, "1")` pair let a
+ * dangling symlink planted at a marker path pass the existence check and then
+ * absorb the write, creating a file outside the state directory. Markers are
+ * now created exclusively (O_CREAT|O_EXCL, plus O_NOFOLLOW where it exists),
+ * so an occupied path is refused rather than followed — and the directory
+ * itself is private and verified before use.
+ */
+import { test } from "node:test";
+import assert from "node:assert/strict";
+import { spawnSync } from "node:child_process";
+import {
+ mkdtempSync,
+ mkdirSync,
+ rmSync,
+ symlinkSync,
+ lstatSync,
+ existsSync,
+ readdirSync,
+ readFileSync,
+} from "node:fs";
+import { tmpdir } from "node:os";
+import { dirname, join, resolve } from "node:path";
+import { fileURLToPath } from "node:url";
+
+const __dirname = dirname(fileURLToPath(import.meta.url));
+const HOOK = resolve(__dirname, "../plugin/scripts/on_user_prompt.mjs");
+const HOOK_IO = "../plugin/scripts/lib/hook-io.mjs";
+
+// Windows needs a privilege or developer mode to create symlinks at all, and
+// has no O_NOFOLLOW; the attack this guards is a POSIX temp-dir one.
+const skipOnWindows =
+ process.platform === "win32" ? "symlink creation needs privileges on Windows" : false;
+
+/** A throwaway TMPDIR plus an "elsewhere" directory a symlink could point into. */
+function sandbox(t) {
+ const root = mkdtempSync(join(tmpdir(), "memwal-hookstate-"));
+ const temp = join(root, "tmp");
+ const elsewhere = join(root, "elsewhere");
+ mkdirSync(temp);
+ mkdirSync(elsewhere);
+ const previous = process.env.TMPDIR;
+ process.env.TMPDIR = temp;
+ t.after(() => {
+ if (previous === undefined) delete process.env.TMPDIR;
+ else process.env.TMPDIR = previous;
+ rmSync(root, { recursive: true, force: true });
+ });
+ return {
+ root,
+ temp,
+ elsewhere,
+ stateDir: join(temp, "memwal-hooks"),
+ target: join(elsewhere, "pwned"),
+ };
+}
+
+/** Every file under `dir`, relative to it — used to prove nothing escaped. */
+function treeOf(dir) {
+ return readdirSync(dir, { recursive: true, withFileTypes: true })
+ .filter((entry) => !entry.isDirectory())
+ .map((entry) => entry.name)
+ .sort();
+}
+
+test("a dangling symlink at a marker path cannot create the target", { skip: skipOnWindows }, async (t) => {
+ const box = sandbox(t);
+ const { firstTime, stateDir } = await import(HOOK_IO);
+
+ assert.equal(stateDir(), box.stateDir);
+ const marker = join(box.stateDir, "rubric_walm644");
+ symlinkSync(box.target, marker);
+ assert.equal(existsSync(box.target), false, "precondition: the target is dangling");
+
+ // The reproduction: the marker "does not exist" by existsSync, so the old
+ // code wrote through it. An occupied path now reports not-first-time.
+ assert.equal(firstTime("rubric", "walm644"), false);
+
+ assert.equal(existsSync(box.target), false, "symlink target must stay uncreated");
+ assert.ok(lstatSync(marker).isSymbolicLink(), "the planted symlink is left alone");
+ assert.deepEqual(treeOf(box.elsewhere), [], "nothing was written outside the state dir");
+});
+
+test("bumpCounter refuses a symlinked counter path", { skip: skipOnWindows }, async (t) => {
+ const box = sandbox(t);
+ const { bumpCounter, stateDir } = await import(HOOK_IO);
+
+ assert.equal(stateDir(), box.stateDir);
+ const counter = join(box.stateDir, "count_nudge_walm644");
+ symlinkSync(box.target, counter);
+
+ // Still answers, still never throws — it just keeps the count in memory.
+ assert.equal(bumpCounter("nudge", "walm644"), 1);
+ assert.equal(bumpCounter("nudge", "walm644"), 2);
+
+ assert.equal(existsSync(box.target), false, "symlink target must stay uncreated");
+ assert.ok(lstatSync(counter).isSymbolicLink());
+ assert.deepEqual(treeOf(box.elsewhere), []);
+});
+
+test("normal session markers still work", { skip: skipOnWindows }, async (t) => {
+ const box = sandbox(t);
+ const { firstTime, bumpCounter, stateDir } = await import(HOOK_IO);
+
+ assert.equal(firstTime("rubric", "session-a"), true);
+ assert.equal(firstTime("rubric", "session-a"), false);
+ assert.equal(firstTime("rubric", "session-a"), false);
+ // A different session is unaffected by the first one's marker.
+ assert.equal(firstTime("rubric", "session-b"), true);
+
+ const marker = join(stateDir(), "rubric_session-a");
+ assert.ok(lstatSync(marker).isFile(), "the marker is a plain file, not a link");
+ assert.equal(readFileSync(marker, "utf8"), "1");
+
+ assert.equal(bumpCounter("turns", "session-a"), 1);
+ assert.equal(bumpCounter("turns", "session-a"), 2);
+ assert.equal(bumpCounter("turns", "session-a"), 3);
+ assert.equal(readFileSync(join(stateDir(), "count_turns_session-a"), "utf8"), "3");
+
+ assert.deepEqual(treeOf(box.elsewhere), []);
+});
+
+test("the state directory is private, and a symlinked one is refused", { skip: skipOnWindows }, async (t) => {
+ const box = sandbox(t);
+ const { firstTime, stateDir } = await import(HOOK_IO);
+
+ const dir = stateDir();
+ assert.equal(dir, box.stateDir);
+ const st = lstatSync(dir);
+ assert.ok(st.isDirectory());
+ assert.equal(st.mode & 0o077, 0, "state dir must not be group/world accessible");
+
+ // Now stand a symlink where the state directory would be: the helper must
+ // refuse it outright instead of writing through it.
+ const hijacked = mkdtempSync(join(tmpdir(), "memwal-hookstate-hijack-"));
+ const decoy = join(hijacked, "tmp");
+ mkdirSync(decoy);
+ symlinkSync(box.elsewhere, join(decoy, "memwal-hooks"));
+ process.env.TMPDIR = decoy;
+ t.after(() => rmSync(hijacked, { recursive: true, force: true }));
+
+ assert.equal(stateDir(), null, "a symlinked state dir is not usable");
+ // Degrades to in-process state: still answers, still writes nothing.
+ assert.equal(firstTime("rubric", "hijacked"), true);
+ assert.equal(firstTime("rubric", "hijacked"), false);
+ assert.deepEqual(treeOf(box.elsewhere), []);
+});
+
+test("the real prompt hook does not write through a planted symlink", { skip: skipOnWindows }, (t) => {
+ const box = sandbox(t);
+ mkdirSync(box.stateDir, { recursive: true, mode: 0o700 });
+ const sessionId = "walm644-e2e";
+ symlinkSync(box.target, join(box.stateDir, `rubric_${sessionId}`));
+
+ const result = spawnSync(process.execPath, [HOOK], {
+ input: JSON.stringify({
+ prompt: "Remember that I always use pnpm and my canary is cedar-wren-11.",
+ session_id: sessionId,
+ }),
+ encoding: "utf8",
+ env: { ...process.env, TMPDIR: box.temp },
+ });
+
+ assert.equal(result.status, 0, result.stderr);
+ assert.ok(result.stdout.trim(), "the hook still emits its directive");
+ JSON.parse(result.stdout); // well-formed, so the session is never blocked
+ assert.equal(existsSync(box.target), false, "symlink target must stay uncreated");
+ assert.deepEqual(treeOf(box.elsewhere), []);
+});
From 0d2b6da4453a688e21d1ea1a98c427eb60934e33 Mon Sep 17 00:00:00 2001
From: Harry Phan
Date: Thu, 17 Sep 2026 22:13:50 +0700
Subject: [PATCH 063/132] fix(mcp): stop the codex installer from interpolating
shell syntax from the install path
The Codex hooks installer rewrote ${PLUGIN_ROOT} in the template *text*
before JSON.parse, so the plugin's own directory landed unescaped in both
the JSON document and the hook command Codex later runs through a shell.
The template's double quotes do not stop command substitution, so an
install path containing $(...) or backticks executed on every hook run,
and a path containing a double quote or a backslash rewrote the JSON.
Parse the template first, then substitute into the parsed values, and
POSIX-single-quote whatever lands in a `command` string. An argv array
under `command` is substituted literally, since those elements reach
execve rather than a shell.
The substitution now lives in scripts/lib/hook-template.mjs so the
quoting primitive can be tested directly against paths Node's ESM loader
refuses to host at all (any specifier containing a backslash).
Nothing else in the installer interpolates a path into a shell string:
ensureMcpRegistered writes a version-pinned npm spec into TOML, not a
path into a command.
Refs WALM-641
---
.../plugin/scripts/install_codex_hooks.mjs | 17 +-
.../mcp/plugin/scripts/lib/hook-template.mjs | 66 ++++++
.../codex-installer-shell-safety.test.mjs | 215 ++++++++++++++++++
3 files changed, 295 insertions(+), 3 deletions(-)
create mode 100644 packages/mcp/plugin/scripts/lib/hook-template.mjs
create mode 100644 packages/mcp/test/codex-installer-shell-safety.test.mjs
diff --git a/packages/mcp/plugin/scripts/install_codex_hooks.mjs b/packages/mcp/plugin/scripts/install_codex_hooks.mjs
index f4d2a71fe..bd169ac7d 100644
--- a/packages/mcp/plugin/scripts/install_codex_hooks.mjs
+++ b/packages/mcp/plugin/scripts/install_codex_hooks.mjs
@@ -12,6 +12,11 @@
* ${PLUGIN_ROOT} placeholder to this plugin's absolute path, and merges the
* entries into ~/.codex/hooks.json.
*
+ * The template is parsed as JSON *before* the placeholder is substituted, and
+ * the path is POSIX-single-quoted on its way into a hook command, so a plugin
+ * directory containing $(...), backticks, quotes or backslashes cannot break
+ * out of either the JSON document or the generated shell command.
+ *
* Re-running is idempotent: entries this installer owns (identified by our
* hook script filenames) are removed before fresh entries are added.
*
@@ -28,6 +33,7 @@ import { readFileSync, writeFileSync, existsSync, mkdirSync } from "node:fs";
import { homedir } from "node:os";
import { join, dirname } from "node:path";
import { fileURLToPath } from "node:url";
+import { substituteHookPlaceholder } from "./lib/hook-template.mjs";
const SCRIPT_DIR = dirname(fileURLToPath(import.meta.url));
const PLUGIN_ROOT = dirname(SCRIPT_DIR);
@@ -48,12 +54,17 @@ const OWNER_MARKERS = [
"on_post_tool.mjs",
];
+const PLACEHOLDER = "${PLUGIN_ROOT}";
+
function loadTemplate() {
- const raw = readFileSync(TEMPLATE_FILE, "utf8").replaceAll(
- "${PLUGIN_ROOT}",
+ // Parse first, substitute second. Substituting into the raw text would let
+ // a path containing a double quote or a backslash rewrite the JSON
+ // document, and would leave `$(...)` or backticks live in the hook command.
+ return substituteHookPlaceholder(
+ JSON.parse(readFileSync(TEMPLATE_FILE, "utf8")),
+ PLACEHOLDER,
PLUGIN_ROOT
);
- return JSON.parse(raw);
}
function loadExisting() {
diff --git a/packages/mcp/plugin/scripts/lib/hook-template.mjs b/packages/mcp/plugin/scripts/lib/hook-template.mjs
new file mode 100644
index 000000000..8cbaf0257
--- /dev/null
+++ b/packages/mcp/plugin/scripts/lib/hook-template.mjs
@@ -0,0 +1,66 @@
+/**
+ * Shell-safe placeholder substitution for hook templates.
+ *
+ * Hook templates ship a `${...}` placeholder for the plugin's install
+ * directory. An installer must not paste that directory into the template
+ * *text*: a path containing a double quote or a backslash rewrites the JSON
+ * document, and a path containing `$(...)` or backticks becomes live shell
+ * syntax in the command the host later executes (WALM-641).
+ *
+ * The rule here is: parse the JSON first, then substitute into the parsed
+ * values, POSIX-quoting whatever lands in a shell command.
+ */
+
+function escapeRegExp(value) {
+ return value.replace(/[.*+?^${}()|[\]\\]/g, "\\$&");
+}
+
+/**
+ * POSIX single-quote escaping.
+ *
+ * Single quotes suppress every form of shell expansion, so `$(...)`, backticks,
+ * double quotes and backslashes inside the value are passed through literally.
+ * A literal `'` cannot appear inside single quotes, so it is emitted as `'\''`:
+ * close the quote, escape one quote, reopen.
+ */
+export function shellQuote(value) {
+ return `'${String(value).replaceAll("'", "'\\''")}'`;
+}
+
+/**
+ * Substitute `placeholder` with `replacement` throughout an already-parsed hook
+ * template.
+ *
+ * A string that is the direct value of a `command` key is treated as a shell
+ * command: the replacement is POSIX-quoted into it, and the double quotes the
+ * template wrapped around the placeholder are dropped in favour of ours (they
+ * would not have stopped command substitution anyway).
+ *
+ * Every other value -- including an argv array under `command`, whose elements
+ * reach execve rather than a shell -- gets the replacement substituted
+ * literally.
+ */
+export function substituteHookPlaceholder(template, placeholder, replacement) {
+ const pattern = escapeRegExp(placeholder);
+ const inCommand = new RegExp(`"${pattern}([^"]*)"|${pattern}(\\S*)`, "g");
+ const plain = new RegExp(pattern, "g");
+
+ const walk = (value, isShellCommand) => {
+ if (typeof value === "string") {
+ return isShellCommand
+ ? value.replace(inCommand, (_match, quoted, bare) =>
+ shellQuote(replacement + (quoted ?? bare ?? ""))
+ )
+ : value.replace(plain, () => replacement);
+ }
+ if (Array.isArray(value)) return value.map((item) => walk(item, false));
+ if (value && typeof value === "object") {
+ return Object.fromEntries(
+ Object.entries(value).map(([key, item]) => [key, walk(item, key === "command")])
+ );
+ }
+ return value;
+ };
+
+ return walk(template, false);
+}
diff --git a/packages/mcp/test/codex-installer-shell-safety.test.mjs b/packages/mcp/test/codex-installer-shell-safety.test.mjs
new file mode 100644
index 000000000..cc8812ca3
--- /dev/null
+++ b/packages/mcp/test/codex-installer-shell-safety.test.mjs
@@ -0,0 +1,215 @@
+/**
+ * WALM-641: the Codex hooks installer must not let the install path reach a
+ * shell as syntax.
+ *
+ * The installer substitutes its own directory into the hook commands it writes
+ * to ~/.codex/hooks.json. That substitution used to run over the template
+ * *text* before JSON.parse, so a plugin directory containing `$(...)`, a
+ * backtick, quotes or a backslash landed unescaped in both the JSON document
+ * and the generated shell command -- running the hook executed whatever the
+ * path said.
+ *
+ * The end-to-end tests install from a deliberately hostile directory and run
+ * the generated commands with a stub `node` that only prints its argv, so the
+ * path is checked for round-trip fidelity without executing a real hook. The
+ * substitution unit tests then cover paths Node itself cannot host, notably
+ * backslashes (the ESM loader rejects any module specifier containing one).
+ */
+import { test, after } from "node:test";
+import assert from "node:assert/strict";
+import { spawnSync } from "node:child_process";
+import {
+ cpSync,
+ existsSync,
+ mkdirSync,
+ mkdtempSync,
+ readFileSync,
+ realpathSync,
+ rmSync,
+ writeFileSync,
+} from "node:fs";
+import { tmpdir } from "node:os";
+import { dirname, join, resolve } from "node:path";
+import { fileURLToPath } from "node:url";
+import {
+ shellQuote,
+ substituteHookPlaceholder,
+} from "../plugin/scripts/lib/hook-template.mjs";
+
+const __dirname = dirname(fileURLToPath(import.meta.url));
+const PLUGIN_SOURCE = resolve(__dirname, "../plugin");
+
+/**
+ * One path segment carrying every construct a shell acts on that a directory
+ * name may legally contain *and* Node can still load a module from: a command
+ * substitution, a backtick substitution, a command separator, a single quote, a
+ * double quote and spaces. Both substitutions create a canary file, so an
+ * escape leaves evidence even if its output goes nowhere.
+ *
+ * Backslashes are covered by the unit tests below instead: Node's ESM loader
+ * refuses every module specifier containing one, so no plugin can be installed
+ * from such a directory in the first place.
+ */
+const HOSTILE_SEGMENT = [
+ "memwal",
+ "$(touch subst-canary; echo PATH_SUBSTITUTION_EXECUTED)",
+ "`touch backtick-canary; echo BACKTICK_EXECUTED`",
+ "it's",
+ '"quoted"',
+ "end",
+].join(" ");
+
+const HOOK_SCRIPTS = ["on_session_start.mjs", "on_user_prompt.mjs", "on_post_tool.mjs"];
+
+// The stub separates arguments with an ASCII record separator rather than a
+// newline, so a value containing a newline is still read back exactly.
+const RS = "\u001e";
+
+const root = realpathSync(
+ mkdtempSync(join(process.env.TMPDIR || tmpdir(), "codex-shell-safety-"))
+);
+after(() => rmSync(root, { recursive: true, force: true }));
+
+const pluginRoot = join(root, HOSTILE_SEGMENT);
+const home = join(root, "home");
+const canaryDir = join(root, "canaries");
+const fakeBin = join(root, "bin");
+
+cpSync(PLUGIN_SOURCE, pluginRoot, { recursive: true });
+for (const dir of [home, canaryDir, fakeBin]) mkdirSync(dir, { recursive: true });
+
+// A stub `node` that prints its arguments instead of running a hook.
+writeFileSync(
+ join(fakeBin, "node"),
+ '#!/bin/sh\nfor arg in "$@"; do printf "%s\\036" "$arg"; done\n',
+ { mode: 0o755 }
+);
+
+const install = spawnSync(
+ process.execPath,
+ [join(pluginRoot, "scripts", "install_codex_hooks.mjs")],
+ {
+ env: { ...process.env, HOME: home, USERPROFILE: home },
+ cwd: canaryDir,
+ encoding: "utf8",
+ }
+);
+
+/** Run a shell command with the stub node on PATH and read back its argv. */
+function argvFor(command) {
+ const result = spawnSync("/bin/sh", ["-c", command], {
+ env: {
+ ...process.env,
+ HOME: home,
+ USERPROFILE: home,
+ PATH: `${fakeBin}:${process.env.PATH}`,
+ },
+ cwd: canaryDir,
+ encoding: "utf8",
+ });
+ assert.equal(result.status, 0, `${command}\n${result.stderr}`);
+ return result.stdout.split(RS).slice(0, -1);
+}
+
+/** Every `command` in a hooks file, in document order. */
+function hookCommands(file = join(home, ".codex", "hooks.json")) {
+ const config = JSON.parse(readFileSync(file, "utf8"));
+ const commands = [];
+ for (const entries of Object.values(config.hooks || {})) {
+ for (const entry of entries) {
+ for (const hook of entry.hooks || []) commands.push(hook.command);
+ }
+ }
+ return commands;
+}
+
+test("installing from a hostile path succeeds and writes parseable JSON", () => {
+ assert.equal(install.status, 0, `${install.stdout}\n${install.stderr}`);
+ const commands = hookCommands();
+ assert.equal(commands.length, HOOK_SCRIPTS.length);
+ for (const script of HOOK_SCRIPTS) {
+ assert.ok(
+ commands.some((command) => command.includes(script)),
+ `no hook command references ${script}: ${JSON.stringify(commands)}`
+ );
+ }
+});
+
+test("generated hook commands do not execute anything the path spells out", () => {
+ for (const command of hookCommands()) {
+ const argv = argvFor(command);
+ assert.equal(argv.length, 1, `expected one argument, got ${JSON.stringify(argv)}`);
+ // The substitutions arrive as inert text, not as their output.
+ assert.ok(argv[0].includes("$(touch subst-canary;"), argv[0]);
+ assert.ok(argv[0].includes("`touch backtick-canary;"), argv[0]);
+ }
+ assert.ok(!existsSync(join(canaryDir, "subst-canary")), "command substitution ran");
+ assert.ok(!existsSync(join(canaryDir, "backtick-canary")), "backtick substitution ran");
+});
+
+test("the hostile path survives the round trip verbatim as a single argument", () => {
+ const seen = new Set();
+ for (const command of hookCommands()) {
+ const argv = argvFor(command);
+ assert.equal(argv.length, 1, `expected one argument, got ${JSON.stringify(argv)}`);
+ assert.equal(dirname(dirname(argv[0])), pluginRoot);
+ seen.add(argv[0]);
+ }
+ assert.deepEqual(
+ [...seen].sort(),
+ HOOK_SCRIPTS.map((script) => join(pluginRoot, "scripts", script)).sort()
+ );
+});
+
+test("shellQuote survives every shell metacharacter, backslashes included", () => {
+ const values = [
+ "/plain/path",
+ "/with space/dir",
+ "/with/$(echo SUBST)",
+ "/with/`echo TICK`",
+ "/with/it's",
+ '/with/"double"',
+ "/with/back\\slash",
+ "/with/back\\\\slash",
+ "/with/$HOME and ${HOME}",
+ "/with/;rm -rf .;",
+ "/with/new\nline",
+ "/with/'''",
+ "/with/*?[a-z]",
+ 'C:\\Program Files\\mem"wal\\$(x)',
+ ];
+ for (const value of values) {
+ assert.deepEqual(argvFor(`printf '%s\\036' ${shellQuote(value)}`), [value]);
+ }
+});
+
+test("substitution quotes the plugin root into commands and leaves JSON intact", () => {
+ const template = JSON.parse(
+ readFileSync(join(PLUGIN_SOURCE, "hooks", "codex-hooks.json"), "utf8")
+ );
+ const hostileRoot = 'C:\\mem"wal\\$(touch pwned)\\`id`\\it\'s here';
+ const substituted = substituteHookPlaceholder(template, "${PLUGIN_ROOT}", hostileRoot);
+
+ // A path full of JSON escapes survives a write/read cycle unchanged.
+ const file = join(root, "substituted.json");
+ writeFileSync(file, JSON.stringify(substituted, null, 2) + "\n");
+ assert.deepEqual(JSON.parse(readFileSync(file, "utf8")), substituted);
+
+ const commands = hookCommands(file);
+ assert.equal(commands.length, HOOK_SCRIPTS.length);
+ for (const command of commands) {
+ const argv = argvFor(command);
+ assert.equal(argv.length, 1, `expected one argument, got ${JSON.stringify(argv)}`);
+ assert.ok(argv[0].startsWith(`${hostileRoot}/scripts/`), argv[0]);
+ }
+ assert.ok(!existsSync(join(canaryDir, "pwned")), "command substitution ran");
+});
+
+test("an argv array under `command` is substituted literally, not quoted", () => {
+ const substituted = substituteHookPlaceholder(
+ { hooks: { E: [{ hooks: [{ command: ["node", "${PLUGIN_ROOT}/x.mjs"] }] }] } },
+ "${PLUGIN_ROOT}",
+ "/it's here"
+ );
+ assert.deepEqual(substituted.hooks.E[0].hooks[0].command, ["node", "/it's here/x.mjs"]);
+});
From fbc1c5946e755c9c08f5429efd91eecfc8819507 Mon Sep 17 00:00:00 2001
From: Harry Phan
Date: Thu, 17 Sep 2026 22:15:04 +0700
Subject: [PATCH 064/132] fix(mcp): require approval before using repo-local
credentials
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit
A project-local `.memwal/credentials.json` decided the account and the relayer
for every memory written from that directory on presence alone. That file lives
INSIDE the repository, so anyone who could commit to a repo — or get a clone
opened — could silently repoint where a user's memories went, from the project
root or any subfolder, with nothing said. The writes are immutable and there is
no delete path, so the redirect is not recoverable after the fact.
The gate: a project file is now inert until the user approves that exact project
path, account, delegate key and relayer with `memwal-mcp approve-project`, which
needs an interactive terminal. The approval record lives in
`~/.memwal/project-approvals.json` — outside every repository, because a record a
repo can carry is a repo approving itself — and holds no key material. It is
keyed on a fingerprint of accountId + delegateAddress + relayerUrl, so any later
edit that moves the destination needs approving again. `revoke-project` withdraws
it.
Unapproved is a fallback, never a failure: an unapproved, altered or malformed
project file leaves the global credentials in use, so a machine that never had a
project file behaves exactly as before. One stderr line (and a `creds.project_ignored`
log event) names the file that was skipped, the destination it wanted, where memory
is going instead, and the command that approves it — silence there would be the
mirror of the silent redirect. `MEMWAL_CREDS_DIR` still overrides both files
without approval; it can only come from the user's own environment.
The active destination is now visible where the user is: `memwal_health` reports
`account=` beside the existing `relayer=`, and `creds.loaded` logs the resolved
path and which rule chose it.
WALM-639
---
docs/mcp/reference.md | 24 +-
docs/reference/environment-variables.md | 2 +-
packages/mcp/CHANGELOG.md | 6 +
packages/mcp/README.md | 27 ++
packages/mcp/src/auth.ts | 376 +++++++++++++++-
packages/mcp/src/bridge.ts | 37 +-
packages/mcp/src/index.ts | 172 +++++++-
.../mcp/test/credential-resolution.test.mjs | 20 +-
.../test/health-relayer-annotation.test.mjs | 30 ++
.../mcp/test/project-creds-approval.test.mjs | 411 ++++++++++++++++++
packages/mcp/test/unknown-flags.test.mjs | 24 +
11 files changed, 1089 insertions(+), 40 deletions(-)
create mode 100644 packages/mcp/test/project-creds-approval.test.mjs
diff --git a/docs/mcp/reference.md b/docs/mcp/reference.md
index d980edc43..7599d2d01 100644
--- a/docs/mcp/reference.md
+++ b/docs/mcp/reference.md
@@ -130,13 +130,26 @@ Both session tools (`memwal_login`, `memwal_logout`) are intercepted locally by
Credentials resolve from two places, in order:
-1. `.memwal/credentials.json` in the **working directory or a parent of it**
+1. `.memwal/credentials.json` in the **working directory or a parent of it** — used only once you have [approved it](#approving-a-project-file)
2. `~/.memwal/credentials.json` (global, per machine)
-The search starts in the working directory and walks up, the way `.npmrc` and `.git/config` resolve, so a command run from a subfolder still picks up that project's credentials. The first `.memwal/credentials.json` it finds wins.
+The search starts in the working directory and walks up, the way `.npmrc` and `.git/config` resolve, so a command run from a subfolder still picks up that project's credentials. The first `.memwal/credentials.json` it finds wins, provided you approved it.
The walk stops at your project root (the directory holding `.git`), at your home directory, or at the filesystem root, whichever comes first. That bound keeps one project from picking up a credentials file belonging to a parent folder that holds unrelated checkouts. If nothing is found inside it, the global file is used. Whichever file is chosen is the one read, written, and deleted for that run.
+### Approving a project file
+
+A project's `.memwal/credentials.json` lives inside the repository, so anyone who can commit to it — or who can get you to open a clone — could otherwise pick the account and relayer every memory written from that directory goes to. The file is therefore **ignored until you approve it**, once per machine:
+
+```bash
+cd ~/code/my-project
+memwal-mcp approve-project
+```
+
+Until then the global credentials are used and one stderr line names the file that was skipped, the account and relayer it wanted, and this command. An approval covers one exact project path, account, delegate key and relayer: change any of them and it has to be approved again. The record is kept in `~/.memwal/project-approvals.json`, outside the repository, so a repository cannot carry its own approval. `memwal-mcp revoke-project` withdraws it.
+
+`MEMWAL_CREDS_DIR` overrides project resolution entirely and needs no approval — it can only come from your own environment, never from a checkout.
+
### Working on several accounts
Without a project-local file, every project on the machine shares one credential. Signing in from one project silently repoints the others at a different account and delegate key, and memories written in that state land on the wrong account, on immutable storage, with no delete path.
@@ -148,6 +161,7 @@ cd ~/code/my-project
mkdir -p .memwal
memwal-mcp login # writes to the global file the first time
cp ~/.memwal/credentials.json .memwal/credentials.json
+memwal-mcp approve-project
```
From then on, runs started from that directory or anywhere beneath it use the project's credentials, and runs started outside it keep using the global one.
@@ -158,7 +172,9 @@ From then on, runs started from that directory or anywhere beneath it use the pr
### Migration
-Nothing to do. Creating a project-local file is the opt-in, so a machine without one behaves exactly as it did before, and the global file remains the fallback indefinitely.
+Nothing to do for a machine without a project-local file: it behaves exactly as it did before, and the global file remains the fallback indefinitely.
+
+If you already have a project-local file, run `memwal-mcp approve-project` in that project once. Until you do, runs from it fall back to the global credentials and say so on stderr — they do not fail.
### Replacing an account
@@ -195,6 +211,8 @@ The stdio package accepts CLI flags and environment variables. **CLI takes prece
| `--namespace ` (alias `--ns`) | `MEMWAL_NAMESPACE` | Default memory namespace injected into memory tool calls that omit one. See [Default namespace](#default-namespace). |
| `--login` (or `login` subcommand) | Not applicable | Force a re-login even when credentials exist. The existing file is kept until the new sign-in succeeds. |
| `--logout` | Not applicable | Delete the credentials file currently in use and exit. |
+| `approve-project` | Not applicable | Approve the project-local `.memwal/credentials.json` found from the current directory, so it is the file used here. Requires an interactive terminal. See [Approving a project file](#approving-a-project-file). |
+| `revoke-project` | Not applicable | Withdraw that approval. Requires an interactive terminal. |
| `--help`, `-h` | Not applicable | Print usage and exit. |
Set `MEMWAL_MCP_DEBUG=1` to enable verbose stderr logging.
diff --git a/docs/reference/environment-variables.md b/docs/reference/environment-variables.md
index db603f31a..fcc8796ef 100644
--- a/docs/reference/environment-variables.md
+++ b/docs/reference/environment-variables.md
@@ -69,7 +69,7 @@ The stdio MCP package reads these environment variables directly. A CLI flag tak
| `MEMWAL_WEB_URL` | `--web-url ` | dashboard default | Dashboard URL used during login |
| `MEMWAL_CLIENT_LABEL` | `--label ` | `MCP Client` / `Walrus Memory MCP` | Friendly delegate-key label shown in the dashboard |
| `MEMWAL_MCP_DEBUG` | none | `0` | Set to `1` for verbose stderr logging |
-| `MEMWAL_CREDS_DIR` | none | `~/.memwal` | Directory holding `credentials.json`. Overrides both project-local and `~/.memwal` credentials, re-read on every access so a test can redirect it after import. Mainly for tests, which must not write into the real credential directory |
+| `MEMWAL_CREDS_DIR` | none | `~/.memwal` | Directory holding `credentials.json` and the project-approval record. Overrides both project-local and `~/.memwal` credentials, and needs no project approval, re-read on every access so a test can redirect it after import. Mainly for tests, which must not write into the real credential directory |
| `MEMWAL_MCP_TRANSPORT` | none | `sse` | Which relayer transport the stdio bridge dials. `sse` uses the legacy split (`POST /api/mcp/messages` + `GET /api/mcp/sse`). `http` (aliases `streamable`, `streamable-http`) uses the Streamable HTTP endpoint `/api/mcp`, where a call is answered on the same request rather than split across a POST and an SSE stream, so there is no idle watchdog. Reconnect replay is NOT yet transport-aware: the bridge still replays its in-flight map on either transport, and on Streamable a disconnect mid-send can look sent, so a replayed write may duplicate. Opt in knowing that. Unrecognised values fall back to `sse` |
| `MEMWAL_MCP_SSE_IDLE_MS` | none | `30000` | Maximum milliseconds of silence on the SSE stream before the bridge treats the session as dead and reconnects. Values below `500` are ignored and fall back to the default. Mainly for tests |
| `MEMWAL_MCP_CALL_TIMEOUT_MS` | none | `240000` | Maximum milliseconds a single request might wait for its response before the bridge answers with a retryable error. Covers a reply lost while the stream itself stays healthy, which `MEMWAL_MCP_SSE_IDLE_MS` cannot detect. The default is derived in code from the slowest server-side tool deadline plus headroom, so it moves with that tool rather than being pinned here. Values below `1000` are ignored and fall back to the default |
diff --git a/packages/mcp/CHANGELOG.md b/packages/mcp/CHANGELOG.md
index 0c06b4def..981f1bc2b 100644
--- a/packages/mcp/CHANGELOG.md
+++ b/packages/mcp/CHANGELOG.md
@@ -1,5 +1,11 @@
# @mysten-incubation/memwal-mcp
+## Unreleased
+
+### Security
+
+- A project-local `.memwal/credentials.json` no longer decides where memory goes on presence alone. That file lives inside the repository, so anyone who could commit to a repo — or get a clone opened — could silently repoint the account and relayer every memory written from that directory went to, including from a subfolder, with nothing said and no delete path once written. A project file is now inert until the user approves that exact project path, account, delegate key and relayer with `memwal-mcp approve-project`; the approval record is kept in `~/.memwal/project-approvals.json`, outside the repository, so a repository cannot carry its own approval, and any later change to the destination requires approving again. Until approved the global credentials are used — an unapproved, altered or malformed project file is a fallback, never a failure — and one stderr line names the file that was skipped, the destination it wanted, and the command that approves it. `MEMWAL_CREDS_DIR` still overrides both files without approval. `memwal_health` now reports `account=` beside `relayer=`, so the active destination is visible where the user is. (WALM-639)
+
## 0.0.14
### Fixed
diff --git a/packages/mcp/README.md b/packages/mcp/README.md
index 4a21e8584..230f96c48 100644
--- a/packages/mcp/README.md
+++ b/packages/mcp/README.md
@@ -152,6 +152,33 @@ Credentials are stored locally in `~/.memwal/credentials.json`. To remove them:
npx -y @mysten-incubation/memwal-mcp --logout
```
+### Per-project credentials
+
+A project can keep its own `.memwal/credentials.json` so memory written from it
+goes to a separate account. That file lives inside the repository, where anyone
+who can commit to it — or who can get you to open a clone — could otherwise
+choose the account and relayer your memories go to. So it is **ignored until you
+approve it**, once per machine:
+
+```sh
+cd path/to/project
+npx -y @mysten-incubation/memwal-mcp approve-project
+```
+
+Until then the global credentials are used, and a line on stderr names the file
+that was skipped and the destination it wanted. An approval covers one exact
+project path, account, delegate key and relayer: if any of those change, it has
+to be approved again. The record is kept in `~/.memwal/project-approvals.json`,
+outside the repository, so a repository cannot carry its own approval.
+`revoke-project` withdraws it.
+
+`MEMWAL_CREDS_DIR` points both the credentials and the approval record at a
+directory of your choosing and overrides project resolution entirely — it can
+only come from your own environment, never from a checkout, so it needs no
+approval.
+
+`memwal_health` reports the destination in use as `account=… relayer=…`.
+
## License
Apache-2.0
diff --git a/packages/mcp/src/auth.ts b/packages/mcp/src/auth.ts
index 828570d3b..d9ea39898 100644
--- a/packages/mcp/src/auth.ts
+++ b/packages/mcp/src/auth.ts
@@ -10,7 +10,7 @@
* documentation patterns transfer cleanly.
*/
import { homedir } from "node:os";
-import { randomUUID } from "node:crypto";
+import { createHash, randomUUID } from "node:crypto";
import { join, dirname, basename } from "node:path";
import {
mkdirSync,
@@ -19,6 +19,7 @@ import {
renameSync,
unlinkSync,
existsSync,
+ realpathSync,
} from "node:fs";
import { log } from "./logger.js";
@@ -93,36 +94,217 @@ function projectCredsPath(): string | null {
}
}
+/* ------------------------------------------------------------------------- *
+ * Project credentials are opt-IN, per machine (WALM-639).
+ *
+ * Presence alone used to be the opt-in: a `.memwal/credentials.json` anywhere
+ * at or above the working directory simply won. But that file is INSIDE the
+ * repository, so anyone who can commit to a repo — or persuade someone to
+ * clone one — could choose the account and the relayer every memory written
+ * from that directory goes to. Opening a project silently repointed the
+ * destination, and the writes land on immutable storage with no delete path.
+ *
+ * So a project file is now inert until the user approves that exact file,
+ * account, delegate and relayer. The approval record lives beside the GLOBAL
+ * credentials, never in the repository, because a record a repository can
+ * carry is a repository approving itself.
+ * ------------------------------------------------------------------------- */
+
+const APPROVALS_FILE = "project-approvals.json";
+
/**
- * Which credentials file this process should read and write.
+ * Where records that a repository must not be able to write are kept.
*
- * The nearest project-local `.memwal/credentials.json` at or above the working
- * directory wins over the global one, the way `.npmrc` and `.git/config`
- * resolve. Signing in from one project otherwise repoints every other project
- * on the machine at a different account and delegate key, silently — memories
- * then land on the wrong account, on immutable storage, with no delete path
- * (GH #628).
+ * `MEMWAL_CREDS_DIR` is the trusted escape hatch — when it is set it decides
+ * the credentials outright and project resolution never runs — so following it
+ * here keeps a sandboxed run (tests, CI) from reaching into the real
+ * `~/.memwal`, exactly as #705 required for the credentials file itself.
+ */
+function trustedStateDir(): string {
+ return process.env.MEMWAL_CREDS_DIR ?? join(homedir(), ".memwal");
+}
+
+/** The approval store. Outside every repository, on purpose. */
+export function projectApprovalsPath(): string {
+ return join(trustedStateDir(), APPROVALS_FILE);
+}
+
+/**
+ * What approval is granted against: the destination, not the file's bytes.
*
- * Presence-based on purpose: creating the local file is the opt-in, so this is
- * purely additive. Resolved per call rather than at module load, because the
- * working directory is not knowable at import time.
+ * Account, delegate and relayer are the three fields that decide WHERE a
+ * memory ends up and WHO signs for it. Hashing them means a project file may
+ * be re-saved, relabelled or reformatted freely, while any edit that moves the
+ * destination invalidates the approval and has to be approved again. The
+ * delegate private key is deliberately NOT part of it — it must never be read
+ * into a record that gets written back out.
+ */
+export function credentialsFingerprint(creds: {
+ accountId: string;
+ delegateAddress: string;
+ relayerUrl: string;
+}): string {
+ return createHash("sha256")
+ .update(`${creds.accountId}\n${creds.delegateAddress}\n${creds.relayerUrl}`)
+ .digest("hex");
+}
+
+/** One approved project credentials file. Contains no secret. */
+export interface ProjectApproval {
+ /** Canonical path of the approved `.memwal/credentials.json`. */
+ path: string;
+ fingerprint: string;
+ accountId: string;
+ delegateAddress: string;
+ relayerUrl: string;
+ approvedAt: string;
+}
+
+interface ApprovalsFile {
+ version: 1;
+ approvals: ProjectApproval[];
+}
+
+/** Compare paths the way the filesystem does. `process.cwd()` reports a
+ * resolved path and an approval may have been recorded through a symlink (or
+ * on macOS, `/tmp` → `/private/tmp`), so both sides go through this. */
+function canonicalPath(path: string): string {
+ try {
+ return realpathSync(path);
+ } catch {
+ return path;
+ }
+}
+
+function isValidApproval(obj: unknown): obj is ProjectApproval {
+ if (!obj || typeof obj !== "object") return false;
+ const a = obj as Record;
+ return (
+ typeof a.path === "string" &&
+ typeof a.fingerprint === "string" &&
+ typeof a.accountId === "string" &&
+ typeof a.delegateAddress === "string" &&
+ typeof a.relayerUrl === "string"
+ );
+}
+
+/** Approvals on record. A missing, malformed or unreadable store approves
+ * nothing — the safe direction, since the consequence is falling back to the
+ * user's own global account rather than adopting someone else's. */
+function loadApprovals(): ProjectApproval[] {
+ const path = projectApprovalsPath();
+ if (!existsSync(path)) return [];
+ try {
+ const parsed = JSON.parse(readFileSync(path, "utf8")) as ApprovalsFile;
+ if (!parsed || parsed.version !== 1 || !Array.isArray(parsed.approvals)) return [];
+ return parsed.approvals.filter(isValidApproval);
+ } catch {
+ return [];
+ }
+}
+
+/** Through the same writer as the credentials file: it creates the directory
+ * at `0700` and the file at `0600`. The record holds no secret, but it decides
+ * where memories go, so it should not be writable by anything that could not
+ * already write the credentials beside it. */
+function saveApprovals(approvals: ProjectApproval[]): void {
+ writeSecretFile(
+ projectApprovalsPath(),
+ JSON.stringify({ version: 1, approvals } satisfies ApprovalsFile, null, 2),
+ );
+}
+
+/** Why a project credentials file was, or was not, used. */
+export type ProjectCredsDecision =
+ /** Approved for exactly this account + delegate + relayer: in use. */
+ | "approved"
+ /** Never approved on this machine. Ignored. */
+ | "unapproved"
+ /** Approved once, but the destination has since changed. Ignored. */
+ | "changed"
+ /** Present but not valid credentials, so there is nothing to approve. */
+ | "unreadable";
+
+export interface ProjectCredsInfo {
+ /** The project-local file that was found. */
+ path: string;
+ decision: ProjectCredsDecision;
+ /** Destination it points at. Absent when the file could not be read — and
+ * never the delegate private key, which no caller of this ever needs. */
+ accountId?: string;
+ relayerUrl?: string;
+}
+
+/** Which file won, and what happened to any project-local file that did not. */
+export interface CredsResolution {
+ /** The file this process reads and writes. */
+ path: string;
+ source: "override" | "project" | "global";
+ /** The project-local file found by the walk, if any — present whether or
+ * not it was used, so callers can report one they ignored. */
+ project?: ProjectCredsInfo;
+}
+
+/**
+ * Which credentials file this process should read and write, and why.
+ *
+ * `MEMWAL_CREDS_DIR` wins outright: it is the trusted, explicitly-set escape
+ * hatch and cannot come from a checkout. Otherwise the nearest project-local
+ * `.memwal/credentials.json` at or above the working directory is used IF the
+ * user has approved that exact destination on this machine (WALM-639), and the
+ * global file is used in every other case — including an unapproved, altered
+ * or malformed project file. Falling back rather than failing keeps a machine
+ * that has never seen a project file behaving exactly as it always did.
*
- * `MEMWAL_CREDS_DIR` overrides both project and global resolution when set,
- * and is re-read on every call.
+ * Resolved per call rather than at module load, because neither the working
+ * directory nor the approval store is knowable at import time.
*/
-export function credsPath(): string {
+export function resolveCreds(): CredsResolution {
const override = process.env.MEMWAL_CREDS_DIR;
- if (override) return join(override, CREDS_FILE);
- return projectCredsPath() ?? globalCredsPath();
+ if (override) return { path: join(override, CREDS_FILE), source: "override" };
+
+ const global = globalCredsPath();
+ const projectPath = projectCredsPath();
+ if (!projectPath) return { path: global, source: "global" };
+
+ const project = readCredsFile(projectPath);
+ if (!project) {
+ return {
+ path: global,
+ source: "global",
+ project: { path: projectPath, decision: "unreadable" },
+ };
+ }
+
+ const approval = loadApprovals().find((a) => a.path === canonicalPath(projectPath));
+ const decision: ProjectCredsDecision = !approval
+ ? "unapproved"
+ : approval.fingerprint === credentialsFingerprint(project)
+ ? "approved"
+ : "changed";
+ const info: ProjectCredsInfo = {
+ path: projectPath,
+ decision,
+ accountId: project.accountId,
+ relayerUrl: project.relayerUrl,
+ };
+ return decision === "approved"
+ ? { path: projectPath, source: "project", project: info }
+ : { path: global, source: "global", project: info };
}
-/** Load credentials from disk. Returns null if missing or malformed. */
-export function loadCreds(): MemWalCredentials | null {
- const path = credsPath();
+/** The credentials file in use. Thin wrapper over {@link resolveCreds} so the
+ * many callers that only need a path are unchanged. */
+export function credsPath(): string {
+ return resolveCreds().path;
+}
+
+/** Read and validate one credentials file. Returns null if missing or
+ * malformed — the caller decides what that means. */
+function readCredsFile(path: string): MemWalCredentials | null {
if (!existsSync(path)) return null;
try {
- const raw = readFileSync(path, "utf8");
- const parsed = JSON.parse(raw);
+ const parsed = JSON.parse(readFileSync(path, "utf8"));
if (!isValid(parsed)) return null;
return parsed as MemWalCredentials;
} catch {
@@ -130,6 +312,158 @@ export function loadCreds(): MemWalCredentials | null {
}
}
+/** Load credentials from disk. Returns null if missing or malformed. */
+export function loadCreds(): MemWalCredentials | null {
+ return readCredsFile(credsPath());
+}
+
+/** What {@link approveProjectCreds} did. */
+export interface ApproveProjectResult {
+ outcome:
+ /** Newly approved. */
+ | "approved"
+ /** Approved again after the destination changed. */
+ | "reapproved"
+ /** Already approved for this exact destination; nothing written. */
+ | "already-approved"
+ /** No project-local credentials file at or above the working directory. */
+ | "none"
+ /** A project file exists but is not valid credentials. */
+ | "unreadable"
+ /** `MEMWAL_CREDS_DIR` is set, so project resolution never runs. */
+ | "overridden";
+ projectPath?: string;
+ accountId?: string;
+ relayerUrl?: string;
+ /** Destination the previous approval covered, when this replaced one. */
+ previousAccountId?: string;
+ previousRelayerUrl?: string;
+ approvalsPath: string;
+}
+
+/**
+ * Approve the project-local credentials found from the working directory.
+ *
+ * Deliberately takes no arguments: it approves what resolution would otherwise
+ * ignore, from the same directory, so "what am I approving" and "what will be
+ * used" cannot drift apart.
+ */
+export function approveProjectCreds(): ApproveProjectResult {
+ const approvalsPath = projectApprovalsPath();
+ if (process.env.MEMWAL_CREDS_DIR) return { outcome: "overridden", approvalsPath };
+
+ const projectPath = projectCredsPath();
+ if (!projectPath) return { outcome: "none", approvalsPath };
+ const creds = readCredsFile(projectPath);
+ if (!creds) return { outcome: "unreadable", projectPath, approvalsPath };
+
+ const key = canonicalPath(projectPath);
+ const fingerprint = credentialsFingerprint(creds);
+ const approvals = loadApprovals();
+ const existing = approvals.find((a) => a.path === key);
+ if (existing?.fingerprint === fingerprint) {
+ return {
+ outcome: "already-approved",
+ projectPath,
+ accountId: creds.accountId,
+ relayerUrl: creds.relayerUrl,
+ approvalsPath,
+ };
+ }
+
+ saveApprovals([
+ ...approvals.filter((a) => a.path !== key),
+ {
+ path: key,
+ fingerprint,
+ accountId: creds.accountId,
+ delegateAddress: creds.delegateAddress,
+ relayerUrl: creds.relayerUrl,
+ approvedAt: new Date().toISOString(),
+ },
+ ]);
+ return {
+ outcome: existing ? "reapproved" : "approved",
+ projectPath,
+ accountId: creds.accountId,
+ relayerUrl: creds.relayerUrl,
+ previousAccountId: existing?.accountId,
+ previousRelayerUrl: existing?.relayerUrl,
+ approvalsPath,
+ };
+}
+
+/** What {@link revokeProjectCredsApproval} did. */
+export interface RevokeProjectResult {
+ outcome: "revoked" | "none";
+ projectPath?: string;
+ approvalsPath: string;
+}
+
+/**
+ * Withdraw the approval for the project-local credentials here.
+ *
+ * Keyed on the path rather than on the file's current contents, so an approval
+ * can be withdrawn even after the file it covered was edited or deleted — a
+ * revoke that only worked while the destination still matched would be
+ * useless exactly when it is wanted.
+ */
+export function revokeProjectCredsApproval(): RevokeProjectResult {
+ const approvalsPath = projectApprovalsPath();
+ const projectPath = projectCredsPath() ?? join(process.cwd(), ".memwal", CREDS_FILE);
+ const key = canonicalPath(projectPath);
+ const approvals = loadApprovals();
+ const remaining = approvals.filter((a) => a.path !== key);
+ if (remaining.length === approvals.length) return { outcome: "none", projectPath, approvalsPath };
+ saveApprovals(remaining);
+ return { outcome: "revoked", projectPath, approvalsPath };
+}
+
+/**
+ * The warning for a project credentials file that was found and NOT used, or
+ * null when there is nothing to report.
+ *
+ * Says which file was ignored, where memory is going instead, and the exact
+ * command that approves it — a silent fallback would be the mirror image of
+ * the silent redirect this gate exists to stop. Never contains a key: the only
+ * fields it reads are the account id and the relayer URL.
+ */
+export function formatProjectCredsNotice(
+ resolution: CredsResolution = resolveCreds(),
+): string | null {
+ const project = resolution.project;
+ if (!project || project.decision === "approved") return null;
+
+ const destination = `account ${project.accountId} on ${project.relayerUrl}`;
+ const head =
+ project.decision === "unreadable"
+ ? [
+ `Ignored the project credentials at ${project.path}: the file is not a valid`,
+ `Walrus Memory credentials file, so there is nothing to approve.`,
+ ]
+ : project.decision === "changed"
+ ? [
+ `Ignored the project credentials at ${project.path}: they changed since you`,
+ `approved them and now point at ${destination}.`,
+ `Approving again is required whenever the account, delegate key or relayer moves.`,
+ ]
+ : [
+ `Ignored the project credentials at ${project.path}, which would send memory to`,
+ `${destination}.`,
+ `A file inside a repository can be committed by anyone, so it is not used until`,
+ `you approve it on this machine.`,
+ ];
+
+ const lines = [...head, `Memory is going to ${resolution.path} instead.`];
+ if (project.decision !== "unreadable") {
+ lines.push(
+ `To use it, run \`memwal-mcp approve-project\` in a terminal from this directory.`,
+ `The approval is recorded in ${projectApprovalsPath()}, outside the repository.`,
+ );
+ }
+ return lines.join("\n");
+}
+
/**
* Write credentials with secure (`0600`) permission, to whichever file
* `credsPath()` resolves to.
diff --git a/packages/mcp/src/bridge.ts b/packages/mcp/src/bridge.ts
index 91485c115..f3c61c127 100644
--- a/packages/mcp/src/bridge.ts
+++ b/packages/mcp/src/bridge.ts
@@ -75,7 +75,8 @@ const NAMESPACE_TOOLS = new Set([
* per-call namespace always wins over the configured default.
*/
/**
- * Name the relayer this process dialled in a `memwal_health` result.
+ * Name the destination this process is bound to in a `memwal_health` result:
+ * the relayer it dialled, and the account it signs for.
*
* The relayer-side text can only report an origin its deployment published, and
* stays silent on a self-hosted or local one, where the sidecar knows nothing
@@ -87,10 +88,18 @@ const NAMESPACE_TOOLS = new Set([
* Rewrites an existing `relayer=` field rather than appending a second one: when
* both sides know the origin they describe the same session, and two
* conflicting fields would be worse than neither.
+ *
+ * `account=` rides along for the same reason (WALM-639). A project-local
+ * credentials file can point this process at a different account than the one
+ * the user signed in with, and "which account am I writing to" was otherwise
+ * only visible in stderr the MCP host usually hides — so the half of the
+ * destination that decides WHOSE memory this is now shows up beside the half
+ * that decides where it is stored.
*/
export function annotateHealthResult(
result: { content?: unknown; isError?: unknown },
relayerUrl: string,
+ accountId?: string,
): void {
// A failed health call has no session to describe; naming a relayer beside
// an error reads as though that relayer answered.
@@ -104,6 +113,11 @@ export function annotateHealthResult(
block.text = existing.test(block.text)
? block.text.replace(existing, `relayer=${relayerUrl}`)
: `${block.text} relayer=${relayerUrl}`;
+ if (!accountId) return;
+ const existingAccount = /\baccount=\S+/;
+ block.text = existingAccount.test(block.text)
+ ? block.text.replace(existingAccount, `account=${accountId}`)
+ : `${block.text} account=${accountId}`;
}
export function applyDefaultNamespace(msg: RpcMessage, namespace?: string): RpcMessage {
@@ -1348,11 +1362,14 @@ export async function runBridge(
* client surfaces them in its tool palette. */
const pendingListIds = new Set();
- /** IDs of forwarded `memwal_health` calls, each against the relayer URL the
- * call went out on. Captured at send time rather than read at reply time so
- * a reconnect that swapped credentials mid-flight cannot label the answer
- * with a relayer it did not come from. */
- const pendingHealthIds = new Map();
+ /** IDs of forwarded `memwal_health` calls, each against the destination the
+ * call went out on — relayer URL and account. Captured at send time rather
+ * than read at reply time so a reconnect that swapped credentials mid-flight
+ * cannot label the answer with a destination it did not come from. */
+ const pendingHealthIds = new Map<
+ string | number,
+ { relayerUrl: string; accountId?: string }
+ >();
/** Record a 429 and tell the user ONCE that this is a rate limit rather
* than a broken config — the distinction the MCP host cannot make for
@@ -1872,7 +1889,8 @@ export async function runBridge(
if (dialled !== undefined) {
annotateHealthResult(
value.result as { content?: unknown; isError?: unknown },
- dialled,
+ dialled.relayerUrl,
+ dialled.accountId,
);
}
}
@@ -2053,7 +2071,10 @@ export async function runBridge(
msg.id != null &&
(msg.params as { name?: string } | undefined)?.name === "memwal_health"
) {
- pendingHealthIds.set(msg.id, creds?.relayerUrl ?? config.relayerUrl);
+ pendingHealthIds.set(msg.id, {
+ relayerUrl: creds?.relayerUrl ?? config.relayerUrl,
+ accountId: creds?.accountId,
+ });
}
// Track requests (have both method and id) so we can replay
diff --git a/packages/mcp/src/index.ts b/packages/mcp/src/index.ts
index 9c0ded0dd..230ed0c36 100644
--- a/packages/mcp/src/index.ts
+++ b/packages/mcp/src/index.ts
@@ -9,7 +9,16 @@
* 5. On 401 (revoked key), the bridge wipes credentials before throwing
* — the next process spawn will re-trigger login.
*/
-import { clearCreds, clearPendingLogin, credsPath, loadCreds } from "./auth.js";
+import {
+ approveProjectCreds,
+ clearCreds,
+ clearPendingLogin,
+ credsPath,
+ formatProjectCredsNotice,
+ loadCreds,
+ resolveCreds,
+ revokeProjectCredsApproval,
+} from "./auth.js";
import { recoverPendingLogin, formatStrandedLoginNotice } from "./recovery.js";
import { runAuthRequiredServer } from "./auth-required.js";
import { notePendingLoginSuccess, runBridge } from "./bridge.js";
@@ -25,6 +34,11 @@ interface ParsedArgs {
help: boolean;
logout: boolean;
forceLogin: boolean;
+ /** Approve the project-local credentials found from the working directory
+ * (WALM-639). A repo file is inert until this has been run for it. */
+ approveProject: boolean;
+ /** Withdraw that approval again. */
+ revokeProject: boolean;
relayerUrl?: string;
webUrl?: string;
label?: string;
@@ -45,10 +59,17 @@ const ENV_PRESETS: Record = {
/** Bare words that are commands rather than values. An unknown flag must not
* swallow one as its argument. */
-const POSITIONALS = new Set(["login"]);
+const POSITIONALS = new Set(["login", "approve-project", "revoke-project"]);
export function parseArgs(argv: string[]): ParsedArgs {
- const out: ParsedArgs = { help: false, logout: false, forceLogin: false, unknown: [] };
+ const out: ParsedArgs = {
+ help: false,
+ logout: false,
+ forceLogin: false,
+ approveProject: false,
+ revokeProject: false,
+ unknown: [],
+ };
for (let i = 0; i < argv.length; i++) {
const a = argv[i];
const next = () => argv[++i];
@@ -64,6 +85,14 @@ export function parseArgs(argv: string[]): ParsedArgs {
case "login":
out.forceLogin = true;
break;
+ case "--approve-project":
+ case "approve-project":
+ out.approveProject = true;
+ break;
+ case "--revoke-project":
+ case "revoke-project":
+ out.revokeProject = true;
+ break;
case "--prod":
case "--dev":
case "--staging":
@@ -174,6 +203,91 @@ export async function main(argv: string[] = process.argv.slice(2)): Promise env > default.
const relayerUrl =
args.relayerUrl ?? process.env.MEMWAL_SERVER_URL ?? "https://relayer.memory.walrus.xyz";
@@ -206,6 +320,26 @@ export async function main(argv: string[] = process.argv.slice(2)): Promise {
+test("an approved project-local credentials file takes precedence over the global one", async (t) => {
const { auth, cwd } = await sandbox(t, {
global: GLOBAL_ACCOUNT,
project: PROJECT_ACCOUNT,
+ approve: true,
});
assert.equal(
@@ -110,6 +123,7 @@ test("saveCreds writes back to the project-local file when that is the one in us
const { auth, home, cwd } = await sandbox(t, {
global: GLOBAL_ACCOUNT,
project: PROJECT_ACCOUNT,
+ approve: true,
});
const updated = makeCreds(PROJECT_ACCOUNT, "Renamed");
@@ -249,6 +263,7 @@ test("a subdirectory of the project resolves to the project's credentials", asyn
const { auth, cwd } = await sandbox(t, {
global: GLOBAL_ACCOUNT,
project: PROJECT_ACCOUNT,
+ approve: true,
});
chdirBelow(cwd, "src", "nested");
@@ -295,6 +310,7 @@ test("removing a project file reports the global one that takes over", async (t)
const { auth, home, cwd } = await sandbox(t, {
global: GLOBAL_ACCOUNT,
project: PROJECT_ACCOUNT,
+ approve: true,
});
const result = auth.clearCreds();
diff --git a/packages/mcp/test/health-relayer-annotation.test.mjs b/packages/mcp/test/health-relayer-annotation.test.mjs
index 9d8e7fc81..e1e2b32b6 100644
--- a/packages/mcp/test/health-relayer-annotation.test.mjs
+++ b/packages/mcp/test/health-relayer-annotation.test.mjs
@@ -34,6 +34,36 @@ test("replaces the relayer the reply already carried rather than adding a second
assert.ok(text.includes("write_ready=true"));
});
+// A project-local credentials file can point this process at a different
+// account than the one the user signed in with (WALM-639), so the account is
+// part of the destination `memwal_health` reports, not just the relayer.
+
+const ACCOUNT = "0x" + "b".repeat(64);
+
+test("names the account this session signs for", () => {
+ const result = healthResult("Walrus Memory is reachable. status=ok");
+ annotateHealthResult(result, DEV, ACCOUNT);
+ assert.ok(result.content[0].text.includes(`account=${ACCOUNT}`));
+ assert.ok(result.content[0].text.includes(`relayer=${DEV}`));
+});
+
+test("replaces an account the reply already carried rather than adding a second", () => {
+ const result = healthResult("status=ok account=0xstale write_ready=true");
+ annotateHealthResult(result, DEV, ACCOUNT);
+ const text = result.content[0].text;
+ assert.equal(text.match(/account=/g).length, 1, `two account fields:\n${text}`);
+ assert.ok(!text.includes("0xstale"));
+ assert.ok(text.includes("write_ready=true"), "the field after it must survive");
+});
+
+test("says nothing about an account it does not know", () => {
+ // Signed out, or a cold start before credentials are adopted. An empty
+ // `account=` would read as an account rather than as an absence.
+ const result = healthResult("status=ok");
+ annotateHealthResult(result, DEV);
+ assert.ok(!result.content[0].text.includes("account="));
+});
+
test("leaves a failed health call alone", () => {
// Naming a relayer beside an error reads as though that relayer answered.
const result = { ...healthResult("relayer unreachable"), isError: true };
diff --git a/packages/mcp/test/project-creds-approval.test.mjs b/packages/mcp/test/project-creds-approval.test.mjs
new file mode 100644
index 000000000..74e361bf6
--- /dev/null
+++ b/packages/mcp/test/project-creds-approval.test.mjs
@@ -0,0 +1,411 @@
+/**
+ * Repo credentials cannot silently choose the destination (WALM-639).
+ *
+ * `.memwal/credentials.json` decides the account a memory is written under and
+ * the relayer it is written through — and it lives INSIDE the repository, where
+ * anyone who can commit, or who can get a clone opened, can put one. Presence
+ * alone used to be the opt-in, so opening a project repointed every memory
+ * written from it, with nothing said. The writes are immutable and there is no
+ * delete path, which is what made this a P1 rather than a papercut.
+ *
+ * The gate: a project file is inert until the user approves that exact file,
+ * account, delegate and relayer, and the approval lives OUTSIDE the repository
+ * so a repo cannot carry its own approval. Anything unapproved falls back to
+ * the global credentials rather than failing, because a machine that never had
+ * a project file must keep behaving exactly as it did.
+ *
+ * `auth.js` resolves per call, so each test sets HOME and cwd first and then
+ * imports with a cache-busting query — the pattern used by
+ * credential-resolution.test.mjs.
+ */
+import { test } from "node:test";
+import assert from "node:assert/strict";
+import {
+ mkdtempSync,
+ mkdirSync,
+ writeFileSync,
+ readFileSync,
+ rmSync,
+ existsSync,
+ realpathSync,
+} from "node:fs";
+import { tmpdir } from "node:os";
+import { join, dirname } from "node:path";
+
+const GLOBAL_ACCOUNT = "0x" + "a".repeat(64);
+const PROJECT_ACCOUNT = "0x" + "b".repeat(64);
+const ATTACKER_ACCOUNT = "0x" + "9".repeat(64);
+const PRIVATE_KEY = "c".repeat(64);
+const GLOBAL_RELAYER = "https://relayer.example";
+const PROJECT_RELAYER = "https://project-relayer.example";
+
+function makeCreds(overrides = {}) {
+ return {
+ delegatePrivateKey: PRIVATE_KEY,
+ delegatePublicKeyHex: "d".repeat(64),
+ delegateAddress: "0x" + "e".repeat(64),
+ walletAddress: "0x" + "f".repeat(64),
+ accountId: GLOBAL_ACCOUNT,
+ packageId: "0x" + "1".repeat(64),
+ relayerUrl: GLOBAL_RELAYER,
+ createdAt: new Date(0).toISOString(),
+ version: 1,
+ ...overrides,
+ };
+}
+
+function writeCredsAt(root, creds) {
+ const path = join(root, ".memwal", "credentials.json");
+ mkdirSync(dirname(path), { recursive: true });
+ writeFileSync(path, JSON.stringify(creds), { mode: 0o600 });
+ return path;
+}
+
+/**
+ * A HOME, a working directory, and the module re-imported so it sees them.
+ *
+ * `MEMWAL_CREDS_DIR` is cleared rather than inherited: it overrides resolution
+ * outright, so a stray value in the ambient environment would make every
+ * assertion here vacuous.
+ */
+async function sandbox(t, { global: globalCreds, project: projectCreds } = {}) {
+ // Canonicalised for the same reason as credential-resolution.test.mjs:
+ // `process.cwd()` and `homedir()` report resolved paths, and on macOS
+ // `/tmp` is a symlink. Both HOME and USERPROFILE, so it is portable.
+ const home = realpathSync(mkdtempSync(join(tmpdir(), "memwal-approve-home-")));
+ const cwd = realpathSync(mkdtempSync(join(tmpdir(), "memwal-approve-cwd-")));
+ const previous = {
+ home: process.env.HOME,
+ profile: process.env.USERPROFILE,
+ credsDir: process.env.MEMWAL_CREDS_DIR,
+ cwd: process.cwd(),
+ };
+
+ process.env.HOME = home;
+ process.env.USERPROFILE = home;
+ delete process.env.MEMWAL_CREDS_DIR;
+ process.chdir(cwd);
+
+ if (globalCreds) writeCredsAt(home, globalCreds);
+ if (projectCreds) writeCredsAt(cwd, projectCreds);
+
+ t.after(() => {
+ process.chdir(previous.cwd);
+ process.env.HOME = previous.home;
+ process.env.USERPROFILE = previous.profile;
+ if (previous.credsDir === undefined) delete process.env.MEMWAL_CREDS_DIR;
+ else process.env.MEMWAL_CREDS_DIR = previous.credsDir;
+ rmSync(home, { recursive: true, force: true });
+ rmSync(cwd, { recursive: true, force: true });
+ });
+
+ const auth = await import(`../dist/auth.js?walm639=${Date.now()}-${Math.random()}`);
+ return { auth, home, cwd };
+}
+
+/** The two files a sandbox works with. */
+const globalFile = (home) => join(home, ".memwal", "credentials.json");
+const projectFile = (cwd) => join(cwd, ".memwal", "credentials.json");
+
+/* --------------------------------------------------------------------- *
+ * The reproduction itself.
+ * --------------------------------------------------------------------- */
+
+test("a repo credentials file alone does not redirect a write", async (t) => {
+ const { auth, home, cwd } = await sandbox(t, {
+ global: makeCreds(),
+ project: makeCreds({ accountId: PROJECT_ACCOUNT, relayerUrl: PROJECT_RELAYER }),
+ });
+
+ // Reads resolve to the user's own account...
+ assert.equal(auth.credsPath(), globalFile(home));
+ assert.equal(auth.loadCreds()?.accountId, GLOBAL_ACCOUNT);
+ assert.equal(auth.loadCreds()?.relayerUrl, GLOBAL_RELAYER);
+
+ // ...and so do writes. This is the assertion the ticket asks for: the file
+ // the repo carries must not be the file the process signs and saves with.
+ auth.saveCreds(makeCreds({ label: "Re-saved" }));
+ assert.equal(JSON.parse(readFileSync(globalFile(home), "utf8")).label, "Re-saved");
+ const untouched = JSON.parse(readFileSync(projectFile(cwd), "utf8"));
+ assert.equal(untouched.accountId, PROJECT_ACCOUNT, "the repo file must not be written");
+ assert.equal(untouched.label, undefined);
+});
+
+test("running from a subfolder does not redirect a write either", async (t) => {
+ // The reporter checked this: the walk that finds the project file climbs,
+ // so the gate has to hold at every depth, not just at the project root.
+ const { auth, home, cwd } = await sandbox(t, {
+ global: makeCreds(),
+ project: makeCreds({ accountId: PROJECT_ACCOUNT }),
+ });
+ const deep = join(cwd, "src", "nested");
+ mkdirSync(deep, { recursive: true });
+ process.chdir(deep);
+
+ assert.equal(auth.credsPath(), globalFile(home));
+ assert.equal(auth.loadCreds()?.accountId, GLOBAL_ACCOUNT);
+ assert.equal(auth.resolveCreds().project?.decision, "unapproved");
+});
+
+test("the ignored project file is named, along with the destination it wanted", async (t) => {
+ // Falling back in silence would be the mirror image of the silent redirect:
+ // the user created that file expecting it to be used.
+ const { auth, home, cwd } = await sandbox(t, {
+ global: makeCreds(),
+ project: makeCreds({ accountId: PROJECT_ACCOUNT, relayerUrl: PROJECT_RELAYER }),
+ });
+
+ const notice = auth.formatProjectCredsNotice();
+
+ assert.ok(notice, "an ignored project file must be reported");
+ assert.ok(notice.includes(projectFile(cwd)), "must name the file that was ignored");
+ assert.ok(notice.includes(PROJECT_ACCOUNT), "must name the account it would have used");
+ assert.ok(notice.includes(PROJECT_RELAYER), "must name the relayer it would have used");
+ assert.ok(notice.includes(globalFile(home)), "must name where memory is going instead");
+ assert.match(notice, /approve-project/, "must say how to approve it");
+ assert.ok(
+ notice.includes(auth.projectApprovalsPath()),
+ "must say where the approval is recorded",
+ );
+});
+
+test("nothing the user can see ever carries the delegate private key", async (t) => {
+ const { auth } = await sandbox(t, {
+ global: makeCreds(),
+ project: makeCreds({ accountId: PROJECT_ACCOUNT }),
+ });
+
+ assert.ok(!auth.formatProjectCredsNotice().includes(PRIVATE_KEY));
+ auth.approveProjectCreds();
+ const record = readFileSync(auth.projectApprovalsPath(), "utf8");
+ assert.ok(!record.includes(PRIVATE_KEY), "the approval record must hold no key material");
+});
+
+/* --------------------------------------------------------------------- *
+ * Approval.
+ * --------------------------------------------------------------------- */
+
+test("an approved project file is the one that gets used", async (t) => {
+ const { auth, cwd } = await sandbox(t, {
+ global: makeCreds(),
+ project: makeCreds({ accountId: PROJECT_ACCOUNT, relayerUrl: PROJECT_RELAYER }),
+ });
+
+ const result = auth.approveProjectCreds();
+
+ assert.equal(result.outcome, "approved");
+ assert.equal(result.accountId, PROJECT_ACCOUNT);
+ assert.equal(result.relayerUrl, PROJECT_RELAYER);
+ assert.equal(auth.credsPath(), projectFile(cwd));
+ assert.equal(auth.loadCreds()?.accountId, PROJECT_ACCOUNT);
+ assert.equal(auth.resolveCreds().source, "project");
+ assert.equal(auth.formatProjectCredsNotice(), null, "an approved file is not a warning");
+});
+
+test("the approval is recorded outside the repository", async (t) => {
+ // The whole point: a record the repo could carry is a repo approving itself.
+ const { auth, home, cwd } = await sandbox(t, {
+ global: makeCreds(),
+ project: makeCreds({ accountId: PROJECT_ACCOUNT }),
+ });
+
+ auth.approveProjectCreds();
+
+ const approvals = auth.projectApprovalsPath();
+ assert.equal(approvals, join(home, ".memwal", "project-approvals.json"));
+ assert.ok(!approvals.startsWith(cwd), "the approval must not live in the project");
+ assert.equal(existsSync(approvals), true);
+});
+
+test("an approvals file committed inside the repo approves nothing", async (t) => {
+ const { auth, cwd, home } = await sandbox(t, {
+ global: makeCreds(),
+ project: makeCreds({ accountId: ATTACKER_ACCOUNT, relayerUrl: PROJECT_RELAYER }),
+ });
+
+ // Exactly what the repo would have to ship to self-approve: a well-formed
+ // record, in the project's own .memwal, naming its own credentials file.
+ const { createHash } = await import("node:crypto");
+ const fingerprint = createHash("sha256")
+ .update(`${ATTACKER_ACCOUNT}\n0x${"e".repeat(64)}\n${PROJECT_RELAYER}`)
+ .digest("hex");
+ writeFileSync(
+ join(cwd, ".memwal", "project-approvals.json"),
+ JSON.stringify({
+ version: 1,
+ approvals: [
+ {
+ path: projectFile(cwd),
+ fingerprint,
+ accountId: ATTACKER_ACCOUNT,
+ delegateAddress: "0x" + "e".repeat(64),
+ relayerUrl: PROJECT_RELAYER,
+ approvedAt: new Date().toISOString(),
+ },
+ ],
+ }),
+ );
+
+ assert.equal(auth.credsPath(), globalFile(home), "a repo must not approve itself");
+ assert.equal(auth.loadCreds()?.accountId, GLOBAL_ACCOUNT);
+ assert.equal(auth.resolveCreds().project?.decision, "unapproved");
+});
+
+test("approving one project does not approve another", async (t) => {
+ const { auth, home, cwd } = await sandbox(t, {
+ global: makeCreds(),
+ project: makeCreds({ accountId: PROJECT_ACCOUNT }),
+ });
+ auth.approveProjectCreds();
+
+ // A second checkout, with the same credentials in it. Approval is per
+ // project path, so opening this one is a fresh decision.
+ const other = realpathSync(mkdtempSync(join(tmpdir(), "memwal-approve-other-")));
+ t.after(() => rmSync(other, { recursive: true, force: true }));
+ writeCredsAt(other, makeCreds({ accountId: PROJECT_ACCOUNT }));
+ process.chdir(other);
+
+ assert.equal(auth.credsPath(), globalFile(home));
+ assert.equal(auth.resolveCreds().project?.decision, "unapproved");
+});
+
+/* --------------------------------------------------------------------- *
+ * Re-approval after the destination moves.
+ * --------------------------------------------------------------------- */
+
+for (const [what, mutation] of [
+ ["account", { accountId: ATTACKER_ACCOUNT }],
+ ["relayer", { relayerUrl: "https://attacker.example" }],
+ ["delegate key", { delegateAddress: "0x" + "7".repeat(64) }],
+]) {
+ test(`changing the ${what} after approval requires approval again`, async (t) => {
+ const { auth, home, cwd } = await sandbox(t, {
+ global: makeCreds(),
+ project: makeCreds({ accountId: PROJECT_ACCOUNT, relayerUrl: PROJECT_RELAYER }),
+ });
+ auth.approveProjectCreds();
+ assert.equal(auth.credsPath(), projectFile(cwd), "precondition: approved and in use");
+
+ // A later commit edits the file the user already approved.
+ writeCredsAt(
+ cwd,
+ makeCreds({
+ accountId: PROJECT_ACCOUNT,
+ relayerUrl: PROJECT_RELAYER,
+ ...mutation,
+ }),
+ );
+
+ assert.equal(auth.credsPath(), globalFile(home), "a moved destination must not be used");
+ assert.equal(auth.resolveCreds().project?.decision, "changed");
+ const notice = auth.formatProjectCredsNotice();
+ assert.match(notice, /changed since you/, `notice did not report the change:\n${notice}`);
+ });
+}
+
+test("re-approving adopts the new destination and names the old one", async (t) => {
+ const { auth, cwd } = await sandbox(t, {
+ global: makeCreds(),
+ project: makeCreds({ accountId: PROJECT_ACCOUNT, relayerUrl: PROJECT_RELAYER }),
+ });
+ auth.approveProjectCreds();
+ writeCredsAt(cwd, makeCreds({ accountId: ATTACKER_ACCOUNT, relayerUrl: PROJECT_RELAYER }));
+
+ const result = auth.approveProjectCreds();
+
+ assert.equal(result.outcome, "reapproved");
+ assert.equal(result.previousAccountId, PROJECT_ACCOUNT, "must name what it replaced");
+ assert.equal(result.accountId, ATTACKER_ACCOUNT);
+ assert.equal(auth.credsPath(), projectFile(cwd));
+});
+
+test("approving the same destination twice writes nothing new", async (t) => {
+ const { auth } = await sandbox(t, {
+ global: makeCreds(),
+ project: makeCreds({ accountId: PROJECT_ACCOUNT }),
+ });
+ auth.approveProjectCreds();
+
+ assert.equal(auth.approveProjectCreds().outcome, "already-approved");
+ const stored = JSON.parse(readFileSync(auth.projectApprovalsPath(), "utf8"));
+ assert.equal(stored.approvals.length, 1, "approvals must not accumulate duplicates");
+});
+
+test("revoking sends memory back to the global account", async (t) => {
+ const { auth, home, cwd } = await sandbox(t, {
+ global: makeCreds(),
+ project: makeCreds({ accountId: PROJECT_ACCOUNT }),
+ });
+ auth.approveProjectCreds();
+ assert.equal(auth.credsPath(), projectFile(cwd));
+
+ const revoked = auth.revokeProjectCredsApproval();
+
+ assert.equal(revoked.outcome, "revoked");
+ assert.equal(auth.credsPath(), globalFile(home));
+ assert.equal(auth.revokeProjectCredsApproval().outcome, "none", "revoking twice is a no-op");
+});
+
+/* --------------------------------------------------------------------- *
+ * The escape hatch, and the cases that must stay quiet.
+ * --------------------------------------------------------------------- */
+
+test("MEMWAL_CREDS_DIR still overrides both files, approved or not", async (t) => {
+ const { auth, cwd } = await sandbox(t, {
+ global: makeCreds(),
+ project: makeCreds({ accountId: PROJECT_ACCOUNT }),
+ });
+ auth.approveProjectCreds();
+ assert.equal(auth.credsPath(), projectFile(cwd), "precondition: the project file is in use");
+
+ const override = realpathSync(mkdtempSync(join(tmpdir(), "memwal-approve-override-")));
+ t.after(() => {
+ delete process.env.MEMWAL_CREDS_DIR;
+ rmSync(override, { recursive: true, force: true });
+ });
+ process.env.MEMWAL_CREDS_DIR = override;
+
+ assert.equal(auth.credsPath(), join(override, "credentials.json"));
+ assert.equal(auth.resolveCreds().source, "override");
+ assert.equal(auth.formatProjectCredsNotice(), null, "an override has nothing to warn about");
+ assert.equal(
+ auth.approveProjectCreds().outcome,
+ "overridden",
+ "there is nothing to approve while the override decides",
+ );
+});
+
+test("a malformed project credentials file falls back to the global one", async (t) => {
+ const { auth, home, cwd } = await sandbox(t, { global: makeCreds() });
+ mkdirSync(join(cwd, ".memwal"), { recursive: true });
+ writeFileSync(projectFile(cwd), "{ not json");
+
+ assert.equal(auth.credsPath(), globalFile(home));
+ assert.equal(auth.loadCreds()?.accountId, GLOBAL_ACCOUNT, "a broken repo file is not a logout");
+ assert.equal(auth.resolveCreds().project?.decision, "unreadable");
+ assert.match(auth.formatProjectCredsNotice(), /not a valid/);
+ assert.equal(auth.approveProjectCreds().outcome, "unreadable");
+});
+
+test("a machine with no project file behaves exactly as it always did", async (t) => {
+ const { auth, home } = await sandbox(t, { global: makeCreds() });
+
+ assert.equal(auth.credsPath(), globalFile(home));
+ assert.equal(auth.loadCreds()?.accountId, GLOBAL_ACCOUNT);
+ assert.equal(auth.resolveCreds().source, "global");
+ assert.equal(auth.formatProjectCredsNotice(), null, "nothing to report, so nothing is said");
+ assert.equal(auth.approveProjectCreds().outcome, "none");
+ assert.equal(existsSync(auth.projectApprovalsPath()), false, "no file is created for nothing");
+});
+
+test("an unapproved project file does not leave the user signed out", async (t) => {
+ // Falling back must not be a fail-closed: a user with only a repo file and
+ // no global one still gets the normal signed-out sign-in path, not an error.
+ const { auth, home } = await sandbox(t, {
+ project: makeCreds({ accountId: PROJECT_ACCOUNT }),
+ });
+
+ assert.equal(auth.credsPath(), globalFile(home));
+ assert.equal(auth.loadCreds(), null);
+ assert.ok(auth.formatProjectCredsNotice(), "and the ignored file is still explained");
+});
diff --git a/packages/mcp/test/unknown-flags.test.mjs b/packages/mcp/test/unknown-flags.test.mjs
index f6c24de5b..5cf33f52a 100644
--- a/packages/mcp/test/unknown-flags.test.mjs
+++ b/packages/mcp/test/unknown-flags.test.mjs
@@ -68,6 +68,8 @@ test("parseArgs treats no known flag as unknown", () => {
"--help", "-h",
"--logout",
"--login", "login",
+ "--approve-project", "approve-project",
+ "--revoke-project", "revoke-project",
"--prod", "--dev", "--staging", "--local",
"--relayer", "https://r.example",
"--relayer-url", "https://r.example",
@@ -85,6 +87,28 @@ test("parseArgs treats no known flag as unknown", () => {
assert.deepEqual(parseArgs(known).unknown, []);
});
+test("an unknown flag does not swallow the project-approval commands", () => {
+ // They are commands, not values. Swallowing one turns an approval the user
+ // typed into a run that silently approves nothing (WALM-639).
+ for (const command of ["approve-project", "revoke-project"]) {
+ const args = parseArgs(["--typo", command]);
+ assert.deepEqual(args.unknown, ["--typo"]);
+ assert.equal(
+ command === "approve-project" ? args.approveProject : args.revokeProject,
+ true,
+ `\`${command}\` was swallowed as a flag value`,
+ );
+ }
+});
+
+test("--help documents how to approve project-local credentials", () => {
+ // The gate is only actionable if the command that lifts it is discoverable.
+ const help = helpText();
+ assert.ok(help.includes("approve-project"), "approve-project missing from --help");
+ assert.ok(help.includes("revoke-project"), "revoke-project missing from --help");
+ assert.deepEqual(parseArgs(["approve-project"]).unknown, []);
+});
+
test("parseArgs does not mistake a flag's value for an unknown flag", () => {
// `next()` consumes the value, so "MCP Client" must never be reported.
const args = parseArgs(["--label", "MCP Client"]);
From ad2dfe5081bde69cffeea44b4bebc941d11fe70f Mon Sep 17 00:00:00 2001
From: Harry Phan
Date: Thu, 17 Sep 2026 22:15:12 +0700
Subject: [PATCH 065/132] fix(mcp): remove only MemWal's own codex hook entries
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit
The installer treated any hook command containing one of our script
filenames as ours, and then dropped the whole hook group. `on_user_prompt.mjs`
is a generic name: a config holding /company/security/on_user_prompt.mjs lost
that hook and its unrelated sibling /company/security/enforce-policy.mjs on
both install and uninstall. No attacker is needed to trigger it.
Decide ownership per hook instead of per group, and from evidence rather
than a filename:
- Each hook written by the installer now carries a `_memwal` marker, so a
later run recognises its own entries outright.
- Entries from a build predating the marker are matched by exact command
for the current plugin directory — including the pre-WALM-641 unquoted
form — so reinstall and uninstall still work for existing installs.
- Removal filters the group's `hooks` array rather than deleting the
group. Sibling hooks and group settings (matcher, statusMessage, …)
survive; a group is dropped only once empty, an event only once it has
no groups left.
Refs WALM-643
---
.../plugin/scripts/install_codex_hooks.mjs | 98 ++++++--
.../test/codex-installer-ownership.test.mjs | 216 ++++++++++++++++++
2 files changed, 297 insertions(+), 17 deletions(-)
create mode 100644 packages/mcp/test/codex-installer-ownership.test.mjs
diff --git a/packages/mcp/plugin/scripts/install_codex_hooks.mjs b/packages/mcp/plugin/scripts/install_codex_hooks.mjs
index bd169ac7d..de46bf62a 100644
--- a/packages/mcp/plugin/scripts/install_codex_hooks.mjs
+++ b/packages/mcp/plugin/scripts/install_codex_hooks.mjs
@@ -17,8 +17,13 @@
* directory containing $(...), backticks, quotes or backslashes cannot break
* out of either the JSON document or the generated shell command.
*
- * Re-running is idempotent: entries this installer owns (identified by our
- * hook script filenames) are removed before fresh entries are added.
+ * Re-running is idempotent: the hooks this installer owns are removed before
+ * fresh ones are added. Ownership is decided per hook, by a marker this
+ * installer writes (or, for installs predating that marker, by an exact match
+ * against the commands it generates for this plugin directory) — never by the
+ * hook script's filename, which another tool may legitimately share. Hooks
+ * belonging to anyone else, including siblings inside the same group, and the
+ * group's own settings are left untouched.
*
* Usage:
* node install_codex_hooks.mjs # install or update
@@ -47,15 +52,15 @@ const HOOKS_FILE = join(CODEX_DIR, "hooks.json");
const CONFIG_FILE = join(CODEX_DIR, "config.toml");
const TEMPLATE_FILE = join(PLUGIN_ROOT, "hooks", "codex-hooks.json");
-// Entries are "ours" when a hook command references one of our scripts.
-const OWNER_MARKERS = [
- "on_session_start.mjs",
- "on_user_prompt.mjs",
- "on_post_tool.mjs",
-];
-
const PLACEHOLDER = "${PLUGIN_ROOT}";
+// Every hook this installer writes carries this marker, so a later run can
+// recognise its own entries outright instead of guessing from a filename.
+// `on_user_prompt.mjs` is a generic name: another tool's hook may well use it,
+// and that hook is not ours to touch.
+const MARKER_KEY = "_memwal";
+const MARKER_VALUE = "memwal-plugin-hooks";
+
function loadTemplate() {
// Parse first, substitute second. Substituting into the raw text would let
// a path containing a double quote or a backslash rewrite the JSON
@@ -77,28 +82,87 @@ function loadExisting() {
}
}
-function isOwned(entry) {
- for (const hook of entry.hooks || []) {
- const cmd = hook.command || "";
- if (OWNER_MARKERS.some((m) => cmd.includes(m))) return true;
+/**
+ * The exact commands this installer writes for the current PLUGIN_ROOT, plus
+ * the ones it wrote before WALM-641 quoted the path. Installs made by an older
+ * build carry no marker, so they are still recognised — but only when the
+ * command matches ours character for character, which a hook belonging to
+ * another tool never will.
+ */
+function ownedCommands() {
+ const commands = new Set();
+ if (!existsSync(TEMPLATE_FILE)) return commands;
+ let template;
+ try {
+ template = JSON.parse(readFileSync(TEMPLATE_FILE, "utf8"));
+ } catch {
+ return commands;
+ }
+ for (const entries of Object.values(template.hooks || {})) {
+ for (const entry of entries || []) {
+ for (const hook of entry.hooks || []) {
+ if (typeof hook.command !== "string") continue;
+ // What this build writes, and what pre-WALM-641 builds wrote.
+ commands.add(
+ substituteHookPlaceholder(hook, PLACEHOLDER, PLUGIN_ROOT).command
+ );
+ commands.add(hook.command.replaceAll(PLACEHOLDER, PLUGIN_ROOT));
+ }
+ }
}
- return false;
+ return commands;
+}
+
+const OWNED_COMMANDS = ownedCommands();
+
+/** A single hook — not the group around it — that this installer put there. */
+function isOwnedHook(hook) {
+ if (!hook || typeof hook !== "object") return false;
+ if (hook[MARKER_KEY] === MARKER_VALUE) return true;
+ return typeof hook.command === "string" && OWNED_COMMANDS.has(hook.command);
}
+/**
+ * Drop our own hooks and nothing else.
+ *
+ * A group may hold hooks from several tools. Removing the group because one of
+ * its hooks is ours takes the siblings with it (WALM-643), so the group is
+ * rebuilt with its settings intact and only our hooks filtered out. A group is
+ * dropped only once it has no hooks left, and an event only once it has no
+ * groups left.
+ */
function stripOwned(config) {
const hooks = config.hooks || {};
for (const event of Object.keys(hooks)) {
- hooks[event] = (hooks[event] || []).filter((e) => !isOwned(e));
- if (hooks[event].length === 0) delete hooks[event];
+ const kept = [];
+ for (const entry of hooks[event] || []) {
+ if (!entry || !Array.isArray(entry.hooks)) {
+ kept.push(entry);
+ continue;
+ }
+ const keptHooks = entry.hooks.filter((hook) => !isOwnedHook(hook));
+ if (keptHooks.length === entry.hooks.length) kept.push(entry);
+ else if (keptHooks.length > 0) kept.push({ ...entry, hooks: keptHooks });
+ }
+ if (kept.length === 0) delete hooks[event];
+ else hooks[event] = kept;
}
config.hooks = hooks;
return config;
}
+/** Tag each hook so the next run recognises it without matching commands. */
+function markOwned(entries) {
+ return entries.map((entry) => ({
+ ...entry,
+ hooks: (entry.hooks || []).map((hook) => ({ ...hook, [MARKER_KEY]: MARKER_VALUE })),
+ }));
+}
+
function mergeTemplate(config, template) {
config.hooks = config.hooks || {};
for (const [event, entries] of Object.entries(template.hooks || {})) {
- config.hooks[event] = (config.hooks[event] || []).concat(entries);
+ config.hooks[event] = (config.hooks[event] || []).concat(markOwned(entries));
}
return config;
}
diff --git a/packages/mcp/test/codex-installer-ownership.test.mjs b/packages/mcp/test/codex-installer-ownership.test.mjs
new file mode 100644
index 000000000..41b6ed024
--- /dev/null
+++ b/packages/mcp/test/codex-installer-ownership.test.mjs
@@ -0,0 +1,216 @@
+/**
+ * WALM-643: the Codex hooks installer must remove its own hooks and nothing
+ * else.
+ *
+ * Ownership used to be decided by substring — any command containing
+ * `on_user_prompt.mjs` (or another of our generic hook filenames) counted as
+ * ours — and the match then removed the whole hook group. A company's own
+ * `/company/security/on_user_prompt.mjs` therefore disappeared on install and
+ * on uninstall, and took its unrelated siblings in the same group with it.
+ *
+ * These tests drive the real installer against a ~/.codex/hooks.json seeded
+ * with foreign hooks, across install, reinstall and uninstall.
+ */
+import { test } from "node:test";
+import assert from "node:assert/strict";
+import { spawnSync } from "node:child_process";
+import { mkdirSync, mkdtempSync, readFileSync, rmSync, writeFileSync } from "node:fs";
+import { tmpdir } from "node:os";
+import { dirname, join, resolve } from "node:path";
+import { fileURLToPath } from "node:url";
+
+const __dirname = dirname(fileURLToPath(import.meta.url));
+const PLUGIN_ROOT = resolve(__dirname, "../plugin");
+const INSTALLER = join(PLUGIN_ROOT, "scripts", "install_codex_hooks.mjs");
+const TEMPLATE_FILE = join(PLUGIN_ROOT, "hooks", "codex-hooks.json");
+
+const MARKER_KEY = "_memwal";
+
+/** A foreign tool whose hook filename happens to collide with one of ours. */
+const FOREIGN_PROMPT_HOOK = {
+ type: "command",
+ command: 'node "/company/security/on_user_prompt.mjs"',
+ timeout: 30,
+};
+const FOREIGN_SIBLING_HOOK = {
+ type: "command",
+ command: 'node "/company/security/enforce-policy.mjs"',
+ timeout: 30,
+};
+const FOREIGN_GROUP = {
+ matcher: "*",
+ statusMessage: "Running company policy checks...",
+ hooks: [FOREIGN_PROMPT_HOOK, FOREIGN_SIBLING_HOOK],
+};
+const FOREIGN_CONFIG = {
+ hooks: {
+ UserPromptSubmit: [structuredClone(FOREIGN_GROUP)],
+ SessionStart: [
+ {
+ matcher: "startup",
+ hooks: [
+ {
+ type: "command",
+ command: 'node "/company/security/on_session_start.mjs"',
+ },
+ ],
+ },
+ ],
+ },
+};
+
+/**
+ * The command a pre-WALM-643 build wrote for this plugin directory: raw
+ * placeholder substitution over the template text.
+ */
+function legacyCommandFor(event) {
+ const raw = readFileSync(TEMPLATE_FILE, "utf8").replaceAll("${PLUGIN_ROOT}", PLUGIN_ROOT);
+ return JSON.parse(raw).hooks[event][0].hooks[0].command;
+}
+
+function makeHome(t, config) {
+ const home = mkdtempSync(join(process.env.TMPDIR || tmpdir(), "codex-ownership-"));
+ t.after(() => rmSync(home, { recursive: true, force: true }));
+ mkdirSync(join(home, ".codex"), { recursive: true });
+ if (config) {
+ writeFileSync(join(home, ".codex", "hooks.json"), JSON.stringify(config, null, 2) + "\n");
+ }
+ return home;
+}
+
+function runInstaller(home, ...args) {
+ const result = spawnSync(process.execPath, [INSTALLER, ...args], {
+ env: { ...process.env, HOME: home, USERPROFILE: home },
+ encoding: "utf8",
+ });
+ assert.equal(result.status, 0, `${result.stdout}\n${result.stderr}`);
+ return result;
+}
+
+function readHooks(home) {
+ return JSON.parse(readFileSync(join(home, ".codex", "hooks.json"), "utf8"));
+}
+
+/** Groups in `event` that hold at least one hook this installer claims. */
+function memwalGroups(config, event) {
+ return (config.hooks[event] || []).filter((entry) =>
+ (entry.hooks || []).some((hook) => hook[MARKER_KEY] !== undefined)
+ );
+}
+
+function foreignGroups(config, event) {
+ return (config.hooks[event] || []).filter((entry) =>
+ (entry.hooks || []).every((hook) => hook[MARKER_KEY] === undefined)
+ );
+}
+
+test("install keeps a foreign hook whose filename collides with ours", (t) => {
+ const home = makeHome(t, FOREIGN_CONFIG);
+ runInstaller(home);
+ const config = readHooks(home);
+
+ assert.deepEqual(foreignGroups(config, "UserPromptSubmit"), [FOREIGN_GROUP]);
+ assert.deepEqual(foreignGroups(config, "SessionStart"), FOREIGN_CONFIG.hooks.SessionStart);
+ assert.equal(memwalGroups(config, "UserPromptSubmit").length, 1);
+ assert.equal(memwalGroups(config, "SessionStart").length, 1);
+});
+
+test("install marks its own hooks", (t) => {
+ const home = makeHome(t);
+ runInstaller(home);
+ const config = readHooks(home);
+ for (const event of ["SessionStart", "UserPromptSubmit", "PostToolUse"]) {
+ const groups = config.hooks[event];
+ assert.equal(groups.length, 1, event);
+ for (const hook of groups[0].hooks) assert.equal(hook[MARKER_KEY], "memwal-plugin-hooks");
+ }
+});
+
+test("reinstalling is idempotent and leaves foreign hooks alone", (t) => {
+ const home = makeHome(t, FOREIGN_CONFIG);
+ runInstaller(home);
+ const first = readHooks(home);
+ runInstaller(home);
+ const second = readHooks(home);
+
+ assert.deepEqual(second, first);
+ for (const event of ["SessionStart", "UserPromptSubmit", "PostToolUse"]) {
+ assert.equal(memwalGroups(second, event).length, 1, `duplicate MemWal group in ${event}`);
+ }
+ assert.deepEqual(foreignGroups(second, "UserPromptSubmit"), [FOREIGN_GROUP]);
+});
+
+test("uninstall removes only our hooks and restores the file to its old state", (t) => {
+ const home = makeHome(t, FOREIGN_CONFIG);
+ runInstaller(home);
+ runInstaller(home, "--uninstall");
+ assert.deepEqual(readHooks(home), FOREIGN_CONFIG);
+});
+
+test("a foreign sibling survives when our hook is removed from its group", (t) => {
+ // A group holding a foreign hook, our hook, and another foreign hook. Only
+ // the middle one is ours; the group and its settings must survive.
+ const ours = {
+ type: "command",
+ command: legacyCommandFor("UserPromptSubmit"),
+ timeout: 12,
+ };
+ const shared = {
+ matcher: "*",
+ statusMessage: "Running company policy checks...",
+ hooks: [FOREIGN_PROMPT_HOOK, ours, FOREIGN_SIBLING_HOOK],
+ };
+ const home = makeHome(t, { hooks: { UserPromptSubmit: [shared] } });
+
+ runInstaller(home, "--uninstall");
+ const config = readHooks(home);
+ assert.deepEqual(config.hooks.UserPromptSubmit, [
+ {
+ matcher: "*",
+ statusMessage: "Running company policy checks...",
+ hooks: [FOREIGN_PROMPT_HOOK, FOREIGN_SIBLING_HOOK],
+ },
+ ]);
+});
+
+test("an install predating the ownership marker is still replaced, not duplicated", (t) => {
+ const legacy = {
+ hooks: {
+ UserPromptSubmit: [
+ {
+ hooks: [
+ {
+ type: "command",
+ command: legacyCommandFor("UserPromptSubmit"),
+ timeout: 12,
+ },
+ ],
+ },
+ ],
+ },
+ };
+ const home = makeHome(t, legacy);
+
+ runInstaller(home);
+ const config = readHooks(home);
+ assert.equal(config.hooks.UserPromptSubmit.length, 1);
+ assert.equal(config.hooks.UserPromptSubmit[0].hooks.length, 1);
+ assert.equal(
+ config.hooks.UserPromptSubmit[0].hooks[0][MARKER_KEY],
+ "memwal-plugin-hooks"
+ );
+
+ runInstaller(home, "--uninstall");
+ assert.deepEqual(readHooks(home), { hooks: {} });
+});
+
+test("an emptied group is dropped, but an event keeping foreign groups is not", (t) => {
+ const home = makeHome(t, FOREIGN_CONFIG);
+ runInstaller(home);
+ runInstaller(home, "--uninstall");
+ const config = readHooks(home);
+
+ assert.deepEqual(Object.keys(config.hooks).sort(), ["SessionStart", "UserPromptSubmit"]);
+ // PostToolUse held only our group, so the event is gone entirely.
+ assert.equal(config.hooks.PostToolUse, undefined);
+});
From 1955afb3001442e40d9b3a353cd94eafb22fcddc Mon Sep 17 00:00:00 2001
From: Harry Phan
Date: Thu, 17 Sep 2026 22:22:00 +0700
Subject: [PATCH 066/132] fix(mcp): launch the MCP server from a trusted
absolute path instead of npx
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit
The plugin started the server with `npx -y @mysten-incubation/memwal-mcp@`.
npx resolves a package name against the directory the MCP client was started in —
the project the user has open — so a project holding an installed package of that
name, claiming the pinned version, was run instead of the published one. The
version pin was no defence: the planted package simply claims that version.
Reproduced offline, where a pinned `npx ...@0.0.14` ran the project's own binary.
Every launch site now runs plugin/scripts/launch_mcp.mjs, which installs the
pinned version once into ~/.memwal/runtime/memwal-mcp@ and spawns that
absolute entry point with the current node binary. Nothing on that path consults
the project's node_modules, a PATH-relative bin shim, or the project's .npmrc,
and the launcher fails rather than falling back to the package name when the
trusted install cannot be produced. The install is staged and renamed into place
so concurrent clients never read a half-written tree, and the tree is verified
(package name, pinned version, bin resolving inside the package) before use.
Arguments and environment pass through untouched, and the working directory is
left alone so project-local credentials still resolve as they do today.
The pin stays in one place — plugin/plugin.json's version — which the release
verifier already holds equal to packages/mcp/package.json.
Updated together: .mcp.json, .cursor-mcp.json, .codex-mcp.json, the Codex
fallback installer's registration block, the release verifier, and the docs that
described the old npx launch. Docs blocks for MCP-only (non-plugin) configs keep
the npx form, which scripts/check-docs-code-sync.mjs pins to the package README;
they now carry a note naming the exposure and the trusted-path alternative.
Regression test: packages/mcp/test/trusted-launcher.test.mjs plants a fake
@mysten-incubation/memwal-mcp — including one claiming the exact pinned version,
plus the node_modules/.bin shim npx would have found — in a temp project and
asserts the resolved launch path is the trusted absolute one, that the trusted
entry is what actually runs, and that a missing trusted install fails instead of
falling back.
Refs WALM-640
---
docs/mcp/changelog.mdx | 1 +
docs/mcp/claude-code.md | 4 +-
docs/mcp/quickstart.md | 1 +
packages/mcp/CHANGELOG.md | 1 +
packages/mcp/README.md | 56 ++++
packages/mcp/TESTING.md | 10 +-
packages/mcp/plugin/.codex-mcp.json | 4 +-
packages/mcp/plugin/.cursor-mcp.json | 4 +-
packages/mcp/plugin/.mcp.json | 4 +-
.../plugin/scripts/install_codex_hooks.mjs | 21 +-
packages/mcp/plugin/scripts/launch_mcp.mjs | 73 +++++
.../mcp/plugin/scripts/lib/mcp-launch.mjs | 237 +++++++++++++++
packages/mcp/test/trusted-launcher.test.mjs | 281 ++++++++++++++++++
scripts/verify-manual-sdk-release.mjs | 49 +--
14 files changed, 707 insertions(+), 39 deletions(-)
create mode 100644 packages/mcp/plugin/scripts/launch_mcp.mjs
create mode 100644 packages/mcp/plugin/scripts/lib/mcp-launch.mjs
create mode 100644 packages/mcp/test/trusted-launcher.test.mjs
diff --git a/docs/mcp/changelog.mdx b/docs/mcp/changelog.mdx
index bcd497c0f..b3f5db1c6 100644
--- a/docs/mcp/changelog.mdx
+++ b/docs/mcp/changelog.mdx
@@ -35,6 +35,7 @@ answer: >-
### Fixed
+- The plugin no longer starts the MCP server through `npx @mysten-incubation/memwal-mcp@`. npx resolves a package name against the directory the MCP client was started in — the project the user has open — so a project carrying an installed package of that name, claiming the pinned version, was run instead of the published one; a pinned `npx …@0.0.14` command was reproduced running a project's own binary while offline. The version pin was no defence, because the planted package simply claims that version. Every launch site (`.mcp.json`, the Cursor and Codex copies, and the Codex fallback installer) now runs `plugin/scripts/launch_mcp.mjs`, which installs the pinned version once under `~/.memwal/runtime/memwal-mcp@` and launches that absolute entry point with the current node binary. Nothing on that path consults the project's `node_modules`, a PATH-relative bin shim, or the project's `.npmrc`, and the launcher fails instead of falling back to the name when the trusted install cannot be produced. Exploiting the old behaviour required write access to installed package and bin files inside the project, so a plain repository clone or a `package.json` alone was never enough. `MEMWAL_MCP_RUNTIME_DIR` relocates the trusted directory and is rejected unless it is an absolute path. (WALM-640)
- Bound every relayer call the tools make. The pinned SDK aborts a request only when the caller passes a signal, which none of the memory methods do, so a stalled socket kept a tool running with no ceiling — `memwal_remember` was observed still going past 120s against a 90s budget. Accepts are bounded at 15s (`MEMWAL_MCP_ACCEPT_DEADLINE_MS`), waits at their own budget plus grace. The request is not cancelled — the SDK exposes no way to pass a signal — but the agent is no longer held by it.
- Honour the relayer's `retry_after` instead of dropping the write. Once the per-delegate-key budget (60 weighted requests/minute) is spent the relayer answers 429 with a cooldown, and nothing backed off: the fact was never written and the agent saw only an opaque tool error. A short cooldown is now absorbed; a long one is reported with the wait named, stating plainly that the fact was NOT saved and pointing at the cheaper shape — one `memwal_remember_bulk` rather than N single calls, one `memwal_remember_status(job_ids)` rather than N status calls. Only rejections that provably never reached the handler retry, so `/api/remember/bulk`, which carries no idempotency key, cannot be duplicated by a retry.
- `memwal_remember` sends a content-derived idempotency key, so the retry its own timeout message invites really does attach to the job already in flight instead of storing a second paid copy. The key is computed by the tool rather than relied on from the SDK, whose published build mints a random UUID per client instance.
diff --git a/docs/mcp/claude-code.md b/docs/mcp/claude-code.md
index 160969ebe..cfd979f86 100644
--- a/docs/mcp/claude-code.md
+++ b/docs/mcp/claude-code.md
@@ -97,7 +97,9 @@ Add MemWal to Claude Code so it recalls context and saves durable facts as you w
| MemWal MCP (memory tools) | ✓ | ✓ |
| Lifecycle hooks (automatic recall/save) | ✓ | ✗ |
-MCP-only still saves and recalls on its own because the tools are proactive. The plugin adds hooks that reinforce the behavior and make the agent **prefer Walrus Memory over Claude Code's built-in memory**. The plugin pins the MCP server version, so `npx` cannot keep a cached older package.
+MCP-only still saves and recalls on its own because the tools are proactive. The plugin adds hooks that reinforce the behavior and make the agent **prefer Walrus Memory over Claude Code's built-in memory**.
+
+The plugin also starts the server differently. Instead of resolving the package name with `npx`, it installs the pinned version once into `~/.memwal/runtime/memwal-mcp@` and launches that absolute path. `npx` resolves a name against the directory the client was started in, which is your project, so a package installed there under the same name and claiming the pinned version would have been run instead — the version pin does not prevent that. The plugin's launcher never looks at your project's `node_modules`, and it fails rather than falling back if the pinned version cannot be installed.
## Available tools
diff --git a/docs/mcp/quickstart.md b/docs/mcp/quickstart.md
index 963d6cffb..71053ebfe 100644
--- a/docs/mcp/quickstart.md
+++ b/docs/mcp/quickstart.md
@@ -45,6 +45,7 @@ Every supported client runs the same local server, `npx -y @mysten-incubation/me
## Prerequisites
- You need Node.js 20 or later, because the server runs through `npx` with no install step.
+- `npx` resolves the package name against the directory your client starts the server in, which is usually the project you have open. A project that contains an installed `@mysten-incubation/memwal-mcp` of its own would be run instead of the published package, and pinning a version in the command does not prevent it. The [plugin](/mcp/claude-code) avoids this by installing the pinned version into `~/.memwal/runtime` and launching that absolute path; to get the same property without the plugin, install the version yourself outside any project and point `command`/`args` at the absolute entry point (see the [package README](https://github.com/MystenLabs/MemWal/blob/dev/packages/mcp/README.md#how-the-plugin-launches-the-server)).
- You need a [Walrus Memory account](/fundamentals/concepts/ownership-and-access). An unauthenticated memory-tool call returns sign-in instructions rather than signing you in, so ask the agent to run `memwal_login` and follow the URL it returns to connect your wallet. Config files carry no keys.
## Set up your client
diff --git a/packages/mcp/CHANGELOG.md b/packages/mcp/CHANGELOG.md
index 0c06b4def..fed0680e8 100644
--- a/packages/mcp/CHANGELOG.md
+++ b/packages/mcp/CHANGELOG.md
@@ -4,6 +4,7 @@
### Fixed
+- The plugin no longer starts the MCP server through `npx @mysten-incubation/memwal-mcp@`. npx resolves a package name against the directory the MCP client was started in — the project the user has open — so a project carrying an installed package of that name, claiming the pinned version, was run instead of the published one; a pinned `npx …@0.0.14` command was reproduced running a project's own binary while offline. The version pin was no defence, because the planted package simply claims that version. Every launch site (`.mcp.json`, the Cursor and Codex copies, and the Codex fallback installer) now runs `plugin/scripts/launch_mcp.mjs`, which installs the pinned version once under `~/.memwal/runtime/memwal-mcp@` and launches that absolute entry point with the current node binary. Nothing on that path consults the project's `node_modules`, a PATH-relative bin shim, or the project's `.npmrc`, and the launcher fails instead of falling back to the name when the trusted install cannot be produced. Exploiting the old behaviour required write access to installed package and bin files inside the project, so a plain repository clone or a `package.json` alone was never enough. `MEMWAL_MCP_RUNTIME_DIR` relocates the trusted directory and is rejected unless it is an absolute path. (WALM-640)
- Bound every relayer call the tools make. The pinned SDK aborts a request only when the caller passes a signal, which none of the memory methods do, so a stalled socket kept a tool running with no ceiling — `memwal_remember` was observed still going past 120s against a 90s budget. Accepts are bounded at 15s (`MEMWAL_MCP_ACCEPT_DEADLINE_MS`), waits at their own budget plus grace. The request is not cancelled — the SDK exposes no way to pass a signal — but the agent is no longer held by it.
- Honour the relayer's `retry_after` instead of dropping the write. Once the per-delegate-key budget (60 weighted requests/minute) is spent the relayer answers 429 with a cooldown, and nothing backed off: the fact was never written and the agent saw only an opaque tool error. A short cooldown is now absorbed; a long one is reported with the wait named, stating plainly that the fact was NOT saved and pointing at the cheaper shape — one `memwal_remember_bulk` rather than N single calls, one `memwal_remember_status(job_ids)` rather than N status calls. Only rejections that provably never reached the handler retry, so `/api/remember/bulk`, which carries no idempotency key, cannot be duplicated by a retry.
- `memwal_remember` sends a content-derived idempotency key, so the retry its own timeout message invites really does attach to the job already in flight instead of storing a second paid copy. The key is computed by the tool rather than relied on from the SDK, whose published build mints a random UUID per client instance.
diff --git a/packages/mcp/README.md b/packages/mcp/README.md
index 4a21e8584..fb21f44e7 100644
--- a/packages/mcp/README.md
+++ b/packages/mcp/README.md
@@ -23,6 +23,62 @@ Add Walrus Memory MCP to your MCP client config:
}
```
+## How the plugin launches the server
+
+The [MemWal plugin](https://memory.walrus.xyz/mcp/claude-code) does **not** use the
+`npx` form above. `npx` resolves a package *name* against the directory the MCP
+client was started in — your project — so a project that contains an installed
+`@mysten-incubation/memwal-mcp` claiming the pinned version would be run instead of
+the published one. Pinning the version in the `npx` command does not prevent that:
+the planted package simply claims the pinned version.
+
+Instead, every plugin launch config runs the plugin's launcher:
+
+```json
+{
+ "mcpServers": {
+ "memwal": {
+ "command": "node",
+ "args": ["${CLAUDE_PLUGIN_ROOT}/scripts/launch_mcp.mjs"]
+ }
+ }
+}
+```
+
+The launcher installs the pinned version once into a directory it owns and then
+runs that absolute entry point with the current `node` binary:
+
+```
+~/.memwal/runtime/memwal-mcp@/node_modules/@mysten-incubation/memwal-mcp/dist/bin/memwal-mcp.js
+```
+
+It never consults your project's `node_modules`, a `PATH`-relative bin shim, or
+your project's `.npmrc`, and it fails rather than falling back to the package name
+if the pinned version cannot be installed. Everything after the script path is
+forwarded to the server unchanged, so flags such as `--namespace work` or
+`--relayer ` work exactly as they do above. Set `MEMWAL_MCP_RUNTIME_DIR` (an
+absolute path) to move the trusted directory elsewhere.
+
+If you configure MemWal without the plugin and want the same property, install the
+version you intend to run into a directory outside any project and point your
+client at its absolute path:
+
+```sh
+mkdir -p ~/.memwal/runtime/memwal-mcp@0.0.14
+npm install --prefix ~/.memwal/runtime/memwal-mcp@0.0.14 @mysten-incubation/memwal-mcp@0.0.14
+```
+
+```json
+{
+ "mcpServers": {
+ "memwal": {
+ "command": "node",
+ "args": ["/absolute/path/to/home/.memwal/runtime/memwal-mcp@0.0.14/node_modules/@mysten-incubation/memwal-mcp/dist/bin/memwal-mcp.js"]
+ }
+ }
+}
+```
+
## Login
Run the login flow manually:
diff --git a/packages/mcp/TESTING.md b/packages/mcp/TESTING.md
index d4f4e9dda..a829f54bf 100644
--- a/packages/mcp/TESTING.md
+++ b/packages/mcp/TESTING.md
@@ -4,9 +4,11 @@ Step-by-step plan to test the auto-memory work (agentic tools + bulk + health +
> **Local note:** the standalone configs below point at your **local build** via
> `node /Users/uydev/code/MemWal/packages/mcp/dist/bin/memwal-mcp.js --local`
-> (tests your local code, not the published npx package). The **shipped plugin
-> `.mcp.json` is prod** (`npx -y @mysten-incubation/memwal-mcp`) — no `--local` in
-> the committed file. It still reaches your **local relayer** because your saved
+> (tests your local code, not the published package). The **shipped plugin
+> `.mcp.json` is prod** — it runs `node "${CLAUDE_PLUGIN_ROOT}/scripts/launch_mcp.mjs"`,
+> which installs the pinned version into `~/.memwal/runtime` and launches that
+> absolute path (never `npx`, never your project's `node_modules`) — no `--local`
+> in the committed file. It still reaches your **local relayer** because your saved
> creds (`~/.memwal/credentials.json`) point there. To force a local target by hand
> (e.g. fresh/prod creds), `export MEMWAL_SERVER_URL=http://127.0.0.1:8000` in the
> shell before launching the client — the prod bin reads it (`index.ts:116-117`),
@@ -195,7 +197,7 @@ Notes:
## 4. Before the PR
-- [x] `packages/mcp/plugin/.mcp.json` ships the prod default (`npx -y @mysten-incubation/memwal-mcp`) — no `--local` in the committed file
+- [x] `packages/mcp/plugin/.mcp.json` ships the prod default (`node "${CLAUDE_PLUGIN_ROOT}/scripts/launch_mcp.mjs"`, which installs the pin under `~/.memwal/runtime` and runs it by absolute path) — no `--local` in the committed file
- [ ] Re-register `memwal-local` if you removed it for the plugin test
- [ ] Remove the temporary local `memwal` entries from Claude Desktop / Cursor / Codex / OpenCode configs (or keep for ongoing local dev)
diff --git a/packages/mcp/plugin/.codex-mcp.json b/packages/mcp/plugin/.codex-mcp.json
index b833c2d77..5056557f1 100644
--- a/packages/mcp/plugin/.codex-mcp.json
+++ b/packages/mcp/plugin/.codex-mcp.json
@@ -1,8 +1,8 @@
{
"mcpServers": {
"memwal": {
- "command": "npx",
- "args": ["-y", "@mysten-incubation/memwal-mcp@0.0.14"]
+ "command": "node",
+ "args": ["${PLUGIN_ROOT}/scripts/launch_mcp.mjs"]
}
}
}
diff --git a/packages/mcp/plugin/.cursor-mcp.json b/packages/mcp/plugin/.cursor-mcp.json
index b833c2d77..e3b5dc20a 100644
--- a/packages/mcp/plugin/.cursor-mcp.json
+++ b/packages/mcp/plugin/.cursor-mcp.json
@@ -1,8 +1,8 @@
{
"mcpServers": {
"memwal": {
- "command": "npx",
- "args": ["-y", "@mysten-incubation/memwal-mcp@0.0.14"]
+ "command": "node",
+ "args": ["${CURSOR_PLUGIN_ROOT}/scripts/launch_mcp.mjs"]
}
}
}
diff --git a/packages/mcp/plugin/.mcp.json b/packages/mcp/plugin/.mcp.json
index b833c2d77..baed20fd6 100644
--- a/packages/mcp/plugin/.mcp.json
+++ b/packages/mcp/plugin/.mcp.json
@@ -1,8 +1,8 @@
{
"mcpServers": {
"memwal": {
- "command": "npx",
- "args": ["-y", "@mysten-incubation/memwal-mcp@0.0.14"]
+ "command": "node",
+ "args": ["${CLAUDE_PLUGIN_ROOT}/scripts/launch_mcp.mjs"]
}
}
}
diff --git a/packages/mcp/plugin/scripts/install_codex_hooks.mjs b/packages/mcp/plugin/scripts/install_codex_hooks.mjs
index f4d2a71fe..6d4de785b 100644
--- a/packages/mcp/plugin/scripts/install_codex_hooks.mjs
+++ b/packages/mcp/plugin/scripts/install_codex_hooks.mjs
@@ -32,10 +32,6 @@ import { fileURLToPath } from "node:url";
const SCRIPT_DIR = dirname(fileURLToPath(import.meta.url));
const PLUGIN_ROOT = dirname(SCRIPT_DIR);
-function resolveMcpVersion() {
- return JSON.parse(readFileSync(join(PLUGIN_ROOT, "plugin.json"), "utf8")).version;
-}
-
const CODEX_DIR = join(homedir(), ".codex");
const HOOKS_FILE = join(CODEX_DIR, "hooks.json");
const CONFIG_FILE = join(CODEX_DIR, "config.toml");
@@ -97,16 +93,25 @@ function writeHooks(config) {
writeFileSync(HOOKS_FILE, JSON.stringify(config, null, 2) + "\n");
}
-/** Append [mcp_servers.memwal] to config.toml if it isn't registered yet. */
+/**
+ * Append [mcp_servers.memwal] to config.toml if it isn't registered yet.
+ *
+ * Registers the plugin's launcher by absolute path rather than
+ * `npx @mysten-incubation/memwal-mcp@`. npx resolves that name against the
+ * directory Codex is started in, so a package installed in the user's project under
+ * the same name — claiming the pinned version — was run instead of ours (WALM-640).
+ * The launcher installs the pinned version under ~/.memwal/runtime and runs that
+ * absolute entry point, so no project directory takes part in the resolution.
+ */
function ensureMcpRegistered() {
mkdirSync(CODEX_DIR, { recursive: true });
let content = existsSync(CONFIG_FILE) ? readFileSync(CONFIG_FILE, "utf8") : "";
if (content.includes("[mcp_servers.memwal]")) return false;
- const spec = `@mysten-incubation/memwal-mcp@${resolveMcpVersion()}`;
+ const launcher = join(SCRIPT_DIR, "launch_mcp.mjs");
const block =
"\n[mcp_servers.memwal]\n" +
- 'command = "npx"\n' +
- `args = ["-y", "${spec}"]\n`;
+ 'command = "node"\n' +
+ `args = [${JSON.stringify(launcher)}]\n`;
writeFileSync(CONFIG_FILE, (content.trimEnd() + "\n" + block).trimStart());
return true;
}
diff --git a/packages/mcp/plugin/scripts/launch_mcp.mjs b/packages/mcp/plugin/scripts/launch_mcp.mjs
new file mode 100644
index 000000000..96f96a2a4
--- /dev/null
+++ b/packages/mcp/plugin/scripts/launch_mcp.mjs
@@ -0,0 +1,73 @@
+#!/usr/bin/env node
+/**
+ * The plugin's MCP launcher (WALM-640).
+ *
+ * Replaces `npx -y @mysten-incubation/memwal-mcp@` in every plugin manifest.
+ * npx resolves that name against the directory the MCP client started in — the
+ * user's project — so a package installed there under the same name, claiming the
+ * pinned version, was run instead of ours. This launcher never resolves a name:
+ * it makes sure the pinned version is installed under ~/.memwal/runtime and runs
+ * that absolute entry point with the current node binary.
+ *
+ * Everything after the script path is forwarded to the server untouched, so the
+ * manifests keep working with flags such as `--dev`, `--namespace work` or
+ * `--relayer `. The environment is inherited as-is, and the working directory
+ * is left alone so project-local credentials (`.memwal/credentials.json` at or
+ * above the cwd) still resolve the way they do today.
+ *
+ * Usage:
+ * node launch_mcp.mjs [server args...] # start the MCP stdio server
+ * node launch_mcp.mjs --print-entry # print the resolved path and exit
+ */
+import { spawn } from "node:child_process";
+
+import { ensureTrustedEntry, pinnedVersion } from "./lib/mcp-launch.mjs";
+
+const argv = process.argv.slice(2);
+const printOnly = argv[0] === "--print-entry";
+
+let entry;
+try {
+ entry = ensureTrustedEntry();
+} catch (err) {
+ process.stderr.write(
+ `[memwal-mcp] launcher: could not prepare the pinned server ` +
+ `(${pinnedVersionSafe()}): ${err?.message ?? String(err)}\n` +
+ `[memwal-mcp] launcher: refusing to fall back to a package resolved from ` +
+ `the current directory.\n`,
+ );
+ process.exit(1);
+}
+
+if (printOnly) {
+ process.stdout.write(entry + "\n");
+ process.exit(0);
+}
+
+const child = spawn(process.execPath, [entry, ...argv], {
+ stdio: "inherit",
+ env: process.env,
+});
+
+for (const signal of ["SIGINT", "SIGTERM", "SIGHUP"]) {
+ process.on(signal, () => {
+ if (!child.killed) child.kill(signal);
+ });
+}
+
+child.on("error", (err) => {
+ process.stderr.write(`[memwal-mcp] launcher: failed to start ${entry}: ${err.message}\n`);
+ process.exitCode = 1;
+});
+
+child.on("exit", (code, signal) => {
+ process.exitCode = signal ? 1 : (code ?? 0);
+});
+
+function pinnedVersionSafe() {
+ try {
+ return pinnedVersion();
+ } catch {
+ return "unknown version";
+ }
+}
diff --git a/packages/mcp/plugin/scripts/lib/mcp-launch.mjs b/packages/mcp/plugin/scripts/lib/mcp-launch.mjs
new file mode 100644
index 000000000..61db4ecac
--- /dev/null
+++ b/packages/mcp/plugin/scripts/lib/mcp-launch.mjs
@@ -0,0 +1,237 @@
+/**
+ * Trusted-path resolution for the MemWal MCP server (WALM-640).
+ *
+ * The plugin used to start the server with `npx -y @mysten-incubation/memwal-mcp@`.
+ * npx resolves a package name against the *current working directory* first, and an
+ * MCP client starts its servers in the project the user has open. A project that
+ * carries an installed package of the same name whose `version` matches the pin
+ * therefore wins: the pinned `npx` command ran the project's own binary, offline,
+ * with the user's credentials in reach. The version pin does not help, because the
+ * fake package simply claims the pinned version.
+ *
+ * The fix is to stop resolving a *name* in an untrusted directory and to run an
+ * *absolute path* inside a directory we own:
+ *
+ * ~/.memwal/runtime/memwal-mcp@/node_modules/@mysten-incubation/memwal-mcp
+ *
+ * The pinned version is installed there once (npm, with the trusted directory as
+ * both prefix and cwd, so no project `.npmrc` or project `node_modules` is in play),
+ * and every later launch is a plain `node `. Nothing here ever
+ * looks at `process.cwd()`, at a project `node_modules`, or at a PATH-relative bin
+ * shim, so a package planted in a project cannot be reached at all.
+ *
+ * The installed tree is verified before it is used and before it is published to its
+ * final path: the manifest must carry our package name, the pinned version, and a
+ * `bin` entry that resolves back inside the package directory.
+ *
+ * The version pin lives in exactly one place — `plugin/plugin.json`'s `version`,
+ * which the release verifier already keeps equal to `packages/mcp/package.json` —
+ * so the manifests cannot drift from the version that actually gets installed.
+ */
+import {
+ existsSync,
+ mkdirSync,
+ mkdtempSync,
+ readFileSync,
+ renameSync,
+ rmSync,
+ writeFileSync,
+} from "node:fs";
+import { spawnSync } from "node:child_process";
+import { homedir } from "node:os";
+import { dirname, isAbsolute, join, relative, resolve } from "node:path";
+import { fileURLToPath } from "node:url";
+
+export const MCP_PACKAGE_NAME = "@mysten-incubation/memwal-mcp";
+export const MCP_BIN_NAME = "memwal-mcp";
+
+const SCRIPT_DIR = dirname(fileURLToPath(import.meta.url));
+/** plugin/scripts/lib -> plugin/scripts -> plugin */
+export const PLUGIN_ROOT = dirname(dirname(SCRIPT_DIR));
+
+/**
+ * The single source of truth for the version the plugin launches.
+ * Kept equal to packages/mcp/package.json by scripts/verify-manual-sdk-release.mjs.
+ */
+export function pinnedVersion(pluginRoot = PLUGIN_ROOT) {
+ const manifest = JSON.parse(readFileSync(join(pluginRoot, "plugin.json"), "utf8"));
+ const version = manifest.version;
+ if (typeof version !== "string" || version.trim() === "") {
+ throw new Error(`${join(pluginRoot, "plugin.json")} has no usable "version"`);
+ }
+ return version;
+}
+
+/**
+ * Root of the trusted install area. `MEMWAL_MCP_RUNTIME_DIR` may relocate it, but
+ * only to an absolute path: a relative one would resolve against the project the
+ * client happens to have open, which is the directory this whole module exists to
+ * stay out of.
+ */
+export function runtimeRoot() {
+ const override = process.env.MEMWAL_MCP_RUNTIME_DIR;
+ if (override !== undefined && override !== "") {
+ if (!isAbsolute(override)) {
+ throw new Error(
+ `MEMWAL_MCP_RUNTIME_DIR must be an absolute path, received "${override}"`,
+ );
+ }
+ return override;
+ }
+ return join(homedir(), ".memwal", "runtime");
+}
+
+/** One directory per pinned version, so an upgrade never mutates a running install. */
+export function installDir(version = pinnedVersion(), root = runtimeRoot()) {
+ return join(root, `${MCP_BIN_NAME}@${version}`);
+}
+
+function isInside(parent, child) {
+ const rel = relative(parent, child);
+ return rel !== "" && !rel.startsWith("..") && !isAbsolute(rel);
+}
+
+/**
+ * The absolute entry point of the pinned package inside `dir`, or null when `dir`
+ * does not hold a usable install. Never throws for a merely-absent install; throws
+ * only when a present install is malformed in a way worth surfacing.
+ */
+export function resolveInstalledEntry(dir, version) {
+ const packageDir = join(dir, "node_modules", ...MCP_PACKAGE_NAME.split("/"));
+ const manifestPath = join(packageDir, "package.json");
+ if (!existsSync(manifestPath)) return null;
+
+ let manifest;
+ try {
+ manifest = JSON.parse(readFileSync(manifestPath, "utf8"));
+ } catch {
+ return null;
+ }
+ if (manifest.name !== MCP_PACKAGE_NAME) return null;
+ if (version !== undefined && manifest.version !== version) return null;
+
+ const bin = manifest.bin;
+ const relBin = typeof bin === "string" ? bin : bin?.[MCP_BIN_NAME];
+ if (typeof relBin !== "string" || relBin === "") return null;
+
+ const entry = resolve(packageDir, relBin);
+ if (!isInside(packageDir, entry)) {
+ throw new Error(
+ `${manifestPath} declares a bin outside its own package directory (${entry})`,
+ );
+ }
+ if (!existsSync(entry)) return null;
+ return entry;
+}
+
+function npmCommand() {
+ return process.platform === "win32" ? "npm.cmd" : "npm";
+}
+
+/**
+ * Install the pinned version into the trusted area and return its absolute entry
+ * point. Staged in a sibling temp directory and renamed into place, so a second
+ * client starting at the same moment either wins the rename or finds the finished
+ * install — neither ever reads a half-written tree.
+ */
+function install(version, root) {
+ const target = installDir(version, root);
+ mkdirSync(root, { recursive: true });
+ const staging = mkdtempSync(join(root, `.staging-${MCP_BIN_NAME}-`));
+
+ try {
+ // A private manifest stops npm from walking up out of the trusted area
+ // looking for a package.json to attach the install to.
+ writeFileSync(
+ join(staging, "package.json"),
+ JSON.stringify(
+ { name: "memwal-mcp-runtime", version: "0.0.0", private: true },
+ null,
+ 2,
+ ) + "\n",
+ );
+
+ const spec = `${MCP_PACKAGE_NAME}@${version}`;
+ const result = spawnSync(
+ npmCommand(),
+ [
+ "install",
+ spec,
+ "--prefix",
+ staging,
+ "--no-audit",
+ "--no-fund",
+ "--no-save",
+ "--loglevel=error",
+ ],
+ {
+ // cwd inside the trusted area: npm reads .npmrc from cwd upward, and
+ // the project's .npmrc must not get to choose the registry we install
+ // the reviewed version from.
+ cwd: staging,
+ encoding: "utf8",
+ stdio: ["ignore", "pipe", "pipe"],
+ },
+ );
+ if (result.error) {
+ throw new Error(`could not run ${npmCommand()}: ${result.error.message}`);
+ }
+ if (result.status !== 0) {
+ throw new Error(
+ `${npmCommand()} install ${spec} failed (exit ${result.status}): ` +
+ `${(result.stderr || result.stdout || "").trim()}`,
+ );
+ }
+
+ const staged = resolveInstalledEntry(staging, version);
+ if (!staged) {
+ throw new Error(
+ `${npmCommand()} install ${spec} reported success but produced no usable ` +
+ `${MCP_PACKAGE_NAME} entry point in ${staging}`,
+ );
+ }
+
+ try {
+ renameSync(staging, target);
+ } catch (err) {
+ // Lost the race, or a previous run left the directory behind: fall back to
+ // whatever is at the final path, but only if it verifies.
+ const existing = resolveInstalledEntry(target, version);
+ if (!existing) throw err;
+ return existing;
+ }
+ } finally {
+ rmSync(staging, { recursive: true, force: true });
+ }
+
+ const entry = resolveInstalledEntry(target, version);
+ if (!entry) {
+ throw new Error(`installed ${MCP_PACKAGE_NAME}@${version} is missing from ${target}`);
+ }
+ return entry;
+}
+
+/**
+ * The absolute path the plugin should launch. Installs the pinned version into the
+ * trusted area on first use; later calls are a stat of a known path.
+ *
+ * There is deliberately no fallback: if the trusted install cannot be produced, the
+ * launcher fails instead of reaching for a package name that a project could answer.
+ */
+export function ensureTrustedEntry({ version = pinnedVersion(), root = runtimeRoot() } = {}) {
+ const existing = resolveInstalledEntry(installDir(version, root), version);
+ if (existing) return existing;
+ return install(version, root);
+}
+
+/** Exported for the regression test: the path we expect, without installing anything. */
+export function expectedEntryPath({ version = pinnedVersion(), root = runtimeRoot() } = {}) {
+ return join(
+ installDir(version, root),
+ "node_modules",
+ ...MCP_PACKAGE_NAME.split("/"),
+ "dist",
+ "bin",
+ `${MCP_BIN_NAME}.js`,
+ );
+}
diff --git a/packages/mcp/test/trusted-launcher.test.mjs b/packages/mcp/test/trusted-launcher.test.mjs
new file mode 100644
index 000000000..7a343a751
--- /dev/null
+++ b/packages/mcp/test/trusted-launcher.test.mjs
@@ -0,0 +1,281 @@
+/**
+ * The plugin must never run an MCP server that came out of the user's project
+ * (WALM-640).
+ *
+ * The plugin used to start the server with `npx -y @mysten-incubation/memwal-mcp@`.
+ * npx resolves that name against the working directory the MCP client was started
+ * in — the project the user has open — so a project carrying an installed package
+ * of the same name that *claims the pinned version* won, offline. The version pin
+ * is no defence: the fake package just says it is that version.
+ *
+ * Every test here plants exactly that: a fake `@mysten-incubation/memwal-mcp` in a
+ * temp project's node_modules, with the real pinned version in its manifest and a
+ * bin that prints LOCAL_PACKAGE_EXECUTED, plus the `node_modules/.bin` shim npx
+ * would have found. The launcher must resolve and run the trusted absolute path
+ * instead — and when the trusted install cannot be produced, it must fail rather
+ * than fall back to the name.
+ */
+import { test } from "node:test";
+import assert from "node:assert/strict";
+import { spawnSync } from "node:child_process";
+import { chmodSync, mkdirSync, mkdtempSync, readFileSync, rmSync, writeFileSync } from "node:fs";
+import { tmpdir } from "node:os";
+import { dirname, join, resolve } from "node:path";
+import { fileURLToPath } from "node:url";
+
+import {
+ MCP_PACKAGE_NAME,
+ expectedEntryPath,
+ installDir,
+ pinnedVersion,
+ resolveInstalledEntry,
+ runtimeRoot,
+} from "../plugin/scripts/lib/mcp-launch.mjs";
+
+const __dirname = dirname(fileURLToPath(import.meta.url));
+const PLUGIN_DIR = resolve(__dirname, "../plugin");
+const LAUNCHER = join(PLUGIN_DIR, "scripts", "launch_mcp.mjs");
+const PIN = pinnedVersion();
+
+const LOCAL_MARKER = "LOCAL_PACKAGE_EXECUTED";
+const TRUSTED_MARKER = "TRUSTED_ENTRY_EXECUTED";
+
+/**
+ * Write a package that looks exactly like the real one to any resolver: same name,
+ * same `bin` layout, and whatever version the caller wants it to claim.
+ */
+function plantPackage(nodeModulesParent, { version, marker }) {
+ const packageDir = join(nodeModulesParent, "node_modules", ...MCP_PACKAGE_NAME.split("/"));
+ const binDir = join(packageDir, "dist", "bin");
+ mkdirSync(binDir, { recursive: true });
+ writeFileSync(
+ join(packageDir, "package.json"),
+ JSON.stringify({
+ name: MCP_PACKAGE_NAME,
+ version,
+ bin: { "memwal-mcp": "dist/bin/memwal-mcp.js" },
+ }),
+ );
+ const entry = join(binDir, "memwal-mcp.js");
+ writeFileSync(
+ entry,
+ `console.log(${JSON.stringify(marker)});\n` +
+ `console.log("ARGV:" + process.argv.slice(2).join(" "));\n`,
+ );
+ return entry;
+}
+
+/** The PATH-relative shim npx would have reached for inside the project. */
+function plantBinShim(projectDir, marker) {
+ const binDir = join(projectDir, "node_modules", ".bin");
+ mkdirSync(binDir, { recursive: true });
+ const shim = join(binDir, "memwal-mcp");
+ writeFileSync(shim, `#!/bin/sh\necho ${marker}\n`);
+ chmodSync(shim, 0o755);
+ return shim;
+}
+
+/**
+ * A project a user might have open, carrying a fake package that claims the exact
+ * version the plugin pins.
+ */
+function makeHostileProject(t, { version = PIN } = {}) {
+ const dir = mkdtempSync(join(tmpdir(), "memwal-hostile-project-"));
+ t.after(() => rmSync(dir, { recursive: true, force: true }));
+ const entry = plantPackage(dir, { version, marker: LOCAL_MARKER });
+ plantBinShim(dir, LOCAL_MARKER);
+ return { dir, entry };
+}
+
+/** A trusted runtime root with the pinned version already installed in it. */
+function makeTrustedRuntime(t, { version = PIN, populate = true } = {}) {
+ const root = mkdtempSync(join(tmpdir(), "memwal-runtime-"));
+ t.after(() => rmSync(root, { recursive: true, force: true }));
+ if (!populate) return { root, entry: null };
+ const target = installDir(version, root);
+ mkdirSync(target, { recursive: true });
+ const entry = plantPackage(target, { version, marker: TRUSTED_MARKER });
+ return { root, entry };
+}
+
+function runLauncher(args, { cwd, env }) {
+ return spawnSync(process.execPath, [LAUNCHER, ...args], {
+ cwd,
+ encoding: "utf8",
+ env: { ...process.env, ...env },
+ });
+}
+
+test("the planted package claims the exact pinned version", (t) => {
+ const project = makeHostileProject(t);
+ const manifest = JSON.parse(
+ readFileSync(
+ join(project.dir, "node_modules", ...MCP_PACKAGE_NAME.split("/"), "package.json"),
+ "utf8",
+ ),
+ );
+ // If this drifts the rest of the file stops testing the reported attack.
+ assert.equal(manifest.version, PIN);
+ assert.equal(manifest.name, MCP_PACKAGE_NAME);
+});
+
+test("resolution ignores the project entirely and lands in the trusted directory", (t) => {
+ const project = makeHostileProject(t);
+ const trusted = makeTrustedRuntime(t);
+
+ const previousCwd = process.cwd();
+ t.after(() => process.chdir(previousCwd));
+ process.chdir(project.dir);
+
+ const resolved = resolveInstalledEntry(installDir(PIN, trusted.root), PIN);
+ assert.equal(resolved, trusted.entry);
+ assert.equal(resolved, expectedEntryPath({ version: PIN, root: trusted.root }));
+ assert.ok(resolved.startsWith(trusted.root), `${resolved} is not under ${trusted.root}`);
+ assert.ok(!resolved.includes(project.dir), `${resolved} points into the project`);
+});
+
+test("--print-entry from inside the hostile project prints the trusted path", (t) => {
+ const project = makeHostileProject(t);
+ const trusted = makeTrustedRuntime(t);
+
+ const result = runLauncher(["--print-entry"], {
+ cwd: project.dir,
+ env: { MEMWAL_MCP_RUNTIME_DIR: trusted.root },
+ });
+
+ assert.equal(result.status, 0, result.stderr);
+ const printed = result.stdout.trim();
+ assert.equal(printed, trusted.entry);
+ assert.notEqual(printed, project.entry);
+ assert.ok(!printed.includes(project.dir), `${printed} points into the project`);
+});
+
+test("launching from the hostile project runs the trusted entry, not the local one", (t) => {
+ const project = makeHostileProject(t);
+ const trusted = makeTrustedRuntime(t);
+
+ const result = runLauncher([], {
+ cwd: project.dir,
+ env: { MEMWAL_MCP_RUNTIME_DIR: trusted.root },
+ });
+
+ assert.equal(result.status, 0, result.stderr);
+ assert.match(result.stdout, new RegExp(TRUSTED_MARKER));
+ assert.doesNotMatch(result.stdout, new RegExp(LOCAL_MARKER));
+});
+
+test("server flags and env are forwarded to the trusted entry unchanged", (t) => {
+ const project = makeHostileProject(t);
+ const trusted = makeTrustedRuntime(t);
+
+ const result = runLauncher(["--dev", "--namespace", "work"], {
+ cwd: project.dir,
+ env: { MEMWAL_MCP_RUNTIME_DIR: trusted.root },
+ });
+
+ assert.equal(result.status, 0, result.stderr);
+ assert.match(result.stdout, /ARGV:--dev --namespace work/);
+});
+
+test("an install under a different version is not accepted for the pin", (t) => {
+ const project = makeHostileProject(t);
+ // Trusted directory holds a stale build; the pinned directory is empty.
+ const trusted = makeTrustedRuntime(t, { version: "0.0.0-stale" });
+
+ assert.equal(resolveInstalledEntry(installDir(PIN, trusted.root), PIN), null);
+ assert.equal(
+ resolveInstalledEntry(installDir("0.0.0-stale", trusted.root), PIN),
+ null,
+ "a directory whose manifest disagrees with the pin must not be used",
+ );
+ assert.ok(!String(trusted.entry).includes(project.dir));
+});
+
+test("with no trusted install and no installer, the launcher fails instead of falling back", (t) => {
+ const project = makeHostileProject(t);
+ const trusted = makeTrustedRuntime(t, { populate: false });
+
+ // An empty PATH makes the `npm` lookup fail the way an offline or broken
+ // toolchain would. The launcher must surface that, never reach for the name.
+ const result = runLauncher([], {
+ cwd: project.dir,
+ env: { MEMWAL_MCP_RUNTIME_DIR: trusted.root, PATH: "" },
+ });
+
+ assert.notEqual(result.status, 0, "launcher must not succeed without a trusted install");
+ assert.doesNotMatch(result.stdout, new RegExp(LOCAL_MARKER));
+ assert.doesNotMatch(result.stderr, new RegExp(LOCAL_MARKER));
+ assert.match(result.stderr, /refusing to fall back/);
+});
+
+test("a relative MEMWAL_MCP_RUNTIME_DIR is rejected, not resolved against the project", (t) => {
+ const project = makeHostileProject(t);
+
+ const result = runLauncher(["--print-entry"], {
+ cwd: project.dir,
+ env: { MEMWAL_MCP_RUNTIME_DIR: "node_modules" },
+ });
+
+ assert.notEqual(result.status, 0);
+ assert.match(result.stderr, /absolute path/);
+ assert.doesNotMatch(result.stdout, new RegExp(LOCAL_MARKER));
+});
+
+test("the default trusted root is ~/.memwal/runtime", (t) => {
+ const project = makeHostileProject(t);
+ const home = mkdtempSync(join(tmpdir(), "memwal-home-"));
+ t.after(() => rmSync(home, { recursive: true, force: true }));
+
+ const target = join(home, ".memwal", "runtime", `memwal-mcp@${PIN}`);
+ mkdirSync(target, { recursive: true });
+ const entry = plantPackage(target, { version: PIN, marker: TRUSTED_MARKER });
+
+ // USERPROFILE alongside HOME: os.homedir() reads USERPROFILE on Windows and
+ // would otherwise escape the sandbox into the real home.
+ const result = runLauncher(["--print-entry"], {
+ cwd: project.dir,
+ env: { HOME: home, USERPROFILE: home, MEMWAL_MCP_RUNTIME_DIR: "" },
+ });
+
+ assert.equal(result.status, 0, result.stderr);
+ assert.equal(result.stdout.trim(), entry);
+});
+
+test("runtimeRoot honours an absolute override and otherwise sits under the home dir", () => {
+ const previous = process.env.MEMWAL_MCP_RUNTIME_DIR;
+ try {
+ const absolute = join(tmpdir(), "memwal-runtime-override");
+ process.env.MEMWAL_MCP_RUNTIME_DIR = absolute;
+ assert.equal(runtimeRoot(), absolute);
+
+ delete process.env.MEMWAL_MCP_RUNTIME_DIR;
+ assert.match(runtimeRoot(), /[\\/]\.memwal[\\/]runtime$/);
+ } finally {
+ if (previous === undefined) delete process.env.MEMWAL_MCP_RUNTIME_DIR;
+ else process.env.MEMWAL_MCP_RUNTIME_DIR = previous;
+ }
+});
+
+test("no plugin launch manifest resolves the server through npx", () => {
+ const manifests = [
+ [".mcp.json", "${CLAUDE_PLUGIN_ROOT}"],
+ [".cursor-mcp.json", "${CURSOR_PLUGIN_ROOT}"],
+ [".codex-mcp.json", "${PLUGIN_ROOT}"],
+ ];
+ for (const [name, root] of manifests) {
+ const server = JSON.parse(readFileSync(join(PLUGIN_DIR, name), "utf8")).mcpServers.memwal;
+ assert.equal(server.command, "node", `${name} must not launch through npx`);
+ assert.deepEqual(server.args, [`${root}/scripts/launch_mcp.mjs`], name);
+ }
+
+ const installer = readFileSync(
+ join(PLUGIN_DIR, "scripts", "install_codex_hooks.mjs"),
+ "utf8",
+ );
+ assert.doesNotMatch(
+ installer,
+ /command\s*=\s*\\?"npx/,
+ "the Codex fallback installer must register the launcher, not npx",
+ );
+ assert.match(installer, /launch_mcp\.mjs/);
+});
diff --git a/scripts/verify-manual-sdk-release.mjs b/scripts/verify-manual-sdk-release.mjs
index 66f89a563..8d6bd3306 100644
--- a/scripts/verify-manual-sdk-release.mjs
+++ b/scripts/verify-manual-sdk-release.mjs
@@ -1,6 +1,6 @@
#!/usr/bin/env node
-import { readFileSync } from "node:fs";
+import { existsSync, readFileSync } from "node:fs";
const releases = [
{
@@ -63,37 +63,46 @@ for (const release of releases) {
console.log(`${release.name} ${release.version}: manifests and changelogs synchronized`);
}
+// The plugin no longer launches the server through `npx @`: npx resolves
+// the name against the project the MCP client is started in, so a package planted
+// there could answer to the pinned spec (WALM-640). Every launch site must run the
+// plugin's launcher, which installs the pin under ~/.memwal/runtime and runs that
+// absolute entry point. The pin itself is plugin/plugin.json's version, already
+// checked against packages/mcp/package.json above.
const mcpVersion = JSON.parse(readFileSync("packages/mcp/package.json", "utf8")).version;
-const expectedPluginArgs = ["-y", `@mysten-incubation/memwal-mcp@${mcpVersion}`];
-for (const pluginPath of [
- "packages/mcp/plugin/.mcp.json",
- "packages/mcp/plugin/.cursor-mcp.json",
- "packages/mcp/plugin/.codex-mcp.json",
+const LAUNCHER = "scripts/launch_mcp.mjs";
+for (const [pluginPath, rootPlaceholder] of [
+ ["packages/mcp/plugin/.mcp.json", "${CLAUDE_PLUGIN_ROOT}"],
+ ["packages/mcp/plugin/.cursor-mcp.json", "${CURSOR_PLUGIN_ROOT}"],
+ ["packages/mcp/plugin/.codex-mcp.json", "${PLUGIN_ROOT}"],
]) {
- const actual = JSON.parse(readFileSync(pluginPath, "utf8")).mcpServers.memwal.args;
- if (JSON.stringify(actual) !== JSON.stringify(expectedPluginArgs)) {
+ const server = JSON.parse(readFileSync(pluginPath, "utf8")).mcpServers.memwal;
+ const expected = { command: "node", args: [`${rootPlaceholder}/${LAUNCHER}`] };
+ if (
+ server.command !== expected.command ||
+ JSON.stringify(server.args) !== JSON.stringify(expected.args)
+ ) {
throw new Error(
- `${pluginPath}: expected ${JSON.stringify(expectedPluginArgs)}, received ${JSON.stringify(actual)}`,
+ `${pluginPath}: expected ${JSON.stringify(expected)}, received ${JSON.stringify({ command: server.command, args: server.args })}`,
);
}
}
const installerPath = "packages/mcp/plugin/scripts/install_codex_hooks.mjs";
const installer = readFileSync(installerPath, "utf8");
-const expectedPin = expectedPluginArgs[1];
-if (installer.includes('["-y", "@mysten-incubation/memwal-mcp"]')) {
+if (/command\s*=\s*\\?"npx/.test(installer)) {
throw new Error(
- `${installerPath}: expected ${JSON.stringify(expectedPluginArgs)}, received ${JSON.stringify(["-y", "@mysten-incubation/memwal-mcp"])}`,
+ `${installerPath}: registers the MCP server through npx; it must register the ` +
+ `absolute path to ${LAUNCHER} (WALM-640)`,
);
}
-if (
- !installer.includes(expectedPin) &&
- !installer.includes("@mysten-incubation/memwal-mcp@${")
-) {
- throw new Error(
- `${installerPath}: expected ${JSON.stringify(expectedPluginArgs)}, received missing version pin`,
- );
+if (!installer.includes("launch_mcp.mjs")) {
+ throw new Error(`${installerPath}: does not register ${LAUNCHER}`);
+}
+const launcherPath = `packages/mcp/plugin/${LAUNCHER}`;
+if (!existsSync(launcherPath)) {
+ throw new Error(`${launcherPath}: missing, but every launch site points at it`);
}
-console.log(`MCP package ${mcpVersion}: plugin npx args pin ${expectedPin}`);
+console.log(`MCP package ${mcpVersion}: every launch site runs ${LAUNCHER} by absolute path`);
function readVersion(content, kind) {
if (kind === "version") return JSON.parse(content).version;
From be50acf0f178044366989c06d2057eff53b6d2d1 Mon Sep 17 00:00:00 2001
From: Harry Phan
Date: Thu, 17 Sep 2026 22:20:27 +0700
Subject: [PATCH 067/132] fix(mcp): filter credentials from automatic memory
and require opt-in
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit
Automatic-save guidance asked for the complete statement and said nothing
about secrets, so a preference stated next to a password was forwarded
whole. Reproduced through the real MCP server with a mocked SDK:
remember, remember_bulk and analyze all passed the full text through
unchanged. Consent was inconsistent too — setup asked before editing
global instructions, but nothing asked before the package started saving
facts nobody requested.
Three parts:
1. One shared source for the secret-exclusion and do-not-save rules,
carried verbatim by the initialize instructions, the tool descriptions
and the plugin hooks. The three packages have no workspace link, so
the block is duplicated byte-for-byte across memory-policy.{ts,mjs}
and pinned by sync tests on both sides: edit one copy and the suite
fails until the others match.
2. A programmatic screen in front of every write. sanitizeFact runs
before any text reaches the SDK on all three write paths, because
Walrus storage is append-only and a stored secret cannot be deleted.
A message mixing a preference with a credential keeps the preference
and loses only the credential span; a text that is nothing but a
secret, or that the user said not to save, is not forwarded at all.
The caller is told which kinds were removed — never the values, which
are not logged, echoed or returned. Detection is shape-based rather
than entropy-based on purpose: blob ids, Sui object ids, commit SHAs
and digests are exactly what a generic high-entropy rule would eat.
The trade-off is documented at the top of redaction.ts.
3. Automatic saving is opt-in, default off. `memwal-mcp auto-save on`
persists the choice to settings.json beside credentials.json, and
MEMWAL_AUTO_SAVE overrides it per process. The state lives on disk
because the hooks are spawned by the client and inherit none of the
MCP server's configuration. With it off, the instructions, both hook
rubrics and the post-tool nudge switch to a save-only-what-you-are-
asked variant. Explicit tool calls and recall are never gated.
Tests: redactor unit coverage (both directions — secrets removed,
legitimate identifiers untouched), write-path coverage asserting what
each handler forwards to the SDK rather than what it says afterwards,
opt-in resolution on both implementations, policy-block byte equality,
and the hooks in both states. TESTING.md gains the manual scenarios for
the model-behaviour half, which cannot be automated.
Fixes WALM-642
---
docs/mcp/overview.md | 10 +-
docs/mcp/reference.md | 4 +-
packages/mcp/AUTO-MEMORY.md | 71 +++-
packages/mcp/README.md | 64 ++-
packages/mcp/TESTING.md | 38 ++
packages/mcp/plugin/scripts/lib/auto-save.mjs | 101 +++++
.../plugin/scripts/lib/decision-rubric.mjs | 67 +++-
.../mcp/plugin/scripts/lib/memory-policy.mjs | 80 ++++
packages/mcp/plugin/scripts/on_post_tool.mjs | 14 +-
.../mcp/plugin/scripts/on_session_start.mjs | 25 +-
.../mcp/plugin/scripts/on_user_prompt.mjs | 16 +-
packages/mcp/src/auth-required.ts | 27 +-
packages/mcp/src/auto-save.ts | 146 +++++++
packages/mcp/src/bridge.ts | 8 +-
packages/mcp/src/index.ts | 67 +++-
packages/mcp/src/instructions.ts | 93 ++++-
packages/mcp/src/memory-policy.ts | 82 ++++
packages/mcp/test/auto-save-optin.test.mjs | 300 ++++++++++++++
packages/mcp/test/memory-policy.test.mjs | 172 +++++++++
packages/mcp/test/user-prompt-hook.test.mjs | 19 +-
.../mcp/__tests__/secret-redaction.test.ts | 264 +++++++++++++
.../__tests__/write-path-redaction.test.ts | 285 ++++++++++++++
services/server/scripts/mcp/server.ts | 15 +
services/server/scripts/mcp/tools/analyze.ts | 42 +-
.../server/scripts/mcp/tools/memory-policy.ts | 80 ++++
.../server/scripts/mcp/tools/redaction.ts | 365 ++++++++++++++++++
.../server/scripts/mcp/tools/remember-bulk.ts | 80 +++-
services/server/scripts/mcp/tools/remember.ts | 38 +-
28 files changed, 2511 insertions(+), 62 deletions(-)
create mode 100644 packages/mcp/plugin/scripts/lib/auto-save.mjs
create mode 100644 packages/mcp/plugin/scripts/lib/memory-policy.mjs
create mode 100644 packages/mcp/src/auto-save.ts
create mode 100644 packages/mcp/src/memory-policy.ts
create mode 100644 packages/mcp/test/auto-save-optin.test.mjs
create mode 100644 packages/mcp/test/memory-policy.test.mjs
create mode 100644 services/server/scripts/mcp/__tests__/secret-redaction.test.ts
create mode 100644 services/server/scripts/mcp/__tests__/write-path-redaction.test.ts
create mode 100644 services/server/scripts/mcp/tools/memory-policy.ts
create mode 100644 services/server/scripts/mcp/tools/redaction.ts
diff --git a/docs/mcp/overview.md b/docs/mcp/overview.md
index bf354de34..b3b98ed94 100644
--- a/docs/mcp/overview.md
+++ b/docs/mcp/overview.md
@@ -43,9 +43,17 @@ There are two ways to use MemWal. The difference is whether you also get the **l
| MemWal MCP: memory tools (`memwal_remember`, `memwal_recall`, …) | ✓ | ✓ |
| Lifecycle hooks: automatic recall/save reminders | ✓ | ✗ |
-- **Plugin** bundles the MCP server **and** lifecycle hooks. The `SessionStart` hook tells the agent to prefer the `memwal_*` tools over any built-in or local memory feature, and when to save without being asked. Automatic memory works with no further instructions from you. Available on **Claude Code**, **Codex**, **Antigravity**, and **Cursor**.
+- **Plugin** bundles the MCP server **and** lifecycle hooks. The `SessionStart` hook tells the agent to prefer the `memwal_*` tools over any built-in or local memory feature, and when to save without being asked. Available on **Claude Code**, **Codex**, **Antigravity**, and **Cursor**.
- **MCP-only** gives the agent the memory tools on **every** MCP client. The tool descriptions encourage proactive use, so agents often do save and recall on their own. Treat that as best-effort: it varies by client and model, and on a client that ships its own memory feature the built-in one commonly wins.
+Saving **without being asked is opt-in on both paths** and off until you turn it
+on with `memwal-mcp auto-save on` (or `MEMWAL_AUTO_SAVE=1` in your client's
+`env` block). Recall, and anything you explicitly ask to be remembered, work
+either way. Credentials — passwords, API keys, tokens, private keys, seed
+phrases, authorization headers, and URLs with an embedded `user:password` — are
+excluded in both modes and stripped before a memory is written, because Walrus
+storage is append-only and a stored secret cannot be deleted.
+
Prefer the plugin wherever you can install it. It is the tested path for reliable automatic save and recall, and it needs no extra instructions from you. On MCP-only clients, paste the client's instruction block (see [Claude Desktop](/mcp/claude-desktop#add-memory-instructions)) to get closer to the same behavior.
diff --git a/docs/mcp/reference.md b/docs/mcp/reference.md
index 7599d2d01..d7ee6e97e 100644
--- a/docs/mcp/reference.md
+++ b/docs/mcp/reference.md
@@ -53,7 +53,9 @@ This is why many first-run sessions show `memwal_login` before the other tools a
### memwal_remember
-Save a durable fact to the user's Walrus Memory. The agent calls this **proactively** when the user states a preference, decision, constraint, correction, identity detail, or recurring workflow, not only when they explicitly ask. Skip one-off tasks, the current file or bug, and small talk. Pass the full statement; do not summarize.
+Save a durable fact to the user's Walrus Memory. The agent calls this **proactively** when the user states a preference, decision, constraint, correction, identity detail, or recurring workflow, not only when they explicitly ask — provided automatic memory is on (`memwal-mcp auto-save on`, or `MEMWAL_AUTO_SAVE=1`; it is off by default). Skip one-off tasks, the current file or bug, and small talk. Pass the full statement; do not summarize.
+
+Credentials are never stored. Passwords, API keys, access and refresh tokens, private keys, seed phrases, authorization headers, session cookies, and URLs with an embedded `user:password` are stripped from the text before the write, on this tool, `memwal_remember_bulk` and `memwal_analyze` alike. A message that mixes a preference with a credential keeps the preference and loses only the credential; the reply names which kinds were removed.
| **Parameter** | **Type** | **Required** | **Description** |
| --- | --- | --- | --- |
diff --git a/packages/mcp/AUTO-MEMORY.md b/packages/mcp/AUTO-MEMORY.md
index 133642c9b..0950f8d38 100644
--- a/packages/mcp/AUTO-MEMORY.md
+++ b/packages/mcp/AUTO-MEMORY.md
@@ -14,6 +14,65 @@ package. `memwal_remember` literally said *"Call ONLY when the user explicitly
asks… agents should not call this proactively."* And `rememberBulk` (in the SDK)
was never exposed as a tool.
+## Opt-in and secret filtering (WALM-642)
+
+Automatic saving is **off by default**. What is gated is narrow: telling the
+model to save something the user did **not** ask it to save. A direct request
+("remember that ...") works either way, and recall is never gated.
+
+**Turning it on**
+
+```sh
+memwal-mcp auto-save on # persists {"autoSave": true} to settings.json
+memwal-mcp auto-save off
+memwal-mcp auto-save # report the current state and where it came from
+```
+
+`MEMWAL_AUTO_SAVE=1` in an MCP client's `env` block does the same for one
+server process and overrides the file.
+
+**Where the state lives** — `settings.json`, next to `credentials.json`, so it
+inherits `credsPath()` resolution: `MEMWAL_CREDS_DIR` override, else the
+nearest project-local `.memwal/`, else `~/.memwal/`. It has to be on disk
+rather than passed as configuration because the **hooks are spawned by the
+client, not by this package**: they inherit the MCP server's `env` block from
+nothing at all. `src/auto-save.ts` and `plugin/scripts/lib/auto-save.mjs` are
+two implementations of the same resolution, pinned against each other by
+`test/auto-save-optin.test.mjs`.
+
+**What changes when it is off** — the `instructions` field, the SessionStart
+rubric, the UserPromptSubmit rubric and the PostToolUse nudge all switch to a
+save-only-what-you-are-asked variant. Nothing is disabled; the guidance that
+drives an unasked-for save is simply not injected.
+
+**One source for the rules** — the secret-exclusion and do-not-save text lives
+in a single block duplicated byte-for-byte across three files that cannot
+import each other:
+
+| Copy | Feeds |
+|---|---|
+| `packages/mcp/src/memory-policy.ts` | `instructions`, cold-start `tools/list` |
+| `packages/mcp/plugin/scripts/lib/memory-policy.mjs` | the three lifecycle hooks |
+| `services/server/scripts/mcp/tools/memory-policy.ts` | live tool descriptions, sidecar `instructions` |
+
+`test/memory-policy.test.mjs` extracts the marked block from each file and
+compares the bytes, so editing one copy fails the suite until the others match.
+
+**The programmatic backstop** — model-facing rules are not enforcement, and
+Walrus storage is append-only: a secret that lands cannot be deleted. So
+`services/server/scripts/mcp/tools/redaction.ts` screens every write **before**
+the text reaches the SDK, on all three write paths (`memwal_remember`,
+`memwal_remember_bulk`, `memwal_analyze`). A mixed message keeps its fact and
+loses only the credential span — the ticket's "save safe facts without
+neighboring credentials" — and a text that is nothing but a secret, or that the
+user said not to save, is not forwarded at all. The caller is told which *kinds*
+were removed; the value is never logged, echoed, or returned.
+
+Detection is shape-based, not entropy-based, on purpose: MemWal's own durable
+facts (blob ids, Sui object ids, git SHAs, digests) are exactly what a generic
+high-entropy rule would eat. The trade-off is written out at the top of
+`redaction.ts`.
+
## Architecture — three layers
1. **Agentic tool descriptions + `memwal_remember_bulk`** — `services/server/scripts/mcp/tools/`.
@@ -30,7 +89,8 @@ was never exposed as a tool.
| Dimension | Before | After |
|---|---|---|
-| Save trigger | "ONLY when user explicitly asks; don't be proactive" | "Save proactively whenever you learn a durable fact" |
+| Save trigger | "ONLY when user explicitly asks; don't be proactive" | "Save proactively whenever you learn a durable fact" — **opt-in since WALM-642; off by default** |
+| Secret handling | none: a preference next to a password was forwarded whole | shared exclusion rules on all three surfaces + a redactor in front of every write |
| Bulk save | not exposed | `memwal_remember_bulk` (wraps SDK `rememberBulkAndWait`, ≤20) |
| Recall trigger | neutral; agent rarely called it unprompted | "Recall proactively at task start / when the user references past work" |
| Reinforcement | none | UserPromptSubmit + PostToolUse hooks (Claude Code + Codex) |
@@ -51,7 +111,11 @@ was never exposed as a tool.
## Decisions (chosen)
-- **Append-only** — no `forget`/`update` tools (relayer dedups embeddings).
+- **Append-only** — no `forget`/`update` tools (relayer dedups embeddings). This
+ is also why WALM-642's credential check runs *before* the write: there is no
+ delete to fall back on.
+- **Automatic saving is opt-in, default off** (WALM-642) — explicit tool use is
+ never gated, and neither is recall.
- **Global `default` namespace** — `MEMWAL_NAMESPACE` overrides for per-project scope.
- **Agent decision rubric** — UserPromptSubmit does not regex-classify remember vs
recall. The agent has the conversation and understands any language or spelling.
@@ -63,6 +127,9 @@ was never exposed as a tool.
## File map
- `services/server/scripts/mcp/tools/{remember,recall,analyze,restore}.ts` — agentic descriptions
+- `services/server/scripts/mcp/tools/redaction.ts` — pre-forward credential screen (WALM-642)
+- `{packages/mcp/src,packages/mcp/plugin/scripts/lib,services/server/scripts/mcp/tools}/memory-policy.*` — the shared rules block, three byte-identical copies
+- `packages/mcp/src/auto-save.ts` + `packages/mcp/plugin/scripts/lib/auto-save.mjs` — the opt-in resolver, server side and hook side
- `services/server/scripts/mcp/tools/remember-bulk.ts` + `index.ts` — new bulk tool
- `packages/mcp/plugin/` — plugin manifest, `.mcp.json`, hooks, Node scripts, Codex installer
- `.claude-plugin/marketplace.json` (repo root) — Claude Code marketplace entry (local source `./packages/mcp/plugin`)
diff --git a/packages/mcp/README.md b/packages/mcp/README.md
index 230f96c48..22b1b6d62 100644
--- a/packages/mcp/README.md
+++ b/packages/mcp/README.md
@@ -38,6 +38,7 @@ The command opens your browser, asks you to connect your Sui wallet, and saves c
```sh
memwal-mcp
memwal-mcp login
+memwal-mcp auto-save on|off
memwal-mcp --logout
memwal-mcp --help
```
@@ -52,9 +53,68 @@ Use CLI flags or environment variables to override the default Walrus Memory end
| `--web-url ` | `MEMWAL_WEB_URL` | Override the web app URL used during login. |
| `--label ` | `MEMWAL_CLIENT_LABEL` | Friendly delegate-key label shown in Walrus Memory. |
| `--namespace ` (alias `--ns`) | `MEMWAL_NAMESPACE` | Default memory namespace applied when the agent omits one. |
+| `auto-save on\|off` | `MEMWAL_AUTO_SAVE` | Let the agent save durable facts unprompted. Off by default; see [Automatic Memory](#automatic-memory-opt-in). |
Enable verbose stderr logging with `MEMWAL_MCP_DEBUG=1`.
+## Automatic Memory (opt-in)
+
+By default the agent saves **only what you ask it to save**. Turning automatic
+memory on lets it save durable facts — preferences, decisions, constraints,
+recurring workflows — without being asked:
+
+```sh
+npx -y @mysten-incubation/memwal-mcp auto-save on
+npx -y @mysten-incubation/memwal-mcp auto-save off
+npx -y @mysten-incubation/memwal-mcp auto-save # report the current setting
+```
+
+The choice is stored as `{"autoSave": true}` in `settings.json` next to your
+credentials file, so it follows the same project-local-beats-global resolution.
+To pin one MCP client instead, set the environment variable — it overrides the
+file:
+
+```json
+{
+ "mcpServers": {
+ "memwal": {
+ "command": "npx",
+ "args": ["-y", "@mysten-incubation/memwal-mcp"],
+ "env": { "MEMWAL_AUTO_SAVE": "1" }
+ }
+ }
+}
+```
+
+Recall is never gated, and neither is an explicit "remember this" — the setting
+only decides whether the agent saves things you did not ask it to save.
+
+### What is never saved
+
+Walrus storage is append-only and encrypted: a memory that lands **cannot be
+edited or deleted**. So credentials are excluded in both modes, by the same
+rules stated in the server instructions, the tool descriptions and the plugin
+hooks — and enforced by a check that runs before any text is sent:
+
+- passwords, API keys, access and refresh tokens, private keys, seed and
+ recovery phrases, authorization headers, session cookies, and connection
+ strings or URLs with an embedded `user:password`;
+- anything you say not to save ("don't save this", "off the record");
+- pasted third-party content — a fenced block, a quoted passage — which is not
+ a fact about you.
+
+When a message mixes a preference with a credential, the **preference is kept**
+and only the credential is removed: "I prefer dark mode, db is
+`postgres://admin:hunter2@db.internal/app`" is stored with the password gone and
+the host intact. The agent is told which kinds were removed; the secret itself
+is never stored, logged, or echoed back.
+
+Detection targets specific credential shapes rather than "looks random", so
+identifiers you *do* want remembered — blob ids, Sui object ids, commit SHAs,
+digests — pass through untouched. The trade-off is that a secret in no
+recognisable shape can still slip past the check, which is why the model-facing
+rules exist alongside it.
+
## Default Namespace
By default the MCP tool schemas expose an optional `namespace` argument and the
@@ -146,7 +206,9 @@ You can also pass explicit URLs:
## Credential Storage
-Credentials are stored locally in `~/.memwal/credentials.json`. To remove them:
+Credentials are stored locally in `~/.memwal/credentials.json`, and the
+automatic-memory setting in `settings.json` beside it. To remove the
+credentials:
```sh
npx -y @mysten-incubation/memwal-mcp --logout
diff --git a/packages/mcp/TESTING.md b/packages/mcp/TESTING.md
index d4f4e9dda..462bf42e5 100644
--- a/packages/mcp/TESTING.md
+++ b/packages/mcp/TESTING.md
@@ -23,6 +23,12 @@ Step-by-step plan to test the auto-memory work (agentic tools + bulk + health +
```
- [ ] Credentials present and on testnet: `~/.memwal/credentials.json` (relayerUrl = `http://127.0.0.1:8000`).
- [ ] MCP package built: `ls packages/mcp/dist/bin/memwal-mcp.js`.
+- [ ] **Automatic saving turned on** — it is OFF by default since WALM-642, and
+ every "agent saves on its own" step below depends on it:
+ ```bash
+ node packages/mcp/dist/bin/memwal-mcp.js auto-save on # → "Automatic memory is ON"
+ ```
+ Leave it off first if you want to confirm the opt-in itself (§1.6).
**How to tell MemWal vs the editor's built-in memory** (use everywhere below):
- ✅ **MemWal** → the step shows `Called memwal…` with a **`blob_id`** + `namespace`.
@@ -77,6 +83,38 @@ Notes:
---
+## 1.6 Automatic-save opt-in and secret filtering (WALM-642)
+
+The redactor and the opt-in resolver have automated coverage
+(`services/server/scripts/mcp/__tests__/{secret-redaction,write-path-redaction}.test.ts`,
+`packages/mcp/test/{memory-policy,auto-save-optin}.test.mjs`). What is NOT
+automatable is the half the ticket calls "model behavior": whether a real model,
+reading the injected rules, actually declines to save a secret it was never
+programmatically stopped from sending. Run these by hand in a real client.
+
+**Opt-in (start with `auto-save off`)**
+
+| # | Action / prompt | Expect | OK? |
+|---|---|---|---|
+| 1 | `auto-save off`, restart the client, then `I prefer pnpm and TypeScript strict mode.` | Agent does **not** call `memwal_remember` on its own | [ ] |
+| 2 | Same session: `remember that I prefer pnpm` | Agent **does** call `memwal_remember` — an explicit ask is never gated | [ ] |
+| 3 | Same session: `what do you remember about my preferences?` | Agent calls `memwal_recall` — recall is never gated | [ ] |
+| 4 | `auto-save on`, restart, repeat #1 | Agent calls `memwal_remember` unprompted again | [ ] |
+
+**Secret filtering (with `auto-save on`)**
+
+| # | Action / prompt | Expect | OK? |
+|---|---|---|---|
+| 5 | `I prefer dark mode, and the staging db is postgres://admin:hunter2@db.internal:5432/app` | A memory IS saved, and it contains the preference and `db.internal:5432/app` but **not** `hunter2`. The tool reply names `url-credentials`. Confirm with a recall in a new chat. | [ ] |
+| 6 | Same, but phrased so the model saves it as several facts | `memwal_remember_bulk` — same result per entry; a bare-secret entry is reported as `NOT SAVED (1)` with its position | [ ] |
+| 7 | Paste a transcript containing a preference and `ghp_…`, ask to analyse it | `memwal_analyze` — the extracted facts never contain the token | [ ] |
+| 8 | `My bank PIN is 4821 — don't save this.` | Nothing is saved; the agent says so | [ ] |
+| 9 | Paste a fenced log/code block and say "save this" | Not saved as a fact about you; the agent asks you to restate it | [ ] |
+| 10 | **Model behavior:** state a secret in a shape the redactor does not match (e.g. `my door code is seven four nine two`) and see whether the model saves it | The rules say not to; a save here is a model failure, not a code failure — record it, it is the residual risk the redactor cannot close | [ ] |
+| 11 | Check `~/.memwal/settings.json` and the relayer logs after #5-#10 | No secret appears in either — the redactor never logs what it removed | [ ] |
+
+---
+
## 2. Per-editor tests
### 2A. Claude Code — Plugin (hooks + "prefer MemWal" steer)
diff --git a/packages/mcp/plugin/scripts/lib/auto-save.mjs b/packages/mcp/plugin/scripts/lib/auto-save.mjs
new file mode 100644
index 000000000..23ce4eb07
--- /dev/null
+++ b/packages/mcp/plugin/scripts/lib/auto-save.mjs
@@ -0,0 +1,101 @@
+/**
+ * Automatic-save opt-in, hook-side (WALM-642).
+ *
+ * The mirror of `packages/mcp/src/auto-save.ts`, in plain ESM with no
+ * dependencies and no network, because a hook is a `.mjs` file the client
+ * spawns directly — it never loads this package's compiled `dist/`, and it
+ * inherits none of the MCP server's configuration (a hook is spawned by Claude
+ * Code or Codex, the MCP server by its own `command`/`env` block). A shared
+ * file on disk is the only thing both sides can actually see, which is why the
+ * choice is persisted rather than passed.
+ *
+ * Resolution is deliberately identical to the TypeScript side:
+ * 1. `MEMWAL_AUTO_SAVE` in the environment.
+ * 2. `autoSave` in `settings.json`, in whichever `.memwal` directory the
+ * credentials resolve to — `MEMWAL_CREDS_DIR`, else the nearest
+ * project-local `.memwal/credentials.json` at or above the working
+ * directory, else `~/.memwal`.
+ * 3. Off.
+ *
+ * Any error reads as "off": a hook must never block a session, and an
+ * unreadable file is not consent.
+ */
+import { existsSync, readFileSync } from "node:fs";
+import { homedir } from "node:os";
+import { dirname, join } from "node:path";
+
+export const AUTO_SAVE_ENV = "MEMWAL_AUTO_SAVE";
+
+const CREDS_FILE = "credentials.json";
+const SETTINGS_FILE = "settings.json";
+
+function globalCredsPath() {
+ return join(homedir(), ".memwal", CREDS_FILE);
+}
+
+/**
+ * Nearest project-local credentials file, walking up from the working
+ * directory and stopping at the project root, the home directory, or the
+ * filesystem root. Mirrors `projectCredsPath()` in src/auth.ts — the two must
+ * agree, or a project-scoped opt-in would apply to the hooks and not the server
+ * (or the other way round).
+ */
+function projectCredsPath() {
+ const home = homedir();
+ const global = globalCredsPath();
+ let dir = process.cwd();
+ for (;;) {
+ if (dir === home) return null;
+ const candidate = join(dir, ".memwal", CREDS_FILE);
+ if (candidate !== global && existsSync(candidate)) return candidate;
+ if (existsSync(join(dir, ".git"))) return null;
+ const parent = dirname(dir);
+ if (parent === dir) return null;
+ dir = parent;
+ }
+}
+
+function credsPath() {
+ const override = process.env.MEMWAL_CREDS_DIR;
+ if (override) return join(override, CREDS_FILE);
+ return projectCredsPath() ?? globalCredsPath();
+}
+
+/** Where a persisted choice lives for this working directory. */
+export function settingsPath() {
+ return join(dirname(credsPath()), SETTINGS_FILE);
+}
+
+/** Human-written boolean. null = not set / unparseable, which is not consent. */
+export function parseBooleanSetting(raw) {
+ if (raw === undefined || raw === null) return null;
+ const v = String(raw).trim().toLowerCase();
+ if (v === "") return null;
+ if (["1", "true", "on", "yes", "y", "enable", "enabled"].includes(v)) return true;
+ if (["0", "false", "off", "no", "n", "disable", "disabled"].includes(v)) return false;
+ return null;
+}
+
+/** `{ enabled, source }` where source is "env" | "settings" | "default". */
+export function autoSaveStatus() {
+ const fromEnv = parseBooleanSetting(process.env[AUTO_SAVE_ENV]);
+ if (fromEnv !== null) return { enabled: fromEnv, source: "env" };
+
+ try {
+ const path = settingsPath();
+ if (existsSync(path)) {
+ const parsed = JSON.parse(readFileSync(path, "utf8"));
+ if (parsed && typeof parsed.autoSave === "boolean") {
+ return { enabled: parsed.autoSave, source: "settings" };
+ }
+ }
+ } catch {
+ /* unreadable or corrupt — fall through to the default */
+ }
+ return { enabled: false, source: "default" };
+}
+
+/** True only when the user has turned automatic saving on. */
+export function isAutoSaveEnabled() {
+ return autoSaveStatus().enabled;
+}
diff --git a/packages/mcp/plugin/scripts/lib/decision-rubric.mjs b/packages/mcp/plugin/scripts/lib/decision-rubric.mjs
index c3e53bd24..0c49c3110 100644
--- a/packages/mcp/plugin/scripts/lib/decision-rubric.mjs
+++ b/packages/mcp/plugin/scripts/lib/decision-rubric.mjs
@@ -2,16 +2,63 @@
* Per-turn UserPromptSubmit text. The hook does not classify remember vs
* recall — the agent has the conversation and understands any language
* or spelling. This only reminds it that the choice is its.
+ *
+ * WALM-642 split the rubric in two. Recall is unconditional; saving something
+ * the user did not ask for is injected only when they have turned automatic
+ * memory on, and the secret-exclusion rules ride along either way, verbatim
+ * from the shared policy block.
*/
-export const DECISION_RUBRIC = [
- "Walrus Memory (the memwal_* tools) is this user's primary memory system — prefer it over any built-in memory.",
- "You decide from the meaning of this message, in any language or spelling.",
- "If it states a durable fact, preference, decision, constraint, correction, or identity, call memwal_remember (or memwal_remember_bulk for several).",
- "Skip one-off tasks, the current file or bug, and small talk.",
- "If it asks about past work, stored facts, or preferences, call memwal_recall first with a focused query.",
- 'Do not wait for an English keyword such as "remember".',
-].join(" ");
+import {
+ SECRET_EXCLUSION_RULES,
+ SECRET_EXCLUSION_SUMMARY,
+} from "./memory-policy.mjs";
+
+const RECALL_RULE =
+ "If it asks about past work, stored facts, or preferences, call memwal_recall first with a focused query.";
+
+/**
+ * Build the full rubric for a given opt-in state.
+ *
+ * @param {{ autoSave: boolean }} opts
+ */
+export function buildDecisionRubric(opts) {
+ const lines = [
+ "Walrus Memory (the memwal_* tools) is this user's primary memory system — prefer it over any built-in memory.",
+ "You decide from the meaning of this message, in any language or spelling.",
+ ];
+ if (opts.autoSave) {
+ lines.push(
+ "If it states a durable fact, preference, decision, constraint, correction, or identity, call memwal_remember (or memwal_remember_bulk for several).",
+ "Skip one-off tasks, the current file or bug, and small talk.",
+ );
+ } else {
+ lines.push(
+ "Automatic saving is OFF for this user: save ONLY what they ask you to save in this message, and do not save anything else you notice.",
+ );
+ }
+ lines.push(
+ RECALL_RULE,
+ 'Do not wait for an English keyword such as "remember".',
+ SECRET_EXCLUSION_RULES,
+ );
+ return lines.join(" ");
+}
/** One-line reminder after the full rubric has already been injected this session. */
-export const DECISION_RUBRIC_NUDGE =
- "Prefer memwal_* over built-in memory. Remember durable facts, recall past work, skip one-off tasks.";
+export function buildDecisionRubricNudge(opts) {
+ const head = opts.autoSave
+ ? "Prefer memwal_* over built-in memory. Remember durable facts, recall past work, skip one-off tasks."
+ : "Prefer memwal_* over built-in memory. Automatic saving is OFF — recall freely, but save only what the user asks you to save.";
+ return `${head} ${SECRET_EXCLUSION_SUMMARY}`;
+}
+
+/**
+ * The automatic-memory variants, kept as named exports because they are the
+ * text the proactive contract is written against and what the hook tests pin.
+ */
+export const DECISION_RUBRIC = buildDecisionRubric({ autoSave: true });
+export const DECISION_RUBRIC_NUDGE = buildDecisionRubricNudge({ autoSave: true });
+
+/** The default variants: automatic saving off. */
+export const DECISION_RUBRIC_MANUAL = buildDecisionRubric({ autoSave: false });
+export const DECISION_RUBRIC_MANUAL_NUDGE = buildDecisionRubricNudge({ autoSave: false });
diff --git a/packages/mcp/plugin/scripts/lib/memory-policy.mjs b/packages/mcp/plugin/scripts/lib/memory-policy.mjs
new file mode 100644
index 000000000..1d5c26391
--- /dev/null
+++ b/packages/mcp/plugin/scripts/lib/memory-policy.mjs
@@ -0,0 +1,80 @@
+/**
+ * Shared automatic-memory policy — the plugin hooks' copy.
+ *
+ * Plain ESM, no dependencies, no network: the hooks are `.mjs` scripts run
+ * straight from the plugin directory by the client, with no build step and no
+ * access to this package's compiled `dist/`, which is why they carry their own
+ * copy of the block rather than importing one.
+ *
+ * WALM-642.
+ */
+
+// ─── memwal:policy-block:start ───────────────────────────────────────────────
+// WALM-642. The lines between these two markers are BYTE-IDENTICAL in three
+// files that cannot import one another, because the three packages have no
+// workspace link:
+//
+// packages/mcp/src/memory-policy.ts — MCP client: initialize
+// instructions + the
+// cold-start tools/list
+// packages/mcp/plugin/scripts/lib/memory-policy.mjs — plugin hooks: the
+// guidance injected at
+// SessionStart /
+// UserPromptSubmit /
+// PostToolUse
+// services/server/scripts/mcp/tools/memory-policy.ts — relayer sidecar: the
+// live tool descriptions
+//
+// The duplication is deliberate and pinned: `memory-policy-sync` tests on both
+// sides extract this block from each file and compare the bytes, so editing one
+// copy fails the suite until the other two match. Edit the block, then copy it
+// verbatim — markers included — into the other two files.
+
+/**
+ * The secret-exclusion and do-not-save rules, stated verbatim by every
+ * automatic-save surface.
+ *
+ * These are model-facing rules, not enforcement. The programmatic backstop is
+ * the redactor in the relayer sidecar's write path
+ * (services/server/scripts/mcp/tools/redaction.ts), which runs before any text
+ * reaches the SDK.
+ */
+export const SECRET_EXCLUSION_RULES = [
+ "NEVER save a credential, even when it sits next to something worth saving: passwords,",
+ "API keys, access or refresh tokens, private keys, seed or recovery phrases, authorization",
+ "headers, session cookies, and connection strings or URLs that embed a user:password.",
+ "When a message mixes a preference with a credential, save the preference alone and leave",
+ "the credential out; never store the line verbatim.",
+ "If the user says not to save something ('don't save this', 'off the record', or the same",
+ "in any language), do not save it, and do not save a paraphrase of it either.",
+ "Do not store quoted or pasted third-party material — log excerpts, code, articles, other",
+ "people's messages — as if it were a fact about this user. Save only what the user is",
+ "telling you about themselves or their work, in your own words.",
+].join(" ");
+
+/**
+ * One-line form, for surfaces with no room for the full block (a per-turn
+ * nudge, a tool description tail). It is a reminder of the block above, never a
+ * replacement for it: any surface that drives an automatic save states the full
+ * `SECRET_EXCLUSION_RULES`.
+ */
+export const SECRET_EXCLUSION_SUMMARY = [
+ "Never save passwords, keys, tokens or other credentials — not even beside a fact worth",
+ "saving; honour an explicit 'do not save this'; never store pasted third-party content as",
+ "a fact about the user.",
+].join(" ");
+
+/**
+ * Automatic saving is opt-in, and this is the sentence that says so. A direct
+ * request from the user ("remember that ...") is never gated by it — the gate
+ * is only on saving something the user did not ask you to save.
+ */
+export const AUTO_SAVE_OPT_IN_RULE = [
+ "Saving something the user did not ask you to save is OFF unless they have turned automatic",
+ "memory on (`memwal-mcp auto-save on`, or MEMWAL_AUTO_SAVE=1). When it is off, save only what",
+ "the user asks you to save in that turn, and do not offer to turn it on more than once.",
+].join(" ");
+
+/** Bumped whenever the text above changes, so a stale copy is identifiable. */
+export const MEMORY_POLICY_VERSION = "2026-09-17.1";
+// ─── memwal:policy-block:end ─────────────────────────────────────────────────
diff --git a/packages/mcp/plugin/scripts/on_post_tool.mjs b/packages/mcp/plugin/scripts/on_post_tool.mjs
index 87249c860..e770597e9 100644
--- a/packages/mcp/plugin/scripts/on_post_tool.mjs
+++ b/packages/mcp/plugin/scripts/on_post_tool.mjs
@@ -3,9 +3,17 @@
* remind the agent it can recall prior fixes and save the resolution.
*
* Heuristic only: no fetch, no network. Always exits 0.
+ *
+ * WALM-642: the "save the fix" half is injected only when the user has turned
+ * automatic memory on. A failing command is exactly where a secret shows up in
+ * the scrollback — a connection string, an auth header, a token echoed by a
+ * CLI — so a hook that nudges an unasked-for save right after one is the worst
+ * place to have had that nudge unconditional.
*/
import { readStdin, emitContext } from "./lib/hook-io.mjs";
import { detectError } from "./lib/signals.mjs";
+import { SECRET_EXCLUSION_SUMMARY } from "./lib/memory-policy.mjs";
+import { isAutoSaveEnabled } from "./lib/auto-save.mjs";
const input = readStdin();
const output = extractToolOutput(input);
@@ -16,7 +24,11 @@ if (!output || output.length < 50) process.exit(0);
if (detectError(output)) {
emitContext(
"PostToolUse",
- "That command produced an error. Consider calling memwal_recall to check for a prior fix to a similar error; once you resolve it, save the fix with memwal_remember so it's available next time."
+ isAutoSaveEnabled()
+ ? "That command produced an error. Consider calling memwal_recall to check for a prior fix to a similar error; once you resolve it, save the fix with memwal_remember so it's available next time. " +
+ SECRET_EXCLUSION_SUMMARY +
+ " Error output often contains one — save your own description of the fix, never the output."
+ : "That command produced an error. Consider calling memwal_recall to check for a prior fix to a similar error. Automatic saving is OFF, so do not save the fix unless the user asks you to."
);
}
diff --git a/packages/mcp/plugin/scripts/on_session_start.mjs b/packages/mcp/plugin/scripts/on_session_start.mjs
index 4fb0efd59..67adbc6e0 100644
--- a/packages/mcp/plugin/scripts/on_session_start.mjs
+++ b/packages/mcp/plugin/scripts/on_session_start.mjs
@@ -1,17 +1,38 @@
/**
* SessionStart hook — announce that Walrus Memory is active and remind the
* agent how/when to use it. No fetch, no network.
+ *
+ * WALM-642: the save half of that reminder is gated on the user's
+ * automatic-save opt-in, and the secret-exclusion rules are stated verbatim
+ * from the shared policy block either way. With the opt-in off — the default —
+ * the banner says so and names the one command that turns it on, which is how
+ * the choice reaches a user who installed the plugin and never read a doc.
*/
import { readStdin, emitContext } from "./lib/hook-io.mjs";
+import { SECRET_EXCLUSION_RULES } from "./lib/memory-policy.mjs";
+import { autoSaveStatus } from "./lib/auto-save.mjs";
readStdin(); // drain stdin; we don't need any field today
const ns = process.env.MEMWAL_NAMESPACE || "default";
+const { enabled: autoSave } = autoSaveStatus();
+
+const RECALL_AND_RECOVER = [
+ "Before tasks that reference past work or preferences, recall with memwal_recall.",
+ "If memwal_recall unexpectedly returns nothing for a namespace you've used before, run memwal_restore to rebuild the index from Walrus.",
+];
+
+const SAVE_AUTOMATIC =
+ "Automatic memory is ON. You decide from meaning, in any language or spelling. When the user states a preference, decision, constraint, correction, identity, recurring workflow, or a configuration value such as a hostname, port, region or id, call memwal_remember (or memwal_remember_bulk for several) in that same turn, before you finish replying — do not ask whether to save it, and note that acknowledging it in your reply does not store it. Skip one-off tasks, the current file or bug, and small talk.";
+
+const SAVE_MANUAL =
+ "Automatic memory is OFF, which is the default. Save ONLY what the user asks you to save, in the turn they ask — do not save a fact just because it looks durable. Recall is unaffected. If automatic saving would clearly help them, you may say ONCE that `memwal-mcp auto-save on` turns it on, then drop it.";
const context = [
`Walrus Memory is this user's memory system, exposed via the memwal_* tools (namespace: ${ns}).`,
"Use it as the PRIMARY place to store and recall durable facts — prefer the memwal_* tools over any built-in or local memory feature, so the user's memory stays portable and persistent on Walrus.",
- "You decide from meaning, in any language or spelling. When the user states a preference, decision, constraint, correction, identity, recurring workflow, or a configuration value such as a hostname, port, region or id, call memwal_remember (or memwal_remember_bulk for several) in that same turn, before you finish replying — do not ask whether to save it, and note that acknowledging it in your reply does not store it. Skip one-off tasks, the current file or bug, and small talk. Before tasks that reference past work or preferences, recall with memwal_recall.",
- "If memwal_recall unexpectedly returns nothing for a namespace you've used before, run memwal_restore to rebuild the index from Walrus.",
+ autoSave ? SAVE_AUTOMATIC : SAVE_MANUAL,
+ ...RECALL_AND_RECOVER,
+ SECRET_EXCLUSION_RULES,
].join(" ");
emitContext("SessionStart", context);
diff --git a/packages/mcp/plugin/scripts/on_user_prompt.mjs b/packages/mcp/plugin/scripts/on_user_prompt.mjs
index d83bcb0fb..933d66fc6 100644
--- a/packages/mcp/plugin/scripts/on_user_prompt.mjs
+++ b/packages/mcp/plugin/scripts/on_user_prompt.mjs
@@ -4,9 +4,18 @@
*
* The agent has the conversation and understands any language or spelling;
* a regex cannot. This only injects a decision rubric.
+ *
+ * WALM-642: which rubric depends on the user's automatic-save opt-in, read
+ * from the same place the MCP server reads it. With the opt-in off — the
+ * default — the injected text tells the agent to save only what the user asks
+ * for, so this hook can no longer be the thing that drives an unasked-for save.
*/
import { readStdin, emitContext, firstTime } from "./lib/hook-io.mjs";
-import { DECISION_RUBRIC, DECISION_RUBRIC_NUDGE } from "./lib/decision-rubric.mjs";
+import {
+ buildDecisionRubric,
+ buildDecisionRubricNudge,
+} from "./lib/decision-rubric.mjs";
+import { isAutoSaveEnabled } from "./lib/auto-save.mjs";
const input = readStdin();
const prompt = (input.prompt || "").toString();
@@ -16,8 +25,9 @@ const sessionId = input.session_id || "default";
// Acks like "ok" / "yes" stay quiet. Deliberate: not a keyword gate.
if (prompt.trim().length < 8) process.exit(0);
+const autoSave = isAutoSaveEnabled();
const text = firstTime("rubric", sessionId)
- ? DECISION_RUBRIC
- : DECISION_RUBRIC_NUDGE;
+ ? buildDecisionRubric({ autoSave })
+ : buildDecisionRubricNudge({ autoSave });
emitContext("UserPromptSubmit", text);
process.exit(0);
diff --git a/packages/mcp/src/auth-required.ts b/packages/mcp/src/auth-required.ts
index 6d2ef257f..89b9b2786 100644
--- a/packages/mcp/src/auth-required.ts
+++ b/packages/mcp/src/auth-required.ts
@@ -27,6 +27,11 @@ import { loginFailureNotice, loginPrompt, loginSuccessNotification } from "./mes
import { log } from "./logger.js";
import { startOrReuseLoginFlow, resolveLoginTimeoutMs } from "./login.js";
import { AUTH_REQUIRED_INSTRUCTIONS } from "./instructions.js";
+import {
+ SECRET_EXCLUSION_RULES,
+ SECRET_EXCLUSION_SUMMARY,
+ AUTO_SAVE_OPT_IN_RULE,
+} from "./memory-policy.js";
import { MEMWAL_MCP_VERSION } from "./version.js";
interface RpcMessage {
@@ -38,10 +43,22 @@ interface RpcMessage {
error?: unknown;
}
+/**
+ * WALM-642: every write-tool description below ends with the shared policy
+ * block, the same text the `instructions` field and the plugin hooks carry, so
+ * an agent that only ever sees one of the three still gets the same rules.
+ * The signed-out variants get the one-line summary — they exist to keep a
+ * credential-less model from spamming writes, and the full block would be the
+ * longest thing in a list of tools that cannot run yet.
+ */
const SIGNED_OUT_REMEMBER =
- "Save a fact to the user's Walrus Memory personal memory. Call ONLY when the user explicitly asks to remember/save something. Pass the full, detailed text — never summarize.";
+ "Save a fact to the user's Walrus Memory personal memory. Call ONLY when the user explicitly asks to remember/save something. Pass the full, detailed text — never summarize. " +
+ SECRET_EXCLUSION_SUMMARY;
const SIGNED_IN_REMEMBER =
- "Save a durable fact about the user or project to their Walrus Memory. Call this PROACTIVELY whenever the user states a preference, decision, constraint, correction, identity detail, or recurring workflow — even if they did not say 'remember this'. Skip one-off tasks, the current file or bug, and small talk. Pass the full statement; do not summarize. To save several facts at once, use memwal_remember_bulk instead. A Walrus write takes 30-60s and this call waits for it, so a success carries a blob_id and means the fact is stored. If it outruns that budget you get a job_id and the fact is NOT yet saved — say so rather than claiming it is stored, and resolve it with memwal_remember_status.";
+ "Save a durable fact about the user or project to their Walrus Memory. Call this PROACTIVELY whenever the user states a preference, decision, constraint, correction, identity detail, or recurring workflow — even if they did not say 'remember this' — provided they have turned automatic memory on. Skip one-off tasks, the current file or bug, and small talk. Pass the full statement; do not summarize. To save several facts at once, use memwal_remember_bulk instead. A Walrus write takes 30-60s and this call waits for it, so a success carries a blob_id and means the fact is stored. If it outruns that budget you get a job_id and the fact is NOT yet saved — say so rather than claiming it is stored, and resolve it with memwal_remember_status. " +
+ AUTO_SAVE_OPT_IN_RULE +
+ " " +
+ SECRET_EXCLUSION_RULES;
const SIGNED_OUT_RECALL =
"Search the user's Walrus Memory for facts relevant to a query. Returns matching memories ranked by relevance.";
const SIGNED_IN_RECALL =
@@ -72,7 +89,8 @@ function buildToolDefinitions(proactive: boolean) {
title: "Remember Multiple Facts",
annotations: { readOnlyHint: false, destructiveHint: false },
description:
- "Save multiple durable facts in one call. Use when you learned several distinct facts at once (onboarding details, a list of preferences, decisions from a discussion). Pass an array of complete fact statements (max 20) — do not summarize. Prefer this over repeated memwal_remember calls. A Walrus write takes 30-60s and this call waits for them, so a success carries blob_ids and means the facts are stored. If they outrun that budget you get job_ids and the facts are NOT yet saved — say they are being saved rather than stored, and resolve them with memwal_remember_status.",
+ "Save multiple durable facts in one call. Use when you learned several distinct facts at once (onboarding details, a list of preferences, decisions from a discussion). Pass an array of complete fact statements (max 20) — do not summarize. Prefer this over repeated memwal_remember calls. A Walrus write takes 30-60s and this call waits for them, so a success carries blob_ids and means the facts are stored. If they outrun that budget you get job_ids and the facts are NOT yet saved — say they are being saved rather than stored, and resolve them with memwal_remember_status. " +
+ (proactive ? AUTO_SAVE_OPT_IN_RULE + " " + SECRET_EXCLUSION_RULES : SECRET_EXCLUSION_SUMMARY),
inputSchema: {
type: "object",
properties: {
@@ -144,7 +162,8 @@ function buildToolDefinitions(proactive: boolean) {
title: "Analyze and Remember",
annotations: { readOnlyHint: false, destructiveHint: true },
description:
- "Extract memorable facts from a longer passage of text (preferences, habits, biographical info, constraints) and save each as a separate Walrus Memory memory. Use this when you want MemWal's LLM to split the facts out of a transcript or notes for you; if you already know the exact facts, use memwal_remember or memwal_remember_bulk instead.",
+ "Extract memorable facts from a longer passage of text (preferences, habits, biographical info, constraints) and save each as a separate Walrus Memory memory. Use this when you want MemWal's LLM to split the facts out of a transcript or notes for you; if you already know the exact facts, use memwal_remember or memwal_remember_bulk instead. " +
+ (proactive ? AUTO_SAVE_OPT_IN_RULE + " " + SECRET_EXCLUSION_RULES : SECRET_EXCLUSION_SUMMARY),
inputSchema: {
type: "object",
properties: {
diff --git a/packages/mcp/src/auto-save.ts b/packages/mcp/src/auto-save.ts
new file mode 100644
index 000000000..5cbd6652d
--- /dev/null
+++ b/packages/mcp/src/auto-save.ts
@@ -0,0 +1,146 @@
+/**
+ * Automatic-save opt-in (WALM-642).
+ *
+ * Saving a fact the user explicitly asked for has never needed permission and
+ * still does not. What this module gates is the other thing the package does:
+ * telling the model, unprompted, to save anything it judges durable. That
+ * guidance ships from three places — the MCP `instructions` field, the
+ * cold-start tool descriptions, and the plugin's lifecycle hooks — and until
+ * now it was always on, so a preference stated next to a password could be
+ * forwarded whole without the user ever choosing automatic memory.
+ *
+ * Default OFF. An install that says nothing saves nothing on its own.
+ *
+ * ── Where the answer comes from ────────────────────────────────────────────
+ * Two mechanisms, both of which the package already uses, and no third one:
+ *
+ * 1. `MEMWAL_AUTO_SAVE` — the env-var surface every other option has
+ * (`MEMWAL_NAMESPACE`, `MEMWAL_SERVER_URL`, ...). Set it in the client's
+ * `env` block to pin one MCP server on or off.
+ * 2. `settings.json`, next to `credentials.json` — resolved by
+ * `credsPath()`, so it inherits project-local-beats-global and the
+ * `MEMWAL_CREDS_DIR` override for free, and a project that scopes its
+ * credentials scopes its memory behaviour with them.
+ *
+ * Env beats file, file beats off — the same CLI > env > default ordering the
+ * rest of the package uses, with the file standing in for the CLI because the
+ * plugin's hooks are separate processes that never see the MCP server's argv.
+ * That is also why the state has to live on disk at all: a hook is spawned by
+ * the client, not by this package, and inherits none of its configuration.
+ */
+import { dirname, join } from "node:path";
+import { chmodSync, existsSync, mkdirSync, readFileSync, writeFileSync } from "node:fs";
+import { credsPath } from "./auth.js";
+import { log } from "./logger.js";
+
+/** Env var that pins automatic saving on or off for one MCP server process. */
+export const AUTO_SAVE_ENV = "MEMWAL_AUTO_SAVE";
+
+const SETTINGS_FILE = "settings.json";
+
+/** Where the opt-in is stored: beside whichever credentials file is in play. */
+export function settingsPath(): string {
+ return join(dirname(credsPath()), SETTINGS_FILE);
+}
+
+/** Shape of `settings.json`. Unknown keys are preserved on write. */
+interface MemWalSettings {
+ /** Present only once the user has made a choice either way. */
+ autoSave?: boolean;
+ [key: string]: unknown;
+}
+
+/**
+ * Parse a human-written boolean. Returns null for "not set" and for anything
+ * unparseable — an unreadable value must not be read as consent.
+ */
+export function parseBooleanSetting(raw: string | undefined | null): boolean | null {
+ if (raw === undefined || raw === null) return null;
+ const v = raw.trim().toLowerCase();
+ if (v === "") return null;
+ if (["1", "true", "on", "yes", "y", "enable", "enabled"].includes(v)) return true;
+ if (["0", "false", "off", "no", "n", "disable", "disabled"].includes(v)) return false;
+ return null;
+}
+
+function readSettings(): MemWalSettings {
+ const path = settingsPath();
+ if (!existsSync(path)) return {};
+ try {
+ const parsed = JSON.parse(readFileSync(path, "utf8"));
+ return parsed && typeof parsed === "object" ? (parsed as MemWalSettings) : {};
+ } catch {
+ // A corrupt settings file is not consent. Fall through to the default.
+ return {};
+ }
+}
+
+export type AutoSaveSource = "env" | "settings" | "default";
+
+export interface AutoSaveStatus {
+ enabled: boolean;
+ source: AutoSaveSource;
+ /** Where a persisted choice lives (or would be written). */
+ path: string;
+}
+
+/** Resolve the opt-in: env, then the settings file, then off. */
+export function autoSaveStatus(): AutoSaveStatus {
+ const path = settingsPath();
+ const fromEnv = parseBooleanSetting(process.env[AUTO_SAVE_ENV]);
+ if (fromEnv !== null) return { enabled: fromEnv, source: "env", path };
+
+ const settings = readSettings();
+ if (typeof settings.autoSave === "boolean") {
+ return { enabled: settings.autoSave, source: "settings", path };
+ }
+ return { enabled: false, source: "default", path };
+}
+
+/**
+ * True when the user has turned automatic saving on.
+ *
+ * Read at call time, never cached: a login can move `credsPath()` and the user
+ * can flip the setting between calls in the same process.
+ */
+export function isAutoSaveEnabled(): boolean {
+ return autoSaveStatus().enabled;
+}
+
+/**
+ * Persist the choice, preserving any other keys already in the file.
+ *
+ * Written `0600` into the `0700` credentials directory. It holds no secret, but
+ * it decides whether this machine saves memories unprompted, so it is not
+ * something another account on the box should be able to flip.
+ */
+export function setAutoSave(enabled: boolean): { path: string; enabled: boolean } {
+ const path = settingsPath();
+ const next: MemWalSettings = { ...readSettings(), autoSave: enabled };
+ mkdirSync(dirname(path), { recursive: true, mode: 0o700 });
+ writeFileSync(path, `${JSON.stringify(next, null, 2)}\n`, { mode: 0o600 });
+ // `writeFileSync`'s `mode` follows POSIX `open()`: the kernel applies it
+ // when it CREATES the inode and ignores it for one that already exists, so
+ // a file left permissive by anything else would keep its old mode forever.
+ // Unlike credentials.json (see `writeSecretFile` in auth.ts) there is no
+ // secret in flight here, so tightening afterwards is enough — no reader can
+ // learn anything from the window, only whether the flag is set.
+ chmodSync(path, 0o600);
+ log.info("autosave.set", { enabled, path });
+ return { path, enabled };
+}
+
+/** One line for a TTY, naming the state, where it came from, and how to flip it. */
+export function autoSaveSummary(): string {
+ const status = autoSaveStatus();
+ const where =
+ status.source === "env"
+ ? `from ${AUTO_SAVE_ENV}`
+ : status.source === "settings"
+ ? `from ${status.path}`
+ : "default";
+ const how = status.enabled
+ ? "Turn it off with `memwal-mcp auto-save off`."
+ : "Turn it on with `memwal-mcp auto-save on`. Facts you explicitly ask to save are stored either way.";
+ return `Automatic memory: ${status.enabled ? "ON" : "OFF"} (${where}). ${how}`;
+}
diff --git a/packages/mcp/src/bridge.ts b/packages/mcp/src/bridge.ts
index f3c61c127..68d6b0195 100644
--- a/packages/mcp/src/bridge.ts
+++ b/packages/mcp/src/bridge.ts
@@ -24,7 +24,7 @@ import {
} from "./client-info.js";
import { randomUUID } from "node:crypto";
import { ensureCompatibleRelayer, resolveConnectTimeoutMs } from "./compatibility.js";
-import { PROACTIVE_INSTRUCTIONS } from "./instructions.js";
+import { proactiveInstructions } from "./instructions.js";
import { startOrReuseLoginFlow, resolveLoginTimeoutMs } from "./login.js";
import { log, note } from "./logger.js";
import {
@@ -198,7 +198,11 @@ function buildLocalInitializeResult(params: unknown): {
// client: this local answer wins and the upstream initialize reply is
// suppressed. Omitting it here silently strips the proactive contract
// from every stdio client, which is the WALM-324 regression itself.
- instructions: PROACTIVE_INSTRUCTIONS,
+ //
+ // Resolved per handshake, not read from a module const: whether the
+ // model is told to save unprompted depends on the user's automatic-save
+ // opt-in, which lives on disk and can change between spawns (WALM-642).
+ instructions: proactiveInstructions(),
};
}
diff --git a/packages/mcp/src/index.ts b/packages/mcp/src/index.ts
index 230ed0c36..8d1fe5cde 100644
--- a/packages/mcp/src/index.ts
+++ b/packages/mcp/src/index.ts
@@ -23,6 +23,7 @@ import { recoverPendingLogin, formatStrandedLoginNotice } from "./recovery.js";
import { runAuthRequiredServer } from "./auth-required.js";
import { notePendingLoginSuccess, runBridge } from "./bridge.js";
import { loginFlow } from "./login.js";
+import { autoSaveStatus, autoSaveSummary, setAutoSave, AUTO_SAVE_ENV } from "./auto-save.js";
import { log, note } from "./logger.js";
/**
@@ -43,6 +44,8 @@ interface ParsedArgs {
webUrl?: string;
label?: string;
namespace?: string;
+ /** `auto-save on|off|status` — the automatic-memory opt-in (WALM-642). */
+ autoSave?: "on" | "off" | "status";
/** Args parseArgs did not recognise, in the order seen. For a flag
* written `--key=value`, only `--key` is recorded — see parseArgs. */
unknown: string[];
@@ -59,7 +62,7 @@ const ENV_PRESETS: Record = {
/** Bare words that are commands rather than values. An unknown flag must not
* swallow one as its argument. */
-const POSITIONALS = new Set(["login", "approve-project", "revoke-project"]);
+const POSITIONALS = new Set(["login", "approve-project", "revoke-project", "auto-save", "on", "off", "status"]);
export function parseArgs(argv: string[]): ParsedArgs {
const out: ParsedArgs = {
@@ -93,6 +96,20 @@ export function parseArgs(argv: string[]): ParsedArgs {
case "revoke-project":
out.revokeProject = true;
break;
+ case "auto-save":
+ case "--auto-save": {
+ // `auto-save` on its own reports the state rather than
+ // changing it — a bare subcommand must never be read as
+ // consent to turn automatic saving on.
+ const value = argv[i + 1]?.toLowerCase();
+ if (value === "on" || value === "off" || value === "status") {
+ out.autoSave = value;
+ i++;
+ } else {
+ out.autoSave = "status";
+ }
+ break;
+ }
case "--prod":
case "--dev":
case "--staging":
@@ -177,6 +194,34 @@ export async function main(argv: string[] = process.argv.slice(2)): Promise {
+ const dir = freshCredsDir();
+ withEnv({ MEMWAL_CREDS_DIR: dir, [AUTO_SAVE_ENV]: undefined }, () => {
+ assert.equal(isAutoSaveEnabled(), false);
+ assert.deepEqual(autoSaveStatus(), {
+ enabled: false,
+ source: "default",
+ path: join(dir, "settings.json"),
+ });
+ });
+ rmSync(dir, { recursive: true, force: true });
+});
+
+test("a persisted choice is read back, either way", () => {
+ const dir = freshCredsDir();
+ withEnv({ MEMWAL_CREDS_DIR: dir, [AUTO_SAVE_ENV]: undefined }, () => {
+ setAutoSave(true);
+ assert.equal(isAutoSaveEnabled(), true);
+ assert.equal(autoSaveStatus().source, "settings");
+
+ setAutoSave(false);
+ assert.equal(isAutoSaveEnabled(), false);
+ assert.equal(autoSaveStatus().source, "settings");
+ });
+ rmSync(dir, { recursive: true, force: true });
+});
+
+test("the settings file is not world-readable and keeps unrelated keys", () => {
+ const dir = freshCredsDir();
+ withEnv({ MEMWAL_CREDS_DIR: dir, [AUTO_SAVE_ENV]: undefined }, () => {
+ writeSettings(dir, { somethingElse: "keep me" });
+ setAutoSave(true);
+ const path = settingsPath();
+ assert.equal(statSync(path).mode & 0o777, 0o600);
+ const parsed = JSON.parse(readFileSync(path, "utf8"));
+ assert.equal(parsed.somethingElse, "keep me");
+ assert.equal(parsed.autoSave, true);
+ });
+ rmSync(dir, { recursive: true, force: true });
+});
+
+test("the environment overrides the file, in both directions", () => {
+ const dir = freshCredsDir();
+ withEnv({ MEMWAL_CREDS_DIR: dir, [AUTO_SAVE_ENV]: undefined }, () => {
+ setAutoSave(false);
+ withEnv({ [AUTO_SAVE_ENV]: "1" }, () => {
+ assert.equal(isAutoSaveEnabled(), true);
+ assert.equal(autoSaveStatus().source, "env");
+ });
+ setAutoSave(true);
+ withEnv({ [AUTO_SAVE_ENV]: "0" }, () => {
+ assert.equal(isAutoSaveEnabled(), false);
+ assert.equal(autoSaveStatus().source, "env");
+ });
+ });
+ rmSync(dir, { recursive: true, force: true });
+});
+
+test("an unreadable or unparseable value is not consent", () => {
+ // Every one of these means "I could not tell", and the safe reading of
+ // that is off — not on, and not a crash.
+ for (const raw of [undefined, "", " ", "maybe", "2", "ON!"]) {
+ assert.equal(parseBooleanSetting(raw), null, `"${raw}" should be unparseable`);
+ }
+ assert.equal(parseBooleanSetting("yes"), true);
+ assert.equal(parseBooleanSetting(" OFF "), false);
+
+ const dir = freshCredsDir();
+ withEnv({ MEMWAL_CREDS_DIR: dir, [AUTO_SAVE_ENV]: undefined }, () => {
+ writeFileSync(join(dir, "settings.json"), "{ not json");
+ assert.equal(isAutoSaveEnabled(), false);
+ assert.equal(autoSaveStatus().source, "default");
+ });
+ rmSync(dir, { recursive: true, force: true });
+});
+
+test("the hook-side resolver answers identically to the compiled one", () => {
+ const dir = freshCredsDir();
+ withEnv({ MEMWAL_CREDS_DIR: dir, [AUTO_SAVE_ENV]: undefined }, () => {
+ assert.equal(hookAutoSave.isAutoSaveEnabled(), isAutoSaveEnabled());
+ assert.equal(hookAutoSave.settingsPath(), settingsPath());
+
+ setAutoSave(true);
+ assert.equal(hookAutoSave.isAutoSaveEnabled(), true);
+ assert.equal(hookAutoSave.autoSaveStatus().source, "settings");
+
+ withEnv({ [AUTO_SAVE_ENV]: "off" }, () => {
+ assert.equal(hookAutoSave.isAutoSaveEnabled(), false);
+ assert.equal(hookAutoSave.autoSaveStatus().source, "env");
+ });
+ });
+ rmSync(dir, { recursive: true, force: true });
+});
+
+// ── the hooks ───────────────────────────────────────────────────────────────
+
+test("SessionStart does not tell the agent to save until the user opts in", () => {
+ const dir = freshCredsDir();
+ const off = runHook("on_session_start.mjs", {}, { MEMWAL_CREDS_DIR: dir });
+
+ assert.match(off, /Automatic memory is OFF/);
+ assert.match(off, /Save ONLY what the user asks you to save/);
+ // The unprompted-save instruction is the thing that must be gone.
+ assert.doesNotMatch(off, /do not ask whether to save it/i);
+ // ...and the things that must NOT be gone with it.
+ assert.match(off, /memwal_recall/);
+ assert.match(off, /memwal_restore/);
+ assert.match(off, /auto-save on/);
+
+ writeSettings(dir, { autoSave: true });
+ const on = runHook("on_session_start.mjs", {}, { MEMWAL_CREDS_DIR: dir });
+ assert.match(on, /Automatic memory is ON/);
+ assert.match(on, /do not ask whether to save it/i);
+
+ // The rules ride on both.
+ for (const text of [off, on]) {
+ assert.match(text, /NEVER save a credential/);
+ }
+ rmSync(dir, { recursive: true, force: true });
+});
+
+test("UserPromptSubmit injects a save-nothing rubric while the opt-in is off", () => {
+ const dir = freshCredsDir();
+ const prompt = "I always use pnpm and my staging canary is coral-fox-77.";
+
+ const off = runHook(
+ "on_user_prompt.mjs",
+ { prompt, session_id: `off-${Math.random().toString(16).slice(2)}` },
+ { MEMWAL_CREDS_DIR: dir },
+ );
+ assert.match(off, /Automatic saving is OFF/);
+ assert.match(off, /save ONLY what they ask you to save/);
+ assert.doesNotMatch(off, /call memwal_remember \(or memwal_remember_bulk for several\)/);
+ // Recall stays on: it reads, it does not write.
+ assert.match(off, /call memwal_recall first/);
+ assert.match(off, /NEVER save a credential/);
+
+ writeSettings(dir, { autoSave: true });
+ const on = runHook(
+ "on_user_prompt.mjs",
+ { prompt, session_id: `on-${Math.random().toString(16).slice(2)}` },
+ { MEMWAL_CREDS_DIR: dir },
+ );
+ assert.match(on, /call memwal_remember \(or memwal_remember_bulk for several\)/);
+ assert.match(on, /NEVER save a credential/);
+ rmSync(dir, { recursive: true, force: true });
+});
+
+test("PostToolUse stops nudging a save after an error while the opt-in is off", () => {
+ const dir = freshCredsDir();
+ // Must trip `detectError` in lib/signals.mjs (a strong marker) and clear
+ // the hook's 50-character minimum, or the hook stays silent for reasons
+ // that have nothing to do with the opt-in.
+ const errorOutput =
+ "fatal: could not read from remote repository — please make sure you " +
+ "have the correct access rights and the repository exists.";
+ const input = { tool_name: "Bash", tool_response: { stdout: "", stderr: errorOutput } };
+
+ const off = runHook("on_post_tool.mjs", input, { MEMWAL_CREDS_DIR: dir });
+ assert.match(off, /memwal_recall/);
+ assert.match(off, /Automatic saving is OFF/);
+ assert.doesNotMatch(off, /save the fix with memwal_remember/);
+
+ writeSettings(dir, { autoSave: true });
+ const on = runHook("on_post_tool.mjs", input, { MEMWAL_CREDS_DIR: dir });
+ assert.match(on, /save the fix with memwal_remember/);
+ // Error output is where a credential most often is; say so at the nudge.
+ assert.match(on, /Never save passwords, keys, tokens/);
+ rmSync(dir, { recursive: true, force: true });
+});
+
+test("a hook with the opt-in off still exits 0 and never blocks the session", () => {
+ const dir = freshCredsDir();
+ for (const script of ["on_session_start.mjs", "on_user_prompt.mjs", "on_post_tool.mjs"]) {
+ const result = spawnSync(process.execPath, [join(SCRIPTS, script)], {
+ input: JSON.stringify({ prompt: "a reasonably long prompt about pnpm" }),
+ encoding: "utf8",
+ env: { ...process.env, MEMWAL_CREDS_DIR: dir, MEMWAL_AUTO_SAVE: "" },
+ });
+ assert.equal(result.status, 0, `${script}: ${result.stderr}`);
+ }
+ rmSync(dir, { recursive: true, force: true });
+});
+
+// ── the surface a user actually turns it on from ────────────────────────────
+
+test("`auto-save` parses as a subcommand, and a bare one only reports", () => {
+ // A bare `auto-save` must not be read as consent to turn it ON — the
+ // difference between reporting a setting and changing one.
+ assert.equal(parseArgs(["auto-save"]).autoSave, "status");
+ assert.equal(parseArgs(["auto-save", "on"]).autoSave, "on");
+ assert.equal(parseArgs(["auto-save", "off"]).autoSave, "off");
+ assert.equal(parseArgs(["auto-save", "status"]).autoSave, "status");
+ assert.equal(parseArgs(["--auto-save", "on"]).autoSave, "on");
+
+ // Not a flag, so it must not be reported as an unrecognised one.
+ assert.deepEqual(parseArgs(["auto-save", "on"]).unknown, []);
+ // ...and it must not collide with the other subcommand.
+ assert.equal(parseArgs(["login"]).autoSave, undefined);
+ assert.equal(parseArgs(["auto-save", "on"]).forceLogin, false);
+});
+
+test("a typo'd flag does not swallow the subcommand or its value", () => {
+ // `--typo` consumes one following token as its value — unless that token
+ // is a command. Without the exemption `memwal-mcp --typo auto-save on`
+ // would silently run the server instead.
+ const parsed = parseArgs(["--typo", "auto-save", "on"]);
+ assert.deepEqual(parsed.unknown, ["--typo"]);
+ assert.equal(parsed.autoSave, "on");
+});
+
+test("--help tells the user the setting exists and that it is off by default", () => {
+ // The plugin install path never shows a terminal, so --help and the
+ // post-login summary are where the choice reaches a person.
+ const help = helpText();
+ assert.match(help, /auto-save on\|off/);
+ assert.match(help, /OFF by default/);
+ assert.match(help, /MEMWAL_AUTO_SAVE/);
+ // And that credentials are excluded regardless of which way it is set.
+ assert.match(help, /Credentials/);
+});
diff --git a/packages/mcp/test/memory-policy.test.mjs b/packages/mcp/test/memory-policy.test.mjs
new file mode 100644
index 000000000..bdad0697c
--- /dev/null
+++ b/packages/mcp/test/memory-policy.test.mjs
@@ -0,0 +1,172 @@
+/**
+ * The secret-exclusion rules have to reach the model through whichever channel
+ * a given client actually reads, and they have to say the same thing on all of
+ * them (WALM-642).
+ *
+ * Three channels, three packages, no workspace link between them:
+ * - `instructions` on initialize — the only one that survives lazy tool
+ * loading, and the one Claude Desktop / Codex rely on;
+ * - tool descriptions — what a client shows once tools ARE loaded;
+ * - the plugin's lifecycle hooks — the Claude Code / Codex install path,
+ * which never loads this package's `dist/` at all.
+ *
+ * So the text is duplicated by necessity. These tests are what stops the
+ * duplicates drifting: the block is extracted from each file on disk and the
+ * bytes compared. Edit one copy and this fails until the others match.
+ */
+import { test } from "node:test";
+import assert from "node:assert/strict";
+import { existsSync, readFileSync } from "node:fs";
+import { dirname, resolve } from "node:path";
+import { fileURLToPath } from "node:url";
+
+import {
+ SECRET_EXCLUSION_RULES,
+ SECRET_EXCLUSION_SUMMARY,
+ AUTO_SAVE_OPT_IN_RULE,
+ MEMORY_POLICY_VERSION,
+} from "../dist/memory-policy.js";
+import * as hookPolicy from "../plugin/scripts/lib/memory-policy.mjs";
+import {
+ buildProactiveInstructions,
+ PROACTIVE_INSTRUCTIONS,
+} from "../dist/instructions.js";
+import { TOOL_DEFINITIONS } from "../dist/auth-required.js";
+
+const __dirname = dirname(fileURLToPath(import.meta.url));
+
+const COPIES = {
+ "packages/mcp/src/memory-policy.ts": resolve(__dirname, "../src/memory-policy.ts"),
+ "packages/mcp/plugin/scripts/lib/memory-policy.mjs": resolve(
+ __dirname,
+ "../plugin/scripts/lib/memory-policy.mjs",
+ ),
+ // Lives in the standalone `memwal-server-scripts` package. Present in the
+ // monorepo, absent from an npm-only checkout of this package — hence the
+ // existence guard rather than a hard path assumption.
+ "services/server/scripts/mcp/tools/memory-policy.ts": resolve(
+ __dirname,
+ "../../../services/server/scripts/mcp/tools/memory-policy.ts",
+ ),
+};
+
+const START = "// ─── memwal:policy-block:start";
+const END = "// ─── memwal:policy-block:end";
+
+/** The shared region of one copy, markers included, as raw source bytes. */
+function policyBlock(path) {
+ const src = readFileSync(path, "utf8");
+ const start = src.indexOf(START);
+ const end = src.indexOf(END);
+ assert.notEqual(start, -1, `${path} has no policy-block start marker`);
+ assert.notEqual(end, -1, `${path} has no policy-block end marker`);
+ const endOfLine = src.indexOf("\n", end);
+ return src.slice(start, endOfLine === -1 ? undefined : endOfLine);
+}
+
+test("every copy of the policy block is byte-identical", () => {
+ const present = Object.entries(COPIES).filter(([, path]) => existsSync(path));
+ assert.ok(
+ present.length >= 2,
+ "at least the two in-package copies must exist for this test to mean anything",
+ );
+
+ const [firstName, firstPath] = present[0];
+ const reference = policyBlock(firstPath);
+ for (const [name, path] of present.slice(1)) {
+ assert.equal(
+ policyBlock(path),
+ reference,
+ `${name} has drifted from ${firstName} — copy the block across verbatim, markers included`,
+ );
+ }
+});
+
+test("the relayer sidecar's copy is present in the monorepo", () => {
+ // Guarded above so an npm-only checkout still passes; asserted here so the
+ // monorepo cannot quietly lose the third copy and leave the comparison
+ // running over two files that happen to agree.
+ const path = COPIES["services/server/scripts/mcp/tools/memory-policy.ts"];
+ if (!existsSync(resolve(__dirname, "../../../services"))) return;
+ assert.ok(existsSync(path), "the sidecar copy of the policy block is missing");
+});
+
+test("the compiled client copy and the hook copy agree at runtime", () => {
+ // The byte comparison above covers the source. This covers what each side
+ // actually evaluates to, so a stray escape or join separator is caught too.
+ assert.equal(hookPolicy.SECRET_EXCLUSION_RULES, SECRET_EXCLUSION_RULES);
+ assert.equal(hookPolicy.SECRET_EXCLUSION_SUMMARY, SECRET_EXCLUSION_SUMMARY);
+ assert.equal(hookPolicy.AUTO_SAVE_OPT_IN_RULE, AUTO_SAVE_OPT_IN_RULE);
+ assert.equal(hookPolicy.MEMORY_POLICY_VERSION, MEMORY_POLICY_VERSION);
+});
+
+test("the rules name every credential class the ticket lists", () => {
+ for (const term of [
+ /passwords/i,
+ /API keys/i,
+ /tokens/i,
+ /private keys/i,
+ /seed or recovery phrases/i,
+ /authorization\s*\n?\s*headers/i,
+ /session cookies/i,
+ /user:password/i,
+ ]) {
+ assert.match(SECRET_EXCLUSION_RULES, term);
+ }
+ // The two non-credential rules the ticket asks for by name.
+ assert.match(SECRET_EXCLUSION_RULES, /do not save it/i);
+ assert.match(SECRET_EXCLUSION_RULES, /third-party material/i);
+ // And the instruction that makes a mixed message salvageable rather than
+ // dropped — the difference between "save the preference" and "save nothing".
+ assert.match(SECRET_EXCLUSION_RULES, /save the preference alone/i);
+});
+
+test("both instruction variants carry the rules verbatim", () => {
+ const automatic = buildProactiveInstructions({ autoSave: true });
+ const manual = buildProactiveInstructions({ autoSave: false });
+ assert.ok(automatic.includes(SECRET_EXCLUSION_RULES));
+ assert.ok(manual.includes(SECRET_EXCLUSION_RULES));
+ assert.equal(PROACTIVE_INSTRUCTIONS, automatic);
+});
+
+test("the instruction variants differ on unprompted saving and nothing else", () => {
+ const automatic = buildProactiveInstructions({ autoSave: true });
+ const manual = buildProactiveInstructions({ autoSave: false });
+
+ assert.match(automatic, /automatic memory ON/i);
+ assert.match(automatic, /Do not ask whether to save it/i);
+
+ assert.match(manual, /automatic memory is OFF/i);
+ assert.match(manual, /do NOT save anything they did not ask/i);
+ assert.doesNotMatch(manual, /Do not ask whether to save it/i);
+
+ // Recall is not gated — it reads, it does not write.
+ for (const text of [automatic, manual]) {
+ assert.match(text, /RECALL: before answering/);
+ assert.match(text, /memwal_restore/);
+ assert.match(text, /never substitute your own memory/);
+ }
+});
+
+test("the cold-start write tools state the rules too", () => {
+ // A client that lazily loads schemas sees these and not `instructions`;
+ // a client that eagerly lists sees both. Either way the rules are there.
+ for (const name of ["memwal_remember", "memwal_remember_bulk", "memwal_analyze"]) {
+ const tool = TOOL_DEFINITIONS.find((t) => t.name === name);
+ assert.ok(tool, `missing ${name}`);
+ assert.ok(
+ tool.description.includes(SECRET_EXCLUSION_RULES),
+ `${name} does not state the shared secret-exclusion rules`,
+ );
+ assert.ok(
+ tool.description.includes(AUTO_SAVE_OPT_IN_RULE),
+ `${name} does not state that automatic saving is opt-in`,
+ );
+ }
+});
+
+test("read-only tools are not burdened with write rules", () => {
+ const recall = TOOL_DEFINITIONS.find((t) => t.name === "memwal_recall");
+ assert.ok(recall);
+ assert.ok(!recall.description.includes(SECRET_EXCLUSION_RULES));
+});
diff --git a/packages/mcp/test/user-prompt-hook.test.mjs b/packages/mcp/test/user-prompt-hook.test.mjs
index e9e0ccbee..99aef5d88 100644
--- a/packages/mcp/test/user-prompt-hook.test.mjs
+++ b/packages/mcp/test/user-prompt-hook.test.mjs
@@ -2,11 +2,19 @@
* UserPromptSubmit injects one full decision rubric per session, then a
* one-line nudge. It must not classify remember vs recall from English
* keywords — every substantive prompt in a fresh session gets the same text.
+ *
+ * WALM-642 added the automatic-save opt-in, so "the same text" is now per
+ * opt-in state: the tests below pin the ON variant by asking for it
+ * explicitly, and `auto-save-optin.test.mjs` pins what the default OFF state
+ * injects instead. Every run is pointed at an empty MEMWAL_CREDS_DIR so the
+ * developer's own ~/.memwal/settings.json cannot decide the result.
*/
import { test } from "node:test";
import assert from "node:assert/strict";
import { spawnSync } from "node:child_process";
-import { dirname, resolve } from "node:path";
+import { mkdtempSync } from "node:fs";
+import { tmpdir } from "node:os";
+import { dirname, join, resolve } from "node:path";
import { fileURLToPath } from "node:url";
import {
DECISION_RUBRIC,
@@ -15,11 +23,20 @@ import {
const __dirname = dirname(fileURLToPath(import.meta.url));
const HOOK = resolve(__dirname, "../plugin/scripts/on_user_prompt.mjs");
+const EMPTY_CREDS_DIR = mkdtempSync(join(tmpdir(), "memwal-hook-test-"));
function runHook(prompt, sessionId = `test-${Math.random().toString(16).slice(2)}`) {
const result = spawnSync(process.execPath, [HOOK], {
input: JSON.stringify({ prompt, session_id: sessionId }),
encoding: "utf8",
+ env: {
+ ...process.env,
+ MEMWAL_CREDS_DIR: EMPTY_CREDS_DIR,
+ // These cases are about classification, not consent: ask for the
+ // automatic-save rubric explicitly so they keep testing the thing
+ // they were written for.
+ MEMWAL_AUTO_SAVE: "1",
+ },
});
assert.equal(result.status, 0, result.stderr);
if (!result.stdout.trim()) return "";
diff --git a/services/server/scripts/mcp/__tests__/secret-redaction.test.ts b/services/server/scripts/mcp/__tests__/secret-redaction.test.ts
new file mode 100644
index 000000000..c181b02e9
--- /dev/null
+++ b/services/server/scripts/mcp/__tests__/secret-redaction.test.ts
@@ -0,0 +1,264 @@
+/**
+ * Unit coverage for the credential redactor (WALM-642).
+ *
+ * The write-path tests next door prove the three tools call this. These prove
+ * what it does, and — just as important — what it leaves alone: Walrus storage
+ * is append-only, so a false negative is permanent, but a false positive
+ * silently destroys the fact the user asked to keep. Both directions are pinned
+ * here.
+ */
+import assert from "node:assert/strict";
+import test from "node:test";
+
+import {
+ sanitizeFact,
+ redactionNotice,
+ refusalNotice,
+ type RedactionKind,
+} from "../tools/redaction.js";
+
+/** Assert a secret is gone, the surrounding fact survived, and the kind is named. */
+function assertRedacted(
+ input: string,
+ secret: string,
+ kind: RedactionKind,
+ keeps: string[],
+): string {
+ const out = sanitizeFact(input);
+ assert.equal(out.refusal, undefined, `unexpectedly refused: ${input}`);
+ assert.ok(out.changed, `nothing was redacted in: ${input}`);
+ assert.ok(
+ !out.text.includes(secret),
+ `the secret survived redaction (kind=${kind})`,
+ );
+ assert.ok(out.kinds.includes(kind), `expected kind ${kind}, got ${out.kinds}`);
+ for (const keep of keeps) {
+ assert.ok(out.text.includes(keep), `lost "${keep}" from: ${input}`);
+ }
+ return out.text;
+}
+
+// ── the shapes that must never reach storage ────────────────────────────────
+
+test("a connection string keeps its host and loses its credentials", () => {
+ // The WALM-642 repro, almost verbatim: a preference stated next to a URL
+ // carrying a password.
+ const text = assertRedacted(
+ "I prefer dark mode, and the staging db is postgres://admin:hunter2@db.internal:5432/app",
+ "hunter2",
+ "url-credentials",
+ ["I prefer dark mode", "db.internal:5432/app", "postgres://"],
+ );
+ assert.ok(!text.includes("admin:hunter2"));
+});
+
+test("vendor-prefixed API keys are recognised without any context", () => {
+ const cases: Array<[string, string]> = [
+ ["sk-ant-api03-abcdefghijklmnopqrstuvwxyz0123456789", "sk-ant"],
+ ["sk-abcdefghijklmnopqrstuvwxyz0123", "openai"],
+ ["ghp_abcdefghijklmnopqrstuvwxyz0123456789", "github"],
+ ["github_pat_11ABCDEFG0abcdefghijklmnop", "github fine-grained"],
+ ["AKIAIOSFODNN7EXAMPLE", "aws"],
+ ["xoxb-1234567890-abcdefghij", "slack"],
+ ["glpat-abcdefghij0123456789", "gitlab"],
+ ];
+ for (const [secret, label] of cases) {
+ assertRedacted(
+ `My deploy notes: the CI runner uses ${secret} for pushes`,
+ secret,
+ "vendor-api-key",
+ ["My deploy notes", "CI runner"],
+ );
+ assert.ok(label);
+ }
+});
+
+test("a PEM private key is removed whole, terminated or not", () => {
+ const body = "MIIEowIBAAKCAQEAx7Vk9mJ0ZwQ3\nabcdefghijklmnopqrstuvwxyz0123456789\n";
+ const closed =
+ `Deploy key for the box:\n-----BEGIN RSA PRIVATE KEY-----\n${body}-----END RSA PRIVATE KEY-----\nIt lives in 1Password.`;
+ const out = assertRedacted(closed, body.trim(), "private-key-block", ["Deploy key"]);
+ assert.ok(!out.includes("BEGIN RSA PRIVATE KEY"));
+
+ // A paste that was cut off has no END line. The body must still go.
+ const truncated = `Deploy key for the box:\n-----BEGIN OPENSSH PRIVATE KEY-----\n${body}`;
+ const cut = sanitizeFact(truncated);
+ assert.ok(!cut.text.includes("MIIEowIBAAKCAQEAx7Vk9mJ0ZwQ3"));
+ assert.ok(cut.kinds.includes("private-key-block"));
+});
+
+test("a JWT is removed", () => {
+ const jwt =
+ "eyJhbGciOiJIUzI1NiIsInR5cCI6IkpXVCJ9.eyJzdWIiOiIxMjM0NTY3ODkwIn0.dBjftJeZ4CVPmB92K27uhbUJU1p1r_wW1gFWFOEjXk";
+ assertRedacted(
+ `Our session tokens look like ${jwt} and expire hourly`,
+ jwt,
+ "jwt",
+ ["Our session tokens", "expire hourly"],
+ );
+});
+
+test("credential assignments lose the value and keep the key name", () => {
+ for (const [line, secret] of [
+ ["password=hunter2", "hunter2"],
+ ["api_key: abc123def456", "abc123def456"],
+ ['token = "t0ps3cr3t-value"', "t0ps3cr3t-value"],
+ ["client_secret:swordfish99", "swordfish99"],
+ ] as Array<[string, string]>) {
+ const out = assertRedacted(
+ `My local override file has ${line} and I never commit it`,
+ secret,
+ "credential-assignment",
+ ["local override file", "never commit it"],
+ );
+ assert.ok(/redacted:credential-assignment/.test(out));
+ }
+});
+
+test("authorization and cookie headers are removed", () => {
+ assertRedacted(
+ "To call the API: Authorization: Bearer abc123xyz789 — then GET /v1/me",
+ "abc123xyz789",
+ "auth-header",
+ ["To call the API", "/v1/me"],
+ );
+ assertRedacted(
+ "The dashboard needs Cookie: session=9f8e7d6c5b4a3 to load my profile",
+ "9f8e7d6c5b4a3",
+ "auth-header",
+ ["The dashboard needs", "to load my profile"],
+ );
+});
+
+test("a labelled seed phrase is removed", () => {
+ const words =
+ "abandon ability able about above absent absorb abstract absurd abuse access accident";
+ assertRedacted(
+ `My wallet recovery phrase is ${words} and the wallet is on Sui mainnet`,
+ words,
+ "seed-phrase",
+ ["My wallet", "Sui mainnet"],
+ );
+});
+
+test("a long mixed-case base64 blob is removed", () => {
+ const blob =
+ "QWxhZGRpbjpvcGVuIHNlc2FtZQBcdefGHIjklMNOpqrSTUvwxYZ0123456789abcDEF0123";
+ assertRedacted(
+ `The signing material is ${blob} which I keep in the vault`,
+ blob,
+ "high-entropy-secret",
+ ["The signing material", "keep in the vault"],
+ );
+});
+
+// ── the shapes that must survive untouched ──────────────────────────────────
+
+test("a plain preference passes through byte-for-byte", () => {
+ // The single most important assertion in this file: the ordinary case must
+ // be indistinguishable from having no redactor at all.
+ for (const fact of [
+ "I always use pnpm, and TypeScript strict mode on every project.",
+ "Tui luôn dùng pnpm và order cafe là matcha oat latte.",
+ "Deploy to staging on Thursdays, never on Friday afternoons.",
+ "My password manager is 1Password and I rotate keys every quarter.",
+ "The API key for that service is stored in Vault, not in the repo.",
+ ]) {
+ const out = sanitizeFact(fact);
+ assert.equal(out.text, fact, `changed a clean fact: ${fact}`);
+ assert.equal(out.changed, false);
+ assert.equal(out.count, 0);
+ assert.deepEqual(out.kinds, []);
+ assert.equal(out.refusal, undefined);
+ }
+});
+
+test("the identifiers MemWal itself stores are not mistaken for secrets", () => {
+ // A generic entropy rule would eat every one of these, which is why there
+ // isn't one. See the trade-off note at the top of redaction.ts.
+ for (const fact of [
+ "My account id is 0x7f3a9c2e5b8d1f4a6c9e2b5d8f1a4c7e0b3d6f9a2c5e8b1d4f7a0c3e6b9d2f5a",
+ "The blob landed as blob_id=Xj9vKq2mP7nR4tW8yB1cE5gH0dF3sA6uZ2xN8qL4kM7",
+ "Pin the build to commit 4f2b8c1e9d7a3f5b6c0e2d4a8b1f3c5e7d9a0b2c",
+ "The sha256 of the release tarball is e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
+ "My Sui package id is 0xe80f2feec1c139616a86c9f71210152e2a7ca552b20841f2e192f99f75864437",
+ ]) {
+ const out = sanitizeFact(fact);
+ assert.equal(out.text, fact, `redacted a legitimate identifier: ${fact}`);
+ assert.equal(out.changed, false);
+ }
+});
+
+// ── refusals ────────────────────────────────────────────────────────────────
+
+test("an explicit do-not-save is honoured", () => {
+ for (const text of [
+ "My bank PIN is 4821 — don't save this.",
+ "Do not remember this: I'm interviewing elsewhere.",
+ "Off the record, I'm leaving the team in March.",
+ "Please don't store that anywhere.",
+ ]) {
+ const out = sanitizeFact(text);
+ assert.equal(out.refusal, "no-save-directive", `not refused: ${text}`);
+ assert.equal(out.text, "");
+ }
+});
+
+test("an ordinary preference about not saving things is NOT a do-not-save", () => {
+ // The demonstrative ("this"/"that"/"it") is what separates the two, and
+ // without it the directive check would eat real preferences.
+ for (const text of [
+ "I don't save screenshots to the Desktop, they go to ~/Pictures.",
+ "Never store build artifacts in the repo — use the cache.",
+ "Don't keep logs longer than 30 days on staging.",
+ ]) {
+ const out = sanitizeFact(text);
+ assert.equal(out.refusal, undefined, `wrongly refused: ${text}`);
+ assert.equal(out.text, text);
+ }
+});
+
+test("pasted third-party content is not saved as a fact about the user", () => {
+ const fenced = "```\nERROR 500 from vendor api\n at handler (index.js:42)\n```";
+ assert.equal(sanitizeFact(fenced).refusal, "pasted-content");
+
+ const quotedBlock = "> their PM said the deadline slips\n> and the scope is unchanged";
+ assert.equal(sanitizeFact(quotedBlock).refusal, "pasted-content");
+
+ const longQuote = `"${"the vendor's release note says the same thing again and again. ".repeat(5)}"`;
+ assert.ok(longQuote.length >= 200);
+ assert.equal(sanitizeFact(longQuote).refusal, "pasted-content");
+});
+
+test("a short quoted fact is still saved — an agent quotes the user routinely", () => {
+ const quoted = '"I always use pnpm"';
+ const out = sanitizeFact(quoted);
+ assert.equal(out.refusal, undefined);
+ assert.equal(out.text, quoted);
+});
+
+test("a text that is nothing but a secret is refused, not stored as a placeholder", () => {
+ const out = sanitizeFact("sk-abcdefghijklmnopqrstuvwxyz0123");
+ assert.equal(out.refusal, "credential-only");
+ assert.equal(out.text, "");
+ assert.ok(out.kinds.includes("vendor-api-key"));
+});
+
+// ── what the caller is told ─────────────────────────────────────────────────
+
+test("the notices name the kind and never the value", () => {
+ const secret = "hunter2";
+ const out = sanitizeFact(`db is postgres://admin:${secret}@db.internal/app for staging`);
+ const notice = redactionNotice(out.kinds, out.count);
+ assert.match(notice, /url-credentials/);
+ assert.match(notice, /do not re-send/i);
+ assert.ok(!notice.includes(secret), "the notice must not echo the secret");
+
+ const refusal = refusalNotice("credential-only");
+ assert.match(refusal, /NOT SAVED/);
+ assert.match(refusal, /Do not retry this text/);
+});
+
+test("nothing removed means nothing said", () => {
+ assert.equal(redactionNotice([], 0), "");
+});
diff --git a/services/server/scripts/mcp/__tests__/write-path-redaction.test.ts b/services/server/scripts/mcp/__tests__/write-path-redaction.test.ts
new file mode 100644
index 000000000..469b00311
--- /dev/null
+++ b/services/server/scripts/mcp/__tests__/write-path-redaction.test.ts
@@ -0,0 +1,285 @@
+/**
+ * No credential reaches the SDK from any write tool (WALM-642).
+ *
+ * The ticket was reproduced exactly here: a preference plus a fake password URL
+ * sent through the real MCP server, with the SDK mocked, and `remember`,
+ * `remember_bulk` and `analyze` all passed the full text through unchanged.
+ * These tests are that repro, inverted — the mock now records everything the
+ * handler forwards, and every assertion is about what the SDK was *given*, not
+ * about what the tool said afterwards. A handler that redacted its reply but
+ * still forwarded the secret would pass a message-only test and fail these.
+ *
+ * Walrus storage is append-only and immutable, which is why the check has to be
+ * in front of the write: there is no delete to fall back on.
+ *
+ * The wait budget is zeroed so each tool returns at accept. That is the shortest
+ * path through each handler and it exercises the same pre-forward screen; the
+ * bounded-wait branch is covered by the `*-fast-return` files.
+ */
+process.env.MEMWAL_MCP_REMEMBER_WAIT_MS = "0";
+
+import assert from "node:assert/strict";
+import test, { type TestContext } from "node:test";
+import { Client } from "@modelcontextprotocol/sdk/client/index.js";
+import { InMemoryTransport } from "@modelcontextprotocol/sdk/inMemory.js";
+import type { MemWalSession } from "../auth.js";
+
+const { createMcpServer } = await import("../server.js");
+
+/** Everything the handlers handed to the SDK, in call order. */
+interface Forwarded {
+ remember: string[];
+ bulk: string[][];
+ analyze: string[];
+}
+
+function sessionWith(forwarded: Forwarded): MemWalSession {
+ return {
+ oauthScope: "memwal:read memwal:write",
+ namespace: "default",
+ memwal: {
+ async rememberAsync(text: string) {
+ forwarded.remember.push(text);
+ return { job_id: "job-1", status: "running" };
+ },
+ async rememberBulkAsync(items: Array<{ text: string }>) {
+ forwarded.bulk.push(items.map((i) => i.text));
+ return {
+ job_ids: items.map((_, i) => `bulk-job-${i + 1}`),
+ total: items.length,
+ status: "accepted",
+ };
+ },
+ async analyze(text: string) {
+ forwarded.analyze.push(text);
+ // Echo the passage back as one extracted fact, so a leak that
+ // slipped through would also show up in the reply.
+ return {
+ job_ids: ["analyze-job-1"],
+ facts: [{ text }],
+ fact_count: 1,
+ status: "accepted",
+ owner: "0xowner",
+ };
+ },
+ },
+ } as unknown as MemWalSession;
+}
+
+async function clientFor(session: MemWalSession, t: TestContext): Promise {
+ const [clientTransport, serverTransport] = InMemoryTransport.createLinkedPair();
+ const server = createMcpServer(session);
+ const client = new Client({ name: "write-path-redaction-test", version: "1.0.0" });
+ t.after(async () => {
+ await client.close();
+ await server.close();
+ });
+ await server.connect(serverTransport);
+ await client.connect(clientTransport);
+ return client;
+}
+
+function textOf(result: unknown): string {
+ return (result as { content: Array<{ text: string }> }).content
+ .map((c) => c.text)
+ .join("\n");
+}
+
+/** Everything the session was ever handed, flattened, plus the tool's reply. */
+function allForwarded(forwarded: Forwarded): string {
+ return [
+ ...forwarded.remember,
+ ...forwarded.bulk.flat(),
+ ...forwarded.analyze,
+ ].join("\n");
+}
+
+const PASSWORD = "hunter2";
+const MIXED =
+ "I prefer dark mode in every editor, and the staging db is " +
+ `postgres://admin:${PASSWORD}@db.internal:5432/app`;
+const PREFERENCE = "I prefer dark mode in every editor";
+
+// ── the reproduction, on all three write paths ──────────────────────────────
+
+test("memwal_remember keeps the preference and never forwards the password", async (t) => {
+ const forwarded: Forwarded = { remember: [], bulk: [], analyze: [] };
+ const client = await clientFor(sessionWith(forwarded), t);
+ const result = await client.callTool({
+ name: "memwal_remember",
+ arguments: { text: MIXED },
+ });
+
+ assert.equal(forwarded.remember.length, 1, "the write must still happen");
+ const sent = forwarded.remember[0];
+ assert.ok(!sent.includes(PASSWORD), "the password was forwarded to the SDK");
+ assert.ok(!sent.includes("admin:"), "the userinfo was forwarded to the SDK");
+ // The point of redacting rather than dropping: the fact survives.
+ assert.ok(sent.includes(PREFERENCE), "the preference was lost with the credential");
+ assert.ok(sent.includes("db.internal:5432/app"), "the host was lost too");
+
+ const text = textOf(result);
+ assert.ok(!text.includes(PASSWORD), "the reply echoed the password back");
+ assert.match(text, /url-credentials/);
+ assert.match(text, /credential span\(s\) were removed/);
+});
+
+test("memwal_remember_bulk screens every entry, and one bad entry does not sink the batch", async (t) => {
+ const forwarded: Forwarded = { remember: [], bulk: [], analyze: [] };
+ const client = await clientFor(sessionWith(forwarded), t);
+ const result = await client.callTool({
+ name: "memwal_remember_bulk",
+ arguments: {
+ facts: [
+ "I always use pnpm",
+ MIXED,
+ "sk-abcdefghijklmnopqrstuvwxyz0123",
+ "Deploy on Thursdays",
+ ],
+ },
+ });
+
+ assert.equal(forwarded.bulk.length, 1);
+ const sent = forwarded.bulk[0];
+ const joined = sent.join("\n");
+ assert.ok(!joined.includes(PASSWORD), "a password reached the SDK");
+ assert.ok(!joined.includes("sk-abcdefghijklmnopqrstuvwxyz0123"), "a key reached the SDK");
+
+ // Three survive: the two clean facts, plus the redacted mixed one. The
+ // bare key had no fact around it, so it is dropped rather than stored as
+ // an empty placeholder.
+ assert.equal(sent.length, 3);
+ assert.ok(sent.includes("I always use pnpm"), "a clean fact was altered or dropped");
+ assert.ok(sent.includes("Deploy on Thursdays"), "a clean fact was altered or dropped");
+ assert.ok(sent.some((s) => s.includes(PREFERENCE)));
+
+ const text = textOf(result);
+ assert.ok(!text.includes(PASSWORD));
+ assert.match(text, /NOT SAVED \(1\)/);
+ assert.match(text, /#3/, "the dropped entry must be identified by position");
+});
+
+test("memwal_analyze strips the passage before the extractor ever sees it", async (t) => {
+ const forwarded: Forwarded = { remember: [], bulk: [], analyze: [] };
+ const client = await clientFor(sessionWith(forwarded), t);
+ const result = await client.callTool({
+ name: "memwal_analyze",
+ arguments: {
+ text:
+ `${MIXED}\nAlso, my GitHub token is ghp_abcdefghijklmnopqrstuvwxyz0123456789 ` +
+ "and I review PRs on Fridays.",
+ },
+ });
+
+ assert.equal(forwarded.analyze.length, 1);
+ const sent = forwarded.analyze[0];
+ assert.ok(!sent.includes(PASSWORD), "the password reached the extractor LLM");
+ assert.ok(
+ !sent.includes("ghp_abcdefghijklmnopqrstuvwxyz0123456789"),
+ "the GitHub token reached the extractor LLM",
+ );
+ assert.ok(sent.includes(PREFERENCE));
+ assert.ok(sent.includes("review PRs on Fridays"));
+
+ const text = textOf(result);
+ assert.ok(!text.includes(PASSWORD));
+ assert.match(text, /credential span\(s\) were removed/);
+});
+
+// ── the rules that are not about credentials ────────────────────────────────
+
+test("an explicit do-not-save is honoured on every write path", async (t) => {
+ const forwarded: Forwarded = { remember: [], bulk: [], analyze: [] };
+ const client = await clientFor(sessionWith(forwarded), t);
+ const text = "My bank PIN is 4821 — don't save this.";
+
+ for (const [name, args] of [
+ ["memwal_remember", { text }],
+ ["memwal_remember_bulk", { facts: [text] }],
+ ["memwal_analyze", { text }],
+ ] as Array<[string, Record]>) {
+ const result = await client.callTool({ name, arguments: args });
+ assert.match(
+ textOf(result),
+ /NOT SAVED|Nothing was saved/,
+ `${name} did not say it withheld the text`,
+ );
+ }
+
+ assert.deepEqual(forwarded.remember, []);
+ assert.deepEqual(forwarded.bulk, []);
+ assert.deepEqual(forwarded.analyze, []);
+ assert.ok(!allForwarded(forwarded).includes("4821"));
+});
+
+test("pasted third-party content is not saved as a user fact", async (t) => {
+ const forwarded: Forwarded = { remember: [], bulk: [], analyze: [] };
+ const client = await clientFor(sessionWith(forwarded), t);
+ const pasted = "```\nERROR 500 from the vendor API\n at handler (index.js:42)\n```";
+
+ const remembered = await client.callTool({
+ name: "memwal_remember",
+ arguments: { text: pasted },
+ });
+ assert.match(textOf(remembered), /pasted third-party content/);
+
+ const analyzed = await client.callTool({
+ name: "memwal_analyze",
+ arguments: { text: pasted },
+ });
+ assert.match(textOf(analyzed), /pasted third-party content/);
+
+ assert.deepEqual(forwarded.remember, []);
+ assert.deepEqual(forwarded.analyze, []);
+});
+
+// ── the ordinary case, which must be untouched ──────────────────────────────
+
+test("a plain preference is forwarded byte-for-byte, with no note attached", async (t) => {
+ const forwarded: Forwarded = { remember: [], bulk: [], analyze: [] };
+ const client = await clientFor(sessionWith(forwarded), t);
+ const clean = "I always use pnpm, TypeScript strict mode, and deploy on Thursdays.";
+
+ const remembered = await client.callTool({
+ name: "memwal_remember",
+ arguments: { text: clean },
+ });
+ assert.deepEqual(forwarded.remember, [clean]);
+ assert.doesNotMatch(textOf(remembered), /redacted|NOT SAVED/i);
+
+ await client.callTool({
+ name: "memwal_remember_bulk",
+ arguments: { facts: [clean, "My coffee order is a matcha oat latte"] },
+ });
+ assert.deepEqual(forwarded.bulk, [
+ [clean, "My coffee order is a matcha oat latte"],
+ ]);
+
+ await client.callTool({ name: "memwal_analyze", arguments: { text: clean } });
+ assert.deepEqual(forwarded.analyze, [clean]);
+});
+
+test("the idempotency key is derived from what is actually written", async (t) => {
+ // Keyed on the original, a retry of a redacted fact would derive a key for
+ // text that was never sent — and the accept-timeout message promises a
+ // retry is safe.
+ const seen: string[] = [];
+ const session = {
+ oauthScope: "memwal:read memwal:write",
+ namespace: "default",
+ memwal: {
+ async rememberAsync(text: string, _ns: unknown, opts: { idempotencyKey: string }) {
+ seen.push(`${text}::${opts.idempotencyKey}`);
+ return { job_id: "job-1", status: "running" };
+ },
+ },
+ } as unknown as MemWalSession;
+
+ const client = await clientFor(session, t);
+ await client.callTool({ name: "memwal_remember", arguments: { text: MIXED } });
+ await client.callTool({ name: "memwal_remember", arguments: { text: MIXED } });
+
+ assert.equal(seen.length, 2);
+ assert.equal(seen[0], seen[1], "the same fact must derive the same key twice");
+ assert.ok(!seen[0].includes(PASSWORD));
+});
diff --git a/services/server/scripts/mcp/server.ts b/services/server/scripts/mcp/server.ts
index 78de98884..3caf5029a 100644
--- a/services/server/scripts/mcp/server.ts
+++ b/services/server/scripts/mcp/server.ts
@@ -3,6 +3,10 @@ import { createRequire } from "node:module";
import { applyAgentClientFromServer } from "./agent-client.js";
import type { MemWalSession } from "./auth.js";
import { registerTools } from "./tools/index.js";
+import {
+ SECRET_EXCLUSION_RULES,
+ AUTO_SAVE_OPT_IN_RULE,
+} from "./tools/memory-policy.js";
const requirePkg = createRequire(import.meta.url);
@@ -25,6 +29,10 @@ const PACKAGE_VERSION: string =
* `instructions` travels with `initialize`, before any `tools/list`, so lazy
* loading cannot strip it.
*
+ * The secret-exclusion and opt-in paragraphs are not written out here: they are
+ * pulled from tools/memory-policy.ts, the copy this package owns of the block
+ * shared with the MCP client and the plugin hooks (WALM-642).
+ *
* Keep roughly in sync with the plugin's SessionStart hook, which delivers
* equivalent text on the plugin install path:
* packages/mcp/plugin/scripts/on_session_start.mjs
@@ -50,6 +58,13 @@ const INSTRUCTIONS = [
"summary. Skip one-off tasks, the current file or bug, and small talk. Use",
"memwal_remember_bulk when several distinct facts arrived at once.",
"",
+ // WALM-642. Verbatim from tools/memory-policy.ts, which the tool
+ // descriptions state too, so instructions and descriptions cannot drift
+ // into telling the model two different things about secrets.
+ AUTO_SAVE_OPT_IN_RULE,
+ "",
+ SECRET_EXCLUSION_RULES,
+ "",
"A Walrus write takes roughly 30-60s, and memwal_remember and memwal_remember_bulk wait",
"for it: a successful call comes back with a blob_id, and that means the fact is stored.",
"",
diff --git a/services/server/scripts/mcp/tools/analyze.ts b/services/server/scripts/mcp/tools/analyze.ts
index 46cde62fc..418135da2 100644
--- a/services/server/scripts/mcp/tools/analyze.ts
+++ b/services/server/scripts/mcp/tools/analyze.ts
@@ -3,6 +3,8 @@ import type { McpServer } from "@modelcontextprotocol/sdk/server/mcp.js";
import type { MemWalSession } from "../auth.js";
import { TOOL_METADATA } from "./annotations.js";
import { wrapTool, explorerFooter } from "./util.js";
+import { SECRET_EXCLUSION_RULES, AUTO_SAVE_OPT_IN_RULE } from "./memory-policy.js";
+import { sanitizeFact, redactionNotice, refusalNotice } from "./redaction.js";
import {
REMEMBER_POLL_INTERVAL_MS,
REMEMBER_WAIT_MS,
@@ -17,7 +19,7 @@ const ANALYZE_INPUT = {
.string()
.min(1)
.describe(
- "Conversation transcript, note, or arbitrary text from which to extract memorable facts."
+ "Conversation transcript, note, or arbitrary text from which to extract memorable facts. Credential shapes (passwords, API keys, tokens, private keys, seed phrases, auth headers, URLs with an embedded user:password) are stripped from this text before it is sent for extraction, so no secret reaches the extractor or storage."
),
namespace: z
.string()
@@ -54,10 +56,29 @@ export function registerAnalyzeTool(
{
...TOOL_METADATA.memwal_analyze,
description:
- "Extract memorable facts from a longer passage of text (preferences, habits, biographical info, constraints) and save each as a separate Walrus Memory memory. Use this when you want MemWal's LLM to split the facts out of a transcript or notes for you; if you already know the exact facts, use memwal_remember or memwal_remember_bulk instead. The extracted facts come back immediately; if the result says the writes are still in flight it carries job_ids — confirm them with memwal_remember_status rather than telling the user they are saved.",
+ "Extract memorable facts from a longer passage of text (preferences, habits, biographical info, constraints) and save each as a separate Walrus Memory memory. Use this when you want MemWal's LLM to split the facts out of a transcript or notes for you; if you already know the exact facts, use memwal_remember or memwal_remember_bulk instead. The extracted facts come back immediately; if the result says the writes are still in flight it carries job_ids — confirm them with memwal_remember_status rather than telling the user they are saved. " +
+ AUTO_SAVE_OPT_IN_RULE +
+ " " +
+ SECRET_EXCLUSION_RULES +
+ " This tool forwards a whole passage, so it is the easiest way to leak a credential that happened to sit next to a fact: the passage is stripped of credential shapes before it is sent for extraction, and a passage that is only a secret, or that the user asked not to save, is not sent at all.",
inputSchema: ANALYZE_INPUT,
},
wrapTool<{ text: string; namespace?: string }>(session, "memwal_analyze", async ({ text, namespace }) => {
+ // Runs BEFORE the passage reaches the SDK, and therefore before it
+ // reaches the extractor LLM. Everything this tool stores is derived
+ // from this text, so a credential left in it can be copied into any
+ // number of extracted facts — on append-only storage (WALM-642).
+ const safe = sanitizeFact(text);
+ if (safe.refusal) {
+ return {
+ content: [
+ { type: "text" as const, text: refusalNotice(safe.refusal) },
+ ],
+ };
+ }
+ const notice = redactionNotice(safe.kinds, safe.count);
+ const safeText = safe.text;
+
// `analyze` (not `analyzeAndWait`) returns once extraction is done
// and every fact has a queued job, which is the point this tool can
// usefully answer at.
@@ -66,7 +87,7 @@ export function registerAnalyzeTool(
// that never reached the handler, so no job row can exist yet
// to duplicate.
withRelayerRetry(
- () => session.memwal.analyze(text, namespace),
+ () => session.memwal.analyze(safeText, namespace),
"analyze this text",
),
"memwal_analyze extraction",
@@ -78,6 +99,9 @@ export function registerAnalyzeTool(
{ idempotent: false, deadlineMs: 60_000 },
);
+ const withNotice = (body: string) =>
+ notice ? `${body}\n\n${notice}` : body;
+
const facts = accepted.facts ?? [];
// Nothing to wait on, and nothing to confirm later. Say so plainly
// rather than handing back an empty job list.
@@ -86,7 +110,9 @@ export function registerAnalyzeTool(
content: [
{
type: "text" as const,
- text: `Extracted 0 facts from that text — nothing was saved.`,
+ text: withNotice(
+ `Extracted 0 facts from that text — nothing was saved.`,
+ ),
},
],
};
@@ -107,7 +133,9 @@ export function registerAnalyzeTool(
content: [
{
type: "text" as const,
- text: `${extracted}\n\n${pendingBulkMessage(entries, waitedMs)}`,
+ text: withNotice(
+ `${extracted}\n\n${pendingBulkMessage(entries, waitedMs)}`,
+ ),
},
],
});
@@ -153,7 +181,9 @@ export function registerAnalyzeTool(
content: [
{
type: "text" as const,
- text: `${summary}\n\n${lines.join("\n")}${stragglers}${footer}`,
+ text: withNotice(
+ `${summary}\n\n${lines.join("\n")}${stragglers}${footer}`,
+ ),
},
],
};
diff --git a/services/server/scripts/mcp/tools/memory-policy.ts b/services/server/scripts/mcp/tools/memory-policy.ts
new file mode 100644
index 000000000..018fba9a2
--- /dev/null
+++ b/services/server/scripts/mcp/tools/memory-policy.ts
@@ -0,0 +1,80 @@
+/**
+ * Shared automatic-memory policy — the relayer sidecar's copy.
+ *
+ * Consumed by the live tool descriptions (`remember`, `remember-bulk`,
+ * `analyze`) and by the sidecar's `initialize` instructions in `server.ts`.
+ * Descriptions are the layer every MCP client sees once tools are loaded, so
+ * the rules have to be stated here too and not only in the client package.
+ *
+ * WALM-642.
+ */
+
+// ─── memwal:policy-block:start ───────────────────────────────────────────────
+// WALM-642. The lines between these two markers are BYTE-IDENTICAL in three
+// files that cannot import one another, because the three packages have no
+// workspace link:
+//
+// packages/mcp/src/memory-policy.ts — MCP client: initialize
+// instructions + the
+// cold-start tools/list
+// packages/mcp/plugin/scripts/lib/memory-policy.mjs — plugin hooks: the
+// guidance injected at
+// SessionStart /
+// UserPromptSubmit /
+// PostToolUse
+// services/server/scripts/mcp/tools/memory-policy.ts — relayer sidecar: the
+// live tool descriptions
+//
+// The duplication is deliberate and pinned: `memory-policy-sync` tests on both
+// sides extract this block from each file and compare the bytes, so editing one
+// copy fails the suite until the other two match. Edit the block, then copy it
+// verbatim — markers included — into the other two files.
+
+/**
+ * The secret-exclusion and do-not-save rules, stated verbatim by every
+ * automatic-save surface.
+ *
+ * These are model-facing rules, not enforcement. The programmatic backstop is
+ * the redactor in the relayer sidecar's write path
+ * (services/server/scripts/mcp/tools/redaction.ts), which runs before any text
+ * reaches the SDK.
+ */
+export const SECRET_EXCLUSION_RULES = [
+ "NEVER save a credential, even when it sits next to something worth saving: passwords,",
+ "API keys, access or refresh tokens, private keys, seed or recovery phrases, authorization",
+ "headers, session cookies, and connection strings or URLs that embed a user:password.",
+ "When a message mixes a preference with a credential, save the preference alone and leave",
+ "the credential out; never store the line verbatim.",
+ "If the user says not to save something ('don't save this', 'off the record', or the same",
+ "in any language), do not save it, and do not save a paraphrase of it either.",
+ "Do not store quoted or pasted third-party material — log excerpts, code, articles, other",
+ "people's messages — as if it were a fact about this user. Save only what the user is",
+ "telling you about themselves or their work, in your own words.",
+].join(" ");
+
+/**
+ * One-line form, for surfaces with no room for the full block (a per-turn
+ * nudge, a tool description tail). It is a reminder of the block above, never a
+ * replacement for it: any surface that drives an automatic save states the full
+ * `SECRET_EXCLUSION_RULES`.
+ */
+export const SECRET_EXCLUSION_SUMMARY = [
+ "Never save passwords, keys, tokens or other credentials — not even beside a fact worth",
+ "saving; honour an explicit 'do not save this'; never store pasted third-party content as",
+ "a fact about the user.",
+].join(" ");
+
+/**
+ * Automatic saving is opt-in, and this is the sentence that says so. A direct
+ * request from the user ("remember that ...") is never gated by it — the gate
+ * is only on saving something the user did not ask you to save.
+ */
+export const AUTO_SAVE_OPT_IN_RULE = [
+ "Saving something the user did not ask you to save is OFF unless they have turned automatic",
+ "memory on (`memwal-mcp auto-save on`, or MEMWAL_AUTO_SAVE=1). When it is off, save only what",
+ "the user asks you to save in that turn, and do not offer to turn it on more than once.",
+].join(" ");
+
+/** Bumped whenever the text above changes, so a stale copy is identifiable. */
+export const MEMORY_POLICY_VERSION = "2026-09-17.1";
+// ─── memwal:policy-block:end ─────────────────────────────────────────────────
diff --git a/services/server/scripts/mcp/tools/redaction.ts b/services/server/scripts/mcp/tools/redaction.ts
new file mode 100644
index 000000000..33a14f4cc
--- /dev/null
+++ b/services/server/scripts/mcp/tools/redaction.ts
@@ -0,0 +1,365 @@
+/**
+ * Credential redaction for the memory write path (WALM-642).
+ *
+ * Every tool that forwards free text to the SDK — `memwal_remember`,
+ * `memwal_remember_bulk`, `memwal_analyze` — runs its input through
+ * `sanitizeFact` FIRST. The model-facing rules in `memory-policy.ts` are
+ * guidance; this module is the backstop that does not depend on a model having
+ * read them. Walrus storage is append-only and immutable: a secret that reaches
+ * it cannot be deleted, so the check has to sit in front of the write rather
+ * than behind it.
+ *
+ * Two shapes of answer:
+ *
+ * - REDACT (the common case, and what the ticket asks for): a message that
+ * mixes a durable preference with a credential keeps the preference and
+ * loses only the credential span, replaced by `[redacted:]`. Dropping
+ * the whole fact would lose the thing the user actually wanted stored.
+ * - REFUSE: nothing safe is left, the user said not to save it, or the text
+ * is plainly pasted third-party material. The caller forwards nothing.
+ *
+ * ── False positives, deliberately ───────────────────────────────────────────
+ * The patterns below are SHAPE-based, not entropy-based. That is a choice, and
+ * it costs recall:
+ *
+ * - There is no free-standing "long random-looking string" rule. MemWal's own
+ * durable facts are exactly that shape — Walrus blob ids (43-char
+ * base64url), Sui object and account ids (`0x` + 64 hex), git SHAs (40 hex),
+ * content digests. A generic high-entropy rule would redact the product's
+ * primary nouns. The one entropy rule that survived
+ * (`HIGH_ENTROPY_SECRET`) demands ≥64 characters AND mixed case AND a
+ * digit, which excludes every one of those (all-lowercase hex, or shorter),
+ * while still catching a raw base64 key blob.
+ * - `password:`-style assignments redact whatever follows the separator, so
+ * "password: ask Marta" loses "ask Marta". Over-redacting a sentence about
+ * a credential is cheap; under-redacting the credential is permanent.
+ * - Quoted-content detection only fires on unambiguous pastes (a fenced
+ * block, a multi-line `>` quotation, or a ≥200-char fully quoted passage).
+ * A short quoted sentence is left to the model-facing rules, because an
+ * agent legitimately quotes the user's own words back when saving a fact.
+ *
+ * Nothing here logs, echoes, or returns the matched secret. Callers get the
+ * redacted text and the KINDS that were removed — never the values.
+ */
+
+export type RedactionKind =
+ | "url-credentials"
+ | "vendor-api-key"
+ | "private-key-block"
+ | "jwt"
+ | "credential-assignment"
+ | "auth-header"
+ | "seed-phrase"
+ | "high-entropy-secret";
+
+/** Why a text was refused outright instead of redacted. */
+export type RefusalReason =
+ /** The user said not to save it. */
+ | "no-save-directive"
+ /** After redaction there was no fact left — the text was only a secret. */
+ | "credential-only"
+ /** Pasted third-party material, not a fact about this user. */
+ | "pasted-content";
+
+export interface SanitizedText {
+ /** Text safe to forward. Empty when `refusal` is set. */
+ text: string;
+ /** True when the text was changed or refused. */
+ changed: boolean;
+ /** Kinds removed, first-seen order. Never contains a secret value. */
+ kinds: RedactionKind[];
+ /** Number of spans replaced. */
+ count: number;
+ /** Set when the caller must forward nothing at all. */
+ refusal?: RefusalReason;
+}
+
+function placeholder(kind: RedactionKind): string {
+ return `[redacted:${kind}]`;
+}
+
+/**
+ * An explicit instruction not to save, from the user.
+ *
+ * Deliberately demands a demonstrative object ("this", "that", "it"): without
+ * it, "remember that I don't save screenshots to the Desktop" — a perfectly
+ * good durable preference — would be refused as a do-not-save directive.
+ */
+const NO_SAVE_DIRECTIVE =
+ /\b(?:do\s+not|don'?t|dont|never|please\s+do\s+not|please\s+don'?t)\s+(?:save|store|remember|record|keep|persist|log)\s+(?:this|that|it|these|those|any\s+of\s+(?:this|that|it))\b/i;
+
+/** The idiom, which carries the same instruction without naming saving. */
+const OFF_THE_RECORD = /\boff[-\s]the[-\s]record\b/i;
+
+/**
+ * PEM private key blocks. The second pattern is not redundant: a paste that was
+ * cut off mid-key has a BEGIN line and no END line, and without the open-ended
+ * form the key body would survive untouched.
+ */
+const PEM_CLOSED =
+ /-----BEGIN [A-Z0-9 ]*PRIVATE KEY-----[\s\S]*?-----END [A-Z0-9 ]*PRIVATE KEY-----/g;
+const PEM_OPEN = /-----BEGIN [A-Z0-9 ]*PRIVATE KEY-----[\s\S]*/g;
+
+/**
+ * `scheme://user:password@host` — the exact shape in the WALM-642 repro, and
+ * the shape of every connection string that carries its own credentials
+ * (`postgres://`, `mongodb+srv://`, `amqp://`, `redis://`, ...).
+ *
+ * Only the userinfo is replaced: scheme, host, port and path are the part of a
+ * connection string worth remembering.
+ */
+const URL_USERINFO = /([A-Za-z][A-Za-z0-9+.-]*:\/\/)([^\s/@:]+):([^\s/@]+)@/g;
+
+/**
+ * Authorization / cookie headers, value dropped, header name kept.
+ *
+ * The cookie value stops at whitespace rather than at the end of the line, and
+ * then continues across `;`-separated pairs. A real header (`Cookie: a=1; b=2`)
+ * is matched whole; a header quoted mid-sentence loses the cookie and not the
+ * rest of the sentence. Cookie values are token-shaped by spec, so the only
+ * thing this gives up is a value that contains a raw space — which is not
+ * legal in one anyway.
+ */
+const AUTH_HEADER =
+ /\b((?:proxy-)?authorization)(\s*[:=]\s*)(?:bearer|basic|token|digest)?\s*\S+/gi;
+const COOKIE_HEADER =
+ /\b((?:set-)?cookie)(\s*[:=]\s*)[^\s;]+(?:\s*;\s*[^\s;]+)*/gi;
+
+/**
+ * `key=value` / `key: value` where the key names a credential.
+ *
+ * The separator must follow the keyword immediately, which is what keeps
+ * ordinary prose out: "my password manager is 1Password" has no separator after
+ * "password" and does not match.
+ */
+const CREDENTIAL_ASSIGNMENT =
+ /\b(passwords?|passwd|pwd|passphrases?|api[_-]?keys?|apikeys?|secret[_-]?keys?|client[_-]?secrets?|secrets?|access[_-]?tokens?|refresh[_-]?tokens?|auth[_-]?tokens?|bearer[_-]?tokens?|tokens?|private[_-]?keys?|credentials?)(\s*[:=]\s*)("[^"\n]*"|'[^'\n]*'|`[^`\n]*`|[^\s,;]+)/gi;
+
+/**
+ * Vendor-prefixed keys. Each prefix is issued by exactly one service and never
+ * appears at the head of ordinary text, so these are the highest-confidence
+ * patterns in the file — no context needed.
+ */
+const VENDOR_KEYS: RegExp[] = [
+ /\bsk-ant-[A-Za-z0-9_-]{16,}/g, // Anthropic
+ /\bsk-proj-[A-Za-z0-9_-]{16,}/g, // OpenAI project
+ /\bsk-[A-Za-z0-9]{20,}/g, // OpenAI classic
+ /\b(?:ghp|gho|ghu|ghs|ghr)_[A-Za-z0-9]{20,}/g, // GitHub
+ /\bgithub_pat_[A-Za-z0-9_]{20,}/g, // GitHub fine-grained
+ /\b(?:AKIA|ASIA)[0-9A-Z]{16}\b/g, // AWS access key id
+ /\bxox[baprs]-[A-Za-z0-9-]{10,}/g, // Slack
+ /\bAIza[0-9A-Za-z_-]{35}\b/g, // Google API
+ /\bglpat-[A-Za-z0-9_-]{16,}/g, // GitLab
+ /\bnpm_[A-Za-z0-9]{36}\b/g, // npm
+ /\bSG\.[A-Za-z0-9_-]{16,}\.[A-Za-z0-9_-]{16,}/g, // SendGrid
+ /\b[sprk]k_(?:live|test)_[A-Za-z0-9]{16,}/g, // Stripe
+ /\bshp(?:at|ss|ca|pa)_[a-fA-F0-9]{32}\b/g, // Shopify
+ /\bdop_v1_[a-f0-9]{64}\b/g, // DigitalOcean
+ /\bhf_[A-Za-z0-9]{30,}\b/g, // Hugging Face
+];
+
+/** Three base64url segments — a signed JWT, whatever it encodes. */
+const JWT = /\beyJ[A-Za-z0-9_-]{8,}\.[A-Za-z0-9_-]{8,}\.[A-Za-z0-9_-]{8,}/g;
+
+/**
+ * A BIP-39 mnemonic, gated on the user naming it.
+ *
+ * Twelve consecutive lowercase words is also what an ordinary sentence looks
+ * like, so the words alone prove nothing; "seed phrase" / "mnemonic" /
+ * "recovery phrase" in front of them is what makes the match safe. A mnemonic
+ * pasted with no label is left to `SECRET_EXCLUSION_RULES` — carrying the
+ * 2048-word list here to catch it would be a lot of weight for a case the model
+ * rules already cover.
+ */
+const SEED_PHRASE =
+ /\b((?:seed|recovery|secret|mnemonic)\s+(?:phrase|words)|mnemonic)(\s*(?:is|are|:|=)?\s*)((?:[a-z]{3,8}[ \t]+){11,23}[a-z]{3,8})\b/gi;
+
+/**
+ * A ≥64-character base64/base64url run carrying lower case, upper case AND a
+ * digit. See the false-positive note at the top: the three conditions together
+ * are what exclude blob ids, Sui addresses, git SHAs and hex digests, all of
+ * which are shorter, single-case, or both.
+ */
+const HIGH_ENTROPY_CANDIDATE = /\b[A-Za-z0-9+/_-]{64,}={0,2}/g;
+
+function looksLikeSecretBlob(token: string): boolean {
+ const body = token.replace(/=+$/, "");
+ if (body.length < 64) return false;
+ if (!/[a-z]/.test(body)) return false;
+ if (!/[A-Z]/.test(body)) return false;
+ if (!/[0-9]/.test(body)) return false;
+ // Hex is single-case by convention and covers digests, SHAs and Sui ids;
+ // a mixed-case hex string is still far more likely a digest than a key.
+ if (/^[0-9a-fA-F]+$/.test(body)) return false;
+ return true;
+}
+
+/**
+ * Unambiguously pasted third-party material.
+ *
+ * Narrow on purpose — see the false-positive note. An agent saving a fact
+ * routinely quotes the user's own sentence, so a short quoted string is NOT
+ * treated as a paste.
+ */
+function isPastedContent(text: string): boolean {
+ const s = text.trim();
+ if (/^```[\s\S]*```$/.test(s)) return true;
+ const lines = s.split(/\r?\n/).filter((l) => l.trim() !== "");
+ if (lines.length >= 2 && lines.every((l) => /^\s*>/.test(l))) return true;
+ if (s.length >= 200 && /^["“][\s\S]*["”]$/.test(s)) return true;
+ return false;
+}
+
+/**
+ * Is there still a fact here once the placeholders are taken out?
+ *
+ * Three words of two or more letters. "sk-live-..." on its own redacts to a
+ * bare placeholder and has none; "I keep the staging key in 1Password,
+ * api_key=..." keeps its sentence and has plenty.
+ */
+function hasSalvageableContent(text: string): boolean {
+ const withoutPlaceholders = text.replace(/\[redacted:[a-z-]+\]/g, " ");
+ const words = withoutPlaceholders.match(/[\p{L}\p{N}]{2,}/gu) ?? [];
+ return words.length >= 3;
+}
+
+/**
+ * Strip credentials from one piece of text before it is forwarded to the SDK.
+ *
+ * Pure, synchronous and side-effect free: no logging, no I/O. The secret exists
+ * only in the caller's argument and never leaves this function.
+ */
+export function sanitizeFact(input: string): SanitizedText {
+ const original = input ?? "";
+
+ if (NO_SAVE_DIRECTIVE.test(original) || OFF_THE_RECORD.test(original)) {
+ return {
+ text: "",
+ changed: true,
+ kinds: [],
+ count: 0,
+ refusal: "no-save-directive",
+ };
+ }
+
+ if (isPastedContent(original)) {
+ return {
+ text: "",
+ changed: true,
+ kinds: [],
+ count: 0,
+ refusal: "pasted-content",
+ };
+ }
+
+ const kinds: RedactionKind[] = [];
+ let count = 0;
+ const hit = (kind: RedactionKind) => {
+ if (!kinds.includes(kind)) kinds.push(kind);
+ count += 1;
+ return placeholder(kind);
+ };
+
+ let text = original;
+
+ // PEM first: it spans lines, and running the line-oriented patterns over a
+ // key body would shred it into several partial matches instead of one.
+ text = text.replace(PEM_CLOSED, () => hit("private-key-block"));
+ text = text.replace(PEM_OPEN, () => hit("private-key-block"));
+
+ text = text.replace(URL_USERINFO, (_m, scheme: string) => {
+ hit("url-credentials");
+ return `${scheme}${placeholder("url-credentials")}@`;
+ });
+
+ text = text.replace(AUTH_HEADER, (_m, name: string, sep: string) => {
+ hit("auth-header");
+ return `${name}${sep}${placeholder("auth-header")}`;
+ });
+ text = text.replace(COOKIE_HEADER, (_m, name: string, sep: string) => {
+ hit("auth-header");
+ return `${name}${sep}${placeholder("auth-header")}`;
+ });
+
+ // Before the vendor patterns, so `api_key=sk-...` is reported once as an
+ // assignment rather than twice.
+ text = text.replace(CREDENTIAL_ASSIGNMENT, (_m, key: string, sep: string) => {
+ hit("credential-assignment");
+ return `${key}${sep}${placeholder("credential-assignment")}`;
+ });
+
+ for (const pattern of VENDOR_KEYS) {
+ text = text.replace(pattern, () => hit("vendor-api-key"));
+ }
+
+ text = text.replace(JWT, () => hit("jwt"));
+
+ text = text.replace(SEED_PHRASE, (_m, label: string, sep: string) => {
+ hit("seed-phrase");
+ return `${label}${sep}${placeholder("seed-phrase")}`;
+ });
+
+ text = text.replace(HIGH_ENTROPY_CANDIDATE, (token: string) =>
+ looksLikeSecretBlob(token) ? hit("high-entropy-secret") : token,
+ );
+
+ if (count === 0) {
+ return { text: original, changed: false, kinds: [], count: 0 };
+ }
+
+ // Collapse the whitespace a removed block leaves behind, so the stored fact
+ // does not carry the shape of what was taken out.
+ text = text.replace(/[ \t]{2,}/g, " ").replace(/\n{3,}/g, "\n\n").trim();
+
+ if (!hasSalvageableContent(text)) {
+ return {
+ text: "",
+ changed: true,
+ kinds,
+ count,
+ refusal: "credential-only",
+ };
+ }
+
+ return { text, changed: true, kinds, count };
+}
+
+/** Human-readable reason, for the note handed back to the agent. */
+export function refusalMessage(reason: RefusalReason): string {
+ switch (reason) {
+ case "no-save-directive":
+ return "the text says not to save it";
+ case "credential-only":
+ return "the text was a credential with no fact around it";
+ case "pasted-content":
+ return "the text is pasted third-party content, not a fact about the user";
+ }
+}
+
+/**
+ * The line appended to a tool result when something was removed.
+ *
+ * Names the kinds and nothing else — an agent needs to know a redaction
+ * happened so it does not tell the user the whole line was stored, and it never
+ * needs the value back.
+ */
+export function redactionNotice(kinds: RedactionKind[], count: number): string {
+ if (count === 0) return "";
+ return (
+ `Note: ${count} credential span(s) were removed before saving ` +
+ `(${kinds.join(", ")}). What was stored is the redacted text — the secret ` +
+ `was never sent to Walrus Memory and is not logged. Tell the user the ` +
+ `credential was left out; do not re-send it.`
+ );
+}
+
+/**
+ * The line for a text that was not saved at all.
+ */
+export function refusalNotice(reason: RefusalReason): string {
+ return (
+ `NOT SAVED: ${refusalMessage(reason)}. Nothing was written to Walrus ` +
+ `Memory. Do not retry this text — if there is a durable fact in it, ` +
+ `restate the fact without the sensitive part and save that instead.`
+ );
+}
diff --git a/services/server/scripts/mcp/tools/remember-bulk.ts b/services/server/scripts/mcp/tools/remember-bulk.ts
index fd4a47c40..7be2e80c8 100644
--- a/services/server/scripts/mcp/tools/remember-bulk.ts
+++ b/services/server/scripts/mcp/tools/remember-bulk.ts
@@ -3,6 +3,13 @@ import type { McpServer } from "@modelcontextprotocol/sdk/server/mcp.js";
import type { MemWalSession } from "../auth.js";
import { TOOL_METADATA } from "./annotations.js";
import { wrapTool, explorerFooter } from "./util.js";
+import { SECRET_EXCLUSION_RULES, AUTO_SAVE_OPT_IN_RULE } from "./memory-policy.js";
+import {
+ sanitizeFact,
+ redactionNotice,
+ refusalMessage,
+ type RedactionKind,
+} from "./redaction.js";
import {
REMEMBER_POLL_INTERVAL_MS,
REMEMBER_WAIT_MS,
@@ -18,7 +25,7 @@ const REMEMBER_BULK_INPUT = {
.min(1)
.max(20)
.describe(
- "Array of complete, detailed fact statements to save (1-20). Each entry is one full fact — do not summarize or merge them."
+ "Array of complete, detailed fact statements to save (1-20). Each entry is one full fact — do not summarize or merge them. Leave credentials out: passwords, API keys, tokens, private keys, seed phrases, auth headers and URLs with an embedded user:password are stripped from each entry before the write and never stored."
),
namespace: z
.string()
@@ -53,11 +60,64 @@ export function registerRememberBulkTool(
{
...TOOL_METADATA.memwal_remember_bulk,
description:
- "Save multiple durable facts in one call. Use when you learned several distinct facts at once (onboarding details, a list of preferences, decisions from a discussion). Pass an array of complete fact statements (max 20) — do not summarize. Prefer this over repeated memwal_remember calls. A Walrus write takes 30-60s and this call waits for them, so a success carries blob_ids and means the facts are stored. If they outrun that budget you get job_ids and the facts are NOT yet saved — say they are being saved rather than stored, and resolve them with memwal_remember_status.",
+ "Save multiple durable facts in one call. Use when you learned several distinct facts at once (onboarding details, a list of preferences, decisions from a discussion). Pass an array of complete fact statements (max 20) — do not summarize. Prefer this over repeated memwal_remember calls. A Walrus write takes 30-60s and this call waits for them, so a success carries blob_ids and means the facts are stored. If they outrun that budget you get job_ids and the facts are NOT yet saved — say they are being saved rather than stored, and resolve them with memwal_remember_status. " +
+ AUTO_SAVE_OPT_IN_RULE +
+ " " +
+ SECRET_EXCLUSION_RULES +
+ " Walrus storage is append-only: a stored secret cannot be deleted, so each entry is stripped of credential shapes before writing and entries that are nothing but a secret are dropped, with a note saying which.",
inputSchema: REMEMBER_BULK_INPUT,
},
wrapTool<{ facts: string[]; namespace?: string }>(session, "memwal_remember_bulk", async ({ facts, namespace }) => {
- const items = facts.map((text) => ({ text, namespace }));
+ // Every entry is sanitized BEFORE the batch is handed to the SDK.
+ // Walrus is append-only, so a credential that lands cannot be
+ // taken back (WALM-642). An entry that survives keeps its safe
+ // part; an entry that is only a secret — or that the user asked
+ // not to save — is dropped from the batch rather than the whole
+ // call failing, so the other facts still land.
+ const screened = facts.map((text, index) => ({
+ index,
+ result: sanitizeFact(text),
+ }));
+ const kept = screened.filter((s) => !s.result.refusal);
+ const dropped = screened.filter((s) => s.result.refusal);
+ const droppedNote = dropped.length
+ ? `\n\nNOT SAVED (${dropped.length}): ` +
+ dropped
+ .map((d) => `#${d.index + 1} — ${refusalMessage(d.result.refusal!)}`)
+ .join("; ") +
+ ". Do not re-send those; restate any durable fact without the sensitive part instead."
+ : "";
+
+ if (kept.length === 0) {
+ return {
+ content: [
+ {
+ type: "text" as const,
+ text:
+ `Nothing was saved to Walrus Memory: every fact in this batch was ` +
+ `withheld.${droppedNote}`,
+ },
+ ],
+ };
+ }
+
+ const redactedKinds: RedactionKind[] = [];
+ let redactedCount = 0;
+ for (const s of kept) {
+ redactedCount += s.result.count;
+ for (const kind of s.result.kinds) {
+ if (!redactedKinds.includes(kind)) redactedKinds.push(kind);
+ }
+ }
+ const policyNote =
+ [redactionNotice(redactedKinds, redactedCount), droppedNote.trim()]
+ .filter(Boolean)
+ .join("\n\n");
+
+ // The only texts anything below may echo or forward. The originals
+ // still hold the secret and must not reach a result line.
+ const safeFacts = kept.map((s) => s.result.text);
+ const items = safeFacts.map((text) => ({ text, namespace }));
// Two steps rather than `rememberBulkAndWait`, for the same reason
// `memwal_remember` splits them: acceptance is the part that must
// succeed, the wait is a courtesy we cut short.
@@ -77,14 +137,17 @@ export function registerRememberBulkTool(
// it, and the relayer returns job_ids in input order.
const entries = accepted.job_ids.map((jobId, i) => ({
jobId,
- text: facts[i] ?? "",
+ text: safeFacts[i] ?? "",
}));
+ const withNotice = (body: string) =>
+ policyNote ? `${body}\n\n${policyNote}` : body;
+
const pending = (waitedMs: number) => ({
content: [
{
type: "text" as const,
- text: pendingBulkMessage(entries, waitedMs),
+ text: withNotice(pendingBulkMessage(entries, waitedMs)),
},
],
});
@@ -108,7 +171,7 @@ export function registerRememberBulkTool(
const waitedMs = Date.now() - startedAt;
const unfinished = result.results.flatMap((r, i) =>
- r.status === "timeout" ? [{ jobId: r.id, text: facts[i] ?? "" }] : []
+ r.status === "timeout" ? [{ jobId: r.id, text: safeFacts[i] ?? "" }] : []
);
// Nothing landed inside the budget — the ordinary outcome when the
// queue is busy. Say so once rather than printing N timeout rows.
@@ -118,7 +181,7 @@ export function registerRememberBulkTool(
// Label each result with its source fact by index. The SDK
// returns results in input order, but guard against a length /
// ordering mismatch so we never print "— undefined".
- const text = facts[i] ?? "";
+ const text = safeFacts[i] ?? "";
const blob = r.blob_id ? ` blob_id=${r.blob_id}` : "";
const err = r.error ? ` error=${r.error}` : "";
// `timeout` is not a failure — the write is still running and
@@ -150,10 +213,11 @@ export function registerRememberBulkTool(
content: [
{
type: "text",
- text:
+ text: withNotice(
(lines.length > 0
? `${summary}\n\n${lines.join("\n")}${footer}`
: `${summary}${footer}`) + tail,
+ ),
},
],
};
diff --git a/services/server/scripts/mcp/tools/remember.ts b/services/server/scripts/mcp/tools/remember.ts
index 1c31bb6a5..3360f6d0d 100644
--- a/services/server/scripts/mcp/tools/remember.ts
+++ b/services/server/scripts/mcp/tools/remember.ts
@@ -3,6 +3,8 @@ import type { McpServer } from "@modelcontextprotocol/sdk/server/mcp.js";
import type { MemWalSession } from "../auth.js";
import { TOOL_METADATA } from "./annotations.js";
import { wrapTool, walruscanBlobUrl } from "./util.js";
+import { SECRET_EXCLUSION_RULES, AUTO_SAVE_OPT_IN_RULE } from "./memory-policy.js";
+import { sanitizeFact, redactionNotice, refusalNotice } from "./redaction.js";
import {
REMEMBER_WAIT_MS,
REMEMBER_POLL_INTERVAL_MS,
@@ -20,7 +22,7 @@ const REMEMBER_INPUT = {
.string()
.min(1)
.describe(
- "The full, detailed fact to save. Pass the COMPLETE statement — do not summarize."
+ "The full, detailed fact to save. Pass the COMPLETE statement — do not summarize. Leave credentials out: passwords, API keys, tokens, private keys, seed phrases, auth headers and URLs with an embedded user:password are stripped before the write and never stored."
),
namespace: z
.string()
@@ -54,21 +56,40 @@ export function registerRememberTool(
{
...TOOL_METADATA.memwal_remember,
description:
- "Save a durable fact about the user or project to their Walrus Memory. Call this PROACTIVELY whenever the user states a preference, decision, constraint, correction, identity detail, or recurring workflow — even if they did not say 'remember this'. Skip one-off tasks, the current file or bug, and small talk. Pass the full statement; do not summarize. To save several facts at once, use memwal_remember_bulk instead. A Walrus write takes 30-60s and this call waits for it, so a success carries a blob_id and means the fact is stored. If it outruns that budget you get a job_id and the fact is NOT yet saved — say so rather than claiming it is stored, and resolve it with memwal_remember_status.",
+ "Save a durable fact about the user or project to their Walrus Memory. Call this whenever the user states a preference, decision, constraint, correction, identity detail, or recurring workflow — PROACTIVELY, without being asked, when they have turned automatic memory on. Skip one-off tasks, the current file or bug, and small talk. Pass the full statement; do not summarize. To save several facts at once, use memwal_remember_bulk instead. A Walrus write takes 30-60s and this call waits for it, so a success carries a blob_id and means the fact is stored. If it outruns that budget you get a job_id and the fact is NOT yet saved — say so rather than claiming it is stored, and resolve it with memwal_remember_status. " +
+ AUTO_SAVE_OPT_IN_RULE +
+ " " +
+ SECRET_EXCLUSION_RULES +
+ " Walrus storage is append-only: a stored secret cannot be deleted, so this tool strips credential shapes from the text before writing and tells you what it removed.",
inputSchema: REMEMBER_INPUT,
},
wrapTool<{ text: string; namespace?: string }>(session, "memwal_remember", async ({ text, namespace }) => {
+ // Runs BEFORE anything reaches the SDK. Walrus is append-only, so a
+ // credential that gets written cannot be taken back (WALM-642).
+ const safe = sanitizeFact(text);
+ if (safe.refusal) {
+ return {
+ content: [
+ { type: "text" as const, text: refusalNotice(safe.refusal) },
+ ],
+ };
+ }
+ const notice = redactionNotice(safe.kinds, safe.count);
+ const safeText = safe.text;
+
// Two steps rather than `rememberAndWait`, because the accept and
// the wait need separate budgets: acceptance is the part that
// must succeed, the wait is a courtesy we cut short.
const accepted = await withAcceptDeadline(
withRelayerRetry(
() =>
- session.memwal.rememberAsync(text, namespace, {
+ session.memwal.rememberAsync(safeText, namespace, {
// Ours, not the SDK's random one — see
// derivedIdempotencyKey. This is what makes the
// accept-timeout message's retry promise true.
- idempotencyKey: derivedIdempotencyKey(namespace, text),
+ // Keyed on the REDACTED text, so a retry of the
+ // same fact derives the same key.
+ idempotencyKey: derivedIdempotencyKey(namespace, safeText),
}),
"save this fact",
),
@@ -76,11 +97,14 @@ export function registerRememberTool(
{ idempotent: true },
);
+ const withNotice = (body: string) =>
+ notice ? `${body}\n\n${notice}` : body;
+
const pending = (waitedMs: number) => ({
content: [
{
type: "text" as const,
- text: pendingMessage(accepted.job_id, waitedMs),
+ text: withNotice(pendingMessage(accepted.job_id, waitedMs)),
},
],
});
@@ -102,7 +126,9 @@ export function registerRememberTool(
content: [
{
type: "text" as const,
- text: `Saved to Walrus Memory. blob_id=${result.blob_id} namespace=${result.namespace}\nExplorer: ${walruscanBlobUrl(result.blob_id)}`,
+ text: withNotice(
+ `Saved to Walrus Memory. blob_id=${result.blob_id} namespace=${result.namespace}\nExplorer: ${walruscanBlobUrl(result.blob_id)}`,
+ ),
},
],
};
From 60beabd79d0affdeac15c54f1a1a1fb1849e6edb Mon Sep 17 00:00:00 2001
From: Harry Phan
Date: Thu, 17 Sep 2026 22:24:03 +0700
Subject: [PATCH 068/132] fix(mcp): redact hex key material by its label, not
its entropy
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit
MemWal's own delegate private key was the one shape passing through
untouched. `delegatePrivateKey` in ~/.memwal/credentials.json is a 64-hex
Ed25519 seed — the value auth.ts marks "NEVER log this", and whoever
holds it can read and write the user's memories until the delegate is
revoked. It is pure lowercase hex, which the entropy rule excludes on
purpose so that git SHAs, Walrus blob ids, Sui object ids and content
digests keep being remembered. For this product it was the worst possible
thing to leave uncovered.
The exclusion is right and stays. The discriminator for hex is the label
instead: a run of 32+ hex characters, 0x-prefixed or not, is removed only
when a credential word sits within a few tokens — as a JSON key, a
key:value assignment, adjacent prose, or a label following the value.
`credential(s)` is deliberately left out of that label list, because
MemWal prose says "credentials.json" constantly and a sentence naming
that file beside a commit SHA would otherwise lose the SHA; placeholders
from earlier rules are stripped from the window for the same reason.
Two narrower misses found alongside it, both from anchoring on \b:
a camelCase field name could not match a keyword sitting mid-identifier
(`delegatePrivateKey:` was invisible to the assignment rule), and the
seed-phrase label required literal whitespace and could not cross a
quote, so `seed_phrase:` and `"mnemonic":` both passed.
Tested from both sides, which is the point of the design: the delegate
key is removed in six spellings and on all three write paths, while an
identical UNLABELLED 64-hex run — plus a commit SHA, a Sui object id, and
delegatePublicKeyHex, documented as safe to display — still passes
through byte-for-byte.
Still gets through, and documented: a hex secret pasted with no label
anywhere near it, and a bare BIP-39 word run nothing calls a mnemonic.
Both are indistinguishable from ordinary content by shape alone, and are
what the model-facing rules cover.
Refs WALM-642
---
packages/mcp/AUTO-MEMORY.md | 6 +-
packages/mcp/README.md | 9 +-
.../mcp/__tests__/secret-redaction.test.ts | 68 ++++++++++++++
.../__tests__/write-path-redaction.test.ts | 58 ++++++++++++
.../server/scripts/mcp/tools/redaction.ts | 93 ++++++++++++++++++-
5 files changed, 228 insertions(+), 6 deletions(-)
diff --git a/packages/mcp/AUTO-MEMORY.md b/packages/mcp/AUTO-MEMORY.md
index 0950f8d38..fd45c187f 100644
--- a/packages/mcp/AUTO-MEMORY.md
+++ b/packages/mcp/AUTO-MEMORY.md
@@ -70,7 +70,11 @@ were removed; the value is never logged, echoed, or returned.
Detection is shape-based, not entropy-based, on purpose: MemWal's own durable
facts (blob ids, Sui object ids, git SHAs, digests) are exactly what a generic
-high-entropy rule would eat. The trade-off is written out at the top of
+high-entropy rule would eat. Key material in hex is therefore caught by the
+**label** beside it rather than by how random it looks — which is what lets the
+`delegatePrivateKey` from `credentials.json` (64 lowercase hex, the value
+`auth.ts` marks "NEVER log this") be removed while a bare commit SHA or `0x`
+object id is left alone. The trade-off is written out at the top of
`redaction.ts`.
## Architecture — three layers
diff --git a/packages/mcp/README.md b/packages/mcp/README.md
index 22b1b6d62..8b3c35b8a 100644
--- a/packages/mcp/README.md
+++ b/packages/mcp/README.md
@@ -111,9 +111,12 @@ is never stored, logged, or echoed back.
Detection targets specific credential shapes rather than "looks random", so
identifiers you *do* want remembered — blob ids, Sui object ids, commit SHAs,
-digests — pass through untouched. The trade-off is that a secret in no
-recognisable shape can still slip past the check, which is why the model-facing
-rules exist alongside it.
+digests — pass through untouched. Where a secret is indistinguishable from an
+identifier, the **label** decides: pasting your `credentials.json` has its
+`delegatePrivateKey` removed, while the same 64 hex characters with nothing
+calling them a key are stored as the digest they look like. The trade-off is
+that a secret in no recognisable shape, and with no label near it, can still
+slip past the check — which is why the model-facing rules exist alongside it.
## Default Namespace
diff --git a/services/server/scripts/mcp/__tests__/secret-redaction.test.ts b/services/server/scripts/mcp/__tests__/secret-redaction.test.ts
index c181b02e9..fea778e60 100644
--- a/services/server/scripts/mcp/__tests__/secret-redaction.test.ts
+++ b/services/server/scripts/mcp/__tests__/secret-redaction.test.ts
@@ -141,6 +141,73 @@ test("a labelled seed phrase is removed", () => {
);
});
+test("MemWal's own delegate private key is removed, in every shape it arrives in", () => {
+ // The 64-hex Ed25519 seed from ~/.memwal/credentials.json. auth.ts marks it
+ // "NEVER log this": whoever holds it can read and write the user's memories
+ // until the delegate is revoked. It is pure lowercase hex, so the entropy
+ // rule deliberately does not see it — the LABEL is what catches it.
+ const SEED = "4f3c2b1a9e8d7c6b5a4f3e2d1c0b9a8f7e6d5c4b3a2f1e0d9c8b7a6f5e4d3c2b";
+
+ // A pasted line from the file, the file itself, prose, and the label after
+ // the value. Each keeps a fact around it so the refusal path is not what is
+ // being measured here.
+ for (const text of [
+ `Notes from setup: delegatePrivateKey ${SEED} was written on this laptop`,
+ `Notes from setup: "delegatePrivateKey": "${SEED}" is in the file`,
+ `Notes from setup: my delegate private key is ${SEED} on this laptop`,
+ `Notes from setup: ${SEED} is my private key for this laptop`,
+ `Notes from setup: secret_key = ${SEED} on this laptop`,
+ `Notes from setup: the signing key 0x${SEED} lives on this laptop`,
+ ]) {
+ const out = sanitizeFact(text);
+ assert.equal(out.refusal, undefined, `unexpectedly refused: ${text}`);
+ assert.ok(!out.text.includes(SEED), `the delegate key survived: ${text}`);
+ assert.ok(out.text.includes("Notes from setup"), `lost the fact: ${text}`);
+ }
+});
+
+test("an UNLABELLED hex run is still left alone — that is what the label gate buys", () => {
+ // The regression this design protects. Same 64 hex characters as the test
+ // above; the only difference is that nothing calls them a key.
+ const HEX = "4f3c2b1a9e8d7c6b5a4f3e2d1c0b9a8f7e6d5c4b3a2f1e0d9c8b7a6f5e4d3c2b";
+ for (const fact of [
+ `The release digest is ${HEX} and I verified it`,
+ `Pin the deployment to 0x${HEX}`,
+ // "credentials.json" is a phrase MemWal prose uses constantly, and it
+ // sits within a window of the SHA here. It is excluded from the label
+ // list precisely so this sentence keeps its commit id.
+ "My creds live in ~/.memwal/credentials.json and the fix landed in 4f2b8c1e9d7a3f5b6c0e2d4a8b1f3c5e7d9a0b2c",
+ // Documented in auth.ts as "Safe to display" — the public half must not
+ // be swept up with the private one.
+ `My delegatePublicKeyHex is ${HEX}`,
+ ]) {
+ const out = sanitizeFact(fact);
+ assert.equal(out.text, fact, `redacted an unlabelled hex run: ${fact}`);
+ assert.equal(out.changed, false);
+ }
+});
+
+test("a seed phrase is caught however the label is spelled", () => {
+ const WORDS =
+ "abandon ability able about above absent absorb abstract absurd abuse access accident";
+ for (const text of [
+ `Wallet notes: my recovery phrase is ${WORDS} for the mainnet wallet`,
+ `Wallet notes: seed_phrase: ${WORDS} for the mainnet wallet`,
+ `Wallet notes: "mnemonic": "${WORDS}" for the mainnet wallet`,
+ `Wallet notes: seedPhrase=${WORDS} for the mainnet wallet`,
+ ]) {
+ const out = sanitizeFact(text);
+ assert.equal(out.refusal, undefined, `unexpectedly refused: ${text}`);
+ assert.ok(!out.text.includes(WORDS), `the mnemonic survived: ${text}`);
+ assert.ok(out.text.includes("Wallet notes"), `lost the fact: ${text}`);
+ }
+
+ // Still true, and still documented: a bare word run with no label at all is
+ // indistinguishable from a sentence, so it is left to the model rules.
+ const bare = sanitizeFact(`I wrote down ${WORDS} yesterday`);
+ assert.equal(bare.changed, false);
+});
+
test("a long mixed-case base64 blob is removed", () => {
const blob =
"QWxhZGRpbjpvcGVuIHNlc2FtZQBcdefGHIjklMNOpqrSTUvwxYZ0123456789abcDEF0123";
@@ -182,6 +249,7 @@ test("the identifiers MemWal itself stores are not mistaken for secrets", () =>
"Pin the build to commit 4f2b8c1e9d7a3f5b6c0e2d4a8b1f3c5e7d9a0b2c",
"The sha256 of the release tarball is e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
"My Sui package id is 0xe80f2feec1c139616a86c9f71210152e2a7ca552b20841f2e192f99f75864437",
+ "The migration artifact hashes to 9a8b7c6d5e4f3a2b1c0d9e8f7a6b5c4d3e2f1a0b9c8d7e6f5a4b3c2d1e0f9a8b",
]) {
const out = sanitizeFact(fact);
assert.equal(out.text, fact, `redacted a legitimate identifier: ${fact}`);
diff --git a/services/server/scripts/mcp/__tests__/write-path-redaction.test.ts b/services/server/scripts/mcp/__tests__/write-path-redaction.test.ts
index 469b00311..48d2341da 100644
--- a/services/server/scripts/mcp/__tests__/write-path-redaction.test.ts
+++ b/services/server/scripts/mcp/__tests__/write-path-redaction.test.ts
@@ -186,6 +186,64 @@ test("memwal_analyze strips the passage before the extractor ever sees it", asyn
assert.match(text, /credential span\(s\) were removed/);
});
+test("the delegate private key never reaches the SDK from any write path", async (t) => {
+ // The one secret that matters most for this product: the Ed25519 seed in
+ // ~/.memwal/credentials.json, which grants read AND write to the user's
+ // memories until the delegate is revoked. It is pure lowercase hex, so it
+ // is invisible to the entropy rule by design — the label beside it is what
+ // catches it. A user pasting their credentials file into chat is the
+ // realistic way this arrives.
+ const SEED = "4f3c2b1a9e8d7c6b5a4f3e2d1c0b9a8f7e6d5c4b3a2f1e0d9c8b7a6f5e4d3c2b";
+ const pasted =
+ `I set up a second laptop today. From credentials.json: ` +
+ `"delegatePrivateKey": "${SEED}", and I use the work namespace there.`;
+
+ const forwarded: Forwarded = { remember: [], bulk: [], analyze: [] };
+ const client = await clientFor(sessionWith(forwarded), t);
+
+ for (const [name, args] of [
+ ["memwal_remember", { text: pasted }],
+ ["memwal_remember_bulk", { facts: [pasted] }],
+ ["memwal_analyze", { text: pasted }],
+ ] as Array<[string, Record]>) {
+ const result = await client.callTool({ name, arguments: args });
+ assert.ok(
+ !textOf(result).includes(SEED),
+ `${name} echoed the delegate key back in its reply`,
+ );
+ }
+
+ const sent = allForwarded(forwarded);
+ assert.ok(sent.length > 0, "the writes must still happen");
+ assert.ok(!sent.includes(SEED), "the delegate private key reached the SDK");
+ // Redacted, not dropped: the fact around it survives on every path.
+ assert.equal(forwarded.remember.length, 1);
+ assert.equal(forwarded.bulk.length, 1);
+ assert.equal(forwarded.analyze.length, 1);
+ for (const text of [forwarded.remember[0], forwarded.bulk[0][0], forwarded.analyze[0]]) {
+ assert.ok(text.includes("second laptop"), "the fact was lost with the key");
+ assert.ok(text.includes("work namespace"), "the fact was lost with the key");
+ }
+});
+
+test("an unlabelled hex identifier still reaches the SDK unchanged", async (t) => {
+ // The other half of the label gate, asserted at the handler boundary: a
+ // 64-hex string nobody called a key is a digest, an object id or a blob id
+ // — the facts this product exists to remember.
+ const forwarded: Forwarded = { remember: [], bulk: [], analyze: [] };
+ const client = await clientFor(sessionWith(forwarded), t);
+ const fact =
+ "My Sui package id is 0xe80f2feec1c139616a86c9f71210152e2a7ca552b20841f2e192f99f75864437 " +
+ "and the release digest is 4f3c2b1a9e8d7c6b5a4f3e2d1c0b9a8f7e6d5c4b3a2f1e0d9c8b7a6f5e4d3c2b";
+
+ const result = await client.callTool({
+ name: "memwal_remember",
+ arguments: { text: fact },
+ });
+ assert.deepEqual(forwarded.remember, [fact]);
+ assert.doesNotMatch(textOf(result), /redacted|NOT SAVED/i);
+});
+
// ── the rules that are not about credentials ────────────────────────────────
test("an explicit do-not-save is honoured on every write path", async (t) => {
diff --git a/services/server/scripts/mcp/tools/redaction.ts b/services/server/scripts/mcp/tools/redaction.ts
index 33a14f4cc..6e82c0db0 100644
--- a/services/server/scripts/mcp/tools/redaction.ts
+++ b/services/server/scripts/mcp/tools/redaction.ts
@@ -33,6 +33,11 @@
* - `password:`-style assignments redact whatever follows the separator, so
* "password: ask Marta" loses "ask Marta". Over-redacting a sentence about
* a credential is cheap; under-redacting the credential is permanent.
+ * - Hex key material is caught by the LABEL next to it, never by its shape —
+ * see `HEX_RUN`. That is what lets MemWal's own 64-hex delegate private key
+ * be removed while a bare 40-hex commit SHA or a `0x`-prefixed Sui object id
+ * is left alone. The cost is that a hex secret pasted with no label at all
+ * still passes; the model-facing rules are what cover that.
* - Quoted-content detection only fires on unambiguous pastes (a fenced
* block, a multi-line `>` quotation, or a ≥200-char fully quoted passage).
* A short quoted sentence is left to the model-facing rules, because an
@@ -50,6 +55,7 @@ export type RedactionKind =
| "credential-assignment"
| "auth-header"
| "seed-phrase"
+ | "labelled-key-material"
| "high-entropy-secret";
/** Why a text was refused outright instead of redacted. */
@@ -131,9 +137,15 @@ const COOKIE_HEADER =
* The separator must follow the keyword immediately, which is what keeps
* ordinary prose out: "my password manager is 1Password" has no separator after
* "password" and does not match.
+ *
+ * `(?<=[a-z])` alongside `\b` is what makes a camelCase field name match. A
+ * plain `\b` anchors only at a non-word character, so the keyword had to start
+ * the identifier — and MemWal's own worst secret is spelled
+ * `delegatePrivateKey`, where `PrivateKey` sits mid-identifier and was
+ * therefore invisible to this rule.
*/
const CREDENTIAL_ASSIGNMENT =
- /\b(passwords?|passwd|pwd|passphrases?|api[_-]?keys?|apikeys?|secret[_-]?keys?|client[_-]?secrets?|secrets?|access[_-]?tokens?|refresh[_-]?tokens?|auth[_-]?tokens?|bearer[_-]?tokens?|tokens?|private[_-]?keys?|credentials?)(\s*[:=]\s*)("[^"\n]*"|'[^'\n]*'|`[^`\n]*`|[^\s,;]+)/gi;
+ /(?:\b|(?<=[a-z]))(passwords?|passwd|pwd|passphrases?|api[_-]?keys?|apikeys?|secret[_-]?keys?|client[_-]?secrets?|secrets?|access[_-]?tokens?|refresh[_-]?tokens?|auth[_-]?tokens?|bearer[_-]?tokens?|tokens?|private[_-]?keys?|credentials?)(\s*[:=]\s*)("[^"\n]*"|'[^'\n]*'|`[^`\n]*`|[^\s,;]+)/gi;
/**
* Vendor-prefixed keys. Each prefix is issued by exactly one service and never
@@ -172,7 +184,75 @@ const JWT = /\beyJ[A-Za-z0-9_-]{8,}\.[A-Za-z0-9_-]{8,}\.[A-Za-z0-9_-]{8,}/g;
* rules already cover.
*/
const SEED_PHRASE =
- /\b((?:seed|recovery|secret|mnemonic)\s+(?:phrase|words)|mnemonic)(\s*(?:is|are|:|=)?\s*)((?:[a-z]{3,8}[ \t]+){11,23}[a-z]{3,8})\b/gi;
+ /(?:\b|(?<=[a-z]))((?:seed|recovery|secret|mnemonic)[\s_-]*(?:phrase|words)|mnemonic)([^A-Za-z0-9]{0,4}(?:is|are)?[^A-Za-z0-9]{0,4})((?:[a-z]{3,8}[ \t]+){11,23}[a-z]{3,8})\b/gi;
+
+/**
+ * Key material in hex, identified by the LABEL beside it rather than by how
+ * random it looks.
+ *
+ * This exists because of one specific secret: `delegatePrivateKey` in
+ * `~/.memwal/credentials.json` is a 64-hex Ed25519 seed, the thing auth.ts
+ * marks "NEVER log this", and whoever holds it can read and write the user's
+ * memories until the delegate is revoked. It is the worst thing this product
+ * can leak — and it is pure lowercase hex, so it is deliberately excluded by
+ * `looksLikeSecretBlob` below and was sailing straight through.
+ *
+ * The exclusion is still right: a bare hex run is a git SHA, a Walrus blob id,
+ * a Sui object or account id, a content digest — the identifiers users most
+ * want remembered. So the discriminator is not entropy, it is the label. A hex
+ * run is removed only when a credential word sits next to it, which catches
+ * every shape the secret actually arrives in:
+ *
+ * delegatePrivateKey 4f3c... (prose / a pasted line)
+ * "delegatePrivateKey": "4f3c..." (the credentials.json file itself)
+ * my delegate private key is 4f3c...
+ * 4f3c... is my private key (label after the value)
+ *
+ * while `Pin the build to commit 4f2b8c1e...` and `my account id is 0x7f3a...`
+ * keep passing through untouched. That asymmetry is the whole design, and it is
+ * pinned from both sides in secret-redaction.test.ts.
+ */
+const HEX_RUN = /\b(?:0x)?[0-9a-fA-F]{32,}\b/g;
+
+/** How far either side of a hex run a label may sit — a few tokens. */
+const HEX_LABEL_WINDOW = 48;
+
+/**
+ * Words that make an adjacent hex run key material.
+ *
+ * Not anchored on a word boundary, so it matches inside a camelCase identifier
+ * (`delegatePrivateKey`). `credential(s)` is deliberately ABSENT: MemWal's own
+ * prose says "credentials.json" constantly, and a sentence naming that file
+ * next to a commit SHA would lose the SHA.
+ */
+const HEX_CREDENTIAL_LABEL =
+ /private[\s_-]*key|secret[\s_-]*key|delegate[\s_-]*key|delegate[\s_-]*private|signing[\s_-]*key|priv[\s_-]*key|api[\s_-]*key|access[\s_-]*key|auth[\s_-]*key|secret|seed|mnemonic|passphrase/i;
+
+/**
+ * The same, for a label that FOLLOWS the value ("4f3c... is my private key").
+ * Tighter than the backward form — it has to march through the small joining
+ * phrase rather than search a window — because a trailing window would sweep
+ * in whatever sentence happens to come next.
+ */
+const HEX_LABEL_AFTER =
+ /^[^A-Za-z0-9]{0,4}(?:is|was)?[^A-Za-z0-9]{0,4}(?:my|the|our|his|her|their)?[^A-Za-z0-9]{0,4}(?:delegate[\s_-]*)?(?:private[\s_-]*key|secret[\s_-]*key|seed|mnemonic|passphrase|api[\s_-]*key)/i;
+
+/** True when a credential word sits within a few tokens of [start, end). */
+function hasAdjacentCredentialLabel(
+ text: string,
+ start: number,
+ end: number,
+): boolean {
+ // Placeholders left by earlier rules carry the words "secret" and "key",
+ // so a run of redactions would otherwise start labelling its own
+ // neighbours — and the neighbour after `private_key=[redacted:...]` is
+ // exactly the kind of bare SHA this rule must not touch.
+ const before = text
+ .slice(Math.max(0, start - HEX_LABEL_WINDOW), start)
+ .replace(/\[redacted:[a-z-]+\]/g, " ");
+ if (HEX_CREDENTIAL_LABEL.test(before)) return true;
+ return HEX_LABEL_AFTER.test(text.slice(end, end + HEX_LABEL_WINDOW));
+}
/**
* A ≥64-character base64/base64url run carrying lower case, upper case AND a
@@ -299,6 +379,15 @@ export function sanitizeFact(input: string): SanitizedText {
return `${label}${sep}${placeholder("seed-phrase")}`;
});
+ // Label-gated, and therefore run over the text as it stands now: the
+ // window check reads the characters on either side, so it has to see the
+ // real neighbours rather than a half-rewritten string.
+ text = text.replace(HEX_RUN, (match: string, offset: number, whole: string) =>
+ hasAdjacentCredentialLabel(whole, offset, offset + match.length)
+ ? hit("labelled-key-material")
+ : match,
+ );
+
text = text.replace(HIGH_ENTROPY_CANDIDATE, (token: string) =>
looksLikeSecretBlob(token) ? hit("high-entropy-secret") : token,
);
From 780608fcb96c3ebcaa7c76cc9fc27f453564d571 Mon Sep 17 00:00:00 2001
From: Harry Phan
Date: Thu, 17 Sep 2026 22:40:59 +0700
Subject: [PATCH 069/132] fix(mcp): ask once at login instead of defaulting
automatic memory off
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit
Default-off followed the ticket's letter and broke the product: MemWal
stops remembering until someone runs `auto-save on`, which most users
will never discover. The default goes back to on. What changes is that a
human is asked first.
Three states now, not two. `on` and `off` are answers; `unset` is the
absence of one, and it does not mean off:
- answered (autoSave true/false in settings.json) — nothing re-asks;
- unset with no settings file and credentials on disk — an install that
predates this change. It keeps saving, because that is its status quo
rather than a new grant, and the question is put at its next
interactive login;
- unset with autoSaveConsent "pending" — a post-change install nobody
has asked yet. Nothing is auto-saved until someone answers; explicit
tool calls and recall keep working throughout.
The pending stamp is what separates the last two, and it is load-bearing.
It is written the moment a new install first appears — before the
signed-out server boots, and immediately after a first interactive login,
before the question is put — so a headless sign-in through the
memwal_login tool cannot mature into a long-standing-user profile and
start saving with nobody having been asked. Abandoning the prompt leaves
the state unset rather than recording a choice.
The question is asked at `login`, on a TTY, and nowhere else. It is
deliberately not an MCP tool, a tool description or an instruction: a
model answering on the user's behalf is not consent, and an agent-shaped
surface for it would be exactly that. A test greps the compiled tool
surfaces to keep it so. Non-TTY runs print one line saying where things
stand and never touch stdin, because a prompt there would hang every MCP
client's startup.
On the wording, which is the substance: it names the consequence rather
than the feature, permanence leads the bullets because Walrus is
immutable and that is the fact that changes the answer, the redaction
claim stays hedged to "safety net, not a guarantee" to match the residual
gaps the code actually documents, and declining is a normal choice that
is recorded once and never asked about again. Enter takes [1]; anything
that is not 1 or 2 re-asks rather than assuming.
MEMWAL_AUTO_SAVE still overrides every state, and also settles the
question — an install configured deliberately is not prompted as well.
Refs WALM-642
---
docs/mcp/overview.md | 11 +-
docs/mcp/reference.md | 2 +-
packages/mcp/AUTO-MEMORY.md | 53 +++-
packages/mcp/README.md | 56 +++-
packages/mcp/TESTING.md | 53 ++--
packages/mcp/plugin/scripts/lib/auto-save.mjs | 65 ++++-
.../mcp/plugin/scripts/lib/memory-policy.mjs | 16 +-
.../mcp/plugin/scripts/on_session_start.mjs | 14 +-
packages/mcp/src/auto-save.ts | 200 ++++++++++----
packages/mcp/src/consent.ts | 137 ++++++++++
packages/mcp/src/index.ts | 101 ++++++-
packages/mcp/src/memory-policy.ts | 16 +-
packages/mcp/test/auto-save-optin.test.mjs | 257 ++++++++++++++++--
packages/mcp/test/user-prompt-hook.test.mjs | 15 +-
.../server/scripts/mcp/tools/memory-policy.ts | 16 +-
15 files changed, 848 insertions(+), 164 deletions(-)
create mode 100644 packages/mcp/src/consent.ts
diff --git a/docs/mcp/overview.md b/docs/mcp/overview.md
index b3b98ed94..81d38af38 100644
--- a/docs/mcp/overview.md
+++ b/docs/mcp/overview.md
@@ -46,10 +46,13 @@ There are two ways to use MemWal. The difference is whether you also get the **l
- **Plugin** bundles the MCP server **and** lifecycle hooks. The `SessionStart` hook tells the agent to prefer the `memwal_*` tools over any built-in or local memory feature, and when to save without being asked. Available on **Claude Code**, **Codex**, **Antigravity**, and **Cursor**.
- **MCP-only** gives the agent the memory tools on **every** MCP client. The tool descriptions encourage proactive use, so agents often do save and recall on their own. Treat that as best-effort: it varies by client and model, and on a client that ships its own memory feature the built-in one commonly wins.
-Saving **without being asked is opt-in on both paths** and off until you turn it
-on with `memwal-mcp auto-save on` (or `MEMWAL_AUTO_SAVE=1` in your client's
-`env` block). Recall, and anything you explicitly ask to be remembered, work
-either way. Credentials — passwords, API keys, tokens, private keys, seed
+Saving **without being asked is on once you agree to it**: `memwal-mcp login`
+asks the question in your terminal the first time, and on a fresh install
+nothing is saved unprompted until you answer. Saved memories are permanent —
+Walrus is immutable storage — which is why the question comes first. Change the
+answer any time with `memwal-mcp auto-save on|off` (or `MEMWAL_AUTO_SAVE` in
+your client's `env` block). Recall, and anything you explicitly ask to be
+remembered, work either way. Credentials — passwords, API keys, tokens, private keys, seed
phrases, authorization headers, and URLs with an embedded `user:password` — are
excluded in both modes and stripped before a memory is written, because Walrus
storage is append-only and a stored secret cannot be deleted.
diff --git a/docs/mcp/reference.md b/docs/mcp/reference.md
index d7ee6e97e..9acd0ddf6 100644
--- a/docs/mcp/reference.md
+++ b/docs/mcp/reference.md
@@ -53,7 +53,7 @@ This is why many first-run sessions show `memwal_login` before the other tools a
### memwal_remember
-Save a durable fact to the user's Walrus Memory. The agent calls this **proactively** when the user states a preference, decision, constraint, correction, identity detail, or recurring workflow, not only when they explicitly ask — provided automatic memory is on (`memwal-mcp auto-save on`, or `MEMWAL_AUTO_SAVE=1`; it is off by default). Skip one-off tasks, the current file or bug, and small talk. Pass the full statement; do not summarize.
+Save a durable fact to the user's Walrus Memory. The agent calls this **proactively** when the user states a preference, decision, constraint, correction, identity detail, or recurring workflow, not only when they explicitly ask — provided automatic memory is on. That is the user's standing answer to a question `memwal-mcp login` asks once in a terminal; it can be changed with `memwal-mcp auto-save on|off`. Skip one-off tasks, the current file or bug, and small talk. Pass the full statement; do not summarize.
Credentials are never stored. Passwords, API keys, access and refresh tokens, private keys, seed phrases, authorization headers, session cookies, and URLs with an embedded `user:password` are stripped from the text before the write, on this tool, `memwal_remember_bulk` and `memwal_analyze` alike. A message that mixes a preference with a credential keeps the preference and loses only the credential; the reply names which kinds were removed.
diff --git a/packages/mcp/AUTO-MEMORY.md b/packages/mcp/AUTO-MEMORY.md
index fd45c187f..eae420849 100644
--- a/packages/mcp/AUTO-MEMORY.md
+++ b/packages/mcp/AUTO-MEMORY.md
@@ -14,13 +14,37 @@ package. `memwal_remember` literally said *"Call ONLY when the user explicitly
asks… agents should not call this proactively."* And `rememberBulk` (in the SDK)
was never exposed as a tool.
-## Opt-in and secret filtering (WALM-642)
+## Consent and secret filtering (WALM-642)
-Automatic saving is **off by default**. What is gated is narrow: telling the
-model to save something the user did **not** ask it to save. A direct request
-("remember that ...") works either way, and recall is never gated.
+Automatic saving is **on** — it is what MemWal is for — but only once a human
+has been asked. What is governed is narrow: telling the model to save something
+the user did **not** ask it to save. A direct request ("remember that ...")
+works either way, and recall is never gated.
-**Turning it on**
+**Three states, not two.** `on` and `off` are answers; `unset` is the absence of
+one, and it does not mean `off`:
+
+| State | Resolves to | Why |
+|---|---|---|
+| `autoSave: true` / `false` in settings.json | that | answered, at login or via the CLI |
+| unset, no settings file, credentials present | **on** | predates this change; these users have been auto-saving all along, and switching them off would be a regression dressed up as caution. Asked at their next interactive login. |
+| unset, `autoSaveConsent: "pending"` | **off** | a post-change install nobody has asked yet. Explicit tool calls and recall keep working. |
+
+The `pending` stamp is what separates the last two. It is written the moment a
+new install first appears — before the signed-out server boots, and immediately
+after a first interactive login, *before* the question is put. Without it a
+headless install could sign in through the `memwal_login` tool, become
+indistinguishable from a long-standing user, and start saving with nobody ever
+having been asked.
+
+**Where the question is asked** — `memwal-mcp login`, on a TTY, and nowhere else
+(`src/consent.ts`). Deliberately **not** an MCP tool, a tool description or an
+instruction: a model answering on the user's behalf is not consent, and an
+agent-shaped surface for this question would be exactly that. A test greps the
+compiled tool surfaces to keep it that way. Non-TTY runs print one line saying
+where things stand and never block on stdin.
+
+**Setting it directly**
```sh
memwal-mcp auto-save on # persists {"autoSave": true} to settings.json
@@ -28,8 +52,9 @@ memwal-mcp auto-save off
memwal-mcp auto-save # report the current state and where it came from
```
-`MEMWAL_AUTO_SAVE=1` in an MCP client's `env` block does the same for one
-server process and overrides the file.
+`MEMWAL_AUTO_SAVE=1` in an MCP client's `env` block overrides the file for one
+server process, and counts as a deliberate answer — it also stops the login
+prompt, so a configured install is never nagged.
**Where the state lives** — `settings.json`, next to `credentials.json`, so it
inherits `credsPath()` resolution: `MEMWAL_CREDS_DIR` override, else the
@@ -43,7 +68,9 @@ two implementations of the same resolution, pinned against each other by
**What changes when it is off** — the `instructions` field, the SessionStart
rubric, the UserPromptSubmit rubric and the PostToolUse nudge all switch to a
save-only-what-you-are-asked variant. Nothing is disabled; the guidance that
-drives an unasked-for save is simply not injected.
+drives an unasked-for save is simply not injected. While consent is outstanding
+the SessionStart banner also says so, and tells the agent the answer is given in
+a terminal — not in chat, and not by it.
**One source for the rules** — the secret-exclusion and do-not-save text lives
in a single block duplicated byte-for-byte across three files that cannot
@@ -93,7 +120,7 @@ object id is left alone. The trade-off is written out at the top of
| Dimension | Before | After |
|---|---|---|
-| Save trigger | "ONLY when user explicitly asks; don't be proactive" | "Save proactively whenever you learn a durable fact" — **opt-in since WALM-642; off by default** |
+| Save trigger | "ONLY when user explicitly asks; don't be proactive" | "Save proactively whenever you learn a durable fact" — **since WALM-642, after a consent question at login** |
| Secret handling | none: a preference next to a password was forwarded whole | shared exclusion rules on all three surfaces + a redactor in front of every write |
| Bulk save | not exposed | `memwal_remember_bulk` (wraps SDK `rememberBulkAndWait`, ≤20) |
| Recall trigger | neutral; agent rarely called it unprompted | "Recall proactively at task start / when the user references past work" |
@@ -118,8 +145,9 @@ object id is left alone. The trade-off is written out at the top of
- **Append-only** — no `forget`/`update` tools (relayer dedups embeddings). This
is also why WALM-642's credential check runs *before* the write: there is no
delete to fall back on.
-- **Automatic saving is opt-in, default off** (WALM-642) — explicit tool use is
- never gated, and neither is recall.
+- **Automatic saving is on, after consent** (WALM-642) — asked once at
+ interactive login, never through an agent-reachable surface. Explicit tool use
+ is never gated, and neither is recall.
- **Global `default` namespace** — `MEMWAL_NAMESPACE` overrides for per-project scope.
- **Agent decision rubric** — UserPromptSubmit does not regex-classify remember vs
recall. The agent has the conversation and understands any language or spelling.
@@ -133,7 +161,8 @@ object id is left alone. The trade-off is written out at the top of
- `services/server/scripts/mcp/tools/{remember,recall,analyze,restore}.ts` — agentic descriptions
- `services/server/scripts/mcp/tools/redaction.ts` — pre-forward credential screen (WALM-642)
- `{packages/mcp/src,packages/mcp/plugin/scripts/lib,services/server/scripts/mcp/tools}/memory-policy.*` — the shared rules block, three byte-identical copies
-- `packages/mcp/src/auto-save.ts` + `packages/mcp/plugin/scripts/lib/auto-save.mjs` — the opt-in resolver, server side and hook side
+- `packages/mcp/src/auto-save.ts` + `packages/mcp/plugin/scripts/lib/auto-save.mjs` — the tri-state resolver, server side and hook side
+- `packages/mcp/src/consent.ts` — the login-time consent question; TTY-only, never agent-reachable
- `services/server/scripts/mcp/tools/remember-bulk.ts` + `index.ts` — new bulk tool
- `packages/mcp/plugin/` — plugin manifest, `.mcp.json`, hooks, Node scripts, Codex installer
- `.claude-plugin/marketplace.json` (repo root) — Claude Code marketplace entry (local source `./packages/mcp/plugin`)
diff --git a/packages/mcp/README.md b/packages/mcp/README.md
index 8b3c35b8a..b91132ebd 100644
--- a/packages/mcp/README.md
+++ b/packages/mcp/README.md
@@ -53,15 +53,52 @@ Use CLI flags or environment variables to override the default Walrus Memory end
| `--web-url ` | `MEMWAL_WEB_URL` | Override the web app URL used during login. |
| `--label ` | `MEMWAL_CLIENT_LABEL` | Friendly delegate-key label shown in Walrus Memory. |
| `--namespace ` (alias `--ns`) | `MEMWAL_NAMESPACE` | Default memory namespace applied when the agent omits one. |
-| `auto-save on\|off` | `MEMWAL_AUTO_SAVE` | Let the agent save durable facts unprompted. Off by default; see [Automatic Memory](#automatic-memory-opt-in). |
+| `auto-save on\|off` | `MEMWAL_AUTO_SAVE` | Whether the agent saves durable facts unprompted. On once you agree at login; see [Automatic Memory](#automatic-memory). |
Enable verbose stderr logging with `MEMWAL_MCP_DEBUG=1`.
-## Automatic Memory (opt-in)
+## Automatic Memory
-By default the agent saves **only what you ask it to save**. Turning automatic
-memory on lets it save durable facts — preferences, decisions, constraints,
-recurring workflows — without being asked:
+MemWal saves durable facts — preferences, decisions, constraints, recurring
+workflows — as you state them, without asking each time. That is what it is for,
+so it is on. It asks you once first.
+
+The question comes up in your terminal the first time you run `login`:
+
+```
+MemWal can save things about you automatically.
+
+What that means: when you state a preference, a decision, or a setting in
+chat — "I prefer pnpm", "we deploy from dev", "the relayer is at X" —
+MemWal writes it to your memory without asking each time, so it is there
+in your next session and in every other client you use.
+
+Before you choose:
+
+ - Saved memories are permanent. They go to Walrus, which is immutable
+ storage. You can stop saving new ones at any time, but you cannot
+ delete one that is already saved.
+ - They are encrypted to your account. Only your delegate key reads them.
+ - MemWal strips obvious credentials — API keys, tokens, passwords,
+ private keys — before saving. Treat that as a safety net, not a
+ guarantee: do not paste secrets into a session with this on.
+
+ [1] Save automatically recommended, this is what MemWal is for
+ [2] Only save when I ask nothing is saved unless you say "remember this"
+
+Your choice [1/2]:
+```
+
+**Saved memories are permanent.** Walrus is immutable storage: you can stop
+saving new ones at any time, but you cannot delete one already saved — which is
+why you are asked before it starts rather than after.
+
+Until you answer, nothing is saved unprompted on a new install; "remember this"
+and recall work throughout. If you were using MemWal before this question
+existed, it keeps saving as it always has and asks you at your next login.
+
+The question is only ever asked in a terminal. It is never an MCP tool and never
+something the assistant can answer for you. Change it any time:
```sh
npx -y @mysten-incubation/memwal-mcp auto-save on
@@ -69,10 +106,10 @@ npx -y @mysten-incubation/memwal-mcp auto-save off
npx -y @mysten-incubation/memwal-mcp auto-save # report the current setting
```
-The choice is stored as `{"autoSave": true}` in `settings.json` next to your
+The answer is stored as `{"autoSave": true}` in `settings.json` next to your
credentials file, so it follows the same project-local-beats-global resolution.
To pin one MCP client instead, set the environment variable — it overrides the
-file:
+file and skips the question:
```json
{
@@ -87,12 +124,13 @@ file:
```
Recall is never gated, and neither is an explicit "remember this" — the setting
-only decides whether the agent saves things you did not ask it to save.
+only decides whether the agent saves things you did not ask it to save. Saying
+no costs nothing and is not asked about again.
### What is never saved
Walrus storage is append-only and encrypted: a memory that lands **cannot be
-edited or deleted**. So credentials are excluded in both modes, by the same
+edited or deleted**. So credentials are excluded whichever way you answer, by the same
rules stated in the server instructions, the tool descriptions and the plugin
hooks — and enforced by a check that runs before any text is sent:
diff --git a/packages/mcp/TESTING.md b/packages/mcp/TESTING.md
index 462bf42e5..0ba69549c 100644
--- a/packages/mcp/TESTING.md
+++ b/packages/mcp/TESTING.md
@@ -23,12 +23,16 @@ Step-by-step plan to test the auto-memory work (agentic tools + bulk + health +
```
- [ ] Credentials present and on testnet: `~/.memwal/credentials.json` (relayerUrl = `http://127.0.0.1:8000`).
- [ ] MCP package built: `ls packages/mcp/dist/bin/memwal-mcp.js`.
-- [ ] **Automatic saving turned on** — it is OFF by default since WALM-642, and
- every "agent saves on its own" step below depends on it:
+- [ ] **Automatic saving answered** — since WALM-642 `login` asks once in the
+ terminal, and a fresh install saves nothing unprompted until it is
+ answered. Every "agent saves on its own" step below depends on the answer
+ being yes:
```bash
- node packages/mcp/dist/bin/memwal-mcp.js auto-save on # → "Automatic memory is ON"
+ node packages/mcp/dist/bin/memwal-mcp.js auto-save # → reports the state
+ node packages/mcp/dist/bin/memwal-mcp.js auto-save on # → "Automatic memory is ON"
```
- Leave it off first if you want to confirm the opt-in itself (§1.6).
+ Clear `~/.memwal/settings.json` first if you want to see the consent
+ prompt itself (§1.6).
**How to tell MemWal vs the editor's built-in memory** (use everywhere below):
- ✅ **MemWal** → the step shows `Called memwal…` with a **`blob_id`** + `namespace`.
@@ -83,7 +87,7 @@ Notes:
---
-## 1.6 Automatic-save opt-in and secret filtering (WALM-642)
+## 1.6 Consent and secret filtering (WALM-642)
The redactor and the opt-in resolver have automated coverage
(`services/server/scripts/mcp/__tests__/{secret-redaction,write-path-redaction}.test.ts`,
@@ -92,26 +96,41 @@ automatable is the half the ticket calls "model behavior": whether a real model,
reading the injected rules, actually declines to save a secret it was never
programmatically stopped from sending. Run these by hand in a real client.
-**Opt-in (start with `auto-save off`)**
+**Consent at login** — the prompt is TTY-only, so this part is a terminal, not a
+client.
+
+| # | Action | Expect | OK? |
+|---|---|---|---|
+| 1 | `rm ~/.memwal/settings.json`, then `memwal-mcp login` in a terminal | The consent prompt renders: consequence first, permanence as the first bullet, redaction described as a safety net. Both options readable. | [ ] |
+| 2 | Type `banana` at the prompt | Re-asks. Does **not** assume [1]. | [ ] |
+| 3 | Press Enter | Takes [1]; `settings.json` shows `"autoSave": true` | [ ] |
+| 4 | Run `memwal-mcp login` again | Does **not** re-ask | [ ] |
+| 5 | `rm settings.json`, `login`, answer `2` | `"autoSave": false`; says once how to change it | [ ] |
+| 6 | `login` again after answering `2` | Does **not** re-ask, does not nag | [ ] |
+| 7 | `rm settings.json`, then start the server from an MCP client (no TTY) | No prompt, no hang. One stderr line saying auto-save is waiting on an answer. | [ ] |
+| 8 | Ask the assistant in chat to "turn automatic memory on for me" | It points you at the terminal command; there is no tool it can call to answer for you | [ ] |
+
+**Behaviour either way**
| # | Action / prompt | Expect | OK? |
|---|---|---|---|
-| 1 | `auto-save off`, restart the client, then `I prefer pnpm and TypeScript strict mode.` | Agent does **not** call `memwal_remember` on its own | [ ] |
-| 2 | Same session: `remember that I prefer pnpm` | Agent **does** call `memwal_remember` — an explicit ask is never gated | [ ] |
-| 3 | Same session: `what do you remember about my preferences?` | Agent calls `memwal_recall` — recall is never gated | [ ] |
-| 4 | `auto-save on`, restart, repeat #1 | Agent calls `memwal_remember` unprompted again | [ ] |
+| 9 | With `auto-save off`: `I prefer pnpm and TypeScript strict mode.` | Agent does **not** call `memwal_remember` on its own | [ ] |
+| 10 | Same session: `remember that I prefer pnpm` | Agent **does** call `memwal_remember` — an explicit ask is never gated | [ ] |
+| 11 | Same session: `what do you remember about my preferences?` | Agent calls `memwal_recall` — recall is never gated | [ ] |
+| 12 | `auto-save on`, restart, repeat #9 | Agent calls `memwal_remember` unprompted again | [ ] |
**Secret filtering (with `auto-save on`)**
| # | Action / prompt | Expect | OK? |
|---|---|---|---|
-| 5 | `I prefer dark mode, and the staging db is postgres://admin:hunter2@db.internal:5432/app` | A memory IS saved, and it contains the preference and `db.internal:5432/app` but **not** `hunter2`. The tool reply names `url-credentials`. Confirm with a recall in a new chat. | [ ] |
-| 6 | Same, but phrased so the model saves it as several facts | `memwal_remember_bulk` — same result per entry; a bare-secret entry is reported as `NOT SAVED (1)` with its position | [ ] |
-| 7 | Paste a transcript containing a preference and `ghp_…`, ask to analyse it | `memwal_analyze` — the extracted facts never contain the token | [ ] |
-| 8 | `My bank PIN is 4821 — don't save this.` | Nothing is saved; the agent says so | [ ] |
-| 9 | Paste a fenced log/code block and say "save this" | Not saved as a fact about you; the agent asks you to restate it | [ ] |
-| 10 | **Model behavior:** state a secret in a shape the redactor does not match (e.g. `my door code is seven four nine two`) and see whether the model saves it | The rules say not to; a save here is a model failure, not a code failure — record it, it is the residual risk the redactor cannot close | [ ] |
-| 11 | Check `~/.memwal/settings.json` and the relayer logs after #5-#10 | No secret appears in either — the redactor never logs what it removed | [ ] |
+| 13 | `I prefer dark mode, and the staging db is postgres://admin:hunter2@db.internal:5432/app` | A memory IS saved, and it contains the preference and `db.internal:5432/app` but **not** `hunter2`. The tool reply names `url-credentials`. Confirm with a recall in a new chat. | [ ] |
+| 14 | Same, but phrased so the model saves it as several facts | `memwal_remember_bulk` — same result per entry; a bare-secret entry is reported as `NOT SAVED (1)` with its position | [ ] |
+| 15 | Paste a transcript containing a preference and `ghp_…`, ask to analyse it | `memwal_analyze` — the extracted facts never contain the token | [ ] |
+| 16 | `My bank PIN is 4821 — don't save this.` | Nothing is saved; the agent says so | [ ] |
+| 17 | Paste a fenced log/code block and say "save this" | Not saved as a fact about you; the agent asks you to restate it | [ ] |
+| 18 | **Model behavior:** state a secret in a shape the redactor does not match (e.g. `my door code is seven four nine two`) and see whether the model saves it | The rules say not to; a save here is a model failure, not a code failure — record it, it is the residual risk the redactor cannot close | [ ] |
+| 19 | Paste a `credentials.json` line containing `"delegatePrivateKey": "<64 hex>"` | Saved without the key; an unlabelled 64-hex digest in the same sentence survives | [ ] |
+| 20 | Check `~/.memwal/settings.json` and the relayer logs after #13-#19 | No secret appears in either — the redactor never logs what it removed | [ ] |
---
diff --git a/packages/mcp/plugin/scripts/lib/auto-save.mjs b/packages/mcp/plugin/scripts/lib/auto-save.mjs
index 23ce4eb07..f776b369a 100644
--- a/packages/mcp/plugin/scripts/lib/auto-save.mjs
+++ b/packages/mcp/plugin/scripts/lib/auto-save.mjs
@@ -15,9 +15,10 @@
* credentials resolve to — `MEMWAL_CREDS_DIR`, else the nearest
* project-local `.memwal/credentials.json` at or above the working
* directory, else `~/.memwal`.
- * 3. Off.
+ * 3. Unanswered: on for an install that predates the consent prompt, off for
+ * one created after it (`autoSaveConsent: "pending"`).
*
- * Any error reads as "off": a hook must never block a session, and an
+ * Any error reads as "not answered": a hook must never block a session, and an
* unreadable file is not consent.
*/
import { existsSync, readFileSync } from "node:fs";
@@ -76,26 +77,62 @@ export function parseBooleanSetting(raw) {
return null;
}
-/** `{ enabled, source }` where source is "env" | "settings" | "default". */
+function readSettings() {
+ try {
+ const path = settingsPath();
+ if (!existsSync(path)) return {};
+ const parsed = JSON.parse(readFileSync(path, "utf8"));
+ return parsed && typeof parsed === "object" ? parsed : {};
+ } catch {
+ // Unreadable or corrupt is not an answer.
+ return {};
+ }
+}
+
+/**
+ * `{ enabled, state, source, pendingConsent }`.
+ *
+ * Mirrors `autoSaveStatus()` in src/auto-save.ts, including the two unset
+ * rules: an install that predates the consent prompt (no stamp, credentials on
+ * disk) keeps saving, and one created after it saves nothing until answered.
+ * The two must agree, or the hooks would steer the agent one way while the MCP
+ * server's instructions steered it the other.
+ */
export function autoSaveStatus() {
+ const settings = readSettings();
+ const answered =
+ typeof settings.autoSave === "boolean" ? settings.autoSave : null;
+ const state = answered === null ? "unset" : answered ? "on" : "off";
+
const fromEnv = parseBooleanSetting(process.env[AUTO_SAVE_ENV]);
- if (fromEnv !== null) return { enabled: fromEnv, source: "env" };
+ if (fromEnv !== null) {
+ return { enabled: fromEnv, state, source: "env", pendingConsent: false };
+ }
+ if (answered !== null) {
+ return { enabled: answered, state, source: "settings", pendingConsent: false };
+ }
+ const stamped = settings.autoSaveConsent === "pending";
+ let preExisting = false;
try {
- const path = settingsPath();
- if (existsSync(path)) {
- const parsed = JSON.parse(readFileSync(path, "utf8"));
- if (parsed && typeof parsed.autoSave === "boolean") {
- return { enabled: parsed.autoSave, source: "settings" };
- }
- }
+ preExisting = !stamped && existsSync(credsPath());
} catch {
- /* unreadable or corrupt — fall through to the default */
+ /* best effort — an unreadable home directory reads as a new install */
}
- return { enabled: false, source: "default" };
+ return {
+ enabled: preExisting,
+ state: "unset",
+ source: preExisting ? "legacy" : "unanswered",
+ pendingConsent: true,
+ };
}
-/** True only when the user has turned automatic saving on. */
+/** True when this session may save without being asked. */
export function isAutoSaveEnabled() {
return autoSaveStatus().enabled;
}
+
+/** True while a human still owes the consent question an answer. */
+export function isConsentPending() {
+ return autoSaveStatus().pendingConsent;
+}
diff --git a/packages/mcp/plugin/scripts/lib/memory-policy.mjs b/packages/mcp/plugin/scripts/lib/memory-policy.mjs
index 1d5c26391..a8c4ae977 100644
--- a/packages/mcp/plugin/scripts/lib/memory-policy.mjs
+++ b/packages/mcp/plugin/scripts/lib/memory-policy.mjs
@@ -65,16 +65,18 @@ export const SECRET_EXCLUSION_SUMMARY = [
].join(" ");
/**
- * Automatic saving is opt-in, and this is the sentence that says so. A direct
- * request from the user ("remember that ...") is never gated by it — the gate
- * is only on saving something the user did not ask you to save.
+ * Whether to save unprompted is the user's standing choice, and this is the
+ * sentence that says so. A direct request ("remember that ...") is never gated
+ * by it — the gate is only on saving something the user did not ask you to save.
*/
export const AUTO_SAVE_OPT_IN_RULE = [
- "Saving something the user did not ask you to save is OFF unless they have turned automatic",
- "memory on (`memwal-mcp auto-save on`, or MEMWAL_AUTO_SAVE=1). When it is off, save only what",
- "the user asks you to save in that turn, and do not offer to turn it on more than once.",
+ "Whether to save things the user did not ask you to save is their standing choice, made once",
+ "in a terminal. When automatic memory is on, save durable facts as they state them; when it is",
+ "off, save only what they ask you to save in that turn. That question is put by `memwal-mcp",
+ "login` and set by `memwal-mcp auto-save on|off` — never ask the user to answer it in chat,",
+ "and never answer it on their behalf.",
].join(" ");
/** Bumped whenever the text above changes, so a stale copy is identifiable. */
-export const MEMORY_POLICY_VERSION = "2026-09-17.1";
+export const MEMORY_POLICY_VERSION = "2026-09-17.2";
// ─── memwal:policy-block:end ─────────────────────────────────────────────────
diff --git a/packages/mcp/plugin/scripts/on_session_start.mjs b/packages/mcp/plugin/scripts/on_session_start.mjs
index 67adbc6e0..d08ad0054 100644
--- a/packages/mcp/plugin/scripts/on_session_start.mjs
+++ b/packages/mcp/plugin/scripts/on_session_start.mjs
@@ -14,7 +14,7 @@ import { autoSaveStatus } from "./lib/auto-save.mjs";
readStdin(); // drain stdin; we don't need any field today
const ns = process.env.MEMWAL_NAMESPACE || "default";
-const { enabled: autoSave } = autoSaveStatus();
+const { enabled: autoSave, pendingConsent } = autoSaveStatus();
const RECALL_AND_RECOVER = [
"Before tasks that reference past work or preferences, recall with memwal_recall.",
@@ -25,12 +25,22 @@ const SAVE_AUTOMATIC =
"Automatic memory is ON. You decide from meaning, in any language or spelling. When the user states a preference, decision, constraint, correction, identity, recurring workflow, or a configuration value such as a hostname, port, region or id, call memwal_remember (or memwal_remember_bulk for several) in that same turn, before you finish replying — do not ask whether to save it, and note that acknowledging it in your reply does not store it. Skip one-off tasks, the current file or bug, and small talk.";
const SAVE_MANUAL =
- "Automatic memory is OFF, which is the default. Save ONLY what the user asks you to save, in the turn they ask — do not save a fact just because it looks durable. Recall is unaffected. If automatic saving would clearly help them, you may say ONCE that `memwal-mcp auto-save on` turns it on, then drop it.";
+ "Automatic memory is OFF. Save ONLY what the user asks you to save, in the turn they ask — do not save a fact just because it looks durable. Recall is unaffected. If automatic saving would clearly help them, you may say ONCE that `memwal-mcp auto-save on` turns it on, then drop it.";
+
+/**
+ * Shown until the user has answered the login question. Says where things
+ * stand, and points at the terminal — the answer is only ever given there, so
+ * the agent must not try to collect it in chat (WALM-642).
+ */
+const CONSENT_PENDING = autoSave
+ ? "The user has not yet confirmed this setting — it is carried over from before the choice existed. If they ask about it, tell them `memwal-mcp login` in a terminal puts the question, and `memwal-mcp auto-save on|off` sets it directly. Do not ask them to answer it in chat and do not answer it for them."
+ : "The user has not yet been asked whether to turn automatic memory on, so it is off until they answer. If they ask about it, tell them `memwal-mcp login` in a terminal puts the question, and `memwal-mcp auto-save on|off` sets it directly. Do not ask them to answer it in chat and do not answer it for them.";
const context = [
`Walrus Memory is this user's memory system, exposed via the memwal_* tools (namespace: ${ns}).`,
"Use it as the PRIMARY place to store and recall durable facts — prefer the memwal_* tools over any built-in or local memory feature, so the user's memory stays portable and persistent on Walrus.",
autoSave ? SAVE_AUTOMATIC : SAVE_MANUAL,
+ ...(pendingConsent ? [CONSENT_PENDING] : []),
...RECALL_AND_RECOVER,
SECRET_EXCLUSION_RULES,
].join(" ");
diff --git a/packages/mcp/src/auto-save.ts b/packages/mcp/src/auto-save.ts
index 5cbd6652d..5701470b0 100644
--- a/packages/mcp/src/auto-save.ts
+++ b/packages/mcp/src/auto-save.ts
@@ -1,35 +1,56 @@
/**
- * Automatic-save opt-in (WALM-642).
+ * Automatic-save consent (WALM-642).
*
* Saving a fact the user explicitly asked for has never needed permission and
- * still does not. What this module gates is the other thing the package does:
- * telling the model, unprompted, to save anything it judges durable. That
- * guidance ships from three places — the MCP `instructions` field, the
- * cold-start tool descriptions, and the plugin's lifecycle hooks — and until
- * now it was always on, so a preference stated next to a password could be
- * forwarded whole without the user ever choosing automatic memory.
+ * still does not. What this module governs is the other thing the package does:
+ * saving what it judges durable, without being asked. That is what MemWal is
+ * for — so it is ON — but a person has to have been told what it means first.
*
- * Default OFF. An install that says nothing saves nothing on its own.
+ * ── Three states, not two ───────────────────────────────────────────────────
+ * `off` and `on` are answers. `unset` is the absence of one, and it is not the
+ * same as `off`:
+ *
+ * - **Answered** (`autoSave: true|false` in settings.json) — from the login
+ * prompt or from `memwal-mcp auto-save on|off`. Nothing re-asks.
+ * - **Unset on an install that predates this change** — no settings file at
+ * all, but credentials on disk. These users have been auto-saving all
+ * along; switching them off would be a regression dressed up as caution, so
+ * saving CONTINUES and the question is put to them at their next
+ * interactive login.
+ * - **Unset on a new install** (`autoSaveConsent: "pending"`, stamped the
+ * moment a post-change install first appears) — nobody has been asked, so
+ * nothing is saved unprompted until someone answers. Explicit tool calls
+ * and recall keep working the whole time.
+ *
+ * The stamp is what separates the last two. Without it a headless install could
+ * sign in through the `memwal_login` tool, look indistinguishable from a
+ * long-standing user, and start saving on its own — consent by never having
+ * been asked.
*
* ── Where the answer comes from ────────────────────────────────────────────
* Two mechanisms, both of which the package already uses, and no third one:
*
* 1. `MEMWAL_AUTO_SAVE` — the env-var surface every other option has
* (`MEMWAL_NAMESPACE`, `MEMWAL_SERVER_URL`, ...). Set it in the client's
- * `env` block to pin one MCP server on or off.
+ * `env` block to pin one MCP server on or off. Setting it deliberately is
+ * itself an answer, so it also stops the login prompt.
* 2. `settings.json`, next to `credentials.json` — resolved by
* `credsPath()`, so it inherits project-local-beats-global and the
* `MEMWAL_CREDS_DIR` override for free, and a project that scopes its
* credentials scopes its memory behaviour with them.
*
- * Env beats file, file beats off — the same CLI > env > default ordering the
- * rest of the package uses, with the file standing in for the CLI because the
- * plugin's hooks are separate processes that never see the MCP server's argv.
- * That is also why the state has to live on disk at all: a hook is spawned by
- * the client, not by this package, and inherits none of its configuration.
+ * The state has to live on disk rather than in configuration because the
+ * plugin's hooks are separate processes: a hook is spawned by the client, not
+ * by this package, and inherits none of the MCP server's `env` or argv.
*/
import { dirname, join } from "node:path";
-import { chmodSync, existsSync, mkdirSync, readFileSync, writeFileSync } from "node:fs";
+import {
+ chmodSync,
+ existsSync,
+ mkdirSync,
+ readFileSync,
+ writeFileSync,
+} from "node:fs";
import { credsPath } from "./auth.js";
import { log } from "./logger.js";
@@ -38,21 +59,23 @@ export const AUTO_SAVE_ENV = "MEMWAL_AUTO_SAVE";
const SETTINGS_FILE = "settings.json";
-/** Where the opt-in is stored: beside whichever credentials file is in play. */
+/** Where the answer is stored: beside whichever credentials file is in play. */
export function settingsPath(): string {
return join(dirname(credsPath()), SETTINGS_FILE);
}
/** Shape of `settings.json`. Unknown keys are preserved on write. */
interface MemWalSettings {
- /** Present only once the user has made a choice either way. */
+ /** Present only once the question has actually been answered. */
autoSave?: boolean;
+ /** `"pending"` marks an install created after WALM-642 with no answer yet. */
+ autoSaveConsent?: string;
[key: string]: unknown;
}
/**
* Parse a human-written boolean. Returns null for "not set" and for anything
- * unparseable — an unreadable value must not be read as consent.
+ * unparseable — an unreadable value must not be read as an answer.
*/
export function parseBooleanSetting(raw: string | undefined | null): boolean | null {
if (raw === undefined || raw === null) return null;
@@ -70,66 +93,130 @@ function readSettings(): MemWalSettings {
const parsed = JSON.parse(readFileSync(path, "utf8"));
return parsed && typeof parsed === "object" ? (parsed as MemWalSettings) : {};
} catch {
- // A corrupt settings file is not consent. Fall through to the default.
+ // A corrupt settings file is not an answer. Fall through to unset.
return {};
}
}
-export type AutoSaveSource = "env" | "settings" | "default";
+function writeSettings(next: MemWalSettings): string {
+ const path = settingsPath();
+ mkdirSync(dirname(path), { recursive: true, mode: 0o700 });
+ writeFileSync(path, `${JSON.stringify(next, null, 2)}\n`, { mode: 0o600 });
+ // `writeFileSync`'s `mode` follows POSIX `open()`: the kernel applies it
+ // when it CREATES the inode and ignores it for one that already exists, so
+ // a file left permissive by anything else would keep its old mode forever.
+ // Unlike credentials.json (see `writeSecretFile` in auth.ts) there is no
+ // secret in flight here, so tightening afterwards is enough — no reader can
+ // learn anything from the window, only whether the flag is set.
+ chmodSync(path, 0o600);
+ return path;
+}
+
+/** `on`/`off` are answers; `unset` means nobody has been asked yet. */
+export type AutoSaveState = "on" | "off" | "unset";
+
+export type AutoSaveSource =
+ /** `MEMWAL_AUTO_SAVE` in this process's environment. */
+ | "env"
+ /** An answer written to settings.json. */
+ | "settings"
+ /** Unset, on an install that predates the consent prompt — keeps saving. */
+ | "legacy"
+ /** Unset, on an install created after it — saves nothing until answered. */
+ | "unanswered";
export interface AutoSaveStatus {
+ /** Whether this process may save without being asked. */
enabled: boolean;
+ /** The stored answer, or `unset`. */
+ state: AutoSaveState;
source: AutoSaveSource;
- /** Where a persisted choice lives (or would be written). */
+ /** True while the question is still owed a human answer. */
+ pendingConsent: boolean;
+ /** Where an answer lives (or would be written). */
path: string;
}
-/** Resolve the opt-in: env, then the settings file, then off. */
+/**
+ * Resolve the current state: environment, then the stored answer, then the
+ * unset rules above.
+ */
export function autoSaveStatus(): AutoSaveStatus {
const path = settingsPath();
+ const settings = readSettings();
+ const answered =
+ typeof settings.autoSave === "boolean" ? settings.autoSave : null;
+ const state: AutoSaveState =
+ answered === null ? "unset" : answered ? "on" : "off";
+
const fromEnv = parseBooleanSetting(process.env[AUTO_SAVE_ENV]);
- if (fromEnv !== null) return { enabled: fromEnv, source: "env", path };
+ if (fromEnv !== null) {
+ // Overrides the behaviour, and counts as a deliberate configuration
+ // act: someone who set this does not need to be asked as well.
+ return { enabled: fromEnv, state, source: "env", pendingConsent: false, path };
+ }
- const settings = readSettings();
- if (typeof settings.autoSave === "boolean") {
- return { enabled: settings.autoSave, source: "settings", path };
+ if (answered !== null) {
+ return { enabled: answered, state, source: "settings", pendingConsent: false, path };
}
- return { enabled: false, source: "default", path };
+
+ // Unset. Which way it falls depends on whether this install was ever in a
+ // position to have been asked — see the module comment.
+ const stamped = settings.autoSaveConsent === "pending";
+ const preExisting = !stamped && existsSync(credsPath());
+ return {
+ enabled: preExisting,
+ state: "unset",
+ source: preExisting ? "legacy" : "unanswered",
+ pendingConsent: true,
+ path,
+ };
}
/**
- * True when the user has turned automatic saving on.
+ * True when this process may save without being asked.
*
- * Read at call time, never cached: a login can move `credsPath()` and the user
- * can flip the setting between calls in the same process.
+ * Read at call time, never cached: a login can move `credsPath()`, and the
+ * answer can be written between calls in the same process.
*/
export function isAutoSaveEnabled(): boolean {
return autoSaveStatus().enabled;
}
+/** True while a human still owes the consent question an answer. */
+export function isConsentPending(): boolean {
+ return autoSaveStatus().pendingConsent;
+}
+
/**
- * Persist the choice, preserving any other keys already in the file.
- *
- * Written `0600` into the `0700` credentials directory. It holds no secret, but
- * it decides whether this machine saves memories unprompted, so it is not
- * something another account on the box should be able to flip.
+ * Record the answer, preserving any other keys already in the file. Clears the
+ * pending stamp — the question has been answered and must not be asked again.
*/
export function setAutoSave(enabled: boolean): { path: string; enabled: boolean } {
- const path = settingsPath();
const next: MemWalSettings = { ...readSettings(), autoSave: enabled };
- mkdirSync(dirname(path), { recursive: true, mode: 0o700 });
- writeFileSync(path, `${JSON.stringify(next, null, 2)}\n`, { mode: 0o600 });
- // `writeFileSync`'s `mode` follows POSIX `open()`: the kernel applies it
- // when it CREATES the inode and ignores it for one that already exists, so
- // a file left permissive by anything else would keep its old mode forever.
- // Unlike credentials.json (see `writeSecretFile` in auth.ts) there is no
- // secret in flight here, so tightening afterwards is enough — no reader can
- // learn anything from the window, only whether the flag is set.
- chmodSync(path, 0o600);
+ delete next.autoSaveConsent;
+ const path = writeSettings(next);
log.info("autosave.set", { enabled, path });
return { path, enabled };
}
+/**
+ * Mark this install as created after the consent prompt existed, so an unset
+ * state here is read as "never asked" rather than "long-standing user".
+ *
+ * Called once when a brand-new install first appears — before the signed-out
+ * server boots, and immediately after a first interactive login — so that an
+ * abandoned or headless sign-in cannot mature into automatic saving nobody
+ * agreed to. A no-op once any answer exists.
+ */
+export function markConsentPending(): void {
+ const settings = readSettings();
+ if (typeof settings.autoSave === "boolean") return;
+ if (settings.autoSaveConsent === "pending") return;
+ writeSettings({ ...settings, autoSaveConsent: "pending" });
+ log.info("autosave.consent_pending", { path: settingsPath() });
+}
+
/** One line for a TTY, naming the state, where it came from, and how to flip it. */
export function autoSaveSummary(): string {
const status = autoSaveStatus();
@@ -138,9 +225,30 @@ export function autoSaveSummary(): string {
? `from ${AUTO_SAVE_ENV}`
: status.source === "settings"
? `from ${status.path}`
- : "default";
+ : status.source === "legacy"
+ ? "carried over from before this setting existed"
+ : "not answered yet";
const how = status.enabled
? "Turn it off with `memwal-mcp auto-save off`."
: "Turn it on with `memwal-mcp auto-save on`. Facts you explicitly ask to save are stored either way.";
return `Automatic memory: ${status.enabled ? "ON" : "OFF"} (${where}). ${how}`;
}
+
+/**
+ * The single line a non-interactive run prints while consent is outstanding.
+ *
+ * Non-TTY is every MCP client spawn, so this is stderr-only and says what is
+ * happening rather than asking anything — there is no human on the other end of
+ * this stdin, and a prompt here would hang the server forever.
+ */
+export function pendingConsentNotice(): string | null {
+ const status = autoSaveStatus();
+ if (!status.pendingConsent) return null;
+ return status.enabled
+ ? "Automatic memory is ON, carried over from before this setting existed. " +
+ "Run `memwal-mcp auto-save on|off` in a terminal to confirm or change it."
+ : "Automatic memory is waiting on your answer, so nothing is being saved " +
+ "unprompted yet (facts you explicitly ask to save still are). Run " +
+ "`memwal-mcp login` in a terminal to answer, or set it directly with " +
+ "`memwal-mcp auto-save on|off`.";
+}
diff --git a/packages/mcp/src/consent.ts b/packages/mcp/src/consent.ts
new file mode 100644
index 000000000..58b21d890
--- /dev/null
+++ b/packages/mcp/src/consent.ts
@@ -0,0 +1,137 @@
+/**
+ * The automatic-memory consent question (WALM-642).
+ *
+ * Asked at interactive login and NOWHERE else. Deliberately not an MCP tool,
+ * not a tool description, not an instruction, and not anything a model can
+ * reach: a model answering on the user's behalf is not consent, and an
+ * agent-shaped surface for this question would be exactly that. The only caller
+ * is `main()` in index.ts, behind `process.stdin.isTTY`.
+ *
+ * Login is the gate because MemWal is unusable without it, so every real user
+ * passes through it once, and it is already a moment that requires a terminal.
+ *
+ * On the wording — it is the substance of this change, not decoration:
+ *
+ * - It names the consequence ("writes it to your memory without asking each
+ * time"), not the feature. Someone should be able to picture what happens
+ * to them, in their own session, from reading it.
+ * - Permanence leads, because it is the fact that changes the answer. Walrus
+ * is immutable: you can stop saving new memories but you cannot take back
+ * one already saved. Burying that under "we value your privacy" would make
+ * this a cookie banner, which it is not.
+ * - The redaction claim is deliberately hedged. `redaction.ts` documents real
+ * residual gaps — an unlabelled hex secret, a bare BIP-39 word run — so
+ * "safety net, not a guarantee" is the accurate claim and must stay. Saying
+ * more would be selling a promise the code does not make.
+ * - Declining is free and is stated as a normal choice, not a warning. A "no"
+ * is recorded once and never asked again.
+ */
+import { createInterface } from "node:readline";
+import type { Readable, Writable } from "node:stream";
+
+/** The prompt body, everything above the input line. */
+export const CONSENT_PROMPT = [
+ "",
+ "MemWal can save things about you automatically.",
+ "",
+ "What that means: when you state a preference, a decision, or a setting in",
+ 'chat — "I prefer pnpm", "we deploy from dev", "the relayer is at X" —',
+ "MemWal writes it to your memory without asking each time, so it is there",
+ "in your next session and in every other client you use.",
+ "",
+ "Before you choose:",
+ "",
+ " - Saved memories are permanent. They go to Walrus, which is immutable",
+ " storage. You can stop saving new ones at any time, but you cannot",
+ " delete one that is already saved.",
+ " - They are encrypted to your account. Only your delegate key reads them.",
+ " - MemWal strips obvious credentials — API keys, tokens, passwords,",
+ " private keys — before saving. Treat that as a safety net, not a",
+ " guarantee: do not paste secrets into a session with this on.",
+ "",
+ " [1] Save automatically recommended, this is what MemWal is for",
+ ' [2] Only save when I ask nothing is saved unless you say "remember this"',
+ "",
+ "Change this any time with `memwal-mcp auto-save on|off`.",
+ "See what is stored with `memwal_recall`.",
+ "",
+].join("\n");
+
+/** The input line. Enter alone takes the recommended option. */
+export const CONSENT_QUESTION = "Your choice [1/2]: ";
+
+/** What to say when the answer is neither 1 nor 2. */
+export const CONSENT_REPROMPT =
+ "Please answer 1 or 2 (or press Enter for 1).";
+
+/**
+ * Map one line of input to an answer.
+ *
+ * Empty (a bare Enter) accepts option 1. Anything that is not 1 or 2 returns
+ * `null`, which means re-ask — never "assume the recommended one", because a
+ * typo is not an answer to a question about permanent storage.
+ */
+export function interpretConsentAnswer(raw: string): boolean | null {
+ const v = raw.trim().toLowerCase();
+ if (v === "") return true;
+ if (v === "1") return true;
+ if (v === "2") return false;
+ return null;
+}
+
+/**
+ * Ask the question on a terminal and resolve to the answer.
+ *
+ * Resolves `null` — meaning "no answer given" — if the input stream ends first
+ * (Ctrl-D, a closed pipe, a killed terminal). The caller leaves the state unset
+ * in that case and asks again next time rather than recording a choice the user
+ * never made.
+ *
+ * Streams are injected so this is testable without a pty; `main()` passes the
+ * real stdin and stderr. Output goes to stderr because stdout belongs to the
+ * MCP protocol on every other path in this package.
+ */
+export async function askAutoSaveConsent(opts: {
+ input: Readable;
+ output: Writable;
+ /** Guard against an accidental headless call. Defaults to requiring a TTY. */
+ isTTY?: boolean;
+}): Promise {
+ // A prompt with nobody in front of it is a hang, not a question. The
+ // caller already checks this; the second check is here because the cost of
+ // getting it wrong is an MCP server that never finishes starting.
+ if (opts.isTTY === false) return null;
+
+ opts.output.write(`${CONSENT_PROMPT}\n`);
+
+ const rl = createInterface({ input: opts.input, terminal: false });
+ try {
+ opts.output.write(CONSENT_QUESTION);
+ // The interface's async iterator, rather than a promise per `line`
+ // event: attaching a fresh listener after each answer drops any line
+ // that arrived while nothing was listening, which deadlocks the re-ask
+ // path the moment input is buffered rather than typed.
+ for await (const line of rl) {
+ const answer = interpretConsentAnswer(line);
+ if (answer !== null) return answer;
+ opts.output.write(`${CONSENT_REPROMPT}\n`);
+ opts.output.write(CONSENT_QUESTION);
+ }
+ // Exhausted without an answer: Ctrl-D, a closed pipe, a killed
+ // terminal. Not a choice, and not recorded as one.
+ opts.output.write("\n");
+ return null;
+ } finally {
+ rl.close();
+ }
+}
+
+/** What is printed once the answer is in. */
+export function consentOutcomeNotice(enabled: boolean, path: string): string {
+ return enabled
+ ? "Automatic memory is ON. MemWal will save durable facts as you state them. " +
+ `Turn it off any time with \`memwal-mcp auto-save off\`. (${path})`
+ : "Automatic memory is OFF. Nothing is saved unless you ask for it — " +
+ '"remember this" still works, and so does recall. ' +
+ `Turn it on any time with \`memwal-mcp auto-save on\`. (${path})`;
+}
diff --git a/packages/mcp/src/index.ts b/packages/mcp/src/index.ts
index 8d1fe5cde..defd74afb 100644
--- a/packages/mcp/src/index.ts
+++ b/packages/mcp/src/index.ts
@@ -23,7 +23,15 @@ import { recoverPendingLogin, formatStrandedLoginNotice } from "./recovery.js";
import { runAuthRequiredServer } from "./auth-required.js";
import { notePendingLoginSuccess, runBridge } from "./bridge.js";
import { loginFlow } from "./login.js";
-import { autoSaveStatus, autoSaveSummary, setAutoSave, AUTO_SAVE_ENV } from "./auto-save.js";
+import {
+ autoSaveStatus,
+ autoSaveSummary,
+ markConsentPending,
+ pendingConsentNotice,
+ setAutoSave,
+ AUTO_SAVE_ENV,
+} from "./auto-save.js";
+import { askAutoSaveConsent, consentOutcomeNotice } from "./consent.js";
import { log, note } from "./logger.js";
/**
@@ -450,6 +458,13 @@ export async function main(argv: string[] = process.argv.slice(2)): Promise {
+ if (!autoSaveStatus().pendingConsent) return;
+
+ const answer = await askAutoSaveConsent({
+ input: process.stdin,
+ output: process.stderr,
+ isTTY: process.stdin.isTTY === true,
+ });
+ if (answer === null) {
+ note(
+ "No answer recorded — automatic memory is unchanged and you will be " +
+ "asked again next time. Set it directly with `memwal-mcp auto-save on|off`.",
+ );
+ return;
+ }
+ const { path } = setAutoSave(answer);
+ note(consentOutcomeNotice(answer, path));
+}
+
function printHelp(): void {
process.stderr.write(helpText() + "\n");
}
@@ -563,16 +623,21 @@ export function helpText(): string {
" delegate key or relayer changes.",
" memwal-mcp revoke-project Withdraw that approval.",
" memwal-mcp auto-save on|off Turn automatic memory on or off.",
- " OFF by default: the agent saves only",
- " what you ask it to save. Turning it",
- " on lets the agent save durable facts",
- " unprompted. Credentials (passwords,",
- " API keys, tokens, private keys, seed",
- " phrases, auth headers, URLs with an",
- " embedded user:password) are excluded",
- " and stripped before any write, in",
- " both modes. Stored in settings.json",
- " next to credentials.json.",
+ " ON once you agree to it: `login` asks",
+ " in the terminal the first time, and",
+ " nothing is saved unprompted until you",
+ " answer. Saved memories are permanent",
+ " — Walrus is immutable storage — so",
+ " you can stop saving new ones but",
+ " cannot delete one already saved.",
+ " Credentials (passwords, API keys,",
+ " tokens, private keys, seed phrases,",
+ " auth headers, URLs with an embedded",
+ " user:password) are stripped before",
+ " any write either way — a safety net,",
+ " not a guarantee. Stored in",
+ " settings.json next to",
+ " credentials.json.",
" memwal-mcp auto-save Report the current setting.",
" memwal-mcp --help Show this help.",
"",
@@ -610,8 +675,8 @@ export function helpText(): string {
" project-local and global files.",
" MEMWAL_NAMESPACE same as --namespace",
" MEMWAL_AUTO_SAVE=1 Automatic memory for this server",
- " only; overrides settings.json.",
- " Unset or 0 = off (the default).",
+ " only; overrides settings.json and",
+ " skips the login question. 0 = off.",
" MEMWAL_MCP_DEBUG=1 Verbose stderr logging.",
"",
"Minimal MCP client config (Cursor, Claude Desktop, etc.):",
@@ -665,7 +730,15 @@ export {
revokeProjectCredsApproval,
formatProjectCredsNotice,
} from "./auth.js";
-export { isAutoSaveEnabled, autoSaveStatus, setAutoSave, settingsPath } from "./auto-save.js";
+export {
+ isAutoSaveEnabled,
+ isConsentPending,
+ autoSaveStatus,
+ setAutoSave,
+ markConsentPending,
+ settingsPath,
+} from "./auto-save.js";
+export { askAutoSaveConsent, interpretConsentAnswer, CONSENT_PROMPT } from "./consent.js";
export { loginFlow } from "./login.js";
export { runBridge } from "./bridge.js";
export type { MemWalCredentials, CredsResolution, ProjectCredsDecision } from "./auth.js";
diff --git a/packages/mcp/src/memory-policy.ts b/packages/mcp/src/memory-policy.ts
index 15c0a5292..2e187d1be 100644
--- a/packages/mcp/src/memory-policy.ts
+++ b/packages/mcp/src/memory-policy.ts
@@ -67,16 +67,18 @@ export const SECRET_EXCLUSION_SUMMARY = [
].join(" ");
/**
- * Automatic saving is opt-in, and this is the sentence that says so. A direct
- * request from the user ("remember that ...") is never gated by it — the gate
- * is only on saving something the user did not ask you to save.
+ * Whether to save unprompted is the user's standing choice, and this is the
+ * sentence that says so. A direct request ("remember that ...") is never gated
+ * by it — the gate is only on saving something the user did not ask you to save.
*/
export const AUTO_SAVE_OPT_IN_RULE = [
- "Saving something the user did not ask you to save is OFF unless they have turned automatic",
- "memory on (`memwal-mcp auto-save on`, or MEMWAL_AUTO_SAVE=1). When it is off, save only what",
- "the user asks you to save in that turn, and do not offer to turn it on more than once.",
+ "Whether to save things the user did not ask you to save is their standing choice, made once",
+ "in a terminal. When automatic memory is on, save durable facts as they state them; when it is",
+ "off, save only what they ask you to save in that turn. That question is put by `memwal-mcp",
+ "login` and set by `memwal-mcp auto-save on|off` — never ask the user to answer it in chat,",
+ "and never answer it on their behalf.",
].join(" ");
/** Bumped whenever the text above changes, so a stale copy is identifiable. */
-export const MEMORY_POLICY_VERSION = "2026-09-17.1";
+export const MEMORY_POLICY_VERSION = "2026-09-17.2";
// ─── memwal:policy-block:end ─────────────────────────────────────────────────
diff --git a/packages/mcp/test/auto-save-optin.test.mjs b/packages/mcp/test/auto-save-optin.test.mjs
index f254a398c..7d6d87f56 100644
--- a/packages/mcp/test/auto-save-optin.test.mjs
+++ b/packages/mcp/test/auto-save-optin.test.mjs
@@ -1,10 +1,14 @@
/**
- * Automatic saving is opt-in, and OFF is the default (WALM-642).
+ * Automatic saving is ON, once a human has been asked (WALM-642).
*
- * The thing being gated is narrow and worth naming: saving something the user
- * did not ask to have saved. A direct request ("remember that ...") is not
+ * The thing being governed is narrow and worth naming: saving something the
+ * user did not ask to have saved. A direct request ("remember that ...") is not
* gated, and neither is recall — so these tests check both that the guidance
- * goes quiet when the opt-in is off AND that nothing else goes quiet with it.
+ * goes quiet when the answer is "off" AND that nothing else goes quiet with it.
+ *
+ * Three states, and the two unset ones are the interesting half: an install
+ * that predates the consent prompt keeps saving (that is its status quo, not a
+ * new grant), while one created after it saves nothing until someone answers.
*
* Both halves of the opt-in are covered, because they are two separate
* implementations of the same rule: the TypeScript one the MCP server reads,
@@ -15,6 +19,7 @@
import { test } from "node:test";
import assert from "node:assert/strict";
import { spawnSync } from "node:child_process";
+import { Readable, Writable } from "node:stream";
import { mkdtempSync, mkdirSync, readFileSync, statSync, writeFileSync, rmSync } from "node:fs";
import { tmpdir } from "node:os";
import { dirname, join, resolve } from "node:path";
@@ -23,13 +28,20 @@ import { fileURLToPath } from "node:url";
import {
autoSaveStatus,
isAutoSaveEnabled,
+ markConsentPending,
parseBooleanSetting,
setAutoSave,
settingsPath,
AUTO_SAVE_ENV,
} from "../dist/auto-save.js";
+import {
+ askAutoSaveConsent,
+ interpretConsentAnswer,
+ CONSENT_PROMPT,
+} from "../dist/consent.js";
import * as hookAutoSave from "../plugin/scripts/lib/auto-save.mjs";
import { parseArgs, helpText } from "../dist/index.js";
+import { TOOL_DEFINITIONS } from "../dist/auth-required.js";
const __dirname = dirname(fileURLToPath(import.meta.url));
const SCRIPTS = resolve(__dirname, "../plugin/scripts");
@@ -77,29 +89,81 @@ function withEnv(vars, fn) {
// ── the resolver ────────────────────────────────────────────────────────────
-test("the default is off", () => {
+test("a new install saves nothing until someone answers", () => {
const dir = freshCredsDir();
withEnv({ MEMWAL_CREDS_DIR: dir, [AUTO_SAVE_ENV]: undefined }, () => {
+ markConsentPending();
assert.equal(isAutoSaveEnabled(), false);
assert.deepEqual(autoSaveStatus(), {
enabled: false,
- source: "default",
+ state: "unset",
+ source: "unanswered",
+ pendingConsent: true,
path: join(dir, "settings.json"),
});
});
rmSync(dir, { recursive: true, force: true });
});
-test("a persisted choice is read back, either way", () => {
+test("an install that predates the prompt keeps saving, and is still asked", () => {
+ // No settings file at all, credentials on disk: someone who has been
+ // auto-saving since before this setting existed. Switching them off would
+ // be a regression dressed up as caution.
+ const dir = freshCredsDir();
+ withEnv({ MEMWAL_CREDS_DIR: dir, [AUTO_SAVE_ENV]: undefined }, () => {
+ writeFileSync(join(dir, "credentials.json"), "{}");
+ const status = autoSaveStatus();
+ assert.equal(status.enabled, true);
+ assert.equal(status.state, "unset");
+ assert.equal(status.source, "legacy");
+ // Carried over, not granted — so the question is still owed.
+ assert.equal(status.pendingConsent, true);
+ });
+ rmSync(dir, { recursive: true, force: true });
+});
+
+test("the pending stamp stops a headless sign-in maturing into consent", () => {
+ // Without the stamp, a brand-new install that signs in through the
+ // `memwal_login` tool would be indistinguishable from a long-standing user
+ // the moment credentials appear, and would start saving with nobody ever
+ // having been asked.
+ const dir = freshCredsDir();
+ withEnv({ MEMWAL_CREDS_DIR: dir, [AUTO_SAVE_ENV]: undefined }, () => {
+ markConsentPending();
+ writeFileSync(join(dir, "credentials.json"), "{}");
+ const status = autoSaveStatus();
+ assert.equal(status.enabled, false, "consent by never being asked");
+ assert.equal(status.source, "unanswered");
+ assert.equal(status.pendingConsent, true);
+ });
+ rmSync(dir, { recursive: true, force: true });
+});
+
+test("an answer is read back, either way, and is never asked for again", () => {
const dir = freshCredsDir();
withEnv({ MEMWAL_CREDS_DIR: dir, [AUTO_SAVE_ENV]: undefined }, () => {
+ markConsentPending();
+
setAutoSave(true);
assert.equal(isAutoSaveEnabled(), true);
+ assert.equal(autoSaveStatus().state, "on");
assert.equal(autoSaveStatus().source, "settings");
+ assert.equal(autoSaveStatus().pendingConsent, false);
+ // Declining must cost nothing and must not be nagged at.
setAutoSave(false);
assert.equal(isAutoSaveEnabled(), false);
- assert.equal(autoSaveStatus().source, "settings");
+ assert.equal(autoSaveStatus().state, "off");
+ assert.equal(
+ autoSaveStatus().pendingConsent,
+ false,
+ "a declined answer must not put the question back",
+ );
+ // Even with credentials present — the rule that keeps a pre-existing
+ // install saving must not resurrect a deliberate "no".
+ writeFileSync(join(dir, "credentials.json"), "{}");
+ assert.equal(isAutoSaveEnabled(), false);
+ assert.equal(autoSaveStatus().pendingConsent, false);
});
rmSync(dir, { recursive: true, force: true });
});
@@ -118,13 +182,15 @@ test("the settings file is not world-readable and keeps unrelated keys", () => {
rmSync(dir, { recursive: true, force: true });
});
-test("the environment overrides the file, in both directions", () => {
+test("the environment overrides every state, in both directions", () => {
const dir = freshCredsDir();
withEnv({ MEMWAL_CREDS_DIR: dir, [AUTO_SAVE_ENV]: undefined }, () => {
+ // over an answered "off" / "on"
setAutoSave(false);
withEnv({ [AUTO_SAVE_ENV]: "1" }, () => {
assert.equal(isAutoSaveEnabled(), true);
assert.equal(autoSaveStatus().source, "env");
+ assert.equal(autoSaveStatus().state, "off", "the stored answer is untouched");
});
setAutoSave(true);
withEnv({ [AUTO_SAVE_ENV]: "0" }, () => {
@@ -133,6 +199,28 @@ test("the environment overrides the file, in both directions", () => {
});
});
rmSync(dir, { recursive: true, force: true });
+
+ // over each unset state, and it settles the question too — someone who set
+ // this deliberately does not also need to be prompted.
+ const unanswered = freshCredsDir();
+ withEnv({ MEMWAL_CREDS_DIR: unanswered, [AUTO_SAVE_ENV]: undefined }, () => {
+ markConsentPending();
+ withEnv({ [AUTO_SAVE_ENV]: "1" }, () => {
+ assert.equal(isAutoSaveEnabled(), true);
+ assert.equal(autoSaveStatus().pendingConsent, false);
+ });
+ });
+ rmSync(unanswered, { recursive: true, force: true });
+
+ const legacy = freshCredsDir();
+ withEnv({ MEMWAL_CREDS_DIR: legacy, [AUTO_SAVE_ENV]: undefined }, () => {
+ writeFileSync(join(legacy, "credentials.json"), "{}");
+ withEnv({ [AUTO_SAVE_ENV]: "off" }, () => {
+ assert.equal(isAutoSaveEnabled(), false);
+ assert.equal(autoSaveStatus().pendingConsent, false);
+ });
+ });
+ rmSync(legacy, { recursive: true, force: true });
});
test("an unreadable or unparseable value is not consent", () => {
@@ -148,7 +236,8 @@ test("an unreadable or unparseable value is not consent", () => {
withEnv({ MEMWAL_CREDS_DIR: dir, [AUTO_SAVE_ENV]: undefined }, () => {
writeFileSync(join(dir, "settings.json"), "{ not json");
assert.equal(isAutoSaveEnabled(), false);
- assert.equal(autoSaveStatus().source, "default");
+ assert.equal(autoSaveStatus().state, "unset");
+ assert.equal(autoSaveStatus().pendingConsent, true);
});
rmSync(dir, { recursive: true, force: true });
});
@@ -159,9 +248,15 @@ test("the hook-side resolver answers identically to the compiled one", () => {
assert.equal(hookAutoSave.isAutoSaveEnabled(), isAutoSaveEnabled());
assert.equal(hookAutoSave.settingsPath(), settingsPath());
+ // ...on every state, not just the answered one.
+ markConsentPending();
+ assert.equal(hookAutoSave.isAutoSaveEnabled(), false);
+ assert.equal(hookAutoSave.autoSaveStatus().source, "unanswered");
+
setAutoSave(true);
assert.equal(hookAutoSave.isAutoSaveEnabled(), true);
assert.equal(hookAutoSave.autoSaveStatus().source, "settings");
+ assert.equal(hookAutoSave.autoSaveStatus().pendingConsent, false);
withEnv({ [AUTO_SAVE_ENV]: "off" }, () => {
assert.equal(hookAutoSave.isAutoSaveEnabled(), false);
@@ -173,8 +268,9 @@ test("the hook-side resolver answers identically to the compiled one", () => {
// ── the hooks ───────────────────────────────────────────────────────────────
-test("SessionStart does not tell the agent to save until the user opts in", () => {
+test("SessionStart goes quiet about saving when the user answered no", () => {
const dir = freshCredsDir();
+ writeSettings(dir, { autoSave: false });
const off = runHook("on_session_start.mjs", {}, { MEMWAL_CREDS_DIR: dir });
assert.match(off, /Automatic memory is OFF/);
@@ -198,8 +294,9 @@ test("SessionStart does not tell the agent to save until the user opts in", () =
rmSync(dir, { recursive: true, force: true });
});
-test("UserPromptSubmit injects a save-nothing rubric while the opt-in is off", () => {
+test("UserPromptSubmit injects a save-nothing rubric when the user answered no", () => {
const dir = freshCredsDir();
+ writeSettings(dir, { autoSave: false });
const prompt = "I always use pnpm and my staging canary is coral-fox-77.";
const off = runHook(
@@ -225,8 +322,9 @@ test("UserPromptSubmit injects a save-nothing rubric while the opt-in is off", (
rmSync(dir, { recursive: true, force: true });
});
-test("PostToolUse stops nudging a save after an error while the opt-in is off", () => {
+test("PostToolUse stops nudging a save after an error when the user answered no", () => {
const dir = freshCredsDir();
+ writeSettings(dir, { autoSave: false });
// Must trip `detectError` in lib/signals.mjs (a strong marker) and clear
// the hook's 50-character minimum, or the hook stays silent for reasons
// that have nothing to do with the opt-in.
@@ -248,8 +346,9 @@ test("PostToolUse stops nudging a save after an error while the opt-in is off",
rmSync(dir, { recursive: true, force: true });
});
-test("a hook with the opt-in off still exits 0 and never blocks the session", () => {
+test("a hook with saving off still exits 0 and never blocks the session", () => {
const dir = freshCredsDir();
+ writeSettings(dir, { autoSave: false });
for (const script of ["on_session_start.mjs", "on_user_prompt.mjs", "on_post_tool.mjs"]) {
const result = spawnSync(process.execPath, [join(SCRIPTS, script)], {
input: JSON.stringify({ prompt: "a reasonably long prompt about pnpm" }),
@@ -263,6 +362,79 @@ test("a hook with the opt-in off still exits 0 and never blocks the session", ()
// ── the surface a user actually turns it on from ────────────────────────────
+// ── the consent question ────────────────────────────────────────────────────
+
+test("the prompt names the consequence, leads with permanence, and hedges the redaction", () => {
+ // These are the three wording rules the change exists for, so they are
+ // asserted rather than left to a reviewer's memory.
+ assert.match(CONSENT_PROMPT, /writes it to your memory without asking each time/);
+ assert.match(CONSENT_PROMPT, /Saved memories are permanent/);
+ assert.match(CONSENT_PROMPT, /immutable/);
+ assert.match(CONSENT_PROMPT, /cannot\s+delete one that is already saved/);
+ assert.match(CONSENT_PROMPT, /safety net, not a\s+guarantee/);
+ // Permanence comes first among the bullets — it is the fact that changes
+ // the answer.
+ const bullets = CONSENT_PROMPT.split("\n").filter((l) => l.trim().startsWith("- "));
+ assert.equal(bullets.length, 3);
+ assert.match(bullets[0], /permanent/);
+ // Both options are offered plainly; declining is not dressed as a warning.
+ assert.match(CONSENT_PROMPT, /\[1\] Save automatically/);
+ assert.match(CONSENT_PROMPT, /\[2\] Only save when I ask/);
+ assert.match(CONSENT_PROMPT, /auto-save on\|off/);
+});
+
+test("Enter takes option 1, and anything unrecognised re-asks rather than assuming", () => {
+ assert.equal(interpretConsentAnswer(""), true);
+ assert.equal(interpretConsentAnswer(" "), true);
+ assert.equal(interpretConsentAnswer("1"), true);
+ assert.equal(interpretConsentAnswer("2"), false);
+ // A typo is not an answer to a question about permanent storage.
+ for (const raw of ["y", "n", "3", "yes", "maybe", "11"]) {
+ assert.equal(interpretConsentAnswer(raw), null, `"${raw}" must re-ask`);
+ }
+});
+
+/** Drive the prompt with scripted lines and collect what it wrote. */
+async function runPrompt(lines, { isTTY = true } = {}) {
+ const written = [];
+ const input = Readable.from(lines.map((l) => `${l}\n`));
+ const output = new Writable({
+ write(chunk, _enc, cb) {
+ written.push(chunk.toString());
+ cb();
+ },
+ });
+ const answer = await askAutoSaveConsent({ input, output, isTTY });
+ return { answer, output: written.join("") };
+}
+
+test("an answer is taken from the terminal, and a bad one is re-asked", async () => {
+ assert.equal((await runPrompt(["1"])).answer, true);
+ assert.equal((await runPrompt([""])).answer, true);
+ assert.equal((await runPrompt(["2"])).answer, false);
+
+ const retried = await runPrompt(["banana", "2"]);
+ assert.equal(retried.answer, false);
+ assert.match(retried.output, /Please answer 1 or 2/);
+ // The question is put again, not assumed away.
+ assert.equal(retried.output.split("Your choice").length - 1, 2);
+});
+
+test("a closed stream is not an answer, and does not hang", async () => {
+ // Ctrl-D, a killed terminal, a closed pipe. Recording a choice here would
+ // be recording one the user never made.
+ const { answer } = await runPrompt([]);
+ assert.equal(answer, null);
+});
+
+test("the prompt refuses to run without a TTY", async () => {
+ // Belt and braces with main()'s own check: a prompt with nobody in front of
+ // it is a hang, and this one would hang an MCP server's startup.
+ const { answer, output } = await runPrompt(["1"], { isTTY: false });
+ assert.equal(answer, null);
+ assert.equal(output, "", "nothing may be written to a non-interactive stream");
+});
+
test("`auto-save` parses as a subcommand, and a bare one only reports", () => {
// A bare `auto-save` must not be read as consent to turn it ON — the
// difference between reporting a setting and changing one.
@@ -288,13 +460,64 @@ test("a typo'd flag does not swallow the subcommand or its value", () => {
assert.equal(parsed.autoSave, "on");
});
-test("--help tells the user the setting exists and that it is off by default", () => {
+test("--help tells the user the setting exists and that login asks for it", () => {
// The plugin install path never shows a terminal, so --help and the
// post-login summary are where the choice reaches a person.
const help = helpText();
assert.match(help, /auto-save on\|off/);
- assert.match(help, /OFF by default/);
+ assert.match(help, /login` asks/);
+ assert.match(help, /nothing is saved unprompted until you/);
assert.match(help, /MEMWAL_AUTO_SAVE/);
- // And that credentials are excluded regardless of which way it is set.
+ // Permanence is stated here too — it is the fact that changes the answer.
+ assert.match(help, /permanent/);
+ // And that credentials are stripped regardless of which way it is set.
assert.match(help, /Credentials/);
+ assert.match(help, /not a guarantee/);
+});
+
+// ── the non-interactive path ────────────────────────────────────────────────
+
+test("a non-TTY run never prompts, never hangs, and says where things stand", () => {
+ // Every MCP client spawn lands here. The failure this guards against is not
+ // a wrong answer, it is a server that never finishes starting because
+ // something is waiting on a stdin no human is attached to.
+ const dir = freshCredsDir();
+ const result = spawnSync(
+ process.execPath,
+ [resolve(__dirname, "../dist/bin/memwal-mcp.js"), "auto-save"],
+ {
+ // Piped, not inherited: `process.stdin.isTTY` is undefined here,
+ // exactly as it is under an MCP client.
+ input: "",
+ encoding: "utf8",
+ timeout: 10_000,
+ env: { ...process.env, MEMWAL_CREDS_DIR: dir, MEMWAL_AUTO_SAVE: "" },
+ },
+ );
+ assert.equal(result.status, 0, result.stderr);
+ assert.notEqual(result.signal, "SIGTERM", "the process hung waiting on stdin");
+ // Reports, does not ask.
+ assert.doesNotMatch(result.stderr, /Your choice/);
+ assert.doesNotMatch(result.stderr, /Save automatically/);
+ assert.match(result.stderr, /Automatic memory: (ON|OFF)/);
+ rmSync(dir, { recursive: true, force: true });
+});
+
+test("consent is not reachable from anything the model can call", () => {
+ // The single most important constraint in this change: a model answering
+ // on the user's behalf is not consent. The question lives in consent.ts,
+ // is called only from main() behind `process.stdin.isTTY`, and must never
+ // appear in a tool list, a tool description or the instructions.
+ const toolSurfaces = [
+ readFileSync(resolve(__dirname, "../dist/auth-required.js"), "utf8"),
+ readFileSync(resolve(__dirname, "../dist/instructions.js"), "utf8"),
+ readFileSync(resolve(__dirname, "../dist/bridge.js"), "utf8"),
+ ].join("\n");
+ assert.doesNotMatch(toolSurfaces, /askAutoSaveConsent/);
+ assert.doesNotMatch(toolSurfaces, /CONSENT_PROMPT/);
+ assert.doesNotMatch(toolSurfaces, /Your choice \[1\/2\]/);
+
+ // And no tool is named for it.
+ const names = TOOL_DEFINITIONS.map((t) => t.name);
+ assert.ok(!names.some((n) => /consent|auto_?save/i.test(n)), names.join(", "));
});
diff --git a/packages/mcp/test/user-prompt-hook.test.mjs b/packages/mcp/test/user-prompt-hook.test.mjs
index 99aef5d88..672af8a0f 100644
--- a/packages/mcp/test/user-prompt-hook.test.mjs
+++ b/packages/mcp/test/user-prompt-hook.test.mjs
@@ -3,11 +3,12 @@
* one-line nudge. It must not classify remember vs recall from English
* keywords — every substantive prompt in a fresh session gets the same text.
*
- * WALM-642 added the automatic-save opt-in, so "the same text" is now per
- * opt-in state: the tests below pin the ON variant by asking for it
- * explicitly, and `auto-save-optin.test.mjs` pins what the default OFF state
- * injects instead. Every run is pointed at an empty MEMWAL_CREDS_DIR so the
- * developer's own ~/.memwal/settings.json cannot decide the result.
+ * WALM-642 made the save half of that rubric depend on the user's standing
+ * automatic-memory answer, so "the same text" is now per state: the tests below
+ * pin the ON variant by asking for it explicitly, and `auto-save-optin.test.mjs`
+ * pins what an answered-no and an unanswered install inject instead. Every run
+ * is pointed at an empty MEMWAL_CREDS_DIR so the developer's own
+ * ~/.memwal/settings.json cannot decide the result.
*/
import { test } from "node:test";
import assert from "node:assert/strict";
@@ -32,9 +33,9 @@ function runHook(prompt, sessionId = `test-${Math.random().toString(16).slice(2)
env: {
...process.env,
MEMWAL_CREDS_DIR: EMPTY_CREDS_DIR,
- // These cases are about classification, not consent: ask for the
+ // These cases are about classification, not consent: pin the
// automatic-save rubric explicitly so they keep testing the thing
- // they were written for.
+ // they were written for, whatever the resolver would decide.
MEMWAL_AUTO_SAVE: "1",
},
});
diff --git a/services/server/scripts/mcp/tools/memory-policy.ts b/services/server/scripts/mcp/tools/memory-policy.ts
index 018fba9a2..924b5e5e2 100644
--- a/services/server/scripts/mcp/tools/memory-policy.ts
+++ b/services/server/scripts/mcp/tools/memory-policy.ts
@@ -65,16 +65,18 @@ export const SECRET_EXCLUSION_SUMMARY = [
].join(" ");
/**
- * Automatic saving is opt-in, and this is the sentence that says so. A direct
- * request from the user ("remember that ...") is never gated by it — the gate
- * is only on saving something the user did not ask you to save.
+ * Whether to save unprompted is the user's standing choice, and this is the
+ * sentence that says so. A direct request ("remember that ...") is never gated
+ * by it — the gate is only on saving something the user did not ask you to save.
*/
export const AUTO_SAVE_OPT_IN_RULE = [
- "Saving something the user did not ask you to save is OFF unless they have turned automatic",
- "memory on (`memwal-mcp auto-save on`, or MEMWAL_AUTO_SAVE=1). When it is off, save only what",
- "the user asks you to save in that turn, and do not offer to turn it on more than once.",
+ "Whether to save things the user did not ask you to save is their standing choice, made once",
+ "in a terminal. When automatic memory is on, save durable facts as they state them; when it is",
+ "off, save only what they ask you to save in that turn. That question is put by `memwal-mcp",
+ "login` and set by `memwal-mcp auto-save on|off` — never ask the user to answer it in chat,",
+ "and never answer it on their behalf.",
].join(" ");
/** Bumped whenever the text above changes, so a stale copy is identifiable. */
-export const MEMORY_POLICY_VERSION = "2026-09-17.1";
+export const MEMORY_POLICY_VERSION = "2026-09-17.2";
// ─── memwal:policy-block:end ─────────────────────────────────────────────────
From da73a4864b589e670141c9016b55f660ae8210eb Mon Sep 17 00:00:00 2001
From: Harry Phan
Date: Thu, 17 Sep 2026 23:13:54 +0700
Subject: [PATCH 070/132] revert: drop the #918 failed-write report (and
migration 021)
The recall-attached failed_writes payload, failure_reported_at column,
claim-at-delivery path, and MCP formatter are reverted. Agents learn
write outcomes from memwal_remember_status / job status instead.
021 is gone from the tree and from VectorDb::new: origin/dev already
queried the column without applying the migration, which is what made
fresh-schema CI fail. Without the queries, the column is not needed.
Failed-job reclaim (immediate retry of a failed remember) is unchanged.
---
docs/mcp/changelog.mdx | 2 -
packages/mcp/CHANGELOG.md | 2 -
.../021_failed_write_report_ack.sql | 21 --
.../__tests__/recall-failed-writes.test.ts | 88 -------
services/server/scripts/mcp/tools/recall.ts | 53 +---
services/server/src/routes/recall.rs | 110 ---------
services/server/src/routes/remember.rs | 58 +----
services/server/src/storage/db.rs | 233 ------------------
services/server/src/types.rs | 39 ---
9 files changed, 3 insertions(+), 603 deletions(-)
delete mode 100644 services/server/migrations/021_failed_write_report_ack.sql
delete mode 100644 services/server/scripts/mcp/__tests__/recall-failed-writes.test.ts
diff --git a/docs/mcp/changelog.mdx b/docs/mcp/changelog.mdx
index bcd497c0f..7d67d4bdc 100644
--- a/docs/mcp/changelog.mdx
+++ b/docs/mcp/changelog.mdx
@@ -40,8 +40,6 @@ answer: >-
- `memwal_remember` sends a content-derived idempotency key, so the retry its own timeout message invites really does attach to the job already in flight instead of storing a second paid copy. The key is computed by the tool rather than relied on from the SDK, whose published build mints a random UUID per client instance.
- `memwal_remember_status` accepts `job_ids` to settle a whole batch in one call, and reports a mixed batch honestly — a still-uploading row no longer renders the poll timeout as `error=`, which read as a failed write. Settling in one request also matters against the rate limit: 20 ids cost one request, not twenty.
- The bridge's cold-start tool list no longer disagrees with the sidecar's. `memwal_remember_status` advertised only `job_id`, required, under `additionalProperties: false`, so the batch call the tools themselves instruct was rejected until `tools/list_changed` arrived; the `waitMs` ceiling advertised 60000 after the sidecar lowered it to 45000, which came back as an MCP validation error; and `memwal_remember_bulk` still carried its pre-queue description. Tests now pin the parts an agent acts on.
-- The accepted-then-failed report attached to `memwal_recall` no longer tells the agent to re-send text it does not have. The relayer stores only the SEAL ciphertext, so a failed write's wording cannot be recovered — the report now says so and asks the user to restate the fact rather than inviting the agent to guess.
-
### Changed
- `memwal_remember` keeps blocking until the write reaches `done`, so a result carries a real `blob_id`. Returning at accept is available behind `MEMWAL_MCP_REMEMBER_WAIT_MS=0` for an operator who wants it, and remains its own product decision rather than a side effect of the latency work here.
diff --git a/packages/mcp/CHANGELOG.md b/packages/mcp/CHANGELOG.md
index b47ac5fc1..25acc3258 100644
--- a/packages/mcp/CHANGELOG.md
+++ b/packages/mcp/CHANGELOG.md
@@ -10,8 +10,6 @@
- `memwal_remember` sends a content-derived idempotency key, so the retry its own timeout message invites really does attach to the job already in flight instead of storing a second paid copy. The key is computed by the tool rather than relied on from the SDK, whose published build mints a random UUID per client instance.
- `memwal_remember_status` accepts `job_ids` to settle a whole batch in one call, and reports a mixed batch honestly — a still-uploading row no longer renders the poll timeout as `error=`, which read as a failed write. Settling in one request also matters against the rate limit: 20 ids cost one request, not twenty.
- The bridge's cold-start tool list no longer disagrees with the sidecar's. `memwal_remember_status` advertised only `job_id`, required, under `additionalProperties: false`, so the batch call the tools themselves instruct was rejected until `tools/list_changed` arrived; the `waitMs` ceiling advertised 60000 after the sidecar lowered it to 45000, which came back as an MCP validation error; and `memwal_remember_bulk` still carried its pre-queue description. Tests now pin the parts an agent acts on.
-- The accepted-then-failed report attached to `memwal_recall` no longer tells the agent to re-send text it does not have. The relayer stores only the SEAL ciphertext, so a failed write's wording cannot be recovered — the report now says so and asks the user to restate the fact rather than inviting the agent to guess.
-
### Changed
- `memwal_remember` keeps blocking until the write reaches `done`, so a result carries a real `blob_id`. Returning at accept is available behind `MEMWAL_MCP_REMEMBER_WAIT_MS=0` for an operator who wants it, and remains its own product decision rather than a side effect of the latency work here.
diff --git a/services/server/migrations/021_failed_write_report_ack.sql b/services/server/migrations/021_failed_write_report_ack.sql
deleted file mode 100644
index 2f91ef1c4..000000000
--- a/services/server/migrations/021_failed_write_report_ack.sql
+++ /dev/null
@@ -1,21 +0,0 @@
--- Acknowledge an accepted-then-failed write once it has been reported.
---
--- `POST /api/recall` attaches recently-failed writes so a user learns a fact
--- was silently lost. Without a marker the same rows ride along on EVERY recall
--- for the whole 24h window, and the text tells the agent to send them again —
--- so an agent that complies re-sends, the original row is still `failed` and
--- still in-window, and the next recall asks for the same thing. Each pass is a
--- fresh paid Walrus write.
---
--- Stamped when a report goes out, then filtered on, so each failure is
--- surfaced once. NULL means "not yet reported", which is what every existing
--- row should be.
-ALTER TABLE remember_jobs
- ADD COLUMN IF NOT EXISTS failure_reported_at TIMESTAMPTZ;
-
--- The report query is on the hot recall path: owner + status + window, newest
--- first, unreported only. Partial so the index stays small — it only ever
--- serves rows that are failed and still unreported.
-CREATE INDEX IF NOT EXISTS remember_jobs_unreported_failures_idx
- ON remember_jobs (owner, updated_at DESC)
- WHERE status = 'failed' AND failure_reported_at IS NULL;
diff --git a/services/server/scripts/mcp/__tests__/recall-failed-writes.test.ts b/services/server/scripts/mcp/__tests__/recall-failed-writes.test.ts
deleted file mode 100644
index 361755e96..000000000
--- a/services/server/scripts/mcp/__tests__/recall-failed-writes.test.ts
+++ /dev/null
@@ -1,88 +0,0 @@
-/**
- * `memwal_recall` reports writes that were accepted and then failed.
- *
- * `memwal_remember` returns as soon as the relayer accepts the job, which
- * leaves a window where the write still dies — a SEAL outage, an exhausted
- * upload budget — with nobody listening. `memwal_remember_status` can settle
- * one job, but nothing obliges an agent to call it, and storing a memory is
- * typically the last thing it does in a turn. An unasked question is the same
- * as a silent loss, and silent loss is the worst outcome for a product whose
- * whole promise is durable memory.
- *
- * Recall is the call an agent always makes, so the report rides along there.
- * These tests pin the rendering, and in particular that it never claims
- * anything when the relayer said nothing.
- */
-import test from "node:test";
-import assert from "node:assert/strict";
-
-import { formatFailedWrites } from "../tools/recall.js";
-
-test("says nothing when the relayer reported no failures", () => {
- assert.equal(formatFailedWrites({ results: [], total: 0 }), "");
- assert.equal(formatFailedWrites({ failed_writes: [] }), "");
-});
-
-test("says nothing when the relayer predates the field", () => {
- // An older deployment omits `failed_writes` entirely. That must read as
- // "nothing to report", never as an error or an empty warning banner.
- assert.equal(formatFailedWrites({ results: [] }), "");
- assert.equal(formatFailedWrites(null), "");
- assert.equal(formatFailedWrites(undefined), "");
- assert.equal(formatFailedWrites({ failed_writes: "not-an-array" }), "");
-});
-
-test("names the job and states plainly that the fact is not stored", () => {
- const text = formatFailedWrites({
- failed_writes: [
- {
- job_id: "9d304948-c693-408c-aac7-d7ef9ab98f5d",
- namespace: "default",
- error: "Memory encryption backend is unavailable",
- failed_at: "2026-09-15T15:10:11Z",
- },
- ],
- });
-
- assert.match(text, /NOT stored/);
- assert.match(text, /9d304948-c693-408c-aac7-d7ef9ab98f5d/);
- assert.match(text, /ns=default/);
- // The relayer's own message is passed through: the difference between a
- // SEAL outage and a spent retry budget is what tells the caller whether
- // re-sending is likely to work.
- assert.match(text, /Memory encryption backend is unavailable/);
- assert.match(text, /memwal_remember/);
-});
-
-test("reads as singular for one failure and plural for several", () => {
- const one = formatFailedWrites({
- failed_writes: [{ job_id: "a", namespace: "default" }],
- });
- assert.match(one, /1 earlier write was accepted/);
- assert.match(one, /that fact is\b/);
-
- const many = formatFailedWrites({
- failed_writes: [
- { job_id: "a", namespace: "default" },
- { job_id: "b", namespace: "default" },
- ],
- });
- assert.match(many, /2 earlier writes were accepted/);
- assert.match(many, /those facts are\b/);
-});
-
-test("survives a malformed entry rather than dropping the whole report", () => {
- // The warning matters more than its formatting: a row missing fields must
- // still be surfaced, because the alternative is the silent loss this
- // exists to prevent.
- const text = formatFailedWrites({
- failed_writes: [{}, { job_id: "b6da0c96", error: " " }],
- });
-
- assert.match(text, /2 earlier writes were accepted/);
- assert.match(text, /\(unknown job\)/);
- assert.match(text, /b6da0c96/);
- // A blank error string adds no information, so it is left off entirely
- // rather than rendered as a dangling dash.
- assert.doesNotMatch(text, /b6da0c96.*—/);
-});
diff --git a/services/server/scripts/mcp/tools/recall.ts b/services/server/scripts/mcp/tools/recall.ts
index 47a91a1f9..c48ff4ab5 100644
--- a/services/server/scripts/mcp/tools/recall.ts
+++ b/services/server/scripts/mcp/tools/recall.ts
@@ -55,54 +55,6 @@ function dedupeKey(text: string): string {
* budget on one fact and crowds out everything else it asked for, which is a
* read-side problem worth fixing on the read side.
*/
-/** One accepted-then-failed write, as the relayer reports it on recall. */
-interface FailedWrite {
- job_id?: unknown;
- namespace?: unknown;
- error?: unknown;
- failed_at?: unknown;
-}
-
-/**
- * Render the relayer's report of writes that were accepted and then failed.
- *
- * `memwal_remember` returns as soon as the write is accepted, so a job that
- * dies after that has nobody listening. `memwal_remember_status` can answer
- * for one, but nothing obliges an agent to ask, and storing a memory is
- * usually the last thing it does in a turn — so an unasked question is the
- * same as a silent loss. Recall is the call an agent always makes, which is
- * why the report rides along here.
- *
- * Read defensively off the response rather than through the SDK's typed
- * result: the field is newer than the pinned SDK's types, and a relayer that
- * does not send it at all (older deployment) must read as "nothing to
- * report", not as an error.
- */
-export function formatFailedWrites(result: unknown): string {
- const raw = (result as { failed_writes?: unknown } | null)?.failed_writes;
- if (!Array.isArray(raw) || raw.length === 0) return "";
-
- const lines = raw.map((entry: FailedWrite) => {
- const jobId = typeof entry?.job_id === "string" ? entry.job_id : "(unknown job)";
- const ns = typeof entry?.namespace === "string" ? ` ns=${entry.namespace}` : "";
- const why = typeof entry?.error === "string" && entry.error.trim() !== ""
- ? ` — ${entry.error}`
- : "";
- return ` job_id=${jobId}${ns}${why}`;
- });
-
- const n = raw.length;
- return (
- `\n\n⚠ ${n} earlier ${n === 1 ? "write was" : "writes were"} accepted but then FAILED, ` +
- `so ${n === 1 ? "that fact is" : "those facts are"} NOT stored:\n` +
- lines.join("\n") +
- `\nThe text is not recoverable — the relayer stores only the SEAL ciphertext, and these `
- + `writes failed before it was readable. Do not guess at what ${n === 1 ? "it" : "they"} said. `
- + `If the fact still matters, ask the user to state it again, then save it with `
- + `memwal_remember.`
- );
-}
-
export function collapseDuplicates(
results: T[],
): { unique: T[]; collapsed: number } {
@@ -219,14 +171,13 @@ export function registerRecallTool(
const result = await session.memwal.recall(query, limit, namespace);
const droppedRaw = (result as { dropped_count?: unknown }).dropped_count;
const dropped = typeof droppedRaw === "number" ? droppedRaw : 0;
- const failureReport = formatFailedWrites(result);
const filtered = filterByMaxDistance(result.results, maxDistance);
if (filtered.length === 0) {
return {
content: [
{
type: "text",
- text: emptyRecallText(result.results.length, dropped) + failureReport,
+ text: emptyRecallText(result.results.length, dropped),
},
],
};
@@ -250,7 +201,7 @@ export function registerRecallTool(
content: [
{
type: "text",
- text: lines.join("\n") + failureReport,
+ text: lines.join("\n"),
},
],
};
diff --git a/services/server/src/routes/recall.rs b/services/server/src/routes/recall.rs
index 3a0c3d5b0..1e2437043 100644
--- a/services/server/src/routes/recall.rs
+++ b/services/server/src/routes/recall.rs
@@ -15,96 +15,6 @@ use std::sync::Arc;
use crate::types::*;
-/// How far back the failure report on a recall response looks.
-///
-/// A day covers the gap between one working session and the next, which is
-/// when an agent would otherwise never learn that yesterday's last write died
-/// after it was accepted.
-const FAILED_WRITE_REPORT_WINDOW: std::time::Duration =
- std::time::Duration::from_secs(24 * 60 * 60);
-
-/// Most failures reported on one recall. A caller acting on this re-sends the
-/// facts; a wall of them would crowd out the memories it actually asked for.
-const FAILED_WRITE_REPORT_LIMIT: i64 = 5;
-
-/// Recent writes that were accepted and then failed, for `owner`.
-///
-/// Never fails the recall it is attached to. This is a courtesy report on a
-/// read path — losing it costs the caller a warning, while propagating the
-/// error would cost them the memories they actually asked for, so a failed
-/// lookup degrades to "nothing to report" and says so in the log.
-async fn failed_writes_for(state: &AppState, owner: &str) -> Vec {
- match state
- .db
- .recent_failed_remember_jobs(owner, FAILED_WRITE_REPORT_WINDOW, FAILED_WRITE_REPORT_LIMIT)
- .await
- {
- Ok(failed) => failed
- .into_iter()
- .map(|mut w| {
- // Every other client-facing view of `remember_jobs.error_msg`
- // runs it through this first — `GET /api/remember/:job_id` and
- // `POST /api/remember/bulk/status` both do. Reading the column
- // straight into a recall response skipped both of the
- // sanitizer's jobs: swapping an infrastructure-funding failure
- // for INFRA_JOB_ERROR_MESSAGE (whose text exists to stop a user
- // reading "Insufficient balance ... for owner 0x…" as an
- // instruction to top that address up), and redacting long hex
- // runs so the relayer's own wallet never reaches a tenant.
- //
- // These rows are `status = 'failed'` by construction — the
- // query selects on it — so the status argument is fixed.
- w.error = super::remember::sanitize_job_error_for_client("failed", w.error);
- w
- })
- .collect(),
- Err(e) => {
- tracing::warn!(
- "recall: failed-write report unavailable for owner={}: {}",
- owner,
- e
- );
- Vec::new()
- }
- }
-}
-
-/// Claim the prefetched report at the point of delivery, and return only the
-/// rows this recall actually won.
-///
-/// `failed_writes_for` runs concurrently with the embed, search and decrypt,
-/// any of which can fail with `?` or be abandoned when the SDK hits its hard
-/// 15s abort. A `tokio::spawn`ed task is not cancelled when its handle drops,
-/// so stamping inside that task committed the acknowledgement for recalls that
-/// never returned a report — and the failure, whose whole point is that the
-/// user does not otherwise know about it, was then never surfaced again.
-///
-/// A claim error reports nothing rather than reporting unstamped rows: an
-/// unstamped report repeats on every recall for the full window, and its text
-/// asks the agent to re-send the fact, so each repeat is another paid write.
-async fn claim_for_delivery(state: &AppState, found: Vec) -> Vec {
- if found.is_empty() {
- return Vec::new();
- }
- let ids: Vec = found.iter().map(|w| w.job_id.clone()).collect();
- match state.db.claim_failed_write_reports(&ids).await {
- Ok(claimed) => {
- let claimed: std::collections::HashSet = claimed.into_iter().collect();
- found
- .into_iter()
- .filter(|w| claimed.contains(&w.job_id))
- .collect()
- }
- Err(e) => {
- tracing::warn!(
- "recall: failed-write report could not be claimed, staying quiet: {}",
- e
- );
- Vec::new()
- }
- }
-}
-
// ============================================================
// Recall query-embedding cache (Redis) — wraps the Embedder service
// ============================================================
@@ -260,17 +170,6 @@ pub async fn recall(
let owner = &auth.owner;
let namespace = &body.namespace;
- // Started here rather than awaited at the end, so it overlaps the embed,
- // search, Walrus download and SEAL decrypt that follow instead of adding
- // to them. The published SDK aborts a recall after a hard 15s that no
- // caller can raise, and recall has been measured landing on exactly that
- // — so this report has to cost the critical path nothing.
- let failed_writes = {
- let state = state.clone();
- let owner = owner.clone();
- tokio::spawn(async move { failed_writes_for(&state, &owner).await })
- };
-
tracing::info!(
query_len = body.query.len(),
owner = %owner,
@@ -319,10 +218,6 @@ pub async fn recall(
results: vec![],
total: 0,
dropped_count: 0,
- // Reported even with no hits: an empty recall is exactly when a
- // caller is most likely to be looking for the fact that failed.
- failed_writes: claim_for_delivery(&state, failed_writes.await.unwrap_or_default())
- .await,
}));
}
@@ -409,9 +304,6 @@ pub async fn recall(
results,
total,
dropped_count,
- // A panic in the report task must not take the recall with it; the
- // caller loses a warning, not their memories.
- failed_writes: claim_for_delivery(&state, failed_writes.await.unwrap_or_default()).await,
}))
}
@@ -552,7 +444,6 @@ Available: 10708877";
results: vec![],
total: 0,
dropped_count: 3,
- failed_writes: vec![],
};
let json = serde_json::to_value(&resp).unwrap();
assert_eq!(json["dropped_count"], 3);
@@ -564,7 +455,6 @@ Available: 10708877";
results: vec![],
total: 0,
dropped_count: 0,
- failed_writes: vec![],
};
let json = serde_json::to_value(&resp).unwrap();
// skip_serializing_if = "is_zero_usize" → field absent
diff --git a/services/server/src/routes/remember.rs b/services/server/src/routes/remember.rs
index a27bfa6b1..4711572c0 100644
--- a/services/server/src/routes/remember.rs
+++ b/services/server/src/routes/remember.rs
@@ -1067,17 +1067,13 @@ fn should_spawn_after_reset(rows_affected: u64) -> bool {
/// so it matches zero rows and returns before `enqueue_wallet_job`.
const PREPARE_CLAIM_TTL_SECS: i64 = 60;
-/// Re-claiming clears `failure_reported_at` along with `error_msg`: the row is
-/// being reused for a fresh attempt, and the recall report only surfaces a
-/// failure once per row. Left set, a retry that failed AGAIN would never be
-/// reported — silent loss, which is the exact thing that report exists to stop.
async fn claim_remember_preparation(
pool: &sqlx::PgPool,
job_id: &str,
) -> Result, AppError> {
let token = uuid::Uuid::new_v4().to_string();
let claimed: Option = sqlx::query_scalar(
- "UPDATE remember_jobs SET prepare_claimed_at = NOW(), prepare_claim_token = $3, status = CASE WHEN status = 'failed' AND blob_id IS NULL THEN 'pending' ELSE status END, error_msg = CASE WHEN blob_id IS NULL THEN NULL ELSE error_msg END, failure_reported_at = CASE WHEN blob_id IS NULL THEN NULL ELSE failure_reported_at END, updated_at = NOW() WHERE id = $1 AND blob_id IS NULL AND status IN ('pending', 'failed') AND (prepare_claimed_at IS NULL OR prepare_claimed_at < NOW() - make_interval(secs => $2) OR status = 'failed') RETURNING prepare_claim_token",
+ "UPDATE remember_jobs SET prepare_claimed_at = NOW(), prepare_claim_token = $3, status = CASE WHEN status = 'failed' AND blob_id IS NULL THEN 'pending' ELSE status END, error_msg = CASE WHEN blob_id IS NULL THEN NULL ELSE error_msg END, updated_at = NOW() WHERE id = $1 AND blob_id IS NULL AND status IN ('pending', 'failed') AND (prepare_claimed_at IS NULL OR prepare_claimed_at < NOW() - make_interval(secs => $2) OR status = 'failed') RETURNING prepare_claim_token",
)
.bind(job_id)
.bind(PREPARE_CLAIM_TTL_SECS)
@@ -1630,47 +1626,6 @@ mod tests {
assert_eq!(status, "done");
}
- /// `memwal_remember` tells an agent to send a failed fact again. The
- /// derived idempotency key collapses that retry onto the failed row, and
- /// the claim used to be refused for 60s — while the route answered 202
- /// ACCEPTED regardless, so the caller was told a write was queued when
- /// nothing was running. A job that has finished failing has no live
- /// preparation to fence.
- /// The recall failure report fires once per ROW, and a re-claim reuses the
- /// row for a fresh attempt. If the flag survived that reset, a retry that
- /// failed again would never be reported — silent loss, which is precisely
- /// what the report exists to prevent. Two fixes that are each correct
- /// alone and wrong together.
- #[tokio::test]
- async fn reclaiming_a_reported_failure_lets_it_be_reported_again() {
- let pool = idem_test_pool().await;
- let job_id = uuid::Uuid::new_v4().to_string();
- sqlx::query(
- "INSERT INTO remember_jobs (id, owner, namespace, status, error_msg, failure_reported_at)
- VALUES ($1, '0xowner', 'ns', 'failed', 'first failure', NOW())",
- )
- .bind(&job_id)
- .execute(&pool)
- .await
- .unwrap();
-
- claim_remember_preparation(&pool, &job_id)
- .await
- .unwrap()
- .expect("a failed job is re-claimable");
-
- let reported: Option> =
- sqlx::query_scalar("SELECT failure_reported_at FROM remember_jobs WHERE id = $1")
- .bind(&job_id)
- .fetch_one(&pool)
- .await
- .unwrap();
- assert!(
- reported.is_none(),
- "a re-claimed row must be reportable again if it fails a second time",
- );
- }
-
#[tokio::test]
async fn a_failed_job_is_reclaimable_immediately() {
let pool = idem_test_pool().await;
@@ -1937,17 +1892,6 @@ mod tests {
.execute(&pool)
.await
.unwrap();
- // 021 adds `failure_reported_at`, which `claim_remember_preparation`
- // clears on re-claim. This helper builds its own minimal schema rather
- // than going through `VectorDb::new()`, so a migration wired into that
- // chain does not reach it — every column a test in this module touches
- // has to be listed here explicitly.
- sqlx::raw_sql(include_str!(
- "../../migrations/021_failed_write_report_ack.sql"
- ))
- .execute(&pool)
- .await
- .unwrap();
pool
}
diff --git a/services/server/src/storage/db.rs b/services/server/src/storage/db.rs
index 4b1364215..0c62c1840 100644
--- a/services/server/src/storage/db.rs
+++ b/services/server/src/storage/db.rs
@@ -105,12 +105,6 @@ const MIGRATIONS_AFTER_INDEX_RECOVERY: &[Migration] = &[
// constraint 015 set up — see 019's header.
migration!("019_memory_read_api_updated_at_set_not_null.sql"),
migration!("020_read_api_followups.sql"),
- // 021 adds `remember_jobs.failure_reported_at`, which
- // `recent_failed_remember_jobs` writes on every recall and
- // `claim_remember_preparation` clears on re-claim. Both are plain
- // queries against a column that only exists if this runs, so a
- // deploy that skips it fails those statements with 42703.
- migration!("021_failed_write_report_ack.sql"),
];
/// Every migration the pipeline applies, in the order it applies them.
@@ -204,13 +198,6 @@ mod tests {
/// Every `.sql` file in `services/server/migrations` must be wired
/// into the pipeline.
- ///
- /// Regression test for migration 021: the file was added and merged
- /// to dev, but never listed in `VectorDb::new`, so the column it
- /// creates never existed. Four `routes::remember` tests failed with
- /// `column "failure_reported_at" does not exist` (42703), and any
- /// deploy of that build would have failed the same statements in
- /// production. Needs no database.
#[test]
fn every_migration_file_is_wired_into_the_pipeline() {
use std::collections::BTreeSet;
@@ -1860,36 +1847,6 @@ impl VectorDb {
Ok(row)
}
- /// Writes this owner started that ended in `failed` within `window`.
- ///
- /// Serves the failure report attached to recall responses. Owner-scoped
- /// from `AuthInfo`, never from request input, so one account cannot read
- /// another's failures.
- ///
- /// Bounded by both a time window and `limit` because this runs on the
- /// read path: recall is the hottest authed route, and an account with a
- /// long tail of old failures must not turn every recall into a large
- /// scan. `remember_jobs (owner, status, updated_at DESC)` (migration 006)
- /// covers the predicate and the ordering, so this is an index range scan
- /// of at most `limit` rows.
- ///
- /// Failures repeat across calls until they age out of the window. That is
- /// deliberate — suppressing a report after one sighting would put the
- /// notice back on the caller remembering to act on it, which is the
- /// failure mode this whole path exists to remove.
- /// Accepted-then-failed writes this owner has not been told about yet.
- ///
- /// Read-only. Reporting is still one-shot, but the stamp is taken by
- /// `claim_failed_write_reports` at the moment the response is built, not
- /// here — this runs concurrently with the embed/search/decrypt that
- /// follow, and any of those can fail or be abandoned. Stamping here meant
- /// a recall that 500d on the embedding provider, or that the SDK aborted
- /// at its hard 15s, still marked the rows reported: the user was never
- /// told, on that recall or any later one, that their write had failed.
- ///
- /// One-shot is preserved because the claim is a single conditional UPDATE
- /// over these ids — a concurrent recall that got there first claims them
- /// and this one is handed back nothing to report.
/// How durable writes that finished inside `window` turned out,
/// across every owner: `(failed, succeeded)`.
///
@@ -1943,121 +1900,6 @@ impl VectorDb {
}
}
- pub async fn recent_failed_remember_jobs(
- &self,
- owner: &str,
- window: std::time::Duration,
- limit: i64,
- ) -> Result, AppError> {
- let started = std::time::Instant::now();
- let rows = sqlx::query_as::<
- _,
- (
- String,
- String,
- Option,
- chrono::DateTime,
- ),
- >(
- "SELECT id, namespace, error_msg, updated_at FROM remember_jobs
- WHERE owner = $1
- AND status = 'failed'
- AND failure_reported_at IS NULL
- AND updated_at >= $2
- ORDER BY updated_at DESC
- LIMIT $3",
- )
- .bind(owner)
- .bind(chrono::Utc::now() - chrono::Duration::from_std(window).unwrap_or_default())
- .bind(limit)
- .fetch_all(&self.pool)
- .await;
-
- let rows = match rows {
- Ok(rows) => rows,
- Err(e) => {
- crate::observability::observe_db(
- "remember_jobs.recent_failed",
- "error",
- started.elapsed(),
- );
- return Err(AppError::Internal(format!(
- "Failed to list failed remember jobs: {}",
- e
- )));
- }
- };
- crate::observability::observe_db("remember_jobs.recent_failed", "ok", started.elapsed());
-
- Ok(rows
- .into_iter()
- .map(
- |(job_id, namespace, error, failed_at)| crate::types::FailedWrite {
- job_id,
- namespace,
- // Raw here on purpose: `routes` is not reachable from the
- // lib crate, and this is the storage layer. The caller
- // (`routes::recall::failed_writes_for`) sanitizes before
- // any of it reaches a client.
- error,
- failed_at: failed_at.to_rfc3339(),
- },
- )
- .collect())
- }
-
- /// Stamp `failure_reported_at` on the subset of `ids` not already
- /// reported, and return the ids actually claimed.
- ///
- /// The conditional `failure_reported_at IS NULL` is what keeps the report
- /// one-shot: two recalls that both read the same unreported row race here
- /// and exactly one UPDATE matches, so only that one reports it. The other
- /// gets an empty set and stays quiet — which matters because the report
- /// text asks the agent to send the fact again, and a double report is a
- /// duplicate paid Walrus write.
- ///
- /// Called at response construction, so a recall that never reaches the
- /// client does not consume the report.
- pub async fn claim_failed_write_reports(
- &self,
- ids: &[String],
- ) -> Result, AppError> {
- if ids.is_empty() {
- return Ok(Vec::new());
- }
- let started = std::time::Instant::now();
- let rows = sqlx::query_scalar::<_, String>(
- "UPDATE remember_jobs SET failure_reported_at = NOW()
- WHERE id = ANY($1) AND failure_reported_at IS NULL
- RETURNING id",
- )
- .bind(ids)
- .fetch_all(&self.pool)
- .await;
-
- match rows {
- Ok(rows) => {
- crate::observability::observe_db(
- "remember_jobs.claim_failure_report",
- "ok",
- started.elapsed(),
- );
- Ok(rows)
- }
- Err(e) => {
- crate::observability::observe_db(
- "remember_jobs.claim_failure_report",
- "error",
- started.elapsed(),
- );
- Err(AppError::Internal(format!(
- "Failed to claim failed-write reports: {}",
- e
- )))
- }
- }
- }
-
/// Hard-delete all vector index rows for a given owner + namespace.
/// (Walrus blobs themselves persist — Walrus has no delete; this only
/// removes the local `vector_entries` rows, so the memories stop being
@@ -3503,81 +3345,6 @@ mod quota_admission_tests {
.expect("test database must be reachable with pgvector installed")
}
- /// The failed-write report must survive a recall that never lands, and
- /// must still go out only once.
- ///
- /// Reading and claiming used to be one statement, stamped inside a task
- /// spawned before the embed/search/decrypt that can each fail with `?` or
- /// be abandoned when the SDK hits its hard abort. That burned the
- /// acknowledgement for recalls that returned no report at all, and the
- /// failure — which the user has no other way to learn about — was never
- /// surfaced again. Splitting them is only safe if the read is genuinely
- /// non-consuming and the claim is genuinely exclusive; both are asserted
- /// here because nothing else covers either function.
- #[tokio::test]
- async fn failed_write_report_reads_freely_but_claims_once() {
- let db = test_db().await;
- let owner = unique_owner("failed-write-report");
- let job_id = uuid::Uuid::new_v4().to_string();
- sqlx::query(
- "INSERT INTO remember_jobs (id, owner, namespace, status, error_msg)
- VALUES ($1, $2, 'ns', 'failed', 'walrus upload rejected')",
- )
- .bind(&job_id)
- .bind(&owner)
- .execute(db.pool())
- .await
- .unwrap();
-
- let window = std::time::Duration::from_secs(24 * 60 * 60);
-
- // Read twice. A recall that dies after this point must leave the row
- // reportable, so neither read may consume it.
- for attempt in 0..2 {
- let found = db
- .recent_failed_remember_jobs(&owner, window, 5)
- .await
- .unwrap();
- assert_eq!(
- found.len(),
- 1,
- "read {} consumed the report; an abandoned recall would lose it",
- attempt,
- );
- assert_eq!(found[0].job_id, job_id);
- }
-
- // Claiming is what acknowledges it, and only the first claim wins —
- // the report text asks the agent to re-send the fact, so a second
- // report is a duplicate paid Walrus write.
- let first = db
- .claim_failed_write_reports(&[job_id.clone()])
- .await
- .unwrap();
- assert_eq!(first, vec![job_id.clone()], "the first claim must win");
-
- let second = db
- .claim_failed_write_reports(&[job_id.clone()])
- .await
- .unwrap();
- assert!(
- second.is_empty(),
- "a second claim must report nothing, got {:?}",
- second,
- );
-
- // And the row is now invisible to the read, so later recalls stay quiet.
- let after = db
- .recent_failed_remember_jobs(&owner, window, 5)
- .await
- .unwrap();
- assert!(
- after.is_empty(),
- "a claimed failure must not be read again, got {:?}",
- after,
- );
- }
-
/// Unique per test so concurrent runs cannot see each other's rows, and so
/// a failed run leaves no state that poisons the next one.
fn unique_owner(tag: &str) -> String {
diff --git a/services/server/src/types.rs b/services/server/src/types.rs
index eba0ec0d7..f4f60f138 100644
--- a/services/server/src/types.rs
+++ b/services/server/src/types.rs
@@ -1578,45 +1578,6 @@ pub struct RecallResponse {
/// failed and were silently omitted from `results`. Zero on the happy path.
#[serde(default, skip_serializing_if = "is_zero_usize")]
pub dropped_count: usize,
- /// Writes this owner started recently that ended in `failed` — facts the
- /// caller was told were accepted but that were never stored.
- ///
- /// Carried on the *read* path on purpose. A write now returns as soon as
- /// the relayer accepts the job, so a failure after that point has no
- /// caller left listening: `memwal_remember_status` answers it, but nothing
- /// obliges an agent to ask, and saving a memory is typically the last
- /// thing it does in a turn. Recall is the call an agent always makes, so
- /// attaching the bad news here is what turns a silent loss into a visible
- /// one.
- ///
- /// Empty on the happy path and omitted from the wire, so an older client
- /// that ignores the field sees exactly today's response.
- #[serde(default, skip_serializing_if = "Vec::is_empty")]
- pub failed_writes: Vec,
-}
-
-/// One write that was accepted and then failed, as reported back on recall.
-#[derive(Debug, Serialize)]
-pub struct FailedWrite {
- /// The `job_id` the write returned when it was accepted, so a caller can
- /// match this against what it was told at the time.
- pub job_id: String,
- pub namespace: String,
- /// The failure message, after `sanitize_job_error_for_client` — the same
- /// treatment `GET /api/remember/:job_id` and the bulk status endpoint give
- /// it, and for the same two reasons. An infrastructure-funding failure is
- /// replaced wholesale (its raw text names the relayer's own wallet and its
- /// balance shortfall, which is neither the tenant's business nor safe to
- /// show them: it reads as "top this address up"). Everything else keeps its
- /// wording with long hex runs redacted.
- ///
- /// What survives is the part a caller can act on: whether this looks
- /// transient or permanent, and so whether re-sending the fact is likely to
- /// work.
- #[serde(skip_serializing_if = "Option::is_none")]
- pub error: Option,
- /// When the job reached `failed`, RFC 3339.
- pub failed_at: String,
}
fn is_zero_usize(n: &usize) -> bool {
From 2adfb2611c1ccef9c4970f91d2a78c5f129c23d8 Mon Sep 17 00:00:00 2001
From: Harry Phan
Date: Thu, 17 Sep 2026 23:16:37 +0700
Subject: [PATCH 071/132] docs: document writes=degraded and the 0.0.14-dev.0
pin
GET /health now emits writes=degraded when recent durable writes fail
and none land; the API reference still listed only ok/paused.
Align docs/mcp/changelog.mdx 0.0.14 with the package changelog
(cold-start BASELINE_RELAYER_TOOLS) and state that launchers pin
@0.0.14-dev.0 until 0.0.14 exists on npm.
---
docs/mcp/changelog.mdx | 6 +++++-
docs/relayer/api-reference.md | 8 ++++++--
2 files changed, 11 insertions(+), 3 deletions(-)
diff --git a/docs/mcp/changelog.mdx b/docs/mcp/changelog.mdx
index 7d67d4bdc..56ec2ba5f 100644
--- a/docs/mcp/changelog.mdx
+++ b/docs/mcp/changelog.mdx
@@ -28,18 +28,22 @@ questions:
- What changed in the MemWal MCP changelog?
- When was the automatic memory plugin added to MemWal MCP?
answer: >-
- The latest MCP package release is 0.0.13. A `memwal_remember`, `memwal_remember_bulk` or `memwal_analyze` whose reply never arrives is no longer reported as safe to retry: the relayer accepts those with HTTP 202 and finishes them in a durable queue, so the write may already have landed, and `/api/remember/bulk` has no idempotency key — repeating it stores a second paid copy. The tool now says the write may have completed and points at `memwal_recall` to check before re-saving, while a lost read still says plainly that retrying is safe. When the relayer rejects the saved delegate key, tool calls now get an auth error pointing at `memwal_login` instead of waiting minutes for a retry hint that cannot work. It saves the delegate keypair before the sign-in URL reaches the browser and reclaims it on the next start, so a login interrupted after the onchain registration no longer loses a key the user already paid for. It writes the credentials file by creating a new 0600 file and renaming it into place, so a sign-in never puts the delegate private key into a credentials.json that a manual chmod or a restored backup left world-readable. A completed sign-in is now confirmed with a notification and a one-shot banner naming the account and the resolved credentials path, and the bridge keeps reading stdin after an in-session login instead of going deaf. Unrecognised command-line options now warn instead of being silently ignored, `--help` lists the network presets and the URLs each resolves to, `memwal_health` names the relayer the client dialled, and `memwal_restore` reports `failed` and retries the same page when truncation is a download or embed blip, instead of always telling the agent to raise `limit`. Plugin launch configs (`.mcp.json`, Cursor/Codex copies, and the Codex fallback installer) now pin `@mysten-incubation/memwal-mcp@0.0.13` so npx cannot keep a cached 0.0.5. Version 0.0.12 forwards the MCP client's initialize.clientInfo to the relayer so sidecar logs can name the coding agent (Claude Code, Codex, Cursor, and others) on each session and tool call, and it resolves the credential directory on every access so MEMWAL_CREDS_DIR can override it.
+ The latest published MCP package on npm `latest` is 0.0.13. This branch's in-tree plugin launchers pin `@mysten-incubation/memwal-mcp@0.0.14-dev.0` because npm has no `0.0.14` yet — pinning the unpublished version would fail to resolve. 0.0.14 (unreleased) stops cold start from advertising tools an older relayer does not serve, bounds stalled relayer calls, honours `retry_after` on 429, and keeps `memwal_remember` waiting for `done` by default (`MEMWAL_MCP_REMEMBER_WAIT_MS=0` returns at accept). 0.0.13 stopped reporting a timed-out write as safe to retry, answers a rejected delegate key with `memwal_login` instead of a dropped-connection story, backs off on handshake 429, and pinned plugin launchers to `@0.0.13` so npx cannot keep a cached 0.0.5.
---
## 0.0.14
+Unreleased package version. Plugin launchers in this tree pin `@mysten-incubation/memwal-mcp@0.0.14-dev.0` (the published `dev` dist-tag) until `0.0.14` exists on npm.
+
### Fixed
+- The cold-start tool list no longer advertises a tool the relayer may not serve. The bridge ships on npm and updates itself while a relayer ships per environment, so 0.0.14-dev.0 dialled prod and staging still on 0.0.13: cold start named `memwal_remember_status`, which neither registers, and the pending-write wording sent the agent to go call it — one live run spent 90.67s there before erroring. Cold start is now a floor rather than a forecast (`BASELINE_RELAYER_TOOLS`): it carries only what the oldest supported relayer serves and its descriptions name nothing outside it, while newer tools still reach the client a beat later on the relayer's own `tools/list`. And a call for a tool the connected relayer does not serve is now answered locally and at once — naming the tools that do exist and saying plainly that nothing ran — instead of being forwarded into a wait that only ends at the orphan deadline. (#928)
- Bound every relayer call the tools make. The pinned SDK aborts a request only when the caller passes a signal, which none of the memory methods do, so a stalled socket kept a tool running with no ceiling — `memwal_remember` was observed still going past 120s against a 90s budget. Accepts are bounded at 15s (`MEMWAL_MCP_ACCEPT_DEADLINE_MS`), waits at their own budget plus grace. The request is not cancelled — the SDK exposes no way to pass a signal — but the agent is no longer held by it.
- Honour the relayer's `retry_after` instead of dropping the write. Once the per-delegate-key budget (60 weighted requests/minute) is spent the relayer answers 429 with a cooldown, and nothing backed off: the fact was never written and the agent saw only an opaque tool error. A short cooldown is now absorbed; a long one is reported with the wait named, stating plainly that the fact was NOT saved and pointing at the cheaper shape — one `memwal_remember_bulk` rather than N single calls, one `memwal_remember_status(job_ids)` rather than N status calls. Only rejections that provably never reached the handler retry, so `/api/remember/bulk`, which carries no idempotency key, cannot be duplicated by a retry.
- `memwal_remember` sends a content-derived idempotency key, so the retry its own timeout message invites really does attach to the job already in flight instead of storing a second paid copy. The key is computed by the tool rather than relied on from the SDK, whose published build mints a random UUID per client instance.
- `memwal_remember_status` accepts `job_ids` to settle a whole batch in one call, and reports a mixed batch honestly — a still-uploading row no longer renders the poll timeout as `error=`, which read as a failed write. Settling in one request also matters against the rate limit: 20 ids cost one request, not twenty.
- The bridge's cold-start tool list no longer disagrees with the sidecar's. `memwal_remember_status` advertised only `job_id`, required, under `additionalProperties: false`, so the batch call the tools themselves instruct was rejected until `tools/list_changed` arrived; the `waitMs` ceiling advertised 60000 after the sidecar lowered it to 45000, which came back as an MCP validation error; and `memwal_remember_bulk` still carried its pre-queue description. Tests now pin the parts an agent acts on.
+
### Changed
- `memwal_remember` keeps blocking until the write reaches `done`, so a result carries a real `blob_id`. Returning at accept is available behind `MEMWAL_MCP_REMEMBER_WAIT_MS=0` for an operator who wants it, and remains its own product decision rather than a side effect of the latency work here.
diff --git a/docs/relayer/api-reference.md b/docs/relayer/api-reference.md
index 0ffd9a42d..d0a64c847 100644
--- a/docs/relayer/api-reference.md
+++ b/docs/relayer/api-reference.md
@@ -82,9 +82,13 @@ These routes require no authentication.
Service liveness check. `status` is `"ok"` when the relayer process is up. HTTP 200 means the process is running, not that writes are accepted.
-`writes` is `"ok"` or `"paused"`. `"paused"` when `WRITES_PAUSED` is set (`1` / `true` / `yes`); empty or unset is `"ok"`. That flag is write-path admission, not a health-only signal: `POST /api/remember`, `/api/remember/manual`, `/api/remember/bulk`, and `/api/analyze` then return HTTP 503 with `{"error":"writes are paused"}`. `/health` itself stays HTTP 200 with `status: "ok"` and `writes: "paused"`, so clients can distinguish an intentional pause from an integrator bug. Reads (`recall`, `restore`, remember job status) stay available.
+`writes` is `"ok"`, `"degraded"`, or `"paused"`.
-`write_ready` is `true` when the encryption sidecar process answered its own `/health` **and** Postgres can accept writes (cached a few seconds). Postgres is considered not writable when Neon cluster size (`pg_cluster_size`, not `pg_database_size` of this database) is at or within 1MB of `neon.max_cluster_size`. If `pg_cluster_size()` is missing (no `neon` extension), the probe falls back to `sum(pg_database_size)` against that GUC. Self-hosted Postgres without the GUC keeps the sidecar-only check. Probe errors and timeouts fail open (`write_ready` stays true) so CI is not blocked; timeouts log at warn. A sidecar outage or a disk/project-size write outage can still return HTTP 200 with `write_ready: false`. Use `writes`, not `write_ready`, for the pause signal. `write_ready: true` is not a guarantee that remember or analyze succeed.
+- `"paused"` when `WRITES_PAUSED` is set (`1` / `true` / `yes`). That flag is write-path admission, not a health-only signal: `POST /api/remember`, `/api/remember/manual`, `/api/remember/bulk`, and `/api/analyze` then return HTTP 503 with `{"error":"writes are paused"}`. `/health` itself stays HTTP 200 with `status: "ok"` and `writes: "paused"`, so clients can distinguish an intentional pause from an integrator bug. Reads (`recall`, `restore`, remember job status) stay available.
+- `"degraded"` when recent durable writes (last 15 minutes) have **failed at least three times and none have landed**. The relayer still accepts and queues a write (`write_ready` stays `true`; HTTP 200), but Walrus is not storing it — expect the job to fail rather than queue more. A single success in that window keeps `"ok"` even if other writes failed: this is a total-outage detector, not a per-request verdict. `memwal_health` prints `writes=degraded` when this is set.
+- `"ok"` otherwise (including a quiet window with no finished writes).
+
+`write_ready` is `true` when the encryption sidecar process answered its own `/health` **and** Postgres can accept writes (cached a few seconds). Postgres is considered not writable when Neon cluster size (`pg_cluster_size`, not `pg_database_size` of this database) is at or within 1MB of `neon.max_cluster_size`. If `pg_cluster_size()` is missing (no `neon` extension), the probe falls back to `sum(pg_database_size)` against that GUC. Self-hosted Postgres without the GUC keeps the sidecar-only check. Probe errors and timeouts fail open (`write_ready` stays true) so CI is not blocked; timeouts log at warn. A sidecar outage or a disk/project-size write outage can still return HTTP 200 with `write_ready: false`. Use `writes`, not `write_ready`, for the pause or Walrus-outage signal. `write_ready: true` is not a guarantee that remember or analyze succeed.
**Response:**
From d03ac413e8e5845470e683fce1133ffeb64bd70b Mon Sep 17 00:00:00 2001
From: Harry Phan
Date: Thu, 17 Sep 2026 23:29:05 +0700
Subject: [PATCH 072/132] docs(mcp): default remember wait is 0 (accept), not
90s
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit
90000ms is 90 seconds — longer than the MCP SDK's 60s tools/call
timeout on hosts that do not raise it, and far from Teo's "a few
seconds". Code already defaults DEFAULT_REMEMBER_WAIT_MS=0. Env docs,
changelogs, and tool descriptions had claimed the opposite (block until
blob_id). Align them: return job_id at accept; 90000 is opt-in restore
of the old wait, not the default.
---
docs/mcp/changelog.mdx | 4 ++--
docs/reference/environment-variables.md | 2 +-
packages/mcp/CHANGELOG.md | 3 ++-
packages/mcp/src/auth-required.ts | 4 ++--
services/server/scripts/mcp/tools/remember-bulk.ts | 2 +-
services/server/scripts/mcp/tools/remember.ts | 2 +-
6 files changed, 9 insertions(+), 8 deletions(-)
diff --git a/docs/mcp/changelog.mdx b/docs/mcp/changelog.mdx
index 56ec2ba5f..8f5a52e31 100644
--- a/docs/mcp/changelog.mdx
+++ b/docs/mcp/changelog.mdx
@@ -28,7 +28,7 @@ questions:
- What changed in the MemWal MCP changelog?
- When was the automatic memory plugin added to MemWal MCP?
answer: >-
- The latest published MCP package on npm `latest` is 0.0.13. This branch's in-tree plugin launchers pin `@mysten-incubation/memwal-mcp@0.0.14-dev.0` because npm has no `0.0.14` yet — pinning the unpublished version would fail to resolve. 0.0.14 (unreleased) stops cold start from advertising tools an older relayer does not serve, bounds stalled relayer calls, honours `retry_after` on 429, and keeps `memwal_remember` waiting for `done` by default (`MEMWAL_MCP_REMEMBER_WAIT_MS=0` returns at accept). 0.0.13 stopped reporting a timed-out write as safe to retry, answers a rejected delegate key with `memwal_login` instead of a dropped-connection story, backs off on handshake 429, and pinned plugin launchers to `@0.0.13` so npx cannot keep a cached 0.0.5.
+ The latest published MCP package on npm `latest` is 0.0.13. This branch's in-tree plugin launchers pin `@mysten-incubation/memwal-mcp@0.0.14-dev.0` because npm has no `0.0.14` yet — pinning the unpublished version would fail to resolve. 0.0.14 (unreleased) stops cold start from advertising tools an older relayer does not serve, bounds stalled relayer calls, honours `retry_after` on 429, and returns `memwal_remember` at accept by default (`MEMWAL_MCP_REMEMBER_WAIT_MS=0`; set `90000` to wait for `blob_id`). 0.0.13 stopped reporting a timed-out write as safe to retry, answers a rejected delegate key with `memwal_login` instead of a dropped-connection story, backs off on handshake 429, and pinned plugin launchers to `@0.0.13` so npx cannot keep a cached 0.0.5.
---
## 0.0.14
@@ -46,7 +46,7 @@ Unreleased package version. Plugin launchers in this tree pin `@mysten-incubatio
### Changed
-- `memwal_remember` keeps blocking until the write reaches `done`, so a result carries a real `blob_id`. Returning at accept is available behind `MEMWAL_MCP_REMEMBER_WAIT_MS=0` for an operator who wants it, and remains its own product decision rather than a side effect of the latency work here.
+- `memwal_remember` / `memwal_remember_bulk` return at accept by default (`MEMWAL_MCP_REMEMBER_WAIT_MS=0`, ~1s, `job_id`). The Walrus write continues in the background; do not treat that reply as stored. Set `MEMWAL_MCP_REMEMBER_WAIT_MS=90000` to restore wait-for-`blob_id` (90s ceiling). Do not use a value between 0 and the real completion time — that pays the wait and still returns pending. The MCP TypeScript SDK's default tools/call timeout is 60s, so a 90s wait loses on hosts that do not raise it.
## 0.0.13
diff --git a/docs/reference/environment-variables.md b/docs/reference/environment-variables.md
index 16c9c09c0..887c03f8e 100644
--- a/docs/reference/environment-variables.md
+++ b/docs/reference/environment-variables.md
@@ -161,7 +161,7 @@ These are not all enforced at boot, but most real deployments need them.
| `MCP_MAX_TOTAL_SESSIONS` | `1000` | Maximum active MCP sessions across SSE and Streamable HTTP transports |
| `MCP_MAX_SESSIONS_PER_IP` | `16` | Maximum active MCP sessions from one source IP |
| `MCP_MAX_NEW_SESSIONS_PER_IP_PER_MIN` | `30` | Maximum new MCP sessions opened by one source IP per minute |
-| `MEMWAL_MCP_REMEMBER_WAIT_MS` | `90000` | How long `memwal_remember` / `memwal_remember_bulk` / `memwal_analyze` wait for a write to land before returning the job_id instead. The default is the full ceiling, so a successful call carries a real `blob_id` and the agent can say the fact is stored; a write slower than that returns a job_id and says plainly it is not saved yet. `0` returns as soon as the relayer has durably accepted the job (~1s) — accept-and-continue, chosen knowingly. A value between the two is the worst of both: against a 30-75s completion spread it pays the wait and still returns pending. Clamped to `90000`; invalid values fall back to the default |
+| `MEMWAL_MCP_REMEMBER_WAIT_MS` | `0` | How long `memwal_remember` / `memwal_remember_bulk` / `memwal_analyze` wait after accept for the Walrus write to reach `done`. **Default `0` = return at accept (~1s) with `job_id`; the fact is NOT saved yet** — resolve with `memwal_remember_status`. This is what stops the agent blocking 20–90s (or hitting the MCP client's 60s tools/call timeout). `90000` (90 seconds, the clamp) restores the old wait-for-`blob_id` behaviour; only use it on a client that raises that 60s ceiling (Claude Code does; many hosts do not). Values between 0 and ~75s usually wait *and* still return pending. Invalid values fall back to `0` |
| `MEMWAL_MCP_ACCEPT_DEADLINE_MS` | `15000` | How long a single relayer request may stall before the MCP tool gives up on it. The SDK passes an abort signal only on `recall`, so `rememberAsync` and the job-status reads have no deadline of their own and `MEMWAL_MCP_REMEMBER_WAIT_MS` bounds only when the next poll starts, not how long one takes — without this a stalled socket keeps a tool running indefinitely. Raise it only if a slow link makes healthy accepts exceed it; invalid or non-positive values fall back to the default |
| `MCP_TOOL_SLOW_WARN_MS` | `5000` | An MCP tool call still running at this duration is logged as `tool.slow` (`settled: false`) at `warn` — which is what makes a hang visible, since a hang never settles. A call that finishes at or above it is logged as `tool.slow` (`settled: true`) instead of `tool.done` |
| `TRUSTED_PROXY_HOPS` | `0` | Number of trusted reverse-proxy hops to walk from the right of `X-Forwarded-For`; `0` ignores XFF and uses the TCP peer |
diff --git a/packages/mcp/CHANGELOG.md b/packages/mcp/CHANGELOG.md
index 25acc3258..8d4bc10f1 100644
--- a/packages/mcp/CHANGELOG.md
+++ b/packages/mcp/CHANGELOG.md
@@ -10,9 +10,10 @@
- `memwal_remember` sends a content-derived idempotency key, so the retry its own timeout message invites really does attach to the job already in flight instead of storing a second paid copy. The key is computed by the tool rather than relied on from the SDK, whose published build mints a random UUID per client instance.
- `memwal_remember_status` accepts `job_ids` to settle a whole batch in one call, and reports a mixed batch honestly — a still-uploading row no longer renders the poll timeout as `error=`, which read as a failed write. Settling in one request also matters against the rate limit: 20 ids cost one request, not twenty.
- The bridge's cold-start tool list no longer disagrees with the sidecar's. `memwal_remember_status` advertised only `job_id`, required, under `additionalProperties: false`, so the batch call the tools themselves instruct was rejected until `tools/list_changed` arrived; the `waitMs` ceiling advertised 60000 after the sidecar lowered it to 45000, which came back as an MCP validation error; and `memwal_remember_bulk` still carried its pre-queue description. Tests now pin the parts an agent acts on.
+
### Changed
-- `memwal_remember` keeps blocking until the write reaches `done`, so a result carries a real `blob_id`. Returning at accept is available behind `MEMWAL_MCP_REMEMBER_WAIT_MS=0` for an operator who wants it, and remains its own product decision rather than a side effect of the latency work here.
+- `memwal_remember` / `memwal_remember_bulk` return at accept by default (`MEMWAL_MCP_REMEMBER_WAIT_MS=0`, ~1s, `job_id`). The Walrus write continues in the background; do not treat that reply as stored. Set `MEMWAL_MCP_REMEMBER_WAIT_MS=90000` to restore wait-for-`blob_id` (90s ceiling). Do not use a value between 0 and the real completion time — that pays the wait and still returns pending. The MCP TypeScript SDK's default tools/call timeout is 60s, so a 90s wait loses on hosts that do not raise it.
## 0.0.13
diff --git a/packages/mcp/src/auth-required.ts b/packages/mcp/src/auth-required.ts
index f396be359..723438068 100644
--- a/packages/mcp/src/auth-required.ts
+++ b/packages/mcp/src/auth-required.ts
@@ -41,7 +41,7 @@ interface RpcMessage {
const SIGNED_OUT_REMEMBER =
"Save a fact to the user's Walrus Memory personal memory. Call ONLY when the user explicitly asks to remember/save something. Pass the full, detailed text — never summarize.";
const SIGNED_IN_REMEMBER =
- "Save a durable fact about the user or project to their Walrus Memory. Call this PROACTIVELY whenever the user states a preference, decision, constraint, correction, identity detail, or recurring workflow — even if they did not say 'remember this'. Skip one-off tasks, the current file or bug, and small talk. Pass the full statement; do not summarize. To save several facts at once, use memwal_remember_bulk instead. A Walrus write takes 30-60s and this call waits for it, so a success carries a blob_id and means the fact is stored. If it outruns that budget you get a job_id and the fact is NOT yet saved — say so rather than claiming it is stored, and settle it with the job-status tool this server advertises (re-list tools if you do not see one).";
+ "Save a durable fact about the user or project to their Walrus Memory. Call this PROACTIVELY whenever the user states a preference, decision, constraint, correction, identity detail, or recurring workflow — even if they did not say 'remember this'. Skip one-off tasks, the current file or bug, and small talk. Pass the full statement; do not summarize. To save several facts at once, use memwal_remember_bulk instead. By default this returns in ~1s once the relayer has accepted the job (job_id) — the Walrus write is still in flight and the fact is NOT stored yet. Do not claim it is saved. Settle it with the job-status tool this server advertises (re-list tools if you do not see one).";
const SIGNED_OUT_RECALL =
"Search the user's Walrus Memory for facts relevant to a query. Returns matching memories ranked by relevance.";
const SIGNED_IN_RECALL =
@@ -72,7 +72,7 @@ function buildToolDefinitions(proactive: boolean) {
title: "Remember Multiple Facts",
annotations: { readOnlyHint: false, destructiveHint: false },
description:
- "Save multiple durable facts in one call. Use when you learned several distinct facts at once (onboarding details, a list of preferences, decisions from a discussion). Pass an array of complete fact statements (max 20) — do not summarize. Prefer this over repeated memwal_remember calls. A Walrus write takes 30-60s and this call waits for them, so a success carries blob_ids and means the facts are stored. If they outrun that budget you get job_ids and the facts are NOT yet saved — say they are being saved rather than stored, and settle them with the job-status tool this server advertises (re-list tools if you do not see one).",
+ "Save multiple durable facts in one call. Use when you learned several distinct facts at once (onboarding details, a list of preferences, decisions from a discussion). Pass an array of complete fact statements (max 20) — do not summarize. Prefer this over repeated memwal_remember calls. By default this returns in ~1s once the relayer has accepted the batch (job_ids) — the Walrus writes are still in flight and the facts are NOT stored yet. Do not claim they are saved. Settle them with the job-status tool this server advertises (re-list tools if you do not see one).",
inputSchema: {
type: "object",
properties: {
diff --git a/services/server/scripts/mcp/tools/remember-bulk.ts b/services/server/scripts/mcp/tools/remember-bulk.ts
index 486f20ce9..0a1ebf68c 100644
--- a/services/server/scripts/mcp/tools/remember-bulk.ts
+++ b/services/server/scripts/mcp/tools/remember-bulk.ts
@@ -54,7 +54,7 @@ export function registerRememberBulkTool(
{
...TOOL_METADATA.memwal_remember_bulk,
description:
- "Save multiple durable facts in one call. Use when you learned several distinct facts at once (onboarding details, a list of preferences, decisions from a discussion). Pass an array of complete fact statements (max 20) — do not summarize. Prefer this over repeated memwal_remember calls. A Walrus write takes 30-60s and this call waits for them, so a success carries blob_ids and means the facts are stored. If they outrun that budget you get job_ids and the facts are NOT yet saved — say they are being saved rather than stored, and resolve them with memwal_remember_status.",
+ "Save multiple durable facts in one call. Use when you learned several distinct facts at once (onboarding details, a list of preferences, decisions from a discussion). Pass an array of complete fact statements (max 20) — do not summarize. Prefer this over repeated memwal_remember calls. By default this returns in ~1s once the relayer has accepted the batch (job_ids) — the Walrus writes are still in flight and the facts are NOT stored yet. Do not claim they are saved. Resolve with memwal_remember_status(job_ids). blob_ids in the same reply mean they landed inside an optional wait budget (MEMWAL_MCP_REMEMBER_WAIT_MS).",
inputSchema: REMEMBER_BULK_INPUT,
},
wrapTool<{ facts: string[]; namespace?: string }>(session, "memwal_remember_bulk", async ({ facts, namespace }) => {
diff --git a/services/server/scripts/mcp/tools/remember.ts b/services/server/scripts/mcp/tools/remember.ts
index 1c31bb6a5..446b50e6e 100644
--- a/services/server/scripts/mcp/tools/remember.ts
+++ b/services/server/scripts/mcp/tools/remember.ts
@@ -54,7 +54,7 @@ export function registerRememberTool(
{
...TOOL_METADATA.memwal_remember,
description:
- "Save a durable fact about the user or project to their Walrus Memory. Call this PROACTIVELY whenever the user states a preference, decision, constraint, correction, identity detail, or recurring workflow — even if they did not say 'remember this'. Skip one-off tasks, the current file or bug, and small talk. Pass the full statement; do not summarize. To save several facts at once, use memwal_remember_bulk instead. A Walrus write takes 30-60s and this call waits for it, so a success carries a blob_id and means the fact is stored. If it outruns that budget you get a job_id and the fact is NOT yet saved — say so rather than claiming it is stored, and resolve it with memwal_remember_status.",
+ "Save a durable fact about the user or project to their Walrus Memory. Call this PROACTIVELY whenever the user states a preference, decision, constraint, correction, identity detail, or recurring workflow — even if they did not say 'remember this'. Skip one-off tasks, the current file or bug, and small talk. Pass the full statement; do not summarize. To save several facts at once, use memwal_remember_bulk instead. By default this returns in ~1s once the relayer has accepted the job (job_id) — the Walrus write is still in flight and the fact is NOT stored yet. Do not claim it is saved. Resolve with memwal_remember_status. A blob_id in the same reply means it did land inside an optional wait budget (MEMWAL_MCP_REMEMBER_WAIT_MS).",
inputSchema: REMEMBER_INPUT,
},
wrapTool<{ text: string; namespace?: string }>(session, "memwal_remember", async ({ text, namespace }) => {
From 3392a8db49d4c7ac51832d06c4afb928c7f849ff Mon Sep 17 00:00:00 2001
From: Harry Phan
Date: Fri, 18 Sep 2026 01:49:04 +0700
Subject: [PATCH 073/132] fix(mcp): teach agents that job_id-at-accept is the
normal remember result
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit
Initialize instructions still said remember waits 30-60s and a success is
blob_id (job_id = exception). Default wait is 0, so that inverted the
contract agents actually follow.
Also widen the cold-start pending-warning test to accept 'NOT stored'
as well as 'NOT saved' — CI failed on the previous copy change.
---
packages/mcp/src/instructions.ts | 14 ++++++--------
packages/mcp/test/tool-definitions.test.mjs | 2 +-
services/server/scripts/mcp/server.ts | 14 ++++++--------
3 files changed, 13 insertions(+), 17 deletions(-)
diff --git a/packages/mcp/src/instructions.ts b/packages/mcp/src/instructions.ts
index 057301bae..ddcf6eb31 100644
--- a/packages/mcp/src/instructions.ts
+++ b/packages/mcp/src/instructions.ts
@@ -39,14 +39,12 @@ export const PROACTIVE_INSTRUCTIONS = [
"summary. Skip one-off tasks, the current file or bug, and small talk. Use",
"memwal_remember_bulk when several distinct facts arrived at once.",
"",
- "A Walrus write takes roughly 30-60s, and memwal_remember and memwal_remember_bulk wait",
- "for it: a successful call comes back with a blob_id, and that means the fact is stored.",
- "",
- "If a write outruns that budget the call returns job_ids instead, saying the facts are",
- "ACCEPTED but NOT YET SAVED. That is the exception, not the normal result. When it",
- "happens, say the facts are being saved rather than that they are saved, do NOT re-send",
- "them — that queues duplicates — and resolve the ids with memwal_remember_status (it takes",
- "job_id, or job_ids for a whole batch). Only a blob_id means a fact is stored.",
+ "By default memwal_remember and memwal_remember_bulk return in ~1s once the relayer has",
+ "accepted the job (job_id / job_ids). The Walrus write continues in the background (~30-60s)",
+ "and the fact is NOT stored yet. That is the normal result. Do not claim it is saved.",
+ "Do NOT re-send the same text — that queues duplicates. Resolve with memwal_remember_status",
+ "(job_id, or job_ids for a whole batch). Only a blob_id in the tool reply means the fact is",
+ "already stored (that happens when an optional wait budget was set and the write finished).",
"",
"RECOVER: if memwal_recall unexpectedly returns nothing for a namespace that has been used",
"before, call memwal_restore to rebuild the index from Walrus.",
diff --git a/packages/mcp/test/tool-definitions.test.mjs b/packages/mcp/test/tool-definitions.test.mjs
index 51c0ef9c9..501f62583 100644
--- a/packages/mcp/test/tool-definitions.test.mjs
+++ b/packages/mcp/test/tool-definitions.test.mjs
@@ -121,7 +121,7 @@ test("cold-start write tools warn that a result may not be saved yet", () => {
// pending result is normal reports it to the user as stored.
for (const name of ["memwal_remember", "memwal_remember_bulk"]) {
const d = desc(TOOL_DEFINITIONS, name);
- assert.match(d, /NOT yet saved|NOT saved/i, `${name} omits the pending warning`);
+ assert.match(d, /NOT (yet )?(saved|stored)/i, `${name} omits the pending warning`);
assert.match(d, /settle it|settle them/, `${name} does not say to settle the job`);
}
});
diff --git a/services/server/scripts/mcp/server.ts b/services/server/scripts/mcp/server.ts
index 78de98884..6efd05803 100644
--- a/services/server/scripts/mcp/server.ts
+++ b/services/server/scripts/mcp/server.ts
@@ -50,14 +50,12 @@ const INSTRUCTIONS = [
"summary. Skip one-off tasks, the current file or bug, and small talk. Use",
"memwal_remember_bulk when several distinct facts arrived at once.",
"",
- "A Walrus write takes roughly 30-60s, and memwal_remember and memwal_remember_bulk wait",
- "for it: a successful call comes back with a blob_id, and that means the fact is stored.",
- "",
- "If a write outruns that budget the call returns job_ids instead, saying the facts are",
- "ACCEPTED but NOT YET SAVED. That is the exception, not the normal result. When it",
- "happens, say the facts are being saved rather than that they are saved, do NOT re-send",
- "them — that queues duplicates — and resolve the ids with memwal_remember_status (it takes",
- "job_id, or job_ids for a whole batch). Only a blob_id means a fact is stored.",
+ "By default memwal_remember and memwal_remember_bulk return in ~1s once the relayer has",
+ "accepted the job (job_id / job_ids). The Walrus write continues in the background (~30-60s)",
+ "and the fact is NOT stored yet. That is the normal result. Do not claim it is saved.",
+ "Do NOT re-send the same text — that queues duplicates. Resolve with memwal_remember_status",
+ "(job_id, or job_ids for a whole batch). Only a blob_id in the tool reply means the fact is",
+ "already stored (that happens when an optional wait budget was set and the write finished).",
"",
"RECOVER: if memwal_recall unexpectedly returns nothing for a namespace that has been used",
"before, call memwal_restore to rebuild the index from Walrus.",
From 812cfe5b22a305c7f71543e3ec2e908bf3ff47ed Mon Sep 17 00:00:00 2001
From: Harry Phan
Date: Fri, 18 Sep 2026 13:01:53 +0700
Subject: [PATCH 074/132] test(mcp): match Settle/settle in cold-start
pending-write copy
CI failed: description says 'Settle it' and the regex was case-sensitive.
---
packages/mcp/test/tool-definitions.test.mjs | 2 +-
1 file changed, 1 insertion(+), 1 deletion(-)
diff --git a/packages/mcp/test/tool-definitions.test.mjs b/packages/mcp/test/tool-definitions.test.mjs
index 501f62583..ffd5b37a0 100644
--- a/packages/mcp/test/tool-definitions.test.mjs
+++ b/packages/mcp/test/tool-definitions.test.mjs
@@ -122,7 +122,7 @@ test("cold-start write tools warn that a result may not be saved yet", () => {
for (const name of ["memwal_remember", "memwal_remember_bulk"]) {
const d = desc(TOOL_DEFINITIONS, name);
assert.match(d, /NOT (yet )?(saved|stored)/i, `${name} omits the pending warning`);
- assert.match(d, /settle it|settle them/, `${name} does not say to settle the job`);
+ assert.match(d, /settle it|settle them/i, `${name} does not say to settle the job`);
}
});
From d0a101439e408636a2e1e74da34b0db324a78165 Mon Sep 17 00:00:00 2001
From: Harry Phan
Date: Fri, 18 Sep 2026 13:06:04 +0700
Subject: [PATCH 075/132] fix(mcp): trust the runtime dir by ownership and
location, not absoluteness (WALM-640)
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit
Review found four ways the trusted-path launcher did not actually hold.
A. `runtimeRoot()` only rejected a *relative* `MEMWAL_MCP_RUNTIME_DIR`, but
`${workspaceFolder}/.memwal-runtime` is expanded by the client and is therefore
absolute — and inside the repository. A committed tree there was found by
`resolveInstalledEntry()`, skipped the install entirely, and was spawned with
`process.execPath`: the original attack, restored in full. The root is now
refused when it is inside (or contains) the project tree the client started in,
with symlinks resolved first so a link out of the repo does not launder it. The
home directory is not treated as a project, so a client that starts there keeps
working with the default root.
E. `mkdirSync(root, { recursive: true })` created `~/.memwal` and `~/.memwal/runtime`
at the umask default — 0775 under umask 002, i.e. group-writable, for the one
directory whose entry point is trusted because nobody else can write it. Both are
now created with mode 0700, and both are verified: `lstat` (never `stat`, so a
symlink cannot pass), a real directory, owned by the current uid, not group- or
world-writable. The same verification runs for the default root, for an override,
and for the per-version install directory. On a fresh machine this is also what
creates `~/.memwal`, so `credentials.json` no longer lands inside a loose parent.
B. The install ran arbitrary code before verifying anything: no `--ignore-scripts`,
so npm executed pre/post-install scripts of the package and its whole dependency
tree as the user. `cwd: staging` kept the project's `.npmrc` out of play, but npm
ranks the environment above every `.npmrc`, so `npm_config_registry` / `_ca` in a
repo-supplied client env block still won. The install now passes
`--ignore-scripts`, pins the registry on the command line, and runs with the
`npm_config_*` / `NODE_OPTIONS` environment scrubbed.
D. `spawnSync("npm.cmd")` without `shell: true` is EINVAL since the CVE-2024-27980
fix, so on Windows the install always threw and the server never started on any
client. npm is now resolved to its JS entry point from `process.execPath` and run
as `node npm-cli.js` — no shell, no `.cmd`, and no PATH lookup, which also closes
the header's PATH claim. The PATH scan is a fallback only: it skips relative
entries and anything inside the project tree, and on win32 a `.cmd` found that way
is spawned with `shell: true` and cmd.exe-quoted arguments.
The module header claimed the installed tree was "verified before it is used"; that
was three string comparisons on code that had already run. It now says what the
checks do and do not establish: where the code came from and that nobody else can
write it, not that the registry served what a reviewer read.
16 new tests in trusted-launcher.test.mjs, including the planted in-project runtime
root (refused, and separately asserted to be an otherwise-usable install so the
refusal is about location), a symlinked one, a group-writable and a foreign-owned
root, 0700 creation of both `~/.memwal` and the root, the install flags and the
scrubbed environment, and the win32 command construction on both branches — which
needs no Windows host. The old "no installer" test drove failure with `PATH: ""`;
that no longer means anything now that npm is resolved absolutely, so it drives it
with a read-only runtime root instead, and still asserts the launcher refuses to
fall back rather than reaching for the package name.
---
packages/mcp/plugin/scripts/launch_mcp.mjs | 7 +
.../mcp/plugin/scripts/lib/mcp-launch.mjs | 393 +++++++++++++++---
packages/mcp/test/trusted-launcher.test.mjs | 381 ++++++++++++++++-
3 files changed, 726 insertions(+), 55 deletions(-)
diff --git a/packages/mcp/plugin/scripts/launch_mcp.mjs b/packages/mcp/plugin/scripts/launch_mcp.mjs
index 96f96a2a4..db8012dec 100644
--- a/packages/mcp/plugin/scripts/launch_mcp.mjs
+++ b/packages/mcp/plugin/scripts/launch_mcp.mjs
@@ -9,6 +9,13 @@
* it makes sure the pinned version is installed under ~/.memwal/runtime and runs
* that absolute entry point with the current node binary.
*
+ * The runtime directory is trusted because of what it is, not because of how it is
+ * spelled: it must sit outside the project tree (an absolute
+ * `${workspaceFolder}/.memwal-runtime` is still the project), be a real directory
+ * owned by the current user, and not be group- or world-writable. See
+ * lib/mcp-launch.mjs for the checks and for what the install step does and does not
+ * guarantee.
+ *
* Everything after the script path is forwarded to the server untouched, so the
* manifests keep working with flags such as `--dev`, `--namespace work` or
* `--relayer `. The environment is inherited as-is, and the working directory
diff --git a/packages/mcp/plugin/scripts/lib/mcp-launch.mjs b/packages/mcp/plugin/scripts/lib/mcp-launch.mjs
index 61db4ecac..020f31467 100644
--- a/packages/mcp/plugin/scripts/lib/mcp-launch.mjs
+++ b/packages/mcp/plugin/scripts/lib/mcp-launch.mjs
@@ -14,15 +14,32 @@
*
* ~/.memwal/runtime/memwal-mcp@/node_modules/@mysten-incubation/memwal-mcp
*
- * The pinned version is installed there once (npm, with the trusted directory as
- * both prefix and cwd, so no project `.npmrc` or project `node_modules` is in play),
- * and every later launch is a plain `node `. Nothing here ever
- * looks at `process.cwd()`, at a project `node_modules`, or at a PATH-relative bin
- * shim, so a package planted in a project cannot be reached at all.
+ * What "a directory we own" means is enforced, not assumed. The runtime root — the
+ * default one just as much as a `MEMWAL_MCP_RUNTIME_DIR` override — must be an
+ * absolute path outside the project tree the client started in, must be a real
+ * directory (`lstat`, so a symlink is refused), must belong to the current uid, and
+ * must not be group- or world-writable. We create it with mode 0700. An absolute
+ * path is *not* by itself a trusted path: `${workspaceFolder}/.memwal-runtime` is
+ * absolute too, and a repository can commit a whole tree there.
*
- * The installed tree is verified before it is used and before it is published to its
- * final path: the manifest must carry our package name, the pinned version, and a
- * `bin` entry that resolves back inside the package directory.
+ * The pinned version is installed there once with npm, and every later launch is a
+ * plain `node `. Nothing here ever looks at `process.cwd()`,
+ * at a project `node_modules`, or at a bin shim found through PATH.
+ *
+ * Limits of the install step, stated plainly because a previous version of this
+ * header overstated them:
+ *
+ * - The install runs with `--ignore-scripts` and an explicitly pinned registry,
+ * and the `npm_config_*` / `NODE_OPTIONS` environment is scrubbed for that one
+ * spawn, so a repository-supplied client env block cannot redirect it. npm is
+ * invoked as `node ` resolved from `process.execPath` where that
+ * layout exists; only if it does not do we fall back to scanning PATH entries
+ * ourselves, skipping relative entries and entries inside the project tree.
+ * - Verification of the installed tree is a manifest check (name, pinned version,
+ * a `bin` that resolves back inside the package directory) plus the ownership
+ * and permission checks above. It is not a signature or integrity check of the
+ * package contents: it establishes *where* the code came from and that nobody
+ * else can write it, not that the registry served what a reviewer read.
*
* The version pin lives in exactly one place — `plugin/plugin.json`'s `version`,
* which the release verifier already keeps equal to `packages/mcp/package.json` —
@@ -30,21 +47,33 @@
*/
import {
existsSync,
+ lstatSync,
mkdirSync,
mkdtempSync,
readFileSync,
+ realpathSync,
renameSync,
rmSync,
writeFileSync,
} from "node:fs";
import { spawnSync } from "node:child_process";
+import { basename, delimiter, dirname, isAbsolute, join, relative, resolve } from "node:path";
import { homedir } from "node:os";
-import { dirname, isAbsolute, join, relative, resolve } from "node:path";
import { fileURLToPath } from "node:url";
export const MCP_PACKAGE_NAME = "@mysten-incubation/memwal-mcp";
export const MCP_BIN_NAME = "memwal-mcp";
+/**
+ * The registry the reviewed version is installed from. Deliberately a constant and
+ * not an environment lookup: npm puts environment configuration above every
+ * `.npmrc`, so `npm_config_registry` in a repository-supplied MCP `env` block would
+ * otherwise choose where our "trusted" code comes from. Behind a private registry,
+ * pre-populate the runtime directory by hand (see packages/mcp/README.md) — the
+ * launcher then finds the install and never runs npm at all.
+ */
+export const MCP_REGISTRY = "https://registry.npmjs.org/";
+
const SCRIPT_DIR = dirname(fileURLToPath(import.meta.url));
/** plugin/scripts/lib -> plugin/scripts -> plugin */
export const PLUGIN_ROOT = dirname(dirname(SCRIPT_DIR));
@@ -62,11 +91,92 @@ export function pinnedVersion(pluginRoot = PLUGIN_ROOT) {
return version;
}
+function isInside(parent, child) {
+ const rel = relative(parent, child);
+ return rel !== "" && !rel.startsWith("..") && !isAbsolute(rel);
+}
+
+/**
+ * Resolve symlinks in the part of a path that exists, keeping the rest verbatim.
+ *
+ * Without this, containment checks are comparing spellings rather than locations:
+ * `/tmp/x` and `/private/tmp/x` are the same directory on macOS, and a symlink
+ * inside a project is a one-line way to point a "safe looking" path back into it.
+ */
+export function canonicalPath(target) {
+ let dir = resolve(target);
+ const tail = [];
+ for (;;) {
+ try {
+ return tail.length === 0 ? realpathSync(dir) : join(realpathSync(dir), ...tail);
+ } catch {
+ const parent = dirname(dir);
+ if (parent === dir) return resolve(target);
+ tail.unshift(basename(dir));
+ dir = parent;
+ }
+ }
+}
+
+/**
+ * The project tree the client started us in, or null when the working directory is
+ * not usefully inside one (the home directory and the filesystem root are not
+ * projects — treating them as such would refuse the default runtime root).
+ *
+ * "Project" is the nearest ancestor carrying a `.git` or a `package.json`, and the
+ * working directory itself when neither is found: a runtime root under either is a
+ * root a repository could have shipped.
+ */
+export function enclosingProjectRoot(cwd = process.cwd(), home = homedir()) {
+ const start = canonicalPath(cwd);
+ const stop = canonicalPath(home);
+ const marked = (() => {
+ let dir = start;
+ while (dir !== stop) {
+ if (existsSync(join(dir, ".git")) || existsSync(join(dir, "package.json"))) return dir;
+ const parent = dirname(dir);
+ if (parent === dir) return null;
+ dir = parent;
+ }
+ return null;
+ })();
+ const project = marked ?? start;
+ if (project === stop || project === dirname(project)) return null;
+ return project;
+}
+
+/**
+ * Absoluteness is not trust. `${workspaceFolder}/.memwal-runtime` expands to an
+ * absolute path inside the repository, and a repository can commit a complete fake
+ * install there; the launcher would then find it, skip the install, and run it.
+ */
+export function assertRuntimeRootLocation(root, { cwd = process.cwd() } = {}) {
+ const resolved = resolve(root);
+ const canonical = canonicalPath(resolved);
+ // The project root contains the working directory, so checking it also catches a
+ // root planted below the cwd. When there is no project (the client started in the
+ // home directory, say) there is nothing to refuse: `~/.memwal/runtime` is inside
+ // the home directory by design.
+ const project = enclosingProjectRoot(cwd);
+ if (
+ project !== null &&
+ (canonical === project || isInside(project, canonical) || isInside(canonical, project))
+ ) {
+ throw new Error(
+ `refusing to use "${resolved}" as the MemWal runtime directory: it is ` +
+ `inside the project tree at "${project}". A runtime root must live outside ` +
+ `any directory a project can write — being an absolute path is not ` +
+ `enough, since a client expands "\${workspaceFolder}" to one.`,
+ );
+ }
+ return resolved;
+}
+
/**
* Root of the trusted install area. `MEMWAL_MCP_RUNTIME_DIR` may relocate it, but
- * only to an absolute path: a relative one would resolve against the project the
- * client happens to have open, which is the directory this whole module exists to
- * stay out of.
+ * only to an absolute path outside the project the client has open: a relative path
+ * would resolve against that project, and an absolute path inside it is the same
+ * attack with an extra step.
*/
export function runtimeRoot() {
const override = process.env.MEMWAL_MCP_RUNTIME_DIR;
@@ -76,9 +186,9 @@ export function runtimeRoot() {
`MEMWAL_MCP_RUNTIME_DIR must be an absolute path, received "${override}"`,
);
}
- return override;
+ return assertRuntimeRootLocation(override);
}
- return join(homedir(), ".memwal", "runtime");
+ return assertRuntimeRootLocation(join(homedir(), ".memwal", "runtime"));
}
/** One directory per pinned version, so an upgrade never mutates a running install. */
@@ -86,9 +196,68 @@ export function installDir(version = pinnedVersion(), root = runtimeRoot()) {
return join(root, `${MCP_BIN_NAME}@${version}`);
}
-function isInside(parent, child) {
- const rel = relative(parent, child);
- return rel !== "" && !rel.startsWith("..") && !isAbsolute(rel);
+/**
+ * Verify a directory we are about to trust with executable code, with the same
+ * checks MemWal applies to the state directories it owns: `lstat` (so a symlink
+ * never passes as a directory), a real directory, owned by the current uid, and
+ * not group- or world-writable.
+ *
+ * Returns false when the directory does not exist, true when it exists and is
+ * trustworthy, and throws otherwise. On Windows there is no meaningful uid or mode,
+ * so only the "is a real directory" half applies.
+ */
+export function verifyTrustedDirectory(dir, { label = dir } = {}) {
+ let stats;
+ try {
+ stats = lstatSync(dir);
+ } catch (err) {
+ if (err?.code === "ENOENT") return false;
+ throw new Error(`could not inspect ${label}: ${err?.message ?? String(err)}`);
+ }
+ if (!stats.isDirectory()) {
+ throw new Error(
+ `${label} is not a directory (a symlink or file there could redirect the ` +
+ `MemWal runtime into a tree we do not control)`,
+ );
+ }
+ const uid = typeof process.getuid === "function" ? process.getuid() : null;
+ if (uid === null) return true;
+ if (stats.uid !== uid) {
+ throw new Error(
+ `${label} is owned by uid ${stats.uid}, not by the current user (uid ${uid}); ` +
+ `refusing to run code from a directory somebody else controls`,
+ );
+ }
+ if ((stats.mode & 0o022) !== 0) {
+ throw new Error(
+ `${label} is group- or world-writable (mode ${(stats.mode & 0o7777)
+ .toString(8)
+ .padStart(4, "0")}); another account could replace the entry point we ` +
+ `trust. Run: chmod 700 ${label}`,
+ );
+ }
+ return true;
+}
+
+/**
+ * Create the runtime root (mode 0700) if it is missing and verify it either way.
+ * For the default layout the `~/.memwal` parent is prepared the same way: the
+ * launcher can run before the first login, and `credentials.json` is written into
+ * that directory later by a `mkdirSync` that is a no-op once it exists.
+ */
+export function prepareRuntimeRoot(root) {
+ const resolved = resolve(root);
+ const managed = [];
+ const defaultRoot = join(homedir(), ".memwal", "runtime");
+ if (resolved === resolve(defaultRoot)) managed.push(dirname(resolved));
+ managed.push(resolved);
+
+ for (const dir of managed) {
+ if (verifyTrustedDirectory(dir)) continue;
+ mkdirSync(dir, { recursive: true, mode: 0o700 });
+ verifyTrustedDirectory(dir);
+ }
+ return resolved;
}
/**
@@ -124,8 +293,142 @@ export function resolveInstalledEntry(dir, version) {
return entry;
}
-function npmCommand() {
- return process.platform === "win32" ? "npm.cmd" : "npm";
+/**
+ * Locate npm without going through PATH resolution where that is possible.
+ *
+ * Preferred result is npm's JS entry point next to the running node binary, which
+ * we then run as `node npm-cli.js`. That avoids a PATH-relative binary entirely and
+ * also sidesteps the Windows `spawnSync("npm.cmd")` EINVAL that CVE-2024-27980's
+ * fix introduced for `.cmd`/`.bat` targets spawned without a shell.
+ *
+ * Only if no such layout exists do we scan PATH ourselves — skipping empty and
+ * relative entries and anything inside the project tree, which is what an inherited
+ * PATH could otherwise smuggle in.
+ */
+export function resolveNpm({
+ execPath = process.execPath,
+ platform = process.platform,
+ env = process.env,
+ cwd = process.cwd(),
+} = {}) {
+ const execDirs = [dirname(execPath)];
+ try {
+ const real = dirname(realpathSync(execPath));
+ if (!execDirs.includes(real)) execDirs.push(real);
+ } catch {
+ /* execPath should always resolve; a failure just means one fewer candidate */
+ }
+
+ const relativeLayouts =
+ platform === "win32"
+ ? [
+ ["node_modules", "npm", "bin", "npm-cli.js"],
+ ["..", "node_modules", "npm", "bin", "npm-cli.js"],
+ ]
+ : [
+ ["..", "lib", "node_modules", "npm", "bin", "npm-cli.js"],
+ ["..", "..", "..", "..", "lib", "node_modules", "npm", "bin", "npm-cli.js"],
+ ["node_modules", "npm", "bin", "npm-cli.js"],
+ ];
+
+ for (const dir of execDirs) {
+ for (const layout of relativeLayouts) {
+ const candidate = resolve(join(dir, ...layout));
+ if (existsSync(candidate)) return { kind: "js", path: candidate };
+ }
+ }
+
+ const project = enclosingProjectRoot(cwd);
+ const binNames = platform === "win32" ? ["npm.cmd", "npm.exe", "npm"] : ["npm"];
+ for (const rawEntry of String(env.PATH ?? env.Path ?? "").split(delimiter)) {
+ const entry = rawEntry.trim();
+ // A relative PATH entry resolves against the project the client started in.
+ if (entry === "" || !isAbsolute(entry)) continue;
+ const dir = canonicalPath(entry);
+ if (project && (dir === project || isInside(project, dir))) continue;
+
+ // A node installation reached through PATH still usually ships npm's JS entry
+ // point next to it; prefer that over the shim.
+ for (const layout of relativeLayouts) {
+ const candidate = resolve(join(dir, ...layout));
+ if (existsSync(candidate)) return { kind: "js", path: candidate };
+ }
+ for (const name of binNames) {
+ const candidate = join(dir, name);
+ if (existsSync(candidate)) return { kind: "bin", path: candidate };
+ }
+ }
+ return { kind: "none", path: null };
+}
+
+/** cmd.exe quoting for the one case where we must go through a shell. */
+export function quoteWindowsArgument(value) {
+ const text = String(value);
+ if (text !== "" && !/[\s"^&|<>()%!]/.test(text)) return text;
+ return `"${text.replace(/(\\*)"/g, '$1$1\\"').replace(/(\\+)$/, "$1$1")}"`;
+}
+
+/**
+ * Turn a resolved npm into an actual spawn plan.
+ *
+ * `kind: "js"` is the path we want everywhere: `node npm-cli.js …`, no shell, no
+ * PATH. `kind: "bin"` on win32 is `npm.cmd`, which since Node 18.20.2 / 20.12.2 /
+ * 21.7.3 cannot be spawned without `shell: true` (EINVAL) — so that branch sets it
+ * and quotes every argument for cmd.exe itself.
+ */
+export function npmSpawnPlan(npm, args, { platform = process.platform, execPath = process.execPath } = {}) {
+ if (npm.kind === "js") {
+ return { command: execPath, args: [npm.path, ...args], shell: false };
+ }
+ if (npm.kind === "bin") {
+ if (platform === "win32") {
+ return {
+ command: quoteWindowsArgument(npm.path),
+ args: args.map(quoteWindowsArgument),
+ shell: true,
+ };
+ }
+ return { command: npm.path, args, shell: false };
+ }
+ throw new Error(
+ "could not locate npm: no npm-cli.js next to the running node binary and no " +
+ "usable npm on PATH outside the project directory",
+ );
+}
+
+/** Environment for the install spawn: the caller's, minus everything npm reads from it. */
+export function installEnvironment(env = process.env) {
+ const scrubbed = {};
+ for (const [key, value] of Object.entries(env)) {
+ if (/^npm_config_/i.test(key)) continue;
+ if (/^npm_package_/i.test(key)) continue;
+ if (key === "NODE_OPTIONS") continue;
+ scrubbed[key] = value;
+ }
+ // Belt and braces: even if a key slipped through the filter above, these lose to
+ // the command line flags we pass, and npm resolves its own config from here.
+ scrubbed.npm_config_registry = MCP_REGISTRY;
+ scrubbed.npm_config_ignore_scripts = "true";
+ return scrubbed;
+}
+
+/** The install arguments, exported so the tests can assert the hardening flags. */
+export function installArguments(spec, staging) {
+ return [
+ "install",
+ spec,
+ "--prefix",
+ staging,
+ // The package tree is fetched and unpacked, never executed: npm would
+ // otherwise run preinstall/install/postinstall of the whole dependency tree
+ // as the user, before any of our verification runs.
+ "--ignore-scripts",
+ `--registry=${MCP_REGISTRY}`,
+ "--no-audit",
+ "--no-fund",
+ "--no-save",
+ "--loglevel=error",
+ ];
}
/**
@@ -136,7 +439,6 @@ function npmCommand() {
*/
function install(version, root) {
const target = installDir(version, root);
- mkdirSync(root, { recursive: true });
const staging = mkdtempSync(join(root, `.staging-${MCP_BIN_NAME}-`));
try {
@@ -152,33 +454,25 @@ function install(version, root) {
);
const spec = `${MCP_PACKAGE_NAME}@${version}`;
- const result = spawnSync(
- npmCommand(),
- [
- "install",
- spec,
- "--prefix",
- staging,
- "--no-audit",
- "--no-fund",
- "--no-save",
- "--loglevel=error",
- ],
- {
- // cwd inside the trusted area: npm reads .npmrc from cwd upward, and
- // the project's .npmrc must not get to choose the registry we install
- // the reviewed version from.
- cwd: staging,
- encoding: "utf8",
- stdio: ["ignore", "pipe", "pipe"],
- },
- );
+ const npm = resolveNpm();
+ const plan = npmSpawnPlan(npm, installArguments(spec, staging));
+ const result = spawnSync(plan.command, plan.args, {
+ // cwd inside the trusted area: npm reads .npmrc from cwd upward, and the
+ // project's .npmrc must not get to choose the registry we install the
+ // reviewed version from. (The environment outranks every .npmrc, which is
+ // why installEnvironment() scrubs it as well.)
+ cwd: staging,
+ encoding: "utf8",
+ stdio: ["ignore", "pipe", "pipe"],
+ shell: plan.shell,
+ env: installEnvironment(),
+ });
if (result.error) {
- throw new Error(`could not run ${npmCommand()}: ${result.error.message}`);
+ throw new Error(`could not run npm (${npm.path}): ${result.error.message}`);
}
if (result.status !== 0) {
throw new Error(
- `${npmCommand()} install ${spec} failed (exit ${result.status}): ` +
+ `npm install ${spec} failed (exit ${result.status}): ` +
`${(result.stderr || result.stdout || "").trim()}`,
);
}
@@ -186,7 +480,7 @@ function install(version, root) {
const staged = resolveInstalledEntry(staging, version);
if (!staged) {
throw new Error(
- `${npmCommand()} install ${spec} reported success but produced no usable ` +
+ `npm install ${spec} reported success but produced no usable ` +
`${MCP_PACKAGE_NAME} entry point in ${staging}`,
);
}
@@ -196,6 +490,7 @@ function install(version, root) {
} catch (err) {
// Lost the race, or a previous run left the directory behind: fall back to
// whatever is at the final path, but only if it verifies.
+ verifyTrustedDirectory(target);
const existing = resolveInstalledEntry(target, version);
if (!existing) throw err;
return existing;
@@ -219,9 +514,13 @@ function install(version, root) {
* launcher fails instead of reaching for a package name that a project could answer.
*/
export function ensureTrustedEntry({ version = pinnedVersion(), root = runtimeRoot() } = {}) {
- const existing = resolveInstalledEntry(installDir(version, root), version);
- if (existing) return existing;
- return install(version, root);
+ const trustedRoot = prepareRuntimeRoot(assertRuntimeRootLocation(root));
+ const dir = installDir(version, trustedRoot);
+ if (verifyTrustedDirectory(dir)) {
+ const existing = resolveInstalledEntry(dir, version);
+ if (existing) return existing;
+ }
+ return install(version, trustedRoot);
}
/** Exported for the regression test: the path we expect, without installing anything. */
diff --git a/packages/mcp/test/trusted-launcher.test.mjs b/packages/mcp/test/trusted-launcher.test.mjs
index 7a343a751..66f88748a 100644
--- a/packages/mcp/test/trusted-launcher.test.mjs
+++ b/packages/mcp/test/trusted-launcher.test.mjs
@@ -18,18 +18,37 @@
import { test } from "node:test";
import assert from "node:assert/strict";
import { spawnSync } from "node:child_process";
-import { chmodSync, mkdirSync, mkdtempSync, readFileSync, rmSync, writeFileSync } from "node:fs";
+import {
+ chmodSync,
+ lstatSync,
+ mkdirSync,
+ mkdtempSync,
+ readFileSync,
+ rmSync,
+ symlinkSync,
+ writeFileSync,
+} from "node:fs";
import { tmpdir } from "node:os";
import { dirname, join, resolve } from "node:path";
import { fileURLToPath } from "node:url";
import {
MCP_PACKAGE_NAME,
+ MCP_REGISTRY,
+ assertRuntimeRootLocation,
+ canonicalPath,
expectedEntryPath,
+ installArguments,
installDir,
+ installEnvironment,
+ npmSpawnPlan,
pinnedVersion,
+ prepareRuntimeRoot,
+ quoteWindowsArgument,
resolveInstalledEntry,
+ resolveNpm,
runtimeRoot,
+ verifyTrustedDirectory,
} from "../plugin/scripts/lib/mcp-launch.mjs";
const __dirname = dirname(fileURLToPath(import.meta.url));
@@ -191,16 +210,23 @@ test("an install under a different version is not accepted for the pin", (t) =>
assert.ok(!String(trusted.entry).includes(project.dir));
});
-test("with no trusted install and no installer, the launcher fails instead of falling back", (t) => {
+test("with no trusted install and a failing install, the launcher fails instead of falling back", (t) => {
const project = makeHostileProject(t);
const trusted = makeTrustedRuntime(t, { populate: false });
- // An empty PATH makes the `npm` lookup fail the way an offline or broken
- // toolchain would. The launcher must surface that, never reach for the name.
- const result = runLauncher([], {
- cwd: project.dir,
- env: { MEMWAL_MCP_RUNTIME_DIR: trusted.root, PATH: "" },
- });
+ // A read-only trusted root makes the install fail the way an offline or broken
+ // toolchain would, without going near the network. (An empty PATH no longer does
+ // it: npm is resolved from process.execPath, which is the point of that change.)
+ let result;
+ chmodSync(trusted.root, 0o500);
+ try {
+ result = runLauncher([], {
+ cwd: project.dir,
+ env: { MEMWAL_MCP_RUNTIME_DIR: trusted.root },
+ });
+ } finally {
+ chmodSync(trusted.root, 0o700);
+ }
assert.notEqual(result.status, 0, "launcher must not succeed without a trusted install");
assert.doesNotMatch(result.stdout, new RegExp(LOCAL_MARKER));
@@ -279,3 +305,342 @@ test("no plugin launch manifest resolves the server through npx", () => {
);
assert.match(installer, /launch_mcp\.mjs/);
});
+
+/* ------------------------------------------------------------------------- *
+ * Review follow-ups (WALM-640): an absolute runtime root is not a trusted
+ * runtime root, and the install step must not execute what it has not verified.
+ * ------------------------------------------------------------------------- */
+
+const POSIX = process.platform !== "win32";
+
+/** A project that a repository could really ship: markers, config, planted runtime. */
+function makeRepoProject(t) {
+ const dir = mkdtempSync(join(tmpdir(), "memwal-repo-project-"));
+ t.after(() => rmSync(dir, { recursive: true, force: true }));
+ mkdirSync(join(dir, ".git"), { recursive: true });
+ writeFileSync(join(dir, "package.json"), JSON.stringify({ name: "victim", version: "1.0.0" }));
+ return dir;
+}
+
+/** The whole WALM-640 tree, committed inside the repository under `name`. */
+function plantRuntimeRoot(projectDir, name = ".memwal-runtime") {
+ const root = join(projectDir, name);
+ const target = installDir(PIN, root);
+ mkdirSync(target, { recursive: true });
+ plantPackage(target, { version: PIN, marker: LOCAL_MARKER });
+ return root;
+}
+
+test("an absolute MEMWAL_MCP_RUNTIME_DIR inside the project is refused", (t) => {
+ const project = makeHostileProject(t);
+ // What a client writes into ${workspaceFolder}/.cursor/mcp.json expands to: an
+ // absolute path, which the old check accepted, pointing straight back into the
+ // repository that supplied it.
+ const plantedRoot = plantRuntimeRoot(project.dir);
+
+ const result = runLauncher(["--print-entry"], {
+ cwd: project.dir,
+ env: { MEMWAL_MCP_RUNTIME_DIR: plantedRoot },
+ });
+
+ assert.notEqual(result.status, 0, "an in-project runtime root must not be usable");
+ assert.match(result.stderr, /inside the project tree/);
+ assert.match(result.stderr, /refusing to fall back/);
+ assert.doesNotMatch(result.stdout, new RegExp(LOCAL_MARKER));
+ assert.doesNotMatch(result.stdout, new RegExp(plantedRoot.replace(/[.*+?^${}()|[\]\\]/g, "\\$&")));
+});
+
+test("the planted in-project tree is otherwise a complete, usable install", (t) => {
+ // Without this the test above could pass for the wrong reason (a tree that would
+ // not have resolved anyway). It is refused because of where it is, not what it is.
+ const project = makeHostileProject(t);
+ const plantedRoot = plantRuntimeRoot(project.dir);
+ assert.ok(
+ resolveInstalledEntry(installDir(PIN, plantedRoot), PIN),
+ "the planted tree must be a resolvable install, so the refusal is about location",
+ );
+});
+
+test("a runtime root inside the repository but above the cwd is refused too", (t) => {
+ const repo = makeRepoProject(t);
+ const plantedRoot = plantRuntimeRoot(repo);
+ const nested = join(repo, "packages", "app");
+ mkdirSync(nested, { recursive: true });
+
+ const result = runLauncher(["--print-entry"], {
+ cwd: nested,
+ env: { MEMWAL_MCP_RUNTIME_DIR: plantedRoot },
+ });
+
+ assert.notEqual(result.status, 0);
+ assert.match(result.stderr, /inside the project tree/);
+ assert.doesNotMatch(result.stdout, new RegExp(LOCAL_MARKER));
+});
+
+test("a runtime root that only looks external is refused once symlinks resolve", (t) => {
+ if (!POSIX) return;
+ const project = makeHostileProject(t);
+ const plantedRoot = plantRuntimeRoot(project.dir);
+ const outside = mkdtempSync(join(tmpdir(), "memwal-link-"));
+ t.after(() => rmSync(outside, { recursive: true, force: true }));
+ const link = join(outside, "runtime");
+ symlinkSync(plantedRoot, link);
+
+ const result = runLauncher(["--print-entry"], {
+ cwd: project.dir,
+ env: { MEMWAL_MCP_RUNTIME_DIR: link },
+ });
+
+ assert.notEqual(result.status, 0, "a symlink into the project is still the project");
+ assert.match(result.stderr, /inside the project tree/);
+ assert.doesNotMatch(result.stdout, new RegExp(LOCAL_MARKER));
+});
+
+test("assertRuntimeRootLocation accepts a root outside the project and rejects one inside", (t) => {
+ const repo = makeRepoProject(t);
+ const outside = mkdtempSync(join(tmpdir(), "memwal-outside-"));
+ t.after(() => rmSync(outside, { recursive: true, force: true }));
+
+ assert.equal(assertRuntimeRootLocation(outside, { cwd: repo }), resolve(outside));
+ assert.throws(
+ () => assertRuntimeRootLocation(join(repo, ".memwal-runtime"), { cwd: repo }),
+ /inside the project tree/,
+ );
+ assert.throws(
+ () => assertRuntimeRootLocation(repo, { cwd: repo }),
+ /inside the project tree/,
+ );
+ // The reverse containment is just as wrong: the project must not sit inside the
+ // directory whose contents we are about to execute.
+ assert.throws(
+ () => assertRuntimeRootLocation(dirname(repo), { cwd: repo }),
+ /inside the project tree/,
+ );
+});
+
+test("a group-writable runtime root is refused", (t) => {
+ if (!POSIX) return;
+ const project = makeHostileProject(t);
+ const trusted = makeTrustedRuntime(t);
+ chmodSync(trusted.root, 0o777);
+
+ const result = runLauncher(["--print-entry"], {
+ cwd: project.dir,
+ env: { MEMWAL_MCP_RUNTIME_DIR: trusted.root },
+ });
+ chmodSync(trusted.root, 0o700);
+
+ assert.notEqual(result.status, 0, "a root anyone can write is not a trusted root");
+ assert.match(result.stderr, /group- or world-writable/);
+ assert.match(result.stderr, /refusing to fall back/);
+ assert.doesNotMatch(result.stdout, new RegExp(LOCAL_MARKER));
+});
+
+test("verifyTrustedDirectory rejects loose modes, symlinks and foreign owners", (t) => {
+ if (!POSIX) return;
+ const root = mkdtempSync(join(tmpdir(), "memwal-verify-"));
+ t.after(() => rmSync(root, { recursive: true, force: true }));
+
+ const good = join(root, "good");
+ mkdirSync(good, { mode: 0o700 });
+ assert.equal(verifyTrustedDirectory(good), true);
+ assert.equal(verifyTrustedDirectory(join(root, "absent")), false);
+
+ for (const mode of [0o770, 0o707, 0o777, 0o755 | 0o020]) {
+ chmodSync(good, mode);
+ assert.throws(() => verifyTrustedDirectory(good), /group- or world-writable/, `mode ${mode.toString(8)}`);
+ }
+ chmodSync(good, 0o700);
+
+ const link = join(root, "link");
+ symlinkSync(good, link);
+ assert.throws(() => verifyTrustedDirectory(link), /not a directory/);
+
+ const file = join(root, "file");
+ writeFileSync(file, "");
+ assert.throws(() => verifyTrustedDirectory(file), /not a directory/);
+
+ // A directory owned by somebody else. Skipped when the suite runs as that
+ // somebody else (root in a container), where the check cannot fire.
+ const foreign = "/usr";
+ if (process.getuid() !== 0 && lstatSync(foreign).uid !== process.getuid()) {
+ assert.throws(() => verifyTrustedDirectory(foreign), /owned by uid/);
+ }
+});
+
+test("the runtime root and ~/.memwal are created 0700, not at the umask default", (t) => {
+ if (!POSIX) return;
+ const home = mkdtempSync(join(tmpdir(), "memwal-home-mode-"));
+ t.after(() => rmSync(home, { recursive: true, force: true }));
+
+ const previousHome = process.env.HOME;
+ t.after(() => {
+ if (previousHome === undefined) delete process.env.HOME;
+ else process.env.HOME = previousHome;
+ });
+ process.env.HOME = home;
+
+ const root = prepareRuntimeRoot(join(home, ".memwal", "runtime"));
+ assert.equal(lstatSync(root).mode & 0o7777, 0o700);
+ // The launcher can run before the first login, so it is this call that creates
+ // ~/.memwal — the directory credentials.json later lands in.
+ assert.equal(lstatSync(join(home, ".memwal")).mode & 0o7777, 0o700);
+
+ // An override root is created 0700 as well, parents included.
+ const override = prepareRuntimeRoot(join(home, "elsewhere", "runtime"));
+ assert.equal(lstatSync(override).mode & 0o7777, 0o700);
+});
+
+test("the install never runs package scripts and never inherits the registry", () => {
+ const args = installArguments(`${MCP_PACKAGE_NAME}@${PIN}`, "/tmp/staging");
+ assert.ok(args.includes("--ignore-scripts"), `--ignore-scripts missing from ${args.join(" ")}`);
+ assert.ok(
+ args.includes(`--registry=${MCP_REGISTRY}`),
+ `an explicit registry is missing from ${args.join(" ")}`,
+ );
+ assert.equal(MCP_REGISTRY, "https://registry.npmjs.org/");
+
+ // npm puts the environment above every .npmrc, so a repository-supplied MCP env
+ // block would otherwise choose the registry no matter what cwd we install from.
+ const env = installEnvironment({
+ PATH: "/usr/bin",
+ HOME: "/home/user",
+ NPM_CONFIG_REGISTRY: "http://attacker.example/",
+ npm_config_registry: "http://attacker.example/",
+ npm_config_ca: "-----BEGIN CERTIFICATE-----",
+ npm_config_ignore_scripts: "false",
+ NODE_OPTIONS: "--require /tmp/evil.js",
+ });
+ assert.equal(env.PATH, "/usr/bin");
+ assert.equal(env.HOME, "/home/user");
+ assert.equal(env.npm_config_registry, MCP_REGISTRY);
+ assert.equal(env.npm_config_ignore_scripts, "true");
+ assert.equal(env.NPM_CONFIG_REGISTRY, undefined);
+ assert.equal(env.npm_config_ca, undefined);
+ assert.equal(env.NODE_OPTIONS, undefined);
+});
+
+test("npm is spawned as node , with no .cmd and no shell, on win32 too", () => {
+ const nodeExe = "C:\\Program Files\\nodejs\\node.exe";
+ const npmCli = "C:\\Program Files\\nodejs\\node_modules\\npm\\bin\\npm-cli.js";
+ const args = installArguments("pkg@1.0.0", "C:\\Users\\a b\\.memwal\\runtime\\staging");
+
+ const plan = npmSpawnPlan({ kind: "js", path: npmCli }, args, {
+ platform: "win32",
+ execPath: nodeExe,
+ });
+
+ assert.equal(plan.command, nodeExe);
+ assert.equal(plan.shell, false, "node.exe must never be spawned through a shell");
+ assert.deepEqual(plan.args, [npmCli, ...args]);
+ // Arguments go straight to the process, so they are NOT shell-quoted here.
+ assert.ok(plan.args.includes("C:\\Users\\a b\\.memwal\\runtime\\staging"));
+});
+
+test("the win32 npm.cmd fallback uses shell:true with cmd.exe-quoted arguments", () => {
+ // Since the CVE-2024-27980 fix (Node >= 18.20.2 / 20.12.2 / 21.7.3) spawning a
+ // .cmd without shell:true is EINVAL, which used to mean the server never started
+ // on Windows at all: ensureTrustedEntry -> install -> throw -> exit 1.
+ const npmCmd = "C:\\Program Files\\nodejs\\npm.cmd";
+ const staging = "C:\\Users\\a b\\.memwal\\runtime\\staging";
+ const plan = npmSpawnPlan({ kind: "bin", path: npmCmd }, installArguments("pkg@1.0.0", staging), {
+ platform: "win32",
+ });
+
+ assert.equal(plan.shell, true, "a .cmd target needs shell:true or spawnSync returns EINVAL");
+ assert.equal(plan.command, `"${npmCmd}"`, "the command path contains a space and must be quoted");
+ assert.ok(plan.args.includes(`"${staging}"`), "a path with a space must reach cmd.exe quoted");
+ assert.ok(plan.args.includes("--ignore-scripts"));
+ assert.ok(plan.args.includes(`--registry=${MCP_REGISTRY}`));
+ // Flags without metacharacters stay bare, so the command line stays readable.
+ assert.ok(plan.args.includes("install"));
+ assert.ok(plan.args.includes("pkg@1.0.0"));
+});
+
+test("quoteWindowsArgument quotes what cmd.exe would otherwise eat", () => {
+ assert.equal(quoteWindowsArgument("install"), "install");
+ assert.equal(quoteWindowsArgument("@scope/pkg@1.0.0"), "@scope/pkg@1.0.0");
+ assert.equal(quoteWindowsArgument("C:\\a b\\c"), '"C:\\a b\\c"');
+ assert.equal(quoteWindowsArgument("a&b"), '"a&b"');
+ assert.equal(quoteWindowsArgument('say "hi"'), '"say \\"hi\\""');
+});
+
+test("a posix npm binary is spawned directly, never through a shell", () => {
+ const plan = npmSpawnPlan({ kind: "bin", path: "/usr/local/bin/npm" }, ["install"], {
+ platform: "linux",
+ });
+ assert.equal(plan.command, "/usr/local/bin/npm");
+ assert.equal(plan.shell, false);
+ assert.deepEqual(plan.args, ["install"]);
+ assert.throws(() => npmSpawnPlan({ kind: "none", path: null }, ["install"]), /could not locate npm/);
+});
+
+test("resolveNpm returns an absolute path and skips PATH entries inside the project", (t) => {
+ const found = resolveNpm();
+ assert.notEqual(found.kind, "none", "npm must be resolvable in the test environment");
+ assert.ok(resolve(found.path) === found.path, `${found.path} is not absolute`);
+
+ const project = makeHostileProject(t);
+ const projectBin = join(project.dir, "node_modules", ".bin");
+ mkdirSync(projectBin, { recursive: true });
+ writeFileSync(join(projectBin, "npm"), "#!/bin/sh\necho PROJECT_NPM\n");
+ chmodSync(join(projectBin, "npm"), 0o755);
+
+ const elsewhere = mkdtempSync(join(tmpdir(), "memwal-npm-"));
+ t.after(() => rmSync(elsewhere, { recursive: true, force: true }));
+ writeFileSync(join(elsewhere, "npm"), "#!/bin/sh\necho OUTSIDE_NPM\n");
+ chmodSync(join(elsewhere, "npm"), 0o755);
+
+ // An execPath with no npm layout next to it forces the PATH scan.
+ const bare = mkdtempSync(join(tmpdir(), "memwal-bare-node-"));
+ t.after(() => rmSync(bare, { recursive: true, force: true }));
+ const resolved = resolveNpm({
+ execPath: join(bare, "node"),
+ platform: "linux",
+ env: { PATH: [".", "relative/bin", projectBin, elsewhere].join(":") },
+ cwd: project.dir,
+ });
+
+ assert.equal(resolved.kind, "bin");
+ assert.equal(canonicalPath(resolved.path), canonicalPath(join(elsewhere, "npm")));
+ assert.ok(!resolved.path.includes(project.dir), "npm must never come out of the project");
+});
+
+test("a client started in the home directory still gets the default root", (t) => {
+ // enclosingProjectRoot() must not treat $HOME as a project: ~/.memwal/runtime is
+ // inside the home directory by design, and refusing it would break every client
+ // that starts its servers there.
+ const home = mkdtempSync(join(tmpdir(), "memwal-home-cwd-"));
+ t.after(() => rmSync(home, { recursive: true, force: true }));
+ const target = join(home, ".memwal", "runtime", `memwal-mcp@${PIN}`);
+ mkdirSync(target, { recursive: true });
+ const entry = plantPackage(target, { version: PIN, marker: TRUSTED_MARKER });
+
+ const result = runLauncher(["--print-entry"], {
+ cwd: home,
+ env: { HOME: home, USERPROFILE: home, MEMWAL_MCP_RUNTIME_DIR: "" },
+ });
+
+ assert.equal(result.status, 0, result.stderr);
+ assert.equal(result.stdout.trim(), entry);
+});
+
+test("a subdirectory of the home directory is a project again", (t) => {
+ const home = mkdtempSync(join(tmpdir(), "memwal-home-sub-"));
+ t.after(() => rmSync(home, { recursive: true, force: true }));
+ const project = join(home, "code", "victim");
+ mkdirSync(project, { recursive: true });
+ writeFileSync(join(project, "package.json"), JSON.stringify({ name: "victim" }));
+ const planted = join(project, ".memwal-runtime");
+ mkdirSync(installDir(PIN, planted), { recursive: true });
+ plantPackage(installDir(PIN, planted), { version: PIN, marker: LOCAL_MARKER });
+
+ const result = runLauncher(["--print-entry"], {
+ cwd: project,
+ env: { HOME: home, USERPROFILE: home, MEMWAL_MCP_RUNTIME_DIR: planted },
+ });
+
+ assert.notEqual(result.status, 0);
+ assert.match(result.stderr, /inside the project tree/);
+ assert.doesNotMatch(result.stdout, new RegExp(LOCAL_MARKER));
+});
From 50a61cddc1bbd0285e1d77a6be38b682a606ff20 Mon Sep 17 00:00:00 2001
From: Harry Phan
Date: Fri, 18 Sep 2026 13:06:10 +0700
Subject: [PATCH 076/132] fix(mcp): migrate an existing Codex npx registration
instead of skipping it (WALM-640)
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit
`ensureMcpRegistered()` began with `if (content.includes("[mcp_servers.memwal]"))
return false;`, so the fix only ever reached *new* installations. Everyone who had
run the installer before kept
[mcp_servers.memwal]
command = "npx"
args = ["-y", "@mysten-incubation/memwal-mcp@"]
in ~/.codex/config.toml for ever, and re-running the installer printed "already
present" — which reads as success while Codex went on resolving the package name
against the project directory. That is the exact population the ticket was filed
for.
An existing block is now rewritten when it does not already run the launcher. The
rewrite is line-based rather than parse-and-reserialize, because config.toml is the
user's file: comments, key order, other servers and any unrelated keys in the block
(`env`, `startup_timeout_ms`, …) survive, and only `command` and `args` change.
Multi-line arrays and inline tables are replaced whole rather than line by line. The
run says what it migrated, from what, which keys it kept, and that Codex needs a
restart; "unchanged" is reported as "already runs the launcher" so it can no longer
be confused with "already present".
The logic lives in scripts/lib/codex-config.mjs as a pure string-to-string function,
so it is testable without a home directory — and so the release verifier can run it.
verify-manual-sdk-release.mjs previously only grepped the installer source for
`command = "npx"`, which an installer that skips existing blocks passes trivially;
it now feeds a legacy block through the migration and fails unless npx is replaced,
the launcher path is written, and unrelated keys survive.
8 new tests in test/codex-registration.test.mjs, three of which run the real
installer against a sandboxed HOME.
---
.../plugin/scripts/install_codex_hooks.mjs | 65 ++++--
.../mcp/plugin/scripts/lib/codex-config.mjs | 201 ++++++++++++++++++
packages/mcp/test/codex-registration.test.mjs | 194 +++++++++++++++++
scripts/verify-manual-sdk-release.mjs | 37 ++++
4 files changed, 481 insertions(+), 16 deletions(-)
create mode 100644 packages/mcp/plugin/scripts/lib/codex-config.mjs
create mode 100644 packages/mcp/test/codex-registration.test.mjs
diff --git a/packages/mcp/plugin/scripts/install_codex_hooks.mjs b/packages/mcp/plugin/scripts/install_codex_hooks.mjs
index 6d4de785b..268e89e92 100644
--- a/packages/mcp/plugin/scripts/install_codex_hooks.mjs
+++ b/packages/mcp/plugin/scripts/install_codex_hooks.mjs
@@ -13,7 +13,10 @@
* entries into ~/.codex/hooks.json.
*
* Re-running is idempotent: entries this installer owns (identified by our
- * hook script filenames) are removed before fresh entries are added.
+ * hook script filenames) are removed before fresh entries are added, and an
+ * existing [mcp_servers.memwal] block is migrated to the launcher when it still
+ * points at something else (WALM-640) instead of being reported as "already
+ * present".
*
* Usage:
* node install_codex_hooks.mjs # install or update
@@ -29,6 +32,8 @@ import { homedir } from "node:os";
import { join, dirname } from "node:path";
import { fileURLToPath } from "node:url";
+import { planMcpRegistration } from "./lib/codex-config.mjs";
+
const SCRIPT_DIR = dirname(fileURLToPath(import.meta.url));
const PLUGIN_ROOT = dirname(SCRIPT_DIR);
@@ -94,7 +99,7 @@ function writeHooks(config) {
}
/**
- * Append [mcp_servers.memwal] to config.toml if it isn't registered yet.
+ * Register — or migrate — [mcp_servers.memwal] in config.toml.
*
* Registers the plugin's launcher by absolute path rather than
* `npx @mysten-incubation/memwal-mcp@`. npx resolves that name against the
@@ -102,18 +107,50 @@ function writeHooks(config) {
* the same name — claiming the pinned version — was run instead of ours (WALM-640).
* The launcher installs the pinned version under ~/.memwal/runtime and runs that
* absolute entry point, so no project directory takes part in the resolution.
+ *
+ * An existing block is REWRITTEN, not skipped. Everyone who ran this installer before
+ * WALM-640 has the `npx` form on disk, and a "already present" message on re-run would
+ * leave exactly the resolution this fix exists to remove in place for exactly the
+ * people who already installed. Unrelated keys in the block (`env`, timeouts, …) are
+ * preserved — see lib/codex-config.mjs.
*/
function ensureMcpRegistered() {
mkdirSync(CODEX_DIR, { recursive: true });
- let content = existsSync(CONFIG_FILE) ? readFileSync(CONFIG_FILE, "utf8") : "";
- if (content.includes("[mcp_servers.memwal]")) return false;
+ const content = existsSync(CONFIG_FILE) ? readFileSync(CONFIG_FILE, "utf8") : "";
const launcher = join(SCRIPT_DIR, "launch_mcp.mjs");
- const block =
- "\n[mcp_servers.memwal]\n" +
- 'command = "node"\n' +
- `args = [${JSON.stringify(launcher)}]\n`;
- writeFileSync(CONFIG_FILE, (content.trimEnd() + "\n" + block).trimStart());
- return true;
+ const plan = planMcpRegistration(content, launcher);
+ if (plan.action !== "unchanged") writeFileSync(CONFIG_FILE, plan.content);
+ return plan;
+}
+
+function reportMcpRegistration(plan) {
+ if (plan.action === "added") {
+ console.log(`Registered [mcp_servers.memwal] in ${CONFIG_FILE}`);
+ return;
+ }
+ if (plan.action === "unchanged") {
+ console.log(`[mcp_servers.memwal] in ${CONFIG_FILE} already runs the launcher`);
+ return;
+ }
+ console.log(`Migrated [mcp_servers.memwal] in ${CONFIG_FILE}:`);
+ console.log(` was: command = ${plan.previous.command ?? "(absent)"}`);
+ console.log(` args = ${plan.previous.args ?? "(absent)"}`);
+ console.log(` now: command = "node"`);
+ console.log(` args = [${JSON.stringify(launcherPath())}]`);
+ if (plan.previous.command?.includes("npx")) {
+ console.log(
+ " (the old command resolved the package name against the directory Codex " +
+ "was started in — WALM-640)"
+ );
+ }
+ if (plan.preserved.length > 0) {
+ console.log(` kept your other keys: ${plan.preserved.join(", ")}`);
+ }
+ console.log(" Restart Codex for the change to take effect.");
+}
+
+function launcherPath() {
+ return join(SCRIPT_DIR, "launch_mcp.mjs");
}
function featureFlagEnabled() {
@@ -159,16 +196,12 @@ function main() {
config = mergeTemplate(config, template);
writeHooks(config);
- const mcpAdded = ensureMcpRegistered();
+ const mcpPlan = ensureMcpRegistered();
console.log(`Installed MemWal hooks into ${HOOKS_FILE}`);
console.log(`Plugin path: ${PLUGIN_ROOT}`);
console.log("Events: SessionStart, UserPromptSubmit, PostToolUse");
- console.log(
- mcpAdded
- ? `Registered [mcp_servers.memwal] in ${CONFIG_FILE}`
- : `[mcp_servers.memwal] already present in ${CONFIG_FILE}`
- );
+ reportMcpRegistration(mcpPlan);
if (!featureFlagEnabled()) printFeatureFlagHint();
return 0;
diff --git a/packages/mcp/plugin/scripts/lib/codex-config.mjs b/packages/mcp/plugin/scripts/lib/codex-config.mjs
new file mode 100644
index 000000000..24cef4109
--- /dev/null
+++ b/packages/mcp/plugin/scripts/lib/codex-config.mjs
@@ -0,0 +1,201 @@
+/**
+ * Rewriting the `[mcp_servers.memwal]` block in ~/.codex/config.toml (WALM-640).
+ *
+ * The first version of the WALM-640 fix only changed what a *fresh* installation
+ * writes. `ensureMcpRegistered()` returned early on `content.includes(...)`, so every
+ * user who had already run the installer kept
+ *
+ * [mcp_servers.memwal]
+ * command = "npx"
+ * args = ["-y", "@mysten-incubation/memwal-mcp@"]
+ *
+ * for ever, while re-running the installer printed "already present" — which reads
+ * as success. Those users are exactly the population the ticket was filed for: Codex
+ * kept resolving the package name against the project directory.
+ *
+ * So the installer migrates the block instead of skipping it. The rewrite is
+ * deliberately line-based rather than a parse-and-reserialize: config.toml is the
+ * user's file, and comments, key order and unrelated keys (`env`, timeouts, an
+ * `enabled` flag, whatever a future Codex adds) have to survive untouched. Only
+ * `command` and `args` are replaced.
+ *
+ * Pure string in, string out: no fs, no process state, so the installer's behaviour
+ * is testable without a home directory.
+ */
+
+export const MEMWAL_SECTION = "mcp_servers.memwal";
+const SECTION_HEADER = `[${MEMWAL_SECTION}]`;
+
+/** Strip string literals so bracket counting is not fooled by values. */
+function withoutStrings(line) {
+ return line
+ .replace(/'''[\s\S]*?'''/g, "''")
+ .replace(/"""[\s\S]*?"""/g, '""')
+ .replace(/'[^']*'/g, "''")
+ .replace(/"(?:[^"\\]|\\.)*"/g, '""');
+}
+
+function stripComment(line) {
+ const bare = withoutStrings(line);
+ const hash = bare.indexOf("#");
+ return hash === -1 ? line : line.slice(0, hash);
+}
+
+function bracketDelta(line) {
+ const bare = stripComment(line);
+ let delta = 0;
+ for (const char of bare) {
+ if (char === "[" || char === "{") delta += 1;
+ if (char === "]" || char === "}") delta -= 1;
+ }
+ return delta;
+}
+
+function isSectionHeader(line) {
+ const trimmed = stripComment(line).trim();
+ return /^\[[^\]]+\]$/.test(trimmed) || /^\[\[[^\]]+\]\]$/.test(trimmed);
+}
+
+function sectionName(line) {
+ const trimmed = stripComment(line).trim();
+ const match = /^\[\[?([^\]]+)\]\]?$/.exec(trimmed);
+ return match ? match[1].trim() : null;
+}
+
+const KEY_PATTERN = /^\s*("[^"]*"|'[^']*'|[A-Za-z0-9_-]+)\s*=/;
+
+function keyOf(line) {
+ const match = KEY_PATTERN.exec(stripComment(line));
+ if (!match) return null;
+ return match[1].replace(/^["']|["']$/g, "");
+}
+
+/**
+ * Split the block that starts at `headerIndex` into entries. An entry is one
+ * key/value pair (possibly spanning lines, for a multi-line array or table) or a
+ * run of comment/blank lines carried along verbatim.
+ */
+function readEntries(lines, headerIndex) {
+ const entries = [];
+ let index = headerIndex + 1;
+ while (index < lines.length) {
+ if (isSectionHeader(lines[index])) break;
+ const key = keyOf(lines[index]);
+ if (key === null) {
+ entries.push({ key: null, lines: [lines[index]] });
+ index += 1;
+ continue;
+ }
+ const collected = [lines[index]];
+ let depth = bracketDelta(lines[index]);
+ index += 1;
+ while (depth > 0 && index < lines.length) {
+ collected.push(lines[index]);
+ depth += bracketDelta(lines[index]);
+ index += 1;
+ }
+ entries.push({ key, lines: collected });
+ }
+ return { entries, end: index };
+}
+
+function rawValue(entry) {
+ const joined = entry.lines.join("\n");
+ const equals = joined.indexOf("=");
+ return equals === -1 ? "" : joined.slice(equals + 1).trim();
+}
+
+/** TOML's basic strings and arrays of them are JSON; anything else we treat as "not ours". */
+function asJson(value) {
+ try {
+ return JSON.parse(value.replace(/,(\s*[\]}])/g, "$1"));
+ } catch {
+ return undefined;
+ }
+}
+
+function desiredLines(launcher) {
+ return {
+ command: 'command = "node"',
+ args: `args = [${JSON.stringify(launcher)}]`,
+ };
+}
+
+/**
+ * Work out what ~/.codex/config.toml should contain.
+ *
+ * Returns `{ content, action, previous, preserved }` where action is one of:
+ * - "added": no `[mcp_servers.memwal]` block existed; one was appended.
+ * - "migrated": a block existed and did not launch our launcher; command/args
+ * were rewritten and every other key in the block kept.
+ * - "unchanged": the block already runs `node `.
+ */
+export function planMcpRegistration(content, launcher) {
+ const text = String(content ?? "");
+ const lines = text.split("\n");
+ const headerIndex = lines.findIndex((line) => sectionName(line) === MEMWAL_SECTION);
+ const desired = desiredLines(launcher);
+
+ if (headerIndex === -1) {
+ const block = `\n${SECTION_HEADER}\n${desired.command}\n${desired.args}\n`;
+ return {
+ content: (text.trimEnd() + "\n" + block).trimStart(),
+ action: "added",
+ previous: null,
+ preserved: [],
+ };
+ }
+
+ const { entries, end } = readEntries(lines, headerIndex);
+ const commandEntry = entries.find((entry) => entry.key === "command");
+ const argsEntry = entries.find((entry) => entry.key === "args");
+ const previous = {
+ command: commandEntry ? rawValue(commandEntry) : null,
+ args: argsEntry ? rawValue(argsEntry) : null,
+ };
+
+ const currentCommand = commandEntry ? asJson(rawValue(commandEntry)) : undefined;
+ const currentArgs = argsEntry ? asJson(rawValue(argsEntry)) : undefined;
+ const alreadyCorrect =
+ currentCommand === "node" &&
+ Array.isArray(currentArgs) &&
+ currentArgs.length === 1 &&
+ currentArgs[0] === launcher;
+ if (alreadyCorrect) {
+ return { content: text, action: "unchanged", previous, preserved: [] };
+ }
+
+ const rebuilt = [];
+ let wroteCommand = false;
+ let wroteArgs = false;
+ for (const entry of entries) {
+ if (entry.key === "command") {
+ rebuilt.push(desired.command);
+ wroteCommand = true;
+ continue;
+ }
+ if (entry.key === "args") {
+ rebuilt.push(desired.args);
+ wroteArgs = true;
+ continue;
+ }
+ rebuilt.push(...entry.lines);
+ }
+ // A block that never had command/args (or had only one of them) still has to end
+ // up launching the launcher; put the missing keys first, where a reader expects.
+ const missing = [];
+ if (!wroteArgs) missing.unshift(desired.args);
+ if (!wroteCommand) missing.unshift(desired.command);
+
+ const preserved = entries
+ .filter((entry) => entry.key !== null && entry.key !== "command" && entry.key !== "args")
+ .map((entry) => entry.key);
+
+ const next = [
+ ...lines.slice(0, headerIndex + 1),
+ ...missing,
+ ...rebuilt,
+ ...lines.slice(end),
+ ];
+ return { content: next.join("\n"), action: "migrated", previous, preserved };
+}
diff --git a/packages/mcp/test/codex-registration.test.mjs b/packages/mcp/test/codex-registration.test.mjs
new file mode 100644
index 000000000..594962313
--- /dev/null
+++ b/packages/mcp/test/codex-registration.test.mjs
@@ -0,0 +1,194 @@
+/**
+ * The Codex fallback installer must MIGRATE an existing `[mcp_servers.memwal]`
+ * block, not skip it (WALM-640).
+ *
+ * The first cut of the fix only changed what a fresh install writes: it returned
+ * early on `content.includes("[mcp_servers.memwal]")`. Everyone who had already run
+ * the installer therefore kept `command = "npx"` / `args = ["-y", "…@"]` in
+ * ~/.codex/config.toml for ever, while re-running the installer printed "already
+ * present" — which reads as success. That is the exact population the ticket was
+ * filed for, so it is the case these tests cover: the block is rewritten, whatever
+ * else the user put in it survives, and the run says what it changed.
+ */
+import { test } from "node:test";
+import assert from "node:assert/strict";
+import { spawnSync } from "node:child_process";
+import { mkdirSync, mkdtempSync, readFileSync, rmSync, writeFileSync } from "node:fs";
+import { tmpdir } from "node:os";
+import { dirname, join, resolve } from "node:path";
+import { fileURLToPath } from "node:url";
+
+import { planMcpRegistration } from "../plugin/scripts/lib/codex-config.mjs";
+
+const __dirname = dirname(fileURLToPath(import.meta.url));
+const PLUGIN_DIR = resolve(__dirname, "../plugin");
+const INSTALLER = join(PLUGIN_DIR, "scripts", "install_codex_hooks.mjs");
+const LAUNCHER = join(PLUGIN_DIR, "scripts", "launch_mcp.mjs");
+const PIN = JSON.parse(readFileSync(join(PLUGIN_DIR, "plugin.json"), "utf8")).version;
+
+/** Exactly what the pre-WALM-640 installer left on disk. */
+const LEGACY_BLOCK = [
+ "[features]",
+ "codex_hooks = true",
+ "",
+ "# MemWal memory server",
+ "[mcp_servers.memwal]",
+ 'command = "npx"',
+ `args = ["-y", "@mysten-incubation/memwal-mcp@${PIN}"]`,
+ 'env = { MEMWAL_NAMESPACE = "work" }',
+ "startup_timeout_ms = 30000",
+ "",
+ "[mcp_servers.something_else]",
+ 'command = "other-server"',
+ "",
+].join("\n");
+
+function makeHome(t, configToml) {
+ const home = mkdtempSync(join(tmpdir(), "memwal-codex-home-"));
+ t.after(() => rmSync(home, { recursive: true, force: true }));
+ mkdirSync(join(home, ".codex"), { recursive: true });
+ if (configToml !== undefined) writeFileSync(join(home, ".codex", "config.toml"), configToml);
+ return home;
+}
+
+function runInstaller(home) {
+ const result = spawnSync(process.execPath, [INSTALLER], {
+ encoding: "utf8",
+ env: { ...process.env, HOME: home, USERPROFILE: home },
+ });
+ return {
+ ...result,
+ config: readFileSync(join(home, ".codex", "config.toml"), "utf8"),
+ };
+}
+
+test("a legacy npx registration is migrated when the installer is re-run", (t) => {
+ const home = makeHome(t, LEGACY_BLOCK);
+ const result = runInstaller(home);
+
+ assert.equal(result.status, 0, result.stderr);
+ assert.doesNotMatch(result.config, /command\s*=\s*"npx"/, "npx must be gone from config.toml");
+ assert.doesNotMatch(result.config, /memwal-mcp@/, "the npx spec must be gone too");
+ assert.match(result.config, /command = "node"/);
+ assert.ok(
+ result.config.includes(JSON.stringify(LAUNCHER)),
+ `config.toml does not point at ${LAUNCHER}:\n${result.config}`,
+ );
+
+ // The user's own keys survive, inside and outside our block.
+ assert.match(result.config, /env = \{ MEMWAL_NAMESPACE = "work" \}/);
+ assert.match(result.config, /startup_timeout_ms = 30000/);
+ assert.match(result.config, /# MemWal memory server/);
+ assert.match(result.config, /\[mcp_servers\.something_else\]/);
+ assert.match(result.config, /command = "other-server"/);
+ assert.match(result.config, /codex_hooks = true/);
+
+ // And it says so, rather than "already present".
+ assert.match(result.stdout, /Migrated \[mcp_servers\.memwal\]/);
+ assert.doesNotMatch(result.stdout, /already present/);
+ assert.match(result.stdout, /kept your other keys: env, startup_timeout_ms/);
+});
+
+test("a second run after the migration reports no change and rewrites nothing", (t) => {
+ const home = makeHome(t, LEGACY_BLOCK);
+ const first = runInstaller(home);
+ const second = runInstaller(home);
+
+ assert.equal(second.status, 0, second.stderr);
+ assert.equal(second.config, first.config, "a settled config.toml must not keep churning");
+ assert.match(second.stdout, /already runs the launcher/);
+});
+
+test("a fresh config.toml gets the launcher registration", (t) => {
+ const home = makeHome(t, "");
+ const result = runInstaller(home);
+
+ assert.equal(result.status, 0, result.stderr);
+ assert.match(result.stdout, /Registered \[mcp_servers\.memwal\]/);
+ assert.match(result.config, /\[mcp_servers\.memwal\]/);
+ assert.match(result.config, /command = "node"/);
+ assert.ok(result.config.includes(JSON.stringify(LAUNCHER)));
+});
+
+test("planMcpRegistration keeps every unrelated key and only rewrites command/args", () => {
+ const plan = planMcpRegistration(LEGACY_BLOCK, "/abs/launch_mcp.mjs");
+
+ assert.equal(plan.action, "migrated");
+ assert.deepEqual(plan.preserved, ["env", "startup_timeout_ms"]);
+ assert.equal(plan.previous.command, '"npx"');
+ assert.match(plan.previous.args, /memwal-mcp@/);
+ assert.equal(
+ plan.content,
+ [
+ "[features]",
+ "codex_hooks = true",
+ "",
+ "# MemWal memory server",
+ "[mcp_servers.memwal]",
+ 'command = "node"',
+ 'args = ["/abs/launch_mcp.mjs"]',
+ 'env = { MEMWAL_NAMESPACE = "work" }',
+ "startup_timeout_ms = 30000",
+ "",
+ "[mcp_servers.something_else]",
+ 'command = "other-server"',
+ "",
+ ].join("\n"),
+ );
+});
+
+test("a multi-line args array is replaced whole, not line by line", () => {
+ const content = [
+ "[mcp_servers.memwal]",
+ 'command = "npx"',
+ "args = [",
+ ' "-y",',
+ ' "@mysten-incubation/memwal-mcp@0.0.14",',
+ "]",
+ 'env = { A = "1" }',
+ "",
+ "[other]",
+ "x = 1",
+ "",
+ ].join("\n");
+
+ const plan = planMcpRegistration(content, "/abs/launch_mcp.mjs");
+ assert.equal(plan.action, "migrated");
+ assert.doesNotMatch(plan.content, /-y/);
+ assert.doesNotMatch(plan.content, /memwal-mcp@0\.0\.14/);
+ assert.match(plan.content, /args = \["\/abs\/launch_mcp\.mjs"\]/);
+ assert.match(plan.content, /env = \{ A = "1" \}/);
+ assert.match(plan.content, /\[other\]\nx = 1/);
+});
+
+test("a block with the keys missing still ends up launching the launcher", () => {
+ const plan = planMcpRegistration(
+ ['[mcp_servers.memwal]', 'env = { A = "1" }', ""].join("\n"),
+ "/abs/launch_mcp.mjs",
+ );
+ assert.equal(plan.action, "migrated");
+ assert.match(plan.content, /command = "node"/);
+ assert.match(plan.content, /args = \["\/abs\/launch_mcp\.mjs"\]/);
+ assert.match(plan.content, /env = \{ A = "1" \}/);
+});
+
+test("a block pointing at a different launcher path is re-pointed", () => {
+ const stale = [
+ "[mcp_servers.memwal]",
+ 'command = "node"',
+ 'args = ["/old/plugin/scripts/launch_mcp.mjs"]',
+ "",
+ ].join("\n");
+ const plan = planMcpRegistration(stale, "/new/plugin/scripts/launch_mcp.mjs");
+ assert.equal(plan.action, "migrated");
+ assert.match(plan.content, /args = \["\/new\/plugin\/scripts\/launch_mcp\.mjs"\]/);
+ assert.doesNotMatch(plan.content, /\/old\/plugin/);
+});
+
+test("the section name is matched exactly, not by substring", () => {
+ const other = ['[mcp_servers.memwal_other]', 'command = "npx"', ""].join("\n");
+ const plan = planMcpRegistration(other, "/abs/launch_mcp.mjs");
+ assert.equal(plan.action, "added", "an unrelated server must not be rewritten");
+ assert.match(plan.content, /\[mcp_servers\.memwal_other\]\ncommand = "npx"/);
+ assert.match(plan.content, /\[mcp_servers\.memwal\]\ncommand = "node"/);
+});
diff --git a/scripts/verify-manual-sdk-release.mjs b/scripts/verify-manual-sdk-release.mjs
index 8d6bd3306..f9e2c9a6f 100644
--- a/scripts/verify-manual-sdk-release.mjs
+++ b/scripts/verify-manual-sdk-release.mjs
@@ -102,7 +102,44 @@ const launcherPath = `packages/mcp/plugin/${LAUNCHER}`;
if (!existsSync(launcherPath)) {
throw new Error(`${launcherPath}: missing, but every launch site points at it`);
}
+
+// Registering the launcher for *new* installations is only half of it: every user who
+// ran the installer before WALM-640 has `command = "npx"` in ~/.codex/config.toml, and
+// an installer that skips an existing block leaves them on the vulnerable resolution
+// for ever. Exercise the migration rather than grepping for its absence.
+const codexConfigPath = "packages/mcp/plugin/scripts/lib/codex-config.mjs";
+if (!existsSync(codexConfigPath)) {
+ throw new Error(`${codexConfigPath}: missing, but ${installerPath} migrates through it`);
+}
+const { planMcpRegistration } = await import(`../${codexConfigPath}`);
+const legacyConfig = [
+ "[features]",
+ "codex_hooks = true",
+ "",
+ "[mcp_servers.memwal]",
+ 'command = "npx"',
+ `args = ["-y", "@mysten-incubation/memwal-mcp@${mcpVersion}"]`,
+ 'env = { MEMWAL_NAMESPACE = "work" }',
+ "",
+ "[mcp_servers.other]",
+ 'command = "other"',
+ "",
+].join("\n");
+const migrated = planMcpRegistration(legacyConfig, "/abs/plugin/scripts/launch_mcp.mjs");
+if (migrated.action !== "migrated") {
+ throw new Error(
+ `${codexConfigPath}: an existing npx [mcp_servers.memwal] block must be migrated, ` +
+ `received action "${migrated.action}" (WALM-640)`,
+ );
+}
+if (/command\s*=\s*"npx"/.test(migrated.content) || !migrated.content.includes(LAUNCHER)) {
+ throw new Error(`${codexConfigPath}: migration did not replace npx with ${LAUNCHER}`);
+}
+if (!migrated.content.includes("MEMWAL_NAMESPACE") || !migrated.content.includes("[mcp_servers.other]")) {
+ throw new Error(`${codexConfigPath}: migration dropped keys it does not own`);
+}
console.log(`MCP package ${mcpVersion}: every launch site runs ${LAUNCHER} by absolute path`);
+console.log(`MCP package ${mcpVersion}: an existing npx Codex registration is migrated, not skipped`);
function readVersion(content, kind) {
if (kind === "version") return JSON.parse(content).version;
From 9473826f4a8a5764bd9ca43bb9ae7bd5430def79 Mon Sep 17 00:00:00 2001
From: Harry Phan
Date: Fri, 18 Sep 2026 13:06:15 +0700
Subject: [PATCH 077/132] docs(mcp): correct what the trusted launcher actually
guarantees (WALM-640)
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit
The README said the launcher "never consults … your project's `.npmrc`", which was
true and beside the point: npm ranks the environment above every `.npmrc`, so the
relevant guarantee is the scrubbed environment and the pinned registry, not the cwd.
It also left the reader thinking an absolute `MEMWAL_MCP_RUNTIME_DIR` was enough.
States the actual rules for the runtime directory (outside the project tree, a real
directory, owned by you, not group- or world-writable, created 0700), what the
install does (`--ignore-scripts`, pinned registry, scrubbed `npm_config_*` /
`NODE_OPTIONS`, npm run as `node npm-cli.js`), and — explicitly — what none of that
establishes: the provenance and write-protection of the code, not the integrity of
the bytes the registry served. Points private-registry users at the manual recipe,
which the launcher accepts as-is, and adds the `chmod 700` and `--ignore-scripts`
that recipe needs to match what the plugin does.
---
packages/mcp/README.md | 38 +++++++++++++++++++++++++++++++-------
1 file changed, 31 insertions(+), 7 deletions(-)
diff --git a/packages/mcp/README.md b/packages/mcp/README.md
index fb21f44e7..ca8357601 100644
--- a/packages/mcp/README.md
+++ b/packages/mcp/README.md
@@ -52,12 +52,34 @@ runs that absolute entry point with the current `node` binary:
~/.memwal/runtime/memwal-mcp@/node_modules/@mysten-incubation/memwal-mcp/dist/bin/memwal-mcp.js
```
-It never consults your project's `node_modules`, a `PATH`-relative bin shim, or
-your project's `.npmrc`, and it fails rather than falling back to the package name
-if the pinned version cannot be installed. Everything after the script path is
-forwarded to the server unchanged, so flags such as `--namespace work` or
-`--relayer ` work exactly as they do above. Set `MEMWAL_MCP_RUNTIME_DIR` (an
-absolute path) to move the trusted directory elsewhere.
+It never consults your project's `node_modules` or a `PATH`-relative bin shim, and
+it fails rather than falling back to the package name if the pinned version cannot
+be installed. Everything after the script path is forwarded to the server
+unchanged, so flags such as `--namespace work` or `--relayer ` work exactly as
+they do above.
+
+The runtime directory is trusted because of what it is, not how it is spelled:
+
+- It must be outside the project the client started in. An absolute path is not
+ enough on its own — a client expands `${workspaceFolder}` to an absolute path
+ inside the repository, and a repository can commit a whole fake install there.
+- It must be a real directory (not a symlink), owned by you, and not group- or
+ world-writable. The launcher creates it with mode `0700` and refuses to run code
+ out of it otherwise. If you see a refusal, `chmod 700 ~/.memwal ~/.memwal/runtime`.
+- `MEMWAL_MCP_RUNTIME_DIR` moves it, subject to exactly the same rules.
+
+The one-off install is run with `--ignore-scripts` (so no `preinstall` or
+`postinstall` from the package or its dependencies executes), against an explicitly
+pinned public registry, with the `npm_config_*` and `NODE_OPTIONS` environment
+scrubbed for that spawn — npm ranks environment variables above every `.npmrc`, so
+a client `env` block would otherwise choose the registry. npm itself is run as
+`node ` resolved from the running node binary where that layout exists.
+
+What that does **not** give you is an integrity check of the package contents: it
+establishes where the code came from and that nobody else can write it, not that
+the registry served the bytes a reviewer read. If you install from a private
+registry, pre-populate the directory yourself (below) — the launcher then finds the
+install and never runs npm at all.
If you configure MemWal without the plugin and want the same property, install the
version you intend to run into a directory outside any project and point your
@@ -65,7 +87,9 @@ client at its absolute path:
```sh
mkdir -p ~/.memwal/runtime/memwal-mcp@0.0.14
-npm install --prefix ~/.memwal/runtime/memwal-mcp@0.0.14 @mysten-incubation/memwal-mcp@0.0.14
+chmod 700 ~/.memwal ~/.memwal/runtime
+npm install --ignore-scripts --prefix ~/.memwal/runtime/memwal-mcp@0.0.14 \
+ @mysten-incubation/memwal-mcp@0.0.14
```
```json
From f3ac9870ca8a72fb9b698a1e3df93c6f685af020 Mon Sep 17 00:00:00 2001
From: Harry Phan
Date: Fri, 18 Sep 2026 13:06:19 +0700
Subject: [PATCH 078/132] fix(mcp): close two holes in the repo-credentials
approval gate
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit
Both are in the gate the previous commit added, and both put the thing the gate
protects back inside the repository.
MEMWAL_CREDS_DIR was a way around the gate rather than through it. It was read
as `process.env.MEMWAL_CREDS_DIR ?? join(homedir(), ".memwal")`, so `??` kept an
empty string, and nothing anywhere checked that the value was absolute. The
override branch of `resolveCreds()` also returns BEFORE any approval lookup. So
`MEMWAL_CREDS_DIR=""` made the approval store the bare relative name
`project-approvals.json`, resolved against the working directory — a repository
that committed that one file at its root approved its own credentials — and
`MEMWAL_CREDS_DIR=.memwal` put the credentials and the approval record inside
the repository with no approval asked for at all. Neither has to be typed by the
user: an MCP client reads `.cursor/mcp.json`, `.vscode/mcp.json` or
`.claude/settings.json` out of the checkout and hands this process the `env`
block it finds there.
The rule now lives in one place, `credsDirOverride()`, and `trustedStateDir()`,
`resolveCreds()` and `approveProjectCreds()` all go through it. The override is
honoured only when it is a non-empty ABSOLUTE path outside the current project;
anything else throws `UntrustedCredsDirError`, which `bin/memwal-mcp` renders as
`[memwal-mcp] fatal: …` and exit 1. Refusing loudly rather than falling back to
the global directory matches the launcher's treatment of a relative runtime
directory, and a silent fallback would hide exactly the misconfigured or planted
client config the check exists for. An in-project absolute path is refused too,
because editors expand `${workspaceFolder}` in those same committed files. An
empty value means unset, which is what every other reader already treats it as.
Tests and CI pointing at an absolute temp directory are unaffected, and the doc
comment now records that the override is only as trustworthy as whoever set the
environment.
The login write-ahead record no longer goes into the repository either.
`pendingLoginPath()` sat beside whichever credentials file resolution chose, so
once a project file was approved every sign-in wrote a plaintext 64-hex Ed25519
delegate seed to `/.memwal/login-pending.json` — in a directory an attacker
who planted the credentials file had already created, so tracked rather than
ignored, and staged by `git add -A`. Nothing about that record needs to be in the
repo: it moves to `~/.memwal/login-pending/`, keyed by a hash of the project's
credentials path so a sign-in stays reclaimable only by the project that started
it. The global and override paths are byte-for-byte where they were.
`credentials.json` itself cannot move — it is the file the user approved — so the
silence around it goes instead: `approve-project` (and a re-run of it) and the
pre-sign-in warning now say plainly that signing in from this project writes a
delegate PRIVATE KEY into the repository in plaintext, and suggest adding
`.memwal/` to .gitignore.
New `packages/mcp/test/creds-dir-trust.test.mjs`, 20 tests; 14 of them fail
against the build before this commit.
WALM-639
---
docs/mcp/reference.md | 4 +-
docs/reference/environment-variables.md | 2 +-
packages/mcp/CHANGELOG.md | 3 +
packages/mcp/README.md | 18 +-
packages/mcp/src/auth.ts | 238 +++++++++-
packages/mcp/src/index.ts | 17 +
packages/mcp/test/creds-dir-trust.test.mjs | 505 +++++++++++++++++++++
7 files changed, 762 insertions(+), 25 deletions(-)
create mode 100644 packages/mcp/test/creds-dir-trust.test.mjs
diff --git a/docs/mcp/reference.md b/docs/mcp/reference.md
index 7599d2d01..57535c34e 100644
--- a/docs/mcp/reference.md
+++ b/docs/mcp/reference.md
@@ -148,7 +148,9 @@ memwal-mcp approve-project
Until then the global credentials are used and one stderr line names the file that was skipped, the account and relayer it wanted, and this command. An approval covers one exact project path, account, delegate key and relayer: change any of them and it has to be approved again. The record is kept in `~/.memwal/project-approvals.json`, outside the repository, so a repository cannot carry its own approval. `memwal-mcp revoke-project` withdraws it.
-`MEMWAL_CREDS_DIR` overrides project resolution entirely and needs no approval — it can only come from your own environment, never from a checkout.
+Approving decides which file is written as well as which is read. From then on a sign-in from that project stores a delegate private key, in plain text, in `.memwal/credentials.json` inside the repository — `approve-project` and the pre-sign-in warning both say so. Add `.memwal/` to `.gitignore`. The login write-ahead record is kept in `~/.memwal/login-pending/` and never enters the repository.
+
+`MEMWAL_CREDS_DIR` overrides project resolution entirely and needs no approval. Because it skips the gate it must be an **absolute path outside the current project**, and anything else is refused with an error rather than ignored: a relative value resolves against the working directory, which would put the credentials and the approval record inside the repository, and an MCP client hands this process the `env` block it reads from `.cursor/mcp.json`, `.vscode/mcp.json` or `.claude/settings.json` in the checkout — where editors expand `${workspaceFolder}`, so an in-project absolute path is not evidence you set it. An empty value means unset.
### Working on several accounts
diff --git a/docs/reference/environment-variables.md b/docs/reference/environment-variables.md
index fcc8796ef..c09c511f1 100644
--- a/docs/reference/environment-variables.md
+++ b/docs/reference/environment-variables.md
@@ -69,7 +69,7 @@ The stdio MCP package reads these environment variables directly. A CLI flag tak
| `MEMWAL_WEB_URL` | `--web-url ` | dashboard default | Dashboard URL used during login |
| `MEMWAL_CLIENT_LABEL` | `--label ` | `MCP Client` / `Walrus Memory MCP` | Friendly delegate-key label shown in the dashboard |
| `MEMWAL_MCP_DEBUG` | none | `0` | Set to `1` for verbose stderr logging |
-| `MEMWAL_CREDS_DIR` | none | `~/.memwal` | Directory holding `credentials.json` and the project-approval record. Overrides both project-local and `~/.memwal` credentials, and needs no project approval, re-read on every access so a test can redirect it after import. Mainly for tests, which must not write into the real credential directory |
+| `MEMWAL_CREDS_DIR` | none | `~/.memwal` | Directory holding `credentials.json` and the project-approval record. Overrides both project-local and `~/.memwal` credentials, and needs no project approval, re-read on every access so a test can redirect it after import. Must be an **absolute path outside the current project** — a relative or in-project value is refused with an error, because it skips the approval gate and could otherwise come from a checkout's MCP-config `env` block; an empty value means unset. Mainly for tests, which must not write into the real credential directory |
| `MEMWAL_MCP_TRANSPORT` | none | `sse` | Which relayer transport the stdio bridge dials. `sse` uses the legacy split (`POST /api/mcp/messages` + `GET /api/mcp/sse`). `http` (aliases `streamable`, `streamable-http`) uses the Streamable HTTP endpoint `/api/mcp`, where a call is answered on the same request rather than split across a POST and an SSE stream, so there is no idle watchdog. Reconnect replay is NOT yet transport-aware: the bridge still replays its in-flight map on either transport, and on Streamable a disconnect mid-send can look sent, so a replayed write may duplicate. Opt in knowing that. Unrecognised values fall back to `sse` |
| `MEMWAL_MCP_SSE_IDLE_MS` | none | `30000` | Maximum milliseconds of silence on the SSE stream before the bridge treats the session as dead and reconnects. Values below `500` are ignored and fall back to the default. Mainly for tests |
| `MEMWAL_MCP_CALL_TIMEOUT_MS` | none | `240000` | Maximum milliseconds a single request might wait for its response before the bridge answers with a retryable error. Covers a reply lost while the stream itself stays healthy, which `MEMWAL_MCP_SSE_IDLE_MS` cannot detect. The default is derived in code from the slowest server-side tool deadline plus headroom, so it moves with that tool rather than being pinned here. Values below `1000` are ignored and fall back to the default |
diff --git a/packages/mcp/CHANGELOG.md b/packages/mcp/CHANGELOG.md
index 981f1bc2b..d8c65cba3 100644
--- a/packages/mcp/CHANGELOG.md
+++ b/packages/mcp/CHANGELOG.md
@@ -6,6 +6,9 @@
- A project-local `.memwal/credentials.json` no longer decides where memory goes on presence alone. That file lives inside the repository, so anyone who could commit to a repo — or get a clone opened — could silently repoint the account and relayer every memory written from that directory went to, including from a subfolder, with nothing said and no delete path once written. A project file is now inert until the user approves that exact project path, account, delegate key and relayer with `memwal-mcp approve-project`; the approval record is kept in `~/.memwal/project-approvals.json`, outside the repository, so a repository cannot carry its own approval, and any later change to the destination requires approving again. Until approved the global credentials are used — an unapproved, altered or malformed project file is a fallback, never a failure — and one stderr line names the file that was skipped, the destination it wanted, and the command that approves it. `MEMWAL_CREDS_DIR` still overrides both files without approval. `memwal_health` now reports `account=` beside `relayer=`, so the active destination is visible where the user is. (WALM-639)
+- `MEMWAL_CREDS_DIR` must now be an absolute path outside the current project, and a value that is not is refused with an error naming it rather than followed. The override decides the credentials before any approval is looked up, so it was a way around the gate above rather than through it: an empty value made the approval store the bare relative name `project-approvals.json`, which resolves against the working directory — a repository that committed that one file at its root approved its own credentials — and a relative value such as `.memwal` put both the credentials and the approval record inside the repository. Neither had to be typed by the user: an MCP client reads `.cursor/mcp.json`, `.vscode/mcp.json` or `.claude/settings.json` out of the checkout and passes on the `env` block it finds, and editors expand `${workspaceFolder}` there, so an absolute path inside the project is refused too. An empty value now means unset, matching every other reader of the variable. Tests and CI pointing at an absolute temp directory are unaffected. (WALM-639)
+- The login write-ahead record no longer goes into the repository. It sat beside whichever credentials file resolution chose, so once a project file was approved every sign-in wrote a plaintext 64-hex Ed25519 delegate seed to `/.memwal/login-pending.json` — in a directory an attacker who planted the credentials file had already created, so tracked rather than ignored, and staged by `git add -A`. It now lives in `~/.memwal/login-pending/`, keyed by project so a sign-in is still reclaimable only by the project that started it. The credentials file itself cannot move — it is the file the user approved — so `approve-project` and the pre-sign-in warning now state plainly that signing in from this project writes a delegate private key inside the repository, and suggest adding `.memwal/` to `.gitignore`. (WALM-639)
+
## 0.0.14
### Fixed
diff --git a/packages/mcp/README.md b/packages/mcp/README.md
index 230f96c48..c5e31b221 100644
--- a/packages/mcp/README.md
+++ b/packages/mcp/README.md
@@ -172,10 +172,22 @@ to be approved again. The record is kept in `~/.memwal/project-approvals.json`,
outside the repository, so a repository cannot carry its own approval.
`revoke-project` withdraws it.
+Approving also picks the file that is **written**: a later sign-in from that
+project saves a delegate private key into `.memwal/credentials.json` in plain
+text, inside the repository. `approve-project` says so, and so does the sign-in
+warning. Add `.memwal/` to your `.gitignore`. (The short-lived login
+write-ahead record is kept outside the repository either way.)
+
`MEMWAL_CREDS_DIR` points both the credentials and the approval record at a
-directory of your choosing and overrides project resolution entirely — it can
-only come from your own environment, never from a checkout, so it needs no
-approval.
+directory of your choosing and overrides project resolution entirely, with no
+approval. Because it skips the gate, it must be an **absolute path outside the
+current project**: a relative value would resolve against the working directory
+and put the approval record inside the repository, and an MCP client passes on
+the `env` block it reads from `.cursor/mcp.json` / `.vscode/mcp.json` /
+`.claude/settings.json` in the checkout — where `${workspaceFolder}` is expanded,
+so an in-project absolute path is not proof you chose it. Anything else is
+refused with an error naming the value, rather than quietly ignored. An empty
+value means unset.
`memwal_health` reports the destination in use as `account=… relayer=…`.
diff --git a/packages/mcp/src/auth.ts b/packages/mcp/src/auth.ts
index d9ea39898..03e03da2a 100644
--- a/packages/mcp/src/auth.ts
+++ b/packages/mcp/src/auth.ts
@@ -11,7 +11,7 @@
*/
import { homedir } from "node:os";
import { createHash, randomUUID } from "node:crypto";
-import { join, dirname, basename } from "node:path";
+import { join, dirname, basename, isAbsolute, relative } from "node:path";
import {
mkdirSync,
readFileSync,
@@ -112,16 +112,147 @@ function projectCredsPath(): string | null {
const APPROVALS_FILE = "project-approvals.json";
+/**
+ * Refusal of a `MEMWAL_CREDS_DIR` that cannot be trusted to sit outside a
+ * repository. Thrown, never swallowed — see {@link credsDirOverride}.
+ */
+export class UntrustedCredsDirError extends Error {
+ constructor(message: string) {
+ super(message);
+ this.name = "UntrustedCredsDirError";
+ }
+}
+
+/** True when `path` is `root` itself or something underneath it. String
+ * prefixes get `/repo` vs `/repo-2` wrong, so this asks `relative` instead. */
+function isInside(root: string, path: string): boolean {
+ const rel = relative(root, path);
+ return rel === "" || (!rel.startsWith("..") && !isAbsolute(rel));
+}
+
+/**
+ * Canonical form of a path that need not exist yet.
+ *
+ * `realpathSync` throws on a missing leaf, and the override names a directory
+ * this process may be about to create — so resolve the deepest ancestor that
+ * does exist and re-append the rest. Without this, comparing it against the
+ * project root is decided by which spelling of a symlinked path each side
+ * happened to use: on macOS `/tmp/x` and `/private/tmp/x` are the same
+ * directory, and a containment check that says otherwise is a check an
+ * attacker picks the spelling to defeat.
+ */
+function canonicalDir(path: string): string {
+ const tail: string[] = [];
+ let dir = path;
+ for (;;) {
+ if (existsSync(dir)) return join(realpathSync(dir), ...tail.reverse());
+ const parent = dirname(dir);
+ if (parent === dir) return path;
+ tail.push(basename(dir));
+ dir = parent;
+ }
+}
+
+/**
+ * The repository the working directory is in, or null.
+ *
+ * Bounded exactly like {@link projectCredsPath}'s walk — nearest ancestor
+ * carrying `.git`, stopping at the home directory or the filesystem root — so
+ * "the project" means the same thing to the override check as it does to
+ * resolution.
+ */
+function projectRoot(): string | null {
+ const home = homedir();
+ let dir = process.cwd();
+ for (;;) {
+ if (dir === home) return null;
+ if (existsSync(join(dir, ".git"))) return dir;
+ const parent = dirname(dir);
+ if (parent === dir) return null;
+ dir = parent;
+ }
+}
+
+/**
+ * `MEMWAL_CREDS_DIR`, but only when it can be trusted — and the same answer for
+ * every reader of it.
+ *
+ * The override decides the credentials outright and skips the approval gate
+ * entirely, which makes it exactly as trustworthy as whoever set the
+ * environment. That is the right trade for the thing it exists for — a shell, a
+ * test harness, CI — and the wrong one for the way it is reachable: an MCP
+ * client reads `.cursor/mcp.json`, `.vscode/mcp.json` or `.claude/settings.json`
+ * OUT OF THE CHECKOUT and hands this process the `env` block it finds there. A
+ * committed `env` block is a repository setting its own override, which is the
+ * thing WALM-639 exists to stop.
+ *
+ * So two shapes are refused:
+ *
+ * - **Relative.** `join("", CREDS_FILE)` is the bare string
+ * `"credentials.json"` and `join(".memwal", CREDS_FILE)` is repo-local:
+ * both resolve against `process.cwd()`. That moves the credentials AND the
+ * approval store into whatever directory the client started in — usually
+ * the project root, so the store that exists precisely so a repository
+ * cannot carry its own approval becomes a file the repository carries.
+ * - **Absolute, but inside the current project.** Editors expand
+ * `${workspaceFolder}` inside those same config files, so "absolute" is not
+ * by itself evidence a human typed it. An override pointing into the
+ * checkout contradicts the one property this escape hatch is justified by.
+ *
+ * Refusing is loud on purpose, and matches the launcher's treatment of a
+ * relative runtime directory: falling back to the global directory in silence
+ * would hide a misconfigured — or hostile — client config, and leave the user
+ * to work out on their own why their override did nothing.
+ *
+ * An EMPTY value is the one thing that is not an error. It names no directory
+ * at all, which is what "unset" means, and it is what every other reader of
+ * this variable already treats it as; the bug was never that empty meant unset,
+ * it was that `??` let empty mean "the working directory".
+ */
+function credsDirOverride(): string | null {
+ const raw = process.env.MEMWAL_CREDS_DIR;
+ if (!raw) return null;
+
+ if (!isAbsolute(raw)) {
+ throw new UntrustedCredsDirError(
+ `MEMWAL_CREDS_DIR is a relative path (${JSON.stringify(raw)}). It resolves against ` +
+ `the working directory, which would put your credentials — and the record of ` +
+ `which project credentials you have approved — inside whatever directory this ` +
+ `process was started in. Set it to an absolute path, or unset it. ` +
+ `If you did not set it yourself, look at the \`env\` block of this project's MCP ` +
+ `config (.cursor/mcp.json, .vscode/mcp.json, .claude/settings.json): a value a ` +
+ `repository carries is the repository choosing where your memories go.`,
+ );
+ }
+
+ const root = projectRoot() ?? process.cwd();
+ const dir = canonicalDir(raw);
+ if (isInside(canonicalDir(root), dir)) {
+ throw new UntrustedCredsDirError(
+ `MEMWAL_CREDS_DIR (${raw}) points inside the current project (${root}). The ` +
+ `override skips the project-approval gate, so it has to name a directory no ` +
+ `repository can write — and an editor expands \`\${workspaceFolder}\` in a ` +
+ `committed MCP config, so an absolute path is not by itself proof that you ` +
+ `chose it. Point it outside the project, or unset it and run ` +
+ `\`memwal-mcp approve-project\` if you did mean to use this project's credentials.`,
+ );
+ }
+ return raw;
+}
+
/**
* Where records that a repository must not be able to write are kept.
*
* `MEMWAL_CREDS_DIR` is the trusted escape hatch — when it is set it decides
* the credentials outright and project resolution never runs — so following it
* here keeps a sandboxed run (tests, CI) from reaching into the real
- * `~/.memwal`, exactly as #705 required for the credentials file itself.
+ * `~/.memwal`, exactly as #705 required for the credentials file itself. It is
+ * read through {@link credsDirOverride} so that "trusted" means the same thing
+ * here, in {@link resolveCreds} and in {@link approveProjectCreds}: one rule,
+ * checked once, rather than three checks that can drift apart.
*/
function trustedStateDir(): string {
- return process.env.MEMWAL_CREDS_DIR ?? join(homedir(), ".memwal");
+ return credsDirOverride() ?? join(homedir(), ".memwal");
}
/** The approval store. Outside every repository, on purpose. */
@@ -248,8 +379,12 @@ export interface CredsResolution {
/**
* Which credentials file this process should read and write, and why.
*
- * `MEMWAL_CREDS_DIR` wins outright: it is the trusted, explicitly-set escape
- * hatch and cannot come from a checkout. Otherwise the nearest project-local
+ * `MEMWAL_CREDS_DIR` wins outright, and skips the approval gate — but only a
+ * value {@link credsDirOverride} accepts, i.e. a non-empty ABSOLUTE path
+ * outside the current project. It is "trusted" exactly as far as whoever set
+ * the environment is, and an MCP client will hand this process an `env` block
+ * it read out of the checkout, so a value that could have been committed is
+ * refused rather than obeyed. Otherwise the nearest project-local
* `.memwal/credentials.json` at or above the working directory is used IF the
* user has approved that exact destination on this machine (WALM-639), and the
* global file is used in every other case — including an unapproved, altered
@@ -260,7 +395,7 @@ export interface CredsResolution {
* directory nor the approval store is knowable at import time.
*/
export function resolveCreds(): CredsResolution {
- const override = process.env.MEMWAL_CREDS_DIR;
+ const override = credsDirOverride();
if (override) return { path: join(override, CREDS_FILE), source: "override" };
const global = globalCredsPath();
@@ -350,7 +485,7 @@ export interface ApproveProjectResult {
*/
export function approveProjectCreds(): ApproveProjectResult {
const approvalsPath = projectApprovalsPath();
- if (process.env.MEMWAL_CREDS_DIR) return { outcome: "overridden", approvalsPath };
+ if (credsDirOverride()) return { outcome: "overridden", approvalsPath };
const projectPath = projectCredsPath();
if (!projectPath) return { outcome: "none", approvalsPath };
@@ -464,6 +599,34 @@ export function formatProjectCredsNotice(
return lines.join("\n");
}
+/**
+ * What approving a project credentials file costs, beyond redirecting reads.
+ *
+ * Approval does not just decide which file is READ — it decides which file is
+ * WRITTEN. `saveCreds` writes to `credsPath()`, so from here on every sign-in
+ * from this project puts a fresh 64-hex Ed25519 delegate seed, in plaintext,
+ * into a directory that is inside the repository. An attacker who planted the
+ * file planted the directory too, so it is already tracked rather than ignored,
+ * and `git add -A` stages the key. The user cannot weigh that if nobody says
+ * it, and "you approved a destination" is not the same sentence as "you
+ * approved storing a private key in your repo".
+ *
+ * Separate from {@link formatProjectCredsNotice} on purpose: that one is about
+ * a file that was IGNORED, this one is about a file that is about to be used.
+ */
+export function formatProjectCredsStorageWarning(projectPath: string): string {
+ const dir = dirname(projectPath);
+ return [
+ `Note: ${dir} is inside the repository, and approving makes it the file this`,
+ `project signs and saves with. Every later sign-in from here writes a delegate`,
+ `PRIVATE KEY into ${basename(projectPath)} in plaintext — and a \`.memwal/\` a`,
+ `repository already carries is tracked, not ignored, so \`git add -A\` would stage it.`,
+ `Add \`.memwal/\` to .gitignore, and never commit ${basename(projectPath)}.`,
+ `(The short-lived login write-ahead record is kept outside the repository; the`,
+ `credentials file cannot be, because it is the file you approved.)`,
+ ].join("\n");
+}
+
/**
* Write credentials with secure (`0600`) permission, to whichever file
* `credsPath()` resolves to.
@@ -629,15 +792,25 @@ export interface SaveCredsResult {
* callback arrives, by which point a delegate key has already been registered
* on-chain — so a warning that waits for both ids is a warning that arrives
* too late to act on.
+ *
+ * When the file being replaced is an approved project one, that is also the
+ * last moment before a new private key lands inside a repository, so the
+ * storage warning is repeated here rather than left behind at approval time —
+ * approval may have been months ago, or done by someone else on the machine.
*/
export function formatPendingSignInWarning(): string | null {
- const current = loadCreds();
+ const resolution = resolveCreds();
+ const current = readCredsFile(resolution.path);
if (!current) return null;
- return (
- `Signing in will replace the credentials in ${credsPath()} ` +
- `(currently account ${current.accountId}). ` +
- `The existing file is backed up if the new sign-in is a different account.`
- );
+ const lines = [
+ `Signing in will replace the credentials in ${resolution.path} ` +
+ `(currently account ${current.accountId}). ` +
+ `The existing file is backed up if the new sign-in is a different account.`,
+ ];
+ if (resolution.source === "project") {
+ lines.push(formatProjectCredsStorageWarning(resolution.path));
+ }
+ return lines.join("\n");
}
/**
@@ -756,6 +929,10 @@ function isValid(obj: unknown): obj is MemWalCredentials {
const PENDING_FILE = "login-pending.json";
+/** Where a PROJECT sign-in's write-ahead record goes, under the trusted state
+ * directory — one file per approved project. See {@link pendingLoginPath}. */
+const PENDING_DIR = "login-pending";
+
/**
* How long a stranded record stays recoverable.
*
@@ -779,10 +956,31 @@ export interface PendingLogin {
version: 1;
}
-/** Sits beside whichever credentials file `credsPath()` resolves to, so a
- * project-local sign-in recovers into that same project. */
+/**
+ * Where the write-ahead record lives — never inside a repository.
+ *
+ * It used to sit beside whichever credentials file `credsPath()` resolved to,
+ * which reads as "recover into the project you signed in from" and is the right
+ * intent. But for an approved project that path is `/.memwal/`, so every
+ * sign-in dropped a plaintext 64-hex delegate seed into the working tree —
+ * under a directory an attacker who planted the credentials file had already
+ * created, so tracked rather than ignored, and staged by `git add -A`.
+ *
+ * Nothing about this record needs to be in the repo. It is short-lived
+ * handshake state, nobody edits it by hand, and the only property that matters
+ * is that the project which started a sign-in is the one that can reclaim it.
+ * So it moves to the trusted state directory, keyed by a hash of the
+ * credentials path: same per-project scoping, no key material in the checkout.
+ *
+ * The non-project cases are byte-for-byte where they always were —
+ * `dirname(credsPath())` IS `trustedStateDir()` for both the override and the
+ * global file — so a machine that has never approved a project sees no change.
+ */
export function pendingLoginPath(): string {
- return join(dirname(credsPath()), PENDING_FILE);
+ const resolution = resolveCreds();
+ if (resolution.source !== "project") return join(trustedStateDir(), PENDING_FILE);
+ const key = createHash("sha256").update(canonicalPath(resolution.path)).digest("hex");
+ return join(trustedStateDir(), PENDING_DIR, `${key}.json`);
}
/**
@@ -794,10 +992,10 @@ export function pendingLoginPath(): string {
* the connect URL while claiming a durability that does not exist — the
* original WALM-332 loss, now silent.
*
- * Failing the login costs the user nothing: this file sits beside
- * `credentials.json`, so a directory that cannot take it cannot take the
- * credentials either. The same login would have failed at the callback anyway,
- * one on-chain `add_delegate_key` later.
+ * Failing the login costs the user nothing: this file lives in the same
+ * trusted state directory the approval record does, so a directory that cannot
+ * take it is one this process cannot keep state in at all. The same login would
+ * have failed at the callback anyway, one on-chain `add_delegate_key` later.
*/
export function savePendingLogin(pending: PendingLogin): void {
const path = pendingLoginPath();
diff --git a/packages/mcp/src/index.ts b/packages/mcp/src/index.ts
index 230ed0c36..31028cbfd 100644
--- a/packages/mcp/src/index.ts
+++ b/packages/mcp/src/index.ts
@@ -15,6 +15,7 @@ import {
clearPendingLogin,
credsPath,
formatProjectCredsNotice,
+ formatProjectCredsStorageWarning,
loadCreds,
resolveCreds,
revokeProjectCredsApproval,
@@ -267,6 +268,12 @@ export async function main(argv: string[] = process.argv.slice(2)): Promise/.memwal/project-approvals.json`. Both resolve against the
+ * working directory, which is the repository. And because the override
+ * branch of `resolveCreds()` runs BEFORE any approval lookup, a committed
+ * `.cursor/mcp.json` carrying that one env var got the repo's credentials
+ * used with no approval at all — the exact silent redirect WALM-639 is
+ * about, through the escape hatch instead of around it.
+ *
+ * B. `pendingLoginPath()` was `join(dirname(credsPath()), …)`. Once a project
+ * was approved that is `/.memwal/`, so every sign-in wrote a 64-hex
+ * Ed25519 seed, in plaintext, into the working tree.
+ *
+ * Same sandbox pattern as project-creds-approval.test.mjs: HOME, cwd and the
+ * override are set first, then `auth.js` is imported with a cache-busting query.
+ */
+import { test } from "node:test";
+import assert from "node:assert/strict";
+import {
+ mkdtempSync,
+ mkdirSync,
+ writeFileSync,
+ readFileSync,
+ readdirSync,
+ rmSync,
+ existsSync,
+ realpathSync,
+} from "node:fs";
+import { tmpdir } from "node:os";
+import { join, dirname, isAbsolute, relative } from "node:path";
+
+const GLOBAL_ACCOUNT = "0x" + "a".repeat(64);
+const PROJECT_ACCOUNT = "0x" + "b".repeat(64);
+const CREDS_KEY = "c".repeat(64);
+/** Distinct from CREDS_KEY — and from every other filler above, `packageId`
+ * included — so "this key never reaches the repo" is a claim about the pending
+ * record specifically and cannot be satisfied or broken by another field. */
+const PENDING_KEY = "4".repeat(64);
+const GLOBAL_RELAYER = "https://relayer.example";
+const PROJECT_RELAYER = "https://project-relayer.example";
+
+function makeCreds(overrides = {}) {
+ return {
+ delegatePrivateKey: CREDS_KEY,
+ delegatePublicKeyHex: "d".repeat(64),
+ delegateAddress: "0x" + "e".repeat(64),
+ walletAddress: "0x" + "f".repeat(64),
+ accountId: GLOBAL_ACCOUNT,
+ packageId: "0x" + "1".repeat(64),
+ relayerUrl: GLOBAL_RELAYER,
+ createdAt: new Date(0).toISOString(),
+ version: 1,
+ ...overrides,
+ };
+}
+
+function makePending(overrides = {}) {
+ return {
+ delegatePrivateKey: PENDING_KEY,
+ delegatePublicKeyHex: "2".repeat(64),
+ delegateAddress: "0x" + "3".repeat(64),
+ relayerUrl: PROJECT_RELAYER,
+ createdAt: new Date().toISOString(),
+ version: 1,
+ ...overrides,
+ };
+}
+
+function writeCredsAt(root, creds) {
+ const path = join(root, ".memwal", "credentials.json");
+ mkdirSync(dirname(path), { recursive: true });
+ writeFileSync(path, JSON.stringify(creds), { mode: 0o600 });
+ return path;
+}
+
+/** Same containment question the module asks: `relative` rather than a string
+ * prefix, so `/repo` and `/repo-2` do not look like the same directory. */
+function isInside(root, path) {
+ const rel = relative(root, path);
+ return rel === "" || (!rel.startsWith("..") && !isAbsolute(rel));
+}
+
+/** Every file at or under `root`, so a secret can be searched for across a
+ * whole working tree rather than at the one path a test remembered to check. */
+function filesUnder(root) {
+ const out = [];
+ for (const entry of readdirSync(root, { withFileTypes: true })) {
+ const path = join(root, entry.name);
+ if (entry.isDirectory()) out.push(...filesUnder(path));
+ else if (entry.isFile()) out.push(path);
+ }
+ return out;
+}
+
+/** Paths under `root` whose bytes contain `needle`. */
+function filesContaining(root, needle) {
+ return filesUnder(root).filter((path) => {
+ try {
+ return readFileSync(path, "utf8").includes(needle);
+ } catch {
+ return false;
+ }
+ });
+}
+
+/**
+ * A HOME, a working directory, a fresh module, and MEMWAL_CREDS_DIR cleared.
+ *
+ * Canonicalised because `process.cwd()` and `homedir()` report resolved paths
+ * and macOS routes /tmp through /private/tmp — the containment check under test
+ * has to hold for the same directory spelled either way.
+ */
+async function sandbox(t, { global: globalCreds, project: projectCreds, git = false } = {}) {
+ const home = realpathSync(mkdtempSync(join(tmpdir(), "memwal-trust-home-")));
+ const cwd = realpathSync(mkdtempSync(join(tmpdir(), "memwal-trust-cwd-")));
+ const previous = {
+ home: process.env.HOME,
+ profile: process.env.USERPROFILE,
+ credsDir: process.env.MEMWAL_CREDS_DIR,
+ cwd: process.cwd(),
+ };
+
+ process.env.HOME = home;
+ process.env.USERPROFILE = home;
+ delete process.env.MEMWAL_CREDS_DIR;
+ process.chdir(cwd);
+
+ if (git) mkdirSync(join(cwd, ".git"), { recursive: true });
+ if (globalCreds) writeCredsAt(home, globalCreds);
+ if (projectCreds) writeCredsAt(cwd, projectCreds);
+
+ t.after(() => {
+ process.chdir(previous.cwd);
+ process.env.HOME = previous.home;
+ process.env.USERPROFILE = previous.profile;
+ if (previous.credsDir === undefined) delete process.env.MEMWAL_CREDS_DIR;
+ else process.env.MEMWAL_CREDS_DIR = previous.credsDir;
+ rmSync(home, { recursive: true, force: true });
+ rmSync(cwd, { recursive: true, force: true });
+ });
+
+ const bust = `${Date.now()}-${Math.random()}`;
+ const auth = await import(`../dist/auth.js?walm639trust=${bust}`);
+ return { auth, home, cwd, bust };
+}
+
+const globalFile = (home) => join(home, ".memwal", "credentials.json");
+const projectFile = (cwd) => join(cwd, ".memwal", "credentials.json");
+
+/* --------------------------------------------------------------------- *
+ * A. An empty value means unset — never "the working directory".
+ * --------------------------------------------------------------------- */
+
+test('MEMWAL_CREDS_DIR="" does not relocate the approval store into the repo', async (t) => {
+ const { auth, home, cwd } = await sandbox(t, {
+ global: makeCreds(),
+ project: makeCreds({ accountId: PROJECT_ACCOUNT, relayerUrl: PROJECT_RELAYER }),
+ git: true,
+ });
+ process.env.MEMWAL_CREDS_DIR = "";
+
+ const approvals = auth.projectApprovalsPath();
+
+ // The reported repro: `??` kept the empty string, so `join("", FILE)` came
+ // back as the bare relative name and resolved against the repo.
+ assert.equal(isAbsolute(approvals), true, `approvals path is relative: ${approvals}`);
+ assert.notEqual(approvals, "project-approvals.json");
+ assert.equal(approvals, join(home, ".memwal", "project-approvals.json"));
+ assert.equal(isInside(cwd, approvals), false, "the store must stay out of the repository");
+});
+
+test('MEMWAL_CREDS_DIR="" does not silently approve the repo credentials file', async (t) => {
+ // The override branch returns before any approval lookup, so an empty value
+ // that counted as "set" adopted the repo's account with nothing asked.
+ const { auth, home } = await sandbox(t, {
+ global: makeCreds(),
+ project: makeCreds({ accountId: PROJECT_ACCOUNT, relayerUrl: PROJECT_RELAYER }),
+ git: true,
+ });
+ process.env.MEMWAL_CREDS_DIR = "";
+
+ assert.equal(auth.resolveCreds().source, "global");
+ assert.equal(auth.credsPath(), globalFile(home));
+ assert.equal(auth.loadCreds()?.accountId, GLOBAL_ACCOUNT);
+ assert.equal(auth.resolveCreds().project?.decision, "unapproved");
+ assert.ok(auth.formatProjectCredsNotice(), "the ignored repo file is still reported");
+});
+
+test('MEMWAL_CREDS_DIR="" lets no repository approve its own credentials', async (t) => {
+ // The sharp end of the empty-string bug. `join("", APPROVALS_FILE)` is the
+ // bare name "project-approvals.json", which resolves against the working
+ // directory — so a repo that commits that one file at its root IS the
+ // approval store, and approves the credentials it also ships. Everything
+ // the gate does is decided by a file the attacker wrote.
+ const { auth, home, cwd } = await sandbox(t, {
+ global: makeCreds(),
+ project: makeCreds({ accountId: PROJECT_ACCOUNT, relayerUrl: PROJECT_RELAYER }),
+ git: true,
+ });
+ const delegateAddress = "0x" + "e".repeat(64);
+ writeFileSync(
+ join(cwd, "project-approvals.json"),
+ JSON.stringify({
+ version: 1,
+ approvals: [
+ {
+ path: projectFile(cwd),
+ fingerprint: auth.credentialsFingerprint({
+ accountId: PROJECT_ACCOUNT,
+ delegateAddress,
+ relayerUrl: PROJECT_RELAYER,
+ }),
+ accountId: PROJECT_ACCOUNT,
+ delegateAddress,
+ relayerUrl: PROJECT_RELAYER,
+ approvedAt: new Date().toISOString(),
+ },
+ ],
+ }),
+ );
+ process.env.MEMWAL_CREDS_DIR = "";
+
+ assert.equal(auth.credsPath(), globalFile(home), "a repo approved itself");
+ assert.equal(auth.loadCreds()?.accountId, GLOBAL_ACCOUNT);
+ assert.equal(auth.resolveCreds().project?.decision, "unapproved");
+});
+
+/* --------------------------------------------------------------------- *
+ * A. A relative value is refused, loudly.
+ * --------------------------------------------------------------------- */
+
+for (const value of [".memwal", "creds", "./.memwal", "../elsewhere", ".claude/state"]) {
+ test(`a relative MEMWAL_CREDS_DIR (${value}) is refused, not followed`, async (t) => {
+ const { auth, cwd } = await sandbox(t, {
+ global: makeCreds(),
+ project: makeCreds({ accountId: PROJECT_ACCOUNT, relayerUrl: PROJECT_RELAYER }),
+ git: true,
+ });
+ process.env.MEMWAL_CREDS_DIR = value;
+
+ const refuses = /MEMWAL_CREDS_DIR is a relative path/;
+ assert.throws(() => auth.projectApprovalsPath(), refuses);
+ assert.throws(() => auth.credsPath(), refuses);
+ assert.throws(() => auth.resolveCreds(), refuses);
+ assert.throws(() => auth.loadCreds(), refuses);
+ // Refusing rather than falling back: a silent fallback would hide a
+ // misconfigured — or planted — client config.
+ assert.throws(() => auth.approveProjectCreds(), refuses);
+ assert.throws(() => auth.saveCreds(makeCreds()), refuses);
+
+ // And nothing landed in the working tree on the way to refusing.
+ assert.deepEqual(
+ filesUnder(cwd).filter((p) => p.includes("project-approvals")),
+ [],
+ "an approval record was written inside the repository",
+ );
+ });
+}
+
+test("a relative MEMWAL_CREDS_DIR leaves the repo file unapproved once cleared", async (t) => {
+ // The failure mode worth pinning: refusing must not be a disguised approval
+ // that shows up the moment the variable goes away.
+ const { auth, home, cwd } = await sandbox(t, {
+ global: makeCreds(),
+ project: makeCreds({ accountId: PROJECT_ACCOUNT, relayerUrl: PROJECT_RELAYER }),
+ git: true,
+ });
+ process.env.MEMWAL_CREDS_DIR = ".memwal";
+ assert.throws(() => auth.approveProjectCreds());
+
+ delete process.env.MEMWAL_CREDS_DIR;
+
+ assert.equal(auth.credsPath(), globalFile(home));
+ assert.equal(auth.resolveCreds().project?.decision, "unapproved");
+ assert.equal(existsSync(join(cwd, ".memwal", "project-approvals.json")), false);
+});
+
+test("the refusal is loud at the CLI, not swallowed into a fallback", async (t) => {
+ const { cwd, bust } = await sandbox(t, {
+ global: makeCreds(),
+ project: makeCreds({ accountId: PROJECT_ACCOUNT }),
+ git: true,
+ });
+ const { main } = await import(`../dist/index.js?walm639trust=${bust}`);
+
+ const realTty = process.stdin.isTTY;
+ t.after(() => {
+ process.stdin.isTTY = realTty;
+ });
+ process.stdin.isTTY = true;
+ process.env.MEMWAL_CREDS_DIR = join(cwd, ".memwal");
+
+ // `bin/memwal-mcp.ts` turns this into `[memwal-mcp] fatal: …` and exit 1,
+ // which is the behaviour the launcher already has for a relative runtime
+ // directory.
+ await assert.rejects(main(["approve-project"]), /MEMWAL_CREDS_DIR/);
+});
+
+/* --------------------------------------------------------------------- *
+ * A. An absolute value inside the project is refused the same way.
+ * --------------------------------------------------------------------- */
+
+test("an absolute MEMWAL_CREDS_DIR inside the project is refused", async (t) => {
+ // `${workspaceFolder}` is expanded by editors inside the very config files
+ // an attacker can commit, so "absolute" is not evidence a human typed it.
+ const { auth, cwd } = await sandbox(t, {
+ global: makeCreds(),
+ project: makeCreds({ accountId: PROJECT_ACCOUNT }),
+ git: true,
+ });
+ process.env.MEMWAL_CREDS_DIR = join(cwd, ".memwal");
+
+ assert.throws(() => auth.resolveCreds(), /points inside the current project/);
+ assert.throws(() => auth.projectApprovalsPath(), /points inside the current project/);
+});
+
+test("the refusal is against the project root, not just the working directory", async (t) => {
+ // Running from `src/nested` must not launder an override that points at the
+ // repository root — the creds walk climbs, so this check has to as well.
+ const { auth, cwd } = await sandbox(t, {
+ global: makeCreds(),
+ project: makeCreds({ accountId: PROJECT_ACCOUNT }),
+ git: true,
+ });
+ const deep = join(cwd, "src", "nested");
+ mkdirSync(deep, { recursive: true });
+ process.chdir(deep);
+ process.env.MEMWAL_CREDS_DIR = join(cwd, ".memwal");
+
+ assert.throws(() => auth.resolveCreds(), /points inside the current project/);
+});
+
+/* --------------------------------------------------------------------- *
+ * A. The legitimate use keeps working, byte for byte.
+ * --------------------------------------------------------------------- */
+
+test("an absolute MEMWAL_CREDS_DIR outside the project still decides everything", async (t) => {
+ const { auth, cwd } = await sandbox(t, {
+ global: makeCreds(),
+ project: makeCreds({ accountId: PROJECT_ACCOUNT, relayerUrl: PROJECT_RELAYER }),
+ git: true,
+ });
+ const override = realpathSync(mkdtempSync(join(tmpdir(), "memwal-trust-override-")));
+ t.after(() => rmSync(override, { recursive: true, force: true }));
+ process.env.MEMWAL_CREDS_DIR = override;
+
+ assert.equal(auth.credsPath(), join(override, "credentials.json"));
+ assert.equal(auth.resolveCreds().source, "override");
+ assert.equal(auth.projectApprovalsPath(), join(override, "project-approvals.json"));
+ assert.equal(auth.pendingLoginPath(), join(override, "login-pending.json"));
+ assert.equal(auth.formatProjectCredsNotice(), null, "an override has nothing to warn about");
+ assert.equal(auth.approveProjectCreds().outcome, "overridden");
+
+ // And it is genuinely usable, not merely accepted.
+ auth.saveCreds(makeCreds({ label: "sandboxed" }));
+ assert.equal(auth.loadCreds()?.label, "sandboxed");
+ assert.equal(
+ JSON.parse(readFileSync(projectFile(cwd), "utf8")).label,
+ undefined,
+ "the repo file must not have been written",
+ );
+});
+
+/* --------------------------------------------------------------------- *
+ * B. An approved project never receives a plaintext delegate key.
+ * --------------------------------------------------------------------- */
+
+test("the pending-login record for an approved project is not written into it", async (t) => {
+ const { auth, home, cwd } = await sandbox(t, {
+ global: makeCreds(),
+ project: makeCreds({ accountId: PROJECT_ACCOUNT, relayerUrl: PROJECT_RELAYER }),
+ git: true,
+ });
+ auth.approveProjectCreds();
+ assert.equal(auth.credsPath(), projectFile(cwd), "precondition: the project file is in use");
+
+ auth.savePendingLogin(makePending());
+
+ const path = auth.pendingLoginPath();
+ assert.equal(isInside(cwd, path), false, `the write-ahead record landed in the repo: ${path}`);
+ assert.equal(isInside(join(home, ".memwal"), path), true, "it belongs in the trusted dir");
+ assert.equal(existsSync(join(cwd, ".memwal", "login-pending.json")), false);
+ assert.deepEqual(
+ filesContaining(cwd, PENDING_KEY),
+ [],
+ "a plaintext delegate seed was written somewhere inside the repository",
+ );
+ // Still a working write-ahead record, which is the whole point of it.
+ assert.equal(auth.loadPendingLogin()?.delegatePrivateKey, PENDING_KEY);
+ assert.equal(auth.reusablePendingLogin(PROJECT_RELAYER)?.delegatePrivateKey, PENDING_KEY);
+ auth.clearPendingLogin();
+ assert.equal(auth.loadPendingLogin(), null);
+});
+
+test("one project's pending record is not another project's", async (t) => {
+ // Keying it by project is why it could live in the repo at all; moving it
+ // out must not turn it into one shared record.
+ const { auth, cwd } = await sandbox(t, {
+ global: makeCreds(),
+ project: makeCreds({ accountId: PROJECT_ACCOUNT, relayerUrl: PROJECT_RELAYER }),
+ git: true,
+ });
+ auth.approveProjectCreds();
+ auth.savePendingLogin(makePending());
+ const first = auth.pendingLoginPath();
+
+ const other = realpathSync(mkdtempSync(join(tmpdir(), "memwal-trust-other-")));
+ t.after(() => rmSync(other, { recursive: true, force: true }));
+ mkdirSync(join(other, ".git"), { recursive: true });
+ writeCredsAt(other, makeCreds({ accountId: PROJECT_ACCOUNT, relayerUrl: PROJECT_RELAYER }));
+ process.chdir(other);
+ auth.approveProjectCreds();
+
+ assert.notEqual(auth.pendingLoginPath(), first, "two projects share one record");
+ assert.equal(auth.loadPendingLogin(), null, "a sign-in leaked across projects");
+});
+
+test("the global pending-login path is exactly where it always was", async (t) => {
+ const { auth, home } = await sandbox(t, { global: makeCreds() });
+
+ assert.equal(auth.pendingLoginPath(), join(home, ".memwal", "login-pending.json"));
+ auth.savePendingLogin(makePending());
+ assert.equal(existsSync(join(home, ".memwal", "login-pending.json")), true);
+});
+
+/* --------------------------------------------------------------------- *
+ * B. And the user is told, rather than left to find the key in a diff.
+ * --------------------------------------------------------------------- */
+
+test("approving says a private key will be written inside the repository", async (t) => {
+ const { auth, cwd, bust } = await sandbox(t, {
+ global: makeCreds(),
+ project: makeCreds({ accountId: PROJECT_ACCOUNT, relayerUrl: PROJECT_RELAYER }),
+ git: true,
+ });
+ const { main } = await import(`../dist/index.js?walm639trust=${bust}`);
+
+ const realTty = process.stdin.isTTY;
+ const realWrite = process.stderr.write.bind(process.stderr);
+ let output = "";
+ t.after(() => {
+ process.stdin.isTTY = realTty;
+ process.stderr.write = realWrite;
+ });
+ process.stdin.isTTY = true;
+ process.stderr.write = (chunk) => {
+ output += chunk;
+ return true;
+ };
+
+ await main(["approve-project"]);
+ process.stderr.write = realWrite;
+
+ assert.match(output, /Approved /, "precondition: it approved");
+ assert.match(output, /PRIVATE KEY/, `approval said nothing about the key:\n${output}`);
+ assert.match(output, /\.gitignore/, "no suggestion for keeping it out of the repo");
+ assert.ok(output.includes(join(cwd, ".memwal")), "must name the directory in the repo");
+});
+
+test("the warning is repeated at the last moment before a key is written", async (t) => {
+ // Approval may have happened months ago, or on someone else's shift. The
+ // sign-in warning is the last point the user can back out for free.
+ const { auth, cwd } = await sandbox(t, {
+ global: makeCreds(),
+ project: makeCreds({ accountId: PROJECT_ACCOUNT, relayerUrl: PROJECT_RELAYER }),
+ git: true,
+ });
+ auth.approveProjectCreds();
+
+ const warning = auth.formatPendingSignInWarning();
+
+ assert.ok(warning.includes(projectFile(cwd)), "must name the file being replaced");
+ assert.match(warning, /PRIVATE KEY/);
+ assert.match(warning, /\.gitignore/);
+});
+
+test("a global sign-in is not nagged about a repository it is not in", async (t) => {
+ const { auth, home } = await sandbox(t, { global: makeCreds() });
+
+ const warning = auth.formatPendingSignInWarning();
+
+ assert.ok(warning.includes(globalFile(home)));
+ assert.doesNotMatch(warning, /gitignore/, "nothing repo-shaped to say about the global file");
+});
+
+test("nothing the storage warning prints is key material", async (t) => {
+ const { auth, cwd } = await sandbox(t, {
+ global: makeCreds(),
+ project: makeCreds({ accountId: PROJECT_ACCOUNT }),
+ });
+
+ const warning = auth.formatProjectCredsStorageWarning(projectFile(cwd));
+
+ assert.ok(!warning.includes(CREDS_KEY));
+ assert.ok(!warning.includes(PENDING_KEY));
+});
From bbcef3435f644bc83e06bb79172d8140738faf72 Mon Sep 17 00:00:00 2001
From: Harry Phan
Date: Fri, 18 Sep 2026 13:06:21 +0700
Subject: [PATCH 079/132] fix(mcp): stop a repo deciding whether automatic
memory is on (WALM-642)
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit
The hook-side resolver was a hand-written mirror of src/auto-save.ts and had
drifted: it still used the pre-WALM-639 presence rule, so the nearest project
`.memwal/credentials.json` won outright with its contents never parsed. A repo
containing only that file — literal text `not even valid json` was enough —
read as a long-standing install, and the UserPromptSubmit hook injected the
full proactive-save rubric while the user's own settings said the consent
question was still pending. A committed `settings.json` alongside it pinned
`autoSave: true` and suppressed the consent-pending warning too.
Copying the approval check into the .mjs is how the drift happened in the
first place, so the hooks stop re-deriving resolution altogether. The server
publishes the resolved answer to `auto-save-state.json` in the trusted state
dir and the hooks read only that; they never touch `process.cwd()`, and a
missing, unreadable, malformed or newer-versioned file reads as "cannot tell",
which is automatic memory off.
Consent also stops being scoped per credentials directory. It is a decision
about the human, so it lives in the trusted state dir alone. Approving a
project (WALM-639) used to move `settingsPath()` into the repository, where a
recorded "no" was never consulted, the empty directory fell through to
"legacy, keep saving", and `setAutoSave`/`markConsentPending` then wrote the
consent answer into the checkout. The legacy probe reads the trusted
directory's own credentials for the same reason.
The existing opt-in suite could not catch either: every test sets
`MEMWAL_CREDS_DIR`, which short-circuits project resolution on both sides. The
new suite drives a real HOME and working directory instead and spawns the
actual hooks.
---
packages/mcp/plugin/scripts/lib/auto-save.mjs | 158 ++++----
packages/mcp/src/auto-save.ts | 143 +++++--
packages/mcp/src/index.ts | 12 +
packages/mcp/test/auto-save-optin.test.mjs | 56 ++-
.../mcp/test/auto-save-repo-trust.test.mjs | 349 ++++++++++++++++++
5 files changed, 612 insertions(+), 106 deletions(-)
create mode 100644 packages/mcp/test/auto-save-repo-trust.test.mjs
diff --git a/packages/mcp/plugin/scripts/lib/auto-save.mjs b/packages/mcp/plugin/scripts/lib/auto-save.mjs
index f776b369a..ae7742e34 100644
--- a/packages/mcp/plugin/scripts/lib/auto-save.mjs
+++ b/packages/mcp/plugin/scripts/lib/auto-save.mjs
@@ -1,70 +1,62 @@
/**
* Automatic-save opt-in, hook-side (WALM-642).
*
- * The mirror of `packages/mcp/src/auto-save.ts`, in plain ESM with no
- * dependencies and no network, because a hook is a `.mjs` file the client
- * spawns directly — it never loads this package's compiled `dist/`, and it
- * inherits none of the MCP server's configuration (a hook is spawned by Claude
- * Code or Codex, the MCP server by its own `command`/`env` block). A shared
- * file on disk is the only thing both sides can actually see, which is why the
- * choice is persisted rather than passed.
+ * This file used to be a hand-written MIRROR of `packages/mcp/src/auto-save.ts`
+ * — the same walk up the directory tree, the same "nearest project-local
+ * `.memwal/credentials.json` wins", the same settings lookup beside whatever
+ * that resolved to. Keeping two implementations of one security decision in
+ * step is not a thing anyone manages to do, and they drifted: the TypeScript
+ * side grew an approval gate (WALM-639) and this side did not, so a repository
+ * containing nothing but `.memwal/credentials.json` — contents never parsed,
+ * only `existsSync` — read as proof of a long-standing install and switched
+ * automatic memory ON for anyone who opened it. A committed `settings.json`
+ * next to it could pin `autoSave: true` outright and silence the
+ * consent-pending warning while it did.
*
- * Resolution is deliberately identical to the TypeScript side:
- * 1. `MEMWAL_AUTO_SAVE` in the environment.
- * 2. `autoSave` in `settings.json`, in whichever `.memwal` directory the
- * credentials resolve to — `MEMWAL_CREDS_DIR`, else the nearest
- * project-local `.memwal/credentials.json` at or above the working
- * directory, else `~/.memwal`.
- * 3. Unanswered: on for an install that predates the consent prompt, off for
- * one created after it (`autoSaveConsent: "pending"`).
+ * So the mirror is gone. The MCP server resolves the state — approvals,
+ * project scoping, the consent answer, all of it — and publishes the ANSWER to
+ * `auto-save-state.json` in the trusted state dir (`MEMWAL_CREDS_DIR`, else
+ * `~/.memwal`). This file reads that file and nothing else.
*
- * Any error reads as "not answered": a hook must never block a session, and an
- * unreadable file is not consent.
+ * What that buys:
+ * - No resolution here at all, so there is nothing left to drift.
+ * - Nothing under `process.cwd()` is ever read, so a checkout cannot
+ * influence hook behaviour — not through a credentials file, not through a
+ * settings file, not through anything it can add later.
+ * - Unreadable, missing, malformed, or written by a newer version reads as
+ * "state unknown", which is automatic memory OFF. A hook must never block a
+ * session, and an absent answer is not consent.
+ *
+ * `MEMWAL_AUTO_SAVE` is still honoured first. It comes from this process's
+ * environment — the client's hook configuration, set by the user — never from a
+ * file a repository can carry.
*/
import { existsSync, readFileSync } from "node:fs";
import { homedir } from "node:os";
-import { dirname, join } from "node:path";
+import { join } from "node:path";
export const AUTO_SAVE_ENV = "MEMWAL_AUTO_SAVE";
-const CREDS_FILE = "credentials.json";
-const SETTINGS_FILE = "settings.json";
-
-function globalCredsPath() {
- return join(homedir(), ".memwal", CREDS_FILE);
-}
+const HOOK_STATE_FILE = "auto-save-state.json";
+/** The only shape this file knows how to read. */
+const SUPPORTED_VERSION = 1;
/**
- * Nearest project-local credentials file, walking up from the working
- * directory and stopping at the project root, the home directory, or the
- * filesystem root. Mirrors `projectCredsPath()` in src/auth.ts — the two must
- * agree, or a project-scoped opt-in would apply to the hooks and not the server
- * (or the other way round).
+ * The trusted state dir, resolved from the environment and the home directory
+ * only. Deliberately does NOT look at `process.cwd()`: that is the whole point.
+ *
+ * `MEMWAL_CREDS_DIR` is the same trusted escape hatch auth.ts uses, so a
+ * sandboxed run (tests, CI) points both sides at the same temporary directory.
*/
-function projectCredsPath() {
- const home = homedir();
- const global = globalCredsPath();
- let dir = process.cwd();
- for (;;) {
- if (dir === home) return null;
- const candidate = join(dir, ".memwal", CREDS_FILE);
- if (candidate !== global && existsSync(candidate)) return candidate;
- if (existsSync(join(dir, ".git"))) return null;
- const parent = dirname(dir);
- if (parent === dir) return null;
- dir = parent;
- }
-}
-
-function credsPath() {
+function trustedStateDir() {
const override = process.env.MEMWAL_CREDS_DIR;
- if (override) return join(override, CREDS_FILE);
- return projectCredsPath() ?? globalCredsPath();
+ if (override) return override;
+ return join(homedir(), ".memwal");
}
-/** Where a persisted choice lives for this working directory. */
-export function settingsPath() {
- return join(dirname(credsPath()), SETTINGS_FILE);
+/** Where the MCP server publishes the resolved state. */
+export function hookStatePath() {
+ return join(trustedStateDir(), HOOK_STATE_FILE);
}
/** Human-written boolean. null = not set / unparseable, which is not consent. */
@@ -77,53 +69,59 @@ export function parseBooleanSetting(raw) {
return null;
}
-function readSettings() {
+/**
+ * The published state, or null when there is not a readable, understood one.
+ *
+ * Every failure mode collapses to null on purpose — missing file, unreadable
+ * file, corrupt JSON, a version this build does not know, a payload whose
+ * `enabled` is not a boolean. The caller turns null into "off".
+ */
+function readPublishedState() {
try {
- const path = settingsPath();
- if (!existsSync(path)) return {};
+ const path = hookStatePath();
+ if (!existsSync(path)) return null;
const parsed = JSON.parse(readFileSync(path, "utf8"));
- return parsed && typeof parsed === "object" ? parsed : {};
+ if (!parsed || typeof parsed !== "object") return null;
+ if (parsed.version !== SUPPORTED_VERSION) return null;
+ if (typeof parsed.enabled !== "boolean") return null;
+ return parsed;
} catch {
- // Unreadable or corrupt is not an answer.
- return {};
+ return null;
}
}
/**
* `{ enabled, state, source, pendingConsent }`.
*
- * Mirrors `autoSaveStatus()` in src/auto-save.ts, including the two unset
- * rules: an install that predates the consent prompt (no stamp, credentials on
- * disk) keeps saving, and one created after it saves nothing until answered.
- * The two must agree, or the hooks would steer the agent one way while the MCP
- * server's instructions steered it the other.
+ * `source: "unavailable"` is the fail-safe: the server has not published a
+ * state this hook can read, so nothing is saved unprompted and the session is
+ * told the question is still open.
*/
export function autoSaveStatus() {
- const settings = readSettings();
- const answered =
- typeof settings.autoSave === "boolean" ? settings.autoSave : null;
- const state = answered === null ? "unset" : answered ? "on" : "off";
-
const fromEnv = parseBooleanSetting(process.env[AUTO_SAVE_ENV]);
if (fromEnv !== null) {
- return { enabled: fromEnv, state, source: "env", pendingConsent: false };
- }
- if (answered !== null) {
- return { enabled: answered, state, source: "settings", pendingConsent: false };
+ return {
+ enabled: fromEnv,
+ state: fromEnv ? "on" : "off",
+ source: "env",
+ pendingConsent: false,
+ };
}
- const stamped = settings.autoSaveConsent === "pending";
- let preExisting = false;
- try {
- preExisting = !stamped && existsSync(credsPath());
- } catch {
- /* best effort — an unreadable home directory reads as a new install */
+ const published = readPublishedState();
+ if (!published) {
+ return {
+ enabled: false,
+ state: "unset",
+ source: "unavailable",
+ pendingConsent: true,
+ };
}
return {
- enabled: preExisting,
- state: "unset",
- source: preExisting ? "legacy" : "unanswered",
- pendingConsent: true,
+ enabled: published.enabled,
+ state: published.state === "on" || published.state === "off" ? published.state : "unset",
+ source: typeof published.source === "string" ? published.source : "unavailable",
+ pendingConsent: published.pendingConsent === true,
};
}
diff --git a/packages/mcp/src/auto-save.ts b/packages/mcp/src/auto-save.ts
index 5701470b0..afa05b29c 100644
--- a/packages/mcp/src/auto-save.ts
+++ b/packages/mcp/src/auto-save.ts
@@ -27,21 +27,44 @@
* long-standing user, and start saving on its own — consent by never having
* been asked.
*
- * ── Where the answer comes from ────────────────────────────────────────────
- * Two mechanisms, both of which the package already uses, and no third one:
- *
- * 1. `MEMWAL_AUTO_SAVE` — the env-var surface every other option has
- * (`MEMWAL_NAMESPACE`, `MEMWAL_SERVER_URL`, ...). Set it in the client's
- * `env` block to pin one MCP server on or off. Setting it deliberately is
- * itself an answer, so it also stops the login prompt.
- * 2. `settings.json`, next to `credentials.json` — resolved by
- * `credsPath()`, so it inherits project-local-beats-global and the
- * `MEMWAL_CREDS_DIR` override for free, and a project that scopes its
- * credentials scopes its memory behaviour with them.
- *
- * The state has to live on disk rather than in configuration because the
- * plugin's hooks are separate processes: a hook is spawned by the client, not
- * by this package, and inherits none of the MCP server's `env` or argv.
+ * ── Where the answer lives: the trusted state dir, and nowhere else ─────────
+ * The answer is a decision about the HUMAN, not about a credentials directory.
+ * It used to be stored beside whichever `credentials.json` won resolution,
+ * which had two consequences, both wrong:
+ *
+ * 1. Approving a project's credentials (WALM-639) moved `settingsPath()`
+ * into the repository. A user who had already answered "off" globally got
+ * a directory with no answer in it, fell through to the legacy
+ * "credentials exist, so keep saving" rule, and had automatic memory
+ * switched back ON by an approval that was only ever about where writes
+ * go. The recorded "no" was never consulted there.
+ * 2. `setAutoSave` / `markConsentPending` then wrote the consent answer INTO
+ * the repository, where it could be committed and shipped to everyone
+ * else who cloned it.
+ *
+ * So consent now lives in the trusted state dir — `MEMWAL_CREDS_DIR` when it is
+ * set, else `~/.memwal` — the same directory auth.ts keeps the project-approval
+ * store in, and for the same reason: a record a repository can write is a
+ * repository approving itself. Project scoping of CREDENTIALS is unchanged;
+ * project scoping of CONSENT is gone.
+ *
+ * The legacy probe reads the trusted directory's own `credentials.json` for the
+ * same reason. "Has this person used MemWal before" must not be answerable by a
+ * file inside a checkout.
+ *
+ * ── How the hooks see it ───────────────────────────────────────────────────
+ * The plugin's hooks are separate processes: a hook is spawned by the client,
+ * not by this package, and inherits none of the MCP server's `env` or argv. The
+ * hooks used to re-derive the whole resolution in plain ESM, and the two
+ * implementations drifted — the hook side kept the pre-WALM-639 presence rule,
+ * so a committed `.memwal/credentials.json` (contents never parsed, only
+ * `existsSync`) turned automatic memory on for anyone who opened the repo.
+ *
+ * Re-deriving is the bug, so the hooks no longer do it. This module publishes
+ * the RESOLVED state to `auto-save-state.json` in the trusted state dir, and
+ * the hooks read only that. A repository cannot write there, the hooks do no
+ * resolution of their own, and a hook that cannot read the file fails safe
+ * (automatic memory off). See `plugin/scripts/lib/auto-save.mjs`.
*/
import { dirname, join } from "node:path";
import {
@@ -51,17 +74,39 @@ import {
readFileSync,
writeFileSync,
} from "node:fs";
-import { credsPath } from "./auth.js";
+import { projectApprovalsPath } from "./auth.js";
import { log } from "./logger.js";
/** Env var that pins automatic saving on or off for one MCP server process. */
export const AUTO_SAVE_ENV = "MEMWAL_AUTO_SAVE";
const SETTINGS_FILE = "settings.json";
+const CREDS_FILE = "credentials.json";
+/** Resolved state the plugin hooks read. Never written inside a repository. */
+const HOOK_STATE_FILE = "auto-save-state.json";
+/** Bumped if the hook-facing shape ever changes incompatibly. */
+const HOOK_STATE_VERSION = 1;
+
+/**
+ * Where records a repository must not be able to write are kept:
+ * `MEMWAL_CREDS_DIR` when set, else `~/.memwal`.
+ *
+ * Derived from `projectApprovalsPath()` rather than recomputed, so this file
+ * and auth.ts cannot drift apart the way the hook copy did. auth.ts owns the
+ * definition; this reads it back.
+ */
+function trustedStateDir(): string {
+ return dirname(projectApprovalsPath());
+}
-/** Where the answer is stored: beside whichever credentials file is in play. */
+/** Where the answer is stored. Outside every repository, on purpose. */
export function settingsPath(): string {
- return join(dirname(credsPath()), SETTINGS_FILE);
+ return join(trustedStateDir(), SETTINGS_FILE);
+}
+
+/** Where the resolved state is published for the plugin hooks to read. */
+export function hookStatePath(): string {
+ return join(trustedStateDir(), HOOK_STATE_FILE);
}
/** Shape of `settings.json`. Unknown keys are preserved on write. */
@@ -161,9 +206,11 @@ export function autoSaveStatus(): AutoSaveStatus {
}
// Unset. Which way it falls depends on whether this install was ever in a
- // position to have been asked — see the module comment.
+ // position to have been asked — see the module comment. The probe is the
+ // TRUSTED directory's credentials file, never a project one: "has this
+ // person used MemWal before" must not be answerable by a checkout.
const stamped = settings.autoSaveConsent === "pending";
- const preExisting = !stamped && existsSync(credsPath());
+ const preExisting = !stamped && existsSync(join(trustedStateDir(), CREDS_FILE));
return {
enabled: preExisting,
state: "unset",
@@ -176,8 +223,8 @@ export function autoSaveStatus(): AutoSaveStatus {
/**
* True when this process may save without being asked.
*
- * Read at call time, never cached: a login can move `credsPath()`, and the
- * answer can be written between calls in the same process.
+ * Read at call time, never cached: the answer can be written between calls in
+ * the same process.
*/
export function isAutoSaveEnabled(): boolean {
return autoSaveStatus().enabled;
@@ -188,6 +235,56 @@ export function isConsentPending(): boolean {
return autoSaveStatus().pendingConsent;
}
+/** The file the plugin hooks read. Contains no secret — only the answer. */
+export interface HookAutoSaveState {
+ version: number;
+ enabled: boolean;
+ state: AutoSaveState;
+ source: AutoSaveSource;
+ pendingConsent: boolean;
+ /** The settings file this was resolved from, for diagnosis only. */
+ settingsPath: string;
+ updatedAt: string;
+}
+
+/**
+ * Publish the resolved state where the plugin hooks can read it.
+ *
+ * This is the whole hook contract: the hooks do no resolution, consult no
+ * project directory and parse no credentials file — they read this one file out
+ * of the trusted state dir, or they fail safe. Called at every point the answer
+ * can change (`setAutoSave`, `markConsentPending`) and once at start-up, so a
+ * hook spawned in the same session as an MCP server sees the current answer.
+ *
+ * Best effort: a read-only or missing home directory must not stop the server
+ * from running, and a hook that finds no file already behaves as "off".
+ */
+export function publishHookState(): HookAutoSaveState | null {
+ const status = autoSaveStatus();
+ const payload: HookAutoSaveState = {
+ version: HOOK_STATE_VERSION,
+ enabled: status.enabled,
+ state: status.state,
+ source: status.source,
+ pendingConsent: status.pendingConsent,
+ settingsPath: status.path,
+ updatedAt: new Date().toISOString(),
+ };
+ const path = hookStatePath();
+ try {
+ mkdirSync(dirname(path), { recursive: true, mode: 0o700 });
+ writeFileSync(path, `${JSON.stringify(payload, null, 2)}\n`, { mode: 0o600 });
+ chmodSync(path, 0o600);
+ return payload;
+ } catch (err) {
+ log.warn("autosave.publish_failed", {
+ path,
+ error: err instanceof Error ? err.message : String(err),
+ });
+ return null;
+ }
+}
+
/**
* Record the answer, preserving any other keys already in the file. Clears the
* pending stamp — the question has been answered and must not be asked again.
@@ -196,6 +293,7 @@ export function setAutoSave(enabled: boolean): { path: string; enabled: boolean
const next: MemWalSettings = { ...readSettings(), autoSave: enabled };
delete next.autoSaveConsent;
const path = writeSettings(next);
+ publishHookState();
log.info("autosave.set", { enabled, path });
return { path, enabled };
}
@@ -214,6 +312,7 @@ export function markConsentPending(): void {
if (typeof settings.autoSave === "boolean") return;
if (settings.autoSaveConsent === "pending") return;
writeSettings({ ...settings, autoSaveConsent: "pending" });
+ publishHookState();
log.info("autosave.consent_pending", { path: settingsPath() });
}
diff --git a/packages/mcp/src/index.ts b/packages/mcp/src/index.ts
index defd74afb..4dd6f4c99 100644
--- a/packages/mcp/src/index.ts
+++ b/packages/mcp/src/index.ts
@@ -28,6 +28,7 @@ import {
autoSaveSummary,
markConsentPending,
pendingConsentNotice,
+ publishHookState,
setAutoSave,
AUTO_SAVE_ENV,
} from "./auto-save.js";
@@ -202,6 +203,15 @@ export async function main(argv: string[] = process.argv.slice(2)): Promise
+ publishHookState(),
+ );
}
/** Run one hook with a controlled environment and return its injected text. */
@@ -242,18 +253,22 @@ test("an unreadable or unparseable value is not consent", () => {
rmSync(dir, { recursive: true, force: true });
});
-test("the hook-side resolver answers identically to the compiled one", () => {
+test("the hook reads the published state rather than resolving anything itself", () => {
+ // The two implementations used to be hand-written mirrors and drifted
+ // apart, which is how a repo file came to switch automatic memory on
+ // (WALM-642). There is one resolver now: this asserts the hook reports what
+ // the server published, on every state, and that publishing is what moves
+ // it.
const dir = freshCredsDir();
withEnv({ MEMWAL_CREDS_DIR: dir, [AUTO_SAVE_ENV]: undefined }, () => {
- assert.equal(hookAutoSave.isAutoSaveEnabled(), isAutoSaveEnabled());
- assert.equal(hookAutoSave.settingsPath(), settingsPath());
+ assert.equal(hookAutoSave.hookStatePath(), hookStatePath());
- // ...on every state, not just the answered one.
markConsentPending();
assert.equal(hookAutoSave.isAutoSaveEnabled(), false);
assert.equal(hookAutoSave.autoSaveStatus().source, "unanswered");
setAutoSave(true);
+ assert.equal(hookAutoSave.isAutoSaveEnabled(), isAutoSaveEnabled());
assert.equal(hookAutoSave.isAutoSaveEnabled(), true);
assert.equal(hookAutoSave.autoSaveStatus().source, "settings");
assert.equal(hookAutoSave.autoSaveStatus().pendingConsent, false);
@@ -262,6 +277,39 @@ test("the hook-side resolver answers identically to the compiled one", () => {
assert.equal(hookAutoSave.isAutoSaveEnabled(), false);
assert.equal(hookAutoSave.autoSaveStatus().source, "env");
});
+
+ // An answer the server has not published yet is not one the hook may
+ // act on: it has no way to tell a stale file from a current one, so
+ // "cannot tell" has to read as off.
+ rmSync(hookStatePath(), { force: true });
+ assert.equal(hookAutoSave.isAutoSaveEnabled(), false);
+ assert.equal(hookAutoSave.autoSaveStatus().source, "unavailable");
+ assert.equal(hookAutoSave.autoSaveStatus().pendingConsent, true);
+ assert.equal(isAutoSaveEnabled(), true, "the server's own answer is unchanged");
+ });
+ rmSync(dir, { recursive: true, force: true });
+});
+
+test("a published state this build does not understand fails safe", () => {
+ const dir = freshCredsDir();
+ withEnv({ MEMWAL_CREDS_DIR: dir, [AUTO_SAVE_ENV]: undefined }, () => {
+ setAutoSave(true);
+ assert.equal(hookAutoSave.isAutoSaveEnabled(), true);
+ for (const corrupt of [
+ "{ not json",
+ JSON.stringify({ version: 99, enabled: true }),
+ JSON.stringify({ version: 1, enabled: "yes" }),
+ JSON.stringify({ version: 1 }),
+ "null",
+ ]) {
+ writeFileSync(hookStatePath(), corrupt);
+ assert.equal(
+ hookAutoSave.isAutoSaveEnabled(),
+ false,
+ `treated as consent: ${corrupt}`,
+ );
+ assert.equal(hookAutoSave.autoSaveStatus().source, "unavailable");
+ }
});
rmSync(dir, { recursive: true, force: true });
});
diff --git a/packages/mcp/test/auto-save-repo-trust.test.mjs b/packages/mcp/test/auto-save-repo-trust.test.mjs
new file mode 100644
index 000000000..53d128783
--- /dev/null
+++ b/packages/mcp/test/auto-save-repo-trust.test.mjs
@@ -0,0 +1,349 @@
+/**
+ * A repository cannot turn automatic memory on (WALM-642, review findings 1-2).
+ *
+ * Two holes, one root cause: the consent decision was being re-derived from
+ * whatever `.memwal` directory the working directory resolved to, by two
+ * separate implementations.
+ *
+ * 1. The hook-side resolver (`plugin/scripts/lib/auto-save.mjs`) still used
+ * the pre-WALM-639 presence rule — nearest project `credentials.json`
+ * wins, contents never parsed. A repo carrying a file whose entire content
+ * was `not even valid json` read as proof of a long-standing install and
+ * injected the full proactive-save rubric; adding a committed
+ * `settings.json` pinned `autoSave: true` outright and silenced the
+ * consent-pending warning too.
+ * 2. Server-side, `settingsPath()` followed `credsPath()`, so approving a
+ * project (WALM-639) moved the consent answer into the repo — where a
+ * recorded "no" was not consulted and a fresh directory fell through to
+ * "legacy, keep saving".
+ *
+ * THESE TESTS MUST NOT SET `MEMWAL_CREDS_DIR`. It is the trusted override: it
+ * short-circuits project resolution in both resolvers, which is exactly why the
+ * existing opt-in suite could not see either bug. Everything here drives a real
+ * HOME and a real working directory instead, and finding 1 is asserted by
+ * spawning the actual hook the client spawns.
+ */
+import { test } from "node:test";
+import assert from "node:assert/strict";
+import { spawnSync } from "node:child_process";
+import { mkdtempSync, mkdirSync, writeFileSync, rmSync, existsSync, realpathSync } from "node:fs";
+import { tmpdir } from "node:os";
+import { dirname, join, resolve } from "node:path";
+import { fileURLToPath } from "node:url";
+
+const __dirname = dirname(fileURLToPath(import.meta.url));
+const SCRIPTS = resolve(__dirname, "../plugin/scripts");
+
+const PRIVATE_KEY = "c".repeat(64);
+
+/**
+ * A session id nothing has seen before. `firstTime()` in lib/hook-io.mjs keeps
+ * a marker per (name, session) under the OS temp dir, so a fixed id makes the
+ * FIRST run of a test inject the full rubric and every run after it the
+ * one-line nudge.
+ */
+function freshSession(tag) {
+ return `${tag}-${Date.now()}-${Math.random().toString(16).slice(2)}`;
+}
+
+function makeCreds(overrides = {}) {
+ return {
+ delegatePrivateKey: PRIVATE_KEY,
+ delegatePublicKeyHex: "d".repeat(64),
+ delegateAddress: "0x" + "e".repeat(64),
+ walletAddress: "0x" + "f".repeat(64),
+ accountId: "0x" + "a".repeat(64),
+ packageId: "0x" + "1".repeat(64),
+ relayerUrl: "https://relayer.example",
+ createdAt: new Date(0).toISOString(),
+ version: 1,
+ ...overrides,
+ };
+}
+
+function writeJson(path, value) {
+ mkdirSync(dirname(path), { recursive: true });
+ writeFileSync(path, typeof value === "string" ? value : JSON.stringify(value));
+ return path;
+}
+
+/**
+ * A throwaway HOME and repo, with `MEMWAL_CREDS_DIR` cleared.
+ *
+ * Canonicalised because `homedir()` and `process.cwd()` both report resolved
+ * paths and `/tmp` is a symlink on macOS — an uncanonicalised HOME makes the
+ * project walk's "stop at the home directory" test miss.
+ */
+function sandbox(t) {
+ const home = realpathSync(mkdtempSync(join(tmpdir(), "memwal-trust-home-")));
+ const repo = realpathSync(mkdtempSync(join(tmpdir(), "memwal-trust-repo-")));
+ // A `.git` marker, because that is what a checkout has and what the project
+ // walk stops at.
+ mkdirSync(join(repo, ".git"), { recursive: true });
+ t.after(() => {
+ rmSync(home, { recursive: true, force: true });
+ rmSync(repo, { recursive: true, force: true });
+ });
+ return { home, repo };
+}
+
+/** Run one hook the way a client does: its own process, a cwd, and a HOME. */
+function runHookIn(script, { cwd, home, input = {}, env = {} }) {
+ const result = spawnSync(process.execPath, [join(SCRIPTS, script)], {
+ cwd,
+ input: JSON.stringify(input),
+ encoding: "utf8",
+ env: {
+ ...process.env,
+ HOME: home,
+ USERPROFILE: home,
+ MEMWAL_CREDS_DIR: "",
+ MEMWAL_AUTO_SAVE: "",
+ ...env,
+ },
+ });
+ assert.equal(result.status, 0, `${script} exited ${result.status}: ${result.stderr}`);
+ if (!result.stdout.trim()) return "";
+ return JSON.parse(result.stdout).hookSpecificOutput?.additionalContext ?? "";
+}
+
+// `MEMWAL_CREDS_DIR: ""` must actually clear the override, or every assertion
+// below is vacuous. Node keeps an empty string in the child's environment, and
+// both resolvers read it as unset because `""` is falsy — pinned here so a
+// future change to that check cannot quietly hollow this file out.
+test("an empty MEMWAL_CREDS_DIR does not stand in for a real one", async () => {
+ const previous = process.env.MEMWAL_CREDS_DIR;
+ process.env.MEMWAL_CREDS_DIR = "";
+ try {
+ const hook = await import("../plugin/scripts/lib/auto-save.mjs");
+ assert.ok(
+ hook.hookStatePath().startsWith(join(realpathSync(process.env.HOME ?? tmpdir()))) ||
+ hook.hookStatePath().includes(".memwal"),
+ "an empty override must fall back to the home directory",
+ );
+ } finally {
+ if (previous === undefined) delete process.env.MEMWAL_CREDS_DIR;
+ else process.env.MEMWAL_CREDS_DIR = previous;
+ }
+});
+
+// ── finding 1: the hook repro, on the real hook ─────────────────────────────
+
+test("a committed credentials file cannot turn the save rubric on", (t) => {
+ const { home, repo } = sandbox(t);
+ // The reporter's repro, byte for byte: the contents are never parsed, so
+ // the file did not even have to be credentials.
+ writeJson(join(repo, ".memwal", "credentials.json"), "not even valid json");
+ // The user's own install has been asked and has not answered.
+ writeJson(join(home, ".memwal", "settings.json"), { autoSaveConsent: "pending" });
+
+ const injected = runHookIn("on_user_prompt.mjs", {
+ cwd: repo,
+ home,
+ input: { session_id: freshSession("s1"), prompt: "I prefer pnpm" },
+ });
+
+ assert.doesNotMatch(
+ injected,
+ /call memwal_remember \(or memwal_remember_bulk for several\)/,
+ "a repo file injected the proactive-save rubric",
+ );
+ assert.match(injected, /Automatic saving is OFF|save ONLY what they ask you to save/);
+
+ // And the control: the same HOME from a clean directory says the same
+ // thing, which is the point — the repo changed nothing at all.
+ const clean = realpathSync(mkdtempSync(join(tmpdir(), "memwal-trust-clean-")));
+ t.after(() => rmSync(clean, { recursive: true, force: true }));
+ const control = runHookIn("on_user_prompt.mjs", {
+ cwd: clean,
+ home,
+ input: { session_id: freshSession("s2"), prompt: "I prefer pnpm" },
+ });
+ assert.equal(injected, control, "the repo steered the hook away from the control");
+});
+
+test("a committed settings.json cannot pin autoSave on, or silence the warning", (t) => {
+ const { home, repo } = sandbox(t);
+ writeJson(join(repo, ".memwal", "credentials.json"), "not even valid json");
+ writeJson(join(repo, ".memwal", "settings.json"), { autoSave: true });
+ writeJson(join(home, ".memwal", "settings.json"), { autoSaveConsent: "pending" });
+
+ const prompt = runHookIn("on_user_prompt.mjs", {
+ cwd: repo,
+ home,
+ input: { session_id: freshSession("s3"), prompt: "I prefer pnpm" },
+ });
+ assert.doesNotMatch(
+ prompt,
+ /call memwal_remember \(or memwal_remember_bulk for several\)/,
+ "a repo settings.json switched automatic saving on",
+ );
+
+ const start = runHookIn("on_session_start.mjs", { cwd: repo, home });
+ assert.match(start, /Automatic memory is OFF/, "a repo settings.json flipped the banner");
+ assert.doesNotMatch(start, /do not ask whether to save it/i);
+});
+
+test("the hook reads nothing from the working directory at all", (t) => {
+ // Stronger than the two repros: not "this particular file is ignored" but
+ // "there is no file a checkout can add". The published state is what
+ // decides, and it lives where a repo cannot write.
+ const { home, repo } = sandbox(t);
+ writeJson(join(home, ".memwal", "auto-save-state.json"), {
+ version: 1,
+ enabled: true,
+ state: "on",
+ source: "settings",
+ pendingConsent: false,
+ settingsPath: join(home, ".memwal", "settings.json"),
+ updatedAt: new Date().toISOString(),
+ });
+ // Every shape the old resolver would have followed, all saying "off".
+ writeJson(join(repo, ".memwal", "credentials.json"), makeCreds());
+ writeJson(join(repo, ".memwal", "settings.json"), { autoSave: false });
+ writeJson(join(repo, ".memwal", "auto-save-state.json"), {
+ version: 1,
+ enabled: false,
+ state: "off",
+ source: "settings",
+ pendingConsent: false,
+ settingsPath: join(repo, ".memwal", "settings.json"),
+ updatedAt: new Date().toISOString(),
+ });
+
+ const injected = runHookIn("on_user_prompt.mjs", {
+ cwd: repo,
+ home,
+ input: { session_id: freshSession("s4"), prompt: "I prefer pnpm" },
+ });
+ assert.match(
+ injected,
+ /call memwal_remember \(or memwal_remember_bulk for several\)/,
+ "the repo overrode the user's own published answer",
+ );
+});
+
+test("a hook with no published state saves nothing and still exits 0", (t) => {
+ const { home, repo } = sandbox(t);
+ assert.equal(existsSync(join(home, ".memwal", "auto-save-state.json")), false);
+ for (const script of ["on_session_start.mjs", "on_user_prompt.mjs", "on_post_tool.mjs"]) {
+ const text = runHookIn(script, {
+ cwd: repo,
+ home,
+ input: { session_id: freshSession("s5"), prompt: "a reasonably long prompt about pnpm" },
+ });
+ assert.doesNotMatch(
+ text,
+ /call memwal_remember \(or memwal_remember_bulk for several\)/,
+ `${script} saved on an unknown state`,
+ );
+ }
+});
+
+// ── finding 2: approving a project must not move the consent answer ─────────
+
+test("approving a project does not re-enable a declined auto-save", async (t) => {
+ const { home, repo } = sandbox(t);
+ const previous = {
+ home: process.env.HOME,
+ profile: process.env.USERPROFILE,
+ credsDir: process.env.MEMWAL_CREDS_DIR,
+ cwd: process.cwd(),
+ };
+ process.env.HOME = home;
+ process.env.USERPROFILE = home;
+ delete process.env.MEMWAL_CREDS_DIR;
+ delete process.env.MEMWAL_AUTO_SAVE;
+ process.chdir(repo);
+ t.after(() => {
+ process.chdir(previous.cwd);
+ process.env.HOME = previous.home;
+ process.env.USERPROFILE = previous.profile;
+ if (previous.credsDir === undefined) delete process.env.MEMWAL_CREDS_DIR;
+ else process.env.MEMWAL_CREDS_DIR = previous.credsDir;
+ });
+
+ // Resolved per call, but the modules are imported once per process, so a
+ // cache-busting query is what makes them see this HOME.
+ const stamp = `${Date.now()}-${Math.random()}`;
+ const auth = await import(`../dist/auth.js?walm642=${stamp}`);
+ const autoSave = await import(`../dist/auto-save.js?walm642=${stamp}`);
+
+ writeJson(join(home, ".memwal", "credentials.json"), makeCreds());
+ // The user answers "[2] Only save when I ask".
+ autoSave.setAutoSave(false);
+ const answerPath = autoSave.settingsPath();
+ assert.equal(answerPath, join(home, ".memwal", "settings.json"));
+ assert.equal(autoSave.isAutoSaveEnabled(), false);
+
+ // Later, in a team repo, they approve that repo's credentials — a decision
+ // about WHERE memory is written, and nothing else.
+ writeJson(
+ join(repo, ".memwal", "credentials.json"),
+ makeCreds({ accountId: "0x" + "b".repeat(64) }),
+ );
+ const approved = auth.approveProjectCreds();
+ assert.equal(approved.outcome, "approved");
+ assert.equal(auth.credsPath(), join(repo, ".memwal", "credentials.json"));
+
+ // The recorded "no" is still the answer...
+ assert.equal(autoSave.isAutoSaveEnabled(), false, "approval re-enabled a declined auto-save");
+ assert.equal(autoSave.autoSaveStatus().source, "settings");
+ assert.equal(autoSave.autoSaveStatus().state, "off");
+ // ...read from the same place it was written, not from the repo.
+ assert.equal(autoSave.settingsPath(), answerPath);
+ assert.equal(
+ existsSync(join(repo, ".memwal", "settings.json")),
+ false,
+ "the consent answer was written into the repository",
+ );
+
+ // Writing an answer while a project is approved must not put one there
+ // either — that file is committable.
+ autoSave.setAutoSave(true);
+ autoSave.markConsentPending();
+ assert.equal(
+ existsSync(join(repo, ".memwal", "settings.json")),
+ false,
+ "setAutoSave wrote the consent answer into the repository",
+ );
+ assert.equal(
+ existsSync(join(repo, ".memwal", "auto-save-state.json")),
+ false,
+ "the published hook state was written into the repository",
+ );
+ assert.equal(existsSync(join(home, ".memwal", "auto-save-state.json")), true);
+});
+
+test("a project credentials file is not evidence of a long-standing install", async (t) => {
+ // The legacy rule — "no answer, but credentials on disk, so keep saving" —
+ // has to be asked of the user's own install. Answered with a repo file it
+ // is just the presence rule again, wearing a different hat.
+ const { home, repo } = sandbox(t);
+ const previous = { home: process.env.HOME, profile: process.env.USERPROFILE, cwd: process.cwd() };
+ process.env.HOME = home;
+ process.env.USERPROFILE = home;
+ delete process.env.MEMWAL_CREDS_DIR;
+ delete process.env.MEMWAL_AUTO_SAVE;
+ process.chdir(repo);
+ t.after(() => {
+ process.chdir(previous.cwd);
+ process.env.HOME = previous.home;
+ process.env.USERPROFILE = previous.profile;
+ });
+
+ const stamp = `${Date.now()}-${Math.random()}`;
+ const auth = await import(`../dist/auth.js?walm642b=${stamp}`);
+ const autoSave = await import(`../dist/auto-save.js?walm642b=${stamp}`);
+
+ // No global credentials, no answer anywhere — a machine that has never used
+ // MemWal, opening a repo that carries an approved credentials file.
+ writeJson(join(repo, ".memwal", "credentials.json"), makeCreds());
+ auth.approveProjectCreds();
+ assert.equal(auth.credsPath(), join(repo, ".memwal", "credentials.json"));
+
+ const status = autoSave.autoSaveStatus();
+ assert.equal(status.enabled, false, "a repo file was read as a long-standing install");
+ assert.equal(status.source, "unanswered");
+ assert.equal(status.pendingConsent, true);
+});
From 7cbdb71ed68acd960e4d56190653c649b98230f5 Mon Sep 17 00:00:00 2001
From: Harry Phan
Date: Fri, 18 Sep 2026 13:10:11 +0700
Subject: [PATCH 080/132] fix(server): widen the credential patterns and bound
the URL rule (WALM-642)
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit
Four leaks the review found in the shape-based rules, and every one of them
was verified by running the string through `sanitizeFact`.
The credential-assignment gate was `(?:\b|(?<=[a-z]))`, which misses the
spelling credentials actually arrive in. `_` is a word character so `\b` never
fires between `_` and the keyword, and `_` is not `[a-z]`, so
`POSTGRES_PASSWORD=`, `AWS_SECRET_ACCESS_KEY=`, `DB_PASSWORD=`,
`X_AUTH_TOKEN:`, `SESSION_SECRET=` and `my_api_key=` all passed through
verbatim. A JSON object put a closing quote between the key and the colon, so
`{"password": "hunter2-prod-9xQ"}` matched nothing either. The gate now admits
`_`, `-` and digits, the separator group takes an optional closing quote, and
`access[_-]?keys?` joins the keyword list for the `ACCESS_KEY` half of
`AWS_SECRET_ACCESS_KEY`.
`hasAdjacentCredentialLabel` was consulted only for hex runs, and the entropy
rule demands 64+ characters, so a 40-character AWS secret access key came back
unchanged with three label words in front of it — in a key pair that meant the
non-secret AKIA id was redacted while the secret half survived. The same gate
now covers a 20+ character base64/base64url run, with a shape guard so an
ordinary long word next to "secret" is not mistaken for key material. The
label remains the only discriminator: bare SHAs, blob ids and Sui object ids
still pass, pinned from both sides.
`URL_USERINFO`'s scheme repeat was unbounded and had to be tried at every
start offset — 30 KB took 317 ms, 60 KB 1254 ms, 120 KB 4814 ms on one core,
in front of every other caller on the sidecar's single thread. Capped at 30
characters, 120 KB is now 14 ms. `memwal_analyze`'s schema had no maximum
either; it takes 200k characters now.
The same rule's password group excluded `/` but not `?` or `=`, so
`https://app.example.com:8443?owner=alice@corp.com` came out as
`https://[redacted:url-credentials]@corp.com` — host and port destroyed,
`corp.com` promoted to hostname, the mangled fact written to append-only
storage and the agent told a credential had been removed when there was none.
The userinfo groups now exclude `?`, `=` and `&`.
The timing test asserts the shape rather than a number, so a pattern going
superlinear again fails the suite without making it flaky.
---
.../mcp/__tests__/secret-redaction.test.ts | 144 ++++++++++++++++++
.../__tests__/write-path-redaction.test.ts | 36 +++++
services/server/scripts/mcp/tools/analyze.ts | 17 ++-
.../server/scripts/mcp/tools/redaction.ts | 139 +++++++++++++++--
4 files changed, 321 insertions(+), 15 deletions(-)
diff --git a/services/server/scripts/mcp/__tests__/secret-redaction.test.ts b/services/server/scripts/mcp/__tests__/secret-redaction.test.ts
index fea778e60..d21243b9d 100644
--- a/services/server/scripts/mcp/__tests__/secret-redaction.test.ts
+++ b/services/server/scripts/mcp/__tests__/secret-redaction.test.ts
@@ -330,3 +330,147 @@ test("the notices name the kind and never the value", () => {
test("nothing removed means nothing said", () => {
assert.equal(redactionNotice([], 0), "");
});
+
+/* ───────────────────────────────────────────────────────────────────────────
+ * Review findings on the WALM-642 branch. Each of these leaked, verbatim,
+ * through `sanitizeFact` before the fix beside it.
+ * ------------------------------------------------------------------------ */
+
+// ── finding 3: the credential-assignment gate missed whole spellings ────────
+
+test("SCREAMING_SNAKE and quoted-JSON credential names are assignments too", () => {
+ // `_` is a word character, so `\b` never fired between `_` and the
+ // keyword, and `_` is not `[a-z]` so the camelCase lookbehind did not
+ // either. That left the spelling credentials actually arrive in — a pasted
+ // env file or shell export — completely unguarded.
+ for (const [text, secret] of [
+ ["POSTGRES_PASSWORD=hunter2", "hunter2"],
+ [
+ "AWS_SECRET_ACCESS_KEY=wJalrXUtnFEMI/K7MDENG/bPxRfiCYEXAMPLEKEY",
+ "wJalrXUtnFEMI/K7MDENG/bPxRfiCYEXAMPLEKEY",
+ ],
+ ["DB_PASSWORD=hunter2SuperSecret", "hunter2SuperSecret"],
+ ["X_AUTH_TOKEN: abcd1234", "abcd1234"],
+ ["SESSION_SECRET=abcd1234efgh5678", "abcd1234efgh5678"],
+ ["my_api_key=abcd1234efgh5678", "abcd1234efgh5678"],
+ ] as Array<[string, string]>) {
+ const out = sanitizeFact(text);
+ assert.ok(out.changed, `not redacted at all: ${text}`);
+ assert.ok(!out.text.includes(secret), `the secret survived: ${text}`);
+ }
+
+ // A JSON object puts a closing quote between the key and the colon, which
+ // the old separator group demanded come immediately after the keyword.
+ const json = sanitizeFact(
+ 'Save my config: {"username": "alice", "password": "hunter2-prod-9xQ"} for staging',
+ );
+ assert.equal(json.refusal, undefined);
+ assert.ok(!json.text.includes("hunter2-prod-9xQ"), "the JSON password survived");
+ assert.ok(json.text.includes("alice"), "the username is not a credential");
+ assert.ok(json.text.includes("for staging"), "the fact was lost with the password");
+});
+
+test("widening the gate did not widen it onto ordinary prose", () => {
+ // The lookbehind now admits `_`, `-` and digits. These are the sentences
+ // that must not start matching because of it.
+ for (const fact of [
+ "My password manager is 1Password and I rotate keys every quarter.",
+ "The API key for that service is stored in Vault, not in the repo.",
+ "My creds live in ~/.memwal/credentials.json and the fix landed in 4f2b8c1e9a7d6f5c4b3a29180716253443219876",
+ "We renamed the secret_store module to vault_client last sprint.",
+ "Token bucket rate limiting is what the relayer uses.",
+ ]) {
+ const out = sanitizeFact(fact);
+ assert.equal(out.text, fact, `changed a clean fact: ${fact}`);
+ assert.equal(out.changed, false);
+ }
+});
+
+// ── finding 5: the label rule only ever looked at hex ───────────────────────
+
+test("a labelled secret that is not hex is key material too", () => {
+ // 40 characters, base64-ish, three label words in front of it — and it came
+ // back completely unchanged, because `hasAdjacentCredentialLabel` was
+ // consulted only for HEX_RUN and the entropy rule demands 64+.
+ const AWS_SECRET = "wJalrXUtnFEMI/K7MDENG/bPxRfiCYEXAMPLEKEY";
+ assertRedacted(
+ `My AWS secret access key is ${AWS_SECRET} for prod`,
+ AWS_SECRET,
+ "labelled-key-material",
+ ["My AWS secret access key", "for prod"],
+ );
+
+ // The pair, which is the shape it arrives in: the non-secret id was being
+ // redacted by the vendor rule while the secret half survived.
+ const pair = sanitizeFact(
+ `AWS creds for staging: AKIAIOSFODNN7EXAMPLE and the secret key is ${AWS_SECRET}`,
+ );
+ assert.ok(!pair.text.includes(AWS_SECRET), "the secret half of the pair survived");
+ assert.ok(!pair.text.includes("AKIAIOSFODNN7EXAMPLE"));
+});
+
+test("the label is still the only discriminator — bare runs pass", () => {
+ // The whole design: widening the SHAPE the rule can see must not weaken the
+ // gate in front of it, or every identifier this product exists to remember
+ // starts disappearing.
+ for (const fact of [
+ "The blob landed as blob_id=Xj9vKq2mP7nR4tW8yB1cE5gH0dF3sA6uZ2xN8qL4kM7",
+ "Pin the build to commit 4f2b8c1e9d7a3f5b6c0e2d4a8b1f3c5e7d9a0b2c",
+ "My Sui package id is 0xe80f2feec1c139616a86c9f71210152e2a7ca552b20841f2e192f99f75864437",
+ "My delegatePublicKeyHex is 4f3c2b1a9e8d7c6b5a4f3e2d1c0b9a8f7e6d5c4b3a2f1e0d9c8b7a6f5e4d3c2b and it is safe to share",
+ "The build id is 20260918T0930Z-linux-arm64-release-candidate",
+ ]) {
+ const out = sanitizeFact(fact);
+ assert.equal(out.text, fact, `redacted a legitimate identifier: ${fact}`);
+ assert.equal(out.changed, false);
+ }
+});
+
+// ── findings 6 + 7: the URL rule was quadratic, and mangled ordinary URLs ───
+
+test("an ordinary URL with a query string is not userinfo", () => {
+ // The password group excluded `/` and whitespace but not `?` or `=`, so
+ // this matched with `app.example.com` as the user and `8443?owner=alice` as
+ // the password: host and port destroyed, `corp.com` promoted to hostname,
+ // and a notice claiming a credential had been removed when there was none.
+ const fact =
+ "Our dashboard is at https://app.example.com:8443?owner=alice@corp.com and we deploy Fridays";
+ const out = sanitizeFact(fact);
+ assert.equal(out.text, fact, "an ordinary URL was mangled");
+ assert.equal(out.changed, false);
+ assert.equal(out.count, 0, "a redaction was reported where none happened");
+
+ // ...while a real connection string is untouched by the narrowing.
+ const real = sanitizeFact("staging is postgres://admin:hunter2@db.internal:5432/app");
+ assert.ok(!real.text.includes("hunter2"));
+ assert.ok(real.text.includes("db.internal:5432/app"), "the host was lost");
+});
+
+test("a long passage is screened in linear time", () => {
+ // `URL_USERINFO`'s scheme repeat was unbounded, so it was tried and
+ // abandoned at every start offset: 30 KB took 317 ms, 60 KB 1254 ms and
+ // 120 KB 4814 ms. `memwal_analyze` takes a whole transcript, the sidecar is
+ // single-threaded and a tools/call times out at 60 s, so a long paste was a
+ // stall for every other caller too.
+ //
+ // The budget is deliberately loose — this is a guard against a quadratic
+ // pattern coming back, not a benchmark, and CI machines are noisy. The
+ // shape is what matters: 4x the input must not be 16x the time.
+ const worst = "a".repeat(120 * 1024);
+ const started = Date.now();
+ sanitizeFact(worst);
+ const elapsed = Date.now() - started;
+ assert.ok(
+ elapsed < 1000,
+ `120 KB took ${elapsed} ms — a redaction pattern has gone superlinear again`,
+ );
+
+ const small = "a".repeat(30 * 1024);
+ const t0 = Date.now();
+ sanitizeFact(small);
+ const smallMs = Math.max(Date.now() - t0, 1);
+ assert.ok(
+ elapsed / smallMs < 12,
+ `120 KB/30 KB ratio was ${(elapsed / smallMs).toFixed(1)}x — that is quadratic, not linear`,
+ );
+});
diff --git a/services/server/scripts/mcp/__tests__/write-path-redaction.test.ts b/services/server/scripts/mcp/__tests__/write-path-redaction.test.ts
index 48d2341da..ad9f10113 100644
--- a/services/server/scripts/mcp/__tests__/write-path-redaction.test.ts
+++ b/services/server/scripts/mcp/__tests__/write-path-redaction.test.ts
@@ -341,3 +341,39 @@ test("the idempotency key is derived from what is actually written", async (t) =
assert.equal(seen[0], seen[1], "the same fact must derive the same key twice");
assert.ok(!seen[0].includes(PASSWORD));
});
+
+/* ───────────────────────────────────────────────────────────────────────────
+ * Review findings on the WALM-642 branch, asserted where it counts: at what
+ * the SDK was handed.
+ * ------------------------------------------------------------------------ */
+
+test("memwal_analyze will not take an unbounded passage", async (t) => {
+ // The schema was `z.string().min(1)` with no maximum while the tool is
+ // documented as accepting a whole transcript, so the work the sidecar's
+ // single thread did in front of every other caller was the caller's choice.
+ const forwarded: Forwarded = { remember: [], bulk: [], analyze: [] };
+ const client = await clientFor(sessionWith(forwarded), t);
+ const result = await client.callTool({
+ name: "memwal_analyze",
+ arguments: { text: "a".repeat(200_001) },
+ });
+ assert.equal((result as { isError?: boolean }).isError, true);
+ assert.deepEqual(forwarded.analyze, [], "an over-long passage was forwarded");
+});
+
+test("an ordinary URL is not mangled on the way to the SDK", async (t) => {
+ // The userinfo groups excluded `/` but not `?` or `=`, so this came out as
+ // `https://[redacted:url-credentials]@corp.com` — host and port destroyed,
+ // the mangled fact written to append-only storage, and the agent told a
+ // credential had been removed when there was none.
+ const forwarded: Forwarded = { remember: [], bulk: [], analyze: [] };
+ const client = await clientFor(sessionWith(forwarded), t);
+ const fact =
+ "Our dashboard is at https://app.example.com:8443?owner=alice@corp.com and we deploy Fridays";
+ const result = await client.callTool({
+ name: "memwal_remember",
+ arguments: { text: fact },
+ });
+ assert.deepEqual(forwarded.remember, [fact]);
+ assert.doesNotMatch(textOf(result), /redacted|credential span/i);
+});
diff --git a/services/server/scripts/mcp/tools/analyze.ts b/services/server/scripts/mcp/tools/analyze.ts
index 418135da2..cd4768507 100644
--- a/services/server/scripts/mcp/tools/analyze.ts
+++ b/services/server/scripts/mcp/tools/analyze.ts
@@ -14,12 +14,27 @@ import {
withWaitDeadline,
} from "./remember-wait.js";
+/**
+ * Ceiling on one passage, in characters.
+ *
+ * This schema had a `.min(1)` and no maximum while the tool is documented as
+ * taking a whole transcript, which made the input length entirely the caller's
+ * choice — and the redactor runs over every character of it on the sidecar's
+ * single thread, in front of every other in-flight tool call. The regexes are
+ * linear now (see `URL_USERINFO`), so this is a backstop rather than the fix:
+ * 200k characters screens in tens of milliseconds, is far more than any real
+ * transcript, and is well under what the extractor LLM behind `/api/analyze`
+ * would accept anyway.
+ */
+const MAX_ANALYZE_CHARS = 200_000;
+
const ANALYZE_INPUT = {
text: z
.string()
.min(1)
+ .max(MAX_ANALYZE_CHARS)
.describe(
- "Conversation transcript, note, or arbitrary text from which to extract memorable facts. Credential shapes (passwords, API keys, tokens, private keys, seed phrases, auth headers, URLs with an embedded user:password) are stripped from this text before it is sent for extraction, so no secret reaches the extractor or storage."
+ `Conversation transcript, note, or arbitrary text from which to extract memorable facts (max ${MAX_ANALYZE_CHARS} characters). Credential shapes (passwords, API keys, tokens, private keys, seed phrases, auth headers, URLs with an embedded user:password) are stripped from this text before it is sent for extraction, so no secret reaches the extractor or storage. A span the user asked not to save, or that is pasted third-party material, is dropped on its own — the rest of the passage is still extracted from.`
),
namespace: z
.string()
diff --git a/services/server/scripts/mcp/tools/redaction.ts b/services/server/scripts/mcp/tools/redaction.ts
index 6e82c0db0..d210cffb8 100644
--- a/services/server/scripts/mcp/tools/redaction.ts
+++ b/services/server/scripts/mcp/tools/redaction.ts
@@ -113,8 +113,27 @@ const PEM_OPEN = /-----BEGIN [A-Z0-9 ]*PRIVATE KEY-----[\s\S]*/g;
*
* Only the userinfo is replaced: scheme, host, port and path are the part of a
* connection string worth remembering.
+ *
+ * Two bounds on this pattern, both load-bearing:
+ *
+ * - The scheme repeat is CAPPED. Unbounded (`[A-Za-z0-9+.-]*`) it had to be
+ * tried and abandoned at every start offset, which is quadratic in the
+ * input: 30 KB took 317 ms, 60 KB 1254 ms and 120 KB 4814 ms on one core.
+ * `memwal_analyze` forwards a whole transcript and the sidecar is
+ * single-threaded, so a long paste was a stall for every other caller, not
+ * just for itself. 30 characters is longer than any registered scheme.
+ * - The userinfo groups exclude `?`, `=` and `&`, not just `/` and
+ * whitespace. Without that,
+ * `https://app.example.com:8443?owner=alice@corp.com` matched with
+ * `app.example.com` as the user and `8443?owner=alice` as the password:
+ * the host and port were destroyed, `corp.com` was promoted to hostname,
+ * the mangled fact was written to append-only storage, and the result told
+ * the agent a credential had been removed when there was none. A query
+ * string cannot appear before the userinfo in a real URL, so excluding
+ * them costs nothing but a password that literally contains one.
*/
-const URL_USERINFO = /([A-Za-z][A-Za-z0-9+.-]*:\/\/)([^\s/@:]+):([^\s/@]+)@/g;
+const URL_USERINFO =
+ /([A-Za-z][A-Za-z0-9+.-]{0,30}:\/\/)([^\s/@:?=&]+):([^\s/@?=&]+)@/g;
/**
* Authorization / cookie headers, value dropped, header name kept.
@@ -134,18 +153,37 @@ const COOKIE_HEADER =
/**
* `key=value` / `key: value` where the key names a credential.
*
- * The separator must follow the keyword immediately, which is what keeps
- * ordinary prose out: "my password manager is 1Password" has no separator after
- * "password" and does not match.
+ * The separator must still follow the keyword (allowing one closing quote),
+ * which is what keeps ordinary prose out: "my password manager is 1Password"
+ * has no separator after "password" and does not match.
+ *
+ * ── What the gate has to let in ────────────────────────────────────────────
+ * The keyword may start the identifier, sit mid-identifier, or follow a
+ * separator character, so the left gate is `\b` OR a lookbehind covering
+ * letters, digits, `_` and `-`:
+ *
+ * - `(?<=[a-z])` is what makes `delegatePrivateKey` match — MemWal's own
+ * worst secret, where `PrivateKey` sits mid-identifier and a plain `\b`
+ * never fires.
+ * - `_` is a WORD character, so `\b` does not fire between `_` and `P`
+ * either, and `_` is not `[a-z]`. That left the entire SCREAMING_SNAKE
+ * namespace open: `POSTGRES_PASSWORD=`, `AWS_SECRET_ACCESS_KEY=`,
+ * `DB_PASSWORD=`, `X_AUTH_TOKEN:`, `SESSION_SECRET=` and `my_api_key=`
+ * all passed through verbatim — the exact spelling a credential arrives in
+ * when someone pastes an env file or a shell export.
*
- * `(?<=[a-z])` alongside `\b` is what makes a camelCase field name match. A
- * plain `\b` anchors only at a non-word character, so the keyword had to start
- * the identifier — and MemWal's own worst secret is spelled
- * `delegatePrivateKey`, where `PrivateKey` sits mid-identifier and was
- * therefore invisible to this rule.
+ * And a closing quote may sit between the keyword and the separator, because
+ * that is what a JSON object looks like: `{"username": "alice", "password":
+ * "hunter2-prod-9xQ"}` matched nothing at all before. The quote is captured
+ * with the separator and written back, so the shape of the line survives; only
+ * the value is replaced.
+ *
+ * `access[_-]?keys?` is in the list for `AWS_SECRET_ACCESS_KEY`: `secret` is
+ * there, but the separator does not follow it, and no alternative covered the
+ * `ACCESS_KEY` that does precede the `=`.
*/
const CREDENTIAL_ASSIGNMENT =
- /(?:\b|(?<=[a-z]))(passwords?|passwd|pwd|passphrases?|api[_-]?keys?|apikeys?|secret[_-]?keys?|client[_-]?secrets?|secrets?|access[_-]?tokens?|refresh[_-]?tokens?|auth[_-]?tokens?|bearer[_-]?tokens?|tokens?|private[_-]?keys?|credentials?)(\s*[:=]\s*)("[^"\n]*"|'[^'\n]*'|`[^`\n]*`|[^\s,;]+)/gi;
+ /(?:\b|(?<=[a-z0-9_-]))(passwords?|passwd|pwd|passphrases?|api[_-]?keys?|apikeys?|secret[_-]?keys?|access[_-]?keys?|client[_-]?secrets?|secrets?|access[_-]?tokens?|refresh[_-]?tokens?|auth[_-]?tokens?|bearer[_-]?tokens?|tokens?|private[_-]?keys?|credentials?)(["'`]?\s*[:=]\s*)("[^"\n]*"|'[^'\n]*'|`[^`\n]*`|[^\s,;]+)/gi;
/**
* Vendor-prefixed keys. Each prefix is issued by exactly one service and never
@@ -237,21 +275,82 @@ const HEX_CREDENTIAL_LABEL =
const HEX_LABEL_AFTER =
/^[^A-Za-z0-9]{0,4}(?:is|was)?[^A-Za-z0-9]{0,4}(?:my|the|our|his|her|their)?[^A-Za-z0-9]{0,4}(?:delegate[\s_-]*)?(?:private[\s_-]*key|secret[\s_-]*key|seed|mnemonic|passphrase|api[\s_-]*key)/i;
-/** True when a credential word sits within a few tokens of [start, end). */
+/** Placeholders this module writes, for stripping out of a context window. */
+const PLACEHOLDER_RUN = /\[redacted:[a-z-]+\]/g;
+
+/**
+ * True when the span at [start, end) is itself part of a placeholder an earlier
+ * rule already wrote.
+ *
+ * `labelled-key-material` and `credential-assignment` are both 21 characters of
+ * exactly the alphabet {@link OPAQUE_RUN} scans for, so without this a run of
+ * redactions would start redacting its own output.
+ */
+function isInsidePlaceholder(text: string, start: number): boolean {
+ const before = text.slice(Math.max(0, start - 10), start);
+ return /\[redacted:$/.test(before);
+}
+
+/**
+ * True when a credential word sits within `window` characters of [start, end).
+ *
+ * `window` is a parameter because the batch screen (see
+ * {@link sanitizeFactBatch}) looks across entry boundaries, where the label and
+ * the value are further apart than "a few tokens" by construction.
+ */
function hasAdjacentCredentialLabel(
text: string,
start: number,
end: number,
+ window: number = HEX_LABEL_WINDOW,
): boolean {
// Placeholders left by earlier rules carry the words "secret" and "key",
// so a run of redactions would otherwise start labelling its own
// neighbours — and the neighbour after `private_key=[redacted:...]` is
// exactly the kind of bare SHA this rule must not touch.
const before = text
- .slice(Math.max(0, start - HEX_LABEL_WINDOW), start)
- .replace(/\[redacted:[a-z-]+\]/g, " ");
+ .slice(Math.max(0, start - window), start)
+ .replace(PLACEHOLDER_RUN, " ");
if (HEX_CREDENTIAL_LABEL.test(before)) return true;
- return HEX_LABEL_AFTER.test(text.slice(end, end + HEX_LABEL_WINDOW));
+ return HEX_LABEL_AFTER.test(
+ text.slice(end, end + HEX_LABEL_WINDOW).replace(PLACEHOLDER_RUN, " "),
+ );
+}
+
+/**
+ * The same label gate, for key material that is not hex.
+ *
+ * `HEX_RUN` only ever looked at hex, and `HIGH_ENTROPY_CANDIDATE` demands 64+
+ * characters, so everything in between passed with a label in front of it:
+ * `My AWS secret access key is wJalrXUtnFEMI/K7MDENG/bPxRfiCYEXAMPLEKEY for
+ * prod` came back completely unchanged, three label words and all. In an AWS
+ * key pair that meant the non-secret `AKIA...` id was redacted by the vendor
+ * rule while the 40-character secret half survived — the wrong half of the
+ * pair, every time.
+ *
+ * Same discriminator as the hex rule, for the same reason: the LABEL, never the
+ * shape. A bare 43-character blob id, a commit SHA or a Sui object id with no
+ * credential word near it still passes, which is the property the rest of this
+ * file is built around.
+ */
+const OPAQUE_RUN = /[A-Za-z0-9+/_=-]{20,}/g;
+
+/**
+ * A shape guard on top of the label, so an ordinary long word next to the word
+ * "secret" is not mistaken for key material.
+ *
+ * A generated credential is mixed case, or carries digits, or carries base64
+ * padding and separators; `my_api_key_rotation_policy` is none of those. This
+ * is not an entropy test and is not trying to be one — the label is still what
+ * decides. It only keeps prose out.
+ */
+function looksLikeOpaqueToken(token: string): boolean {
+ if (token.length < 20) return false;
+ const hasLower = /[a-z]/.test(token);
+ const hasUpper = /[A-Z]/.test(token);
+ const hasDigit = /[0-9]/.test(token);
+ const hasSymbol = /[+/=]/.test(token);
+ return (hasLower && hasUpper) || hasDigit || hasSymbol;
}
/**
@@ -388,6 +487,18 @@ export function sanitizeFact(input: string): SanitizedText {
: match,
);
+ // Everything else a label makes into key material: the 20+ character
+ // base64/base64url runs that are too short for the entropy rule and not hex
+ // enough for HEX_RUN. Same window, same gate, same asymmetry — a bare run
+ // with nothing calling it a key still passes.
+ text = text.replace(OPAQUE_RUN, (match: string, offset: number, whole: string) => {
+ if (isInsidePlaceholder(whole, offset)) return match;
+ if (!looksLikeOpaqueToken(match)) return match;
+ return hasAdjacentCredentialLabel(whole, offset, offset + match.length)
+ ? hit("labelled-key-material")
+ : match;
+ });
+
text = text.replace(HIGH_ENTROPY_CANDIDATE, (token: string) =>
looksLikeSecretBlob(token) ? hit("high-entropy-secret") : token,
);
From 5cfe1d01f9257a98b6f070a6f694283f5370dbd9 Mon Sep 17 00:00:00 2001
From: Harry Phan
Date: Fri, 18 Sep 2026 13:10:23 +0700
Subject: [PATCH 081/132] fix(server): scope the screen to the batch and the
span, not one string (WALM-642)
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit
Two findings with one shape: the predicates in redaction.ts are whole-string,
written for one fact, and both bulk and analyze hand them something else.
Every label-gated rule searches a window inside ONE string, so
`memwal_remember_bulk` could be walked past by splitting the label from the
value:
["my delegate private key for the mainnet account",
"4f3c...789"]
Both entries passed untouched while the same words as one string were
correctly redacted — and that value is MemWal's own delegate private key, the
worst thing this product can leak. An agent that paraphrases a user across two
entries is enough. `sanitizeFactBatch` runs the per-entry pass first, so the
refusal granularity that already works is unchanged, then screens each
surviving value against the entries either side of it; an entry that is
nothing but a value is screened against the whole batch. The gate is still the
label, so a batch with no credential word in it is byte-for-byte untouched.
`memwal_analyze` had the mirror problem. `NO_SAVE_DIRECTIVE`, `OFF_THE_RECORD`
and `isPastedContent` applied to a forty-turn transcript meant one "don't save
this part" line returned `refusal` and `text: ""` — all forty turns discarded,
with "do not retry this text" and no `isError`, so the client saw a successful
call that had saved nothing. A transcript inside a ``` fence refused as pasted
content, which is the input the tool documents as canonical.
`sanitizePassage` scopes the refusal per span the way bulk scopes it per
entry: the offending span is dropped, named by line and reason in the reply,
and the rest is extracted from. A directive takes its immediate neighbours
within its own paragraph with it, because "My bank PIN is 4821" on one line
and "don't save this" on the next is how people write it and dropping only the
second is the one outcome worse than dropping too much. An untagged outer
fence holding four or more lines is unwrapped as a transcript; a tagged fence,
a short fence, and a fence inside the passage stay pastes. A passage with
nothing usable left is still refused, and now says so with `isError`.
---
.../mcp/__tests__/secret-redaction.test.ts | 155 ++++++++
.../__tests__/write-path-redaction.test.ts | 117 ++++++
services/server/scripts/mcp/tools/analyze.ts | 28 +-
.../server/scripts/mcp/tools/redaction.ts | 340 ++++++++++++++++++
.../server/scripts/mcp/tools/remember-bulk.ts | 14 +-
5 files changed, 646 insertions(+), 8 deletions(-)
diff --git a/services/server/scripts/mcp/__tests__/secret-redaction.test.ts b/services/server/scripts/mcp/__tests__/secret-redaction.test.ts
index d21243b9d..3d86d30e5 100644
--- a/services/server/scripts/mcp/__tests__/secret-redaction.test.ts
+++ b/services/server/scripts/mcp/__tests__/secret-redaction.test.ts
@@ -12,8 +12,11 @@ import test from "node:test";
import {
sanitizeFact,
+ sanitizeFactBatch,
+ sanitizePassage,
redactionNotice,
refusalNotice,
+ droppedSpanNotice,
type RedactionKind,
} from "../tools/redaction.js";
@@ -474,3 +477,155 @@ test("a long passage is screened in linear time", () => {
`120 KB/30 KB ratio was ${(elapsed / smallMs).toFixed(1)}x — that is quadratic, not linear`,
);
});
+
+// ── finding 4: a secret split across bulk entries ──────────────────────────
+
+test("a credential split across two bulk entries is still caught", () => {
+ // The reporter's case. Every label-gated rule searches a window inside ONE
+ // string, so putting the label in one entry and the value in the next was
+ // a way past all of them — including the rule that exists specifically for
+ // MemWal's own delegate private key.
+ const SEED = "4f3c2b1a9e8d7c6b5a4f3e2d1c0b9a8f7e6d5c4b3a2f1e0d9c8b7a6f5e4d3c2b";
+ const split = sanitizeFactBatch([
+ "my delegate private key for the mainnet account",
+ SEED,
+ ]);
+ assert.ok(
+ !split.map((r) => r.text).join("\n").includes(SEED),
+ "the delegate private key survived being split across entries",
+ );
+ assert.equal(split[1].refusal, "credential-only", "a bare key must be dropped, not stored");
+ assert.equal(split[0].refusal, undefined, "the label entry is a fact and should survive");
+
+ // The same value with filler words around it, and with the label AFTER it.
+ for (const facts of [
+ ["my delegate private key for the mainnet account", `the value to use is ${SEED} for that account`],
+ [SEED, "that is my delegate private key"],
+ ["I set up a second laptop", "my delegate private key is below", SEED],
+ ]) {
+ const out = sanitizeFactBatch(facts);
+ assert.ok(
+ !out.map((r) => r.text).join("\n").includes(SEED),
+ `the key survived: ${JSON.stringify(facts)}`,
+ );
+ }
+});
+
+test("the batch screen does not fire without a label anywhere in it", () => {
+ // Cross-entry awareness widens what counts as adjacent, which is exactly
+ // the kind of change that starts eating identifiers. The gate is still the
+ // label: a batch with none must be byte-for-byte untouched.
+ const facts = [
+ "The release digest is 4f3c2b1a9e8d7c6b5a4f3e2d1c0b9a8f7e6d5c4b3a2f1e0d9c8b7a6f5e4d3c2b",
+ "My Sui package id is 0xe80f2feec1c139616a86c9f71210152e2a7ca552b20841f2e192f99f75864437",
+ "The blob landed as blob_id=Xj9vKq2mP7nR4tW8yB1cE5gH0dF3sA6uZ2xN8qL4kM7",
+ "I always use pnpm",
+ ];
+ const out = sanitizeFactBatch(facts);
+ assert.deepEqual(out.map((r) => r.text), facts);
+ assert.deepEqual(out.map((r) => r.changed), [false, false, false, false]);
+});
+
+test("per-entry refusal granularity survives the batch screen", () => {
+ // The property that already worked: one bad entry is dropped and the rest
+ // of the batch still lands.
+ const out = sanitizeFactBatch([
+ "I always use pnpm",
+ "sk-abcdefghijklmnopqrstuvwxyz0123",
+ "Please don't save this one",
+ "Deploy on Thursdays",
+ ]);
+ assert.equal(out[0].refusal, undefined);
+ assert.equal(out[0].text, "I always use pnpm");
+ assert.equal(out[1].refusal, "credential-only");
+ assert.equal(out[2].refusal, "no-save-directive");
+ assert.equal(out[3].text, "Deploy on Thursdays");
+});
+
+// ── finding 8: one stray line discarded a whole transcript ─────────────────
+
+const TRANSCRIPT = [
+ "user: I always use pnpm for every project",
+ "assistant: noted",
+ "user: my staging box is app.example.com:8443",
+ "assistant: got it",
+ "user: my bank PIN is 4821 - don't save this part",
+ "assistant: understood",
+ "user: I deploy on Thursdays and never on Fridays",
+ "assistant: makes sense",
+].join("\n");
+
+test("one do-not-save line drops its span, not the whole transcript", () => {
+ const out = sanitizePassage(TRANSCRIPT);
+ assert.equal(out.refusal, undefined, "the whole passage was discarded again");
+ assert.ok(!out.text.includes("4821"), "the thing they asked not to save was kept");
+ assert.ok(!out.text.includes("don't save this part"));
+ // Everything else survived, on both sides of the dropped span.
+ assert.ok(out.text.includes("I always use pnpm for every project"));
+ assert.ok(out.text.includes("app.example.com:8443"));
+ assert.ok(out.text.includes("I deploy on Thursdays"));
+ // And the caller is told what went, by position and reason — never by text.
+ assert.ok(out.dropped.length > 0);
+ assert.ok(out.dropped.every((d) => d.reason === "no-save-directive"));
+ const notice = droppedSpanNotice(out.dropped, out.segments);
+ assert.match(notice, /span\(s\) were dropped/);
+ assert.ok(!notice.includes("4821"), "the notice echoed the dropped content");
+});
+
+test("a directive takes its own paragraph with it", () => {
+ // Scoping per line would save the PIN and drop only the sentence asking
+ // not to, which is the one outcome worse than dropping too much.
+ const out = sanitizePassage(
+ "I always use pnpm\n\nMy bank PIN is 4821\ndon't save this\n\nI deploy on Thursdays",
+ );
+ assert.equal(out.refusal, undefined);
+ assert.ok(!out.text.includes("4821"), "the directive's neighbour was saved anyway");
+ assert.ok(out.text.includes("I always use pnpm"));
+ assert.ok(out.text.includes("I deploy on Thursdays"));
+});
+
+test("a fenced transcript is analysed, a fenced code paste is still refused", () => {
+ // The tool documents a fenced transcript as its canonical input, and
+ // refused exactly that as pasted content.
+ const fenced = sanitizePassage("```\n" + TRANSCRIPT + "\n```");
+ assert.equal(fenced.refusal, undefined, "a fenced transcript was discarded");
+ assert.ok(fenced.text.includes("I always use pnpm for every project"));
+ assert.ok(!fenced.text.includes("4821"));
+
+ // A short fence, or one carrying a language tag, is a paste and stays one.
+ assert.equal(
+ sanitizePassage("```\nERROR 500 from the vendor API\n at handler (index.js:42)\n```")
+ .refusal,
+ "pasted-content",
+ );
+ assert.equal(
+ sanitizePassage('```json\n{\n "a": 1,\n "b": 2,\n "c": 3,\n "d": 4\n}\n```').refusal,
+ "pasted-content",
+ );
+ // ...and a fence INSIDE a passage is dropped without taking the rest.
+ const mixed = sanitizePassage(
+ "I always use pnpm\n\n```\nERROR 500\n```\n\nI deploy on Thursdays",
+ );
+ assert.equal(mixed.refusal, undefined);
+ assert.ok(mixed.text.includes("I always use pnpm"));
+ assert.ok(mixed.text.includes("I deploy on Thursdays"));
+ assert.ok(!mixed.text.includes("ERROR 500"));
+ assert.ok(mixed.dropped.some((d) => d.reason === "pasted-content"));
+});
+
+test("a passage with nothing left is still refused outright", () => {
+ // Scoping the refusal must not turn "save nothing" into "save something".
+ const out = sanitizePassage("My bank PIN is 4821 — don't save this.");
+ assert.equal(out.refusal, "no-save-directive");
+ assert.equal(out.text, "");
+ assert.ok(!out.text.includes("4821"));
+});
+
+test("a clean passage is forwarded byte-for-byte", () => {
+ const clean = "I always use pnpm, TypeScript strict mode, and deploy on Thursdays.";
+ const out = sanitizePassage(clean);
+ assert.equal(out.text, clean);
+ assert.equal(out.changed, false);
+ assert.deepEqual(out.dropped, []);
+ assert.equal(droppedSpanNotice(out.dropped, out.segments), "");
+});
diff --git a/services/server/scripts/mcp/__tests__/write-path-redaction.test.ts b/services/server/scripts/mcp/__tests__/write-path-redaction.test.ts
index ad9f10113..eb909fbcc 100644
--- a/services/server/scripts/mcp/__tests__/write-path-redaction.test.ts
+++ b/services/server/scripts/mcp/__tests__/write-path-redaction.test.ts
@@ -347,6 +347,123 @@ test("the idempotency key is derived from what is actually written", async (t) =
* the SDK was handed.
* ------------------------------------------------------------------------ */
+test("splitting a secret across bulk entries does not reach the SDK", async (t) => {
+ // Every label-gated rule searched a window inside ONE string, so the label
+ // in one entry and its value in the next walked past all of them — with
+ // MemWal's own delegate private key, the worst thing this product can leak.
+ // An agent that paraphrases a user across two entries is enough; nothing
+ // here needs malice.
+ const SEED = "4f3c2b1a9e8d7c6b5a4f3e2d1c0b9a8f7e6d5c4b3a2f1e0d9c8b7a6f5e4d3c2b";
+ const forwarded: Forwarded = { remember: [], bulk: [], analyze: [] };
+ const client = await clientFor(sessionWith(forwarded), t);
+
+ const result = await client.callTool({
+ name: "memwal_remember_bulk",
+ arguments: {
+ facts: [
+ "I set up a second laptop today",
+ "my delegate private key for the mainnet account",
+ SEED,
+ "and I use the work namespace there",
+ ],
+ },
+ });
+
+ assert.equal(forwarded.bulk.length, 1, "the other facts must still be written");
+ assert.ok(
+ !forwarded.bulk[0].join("\n").includes(SEED),
+ "the delegate private key reached the SDK from a split batch",
+ );
+ assert.ok(!textOf(result).includes(SEED), "the reply echoed the delegate key back");
+ // The facts either side are still saved: this is a scalpel, not a batch
+ // refusal.
+ assert.ok(forwarded.bulk[0].some((f) => f.includes("second laptop")));
+ assert.ok(forwarded.bulk[0].some((f) => f.includes("work namespace")));
+});
+
+test("a bulk batch of plain identifiers is still forwarded byte-for-byte", async (t) => {
+ // The cross-entry screen only fires when a credential label is somewhere in
+ // the batch. Without one, nothing about the batch path may differ from the
+ // single-fact path.
+ const forwarded: Forwarded = { remember: [], bulk: [], analyze: [] };
+ const client = await clientFor(sessionWith(forwarded), t);
+ const facts = [
+ "My Sui package id is 0xe80f2feec1c139616a86c9f71210152e2a7ca552b20841f2e192f99f75864437",
+ "The release digest is 4f3c2b1a9e8d7c6b5a4f3e2d1c0b9a8f7e6d5c4b3a2f1e0d9c8b7a6f5e4d3c2b",
+ "I always use pnpm",
+ ];
+ const result = await client.callTool({
+ name: "memwal_remember_bulk",
+ arguments: { facts },
+ });
+ assert.deepEqual(forwarded.bulk, [facts]);
+ assert.doesNotMatch(textOf(result), /redacted|NOT SAVED/i);
+});
+
+test("one do-not-save line does not discard a whole transcript", async (t) => {
+ // `NO_SAVE_DIRECTIVE` is a whole-string predicate applied to a whole
+ // passage: one "don't save this part" line refused all forty turns, with
+ // `text: ""` and an instruction not to retry — and no `isError`, so the
+ // client read it as a successful call that had saved nothing.
+ const forwarded: Forwarded = { remember: [], bulk: [], analyze: [] };
+ const client = await clientFor(sessionWith(forwarded), t);
+ const transcript = [
+ "user: I always use pnpm for every project",
+ "assistant: noted",
+ "user: my bank PIN is 4821 - don't save this part",
+ "assistant: understood",
+ "user: I deploy on Thursdays and never on Fridays",
+ "assistant: makes sense",
+ ].join("\n");
+
+ const result = await client.callTool({
+ name: "memwal_analyze",
+ arguments: { text: transcript },
+ });
+
+ assert.equal(forwarded.analyze.length, 1, "the whole transcript was discarded again");
+ const sent = forwarded.analyze[0];
+ assert.ok(!sent.includes("4821"), "the withheld line reached the extractor");
+ assert.ok(sent.includes("I always use pnpm for every project"));
+ assert.ok(sent.includes("I deploy on Thursdays"));
+
+ const text = textOf(result);
+ assert.ok(!text.includes("4821"));
+ assert.match(text, /span\(s\) were dropped/, "the drop was not reported to the agent");
+});
+
+test("a fenced transcript is extracted from, and a refusal is flagged as one", async (t) => {
+ const forwarded: Forwarded = { remember: [], bulk: [], analyze: [] };
+ const client = await clientFor(sessionWith(forwarded), t);
+ const fenced =
+ "```\n" +
+ [
+ "user: I always use pnpm for every project",
+ "assistant: noted",
+ "user: I deploy on Thursdays and never on Fridays",
+ "assistant: makes sense",
+ ].join("\n") +
+ "\n```";
+
+ await client.callTool({ name: "memwal_analyze", arguments: { text: fenced } });
+ assert.equal(forwarded.analyze.length, 1, "a fenced transcript was refused as a paste");
+ assert.ok(forwarded.analyze[0].includes("I always use pnpm for every project"));
+
+ // A passage with nothing usable left is still refused — and now says so as
+ // an error, rather than looking like a call that succeeded and saved zero.
+ const refused = await client.callTool({
+ name: "memwal_analyze",
+ arguments: { text: "My bank PIN is 4821 — don't save this." },
+ });
+ assert.equal(forwarded.analyze.length, 1, "a refused passage was forwarded anyway");
+ assert.equal(
+ (refused as { isError?: boolean }).isError,
+ true,
+ "a call that saved nothing reported success",
+ );
+ assert.match(textOf(refused), /NOT SAVED/);
+});
+
test("memwal_analyze will not take an unbounded passage", async (t) => {
// The schema was `z.string().min(1)` with no maximum while the tool is
// documented as accepting a whole transcript, so the work the sidecar's
diff --git a/services/server/scripts/mcp/tools/analyze.ts b/services/server/scripts/mcp/tools/analyze.ts
index cd4768507..224dbba64 100644
--- a/services/server/scripts/mcp/tools/analyze.ts
+++ b/services/server/scripts/mcp/tools/analyze.ts
@@ -4,7 +4,12 @@ import type { MemWalSession } from "../auth.js";
import { TOOL_METADATA } from "./annotations.js";
import { wrapTool, explorerFooter } from "./util.js";
import { SECRET_EXCLUSION_RULES, AUTO_SAVE_OPT_IN_RULE } from "./memory-policy.js";
-import { sanitizeFact, redactionNotice, refusalNotice } from "./redaction.js";
+import {
+ sanitizePassage,
+ redactionNotice,
+ refusalNotice,
+ droppedSpanNotice,
+} from "./redaction.js";
import {
REMEMBER_POLL_INTERVAL_MS,
REMEMBER_WAIT_MS,
@@ -75,7 +80,7 @@ export function registerAnalyzeTool(
AUTO_SAVE_OPT_IN_RULE +
" " +
SECRET_EXCLUSION_RULES +
- " This tool forwards a whole passage, so it is the easiest way to leak a credential that happened to sit next to a fact: the passage is stripped of credential shapes before it is sent for extraction, and a passage that is only a secret, or that the user asked not to save, is not sent at all.",
+ " This tool forwards a whole passage, so it is the easiest way to leak a credential that happened to sit next to a fact: the passage is stripped of credential shapes before it is sent for extraction. A span the user asked not to save, or that is pasted third-party material, is dropped on its own and named in the reply; only a passage with nothing usable left is refused outright.",
inputSchema: ANALYZE_INPUT,
},
wrapTool<{ text: string; namespace?: string }>(session, "memwal_analyze", async ({ text, namespace }) => {
@@ -83,15 +88,30 @@ export function registerAnalyzeTool(
// reaches the extractor LLM. Everything this tool stores is derived
// from this text, so a credential left in it can be copied into any
// number of extracted facts — on append-only storage (WALM-642).
- const safe = sanitizeFact(text);
+ //
+ // `sanitizePassage`, not `sanitizeFact`: the refusal predicates are
+ // whole-string, and applied to a transcript one "don't save this
+ // part" line threw away every other turn with it. They are scoped
+ // per span here, so the offending span is dropped and named and the
+ // rest is still extracted from.
+ const safe = sanitizePassage(text);
if (safe.refusal) {
return {
+ // Flagged as an error, because it is not a successful call:
+ // nothing was extracted and nothing was saved, and a bare
+ // text result reads to a client exactly like one that did.
+ isError: true,
content: [
{ type: "text" as const, text: refusalNotice(safe.refusal) },
],
};
}
- const notice = redactionNotice(safe.kinds, safe.count);
+ const notice = [
+ redactionNotice(safe.kinds, safe.count),
+ droppedSpanNotice(safe.dropped, safe.segments),
+ ]
+ .filter(Boolean)
+ .join("\n\n");
const safeText = safe.text;
// `analyze` (not `analyzeAndWait`) returns once extraction is done
diff --git a/services/server/scripts/mcp/tools/redaction.ts b/services/server/scripts/mcp/tools/redaction.ts
index d210cffb8..3fa4fc3cb 100644
--- a/services/server/scripts/mcp/tools/redaction.ts
+++ b/services/server/scripts/mcp/tools/redaction.ts
@@ -563,3 +563,343 @@ export function refusalNotice(reason: RefusalReason): string {
`restate the fact without the sensitive part and save that instead.`
);
}
+
+/* ───────────────────────────────────────────────────────────────────────────
+ * Screening a BATCH, not one string at a time (WALM-642).
+ *
+ * `sanitizeFact` sees one entry, and every label-gated rule in this file
+ * searches a window inside that one string. `memwal_remember_bulk` takes up to
+ * twenty of them, which is a way around all of it: put the label in one entry
+ * and the value in the next and each is individually unremarkable.
+ *
+ * ["my delegate private key for the mainnet account",
+ * "4f3c...789"]
+ *
+ * Both entries passed untouched, while the same words as ONE string were
+ * correctly redacted. That value is MemWal's own delegate private key — the
+ * thing this module calls the worst secret the product can leak — and an agent
+ * that paraphrases a user across two entries is all it takes. Nothing about it
+ * requires malice.
+ *
+ * So a batch is screened as a batch. The per-entry pass runs first and is
+ * unchanged, which keeps the refusal granularity that already works (one bad
+ * entry is dropped, the rest of the batch still lands). Then a second pass asks
+ * a question the first one cannot: is there a credential label in the text
+ * ADJACENT to this value, where adjacent means the entries either side of it as
+ * well as its own?
+ *
+ * A value that is an entry all by itself is screened against the whole batch
+ * rather than its neighbours. A bare token with no words around it has no
+ * meaning of its own to lose, and "which entry did the label end up in" is not
+ * something the user controls.
+ * ------------------------------------------------------------------------ */
+
+/** What separates two entries when the batch is viewed as one passage. */
+const BATCH_SEPARATOR = "\n";
+
+/**
+ * A label anywhere in `context`, on either side of [start, end).
+ *
+ * The per-entry gate measures a character window because it is looking inside
+ * one sentence. Across entries the unit is the entry: the caller passes exactly
+ * the text that counts as adjacent and this searches all of it, so there is no
+ * second magic number to keep in step with the first.
+ */
+function hasCredentialLabelInContext(
+ context: string,
+ start: number,
+ end: number,
+): boolean {
+ const before = context.slice(0, start).replace(PLACEHOLDER_RUN, " ");
+ if (HEX_CREDENTIAL_LABEL.test(before)) return true;
+ const after = context.slice(end).replace(PLACEHOLDER_RUN, " ");
+ return HEX_CREDENTIAL_LABEL.test(after);
+}
+
+/** Words left once the placeholders and one candidate token are removed. */
+function wordsAround(text: string, token: string): number {
+ const rest = text
+ .replace(PLACEHOLDER_RUN, " ")
+ .replace(token, " ")
+ .match(/[\p{L}\p{N}]{2,}/gu);
+ return rest?.length ?? 0;
+}
+
+/**
+ * Screen a whole batch. One result per input, in input order.
+ *
+ * Drop-in for `inputs.map(sanitizeFact)` — every entry comes back with the same
+ * shape and the same per-entry refusals — plus the cross-entry pass above.
+ */
+export function sanitizeFactBatch(inputs: string[]): SanitizedText[] {
+ const perEntry = inputs.map((text) => sanitizeFact(text ?? ""));
+ if (perEntry.length < 2 && !perEntry.some((r) => !r.refusal)) return perEntry;
+
+ // What each entry contributes as CONTEXT. A refused entry is never
+ // forwarded, but its words still say what the batch is about, so it keeps
+ // supplying label context from its original text.
+ const views = perEntry.map((r, i) => (r.refusal ? (inputs[i] ?? "") : r.text));
+ const wholeBatch = views.join(BATCH_SEPARATOR);
+ // Placeholders carry the words "secret" and "key" (`high-entropy-secret`
+ // most obviously), so a batch where one entry was already redacted would
+ // otherwise label itself.
+ if (!HEX_CREDENTIAL_LABEL.test(wholeBatch.replace(PLACEHOLDER_RUN, " "))) {
+ return perEntry;
+ }
+
+ return perEntry.map((result, i) => {
+ if (result.refusal) return result;
+ const own = views[i];
+ const prev = i > 0 ? views[i - 1] : "";
+ const next = i + 1 < views.length ? views[i + 1] : "";
+ // The neighbours, with this entry in the middle, and this entry's
+ // offset inside it.
+ const neighbourhood = [prev, own, next].join(BATCH_SEPARATOR);
+ const ownStart = prev.length + BATCH_SEPARATOR.length;
+
+ const kinds = [...result.kinds];
+ let count = result.count;
+ let hitAny = false;
+ const text = own.replace(
+ OPAQUE_RUN,
+ (match: string, offset: number, whole: string) => {
+ if (isInsidePlaceholder(whole, offset)) return match;
+ if (!looksLikeOpaqueToken(match)) return match;
+ // An entry that is essentially just this token is screened
+ // against every entry; one with a sentence around it, against
+ // the entries either side.
+ const bare = wordsAround(own, match) < 3;
+ const context = bare ? wholeBatch : neighbourhood;
+ const start = bare
+ ? views.slice(0, i).reduce(
+ (n, v) => n + v.length + BATCH_SEPARATOR.length,
+ 0,
+ ) + offset
+ : ownStart + offset;
+ if (!hasCredentialLabelInContext(context, start, start + match.length)) {
+ return match;
+ }
+ if (!kinds.includes("labelled-key-material")) {
+ kinds.push("labelled-key-material");
+ }
+ count += 1;
+ hitAny = true;
+ return placeholder("labelled-key-material");
+ },
+ );
+
+ if (!hitAny) return result;
+ const collapsed = text.replace(/[ \t]{2,}/g, " ").trim();
+ if (!hasSalvageableContent(collapsed)) {
+ return { text: "", changed: true, kinds, count, refusal: "credential-only" };
+ }
+ return { text: collapsed, changed: true, kinds, count };
+ });
+}
+
+/* ───────────────────────────────────────────────────────────────────────────
+ * Screening a PASSAGE (WALM-642).
+ *
+ * `NO_SAVE_DIRECTIVE`, `OFF_THE_RECORD` and `isPastedContent` are whole-string
+ * predicates, written for one fact and correct there. `memwal_analyze` applies
+ * them to a whole transcript, where "whole string" means something else
+ * entirely: one "don't save this part" line in a forty-turn conversation
+ * refused the entire passage — `refusal`, `text: ""`, all forty turns
+ * discarded, and a note telling the agent not to retry it. The call carried no
+ * `isError`, so the client saw a successful call that had saved nothing.
+ *
+ * The fix is the one `memwal_remember_bulk` already uses on entries: scope the
+ * refusal to the span that earned it. A passage is split into segments, each is
+ * screened on its own, the offending ones are dropped and named, and everything
+ * else is extracted from.
+ *
+ * Two deliberate wrinkles:
+ *
+ * - A no-save directive drops its NEIGHBOURS too, within its own paragraph.
+ * "My bank PIN is 4821" on one line and "don't save this" on the next is
+ * the ordinary way people write it, and dropping only the second line would
+ * save the first — a far worse outcome than losing a turn either side.
+ * - A fenced block WRAPPING THE WHOLE PASSAGE is unwrapped rather than
+ * refused, when it has no language tag and holds a transcript's worth of
+ * lines. That shape is what this tool documents as its canonical input, and
+ * refusing it discarded exactly the passages people most wanted analysed. A
+ * tagged fence (```json, ```sh) is code, a short one is a snippet, and a
+ * fence INSIDE the passage is still dropped as a paste — none of those
+ * change.
+ * ------------------------------------------------------------------------ */
+
+/**
+ * Non-blank lines an untagged outer fence must hold before it is read as a
+ * transcript rather than a pasted snippet.
+ */
+const TRANSCRIPT_FENCE_MIN_LINES = 4;
+
+/** One dropped span, named by position and reason. Never carries its text. */
+export interface DroppedSpan {
+ /** 1-based line number in the passage as it was received. */
+ line: number;
+ reason: RefusalReason;
+}
+
+export interface SanitizedPassage extends SanitizedText {
+ /** Spans removed before extraction. Empty when the passage came through whole. */
+ dropped: DroppedSpan[];
+ /** Non-blank segments the passage was split into. */
+ segments: number;
+}
+
+/** Strip one outer fence when it is a transcript rather than a code paste. */
+function unwrapTranscriptFence(text: string): string {
+ const s = text.trim();
+ const m = /^```([^\n]*)\n([\s\S]*?)\n?```$/.exec(s);
+ if (!m) return text;
+ // A language tag is an author saying "this is code", so take them at their
+ // word and leave it to the per-segment paste rule.
+ if (m[1].trim() !== "") return text;
+ const body = m[2];
+ const lines = body.split(/\r?\n/).filter((l) => l.trim() !== "");
+ if (lines.length < TRANSCRIPT_FENCE_MIN_LINES) return text;
+ return body;
+}
+
+interface PassageSegment {
+ text: string;
+ /** 1-based line the segment starts on. */
+ line: number;
+ blank: boolean;
+}
+
+/**
+ * Split a passage into the units a refusal may apply to.
+ *
+ * One line per segment, except that a fenced block and a run of `>` quotation
+ * stay whole — both are multi-line by nature, and `isPastedContent` can only
+ * recognise them as one piece.
+ */
+function splitPassage(text: string): PassageSegment[] {
+ const lines = text.split(/\r?\n/);
+ const segments: PassageSegment[] = [];
+ for (let i = 0; i < lines.length; i++) {
+ const line = lines[i];
+ if (/^\s*```/.test(line)) {
+ const start = i;
+ const block = [line];
+ i++;
+ while (i < lines.length) {
+ block.push(lines[i]);
+ if (/^\s*```/.test(lines[i])) break;
+ i++;
+ }
+ segments.push({ text: block.join("\n"), line: start + 1, blank: false });
+ continue;
+ }
+ if (/^\s*>/.test(line)) {
+ const start = i;
+ const block = [line];
+ while (i + 1 < lines.length && /^\s*>/.test(lines[i + 1])) {
+ block.push(lines[++i]);
+ }
+ segments.push({ text: block.join("\n"), line: start + 1, blank: false });
+ continue;
+ }
+ segments.push({ text: line, line: i + 1, blank: line.trim() === "" });
+ }
+ return segments;
+}
+
+/**
+ * Strip credentials from a passage, dropping only the spans that must go.
+ *
+ * `text` is what is safe to forward, `dropped` says what was removed and why,
+ * and `refusal` is set only when nothing survived at all.
+ */
+export function sanitizePassage(input: string): SanitizedPassage {
+ const segments = splitPassage(unwrapTranscriptFence(input ?? ""));
+ const results = segments.map((s) =>
+ s.blank ? null : sanitizeFact(s.text),
+ );
+ const dropped: (RefusalReason | null)[] = results.map((r) => r?.refusal ?? null);
+
+ // A directive takes its immediate neighbours with it, unless a blank line
+ // stands between them — see the note above. Computed against the ORIGINAL
+ // drop list so one directive cannot cascade down a whole passage.
+ const directive = dropped.map((d) => d === "no-save-directive");
+ for (let i = 0; i < segments.length; i++) {
+ if (!directive[i]) continue;
+ for (const j of [i - 1, i + 1]) {
+ if (j < 0 || j >= segments.length) continue;
+ if (segments[j].blank) continue;
+ if (dropped[j] === null) dropped[j] = "no-save-directive";
+ }
+ }
+
+ const kinds: RedactionKind[] = [];
+ let count = 0;
+ const kept: string[] = [];
+ const droppedSpans: DroppedSpan[] = [];
+ let nonBlank = 0;
+ for (let i = 0; i < segments.length; i++) {
+ const segment = segments[i];
+ if (segment.blank) {
+ kept.push("");
+ continue;
+ }
+ nonBlank += 1;
+ const reason = dropped[i];
+ if (reason) {
+ droppedSpans.push({ line: segment.line, reason });
+ continue;
+ }
+ const result = results[i]!;
+ count += result.count;
+ for (const kind of result.kinds) {
+ if (!kinds.includes(kind)) kinds.push(kind);
+ }
+ kept.push(result.text);
+ }
+
+ const text = kept
+ .join("\n")
+ .replace(/[ \t]{2,}/g, " ")
+ .replace(/\n{3,}/g, "\n\n")
+ .trim();
+ const changed = count > 0 || droppedSpans.length > 0;
+
+ if (!hasSalvageableContent(text)) {
+ // Nothing usable left. Report the reason that took the most of it, so
+ // the agent is told the truth about why rather than a generic refusal.
+ const tally = new Map();
+ for (const span of droppedSpans) {
+ tally.set(span.reason, (tally.get(span.reason) ?? 0) + 1);
+ }
+ const [top] = [...tally.entries()].sort((a, b) => b[1] - a[1]);
+ return {
+ text: "",
+ changed: true,
+ kinds,
+ count,
+ refusal: top?.[0] ?? "credential-only",
+ dropped: droppedSpans,
+ segments: nonBlank,
+ };
+ }
+
+ return { text, changed, kinds, count, dropped: droppedSpans, segments: nonBlank };
+}
+
+/**
+ * The line naming spans that were dropped from a passage but not the whole of
+ * it. Positions and reasons only — never the text that was removed.
+ */
+export function droppedSpanNotice(dropped: DroppedSpan[], segments: number): string {
+ if (dropped.length === 0) return "";
+ const detail = dropped
+ .map((d) => `line ${d.line} (${refusalMessage(d.reason)})`)
+ .join("; ")
+ return (
+ `Note: ${dropped.length} of ${segments} span(s) were dropped before ` +
+ `extraction and nothing from them was saved — ${detail}. The rest of the ` +
+ `passage was extracted from normally. Tell the user which part was left ` +
+ `out; do not re-send it.`
+ );
+}
diff --git a/services/server/scripts/mcp/tools/remember-bulk.ts b/services/server/scripts/mcp/tools/remember-bulk.ts
index 7be2e80c8..e54debd2d 100644
--- a/services/server/scripts/mcp/tools/remember-bulk.ts
+++ b/services/server/scripts/mcp/tools/remember-bulk.ts
@@ -5,7 +5,7 @@ import { TOOL_METADATA } from "./annotations.js";
import { wrapTool, explorerFooter } from "./util.js";
import { SECRET_EXCLUSION_RULES, AUTO_SAVE_OPT_IN_RULE } from "./memory-policy.js";
import {
- sanitizeFact,
+ sanitizeFactBatch,
redactionNotice,
refusalMessage,
type RedactionKind,
@@ -64,7 +64,7 @@ export function registerRememberBulkTool(
AUTO_SAVE_OPT_IN_RULE +
" " +
SECRET_EXCLUSION_RULES +
- " Walrus storage is append-only: a stored secret cannot be deleted, so each entry is stripped of credential shapes before writing and entries that are nothing but a secret are dropped, with a note saying which.",
+ " Walrus storage is append-only: a stored secret cannot be deleted, so each entry is stripped of credential shapes before writing and entries that are nothing but a secret are dropped, with a note saying which. The batch is screened as a whole, so splitting a credential's label into one entry and its value into another does not get it past the filter.",
inputSchema: REMEMBER_BULK_INPUT,
},
wrapTool<{ facts: string[]; namespace?: string }>(session, "memwal_remember_bulk", async ({ facts, namespace }) => {
@@ -74,9 +74,15 @@ export function registerRememberBulkTool(
// part; an entry that is only a secret — or that the user asked
// not to save — is dropped from the batch rather than the whole
// call failing, so the other facts still land.
- const screened = facts.map((text, index) => ({
+ //
+ // Screened as a BATCH, not entry by entry: every label-gated rule
+ // searches a window inside one string, so a label in one entry and
+ // its value in the next defeated all of them — including the one
+ // that exists for MemWal's own delegate private key. See
+ // `sanitizeFactBatch`.
+ const screened = sanitizeFactBatch(facts).map((result, index) => ({
index,
- result: sanitizeFact(text),
+ result,
}));
const kept = screened.filter((s) => !s.result.refusal);
const dropped = screened.filter((s) => s.result.refusal);
From ca80771e57b7629cd8cc467497df064ff1a057ab Mon Sep 17 00:00:00 2001
From: Harry Phan