diff --git a/.claude-plugin/marketplace.json b/.claude-plugin/marketplace.json index faf2536a7..1070950e9 100644 --- a/.claude-plugin/marketplace.json +++ b/.claude-plugin/marketplace.json @@ -11,7 +11,7 @@ "name": "memwal", "source": "./packages/mcp/plugin", "description": "Automatic Walrus Memory — proactive recall and durable-fact saving via the MemWal MCP + lifecycle hooks.", - "version": "0.0.12" + "version": "0.0.13" } ] } diff --git a/.cursor-plugin/marketplace.json b/.cursor-plugin/marketplace.json index c99617559..a4b700e2b 100644 --- a/.cursor-plugin/marketplace.json +++ b/.cursor-plugin/marketplace.json @@ -11,7 +11,7 @@ "name": "memwal", "source": "./packages/mcp/plugin", "description": "Automatic Walrus Memory — proactive recall and durable-fact saving via the MemWal MCP + lifecycle hooks.", - "version": "0.0.12" + "version": "0.0.13" } ] } diff --git a/.github/workflows/test.yml b/.github/workflows/test.yml index b2e6ab3ef..e029a0577 100644 --- a/.github/workflows/test.yml +++ b/.github/workflows/test.yml @@ -34,6 +34,8 @@ jobs: script: scripts/check-docs-freshness.mjs - name: SEAL cross-account synthetic parser script: scripts/synthetic-seal-cross-account.test.mjs + - name: Migration / Completion artifact + script: scripts/write-migration-completion-artifact.mjs --self-test steps: - uses: actions/checkout@v4 diff --git a/SKILL.md b/SKILL.md index 9957b0ffc..062f727e4 100644 --- a/SKILL.md +++ b/SKILL.md @@ -155,7 +155,7 @@ const stored = await memwal.waitForRememberJob(accepted.job_id, { | `recall({ query, limit?, topK?, namespace?, maxDistance? })` *(preferred)* or `recall(query, limit?, namespace?)` | Semantic search for memories | `{ results: [{ blob_id, text, distance }], total }` | | `analyze(text, namespace?)` | Extract facts and accept one memory job per fact | `{ job_ids, facts, fact_count, status, owner }` | | `analyzeAndWait(text, namespace?, opts?)` | Extract facts and wait for all fact jobs to complete | `{ results, facts, total, succeeded, failed, owner }` | -| `restore(namespace, limit?)` | Rebuild missing index entries from Walrus | `{ restored, skipped, total, namespace, owner, truncated }` | +| `restore(namespace, limit?)` | Rebuild missing index entries from Walrus | `{ restored, skipped, failed, total, namespace, owner, truncated }` | | `health()` | Check relayer health | `{ status, version }` | | `getPublicKeyHex()` | Get hex-encoded public key | `string` | @@ -271,6 +271,7 @@ interface EmbedResult { interface RestoreResult { restored: number; skipped: number; + failed: number; total: number; namespace: string; owner: string; @@ -363,7 +364,8 @@ Cross-namespace and cross-owner reads are not just filtered out of results — t | Field | Counts | Notes | |---|---|---| | `restored` | Blobs the relayer just rebuilt this call | Pulled from Walrus → SEAL decrypted → re-embedded → inserted as a new row | -| `skipped` | On-chain blobs already in the local index | No work needed; relayer left them as-is | +| `skipped` | On-chain blobs already in the local **success** index | No work needed; relayer left them as-is. Does not include decrypt/UTF-8 failures. | +| `failed` | Permanent decrypt/UTF-8 failures | On-chain blobs in this page that are negative-cached, plus new permanent failures this call. Older relayers omit the field; SDKs default it to `0`. | | `total` | All on-chain blobs the relayer saw for `(owner, namespace)` | Before the limit was applied | | `namespace` | Echo of the request | | | `owner` | Resolved owner address | | @@ -371,7 +373,7 @@ Cross-namespace and cross-owner reads are not just filtered out of results — t `truncated=true` means this restore is **known-retryable-incomplete**: more missing blobs than `limit` allowed this call to restore, **or** the sidecar's owner-wide candidate fetch hit its cap **and** raising `limit` can still expand that fetch (`limit < 20`). Once the sidecar cap is saturated (`limit >= 20`, cap pinned at 100), truncation follows this call's missing-blob page length, not onchain `total`. A fully restored namespace does not loop. `truncated=false` is **not** proof the sidecar saw every onchain blob; blobs beyond the owner-wide sidecar candidate cap can still be missing. WALM-451 tracks a `sourceCapped` field for that case. Relayers older than WALM-319 omit `truncated`; SDKs default it to `false`. -**Silent drops.** A blob that *cannot* be decrypted or embedded (e.g. wrong delegate key, malformed ciphertext, embedding API down) is dropped without counting in `restored` *or* `skipped`. `restored + skipped` is therefore a lower bound on healthy entries, not a strict equality with `total`. +Permanent decrypt or invalid-UTF-8 failures count in `failed`, not `skipped`. Transient download/decrypt/embed errors are still not counted in `restored`, `skipped`, or `failed` and may be retried (`truncated=true` when a page yields only those). `restored + skipped + failed` therefore never exceeds `total`, and falls short of it whenever transient errors leave blobs uncounted. #### Default and limit diff --git a/apps/chatbot/app/(auth)/api/auth/guest/route.ts b/apps/chatbot/app/(auth)/api/auth/guest/route.ts index 682f6542b..0b5fe56c1 100644 --- a/apps/chatbot/app/(auth)/api/auth/guest/route.ts +++ b/apps/chatbot/app/(auth)/api/auth/guest/route.ts @@ -1,43 +1,22 @@ import { NextResponse } from "next/server"; import { signIn } from "@/app/(auth)/auth"; +import { isSafeRedirectUrl, publicRequestUrl } from "@/lib/public-request-url"; import { getSessionToken } from "@/lib/session-token"; -/** - * Validate a redirect target before forwarding to auth. - * Allows only: - * - Relative paths beginning with "/" (but not "//", which is protocol-relative) - * - Absolute URLs whose origin matches the request origin (same-origin) - * Anything else (external hosts, javascript:, data:, //evil.com) falls back to "/". - */ -function isSafeRedirectUrl(redirectUrl: string, requestUrl: string): boolean { - // Relative path — safe as long as it isn't protocol-relative ("//host/...") - if (redirectUrl.startsWith("/") && !redirectUrl.startsWith("//")) { - return true; - } - // Absolute URL — must share the same origin as the request - try { - const redirectOrigin = new URL(redirectUrl).origin; - const requestOrigin = new URL(requestUrl).origin; - return redirectOrigin === requestOrigin; - } catch { - // Unparseable URL (e.g. "javascript:alert(1)") — reject - return false; - } -} - export async function GET(request: Request) { const { searchParams } = new URL(request.url); const rawRedirectUrl = searchParams.get("redirectUrl") || "/"; + const publicUrl = publicRequestUrl(request); - // Reject cross-origin or protocol-relative redirect targets - const redirectUrl = isSafeRedirectUrl(rawRedirectUrl, request.url) + // Reject cross-origin, bind-address, or protocol-relative redirect targets + const redirectUrl = isSafeRedirectUrl(rawRedirectUrl, request) ? rawRedirectUrl : "/"; const token = await getSessionToken(request); if (token) { - return NextResponse.redirect(new URL("/", request.url)); + return NextResponse.redirect(new URL("/", publicUrl)); } return signIn("guest", { redirect: true, redirectTo: redirectUrl }); diff --git a/apps/chatbot/app/(auth)/auth.config.ts b/apps/chatbot/app/(auth)/auth.config.ts index b8bc9e1f1..434c00a8a 100644 --- a/apps/chatbot/app/(auth)/auth.config.ts +++ b/apps/chatbot/app/(auth)/auth.config.ts @@ -1,6 +1,8 @@ import type { NextAuthConfig } from "next-auth"; export const authConfig = { + // Railway / Docker set HOSTNAME=0.0.0.0; trust the incoming Host header. + trustHost: true, pages: { signIn: "/login", newUser: "/", diff --git a/apps/chatbot/lib/public-request-url.ts b/apps/chatbot/lib/public-request-url.ts new file mode 100644 index 000000000..08326068d --- /dev/null +++ b/apps/chatbot/lib/public-request-url.ts @@ -0,0 +1,79 @@ +const BIND_HOSTNAMES = new Set(["0.0.0.0", "::", "[::]"]); + +function isBindHostname(hostname: string): boolean { + return BIND_HOSTNAMES.has(hostname.toLowerCase()); +} + +function firstHeader(headers: Headers, name: string): string | null { + return headers.get(name)?.split(",")[0]?.trim() || null; +} + +function usablePublicHost(host: string | null): string | null { + if (!host) { + return null; + } + try { + const hostname = new URL(`http://${host}`).hostname; + return hostname && !isBindHostname(hostname) ? host : null; + } catch { + return null; + } +} + +export function publicRequestUrl(request: Request): URL { + const url = new URL(request.url); + const forwardedHost = usablePublicHost( + firstHeader(request.headers, "x-forwarded-host") + ); + const publicHost = + forwardedHost ?? + (isBindHostname(url.hostname) + ? usablePublicHost(firstHeader(request.headers, "host")) + : null); + + if (!publicHost) { + return url; + } + + const proto = firstHeader(request.headers, "x-forwarded-proto")?.toLowerCase(); + const protocol = + proto === "http" || proto === "https" + ? proto + : url.protocol.replace(/:$/, ""); + + // Reconstruct; assigning URL.host keeps :3000 from the bind address. + try { + return new URL(`${protocol}://${publicHost}${url.pathname}${url.search}`); + } catch { + return url; + } +} + +export function guestReturnPath(request: Request): string { + const url = new URL(request.url); + const path = `${url.pathname}${url.search}`; + return path.startsWith("/") && !path.startsWith("//") ? path : "/"; +} + +export function isSafeRedirectUrl( + redirectUrl: string, + request: Request +): boolean { + if (redirectUrl.startsWith("/") && !redirectUrl.startsWith("//")) { + return true; + } + + try { + const redirect = new URL(redirectUrl); + const publicUrl = publicRequestUrl(request); + if ( + isBindHostname(redirect.hostname) || + isBindHostname(publicUrl.hostname) + ) { + return false; + } + return redirect.origin === publicUrl.origin; + } catch { + return false; + } +} diff --git a/apps/chatbot/lib/public-request-url.unit.test.ts b/apps/chatbot/lib/public-request-url.unit.test.ts new file mode 100644 index 000000000..3be9cdb7b --- /dev/null +++ b/apps/chatbot/lib/public-request-url.unit.test.ts @@ -0,0 +1,112 @@ +import { describe, expect, it } from "vitest"; +import { + guestReturnPath, + isSafeRedirectUrl, + publicRequestUrl, +} from "./public-request-url"; + +const STAGING_HOST = "chatbot-demo-staging.memory.walrus.xyz"; + +function bindRequest(path = "/chat/abc", headers?: HeadersInit): Request { + return new Request(`https://0.0.0.0:3000${path}`, { headers }); +} + +describe("publicRequestUrl", () => { + it("uses x-forwarded-host and proto when request.url is a bind address", () => { + const request = bindRequest("/chat/abc", { + "x-forwarded-host": STAGING_HOST, + "x-forwarded-proto": "https", + }); + + const publicUrl = publicRequestUrl(request); + expect(publicUrl.origin).toBe(`https://${STAGING_HOST}`); + expect(publicUrl.hostname).not.toBe("0.0.0.0"); + expect(publicUrl.pathname).toBe("/chat/abc"); + }); + + it("prefers a non-bind forwarded host over request.url", () => { + const request = new Request("http://localhost:3000/login", { + headers: { + "x-forwarded-host": STAGING_HOST, + "x-forwarded-proto": "https", + }, + }); + + expect(publicRequestUrl(request).origin).toBe(`https://${STAGING_HOST}`); + }); + + it("falls back to Host when the URL is a bind address", () => { + const request = bindRequest("/chat/abc", { + host: STAGING_HOST, + "x-forwarded-proto": "https", + }); + + expect(publicRequestUrl(request).origin).toBe(`https://${STAGING_HOST}`); + }); +}); + +describe("guestReturnPath", () => { + it("returns pathname and search as a relative path", () => { + const forwarded = { + "x-forwarded-host": STAGING_HOST, + "x-forwarded-proto": "https", + }; + + expect(guestReturnPath(bindRequest("/chat/abc", forwarded))).toBe( + "/chat/abc" + ); + expect(guestReturnPath(bindRequest("/chat/abc?foo=1", forwarded))).toBe( + "/chat/abc?foo=1" + ); + }); + + it("rejects protocol-relative pathnames and falls back to /", () => { + expect(guestReturnPath(bindRequest("//evil.example"))).toBe("/"); + }); +}); + +describe("isSafeRedirectUrl", () => { + it("allows relative paths", () => { + const request = bindRequest("/", { + "x-forwarded-host": STAGING_HOST, + "x-forwarded-proto": "https", + }); + + expect(isSafeRedirectUrl("/", request)).toBe(true); + expect(isSafeRedirectUrl("/chat/1", request)).toBe(true); + }); + + it("rejects cross-origin and protocol-relative targets", () => { + const request = bindRequest("/chat/abc", { + "x-forwarded-host": STAGING_HOST, + "x-forwarded-proto": "https", + }); + + expect(isSafeRedirectUrl("https://evil.example", request)).toBe(false); + expect(isSafeRedirectUrl("//evil.example", request)).toBe(false); + }); + + it("allows same-origin absolute URLs against the public origin", () => { + const request = bindRequest("/chat/abc", { + "x-forwarded-host": STAGING_HOST, + "x-forwarded-proto": "https", + }); + + expect( + isSafeRedirectUrl(`https://${STAGING_HOST}/chat/abc`, request) + ).toBe(true); + expect(isSafeRedirectUrl("https://0.0.0.0:3000/chat/abc", request)).toBe( + false + ); + }); + + it("does not treat a bind address as a safe absolute redirect target", () => { + const request = bindRequest("/chat/abc"); + + expect(isSafeRedirectUrl("https://0.0.0.0:3000/chat/abc", request)).toBe( + false + ); + expect(isSafeRedirectUrl("https://0.0.0.0:3000/", request)).toBe(false); + expect(isSafeRedirectUrl("/", request)).toBe(true); + }); +}); diff --git a/apps/chatbot/proxy.ts b/apps/chatbot/proxy.ts index ccad67efb..5dd48b30b 100644 --- a/apps/chatbot/proxy.ts +++ b/apps/chatbot/proxy.ts @@ -1,5 +1,6 @@ import { type NextRequest, NextResponse } from "next/server"; import { guestRegex } from "./lib/constants"; +import { guestReturnPath, publicRequestUrl } from "./lib/public-request-url"; import { getSessionToken } from "./lib/session-token"; export async function proxy(request: NextRequest) { @@ -20,17 +21,20 @@ export async function proxy(request: NextRequest) { const token = await getSessionToken(request); if (!token) { - const redirectUrl = encodeURIComponent(request.url); + const redirectUrl = encodeURIComponent(guestReturnPath(request)); return NextResponse.redirect( - new URL(`/api/auth/guest?redirectUrl=${redirectUrl}`, request.url) + new URL( + `/api/auth/guest?redirectUrl=${redirectUrl}`, + publicRequestUrl(request) + ) ); } const isGuest = guestRegex.test(token?.email ?? ""); if (token && !isGuest && ["/login", "/register"].includes(pathname)) { - return NextResponse.redirect(new URL("/", request.url)); + return NextResponse.redirect(new URL("/", publicRequestUrl(request))); } return NextResponse.next(); diff --git a/apps/researcher/components/enoki-login-card.tsx b/apps/researcher/components/enoki-login-card.tsx index e3650c572..333a9557e 100644 --- a/apps/researcher/components/enoki-login-card.tsx +++ b/apps/researcher/components/enoki-login-card.tsx @@ -7,15 +7,20 @@ import { useCurrentAccount, useSignPersonalMessage, useSignTransaction, - useSuiClient, } from "@mysten/dapp-kit"; import { isEnokiWallet } from "@mysten/enoki"; +import type { SuiGrpcClient } from "@mysten/sui/grpc"; import { Transaction } from "@mysten/sui/transactions"; import { createSponsorAuthorization } from "@mysten-incubation/memwal"; import { Loader2 } from "lucide-react"; import { useRouter } from "next/navigation"; import { Button } from "@/components/ui/button"; import { enokiConfig } from "@/lib/enoki/config"; +import { getSuiGrpcClient } from "@/lib/sui/grpc-client"; +import { + fetchAccountIdForOwner, + findCreatedObjectByType, +} from "@/lib/sui/account-lookup"; type Step = | "idle" @@ -57,14 +62,14 @@ function uint8ArrayToBase64(bytes: Uint8Array): string { async function sponsoredSignAndExecute( transaction: Transaction, sender: string, - suiClient: ReturnType, + suiClient: SuiGrpcClient, signTransaction: (args: { - transaction: Transaction; + transaction: Transaction | string; }) => Promise<{ signature: string }>, signPersonalMessage: (message: Uint8Array) => Promise<{ signature: string }>, ): Promise<{ digest: string }> { const kindBytes = await transaction.build({ - client: suiClient as any, + client: suiClient, onlyTransactionKind: true, }); const authorization = await createSponsorAuthorization( @@ -93,7 +98,16 @@ async function sponsoredSignAndExecute( const sponsored = await sponsorRes.json(); const sponsoredTx = Transaction.from(sponsored.bytes); - const { signature } = await signTransaction({ transaction: sponsoredTx }); + // dapp-kit's useSignTransaction resolves move-call ABIs via the ambient + // client from SuiClientProvider, which is JSON-RPC (deprecated, no longer + // CORS-enabled for browser origins) — that resolution is what fails as + // `getNormalizedMoveFunction: Failed to fetch`. Pre-serializing with our + // gRPC client and handing off the resulting string short-circuits it: + // dapp-kit passes a string through as-is. + const sponsoredTxJson = await sponsoredTx.toJSON({ client: suiClient }); + const { signature } = await signTransaction({ + transaction: sponsoredTxJson, + }); const execRes = await fetch( `${enokiConfig.memwalServerUrl}/sponsor/execute`, @@ -127,7 +141,7 @@ export function EnokiLoginCard() { const wallets = useWallets(); const { mutateAsync: connect } = useConnectWallet(); const currentAccount = useCurrentAccount(); - const suiClient = useSuiClient(); + const suiClient = getSuiGrpcClient(); const { mutateAsync: signTransaction } = useSignTransaction(); const { mutateAsync: signPersonalMessage } = useSignPersonalMessage(); @@ -222,37 +236,18 @@ export function EnokiLoginCard() { // Check if a Walrus Memory account already exists for this address try { - const registryObj = await suiClient.getObject({ - id: enokiConfig.memwalRegistryId, - options: { showContent: true }, - }); - if ( - registryObj?.data?.content && - "fields" in registryObj.data.content - ) { - const fields = registryObj.data.content.fields as any; - const tableId = fields?.accounts?.fields?.id?.id; - if (tableId) { - const dynField = await suiClient.getDynamicFieldObject({ - parentId: tableId, - name: { type: "address", value: address }, - }); - if ( - dynField?.data?.content && - "fields" in dynField.data.content - ) { - knownAccountId = (dynField.data.content.fields as any) - .value as string; - } - } - } + knownAccountId = await fetchAccountIdForOwner( + suiClient, + enokiConfig.memwalRegistryId, + address, + ); } catch { // Dynamic field not found → no account yet } const pubKeyBytes = Array.from(publicKeyRaw); - const sign = (args: { transaction: Transaction }) => + const sign = (args: { transaction: Transaction | string }) => signTransaction(args); if (knownAccountId) { @@ -298,19 +293,11 @@ export function EnokiLoginCard() { }); // Find the created account object - const txDetails = await suiClient.getTransactionBlock({ - digest: createResult.digest, - options: { showObjectChanges: true }, - }); - const createdObj = txDetails.objectChanges?.find( - (c) => - c.type === "created" && - "objectType" in c && - c.objectType.includes("MemWalAccount"), + knownAccountId = await findCreatedObjectByType( + suiClient, + createResult.digest, + "MemWalAccount", ); - if (createdObj && "objectId" in createdObj) { - knownAccountId = createdObj.objectId; - } if (!knownAccountId) { throw new Error( diff --git a/apps/researcher/components/sui-providers.tsx b/apps/researcher/components/sui-providers.tsx index 92003d0c0..e8c25975b 100644 --- a/apps/researcher/components/sui-providers.tsx +++ b/apps/researcher/components/sui-providers.tsx @@ -5,12 +5,12 @@ import { createNetworkConfig, SuiClientProvider, WalletProvider, - useSuiClientContext, } from "@mysten/dapp-kit"; import { isEnokiNetwork, registerEnokiWallets } from "@mysten/enoki"; import { getJsonRpcFullnodeUrl } from "@mysten/sui/jsonRpc"; import { QueryClient, QueryClientProvider } from "@tanstack/react-query"; import { enokiConfig } from "@/lib/enoki/config"; +import { getSuiGrpcClient } from "@/lib/sui/grpc-client"; const { networkConfig } = createNetworkConfig({ testnet: { url: getJsonRpcFullnodeUrl("testnet"), network: "testnet" }, @@ -19,11 +19,20 @@ const { networkConfig } = createNetworkConfig({ const queryClient = new QueryClient(); -/** Registers Enoki wallets (Google OAuth) with dapp-kit on mount. No-op if env vars are missing. */ +/** + * Registers Enoki wallets (Google OAuth) with dapp-kit on mount. No-op if env + * vars are missing. + * + * Uses a standalone SuiGrpcClient rather than SuiClientProvider's client: + * dapp-kit's SuiClientProvider is hard-typed to SuiJsonRpcClient (even in the + * latest published version), and Sui's public JSON-RPC fullnodes no longer + * serve browser JSON-RPC — so useSuiClientContext()'s client can't be used + * here. Enoki's `client` option accepts the same ClientWithCoreApi interface a + * gRPC client satisfies, so this is otherwise a drop-in swap. + */ function RegisterEnokiWallets() { - const { client, network } = useSuiClientContext(); - useEffect(() => { + const network = enokiConfig.suiNetwork; if (!isEnokiNetwork(network)) return; if (!enokiConfig.enokiApiKey || !enokiConfig.googleClientId) return; @@ -32,12 +41,12 @@ function RegisterEnokiWallets() { providers: { google: { clientId: enokiConfig.googleClientId }, }, - client, + client: getSuiGrpcClient(), network, }); return unregister; - }, [client, network]); + }, []); return null; } diff --git a/apps/researcher/lib/sui/account-lookup.ts b/apps/researcher/lib/sui/account-lookup.ts new file mode 100644 index 000000000..2cbef41cc --- /dev/null +++ b/apps/researcher/lib/sui/account-lookup.ts @@ -0,0 +1,81 @@ +/** + * On-chain reads for the Enoki registration flow, over gRPC. + * + * These deliberately use gRPC's `include: { json: true }` rather than decoding + * BCS with hand-written struct schemas (the approach in + * apps/noter/lib/sui/account-bcs.ts). BCS schemas must list every field of a + * struct in declaration order; when the published package grows a field, a + * schema that predates it decodes *silently wrong* rather than erroring. + * account.move has already grown fields on both AccountRegistry and + * MemWalAccount that noter's schemas don't model — that's a latent bug waiting + * on the next publish. Reading `json` keeps these lookups correct across + * contract upgrades, since new fields just arrive as extra keys. + * + * The SDK notes the `json` shape may differ between JSON-RPC/gRPC/GraphQL + * backends. That doesn't apply here — this module only ever talks to the gRPC + * client from ./grpc-client — but the table-id read below still tolerates both + * renderings of a `UID`, since that is the one field whose shape has actually + * varied in practice. + */ +import type { SuiGrpcClient } from "@mysten/sui/grpc"; +import { fromHex, normalizeSuiAddress, toHex } from "@mysten/sui/utils"; + +/** + * Look up the MemWalAccount object id owned by `ownerAddress`, or null if the + * owner has no account yet. + */ +export async function fetchAccountIdForOwner( + client: SuiGrpcClient, + registryId: string, + ownerAddress: string, +): Promise { + const registry = await client.getObject({ + objectId: registryId, + include: { json: true }, + }); + + // AccountRegistry.accounts is a sui::table::Table; its entries live as + // dynamic fields on the table's own UID, not inlined in the struct. + const accounts = registry.object.json?.accounts as + | { id?: string | { id?: string } } + | undefined; + const rawTableId = accounts?.id; + const tableId = typeof rawTableId === "string" ? rawTableId : rawTableId?.id; + if (!tableId) return null; + + const response = await client.getDynamicField({ + parentId: tableId, + name: { + type: "address", + bcs: fromHex(normalizeSuiAddress(ownerAddress)), + }, + }); + + // The value is a Move `ID` — a bare 32-byte address, so it needs no struct + // schema to decode. + const value = response.dynamicField?.value?.bcs; + return value?.length === 32 ? `0x${toHex(value)}` : null; +} + +/** + * Find the object created by `digest` whose type contains `objectType`, or null + * if the transaction created no such object. + */ +export async function findCreatedObjectByType( + client: SuiGrpcClient, + digest: string, + objectType: string, +): Promise { + const response = await client.getTransaction({ + digest, + include: { effects: true, objectTypes: true }, + }); + + const transaction = response.Transaction ?? response.FailedTransaction; + const created = transaction.effects?.changedObjects.find( + (change) => + change.idOperation === "Created" && + transaction.objectTypes?.[change.objectId]?.includes(objectType), + ); + return created?.objectId ?? null; +} diff --git a/apps/researcher/lib/sui/grpc-client.ts b/apps/researcher/lib/sui/grpc-client.ts new file mode 100644 index 000000000..4c2a52443 --- /dev/null +++ b/apps/researcher/lib/sui/grpc-client.ts @@ -0,0 +1,36 @@ +/** + * Sui gRPC client — used for Enoki's on-chain registration flow. + * + * Sui's public JSON-RPC fullnodes were deprecated in 2026 in favor of gRPC and + * no longer answer browser preflights, so any JSON-RPC read from the browser + * fails CORS. @mysten/dapp-kit's SuiClientProvider/useSuiClient are still + * hard-typed to SuiJsonRpcClient (confirmed against dapp-kit 1.1.17) and can't + * be swapped for a gRPC client, so this bypasses that provider entirely for the + * one place researcher needs live chain reads: the on-chain account + * lookup/creation in enoki-login-card.tsx. + * + * Mirrors apps/noter/lib/sui/grpc-client.ts, which has run this way in + * production since #684. + */ +import { SuiGrpcClient } from "@mysten/sui/grpc"; +import { enokiConfig } from "@/lib/enoki/config"; + +// Same hostnames Sui's own JSON-RPC used — gRPC-web is served from the same +// fullnode, dispatched by content-type/path rather than a separate host. +const GRPC_BASE_URLS = { + testnet: "https://fullnode.testnet.sui.io:443", + mainnet: "https://fullnode.mainnet.sui.io:443", +} as const; + +let cached: SuiGrpcClient | null = null; +let cachedNetwork: keyof typeof GRPC_BASE_URLS | null = null; + +/** Memoized SuiGrpcClient for the app's configured network. */ +export function getSuiGrpcClient(): SuiGrpcClient { + const network = enokiConfig.suiNetwork; + if (cached && cachedNetwork === network) return cached; + + cached = new SuiGrpcClient({ network, baseUrl: GRPC_BASE_URLS[network] }); + cachedNetwork = network; + return cached; +} diff --git a/docs/llms-full.txt b/docs/llms-full.txt index 7d6018b76..5e8d7ed0d 100644 --- a/docs/llms-full.txt +++ b/docs/llms-full.txt @@ -151,7 +151,8 @@ Returns: ```ts { restored: number; // Entries newly indexed - skipped: number; // Entries already in DB + skipped: number; // On-chain blobs already in the local success index + failed: number; // Permanent decrypt/UTF-8 failures (defaults to 0) total: number; // Total blobs found on-chain namespace: string; owner: string; diff --git a/docs/mcp/changelog.mdx b/docs/mcp/changelog.mdx index 9bc31ad4b..cd8cc258a 100644 --- a/docs/mcp/changelog.mdx +++ b/docs/mcp/changelog.mdx @@ -28,9 +28,30 @@ questions: - What changed in the MemWal MCP changelog? - When was the automatic memory plugin added to MemWal MCP? answer: >- - The latest MCP package release is 0.0.12. It forwards the MCP client's initialize.clientInfo to the relayer so sidecar logs can name the coding agent (Claude Code, Codex, Cursor, and others) on each session and tool call, and it resolves the credential directory on every access so MEMWAL_CREDS_DIR can override it. Version 0.0.11 makes memwal_logout cut the running bridge session so memory tools stop after sign-out. + The latest MCP package release is 0.0.13. A `memwal_remember`, `memwal_remember_bulk` or `memwal_analyze` whose reply never arrives is no longer reported as safe to retry: the relayer accepts those with HTTP 202 and finishes them in a durable queue, so the write may already have landed, and `/api/remember/bulk` has no idempotency key — repeating it stores a second paid copy. The tool now says the write may have completed and points at `memwal_recall` to check before re-saving, while a lost read still says plainly that retrying is safe. When the relayer rejects the saved delegate key, tool calls now get an auth error pointing at `memwal_login` instead of waiting minutes for a retry hint that cannot work. It writes the credentials file by creating a new 0600 file and renaming it into place, so a sign-in never puts the delegate private key into a credentials.json that a manual chmod or a restored backup left world-readable. A completed sign-in is now confirmed with a notification and a one-shot banner naming the account and the resolved credentials path, and the bridge keeps reading stdin after an in-session login instead of going deaf. Unrecognised command-line options now warn instead of being silently ignored, `--help` lists the network presets and the URLs each resolves to, `memwal_health` names the relayer the client dialled, and `memwal_restore` reports `failed` and retries the same page when truncation is a download or embed blip, instead of always telling the agent to raise `limit`. Plugin launch configs (`.mcp.json`, Cursor/Codex copies, and the Codex fallback installer) now pin `@mysten-incubation/memwal-mcp@0.0.13` so npx cannot keep a cached 0.0.5. Version 0.0.12 forwards the MCP client's initialize.clientInfo to the relayer so sidecar logs can name the coding agent (Claude Code, Codex, Cursor, and others) on each session and tool call, and it resolves the credential directory on every access so MEMWAL_CREDS_DIR can override it. --- +## 0.0.13 + +This release stops a write whose reply was lost from being reported as safe to retry — repeating one can store a second paid copy — answers tool calls with an auth error pointing at `memwal_login` when the relayer rejects the saved delegate key, writes the credentials file through a fresh `0600` file that it renames into place, confirms a completed sign-in and keeps the bridge reading stdin afterwards, warns on unrecognised command-line options instead of ignoring them, documents the network presets in `--help`, names the relayer in `memwal_health`, reports restore `failed` counts when truncation is a transient download or embed blip, and pins plugin launch configs (`.mcp.json`, Cursor/Codex copies, and the Codex fallback installer) so npx cannot keep a cached 0.0.5. + +### Fixed + +- Stop telling the user to retry a write whose reply was lost. A `memwal_remember`, `memwal_remember_bulk` or `memwal_analyze` that was POSTed and then timed out came back as "the connection to the relayer dropped before the result came back. Please retry." — but the relayer answers those with HTTP 202 and finishes the work in a durable queue, so a client-side deadline cancels nothing and the write may already have landed. `/api/remember/bulk` carries no idempotency key, unlike the single path, so following that advice stores a second paid copy that `recall` then hides behind the first. A sent write now says it may have completed, that the timeout did not undo it, and to check with `memwal_recall` before re-saving. A sent read still says plainly that retrying is safe. (WALM-618 follow-up) +- Answer tool calls with an auth error when the relayer rejects the saved delegate key, instead of parking them until the call deadline. A 401 on the SSE handshake was treated like any other connect failure, so the bridge retried a key that could never be accepted while the queued `memwal_recall` waited out the orphan sweeper — up to four minutes — and then came back as "the connection to the relayer dropped, please retry", advice that cannot work. The bridge now names the rejection and points at `memwal_login`, whether the key is rejected at startup or revoked mid-session, and refuses later requests immediately while it stays rejected. Any accepted handshake resumes normal buffering, so both a re-login and a transient WAF or rate-limit 401 recover on their own. Credentials are still never wiped automatically. (#365, WALM-602) +- Back off when the relayer refuses the SSE handshake with HTTP 429 instead of retrying ~500ms later. The bridge now honours a `Retry-After` (clamped to 60s) and falls back to a 5s floor when the header is absent — the `ip_active_cap` shape — and carries the deadline across reconnect attempts. It also prints one stderr line saying this is a rate limit rather than a bad config or bad credentials, so a throttled bridge no longer reads as a broken one. (WALM-386) +- Answer a request that is still buffered while the handshake keeps failing, instead of holding it for the full call timeout and then blaming a dropped reply. A request that was never sent cannot have executed, so after 90s of consecutive handshake failures the bridge fails it with the reason the handshake actually gave — rather than parking it four minutes and returning "the connection to the relayer dropped, please retry", which named the wrong layer and invited a retry of a `remember` that had never left the process. The full deadline still applies while the handshake is healthy. Override with `MEMWAL_MCP_STALLED_HANDSHAKE_MS`. (WALM-618) +- `memwal_restore` reports `failed` and retries the same page when `truncated` is a download/embed blip (`restored=0` and `skipped+failed < total`), instead of always telling the agent to raise `limit` (WALM-480). +- Unrecognised options now warn on stderr and in the structured log (`cli.unrecognised_arg`) instead of being dropped in silence, so a typo'd `--namesapce work` no longer writes to the default namespace with nothing to say it had. The warning names the option key only, keeping a mistyped value-taking flag (`--tokenn=hunter2`) from putting the secret on stderr, and it warns rather than exits so an option from a newer config cannot brick the server. (#630) +- `--help` now lists the network presets (`--prod`, `--dev`, `--staging`, `--local`) with the relayer and web URLs each resolves to, rendered from the preset table rather than retyped so a new preset cannot ship undocumented the way `--prod` did. The `--label` default is corrected to "MCP Client", which is what the code actually falls back to. (#630) +- `memwal_health` reports `relayer=`, naming the origin this process dialled, so a client bound to the wrong network finds out there instead of by noticing its memories are missing. The URL is captured when the call is sent rather than when the reply lands, so a reconnect mid-flight cannot label the answer with a relayer it did not come from. (#630) +- Keep reading stdin after an in-session `memwal_login`. The auth-required stub hands off to the bridge by pausing stdin, and a stream paused that way does not resume when a new `data` listener attaches, so the bridge served only the request replayed from the hand-off and then went deaf. Signing in appeared to work, the retry succeeded, and every call after it hung until the client timed it out. (#633) +- Confirm a completed sign-in instead of only writing it to the log file. The failure path already reported itself twice (a notification and a notice on the next tool call) while success reported nothing, so a user who approved in the browser could not tell whether credentials had landed, the bridge had adopted them, or a retry was worth trying. Success now sends the matching notification and prefixes a one-shot banner naming the account, the delegate, and the resolved credentials path onto the next tool result. (#633) +- Stop the `memwal_login` prompt claiming you are already signed in when you are not. The bridge assumed it only ever runs with credentials on disk, but `memwal_logout` deletes them and login is intercepted before the signed-out guard, so a login after a logout in the same session announced that a stored delegate key would be replaced, reading as though the logout had not taken. Both prompts now read the credentials file instead of assuming the mode. (#633) +- Serve one `memwal_login` prompt in both modes. The signed-out stub and the signed-in bridge had drifted into different assistant instructions, step wording, and closing lines, and both claimed credentials land at `~/.memwal/credentials.json` even when the resolved path was project-local. (#633, #628) +- Write the credentials file by creating a new `0600` file and renaming it into place, instead of writing the delegate private key to the existing path and tightening the permission afterwards. `writeFileSync`'s `mode` only applies when it creates the file, so a `credentials.json` left world-readable by anything outside the CLI (a manual `chmod`, a restored backup, another tool) received the new key under the old permission until the following `chmod` landed. The CLI now writes the backup it takes when signing in as a different account the same way; that backup holds the same plaintext key, and the CLI previously created it under the process umask. On Windows, where `rename` cannot replace a destination another process holds open and POSIX mode bits are not enforced at all, the CLI retries briefly and then writes in place rather than failing the sign-in. (#520) +- Plugin launch configs (`.mcp.json`, Cursor/Codex copies, and the Codex fallback installer) now pin `@mysten-incubation/memwal-mcp@0.0.13` so npx cannot keep a cached 0.0.5. (WALM-627) + ## 0.0.12 This release forwards the MCP client's identity to the relayer so sidecar logs can name the coding agent, and it resolves the credential directory on every access. diff --git a/docs/mcp/claude-code.md b/docs/mcp/claude-code.md index e442b4779..160969ebe 100644 --- a/docs/mcp/claude-code.md +++ b/docs/mcp/claude-code.md @@ -97,7 +97,7 @@ Add MemWal to Claude Code so it recalls context and saves durable facts as you w | MemWal MCP (memory tools) | ✓ | ✓ | | Lifecycle hooks (automatic recall/save) | ✓ | ✗ | -MCP-only still saves and recalls on its own because the tools are proactive. The plugin adds hooks that reinforce the behavior and make the agent **prefer Walrus Memory over Claude Code's built-in memory**. +MCP-only still saves and recalls on its own because the tools are proactive. The plugin adds hooks that reinforce the behavior and make the agent **prefer Walrus Memory over Claude Code's built-in memory**. The plugin pins the MCP server version, so `npx` cannot keep a cached older package. ## Available tools diff --git a/docs/ops/migration-completion-artifact.md b/docs/ops/migration-completion-artifact.md new file mode 100644 index 000000000..a32efb1fb --- /dev/null +++ b/docs/ops/migration-completion-artifact.md @@ -0,0 +1,78 @@ +## Migration completion artifact + +After a security migration, write a completion artifact that records the target +package, reviewed manifest digest, imported counts, verification result, and +approver. Keep the file with the operator record. This is not a full release +checklist and it is not the in-cluster completion-report consumed as +`COMPLETION_EVIDENCE_JSON`. Security RC classification and review are in +`docs/ops/security-rc-workflow.md`. + +`scripts/build-finalize-tx.ts` still requires `MANIFEST_SHA256` and a fresh +completion-report. Ceremony GitHub Environments are checked by +`scripts/verify-migration-environments.sh` and +`.github/workflows/verify-migration-environments.yml`. + +## Schema + +```json +{ + "packageId": "0x…", + "manifestSha256": "0123456789abcdef0123456789abcdef0123456789abcdef0123456789abcdef", + "imported": 12, + "skipped": 0, + "verified": true, + "approver": "user:alice", + "timestamp": "2026-08-31T12:00:00.000Z" +} +``` + +| Field | Meaning | +| --- | --- | +| `packageId` | Target (destination) package id, in the form `build-finalize-tx.ts` normalizes `PACKAGE_ID` to: lowercase, `0x`-prefixed, 32 bytes of hex | +| `manifestSha256` | Independently reviewed migration manifest digest (same value as `MANIFEST_SHA256` for finalize-tx) | +| `imported` | Count of imported records | +| `skipped` | Count of skipped records | +| `verified` | Verification result (`true` if checks passed) | +| `approver` | Reviewer who approved the ceremony | +| `timestamp` | UTC time the artifact was written (ISO-8601) | + +## How to fill + +1. Confirm ceremony environments still match the reviewer allowlist (`scripts/verify-migration-environments.sh`). +2. Copy `packageId` from the destination package used with `scripts/build-finalize-tx.ts` (`PACKAGE_ID`). +3. Copy `manifestSha256` from the independently reviewed digest (`MANIFEST_SHA256`). +4. Set `imported` and `skipped` from the import totals. +5. Set `verified` from the verification result. +6. Set `approver` to the reviewer who approved the ceremony (not the initiator). +7. Write the file: + +```bash +node scripts/write-migration-completion-artifact.mjs \ + --package-id "$PACKAGE_ID" \ + --manifest-sha256 "$MANIFEST_SHA256" \ + --imported "$IMPORTED" \ + --skipped "$SKIPPED" \ + --verified true \ + --approver "user:harrymove-ctrl" \ + --out ./migration-completion-artifact.json +``` + +Flags override env of the same name (`PACKAGE_ID`, `MANIFEST_SHA256`, +`IMPORTED`, `SKIPPED`, `VERIFIED`, `APPROVER`, `OUT`). Missing required fields +exit 1, and so does an unrecognized flag: a misspelled `--approver` would +otherwise fall back to `APPROVER` and record an approver nobody typed. + +The writer rejects a `packageId` that `scripts/build-finalize-tx.ts` would +reject, and records it in the same normalized form, so the artifact and the +transaction name the target package identically. It also refuses to overwrite an +existing `--out` file: pass `--force` to replace one deliberately. `--force` is +a flag only, with no env equivalent. + +The artifact is an operator record, not a control the tooling enforces. Nothing checks that +`approver` is a real reviewer or that it differs from the operator running the +command; step 6 is a procedure the ceremony follows, and the ceremony +environments in `scripts/verify-migration-environments.sh` are what actually +prevent self-review. + +Signing and submitting finalize-tx remains a separate offline step; see +`scripts/build-finalize-tx.ts` and `.github/workflows/finalize-tx.yml`. diff --git a/docs/ops/security-rc-workflow.md b/docs/ops/security-rc-workflow.md new file mode 100644 index 000000000..1b86140d0 --- /dev/null +++ b/docs/ops/security-rc-workflow.md @@ -0,0 +1,38 @@ +## Security RC review workflow + +Every release-candidate batch gets Security review capacity (human, AI, or +combined) before it ships. Classify the change set, keep one candidate in one +PR, and request Security review when the change requires it. + +## One candidate per PR + +One candidate release, or one security-sensitive change set, is one PR. Do not +mix contract, demo, and docs work in an RC. + +## Classification + +**Security review required:** request Security review on the PR: + +- Move contract / SEAL policy +- Relayer auth +- Sidecar SEAL encrypt / decrypt +- Migration ceremony (manifest, finalize-tx, environments) +- Anything that changes who can decrypt + +**Security review not required:** normal Eng review: + +- Docs-only changes +- Demo apps +- Changelog / version dump +- Tests that do not change production policy + +## How + +Request review from Security (or the postmortem Security owners) on that single +PR. When the change is a security migration, record the approver on the +migration completion artifact (`docs/ops/migration-completion-artifact.md`). + +## TDD bar + +The PR body lists test proof, method, and reproduction. Prefer unit tests plus +connected-surface integration / e2e over coverage slogans. diff --git a/docs/python-sdk/api-reference.md b/docs/python-sdk/api-reference.md index 1c31f0b76..b7816cb41 100644 --- a/docs/python-sdk/api-reference.md +++ b/docs/python-sdk/api-reference.md @@ -174,11 +174,11 @@ AskResult( Rebuild missing indexed entries for one namespace from Walrus. Incremental. - `limit` defaults to `10` and caps the inspected blob set, newest-first -- `restored` counts blobs re-indexed in this call; `skipped` counts blobs already in the local index +- `restored` counts blobs re-indexed in this call; `skipped` counts onchain blobs already in the local success index; `failed` counts permanent decrypt/UTF-8 failures (defaults to `0`) - There is no pagination cursor; use a larger `limit` for larger one-shot restores ```python -RestoreResult(restored: int, skipped: int, total: int, namespace: str, owner: str, truncated: bool = False) +RestoreResult(restored: int, skipped: int, total: int, namespace: str, owner: str, truncated: bool = False, failed: int = 0) ``` `truncated=true` is known-retryable-incomplete (this call's `limit`, or a still-expandable sidecar candidate fetch); `truncated=false` is not proof the sidecar saw every onchain blob (WALM-451 `sourceCapped`). diff --git a/docs/python-sdk/changelog.mdx b/docs/python-sdk/changelog.mdx index bd4214549..28977755e 100644 --- a/docs/python-sdk/changelog.mdx +++ b/docs/python-sdk/changelog.mdx @@ -29,13 +29,21 @@ questions: - What changes were made in memwal 0.1.4? - Where can I find the release history for the Walrus Memory Python SDK? answer: >- - The latest Python SDK release is 0.1.9. It reports HTTP 503 as a retryable upstream outage instead of a credential failure, rejects empty `remember_bulk_async` batches and misaligned relayer `job_ids`, aligns restore `truncated` docs with WALM-431 retryable semantics, and warns when `server_url` uses plaintext HTTP on a non-localhost host without logging URL credentials. 0.1.8 added `dropped_count` on recall results, `write_ready` on health, `MemWalClockDriftError` for clock-drift 401s, and the `dev` relayer preset. + The latest Python SDK release is 0.1.10. `restore()` results include `failed` (default `0`) for permanent decrypt/UTF-8 failures instead of folding them into `skipped`. 0.1.9 reports HTTP 503 as a retryable upstream outage instead of a credential failure, rejects empty `remember_bulk_async` batches and misaligned relayer `job_ids`, aligns restore `truncated` docs with WALM-431 retryable semantics, and warns when `server_url` uses plaintext HTTP on a non-localhost host without logging URL credentials. --- Track what's new, changed, and fixed in `memwal` (Python). For the latest version, see the [PyPI project page](https://pypi.org/project/memwal/). +## 0.1.10 + +This release adds `failed` on `restore()` results for permanent decrypt and UTF-8 failures. + +### Added + +- `restore()` results include `failed` (default `0`) for permanent decrypt/UTF-8 failures instead of folding them into `skipped` or dropping them silently. + ## 0.1.9 This release reports HTTP 503 as a retryable upstream outage, aligns Python bulk remember with the TypeScript SDK by refusing empty batches and mismatched job ids, and warns when `server_url` uses plaintext HTTP on a non-localhost host. diff --git a/docs/reference/environment-variables.md b/docs/reference/environment-variables.md index 55b554223..cb6d095cf 100644 --- a/docs/reference/environment-variables.md +++ b/docs/reference/environment-variables.md @@ -72,6 +72,8 @@ The stdio MCP package reads these environment variables directly. A CLI flag tak | `MEMWAL_CREDS_DIR` | none | `~/.memwal` | Directory holding `credentials.json`. Overrides both project-local and `~/.memwal` credentials, re-read on every access so a test can redirect it after import. Mainly for tests, which must not write into the real credential directory | | `MEMWAL_MCP_SSE_IDLE_MS` | none | `30000` | Maximum milliseconds of silence on the SSE stream before the bridge treats the session as dead and reconnects. Values below `500` are ignored and fall back to the default. Mainly for tests | | `MEMWAL_MCP_CALL_TIMEOUT_MS` | none | `240000` | Maximum milliseconds a single request might wait for its response before the bridge answers with a retryable error. Covers a reply lost while the stream itself stays healthy, which `MEMWAL_MCP_SSE_IDLE_MS` cannot detect. The default is derived in code from the slowest server-side tool deadline plus headroom, so it moves with that tool rather than being pinned here. Values below `1000` are ignored and fall back to the default | +| `MEMWAL_MCP_THROTTLE_FLOOR_MS` | none | `5000` | Minimum milliseconds the bridge waits before retrying an SSE handshake the relayer refused with HTTP 429 and no `Retry-After`. A `Retry-After` on the response wins instead. Either way the wait is capped at `60000`. Non-numeric or negative values are ignored and fall back to the default. Mainly for tests | +| `MEMWAL_MCP_STALLED_HANDSHAKE_MS` | none | `90000` | Milliseconds a request may stay buffered while the SSE handshake keeps failing before the bridge answers it with the handshake's own error. Applies only to requests that were never sent, and only while the handshake is failing; a request buffered behind a healthy connection keeps `MEMWAL_MCP_CALL_TIMEOUT_MS`. Clamped to never exceed that call timeout. Values below `1000` are ignored and fall back to the default. Mainly for tests | | `MEMWAL_MCP_LOGIN_TIMEOUT_MS` | none | `300000` | Maximum milliseconds the local sign-in listener stays bound waiting for the browser callback. The default gives you time to review a wallet prompt; shorten it only in tests. Values below `100` are ignored and fall back to the default | ## Self-hosted relayer @@ -146,10 +148,13 @@ These are not all enforced at boot, but most real deployments need them. | `WALLET_BALANCE_LOW_THRESHOLD_SUI` | `5000000000` | Uploader SUI address-balance threshold in MIST (5 SUI). Load-bearing during phase 1, when durable register pays gas from the uploader wallet | | `SPONSOR_BALANCE_LOW_THRESHOLD_SUI` | `5000000000` | Sponsor wallet SUI address-balance threshold in MIST (5 SUI) | | `WALLET_BALANCE_LOW_ALERT_DEDUP_SECS` | `43200` | Dedup window for wallet low-balance Slack alerts, per `(network, wallet type, token, address)` | -| `MEMWAL_RELAYER_URL` | `http://127.0.0.1:$PORT` | Relayer URL passed from the Rust server to the sidecar for MCP tool calls | +| `MEMWAL_RELAYER_URL` | unset | The deployment's **public identity**. It is the MCP OAuth issuer, and the network `memwal_health` names. Keep it set on any deployment with MCP OAuth — unsetting it disables OAuth entirely. It does **not** decide what the sidecar dials | +| `MEMWAL_SIDECAR_RELAYER_URL` | `http://127.0.0.1:$PORT` | Address the sidecar **dials** for every MCP tool call. Loopback, because the managed sidecar is a child of the relayer process. Set it only if this sidecar genuinely serves a relayer in another process — a non-loopback value sends each `memwal_*` call out of the container and back through the public edge, and the server warns at startup when it sees one | +| `MEMWAL_PUBLIC_RELAYER_URL` | falls back to `MEMWAL_RELAYER_URL` | Public origin `memwal_health` reports, when it should differ from `MEMWAL_RELAYER_URL` | | `MCP_MAX_TOTAL_SESSIONS` | `1000` | Maximum active MCP sessions across SSE and Streamable HTTP transports | | `MCP_MAX_SESSIONS_PER_IP` | `16` | Maximum active MCP sessions from one source IP | | `MCP_MAX_NEW_SESSIONS_PER_IP_PER_MIN` | `30` | Maximum new MCP sessions opened by one source IP per minute | +| `MCP_TOOL_SLOW_WARN_MS` | `5000` | An MCP tool call still running at this duration is logged as `tool.slow` (`settled: false`) at `warn` — which is what makes a hang visible, since a hang never settles. A call that finishes at or above it is logged as `tool.slow` (`settled: true`) instead of `tool.done` | | `TRUSTED_PROXY_HOPS` | `0` | Number of trusted reverse-proxy hops to walk from the right of `X-Forwarded-For`; `0` ignores XFF and uses the TCP peer | | `WRITES_PAUSED` | `false` | When `1` / `true` / `yes` / `on`, write routes (`POST /api/remember`, `/api/remember/manual`, `/api/remember/bulk`, `/api/analyze`) return HTTP 503 `{"error":"writes are paused"}`. `GET /health` stays HTTP 200 with `status: "ok"` and `writes: "paused"`. Reads (`recall`, `restore`, health) stay available | @@ -174,7 +179,9 @@ These are not all enforced at boot, but most real deployments need them. - The sidecar `POST /walrus/upload` route defaults Walrus storage epochs by network: `50` on `testnet` (about 50 days) and `2` on `mainnet` (about 4 weeks), unless the request explicitly passes `epochs`. - `MEMWAL_PACKAGE_ID` and `MEMWAL_REGISTRY_ID` are server env vars. Do not replace them with `VITE_*` app env vars. - For network-specific `MEMWAL_PACKAGE_ID` and `MEMWAL_REGISTRY_ID` values, see [Contract Overview](/contract/overview). -- `MEMWAL_RELAYER_URL` is only needed when the sidecar should call a different relayer URL than the Rust server's local port. The Rust server sets it automatically to `http://127.0.0.1:$PORT` for the managed sidecar when it starts. +- `MEMWAL_SIDECAR_RELAYER_URL` is the only variable that changes where the sidecar connects, and it is only needed when the sidecar must reach a relayer in another process. Leave it unset: the managed sidecar is a child of the Rust server, which points it at `http://127.0.0.1:$PORT`. Do not reach for `MEMWAL_RELAYER_URL` to redirect it — that variable is the deployment's public identity and the OAuth issuer, and unsetting it takes `McpOAuthConfig::from_env` to `None`, which refuses every OAuth handshake with `OauthNotConfigured`. +- `MEMWAL_RELAYER_URL` names the deployment; it does not route anything. It is the OAuth issuer (see `McpOAuthConfig::from_env`) and the origin `memwal_health` reports, and MCP OAuth stops working if it is unset — so leave it set, and do not reach for it to change where the sidecar connects. Only `MEMWAL_SIDECAR_RELAYER_URL` moves the dial, and it should stay unset: the managed sidecar runs as a child of the relayer process, so loopback is the relayer it needs. Pointing it at a public hostname makes every `memwal_*` tool call leave the container and return through the edge. +- `memwal_health` reports only a stated public origin, never the loopback dial address, because an address that names no network would make a client bound to the wrong relayer read as correctly configured. Clients run through the `memwal-mcp` stdio package always see the relayer that package dialled, whether or not this is set. ## Frontend apps diff --git a/docs/relayer/api-reference.md b/docs/relayer/api-reference.md index df431351f..55a6446a7 100644 --- a/docs/relayer/api-reference.md +++ b/docs/relayer/api-reference.md @@ -84,7 +84,7 @@ Service liveness check. `status` is `"ok"` when the relayer process is up. HTTP `writes` is `"ok"` or `"paused"`. `"paused"` when `WRITES_PAUSED` is set (`1` / `true` / `yes`); empty or unset is `"ok"`. That flag is write-path admission, not a health-only signal: `POST /api/remember`, `/api/remember/manual`, `/api/remember/bulk`, and `/api/analyze` then return HTTP 503 with `{"error":"writes are paused"}`. `/health` itself stays HTTP 200 with `status: "ok"` and `writes: "paused"`, so clients can distinguish an intentional pause from an integrator bug. Reads (`recall`, `restore`, remember job status) stay available. -`write_ready` is `true` when the encryption sidecar process answered its own `/health` (cached a few seconds). That is sidecar liveness only, not a write-pause flag and not a guarantee that remember or analyze succeed. A sidecar outage can still return HTTP 200 with `write_ready: false`. Use `writes`, not `write_ready`, for the pause signal. +`write_ready` is `true` when the encryption sidecar process answered its own `/health` **and** Postgres can accept writes (cached a few seconds). Postgres is considered not writable when Neon cluster size (`pg_cluster_size`, not `pg_database_size` of this database) is at or within 1MB of `neon.max_cluster_size`. If `pg_cluster_size()` is missing (no `neon` extension), the probe falls back to `sum(pg_database_size)` against that GUC. Self-hosted Postgres without the GUC keeps the sidecar-only check. Probe errors and timeouts fail open (`write_ready` stays true) so CI is not blocked; timeouts log at warn. A sidecar outage or a disk/project-size write outage can still return HTTP 200 with `write_ready: false`. Use `writes`, not `write_ready`, for the pause signal. `write_ready: true` is not a guarantee that remember or analyze succeed. **Response:** @@ -462,6 +462,7 @@ Rebuild missing vector entries for one namespace. Queries onchain blobs by owner { "restored": 3, "skipped": 7, + "failed": 0, "total": 10, "namespace": "demo", "owner": "0x...", @@ -469,6 +470,8 @@ Rebuild missing vector entries for one namespace. Queries onchain blobs by owner } ``` +`skipped` is onchain blobs already in the local success index. `failed` is permanent decrypt/UTF-8 failures on this onchain page (negative-cache hits plus new permanent failures this call), so it never exceeds `total`. Transient download, decrypt, or embed errors are not counted in `failed`; when a page yields only those, `truncated` is true so the caller retries. + `truncated=true` means this restore is **known-retryable-incomplete**: more missing blobs than `limit` allowed this call to restore, **or** the sidecar's owner-wide candidate fetch hit its cap **and** raising `limit` can still expand that fetch (`limit < 20`). Once the sidecar cap is saturated (`limit >= 20`, cap pinned at 100), truncation follows this call's missing-blob page length, not onchain `total`. A fully restored namespace does not loop. `truncated=false` is **not** proof the sidecar saw every onchain blob; blobs beyond the owner-wide sidecar candidate cap can still be missing. WALM-451 tracks a `sourceCapped` field for that case ([WALM-451](https://linear.app/mysten-labs/issue/WALM-451)). Relayers older than WALM-319 omit `truncated`; SDKs default it to `false`. ### `POST /api/forget` diff --git a/docs/sdk/api-reference.md b/docs/sdk/api-reference.md index ffe67e6c2..3e3054c66 100644 --- a/docs/sdk/api-reference.md +++ b/docs/sdk/api-reference.md @@ -262,7 +262,8 @@ Rebuild missing indexed entries for one namespace from Walrus. Incremental — o ```ts { restored: number; // Entries newly indexed - skipped: number; // Entries already in DB + skipped: number; // On-chain blobs already in the local success index + failed: number; // Permanent decrypt/UTF-8 failures (defaults to 0) total: number; // Total blobs found on-chain namespace: string; owner: string; diff --git a/docs/sdk/changelog.mdx b/docs/sdk/changelog.mdx index 8e66ac7ae..27064137e 100644 --- a/docs/sdk/changelog.mdx +++ b/docs/sdk/changelog.mdx @@ -28,9 +28,24 @@ questions: - When was bulk remember added to the Walrus Memory SDK? - What security improvements have been made to the MemWal SDK? answer: >- - The latest TypeScript SDK release is 0.1.6. It adds optional `created_at` on `recall()` results, plus `sort` and `scoringWeights` on `RecallOptions`. HTTP 503 from the relayer is reported as a retryable upstream outage instead of a sign-in failure. 0.1.5 added `dropped_count` on recall results and `write_ready` on health, switched `rememberManual` to sending `encryptedData`, and hardened hex decoding and error redaction. + The latest TypeScript SDK release is 0.1.7. Request bodies are hashed with `@noble/hashes` rather than WebCrypto-or-`node:crypto`, so the SDK no longer imports a Node builtin that Vite silently externalises into a runtime crash in the browser, and it declares a Node 20 floor. `restore()` results include `failed` (required like `truncated`; SDK defaults omitted to `0`) for permanent decrypt/UTF-8 failures instead of folding them into `skipped`. Empty-body 401s use the AUTH_REJECTED troubleshooting message instead of telling callers to run `memwal_login`. Account and manual PTBs use typed `tx.pure` helpers so they work with modern `@mysten/sui`. 0.1.6 added optional `created_at` on `recall()` results, plus `sort` and `scoringWeights` on `RecallOptions`, and reports HTTP 503 as a retryable upstream outage instead of a sign-in failure. --- +## 0.1.7 + +This release removes the Node `crypto` import that crashed bundled browser builds at runtime, adds `failed` on `restore()` results, declares a Node 20 floor, stops telling headless SDK clients to call `memwal_login` on empty-body 401s, and switches account and manual PTBs to typed `tx.pure` helpers. + +### Added + +- `restore()` results include `failed` (required like `truncated`; SDK defaults omitted to `0`) for permanent decrypt/UTF-8 failures instead of folding them into `skipped` or dropping them silently. + +### Fixed + +- Hash request bodies with `@noble/hashes` instead of WebCrypto-or-`node:crypto`, so the package no longer imports a Node builtin on a browser-reachable path. Vite externalises such an import without warning: the app builds clean and the browser crashes the first time the path runs. The fallback could never have helped a browser anyway — `crypto.subtle` is absent precisely when the page is not a secure context, where `node:crypto` is absent too — so it only served Node <19 while being the sole source of the exposure. `sha256hex` sits on the signed-request path, so every remember and recall reached it. (#322, WALM-136) +- Declare `engines.node >= 20.0.0`, matching `memwal-mcp` and `openclaw-memory-memwal`. The SDK was the only published package without a floor. (WALM-599) +- Empty-body 401s now use the same AUTH_REJECTED troubleshooting message as credential 401s instead of telling callers to run `memwal_login`. Headless SDK clients do not have that MCP tool. +- `account.ts` and `manual.ts` PTBs use typed `tx.pure` helpers instead of the legacy untyped moveCall argument syntax that fails under modern `@mysten/sui`. + ## 0.1.6 This release adds write-time on recall results, lets callers sort by recency or pass scoring weights, and stops mapping relayer 503s to a sign-in failure. diff --git a/packages/mcp/CHANGELOG.md b/packages/mcp/CHANGELOG.md index 14e24f162..309ffa8b0 100644 --- a/packages/mcp/CHANGELOG.md +++ b/packages/mcp/CHANGELOG.md @@ -1,5 +1,24 @@ # @mysten-incubation/memwal-mcp +## 0.0.13 + +### Fixed + +- Stop telling the user to retry a write whose reply was lost. A `memwal_remember`, `memwal_remember_bulk` or `memwal_analyze` that was POSTed and then timed out came back as "the connection to the relayer dropped before the result came back. Please retry." — but the relayer answers those with HTTP 202 and finishes the work in a durable queue, so a client-side deadline cancels nothing and the write may already have landed. `/api/remember/bulk` carries no idempotency key, unlike the single path, so following that advice stores a second paid copy that `recall` then hides behind the first. A sent write now says it may have completed, that the timeout did not undo it, and to check with `memwal_recall` before re-saving. A sent read still says plainly that retrying is safe. (WALM-618 follow-up) +- Answer tool calls with an auth error when the relayer rejects the saved delegate key, instead of parking them until the call deadline. A 401 on the SSE handshake was treated like any other connect failure, so the bridge retried a key that could never be accepted while the queued `memwal_recall` waited out the orphan sweeper — up to four minutes — and then came back as "the connection to the relayer dropped, please retry", advice that cannot work. The bridge now names the rejection and points at `memwal_login`, whether the key is rejected at startup or revoked mid-session, and refuses later requests immediately while it stays rejected. Any accepted handshake resumes normal buffering, so both a re-login and a transient WAF or rate-limit 401 recover on their own. Credentials are still never wiped automatically. (#365, WALM-602) +- Back off when the relayer refuses the SSE handshake with HTTP 429 instead of retrying ~500ms later. The bridge now honours a `Retry-After` (clamped to 60s) and falls back to a 5s floor when the header is absent — the `ip_active_cap` shape — and carries the deadline across reconnect attempts. It also prints one stderr line saying this is a rate limit rather than a bad config or bad credentials, so a throttled bridge no longer reads as a broken one. (WALM-386) +- Answer a request that is still buffered while the handshake keeps failing, instead of holding it for the full call timeout and then blaming a dropped reply. A request that was never sent cannot have executed, so after 90s of consecutive handshake failures the bridge fails it with the reason the handshake actually gave — rather than parking it four minutes and returning "the connection to the relayer dropped, please retry", which named the wrong layer and invited a retry of a `remember` that had never left the process. The full deadline still applies while the handshake is healthy. Override with `MEMWAL_MCP_STALLED_HANDSHAKE_MS`. (WALM-618) +- `memwal_restore` reports `failed` and retries the same page when `truncated` is a download/embed blip (`restored=0` and `skipped+failed < total`), instead of always telling the agent to raise `limit` (WALM-480). +- Unrecognised options now warn on stderr and in the structured log (`cli.unrecognised_arg`) instead of being dropped in silence, so a typo'd `--namesapce work` no longer writes to the default namespace with nothing to say it had. The warning names the option key only, keeping a mistyped value-taking flag (`--tokenn=hunter2`) from putting the secret on stderr, and it warns rather than exits so an option from a newer config cannot brick the server. (#630) +- `--help` now lists the network presets (`--prod`, `--dev`, `--staging`, `--local`) with the relayer and web URLs each resolves to, rendered from the preset table rather than retyped so a new preset cannot ship undocumented the way `--prod` did. The `--label` default is corrected to "MCP Client", which is what the code actually falls back to. (#630) +- `memwal_health` reports `relayer=`, naming the origin this process dialled, so a client bound to the wrong network finds out there instead of by noticing its memories are missing. The URL is captured when the call is sent rather than when the reply lands, so a reconnect mid-flight cannot label the answer with a relayer it did not come from. (#630) +- Keep reading stdin after an in-session `memwal_login`. The auth-required stub hands off to the bridge by pausing stdin, and a stream paused that way does not resume when a new `data` listener attaches, so the bridge served only the request replayed from the hand-off and then went deaf. Signing in appeared to work, the retry succeeded, and every call after it hung until the client timed it out. (#633) +- Confirm a completed sign-in instead of only writing it to the log file. The failure path already reported itself twice (a notification and a notice on the next tool call) while success reported nothing, so a user who approved in the browser could not tell whether credentials had landed, the bridge had adopted them, or a retry was worth trying. Success now sends the matching notification and prefixes a one-shot banner naming the account, the delegate, and the resolved credentials path onto the next tool result. (#633) +- Stop the `memwal_login` prompt claiming you are already signed in when you are not. The bridge assumed it only ever runs with credentials on disk, but `memwal_logout` deletes them and login is intercepted before the signed-out guard, so a login after a logout in the same session announced that a stored delegate key would be replaced, reading as though the logout had not taken. Both prompts now read the credentials file instead of assuming the mode. (#633) +- Serve one `memwal_login` prompt in both modes. The signed-out stub and the signed-in bridge had drifted into different assistant instructions, step wording, and closing lines, and both claimed credentials land at `~/.memwal/credentials.json` even when the resolved path was project-local. (#633, #628) +- Write the credentials file by creating a new `0600` file and renaming it into place, instead of writing the delegate private key to the existing path and tightening the permission afterwards. `writeFileSync`'s `mode` only applies when it creates the file, so a `credentials.json` left world-readable by anything outside the CLI (a manual `chmod`, a restored backup, another tool) received the new key under the old permission until the following `chmod` landed. The CLI now writes the backup it takes when signing in as a different account the same way; that backup holds the same plaintext key, and the CLI previously created it under the process umask. On Windows, where `rename` cannot replace a destination another process holds open and POSIX mode bits are not enforced at all, the CLI retries briefly and then writes in place rather than failing the sign-in. (#520) +- Plugin launch configs (`.mcp.json`, Cursor/Codex copies, and the Codex fallback installer) now pin `@mysten-incubation/memwal-mcp@0.0.13` so npx cannot keep a cached 0.0.5. (WALM-627) + ## 0.0.12 ### Added diff --git a/packages/mcp/package.json b/packages/mcp/package.json index 20be56068..44738b58d 100644 --- a/packages/mcp/package.json +++ b/packages/mcp/package.json @@ -1,6 +1,6 @@ { "name": "@mysten-incubation/memwal-mcp", - "version": "0.0.12", + "version": "0.0.13", "description": "Walrus Memory MCP client — single-binary stdio MCP server that bridges Cursor / Claude Desktop / Antigravity / Claude Code to the Walrus Memory relayer. Handles browser-based wallet login on first run.", "type": "module", "engines": { diff --git a/packages/mcp/plugin/.claude-plugin/plugin.json b/packages/mcp/plugin/.claude-plugin/plugin.json index 94ce91fb3..66925ea5d 100644 --- a/packages/mcp/plugin/.claude-plugin/plugin.json +++ b/packages/mcp/plugin/.claude-plugin/plugin.json @@ -1,6 +1,6 @@ { "name": "memwal", - "version": "0.0.12", + "version": "0.0.13", "description": "Automatic Walrus Memory for Claude Code — proactive recall and durable-fact saving via the MemWal MCP + lifecycle hooks.", "author": { "name": "Mysten Labs" diff --git a/packages/mcp/plugin/.codex-mcp.json b/packages/mcp/plugin/.codex-mcp.json index 6051f56e7..345b9c487 100644 --- a/packages/mcp/plugin/.codex-mcp.json +++ b/packages/mcp/plugin/.codex-mcp.json @@ -2,7 +2,7 @@ "mcpServers": { "memwal": { "command": "npx", - "args": ["-y", "@mysten-incubation/memwal-mcp"] + "args": ["-y", "@mysten-incubation/memwal-mcp@0.0.13"] } } } diff --git a/packages/mcp/plugin/.codex-plugin/plugin.json b/packages/mcp/plugin/.codex-plugin/plugin.json index b7db62a44..bdbfa5d91 100644 --- a/packages/mcp/plugin/.codex-plugin/plugin.json +++ b/packages/mcp/plugin/.codex-plugin/plugin.json @@ -1,6 +1,6 @@ { "name": "memwal", - "version": "0.0.12", + "version": "0.0.13", "description": "Persistent Walrus Memory for Codex. Remembers decisions, preferences, and project context across sessions.", "author": { "name": "Mysten Labs", diff --git a/packages/mcp/plugin/.cursor-mcp.json b/packages/mcp/plugin/.cursor-mcp.json index 6051f56e7..345b9c487 100644 --- a/packages/mcp/plugin/.cursor-mcp.json +++ b/packages/mcp/plugin/.cursor-mcp.json @@ -2,7 +2,7 @@ "mcpServers": { "memwal": { "command": "npx", - "args": ["-y", "@mysten-incubation/memwal-mcp"] + "args": ["-y", "@mysten-incubation/memwal-mcp@0.0.13"] } } } diff --git a/packages/mcp/plugin/.cursor-plugin/plugin.json b/packages/mcp/plugin/.cursor-plugin/plugin.json index a4c5c8adf..68bfe76ff 100644 --- a/packages/mcp/plugin/.cursor-plugin/plugin.json +++ b/packages/mcp/plugin/.cursor-plugin/plugin.json @@ -1,6 +1,6 @@ { "name": "memwal", - "version": "0.0.12", + "version": "0.0.13", "description": "Automatic Walrus Memory for Cursor — proactive recall and durable-fact saving via the MemWal MCP + lifecycle hooks.", "author": { "name": "Mysten Labs" }, "homepage": "https://memory.walrus.xyz", diff --git a/packages/mcp/plugin/.mcp.json b/packages/mcp/plugin/.mcp.json index 6051f56e7..345b9c487 100644 --- a/packages/mcp/plugin/.mcp.json +++ b/packages/mcp/plugin/.mcp.json @@ -2,7 +2,7 @@ "mcpServers": { "memwal": { "command": "npx", - "args": ["-y", "@mysten-incubation/memwal-mcp"] + "args": ["-y", "@mysten-incubation/memwal-mcp@0.0.13"] } } } diff --git a/packages/mcp/plugin/plugin.json b/packages/mcp/plugin/plugin.json index 4470d50a5..17b1993e0 100644 --- a/packages/mcp/plugin/plugin.json +++ b/packages/mcp/plugin/plugin.json @@ -1,7 +1,7 @@ { "id": "memwal", "name": "memwal", - "version": "0.0.12", + "version": "0.0.13", "description": "Automatic Walrus Memory for Antigravity — proactive recall and durable-fact saving via the MemWal MCP + lifecycle hooks.", "author": { "name": "Mysten Labs" }, "homepage": "https://memory.walrus.xyz", diff --git a/packages/mcp/plugin/scripts/install_codex_hooks.mjs b/packages/mcp/plugin/scripts/install_codex_hooks.mjs index cdf7a4c99..f4d2a71fe 100644 --- a/packages/mcp/plugin/scripts/install_codex_hooks.mjs +++ b/packages/mcp/plugin/scripts/install_codex_hooks.mjs @@ -32,6 +32,10 @@ import { fileURLToPath } from "node:url"; const SCRIPT_DIR = dirname(fileURLToPath(import.meta.url)); const PLUGIN_ROOT = dirname(SCRIPT_DIR); +function resolveMcpVersion() { + return JSON.parse(readFileSync(join(PLUGIN_ROOT, "plugin.json"), "utf8")).version; +} + const CODEX_DIR = join(homedir(), ".codex"); const HOOKS_FILE = join(CODEX_DIR, "hooks.json"); const CONFIG_FILE = join(CODEX_DIR, "config.toml"); @@ -98,10 +102,11 @@ function ensureMcpRegistered() { mkdirSync(CODEX_DIR, { recursive: true }); let content = existsSync(CONFIG_FILE) ? readFileSync(CONFIG_FILE, "utf8") : ""; if (content.includes("[mcp_servers.memwal]")) return false; + const spec = `@mysten-incubation/memwal-mcp@${resolveMcpVersion()}`; const block = "\n[mcp_servers.memwal]\n" + 'command = "npx"\n' + - 'args = ["-y", "@mysten-incubation/memwal-mcp"]\n'; + `args = ["-y", "${spec}"]\n`; writeFileSync(CONFIG_FILE, (content.trimEnd() + "\n" + block).trimStart()); return true; } diff --git a/packages/mcp/src/auth-required.ts b/packages/mcp/src/auth-required.ts index beec42d7f..ed6bdfda2 100644 --- a/packages/mcp/src/auth-required.ts +++ b/packages/mcp/src/auth-required.ts @@ -21,8 +21,9 @@ * MCP spec 2025-06 — see ENG-1750. The two paths cover different surfaces * and coexist. */ -import { loadCreds, type MemWalCredentials } from "./auth.js"; +import { credsPath, loadCreds, type MemWalCredentials } from "./auth.js"; import { rememberInitializeClientInfo } from "./client-info.js"; +import { loginFailureNotice, loginPrompt, loginSuccessNotification } from "./messages.js"; import { log } from "./logger.js"; import { startOrReuseLoginFlow, resolveLoginTimeoutMs } from "./login.js"; import { AUTH_REQUIRED_INSTRUCTIONS } from "./instructions.js"; @@ -130,7 +131,7 @@ function buildToolDefinitions(proactive: boolean) { title: "Restore Memory Index", annotations: { readOnlyHint: false, destructiveHint: false }, description: - "Recovery tool. Re-index a namespace from Walrus blobs back into the relayer's search index \u2014 use when memwal_recall unexpectedly returns nothing even though facts were saved before (e.g. on a new machine, a fresh relayer, or after switching servers). Returns counts plus truncated status \u2014 does not return memory texts. truncated=true is known-retryable-incomplete: raising limit expands the sidecar cap only while limit < 20; after the cap saturates, truncation follows this call's missing-blob page. truncated=false is not completeness; WALM-451 will add sourceCapped. Call memwal_recall afterwards to query the rebuilt index.", + "Recovery tool. Re-index a namespace from Walrus blobs back into the relayer's search index \u2014 use when memwal_recall unexpectedly returns nothing even though facts were saved before (e.g. on a new machine, a fresh relayer, or after switching servers). Returns restored/skipped/failed/total plus truncated \u2014 does not return memory texts. truncated=true is known-retryable-incomplete: retry the same limit on a download/embed blip; raising limit expands the sidecar cap only while limit < 20; after the cap saturates, truncation follows this call's missing-blob page. truncated=false is not completeness; WALM-451 will add sourceCapped. Call memwal_recall afterwards to query the rebuilt index.", inputSchema: { type: "object", properties: { @@ -203,24 +204,6 @@ const LOGIN_INSTRUCTION = [ * already returned the URL by then, so this is the only place left to say so. */ let lastLoginFailure: string | null = null; -/** Prefix explaining that a sign-in was attempted and did not complete. */ -function loginFailureNotice(): string { - if (!lastLoginFailure) return ""; - return [ - "⚠️ A sign-in was started but never completed, so there are still no credentials.", - "", - `Reason: ${lastLoginFailure}`, - "", - "The unused key from this attempt may already be registered on your account. Remove it", - "from the dashboard if you are not using it. Sign in again and open the new link", - "straight away. A retry only helps once the MCP client is left running through the", - "wallet prompt.", - "", - "---", - "", - ].join("\n"); -} - function writeStdoutMessage(msg: RpcMessage): void { process.stdout.write(JSON.stringify(msg) + "\n"); } @@ -306,6 +289,18 @@ async function handleLoginToolCall( accountId: creds.accountId, delegateAddress: creds.delegateAddress, }); + // Symmetric with the failure branch below: the tool call returned + // the URL immediately, so nothing is left to carry the outcome + // except this notification and the banner the bridge prefixes onto + // the next tool result. + sendLogMessage( + "info", + loginSuccessNotification({ + accountId: creds.accountId, + delegateAddress: creds.delegateAddress, + credentialsPath: credsPath(), + }), + ); }, (err) => { const msg = err instanceof Error ? err.message : String(err); @@ -344,34 +339,17 @@ async function handleLoginToolCall( } log.info("memwal_login.tool.url_ready", { url }); - // The URL is included MULTIPLE times in different formats so agents - // that try to summarize the result can't strip all of them. Some MCP - // clients (Claude Code) paraphrase tool output aggressively — by - // repeating the URL in plain, code-block, and markdown-link form, at - // least one survives the agent's response template. return { isError: false, - text: [ - `## ⚠️ ACTION REQUIRED: User must click this URL to sign in`, - ``, - `**URL:** ${url}`, - ``, - `\`\`\``, + // Read from disk rather than assuming the stub only runs signed out: a + // completed callback writes credentials before the hand-off, so a + // second `memwal_login` in that window really would replace a stored + // key and must say so. + text: loginPrompt({ url, - `\`\`\``, - ``, - `[Click here to open Walrus Memory sign-in](${url})`, - ``, - `**IMPORTANT for the assistant**: do NOT summarize or omit the URL above.`, - `The user CANNOT proceed without seeing the exact URL. Surface it verbatim`, - `in your reply, then explain the steps:`, - ``, - `1. Open the URL in any browser (it may have already opened automatically)`, - `2. Click **Connect Sui Wallet** and approve the on-chain \`add_delegate_key\` transaction`, - `3. Once "Connected" appears in the browser, the assistant should retry the original request — the other memwal_* tools will then have credentials at \`~/.memwal/credentials.json\``, - ``, - `_The login link stays valid for 5 minutes. If it expires, call \`memwal_login\` again to get a fresh URL._`, - ].join("\n"), + credentialsPath: credsPath(), + signedIn: loadCreds() !== null, + }), }; } @@ -495,7 +473,7 @@ function handleAuthLine( id, result: { content: [ - { type: "text", text: `${loginFailureNotice()}${LOGIN_INSTRUCTION}` }, + { type: "text", text: `${loginFailureNotice(lastLoginFailure)}${LOGIN_INSTRUCTION}` }, ], isError: true, }, diff --git a/packages/mcp/src/auth.ts b/packages/mcp/src/auth.ts index 6dab05a48..be9d3bf73 100644 --- a/packages/mcp/src/auth.ts +++ b/packages/mcp/src/auth.ts @@ -10,15 +10,15 @@ * documentation patterns transfer cleanly. */ import { homedir } from "node:os"; -import { join, dirname } from "node:path"; +import { randomUUID } from "node:crypto"; +import { join, dirname, basename } from "node:path"; import { mkdirSync, readFileSync, writeFileSync, - chmodSync, + renameSync, unlinkSync, existsSync, - copyFileSync, } from "node:fs"; export interface MemWalCredentials { @@ -142,16 +142,133 @@ export function loadCreds(): MemWalCredentials | null { export function saveCreds(creds: MemWalCredentials): SaveCredsResult { const path = credsPath(); const replaced = backupIfReplacingAnotherAccount(path, creds.accountId); - mkdirSync(dirname(path), { recursive: true, mode: 0o700 }); - writeFileSync(path, JSON.stringify(creds, null, 2), { encoding: "utf8", mode: 0o600 }); - // writeFileSync's `mode` argument is only honored on file creation; ensure - // the permission on an existing file matches. + writeSecretFile(path, JSON.stringify(creds, null, 2)); + return { path, ...replaced }; +} + +/** + * Write a file whose bytes are only ever reachable through an inode this call + * created at `0600`. + * + * The obvious version — write to the final path, then `chmod` it — does not + * hold that property. `writeFileSync`'s `mode` follows POSIX `open()`: the + * kernel applies it when it creates the inode and ignores it for one that + * already exists. So a credentials file that anything outside this code left + * world-readable (a manual `chmod`, a restored backup, another tool) would + * receive the plaintext delegate key under the *old* mode, with a second, + * separate syscall to tighten it afterwards. Anyone reading the path in + * between gets the key. + * + * Writing a fresh file and renaming removes that window instead of shortening + * it. `rename(2)` repoints the name atomically, so a reader sees either the + * whole old file or the whole new one, never a permissive inode holding a new + * secret. `wx` (`O_EXCL`) makes a temp path that already exists — an + * interrupted earlier run, or a file planted by someone else — a hard failure + * rather than a write through a file this code did not create. + */ +function writeSecretFile(path: string, contents: string): void { + const dir = dirname(path); + mkdirSync(dir, { recursive: true, mode: 0o700 }); + // Named off the target and randomised, so concurrent saves cannot collide + // on it and no one can guess it ahead of time. Dot-prefixed to keep a + // crashed run's leftovers out of the way of directory listings. + const tmp = join(dir, `.${basename(path)}.${process.pid}.${randomUUID()}.tmp`); try { - chmodSync(path, 0o600); - } catch { - /* Windows etc. — best effort */ + writeFileSync(tmp, contents, { encoding: "utf8", mode: 0o600, flag: "wx" }); + replaceWithTemp(tmp, path, contents); + } catch (err) { + // Never leave a temp file holding the secret behind on a failed write. + try { + unlinkSync(tmp); + } catch { + /* already gone, or never created */ + } + throw err; + } +} + +/** Windows errors for "someone else holds the destination open". */ +const WIN32_LOCKED_CODES = new Set(["EPERM", "EACCES", "EBUSY"]); +const WIN32_RENAME_ATTEMPTS = 5; +const WIN32_RENAME_BACKOFF_MS = 20; + +/** Block the calling thread. `saveCreds` is synchronous all the way up. */ +function sleepSync(ms: number): void { + Atomics.wait(new Int32Array(new SharedArrayBuffer(4)), 0, 0, ms); +} + +/** + * Move `tmp` onto `path`, atomically where the platform can. + * + * POSIX `rename(2)` replaces a destination regardless of who has it open, so + * there is nothing to handle there and any error is a real one. Windows + * implements the same call as `MoveFileEx(MOVEFILE_REPLACE_EXISTING)`, which + * refuses with EPERM / EACCES / EBUSY while another handle holds the + * destination — an antivirus scan or a backup agent touching + * `credentials.json` is enough. Before this file wrote through a temp inode, + * `writeFileSync` to the final path survived that; `login.ts` turns a thrown + * `saveCreds` into an HTTP 500, so a lock that lasts a few milliseconds would + * otherwise become a failed sign-in. + * + * So on Windows: retry briefly, then write in place rather than fail. That + * fallback gives up the atomic swap, but not the property this function exists + * for — Windows does not enforce POSIX mode bits at all, so `0600` was never + * doing the work there; NTFS ACLs are, and they are inherited from the + * directory either way. On POSIX, where the mode IS the protection, there is no + * fallback and no retry. + * + * `deps` is a seam for tests. CI has no Windows runner, and the fallback is the + * one branch here that can leave a second plaintext copy of the delegate key on + * disk, so it must be exercisable off Windows. + */ +export function replaceWithTemp( + tmp: string, + path: string, + contents: string, + deps: { + platform?: string; + rename?: (from: string, to: string) => void; + sleep?: (ms: number) => void; + } = {}, +): void { + const platform = deps.platform ?? process.platform; + const rename = deps.rename ?? renameSync; + const sleep = deps.sleep ?? sleepSync; + + if (platform !== "win32") { + rename(tmp, path); + return; + } + for (let attempt = 1; ; attempt++) { + try { + rename(tmp, path); + return; + } catch (err) { + const code = (err as NodeJS.ErrnoException).code ?? ""; + if (!WIN32_LOCKED_CODES.has(code)) throw err; + if (attempt < WIN32_RENAME_ATTEMPTS) { + sleep(WIN32_RENAME_BACKOFF_MS * attempt); + continue; + } + // Still locked. Write through the existing handle's inode rather + // than failing the sign-in. + writeFileSync(path, contents, { encoding: "utf8", mode: 0o600 }); + // This returns SUCCESSFULLY, so `writeSecretFile`'s catch never + // runs and nothing else will remove `tmp` — which still holds the + // plaintext delegate key. Every locked save would otherwise leave + // another copy of it beside the credentials file, which is the + // opposite of what this whole helper is for. + // + // Best-effort: a temp that cannot be unlinked must not fail a save + // that has already landed. + try { + unlinkSync(tmp); + } catch { + /* nothing more to do; the destination write already succeeded */ + } + return; + } } - return { path, ...replaced }; } /** What `saveCreds` did, so the caller can tell the user precisely — naming @@ -223,8 +340,10 @@ function backupIfReplacingAnotherAccount( const stamp = new Date().toISOString().replace(/[:.]/g, "-"); const backup = join(dirname(path), `credentials.backup-${stamp}.json`); try { - copyFileSync(path, backup); - chmodSync(backup, 0o600); + // Through the same writer as the credentials file itself: the backup is + // a second copy of the same plaintext delegate key, and `copyFileSync` + // would create it under the process umask before any tightening. + writeSecretFile(backup, readFileSync(path, "utf8")); return { replacedAccountId: current.accountId, backedUpTo: backup }; } catch { // Never block sign-in on a failed backup — but still report the diff --git a/packages/mcp/src/bridge.ts b/packages/mcp/src/bridge.ts index bddf00895..6b9319ee3 100644 --- a/packages/mcp/src/bridge.ts +++ b/packages/mcp/src/bridge.ts @@ -15,17 +15,24 @@ * Re-auth requires an explicit `memwal-mcp login` from the user. */ import type { MemWalCredentials } from "./auth.js"; -import { clearCreds, credsPath } from "./auth.js"; +import { clearCreds, credsPath, loadCreds } from "./auth.js"; import { TOOL_DEFINITIONS } from "./auth-required.js"; import { clientInfoHeaders, lastClientInfoHeaders, rememberInitializeClientInfo, } from "./client-info.js"; +import { randomUUID } from "node:crypto"; import { ensureCompatibleRelayer, resolveConnectTimeoutMs } from "./compatibility.js"; import { PROACTIVE_INSTRUCTIONS } from "./instructions.js"; import { startOrReuseLoginFlow, resolveLoginTimeoutMs } from "./login.js"; import { log, note } from "./logger.js"; +import { + loginPrompt, + loginSuccessNotice, + loginSuccessNotification, + type LoginSuccessInfo, +} from "./messages.js"; import { MEMWAL_MCP_VERSION } from "./version.js"; /** Bridge mode runtime config — the URLs / label resolved at boot from @@ -66,6 +73,38 @@ const NAMESPACE_TOOLS = new Set([ * - the caller already supplied a non-empty `namespace` — an explicit * per-call namespace always wins over the configured default. */ +/** + * Name the relayer this process dialled in a `memwal_health` result. + * + * The relayer-side text can only report an origin its deployment published, and + * stays silent on a self-hosted or local one, where the sidecar knows nothing + * but the loopback address it dials. This side always knows the URL it + * connected to — it is exactly what `--prod` / `--relayer` / `MEMWAL_SERVER_URL` + * selected — so a client bound to the wrong network sees that here instead of + * by noticing its memories are missing. + * + * Rewrites an existing `relayer=` field rather than appending a second one: when + * both sides know the origin they describe the same session, and two + * conflicting fields would be worse than neither. + */ +export function annotateHealthResult( + result: { content?: unknown; isError?: unknown }, + relayerUrl: string, +): void { + // A failed health call has no session to describe; naming a relayer beside + // an error reads as though that relayer answered. + if (result.isError) return; + if (!Array.isArray(result.content)) return; + const block = (result.content as { type?: string; text?: string }[]).find( + (c) => c?.type === "text" && typeof c.text === "string", + ); + if (!block || typeof block.text !== "string") return; + const existing = /\brelayer=\S+/; + block.text = existing.test(block.text) + ? block.text.replace(existing, `relayer=${relayerUrl}`) + : `${block.text} relayer=${relayerUrl}`; +} + export function applyDefaultNamespace(msg: RpcMessage, namespace?: string): RpcMessage { if (!namespace) return msg; if (msg.method !== "tools/call") return msg; @@ -166,6 +205,21 @@ const SIGNED_OUT_FAILURE = { errorMessage: SIGNED_OUT_TEXT, } as const; +/** Reply for every request once the relayer has rejected the saved delegate + * key. An empty recall and a rejected key used to be indistinguishable to the + * agent — the queued call simply waited out the orphan sweeper and came back as + * "connection dropped, please retry", which is advice that cannot work. Name + * the cause and the way back in instead (GH #365 / WALM-602). */ +const UNAUTHORIZED_TEXT = + "❌ Walrus Memory rejected the saved credentials (HTTP 401). The delegate key may have been revoked or is no longer registered on this account. Call `memwal_login` to sign in again — saved credentials were NOT modified."; + +/** `failRequest` options for every credentials-rejected refusal, so one refused + * at handshake time and one refused on arrival afterwards read identically. */ +const UNAUTHORIZED_FAILURE = { + toolText: UNAUTHORIZED_TEXT, + errorMessage: UNAUTHORIZED_TEXT, +} as const; + /** The `tools/list` we serve LOCALLY at cold start: the memory tools (from the * same source as auth-required mode) plus the locally-handled login/logout * tools. We strip any locally-served name from the imported list first — @@ -229,6 +283,108 @@ function resolveCallTimeoutMs(): number { return n; } +/** How long to wait before retrying a handshake the relayer refused with a 429 + * that carried NO `Retry-After`. That is the relayer's concurrent-session cap + * (`ip_active_cap`), which deliberately sends no header because it clears when + * some other session closes, not on a timer — so the ordinary sub-second + * geometric retry is pure noise against it. + * + * Override via `MEMWAL_MCP_THROTTLE_FLOOR_MS` (mostly for tests). */ +const DEFAULT_THROTTLE_FLOOR_MS = 5_000; + +/** A relayer-supplied interval is a remote-controlled sleep, so cap it: a + * misconfigured (or hostile) `Retry-After: 86400` must not park the bridge for + * a day. Past this we retry anyway and take another 429 if we were wrong. */ +const MAX_THROTTLE_WAIT_MS = 60_000; + +function resolveThrottleFloorMs(): number { + const raw = process.env.MEMWAL_MCP_THROTTLE_FLOOR_MS; + if (!raw) return DEFAULT_THROTTLE_FLOOR_MS; + const n = Number(raw); + if (!Number.isFinite(n) || n < 0) return DEFAULT_THROTTLE_FLOOR_MS; + return Math.min(n, MAX_THROTTLE_WAIT_MS); +} + +/** Parse a `Retry-After` value into ms. The header is legally either + * delta-seconds or an HTTP-date (this relayer only ever emits the former, but + * a proxy in the path may rewrite it). Returns null for absent / unparseable + * values so the caller falls back to the floor — never NaN, which would poison + * the backoff arithmetic and break the retry loop outright. */ +function parseRetryAfterMs(raw: string | null): number | null { + if (!raw) return null; + const trimmed = raw.trim(); + if (trimmed === "") return null; + if (/^\d+$/.test(trimmed)) { + const seconds = Number(trimmed); + // Non-positive is not advice. `Retry-After: 0` is a real thing to + // receive (some intermediaries emit it for "unknown"), and taking it + // literally puts us back on the ~500ms geometric backoff that + // WALM-386 exists to stop — while still reporting `serverAdvised`, + // which would also suppress the one hint the user can act on. Treat + // it as no usable header and fall back to the floor. + return Number.isFinite(seconds) && seconds > 0 ? seconds * 1000 : null; + } + const at = Date.parse(trimmed); + if (!Number.isFinite(at)) return null; + // Same for an HTTP-date already in the past — which a correct server can + // produce simply by being a second behind the client's clock. + const waitMs = at - Date.now(); + return waitMs > 0 ? waitMs : null; +} + +/** The relayer refused the handshake with HTTP 429. Carried as a typed error so + * the retry loops can honour the throttle interval instead of re-deriving it + * from a message string — the whole point of WALM-386. `retryAfterMs` is + * already resolved (header, else floor) and clamped, so callers just sleep it. */ +class RelayerThrottledError extends Error { + readonly status = 429; + /** How long to wait before the next attempt, ms. Always a finite number. */ + readonly retryAfterMs: number; + /** True when the relayer actually sent a usable `Retry-After`. False means + * we applied the floor — the `ip_active_cap` shape, which has no ETA. */ + readonly serverAdvised: boolean; + + constructor(message: string, retryAfterHeader: string | null) { + super(message); + this.name = "RelayerThrottledError"; + const advised = parseRetryAfterMs(retryAfterHeader); + this.serverAdvised = advised !== null; + this.retryAfterMs = Math.min( + MAX_THROTTLE_WAIT_MS, + Math.max(0, advised ?? resolveThrottleFloorMs()), + ); + } +} + +/** Deadline for a request that is still buffered while the handshake has been + * failing for at least this long — i.e. one we can prove never left this + * process. + * + * It is much shorter than `callTimeoutMs` because the two cases carry + * different risk, not because the wait is less important. A request that was + * SENT might have been executed, so failing it early invites the agent to + * retry a `remember` that already landed. A request that was never sent + * cannot have executed: failing it is provably a no-op, and the agent's retry + * costs one round trip. + * + * 90s is well past a relayer cold start and past six reconnect attempts at the + * capped 15s backoff, so it does not fire on a slow-but-recovering relayer — + * and it does not apply at all while the handshake is healthy (a request + * buffered behind an in-progress flush keeps the full deadline). What it ends + * is the case from WALM-618: no working connection, nothing sent, and four + * minutes of silence before the user is told anything. */ +const DEFAULT_STALLED_HANDSHAKE_MS = 90_000; + +/** Same override shape as the call timeout, mostly for tests. Never longer + * than the call timeout itself: this deadline exists to fire sooner. */ +function resolveStalledHandshakeMs(callTimeoutMs: number): number { + const raw = process.env.MEMWAL_MCP_STALLED_HANDSHAKE_MS; + const n = raw ? Number(raw) : DEFAULT_STALLED_HANDSHAKE_MS; + const resolved = + Number.isFinite(n) && n >= MIN_CALL_TIMEOUT_MS ? n : DEFAULT_STALLED_HANDSHAKE_MS; + return Math.min(resolved, callTimeoutMs); +} + interface RpcMessage { jsonrpc: "2.0"; id?: number | string | null; @@ -242,6 +398,26 @@ interface RpcMessage { interface InFlightEntry { msg: RpcMessage; startedAt: number; + /** Set once a POST has been issued for this request. + * + * This is what separates "cannot have executed" from "might have + * executed", and it has to live on the entry: `pendingForward` only holds + * requests buffered before the first successful connect, so in a + * mid-session outage — the ordinary case — a request that never left the + * process was indistinguishable from one already sent. */ + sent?: boolean; +} + +/** The relayer rejected the saved delegate key (HTTP 401 on the handshake). + * Distinct from every other connect failure because retrying cannot fix it: + * the caller must re-authenticate. Carrying it as a type keeps the background + * connect loop from backing off forever on a key that will never be accepted, + * which left tool calls parked until the orphan sweeper's deadline (WALM-602). */ +class RelayerUnauthorizedError extends Error { + constructor(message: string) { + super(message); + this.name = "RelayerUnauthorizedError"; + } } interface SseHandshakeResult { @@ -326,6 +502,14 @@ async function openSseStream( throw err; } + // Every non-OK exit below DRAINS the body and deliberately does NOT abort + // `controller`. Aborting a handshake response we have already read is what + // produced the Windows libuv assertion in WALM-386 + // (`!(handle->flags & UV_HANDLE_CLOSING)`, src/win/async.c:76); 45b0ad87 + // removed those aborts on purpose ("the stdio bridge drains handshake error + // bodies instead of aborting the socket", CHANGELOG 0.0.11). Draining to + // completion lets undici return the socket to its pool normally. Do not + // re-add `controller.abort()` here. if (resp.status === 401) { clearConnectTimer(); if (resp.body) { @@ -340,7 +524,7 @@ async function openSseStream( // Auto-wiping the seed turns any one of those into a permanent // outage that forces re-login. Force-fail loud instead; the user // runs `memwal-mcp login` if they want to actually rotate. - throw new Error( + throw new RelayerUnauthorizedError( "Walrus Memory relayer rejected credentials (HTTP 401). " + "Delegate key may have been revoked, the relayer may be " + "rate-limiting, or a proxy may be interposed. Saved " + @@ -351,15 +535,20 @@ async function openSseStream( if (resp.status === 429) { clearConnectTimer(); const retryAfter = resp.headers.get("retry-after"); - const body = resp.body ? await resp.text() : ""; - throw new Error( + const body = resp.body ? await resp.text().catch(() => "") : ""; + // Throw a TYPED error: the interval has to survive as a number for the + // retry loops to honour it. Stringifying it into the message (what this + // used to do) left both loops guessing, so they retried a throttled + // handshake after 500ms — WALM-386. + throw new RelayerThrottledError( `Walrus Memory relayer SSE handshake rate-limited (HTTP 429` + - `${retryAfter ? `, retry after ${retryAfter}s` : ""}). ${body.slice(0, 200)}`.trim() + `${retryAfter ? `, retry after ${retryAfter}s` : ""}). ${body.slice(0, 200)}`.trim(), + retryAfter, ); } if (!resp.ok || !resp.body) { clearConnectTimer(); - const body = resp.body ? await resp.text() : ""; + const body = resp.body ? await resp.text().catch(() => "") : ""; throw new Error( `Walrus Memory relayer SSE handshake failed: HTTP ${resp.status} ${body.slice(0, 200)}` ); @@ -567,6 +756,14 @@ function readStdinLines(onLine: (line: string) => void): Promise { }); process.stdin.on("end", () => resolve()); process.stdin.on("close", () => resolve()); + // Attaching a `data` listener only starts the flow on a stream that was + // never explicitly paused. The auth-required stub hands off by calling + // `process.stdin.pause()`, and a stream paused that way stays paused no + // matter how many listeners attach — so after an in-session + // `memwal_login` the bridge read NOTHING beyond the requests replayed + // from `pendingLines`, and every later call hung unanswered. Harmless + // on the cold path, where stdin is already flowing. + process.stdin.resume(); }); } @@ -596,6 +793,22 @@ async function handleLocalLogin( accountId: creds.accountId, delegateAddress: creds.delegateAddress, }); + // The tool call returned the URL long ago, so — exactly as on the + // failure path below — this notification and the banner on the next + // tool result are the only ways left to say the sign-in landed. + writeStdoutMessage({ + jsonrpc: "2.0", + method: "notifications/message", + params: { + level: "info", + logger: "memwal-mcp", + data: loginSuccessNotification({ + accountId: creds.accountId, + delegateAddress: creds.delegateAddress, + credentialsPath: credsPath(), + }), + }, + }); }, (err) => { const msg = err instanceof Error ? err.message : String(err); @@ -631,27 +844,15 @@ async function handleLocalLogin( return { isError: false, - text: [ - `## ⚠️ ACTION REQUIRED: User must click this URL to sign in`, - ``, - `**URL:** ${url}`, - ``, - `\`\`\``, + // Read from disk, never assumed from the mode. The bridge usually runs + // with credentials, but `memwal_logout` in this same session deletes + // them and login is intercepted before the signed-out guard — claiming + // "already signed in" there tells the user logout did not take. + text: loginPrompt({ url, - `\`\`\``, - ``, - `[Click here to open Walrus Memory sign-in](${url})`, - ``, - `**IMPORTANT for the assistant**: do NOT summarize or omit the URL above.`, - `Surface it verbatim so the user can click it.`, - ``, - `Steps:`, - `1. Open the URL in any browser`, - `2. Click **Connect Sui Wallet** and approve the on-chain \`add_delegate_key\` transaction`, - `3. Once "Connected" appears, retry the previous request — credentials at \`~/.memwal/credentials.json\` get overwritten with the new wallet's delegate key`, - ``, - `_The login link stays valid for 5 minutes._`, - ].join("\n"), + credentialsPath: credsPath(), + signedIn: loadCreds() !== null, + }), }; } @@ -703,6 +904,52 @@ function handleLocalLogout(): { text: string; isError: boolean } { } } +/** + * A completed sign-in waiting to be reported to the client. + * + * Set when credentials are adopted — either mid-session via `adoptCredentials` + * or on the cold hand-off from the auth-required stub, which is why this is + * module state with a setter rather than a local inside `runBridge`: on the + * cold path the sign-in happens before the bridge exists. + * + * Consumed by {@link takePendingLoginSuccess}, so it can only ever be reported + * once. + */ +let pendingLoginSuccess: LoginSuccessInfo | null = null; + +/** Record a completed sign-in for the next tool result to carry. */ +export function notePendingLoginSuccess(info: LoginSuccessInfo): void { + pendingLoginSuccess = info; +} + +/** Read the pending sign-in AND clear it — the read is the consumption. */ +function takePendingLoginSuccess(): LoginSuccessInfo | null { + const pending = pendingLoginSuccess; + pendingLoginSuccess = null; + return pending; +} + +/** + * Prefix the sign-in banner onto a tool result, if one is pending. + * + * Only ever called for a `tools/call` reply. A `tools/list` or `ping` response + * would consume the banner into somewhere the user never reads it, so the + * caller checks which request is being answered first. + */ +function applyPendingLoginSuccess(value: RpcMessage): void { + const result = value.result as { content?: unknown } | undefined; + if (!result || typeof result !== "object" || !Array.isArray(result.content)) return; + + const first = result.content[0] as { type?: string; text?: string } | undefined; + if (!first || first.type !== "text" || typeof first.text !== "string") return; + + const pending = takePendingLoginSuccess(); + if (!pending) return; + + first.text = `${loginSuccessNotice(pending)}${first.text}`; + log.info("bridge.login_success_notice_attached", { accountId: pending.accountId }); +} + /** * Open the SSE bridge and forward stdio ↔ relayer until stdin closes. * @@ -768,6 +1015,10 @@ export async function runBridge( * checks this alongside `stdinClosed`; `adoptCredentials` clears it when a * new login lands. */ let loggedOut = false; + /** Set when the relayer 401s the handshake, cleared on the next successful + * connect. While set, requests fail fast with `UNAUTHORIZED_FAILURE` rather + * than parking in `pendingForward` behind a connect that cannot succeed. */ + let credentialsRejected = false; /** Resolves once a post-logout `memwal_login` has published a fresh session * (or stdin closed). The server pump parks on this instead of exiting, so * signing back in resumes streaming without the user restarting their MCP @@ -777,6 +1028,55 @@ export async function runBridge( let reconnectAttempt = 0; let reconnectPromise: Promise | null = null; let firstConnectDone = false; + /** Wall-clock instant before which the relayer told us (HTTP 429) not to + * open another session. Both retry loops floor their backoff at this, so a + * throttle survives across the separate `reconnect()` calls that would + * otherwise each start from a fresh 500ms. 0 = not throttled. */ + let throttledUntilMs = 0; + /** One "we are being throttled" note per throttle episode. A sustained cap + * would otherwise print a line per retry cycle for as long as it lasts. */ + let throttleNoticed = false; + /** Why the last handshake attempt failed, and when the run of failures + * started. Kept so a request that ages out while buffered can say what + * it was actually waiting on instead of a generic "unavailable" — the + * user-visible half of WALM-618, where the bridge retried in silence and + * a `remember` looked like it was just slow. Cleared on every success. */ + let lastHandshakeError: string | null = null; + let handshakeFailingSince: number | null = null; + /** Identifies one "I need a session" episode, and stays the same across + * every retry inside it. Sent on each handshake as `x-memwal-connect-id`. + * + * Without it the relayer sees N unrelated sub-second requests and cannot + * tell they were one user waiting: its per-request id is minted fresh each + * time, so a four-minute wait leaves no four-minute anything in its logs, + * only a scatter of fast 401s and 429s. With it, one grep returns the + * whole episode and the span between first and last line IS the wait. + * Cleared on success, so the next outage starts a new episode. */ + let connectEpisodeId: string | null = null; + const connectHeaders = (): Record => { + connectEpisodeId ??= randomUUID(); + return { + // `x-memwal-client` is only known after `initialize`, and a first + // connect happens before stdin is even wired — so on the attempt + // that matters most the relayer has no idea who is calling. The + // bridge's own version it always knows, and "which build is + // looping" is the actionable half anyway. + "x-memwal-bridge-version": MEMWAL_MCP_VERSION, + ...extraHeaders, + "x-memwal-connect-id": connectEpisodeId, + }; + }; + const endConnectEpisode = (): void => { + connectEpisodeId = null; + }; + const noteHandshakeFailure = (reason: string): void => { + lastHandshakeError = reason; + handshakeFailingSince ??= Date.now(); + }; + const clearHandshakeFailure = (): void => { + lastHandshakeError = null; + handshakeFailingSince = null; + }; /** Bumped when the live SSE session is aborted or replaced so queued * POSTs captured against a stale URL are skipped (reconnect replays). */ let sessionEpoch = 0; @@ -797,7 +1097,32 @@ export async function runBridge( postCreds: MemWalCredentials, ): Promise { if (epoch !== sessionEpoch) return Promise.resolve(0); - return postMessage(postUrl, msg, postCreds, extraHeaders); + // Marked before the await, not after. Once the POST is issued we can + // no longer prove the call did not run, so it must keep the full call + // timeout even if the socket then fails — failing it early is what + // invites a duplicate `remember`. + if (msg.id !== undefined && msg.id !== null) { + const tracked = inFlight.get(msg.id); + if (tracked) tracked.sent = true; + } + return postMessage(postUrl, msg, postCreds, extraHeaders).then((status) => { + // 404 is the relayer saying that session does not exist, so the + // message was discarded rather than routed: it provably did not + // run, and the request goes back to being never-sent. + // + // This is not a corner case. `sse` is not cleared when the server + // pump hits EOF — it keeps pointing at the dead session until a + // reconnect succeeds — so a call arriving during a mid-session + // outage takes the POST path, posts to the stale URL, and gets + // exactly this. Without the reset it would be marked sent and + // wait out the full call timeout, which is the WALM-618 symptom + // the stalled deadline exists to remove. + if (status === 404 && msg.id !== undefined && msg.id !== null) { + const tracked = inFlight.get(msg.id); + if (tracked) tracked.sent = false; + } + return status; + }); } let credentialGeneration = 0; let activeCredentialGeneration = 0; @@ -908,6 +1233,7 @@ export async function runBridge( // would otherwise keep pushing the deadline out. const inFlight = new Map(); const callTimeoutMs = resolveCallTimeoutMs(); + const stalledHandshakeMs = resolveStalledHandshakeMs(callTimeoutMs); /** IDs of `tools/list` requests we've forwarded to the relayer. When * the response comes back through the SSE pump, we splice in the @@ -915,6 +1241,36 @@ export async function runBridge( * client surfaces them in its tool palette. */ const pendingListIds = new Set(); + /** IDs of forwarded `memwal_health` calls, each against the relayer URL the + * call went out on. Captured at send time rather than read at reply time so + * a reconnect that swapped credentials mid-flight cannot label the answer + * with a relayer it did not come from. */ + const pendingHealthIds = new Map(); + + /** Record a 429 and tell the user ONCE that this is a rate limit rather + * than a broken config — the distinction the MCP host cannot make for + * itself, and the reason a throttled bridge reads as "memwal is down". */ + function noteThrottled(err: RelayerThrottledError): void { + throttledUntilMs = Math.max(throttledUntilMs, Date.now() + err.retryAfterMs); + log.warn("bridge.relayer_throttled", { + retryAfterMs: err.retryAfterMs, + serverAdvised: err.serverAdvised, + err: err.message, + }); + if (throttleNoticed) return; + throttleNoticed = true; + const seconds = Math.max(1, Math.round(err.retryAfterMs / 1000)); + note( + `Relayer is rate-limiting new MCP sessions (HTTP 429). This is a ` + + `throttle, not a bad config or bad credentials — retrying in ` + + `${seconds}s. Memory tools start working once a session opens.` + + (err.serverAdvised + ? "" + : " The cap counts concurrent sessions, so closing another " + + "MCP client using this account clears it sooner."), + ); + } + /** Reopen the SSE stream and replay outstanding `inFlight` requests against * the fresh session. All callers await the SAME reconnect via * `reconnectPromise` — returning immediately while one is active would let @@ -935,9 +1291,15 @@ export async function runBridge( } catch { /* already dead */ } + // `immediate` (a login credential swap) still bypasses everything, + // throttle included: that path trades a possible extra 429 for a + // re-login that doesn't stall behind a multi-second floor. const backoff = immediate ? 0 - : Math.min(15_000, 500 * Math.pow(2, reconnectAttempt)); + : Math.max( + Math.min(15_000, 500 * Math.pow(2, reconnectAttempt)), + throttledUntilMs - Date.now(), + ); reconnectAttempt += 1; log.warn("bridge.reconnecting", { reason, @@ -960,9 +1322,13 @@ export async function runBridge( }); }); } + // Hoisted so the catch below can ask whether the credentials moved + // since the handshake that threw was opened — a 401 for a key a + // login has already replaced says nothing about the new one. + let openingGeneration = credentialGeneration; try { while (!stdinClosed && !loggedOut) { - const openingGeneration = credentialGeneration; + openingGeneration = credentialGeneration; const openingCreds = creds; // Signed out between the guard above and here: the key is // gone, so there is nothing to authorize a new session @@ -971,7 +1337,7 @@ export async function runBridge( const candidate = await openSseStream( openingCreds.relayerUrl, openingCreds, - extraHeaders, + connectHeaders(), ); // Logout can also land mid-handshake. Same reasoning as the @@ -1001,6 +1367,22 @@ export async function runBridge( firstConnectDone = true; activeCredentialGeneration = openingGeneration; reconnectAttempt = 0; + throttledUntilMs = 0; + throttleNoticed = false; + clearHandshakeFailure(); + endConnectEpisode(); + // An accepted handshake retires any earlier rejection — + // `memwal_login` re-registers a key and lands here, not on + // the background connect's publish path, so clearing only + // there would leave every later request refused (WALM-602). + credentialsRejected = false; + // Usually a no-op: the pump is past `firstConnect` by the + // time anything reconnects. It is NOT a no-op when this is + // the first session to exist at all — a login after the + // saved key was rejected — and without it the pump would + // stay parked until the background connect's backoff + // happened to expire, with nothing draining this stream. + signalFirstConnect(); log.info("bridge.reconnected", { relayer: openingCreds.relayerUrl, replayCount: inFlight.size, @@ -1076,9 +1458,35 @@ export async function runBridge( break; } } catch (err) { - log.error("bridge.reconnect_failed", { - err: err instanceof Error ? err.message : String(err), - }); + const reason = err instanceof Error ? err.message : String(err); + // A 429 must outlive this call: reconnect() gives up after one + // failure, so without recording the deadline the next caller + // would compute a fresh sub-second backoff and hammer the cap. + if (err instanceof RelayerThrottledError) noteThrottled(err); + noteHandshakeFailure(reason); + log.error("bridge.reconnect_failed", { err: reason }); + // A key revoked mid-session lands here rather than on the + // background connect, and retrying cannot fix it either. Answer + // the replay set now instead of letting the orphan sweeper hand + // back "connection dropped, please retry" four minutes later — + // the same WALM-602 symptom, one path over. + // + // This does not strand the transient case: the server pump is + // still looping on the dead stream, so it keeps driving + // `reconnect()` on its own growing backoff, and the publish + // above clears the flag the moment a handshake is accepted. + // + // It also takes precedence over the stalled-handshake deadline + // added here: a 401 is terminal until the user logs in again, + // so there is nothing to gain by waiting out even the short + // deadline for it. + if ( + err instanceof RelayerUnauthorizedError && + openingGeneration === credentialGeneration + ) { + credentialsRejected = true; + failInFlightRequests("credentials rejected", UNAUTHORIZED_FAILURE); + } // Try again on the next stdin message rather than spinning. } })(); @@ -1119,6 +1527,7 @@ export async function runBridge( const purge = (msg: RpcMessage): void => { if (msg.id == null) return; // notification — nothing to reply to pendingListIds.delete(msg.id); + pendingHealthIds.delete(msg.id); if (msg.method === "initialize") { return; } @@ -1174,6 +1583,15 @@ export async function runBridge( releaseLogoutPark?.(); releaseLogoutPark = null; logoutPark = null; + + // Queue the confirmation only once the session is actually live, so + // the banner cannot claim an authenticated connection before there is + // one. It rides out on the next `tools/call` result. + notePendingLoginSuccess({ + accountId: creds.accountId, + delegateAddress: creds.delegateAddress, + credentialsPath: credsPath(), + }); } /** @@ -1291,6 +1709,16 @@ export async function runBridge( inFlight.delete(value.id); continue; } + // Which request this reply answers. Captured BEFORE the + // `inFlight.delete` below drops the entry, so the sign-in + // banner can tell a `tools/call` result from a `tools/list` + // or a `ping` and avoid being consumed by a response the + // user never reads. + const answeredMethod = + value && value.id !== undefined && value.id !== null + ? inFlight.get(value.id)?.msg.method + : undefined; + // Clear in-flight tracking once the response lands. if ( value && @@ -1324,6 +1752,31 @@ export async function runBridge( result.tools = [...upstream, ...LOCAL_TOOL_DEFINITIONS]; } } + if ( + value && + value.id !== undefined && + value.id !== null && + pendingHealthIds.has(value.id) && + value.result && + typeof value.result === "object" + ) { + const dialled = pendingHealthIds.get(value.id); + pendingHealthIds.delete(value.id); + if (dialled !== undefined) { + annotateHealthResult( + value.result as { content?: unknown; isError?: unknown }, + dialled, + ); + } + } + // Health annotation runs BEFORE the sign-in banner. Both + // rewrite the same first text block, and annotateHealthResult + // replaces the first `relayer=` it finds — so a banner + // prefixed first would be the thing it rewrote if that text + // ever names a relayer. + if (answeredMethod === "tools/call") { + applyPendingLoginSuccess(value); + } writeStdoutMessage(value); } } catch (err) { @@ -1374,11 +1827,12 @@ export async function runBridge( id: msg.id, result: buildLocalInitializeResult(msg.params), }); - // Signed out: the local reply is the whole answer. We will - // not forward upstream, so do not arm a suppression that no - // reply can ever consume — a leaked arm would swallow the - // real reply if the client later reuses this id. - if (loggedOut) return; + // Signed out, or the key was rejected: the local reply is the + // whole answer. Both refuse further down instead of + // forwarding, so do not arm a suppression that no reply can + // ever consume — a leaked arm would swallow the real reply + // if the client later reuses this id. + if (loggedOut || credentialsRejected) return; // Expect exactly one upstream reply to drop for this forward. expectSuppressedReply(msg.id); // Fall through: forward/buffer the initialize upstream too. @@ -1463,6 +1917,17 @@ export async function runBridge( return; } + // Credentials rejected: same reasoning as `loggedOut` above. + // `memwal_login` returned locally already, so refusing here + // still leaves the user a way back in. Falling through would + // park the request in `pendingForward` behind a connect loop + // that keeps 401ing, and the client would learn nothing until + // the orphan sweeper's deadline — the WALM-602 symptom. + if (credentialsRejected) { + failRequest(msg, "credentials rejected", UNAUTHORIZED_FAILURE); + return; + } + // Fill in the configured default namespace for memory tool // calls that didn't pass one. Mutates msg in place so the // forwarded — and any replayed-on-reconnect — copy carries it. @@ -1474,6 +1939,16 @@ export async function runBridge( pendingListIds.add(msg.id); } + // Same idea for `memwal_health`: record the relayer this + // session is bound to so the pump can name it on the reply. + if ( + msg.method === "tools/call" && + msg.id != null && + (msg.params as { name?: string } | undefined)?.name === "memwal_health" + ) { + pendingHealthIds.set(msg.id, creds?.relayerUrl ?? config.relayerUrl); + } + // Track requests (have both method and id) so we can replay // them on reconnect. Notifications and responses are not // tracked. @@ -1682,9 +2157,12 @@ export async function runBridge( } } - function failPendingForward(reason: string): void { + function failPendingForward( + reason: string, + opts: { toolText?: string; errorMessage?: string } = {}, + ): void { const queued = pendingForward.splice(0, pendingForward.length); - for (const msg of queued) failRequest(msg, reason); + for (const msg of queued) failRequest(msg, reason, opts); } /** Close out requests that reached `inFlight` but were never delivered a @@ -1692,8 +2170,134 @@ export async function runBridge( * closes mid-flush: items already shifted out of `pendingForward` and posted * to a torn-down session would otherwise hang, since no upstream reply is * coming. Idempotent w.r.t. ids already closed out (delete-then-skip). */ - function failInFlightRequests(reason: string): void { - for (const entry of Array.from(inFlight.values())) failRequest(entry.msg, reason); + function failInFlightRequests( + reason: string, + opts: { toolText?: string; errorMessage?: string } = {}, + ): void { + for (const entry of Array.from(inFlight.values())) failRequest(entry.msg, reason, opts); + } + + /** How long the current run of handshake failures has lasted, or `null` + * when the last attempt succeeded. */ + function handshakeStalledForMs(now: number): number | null { + return handshakeFailingSince === null ? null : now - handshakeFailingSince; + } + + /** How a request that just hit its deadline should be explained. + * + * Three cases, where the old wording only described one. A request for + * which no POST was ever issued never left this process: no session ever + * carried it. Telling the user the connection "dropped before the result + * came back" points them at the relayer, or at a half-written memory, when + * the truth is that nothing was attempted (WALM-618 — the bridge retried + * in silence, so a `remember` looked like it was merely slow for minutes). + * And a buffered request is only evidence of a *failing* connection when + * one is actually failing: post-connect, `handleClientLine` also buffers + * behind an in-progress flush, on a perfectly healthy session. + * + * Pure: the caller is responsible for dropping a `neverSent` message from + * the buffer, which it must, or a later flush would run the call we just + * said never ran. */ + /** Tools whose call, once POSTed, may have written to Walrus. + * + * The relayer answers these with HTTP 202 and finishes the work in a + * durable queue, so a client-side deadline cancels nothing: the write can + * still land minutes after we have given up waiting for the reply. And + * `/api/remember/bulk` carries no idempotency key — unlike the single + * path — so a blind retry mints a second paid blob that `recall` will then + * hide behind the first. Telling the user to "please retry" here is how a + * lost reply turns into duplicate paid storage. */ + const MUTATING_TOOLS = new Set([ + "memwal_remember", + "memwal_remember_bulk", + "memwal_analyze", + ]); + + /** Name of the tool a tracked request was calling, when it was one. */ + function toolNameOf(msg: RpcMessage): string | null { + if (msg.method !== "tools/call") return null; + const params = msg.params as { name?: unknown } | undefined; + return typeof params?.name === "string" ? params.name : null; + } + + function expiredRequestReport( + neverSent: boolean, + now: number, + tool: string | null, + ): { + reason: string; + opts: { toolText: string; errorMessage: string }; + } { + if (!neverSent) { + // The request reached the relayer. What is missing is the reply, + // and for a write that distinction is the whole message: the work + // may have completed, may still be running, and cannot be assumed + // undone. "Please retry" is only safe advice for a read. + if (tool !== null && MUTATING_TOOLS.has(tool)) { + return { + reason: "no response to a sent write", + opts: { + toolText: + `⚠️ Walrus Memory accepted this ${tool} call but did not return a ` + + "result in time. The write was sent, so it may have completed or may " + + "still be finishing in the background — a timeout here does not cancel " + + "it and does not mean nothing was stored. Do NOT simply repeat the " + + "call: run `memwal_recall` for this content first, and only re-save " + + "what is genuinely missing. Repeating a bulk save that already " + + "landed stores a second paid copy.", + errorMessage: + `Walrus Memory ${tool} was sent but its reply never arrived. The write ` + + "may have completed; verify with recall before retrying.", + }, + }; + } + return { + reason: "no response", + opts: { + toolText: + "❌ Walrus Memory did not answer this call. The request reached the " + + "relayer but the reply never came back. This call only reads, so it is " + + "safe to retry.", + errorMessage: + "Walrus Memory call was orphaned by a reconnect and never " + + "received a response. Safe to retry: this call only reads.", + }, + }; + } + + const stalledForMs = handshakeStalledForMs(now); + if (stalledForMs === null) { + // Buffered on a live session (a flush was draining) and still + // unsent at the deadline. Nothing ran, but nothing is failing + // either — do not invent an outage. + return { + reason: "never left the queue", + opts: { + toolText: + "❌ Walrus Memory never sent this call — it was still queued when " + + "the call timed out, so nothing was stored. Please retry.", + errorMessage: + "Walrus Memory call was still queued when it timed out and was " + + "never sent. Please retry.", + }, + }; + } + + const waited = `for ${Math.round(stalledForMs / 1000)}s`; + const detail = lastHandshakeError ? ` Last handshake error: ${lastHandshakeError}` : ""; + return { + reason: "never reached the relayer", + opts: { + toolText: + `❌ Walrus Memory could not reach the relayer — the MCP connection has ` + + `been failing ${waited}, so this call never ran and nothing was stored.` + + `${detail} Check the relayer, or run \`memwal-mcp login\` if the delegate ` + + `key was revoked, then retry.`, + errorMessage: + `Walrus Memory call never reached the relayer: the MCP connection has ` + + `been failing ${waited}.${detail}`, + }, + }; } /** Close out requests whose deadline has passed. Without this a reply lost @@ -1705,22 +2309,58 @@ export async function runBridge( ); const orphanSweeper = setInterval(() => { const now = Date.now(); + const handshakeStalledMs = handshakeStalledForMs(now); for (const [id, entry] of Array.from(inFlight.entries())) { const elapsedMs = now - entry.startedAt; - if (elapsedMs <= callTimeoutMs) continue; + // Never sent = no POST was ever issued for it. Read from the entry + // rather than from `pendingForward` membership, which only ever + // covered the cold-start window. + const neverSent = entry.sent !== true; + // A call we can prove never left this process, while no working + // connection has existed for `stalledHandshakeMs`, does not need + // the full `callTimeoutMs`: it cannot have executed, so answering + // it early is a no-op the agent can safely retry. Anything that + // was actually sent — or that is queued on a healthy session — + // keeps the full deadline, because there a premature failure + // invites a duplicate write. + const handshakeIsStalled = + handshakeStalledMs !== null && handshakeStalledMs > stalledHandshakeMs; + const deadlineMs = + neverSent && handshakeIsStalled ? stalledHandshakeMs : callTimeoutMs; + if (elapsedMs <= deadlineMs) continue; + // Built only for what actually expired: this walks `pendingForward` + // and interpolates two user-facing strings, and the branch it + // serves fires roughly never. + const { reason, opts } = expiredRequestReport( + neverSent, + now, + toolNameOf(entry.msg), + ); + // Drop it from the buffer before answering: a later successful + // connect would otherwise flush and actually run the call we are + // about to report as never having run. + // + // `initialize` is the exception, as everywhere else here: it was + // answered locally and is only buffered so the relayer session can + // still negotiate capabilities, and `failRequest` writes it no + // reply. Removing it would silently cost that negotiation on the + // first connect after a long outage. + if (neverSent && entry.msg.method !== "initialize") { + // Only buffered requests are in there at all now, so the miss + // is ordinary — `splice(-1, 1)` would drop the last entry. + const queuedAt = pendingForward.indexOf(entry.msg); + if (queuedAt >= 0) pendingForward.splice(queuedAt, 1); + } log.warn("bridge.call_orphaned", { id, method: entry.msg.method ?? null, elapsedMs, + deadlineMs, + reason, + handshakeStalledMs, + lastHandshakeError, }); - failRequest(entry.msg, "no response", { - toolText: - "❌ Walrus Memory did not answer this call. The connection to " + - "the relayer dropped before the result came back. Please retry.", - errorMessage: - "Walrus Memory call was orphaned by a reconnect and never " + - "received a response. Please retry.", - }); + failRequest(entry.msg, reason, opts); } }, sweepIntervalMs); // unref so the sweeper never holds the event loop open during shutdown. @@ -1735,8 +2375,11 @@ export async function runBridge( // NOT fail buffered requests between attempts: a request that the next // attempt would serve must not get a spurious "unavailable" error (that // would also drop the auth-required hot-handoff request). Buffered tool - // calls stay queued and are flushed on the first SUCCESS; if they never - // connect, the client's own per-tool timeout fires (graceful) — and on + // calls stay queued and are flushed on the first SUCCESS. They are no + // longer left to the client's own per-tool timeout, though: once no + // connection has existed for `stalledHandshakeMs` the orphan sweeper + // answers them (see `DEFAULT_STALLED_HANDSHAKE_MS`), because a call that + // was never sent cannot have executed and silence helps nobody. On // shutdown `failPendingForward` closes out anything still open. `initialize` // is answered locally, so it never blocks and is only forwarded, not failed. // First connect stays on `openSseStream` + `flushPendingForward` so a @@ -1760,7 +2403,7 @@ export async function runBridge( } const openingGeneration = credentialGeneration; try { - const candidate = await openSseStream(creds.relayerUrl, creds, extraHeaders); + const candidate = await openSseStream(creds.relayerUrl, creds, connectHeaders()); if (stdinClosed) { candidate.abort(); break; @@ -1780,6 +2423,14 @@ export async function runBridge( sessionEpoch += 1; sse = candidate; firstConnectDone = true; + throttledUntilMs = 0; + throttleNoticed = false; + clearHandshakeFailure(); + endConnectEpisode(); + // A key that was rejected earlier is evidently accepted now + // (re-registered, or the 401 was a transient WAF/rate-limit + // false positive), so stop failing requests fast. + credentialsRejected = false; note(`Connected. Bridging stdio MCP ↔ ${creds.relayerUrl}`); log.info("bridge.connected", { relayer: creds.relayerUrl }); signalFirstConnect(); @@ -1787,10 +2438,54 @@ export async function runBridge( return; } catch (err) { const reason = err instanceof Error ? err.message : String(err); + noteHandshakeFailure(reason); attempt += 1; + if (err instanceof RelayerThrottledError) noteThrottled(err); log.error("bridge.initial_connect_failed", { err: reason, attempt }); + // A rejected key will not start working on the next attempt, so + // answer everything queued instead of leaving it to the orphan + // sweeper. Keep looping: `memwal_login` re-registers a key on + // this same relayer, and whichever path publishes the next + // session clears the flag and resumes normal buffering. + // + // Do NOT signal `firstConnect` here. It means "a session + // exists", and none does — the pump would fall straight through + // its `break; // stdin closed before we ever connected`, win the + // shutdown race in `runBridge`, and `markStdinClosed()` would + // disable the very `reconnect()` the error text tells the user + // to reach via `memwal_login`. `failPendingForward` writes to + // stdout directly and needs no pump. + // + // Same staleness test as the publish path above: a 401 for the + // key a login already replaced says nothing about the new one, + // and latching the flag on it would refuse every request against + // a session that is live and fine. + if ( + err instanceof RelayerUnauthorizedError && + !sse && + openingGeneration === credentialGeneration + ) { + credentialsRejected = true; + // Everything still queued never left the process, so no + // upstream initialize reply will arrive to consume its arm. + // `failRequest` keeps initialize arms for replies that CAN + // still arrive; a leaked one here would swallow the reply to + // a reused id after `memwal_login`. + for (const msg of pendingForward) { + if (msg.method === "initialize" && msg.id != null) { + suppressUpstreamReplies.delete(msg.id); + } + } + failPendingForward("credentials rejected", UNAUTHORIZED_FAILURE); + } if (stdinClosed) break; - const backoff = Math.min(15_000, 500 * Math.pow(2, attempt - 1)); + // Floor the geometric backoff at whatever throttle window is + // still open. Without this the first retry after a 429 lands + // 500ms later, well inside the interval the relayer asked for. + const backoff = Math.max( + Math.min(15_000, 500 * Math.pow(2, attempt - 1)), + throttledUntilMs - Date.now(), + ); await new Promise((resolve) => { const timer = setTimeout(() => { unregister(); diff --git a/packages/mcp/src/index.ts b/packages/mcp/src/index.ts index 3a30f1d15..02b9b3fff 100644 --- a/packages/mcp/src/index.ts +++ b/packages/mcp/src/index.ts @@ -11,7 +11,7 @@ */ import { clearCreds, credsPath, loadCreds } from "./auth.js"; import { runAuthRequiredServer } from "./auth-required.js"; -import { runBridge } from "./bridge.js"; +import { notePendingLoginSuccess, runBridge } from "./bridge.js"; import { loginFlow } from "./login.js"; import { log, note } from "./logger.js"; @@ -28,6 +28,9 @@ interface ParsedArgs { webUrl?: string; label?: string; namespace?: string; + /** Args parseArgs did not recognise, in the order seen. For a flag + * written `--key=value`, only `--key` is recorded — see parseArgs. */ + unknown: string[]; } /** Per-environment URL shortcuts. `--dev`/`--staging`/`--local` set both @@ -39,8 +42,12 @@ const ENV_PRESETS: Record = { local: { relayer: "http://127.0.0.1:8000", web: "http://localhost:5173" }, }; -function parseArgs(argv: string[]): ParsedArgs { - const out: ParsedArgs = { help: false, logout: false, forceLogin: false }; +/** Bare words that are commands rather than values. An unknown flag must not + * swallow one as its argument. */ +const POSITIONALS = new Set(["login"]); + +export function parseArgs(argv: string[]): ParsedArgs { + const out: ParsedArgs = { help: false, logout: false, forceLogin: false, unknown: [] }; for (let i = 0; i < argv.length; i++) { const a = argv[i]; const next = () => argv[++i]; @@ -89,7 +96,33 @@ function parseArgs(argv: string[]): ParsedArgs { else if (a?.startsWith("--label=")) out.label = a.split("=", 2)[1]; else if (a?.startsWith("--namespace=")) out.namespace = a.split("=", 2)[1]; else if (a?.startsWith("--ns=")) out.namespace = a.split("=", 2)[1]; - // Unknown flag: ignore silently. + // Anything still unmatched is a typo, or a flag from a newer + // build. Values of KNOWN value-taking flags never reach this + // branch — `next()` already consumed them. + else if (a !== undefined) { + // Record the key only. A mistyped value-taking flag written + // `--tokenn=hunter2` would otherwise put the user's secret + // on stderr, which is the one place this warning must not + // put it. + const eq = a.indexOf("="); + out.unknown.push(eq === -1 ? a : a.slice(0, eq)); + // An unknown flag may take its value as the next token, so + // consume one — `--namesapce work` should warn once about + // `--namesapce`, not a second time naming the user's data. + // POSITIONALS are exempt: they are commands, not values, and + // swallowing one would turn `memwal-mcp --typo login` into a + // run that never logs in. + const value = argv[i + 1]; + if ( + a.startsWith("-") && + eq === -1 && + value !== undefined && + !value.startsWith("-") && + !POSITIONALS.has(value) + ) { + i++; + } + } break; } } @@ -99,6 +132,17 @@ function parseArgs(argv: string[]): ParsedArgs { export async function main(argv: string[] = process.argv.slice(2)): Promise { const args = parseArgs(argv); + // Runs before the --help branch so `memwal-mcp --typo --help` still calls + // the typo out. Warn, never exit: an unknown flag from a newer config must + // not brick the server. + for (const flag of args.unknown) { + log.warn("cli.unrecognised_arg", { arg: flag }); + note( + `Unrecognised option \`${flag}\` — ignored. ` + + `Run \`memwal-mcp --help\` for the supported options.` + ); + } + if (args.help) { printHelp(); return; @@ -208,6 +252,16 @@ export async function main(argv: string[] = process.argv.slice(2)): Promise [ + ` ${`--${name}`.padEnd(33)}relayer: ${urls.relayer}`, + ` ${"".padEnd(33)}web: ${urls.web}`, + ]); const help = [ "memwal-mcp — Walrus Memory Model Context Protocol client", "", @@ -279,7 +345,7 @@ function printHelp(): void { " Default: https://memory.walrus.xyz", " --label Friendly delegate-key label", " registered on-chain. Default:", - ' "Walrus Memory MCP"', + ' "MCP Client"', " --namespace Default memory namespace applied", " to memwal_remember / recall /", " analyze / restore when the agent", @@ -288,6 +354,13 @@ function printHelp(): void { ' relayer uses its "default".', " Alias: --ns", "", + "Network presets (set --relayer and --web-url together):", + ...presetLines, + "", + " An explicit --relayer or --web-url", + " wins over a preset, whichever", + " order they are written in.", + "", "Environment (equivalent to options):", " MEMWAL_SERVER_URL same as --relayer", " MEMWAL_WEB_URL same as --web-url", @@ -332,7 +405,7 @@ function printHelp(): void { " }", "", ].join("\n"); - process.stderr.write(help + "\n"); + return help; } // Re-exports — handy if someone wants to embed this in another tool. diff --git a/packages/mcp/src/messages.ts b/packages/mcp/src/messages.ts new file mode 100644 index 000000000..11e7d6176 --- /dev/null +++ b/packages/mcp/src/messages.ts @@ -0,0 +1,136 @@ +/** + * User-facing copy for the sign-in lifecycle. + * + * The same events are reported from two places — the auth-required stub + * (signed out) and the bridge (signed in) — and the wording had already + * drifted between them. Keeping the strings here means one voice regardless + * of which mode the user happens to be in. + */ + +/** What a completed sign-in produced, as the user needs it described. */ +export interface LoginSuccessInfo { + accountId: string; + delegateAddress: string; + /** The file `saveCreds` actually wrote — project-local or global. Never a + * hardcoded `~/.memwal/credentials.json`: which one it lands in is exactly + * the confusion behind GH #628, so the copy states the resolved path. */ + credentialsPath: string; +} + +/** What a sign-in prompt needs to describe the flow accurately. */ +export interface LoginPromptInfo { + /** The connect URL the user must open. */ + url: string; + /** Where credentials will land — resolved, never assumed to be global. */ + credentialsPath: string; + /** True when credentials already exist, so signing in REPLACES them. */ + signedIn: boolean; +} + +/** + * The `memwal_login` tool result, for both modes. + * + * The URL appears three times — plain, fenced, and as a link — on purpose. + * Some clients paraphrase tool output aggressively, and repeating it in three + * forms means at least one survives into what the user actually sees. Without + * the URL the user cannot proceed at all, so this is the one place where + * redundancy beats tidiness. + */ +export function loginPrompt(info: LoginPromptInfo): string { + return [ + `## ⚠️ ACTION REQUIRED: User must click this URL to sign in`, + ``, + `**URL:** ${info.url}`, + ``, + "```", + info.url, + "```", + ``, + `[Click here to open Walrus Memory sign-in](${info.url})`, + ``, + `**IMPORTANT for the assistant**: do NOT summarize or omit the URL above.`, + `The user CANNOT proceed without seeing the exact URL. Surface it verbatim`, + `in your reply, then explain the steps:`, + ``, + `1. Open the URL in any browser (it may have already opened automatically)`, + `2. Click **Connect Sui Wallet** and approve the on-chain \`add_delegate_key\` transaction`, + `3. Once "Connected" appears in the browser, retry the original request — the other memwal_* tools will then have credentials at \`${info.credentialsPath}\``, + ...(info.signedIn + ? [ + ``, + `**Note:** you are already signed in. Completing this replaces the stored delegate key at \`${info.credentialsPath}\` with the new wallet's.`, + ] + : []), + ``, + `_The login link stays valid for 5 minutes. If it expires, call \`memwal_login\` again to get a fresh URL._`, + ].join("\n"); +} + +/** Sui ids are 66 characters; showing one whole swamps the message. */ +function shortId(id: string): string { + return id.length > 20 ? `${id.slice(0, 10)}…${id.slice(-6)}` : id; +} + +/** + * Banner prefixed onto the first tool result after a sign-in completes. + * + * Deliberately a ONE-SHOT, unlike {@link loginFailureNotice}, which repeats + * until the user fixes it: a failed sign-in is a state that persists until + * they act, but a successful one is an event. Repeating it on every recall + * would be noise on top of every result the user asked for. + */ +export function loginSuccessNotice(info: LoginSuccessInfo): string { + return [ + "✅ Signed in to Walrus Memory.", + "", + `Account: ${shortId(info.accountId)}`, + `Delegate: ${shortId(info.delegateAddress)}`, + `Saved to: ${info.credentialsPath}`, + "", + "This connection is now authenticated — no client restart needed.", + "", + "---", + "", + ].join("\n"); +} + +/** + * The `notifications/message` twin of the banner, mirroring the warning the + * failure path already sends. Clients differ in which surface they show — + * some render notifications inline, others drop them — so the confirmation + * goes out on both and neither depends on the other. + */ +export function loginSuccessNotification(info: LoginSuccessInfo): string { + return ( + `Walrus Memory sign-in complete — account ${shortId(info.accountId)}, ` + + `credentials saved to ${info.credentialsPath}.` + ); +} + +/** + * Prefix explaining that a sign-in was attempted and did not complete. + * + * Repeats on every refused tool call, unlike {@link loginSuccessNotice}: this + * describes a state the user is still in, and the tool call that started the + * sign-in returned its URL long before the failure was known, so there is no + * earlier surface left to report it on. + * + * `reason` null — no attempt on record — yields the empty string, so callers + * can prefix unconditionally. + */ +export function loginFailureNotice(reason: string | null): string { + if (!reason) return ""; + return [ + "⚠️ A sign-in was started but never completed, so there are still no credentials.", + "", + `Reason: ${reason}`, + "", + "The unused key from this attempt may already be registered on your account. Remove it", + "from the dashboard if you are not using it. Sign in again and open the new link", + "straight away. A retry only helps once the MCP client is left running through the", + "wallet prompt.", + "", + "---", + "", + ].join("\n"); +} diff --git a/packages/mcp/test/credential-file-permissions.test.mjs b/packages/mcp/test/credential-file-permissions.test.mjs new file mode 100644 index 000000000..f85f583e2 --- /dev/null +++ b/packages/mcp/test/credential-file-permissions.test.mjs @@ -0,0 +1,319 @@ +/** + * Credential file permissions (GH #520 / WALM-312). + * + * `saveCreds` wrote the new delegate private key straight to the final path and + * only then called `chmodSync(0600)`. `writeFileSync`'s `mode` follows POSIX + * `open()` — it applies when the kernel creates the inode, never to an existing + * one. So a `credentials.json` left at broader permissions by anything outside + * this code (a manual chmod, a restored backup, another tool) received the new + * secret under the *old* mode, and only the second, non-atomic syscall tightened + * it. + * + * The window itself is a race and cannot be asserted by watching for it. What + * can be asserted is the property that closes it: the new secret is never + * written through the old inode at all. A reader that already holds that inode + * open — the attacker in the report — is the observer that makes this + * deterministic. It sees the old bytes forever if the write went to a fresh + * 0600 inode that was then renamed over the name, and the new key the moment + * the write went through the old permissive inode in place. + * + * `auth.js` resolves paths at call time, so each test sets HOME and cwd first + * and then imports with a cache-busting query — the pattern used by + * credential-resolution.test.mjs. + */ +import { test } from "node:test"; +import assert from "node:assert/strict"; +import { + mkdtempSync, + mkdirSync, + writeFileSync, + readFileSync, + readdirSync, + readSync, + openSync, + closeSync, + statSync, + rmSync, + realpathSync, + renameSync, + existsSync, +} from "node:fs"; +import { tmpdir } from "node:os"; +import { join, dirname } from "node:path"; + +const ACCOUNT = "0x" + "a".repeat(64); +const OTHER_ACCOUNT = "0x" + "b".repeat(64); +const OLD_KEY = "1".repeat(64); +const NEW_KEY = "2".repeat(64); + +function makeCreds(accountId, delegatePrivateKey) { + return { + delegatePrivateKey, + delegatePublicKeyHex: "d".repeat(64), + delegateAddress: "0x" + "e".repeat(64), + walletAddress: "0x" + "f".repeat(64), + accountId, + packageId: "0x" + "1".repeat(64), + relayerUrl: "https://relayer.example", + label: "Test", + createdAt: new Date(0).toISOString(), + version: 1, + }; +} + +/** Permission bits only — `statSync().mode` carries the file type as well. */ +function modeOf(path) { + return statSync(path).mode & 0o777; +} + +// Windows does not enforce POSIX mode bits. `statSync().mode` there is +// synthesized from the read-only attribute, so a `0o600` assertion tests +// nothing, and `writeSecretFile` falls back to an in-place write when the +// destination is locked. NTFS ACLs carry the protection instead, inherited +// from the containing directory. Tests whose premise IS the mode bit are +// skipped rather than weakened into passing everywhere. +const POSIX_ONLY = { + skip: process.platform === "win32" ? "POSIX mode bits are not enforced on Windows" : false, +}; + +/** + * Fresh HOME with the module re-imported so it observes it. The working + * directory is moved to an empty sandbox too, so no project-local + * `.memwal` from the real checkout can win over the global file under test. + */ +async function sandbox(t, { existingFileMode } = {}) { + // Canonicalised for the same reason as credential-resolution.test.mjs: + // `homedir()` and `process.cwd()` report resolved paths, and on macOS + // `/var` is a symlink to `/private/var`. + const home = realpathSync(mkdtempSync(join(tmpdir(), "memwal-perm-home-"))); + const cwd = realpathSync(mkdtempSync(join(tmpdir(), "memwal-perm-cwd-"))); + const prevHome = process.env.HOME; + const prevProfile = process.env.USERPROFILE; + const prevCwd = process.cwd(); + + process.env.HOME = home; + process.env.USERPROFILE = home; + process.chdir(cwd); + + const path = join(home, ".memwal", "credentials.json"); + if (existingFileMode !== undefined) { + mkdirSync(dirname(path), { recursive: true, mode: 0o700 }); + writeFileSync(path, JSON.stringify(makeCreds(ACCOUNT, OLD_KEY)), { + mode: existingFileMode, + }); + } + + t.after(() => { + process.chdir(prevCwd); + process.env.HOME = prevHome; + process.env.USERPROFILE = prevProfile; + rmSync(home, { recursive: true, force: true }); + rmSync(cwd, { recursive: true, force: true }); + }); + + const auth = await import(`../dist/auth.js?walm312=${Date.now()}-${Math.random()}`); + return { auth, home, path }; +} + +/** Read through an already-open descriptor, which follows the inode rather + * than the name — so this reports what a holder of the *old* file sees, even + * after the name has been repointed at a different inode. */ +function readThroughOpenFd(fd) { + const buffer = Buffer.alloc(4096); + const bytes = readSync(fd, buffer, 0, buffer.length, 0); + return buffer.subarray(0, bytes).toString("utf8"); +} + +test("saveCreds never writes the new secret through a pre-existing permissive file", POSIX_ONLY, async (t) => { + const { auth, path } = await sandbox(t, { existingFileMode: 0o644 }); + + // The attacker's handle, opened while the file is still world-readable and + // held across the save. Same accountId as the file already on disk, so this + // is a plain same-account key rotation with no backup in the way. + const attackerFd = openSync(path, "r"); + t.after(() => closeSync(attackerFd)); + + auth.saveCreds(makeCreds(ACCOUNT, NEW_KEY)); + + const seenByAttacker = readThroughOpenFd(attackerFd); + assert.ok( + !seenByAttacker.includes(NEW_KEY), + "the new delegate private key must never be readable through the pre-existing 0644 inode", + ); + assert.ok( + seenByAttacker.includes(OLD_KEY), + "the displaced inode should still hold the old content, proving it was replaced rather than truncated in place", + ); + + // Positive control: the save really did happen, at the right permission. + assert.equal(JSON.parse(readFileSync(path, "utf8")).delegatePrivateKey, NEW_KEY); + assert.equal(modeOf(path), 0o600, "the file in place after the save must be 0600"); +}); + +test("saveCreds creates a new credentials file at 0600", POSIX_ONLY, async (t) => { + const { auth, path } = await sandbox(t); + + auth.saveCreds(makeCreds(ACCOUNT, NEW_KEY)); + + assert.equal(modeOf(path), 0o600); + assert.equal(modeOf(dirname(path)), 0o700, "the containing directory stays owner-only"); +}); + +test("the backup of a displaced account is written at 0600", async (t) => { + const { auth } = await sandbox(t, { existingFileMode: 0o600 }); + + const saved = auth.saveCreds(makeCreds(OTHER_ACCOUNT, NEW_KEY)); + + assert.equal(saved.replacedAccountId, ACCOUNT, "the outgoing account should be reported"); + assert.ok(saved.backedUpTo, "a different incoming account should be backed up"); + if (process.platform !== "win32") { + assert.equal(modeOf(saved.backedUpTo), 0o600, "the backup holds the same plaintext key"); + } + assert.equal(JSON.parse(readFileSync(saved.backedUpTo, "utf8")).delegatePrivateKey, OLD_KEY); +}); + +test("saveCreds leaves no temporary file behind", async (t) => { + const { auth, home } = await sandbox(t, { existingFileMode: 0o644 }); + + auth.saveCreds(makeCreds(ACCOUNT, NEW_KEY)); + + const stray = readdirSync(join(home, ".memwal")).filter((name) => name.endsWith(".tmp")); + assert.deepEqual(stray, [], "a completed save should not leave a temporary file in the directory"); +}); + +/** + * The Windows locked-destination fallback. + * + * CI has no Windows runner, so these drive `replaceWithTemp` directly with an + * injected platform and a `rename` that fails the way `MoveFileEx` does when + * another handle holds the destination. The fallback returns SUCCESSFULLY, so + * nothing upstream cleans up after it — a leaked temp here is a second + * plaintext copy of the delegate key, which is the exact class of bug this + * file exists to prevent. + */ +const lockedRename = (code) => () => { + const err = new Error(`${code}: locked`); + err.code = code; + throw err; +}; + +/** A temp file already written at 0600, as `writeSecretFile` leaves it. */ +function stageTemp(t, contents = "SECRET_KEY_MATERIAL") { + const dir = realpathSync(mkdtempSync(join(tmpdir(), "memwal-replace-"))); + t.after(() => rmSync(dir, { recursive: true, force: true })); + const tmp = join(dir, ".credentials.json.123.abc.tmp"); + const dest = join(dir, "credentials.json"); + writeFileSync(tmp, contents, { mode: 0o600 }); + return { dir, tmp, dest, contents }; +} + +for (const code of ["EPERM", "EACCES", "EBUSY"]) { + test(`a destination locked with ${code} still lands, leaving no temp behind`, async (t) => { + const { auth } = await sandbox(t); + const { dir, tmp, dest, contents } = stageTemp(t); + + auth.replaceWithTemp(tmp, dest, contents, { + platform: "win32", + rename: lockedRename(code), + sleep: () => {}, + }); + + assert.equal(readFileSync(dest, "utf8"), contents, "the save must still land"); + assert.equal( + existsSync(tmp), + false, + "the temp still holds the plaintext key — it must not survive the fallback", + ); + assert.deepEqual( + readdirSync(dir).filter((n) => n.endsWith(".tmp")), + [], + "no temporary file may remain in the credentials directory", + ); + }); +} + +test("repeated locked saves do not accumulate copies of the key", async (t) => { + // The regression the fallback introduced: the old in-place writeFileSync + // never created a sibling file, so nothing used to pile up here. + const { auth } = await sandbox(t); + const dir = realpathSync(mkdtempSync(join(tmpdir(), "memwal-replace-many-"))); + t.after(() => rmSync(dir, { recursive: true, force: true })); + const dest = join(dir, "credentials.json"); + + for (let i = 0; i < 3; i++) { + const tmp = join(dir, `.credentials.json.123.run${i}.tmp`); + writeFileSync(tmp, `SECRET_${i}`, { mode: 0o600 }); + auth.replaceWithTemp(tmp, dest, `SECRET_${i}`, { + platform: "win32", + rename: lockedRename("EPERM"), + sleep: () => {}, + }); + } + + assert.equal(readFileSync(dest, "utf8"), "SECRET_2", "the last save wins"); + assert.deepEqual( + readdirSync(dir).filter((n) => n.endsWith(".tmp")), + [], + "three locked saves must not leave three copies of the delegate key", + ); +}); + +test("a lock that clears before the attempts run out renames instead of falling back", async (t) => { + const { auth } = await sandbox(t); + const { tmp, dest, contents } = stageTemp(t); + + let calls = 0; + auth.replaceWithTemp(tmp, dest, contents, { + platform: "win32", + sleep: () => {}, + rename: (from, to) => { + calls++; + if (calls < 3) lockedRename("EPERM")(); + renameSync(from, to); + }, + }); + + assert.equal(calls, 3, "should have retried rather than given up on the first refusal"); + assert.equal(readFileSync(dest, "utf8"), contents); + assert.equal(existsSync(tmp), false, "the rename consumed the temp"); +}); + +test("a non-lock rename error is not swallowed by the Windows path", async (t) => { + // Only lock codes get the retry-and-fall-back treatment. Anything else is + // a real failure and must reach `writeSecretFile`, which removes the temp. + const { auth } = await sandbox(t); + const { tmp, dest, contents } = stageTemp(t); + + assert.throws( + () => + auth.replaceWithTemp(tmp, dest, contents, { + platform: "win32", + rename: lockedRename("ENOSPC"), + sleep: () => {}, + }), + /ENOSPC/, + ); + assert.equal(existsSync(dest), false, "nothing should have been written"); +}); + +test("POSIX does not retry or fall back", async (t) => { + // There the mode IS the protection, and rename(2) replaces a destination + // regardless of who holds it open, so a refusal is a real error. + const { auth } = await sandbox(t); + const { tmp, dest, contents } = stageTemp(t); + + let calls = 0; + assert.throws( + () => + auth.replaceWithTemp(tmp, dest, contents, { + platform: "linux", + rename: () => { + calls++; + lockedRename("EPERM")(); + }, + }), + /EPERM/, + ); + assert.equal(calls, 1, "POSIX must not retry"); + assert.equal(existsSync(dest), false, "and must not write in place"); +}); diff --git a/packages/mcp/test/expired-credentials-recall.test.mjs b/packages/mcp/test/expired-credentials-recall.test.mjs new file mode 100644 index 000000000..c84dd2d7c --- /dev/null +++ b/packages/mcp/test/expired-credentials-recall.test.mjs @@ -0,0 +1,831 @@ +/** + * WALM-602 / GH #365 — an expired session must be distinguishable from an + * empty namespace. + * + * The original report was "recall silently returns empty instead of an auth + * error". The server half of that closed in 0.0.11 (`45b0ad87` made the MCP + * proxy require a registered delegate, so an unregistered key no longer opens + * a session that then honestly reports zero rows). What remains is the client + * half: a relayer that rejects the credentials 401s the SSE handshake, and the + * bridge's background connect treats that like any other connect failure — + * exponential-backoff retry — so the queued tool call waits out the orphan + * sweeper instead of being told the credentials were rejected. + * + * These tests pin the distinction the ticket asks for, and the way back out of + * it: + * - rejected credentials -> an auth error naming the way back in + * - valid creds, no hits -> an ordinary empty result, NOT an error + * - still rejected -> refused again, fast, with the bridge alive + * - `memwal_login` afterwards -> service restored, promptly + * - revoked mid-session -> the in-flight call answered, not orphaned + * - transient 401 mid-session -> recovers with no client intervention + */ +import { test } from "node:test"; +import assert from "node:assert/strict"; +import http from "node:http"; +import { spawn } from "node:child_process"; +import { mkdtempSync, mkdirSync, writeFileSync, rmSync } from "node:fs"; +import { tmpdir } from "node:os"; +import { join, dirname, resolve } from "node:path"; +import { fileURLToPath } from "node:url"; + +const __dirname = dirname(fileURLToPath(import.meta.url)); +const BIN = resolve(__dirname, "../dist/bin/memwal-mcp.js"); +const BEARER = "a".repeat(64); +const ACCOUNT = "0x" + "3".repeat(64); + +/** Bound the whole exchange. Long enough for a couple of reconnect backoffs, + * short enough that a hang fails the test instead of stalling the suite. */ +const CALL_TIMEOUT_MS = 4000; + +function serveVersion(res) { + res.writeHead(200, { "content-type": "application/json" }); + res.end( + JSON.stringify({ + apiVersion: "1.0.0", + relayerVersion: "1.0.0", + minSupportedSdk: { mcp: "0.0.1" }, + }), + ); +} + +/** + * Relayer that rejects the delegate key on the SSE handshake — what the proxy + * now does for a revoked or never-registered delegate. + */ +function startRejectingRelayer() { + let sseAttempts = 0; + const server = http.createServer((req, res) => { + const url = new URL(req.url, "http://127.0.0.1"); + if (req.method === "GET" && url.pathname === "/version") { + serveVersion(res); + return; + } + if (req.method === "GET" && url.pathname === "/api/mcp/sse") { + sseAttempts += 1; + res.writeHead(401, { "content-type": "application/json" }); + res.end(JSON.stringify({ error: "delegate key is not registered" })); + return; + } + res.writeHead(404); + res.end(); + }); + return new Promise((ready) => { + server.listen(0, "127.0.0.1", () => { + const { port } = server.address(); + ready({ + server, + base: `http://127.0.0.1:${port}`, + sseAttempts: () => sseAttempts, + }); + }); + }); +} + +/** + * Healthy relayer whose namespace simply holds nothing — the contrast case. + * Mirrors the sidecar's own wording for a genuinely empty namespace. + */ +function startEmptyNamespaceRelayer() { + let sseRes = null; + const server = http.createServer((req, res) => { + const url = new URL(req.url, "http://127.0.0.1"); + if (req.method === "GET" && url.pathname === "/version") { + serveVersion(res); + return; + } + if (req.method === "GET" && url.pathname === "/api/mcp/sse") { + res.writeHead(200, { + "content-type": "text/event-stream", + "cache-control": "no-cache", + connection: "keep-alive", + }); + res.write("event: endpoint\ndata: /api/mcp/messages?sessionId=test\n\n"); + sseRes = res; + return; + } + if (req.method === "POST" && url.pathname === "/api/mcp/messages") { + let body = ""; + req.on("data", (c) => (body += c)); + req.on("end", () => { + res.writeHead(202); + res.end(); + let msg; + try { + msg = JSON.parse(body); + } catch { + return; + } + if (msg.method === "tools/call") { + sseRes?.write( + `event: message\ndata: ${JSON.stringify({ + jsonrpc: "2.0", + id: msg.id, + result: { + content: [{ type: "text", text: "No matching memories found." }], + isError: false, + }, + })}\n\n`, + ); + } + }); + return; + } + res.writeHead(404); + res.end(); + }); + return new Promise((ready) => { + server.listen(0, "127.0.0.1", () => { + const { port } = server.address(); + ready({ server, base: `http://127.0.0.1:${port}` }); + }); + }); +} + +/** Spawn the bridge against `base` with credentials on disk, wired for stdio. */ +function startBridge(base) { + const home = mkdtempSync(join(tmpdir(), "memwal-test-")); + mkdirSync(join(home, ".memwal")); + writeFileSync( + join(home, ".memwal", "credentials.json"), + JSON.stringify({ + delegatePrivateKey: BEARER, + delegatePublicKeyHex: "b".repeat(64), + delegateAddress: "0x" + "1".repeat(64), + walletAddress: "0x" + "2".repeat(64), + accountId: ACCOUNT, + packageId: "0x" + "4".repeat(64), + relayerUrl: base, + label: "test", + createdAt: new Date(0).toISOString(), + version: 1, + }), + ); + + const child = spawn(process.execPath, [BIN, "--relayer", base, "--web-url", base], { + env: { + ...process.env, + HOME: home, + USERPROFILE: home, + MEMWAL_MCP_CALL_TIMEOUT_MS: String(CALL_TIMEOUT_MS), + }, + stdio: ["pipe", "pipe", "pipe"], + }); + + const received = []; + const listeners = new Set(); + let buf = ""; + child.stdout.on("data", (d) => { + buf += d.toString(); + let nl; + while ((nl = buf.indexOf("\n")) >= 0) { + const line = buf.slice(0, nl); + buf = buf.slice(nl + 1); + if (!line.trim()) continue; + let msg; + try { + msg = JSON.parse(line); + } catch { + continue; + } + received.push(msg); + for (const l of [...listeners]) l(msg); + } + }); + + const send = (obj) => child.stdin.write(JSON.stringify(obj) + "\n"); + const waitFor = (pred, ms) => { + const hit = received.find(pred); + if (hit) return Promise.resolve(hit); + return new Promise((res, rej) => { + const timer = setTimeout(() => { + listeners.delete(l); + rej(new Error("timed out waiting for message")); + }, ms); + const l = (m) => { + if (pred(m)) { + clearTimeout(timer); + listeners.delete(l); + res(m); + } + }; + listeners.add(l); + }); + }; + + return { + send, + waitFor, + cleanup: () => { + child.kill("SIGKILL"); + rmSync(home, { recursive: true, force: true }); + }, + }; +} + +function textOf(msg) { + const content = msg?.result?.content; + if (!Array.isArray(content)) return ""; + return content.map((c) => c?.text ?? "").join("\n"); +} + +test("recall on rejected credentials reports an auth error, not empty results", async (t) => { + const { server, base } = await startRejectingRelayer(); + const bridge = startBridge(base); + t.after(() => { + bridge.cleanup(); + server.close(); + }); + + bridge.send({ + jsonrpc: "2.0", + id: 1, + method: "initialize", + params: { + protocolVersion: "2024-11-05", + capabilities: {}, + clientInfo: { name: "test", version: "0" }, + }, + }); + await bridge.waitFor((m) => m.id === 1 && m.result, 15000); + + bridge.send({ + jsonrpc: "2.0", + id: 2, + method: "tools/call", + params: { name: "memwal_recall", arguments: { query: "anything", limit: 5 } }, + }); + + // Generous relative to CALL_TIMEOUT_MS so a slow machine doesn't flake, but + // far below the 240s production default: the point is that the answer comes + // from the 401, not from waiting out the orphan sweeper. + const reply = await bridge.waitFor((m) => m.id === 2 && (m.result || m.error), 20000); + const text = `${textOf(reply)} ${reply?.error?.message ?? ""}`.toLowerCase(); + + assert.ok( + reply.error || reply.result?.isError, + `recall against rejected credentials must be an error, got: ${JSON.stringify(reply)}`, + ); + assert.ok( + !text.includes("no matching memories"), + "rejected credentials must not read as an empty namespace", + ); + assert.ok( + /401|credential|unauthorized|signed out|memwal_login/.test(text), + `error must name the auth failure and the way back in, got: ${text}`, + ); +}); + +test("recall on an empty namespace reports empty results, not an auth error", async (t) => { + const { server, base } = await startEmptyNamespaceRelayer(); + const bridge = startBridge(base); + t.after(() => { + bridge.cleanup(); + server.close(); + }); + + bridge.send({ + jsonrpc: "2.0", + id: 1, + method: "initialize", + params: { + protocolVersion: "2024-11-05", + capabilities: {}, + clientInfo: { name: "test", version: "0" }, + }, + }); + await bridge.waitFor((m) => m.id === 1 && m.result, 15000); + + bridge.send({ + jsonrpc: "2.0", + id: 2, + method: "tools/call", + params: { name: "memwal_recall", arguments: { query: "anything", limit: 5 } }, + }); + + const reply = await bridge.waitFor((m) => m.id === 2 && (m.result || m.error), 20000); + const text = textOf(reply); + + assert.equal(reply.error, undefined, `empty namespace must not error: ${JSON.stringify(reply)}`); + assert.notEqual(reply.result?.isError, true, "empty namespace must not be an error result"); + assert.match(text, /no matching memories/i); + assert.ok( + !/401|unauthorized|signed out/i.test(text), + `empty namespace must not read as an auth failure, got: ${text}`, + ); +}); + +/** + * Relayer that 401s one specific delegate key and accepts every other one — + * what a revoked key looks like once `memwal_login` has registered a fresh one. + * Sessions that DO open answer `tools/call` with an ordinary empty result, so + * "recovered" is distinguishable from "still refusing". + * + * `holdRejections` parks each 401 until `releaseRejections()`, so a test can + * queue requests before the bridge learns the key is rejected. + */ +function startRevokedKeyRelayer(revokedBearer, { holdRejections = false } = {}) { + let sseRes = null; + let rejections = 0; + let accepted = 0; + let held = []; + const reject = (res) => { + rejections += 1; + res.writeHead(401, { "content-type": "application/json" }); + res.end(JSON.stringify({ error: "delegate key is not registered" })); + }; + const server = http.createServer((req, res) => { + const url = new URL(req.url, "http://127.0.0.1"); + if (req.method === "GET" && url.pathname === "/version") { + serveVersion(res); + return; + } + if (req.method === "GET" && url.pathname === "/api/mcp/sse") { + const bearer = (req.headers.authorization ?? "").replace(/^Bearer\s+/i, ""); + if (bearer === revokedBearer) { + if (holdRejections) held.push(res); + else reject(res); + return; + } + res.writeHead(200, { + "content-type": "text/event-stream", + "cache-control": "no-cache", + connection: "keep-alive", + }); + accepted += 1; + res.write("event: endpoint\ndata: /api/mcp/messages?sessionId=recovered\n\n"); + const heartbeat = setInterval(() => res.write(": keepalive\n\n"), 250); + heartbeat.unref?.(); + res.on("close", () => clearInterval(heartbeat)); + sseRes = res; + return; + } + if (req.method === "POST" && url.pathname === "/api/mcp/messages") { + let body = ""; + req.on("data", (c) => (body += c)); + req.on("end", () => { + res.writeHead(202); + res.end(); + let msg; + try { + msg = JSON.parse(body); + } catch { + return; + } + if (msg.id == null) return; + sseRes?.write( + `event: message\ndata: ${JSON.stringify({ + jsonrpc: "2.0", + id: msg.id, + result: + msg.method === "tools/call" + ? { + content: [ + { type: "text", text: "No matching memories found." }, + ], + isError: false, + } + : {}, + })}\n\n`, + ); + }); + return; + } + res.writeHead(404); + res.end(); + }); + return new Promise((ready) => { + server.listen(0, "127.0.0.1", () => { + const { port } = server.address(); + ready({ + server, + base: `http://127.0.0.1:${port}`, + rejections: () => rejections, + accepted: () => accepted, + releaseRejections: () => { + holdRejections = false; + for (const res of held.splice(0)) reject(res); + }, + }); + }); + }); +} + +/** Drive the browser half of `memwal_login` against the bridge's own localhost + * listener — same handshake the dashboard performs (preflight, then callback). + * Mirrors `live-login-credentials.test.mjs`. */ +async function completeLogin(connectUrl, accountId) { + const url = new URL(connectUrl); + const callbackBase = `http://127.0.0.1:${url.searchParams.get("port")}`; + const headers = { origin: url.origin, "content-type": "application/json" }; + const body = { + state: url.searchParams.get("connectState"), + publicKey: url.searchParams.get("publicKey"), + relayer: url.searchParams.get("relayer"), + }; + + const preflight = await fetch(`${callbackBase}/preflight`, { + method: "POST", + headers, + body: JSON.stringify(body), + }); + assert.equal(preflight.status, 200); + + const callback = await fetch(`${callbackBase}/callback`, { + method: "POST", + headers, + body: JSON.stringify({ + state: body.state, + accountId, + walletAddress: "0x" + "2".repeat(64), + packageId: "0x" + "4".repeat(64), + }), + }); + assert.equal(callback.status, 200); +} + +/** Poll until `predicate` holds. Same shape as `live-login-credentials`. */ +async function waitUntil(predicate, timeoutMs = 10_000) { + const started = Date.now(); + while (!predicate()) { + if (Date.now() - started > timeoutMs) throw new Error("timed out waiting for condition"); + await new Promise((r) => setTimeout(r, 25)); + } +} + +test("a rejected key keeps failing fast, and memwal_login restores service", async (t) => { + const relayer = await startRevokedKeyRelayer(BEARER); + const { server, base } = relayer; + const bridge = startBridge(base); + t.after(() => { + bridge.cleanup(); + server.close(); + }); + + bridge.send({ + jsonrpc: "2.0", + id: 1, + method: "initialize", + params: { + protocolVersion: "2024-11-05", + capabilities: {}, + clientInfo: { name: "test", version: "0" }, + }, + }); + await bridge.waitFor((m) => m.id === 1 && m.result, 15000); + + const recall = (id) => { + bridge.send({ + jsonrpc: "2.0", + id, + method: "tools/call", + params: { name: "memwal_recall", arguments: { query: "anything", limit: 5 } }, + }); + return bridge.waitFor((m) => m.id === id && (m.result || m.error), 20000); + }; + + const first = await recall(2); + assert.ok(first.result?.isError || first.error, "first recall must be an auth error"); + + // The bridge must still be reading stdin after the 401 answered the first + // call. A second recall is refused ON ARRIVAL, so it comes back well inside + // CALL_TIMEOUT_MS — anything near that deadline means it parked instead. + const startedAt = Date.now(); + const second = await recall(3); + const elapsed = Date.now() - startedAt; + assert.ok( + second.result?.isError || second.error, + `second recall must also be an auth error, got: ${JSON.stringify(second)}`, + ); + assert.ok( + elapsed < CALL_TIMEOUT_MS / 2, + `second recall must fail fast, took ${elapsed}ms (deadline ${CALL_TIMEOUT_MS}ms)`, + ); + + // Let the background connect back off a few times before signing in — a + // real user takes seconds to click the link. By the 4th rejection the loop + // is asleep for ~4s, which is long enough that "the pump woke because the + // login published a session" and "the pump woke because the backoff + // happened to expire" are no longer the same measurement. + await waitUntil(() => relayer.rejections() >= 4, 15_000); + + // `memwal_login` is answered locally, so it must still work while the saved + // key is being refused — it is the only way back in. + bridge.send({ + jsonrpc: "2.0", + id: 4, + method: "tools/call", + params: { name: "memwal_login", arguments: {} }, + }); + const loginReply = await bridge.waitFor((m) => m.id === 4 && m.result, 20000); + const connectUrl = /\*\*URL:\*\* (\S+)/.exec(textOf(loginReply))?.[1]; + assert.ok(connectUrl, `memwal_login must return the browser URL, got: ${textOf(loginReply)}`); + await completeLogin(connectUrl, ACCOUNT); + // The login's own reconnect owns the new handshake; wait for the relayer to + // accept it before asking for the recall, so the assertion below is about + // the flag being cleared and not about who won a race. + await waitUntil(() => relayer.accepted() > 0); + + // The new key is accepted, so the bridge must resume normal buffering: an + // ordinary empty result, not the credentials-rejected refusal. It must also + // land promptly: the login's own reconnect has to release the server pump, + // because nothing else is draining this stream until the background + // connect's backoff — up to 15s in production — next expires. + const recoveredAt = Date.now(); + const recovered = await recall(5); + const recoveredIn = Date.now() - recoveredAt; + assert.equal( + recovered.error, + undefined, + `recall after re-login must not error: ${JSON.stringify(recovered)}`, + ); + assert.notEqual( + recovered.result?.isError, + true, + `recall after re-login must not be refused: ${JSON.stringify(recovered)}`, + ); + assert.match(textOf(recovered), /no matching memories/i); + assert.ok( + recoveredIn < 1500, + `recall after re-login must not wait for the connect backoff, took ${recoveredIn}ms`, + ); + // The saved key really was refused throughout, rather than the relayer + // having quietly accepted it at some point. + assert.ok(relayer.rejections() > 0, "the revoked key must have been 401'd"); +}); + +test("an initialize queued before the 401 does not swallow a reused id after memwal_login", async (t) => { + const relayer = await startRevokedKeyRelayer(BEARER, { holdRejections: true }); + const { server, base } = relayer; + const bridge = startBridge(base); + t.after(() => { + bridge.cleanup(); + server.close(); + }); + + // The bridge answers initialize locally and arms a suppression for the + // upstream reply it expects once the initialize is forwarded. With the 401 + // held, both requests are still queued when the key is rejected, so neither + // is ever forwarded and no upstream reply comes to consume that arm. + bridge.send({ + jsonrpc: "2.0", + id: 1, + method: "initialize", + params: { + protocolVersion: "2024-11-05", + capabilities: {}, + clientInfo: { name: "test", version: "0" }, + }, + }); + await bridge.waitFor((m) => m.id === 1 && m.result, 15000); + bridge.send({ + jsonrpc: "2.0", + id: 2, + method: "tools/call", + params: { name: "memwal_recall", arguments: { query: "anything", limit: 5 } }, + }); + relayer.releaseRejections(); + const refused = await bridge.waitFor((m) => m.id === 2 && (m.result || m.error), 20000); + assert.ok(refused.result?.isError || refused.error, "queued recall must be an auth error"); + + bridge.send({ + jsonrpc: "2.0", + id: 3, + method: "tools/call", + params: { name: "memwal_login", arguments: {} }, + }); + const loginReply = await bridge.waitFor((m) => m.id === 3 && m.result, 20000); + const connectUrl = /\*\*URL:\*\* (\S+)/.exec(textOf(loginReply))?.[1]; + assert.ok(connectUrl, `memwal_login must return the browser URL, got: ${textOf(loginReply)}`); + await completeLogin(connectUrl, ACCOUNT); + await waitUntil(() => relayer.accepted() > 0); + + // JSON-RPC lets a client reuse an id once its request is answered. A + // leftover arm drops this genuine reply and untracks the id, so not even + // the orphan sweeper answers it: the call hangs. + bridge.send({ + jsonrpc: "2.0", + id: 1, + method: "tools/call", + params: { name: "memwal_recall", arguments: { query: "anything", limit: 5 } }, + }); + const reply = await bridge.waitFor( + (m) => m.id === 1 && Array.isArray(m.result?.content), + 20000, + ); + assert.notEqual( + reply.result.isError, + true, + `reused id must get the relayer's reply, got: ${JSON.stringify(reply)}`, + ); + assert.match(textOf(reply), /no matching memories/i); +}); + +/** + * Relayer whose key is revoked WHILE a session is live: the open stream is cut + * and every later handshake 401s. `restore()` puts it back, standing in for a + * WAF or rate-limit 401 that clears on its own. + */ +function startMidSessionRevokeRelayer() { + let sseRes = null; + let rejecting = false; + let parkCalls = false; + let accepted = 0; + let rejections = 0; + let calls = 0; + + const server = http.createServer((req, res) => { + const url = new URL(req.url, "http://127.0.0.1"); + if (req.method === "GET" && url.pathname === "/version") { + serveVersion(res); + return; + } + if (req.method === "GET" && url.pathname === "/api/mcp/sse") { + if (rejecting) { + rejections += 1; + res.writeHead(401, { "content-type": "application/json" }); + res.end(JSON.stringify({ error: "delegate key was revoked" })); + return; + } + accepted += 1; + res.writeHead(200, { + "content-type": "text/event-stream", + "cache-control": "no-cache", + connection: "keep-alive", + }); + res.write(`event: endpoint\ndata: /api/mcp/messages?sessionId=s${accepted}\n\n`); + const heartbeat = setInterval(() => res.write(": keepalive\n\n"), 250); + heartbeat.unref?.(); + res.on("close", () => clearInterval(heartbeat)); + sseRes = res; + return; + } + if (req.method === "POST" && url.pathname === "/api/mcp/messages") { + let body = ""; + req.on("data", (c) => (body += c)); + req.on("end", () => { + res.writeHead(202); + res.end(); + let msg; + try { + msg = JSON.parse(body); + } catch { + return; + } + if (msg.id == null) return; + if (msg.method === "tools/call") { + calls += 1; + // Park it: the point of the revocation case is a call that + // is already in flight when the key stops being accepted. + if (parkCalls) return; + } + sseRes?.write( + `event: message\ndata: ${JSON.stringify({ + jsonrpc: "2.0", + id: msg.id, + result: + msg.method === "tools/call" + ? { + content: [ + { type: "text", text: "No matching memories found." }, + ], + isError: false, + } + : {}, + })}\n\n`, + ); + }); + return; + } + res.writeHead(404); + res.end(); + }); + + return new Promise((ready) => { + server.listen(0, "127.0.0.1", () => { + const { port } = server.address(); + ready({ + server, + base: `http://127.0.0.1:${port}`, + accepted: () => accepted, + rejections: () => rejections, + calls: () => calls, + park: () => { + parkCalls = true; + }, + revoke: () => { + rejecting = true; + parkCalls = false; + sseRes?.destroy(); + sseRes = null; + }, + restore: () => { + rejecting = false; + }, + }); + }); + }); +} + +test("a key revoked mid-session answers the in-flight call instead of orphaning it", async (t) => { + const relayer = await startMidSessionRevokeRelayer(); + const bridge = startBridge(relayer.base); + t.after(() => { + bridge.cleanup(); + relayer.server.close(); + }); + + bridge.send({ + jsonrpc: "2.0", + id: 1, + method: "initialize", + params: { + protocolVersion: "2024-11-05", + capabilities: {}, + clientInfo: { name: "test", version: "0" }, + }, + }); + await bridge.waitFor((m) => m.id === 1 && m.result, 15000); + await waitUntil(() => relayer.accepted() > 0); + + // In flight against a live session, with no reply coming. + relayer.park(); + bridge.send({ + jsonrpc: "2.0", + id: 2, + method: "tools/call", + params: { name: "memwal_recall", arguments: { query: "anything", limit: 5 } }, + }); + await waitUntil(() => relayer.calls() > 0); + + // The key is revoked underneath it: the stream is cut and the reconnect + // that follows is 401'd. + const revokedAt = Date.now(); + relayer.revoke(); + + const reply = await bridge.waitFor((m) => m.id === 2 && (m.result || m.error), 20000); + const elapsed = Date.now() - revokedAt; + const text = `${textOf(reply)} ${reply?.error?.message ?? ""}`.toLowerCase(); + + assert.ok(reply.error || reply.result?.isError, "the in-flight call must be answered as error"); + assert.match( + text, + /401|credential|unauthorized|memwal_login/, + `the in-flight call must name the rejection, got: ${text}`, + ); + assert.ok( + !text.includes("please retry"), + `"please retry" is the orphan sweeper's advice and cannot work here, got: ${text}`, + ); + assert.ok( + elapsed < CALL_TIMEOUT_MS, + `must beat the orphan sweeper's ${CALL_TIMEOUT_MS}ms deadline, took ${elapsed}ms`, + ); +}); + +test("a transient mid-session 401 recovers on its own, without memwal_login", async (t) => { + const relayer = await startMidSessionRevokeRelayer(); + const bridge = startBridge(relayer.base); + t.after(() => { + bridge.cleanup(); + relayer.server.close(); + }); + + bridge.send({ + jsonrpc: "2.0", + id: 1, + method: "initialize", + params: { + protocolVersion: "2024-11-05", + capabilities: {}, + clientInfo: { name: "test", version: "0" }, + }, + }); + await bridge.waitFor((m) => m.id === 1 && m.result, 15000); + await waitUntil(() => relayer.accepted() > 0); + + // A WAF or rate-limit blip: 401 for a while, then fine again. Nothing here + // calls `memwal_login` — the saved key was always good. + relayer.revoke(); + await waitUntil(() => relayer.rejections() >= 2, 15_000); + relayer.restore(); + + // The server pump keeps driving `reconnect()` on the dead stream, so the + // bridge must find its own way back without the client intervening. + await waitUntil(() => relayer.accepted() >= 2, 20_000); + + bridge.send({ + jsonrpc: "2.0", + id: 3, + method: "tools/call", + params: { name: "memwal_recall", arguments: { query: "anything", limit: 5 } }, + }); + const reply = await bridge.waitFor((m) => m.id === 3 && (m.result || m.error), 20000); + assert.equal(reply.error, undefined, `recovered recall must not error: ${JSON.stringify(reply)}`); + assert.notEqual( + reply.result?.isError, + true, + `recovered recall must not still be refused: ${JSON.stringify(reply)}`, + ); + assert.match(textOf(reply), /no matching memories/i); +}); diff --git a/packages/mcp/test/health-relayer-annotation.test.mjs b/packages/mcp/test/health-relayer-annotation.test.mjs new file mode 100644 index 000000000..9d8e7fc81 --- /dev/null +++ b/packages/mcp/test/health-relayer-annotation.test.mjs @@ -0,0 +1,48 @@ +import assert from "node:assert/strict"; +import test from "node:test"; + +import { annotateHealthResult } from "../dist/bridge.js"; + +// The relayer can only name an origin its deployment published, and says +// nothing on a self-hosted or local one — the sidecar there knows only the +// loopback address it dials. The bridge always knows the URL it connected to, +// which is exactly what `--prod` / `--relayer` / MEMWAL_SERVER_URL selected. + +const DEV = "https://relayer.dev.memwal.ai"; + +const healthResult = (text) => ({ content: [{ type: "text", text }] }); + +test("names the dialled relayer when the reply carries none", () => { + const result = healthResult("Walrus Memory is reachable. status=ok version=1.2.3"); + annotateHealthResult(result, DEV); + assert.ok(result.content[0].text.includes(`relayer=${DEV}`)); + // Existing fields must survive. + assert.ok(result.content[0].text.includes("status=ok")); + assert.ok(result.content[0].text.includes("version=1.2.3")); +}); + +test("replaces the relayer the reply already carried rather than adding a second", () => { + const result = healthResult( + "Walrus Memory is reachable. status=ok version=1.2.3 relayer=https://stale.example write_ready=true", + ); + annotateHealthResult(result, DEV); + const text = result.content[0].text; + assert.equal(text.match(/relayer=/g).length, 1, `two relayer fields:\n${text}`); + assert.ok(text.includes(`relayer=${DEV}`)); + assert.ok(!text.includes("stale.example")); + // The field after it must not be eaten by the replacement. + assert.ok(text.includes("write_ready=true")); +}); + +test("leaves a failed health call alone", () => { + // Naming a relayer beside an error reads as though that relayer answered. + const result = { ...healthResult("relayer unreachable"), isError: true }; + annotateHealthResult(result, DEV); + assert.equal(result.content[0].text, "relayer unreachable"); +}); + +test("tolerates a result shape it does not recognise", () => { + for (const result of [{}, { content: [] }, { content: "nope" }, { content: [{ type: "image" }] }]) { + assert.doesNotThrow(() => annotateHealthResult(result, DEV)); + } +}); diff --git a/packages/mcp/test/live/teo-report.live.mjs b/packages/mcp/test/live/teo-report.live.mjs new file mode 100644 index 000000000..f94b205a8 --- /dev/null +++ b/packages/mcp/test/live/teo-report.live.mjs @@ -0,0 +1,278 @@ +/** + * LIVE acceptance run for the 2026-09-14 field report, case by case. + * + * Opt-in, never run by `npm test`: the filename ends in `.live.mjs` so the + * `test/**\/*.test.mjs` glob skips it, and it exits early without credentials. + * + * Why this exists: the hermetic suite proves the bridge's intent against a mock + * relayer. It cannot tell you whether a deployment actually answers, and the + * report is entirely about a deployment that does not. This script runs the + * reported sequence against a real relayer and prints a scorecard, so "is it + * fixed" stops being a matter of opinion. + * + * Reads are always run. Writes cost a real Walrus blob, so they are opt-in: + * + * # read-only (safe anywhere, including production) + * MEMWAL_LIVE_HOME=/path/to/home \ + * MEMWAL_LIVE_RELAYER=https://relayer.dev.memwal.ai \ + * node test/live/teo-report.live.mjs + * + * # full run, including the two write cases — testnet deployments only + * MEMWAL_LIVE_WRITE=1 MEMWAL_LIVE_HOME=... MEMWAL_LIVE_RELAYER=... \ + * node test/live/teo-report.live.mjs + * + * `MEMWAL_LIVE_HOME` must contain `.memwal/credentials.json`. Nothing is + * deleted or overwritten — the credentials are read, never rewritten. + * + * Exit code is 0 only if every case that ran met its threshold. + */ +import { spawn } from "node:child_process"; +import { existsSync, readFileSync } from "node:fs"; +import { dirname, join, resolve } from "node:path"; +import { fileURLToPath } from "node:url"; + +const __dirname = dirname(fileURLToPath(import.meta.url)); +const BIN = resolve(__dirname, "../../dist/bin/memwal-mcp.js"); + +const HOME_DIR = process.env.MEMWAL_LIVE_HOME; +const RELAYER = process.env.MEMWAL_LIVE_RELAYER; +const RUN_WRITES = process.env.MEMWAL_LIVE_WRITE === "1"; +const NAMESPACE = process.env.MEMWAL_LIVE_NAMESPACE ?? "memwal-live-acceptance"; + +if (!HOME_DIR || !RELAYER) { + console.log("skip: set MEMWAL_LIVE_HOME and MEMWAL_LIVE_RELAYER to run this"); + process.exit(0); +} +const CREDS = join(HOME_DIR, ".memwal", "credentials.json"); +if (!existsSync(CREDS)) { + console.error(`fatal: no credentials at ${CREDS}`); + process.exit(1); +} +if (!existsSync(BIN)) { + console.error(`fatal: no build at ${BIN} — run \`npm run build\` in packages/mcp first`); + process.exit(1); +} + +/** + * Thresholds. These are the report's own expectations, not aspirations: + * "a single-fact save returns in a few seconds", and health is documented as + * the lightweight check. A case that exceeds its budget is a FAIL even when + * it eventually returns — "slow" is the complaint. + */ +const BUDGET_MS = { + connect: 30_000, + health: 5_000, + recall: 15_000, + remember: 15_000, + rememberBulk: 60_000, +}; + +const t0 = Date.now(); +const rel = () => `${((Date.now() - t0) / 1000).toFixed(1).padStart(6)}s`; +const child = spawn(process.execPath, [BIN], { + stdio: ["pipe", "pipe", "pipe"], + env: { + ...process.env, + HOME: HOME_DIR, + USERPROFILE: HOME_DIR, + MEMWAL_CREDS_DIR: join(HOME_DIR, ".memwal"), + MEMWAL_SERVER_URL: RELAYER, + }, +}); + +const stderrLines = []; +let connectedAtMs = null; +child.stderr.on("data", (d) => { + for (const line of d.toString().split("\n")) { + if (!line.trim()) continue; + stderrLines.push(line); + if (line.includes("Connected. Bridging") && connectedAtMs === null) { + connectedAtMs = Date.now() - t0; + } + } +}); + +let buf = ""; +const pending = new Map(); +child.stdout.on("data", (d) => { + buf += d.toString(); + let i; + while ((i = buf.indexOf("\n")) >= 0) { + const line = buf.slice(0, i).trim(); + buf = buf.slice(i + 1); + if (!line) continue; + let msg; + try { + msg = JSON.parse(line); + } catch { + continue; + } + const resolveFn = msg.id !== undefined ? pending.get(msg.id) : undefined; + if (resolveFn) { + pending.delete(msg.id); + resolveFn(msg); + } + } +}); + +let nextId = 1; +/** Send one JSON-RPC request. `capMs` bounds THIS script, not the bridge — the + * bridge's own 240s deadline is part of what is under test, so the cap sits + * above it. */ +function send(method, params, capMs = 300_000) { + const id = nextId++; + const startedAt = Date.now(); + return new Promise((resolveP) => { + const timer = setTimeout(() => { + pending.delete(id); + resolveP({ __noReply: true, __ms: Date.now() - startedAt }); + }, capMs); + pending.set(id, (m) => { + clearTimeout(timer); + resolveP({ ...m, __ms: Date.now() - startedAt }); + }); + child.stdin.write(JSON.stringify({ jsonrpc: "2.0", id, method, params }) + "\n"); + }); +} + +const textOf = (r) => (r.result?.content ?? []).map((c) => c.text).join("\n"); + +const results = []; +function record(id, title, ok, detail, ms) { + results.push({ id, title, ok, detail, ms }); + const mark = ok === null ? "SKIP" : ok ? "PASS" : "FAIL"; + const took = ms === null ? "" : ` ${(ms / 1000).toFixed(1)}s`; + console.log(`${rel()} [${mark}] ${id} ${title}${took}`); + if (detail) console.log(` ${detail.replace(/\n/g, "\n ")}`); +} + +// ~3.5 KB over 5 facts, the reported payload shape. +const FACTS = Array.from({ length: 5 }, (_, i) => + `Acceptance fact ${i + 1} of 5 for the MemWal field report, namespace ${NAMESPACE}. ` + + "Padding so the batch matches the reported payload size and exercises the same path: " + + "lorem ipsum dolor sit amet consectetur adipiscing elit sed do eiusmod tempor ".repeat(8) +); + +(async () => { + console.log(`relayer: ${RELAYER}`); + console.log(`namespace: ${NAMESPACE}`); + console.log(`writes: ${RUN_WRITES ? "ENABLED (will mint real blobs)" : "skipped (set MEMWAL_LIVE_WRITE=1)"}`); + console.log(`payload: ${FACTS.length} facts, ${Buffer.byteLength(FACTS.join(""))} bytes\n`); + + await send("initialize", { + protocolVersion: "2024-11-05", + capabilities: {}, + clientInfo: { name: "memwal-live-acceptance", version: "1.0.0" }, + }); + child.stdin.write(JSON.stringify({ jsonrpc: "2.0", method: "notifications/initialized" }) + "\n"); + + // T1 — §2. The session must reach a live relayer session, and a throttle + // must be named as one rather than looking like broken credentials. + const tools = await send("tools/list", {}); + const toolNames = (tools.result?.tools ?? []).map((t) => t.name); + const throttled = stderrLines.filter((l) => l.includes("rate-limiting new MCP sessions")); + await new Promise((r) => setTimeout(r, 2000)); + const connectOk = connectedAtMs !== null && connectedAtMs <= BUDGET_MS.connect; + record( + "T1", + "§2 handshake reaches a session (and a 429 is named as a throttle)", + connectOk, + connectedAtMs === null + ? `never connected; last stderr: ${stderrLines.slice(-1)[0] ?? "(none)"}` + : `connected in ${(connectedAtMs / 1000).toFixed(1)}s, ${toolNames.length} tools` + + (throttled.length ? `, ${throttled.length} throttle notice(s) — cap was hit but explained` : ""), + connectedAtMs + ); + + // T2 — the canary. Health is an unsigned GET with no SEAL preamble, so it + // is the one call that still answers on a deployment whose signed path is + // broken. Health passing while T3 fails is exactly that shape. + const health = await send("tools/call", { name: "memwal_health", arguments: {} }); + record( + "T2", + "memwal_health answers promptly", + !health.result?.isError && health.__ms <= BUDGET_MS.health, + textOf(health).slice(0, 200), + health.__ms + ); + + // T3 — §1/§4. The first signed round trip. On a deployment where the + // sidecar cannot reach its own relayer this is where the 504 appears. + const recall = await send("tools/call", { + name: "memwal_recall", + arguments: { query: "acceptance fact", limit: 3, namespace: NAMESPACE }, + }); + record( + "T3", + "memwal_recall completes over the signed path", + !recall.result?.isError && recall.__ms <= BUDGET_MS.recall, + textOf(recall).slice(0, 200), + recall.__ms + ); + + if (!RUN_WRITES) { + record("T4", "§1a memwal_remember latency", null, "writes disabled", null); + record("T5", "§1b memwal_remember_bulk returns a result", null, "writes disabled", null); + record("T6", "§1b what actually landed", null, "writes disabled", null); + } else { + // T4 — §1a. The report measures 19s-1m1s and expects "a few seconds". + const single = await send("tools/call", { + name: "memwal_remember", + arguments: { text: `Acceptance single-fact write, namespace ${NAMESPACE}.`, namespace: NAMESPACE }, + }); + record( + "T4", + "§1a memwal_remember returns within budget", + !single.result?.isError && single.__ms <= BUDGET_MS.remember, + textOf(single).slice(0, 200), + single.__ms + ); + + // T5 — §1b. The reported failure: no reply at all, expiring at the + // bridge's 240s deadline. Any answer inside the cap beats that; an + // answer inside budget is the actual goal. + const bulk = await send("tools/call", { + name: "memwal_remember_bulk", + arguments: { facts: FACTS, namespace: NAMESPACE }, + }); + const bulkText = textOf(bulk); + record( + "T5", + "§1b memwal_remember_bulk returns a result within budget", + !bulk.__noReply && !bulk.result?.isError && bulk.__ms <= BUDGET_MS.rememberBulk, + bulk.__noReply ? "no reply at all" : bulkText.slice(0, 400), + bulk.__ms + ); + + // T6 — the question the report asks and no log could answer: when a + // bulk reports failure, was anything actually stored? Recall is the + // only honest way to find out, and the answer decides whether a retry + // would have duplicated paid blobs. + await new Promise((r) => setTimeout(r, 5000)); + const verify = await send("tools/call", { + name: "memwal_recall", + arguments: { query: "Acceptance fact for the MemWal field report", limit: 10, namespace: NAMESPACE }, + }); + const verifyText = textOf(verify); + const landed = (verifyText.match(/\[score=/g) ?? []).length; + record( + "T6", + "§1b recall agrees with what the bulk reported", + !verify.result?.isError, + `recall sees ${landed} of ${FACTS.length} acceptance facts. ` + + (landed > 0 && /failed=\d/.test(bulkText) + ? "NOTE: the bulk reported failures while the facts are present — a retry would have duplicated paid blobs." + : ""), + verify.__ms + ); + } + + const ran = results.filter((r) => r.ok !== null); + const failed = ran.filter((r) => !r.ok); + console.log(`\n${"=".repeat(66)}`); + console.log(`${ran.length - failed.length}/${ran.length} passed` + (failed.length ? ` — FAILED: ${failed.map((f) => f.id).join(", ")}` : "")); + console.log(`credentials: ${JSON.parse(readFileSync(CREDS, "utf8")).accountId?.slice(0, 12)}…`); + + child.kill("SIGTERM"); + setTimeout(() => process.exit(failed.length ? 1 : 0), 500); +})(); diff --git a/packages/mcp/test/login-handoff-stdin.test.mjs b/packages/mcp/test/login-handoff-stdin.test.mjs new file mode 100644 index 000000000..9a1367e70 --- /dev/null +++ b/packages/mcp/test/login-handoff-stdin.test.mjs @@ -0,0 +1,244 @@ +/** + * After an in-session `memwal_login`, the bridge must keep serving stdin. + * + * The auth-required stub hands off by detaching its own listeners and calling + * `process.stdin.pause()`. An explicitly paused stream does NOT resume just + * because a new `data` listener is attached, so the bridge's reader has to ask + * for it. Without that, the ONLY request served after signing in was the one + * replayed from `pendingLines` — every later call was read by nobody and hung + * until the client timed it out. + * + * That is WALM-394's "the next call timed out": the sign-in genuinely worked, + * and the connection was deaf from the second call onward. + */ +import { test } from "node:test"; +import assert from "node:assert/strict"; +import http from "node:http"; +import { spawn } from "node:child_process"; +import { mkdtempSync, rmSync } from "node:fs"; +import { tmpdir } from "node:os"; +import { join, dirname, resolve } from "node:path"; +import { fileURLToPath } from "node:url"; + +const __dirname = dirname(fileURLToPath(import.meta.url)); +const BIN = resolve(__dirname, "../dist/bin/memwal-mcp.js"); + +/** + * Version probe + SSE + a relayer that actually ANSWERS forwarded calls. + * + * The failure-path fixtures never need a reply (nothing gets that far), but + * the banner rides on a real `tools/call` result, so this one has to complete + * the round-trip: read the POSTed request, push a matching JSON-RPC result + * back down the SSE stream. + */ +function startAnsweringRelayer() { + let sseRes = null; + const server = http.createServer((req, res) => { + const url = new URL(req.url, "http://127.0.0.1"); + + if (req.method === "GET" && url.pathname === "/version") { + res.writeHead(200, { "content-type": "application/json" }); + res.end( + JSON.stringify({ + apiVersion: "1.0.0", + relayerVersion: "1.0.0", + minSupportedSdk: { mcp: "0.0.1" }, + }), + ); + return; + } + + if (req.method === "GET" && url.pathname === "/api/mcp/sse") { + res.writeHead(200, { + "content-type": "text/event-stream", + "cache-control": "no-cache", + connection: "keep-alive", + }); + res.write("event: endpoint\ndata: /api/mcp/messages?sessionId=test\n\n"); + sseRes = res; + return; + } + + if (req.method === "POST" && url.pathname === "/api/mcp/messages") { + let body = ""; + req.on("data", (d) => { + body += d; + }); + req.on("end", () => { + res.writeHead(202); + res.end(); + + let msg; + try { + msg = JSON.parse(body); + } catch { + return; + } + if (msg.id === undefined || msg.id === null) return; + + // Distinguishable payload so the assertion proves the banner + // was prefixed onto a REAL upstream result, not substituted + // for one. + const result = + msg.method === "initialize" + ? { protocolVersion: "2024-11-05", capabilities: {}, serverInfo: { name: "mock", version: "1.0.0" } } + : { content: [{ type: "text", text: "UPSTREAM_RECALL_RESULT" }], isError: false }; + + sseRes?.write( + `event: message\ndata: ${JSON.stringify({ jsonrpc: "2.0", id: msg.id, result })}\n\n`, + ); + }); + return; + } + + res.writeHead(404); + res.end(); + }); + + return new Promise((res) => { + server.listen(0, "127.0.0.1", () => { + res({ + server, + base: `http://127.0.0.1:${server.address().port}`, + closeSse: () => sseRes?.end(), + }); + }); + }); +} + +function attachStdio(child) { + const received = []; + const listeners = new Set(); + let buf = ""; + child.stdout.on("data", (d) => { + buf += d.toString(); + let nl; + while ((nl = buf.indexOf("\n")) >= 0) { + const line = buf.slice(0, nl); + buf = buf.slice(nl + 1); + if (!line.trim()) continue; + let msg; + try { + msg = JSON.parse(line); + } catch { + continue; + } + received.push(msg); + for (const l of [...listeners]) l(msg); + } + }); + const send = (obj) => child.stdin.write(JSON.stringify(obj) + "\n"); + const waitFor = (pred, ms = 15000) => { + const hit = received.find(pred); + if (hit) return Promise.resolve(hit); + return new Promise((res, rej) => { + const timer = setTimeout(() => { + listeners.delete(l); + const seen = received + .map((m) => (m.id !== undefined ? `id=${m.id}` : m.method)) + .join(", "); + rej(new Error(`timed out waiting for message; received: [${seen}]`)); + }, ms); + const l = (m) => { + if (pred(m)) { + clearTimeout(timer); + listeners.delete(l); + res(m); + } + }; + listeners.add(l); + }); + }; + return { send, waitFor }; +} + +/** + * Drive the browser half of the sign-in: the same preflight-then-callback + * handshake a real wallet approval performs (pinned by login-preflight), so + * no browser or on-chain transaction is involved. + */ +async function completeSignIn(loginText, webUrl) { + const match = loginText.match(/http:\/\/127\.0\.0\.1:\d+\/connect\/mcp\?\S+/); + assert.ok(match, `login result should carry a connect URL, got: ${loginText.slice(0, 300)}`); + const connectUrl = new URL(match[0].replace(/[)`\s]+$/, "")); + + const port = connectUrl.searchParams.get("port"); + const publicKey = connectUrl.searchParams.get("publicKey"); + const state = connectUrl.searchParams.get("connectState"); + assert.match(port ?? "", /^\d+$/); + + const post = (path, body) => + fetch(`http://127.0.0.1:${port}${path}`, { + method: "POST", + headers: { "content-type": "application/json", origin: webUrl }, + body: JSON.stringify(body), + }); + + const preflight = await post("/preflight", { state, publicKey, relayer: webUrl }); + assert.equal(preflight.status, 200, "preflight should be accepted"); + + const callback = await post("/callback", { + state, + accountId: `0x${"1".repeat(64)}`, + walletAddress: `0x${"2".repeat(64)}`, + packageId: `0x${"3".repeat(64)}`, + label: "Test MCP", + }); + assert.equal(callback.status, 200, "callback should be accepted"); +} + +function spawnSignedOut(base, credsDir) { + return spawn(process.execPath, [BIN, "--relayer", base, "--web-url", base], { + env: { + ...process.env, + // MEMWAL_CREDS_DIR rather than HOME alone: os.homedir() ignores + // HOME on Windows, which once let this suite overwrite a real + // credentials.json (see CHANGELOG #705). + MEMWAL_CREDS_DIR: credsDir, + HOME: credsDir, + USERPROFILE: credsDir, + MEMWAL_MCP_LOGIN_TIMEOUT_MS: "15000", + }, + stdio: ["pipe", "pipe", "pipe"], + }); +} + +test("the bridge keeps reading stdin after an in-session sign-in", async (t) => { + const { server, base, closeSse } = await startAnsweringRelayer(); + const credsDir = mkdtempSync(join(tmpdir(), "memwal-handoff-stdin-")); + const child = spawnSignedOut(base, credsDir); + const { send, waitFor } = attachStdio(child); + + t.after(() => { + child.kill("SIGKILL"); + closeSse(); + server.close(); + rmSync(credsDir, { recursive: true, force: true }); + }); + + send({ jsonrpc: "2.0", id: 1, method: "initialize", params: { protocolVersion: "2024-11-05" } }); + await waitFor((m) => m.id === 1 && m.result); + + send({ jsonrpc: "2.0", id: 2, method: "tools/call", params: { name: "memwal_login" } }); + const login = await waitFor((m) => m.id === 2 && m.result); + await completeSignIn(login.result.content[0].text, base); + + // Served from `pendingLines` — this one worked even with stdin paused. + send({ + jsonrpc: "2.0", + id: 3, + method: "tools/call", + params: { name: "memwal_recall", arguments: { query: "replayed" } }, + }); + await waitFor((m) => m.id === 3 && m.result); + + // Read from the live stream. This is the one that used to hang forever. + send({ + jsonrpc: "2.0", + id: 4, + method: "tools/call", + params: { name: "memwal_recall", arguments: { query: "live" } }, + }); + const live = await waitFor((m) => m.id === 4 && m.result); + assert.match(live.result.content[0].text, /UPSTREAM_RECALL_RESULT/); +}); diff --git a/packages/mcp/test/login-prompt-unified.test.mjs b/packages/mcp/test/login-prompt-unified.test.mjs new file mode 100644 index 000000000..8b363a229 --- /dev/null +++ b/packages/mcp/test/login-prompt-unified.test.mjs @@ -0,0 +1,79 @@ +/** + * One sign-in prompt, whichever mode the user is in. + * + * `memwal_login` is answered locally in two places — the auth-required stub + * when signed out, and the bridge when already signed in — and the two copies + * had drifted: different assistant instructions, different step wording, + * different closing line. Same tool, same user, two voices. + * + * They differ legitimately in exactly one respect: signing in while already + * signed in REPLACES the stored delegate key, and that is worth saying. This + * pins everything else as shared, so the next edit to one cannot silently + * fork the other again. + */ +import { test } from "node:test"; +import assert from "node:assert/strict"; + +const { loginPrompt } = await import("../dist/messages.js"); + +const URL_ = "https://memory.example/connect/mcp?port=1&publicKey=ab&connectState=cd"; +const CREDS = "/tmp/sandbox/.memwal/credentials.json"; + +const signedOut = loginPrompt({ url: URL_, credentialsPath: CREDS, signedIn: false }); +const signedIn = loginPrompt({ url: URL_, credentialsPath: CREDS, signedIn: true }); + +test("both modes repeat the URL in all three forms", () => { + // Deliberate armor against clients that paraphrase tool output: plain, + // code-block, and link form, so at least one survives. Neither mode may + // quietly drop it. + for (const [name, text] of [["signed out", signedOut], ["signed in", signedIn]]) { + assert.ok(text.includes(`**URL:** ${URL_}`), `${name}: plain URL`); + assert.ok(text.includes(`\`\`\`\n${URL_}\n\`\`\``), `${name}: code-block URL`); + assert.ok(text.includes(`](${URL_})`), `${name}: markdown link URL`); + } +}); + +test("both modes give the assistant the same instruction and closing line", () => { + const instruction = "**IMPORTANT for the assistant**"; + const closing = "_The login link stays valid for 5 minutes"; + + const lineWith = (text, needle) => + text.split("\n").filter((l) => l.includes(needle)).join("\n"); + + assert.equal(lineWith(signedOut, instruction), lineWith(signedIn, instruction)); + assert.equal(lineWith(signedOut, closing), lineWith(signedIn, closing)); +}); + +test("both modes name the resolved credentials path, never a hardcoded home", () => { + // Which file a sign-in lands in is exactly the confusion behind GH #628, + // so neither mode may claim `~/.memwal/credentials.json` when the real + // path is elsewhere. + for (const [name, text] of [["signed out", signedOut], ["signed in", signedIn]]) { + assert.ok(text.includes(CREDS), `${name}: should state the resolved path`); + assert.ok( + !text.includes("~/.memwal/credentials.json"), + `${name}: should not hardcode the global path`, + ); + } +}); + +test("only the signed-in prompt warns that the stored key is replaced", () => { + assert.match(signedIn, /replac/i, "re-signing in overwrites the delegate key — say so"); + assert.doesNotMatch( + signedOut, + /replac/i, + "a first sign-in replaces nothing; the warning would be a lie", + ); +}); + +test("the two prompts differ ONLY in that warning", () => { + // Drop the warning, then collapse the blank line that separated it — the + // claim under test is that no other CONTENT differs. + const strip = (text) => + text + .split("\n") + .filter((l) => !/replac/i.test(l)) + .join("\n") + .replace(/\n{3,}/g, "\n\n"); + assert.equal(strip(signedOut), strip(signedIn)); +}); diff --git a/packages/mcp/test/login-success-notice.test.mjs b/packages/mcp/test/login-success-notice.test.mjs new file mode 100644 index 000000000..526d2ec8b --- /dev/null +++ b/packages/mcp/test/login-success-notice.test.mjs @@ -0,0 +1,402 @@ +/** + * A sign-in that DOES complete must say so. + * + * The failure path already reports itself twice — a `notifications/message` + * and a notice prefixed onto the next tool call (see login-failure-notice). + * Success reported nothing at all: it only wrote to the log file, so the user + * who approved in the browser had no way to tell the credentials landed, the + * bridge adopted them, or the retry would work. This pins the confirmation. + * + * The banner is a ONE-SHOT. Success is an event, not a state, so unlike the + * failure notice it is consumed by the call that shows it and never repeats. + */ +import { test } from "node:test"; +import assert from "node:assert/strict"; +import http from "node:http"; +import { spawn } from "node:child_process"; +import { mkdtempSync, mkdirSync, writeFileSync, rmSync } from "node:fs"; +import { tmpdir } from "node:os"; +import { join, dirname, resolve } from "node:path"; +import { fileURLToPath } from "node:url"; + +const __dirname = dirname(fileURLToPath(import.meta.url)); +const BIN = resolve(__dirname, "../dist/bin/memwal-mcp.js"); + +/** + * Version probe + SSE + a relayer that actually ANSWERS forwarded calls. + * + * The failure-path fixtures never need a reply (nothing gets that far), but + * the banner rides on a real `tools/call` result, so this one has to complete + * the round-trip: read the POSTed request, push a matching JSON-RPC result + * back down the SSE stream. + */ +function startAnsweringRelayer() { + let sseRes = null; + const server = http.createServer((req, res) => { + const url = new URL(req.url, "http://127.0.0.1"); + + if (req.method === "GET" && url.pathname === "/version") { + res.writeHead(200, { "content-type": "application/json" }); + res.end( + JSON.stringify({ + apiVersion: "1.0.0", + relayerVersion: "1.0.0", + minSupportedSdk: { mcp: "0.0.1" }, + }), + ); + return; + } + + if (req.method === "GET" && url.pathname === "/api/mcp/sse") { + res.writeHead(200, { + "content-type": "text/event-stream", + "cache-control": "no-cache", + connection: "keep-alive", + }); + res.write("event: endpoint\ndata: /api/mcp/messages?sessionId=test\n\n"); + sseRes = res; + return; + } + + if (req.method === "POST" && url.pathname === "/api/mcp/messages") { + let body = ""; + req.on("data", (d) => { + body += d; + }); + req.on("end", () => { + res.writeHead(202); + res.end(); + + let msg; + try { + msg = JSON.parse(body); + } catch { + return; + } + if (msg.id === undefined || msg.id === null) return; + + // Distinguishable payload so the assertion proves the banner + // was prefixed onto a REAL upstream result, not substituted + // for one. + const result = + msg.method === "initialize" + ? { protocolVersion: "2024-11-05", capabilities: {}, serverInfo: { name: "mock", version: "1.0.0" } } + : { content: [{ type: "text", text: "UPSTREAM_RECALL_RESULT" }], isError: false }; + + sseRes?.write( + `event: message\ndata: ${JSON.stringify({ jsonrpc: "2.0", id: msg.id, result })}\n\n`, + ); + }); + return; + } + + res.writeHead(404); + res.end(); + }); + + return new Promise((res) => { + server.listen(0, "127.0.0.1", () => { + res({ + server, + base: `http://127.0.0.1:${server.address().port}`, + closeSse: () => sseRes?.end(), + }); + }); + }); +} + +function attachStdio(child) { + const received = []; + const listeners = new Set(); + let buf = ""; + child.stdout.on("data", (d) => { + buf += d.toString(); + let nl; + while ((nl = buf.indexOf("\n")) >= 0) { + const line = buf.slice(0, nl); + buf = buf.slice(nl + 1); + if (!line.trim()) continue; + let msg; + try { + msg = JSON.parse(line); + } catch { + continue; + } + received.push(msg); + for (const l of [...listeners]) l(msg); + } + }); + const send = (obj) => child.stdin.write(JSON.stringify(obj) + "\n"); + const waitFor = (pred, ms = 15000) => { + const hit = received.find(pred); + if (hit) return Promise.resolve(hit); + return new Promise((res, rej) => { + const timer = setTimeout(() => { + listeners.delete(l); + const seen = received + .map((m) => (m.id !== undefined ? `id=${m.id}` : m.method)) + .join(", "); + rej(new Error(`timed out waiting for message; received: [${seen}]`)); + }, ms); + const l = (m) => { + if (pred(m)) { + clearTimeout(timer); + listeners.delete(l); + res(m); + } + }; + listeners.add(l); + }); + }; + return { send, waitFor }; +} + +/** + * Drive the browser half of the sign-in: the same preflight-then-callback + * handshake a real wallet approval performs (pinned by login-preflight), so + * no browser or on-chain transaction is involved. + */ +async function completeSignIn(loginText, webUrl) { + const match = loginText.match(/http:\/\/127\.0\.0\.1:\d+\/connect\/mcp\?\S+/); + assert.ok(match, `login result should carry a connect URL, got: ${loginText.slice(0, 300)}`); + const connectUrl = new URL(match[0].replace(/[)`\s]+$/, "")); + + const port = connectUrl.searchParams.get("port"); + const publicKey = connectUrl.searchParams.get("publicKey"); + const state = connectUrl.searchParams.get("connectState"); + assert.match(port ?? "", /^\d+$/); + + const post = (path, body) => + fetch(`http://127.0.0.1:${port}${path}`, { + method: "POST", + headers: { "content-type": "application/json", origin: webUrl }, + body: JSON.stringify(body), + }); + + const preflight = await post("/preflight", { state, publicKey, relayer: webUrl }); + assert.equal(preflight.status, 200, "preflight should be accepted"); + + const callback = await post("/callback", { + state, + accountId: `0x${"1".repeat(64)}`, + walletAddress: `0x${"2".repeat(64)}`, + packageId: `0x${"3".repeat(64)}`, + label: "Test MCP", + }); + assert.equal(callback.status, 200, "callback should be accepted"); +} + +function spawnSignedOut(base, credsDir) { + return spawn(process.execPath, [BIN, "--relayer", base, "--web-url", base], { + env: { + ...process.env, + // MEMWAL_CREDS_DIR rather than HOME alone: os.homedir() ignores + // HOME on Windows, which once let this suite overwrite a real + // credentials.json (see CHANGELOG #705). + MEMWAL_CREDS_DIR: credsDir, + HOME: credsDir, + USERPROFILE: credsDir, + MEMWAL_MCP_LOGIN_TIMEOUT_MS: "15000", + }, + stdio: ["pipe", "pipe", "pipe"], + }); +} + +/** Credentials on disk, so the real bridge runs instead of the auth-required + * stub. Same account the callback below reports, keeping this a plain key + * rotation rather than an account switch. */ +function seedCreds(credsDir, relayerUrl) { + // Flat, not `.memwal/` — `spawnSignedOut` sets MEMWAL_CREDS_DIR, which the + // CLI uses as the credentials directory itself. + const path = join(credsDir, "credentials.json"); + mkdirSync(credsDir, { recursive: true }); + writeFileSync( + path, + JSON.stringify({ + delegatePrivateKey: "a".repeat(64), + delegatePublicKeyHex: "b".repeat(64), + delegateAddress: `0x${"4".repeat(64)}`, + walletAddress: `0x${"2".repeat(64)}`, + accountId: `0x${"1".repeat(64)}`, + packageId: `0x${"3".repeat(64)}`, + relayerUrl, + label: "Existing MCP", + createdAt: new Date(0).toISOString(), + version: 1, + }), + { mode: 0o600 }, + ); +} + +/** + * The re-login path: already signed in, so `memwal_login` is answered by the + * bridge's `handleLocalLogin` and the callback lands in `adoptCredentials` — + * a different pair of surfaces from the signed-out hand-off the tests above + * drive. A regression that dropped either one would pass every one of them. + */ +test("re-signing in while already signed in is confirmed on both surfaces", async (t) => { + const { server, base, closeSse } = await startAnsweringRelayer(); + const credsDir = mkdtempSync(join(tmpdir(), "memwal-success-relogin-")); + seedCreds(credsDir, base); + const child = spawnSignedOut(base, credsDir); + const { send, waitFor } = attachStdio(child); + + t.after(() => { + child.kill("SIGKILL"); + closeSse(); + server.close(); + rmSync(credsDir, { recursive: true, force: true }); + }); + + send({ jsonrpc: "2.0", id: 1, method: "initialize", params: { protocolVersion: "2024-11-05" } }); + await waitFor((m) => m.id === 1 && m.result); + + send({ jsonrpc: "2.0", id: 2, method: "tools/call", params: { name: "memwal_login" } }); + const login = await waitFor((m) => m.id === 2 && m.result); + assert.equal(login.result.isError, false); + + // Credentials really are on disk here, so this is the one case where the + // replacement warning is true and must appear. + assert.match( + login.result.content[0].text, + /already signed in/i, + "a stored key IS about to be replaced; the prompt has to say so", + ); + + await completeSignIn(login.result.content[0].text, base); + + const announced = await waitFor( + (m) => + m.method === "notifications/message" && + String(m.params?.data).includes("sign-in complete"), + ); + assert.match(String(announced.params.data), /0x1{4}/, "should name the account signed in as"); + + send({ + jsonrpc: "2.0", + id: 3, + method: "tools/call", + params: { name: "memwal_recall", arguments: { query: "after re-login" } }, + }); + const after = await waitFor((m) => m.id === 3 && m.result); + const text = after.result.content[0].text; + assert.match(text, /Signed in to Walrus Memory/, "the re-login should carry the banner too"); + assert.match(text, /UPSTREAM_RECALL_RESULT/, "prefixed onto the real result, not instead of it"); + + send({ + jsonrpc: "2.0", + id: 4, + method: "tools/call", + params: { name: "memwal_recall", arguments: { query: "one banner only" } }, + }); + const second = await waitFor((m) => m.id === 4 && m.result); + assert.doesNotMatch( + second.result.content[0].text, + /Signed in to Walrus Memory/, + "the banner is a one-shot on the re-login path as well", + ); +}); + +test("a completed sign-in is confirmed on the next tool call", async (t) => { + const { server, base, closeSse } = await startAnsweringRelayer(); + const credsDir = mkdtempSync(join(tmpdir(), "memwal-success-")); + const child = spawnSignedOut(base, credsDir); + const { send, waitFor } = attachStdio(child); + + t.after(() => { + child.kill("SIGKILL"); + closeSse(); + server.close(); + rmSync(credsDir, { recursive: true, force: true }); + }); + + send({ jsonrpc: "2.0", id: 1, method: "initialize", params: { protocolVersion: "2024-11-05" } }); + await waitFor((m) => m.id === 1 && m.result); + + send({ jsonrpc: "2.0", id: 2, method: "tools/call", params: { name: "memwal_login" } }); + const login = await waitFor((m) => m.id === 2 && m.result); + assert.equal(login.result.isError, false); + + await completeSignIn(login.result.content[0].text, base); + + // The callback landed with nobody awaiting it, exactly as a real browser + // approval does. This is the moment that used to be silent. + const announced = await waitFor( + (m) => + m.method === "notifications/message" && + String(m.params?.data).includes("sign-in complete"), + ); + assert.match(String(announced.params.data), /0x1{4}/, "should name the account signed in as"); + + send({ + jsonrpc: "2.0", + id: 3, + method: "tools/call", + params: { name: "memwal_recall", arguments: { query: "anything" } }, + }); + const after = await waitFor((m) => m.id === 3 && m.result); + const text = after.result.content[0].text; + + assert.match(text, /Signed in to Walrus Memory/); + assert.match(text, /0x1{4}/, "banner should name the account"); + assert.match(text, /credentials\.json/, "banner should name where credentials landed"); + assert.match(text, /no client restart needed/i); + // Prefixed onto the real result, never in place of it. + assert.match(text, /UPSTREAM_RECALL_RESULT/); + assert.equal(after.result.isError, false); +}); + +test("the sign-in confirmation is not repeated on later calls", async (t) => { + const { server, base, closeSse } = await startAnsweringRelayer(); + const credsDir = mkdtempSync(join(tmpdir(), "memwal-success-once-")); + const child = spawnSignedOut(base, credsDir); + const { send, waitFor } = attachStdio(child); + + t.after(() => { + child.kill("SIGKILL"); + closeSse(); + server.close(); + rmSync(credsDir, { recursive: true, force: true }); + }); + + send({ jsonrpc: "2.0", id: 1, method: "initialize", params: { protocolVersion: "2024-11-05" } }); + await waitFor((m) => m.id === 1 && m.result); + + send({ jsonrpc: "2.0", id: 2, method: "tools/call", params: { name: "memwal_login" } }); + const login = await waitFor((m) => m.id === 2 && m.result); + await completeSignIn(login.result.content[0].text, base); + await waitFor( + (m) => + m.method === "notifications/message" && + String(m.params?.data).includes("sign-in complete"), + ); + + send({ + jsonrpc: "2.0", + id: 3, + method: "tools/call", + params: { name: "memwal_recall", arguments: { query: "first" } }, + }); + const first = await waitFor((m) => m.id === 3 && m.result); + assert.match( + first.result.content[0].text, + /Signed in to Walrus Memory/, + "precondition: the first call carries the banner", + ); + + send({ + jsonrpc: "2.0", + id: 4, + method: "tools/call", + params: { name: "memwal_recall", arguments: { query: "second" } }, + }); + const second = await waitFor((m) => m.id === 4 && m.result); + const text = second.result.content[0].text; + + assert.doesNotMatch( + text, + /Signed in to Walrus Memory/, + "the banner is consumed by the call that shows it — a signed-in session must not repeat it", + ); + assert.match(text, /UPSTREAM_RECALL_RESULT/, "the real result still comes through"); +}); diff --git a/packages/mcp/test/logout-invalidation.test.mjs b/packages/mcp/test/logout-invalidation.test.mjs index 1a4e55a1b..e8b7020f5 100644 --- a/packages/mcp/test/logout-invalidation.test.mjs +++ b/packages/mcp/test/logout-invalidation.test.mjs @@ -473,12 +473,44 @@ test("signing back in after logout restores memory tools without a client restar params: { name: "memwal_login", arguments: {} }, }); const login = await waitFor((m) => m.id === 5, 10_000); - const connectUrl = login.result?.content?.[0]?.text?.match(/\*\*URL:\*\* (http[^\n]+)/)?.[1]; + const prompt = login.result?.content?.[0]?.text ?? ""; + const connectUrl = prompt.match(/\*\*URL:\*\* (http[^\n]+)/)?.[1]; assert.ok(connectUrl, "memwal_login should return the browser URL"); + // The bridge used to hardcode `signedIn: true` on the assumption that it + // only ever runs with credentials. Logout deletes them, and login is + // intercepted before the signed-out guard, so the prompt claimed the user + // was still signed in and that a stored key would be replaced — which + // reads as though the logout they just performed did not take. + assert.doesNotMatch( + prompt, + /already signed in/i, + "the credentials were just deleted; this prompt must not claim otherwise", + ); + assert.doesNotMatch( + prompt, + /replaces the stored delegate key/i, + "there is no stored key left to replace after logout", + ); + await completeLogin(connectUrl, ACCOUNT_B); await waitUntil(() => mock.getHandshakes().some((h) => h.accountId === ACCOUNT_B)); + // Signing back in is a completed sign-in like any other, so it is announced + // on both surfaces. Without this the notification could be dropped from the + // re-login path and every assertion below would still pass. + const announced = await waitFor( + (m) => + m.method === "notifications/message" && + String(m.params?.data).includes("sign-in complete"), + 10_000, + ); + // Shortened for readability by `shortId`, so match the head, not the whole id. + assert.ok( + String(announced.params.data).includes(ACCOUNT_B.slice(0, 10)), + `should name the new account, got: ${announced.params.data}`, + ); + // The real assertion: a memory tool works again, end to end, on the new // session. Without a resumable pump this reply never reaches stdout and the // wait below times out. @@ -492,6 +524,26 @@ test("signing back in after logout restores memory tools without a client restar assert.notEqual(after.result?.isError, true, "memory tools should work again after re-login"); assert.match(JSON.stringify(after.result), /RECALL_OK/); assert.equal(mock.getRecallCount(), 2, "the post-login recall should reach the relayer"); + + // Prefixed onto the real result rather than replacing it. This assertion is + // what would catch `adoptCredentials` dropping its banner queue: RECALL_OK + // above passes with or without the confirmation. + const afterText = after.result?.content?.[0]?.text ?? ""; + assert.match(afterText, /Signed in to Walrus Memory/, "the re-login should be confirmed"); + assert.match(afterText, /RECALL_OK/); + + send({ + jsonrpc: "2.0", + id: 7, + method: "tools/call", + params: { name: "memwal_recall", arguments: { query: "one banner only" } }, + }); + const second = await waitFor((m) => m.id === 7, 10_000); + assert.doesNotMatch( + second.result?.content?.[0]?.text ?? "", + /Signed in to Walrus Memory/, + "the banner is a one-shot — it must not repeat on later calls", + ); }); /** diff --git a/packages/mcp/test/orphaned-call.test.mjs b/packages/mcp/test/orphaned-call.test.mjs index 8df522467..79a304496 100644 --- a/packages/mcp/test/orphaned-call.test.mjs +++ b/packages/mcp/test/orphaned-call.test.mjs @@ -184,7 +184,7 @@ function makeCreds(relayerUrl) { }; } -test("a call whose reply never arrives is closed out with a retryable error", async (t) => { +test("a sent write whose reply never arrives is closed out without inviting a duplicate", async (t) => { const mock = await startMockRelayer(); const home = mkdtempSync(join(tmpdir(), "memwal-orphan-test-")); const credsPath = join(home, ".memwal", "credentials.json"); @@ -272,13 +272,33 @@ test("a call whose reply never arrives is closed out with a retryable error", as // Before the fix this never resolved. const orphaned = await waitFor((m) => m.id === 2, 10_000); assert.equal(orphaned.result?.isError, true, "expected a tool-result error envelope"); + const orphanedText = JSON.stringify(orphaned.result); + + // `memwal_remember` is a write, and this one WAS sent — the relayer + // accepted it and answered 202 before finishing in a durable queue, so the + // lost reply says nothing about whether it landed. The message must not + // invite a blind repeat: `/api/remember/bulk` carries no idempotency key, + // so repeating a batch that already landed buys a second paid blob. + assert.match( + orphanedText, + /may have completed|does not cancel it/i, + "a sent write must say it may already have landed", + ); assert.match( - JSON.stringify(orphaned.result), - /retry/i, - "the message should tell the caller it is safe to retry", + orphanedText, + /memwal_recall/, + "a sent write must point at recall as the way to check before re-saving", + ); + // Careful with the negative: the message deliberately says it "does not + // mean nothing was stored", which is the opposite of claiming it. What must + // never appear is the instruction to repeat the call. + assert.doesNotMatch( + orphanedText, + /please retry/i, + "a sent write must not invite a blind retry", ); assert.doesNotMatch( - JSON.stringify(orphaned.result), + orphanedText, /relayer unavailable/i, "the relayer was healthy — saying otherwise sends debugging the wrong way", ); diff --git a/packages/mcp/test/pending-forward-stalled.test.mjs b/packages/mcp/test/pending-forward-stalled.test.mjs new file mode 100644 index 000000000..ca978408d --- /dev/null +++ b/packages/mcp/test/pending-forward-stalled.test.mjs @@ -0,0 +1,605 @@ +/** + * Regression test for WALM-618 — a tool call that expires while still buffered + * must be explained as what it is: a call that never left this process. + * + * Repro (the shape Dio hit: 15 session opens, 6 calls that ever reached the + * relayer, minutes of silence in between): + * - Mock relayer answers GET /version, then 503s every SSE handshake, the + * way the real relayer does when the on-chain delegate verify cannot + * reach a throttled fullnode. + * - A `tools/call` arrives before any session exists, so it lands in + * `pendingForward` and waits there while the bridge retries. + * + * The orphan sweeper did already bound this wait. What it got wrong was the + * answer: every expiry was reported as "the connection to the relayer dropped + * before the result came back", which points the user at the relayer — or at a + * possibly half-written memory — when in fact nothing was ever sent and the + * handshake was the thing failing. + * + * Asserts: + * - `initialize` is still answered locally, exactly once. + * - the buffered call is NOT eager-failed between retries (the property + * `coldstart-timeout.test.mjs` locks down — this stays a deadline, not a + * per-attempt failure). + * - once the deadline passes it is answered as a tool error naming the + * failing connection, saying nothing was stored, and carrying the last + * handshake error. + * - the process stays alive throughout. + */ +import { test } from "node:test"; +import assert from "node:assert/strict"; +import http from "node:http"; +import { spawn } from "node:child_process"; +import { mkdtempSync, mkdirSync, writeFileSync, rmSync } from "node:fs"; +import { tmpdir } from "node:os"; +import { join, dirname, resolve } from "node:path"; +import { fileURLToPath } from "node:url"; + +const __dirname = dirname(fileURLToPath(import.meta.url)); +const BIN = resolve(__dirname, "../dist/bin/memwal-mcp.js"); +const EXPECTED_BEARER = "a".repeat(64); +const EXPECTED_ACCOUNT_ID = "0x" + "3".repeat(64); + +/** The deadline under test: the short one that applies only to a call which + * never left the bridge while no connection has existed. Short enough to run, + * long enough that several connect-retry cycles fit inside it — otherwise + * "not eager-failed between retries" would pass for the wrong reason. */ +const STALLED_HANDSHAKE_MS = 3_000; + +/** Deliberately far larger, so an answer arriving near STALLED_HANDSHAKE_MS + * proves the stalled-handshake deadline fired and not the ordinary call + * timeout, which is what used to leave the user waiting ~4 minutes. */ +const CALL_TIMEOUT_MS = 60_000; +const CONNECT_TIMEOUT_MS = 400; + +function hasBridgeAuth(req) { + return ( + req.headers.authorization === `Bearer ${EXPECTED_BEARER}` && + req.headers["x-memwal-account-id"] === EXPECTED_ACCOUNT_ID + ); +} + +/** Mock relayer that 503s every SSE handshake until `heal()` is called — an + * infrastructure failure, not an auth rejection, which is exactly what a + * throttled fullnode produces via the relayer's `upstream_unavailable()`. + * Once healed it behaves like a normal session, and records every JSON-RPC + * envelope it is posted so a test can prove what did (and did not) reach it. */ +function startUnavailableRelayer() { + let sseGetCount = 0; + let healthy = false; + const sessions = new Map(); + const posted = []; + const server = http.createServer((req, res) => { + const url = new URL(req.url, "http://127.0.0.1"); + if (req.method === "GET" && url.pathname === "/version") { + res.writeHead(200, { "content-type": "application/json" }); + res.end( + JSON.stringify({ + apiVersion: "1.0.0", + relayerVersion: "1.0.0", + minSupportedSdk: { mcp: "0.0.1" }, + }), + ); + return; + } + if (req.method === "GET" && url.pathname === "/api/mcp/sse") { + if (!hasBridgeAuth(req)) { + res.writeHead(401); + res.end(); + return; + } + sseGetCount += 1; + if (!healthy) { + res.writeHead(503, { "content-type": "text/plain" }); + res.end("Account resolution unavailable: on-chain re-verify unavailable"); + return; + } + const sessionId = `session-${sseGetCount}`; + res.writeHead(200, { + "content-type": "text/event-stream", + "cache-control": "no-cache", + connection: "keep-alive", + }); + res.write(`event: endpoint\ndata: /api/mcp/messages?sessionId=${sessionId}\n\n`); + sessions.set(sessionId, { res }); + const hb = setInterval(() => { + if (res.writableEnded) { + clearInterval(hb); + return; + } + res.write(":\n\n"); + }, 200); + hb.unref?.(); + res.on("close", () => clearInterval(hb)); + return; + } + if (req.method === "POST" && url.pathname === "/api/mcp/messages") { + const session = sessions.get(url.searchParams.get("sessionId")); + let body = ""; + req.on("data", (c) => (body += c)); + req.on("end", () => { + if (!session) { + res.writeHead(404); + res.end(); + return; + } + try { + posted.push(JSON.parse(body)); + } catch { + /* not JSON — not something this test asserts on */ + } + res.writeHead(202); + res.end(); + }); + return; + } + res.writeHead(404); + res.end(); + }); + return new Promise((res) => { + server.listen(0, "127.0.0.1", () => { + const { port } = server.address(); + res({ + server, + base: `http://127.0.0.1:${port}`, + getSseGetCount: () => sseGetCount, + getPosted: () => posted, + heal: () => { + healthy = true; + }, + closeStreams: () => + sessions.forEach((s) => { + if (!s.res.writableEnded) s.res.end(); + }), + }); + }); + }); +} + +function makeCreds(relayerUrl) { + return { + delegatePrivateKey: EXPECTED_BEARER, + delegatePublicKeyHex: "b".repeat(64), + delegateAddress: "0x" + "1".repeat(64), + walletAddress: "0x" + "2".repeat(64), + accountId: EXPECTED_ACCOUNT_ID, + packageId: "0x" + "4".repeat(64), + relayerUrl, + label: "Pending Forward Stalled Test", + createdAt: new Date(0).toISOString(), + version: 1, + }; +} + +test("a call buffered behind a failing handshake is answered, and says why", async (t) => { + const mock = await startUnavailableRelayer(); + const home = mkdtempSync(join(tmpdir(), "memwal-pending-stalled-test-")); + const credsPath = join(home, ".memwal", "credentials.json"); + mkdirSync(dirname(credsPath), { recursive: true }); + writeFileSync(credsPath, JSON.stringify(makeCreds(mock.base)), { mode: 0o600 }); + + const child = spawn(process.execPath, [BIN, "--relayer", mock.base, "--web-url", mock.base], { + env: { + ...process.env, + HOME: home, + USERPROFILE: home, + MEMWAL_MCP_CONNECT_TIMEOUT_MS: String(CONNECT_TIMEOUT_MS), + MEMWAL_MCP_CALL_TIMEOUT_MS: String(CALL_TIMEOUT_MS), + MEMWAL_MCP_STALLED_HANDSHAKE_MS: String(STALLED_HANDSHAKE_MS), + }, + stdio: ["pipe", "pipe", "pipe"], + }); + + const received = []; + const listeners = new Set(); + let buf = ""; + child.stdout.on("data", (d) => { + buf += d.toString(); + let nl; + while ((nl = buf.indexOf("\n")) >= 0) { + const line = buf.slice(0, nl); + buf = buf.slice(nl + 1); + if (!line.trim()) continue; + let msg; + try { + msg = JSON.parse(line); + } catch { + continue; + } + received.push(msg); + for (const l of [...listeners]) l(msg); + } + }); + let stderrBuf = ""; + child.stderr.on("data", (d) => (stderrBuf += d.toString())); + + const send = (obj) => child.stdin.write(JSON.stringify(obj) + "\n"); + const waitFor = (pred, ms = 15_000) => { + const hit = received.find(pred); + if (hit) return Promise.resolve(hit); + return new Promise((res, rej) => { + const timer = setTimeout(() => { + listeners.delete(l); + rej( + new Error( + `timed out waiting for message\n--- stderr ---\n${stderrBuf}\n--- received ---\n${received.map((m) => JSON.stringify(m)).join("\n")}`, + ), + ); + }, ms); + const l = (m) => { + if (pred(m)) { + clearTimeout(timer); + listeners.delete(l); + res(m); + } + }; + listeners.add(l); + }); + }; + + t.after(() => { + child.kill("SIGKILL"); + mock.closeStreams(); + mock.server.close(); + rmSync(home, { recursive: true, force: true }); + }); + + send({ jsonrpc: "2.0", id: 1, method: "initialize", params: {} }); + const init = await waitFor((m) => m.id === 1 && m.result, 5_000); + assert.equal(init.result.serverInfo.name, "memwal"); + + const sentAt = Date.now(); + send({ + jsonrpc: "2.0", + id: 2, + method: "tools/call", + params: { name: "memwal_remember", arguments: { text: "anything" } }, + }); + + // Half the deadline in, several connect attempts have already failed and + // the call must still be waiting — the fix adds a deadline, it does not + // eager-fail a call the next attempt might serve. + await new Promise((r) => setTimeout(r, STALLED_HANDSHAKE_MS / 2)); + assert.ok( + mock.getSseGetCount() >= 2, + `expected the handshake to have been retried by now, saw ${mock.getSseGetCount()} attempts`, + ); + assert.ok( + !received.some((m) => m.id === 2), + `id=2 must still be buffered mid-deadline, got: ${JSON.stringify(received.find((m) => m.id === 2))}`, + ); + + // Past the deadline it is answered rather than left hanging forever. + const reply = await waitFor((m) => m.id === 2, 15_000); + const waitedMs = Date.now() - sentAt; + assert.ok( + waitedMs >= STALLED_HANDSHAKE_MS, + `must not be answered before its deadline; waited only ${waitedMs}ms`, + ); + assert.ok( + waitedMs < CALL_TIMEOUT_MS, + `must be answered on the stalled-handshake deadline, not the ordinary ${CALL_TIMEOUT_MS}ms ` + + `call timeout — that long wait with no feedback is the reported bug; waited ${waitedMs}ms`, + ); + + assert.equal( + reply.result?.isError, + true, + `expected a tool-error envelope, got ${JSON.stringify(reply)}`, + ); + const text = JSON.stringify(reply.result); + assert.match( + text, + /could not reach the relayer/i, + `the answer must name the failing connection, got ${text}`, + ); + assert.match( + text, + /nothing was\\?\s*stored/i, + `the answer must say the call never ran, got ${text}`, + ); + assert.match( + text, + /503/, + `the answer must carry the handshake error the call was stuck behind, got ${text}`, + ); + + assert.equal(child.exitCode, null, "bridge should still be running, not exited"); + + // initialize answered exactly once — the sweep must never write a second + // envelope for an id that was answered locally. + const initReplies = received.filter((m) => m.id === 1 && (m.result || m.error)); + assert.equal( + initReplies.length, + 1, + `initialize (id=1) must be answered exactly once; saw ${initReplies.length}`, + ); + + // The hazard the expiry has to close: the answered call is still a plain + // object sitting in `pendingForward`, and the flush that follows the next + // successful connect forwards whatever it finds there. Left in, a + // `remember` we just reported as never having run would run for real — + // after the agent was told nothing was stored, so it may well have retried + // by then. Let the relayer recover and prove the call is gone. + mock.heal(); + const connectedAt = Date.now(); + while (!stderrBuf.includes("Connected. Bridging") && Date.now() - connectedAt < 15_000) { + await new Promise((r) => setTimeout(r, 100)); + } + assert.ok( + stderrBuf.includes("Connected. Bridging"), + `the relayer must recover for this assertion to mean anything; stderr:\n${stderrBuf}`, + ); + + // A fresh call proves the session really is carrying traffic, so "id=2 + // never posted" below is evidence rather than an artefact of a dead link. + send({ + jsonrpc: "2.0", + id: 3, + method: "tools/call", + params: { name: "memwal_remember", arguments: { text: "after recovery" } }, + }); + const postedAt = Date.now(); + while ( + !mock.getPosted().some((m) => m.id === 3) && + Date.now() - postedAt < 10_000 + ) { + await new Promise((r) => setTimeout(r, 100)); + } + assert.ok( + mock.getPosted().some((m) => m.id === 3), + `a call sent after recovery must reach the relayer; posted: ${JSON.stringify(mock.getPosted())}`, + ); + + assert.ok( + !mock.getPosted().some((m) => m.id === 2), + `the expired call must never reach the relayer after being answered, but saw: ${JSON.stringify( + mock.getPosted().filter((m) => m.id === 2), + )}`, + ); + const callReplies = received.filter((m) => m.id === 2); + assert.equal( + callReplies.length, + 1, + `id=2 must be answered exactly once; saw ${callReplies.length}: ${JSON.stringify(callReplies)}`, + ); + + // `initialize` is buffered, not answered upstream — the expiry must leave + // it in place so the recovered session still negotiates capabilities. + assert.ok( + mock.getPosted().some((m) => m.method === "initialize"), + `initialize must still be forwarded once the session comes up; posted: ${JSON.stringify( + mock.getPosted().map((m) => m.method ?? m.id), + )}`, + ); +}); + +/** Serves exactly one healthy session, then refuses every reconnect and 404s + * any POST against the dead session — the way the real relayer does. The + * sibling mock above fails from the first handshake, which lands the request + * in `pendingForward`; that path is already covered and cannot reach the + * mid-session case. */ +function startHealthyThenDeadRelayer() { + let sseGetCount = 0; + let postCount = 0; + let sessionAlive = false; + let liveSession = null; + let liveHeartbeat = null; + const server = http.createServer((req, res) => { + const url = new URL(req.url, "http://127.0.0.1"); + if (req.method === "GET" && url.pathname === "/version") { + res.writeHead(200, { "content-type": "application/json" }); + res.end( + JSON.stringify({ + apiVersion: "1.0.0", + relayerVersion: "1.0.0", + minSupportedSdk: { mcp: "0.0.1" }, + }), + ); + return; + } + if (req.method === "GET" && url.pathname === "/api/mcp/sse") { + if (!hasBridgeAuth(req)) { + res.writeHead(401); + res.end(); + return; + } + sseGetCount += 1; + if (sseGetCount > 1) { + res.writeHead(503, { "content-type": "text/plain" }); + res.end("upstream unavailable"); + return; + } + res.writeHead(200, { + "content-type": "text/event-stream", + "cache-control": "no-cache", + connection: "keep-alive", + }); + res.write("event: endpoint\ndata: /api/mcp/messages?sessionId=session-1\n\n"); + liveSession = res; + sessionAlive = true; + liveHeartbeat = setInterval(() => { + if (!res.writableEnded) res.write(":\n\n"); + }, 200); + liveHeartbeat.unref?.(); + res.on("close", () => clearInterval(liveHeartbeat)); + return; + } + if (req.method === "POST" && url.pathname === "/api/mcp/messages") { + postCount += 1; + // A POST against a session that no longer exists is a 404 here, as + // it is on the relayer. Answering 202 would let a stray post look + // delivered and quietly flip the request to "sent". + res.writeHead(!hasBridgeAuth(req) ? 401 : sessionAlive ? 202 : 404); + res.end(); + return; + } + res.writeHead(404); + res.end(); + }); + return new Promise((ready) => { + server.listen(0, "127.0.0.1", () => { + const { port } = server.address(); + ready({ + server, + base: `http://127.0.0.1:${port}`, + getSseGetCount: () => sseGetCount, + getPostCount: () => postCount, + killSession: () => { + sessionAlive = false; + if (liveHeartbeat) clearInterval(liveHeartbeat); + if (liveSession && !liveSession.writableEnded) liveSession.end(); + }, + closeStreams: () => { + sessionAlive = false; + if (liveHeartbeat) clearInterval(liveHeartbeat); + if (liveSession && !liveSession.writableEnded) liveSession.end(); + }, + }); + }); + }); +} + +test("a call issued after the session dies is answered on the stalled deadline", async (t) => { + // `neverSent` used to be read from `pendingForward` membership, and nothing + // refills that buffer once `firstConnectDone` is set — so in the reported + // shape (the bridge worked, then the relayer stopped answering) the call + // looked sent, kept the full call timeout, and blamed a dropped connection + // for a request that never left the process. + // + // Every step below is confirmed from the bridge's own stderr before the + // next one runs. Two earlier attempts at this test raced an internal state + // transition the mock cannot see, and failed in ways the output could not + // explain; if this one fails, the assertion says which precondition broke. + const mock = await startHealthyThenDeadRelayer(); + const home = mkdtempSync(join(tmpdir(), "memwal-midsession-stalled-test-")); + const credsPath = join(home, ".memwal", "credentials.json"); + mkdirSync(dirname(credsPath), { recursive: true }); + writeFileSync(credsPath, JSON.stringify(makeCreds(mock.base)), { mode: 0o600 }); + + const child = spawn(process.execPath, [BIN, "--relayer", mock.base, "--web-url", mock.base], { + env: { + ...process.env, + HOME: home, + USERPROFILE: home, + MEMWAL_MCP_CONNECT_TIMEOUT_MS: String(CONNECT_TIMEOUT_MS), + MEMWAL_MCP_CALL_TIMEOUT_MS: String(CALL_TIMEOUT_MS), + MEMWAL_MCP_STALLED_HANDSHAKE_MS: String(STALLED_HANDSHAKE_MS), + }, + stdio: ["pipe", "pipe", "pipe"], + }); + + const received = []; + const listeners = new Set(); + let buf = ""; + child.stdout.on("data", (d) => { + buf += d.toString(); + let nl; + while ((nl = buf.indexOf("\n")) >= 0) { + const line = buf.slice(0, nl); + buf = buf.slice(nl + 1); + if (!line.trim()) continue; + let msg; + try { + msg = JSON.parse(line); + } catch { + continue; + } + received.push(msg); + for (const l of [...listeners]) l(msg); + } + }); + let stderrBuf = ""; + child.stderr.on("data", (d) => (stderrBuf += d.toString())); + + const dump = (what) => + `${what}\n--- stderr ---\n${stderrBuf}\n--- received ---\n${received.map((m) => JSON.stringify(m)).join("\n")}`; + const send = (obj) => child.stdin.write(JSON.stringify(obj) + "\n"); + const waitFor = (pred, ms, what) => { + const hit = received.find(pred); + if (hit) return Promise.resolve(hit); + return new Promise((res, rej) => { + const timer = setTimeout(() => { + listeners.delete(l); + rej(new Error(dump(`timed out waiting for ${what}`))); + }, ms); + const l = (m) => { + if (pred(m)) { + clearTimeout(timer); + listeners.delete(l); + res(m); + } + }; + listeners.add(l); + }); + }; + /** Poll a condition, and fail with the full bridge output naming it. */ + const until = async (pred, ms, what) => { + const deadline = Date.now() + ms; + while (Date.now() < deadline) { + if (pred()) return; + await new Promise((r) => setTimeout(r, 50)); + } + throw new Error(dump(`precondition never held: ${what}`)); + }; + + t.after(() => { + child.kill("SIGKILL"); + mock.closeStreams(); + mock.server.close(); + rmSync(home, { recursive: true, force: true }); + }); + + // 1. initialize is answered locally. + send({ jsonrpc: "2.0", id: 1, method: "initialize", params: {} }); + await waitFor((m) => m.id === 1 && m.result, 10_000, "the local initialize reply"); + + // 2. the first session is genuinely up, so `firstConnectDone` is set and + // this cannot degenerate into the cold-start case. + await until(() => /"event":"bridge\.connected"/.test(stderrBuf), 15_000, "bridge.connected"); + + // 3. the relayer goes away and refuses every reconnect. + mock.killSession(); + await until( + () => /"event":"bridge\.reconnect_failed"/.test(stderrBuf), + 20_000, + "bridge.reconnect_failed (so sse is null and the handshake is on record as failing)", + ); + + // 4. only now is the call issued — it must take the never-sent path. + const sentAt = Date.now(); + send({ + jsonrpc: "2.0", + id: 2, + method: "tools/call", + params: { name: "memwal_remember", arguments: { text: "anything" } }, + }); + + const reply = await waitFor( + (m) => m.id === 2, + Math.floor(CALL_TIMEOUT_MS * 0.6), + "the tool-call answer on the stalled-handshake deadline", + ); + const waitedMs = Date.now() - sentAt; + + // `sse` is NOT cleared on server-pump EOF — it keeps pointing at the dead + // session until a reconnect succeeds — so the call does take the POST + // path and is 404'd by the relayer. That 404 is what returns it to + // never-sent: the session did not exist, so the message was discarded + // rather than routed, and it provably did not run. + assert.match( + stderrBuf, + /"event":"bridge\.session_stale"/, + dump("expected the stale-session 404 that returns the call to never-sent"), + ); + assert.ok( + waitedMs < CALL_TIMEOUT_MS, + `must expire on the ${STALLED_HANDSHAKE_MS}ms stalled deadline, not the ${CALL_TIMEOUT_MS}ms ` + + `call timeout — waiting out the latter with no feedback is the reported bug; waited ${waitedMs}ms`, + ); + assert.equal(reply.result?.isError, true, dump("expected a tool-error envelope")); + const text = JSON.stringify(reply.result); + assert.match(text, /could not reach the relayer/i, dump("must name the failing connection")); + assert.match(text, /nothing was\\?\s*stored/i, dump("must say the call never ran")); + assert.equal(child.exitCode, null, "bridge should still be running, not exited"); +}); diff --git a/packages/mcp/test/sse-handshake-429.test.mjs b/packages/mcp/test/sse-handshake-429.test.mjs new file mode 100644 index 000000000..30651153c --- /dev/null +++ b/packages/mcp/test/sse-handshake-429.test.mjs @@ -0,0 +1,404 @@ +/** + * Regression test for WALM-386 — a 429 on the SSE handshake must be honoured as + * a THROTTLE, not retried blind. + * + * Bug being guarded against: `openSseStream` read the `retry-after` header and + * then interpolated it into an Error *message string*, so nothing + * machine-readable survived. Both retry loops (`connectInBackground` and + * `reconnect`) fell back to the generic geometric backoff, i.e. the first retry + * after a 429 landed ~500ms later — well inside the window the relayer had just + * asked for, and pure noise against `ip_active_cap`, which is a CONCURRENT cap + * that only clears when some other session closes. + * + * Repro: + * - Mock relayer answers GET /version, then 429s the SSE GET for the first N + * attempts (with or without a `retry-after` header), then serves a real + * event-stream. Every SSE GET is timestamped. + * + * Asserts: + * - a `retry-after: 2` is actually waited out (gap between attempts ≈ 2s, not + * 500ms), and the attempt COUNT over the throttle window stays low; + * - a 429 with NO `retry-after` (the `ip_active_cap` shape) falls back to the + * throttle floor rather than the sub-second geometric backoff; + * - the bridge does not exit and `initialize` is still answered locally + * exactly once; + * - a tool call buffered during the throttle is served for real once the + * relayer stops throttling (the buffering path is untouched); + * - stderr says "rate-limiting … not a bad config or bad credentials", so the + * user can tell a throttle from a misconfiguration. + */ +import { test } from "node:test"; +import assert from "node:assert/strict"; +import http from "node:http"; +import { spawn } from "node:child_process"; +import { mkdtempSync, mkdirSync, writeFileSync, rmSync } from "node:fs"; +import { tmpdir } from "node:os"; +import { join, dirname, resolve } from "node:path"; +import { fileURLToPath } from "node:url"; + +const __dirname = dirname(fileURLToPath(import.meta.url)); +const BIN = resolve(__dirname, "../dist/bin/memwal-mcp.js"); +const EXPECTED_BEARER = "a".repeat(64); +const EXPECTED_ACCOUNT_ID = "0x" + "3".repeat(64); + +function hasBridgeAuth(req) { + return ( + req.headers.authorization === `Bearer ${EXPECTED_BEARER}` && + req.headers["x-memwal-account-id"] === EXPECTED_ACCOUNT_ID + ); +} + +/** + * Mock relayer that 429s the SSE handshake `throttleCount` times, then serves a + * working stream. `retryAfterSeconds: null` reproduces the header-less + * `ip_active_cap` denial the real relayer sends for a concurrent cap. + */ +function startThrottlingRelayer({ throttleCount, retryAfterSeconds }) { + /** ms-since-start of every SSE GET, so the test can measure the gaps. */ + const sseGetAt = []; + const startedAt = Date.now(); + const sessions = new Map(); + let sseGetCount = 0; + const openStreams = []; + + const server = http.createServer((req, res) => { + const u = new URL(req.url, "http://127.0.0.1"); + if (req.method === "GET" && u.pathname === "/version") { + res.writeHead(200, { "content-type": "application/json" }); + res.end( + JSON.stringify({ + apiVersion: "1.0.0", + relayerVersion: "1.0.0", + minSupportedSdk: { mcp: "0.0.1" }, + }), + ); + return; + } + if (req.method === "GET" && u.pathname === "/api/mcp/sse") { + if (!hasBridgeAuth(req)) { + res.writeHead(401); + res.end(); + return; + } + sseGetCount += 1; + sseGetAt.push(Date.now() - startedAt); + if (sseGetCount <= throttleCount) { + // Same envelope the relayer's `rateLimitDeny` sends. + const headers = { "content-type": "application/json" }; + if (retryAfterSeconds != null) { + headers["retry-after"] = String(retryAfterSeconds); + } + res.writeHead(429, headers); + res.end( + JSON.stringify({ + jsonrpc: "2.0", + error: { + code: -32000, + message: + retryAfterSeconds == null + ? "MCP rate limit: ip_active_cap. Close another MCP session, then retry." + : `MCP rate limit: ip_burst_cap. Try again in ${retryAfterSeconds}s.`, + }, + id: null, + }), + ); + return; + } + const sessionId = `session-${sseGetCount}`; + res.writeHead(200, { + "content-type": "text/event-stream", + "cache-control": "no-cache", + connection: "keep-alive", + }); + res.write(`event: endpoint\ndata: /api/mcp/messages?sessionId=${sessionId}\n\n`); + sessions.set(sessionId, { res }); + openStreams.push(res); + const hb = setInterval(() => { + if (!res.writableEnded) res.write(":\n\n"); + else clearInterval(hb); + }, 200); + hb.unref?.(); + res.on("close", () => clearInterval(hb)); + return; + } + if (req.method === "POST" && u.pathname === "/api/mcp/messages") { + if (!hasBridgeAuth(req)) { + res.writeHead(401); + res.end(); + return; + } + const session = sessions.get(u.searchParams.get("sessionId")); + let body = ""; + req.on("data", (c) => (body += c)); + req.on("end", () => { + let msg; + try { + msg = JSON.parse(body); + } catch { + res.writeHead(202); + res.end(); + return; + } + if (!session) { + res.writeHead(404); + res.end(); + return; + } + res.writeHead(202); + res.end(); + if (msg.method === "initialize") return; // suppressed by the bridge + if (msg.method === "tools/call") { + session.res.write( + `event: message\ndata: ${JSON.stringify({ + jsonrpc: "2.0", + id: msg.id, + result: { + content: [{ type: "text", text: "RECALLED" }], + isError: false, + }, + })}\n\n`, + ); + } + }); + return; + } + res.writeHead(404); + res.end(); + }); + + return new Promise((res) => { + server.listen(0, "127.0.0.1", () => { + res({ + server, + base: `http://127.0.0.1:${server.address().port}`, + sseGetAt, + getSseGetCount: () => sseGetCount, + closeStreams: () => openStreams.forEach((r) => r.end()), + }); + }); + }); +} + +function makeCreds(relayerUrl) { + return { + delegatePrivateKey: EXPECTED_BEARER, + delegatePublicKeyHex: "b".repeat(64), + delegateAddress: "0x" + "1".repeat(64), + walletAddress: "0x" + "2".repeat(64), + accountId: EXPECTED_ACCOUNT_ID, + packageId: "0x" + "4".repeat(64), + relayerUrl, + label: "SSE 429 Test", + createdAt: new Date(0).toISOString(), + version: 1, + }; +} + +/** Spawn the bridge against `mock`, wired with the usual line-splitter. */ +function startBridge(t, mock, env = {}) { + const home = mkdtempSync(join(tmpdir(), "memwal-sse-429-test-")); + const credsPath = join(home, ".memwal", "credentials.json"); + mkdirSync(dirname(credsPath), { recursive: true }); + writeFileSync(credsPath, JSON.stringify(makeCreds(mock.base)), { mode: 0o600 }); + + const child = spawn(process.execPath, [BIN, "--relayer", mock.base, "--web-url", mock.base], { + env: { ...process.env, HOME: home, USERPROFILE: home, ...env }, + stdio: ["pipe", "pipe", "pipe"], + }); + + const received = []; + const listeners = new Set(); + let buf = ""; + child.stdout.on("data", (d) => { + buf += d.toString(); + let nl; + while ((nl = buf.indexOf("\n")) >= 0) { + const line = buf.slice(0, nl); + buf = buf.slice(nl + 1); + if (!line.trim()) continue; + let msg; + try { + msg = JSON.parse(line); + } catch { + continue; + } + received.push(msg); + for (const l of [...listeners]) l(msg); + } + }); + let stderrBuf = ""; + child.stderr.on("data", (d) => (stderrBuf += d.toString())); + + t.after(() => { + child.kill("SIGKILL"); + mock.closeStreams(); + mock.server.close(); + rmSync(home, { recursive: true, force: true }); + }); + + return { + child, + received, + send: (obj) => child.stdin.write(JSON.stringify(obj) + "\n"), + stderr: () => stderrBuf, + waitFor: (pred, ms = 15000) => { + const hit = received.find(pred); + if (hit) return Promise.resolve(hit); + return new Promise((res, rej) => { + const timer = setTimeout(() => { + listeners.delete(l); + rej( + new Error( + `timed out waiting for message\n--- stderr ---\n${stderrBuf}\n--- received ---\n${received.map((m) => JSON.stringify(m)).join("\n")}`, + ), + ); + }, ms); + const l = (m) => { + if (pred(m)) { + clearTimeout(timer); + listeners.delete(l); + res(m); + } + }; + listeners.add(l); + }); + }, + }; +} + +test("a 429 with Retry-After is waited out, not retried after 500ms", async (t) => { + const mock = await startThrottlingRelayer({ throttleCount: 2, retryAfterSeconds: 2 }); + // Floor set well BELOW the advertised interval so a passing gap can only + // come from the header, never from the no-header fallback. + const bridge = startBridge(t, mock, { MEMWAL_MCP_THROTTLE_FLOOR_MS: "250" }); + + // initialize is answered locally even while the relayer is throttling — the + // whole reason a throttled bridge should not look like a broken one. + bridge.send({ jsonrpc: "2.0", id: 1, method: "initialize", params: {} }); + const init = await bridge.waitFor((m) => m.id === 1 && m.result, 5_000); + assert.equal(init.result.serverInfo.name, "memwal"); + + // Buffered during the throttle; must be served for real after recovery. + bridge.send({ + jsonrpc: "2.0", + id: 2, + method: "tools/call", + params: { name: "memwal_recall", arguments: { query: "anything" } }, + }); + + // 2 denials × 2s ≈ 4s before the third attempt succeeds. + const recall = await bridge.waitFor((m) => m.id === 2, 20_000); + + // The regression guard. Pre-fix the gaps were ~500ms / ~1s (geometric). + const gaps = mock.sseGetAt.slice(1).map((t2, i) => t2 - mock.sseGetAt[i]); + assert.ok( + gaps.length >= 2, + `expected at least 3 SSE attempts, saw ${mock.sseGetAt.length}: ${JSON.stringify(mock.sseGetAt)}`, + ); + for (const [i, gap] of gaps.entries()) { + assert.ok( + gap >= 1_600, + `attempt ${i + 2} came ${gap}ms after attempt ${i + 1}; a retry-after of 2s must be honoured (gaps: ${JSON.stringify(gaps)})`, + ); + } + // Attempt count is the flake-resistant half of the same signal: a 500ms + // geometric backoff would have burned ~7 attempts by the time this lands. + assert.ok( + mock.getSseGetCount() <= 4, + `expected the throttle to be respected, but the bridge made ${mock.getSseGetCount()} SSE attempts`, + ); + + // Recovery: the buffered call is served, not error-enveloped. + assert.equal( + recall.result?.isError, + false, + `buffered call should be served after the throttle clears, got ${JSON.stringify(recall)}`, + ); + + // Still alive, and initialize answered exactly once. + assert.equal(bridge.child.exitCode, null, "bridge should still be running, not exited"); + const initReplies = bridge.received.filter((m) => m.id === 1 && (m.result || m.error)); + assert.equal( + initReplies.length, + 1, + `initialize (id=1) must be answered exactly once; saw ${initReplies.length}`, + ); + + // The user must be able to tell "throttled" from "misconfigured". + const stderr = bridge.stderr(); + assert.match(stderr, /rate-limiting new MCP sessions \(HTTP 429\)/); + assert.match(stderr, /not a bad config or bad credentials/); + assert.doesNotMatch( + stderr, + /rejected credentials \(HTTP 401\)/, + "a throttle must not be reported as a credential problem", + ); +}); + +test("a 429 with Retry-After: 0 falls back to the floor, not to 500ms", async (t) => { + // `0` parses, so it used to satisfy `advised ?? floor` and set the wait to + // zero — the backoff collapsed to the ~500ms geometric retry this whole + // feature exists to remove, and `serverAdvised` stayed true, suppressing + // the concurrent-cap hint as well. It is only reachable in production + // since the relayer started forwarding `retry-after` at all. + const mock = await startThrottlingRelayer({ throttleCount: 1, retryAfterSeconds: 0 }); + const bridge = startBridge(t, mock, { MEMWAL_MCP_THROTTLE_FLOOR_MS: "2500" }); + + bridge.send({ jsonrpc: "2.0", id: 1, method: "initialize", params: {} }); + await bridge.waitFor((m) => m.id === 1 && m.result, 5_000); + + bridge.send({ + jsonrpc: "2.0", + id: 2, + method: "tools/call", + params: { name: "memwal_recall", arguments: { query: "anything" } }, + }); + await bridge.waitFor((m) => m.id === 2, 20_000); + + assert.ok( + mock.sseGetAt.length >= 2, + `expected a retry after the 429, saw ${mock.sseGetAt.length} attempts`, + ); + const gap = mock.sseGetAt[1] - mock.sseGetAt[0]; + assert.ok( + gap >= 2_000, + `a zero Retry-After must be ignored in favour of the floor; retry came after ${gap}ms`, + ); + + assert.equal(bridge.child.exitCode, null, "bridge should still be running, not exited"); + // Treating it as no usable header also restores `serverAdvised: false`, + // so the user still gets the one remediation that clears a live cap. + assert.match(bridge.stderr(), /closing another\s+MCP client/); +}); + +test("a 429 with no Retry-After falls back to the throttle floor", async (t) => { + // The ip_active_cap shape: a concurrent cap, so the relayer deliberately + // sends no header — there is no honest ETA to give. + const mock = await startThrottlingRelayer({ throttleCount: 1, retryAfterSeconds: null }); + const bridge = startBridge(t, mock, { MEMWAL_MCP_THROTTLE_FLOOR_MS: "2500" }); + + bridge.send({ jsonrpc: "2.0", id: 1, method: "initialize", params: {} }); + await bridge.waitFor((m) => m.id === 1 && m.result, 5_000); + + bridge.send({ + jsonrpc: "2.0", + id: 2, + method: "tools/call", + params: { name: "memwal_recall", arguments: { query: "anything" } }, + }); + await bridge.waitFor((m) => m.id === 2, 20_000); + + assert.ok( + mock.sseGetAt.length >= 2, + `expected a retry after the 429, saw ${mock.sseGetAt.length} attempts`, + ); + const gap = mock.sseGetAt[1] - mock.sseGetAt[0]; + assert.ok( + gap >= 2_000, + `header-less 429 must fall back to the throttle floor; retry came after ${gap}ms`, + ); + + assert.equal(bridge.child.exitCode, null, "bridge should still be running, not exited"); + // The no-ETA branch tells the user what actually clears a concurrent cap. + assert.match(bridge.stderr(), /closing another\s+MCP client/); +}); diff --git a/packages/mcp/test/unknown-flags.test.mjs b/packages/mcp/test/unknown-flags.test.mjs new file mode 100644 index 000000000..f6c24de5b --- /dev/null +++ b/packages/mcp/test/unknown-flags.test.mjs @@ -0,0 +1,126 @@ +import assert from "node:assert/strict"; +import test from "node:test"; + +import { helpText, parseArgs } from "../dist/index.js"; + +// An unrecognised flag used to fall through parseArgs' default branch and +// vanish: a typo'd `--namesapce` still wrote to the relayer's "default" +// namespace with nothing on stderr to explain why. parseArgs now collects what +// it did not understand so main() can name it. + +test("parseArgs collects a typo'd flag instead of dropping it", () => { + const args = parseArgs(["--namesapce", "work"]); + assert.deepEqual(args.unknown, ["--namesapce"]); + // The typo must NOT have set the real namespace. + assert.equal(args.namespace, undefined); +}); + +test("parseArgs reports every unknown flag, not just the first", () => { + const args = parseArgs(["--nope", "--alsobad"]); + assert.deepEqual(args.unknown, ["--nope", "--alsobad"]); +}); + +test("an unknown flag swallows its value rather than reporting it too", () => { + // Warning once about `--namesapce` beats warning twice, the second time + // naming the user's data. Also keeps a mistyped secret out of the logs. + assert.deepEqual(parseArgs(["--tokenn", "hunter2"]).unknown, ["--tokenn"]); +}); + +test("an unknown flag does not swallow the flag that follows it", () => { + const args = parseArgs(["--typo", "--prod"]); + assert.deepEqual(args.unknown, ["--typo"]); + assert.equal(args.relayerUrl, "https://relayer.memory.walrus.xyz"); +}); + +test("a known flag after an unknown flag's value still applies", () => { + const args = parseArgs(["--typo", "value", "--ns", "work"]); + assert.deepEqual(args.unknown, ["--typo"]); + assert.equal(args.namespace, "work"); +}); + +test("an unknown flag does not swallow the `login` command", () => { + // `login` is a command, not a value. Consuming it turned + // `memwal-mcp --typo login` into a run that never logged in. + const args = parseArgs(["--typo", "login"]); + assert.deepEqual(args.unknown, ["--typo"]); + assert.equal(args.forceLogin, true, "`login` was swallowed as a flag value"); +}); + +test("an unknown `--key=value` flag reports the key and never the value", () => { + // The warning goes to stderr, so a mistyped secret must not survive into it. + const args = parseArgs(["--tokenn=hunter2"]); + assert.deepEqual(args.unknown, ["--tokenn"]); + assert.ok( + !args.unknown.some((u) => u.includes("hunter2")), + "the flag's value reached the warning", + ); +}); + +test("an unknown `--key=value` flag does not also swallow the next token", () => { + // Its value is already attached, so the following token is someone else's. + const args = parseArgs(["--tokenn=hunter2", "login"]); + assert.deepEqual(args.unknown, ["--tokenn"]); + assert.equal(args.forceLogin, true); +}); + +test("parseArgs treats no known flag as unknown", () => { + const known = [ + "--help", "-h", + "--logout", + "--login", "login", + "--prod", "--dev", "--staging", "--local", + "--relayer", "https://r.example", + "--relayer-url", "https://r.example", + "--web-url", "https://w.example", + "--web", "https://w.example", + "--label", "my label", + "--namespace", "ns", + "--ns", "ns", + "--relayer=https://r.example", + "--web-url=https://w.example", + "--label=my-label", + "--namespace=ns", + "--ns=ns", + ]; + assert.deepEqual(parseArgs(known).unknown, []); +}); + +test("parseArgs does not mistake a flag's value for an unknown flag", () => { + // `next()` consumes the value, so "MCP Client" must never be reported. + const args = parseArgs(["--label", "MCP Client"]); + assert.deepEqual(args.unknown, []); + assert.equal(args.label, "MCP Client"); +}); + +test("env presets still resolve both URLs (regression guard)", () => { + const args = parseArgs(["--prod"]); + assert.equal(args.relayerUrl, "https://relayer.memory.walrus.xyz"); + assert.equal(args.webUrl, "https://memory.walrus.xyz"); + assert.deepEqual(args.unknown, []); +}); + +// Help must list every preset the parser honours, and stay listing them as +// presets are added. + +test("--help documents every network preset the parser accepts", () => { + const help = helpText(); + for (const preset of ["--prod", "--dev", "--staging", "--local"]) { + // Not merely mentioned somewhere — parseArgs must accept it too. + assert.deepEqual(parseArgs([preset]).unknown, [], `${preset} not accepted`); + assert.ok(help.includes(preset), `${preset} missing from --help`); + } + // The URLs a preset resolves to are what tell you which network you're on. + assert.ok(help.includes("https://relayer.dev.memwal.ai")); + assert.ok(help.includes("http://127.0.0.1:8000")); +}); + +test("--help does not promise that flag order decides a preset override", () => { + // Preset application is `??=`, so an explicit URL wins from either side. + // Help used to say the flag overrides "the preset it follows". + const help = helpText(); + assert.ok(!help.includes("the preset it follows"), "help still implies order matters"); + const before = parseArgs(["--relayer", "https://custom.example", "--prod"]); + const after = parseArgs(["--prod", "--relayer", "https://custom.example"]); + assert.equal(before.relayerUrl, "https://custom.example"); + assert.equal(after.relayerUrl, "https://custom.example"); +}); diff --git a/packages/python-sdk-memwal/CHANGELOG.md b/packages/python-sdk-memwal/CHANGELOG.md index 007dd8aca..9593c4d2d 100644 --- a/packages/python-sdk-memwal/CHANGELOG.md +++ b/packages/python-sdk-memwal/CHANGELOG.md @@ -1,5 +1,11 @@ # memwal +## 0.1.10 + +### Added + +- `restore()` results include `failed` (default `0`) for permanent decrypt/UTF-8 failures instead of folding them into `skipped` or dropping them silently. + ## 0.1.9 ### Fixed diff --git a/packages/python-sdk-memwal/memwal/__init__.py b/packages/python-sdk-memwal/memwal/__init__.py index 974c3903d..18c7f6e78 100644 --- a/packages/python-sdk-memwal/memwal/__init__.py +++ b/packages/python-sdk-memwal/memwal/__init__.py @@ -122,4 +122,4 @@ "RecallManualResult", ] -__version__ = "0.1.9" +__version__ = "0.1.10" diff --git a/packages/python-sdk-memwal/memwal/client.py b/packages/python-sdk-memwal/memwal/client.py index cb964b86a..e6faa2d62 100644 --- a/packages/python-sdk-memwal/memwal/client.py +++ b/packages/python-sdk-memwal/memwal/client.py @@ -905,9 +905,15 @@ async def restore(self, namespace: str, limit: int = 10) -> RestoreResult: * ``restored`` — blobs that completed the full download → decrypt → embed → DB insert pipeline this call. - * ``skipped`` — on-chain blobs already present in the local index - (no work needed). Decrypt / embed failures are dropped silently and - do **not** count as either restored or skipped. + * ``skipped`` — on-chain blobs already present in the local success + index (no work needed). Does not include permanent decrypt/UTF-8 + failures. + * ``failed`` — permanent decrypt/UTF-8 failures on this on-chain page: + negative-cache hits plus any new permanent failures this call. + Transient download/decrypt/embed errors are not counted here; when + a page yields only those, ``truncated`` is true so the caller + retries. Defaults to ``0`` when talking to a relayer older than + COMG-719 that omits the field. * ``total`` — count of on-chain blobs the relayer saw for ``(owner, namespace)`` before the limit was applied. * ``truncated`` — True when this restore is known-incomplete (limit @@ -952,6 +958,8 @@ async def restore(self, namespace: str, limit: int = 10) -> RestoreResult: # treat "not present" as "not known to be truncated" rather # than require every relayer version to send it. truncated=data.get("truncated", False), + # Relayers older than COMG-719 omit `failed`; default to 0. + failed=data.get("failed", 0), ) async def health(self) -> HealthResult: diff --git a/packages/python-sdk-memwal/memwal/mock.py b/packages/python-sdk-memwal/memwal/mock.py index f4950e81b..06da3d105 100644 --- a/packages/python-sdk-memwal/memwal/mock.py +++ b/packages/python-sdk-memwal/memwal/mock.py @@ -341,6 +341,7 @@ async def restore(self, namespace: str, limit: int = 10) -> RestoreResult: namespace=namespace, owner=self._owner, truncated=False, + failed=0, ) async def health(self) -> HealthResult: diff --git a/packages/python-sdk-memwal/memwal/types.py b/packages/python-sdk-memwal/memwal/types.py index 9b7744dcd..09fdc49d8 100644 --- a/packages/python-sdk-memwal/memwal/types.py +++ b/packages/python-sdk-memwal/memwal/types.py @@ -212,7 +212,8 @@ class RestoreResult: #: ``limit`` can still expand that fetch (``limit < 20``). Once the cap #: is saturated, truncation follows this call's missing-blob page, not #: on-chain ``total``, so a fully restored namespace does not loop - #: (WALM-431 / GH #762). + #: (WALM-431 / GH #762). Also true when an inspected page produced only + #: transients (download/decrypt/embed) so the caller retries (WALM-480). #: #: ``truncated=False`` is not proof the sidecar saw every on-chain blob. #: Blobs beyond the owner-wide sidecar candidate cap can still be @@ -221,6 +222,10 @@ class RestoreResult: #: Relayers older than WALM-319 don't send this field at all; the SDK #: defaults it to ``False`` in that case rather than requiring it. truncated: bool = False + #: Permanent decrypt/UTF-8 failures on this on-chain page: negative-cache + #: hits plus any new permanent failures this call. Relayers older than + #: COMG-719 omit this field; the SDK defaults it to ``0``. + failed: int = 0 @dataclass diff --git a/packages/python-sdk-memwal/pyproject.toml b/packages/python-sdk-memwal/pyproject.toml index 355221fc3..549018172 100644 --- a/packages/python-sdk-memwal/pyproject.toml +++ b/packages/python-sdk-memwal/pyproject.toml @@ -4,7 +4,7 @@ build-backend = "hatchling.build" [project] name = "memwal" -version = "0.1.9" +version = "0.1.10" description = "Python SDK for Walrus Memory — Privacy-first AI memory with Ed25519 signing" readme = "README.md" license = "MIT" diff --git a/packages/python-sdk-memwal/tests/test_client.py b/packages/python-sdk-memwal/tests/test_client.py index 8ec4c8ee7..44adedd90 100644 --- a/packages/python-sdk-memwal/tests/test_client.py +++ b/packages/python-sdk-memwal/tests/test_client.py @@ -854,6 +854,7 @@ async def test_restore(self, memwal_client: MemWal) -> None: assert body["limit"] == 100 assert result.restored == 5 assert result.skipped == 2 + assert result.failed == 0 assert result.truncated is False @respx.mock @@ -906,6 +907,55 @@ async def test_restore_truncated_defaults_false_when_omitted( assert result.truncated is False + @respx.mock + async def test_restore_preserves_failed( + self, memwal_client: MemWal + ) -> None: + mock_seal_session_prereqs() + respx.post(f"{_TEST_SERVER}/api/restore").mock( + return_value=httpx.Response( + 200, + json={ + "restored": 5, + "skipped": 2, + "failed": 3, + "total": 10, + "namespace": "my-app", + "owner": "0xowner", + "truncated": False, + }, + ) + ) + + result = await memwal_client.restore("my-app", limit=100) + + assert result.failed == 3 + + @respx.mock + async def test_restore_failed_defaults_zero_when_omitted( + self, memwal_client: MemWal + ) -> None: + """Relayers older than COMG-719 omit `failed` — the SDK must + default it to 0 rather than requiring the field.""" + mock_seal_session_prereqs() + respx.post(f"{_TEST_SERVER}/api/restore").mock( + return_value=httpx.Response( + 200, + json={ + "restored": 5, + "skipped": 2, + "total": 7, + "namespace": "my-app", + "owner": "0xowner", + "truncated": False, + }, + ) + ) + + result = await memwal_client.restore("my-app", limit=100) + + assert result.failed == 0 + class TestHealth: @respx.mock diff --git a/packages/sdk/CHANGELOG.md b/packages/sdk/CHANGELOG.md index 52d317df9..75fa71aa5 100644 --- a/packages/sdk/CHANGELOG.md +++ b/packages/sdk/CHANGELOG.md @@ -1,5 +1,18 @@ # @mysten-incubation/memwal +## 0.1.7 + +### Added + +- `restore()` results include `failed` (required like `truncated`; SDK defaults omitted to `0`) for permanent decrypt/UTF-8 failures instead of folding them into `skipped` or dropping them silently. + +### Fixed + +- Hash request bodies with `@noble/hashes` instead of WebCrypto-or-`node:crypto`, so the package no longer imports a Node builtin on a browser-reachable path. Vite externalises such an import without warning: the app builds clean and the browser crashes the first time the path runs. The fallback could never have helped a browser anyway — `crypto.subtle` is absent precisely when the page is not a secure context, where `node:crypto` is absent too — so it only served Node <19 while being the sole source of the exposure. `sha256hex` sits on the signed-request path, so every remember and recall reached it. (#322, WALM-136) +- Declare `engines.node >= 20.0.0`, matching `memwal-mcp` and `openclaw-memory-memwal`. The SDK was the only published package without a floor. (WALM-599) +- Empty-body 401s now use the same AUTH_REJECTED troubleshooting message as credential 401s instead of telling callers to run `memwal_login`. Headless SDK clients do not have that MCP tool. +- `account.ts` and `manual.ts` PTBs use typed `tx.pure` helpers instead of the legacy untyped moveCall argument syntax that fails under modern `@mysten/sui`. + ## 0.1.6 ### Added diff --git a/packages/sdk/package.json b/packages/sdk/package.json index 4cc47eec5..ca57884f5 100644 --- a/packages/sdk/package.json +++ b/packages/sdk/package.json @@ -1,8 +1,11 @@ { "name": "@mysten-incubation/memwal", - "version": "0.1.6", + "version": "0.1.7", "description": "Walrus Memory — Privacy-first AI memory SDK with Ed25519 delegate key auth", "type": "module", + "engines": { + "node": ">=20.0.0" + }, "main": "./dist/index.js", "types": "./dist/index.d.ts", "exports": { diff --git a/packages/sdk/src/account.ts b/packages/sdk/src/account.ts index f4ab482b7..f62ec9640 100644 --- a/packages/sdk/src/account.ts +++ b/packages/sdk/src/account.ts @@ -287,8 +287,8 @@ export async function addDelegateKey( arguments: [ tx.object(opts.accountId), tx.object(opts.registryId), - tx.pure("vector", Array.from(pkBytes)), - tx.pure("string", opts.label), + tx.pure.vector("u8", Array.from(pkBytes)), + tx.pure.string(opts.label), tx.object(SUI_CLOCK), ], }); @@ -338,7 +338,7 @@ export async function removeDelegateKey( arguments: [ tx.object(opts.accountId), tx.object(opts.registryId), - tx.pure("vector", Array.from(pkBytes)), + tx.pure.vector("u8", Array.from(pkBytes)), ], }); diff --git a/packages/sdk/src/manual.ts b/packages/sdk/src/manual.ts index 65205bdb5..deb77d41c 100644 --- a/packages/sdk/src/manual.ts +++ b/packages/sdk/src/manual.ts @@ -505,7 +505,7 @@ export class MemWalManual { tx.moveCall({ target: `${this.config.sealPolicyPackageId ?? this.config.packageId}::account::seal_approve`, arguments: [ - tx.pure("vector", idBytes), + tx.pure.vector("u8", idBytes), tx.object(this.config.registryId), tx.object(this.config.accountId), ], @@ -932,6 +932,11 @@ export class MemWalManual { // Relayers older than WALM-319 omit `truncated` entirely — treat // "not present" as "not known to be truncated" rather than drop // the field or require every relayer version to send it. - return { ...result, truncated: result.truncated ?? false }; + // Relayers older than COMG-719 omit `failed`; default to 0. + return { + ...result, + truncated: result.truncated ?? false, + failed: result.failed ?? 0, + }; } } diff --git a/packages/sdk/src/memwal.ts b/packages/sdk/src/memwal.ts index 5217a0122..c49c90481 100644 --- a/packages/sdk/src/memwal.ts +++ b/packages/sdk/src/memwal.ts @@ -877,8 +877,12 @@ export class MemWal { * **Response semantics**: * - `restored` — blobs that completed the full * download → decrypt → embed → DB insert pipeline this call. - * - `skipped` — on-chain blobs already in the local index (no work needed). - * Decrypt / embed failures are dropped silently and count as neither. + * - `skipped` — on-chain blobs already in the local success index + * (no work needed). Does not include permanent decrypt/UTF-8 failures. + * - `failed` — permanent decrypt/UTF-8 failures on this on-chain page: + * negative-cache hits plus any new permanent failures this call. + * Transient download/decrypt/embed errors are not counted here; when + * a page yields only those, `truncated` is true so the caller retries. * - `total` — on-chain blobs the relayer saw for `(owner, namespace)` * before the limit was applied. * @@ -899,12 +903,12 @@ export class MemWal { * * @param namespace - Namespace to restore (exact match; no prefix/hierarchy) * @param limit - Max blobs to inspect this call (default: 10) - * @returns RestoreResult with restored / skipped / total counts + * @returns RestoreResult with restored / skipped / failed / total counts * * @example * ```typescript * const result = await memwal.restore("my-app"); - * console.log(`restored=${result.restored} skipped=${result.skipped} total=${result.total}`); + * console.log(`restored=${result.restored} skipped=${result.skipped} failed=${result.failed} total=${result.total}`); * ``` */ async restore(namespace: string, limit: number = 10): Promise { @@ -915,7 +919,12 @@ export class MemWal { // Relayers older than WALM-319 omit `truncated` entirely — treat // "not present" as "not known to be truncated" rather than drop // the field or require every relayer version to send it. - return { ...result, truncated: result.truncated ?? false }; + // Relayers older than COMG-719 omit `failed`; default to 0. + return { + ...result, + truncated: result.truncated ?? false, + failed: result.failed ?? 0, + }; } /** diff --git a/packages/sdk/src/mock.ts b/packages/sdk/src/mock.ts index d3161b781..c9c48791b 100644 --- a/packages/sdk/src/mock.ts +++ b/packages/sdk/src/mock.ts @@ -377,6 +377,7 @@ export class MemWalMock { return { restored: 0, skipped: 0, + failed: 0, total: 0, namespace, owner: this.owner, diff --git a/packages/sdk/src/types.ts b/packages/sdk/src/types.ts index eb140ee7f..e6b374c50 100644 --- a/packages/sdk/src/types.ts +++ b/packages/sdk/src/types.ts @@ -449,6 +449,12 @@ export interface ListNamespacesOptions { export interface RestoreResult { restored: number; skipped: number; + /** + * Permanent decrypt/UTF-8 failures on this on-chain page: negative-cache + * hits plus any new permanent failures this call. Relayers older than + * COMG-719 omit this field; the SDK defaults it to `0`. + */ + failed: number; total: number; namespace: string; owner: string; @@ -459,7 +465,8 @@ export interface RestoreResult { * `limit` can still expand that fetch (`limit < 20`). Once the cap is * saturated, truncation follows this call's missing-blob page, not * on-chain `total`, so a fully restored namespace does not loop - * (WALM-431 / GH #762). + * (WALM-431 / GH #762). Also true when an inspected page produced only + * transients (download/decrypt/embed) so the caller retries (WALM-480). * * `truncated=false` is not proof the sidecar saw every on-chain blob. * Blobs beyond the owner-wide sidecar candidate cap can still be diff --git a/packages/sdk/src/utils.ts b/packages/sdk/src/utils.ts index e8190b24a..54c33eb4d 100644 --- a/packages/sdk/src/utils.ts +++ b/packages/sdk/src/utils.ts @@ -11,20 +11,26 @@ import type { ScoringWeights } from "./types.js"; // ============================================================ /** - * Isomorphic SHA-256 hash — uses Web Crypto API (browser) or Node.js crypto (server). + * Isomorphic SHA-256 hash. + * + * Hashes in userland rather than reaching for a platform digest, so there is no + * Node builtin to import and nothing for a browser bundler to externalise — + * the WALM-136 / GH #322 landmine, where Vite quietly stubs `crypto`, the app + * builds clean, and the browser crashes the first time the path runs. + * + * This previously preferred WebCrypto and fell back to `node:crypto`. The + * fallback could never help a browser — when `crypto.subtle` is missing it is + * because the page is not a secure context, and `node:crypto` is not there + * either — so it only served Node <19 (EOL April 2025) while being the sole + * source of the bundler exposure. `sha256hex` is on the signed-request path, + * so every remember and recall runs through it. + * + * Stays `async` though `sha256` is synchronous: the signature is public API and + * callers already await it. */ export async function sha256hex(data: string): Promise { - const bytes = new TextEncoder().encode(data); - // Try Web Crypto API first (browser + modern Node.js) - if (typeof globalThis.crypto?.subtle?.digest === "function") { - const hashBuf = await globalThis.crypto.subtle.digest("SHA-256", bytes); - return Array.from(new Uint8Array(hashBuf)) - .map((b) => b.toString(16).padStart(2, "0")) - .join(""); - } - // Fallback to Node.js crypto - const crypto = await import("crypto"); - return crypto.createHash("sha256").update(data).digest("hex"); + const { sha256 } = await import("@noble/hashes/sha2.js"); + return bytesToHex(sha256(new TextEncoder().encode(data))); } // ============================================================ @@ -308,15 +314,11 @@ export function sanitizeServerError( ): { message: string; raw: string; serverCode?: string } { // Number() so a string "401" (some MCP / HTTP paths) still hits this branch. if (Number(status) === 401) { - // Empty body = no-session / bare relayer 401. Non-empty keeps - // the AUTH_REJECTED triage (wrong key / account / network). - const empty = !String(rawBody ?? "").trim(); return { - message: empty - ? "Walrus Memory isn't signed in. Call the memwal_login tool, then retry." - : "401 from relayer: typically wrong private key, key not registered on this account, " + - "account ID mismatch, or staging/mainnet mismatch. Check .env.local and dashboard credentials. " + - "Full troubleshooting: https://docs.wal.app/walrus-memory/troubleshooting/overview#401-auth_rejected-errors", + message: + "401 from relayer: typically wrong private key, key not registered on this account, " + + "account ID mismatch, or staging/mainnet mismatch. Check .env.local and dashboard credentials. " + + "Full troubleshooting: https://docs.wal.app/walrus-memory/troubleshooting/overview#401-auth_rejected-errors", raw: rawBody, serverCode: "AUTH_REJECTED", }; diff --git a/packages/sdk/test/no-node-builtins.test.mjs b/packages/sdk/test/no-node-builtins.test.mjs new file mode 100644 index 000000000..7fd9b7ec9 --- /dev/null +++ b/packages/sdk/test/no-node-builtins.test.mjs @@ -0,0 +1,90 @@ +/** + * WALM-136 (GH #322) — the SDK must not pull Node builtins into a browser bundle. + * + * The reported failure is a quiet one: Vite externalises an imported Node + * builtin without warning, the app builds cleanly, and the browser crashes at + * runtime when the code path is finally taken. A green build proves nothing, + * so the guard has to be on what the package actually ships. + * + * `sha256hex` was the last such import — a `crypto` fallback that could never + * help a browser anyway (if `crypto.subtle` is missing because the page is not + * a secure context, `node:crypto` is not there either) and only served Node + * <19, which is EOL. It sits on the signed-request path, so every remember and + * recall reached it. + * + * The first test is the regression guard, and it covers the whole class rather + * than one line: any future `fs`/`path`/`crypto` import in the SDK fails it. + * The second is a correctness companion — it passed before the swap too (Node + * resolved the fallback happily), so it is not the guard, it just proves the + * hash itself stayed right once the branch was removed. + */ +import { test } from "node:test"; +import assert from "node:assert/strict"; +import { builtinModules } from "node:module"; +import { readdirSync, readFileSync, existsSync } from "node:fs"; +import { join, dirname, resolve, relative } from "node:path"; +import { fileURLToPath } from "node:url"; + +const __dirname = dirname(fileURLToPath(import.meta.url)); +const DIST = resolve(__dirname, "../dist"); + +const BUILTINS = new Set(builtinModules); + +/** `from "x"`, `import("x")`, `require("x")` — the three ways a specifier can + * reach a bundler. Deliberately does NOT match `globalThis.crypto`, which is a + * global read and has nothing to resolve. */ +const SPECIFIER = /(?:\bfrom\s*|\bimport\s*\(\s*|\brequire\s*\(\s*)["']([^"']+)["']/g; + +function jsFiles(dir) { + const out = []; + for (const entry of readdirSync(dir, { withFileTypes: true })) { + const full = join(dir, entry.name); + if (entry.isDirectory()) out.push(...jsFiles(full)); + else if (entry.name.endsWith(".js")) out.push(full); + } + return out; +} + +test("the built SDK imports no Node builtins", () => { + assert.ok(existsSync(DIST), `dist/ missing — run \`pnpm run build\` first (looked in ${DIST})`); + + const offenders = []; + for (const file of jsFiles(DIST)) { + const src = readFileSync(file, "utf8"); + for (const [, spec] of src.matchAll(SPECIFIER)) { + const bare = spec.startsWith("node:") ? spec.slice("node:".length) : spec; + if (BUILTINS.has(bare)) { + offenders.push(`${relative(DIST, file)} imports "${spec}"`); + } + } + } + + assert.deepEqual( + offenders, + [], + `SDK ships Node builtin imports; a browser bundler will externalise these ` + + `silently and the page crashes when the path runs:\n ${offenders.join("\n ")}`, + ); +}); + +test("sha256hex hashes correctly without globalThis.crypto", async () => { + const original = Object.getOwnPropertyDescriptor(globalThis, "crypto"); + // Simulate a browser that is not a secure context: `crypto.subtle` is + // undefined there, which is exactly when the old fallback fired. + Object.defineProperty(globalThis, "crypto", { value: undefined, configurable: true }); + try { + const { sha256hex } = await import("../dist/utils.js"); + // Known-answer vectors, so a wrong-but-plausible digest cannot pass. + assert.equal( + await sha256hex("hello"), + "2cf24dba5fb0a30e26e83b2ac5b9e29e1b161e5c1fa7425e73043362938b9824", + ); + assert.equal( + await sha256hex(""), + "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855", + ); + } finally { + if (original) Object.defineProperty(globalThis, "crypto", original); + else delete globalThis.crypto; + } +}); diff --git a/packages/sdk/test/restore-truncated.test.mjs b/packages/sdk/test/restore-truncated.test.mjs index f09d67fd1..c9b3116c3 100644 --- a/packages/sdk/test/restore-truncated.test.mjs +++ b/packages/sdk/test/restore-truncated.test.mjs @@ -60,6 +60,26 @@ test("MemWal.restore() defaults truncated to false when an older relayer omits i assert.equal(result.truncated, false); }); +test("MemWal.restore() preserves failed from the relayer response", async () => { + const memwal = client(); + memwal.signedRequest = async () => baseResponse({ failed: 4 }); + + const result = await memwal.restore("demo"); + + assert.equal(result.failed, 4); +}); + +test("MemWal.restore() defaults failed to 0 when an older relayer omits it", async () => { + const memwal = client(); + const raw = baseResponse(); + delete raw.failed; + memwal.signedRequest = async () => raw; + + const result = await memwal.restore("demo"); + + assert.equal(result.failed, 0); +}); + test("MemWalManual.restore() preserves truncated=true from the relayer response", async () => { const manual = manualClient(); manual.signedRequest = async () => baseResponse({ truncated: true }); @@ -79,3 +99,23 @@ test("MemWalManual.restore() defaults truncated to false when an older relayer o assert.equal(result.truncated, false); }); + +test("MemWalManual.restore() preserves failed from the relayer response", async () => { + const manual = manualClient(); + manual.signedRequest = async () => baseResponse({ failed: 4 }); + + const result = await manual.restore("demo"); + + assert.equal(result.failed, 4); +}); + +test("MemWalManual.restore() defaults failed to 0 when an older relayer omits it", async () => { + const manual = manualClient(); + const raw = baseResponse(); + delete raw.failed; + manual.signedRequest = async () => raw; + + const result = await manual.restore("demo"); + + assert.equal(result.failed, 0); +}); diff --git a/packages/sdk/test/sanitize-server-error.test.mjs b/packages/sdk/test/sanitize-server-error.test.mjs index 23db9fd14..623afdad4 100644 --- a/packages/sdk/test/sanitize-server-error.test.mjs +++ b/packages/sdk/test/sanitize-server-error.test.mjs @@ -3,30 +3,31 @@ import test from "node:test"; import { sanitizeServerError } from "../dist/utils.js"; -const LOGIN = - "Walrus Memory isn't signed in. Call the memwal_login tool, then retry."; +const AUTH_REJECTED = + "401 from relayer: typically wrong private key, key not registered on this account, " + + "account ID mismatch, or staging/mainnet mismatch. Check .env.local and dashboard credentials. " + + "Full troubleshooting: https://docs.wal.app/walrus-memory/troubleshooting/overview#401-auth_rejected-errors"; -test("empty-body 401 points at memwal_login instead of ", () => { +test("empty-body 401 uses AUTH_REJECTED troubleshooting instead of memwal_login", () => { const { message, serverCode } = sanitizeServerError(401, ""); assert.equal(serverCode, "AUTH_REJECTED"); - assert.equal(message, LOGIN); + assert.equal(message, AUTH_REJECTED); assert.doesNotMatch(message, //); + assert.doesNotMatch(message, /memwal_login/); }); -test("string status \"401\" with an empty body uses the login hint", () => { +test("string status \"401\" with an empty body uses AUTH_REJECTED troubleshooting", () => { const { message, serverCode } = sanitizeServerError("401", " "); assert.equal(serverCode, "AUTH_REJECTED"); - assert.equal(message, LOGIN); + assert.equal(message, AUTH_REJECTED); assert.doesNotMatch(message, //); + assert.doesNotMatch(message, /memwal_login/); }); test("non-empty 401 keeps the AUTH_REJECTED troubleshooting URL", () => { const { message, serverCode } = sanitizeServerError(401, "auth rejected"); assert.equal(serverCode, "AUTH_REJECTED"); - assert.match( - message, - /docs\.wal\.app\/walrus-memory\/troubleshooting\/overview/, - ); + assert.equal(message, AUTH_REJECTED); assert.doesNotMatch(message, /memwal_login/); }); diff --git a/packages/sdk/test/typed-pure-args.test.mjs b/packages/sdk/test/typed-pure-args.test.mjs new file mode 100644 index 000000000..e781736d1 --- /dev/null +++ b/packages/sdk/test/typed-pure-args.test.mjs @@ -0,0 +1,36 @@ +import assert from "node:assert/strict"; +import { readFileSync } from "node:fs"; +import test from "node:test"; + +// WALM-442 / GH #799: account.ts and manual.ts used the legacy two-arg +// `tx.pure("vector", …)` form. Modern `@mysten/sui` documents the typed +// helpers (`tx.pure.vector("u8", …)`, `tx.pure.string(…)`). Pin the call +// sites so the legacy form cannot land again. + +const FILES = ["account.ts", "manual.ts"]; + +test("account.ts and manual.ts use typed tx.pure helpers, not legacy two-arg form", () => { + for (const file of FILES) { + const src = readFileSync(new URL(`../src/${file}`, import.meta.url), "utf8"); + assert.equal( + src.includes('tx.pure("'), + false, + `${file} still contains legacy tx.pure("type", value)`, + ); + assert.equal( + src.includes("tx.pure('"), + false, + `${file} still contains legacy tx.pure('type', value)`, + ); + assert.match( + src, + /tx\.pure\.vector\(\s*["']u8["']/, + `${file} must pass vector via tx.pure.vector`, + ); + } +}); + +test("addDelegateKey / removeDelegateKey labels use tx.pure.string", () => { + const src = readFileSync(new URL("../src/account.ts", import.meta.url), "utf8"); + assert.match(src, /tx\.pure\.string\(\s*opts\.label\s*\)/); +}); diff --git a/scripts/build-finalize-tx.ts b/scripts/build-finalize-tx.ts index aa353319f..0f5a9d510 100644 --- a/scripts/build-finalize-tx.ts +++ b/scripts/build-finalize-tx.ts @@ -14,6 +14,7 @@ * Burning the spent MigrationCap is a separate tx signed by the wallet that owns * the cap (the controller's), not this one. * A fresh completion report from the in-cluster one-shot Job is mandatory. + * After a security migration, write the operator completion artifact with scripts/write-migration-completion-artifact.mjs (docs/ops/migration-completion-artifact.md). * * Gas is auto-selected by build(): address balance when available, otherwise an * owned SUI coin. The transaction is chain-bound to the current Sui epoch. Sui diff --git a/scripts/verify-manual-sdk-release.mjs b/scripts/verify-manual-sdk-release.mjs index e13df509b..314758341 100644 --- a/scripts/verify-manual-sdk-release.mjs +++ b/scripts/verify-manual-sdk-release.mjs @@ -5,13 +5,13 @@ import { readFileSync } from "node:fs"; const releases = [ { name: "TypeScript SDK", - version: "0.1.6", + version: "0.1.7", manifests: [["packages/sdk/package.json", "version"]], changelogs: ["packages/sdk/CHANGELOG.md", "docs/sdk/changelog.mdx"], }, { name: "Python SDK", - version: "0.1.9", + version: "0.1.10", manifests: [ ["packages/python-sdk-memwal/pyproject.toml", "toml-version"], ["packages/python-sdk-memwal/memwal/__init__.py", "python-version"], @@ -23,7 +23,7 @@ const releases = [ }, { name: "MCP package", - version: "0.0.12", + version: "0.0.13", manifests: [ ["packages/mcp/package.json", "version"], [".claude-plugin/marketplace.json", "plugin-version"], @@ -63,6 +63,38 @@ for (const release of releases) { console.log(`${release.name} ${release.version}: manifests and changelogs synchronized`); } +const mcpVersion = JSON.parse(readFileSync("packages/mcp/package.json", "utf8")).version; +const expectedPluginArgs = ["-y", `@mysten-incubation/memwal-mcp@${mcpVersion}`]; +for (const pluginPath of [ + "packages/mcp/plugin/.mcp.json", + "packages/mcp/plugin/.cursor-mcp.json", + "packages/mcp/plugin/.codex-mcp.json", +]) { + const actual = JSON.parse(readFileSync(pluginPath, "utf8")).mcpServers.memwal.args; + if (JSON.stringify(actual) !== JSON.stringify(expectedPluginArgs)) { + throw new Error( + `${pluginPath}: expected ${JSON.stringify(expectedPluginArgs)}, received ${JSON.stringify(actual)}`, + ); + } +} +const installerPath = "packages/mcp/plugin/scripts/install_codex_hooks.mjs"; +const installer = readFileSync(installerPath, "utf8"); +const expectedPin = expectedPluginArgs[1]; +if (installer.includes('["-y", "@mysten-incubation/memwal-mcp"]')) { + throw new Error( + `${installerPath}: expected ${JSON.stringify(expectedPluginArgs)}, received ${JSON.stringify(["-y", "@mysten-incubation/memwal-mcp"])}`, + ); +} +if ( + !installer.includes(expectedPin) && + !installer.includes("@mysten-incubation/memwal-mcp@${") +) { + throw new Error( + `${installerPath}: expected ${JSON.stringify(expectedPluginArgs)}, received missing version pin`, + ); +} +console.log(`MCP package ${mcpVersion}: plugin npx args pin ${expectedPin}`); + function readVersion(content, kind) { if (kind === "version") return JSON.parse(content).version; if (kind === "plugin-version") return JSON.parse(content).plugins[0].version; diff --git a/scripts/write-migration-completion-artifact.mjs b/scripts/write-migration-completion-artifact.mjs new file mode 100644 index 000000000..f42af7d8d --- /dev/null +++ b/scripts/write-migration-completion-artifact.mjs @@ -0,0 +1,416 @@ +#!/usr/bin/env node +/** + * Write a migration completion artifact JSON file. + * + * node scripts/write-migration-completion-artifact.mjs \ + * --package-id 0x… --manifest-sha256 <64-hex> \ + * --imported N --skipped N --verified true \ + * --approver user:alice --out artifact.json + * + * Flags override env of the same name. Missing required fields exit 1. + * See docs/ops/migration-completion-artifact.md. + */ + +import { spawnSync } from "node:child_process"; +import { mkdtempSync, mkdirSync, readFileSync, rmSync, statSync, writeFileSync } from "node:fs"; +import { tmpdir } from "node:os"; +import path from "node:path"; +import { fileURLToPath } from "node:url"; + +const USAGE = `Write a migration completion artifact JSON file. + +Usage: + node scripts/write-migration-completion-artifact.mjs \\ + --package-id --manifest-sha256 <64-hex> \\ + --imported --skipped --verified \\ + --approver --out + +Flags (override env): + --package-id PACKAGE_ID target package id + --manifest-sha256 MANIFEST_SHA256 reviewed manifest digest + --imported IMPORTED imported count + --skipped SKIPPED skipped count + --verified VERIFIED verification result (true|false) + --approver APPROVER ceremony approver + --out OUT output JSON path + + --force replace an existing --out file (flag only, no env) + + --help, -h print this help + --self-test write/read round-trip and missing-field checks +`; + +const REQUIRED = [ + ["package-id", "PACKAGE_ID"], + ["manifest-sha256", "MANIFEST_SHA256"], + ["imported", "IMPORTED"], + ["skipped", "SKIPPED"], + ["verified", "VERIFIED"], + ["approver", "APPROVER"], + ["out", "OUT"], +]; + +// Every flag the parser accepts. An unrecognized --flag is a typo, and a typo +// on a value flag would otherwise fall through to the env var of the same name +// and record something the operator never typed. +const KNOWN_FLAGS = new Set([ + ...REQUIRED.map(([flag]) => flag), + "force", + "help", + "h", + "self-test", +]); + +function main(argv = process.argv.slice(2), env = process.env) { + const flags = parseArgv(argv); + if (flags.has("help") || flags.has("h")) { + process.stdout.write(USAGE); + return 0; + } + if (flags.has("self-test")) { + selfTest(); + return 0; + } + + const raw = Object.fromEntries( + REQUIRED.map(([flag, envName]) => [flag, valueOf(flags, env, flag, envName)]), + ); + const missing = REQUIRED.filter(([flag]) => raw[flag] === "").map( + ([flag, envName]) => `--${flag} (or ${envName})`, + ); + if (missing.length > 0) { + process.stderr.write(`missing required fields: ${missing.join(", ")}\n`); + return 1; + } + + let artifact; + try { + artifact = buildArtifact(raw); + } catch (error) { + process.stderr.write(`${error.message}\n`); + return 1; + } + + const outPath = path.resolve(raw.out); + if (statSync(outPath, { throwIfNoEntry: false })?.isDirectory()) { + // "wx" reports EEXIST for a directory, so without this the operator is + // told to pass --force, which then fails with EISDIR. + process.stderr.write(`--out is a directory, not a file: ${outPath}\n`); + return 1; + } + mkdirSync(path.dirname(outPath), { recursive: true }); + try { + // "wx" fails if the path exists: a completion artifact is an audit + // record, so replacing one has to be deliberate. + writeFileSync(outPath, `${JSON.stringify(artifact, null, 2)}\n`, { + flag: flags.has("force") ? "w" : "wx", + }); + } catch (error) { + if (error.code !== "EEXIST") throw error; + process.stderr.write( + `refusing to overwrite existing artifact: ${outPath}\n` + + "pass --force to replace it\n", + ); + return 1; + } + process.stdout.write(`wrote ${outPath}\n`); + return 0; +} + +function parseArgv(argv) { + const flags = new Map(); + for (let i = 0; i < argv.length; i += 1) { + const arg = argv[i]; + if (arg === "--help" || arg === "-h") { + flags.set("help", "true"); + continue; + } + if (arg === "--self-test") { + flags.set("self-test", "true"); + continue; + } + if (arg === "--force") { + flags.set("force", "true"); + continue; + } + if (!arg.startsWith("--")) { + throw new Error(`unexpected argument: ${arg}`); + } + const eq = arg.indexOf("="); + if (eq !== -1) { + const name = assertKnown(arg.slice(2, eq)); + flags.set(name, arg.slice(eq + 1)); + continue; + } + const name = assertKnown(arg.slice(2)); + const next = argv[i + 1]; + if (next === undefined || next.startsWith("--")) { + flags.set(name, ""); + continue; + } + flags.set(name, next); + i += 1; + } + return flags; +} + +function assertKnown(name) { + if (!KNOWN_FLAGS.has(name)) { + throw new Error(`unknown flag: --${name}`); + } + return name; +} + +function valueOf(flags, env, flag, envName) { + if (flags.has(flag)) return String(flags.get(flag) ?? "").trim(); + return String(env[envName] ?? "").trim(); +} + +function buildArtifact(raw) { + const packageId = parsePackageId(raw["package-id"]); + + const manifestSha256 = parseManifestSha256(raw["manifest-sha256"]); + const imported = parseCount(raw.imported, "imported"); + const skipped = parseCount(raw.skipped, "skipped"); + const verified = parseBoolean(raw.verified, "verified"); + const approver = raw.approver; + if (!approver) throw new Error("approver is required"); + + return { + packageId, + manifestSha256, + imported, + skipped, + verified, + approver, + timestamp: new Date().toISOString(), + }; +} + +/** + * Mirrors assertObjectId() in scripts/assertions.ts, which is + * isValidSuiObjectId() + normalizeSuiAddress() from @mysten/sui/utils: a Sui + * object id is exactly 32 bytes of hex, optionally 0x-prefixed, and normalizes + * to lowercase with the prefix. Reimplemented rather than imported because the + * CI job runs this under bare `node` with no install step. An artifact whose + * packageId does not round-trip to what build-finalize-tx.ts used is not + * evidence of anything, so reject instead of recording it verbatim. + */ +function parsePackageId(value) { + const hex = /^0[xX]/.test(value) ? value.slice(2) : value; + if (!/^[0-9a-fA-F]{64}$/.test(hex)) { + throw new Error( + "packageId must be a Sui object id: 32 bytes of hex (64 characters)," + + " optionally 0x-prefixed", + ); + } + return `0x${hex.toLowerCase()}`; +} + +function parseManifestSha256(value) { + if (!/^[0-9a-fA-F]{64}$/.test(value)) { + throw new Error("manifestSha256 must be a 64-character hex digest"); + } + return value.toLowerCase(); +} + +function parseCount(value, field) { + if (!/^(0|[1-9][0-9]*)$/.test(value)) { + throw new Error(`${field} must be a non-negative integer`); + } + const n = Number(value); + if (!Number.isSafeInteger(n)) { + throw new Error(`${field} must be a non-negative integer`); + } + return n; +} + +function parseBoolean(value, field) { + const normalized = value.toLowerCase(); + if (normalized === "true" || normalized === "1" || normalized === "yes") return true; + if (normalized === "false" || normalized === "0" || normalized === "no") return false; + throw new Error(`${field} must be true or false`); +} + +function selfTest() { + const self = fileURLToPath(import.meta.url); + const env = { ...process.env }; + for (const [, envName] of REQUIRED) delete env[envName]; + + const help = spawnSync(process.execPath, [self, "--help"], { + encoding: "utf8", + env, + }); + assert(help.status === 0, `help exit ${help.status}: ${help.stderr}`); + assert(help.stdout.includes("--package-id"), "help missing --package-id"); + assert(help.stdout.includes("--manifest-sha256"), "help missing --manifest-sha256"); + + const missing = spawnSync(process.execPath, [self], { encoding: "utf8", env }); + assert(missing.status === 1, `missing fields exit ${missing.status}`); + assert( + missing.stderr.includes("missing required fields"), + `missing-field stderr: ${missing.stderr}`, + ); + + const dir = mkdtempSync(path.join(tmpdir(), "migration-completion-artifact-")); + try { + const out = path.join(dir, "artifact.json"); + const packageId = `0x${"ab".repeat(32)}`; + const manifestSha256 = "a".repeat(64); + const write = spawnSync( + process.execPath, + [ + self, + "--package-id", + packageId, + "--manifest-sha256", + manifestSha256, + "--imported", + "3", + "--skipped", + "1", + "--verified", + "true", + "--approver", + "user:alice", + "--out", + out, + ], + { encoding: "utf8", env }, + ); + assert(write.status === 0, `write exit ${write.status}: ${write.stderr}`); + const before = readFileSync(out, "utf8"); + const artifact = JSON.parse(before); + assert(artifact.packageId === packageId, "packageId mismatch"); + assert(artifact.manifestSha256 === manifestSha256, "manifestSha256 mismatch"); + assert(artifact.imported === 3, "imported mismatch"); + assert(artifact.skipped === 1, "skipped mismatch"); + assert(artifact.verified === true, "verified mismatch"); + assert(artifact.approver === "user:alice", "approver mismatch"); + assert( + typeof artifact.timestamp === "string" && + /^\d{4}-\d{2}-\d{2}T\d{2}:\d{2}:\d{2}\.\d{3}Z$/.test(artifact.timestamp), + `timestamp invalid: ${artifact.timestamp}`, + ); + assert( + JSON.stringify(Object.keys(artifact).sort()) === + JSON.stringify([ + "approver", + "imported", + "manifestSha256", + "packageId", + "skipped", + "timestamp", + "verified", + ]), + `unexpected keys: ${Object.keys(artifact)}`, + ); + + // Same valid inputs as above, with per-case overrides appended. + const run = (extra) => + spawnSync( + process.execPath, + [ + self, + "--package-id", + packageId, + "--manifest-sha256", + manifestSha256, + "--imported", + "3", + "--skipped", + "1", + "--verified", + "true", + "--approver", + "user:alice", + ...extra, + ], + { encoding: "utf8", env }, + ); + + // A second write to the same path must not clobber the record. + const clobber = run(["--out", out]); + assert(clobber.status === 1, `clobber exit ${clobber.status}`); + assert( + clobber.stderr.includes("refusing to overwrite existing artifact"), + `clobber stderr: ${clobber.stderr}`, + ); + assert( + readFileSync(out, "utf8") === before, + "refused write still modified the artifact", + ); + + // --force is the deliberate replace. + const forced = run(["--out", out, "--skipped", "2", "--force"]); + assert(forced.status === 0, `force exit ${forced.status}: ${forced.stderr}`); + assert( + JSON.parse(readFileSync(out, "utf8")).skipped === 2, + "--force did not replace the artifact", + ); + + // A typo in a value flag must not fall through to the env var of the + // same name and record something the operator never typed. + const unknown = run(["--out", path.join(dir, "unknown.json"), "--aprover", "user:bob"]); + assert(unknown.status === 1, `unknown flag exit ${unknown.status}`); + assert( + unknown.stderr.includes("unknown flag: --aprover"), + `unknown flag stderr: ${unknown.stderr}`, + ); + + // A directory --out is caught before the overwrite check, which would + // otherwise tell the operator to pass --force. + const dirOut = run(["--out", dir]); + assert(dirOut.status === 1, `directory --out exit ${dirOut.status}`); + assert( + dirOut.stderr.includes("--out is a directory"), + `directory --out stderr: ${dirOut.stderr}`, + ); + + // packageId mirrors assertObjectId: reject non-ids and wrong lengths. + const badIds = ["not-a-package-id", "0x2", `0x${"ab".repeat(31)}`, `0x${"a".repeat(65)}`]; + for (const bad of badIds) { + const rejected = run(["--package-id", bad, "--out", path.join(dir, "bad.json")]); + assert(rejected.status === 1, `bad packageId ${bad} exit ${rejected.status}`); + assert( + rejected.stderr.includes("packageId must be a Sui object id"), + `bad packageId ${bad} stderr: ${rejected.stderr}`, + ); + } + + // ...and normalizes case and the 0x prefix the way finalize-tx does. + const normOut = path.join(dir, "normalized.json"); + const normalized = run([ + "--package-id", + "AB".repeat(32), + "--out", + normOut, + ]); + assert( + normalized.status === 0, + `normalize exit ${normalized.status}: ${normalized.stderr}`, + ); + assert( + JSON.parse(readFileSync(normOut, "utf8")).packageId === `0x${"ab".repeat(32)}`, + "packageId was not normalized to lowercase 0x form", + ); + } finally { + rmSync(dir, { recursive: true, force: true }); + } + + process.stdout.write("self-test OK\n"); +} + +function assert(condition, message) { + if (!condition) throw new Error(`self-test failed: ${message}`); +} + +const isMain = process.argv[1] && path.resolve(process.argv[1]) === fileURLToPath(import.meta.url); +if (isMain) { + try { + process.exit(main()); + } catch (error) { + process.stderr.write(`${error.message}\n`); + process.exit(1); + } +} diff --git a/services/server/Cargo.lock b/services/server/Cargo.lock index 698037809..1de3f58e5 100644 --- a/services/server/Cargo.lock +++ b/services/server/Cargo.lock @@ -429,6 +429,15 @@ dependencies = [ "syn 2.0.117", ] +[[package]] +name = "backon" +version = "1.6.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "cffb0e931875b666fc4fcb20fee52e9bbd1ef836fd9e9e04ec21555f9f85f7ef" +dependencies = [ + "fastrand", +] + [[package]] name = "base16ct" version = "0.2.0" @@ -2410,8 +2419,10 @@ checksum = "09d8f99a4090c89cc489a94833c901ead69bfbf3877b4867d5482e321ee875bc" dependencies = [ "arc-swap", "async-trait", + "backon", "bytes", "combine", + "futures", "futures-util", "itertools 0.13.0", "itoa", diff --git a/services/server/Cargo.toml b/services/server/Cargo.toml index 42cb134ba..7f36b72e5 100644 --- a/services/server/Cargo.toml +++ b/services/server/Cargo.toml @@ -79,7 +79,7 @@ futures = "0.3" async-trait = "0.1" # Rate limiting (Redis-backed) -redis = { version = "0.27", features = ["tokio-comp"] } +redis = { version = "0.27", features = ["tokio-comp", "connection-manager"] } # Utils uuid = { version = "1", features = ["v4", "serde"] } diff --git a/services/server/scripts/mcp/__tests__/health-relayer.test.ts b/services/server/scripts/mcp/__tests__/health-relayer.test.ts new file mode 100644 index 000000000..301421e3d --- /dev/null +++ b/services/server/scripts/mcp/__tests__/health-relayer.test.ts @@ -0,0 +1,100 @@ +import assert from "node:assert/strict"; +import test, { type TestContext } from "node:test"; +import { Client } from "@modelcontextprotocol/sdk/client/index.js"; +import { InMemoryTransport } from "@modelcontextprotocol/sdk/inMemory.js"; +import type { MemWalSession } from "../auth.js"; +import { resolveAuth } from "../auth.js"; +import { createMcpServer } from "../server.js"; + +// memwal_health reported status + version only. Nothing said WHICH relayer +// answered, so a config pointing at the wrong network looked perfectly healthy +// right up until the memories were missing. +// +// The value it reports must be a network identity. `relayerUrl` is not one: +// it is the address the sidecar dials, and the Rust parent fills it with +// loopback whenever an operator did not override it. Only an operator-supplied +// public origin reaches `publicRelayerUrl`, and only that is printed. + +const PUBLIC_ORIGIN = "https://relayer-staging.memory.walrus.xyz"; +const LOOPBACK = "http://127.0.0.1:8000"; + +const TOKEN = "test-sidecar-token-0123456789"; +const DELEGATE_KEY = "a".repeat(64); +const ACCOUNT_ID = `0x${"b".repeat(64)}`; + +function mcpHeaders(): Headers { + return new Headers({ + authorization: `Bearer ${DELEGATE_KEY}`, + "x-memwal-account-id": ACCOUNT_ID, + "x-memwal-internal-sidecar-token": TOKEN, + "x-memwal-internal-oauth-scope": "memwal:read", + }); +} + +/** Stubbed relayer health so the tool call stays offline. */ +const HEALTH_STUB = { health: async () => ({ status: "ok", version: "1.2.3" }) }; + +async function callHealth(t: TestContext, session: Partial): Promise { + const [clientTransport, serverTransport] = InMemoryTransport.createLinkedPair(); + const server = createMcpServer({ + oauthScope: "memwal:read", + memwal: HEALTH_STUB, + ...session, + } as unknown as MemWalSession); + const client = new Client({ name: "health-test", version: "1.0.0" }); + t.after(async () => { + await client.close(); + await server.close(); + }); + await server.connect(serverTransport); + await client.connect(clientTransport); + + const res = (await client.callTool({ name: "memwal_health", arguments: {} })) as { + content: { type: string; text: string }[]; + }; + return res.content.map((c) => c.text).join("\n"); +} + +/** + * Health text for a session built the way a real request builds one — through + * `resolveAuth`, not hand-assembled — with only the relayer round-trip stubbed. + */ +async function callHealthThroughResolveAuth( + t: TestContext, + serverUrl: string, + publicRelayerUrl?: string +): Promise { + process.env.SIDECAR_AUTH_TOKEN = TOKEN; + const { session } = await resolveAuth(mcpHeaders(), serverUrl, publicRelayerUrl); + return callHealth(t, { + ...session, + memwal: HEALTH_STUB as unknown as MemWalSession["memwal"], + }); +} + +test("memwal_health names the relayer origin the deployment published", async (t) => { + const text = await callHealthThroughResolveAuth(t, LOOPBACK, PUBLIC_ORIGIN); + assert.ok(text.includes(PUBLIC_ORIGIN), `public origin missing from health output:\n${text}`); + // Existing contract must survive. + assert.ok(text.includes("status=ok")); + assert.ok(text.includes("version=1.2.3")); +}); + +test("memwal_health does not report the loopback address as a network", async (t) => { + // The default deployment: the sidecar dials loopback and no operator + // supplied a public origin. Printing `relayer=http://127.0.0.1:8000` here + // is the "healthy on the wrong network" failure this tool exists to catch, + // so health must stay silent about the relayer instead. + const text = await callHealthThroughResolveAuth(t, LOOPBACK); + assert.ok(text.includes("status=ok"), `health broke without a public origin:\n${text}`); + assert.ok( + !text.includes("relayer="), + `health named a relayer it cannot vouch for:\n${text}` + ); + assert.ok(!text.includes(LOOPBACK), `health leaked the loopback dial address:\n${text}`); +}); + +test("memwal_health still answers when the session carries no relayer URL", async (t) => { + const text = await callHealth(t, { relayerUrl: undefined, publicRelayerUrl: undefined }); + assert.ok(text.includes("status=ok"), `health broke without a relayer URL:\n${text}`); +}); diff --git a/services/server/scripts/mcp/__tests__/restore.test.ts b/services/server/scripts/mcp/__tests__/restore.test.ts index 879d5ad63..2871eaa5c 100644 --- a/services/server/scripts/mcp/__tests__/restore.test.ts +++ b/services/server/scripts/mcp/__tests__/restore.test.ts @@ -68,3 +68,69 @@ test("memwal_restore treats an omitted legacy truncated field as false", () => { assert.match(text, /truncated=false/); assert.match(text, /not proof the sidecar saw every blob/); }); + +test("memwal_restore prints failed next to the other counts", () => { + const text = formatRestoreResult({ + namespace: "my-app", + total: 10, + restored: 7, + skipped: 0, + failed: 3, + truncated: false, + }); + + assert.match(text, /failed=3/); + assert.match(text, /restored=7/); + assert.match(text, /skipped=0/); +}); + +test("memwal_restore defaults omitted failed to 0", () => { + const text = formatRestoreResult({ + namespace: "legacy", + total: 1, + restored: 1, + skipped: 0, + truncated: false, + }); + + assert.match(text, /failed=0/); +}); + +test("memwal_restore hints retry not raise-limit when the page is only transients", () => { + const text = formatRestoreResult( + { + namespace: "my-app", + total: 10, + restored: 0, + skipped: 0, + failed: 0, + truncated: true, + }, + 10, + ); + + assert.match(text, /^Restore partially complete/); + assert.match(text, /failed=0/); + assert.match(text, /download\/embed blip/); + assert.match(text, /retry the same limit/); + assert.doesNotMatch(text, /increase limit and call again/); +}); + +test("memwal_restore still tells agents to raise limit for WALM-431 cap truncation", () => { + // Empty namespace, sidecar cap still expandable (limit < 20): skipped+failed + // is not short of total, so this is page/cap truncation, not an embed blip. + const text = formatRestoreResult( + { + namespace: "my-app", + total: 0, + restored: 0, + skipped: 0, + failed: 0, + truncated: true, + }, + 10, + ); + + assert.match(text, /increase limit and call again/); + assert.doesNotMatch(text, /download\/embed blip/); +}); diff --git a/services/server/scripts/mcp/__tests__/tool-duration-log.test.ts b/services/server/scripts/mcp/__tests__/tool-duration-log.test.ts new file mode 100644 index 000000000..591719159 --- /dev/null +++ b/services/server/scripts/mcp/__tests__/tool-duration-log.test.ts @@ -0,0 +1,204 @@ +/** + * A slow hop between the sidecar and the relayer used to leave no trace. + * + * The relayer's own latency histogram cannot show one: it starts counting when + * the request lands, so a minute spent reaching it reads there as a healthy few + * milliseconds. The sidecar logged `tool.call` on the way in and nothing on the + * way out, so the wait was invisible from both ends — it took polling the + * relayer's Prometheus counter against a wall clock to even locate it. + * + * `wrapTool` now times every call and says how long it took, loudly past a + * threshold. These tests pin that. + * + * IMPORTANT: `MCP_TOOL_SLOW_WARN_MS` is read when the module loads, so it is + * set before the dynamic import below rather than at the top of the file. + */ +import assert from "node:assert/strict"; +import test from "node:test"; +import type { MemWalSession } from "../auth.js"; + +const SLOW_THRESHOLD_MS = 150; +process.env.MCP_TOOL_SLOW_WARN_MS = String(SLOW_THRESHOLD_MS); +const { wrapTool } = await import("../tools/util.js"); + +const SESSION = { + accountId: `0x${"b".repeat(64)}`, + relayerUrl: "http://127.0.0.1:8000", + agentClient: "claude-code", +} as unknown as MemWalSession; + +interface LogLine { + level: string; + event: string; + [key: string]: unknown; +} + +/** Run `fn` with stderr captured, returning the structured lines it wrote. */ +async function capturingLogs(fn: () => Promise): Promise<{ + result: T; + lines: LogLine[]; +}> { + const written: string[] = []; + const original = process.stderr.write.bind(process.stderr); + (process.stderr as unknown as { write: unknown }).write = (chunk: unknown) => { + written.push(String(chunk)); + return true; + }; + try { + const result = await fn(); + return { result, lines: parse(written) }; + } finally { + (process.stderr as unknown as { write: unknown }).write = original; + } +} + +function parse(written: string[]): LogLine[] { + return written + .join("") + .split("\n") + .filter((l) => l.trim().startsWith("{")) + .map((l) => JSON.parse(l) as LogLine); +} + +const ok = async () => ({ content: [{ type: "text" as const, text: "fine" }] }); + +test("a completed tool call reports how long it took", async () => { + const { lines } = await capturingLogs(() => + wrapTool(SESSION, "memwal_health", ok)({}) + ); + + const done = lines.find((l) => l.event === "tool.done"); + assert.ok(done, `no tool.done line:\n${JSON.stringify(lines, null, 2)}`); + assert.equal(done.level, "info"); + assert.equal(done.tool, "memwal_health"); + assert.equal(typeof done.durationMs, "number"); + // The address dialled is the thing an operator has to change, so the line + // that reports the latency has to name it. + assert.equal(done.relayerUrl, "http://127.0.0.1:8000"); +}); + +test("a call slower than the threshold is warned about, not filed as normal", async () => { + const slow = async () => { + await new Promise((r) => setTimeout(r, SLOW_THRESHOLD_MS * 2)); + return ok(); + }; + + const { lines } = await capturingLogs(() => + wrapTool(SESSION, "memwal_health", slow)({}) + ); + + // Assert on the SETTLED line specifically. The in-flight line is emitted by + // the timer at the threshold itself, so its own durationMs sits within a + // millisecond or two of the threshold and can land just under it — that is + // timer resolution, not a defect, and asserting on it made this flaky in CI + // (`durationMs 149 is under the threshold` at a 150ms threshold). + const warned = lines.find((l) => l.event === "tool.slow" && l.settled === true); + assert.ok(warned, `no settled tool.slow line:\n${JSON.stringify(lines, null, 2)}`); + assert.equal(warned.level, "warn"); + assert.equal(warned.thresholdMs, SLOW_THRESHOLD_MS); + assert.ok( + (warned.durationMs as number) >= SLOW_THRESHOLD_MS, + `durationMs ${warned.durationMs} is under the threshold that triggered it` + ); + // A slow call is reported as slow — never also as a healthy one. + assert.equal(lines.filter((l) => l.event === "tool.done").length, 0); +}); + +test("a call still running past the threshold is reported BEFORE it settles", async () => { + // The whole point. The incident that motivated this left a tool call + // outstanding for 61s; a log that only fires on settle says nothing for the + // entire minute an operator is staring at the service. + let release; + const hang = () => + new Promise((resolve) => { + release = () => resolve(ok()); + }); + + const written = []; + const original = process.stderr.write.bind(process.stderr); + (process.stderr as unknown as { write: unknown }).write = (chunk: unknown) => { + written.push(String(chunk)); + return true; + }; + + let lines; + try { + const call = wrapTool(SESSION, "memwal_health", hang as never)({}); + // Wait past the threshold while the call is deliberately still pending. + await new Promise((r) => setTimeout(r, SLOW_THRESHOLD_MS * 2)); + lines = parse(written); + + const inflight = lines.find((l) => l.event === "tool.slow"); + assert.ok( + inflight, + `nothing was reported while the call was still running:\n${JSON.stringify(lines, null, 2)}` + ); + assert.equal(inflight.level, "warn"); + assert.equal(inflight.settled, false, "the in-flight line must say it has not settled"); + assert.equal(inflight.tool, "memwal_health"); + + release(); + await call; + } finally { + (process.stderr as unknown as { write: unknown }).write = original; + } + + // And when it finally lands, the settle line marks that it was already + // reported, so an operator counting warns does not double-count one call. + const settled = parse(written).filter((l) => l.event === "tool.slow" && l.settled === true); + assert.equal(settled.length, 1); + assert.equal(settled[0].alreadyWarned, true); +}); + +test("a failing tool call still reports its duration", async () => { + const boom = async (): Promise => { + throw new Error("relayer unreachable"); + }; + + const { result, lines } = await capturingLogs(() => + wrapTool(SESSION, "memwal_recall", boom)({}) + ); + + const failed = lines.find((l) => l.event === "tool.failed"); + assert.ok(failed, `no tool.failed line:\n${JSON.stringify(lines, null, 2)}`); + assert.equal(failed.level, "warn"); + assert.equal(failed.tool, "memwal_recall"); + assert.equal(typeof failed.durationMs, "number"); + // The structured line has to name the failure, or an operator reading logs + // learns only that something failed and must go hunting for the reason. + assert.equal(failed.errMessage, "relayer unreachable"); + assert.equal(failed.errName, "Error"); + // The existing error envelope is untouched. + assert.equal(result.isError, true); + assert.ok(result.content[0].text.includes("relayer unreachable")); +}); + +test("a credential in the dialled URL never reaches a log line", async () => { + // `session.relayerUrl` is whatever MEMWAL_SIDECAR_RELAYER_URL was set to, + // and it is echoed on every outcome line. The Rust side redacts its own + // startup lines; this is the per-call path, which is far noisier. + const withSecret = { + ...SESSION, + relayerUrl: "https://ops:hunter2@relayer.internal:8000", + } as unknown as MemWalSession; + + const { lines } = await capturingLogs(() => + wrapTool(withSecret, "memwal_health", ok)({}) + ); + + const done = lines.find((l) => l.event === "tool.done"); + assert.ok(done, "no tool.done line"); + assert.ok( + !JSON.stringify(lines).includes("hunter2"), + `a credential reached the log:\n${JSON.stringify(lines, null, 2)}` + ); + // Still useful: the host an operator has to change is preserved. + assert.match(String(done.relayerUrl), /relayer\.internal:8000/); +}); + +test("an unparseable dial URL is dropped rather than echoed", async () => { + const bad = { ...SESSION, relayerUrl: "not a url" } as unknown as MemWalSession; + const { lines } = await capturingLogs(() => wrapTool(bad, "memwal_health", ok)({})); + const done = lines.find((l) => l.event === "tool.done"); + assert.equal(done?.relayerUrl, null); +}); diff --git a/services/server/scripts/mcp/auth.ts b/services/server/scripts/mcp/auth.ts index bd862484b..451df2991 100644 --- a/services/server/scripts/mcp/auth.ts +++ b/services/server/scripts/mcp/auth.ts @@ -24,6 +24,16 @@ export interface MemWalSession { delegatePubKeyHex: string; namespace?: string; memwal: MemWal; + /** Relayer base URL the SDK dials. Loopback unless the deployment + * overrides it, so it is NOT a network identity — see + * `publicRelayerUrl`. MemWal keeps its own copy private, so we carry + * one alongside. */ + relayerUrl: string; + /** The relayer's public origin, when the deployment states one. + * `memwal_health` reports it so a client pointed at the wrong network + * sees that, rather than discovering it via missing memories. Unset + * when the sidecar only knows the loopback address it dials. */ + publicRelayerUrl?: string; authMethod: "delegate-key"; oauthScope?: string; /** Stable coding-agent id (`claude-code`, `codex`, `other`, …). */ @@ -106,7 +116,8 @@ function bytesToHex(b: Uint8Array): string { */ export async function resolveAuth( headers: Headers, - serverUrl: string + serverUrl: string, + publicRelayerUrl?: string ): Promise { // Runs before anything else reads the request: `x-memwal-internal-*` // headers carry decisions the relayer already made, so a caller that @@ -156,6 +167,8 @@ export async function resolveAuth( delegatePubKeyHex, namespace, memwal, + relayerUrl: serverUrl, + publicRelayerUrl, authMethod: "delegate-key", oauthScope, }; diff --git a/services/server/scripts/mcp/index.ts b/services/server/scripts/mcp/index.ts index 8c14b5c3c..fd2631d88 100644 --- a/services/server/scripts/mcp/index.ts +++ b/services/server/scripts/mcp/index.ts @@ -167,7 +167,8 @@ function expressHeadersToWeb(req: Request): Headers { async function handleSse( req: Request, res: Response, - relayerUrl: string + relayerUrl: string, + publicRelayerUrl: string | undefined ): Promise { // Rate limit BEFORE resolveAuth — see comment on `rateLimiter` above. // resolveAuth only checks header shape, so we must cap concurrent SSE @@ -187,7 +188,7 @@ async function handleSse( let auth: AuthResolution; try { - auth = await resolveAuth(expressHeadersToWeb(req), relayerUrl); + auth = await resolveAuth(expressHeadersToWeb(req), relayerUrl, publicRelayerUrl); } catch (err) { releaseSlot(); if (err instanceof McpAuthError) { @@ -287,7 +288,8 @@ async function handleSse( async function handlePostMessage( req: Request, res: Response, - relayerUrl: string + relayerUrl: string, + publicRelayerUrl: string | undefined ): Promise { const sessionId = typeof req.query.sessionId === "string" ? req.query.sessionId : undefined; if (!sessionId) { @@ -300,7 +302,7 @@ async function handlePostMessage( let auth: AuthResolution; try { - auth = await resolveAuth(expressHeadersToWeb(req), relayerUrl); + auth = await resolveAuth(expressHeadersToWeb(req), relayerUrl, publicRelayerUrl); } catch (err) { if (err instanceof McpAuthError) { res.setHeader( @@ -357,14 +359,15 @@ async function handlePostMessage( async function handleStreamableHttp( req: Request, res: Response, - relayerUrl: string + relayerUrl: string, + publicRelayerUrl: string | undefined ): Promise { // 1) Auth — bearer + accountId same as SSE path. Cheap to re-run per // request; resolveAuth's on-chain lookup is cached by the SDK once // we mint the Walrus Memory client per session. let auth: AuthResolution; try { - auth = await resolveAuth(expressHeadersToWeb(req), relayerUrl); + auth = await resolveAuth(expressHeadersToWeb(req), relayerUrl, publicRelayerUrl); } catch (err) { if (err instanceof McpAuthError) { res.setHeader( @@ -547,8 +550,20 @@ async function handleStreamableHttp( } export interface MountMcpOptions { - /** Relayer base URL that tool calls hit. Default: `http://localhost:3001`. */ + /** + * Relayer base URL that tool calls hit. Default: `http://127.0.0.1:3001` + * — loopback, because that is where the relayer this sidecar belongs to + * listens. Pointing it at a public origin sends every tool call out of + * the process and back through the edge. + */ relayerUrl?: string; + /** + * The relayer's public origin, when the deployment states one. Reported by + * `memwal_health` as the network the session is bound to. Deliberately + * separate from `relayerUrl`, which is only the address the sidecar dials + * and is loopback on every deployment that does not override it. + */ + publicRelayerUrl?: string; } /** @@ -567,11 +582,12 @@ export function mountMcpRoutes( app: Router, options: MountMcpOptions = {} ): void { - const relayerUrl = options.relayerUrl ?? "http://localhost:3001"; + const relayerUrl = options.relayerUrl ?? "http://127.0.0.1:3001"; + const publicRelayerUrl = options.publicRelayerUrl; app.get("/mcp/sse", async (req, res) => { try { - await handleSse(req, res, relayerUrl); + await handleSse(req, res, relayerUrl, publicRelayerUrl); } catch (err) { log.error("mcp.sse.error", { err: err instanceof Error ? err.message : String(err), @@ -589,7 +605,7 @@ export function mountMcpRoutes( // transport's internal raw-body parser. async (req, res) => { try { - await handlePostMessage(req, res, relayerUrl); + await handlePostMessage(req, res, relayerUrl, publicRelayerUrl); } catch (err) { log.error("mcp.post.error", { err: err instanceof Error ? err.message : String(err), @@ -613,7 +629,7 @@ export function mountMcpRoutes( // req.method. const streamableHandler = async (req: Request, res: Response) => { try { - await handleStreamableHttp(req, res, relayerUrl); + await handleStreamableHttp(req, res, relayerUrl, publicRelayerUrl); } catch (err) { log.error("mcp.streamable.error", { err: err instanceof Error ? err.message : String(err), @@ -634,6 +650,7 @@ export function mountMcpRoutes( "GET|POST|DELETE /mcp (streamable HTTP)", ], relayerUrl, + publicRelayerUrl: publicRelayerUrl ?? null, }); } diff --git a/services/server/scripts/mcp/tools/health.ts b/services/server/scripts/mcp/tools/health.ts index 6c73d5e96..b1ca2a8d0 100644 --- a/services/server/scripts/mcp/tools/health.ts +++ b/services/server/scripts/mcp/tools/health.ts @@ -18,23 +18,35 @@ export function registerHealthTool( { ...TOOL_METADATA.memwal_health, description: - "Quick connectivity check for Walrus Memory. Calls the relayer's lightweight health endpoint (no search, no decryption) and returns its status and version. Use this to confirm the server is reachable — do NOT use memwal_recall for health checks, which is a full and slow retrieval.", + "Quick connectivity check for Walrus Memory. Calls the relayer's lightweight health endpoint (no search, no decryption) and returns its status and version, plus the relayer origin when the deployment publishes one (use it to confirm which network — prod / staging / dev / local — this client is bound to). Use this to confirm the server is reachable — do NOT use memwal_recall for health checks, which is a full and slow retrieval.", inputSchema: {}, }, wrapTool>(session, "memwal_health", async () => { const result = await session.memwal.health(); - const extra = result as { write_ready?: boolean }; - const writeNote = + const extra = result as { + write_ready?: boolean; + writes?: string; + }; + const readyNote = extra.write_ready === false - ? " write_ready=false (relayer is up; encryption sidecar did not answer health)" + ? " write_ready=false (writes unavailable)" : extra.write_ready === true ? " write_ready=true" : ""; + // Only a deployment-supplied public origin, never `relayerUrl` + // — that one is the address this process dials, which is loopback + // unless overridden. Printing loopback as the network is how a + // client bound to the wrong relayer reads as correctly configured. + const relayerNote = session.publicRelayerUrl + ? ` relayer=${session.publicRelayerUrl}` + : ""; + const pausedNote = extra.writes === "paused" ? " writes=paused" : ""; + const writeNote = `${readyNote}${pausedNote}`; return { content: [ { type: "text", - text: `Walrus Memory is reachable. status=${result.status} version=${result.version}${writeNote}`, + text: `Walrus Memory is reachable. status=${result.status} version=${result.version}${relayerNote}${writeNote}`, }, ], }; diff --git a/services/server/scripts/mcp/tools/restore.ts b/services/server/scripts/mcp/tools/restore.ts index c77d1a3e8..bcf933f08 100644 --- a/services/server/scripts/mcp/tools/restore.ts +++ b/services/server/scripts/mcp/tools/restore.ts @@ -36,19 +36,29 @@ export function formatRestoreResult( total: number; restored: number; skipped: number; + failed?: number; truncated?: boolean; }, limit = 10, ): string { const truncated = result.truncated === true; + const failed = result.failed ?? 0; + // WALM-480: truncated + no success + uncounted blobs → transients + // (download/embed blip). Raising limit does not fix that; retry does. + const transientPage = + truncated && + result.restored === 0 && + result.skipped + failed < result.total; const hint = !truncated ? "\n truncated=false is not proof the sidecar saw every blob." - : limit < SIDECAR_CAP_SATURATES_AT_LIMIT - ? "\n ⚠️ More blobs remain to restore — increase limit and call again." - : "\n ⚠️ Sidecar cap is saturated — truncation follows this call's missing-blob page; truncated is not completeness (WALM-451 sourceCapped)."; + : transientPage + ? "\n ⚠️ This page did not restore (download/embed blip) — retry the same limit." + : limit < SIDECAR_CAP_SATURATES_AT_LIMIT + ? "\n ⚠️ More blobs remain to restore — increase limit and call again." + : "\n ⚠️ Sidecar cap is saturated — truncation follows this call's missing-blob page; truncated is not completeness (WALM-451 sourceCapped)."; return ( `${truncated ? "Restore partially complete" : "Restore page finished"} for namespace "${result.namespace}":\n` + - ` total=${result.total} restored=${result.restored} skipped=${result.skipped} truncated=${truncated}` + + ` total=${result.total} restored=${result.restored} skipped=${result.skipped} failed=${failed} truncated=${truncated}` + hint ); } @@ -62,7 +72,7 @@ export function registerRestoreTool( { ...TOOL_METADATA.memwal_restore, description: - "Recovery tool. Re-index a namespace from Walrus blobs back into the relayer's search index — use when memwal_recall unexpectedly returns nothing even though facts were saved before (e.g. on a new machine, a fresh relayer, or after switching servers). Returns counts plus truncated status — does not return memory texts. truncated=true is known-retryable-incomplete: raising limit expands the sidecar cap only while limit < 20; after the cap saturates, truncation follows this call's missing-blob page. truncated=false is not completeness; WALM-451 will add sourceCapped. Call memwal_recall afterwards to query the rebuilt index.", + "Recovery tool. Re-index a namespace from Walrus blobs back into the relayer's search index — use when memwal_recall unexpectedly returns nothing even though facts were saved before (e.g. on a new machine, a fresh relayer, or after switching servers). Returns restored/skipped/failed/total plus truncated — does not return memory texts. truncated=true is known-retryable-incomplete: retry the same limit on a download/embed blip; raising limit expands the sidecar cap only while limit < 20; after the cap saturates, truncation follows this call's missing-blob page. truncated=false is not completeness; WALM-451 will add sourceCapped. Call memwal_recall afterwards to query the rebuilt index.", inputSchema: RESTORE_INPUT, }, wrapTool<{ namespace: string; limit: number }>(session, "memwal_restore", async ({ namespace, limit }) => { diff --git a/services/server/scripts/mcp/tools/util.ts b/services/server/scripts/mcp/tools/util.ts index e253f5d60..7068c7fa2 100644 --- a/services/server/scripts/mcp/tools/util.ts +++ b/services/server/scripts/mcp/tools/util.ts @@ -46,6 +46,57 @@ export function explorerFooter(): string { return `Explorer: ${walruscanBlobUrl("")} for any blob_id above.`; } +const DEFAULT_SLOW_TOOL_WARN_MS = 5000; + +/** + * Above this, a tool call is reported at `warn` rather than `info`. + * Tuned to sit above a healthy `memwal_health` (single unsigned GET to the + * relayer, tens of milliseconds) and below a healthy `memwal_remember` + * (embed + SEAL + Walrus), so the threshold catches a slow hop to the + * relayer without crying about work that is slow by nature. + * + * Read once, because it cannot change mid-process — but validated, because a + * typo'd value must not silently disable the only signal this file adds. + * `Number.parseInt` alone accepts "5s" as 5 and yields NaN for "" or "abc", + * and every NaN comparison is false, so an unvalidated parse turns the warning + * off without saying anything. + */ +const SLOW_TOOL_WARN_MS = (() => { + const raw = process.env.MCP_TOOL_SLOW_WARN_MS; + if (raw === undefined || raw.trim() === "") return DEFAULT_SLOW_TOOL_WARN_MS; + const parsed = Number(raw); + if (!Number.isFinite(parsed) || parsed <= 0) { + log.warn("tool.slow_threshold_invalid", { + value: raw, + usingMs: DEFAULT_SLOW_TOOL_WARN_MS, + }); + return DEFAULT_SLOW_TOOL_WARN_MS; + } + return parsed; +})(); + +/** + * Strip any `user:password@` before a URL reaches a log line. + * + * The dial address is echoed on every `tool.done` / `tool.slow` / `tool.failed` + * line, and an operator is free to have put a credential in + * `MEMWAL_SIDECAR_RELAYER_URL`. Mirrors `redact_url_userinfo` in main.rs, which + * only covers the Rust-side startup lines. An unparseable value is dropped + * rather than echoed: it cannot be redacted, so it cannot be shown. + */ +export function redactUrlUserinfo(url: string | undefined | null): string | null { + if (!url) return null; + try { + const parsed = new URL(url); + if (!parsed.username && !parsed.password) return url; + parsed.username = ""; + parsed.password = ""; + return parsed.toString(); + } catch { + return null; + } +} + export function wrapTool( session: MemWalSession, tool: string, @@ -58,9 +109,68 @@ export function wrapTool( clientName: session.clientName ?? null, accountId: session.accountId ?? null, }); + // How long the call takes is the only number that shows a slow hop to + // the relayer. Per-request latency on the relayer side cannot: it + // measures the request once it lands, so a minute spent reaching it + // reads as a healthy few milliseconds there and as silence here. + const startedAt = Date.now(); + const durationMs = () => Date.now() - startedAt; + const outcomeFields = () => ({ + tool, + durationMs: durationMs(), + relayerUrl: redactUrlUserinfo(session.relayerUrl), + agentClient: session.agentClient ?? null, + accountId: session.accountId ?? null, + }); + + // Fire WHILE the call is still running, not when it settles. A hang is + // precisely the case that never settles: the incident this timing was + // written for left a tool call outstanding for 61s and the process + // emitted nothing until it finally returned. A settle-only log would + // have stayed silent for the whole minute an operator was looking. + // `unref` so a pending timer can never hold the sidecar open. + // Set by the timer itself. `Timeout.hasRef()` cannot stand in for it: + // that reports false from the moment `unref()` is called, while the + // timer is still pending, so reading it would mark every settled call + // as already-warned. + let warnedInFlight = false; + const watchdog = setTimeout(() => { + warnedInFlight = true; + log.warn("tool.slow", { + ...outcomeFields(), + thresholdMs: SLOW_TOOL_WARN_MS, + settled: false, + }); + }, SLOW_TOOL_WARN_MS); + watchdog.unref?.(); + const stopWatchdog = () => clearTimeout(watchdog); + try { - return await handler(args); + const result = await handler(args); + stopWatchdog(); + const elapsed = durationMs(); + if (elapsed >= SLOW_TOOL_WARN_MS) { + log.warn("tool.slow", { + ...outcomeFields(), + thresholdMs: SLOW_TOOL_WARN_MS, + settled: true, + alreadyWarned: warnedInFlight, + }); + } else { + log.info("tool.done", outcomeFields()); + } + return result; } catch (err: any) { + stopWatchdog(); + // Name the failure in the structured line too. Without this the log + // says a call failed and the operator still has to go find the + // separate console.error below to learn how. + log.warn("tool.failed", { + ...outcomeFields(), + errName: err?.constructor?.name ?? "Error", + errMessage: err?.message ?? String(err), + causeCode: err?.cause?.code ?? null, + }); const name = err?.constructor?.name ?? "Error"; const msg = err?.message ?? String(err); const cause = err?.cause; diff --git a/services/server/scripts/sidecar/app.ts b/services/server/scripts/sidecar/app.ts index 8960358ef..b9b9eadb2 100644 --- a/services/server/scripts/sidecar/app.ts +++ b/services/server/scripts/sidecar/app.ts @@ -56,7 +56,16 @@ export function createSidecarApp(mode: "full" | "writer" = SIDECAR_ROUTE_MODE): // enough to claim relayer-issued privileges (GH #685). if (mode === "full") { mountMcpRoutes(app, { - relayerUrl: process.env.MEMWAL_RELAYER_URL ?? "http://localhost:3001", + // `127.0.0.1`, not `localhost`: the managed sidecar always gets an + // explicit value from the Rust parent, so this default only covers + // a standalone run — and there a dual-stack `localhost` can resolve + // to an address nothing answers on, turning every tool call into a + // connect timeout. + relayerUrl: process.env.MEMWAL_RELAYER_URL ?? "http://127.0.0.1:3001", + // The origin `memwal_health` may name. Independent of the address + // above, so naming the network never redirects tool calls through + // the public edge. Absent means this deployment names no network. + publicRelayerUrl: process.env.MEMWAL_PUBLIC_RELAYER_URL, }); } diff --git a/services/server/src/alerts.rs b/services/server/src/alerts.rs index b1e90f63e..208a7883d 100644 --- a/services/server/src/alerts.rs +++ b/services/server/src/alerts.rs @@ -19,6 +19,8 @@ const WALRUS_QUEUE_SATURATION_ALERT_DEDUP_SECS_ENV: &str = const WALRUS_QUEUE_SATURATION_ALERT_DEDUP_DEFAULT: Duration = Duration::from_secs(1800); const WALLET_BALANCE_LOW_ALERT_DEDUP_SECS_ENV: &str = "WALLET_BALANCE_LOW_ALERT_DEDUP_SECS"; const WALLET_BALANCE_LOW_ALERT_DEDUP_DEFAULT: Duration = Duration::from_secs(43200); +const POSTGRES_STORAGE_ALERT_DEDUP_SECS_ENV: &str = "POSTGRES_STORAGE_ALERT_DEDUP_SECS"; +const POSTGRES_STORAGE_ALERT_DEDUP_DEFAULT: Duration = Duration::from_secs(1800); /// Mirrors the `@mysten/walrus` dep version in /// `services/server/scripts/package.json`. Bump this constant in lockstep @@ -97,6 +99,9 @@ pub struct AlertManager { /// Suppresses wallet balance low spam. Keyed by `(wallet_type:token, address)` /// so WAL and SUI can each alert once for the same wallet per dedup window. wallet_balance_low_dedup: AlertDedup, + /// Suppresses Postgres disk / Neon project-size-cap spam. The failure is + /// cluster-wide, so one notification per network per window — not per job. + postgres_storage_dedup: AlertDedup, } impl AlertManager { @@ -131,6 +136,10 @@ impl AlertManager { WALLET_BALANCE_LOW_ALERT_DEDUP_SECS_ENV, WALLET_BALANCE_LOW_ALERT_DEDUP_DEFAULT, )), + postgres_storage_dedup: AlertDedup::new(dedup_window_from_env( + POSTGRES_STORAGE_ALERT_DEDUP_SECS_ENV, + POSTGRES_STORAGE_ALERT_DEDUP_DEFAULT, + )), } } @@ -265,6 +274,25 @@ impl AlertManager { slack.send_payload(&payload).await } + pub async fn notify_postgres_storage_exhausted( + &self, + alert: PostgresStorageExhaustedAlert, + ) -> Result<(), AlertError> { + let Some(slack) = &self.slack else { + return Ok(()); + }; + // Cluster-wide cap: one notification per network per window. Concurrent + // remember/analyze jobs all hit the same Neon/Postgres size limit. + if self + .postgres_storage_dedup + .should_suppress(postgres_storage_dedup_key(&alert.sui_network)) + { + return Ok(()); + } + let payload = SlackPayload::for_postgres_storage_exhausted(&alert); + slack.send_payload(&payload).await + } + fn should_suppress_wallet_balance_low(&self, alert: &WalletBalanceLowAlert) -> bool { self.wallet_balance_low_dedup .should_suppress(wallet_balance_low_dedup_key(alert)) @@ -281,6 +309,57 @@ fn wallet_balance_low_dedup_key(alert: &WalletBalanceLowAlert) -> (String, Strin ) } +fn postgres_storage_dedup_key(sui_network: &str) -> (String, String) { + (sui_network.to_string(), "postgres-storage".to_string()) +} + +/// True when Postgres (or Neon) refused a write because the disk / project +/// size cap is exhausted. Matches the prod Neon message +/// `could not extend file because project size limit (3072 MB) has been exceeded` +/// plus vanilla `no space left on device`. sqlx 0.8 `Display` is message-only, +/// so SQLSTATE `53100` is matched via `DatabaseError::code` — not as a +/// substring of the message. +pub fn is_postgres_storage_exhausted(msg: &str) -> bool { + let lower = msg.to_ascii_lowercase(); + lower.contains("could not extend file") + || lower.contains("project size limit") + || lower.contains("no space left on device") +} + +pub fn sqlx_error_is_postgres_storage_exhausted(err: &sqlx::Error) -> bool { + if let Some(db) = err.as_database_error() { + if db.code().as_deref() == Some("53100") { + return true; + } + if is_postgres_storage_exhausted(db.message()) { + return true; + } + } + is_postgres_storage_exhausted(&err.to_string()) +} + +/// Slack the cluster-wide disk / Neon size-cap incident when `err` is +/// SQLSTATE `53100` or matches the string classifier. +pub async fn maybe_alert_sqlx_postgres_storage_exhausted( + alerts: &AlertManager, + sui_network: &str, + err: &sqlx::Error, +) { + if !sqlx_error_is_postgres_storage_exhausted(err) { + return; + } + let alert = PostgresStorageExhaustedAlert { + sui_network: sui_network.to_string(), + error: err.to_string(), + }; + if let Err(alert_err) = alerts.notify_postgres_storage_exhausted(alert).await { + tracing::warn!( + "failed to send Slack alert for Postgres storage exhaustion: {}", + alert_err + ); + } +} + /// Read a dedup window (seconds) from `env_var`, falling back to `default` /// when unset, unparseable, or zero. fn dedup_window_from_env(env_var: &str, default: Duration) -> Duration { @@ -450,6 +529,15 @@ pub struct WalletBalanceLowAlert { pub wallet_index: Option, } +/// Fired when Postgres cannot extend a file — Neon `project size limit` +/// / SQLSTATE `53100` / `no space left on device`. Cluster-wide, not a +/// per-user storage quota. +#[derive(Debug, Clone)] +pub struct PostgresStorageExhaustedAlert { + pub sui_network: String, + pub error: String, +} + #[derive(Debug)] pub enum AlertError { Transport(String), @@ -837,6 +925,42 @@ If the wallet is being topped up, rotate or temporarily remove that key from poo ], } } + + fn for_postgres_storage_exhausted(alert: &PostgresStorageExhaustedAlert) -> Self { + let title = "MemWal Postgres storage exhausted".to_string(); + let summary = format!( + "Postgres cannot accept writes on {}: the database disk/project size cap has been reached. \ + Writes (remember/analyze) are failing. GET /health write_ready will be false. \ + This is the database disk/project size cap, not a user quota; do not tell users to send SUI/WAL.", + alert.sui_network, + ); + let action = "*Action (ops):* raise the Neon project size limit or reclaim disk. \ +This is not a per-user storage quota and is not a Walrus/SUI/WAL funding issue." + .to_string(); + let details = format!( + "*Network:* `{}`\n*Error:* ```{}```", + alert.sui_network, + truncate(&alert.error, MAX_SLACK_ERROR_LEN), + ); + + Self { + text: summary.clone(), + blocks: vec![ + SlackBlock::Header { + text: plain_text(title), + }, + SlackBlock::Section { + text: mrkdwn(summary), + }, + SlackBlock::Section { + text: mrkdwn(action), + }, + SlackBlock::Section { + text: mrkdwn(details), + }, + ], + } + } } fn plain_text(text: String) -> SlackText { @@ -1321,4 +1445,113 @@ mod tests { let amount = format_token_amount(1_100_000_000); assert_eq!(amount, "1.1"); } + + #[test] + fn is_postgres_storage_exhausted_matches_prod_neon_message() { + let prod = "could not extend file because project size limit (3072 MB) has been exceeded"; + assert!(is_postgres_storage_exhausted(prod)); + assert!(is_postgres_storage_exhausted(&prod.to_ascii_uppercase())); + assert!(is_postgres_storage_exhausted(&format!( + "Internal Error: Failed to insert reservation: error returned from database: {prod}" + ))); + assert!(is_postgres_storage_exhausted( + "ERROR: could not extend file \"base/16384/12345\": No space left on device" + )); + assert!(!is_postgres_storage_exhausted("sqlstate 53100 disk_full")); + assert!(!is_postgres_storage_exhausted( + "duplicate key value violates unique constraint" + )); + assert!(!is_postgres_storage_exhausted("Storage quota exceeded")); + } + + #[test] + fn sqlx_error_is_postgres_storage_exhausted_matches_sqlstate() { + let disk_full = sqlx::Error::Database(Box::new(FakePgError { + message: "the wording changed in a future postgres", + code: Some("53100"), + })); + assert!(sqlx_error_is_postgres_storage_exhausted(&disk_full)); + + let other = sqlx::Error::Database(Box::new(FakePgError { + message: "duplicate key value violates unique constraint", + code: Some("23505"), + })); + assert!(!sqlx_error_is_postgres_storage_exhausted(&other)); + } + + #[test] + fn postgres_storage_exhausted_payload_names_write_outage_not_user_quota() { + let payload = + SlackPayload::for_postgres_storage_exhausted(&PostgresStorageExhaustedAlert { + sui_network: "mainnet".into(), + error: + "could not extend file because project size limit (3072 MB) has been exceeded" + .into(), + }); + + let json = serde_json::to_string(&payload).unwrap(); + assert!(json.contains("MemWal Postgres storage exhausted")); + assert!(json.contains("mainnet")); + assert!(json.contains("remember/analyze")); + assert!(json.contains("write_ready")); + assert!(json.contains("not a user quota")); + assert!(json.contains("do not tell users to send SUI/WAL")); + assert!(json.contains("project size limit (3072 MB)")); + assert!(!json.to_lowercase().contains("exhausted retries")); + } + + #[test] + fn postgres_storage_dedup_is_per_network_not_per_job() { + assert_eq!( + postgres_storage_dedup_key("mainnet"), + ("mainnet".to_string(), "postgres-storage".to_string()) + ); + + // Do not read POSTGRES_STORAGE_ALERT_DEDUP_SECS: a real env value + // would change the window (or make the second fire miss the window). + let dedup = AlertDedup::new(POSTGRES_STORAGE_ALERT_DEDUP_DEFAULT); + assert!(!dedup.should_suppress(postgres_storage_dedup_key("mainnet"))); + assert!(dedup.should_suppress(postgres_storage_dedup_key("mainnet"))); + assert!(!dedup.should_suppress(postgres_storage_dedup_key("testnet"))); + } + + #[derive(Debug)] + struct FakePgError { + message: &'static str, + code: Option<&'static str>, + } + + impl std::fmt::Display for FakePgError { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + f.write_str(self.message) + } + } + + impl std::error::Error for FakePgError {} + + impl sqlx::error::DatabaseError for FakePgError { + fn message(&self) -> &str { + self.message + } + + fn kind(&self) -> sqlx::error::ErrorKind { + sqlx::error::ErrorKind::Other + } + + fn code(&self) -> Option> { + self.code.map(std::borrow::Cow::Borrowed) + } + + fn as_error(&self) -> &(dyn std::error::Error + Send + Sync + 'static) { + self + } + + fn as_error_mut(&mut self) -> &mut (dyn std::error::Error + Send + Sync + 'static) { + self + } + + fn into_error(self: Box) -> Box { + self + } + } } diff --git a/services/server/src/auth.rs b/services/server/src/auth.rs index 4317afbf4..766e73615 100644 --- a/services/server/src/auth.rs +++ b/services/server/src/auth.rs @@ -11,7 +11,7 @@ use std::sync::Arc; use crate::owner_token_auth; use crate::storage::sui::{ - find_account_by_delegate_key, verify_delegate_key_onchain, OnchainVerifyError, + find_account_by_delegate_key, verify_delegate_key_cached, OnchainVerifyError, }; use crate::types::{AppState, AuthInfo}; @@ -441,12 +441,20 @@ async fn resolve_account( if let Ok(Some((cached_account_id, _cached_owner))) = state.db.get_cached_account(public_key_hex).await { - // Re-verify the cached mapping on-chain when Sui is reachable. - // A transient RPC failure is *not* a revoke: keep the row, but - // fail closed with 503 so a revoked key cannot ride a 24h cache - // through a Sui outage. Definitive misses evict. + // Re-verify the cached mapping, through the in-memory verify cache. + // A hit inside `DELEGATE_VERIFY_CACHE_TTL` answers without touching + // Sui at all — including during an outage — so the fail-closed rule + // below now applies to a live read, not to every request. That 30s + // window is the stated revocation bound. + // + // On a live miss: a transient RPC failure is *not* a revoke, so keep + // the Postgres row and fail closed with 503 rather than let a revoked + // key ride the 24h cache through an outage. A definitive miss evicts + // both caches. match cache_reverify_action( - verify_delegate_key_onchain( + verify_delegate_key_cached( + &state.delegate_verify_cache, + &state.delegate_reject_cache, &state.http_client, &state.config.sui_rpc_url, state.sui_grpc_client.as_ref(), @@ -494,7 +502,9 @@ async fn resolve_account( .as_deref() .or(state.config.memwal_account_id.as_deref()) { - match verify_delegate_key_onchain( + match verify_delegate_key_cached( + &state.delegate_verify_cache, + &state.delegate_reject_cache, &state.http_client, &state.config.sui_rpc_url, state.sui_grpc_client.as_ref(), diff --git a/services/server/src/engine/walrus_seal.rs b/services/server/src/engine/walrus_seal.rs index aa9f85fbc..54c0e25a6 100644 --- a/services/server/src/engine/walrus_seal.rs +++ b/services/server/src/engine/walrus_seal.rs @@ -50,7 +50,7 @@ pub struct WalrusSealEngine { http_client: reqwest::Client, key_pool: Arc, config: Arc, - redis: redis::aio::MultiplexedConnection, + redis: redis::aio::ConnectionManager, /// Blob ciphertext cache TTL. Zero disables write-back. blob_cache_ttl: Duration, /// Max ciphertext size kept in the Redis cache. Reads ignore @@ -66,7 +66,7 @@ impl WalrusSealEngine { http_client: reqwest::Client, key_pool: Arc, config: Arc, - redis: redis::aio::MultiplexedConnection, + redis: redis::aio::ConnectionManager, blob_cache_ttl: Duration, blob_cache_max_bytes: usize, ) -> Self { diff --git a/services/server/src/main.rs b/services/server/src/main.rs index 5d4614dfc..292bdd6c1 100644 --- a/services/server/src/main.rs +++ b/services/server/src/main.rs @@ -94,6 +94,225 @@ fn relayer_cors(origins: Vec) -> CorsLayer { ]) } +/// Where the managed sidecar dials this relayer, and which public origin it names. +/// +/// These are two different questions and they used to share one answer: +/// `MEMWAL_RELAYER_URL` was both. That variable cannot simply be unset to break +/// the tie — it is also the MCP OAuth issuer (`oauth.rs:161`, where a missing +/// value makes `McpOAuthConfig::from_env` return `None` and every OAuth +/// handshake refuse with `OauthNotConfigured`), so every deployment must keep +/// it set. The consequence was that every deployment also dialled its own +/// public hostname for every MCP tool call: each `memwal_health`, +/// `memwal_recall` and `memwal_remember` left the container, crossed the public +/// edge, and came back to the process it started from. +/// +/// So the dial address is the half that moves. It is now loopback unless an +/// operator explicitly overrides it with `MEMWAL_SIDECAR_RELAYER_URL`, which +/// almost nobody should: the managed sidecar is a child of this process and the +/// relayer it needs is this one. `MEMWAL_RELAYER_URL` keeps its meaning as the +/// deployment's public identity, so OAuth and `memwal_health` are untouched and +/// no deployment has to change an environment variable to stop paying the round +/// trip. +#[derive(Debug, PartialEq, Eq)] +struct SidecarRelayerUrls { + /// Address the sidecar dials for every MCP tool call. Loopback unless + /// explicitly overridden. + dial: String, + /// Public origin `memwal_health` may report. Never the loopback default: + /// an address that names no network is how a client bound to the wrong + /// relayer reads as correctly configured. + public: Option, + /// Startup lines to log at `warn`. Non-empty only when an operator has + /// explicitly pointed the dial off this host, which costs a public round + /// trip on every single tool call. + warnings: Vec, +} + +/// True when `url`'s host is this machine, so dialling it stays in-process. +/// +/// Handles the spellings this system actually produces: a bare `localhost` +/// (with or without the trailing dot a resolver may hand back), dotted IPv4 +/// anywhere in `127.0.0.0/8`, `[::1]`, and the IPv4-mapped `[::ffff:127.0.0.1]` +/// form that a dual-stack listener reports for an IPv4 peer. +fn is_loopback_relayer_url(url: &str) -> bool { + let Ok(parsed) = url::Url::parse(url) else { + return false; + }; + match parsed.host() { + Some(url::Host::Domain(host)) => { + let host = host.strip_suffix('.').unwrap_or(host); + host.eq_ignore_ascii_case("localhost") + } + Some(url::Host::Ipv4(ip)) => ip.is_loopback(), + Some(url::Host::Ipv6(ip)) => { + ip.is_loopback() || ip.to_ipv4_mapped().is_some_and(|v4| v4.is_loopback()) + } + None => false, + } +} + +/// Strip any `user:password@` before a URL is logged. The dial address is +/// echoed in startup warnings and in every per-call sidecar log line, and an +/// operator is free to have put a credential in it. +fn redact_url_userinfo(url: &str) -> String { + match url::Url::parse(url) { + Ok(mut parsed) if !parsed.username().is_empty() || parsed.password().is_some() => { + let _ = parsed.set_username(""); + let _ = parsed.set_password(None); + parsed.to_string() + } + _ => url.to_string(), + } +} + +fn resolve_sidecar_relayer_urls( + dial_override_env: Option, + public_env: Option, + relayer_url_env: Option, + port: u16, +) -> SidecarRelayerUrls { + // Loopback unless an operator explicitly asked for something else. This is + // the behaviour change: `MEMWAL_RELAYER_URL` no longer steers the dial. + let dial = + dial_override_env + .clone() + .unwrap_or_else(|| format!("http://127.0.0.1:{}", port)); + + // An explicit public origin wins; `MEMWAL_RELAYER_URL` remains the fallback + // so `memwal_health` names exactly what it names today on every deployment. + // The loopback default is never reported. + let public = public_env.or(relayer_url_env); + + let mut warnings = Vec::new(); + if !is_loopback_relayer_url(&dial) { + let shown = redact_url_userinfo(&dial); + warnings.push(format!( + "⚠️ MEMWAL_SIDECAR_RELAYER_URL={shown} is not loopback — the sidecar dials it for EVERY MCP tool call." + )); + warnings.push( + "⚠️ Each memwal_* call then leaves this container and returns through the public edge." + .to_string(), + ); + warnings.push( + "⚠️ Unset it unless this sidecar genuinely serves a relayer in another process." + .to_string(), + ); + } + + SidecarRelayerUrls { + dial, + public, + warnings, + } +} + +#[cfg(test)] +mod sidecar_relayer_url_tests { + use super::*; + + const PUBLIC: &str = "https://relayer.dev.memwal.ai"; + + #[test] + fn the_deployed_shape_now_dials_loopback_while_naming_the_same_network() { + // Every deployed environment sets MEMWAL_RELAYER_URL to its own public + // hostname and nothing else. Before this change that address was also + // the dial target, so every tool call took a public round trip. + let urls = resolve_sidecar_relayer_urls(None, None, Some(PUBLIC.to_string()), 3001); + assert_eq!(urls.dial, "http://127.0.0.1:3001"); + // Unchanged: health still names exactly what it named before, and the + // OAuth issuer that reads the same variable is untouched. + assert_eq!(urls.public.as_deref(), Some(PUBLIC)); + assert!(urls.warnings.is_empty()); + } + + #[test] + fn nothing_set_dials_loopback_and_names_no_network() { + let urls = resolve_sidecar_relayer_urls(None, None, None, 8000); + assert_eq!(urls.dial, "http://127.0.0.1:8000"); + // Reporting loopback as the network is how a client bound to the wrong + // relayer reads as correctly configured. + assert_eq!(urls.public, None); + assert!(urls.warnings.is_empty()); + } + + #[test] + fn an_explicit_public_origin_wins_over_the_relayer_url() { + let urls = resolve_sidecar_relayer_urls( + None, + Some("https://memory.example".to_string()), + Some(PUBLIC.to_string()), + 8000, + ); + assert_eq!(urls.public.as_deref(), Some("https://memory.example")); + assert_eq!(urls.dial, "http://127.0.0.1:8000"); + } + + #[test] + fn only_the_dedicated_override_can_move_the_dial_off_this_host() { + let urls = resolve_sidecar_relayer_urls( + Some("https://relayer.elsewhere.test".to_string()), + None, + Some(PUBLIC.to_string()), + 8000, + ); + assert_eq!(urls.dial, "https://relayer.elsewhere.test"); + // And it says so, because it costs a round trip per tool call. + assert_eq!(urls.warnings.len(), 3); + assert!(urls.warnings[0].contains("EVERY MCP tool call")); + assert!(urls + .warnings + .iter() + .any(|w| w.contains("MEMWAL_SIDECAR_RELAYER_URL"))); + } + + #[test] + fn an_explicit_loopback_override_is_not_warned_about() { + let urls = + resolve_sidecar_relayer_urls(Some("http://localhost:9000".to_string()), None, None, 8000); + assert_eq!(urls.dial, "http://localhost:9000"); + assert!(urls.warnings.is_empty()); + } + + #[test] + fn a_credential_in_the_dial_url_is_never_logged() { + let urls = resolve_sidecar_relayer_urls( + Some("https://ops:hunter2@relayer.elsewhere.test".to_string()), + None, + None, + 8000, + ); + // The sidecar still dials the real thing... + assert!(urls.dial.contains("hunter2")); + // ...but nothing that reaches a log carries the secret. + for line in &urls.warnings { + assert!(!line.contains("hunter2"), "warning leaked userinfo: {line}"); + } + } + + #[test] + fn every_loopback_spelling_is_recognised_as_in_process() { + for url in [ + "http://127.0.0.1:8000", + "http://127.0.0.2:8000", + "http://localhost:3001", + "http://LOCALHOST:3001", + "http://localhost.:3001", + "http://[::1]:8000", + "http://[::ffff:127.0.0.1]:8000", + ] { + assert!(is_loopback_relayer_url(url), "{url} should be loopback"); + } + for url in [ + "https://relayer.dev.memwal.ai", + "http://relayer.railway.internal:8000", + "http://100.64.0.1:8000", + "not a url", + ] { + assert!(!is_loopback_relayer_url(url), "{url} should not be loopback"); + } + } +} + #[cfg(test)] mod cors_tests { use super::*; @@ -706,14 +925,35 @@ async fn main() { let scripts_dir = std::env::var("SIDECAR_SCRIPTS_DIR") .map(std::path::PathBuf::from) .unwrap_or_else(|_| std::path::Path::new(env!("CARGO_MANIFEST_DIR")).join("scripts")); - let mcp_relayer_url = std::env::var("MEMWAL_RELAYER_URL") - .unwrap_or_else(|_| format!("http://127.0.0.1:{}", config.port)); - let mut sidecar_child = tokio::process::Command::new("npx") + // `MEMWAL_RELAYER_URL` is this deployment's public identity — the OAuth + // issuer and the network `memwal_health` names — and stays that. What the + // sidecar DIALS is now loopback unless `MEMWAL_SIDECAR_RELAYER_URL` says + // otherwise, so no deployment pays a public round trip per tool call and + // none has to change an env var to stop. + let relayer_urls = resolve_sidecar_relayer_urls( + std::env::var("MEMWAL_SIDECAR_RELAYER_URL").ok(), + std::env::var("MEMWAL_PUBLIC_RELAYER_URL").ok(), + std::env::var("MEMWAL_RELAYER_URL").ok(), + config.port, + ); + for line in &relayer_urls.warnings { + tracing::warn!("{}", line); + } + tracing::info!( + " sidecar: dialling relayer at {}", + redact_url_userinfo(&relayer_urls.dial) + ); + let mut sidecar_command = tokio::process::Command::new("npx"); + sidecar_command .args(["tsx", "sidecar-server.ts"]) .current_dir(&scripts_dir) - .env("MEMWAL_RELAYER_URL", mcp_relayer_url) + .env("MEMWAL_RELAYER_URL", &relayer_urls.dial) .stdout(std::process::Stdio::inherit()) - .stderr(std::process::Stdio::inherit()) + .stderr(std::process::Stdio::inherit()); + if let Some(public_relayer_url) = &relayer_urls.public { + sidecar_command.env("MEMWAL_PUBLIC_RELAYER_URL", public_relayer_url); + } + let mut sidecar_child = sidecar_command .spawn() .expect("Failed to start TS sidecar. Is Node.js installed?"); @@ -808,12 +1048,15 @@ async fn main() { } }); + let alerts = Arc::new(AlertManager::from_env(http_client.clone())); + // Initialize database (PostgreSQL + pgvector). // `Arc` so the MemoryEngine impl shares the same pool as the handlers. let db = Arc::new( VectorDb::new(&config.database_url) .await - .expect("Failed to connect to PostgreSQL"), + .expect("Failed to connect to PostgreSQL") + .with_storage_alerts(Arc::clone(&alerts), config.sui_network.clone()), ); let security_delete_component_enabled = config.enable_security_delete || config.deletion_reconciler_enabled @@ -988,8 +1231,6 @@ async fn main() { // CompositeRanker is stateless — one shared instance is fine. let ranker: Arc = Arc::new(CompositeRanker); - let alerts = Arc::new(AlertManager::from_env(http_client.clone())); - // General delegate-key verification and the boot-time SEAL policy check // share this independent gRPC client; security deletion owns a separate // quota-gated client below. @@ -1153,6 +1394,9 @@ async fn main() { http_client, sui_grpc_client, delegate_keys_cache: crate::storage::sui::new_delegate_keys_cache(), + delegate_verify_cache: crate::storage::sui::new_delegate_verify_cache(), + delegate_reject_cache: crate::storage::sui::new_delegate_reject_cache(), + mcp_connect_episodes: crate::observability::new_mcp_connect_episodes(), key_pool, alerts, engine, @@ -1427,6 +1671,78 @@ async fn main() { before - evicted ); } + + // Same reasoning for the verify-result cache (WALM-618): its + // TTL only gates trust-on-hit, so the map itself needs sweeping + // or it grows one entry per (account, delegate key) pair ever + // seen. It sweeps on TTL + `DELEGATE_VERIFY_STALE_GRACE`, not on + // the TTL alone: an entry past the TTL is still servable while + // the chain is unreachable, and sweeping it at 30s would delete + // exactly the entries that outage path exists to serve. + let mut verify_cache = delegate_cache_sweep_state + .delegate_verify_cache + .entries + .write() + .await; + let before = verify_cache.len(); + verify_cache.retain(|_, v| v.is_servable_while_unavailable()); + let evicted = before - verify_cache.len(); + drop(verify_cache); + if evicted > 0 { + tracing::debug!( + "delegate_verify_cache sweep: evicted {} stale entries ({} remaining)", + evicted, + before - evicted + ); + } + + // The rejection cache is keyed by what callers send rather than + // by what exists on chain, so sweeping it is what keeps its cap + // from being reached by ordinary churn instead of by abuse. + let mut reject_cache = delegate_cache_sweep_state + .delegate_reject_cache + .write() + .await; + let before = reject_cache.len(); + reject_cache + .retain(|_, rejected_at| storage::sui::reject_entry_is_fresh(*rejected_at)); + let evicted = before - reject_cache.len(); + drop(reject_cache); + if evicted > 0 { + tracing::debug!( + "delegate_reject_cache sweep: evicted {} expired entries ({} remaining)", + evicted, + before - evicted + ); + } + + // Connect episodes whose client gave up (or was killed) never see + // the success that would remove them. Sweeping is what keeps an + // abandoned episode from holding a slot against the cap. + let mut episodes = delegate_cache_sweep_state + .mcp_connect_episodes + .write() + .await; + let before = episodes.len(); + episodes.retain(|_, started| observability::connect_episode_is_fresh(*started)); + let evicted = before - episodes.len(); + drop(episodes); + if evicted > 0 { + tracing::debug!( + "mcp_connect_episodes sweep: evicted {} abandoned episodes ({} remaining)", + evicted, + before - evicted + ); + } + + // The two log samplers are the last per-account state that was + // not swept here. They expire on insert, but only at the cap, so + // an account that went quiet an hour ago holds its slot until + // some unrelated overflow reclaims it. + let evicted = observability::sweep_log_samplers(); + if evicted > 0 { + tracing::debug!("log sampler sweep: evicted {} idle accounts", evicted); + } } }); diff --git a/services/server/src/mcp_proxy.rs b/services/server/src/mcp_proxy.rs index 2d5390820..cf5fac3ad 100644 --- a/services/server/src/mcp_proxy.rs +++ b/services/server/src/mcp_proxy.rs @@ -118,6 +118,174 @@ fn out_set( // needs zero changes. See `oauth.rs` for the crypto/DB side. // --------------------------------------------------------------------- +/// Client-supplied handshake identity. None of it is trusted for any +/// decision — it exists so a refused handshake can be attributed to a person +/// and a client build instead of appearing as an anonymous status code. +const CONNECT_ID_HEADER: &str = "x-memwal-connect-id"; +const CLIENT_NAME_HEADER: &str = "x-memwal-client"; +const CLIENT_VERSION_HEADER: &str = "x-memwal-client-version"; +const BRIDGE_VERSION_HEADER: &str = "x-memwal-bridge-version"; + +/// Which `/api/mcp/*` entry point a handshake outcome came from. +/// +/// A label rather than three metrics, so the SSE refusal ratio can be +/// queried on its own. `classify_and_resolve` runs on every JSON-RPC +/// envelope as well as on the handshake, so without this the `ok` bucket +/// counts a live session's POSTs and the ratio WALM-618 was diagnosed from +/// reads far healthier than it is. +const ROUTE_SSE: &str = "sse"; +const ROUTE_MESSAGES: &str = "messages"; +const ROUTE_STREAMABLE: &str = "streamable"; + +/// Why a handshake was refused. Stable, low-cardinality strings: they are a +/// Prometheus label and a log field, and they never carry a token, a key, or +/// anything else caller-supplied. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +enum HandshakeRejection { + /// No `Authorization: Bearer`, or an empty one. + NoBearer, + /// A delegate-shaped bearer with no `X-MemWal-Account-Id` to check it against. + NoAccountHeader, + /// 64 hex characters that are not a usable ed25519 secret. + MalformedDelegateKey, + /// Well-formed key, real account, but the key is not registered on it. + NotRegistered, + /// Not a delegate key, and this deployment has no OAuth configured. + OauthNotConfigured, + /// Not a delegate key and not an OAuth token either. + NotOauthToken, + /// An OAuth token that is expired, revoked, or otherwise refused. + OauthRejected, +} + +impl HandshakeRejection { + fn code(self) -> &'static str { + match self { + Self::NoBearer => "no_bearer", + Self::NoAccountHeader => "no_account_header", + Self::MalformedDelegateKey => "malformed_delegate_key", + Self::NotRegistered => "not_registered", + Self::OauthNotConfigured => "oauth_not_configured", + Self::NotOauthToken => "not_oauth_token", + Self::OauthRejected => "oauth_rejected", + } + } +} + +fn header_str<'a>(headers: &'a HeaderMap, name: &str) -> Option<&'a str> { + headers + .get(name) + .and_then(|v| v.to_str().ok()) + .map(str::trim) + .filter(|v| !v.is_empty()) +} + +/// Client identity is logged, so cap it and keep it printable — it is +/// caller-supplied and must not be able to inject newlines into the log or +/// blow up a line. +fn sanitized_client(headers: &HeaderMap, name: &str) -> String { + header_str(headers, name) + .map(|v| { + v.chars() + .filter(|c| c.is_ascii_graphic() || *c == ' ') + .take(64) + .collect::() + }) + .filter(|v| !v.is_empty()) + .unwrap_or_else(|| "-".to_string()) +} + +/// Log and count one refused handshake, then produce the outcome. +/// +/// Every refusal goes through here. Before this, four of the five ways to be +/// refused logged nothing at all and none of them touched a metric, so 401 — +/// 70% of this route's traffic — was invisible in both logs and dashboards. +fn refuse( + reason: HandshakeRejection, + route: &str, + headers: &HeaderMap, + oauth_err: Option, +) -> McpAuthOutcome { + // Counter first and unconditionally — it is the signal a dashboard reads, + // and it must not depend on whether this particular refusal was sampled. + crate::observability::record_mcp_handshake(route, "unauthorized", reason.code()); + crate::observability::record_app_error("mcp_unauthorized"); + // The line carries what the counter cannot, but this route has no rate + // limit in front of it and a stuck client retries forever, so it is + // sampled per account rather than written per request. + let account_id = account_id_header(headers).unwrap_or("-"); + if crate::observability::should_log_refusal(account_id) { + tracing::warn!( + reason = reason.code(), + account_id = %account_id, + connect_id = header_str(headers, CONNECT_ID_HEADER).unwrap_or("-"), + client = %sanitized_client(headers, CLIENT_NAME_HEADER), + client_version = %sanitized_client(headers, CLIENT_VERSION_HEADER), + bridge_version = %sanitized_client(headers, BRIDGE_VERSION_HEADER), + "mcp handshake refused (sampled; see memwal_mcp_handshake_total for the rate)" + ); + } + McpAuthOutcome::Unauthorized(oauth_err) +} + +/// Start timing this client's connect episode, if it named one. +/// +/// Called on every attempt; only the first one for an id records anything, so +/// the measured span runs from the client's first try to the one that works — +/// which is the interval a user perceives, and the one no per-request metric +/// can see, because each individual request here is fast. +async fn note_connect_attempt(state: &AppState, headers: &HeaderMap) { + let Some(id) = header_str(headers, CONNECT_ID_HEADER) else { + return; + }; + let mut episodes = state.mcp_connect_episodes.write().await; + if episodes.contains_key(id) { + return; + } + // Expire before consulting the cap, as the rejection cache and the log + // samplers do. This runs before authentication on a route with no rate + // limit, keyed on a header the caller chooses, so without it a few + // thousand made-up connect ids hold every slot until the 300s sweep — + // and while they do, no legitimate client is timed at all, which blinds + // `time_to_session` during exactly the incident it exists to measure. + // + // It self-heals once the spam stops; it does not stop a caller who keeps + // it up. Bounding that needs an authenticated key, or the measurement + // moved to the bridge, which already knows its own elapsed time. + if episodes.len() >= crate::observability::MCP_CONNECT_EPISODE_MAX { + episodes.retain(|_, started| crate::observability::connect_episode_is_fresh(*started)); + } + if episodes.len() >= crate::observability::MCP_CONNECT_EPISODE_MAX { + return; + } + episodes.insert(id.to_string(), std::time::Instant::now()); +} + +/// Close the episode and record how long the client waited in total. +async fn finish_connect_episode(state: &AppState, headers: &HeaderMap) { + let Some(id) = header_str(headers, CONNECT_ID_HEADER) else { + return; + }; + let started = state.mcp_connect_episodes.write().await.remove(id); + let Some(started) = started.filter(|s| crate::observability::connect_episode_is_fresh(*s)) else { + return; + }; + let waited = started.elapsed(); + // Anything past a couple of seconds means the client was retrying, which + // is the WALM-618 shape. Say so at `info` with the id, so one grep gives + // the whole episode including the refusals that led here. + if waited > std::time::Duration::from_secs(2) { + tracing::info!( + connect_id = %id, + account_id = account_id_header(headers).unwrap_or("-"), + waited_ms = waited.as_millis(), + client = %sanitized_client(headers, CLIENT_NAME_HEADER), + "mcp session opened after retries" + ); + } + crate::observability::record_mcp_time_to_session(waited); +} + enum McpAuthOutcome { /// The bearer is the legacy 64-hex delegate key — forward exactly as /// today, byte for byte (OAuth tokens are never valid here). @@ -168,14 +336,19 @@ async fn legacy_delegate_registered( state: &AppState, headers: &HeaderMap, token: &str, + route: &str, ) -> McpAuthOutcome { let Some(account_id) = account_id_header(headers) else { - return McpAuthOutcome::Unauthorized(None); + return refuse(HandshakeRejection::NoAccountHeader, route, headers, None); }; let Some(pk) = public_key_from_delegate_hex(token) else { - return McpAuthOutcome::Unauthorized(None); + return refuse(HandshakeRejection::MalformedDelegateKey, route, headers, None); }; - match crate::storage::sui::verify_delegate_key_onchain( + // Cached: this runs on the SSE handshake and on every JSON-RPC envelope, + // so an uncached read here is one fullnode call per envelope. + match crate::storage::sui::verify_delegate_key_cached( + &state.delegate_verify_cache, + &state.delegate_reject_cache, &state.http_client, &state.config.sui_rpc_url, state.sui_grpc_client.as_ref(), @@ -185,37 +358,57 @@ async fn legacy_delegate_registered( ) .await { - Ok(_) => McpAuthOutcome::Passthrough, + Ok(_) => { + crate::observability::record_mcp_handshake(route, "ok", "none"); + McpAuthOutcome::Passthrough + } Err(err) if err.is_unavailable() => { - tracing::warn!(error = %err, "mcp delegate on-chain verify unavailable"); + tracing::warn!( + account_id = %account_id, + connect_id = header_str(headers, CONNECT_ID_HEADER).unwrap_or("-"), + client = %sanitized_client(headers, CLIENT_NAME_HEADER), + error = %err, + "mcp delegate on-chain verify unavailable" + ); + crate::observability::record_mcp_handshake(route, "unavailable", "sui_unavailable"); + crate::observability::record_app_error("mcp_upstream_unavailable"); McpAuthOutcome::Unavailable } Err(err) => { - tracing::debug!("mcp delegate rejected: {err}"); - McpAuthOutcome::Unauthorized(None) + tracing::debug!(error = %err, "mcp delegate rejected on chain"); + refuse(HandshakeRejection::NotRegistered, route, headers, None) } } } -async fn classify_and_resolve(state: &AppState, headers: &HeaderMap) -> McpAuthOutcome { +async fn classify_and_resolve( + state: &AppState, + headers: &HeaderMap, + route: &str, +) -> McpAuthOutcome { let Some(token) = bearer_token(headers) else { - return McpAuthOutcome::Unauthorized(None); + return refuse(HandshakeRejection::NoBearer, route, headers, None); }; if is_legacy_delegate_bearer(token) { - return legacy_delegate_registered(state, headers, token).await; + return legacy_delegate_registered(state, headers, token, route).await; } if state.config.mcp_oauth.is_none() { - return McpAuthOutcome::Unauthorized(None); + return refuse(HandshakeRejection::OauthNotConfigured, route, headers, None); } match crate::oauth::resolve_oauth_bearer(state, token).await { - Ok(identity) => McpAuthOutcome::Oauth(Box::new(identity)), - Err(crate::oauth::OAuthBearerError::NotOAuthToken) => McpAuthOutcome::Unauthorized(None), + Ok(identity) => { + crate::observability::record_mcp_handshake(route, "ok", "none"); + McpAuthOutcome::Oauth(Box::new(identity)) + } + Err(crate::oauth::OAuthBearerError::NotOAuthToken) => { + refuse(HandshakeRejection::NotOauthToken, route, headers, None) + } Err(err) => { - tracing::debug!("mcp_proxy oauth bearer rejected: {:?}", err); - McpAuthOutcome::Unauthorized(Some(err)) + tracing::debug!("mcp_proxy oauth bearer detail: {:?}", err); + refuse(HandshakeRejection::OauthRejected, route, headers, Some(err)) } } } @@ -346,9 +539,16 @@ pub async fn sse_proxy( peer, state.config.trusted_proxy_hops, ); - let identity = match classify_and_resolve(&state, &headers).await { - McpAuthOutcome::Passthrough => None, - McpAuthOutcome::Oauth(identity) => Some(identity), + note_connect_attempt(&state, &headers).await; + let identity = match classify_and_resolve(&state, &headers, ROUTE_SSE).await { + McpAuthOutcome::Passthrough => { + finish_connect_episode(&state, &headers).await; + None + } + McpAuthOutcome::Oauth(identity) => { + finish_connect_episode(&state, &headers).await; + Some(identity) + } McpAuthOutcome::Unauthorized(err) => { return oauth_unauthorized_response(&state, err.as_ref()) } @@ -390,7 +590,16 @@ pub async fn sse_proxy( let lname = name.as_str().to_ascii_lowercase(); if matches!( lname.as_str(), - "content-type" | "cache-control" | "www-authenticate" | "connection" + // `retry-after` is load-bearing: the sidecar sets it on an + // `ip_burst_cap` 429 and the bridge honours it (WALM-386). + // Dropping it here left the client with no ETA and the wrong + // remediation, since it could not tell a timed cap from a + // concurrency one. + "content-type" + | "cache-control" + | "www-authenticate" + | "connection" + | "retry-after" ) { if let (Ok(n), Ok(v)) = ( HeaderName::from_bytes(name.as_str().as_bytes()), @@ -457,9 +666,16 @@ pub async fn messages_proxy( peer, state.config.trusted_proxy_hops, ); - let identity = match classify_and_resolve(&state, &headers).await { - McpAuthOutcome::Passthrough => None, - McpAuthOutcome::Oauth(identity) => Some(identity), + note_connect_attempt(&state, &headers).await; + let identity = match classify_and_resolve(&state, &headers, ROUTE_MESSAGES).await { + McpAuthOutcome::Passthrough => { + finish_connect_episode(&state, &headers).await; + None + } + McpAuthOutcome::Oauth(identity) => { + finish_connect_episode(&state, &headers).await; + Some(identity) + } McpAuthOutcome::Unauthorized(err) => { return oauth_unauthorized_response(&state, err.as_ref()) } @@ -500,18 +716,30 @@ pub async fn messages_proxy( .and_then(|v| v.to_str().ok()) .unwrap_or("application/json") .to_string(); + // Same reason as the SSE and streamable allowlists: the sidecar rate + // limits this route too, and dropping `retry-after` leaves the client + // unable to tell a cap that clears on a timer from one that clears when + // somebody else disconnects (WALM-386). Captured before `bytes()` + // consumes the response. + let retry_after = upstream + .headers() + .get(reqwest::header::RETRY_AFTER) + .and_then(|v| v.to_str().ok()) + .and_then(|v| HeaderValue::from_str(v).ok()); match upstream.bytes().await { - Ok(bytes) => ( - status, - [( + Ok(bytes) => { + let mut headers = HeaderMap::new(); + headers.insert( axum::http::header::CONTENT_TYPE, HeaderValue::from_str(&content_type) .unwrap_or_else(|_| HeaderValue::from_static("application/json")), - )], - bytes, - ) - .into_response(), + ); + if let Some(value) = retry_after { + headers.insert(axum::http::header::RETRY_AFTER, value); + } + (status, headers, bytes).into_response() + } Err(err) => ( StatusCode::BAD_GATEWAY, format!("MCP sidecar read failed: {}", err), @@ -575,9 +803,16 @@ pub async fn streamable_proxy( peer, state.config.trusted_proxy_hops, ); - let identity = match classify_and_resolve(&state, &headers).await { - McpAuthOutcome::Passthrough => None, - McpAuthOutcome::Oauth(identity) => Some(identity), + note_connect_attempt(&state, &headers).await; + let identity = match classify_and_resolve(&state, &headers, ROUTE_STREAMABLE).await { + McpAuthOutcome::Passthrough => { + finish_connect_episode(&state, &headers).await; + None + } + McpAuthOutcome::Oauth(identity) => { + finish_connect_episode(&state, &headers).await; + Some(identity) + } McpAuthOutcome::Unauthorized(err) => { return oauth_unauthorized_response(&state, err.as_ref()) } @@ -627,6 +862,7 @@ pub async fn streamable_proxy( | "connection" | "mcp-session-id" | "mcp-protocol-version" + | "retry-after" ) { if let (Ok(n), Ok(v)) = ( HeaderName::from_bytes(name.as_str().as_bytes()), @@ -678,6 +914,113 @@ mod tests { .map(|s| s.to_string()) } + #[test] + fn every_rejection_reason_has_a_distinct_stable_code() { + // These are Prometheus label values and log fields. A duplicate would + // silently merge two causes into one series; a rename breaks every + // saved query. Both are worth a test. + use HandshakeRejection::*; + let all = [ + NoBearer, + NoAccountHeader, + MalformedDelegateKey, + NotRegistered, + OauthNotConfigured, + NotOauthToken, + OauthRejected, + ]; + let codes: Vec<&str> = all.iter().map(|r| r.code()).collect(); + let mut unique = codes.clone(); + unique.sort_unstable(); + unique.dedup(); + assert_eq!(unique.len(), codes.len(), "reason codes must be distinct"); + assert_eq!( + codes, + vec![ + "no_bearer", + "no_account_header", + "malformed_delegate_key", + "not_registered", + "oauth_not_configured", + "not_oauth_token", + "oauth_rejected", + ] + ); + assert!( + codes.iter().all(|c| c + .chars() + .all(|ch| ch.is_ascii_lowercase() || ch == '_')), + "low-cardinality snake_case only — never a token or an account" + ); + } + + #[test] + fn client_identity_is_sanitized_before_it_reaches_a_log_line() { + // Caller-supplied and logged, so it must not be able to forge a second + // log line or run away with the line length. + let h = axum_headers(&[("x-memwal-client", "claude-code")]); + assert_eq!(sanitized_client(&h, "x-memwal-client"), "claude-code"); + + let missing = axum_headers(&[]); + assert_eq!( + sanitized_client(&missing, "x-memwal-client"), + "-", + "absent must read as absent, not as an empty field" + ); + + let long = "a".repeat(500); + let h = axum_headers(&[("x-memwal-client", &long)]); + assert_eq!(sanitized_client(&h, "x-memwal-client").len(), 64); + } + + #[test] + fn expired_episodes_do_not_hold_the_cap_against_a_real_client() { + // `note_connect_attempt` runs BEFORE authentication, on a route with + // no rate limit, keyed on a header the caller picks. Without expiring + // at the cap, a few thousand made-up connect ids hold every slot for + // the full 300s sweep interval — and while they do, no legitimate + // client is ever recorded, so `time_to_session` observes nothing + // during exactly the incident it was added to measure. + // + // This pins the policy; the retain itself is inline in + // `note_connect_attempt`, which needs an `AppState`. Keep them in step. + let mut episodes: std::collections::HashMap = + std::collections::HashMap::new(); + let abandoned = std::time::Instant::now() + - (crate::observability::MCP_CONNECT_EPISODE_TTL + std::time::Duration::from_secs(1)); + for i in 0..crate::observability::MCP_CONNECT_EPISODE_MAX { + episodes.insert(format!("spam-{i}"), abandoned); + } + assert!( + episodes.len() >= crate::observability::MCP_CONNECT_EPISODE_MAX, + "precondition: the cap is full, so a real client would be turned away" + ); + + episodes.retain(|_, started| crate::observability::connect_episode_is_fresh(*started)); + + assert!( + episodes.is_empty(), + "every seeded episode is past the TTL, so none may be kept" + ); + assert!( + episodes.len() < crate::observability::MCP_CONNECT_EPISODE_MAX, + "after expiring, a real client can be timed again" + ); + } + + #[test] + fn a_connect_episode_expires_so_an_abandoned_one_cannot_hold_a_slot() { + let fresh = std::time::Instant::now(); + assert!(crate::observability::connect_episode_is_fresh(fresh)); + + let abandoned = std::time::Instant::now() + - (crate::observability::MCP_CONNECT_EPISODE_TTL + std::time::Duration::from_secs(1)); + assert!( + !crate::observability::connect_episode_is_fresh(abandoned), + "past the TTL it is swept, and never reported as a time_to_session" + ); + } + #[test] fn should_forward_allows_authorization_and_mcp_headers() { for h in [ diff --git a/services/server/src/observability.rs b/services/server/src/observability.rs index 14544d284..0f9707d32 100644 --- a/services/server/src/observability.rs +++ b/services/server/src/observability.rs @@ -115,6 +115,29 @@ static ERRORS_TOTAL: LazyLock = LazyLock::new(|| { .expect("register memwal_errors_total") }); +static MCP_HANDSHAKE_TOTAL: LazyLock = LazyLock::new(|| { + prometheus::register_int_counter_vec!( + "memwal_mcp_handshake_total", + "MCP handshake attempts by route and outcome and, when refused, why.", + &["route", "outcome", "reason"] + ) + .expect("register memwal_mcp_handshake_total") +}); + +static MCP_TIME_TO_SESSION_SECONDS: LazyLock = LazyLock::new(|| { + prometheus::register_histogram!(HistogramOpts::new( + "memwal_mcp_time_to_session_seconds", + "Wall-clock from a client's first handshake attempt to the one that \ + succeeded, keyed by its connect-episode id. This is the number a user \ + experiences as \"nothing is happening\": no single request is slow, so \ + the per-request latency histogram cannot show it." + ) + .buckets(vec![ + 0.5, 1.0, 2.0, 5.0, 10.0, 30.0, 60.0, 120.0, 300.0, 600.0 + ])) + .expect("register memwal_mcp_time_to_session_seconds") +}); + static RATE_LIMIT_DENIALS_TOTAL: LazyLock = LazyLock::new(|| { prometheus::register_int_counter_vec!( "memwal_rate_limit_denials_total", @@ -588,6 +611,148 @@ pub fn record_app_error(kind: &'static str) { ERRORS_TOTAL.with_label_values(&[kind, &route]).inc(); } +// ── Refusal log sampling ──────────────────────────────────────────── +// +// `/api/mcp/*` has no rate limit ahead of it and a stuck client retries +// forever, so one log line per refusal is one log line per retry — the +// 784,627 × 401 this PR measures would have become 784,627 warn lines. +// The counter is the always-on signal; the line is a sample carrying the +// detail a counter cannot (which account, which client, which reason). + +/// At most one refusal line per account per window. +pub const MCP_REFUSAL_LOG_INTERVAL: std::time::Duration = std::time::Duration::from_secs(60); + +/// Bounded for the usual reason: the key is a caller-supplied header. At the +/// cap we stop tracking new accounts and simply do not log them — the metric +/// still counts every refusal, so nothing is lost that a dashboard needs. +pub const MCP_REFUSAL_LOG_MAX_ACCOUNTS: usize = 4_096; + +static MCP_REFUSAL_LOG_SEEN: LazyLock< + std::sync::Mutex>, +> = LazyLock::new(|| std::sync::Mutex::new(std::collections::HashMap::new())); + +/// Whether this refusal should be written out, given what was logged before. +/// Pure given the map, so the policy is unit-testable. +pub fn should_log_refusal_at( + seen: &mut std::collections::HashMap, + account: &str, + now: std::time::Instant, +) -> bool { + match seen.get(account) { + Some(last) if now.duration_since(*last) < MCP_REFUSAL_LOG_INTERVAL => false, + Some(_) => { + seen.insert(account.to_string(), now); + true + } + None => { + // Expire on insert: this map is only written on the sampled path, + // so it is cheap, and it keeps an idle account from holding a slot. + if seen.len() >= MCP_REFUSAL_LOG_MAX_ACCOUNTS { + seen.retain(|_, last| now.duration_since(*last) < MCP_REFUSAL_LOG_INTERVAL); + } + if seen.len() >= MCP_REFUSAL_LOG_MAX_ACCOUNTS { + return false; + } + seen.insert(account.to_string(), now); + true + } + } +} + +pub fn should_log_refusal(account: &str) -> bool { + let Ok(mut seen) = MCP_REFUSAL_LOG_SEEN.lock() else { + return false; + }; + should_log_refusal_at(&mut seen, account, std::time::Instant::now()) +} + +static MCP_STALE_SERVE_LOG_SEEN: LazyLock< + std::sync::Mutex>, +> = LazyLock::new(|| std::sync::Mutex::new(std::collections::HashMap::new())); + +/// Whether this stale-grace serve should be written out. +/// +/// Same policy and same window as the refusal line, and for the same reason: +/// it fires on the hot path (every signed request and every MCP envelope) and +/// fires *hardest* during a Sui outage, when many requests fall past the TTL +/// at once — the 155,874 x 503 shape this PR measures. One line per request +/// there is the same unbounded repetition the refusal sampler was added to +/// stop. +/// +/// Its own map, though. An account being refused must not suppress the very +/// different fact that it is being authenticated from a stale verification, +/// and vice versa — sharing one map would let either hide the other. +pub fn should_log_stale_serve(account: &str) -> bool { + let Ok(mut seen) = MCP_STALE_SERVE_LOG_SEEN.lock() else { + return false; + }; + should_log_refusal_at(&mut seen, account, std::time::Instant::now()) +} + +/// Drop sampler entries that have aged out of their window. +/// +/// Both maps already expire on insert, but only once they reach the cap, so +/// an account that was noisy an hour ago holds its slot until some unrelated +/// overflow reclaims it. Every other per-account map this service keeps is +/// swept periodically; this makes these two consistent with them. Returns how +/// many entries were dropped, for the sweep log. +pub fn sweep_log_samplers() -> usize { + let now = std::time::Instant::now(); + let mut evicted = 0; + for map in [&*MCP_REFUSAL_LOG_SEEN, &*MCP_STALE_SERVE_LOG_SEEN] { + let Ok(mut seen) = map.lock() else { continue }; + let before = seen.len(); + seen.retain(|_, last| now.duration_since(*last) < MCP_REFUSAL_LOG_INTERVAL); + evicted += before - seen.len(); + } + evicted +} + +// ── MCP connect episodes ──────────────────────────────────────────── +// +// State for `time_to_session`. Lives here rather than in `mcp_proxy` +// because `AppState` is in the library crate and `mcp_proxy` is not. + +/// Longest an unfinished connect episode is remembered. A client that gives +/// up, or is killed, leaves an entry behind; past this it is swept. Also the +/// ceiling on any single `time_to_session` observation. +pub const MCP_CONNECT_EPISODE_TTL: std::time::Duration = std::time::Duration::from_secs(900); + +/// Hard ceiling on tracked episodes, for the same reason the rejection cache +/// has one: the key comes from the caller. At the cap new episodes are simply +/// not timed — the metric loses samples, nothing else degrades. +pub const MCP_CONNECT_EPISODE_MAX: usize = 4_096; + +pub type McpConnectEpisodes = + std::sync::Arc>>; + +pub fn new_mcp_connect_episodes() -> McpConnectEpisodes { + std::sync::Arc::new(tokio::sync::RwLock::new(HashMap::new())) +} + +pub fn connect_episode_is_fresh(started: std::time::Instant) -> bool { + started.elapsed() < MCP_CONNECT_EPISODE_TTL +} + +/// Count one MCP handshake. `reason` is `"none"` on success — Prometheus +/// label sets must be uniform, and an empty string reads as missing data. +/// +/// `route` names the `/api/mcp/*` entry point. Without it the SSE refusal +/// ratio cannot be read at all: `classify_and_resolve` runs on the handshake +/// *and* on every JSON-RPC envelope, so one live session's POSTs outnumber +/// the GET this metric exists to measure. +pub fn record_mcp_handshake(route: &str, outcome: &str, reason: &str) { + MCP_HANDSHAKE_TOTAL + .with_label_values(&[route, outcome, reason]) + .inc(); +} + +/// Record how long a client spent getting a session. Only called on the +/// attempt that succeeded, so the histogram counts episodes, not requests. +pub fn record_mcp_time_to_session(elapsed: std::time::Duration) { + MCP_TIME_TO_SESSION_SECONDS.observe(elapsed.as_secs_f64()); +} + pub fn record_rate_limit_denial(bucket: &str) { let route = current_route(); RATE_LIMIT_DENIALS_TOTAL @@ -784,3 +949,102 @@ mod tests { ); } } + +#[cfg(test)] +mod refusal_log_tests { + use super::*; + use std::collections::HashMap; + use std::time::{Duration, Instant}; + + #[test] + fn a_refusal_and_a_stale_serve_do_not_share_a_sampling_slot() { + // Same policy, same window, deliberately different maps. They report + // unrelated things — "this key was refused" versus "this key is being + // authenticated from a verification we could not refresh" — and an + // account in trouble tends to produce both. One map would let + // whichever fired first silence the other for the whole window, which + // is exactly the attribution problem this PR set out to fix. + let account = "0xsampler-independence-probe"; + assert!( + should_log_refusal(account), + "first refusal for this account must be written" + ); + assert!( + should_log_stale_serve(account), + "the stale-serve line must not be suppressed by the refusal that just fired" + ); + assert!( + !should_log_stale_serve(account), + "but it is still sampled within its own window" + ); + // Fresh entries are never swept, so this is deterministic regardless + // of what else ran before it. + assert_eq!( + sweep_log_samplers(), + 0, + "entries inside the window must survive the periodic sweep" + ); + } + + #[test] + fn one_account_is_logged_once_per_window_however_hard_it_retries() { + // The reason this exists: `/api/mcp/*` has no rate limit in front of + // it and a stuck client retries forever, so an unsampled line is one + // line per retry — 784,627 of them in the window this PR measures. + let mut seen = HashMap::new(); + let t0 = Instant::now(); + assert!(should_log_refusal_at(&mut seen, "0xacct", t0)); + for i in 1..100 { + assert!( + !should_log_refusal_at(&mut seen, "0xacct", t0 + Duration::from_millis(i * 500)), + "retry {i} must not produce a second line inside the window" + ); + } + assert!(should_log_refusal_at( + &mut seen, + "0xacct", + t0 + MCP_REFUSAL_LOG_INTERVAL + Duration::from_secs(1) + )); + } + + #[test] + fn accounts_are_sampled_independently() { + let mut seen = HashMap::new(); + let t0 = Instant::now(); + assert!(should_log_refusal_at(&mut seen, "0xdio", t0)); + assert!( + should_log_refusal_at(&mut seen, "0xteo", t0), + "one noisy account must not silence everyone else" + ); + } + + #[test] + fn the_sampling_map_cannot_be_grown_without_bound() { + // Keyed by an unauthenticated header, so the cap matters. Past it we + // stop logging new accounts; the counter still counts every refusal. + let mut seen = HashMap::new(); + let t0 = Instant::now(); + for i in 0..MCP_REFUSAL_LOG_MAX_ACCOUNTS { + assert!(should_log_refusal_at(&mut seen, &format!("0x{i}"), t0)); + } + assert!( + !should_log_refusal_at(&mut seen, "0xoverflow", t0), + "at the cap a new account is counted but not logged" + ); + assert_eq!(seen.len(), MCP_REFUSAL_LOG_MAX_ACCOUNTS); + } + + #[test] + fn expired_accounts_free_their_slot_for_a_new_one() { + let mut seen = HashMap::new(); + let t0 = Instant::now(); + for i in 0..MCP_REFUSAL_LOG_MAX_ACCOUNTS { + assert!(should_log_refusal_at(&mut seen, &format!("0x{i}"), t0)); + } + let later = t0 + MCP_REFUSAL_LOG_INTERVAL + Duration::from_secs(1); + assert!( + should_log_refusal_at(&mut seen, "0xoverflow", later), + "once the old entries age out the cap must not stay wedged" + ); + } +} diff --git a/services/server/src/rate_limit.rs b/services/server/src/rate_limit.rs index d17600fae..54c8f2a5d 100644 --- a/services/server/src/rate_limit.rs +++ b/services/server/src/rate_limit.rs @@ -11,7 +11,6 @@ use std::time::Duration; use uuid::Uuid; use crate::{ - client_ip::canonical_client_ip, storage::db::{StorageAdmission, StorageReservationRequest}, types::{AppError, AppState, AuthInfo}, }; @@ -171,19 +170,22 @@ fn endpoint_weight(path: &str) -> i64 { // Redis Client // ============================================================ -/// Create a Redis multiplexed connection for shared use across the app. -pub async fn create_redis_client( - redis_url: &str, -) -> Result { +/// Reconnect so a dropped Redis connection does not fail-close +/// unauthenticated limiters for the process lifetime. +pub async fn create_redis_client(redis_url: &str) -> Result { let client = redis::Client::open(redis_url) .map_err(|e| format!("Failed to create Redis client: {}", e))?; - let conn = client - .get_multiplexed_async_connection() + // Bound reconnect so a dead Redis still 503s promptly instead of + // stalling fail-closed routes. + let config = redis::aio::ConnectionManagerConfig::new() + .set_connection_timeout(Duration::from_secs(2)) + .set_response_timeout(Duration::from_secs(2)) + .set_max_delay(2_000) + .set_number_of_retries(3); + redis::aio::ConnectionManager::new_with_config(client, config) .await - .map_err(|e| format!("Failed to connect to Redis: {}", e))?; - - Ok(conn) + .map_err(|e| format!("Failed to connect to Redis: {}", e)) } // ============================================================ @@ -247,15 +249,18 @@ enum WindowCheckResult { /// The Lua script executes as a single atomic Redis operation, preventing the /// TOCTOU race where two concurrent requests could both pass the check before /// either records, then both record and collectively exceed the limit. -async fn check_and_record_window( - redis: &mut redis::aio::MultiplexedConnection, +async fn check_and_record_window( + redis: &mut C, key: &str, window_start: f64, now: f64, limit: i64, weight: i64, ttl_seconds: i64, -) -> Result { +) -> Result +where + C: redis::aio::ConnectionLike, +{ // The UUID keeps members distinct across concurrent requests, processes, // and replicas even when they share an identical millisecond timestamp. let request_id = Uuid::new_v4().to_string(); @@ -891,8 +896,8 @@ fn stable_hash_i64(s: &str) -> i64 { /// check/log/fail-open boilerplate that already exists inline in /// `rate_limit_middleware`'s per-account layer and in the global /// sponsor/account limiters below. -async fn check_owner_window_limit( - redis: &mut redis::aio::MultiplexedConnection, +async fn check_owner_window_limit( + redis: &mut C, scope: &str, key: &str, window_start: f64, @@ -902,7 +907,10 @@ async fn check_owner_window_limit( ttl_seconds: i64, owner: &str, deny_message: impl FnOnce() -> String, -) -> Result<(), AppError> { +) -> Result<(), AppError> +where + C: redis::aio::ConnectionLike, +{ match check_and_record_window(redis, key, window_start, now, limit, weight, ttl_seconds).await { Ok(WindowCheckResult::Denied) => { crate::observability::record_rate_limit_denial(scope); @@ -1126,6 +1134,21 @@ pub async fn charge_explicit_weight( Ok(()) } +// ============================================================ +// Client IP for unauthenticated IP limiters +// ============================================================ + +/// Resolve the rate-limit client IP. Missing `ConnectInfo` uses `0.0.0.0` +/// so hops=0 still shares one unknown bucket and hops>0 still read XFF. +fn rate_limit_peer_addr(request: &Request, trusted_proxy_hops: usize) -> std::net::IpAddr { + let peer = request + .extensions() + .get::>() + .map(|ci| ci.0) + .unwrap_or_else(|| std::net::SocketAddr::from(([0, 0, 0, 0], 0))); + crate::client_ip::canonical_client_ip(request.headers(), peer, trusted_proxy_hops) +} + // ============================================================ // Sponsor Rate Limit Middleware (IP-based, unauthenticated) // ============================================================ @@ -1151,18 +1174,7 @@ pub async fn sponsor_rate_limit_middleware( // XFF is ignored by default. Only walk back through the explicitly // configured number of trusted proxy hops, using the same resolver as // the MCP proxy path. - let ip = match request - .extensions() - .get::>() - .map(|ci| canonical_client_ip(request.headers(), ci.0, state.config.trusted_proxy_hops)) - { - Some(ip) => ip.to_string(), - None => { - // Cannot determine IP — fail-closed: deny rather than allow unknown callers. - tracing::warn!("sponsor_rate_limit_middleware: cannot determine client IP, denying"); - return rate_limiter_unavailable_response(); - } - }; + let ip = rate_limit_peer_addr(&request, state.config.trusted_proxy_hops).to_string(); let config = &state.config.sponsor_rate_limit; let mut redis = state.redis.clone(); @@ -1375,18 +1387,7 @@ pub async fn accounts_rate_limit_middleware( // XFF is ignored by default. Only walk back through the explicitly // configured number of trusted proxy hops, using the same resolver as // the sponsor and MCP proxy paths. - let ip = match request - .extensions() - .get::>() - .map(|ci| canonical_client_ip(request.headers(), ci.0, state.config.trusted_proxy_hops)) - { - Some(ip) => ip.to_string(), - None => { - // Cannot determine IP — fail-closed: deny rather than allow unknown callers. - tracing::warn!("accounts_rate_limit_middleware: cannot determine client IP, denying"); - return rate_limiter_unavailable_response(); - } - }; + let ip = rate_limit_peer_addr(&request, state.config.trusted_proxy_hops).to_string(); let config = &state.config.accounts_rate_limit; let mut redis = state.redis.clone(); @@ -1535,19 +1536,7 @@ pub async fn owner_token_ip_rate_limit_middleware( return next.run(request).await; } - let ip = match request - .extensions() - .get::>() - .map(|ci| canonical_client_ip(request.headers(), ci.0, state.config.trusted_proxy_hops)) - { - Some(ip) => ip.to_string(), - None => { - tracing::warn!( - "owner_token_ip_rate_limit_middleware: cannot determine client IP, denying" - ); - return rate_limiter_unavailable_response(); - } - }; + let ip = rate_limit_peer_addr(&request, state.config.trusted_proxy_hops).to_string(); let config = &state.config.owner_token_rate_limit; let mut redis = state.redis.clone(); @@ -1917,6 +1906,64 @@ mod tests { assert!(resp.headers().contains_key("retry-after")); } + // ---- Missing ConnectInfo must still be rate-limited (WALM-626) ---- + + fn rate_limit_request(peer: Option, xff: Option<&str>) -> Request { + let mut builder = axum::http::Request::builder().uri("/"); + if let Some(xff) = xff { + builder = builder.header("x-forwarded-for", xff); + } + let mut request = builder.body(axum::body::Body::empty()).unwrap(); + if let Some(peer) = peer { + request + .extensions_mut() + .insert(axum::extract::ConnectInfo(peer)); + } + request + } + + #[test] + fn missing_connect_info_with_zero_hops_uses_unspecified_peer() { + let request = rate_limit_request(None, Some("198.51.100.7")); + assert_eq!( + rate_limit_peer_addr(&request, 0), + "0.0.0.0".parse::().unwrap() + ); + } + + #[test] + fn connect_info_present_with_zero_hops_ignores_xff() { + let peer = "203.0.113.9:443".parse().unwrap(); + let request = rate_limit_request(Some(peer), Some("198.51.100.7")); + assert_eq!( + rate_limit_peer_addr(&request, 0), + "203.0.113.9".parse::().unwrap() + ); + } + + #[test] + fn missing_connect_info_with_trusted_hop_uses_xff() { + let request = rate_limit_request(None, Some("198.51.100.7")); + assert_eq!( + rate_limit_peer_addr(&request, 1), + "198.51.100.7".parse::().unwrap() + ); + } + + #[test] + fn missing_connect_info_with_trusted_hop_falls_back_without_xff() { + let no_xff = rate_limit_request(None, None); + let malformed = rate_limit_request(None, Some("not-an-ip")); + assert_eq!( + rate_limit_peer_addr(&no_xff, 1), + "0.0.0.0".parse::().unwrap() + ); + assert_eq!( + rate_limit_peer_addr(&malformed, 1), + "0.0.0.0".parse::().unwrap() + ); + } + // ---- Read API rate limit config + response shape ---- #[test] diff --git a/services/server/src/routes/admin.rs b/services/server/src/routes/admin.rs index 82e453974..501e7b03e 100644 --- a/services/server/src/routes/admin.rs +++ b/services/server/src/routes/admin.rs @@ -149,28 +149,44 @@ pub async fn health(State(state): State>) -> Json extract: crate::services::extractor::FACT_EXTRACTION_PROMPT_VERSION.to_string(), ask: ASK_SYSTEM_PROMPT_VERSION.to_string(), }, - write_ready: sidecar_write_ready(&state).await, + write_ready: write_ready(&state).await, writes: writes_health_status(state.config.writes_paused), }) } -async fn sidecar_write_ready(state: &std::sync::Arc) -> bool { - // Reuse a short TTL so unsigned /health probes do not fan out to the - // sidecar on every load-balancer tick. +const WRITE_READY_CACHE_TTL: std::time::Duration = std::time::Duration::from_secs(2); +const WRITE_READY_PROBE_TIMEOUT: std::time::Duration = std::time::Duration::from_secs(1); +/// Neon refuses `smgrextend` once cluster size is at the cap; treat less +/// than 1MB remaining as not writable so `/health` trips before the next +/// page allocation fails. +const POSTGRES_EXTEND_HEADROOM_BYTES: i64 = 1024 * 1024; + +/// Sidecar liveness AND Postgres can accept writes. Cached together so +/// unsigned `/health` probes do not fan out on every load-balancer tick. +async fn write_ready(state: &std::sync::Arc) -> bool { { let cache = WRITE_READY_CACHE.lock().unwrap_or_else(|e| e.into_inner()); if let Some((at, ready)) = *cache { - if at.elapsed() < std::time::Duration::from_secs(2) { + if at.elapsed() < WRITE_READY_CACHE_TTL { return ready; } } } + let (sidecar, postgres) = tokio::join!(sidecar_write_ready(state), postgres_write_ready(state)); + let ready = sidecar && postgres; + if let Ok(mut cache) = WRITE_READY_CACHE.lock() { + *cache = Some((std::time::Instant::now(), ready)); + } + ready +} + +async fn sidecar_write_ready(state: &std::sync::Arc) -> bool { let url = format!("{}/health", state.config.sidecar_url.trim_end_matches('/')); - let ready = match state + match state .http_client .get(&url) - .timeout(std::time::Duration::from_millis(300)) + .timeout(WRITE_READY_PROBE_TIMEOUT) .send() .await { @@ -179,11 +195,122 @@ async fn sidecar_write_ready(state: &std::sync::Arc) -> bool { tracing::debug!(error = %err, "sidecar health probe failed"); false } + } +} + +/// Self-hosted Postgres without `neon.max_cluster_size` stays ready (sidecar +/// still applies). Missing `public.pg_cluster_size` falls back to +/// `sum(pg_database_size)` against that GUC. Other probe/pool failures and +/// timeouts fail open at `warn` so CI `wait-for-relayer` is not blocked. +async fn postgres_write_ready(state: &std::sync::Arc) -> bool { + match tokio::time::timeout( + WRITE_READY_PROBE_TIMEOUT, + probe_postgres_write_ready(state.db.pool()), + ) + .await + { + Ok(Ok(ready)) => ready, + Ok(Err(err)) => { + tracing::warn!( + error = %err, + "postgres write-ready probe failed; treating writes as ready" + ); + true + } + Err(_) => { + tracing::warn!("postgres write-ready probe timed out"); + true + } + } +} + +static NEON_MAX_CLUSTER_SIZE_BYTES: tokio::sync::OnceCell> = + tokio::sync::OnceCell::const_new(); + +async fn cached_neon_max_cluster_size_bytes( + pool: &sqlx::PgPool, +) -> Result, sqlx::Error> { + NEON_MAX_CLUSTER_SIZE_BYTES + .get_or_try_init(|| async { + let max_setting: Option = sqlx::query_scalar( + "SELECT setting FROM pg_catalog.pg_settings WHERE name = 'neon.max_cluster_size'", + ) + .fetch_optional(pool) + .await?; + Ok(neon_max_cluster_size_bytes(max_setting.as_deref())) + }) + .await + .copied() +} + +async fn probe_postgres_write_ready(pool: &sqlx::PgPool) -> Result { + let Some(max_bytes) = cached_neon_max_cluster_size_bytes(pool).await? else { + return Ok(true); }; - if let Ok(mut cache) = WRITE_READY_CACHE.lock() { - *cache = Some((std::time::Instant::now(), ready)); + + let used_bytes = cluster_used_bytes(pool).await?; + Ok(postgres_can_accept_writes(used_bytes, max_bytes)) +} + +static PG_CLUSTER_SIZE_MISSING: std::sync::atomic::AtomicBool = + std::sync::atomic::AtomicBool::new(false); + +/// Neon gates smgrextend on cluster size, not this database's +/// `pg_database_size`. Qualify `public.pg_cluster_size` for empty +/// search_path through PgBouncer. Missing function (no `neon` extension) +/// falls back to `sum(pg_database_size)` vs the same GUC. +async fn cluster_used_bytes(pool: &sqlx::PgPool) -> Result { + if PG_CLUSTER_SIZE_MISSING.load(std::sync::atomic::Ordering::Relaxed) { + return sum_database_size_bytes(pool).await; + } + match sqlx::query_scalar::<_, i64>("SELECT public.pg_cluster_size()::bigint") + .fetch_one(pool) + .await + { + Ok(n) => Ok(n), + Err(e) if pg_cluster_size_unavailable(&e) => { + PG_CLUSTER_SIZE_MISSING.store(true, std::sync::atomic::Ordering::Relaxed); + tracing::warn!( + error = %e, + "public.pg_cluster_size() is missing; falling back to sum(pg_catalog.pg_database_size(datname)) vs neon.max_cluster_size" + ); + sum_database_size_bytes(pool).await + } + Err(e) => Err(e), } - ready +} + +async fn sum_database_size_bytes(pool: &sqlx::PgPool) -> Result { + sqlx::query_scalar::<_, i64>( + "SELECT COALESCE(SUM(pg_catalog.pg_database_size(datname)), 0)::bigint \ + FROM pg_catalog.pg_database", + ) + .fetch_one(pool) + .await +} + +/// Postgres `undefined_function` (SQLSTATE 42883) — missing +/// `public.pg_cluster_size` when the `neon` extension is not installed. +fn pg_cluster_size_unavailable(err: &sqlx::Error) -> bool { + err.as_database_error().and_then(|db| db.code()).as_deref() == Some("42883") +} + +/// `None` = no cap (self-host / unset / unparseable / unlimited `-1`). +/// Neon `neon.max_cluster_size` is MB. +fn neon_max_cluster_size_bytes(setting: Option<&str>) -> Option { + let setting = setting?.trim(); + if setting.is_empty() { + return None; + } + let n = setting.parse::().ok()?; + if n <= 0 { + return None; + } + n.checked_mul(1024 * 1024) +} + +fn postgres_can_accept_writes(used_bytes: i64, max_bytes: i64) -> bool { + used_bytes.saturating_add(POSTGRES_EXTEND_HEADROOM_BYTES) < max_bytes } static WRITE_READY_CACHE: std::sync::Mutex> = @@ -546,6 +673,85 @@ fn clamp_restore_limit(limit: usize) -> usize { limit.clamp(1, 100) } +/// Count restore `skipped` / `failed` over the on-chain page. +/// +/// `skipped` is on-chain blobs already in the local **success** index +/// (`existing_blob_ids`). Negative-cached blob IDs are not skipped. +/// +/// `failed` is on-chain blobs in `failed_blob_ids` (the owner+namespace +/// negative cache). Both counts share `on_chain_blob_ids` as their domain, +/// so neither can exceed `total`. New permanent failures this call are +/// added by the caller after inspection. +fn restore_skip_fail_counts( + on_chain_blob_ids: &[String], + existing_blob_ids: &[String], + failed_blob_ids: &[String], +) -> (usize, usize) { + let existing_set: std::collections::HashSet<&str> = + existing_blob_ids.iter().map(|s| s.as_str()).collect(); + let failed_set: std::collections::HashSet<&str> = + failed_blob_ids.iter().map(|s| s.as_str()).collect(); + let skipped = on_chain_blob_ids + .iter() + .filter(|id| existing_set.contains(id.as_str())) + .count(); + let failed = on_chain_blob_ids + .iter() + .filter(|id| failed_set.contains(id.as_str())) + .count(); + (skipped, failed) +} + +/// Force `truncated` when an inspected page produced only transients +/// (download / SEAL infra / embed). Permanent failures are counted in +/// `failed` and must not be retried; embed failures are not negative-cached. +fn restore_truncated_after_page( + truncated: bool, + restored: usize, + newly_failed: usize, + transient_unresolved: usize, +) -> bool { + truncated || (restored == 0 && newly_failed == 0 && transient_unresolved > 0) +} + +enum RestoreDecrypt { + Ok(String, String), + PermanentFail, + TransientFail, +} + +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +enum RestoreFailStage { + InvalidUtf8, + Decrypt, +} + +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +enum RestoreFailClass { + Permanent, + Transient, +} + +/// Classify a restore decrypt/UTF-8 failure. Swapping permanent/transient +/// here would negative-cache blobs during a SEAL infra blip. +fn restore_fail_class(stage: RestoreFailStage, decrypt_err: Option<&str>) -> RestoreFailClass { + match stage { + RestoreFailStage::InvalidUtf8 => RestoreFailClass::Permanent, + RestoreFailStage::Decrypt => match decrypt_err { + Some(err) if seal::DecryptOutcome::permanent_from_error(err) => { + RestoreFailClass::Permanent + } + _ => RestoreFailClass::Transient, + }, + } +} + +enum RestoreDownload { + Ok(String, Vec), + Expired, + Transient, +} + /// POST /api/restore /// /// Restore a namespace from Walrus: @@ -649,6 +855,7 @@ async fn restore_unbounded( return Ok(Json(RestoreResponse { restored: 0, skipped: 0, + failed: 0, total: 0, namespace: namespace.clone(), owner: owner.clone(), @@ -661,18 +868,18 @@ async fn restore_unbounded( // (GH #501 / WALM-299 negative cache — see `db.record_restore_failure`). // A foreign/attacker blob that already failed SEAL decrypt or UTF-8 // validation for this owner+namespace is never re-downloaded and - // re-decrypt-attempted on a later call; it's already correctly reported - // as "skipped", same as any other missing-but-excluded blob. + // re-decrypt-attempted on a later call; it counts as `failed`, not + // `skipped` (COMG-719 / GH #399). let existing_blob_ids = state.db.get_blobs_by_namespace(owner, namespace).await?; let failed_blob_ids = state.db.get_failed_blob_ids(owner, namespace).await?; - let existing_set: std::collections::HashSet<&str> = existing_blob_ids + let exclude_set: std::collections::HashSet<&str> = existing_blob_ids .iter() .map(|s| s.as_str()) .chain(failed_blob_ids.iter().map(|s| s.as_str())) .collect(); let all_missing: Vec = all_blob_ids .iter() - .filter(|id| !existing_set.contains(id.as_str())) + .filter(|id| !exclude_set.contains(id.as_str())) .cloned() .collect(); // Apply limit — query-blobs' on-chain ordering is unspecified (the @@ -693,12 +900,15 @@ async fn restore_unbounded( missing_blob_ids.len(), limit, ); - let skipped = total - missing_blob_ids.len(); + let (skipped, failed) = + restore_skip_fail_counts(&all_blob_ids, &existing_blob_ids, &failed_blob_ids); tracing::info!( - "restore: total={} on-chain, existing={}, negative-cached={}, missing={} (limited to {}, truncated={}, source_capped={}) for ns={}", + "restore: total={} on-chain, existing={}, negative-cached={}, skipped={}, failed={}, missing={} (limited to {}, truncated={}, source_capped={}) for ns={}", total, existing_blob_ids.len(), failed_blob_ids.len(), + skipped, + failed, missing_blob_ids.len(), limit, truncated, @@ -710,6 +920,7 @@ async fn restore_unbounded( return Ok(Json(RestoreResponse { restored: 0, skipped, + failed, total, namespace: namespace.clone(), owner: owner.clone(), @@ -740,7 +951,7 @@ async fn restore_unbounded( ) .await { - Ok(data) => Some((blob_id, data)), + Ok(data) => RestoreDownload::Ok(blob_id, data), Err(AppError::BlobNotFound(msg)) => { tracing::warn!("restore: blob expired, skipping: {}", msg); cleanup_expired_blob( @@ -750,11 +961,11 @@ async fn restore_unbounded( &namespace_for_cleanup, ) .await; - None + RestoreDownload::Expired } Err(e) => { tracing::warn!("restore: download failed for {}: {}", blob_id, e); - None + RestoreDownload::Transient } } } @@ -765,11 +976,19 @@ async fn restore_unbounded( // OOM when restoring large namespaces. join_all() with hundreds of blobs // would spawn all downloads simultaneously → memory spike. // We use buffer_unordered(10) to cap parallelism at 10 concurrent downloads. - let downloaded: Vec<(String, Vec)> = stream::iter(download_tasks) + let download_results: Vec = stream::iter(download_tasks) .buffer_unordered(10) - .filter_map(|opt| async move { opt }) .collect() .await; + let mut downloaded = Vec::with_capacity(download_results.len()); + let mut transient_unresolved = 0usize; + for result in download_results { + match result { + RestoreDownload::Ok(blob_id, data) => downloaded.push((blob_id, data)), + RestoreDownload::Transient => transient_unresolved += 1, + RestoreDownload::Expired => {} + } + } // Preserve encrypted blob sizes so restored rows still contribute to storage quota. let blob_sizes: std::collections::HashMap = downloaded @@ -778,9 +997,11 @@ async fn restore_unbounded( .collect(); if downloaded.is_empty() { + let truncated = restore_truncated_after_page(truncated, 0, 0, transient_unresolved); return Ok(Json(RestoreResponse { restored: 0, skipped, + failed, total, namespace: namespace.clone(), owner: owner.clone(), @@ -795,7 +1016,7 @@ async fn restore_unbounded( ); // Step 4: SEAL decrypt with bounded concurrency (3 at a time). - let decrypt_results: Vec> = stream::iter(downloaded) + let decrypt_results: Vec = stream::iter(downloaded) .map(|(blob_id, encrypted_data)| { let http_client = &state.http_client; let sidecar_url = state.config.sidecar_url.clone(); @@ -822,23 +1043,33 @@ async fn restore_unbounded( .await { Ok(plaintext) => match String::from_utf8(plaintext) { - Ok(text) => Some((blob_id, text)), + Ok(text) => RestoreDecrypt::Ok(blob_id, text), Err(e) => { tracing::warn!("restore: invalid UTF-8 for {}: {}", blob_id, e); // Decrypt already succeeded here, so invalid UTF-8 // is inherently deterministic for this blob — always // safe to negative-cache (GH #501 / WALM-299). - if let Err(db_err) = db - .record_restore_failure(&owner, &namespace, &blob_id, "invalid_utf8") - .await - { - tracing::warn!( - "restore: failed to record invalid-UTF-8 negative cache for {}: {}", - blob_id, - db_err - ); + match restore_fail_class(RestoreFailStage::InvalidUtf8, None) { + RestoreFailClass::Permanent => { + if let Err(db_err) = db + .record_restore_failure( + &owner, + &namespace, + &blob_id, + "invalid_utf8", + ) + .await + { + tracing::warn!( + "restore: failed to record invalid-UTF-8 negative cache for {}: {}", + blob_id, + db_err + ); + } + RestoreDecrypt::PermanentFail + } + RestoreFailClass::Transient => RestoreDecrypt::TransientFail, } - None } }, Err(e) => { @@ -849,24 +1080,30 @@ async fn restore_unbounded( // rate limit) must keep being retried; caching those // could permanently and wrongly blacklist a // legitimate blob during an infra blip. - if seal::DecryptOutcome::permanent_from_error(&e.to_string()) { - if let Err(db_err) = db - .record_restore_failure( - &owner, - &namespace, - &blob_id, - "decrypt_permanent", - ) - .await - { - tracing::warn!( - "restore: failed to record decrypt-permanent negative cache for {}: {}", - blob_id, - db_err - ); + match restore_fail_class( + RestoreFailStage::Decrypt, + Some(&e.to_string()), + ) { + RestoreFailClass::Permanent => { + if let Err(db_err) = db + .record_restore_failure( + &owner, + &namespace, + &blob_id, + "decrypt_permanent", + ) + .await + { + tracing::warn!( + "restore: failed to record decrypt-permanent negative cache for {}: {}", + blob_id, + db_err + ); + } + RestoreDecrypt::PermanentFail } + RestoreFailClass::Transient => RestoreDecrypt::TransientFail, } - None } } } @@ -875,7 +1112,22 @@ async fn restore_unbounded( .collect() .await; - let decrypted_texts: Vec<(String, String)> = decrypt_results.into_iter().flatten().collect(); + let newly_failed = decrypt_results + .iter() + .filter(|r| matches!(r, RestoreDecrypt::PermanentFail)) + .count(); + transient_unresolved += decrypt_results + .iter() + .filter(|r| matches!(r, RestoreDecrypt::TransientFail)) + .count(); + let failed = failed + newly_failed; + let decrypted_texts: Vec<(String, String)> = decrypt_results + .into_iter() + .filter_map(|r| match r { + RestoreDecrypt::Ok(blob_id, text) => Some((blob_id, text)), + RestoreDecrypt::PermanentFail | RestoreDecrypt::TransientFail => None, + }) + .collect(); tracing::info!( "restore: decrypted {}/{} blobs", decrypted_texts.len(), @@ -910,7 +1162,10 @@ async fn restore_unbounded( .collect(); // Step 6: Insert only new entries (no delete!) + transient_unresolved += decrypted_texts.len().saturating_sub(results.len()); let restored = results.len(); + let truncated = + restore_truncated_after_page(truncated, restored, newly_failed, transient_unresolved); for (blob_id, vector) in &results { let id = uuid::Uuid::new_v4().to_string(); let blob_size = blob_sizes.get(blob_id).copied().unwrap_or_else(|| { @@ -957,9 +1212,10 @@ async fn restore_unbounded( } tracing::info!( - "restore complete: restored={} skipped={} total={} owner={} ns={}", + "restore complete: restored={} skipped={} failed={} total={} owner={} ns={}", restored, skipped, + failed, total, owner, namespace @@ -968,6 +1224,7 @@ async fn restore_unbounded( Ok(Json(RestoreResponse { restored, skipped, + failed, total, namespace: namespace.clone(), owner: owner.clone(), @@ -1036,6 +1293,102 @@ mod tests { } } + // ── /health write_ready Postgres size cap (WALM-612) ────────────── + + #[test] + fn neon_max_cluster_size_bytes_parses_mb_and_unlimited() { + assert!(super::neon_max_cluster_size_bytes(None).is_none()); + assert!(super::neon_max_cluster_size_bytes(Some("-1")).is_none()); + assert!(super::neon_max_cluster_size_bytes(Some("0")).is_none()); + assert_eq!( + super::neon_max_cluster_size_bytes(Some("3072")), + Some(3072 * 1024 * 1024) + ); + } + + #[test] + fn postgres_can_accept_writes_false_at_or_within_1mb_of_cap() { + let max = 3072 * 1024 * 1024; + assert!(!super::postgres_can_accept_writes(max, max)); + assert!(!super::postgres_can_accept_writes( + max - super::POSTGRES_EXTEND_HEADROOM_BYTES, + max + )); + assert!(super::postgres_can_accept_writes( + max - super::POSTGRES_EXTEND_HEADROOM_BYTES - 1, + max + )); + } + + #[test] + fn postgres_write_ready_probe_timeout_is_one_second() { + assert_eq!( + super::WRITE_READY_PROBE_TIMEOUT, + std::time::Duration::from_secs(1) + ); + } + + #[test] + fn missing_pg_cluster_size_falls_back_instead_of_fail_open() { + let missing = sqlx::Error::Database(Box::new(FakePgError { + message: "function public.pg_cluster_size() does not exist", + code: Some("42883"), + })); + assert!(super::pg_cluster_size_unavailable(&missing)); + + let other = sqlx::Error::Database(Box::new(FakePgError { + message: "connection reset", + code: Some("08006"), + })); + assert!(!super::pg_cluster_size_unavailable(&other)); + + let disk_full = sqlx::Error::Database(Box::new(FakePgError { + message: "could not extend file because project size limit (3072 MB) has been exceeded", + code: Some("53100"), + })); + assert!(!super::pg_cluster_size_unavailable(&disk_full)); + } + + #[derive(Debug)] + struct FakePgError { + message: &'static str, + code: Option<&'static str>, + } + + impl std::fmt::Display for FakePgError { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + f.write_str(self.message) + } + } + + impl std::error::Error for FakePgError {} + + impl sqlx::error::DatabaseError for FakePgError { + fn message(&self) -> &str { + self.message + } + + fn kind(&self) -> sqlx::error::ErrorKind { + sqlx::error::ErrorKind::Other + } + + fn code(&self) -> Option> { + self.code.map(std::borrow::Cow::Borrowed) + } + + fn as_error(&self) -> &(dyn std::error::Error + Send + Sync + 'static) { + self + } + + fn as_error_mut(&mut self) -> &mut (dyn std::error::Error + Send + Sync + 'static) { + self + } + + fn into_error(self: Box) -> Box { + self + } + } + // ── /api/restore body.limit cap (GH #501 / WALM-299) ──────────────── // // `RestoreRequest.limit` (plain `usize`, serde default 10 — unlike @@ -1140,6 +1493,7 @@ mod tests { let resp = RestoreResponse { restored: 5, skipped: 2, + failed: 0, total: 20, namespace: "ns".to_string(), owner: "0xabc".to_string(), @@ -1148,6 +1502,122 @@ mod tests { assert!(resp.truncated); } + #[test] + fn restore_response_serializes_failed_field() { + let resp = RestoreResponse { + restored: 5, + skipped: 2, + failed: 3, + total: 20, + namespace: "ns".to_string(), + owner: "0xabc".to_string(), + truncated: false, + }; + let json = serde_json::to_value(&resp).unwrap(); + assert_eq!(json["failed"], 3); + assert_eq!(json["skipped"], 2); + assert!(json.get("failed").is_some()); + } + + #[test] + fn restore_skip_fail_counts_excludes_negative_cache_from_skipped() { + let on_chain = vec!["a", "b", "c", "d"] + .into_iter() + .map(String::from) + .collect::>(); + let existing = vec!["a", "b"] + .into_iter() + .map(String::from) + .collect::>(); + let failed = vec!["c".to_string()]; + + let (skipped, failed_count) = + super::restore_skip_fail_counts(&on_chain, &existing, &failed); + + assert_eq!(skipped, 2, "skipped is on-chain success index only"); + assert_eq!(failed_count, 1, "failed is page ∩ negative cache"); + } + + #[test] + fn restore_skip_fail_counts_does_not_count_off_chain_existing() { + let on_chain = vec!["a".to_string()]; + let existing = vec!["a".to_string(), "ghost".to_string()]; + let none: Vec = vec![]; + + let (skipped, failed_count) = super::restore_skip_fail_counts(&on_chain, &existing, &none); + + assert_eq!(skipped, 1); + assert_eq!(failed_count, 0); + } + + #[test] + fn restore_skip_fail_counts_intersects_failed_with_page() { + let on_chain = vec!["a".to_string(), "b".to_string()]; + let none: Vec = vec![]; + let failed = vec![ + "old-fail".to_string(), + "a".to_string(), + "also-old".to_string(), + ]; + + let (skipped, failed_count) = super::restore_skip_fail_counts(&on_chain, &none, &failed); + + assert_eq!(skipped, 0); + assert_eq!( + failed_count, 1, + "historical cache still skips re-download, but failed counts only this page" + ); + assert!(failed_count <= on_chain.len()); + } + + #[test] + fn restore_truncated_after_page_signals_retry_on_transients_only() { + // Embedder down / download blip: inspected page yielded neither a + // restore nor a permanent failure. source_capped=false would otherwise + // leave truncated=false and the caller would not retry (WALM-480). + assert!(super::restore_truncated_after_page(false, 0, 0, 10)); + assert!(super::restore_truncated_after_page(false, 0, 0, 1)); + assert!(super::restore_truncated_after_page(true, 5, 0, 0)); + assert!(!super::restore_truncated_after_page(false, 1, 0, 9)); + assert!(!super::restore_truncated_after_page(false, 0, 10, 0)); + assert!(!super::restore_truncated_after_page(false, 0, 1, 9)); + assert!(!super::restore_truncated_after_page(false, 0, 0, 0)); + } + + #[test] + fn restore_fail_class_pins_permanent_vs_transient() { + use super::{restore_fail_class, RestoreFailClass, RestoreFailStage}; + + assert_eq!( + restore_fail_class(RestoreFailStage::InvalidUtf8, None), + RestoreFailClass::Permanent, + "invalid UTF-8 is deterministic for the blob" + ); + assert_eq!( + restore_fail_class(RestoreFailStage::Decrypt, Some("InvalidCiphertext")), + RestoreFailClass::Permanent + ); + assert_eq!( + restore_fail_class( + RestoreFailStage::Decrypt, + Some( + "seal decrypt failed: seal/decrypt failed during fetch_keys: \ + NoAccessError: user does not have access to one or more of \ + the requested keys (traceId=abc123, timeoutMs=10000)" + ) + ), + RestoreFailClass::Permanent + ); + assert_eq!( + restore_fail_class( + RestoreFailStage::Decrypt, + Some("TimeoutError: The operation was aborted due to timeout") + ), + RestoreFailClass::Transient, + "SEAL infra blips must not be negative-cached" + ); + } + // ── /api/forget + /api/stats empty-namespace validation ───────────── // // Both handlers reject an empty namespace with `AppError::BadRequest` diff --git a/services/server/src/routes/mod.rs b/services/server/src/routes/mod.rs index 5f4349341..abd037568 100644 --- a/services/server/src/routes/mod.rs +++ b/services/server/src/routes/mod.rs @@ -72,15 +72,28 @@ pub async fn enqueue_wallet_job( operation: WalletOperation, ) -> Result { let mut storage = state.wallet_storage.clone(); - storage + match storage .push_request(wallet_job_request(WalletJob { wallet_index, congestion_requeues: 0, operation, })) .await - .map_err(|e| AppError::Internal(format!("Failed to enqueue WalletJob: {}", e)))?; - Ok(wallet_index) + { + Ok(_) => Ok(wallet_index), + Err(e) => { + crate::alerts::maybe_alert_sqlx_postgres_storage_exhausted( + &state.alerts, + &state.config.sui_network, + &e, + ) + .await; + Err(AppError::Internal(format!( + "Failed to enqueue WalletJob: {}", + e + ))) + } + } } // ============================================================ diff --git a/services/server/src/routes/remember.rs b/services/server/src/routes/remember.rs index 73bf3f2c9..62a99dd31 100644 --- a/services/server/src/routes/remember.rs +++ b/services/server/src/routes/remember.rs @@ -903,7 +903,7 @@ pub async fn remember( // stays `pending` → the guard takes the plain Upload path, no on-chain // reconcile round-trip on the happy path. (The 202 response is still // "running" for API compatibility — see below.) - let inserted = sqlx::query( + let inserted = match sqlx::query( "INSERT INTO remember_jobs (id, owner, namespace, status, idempotency_key, request_fingerprint) VALUES ($1, $2, $3, 'pending', $4, $5) ON CONFLICT (owner, idempotency_key) WHERE idempotency_key IS NOT NULL DO NOTHING", ) @@ -914,7 +914,18 @@ pub async fn remember( .bind(body.idempotency_key.as_ref().map(|_| fingerprint.as_str())) .execute(state.db.pool()) .await - .map_err(|e| AppError::Internal(format!("Failed to create job row: {}", e)))?; + { + Ok(inserted) => inserted, + Err(e) => { + crate::alerts::maybe_alert_sqlx_postgres_storage_exhausted( + &state.alerts, + &state.config.sui_network, + &e, + ) + .await; + return Err(AppError::Internal(format!("Failed to create job row: {}", e))); + } + }; // Lost the race against a concurrent same-key request — return the winner's // job rather than spawning a duplicate write. @@ -1295,7 +1306,7 @@ pub async fn remember_bulk( for item in body.items { let job_id = uuid::Uuid::new_v4().to_string(); - sqlx::query( + if let Err(e) = sqlx::query( // `pending` (not `running`) so a fresh job takes the plain Upload // path; only a retry of an in-flight job (worker-set `running`) // triggers the crash-window reconcile. See the single-remember insert. @@ -1306,7 +1317,18 @@ pub async fn remember_bulk( .bind(&item.namespace) .execute(state.db.pool()) .await - .map_err(|e| AppError::Internal(format!("Failed to create bulk job row: {}", e)))?; + { + crate::alerts::maybe_alert_sqlx_postgres_storage_exhausted( + &state.alerts, + &state.config.sui_network, + &e, + ) + .await; + return Err(AppError::Internal(format!( + "Failed to create bulk job row: {}", + e + ))); + } pending_items.push(PendingBulkRememberItem { job_id: job_id.clone(), diff --git a/services/server/src/security_delete_auth.rs b/services/server/src/security_delete_auth.rs index bdde0e170..8f8b3b3fd 100644 --- a/services/server/src/security_delete_auth.rs +++ b/services/server/src/security_delete_auth.rs @@ -8,7 +8,7 @@ use axum::Json; use base64::engine::general_purpose::{STANDARD, URL_SAFE_NO_PAD}; use base64::Engine; use hmac::{Hmac, Mac}; -use redis::aio::MultiplexedConnection; +use redis::aio::ConnectionManager; use serde::{Deserialize, Serialize}; use sha2::Sha256; use sui_crypto::{simple::SimpleVerifier, SuiVerifier}; @@ -168,11 +168,11 @@ pub trait NonceStore: Send + Sync { } pub struct RedisNonceStore { - connection: MultiplexedConnection, + connection: ConnectionManager, } impl RedisNonceStore { - pub fn new(connection: MultiplexedConnection) -> Self { + pub fn new(connection: ConnectionManager) -> Self { Self { connection } } } @@ -405,7 +405,7 @@ mod tests { return; }; let client = redis::Client::open(url).unwrap(); - let connection = client.get_multiplexed_async_connection().await.unwrap(); + let connection = redis::aio::ConnectionManager::new(client).await.unwrap(); let store = RedisNonceStore::new(connection); let id = Uuid::new_v4().to_string(); store.issue(&id, r#"{"ok":true}"#, 30).await.unwrap(); diff --git a/services/server/src/storage/db.rs b/services/server/src/storage/db.rs index 09833120b..90e8cd18c 100644 --- a/services/server/src/storage/db.rs +++ b/services/server/src/storage/db.rs @@ -1,7 +1,10 @@ +use std::sync::Arc; + use pgvector::Vector; use sqlx::postgres::PgPoolOptions; use sqlx::PgPool; +use crate::alerts::AlertManager; use crate::types::{AppError, SearchHit}; /// Tombstone retention for both the read-API `must_resync` clock and the @@ -10,12 +13,32 @@ pub const TOMBSTONE_RETENTION: chrono::Duration = chrono::Duration::days(30); pub struct VectorDb { pool: PgPool, + storage_alerts: Option<(Arc, String)>, +} + +impl VectorDb { + pub fn with_storage_alerts(self, alerts: Arc, sui_network: String) -> Self { + Self { + storage_alerts: Some((alerts, sui_network)), + ..self + } + } + + async fn maybe_alert_storage_exhausted(&self, err: &sqlx::Error) { + let Some((alerts, network)) = &self.storage_alerts else { + return; + }; + crate::alerts::maybe_alert_sqlx_postgres_storage_exhausted(alerts, network, err).await; + } } #[cfg(test)] impl VectorDb { pub(crate) fn from_pool(pool: PgPool) -> Self { - Self { pool } + Self { + pool, + storage_alerts: None, + } } } @@ -89,7 +112,10 @@ mod tests { sqlx::raw_sql(migration).execute(&pool).await.unwrap(); } - Some(VectorDb { pool }) + Some(VectorDb { + pool, + storage_alerts: None, + }) } /// Regression test for the migration-order fixes: batched Rust @@ -1491,7 +1517,10 @@ impl VectorDb { tracing::info!("database connected and migrations applied"); - Ok(Self { pool }) + Ok(Self { + pool, + storage_alerts: None, + }) } /// Expose a reference to the underlying `PgPool` so job handlers @@ -1556,10 +1585,17 @@ impl VectorDb { .bind(package_id) .bind(end_epoch) .execute(&mut *tx) - .await - .map_err(|e| AppError::Internal(format!("Failed to insert vector: {}", e))); - crate::observability::observe_db("vector.insert", db_status(&result), started.elapsed()); - result?; + .await; + if let Err(e) = result { + drop(tx); + self.maybe_alert_storage_exhausted(&e).await; + crate::observability::observe_db("vector.insert", "error", started.elapsed()); + return Err(AppError::Internal(format!( + "Failed to insert vector: {}", + e + ))); + } + crate::observability::observe_db("vector.insert", "ok", started.elapsed()); sqlx::query("DELETE FROM memory_tombstones WHERE memory_id = $1") .bind(id) .execute(&mut *tx) diff --git a/services/server/src/storage/sui.rs b/services/server/src/storage/sui.rs index 9a4bcc88b..e48906014 100644 --- a/services/server/src/storage/sui.rs +++ b/services/server/src/storage/sui.rs @@ -251,6 +251,416 @@ pub async fn list_delegate_keys_cached( Ok(keys) } +// ============================================================ +// Delegate key verification — short-TTL in-memory result cache +// ============================================================ +// +// Positive verifications are trusted for `DELEGATE_VERIFY_CACHE_TTL`, so a +// burst of requests carrying the same credentials costs one `GetObject` +// rather than one each. Only `Ok` is stored, keyed by +// `(account_object_id, public_key_bytes)`. An unavailable RPC does not +// evict; see `VerifyCacheMissAction`. + +/// How long a successful on-chain verification is trusted without +/// re-reading the account object. Matches `DELEGATE_KEYS_CACHE_TTL`. +/// +/// This is the upper bound on delegate-key revocation latency at the +/// relayer: a key revoked on-chain keeps authenticating for at most this +/// long. 30s is the same staleness the `/agents` listing already accepts, +/// and is the deliberate trade for removing the retry amplifier. +pub const DELEGATE_VERIFY_CACHE_TTL: std::time::Duration = std::time::Duration::from_secs(30); + +/// How far past `DELEGATE_VERIFY_CACHE_TTL` an entry may still be served — +/// but *only* when the chain itself is unreachable. +/// +/// The 30s TTL assumes dense traffic: several requests carrying the same +/// credentials inside one window. Real MCP usage is not dense. A user who +/// calls a tool every few minutes misses the cache every single time, so +/// while the public fullnode is throttling they take a 503 on each attempt +/// even though their key verified cleanly minutes ago — measured on +/// production as 155,874 × 503 against 50,958 × 200 on `/api/mcp/sse`, and +/// six consecutive failed handshakes with a valid registered key. +/// +/// Serving the stale entry in exactly that case turns a hard 503 into a +/// successful call. The trade is bounded and narrow: revocation latency +/// stays 30s whenever the chain answers, and stretches to 10 minutes only +/// while the chain cannot be read at all — a window in which the relayer +/// could not have observed the revoke anyway. +pub const DELEGATE_VERIFY_STALE_GRACE: std::time::Duration = + std::time::Duration::from_secs(600); + +#[derive(Clone)] +pub struct TimedVerifiedOwner { + /// Owner address returned by the verification that populated this entry. + pub owner: String, + pub verified_at: std::time::Instant, +} + +impl TimedVerifiedOwner { + /// Whether this entry may be served on the ordinary path — the window + /// in which a verification is trusted without re-reading the chain. + /// The sweeper uses `is_servable_while_unavailable` instead, because an + /// entry past this point is still worth keeping for the outage path. + pub fn is_fresh(&self) -> bool { + self.verified_at.elapsed() < DELEGATE_VERIFY_CACHE_TTL + } + + /// Whether this entry may be served *because the chain is unreachable*. + /// Never consulted on the healthy path: a caller reaches this only after + /// a live read already failed with an unavailable error. + pub fn is_servable_while_unavailable(&self) -> bool { + self.verified_at.elapsed() < DELEGATE_VERIFY_CACHE_TTL + DELEGATE_VERIFY_STALE_GRACE + } +} + +/// Keyed by `(account_object_id, public_key_bytes)` so one account's +/// entry can never authenticate a different delegate key. +/// +/// Only *successful* verifications are stored, so an entry always +/// corresponds to a delegate key that is really registered on an account: +/// the map is bounded by real accounts, not by what callers send. A +/// rejection records nothing, which is also why an unregistered key +/// cannot be used to grow this map. +/// +/// `expected_type_origin_package_id` is deliberately not part of the key: +/// it comes from `Config::package_id`, which is fixed for the life of the +/// process, so it cannot vary between a cache write and a later hit. +/// Borrowed view of an `(account_object_id, public_key_bytes)` key. +/// +/// `HashMap<(String, Vec), _>` cannot be probed with `(&str, &[u8])`, and +/// this lookup now runs on every signed request *and* every MCP envelope, so +/// both caches below are keyed through this trait object rather than +/// allocating a `String` and a `Vec` per hit. `(String, Vec)` and +/// `(&str, &[u8])` hash identically — tuples hash element-wise, `String` +/// hashes as its `str`, and `Vec` as its `[u8]` — so the borrowed probe +/// finds the owned key. Same shape as `DelegatePairKey` in #882; this is the +/// two-map version, since the verify cache and the rejection cache share a +/// key. `Send + Sync` because the probe is held across an `.await`, and a +/// non-`Sync` referent there would make the surrounding futures non-`Send`. +pub trait DelegateAccountKey: Send + Sync { + fn parts(&self) -> (&str, &[u8]); +} + +impl DelegateAccountKey for (String, Vec) { + fn parts(&self) -> (&str, &[u8]) { + (self.0.as_str(), self.1.as_slice()) + } +} + +impl DelegateAccountKey for (&str, &[u8]) { + fn parts(&self) -> (&str, &[u8]) { + (self.0, self.1) + } +} + +impl std::hash::Hash for dyn DelegateAccountKey + '_ { + fn hash(&self, state: &mut H) { + self.parts().hash(state); + } +} + +impl PartialEq for dyn DelegateAccountKey + '_ { + fn eq(&self, other: &Self) -> bool { + self.parts() == other.parts() + } +} + +impl Eq for dyn DelegateAccountKey + '_ {} + +impl<'a> std::borrow::Borrow for (String, Vec) { + fn borrow(&self) -> &(dyn DelegateAccountKey + 'a) { + self + } +} + +pub struct DelegateVerifyCacheState { + pub entries: + tokio::sync::RwLock), TimedVerifiedOwner>>, + /// Bumped by every definitive eviction. + /// + /// Cold misses are deliberately not single-flighted, so two requests for + /// the same pair can be in the chain at once. Without this, request A can + /// start a read, request B can observe a revoke and evict, and A's older + /// success can then land and re-open a full trust window on a key that is + /// already gone. An insert refuses when the generation moved under it, so + /// the revoke wins and the next caller reads the chain again. + pub evictions: std::sync::atomic::AtomicU64, +} + +pub type DelegateVerifyCache = std::sync::Arc; + +pub fn new_delegate_verify_cache() -> DelegateVerifyCache { + std::sync::Arc::new(DelegateVerifyCacheState { + entries: tokio::sync::RwLock::new(std::collections::HashMap::new()), + evictions: std::sync::atomic::AtomicU64::new(0), + }) +} + +// ── Rejections ────────────────────────────────────────────────────────── +// +// Definitive rejections are cached too, briefly. A client holding a key the +// relayer will never accept retries forever, and an uncached rejection makes +// each retry another fullnode read. + +/// How long a definitive rejection is remembered. +/// +/// Deliberately much shorter than the positive TTL, because the cost of +/// being wrong is asymmetric: a stale positive authenticates a revoked key, +/// while a stale negative only delays a key that just became valid. +/// +/// It does not delay an ordinary `memwal_login`: that registers a freshly +/// generated delegate key, so the `(account, pk)` pair has never been +/// rejected and has no entry. What it can delay by up to this long is the +/// narrower case of retrying a key that was tried *before* its registration +/// landed — the interrupted-login path. +pub const DELEGATE_REJECT_CACHE_TTL: std::time::Duration = std::time::Duration::from_secs(10); + +/// Hard ceiling on remembered rejections. +/// +/// Unlike the positive cache, this one is keyed by what *callers send*, not +/// by what exists on chain, so it would otherwise grow one entry per made-up +/// `(account, key)` pair anyone cares to try. At the cap we stop inserting +/// and fall back to the live read — degrading to today's behaviour rather +/// than trading a throttle for unbounded memory. +pub const DELEGATE_REJECT_CACHE_MAX_ENTRIES: usize = 4_096; + +pub type DelegateRejectCache = std::sync::Arc< + tokio::sync::RwLock), std::time::Instant>>, +>; + +pub fn new_delegate_reject_cache() -> DelegateRejectCache { + std::sync::Arc::new(tokio::sync::RwLock::new(std::collections::HashMap::new())) +} + +/// Whether a rejection recorded at `rejected_at` may still be reused. +pub fn reject_entry_is_fresh(rejected_at: std::time::Instant) -> bool { + rejected_at.elapsed() < DELEGATE_REJECT_CACHE_TTL +} + +/// Whether a fresh rejection may be recorded, given the map's current size +/// and whether this pair is already present. Pure so the cap is testable +/// without a chain: refreshing an existing entry is always allowed (it +/// cannot grow the map), a new one only below the cap. +pub fn should_record_rejection(current_len: usize, already_present: bool) -> bool { + already_present || current_len < DELEGATE_REJECT_CACHE_MAX_ENTRIES +} + +/// Cached wrapper around `verify_delegate_key_onchain`. +/// +/// A hit within `DELEGATE_VERIFY_CACHE_TTL` returns the recorded owner +/// without touching the chain. Only successes are cached: a rejection is +/// always a live read, so adding a delegate key (the tail of `login`) +/// takes effect immediately rather than after a TTL. A definitive +/// rejection also evicts any entry for that pair, so an observed revoke +/// cannot be overtaken by a positive still inside its window. +/// +/// When the live read fails *because the chain is unreachable*, a stale +/// entry within `DELEGATE_VERIFY_STALE_GRACE` is served rather than +/// surfacing the outage to a caller whose key is known good. Only an +/// unavailable error takes this path — a definitive rejection is still +/// returned, and still evicts. Without it the TTL helps only callers who +/// repeat inside 30s, which is not how the MCP clients that hit this +/// actually behave (WALM-618). +pub async fn verify_delegate_key_cached( + cache: &DelegateVerifyCache, + reject_cache: &DelegateRejectCache, + http_client: &reqwest::Client, + rpc_url: &str, + grpc_client: Option<&sui_rpc::Client>, + account_object_id: &str, + public_key_bytes: &[u8], + expected_type_origin_package_id: &str, +) -> Result { + // Borrowed probe: a hit — the common case on both hot paths — allocates + // nothing. The owned key is built only where the map is actually written. + let probe: &dyn DelegateAccountKey = &(account_object_id, public_key_bytes); + + if let Some(cached) = cache + .entries + .read() + .await + .get(probe) + .filter(|c| c.is_fresh()) + { + return Ok(cached.owner.clone()); + } + + // Read before the chain call, compared after it. See `evictions`. + let generation_before = cache.evictions.load(std::sync::atomic::Ordering::Acquire); + // Stamped from BEFORE the read, not after: a slow `GetObject` would + // otherwise extend the stated revocation bound by its own duration. + let verify_started = std::time::Instant::now(); + + // A pair we refused moments ago is refused again without a chain read. + // Checked after the positive lookup so a key that has since been + // registered and verified is never held back by an older rejection. + if reject_cache + .read() + .await + .get(probe) + .copied() + .is_some_and(reject_entry_is_fresh) + { + return Err(OnchainVerifyError::KeyNotFound(format!( + "delegate key not registered on account {account_object_id} (cached)" + ))); + } + + match verify_delegate_key_onchain( + http_client, + rpc_url, + grpc_client, + account_object_id, + public_key_bytes, + expected_type_origin_package_id, + ) + .await + { + Ok(owner) => { + let key = (account_object_id.to_string(), public_key_bytes.to_vec()); + // The unconditional `reject_cache.remove` that used to stand here + // is deleted rather than made conditional. Any entry present for + // this pair at this point was either stamped before this read — + // in which case the lookup above already stepped past it, so it + // was expired and inert — or stamped *during* it, which means a + // concurrent request saw the chain refuse this pair. Removing the + // second kind is how a revoke ended up recorded in neither map: + // no positive entry (correct, the generation check below declines + // it) and no rejection either (wrong). + let mut entries = cache.entries.write().await; + if may_store_verification( + generation_before, + cache.evictions.load(std::sync::atomic::Ordering::Acquire), + ) { + entries.insert( + key, + TimedVerifiedOwner { + owner: owner.clone(), + verified_at: verify_started, + }, + ); + } + // Otherwise a concurrent request saw something definitive while + // this read was in flight. Answer this caller — the read did + // succeed — but do not cache a result the chain has since + // contradicted. + Ok(owner) + } + Err(err) => { + if verify_cache_miss_action(&err) == VerifyCacheMissAction::Evict { + // One guard across the removal, the bump and the rejection + // record. A concurrent success takes this same `entries` lock + // to insert and reads the generation while holding it, so it + // either runs before this block (and its entry is deleted by + // the remove) or after it (and sees the bumped generation and + // declines). Bumping after the guard dropped left a window + // where it could do neither, which is the race `evictions` + // exists to close. + let mut entries = cache.entries.write().await; + entries.remove(probe); + cache + .evictions + .fetch_add(1, std::sync::atomic::Ordering::AcqRel); + // Remember the refusal so a client looping on a key that can + // never be accepted stops costing one fullnode read per retry. + let mut rejects = reject_cache.write().await; + let present = rejects.contains_key(probe); + // Expire before consulting the cap. The TTL is 10s and the + // sweeper runs every 300s, so the raw length counts up to + // thirty generations of entries that can no longer be served + // by anyone. Without this, a few thousand distinct made-up + // pairs hold every slot for five minutes, during which no + // genuine rejection is recorded at all and every looping + // client is back to one fullnode read per retry — the exact + // amplifier this cache was added to remove. Same policy as + // the refusal-log sampler in `observability`. + if !present && rejects.len() >= DELEGATE_REJECT_CACHE_MAX_ENTRIES { + rejects.retain(|_, rejected_at| reject_entry_is_fresh(*rejected_at)); + } + // At the cap we still simply do not record — the next attempt + // reads the chain exactly as it does today. + if should_record_rejection(rejects.len(), present) { + rejects.insert( + (account_object_id.to_string(), public_key_bytes.to_vec()), + std::time::Instant::now(), + ); + } + drop(rejects); + drop(entries); + return Err(err); + } + // Unavailable: the chain proved nothing about this key, so a + // recent success is still the best evidence we have. Serving it + // is what keeps a valid caller working through a fullnode + // throttle instead of collecting a 503 per attempt. + let stale = cache + .entries + .read() + .await + .get(probe) + .filter(|c| c.is_servable_while_unavailable()) + .map(|c| (c.owner.clone(), c.verified_at.elapsed())); + match stale { + Some((owner, age)) => { + // Sampled per account, like the refusal line. This runs on + // every signed request and every MCP envelope, and it fires + // hardest exactly when an outage is pushing many requests + // past the TTL at once — one line per request there is the + // repetition the sampler exists to stop. + if crate::observability::should_log_stale_serve(account_object_id) { + tracing::warn!( + account_id = %account_object_id, + age_secs = age.as_secs(), + error = %err, + "serving stale delegate verification while Sui is unavailable (sampled)" + ); + } + Ok(owner) + } + None => Err(err), + } + } + } +} + +/// What a failed live verification means for any cached entry on the same +/// `(account, key)` pair. Pure so the policy is testable without a chain. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum VerifyCacheMissAction { + /// The rejection is definitive (revoked, deactivated, wrong object) — + /// drop the pair so a positive still inside its TTL cannot outlive the + /// revoke we just observed. + Evict, + /// An unavailable RPC proves nothing about the key, so leave the entry + /// alone. Load-bearing in two distinct ways, neither of them obvious: + /// + /// - The entry this thread missed on is what + /// `DELEGATE_VERIFY_STALE_GRACE` goes on to serve, so evicting here + /// would delete exactly what the outage path exists to use. + /// - A *different* request may have verified successfully in the window + /// between this thread's miss and its failed read. Evicting on an + /// unavailable error would throw away that fresh, valid entry on the + /// strength of an RPC failure that says nothing about the key. + Keep, +} + +/// Whether a completed verification may still be stored. +/// +/// False when a definitive eviction landed while the read was in flight: the +/// chain has since contradicted this answer, so caching it would re-open a +/// trust window on a key another request already saw revoked. +pub fn may_store_verification(generation_before: u64, generation_now: u64) -> bool { + generation_before == generation_now +} + +pub fn verify_cache_miss_action(err: &OnchainVerifyError) -> VerifyCacheMissAction { + if err.is_unavailable() { + VerifyCacheMissAction::Keep + } else { + VerifyCacheMissAction::Evict + } +} + /// Parse the `delegate_keys` array out of a MemWalAccount's `fields` map. /// Pure function — no I/O — so it's unit-testable without a live chain. pub fn parse_delegate_keys( @@ -1600,6 +2010,640 @@ mod tests { ); } + // ── verify_delegate_key_cached (WALM-618) ─────────────────────────── + // + // Same no-mock-HTTP technique as the block above: a cache HIT returns + // before any network attempt (so it succeeds against an unreachable RPC + // URL), a MISS falls through to the real request (so it fails). That is + // exactly the branch the fix turns on — an uncached verify ran on every + // signed API call and every MCP envelope, ~10 fullnode reads per tool + // call, which is what the public fullnode was throttling. + + fn sample_pk() -> Vec { + vec![7u8; 32] + } + + async fn seed_verify_cache( + cache: &DelegateVerifyCache, + account_id: &str, + pk: &[u8], + age: std::time::Duration, + ) -> String { + let owner = "0xowner-from-cache".to_string(); + cache.entries.write().await.insert( + (account_id.to_string(), pk.to_vec()), + TimedVerifiedOwner { + owner: owner.clone(), + verified_at: std::time::Instant::now() - age, + }, + ); + owner + } + + #[tokio::test] + async fn verify_delegate_key_cached_returns_cached_owner_within_ttl() { + let cache = new_delegate_verify_cache(); + let account_id = "0xaccount-verify-fresh"; + let pk = sample_pk(); + let owner = seed_verify_cache(&cache, account_id, &pk, std::time::Duration::ZERO).await; + + let client = reqwest::Client::new(); + let result = verify_delegate_key_cached( + &cache, + &new_delegate_reject_cache(), + &client, + unreachable_rpc_url(), + None, + account_id, + &pk, + "0xpkg", + ) + .await; + + assert_eq!( + result.ok(), + Some(owner), + "a fresh entry must be served without attempting the on-chain read" + ); + } + + #[tokio::test] + async fn verify_delegate_key_cached_reverifies_after_ttl_expiry() { + // An expired entry must not be served on the ordinary path: the read + // is attempted for real. Here the RPC is unreachable, so the attempt + // fails and the stale-grace path below decides what happens next — + // this test only pins that the live read was actually made. + let cache = new_delegate_verify_cache(); + let account_id = "0xaccount-verify-stale"; + let pk = sample_pk(); + seed_verify_cache( + &cache, + account_id, + &pk, + DELEGATE_VERIFY_CACHE_TTL + std::time::Duration::from_secs(1), + ) + .await; + + let before = cache + .entries + .read() + .await + .get(&(account_id.to_string(), pk.clone())) + .map(|c| c.verified_at); + + let client = reqwest::Client::new(); + let _ = verify_delegate_key_cached( + &cache, + &new_delegate_reject_cache(), + &client, + unreachable_rpc_url(), + None, + account_id, + &pk, + "0xpkg", + ) + .await; + + let after = cache + .entries + .read() + .await + .get(&(account_id.to_string(), pk.clone())) + .map(|c| c.verified_at); + assert_eq!( + before, after, + "a failed re-verify must not refresh the entry's timestamp — otherwise a key \ + could be renewed indefinitely by an outage and never re-checked" + ); + } + + #[tokio::test] + async fn verify_delegate_key_cached_serves_stale_entry_while_chain_unavailable() { + // The WALM-618 case: a valid key, verified minutes ago, used again + // while the public fullnode is throttling. Before this, every such + // call was a 503 even though nothing about the key had changed. + let cache = new_delegate_verify_cache(); + let account_id = "0xaccount-verify-grace"; + let pk = sample_pk(); + let owner = seed_verify_cache( + &cache, + account_id, + &pk, + DELEGATE_VERIFY_CACHE_TTL + std::time::Duration::from_secs(60), + ) + .await; + + let client = reqwest::Client::new(); + let result = verify_delegate_key_cached( + &cache, + &new_delegate_reject_cache(), + &client, + unreachable_rpc_url(), + None, + account_id, + &pk, + "0xpkg", + ) + .await; + + assert_eq!( + result.ok(), + Some(owner), + "an entry inside the stale grace must be served when the chain cannot be read" + ); + } + + #[tokio::test] + async fn verify_delegate_key_cached_refuses_stale_entry_past_the_grace() { + // The grace is bounded. Past it the outage is no longer an excuse and + // the caller gets the unavailable error, so a key revoked during a + // long outage cannot authenticate forever. + let cache = new_delegate_verify_cache(); + let account_id = "0xaccount-verify-past-grace"; + let pk = sample_pk(); + seed_verify_cache( + &cache, + account_id, + &pk, + DELEGATE_VERIFY_CACHE_TTL + + DELEGATE_VERIFY_STALE_GRACE + + std::time::Duration::from_secs(1), + ) + .await; + + let client = reqwest::Client::new(); + let result = verify_delegate_key_cached( + &cache, + &new_delegate_reject_cache(), + &client, + unreachable_rpc_url(), + None, + account_id, + &pk, + "0xpkg", + ) + .await; + + assert!( + result.is_err(), + "past TTL + grace the entry must not be served, outage or not" + ); + } + + #[test] + fn a_borrowed_probe_finds_the_key_an_owned_insert_wrote() { + // If `(String, Vec)` and `(&str, &[u8])` ever hashed differently, + // every lookup would miss silently: no error, no panic, just a cache + // that never hits and a `GetObject` per request — the exact thing this + // PR exists to remove. Worth an explicit assertion rather than trust. + use std::collections::HashMap; + + let mut map: HashMap<(String, Vec), &str> = HashMap::new(); + map.insert(("0xaccount".to_string(), vec![7u8; 32]), "owner"); + + let probe: &dyn DelegateAccountKey = &("0xaccount", &[7u8; 32][..]); + assert_eq!(map.get(probe).copied(), Some("owner")); + + let wrong_account: &dyn DelegateAccountKey = &("0xother", &[7u8; 32][..]); + assert_eq!(map.get(wrong_account), None); + + let wrong_key: &dyn DelegateAccountKey = &("0xaccount", &[9u8; 32][..]); + assert_eq!( + map.get(wrong_key), + None, + "a different delegate key on the same account must not collide" + ); + + // Removal through the borrowed probe has to reach the owned entry too, + // or a revoke would be observed and then quietly not applied. + assert_eq!(map.remove(probe), Some("owner")); + assert!(map.is_empty()); + } + + #[test] + fn an_unavailable_error_never_evicts_so_a_concurrent_success_survives() { + // The race this pins: thread A misses (no entry, or a stale one), + // thread B verifies successfully and writes a FRESH entry, then A's + // own read fails with an unavailable RPC. Evicting on that error + // would delete B's valid entry on the strength of a failure that says + // nothing about the key. `Keep` is unconditional, so no interleaving + // is needed to guarantee it — the policy itself is the guarantee. + assert_eq!( + verify_cache_miss_action(&OnchainVerifyError::RpcError("throttled".into())), + VerifyCacheMissAction::Keep, + "an unavailable read must never remove an entry it did not observe" + ); + assert_eq!( + verify_cache_miss_action(&OnchainVerifyError::ScanCapExceeded("cap".into())), + VerifyCacheMissAction::Keep + ); + } + + #[test] + fn stale_grace_is_only_reachable_through_the_unavailable_branch() { + // Guards the pairing the outage path depends on: the only error class + // that keeps an entry is the one the stale read is allowed to serve. + // If a definitive rejection ever became `Keep`, a revoked key would + // start riding the grace window. + assert_eq!( + verify_cache_miss_action(&OnchainVerifyError::KeyNotFound("k".into())), + VerifyCacheMissAction::Evict + ); + assert_eq!( + verify_cache_miss_action(&OnchainVerifyError::AccountDeactivated("a".into())), + VerifyCacheMissAction::Evict + ); + assert_eq!( + verify_cache_miss_action(&OnchainVerifyError::RpcError("throttled".into())), + VerifyCacheMissAction::Keep + ); + } + + async fn seed_reject_cache( + cache: &DelegateRejectCache, + account_id: &str, + pk: &[u8], + age: std::time::Duration, + ) { + cache.write().await.insert( + (account_id.to_string(), pk.to_vec()), + std::time::Instant::now() - age, + ); + } + + #[tokio::test] + async fn a_remembered_rejection_is_reused_without_touching_the_chain() { + // The 401 loop: ~70% of `/api/mcp/sse` traffic is a bridge retrying a + // key that will never be accepted, and each retry cost one fullnode + // read. `KeyNotFound` here (rather than the `RpcError` the + // unreachable URL would produce) proves no read was attempted. + let cache = new_delegate_verify_cache(); + let rejects = new_delegate_reject_cache(); + let account_id = "0xaccount-reject-fresh"; + let pk = sample_pk(); + seed_reject_cache(&rejects, account_id, &pk, std::time::Duration::ZERO).await; + + let client = reqwest::Client::new(); + let err = verify_delegate_key_cached( + &cache, + &rejects, + &client, + unreachable_rpc_url(), + None, + account_id, + &pk, + "0xpkg", + ) + .await + .expect_err("a remembered rejection must still be a rejection"); + + assert!( + !err.is_unavailable(), + "served from the reject cache, so it must not look like an RPC failure: {err}" + ); + } + + #[tokio::test] + async fn an_expired_rejection_goes_back_to_the_chain() { + let cache = new_delegate_verify_cache(); + let rejects = new_delegate_reject_cache(); + let account_id = "0xaccount-reject-expired"; + let pk = sample_pk(); + seed_reject_cache( + &rejects, + account_id, + &pk, + DELEGATE_REJECT_CACHE_TTL + std::time::Duration::from_secs(1), + ) + .await; + + let client = reqwest::Client::new(); + let err = verify_delegate_key_cached( + &cache, + &rejects, + &client, + unreachable_rpc_url(), + None, + account_id, + &pk, + "0xpkg", + ) + .await + .expect_err("the unreachable RPC still fails"); + + assert!( + err.is_unavailable(), + "past the TTL the chain must be consulted again, so the error is the RPC's: {err}" + ); + } + + #[tokio::test] + async fn a_registered_key_is_never_held_back_by_an_older_rejection() { + // Ordering guard: the positive lookup runs first, so a key that was + // refused before its registration landed starts working the moment a + // verification succeeds, without waiting out the rejection TTL. + let cache = new_delegate_verify_cache(); + let rejects = new_delegate_reject_cache(); + let account_id = "0xaccount-reject-then-registered"; + let pk = sample_pk(); + let owner = seed_verify_cache(&cache, account_id, &pk, std::time::Duration::ZERO).await; + seed_reject_cache(&rejects, account_id, &pk, std::time::Duration::ZERO).await; + + let client = reqwest::Client::new(); + let result = verify_delegate_key_cached( + &cache, + &rejects, + &client, + unreachable_rpc_url(), + None, + account_id, + &pk, + "0xpkg", + ) + .await; + + assert_eq!(result.ok(), Some(owner), "the positive entry must win"); + } + + #[test] + fn the_rejection_cache_cap_bounds_what_callers_can_grow() { + // Keyed by what callers send, so without a cap anyone could grow it + // one entry per made-up pair. Refreshing an entry that already exists + // cannot grow the map and stays allowed at the cap. + assert!(should_record_rejection(0, false)); + assert!(should_record_rejection( + DELEGATE_REJECT_CACHE_MAX_ENTRIES - 1, + false + )); + assert!( + !should_record_rejection(DELEGATE_REJECT_CACHE_MAX_ENTRIES, false), + "a new pair at the cap must fall back to the live read, not evict something" + ); + assert!( + should_record_rejection(DELEGATE_REJECT_CACHE_MAX_ENTRIES, true), + "refreshing an existing entry does not grow the map" + ); + } + + #[tokio::test] + async fn expired_rejections_do_not_hold_the_cap_against_a_genuine_one() { + // The cap is consulted with the map's raw length, but the TTL is 10s + // and the sweeper runs every 300s — so without expiring first, up to + // thirty generations of entries nobody can be served from still hold + // every slot. A few thousand made-up pairs then block every genuine + // rejection for five minutes, and each looping client goes back to + // one fullnode read per retry. + // + // This pins the policy, not the call site: the retain is inline in + // `verify_delegate_key_cached`'s Evict arm, which needs a chain that + // answers definitively and so cannot run offline. Keep the two in + // step by hand. + let rejects = new_delegate_reject_cache(); + { + let mut map = rejects.write().await; + let dead = std::time::Instant::now() - (DELEGATE_REJECT_CACHE_TTL + + std::time::Duration::from_secs(1)); + for i in 0..DELEGATE_REJECT_CACHE_MAX_ENTRIES { + map.insert((format!("0xspam-{i}"), sample_pk()), dead); + } + } + + assert!( + !should_record_rejection(rejects.read().await.len(), false), + "precondition: the cap is full, so a new pair would be turned away" + ); + + rejects + .write() + .await + .retain(|_, rejected_at| reject_entry_is_fresh(*rejected_at)); + + assert!( + rejects.read().await.is_empty(), + "every seeded entry is past the TTL, so none may be kept" + ); + assert!( + should_record_rejection(rejects.read().await.len(), false), + "after expiring, a genuine rejection has room again" + ); + } + + #[test] + fn a_verification_overtaken_by_an_eviction_is_not_stored() { + // Cold misses are not single-flighted, so A can be reading while B + // observes a revoke and evicts. Without this check A's older success + // lands afterwards and re-opens a full trust window on a dead key. + assert!(may_store_verification(7, 7), "nothing moved, safe to store"); + assert!( + !may_store_verification(7, 8), + "an eviction landed mid-read; the chain has contradicted this answer" + ); + } + + #[test] + fn a_rejection_is_forgotten_sooner_than_a_success_is_trusted() { + // The asymmetry that makes the negative cache safe: a stale positive + // authenticates a revoked key, a stale negative only delays one that + // just became valid. + assert!( + DELEGATE_REJECT_CACHE_TTL < DELEGATE_VERIFY_CACHE_TTL, + "a rejection must never outlive the trust window for a success" + ); + } + + #[test] + fn stale_grace_outlives_the_ttl_so_the_sweeper_has_something_to_serve() { + // `main.rs` sweeps on `is_servable_while_unavailable`. If that ever + // collapsed back to the TTL the outage path would still compile and + // still be dead, because the entry would already have been evicted. + let entry = TimedVerifiedOwner { + owner: "0xowner".into(), + verified_at: std::time::Instant::now() + - (DELEGATE_VERIFY_CACHE_TTL + std::time::Duration::from_secs(1)), + }; + assert!(!entry.is_fresh(), "past the TTL on the ordinary path"); + assert!( + entry.is_servable_while_unavailable(), + "but still held for the outage path" + ); + } + + #[tokio::test] + async fn verify_delegate_key_cached_entry_is_scoped_to_account_and_key() { + let cache = new_delegate_verify_cache(); + let account_id = "0xaccount-verify-scope"; + let pk = sample_pk(); + seed_verify_cache(&cache, account_id, &pk, std::time::Duration::ZERO).await; + + let client = reqwest::Client::new(); + + let other_key = vec![9u8; 32]; + assert!( + verify_delegate_key_cached( + &cache, + &new_delegate_reject_cache(), + &client, + unreachable_rpc_url(), + None, + account_id, + &other_key, + "0xpkg", + ) + .await + .is_err(), + "a different delegate key on the same account must not ride this entry" + ); + + assert!( + verify_delegate_key_cached( + &cache, + &new_delegate_reject_cache(), + &client, + unreachable_rpc_url(), + None, + "0xsome-other-account", + &pk, + "0xpkg", + ) + .await + .is_err(), + "the same delegate key on a different account must not ride this entry" + ); + } + + #[tokio::test] + async fn verify_delegate_key_cached_keeps_entry_when_rpc_is_unavailable() { + let cache = new_delegate_verify_cache(); + let account_id = "0xaccount-verify-unavailable"; + let pk = sample_pk(); + seed_verify_cache( + &cache, + account_id, + &pk, + DELEGATE_VERIFY_CACHE_TTL + std::time::Duration::from_secs(1), + ) + .await; + + let client = reqwest::Client::new(); + let _ = verify_delegate_key_cached( + &cache, + &new_delegate_reject_cache(), + &client, + unreachable_rpc_url(), + None, + account_id, + &pk, + "0xpkg", + ) + .await; + + assert!( + cache + .entries + .read() + .await + .contains_key(&(account_id.to_string(), pk.clone())), + "a transport failure is not a revoke, so it must not evict the pair" + ); + } + + #[test] + fn verify_cache_miss_action_evicts_only_on_a_definitive_rejection() { + for err in [ + OnchainVerifyError::KeyNotFound("revoked".into()), + OnchainVerifyError::AccountDeactivated("deactivated".into()), + OnchainVerifyError::NotFound("missing object".into()), + OnchainVerifyError::WrongObjectType("lookalike".into()), + ] { + assert_eq!( + verify_cache_miss_action(&err), + VerifyCacheMissAction::Evict, + "{err}" + ); + } + for err in [ + OnchainVerifyError::RpcError("429 Too Many Requests".into()), + OnchainVerifyError::ScanCapExceeded("cap".into()), + ] { + assert_eq!( + verify_cache_miss_action(&err), + VerifyCacheMissAction::Keep, + "{err}" + ); + } + } + + #[tokio::test] + async fn delegate_verify_cache_sweep_keeps_what_the_outage_path_can_serve() { + let cache = new_delegate_verify_cache(); + seed_verify_cache(&cache, "0xfresh", &sample_pk(), std::time::Duration::ZERO).await; + // Past the TTL, so `is_fresh` is already false and the ordinary lookup + // will not serve it — but inside the grace, which is precisely what + // the unavailable branch falls back to. + seed_verify_cache( + &cache, + "0xin-grace", + &sample_pk(), + DELEGATE_VERIFY_CACHE_TTL + std::time::Duration::from_secs(1), + ) + .await; + seed_verify_cache( + &cache, + "0xpast-grace", + &sample_pk(), + DELEGATE_VERIFY_CACHE_TTL + + DELEGATE_VERIFY_STALE_GRACE + + std::time::Duration::from_secs(1), + ) + .await; + + // The predicate `main.rs`'s sweep task uses. Sweeping on `is_fresh` + // instead would evict `0xin-grace` 30s after it was verified — the + // entry the stale-serve path exists to use — so the sweep bound has + // to be the grace, not the TTL. + cache + .entries + .write() + .await + .retain(|_, v| v.is_servable_while_unavailable()); + + let remaining = cache.entries.read().await; + assert!(remaining.contains_key(&("0xfresh".to_string(), sample_pk()))); + assert!( + remaining.contains_key(&("0xin-grace".to_string(), sample_pk())), + "sweeping on the TTL would delete exactly what the unavailable branch serves" + ); + assert!(!remaining.contains_key(&("0xpast-grace".to_string(), sample_pk()))); + } + + #[tokio::test] + async fn a_rejected_key_records_nothing_so_it_cannot_grow_the_map() { + // The MCP proxy takes `x-memwal-account-id` from an unauthenticated + // header and only checks that it is non-empty, and `/api/mcp/*` has no + // rate limit ahead of the verify. Caching rejections would therefore + // let an anonymous caller mint one entry per made-up account id. Only + // successes are stored, so the map stays bounded by real accounts. + let cache = new_delegate_verify_cache(); + let client = reqwest::Client::new(); + for i in 0..5 { + let _ = verify_delegate_key_cached( + &cache, + &new_delegate_reject_cache(), + &client, + unreachable_rpc_url(), + None, + &format!("0xmade-up-{i}"), + &sample_pk(), + "0xpkg", + ) + .await; + } + assert!( + cache.entries.read().await.is_empty(), + "a failed verification must leave no entry behind" + ); + } + // ── DelegateKeysCache periodic sweep (nothing else ever removed a map // slot — only the TTL above gated trust-on-hit) ─────────────────── // diff --git a/services/server/src/types.rs b/services/server/src/types.rs index 93fc43802..61c113537 100644 --- a/services/server/src/types.rs +++ b/services/server/src/types.rs @@ -211,6 +211,16 @@ pub struct AppState { /// id. Backs `GET /v1/owners/{owner}/agents` so repeated calls within /// the TTL window don't re-hit the chain. pub delegate_keys_cache: crate::storage::sui::DelegateKeysCache, + /// Short-TTL (`storage::sui::DELEGATE_VERIFY_CACHE_TTL`) in-memory + /// cache of successful delegate-key verifications, keyed by + /// `(account object id, public key)`. Shared by the signed-request + /// auth middleware and the MCP proxy so a burst of requests carrying + /// the same credentials costs one `GetObject` instead of one each + /// (WALM-618). + pub delegate_verify_cache: crate::storage::sui::DelegateVerifyCache, + pub delegate_reject_cache: crate::storage::sui::DelegateRejectCache, + /// In-flight MCP connect episodes, for `time_to_session`. + pub mcp_connect_episodes: crate::observability::McpConnectEpisodes, /// Alert dispatchers for operational notifications. Individual alert /// paths decide when failures are terminal enough to notify. pub alerts: Arc, @@ -233,8 +243,8 @@ pub struct AppState { /// when the request body sets `scoring_weights`; default weights /// preserve the pgvector cosine order exactly. pub ranker: Arc, - /// Redis multiplexed connection for rate limiting - pub redis: redis::aio::MultiplexedConnection, + /// Redis connection manager for rate limiting (reconnects after a drop) + pub redis: redis::aio::ConnectionManager, /// In-memory token bucket fallback for when Redis is unavailable pub fallback_rate_limit: tokio::sync::Mutex, /// Bounds concurrent AccountRegistry fallback scans (auth Strategy 3). @@ -1826,6 +1836,11 @@ pub struct RestoreRequest { pub struct RestoreResponse { pub restored: usize, pub skipped: usize, + /// Permanent decrypt/UTF-8 failures on this on-chain page: negative-cache + /// hits plus any new permanent failures this call. Transient download, + /// decrypt, or embed errors are not counted here. Additive JSON field + /// (COMG-719 / WALM-480). + pub failed: usize, pub total: usize, pub namespace: String, pub owner: String, @@ -1837,7 +1852,8 @@ pub struct RestoreResponse { /// namespaces can starve this one. Once the sidecar cap is saturated /// (`limit >= 20`), truncation follows this call's missing-blob page, /// not on-chain `total`, so a fully restored namespace does not loop - /// (WALM-431 / GH #762). + /// (WALM-431 / GH #762). Also true when an inspected page produced only + /// transients (download/decrypt/embed) so the caller retries (WALM-480). pub truncated: bool, } @@ -1901,9 +1917,13 @@ pub struct HealthResponse { /// at from git history. Both fields are always populated — there is /// no "version unknown" state for a running server. pub prompt_versions: PromptVersions, - /// Whether the encryption sidecar process answered its own `/health`. - /// This is sidecar liveness, not a guarantee that remember/analyze will - /// succeed. `status` stays `"ok"` while the relayer process is up. + /// Whether the encryption sidecar answered `/health` AND Postgres can + /// accept writes (Neon `neon.max_cluster_size` cap). Prefer + /// `public.pg_cluster_size()`; if that function is missing, fall back + /// to `sum(pg_database_size)` against the same GUC. Self-hosted + /// Postgres without the GUC is sidecar-only. Probe errors and timeouts + /// fail open so CI `wait-for-relayer` does not hang. `status` stays + /// `"ok"` while the relayer process is up. pub write_ready: bool, /// Write-path admission: `"ok"` or `"paused"`. `"paused"` when /// `WRITES_PAUSED` is set; write routes then return HTTP 503.