From 6433369259a489ce4272918d6aac8620b1607ed0 Mon Sep 17 00:00:00 2001 From: Ashwin-3cS Date: Fri, 27 Mar 2026 01:49:28 +0530 Subject: [PATCH 01/29] feat(noter): add server-issued challenge nonce for wallet auth replay protection MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Replace client-generated auth messages with a server-issued one-time challenge nonce. The server generates a random nonce stored in a new walletChallenges table with a 5-minute TTL. The client signs the nonce, and the server deletes it on use — replaying a captured signature fails because the challenge no longer exists. --- apps/noter/package/feature/auth/api/input.ts | 2 +- apps/noter/package/feature/auth/api/route.ts | 67 ++++++++++++++++--- apps/noter/package/feature/auth/constant.ts | 2 +- .../package/feature/auth/hook/use-auth.ts | 3 +- .../package/feature/auth/lib/wallet-client.ts | 14 ---- .../feature/auth/ui/auth-button-group.tsx | 13 ++-- .../package/feature/auth/ui/wallet-button.tsx | 15 +++-- apps/noter/package/shared/db/schema.ts | 21 ++++++ 8 files changed, 98 insertions(+), 39 deletions(-) diff --git a/apps/noter/package/feature/auth/api/input.ts b/apps/noter/package/feature/auth/api/input.ts index 76d0620d4..031eca558 100644 --- a/apps/noter/package/feature/auth/api/input.ts +++ b/apps/noter/package/feature/auth/api/input.ts @@ -79,7 +79,7 @@ export const connectWalletInput = z.object({ walletType: walletSessionInsertSchema.shape.walletType.pipe(z.enum(["slush"])), // Subset validation address: walletSessionInsertSchema.shape.walletAddress, // Maps to walletAddress in DB signature: walletSessionInsertSchema.shape.signature, - message: walletSessionInsertSchema.shape.signedMessage, // Maps to signedMessage in DB + challengeId: uuidv7Schema, // Server-issued challenge ID (replaces client message) }); export type ConnectWalletInput = z.infer; diff --git a/apps/noter/package/feature/auth/api/route.ts b/apps/noter/package/feature/auth/api/route.ts index 12b2e9482..40615fc7e 100644 --- a/apps/noter/package/feature/auth/api/route.ts +++ b/apps/noter/package/feature/auth/api/route.ts @@ -6,6 +6,7 @@ import { router, procedure } from "@/shared/lib/trpc/init"; import { TRPCError } from "@trpc/server"; import { verifyPersonalMessageSignature } from "@mysten/sui/verify"; +import { randomBytes } from "crypto"; import { uuidv7 } from "uuidv7"; import { initiateLoginInput, @@ -32,7 +33,7 @@ import { } from "../lib/zklogin-client"; import { OAUTH_PROVIDERS, OAUTH_SCOPES, AUTH_ERRORS } from "../constant"; import { buildOAuthUrl } from "../domain/zklogin"; -import { zkLoginSessions, walletSessions } from "@/shared/db/schema"; +import { zkLoginSessions, walletSessions, walletChallenges } from "@/shared/db/schema"; import { eq } from "drizzle-orm"; import * as authService from "../domain/service"; @@ -238,35 +239,82 @@ export const authRouter = router({ return { success: true }; }), + /** + * Get a one-time challenge nonce for wallet authentication + * The nonce must be signed by the wallet and returned via connectWallet + */ + getChallenge: procedure + .mutation(async ({ ctx }) => { + const challengeId = uuidv7(); + const nonce = randomBytes(32).toString("hex"); + const expiresAt = new Date(Date.now() + 5 * 60 * 1000); // 5 minutes + + await ctx.db.insert(walletChallenges).values({ + id: challengeId, + nonce, + expiresAt, + }); + + return { challengeId, nonce, expiresAt }; + }), + /** * Connect wallet - authenticate with Sui wallet (Slush, Sui Wallet) - * Verifies signature and creates session + * Verifies signature against a server-issued challenge nonce */ connectWallet: procedure .input(connectWalletInput) .mutation(async ({ ctx, input }) => { - const { walletType, address, signature, message } = input; + const { challengeId, walletType, address, signature } = input; try { - // Verify the wallet signature before creating a session + // 1. Fetch and validate challenge + const [challenge] = await ctx.db + .select() + .from(walletChallenges) + .where(eq(walletChallenges.id, challengeId)) + .limit(1); + + if (!challenge) { + throw new TRPCError({ + code: "NOT_FOUND", + message: "Challenge not found or already used", + }); + } + + if (challenge.expiresAt < new Date()) { + await ctx.db.delete(walletChallenges).where(eq(walletChallenges.id, challengeId)); + throw new TRPCError({ + code: "UNAUTHORIZED", + message: "Challenge expired", + }); + } + + // 2. Consume challenge (delete before verify to prevent timing-based replay) + await ctx.db.delete(walletChallenges).where(eq(walletChallenges.id, challengeId)); + + // 3. Verify signature against server-issued nonce const signerAddress = await verifyPersonalMessageSignature( - new TextEncoder().encode(message), + new TextEncoder().encode(challenge.nonce), signature, ).catch(() => { throw new TRPCError({ code: "UNAUTHORIZED", message: "Invalid signature" }); }); if (signerAddress.toSuiAddress() !== address) { - throw new TRPCError({ code: "UNAUTHORIZED", message: "Signature does not match address" }); + throw new TRPCError({ + code: "UNAUTHORIZED", + message: "Signature does not match address", + }); } - // Create or update user via service + // 4. Create or update user const user = await authService.upsertWalletUser(ctx.db, { address, walletType, }); - // Create wallet session + // 5. Create wallet session const sessionId = uuidv7(); const expiresAt = new Date(); expiresAt.setHours(expiresAt.getHours() + 24); // 24 hour session @@ -276,13 +324,12 @@ export const authRouter = router({ userId: user.id, walletAddress: address, walletType, - signedMessage: message, + signedMessage: challenge.nonce, signature, signedAt: new Date(), expiresAt, }); - // Return wallet session data (no ephemeral keys for wallet auth) return { user, sessionId, diff --git a/apps/noter/package/feature/auth/constant.ts b/apps/noter/package/feature/auth/constant.ts index ebd0d0597..2def6e91d 100644 --- a/apps/noter/package/feature/auth/constant.ts +++ b/apps/noter/package/feature/auth/constant.ts @@ -69,7 +69,7 @@ export const OAUTH_SCOPES = { export const STORAGE_KEYS = { ephemeralPrivateKey: "zklogin:ephemeral:private", ephemeralPublicKey: "zklogin:ephemeral:public", - sessionId: "zklogin:session:id", + sessionId: "auth:session:id", nonce: "zklogin:nonce", maxEpoch: "zklogin:maxEpoch", randomness: "zklogin:randomness", diff --git a/apps/noter/package/feature/auth/hook/use-auth.ts b/apps/noter/package/feature/auth/hook/use-auth.ts index 66611d573..fc7defd9f 100644 --- a/apps/noter/package/feature/auth/hook/use-auth.ts +++ b/apps/noter/package/feature/auth/hook/use-auth.ts @@ -27,6 +27,7 @@ export function useAuth() { // tRPC mutations const initiateLoginMutation = trpc.auth.initiateLogin.useMutation(); const completeLoginMutation = trpc.auth.completeLogin.useMutation(); + const getChallengeMutation = trpc.auth.getChallenge.useMutation(); const connectWalletMutation = trpc.auth.connectWallet.useMutation(); const logoutMutation = trpc.auth.logout.useMutation(); @@ -141,7 +142,7 @@ export function useAuth() { walletType: "slush"; address: string; signature: string; - message: string; + challengeId: string; }) => { try { setLoading(true); diff --git a/apps/noter/package/feature/auth/lib/wallet-client.ts b/apps/noter/package/feature/auth/lib/wallet-client.ts index 87e3e0862..67325407e 100644 --- a/apps/noter/package/feature/auth/lib/wallet-client.ts +++ b/apps/noter/package/feature/auth/lib/wallet-client.ts @@ -180,17 +180,3 @@ export async function disconnectWallet(type: WalletType): Promise { } } -/** - * Generate authentication message - */ -export function generateAuthMessage(): string { - const timestamp = Date.now(); - const nonce = Math.random().toString(36).substring(7); - - return `Sign this message to authenticate with Noter - -Timestamp: ${timestamp} -Nonce: ${nonce} - -This will not trigger any blockchain transaction or cost any gas fees.`; -} diff --git a/apps/noter/package/feature/auth/ui/auth-button-group.tsx b/apps/noter/package/feature/auth/ui/auth-button-group.tsx index 113839b7c..5257980e4 100644 --- a/apps/noter/package/feature/auth/ui/auth-button-group.tsx +++ b/apps/noter/package/feature/auth/ui/auth-button-group.tsx @@ -20,10 +20,10 @@ import { WALLET_INSTALL_URLS, type WalletType } from "../constant"; import { useAuth } from "../hook/use-auth"; import { connectWallet, - generateAuthMessage, isWalletInstalled, signMessage, } from "../lib/wallet-client"; +import { trpc } from "@/shared/lib/trpc/client"; import { LoginButton } from "./login-button"; export function AuthButtonGroup() { @@ -31,6 +31,7 @@ export function AuthButtonGroup() { const [isWalletConnecting, setIsWalletConnecting] = useState(false); const [walletError, setWalletError] = useState(null); const { connectWalletAuth, isLoginPending } = useAuth(); + const getChallenge = trpc.auth.getChallenge.useMutation(); const slushInstalled = isWalletInstalled("slush"); @@ -50,18 +51,18 @@ export function AuthButtonGroup() { // 1. Connect to wallet const account = await connectWallet(walletType); - // 2. Generate message to sign - const message = generateAuthMessage(); + // 2. Get server-issued challenge nonce + const { challengeId, nonce } = await getChallenge.mutateAsync(); - // 3. Sign message - const { signature } = await signMessage(walletType, message, account); + // 3. Sign the challenge nonce + const { signature } = await signMessage(walletType, nonce, account); // 4. Authenticate with backend await connectWalletAuth({ walletType, address: account.address, signature, - message, + challengeId, }); } catch (err) { console.error(`[AuthButtonGroup] Failed to connect ${walletType}:`, err); diff --git a/apps/noter/package/feature/auth/ui/wallet-button.tsx b/apps/noter/package/feature/auth/ui/wallet-button.tsx index af4322fd3..b115bc60e 100644 --- a/apps/noter/package/feature/auth/ui/wallet-button.tsx +++ b/apps/noter/package/feature/auth/ui/wallet-button.tsx @@ -12,9 +12,9 @@ import { useState } from "react"; import { connectWallet, signMessage, - generateAuthMessage, isWalletInstalled, } from "../lib/wallet-client"; +import { trpc } from "@/shared/lib/trpc/client"; import { WALLET_NAMES, WALLET_INSTALL_URLS, type WalletType } from "../constant"; export type WalletButtonProps = { @@ -31,6 +31,7 @@ export function WalletButton({ size = "default", }: WalletButtonProps) { const { connectWalletAuth, isLoginPending } = useAuth(); + const getChallenge = trpc.auth.getChallenge.useMutation(); const [isConnecting, setIsConnecting] = useState(false); const [error, setError] = useState(null); @@ -44,17 +45,19 @@ export function WalletButton({ try { // 1. Connect to wallet const account = await connectWallet(wallet); - // 2. Generate message to sign - const message = generateAuthMessage(); - // 3. Sign message - const { signature } = await signMessage(wallet, message, account); + // 2. Get server-issued challenge nonce + const { challengeId, nonce } = await getChallenge.mutateAsync(); + + // 3. Sign the challenge nonce + const { signature } = await signMessage(wallet, nonce, account); + // 4. Authenticate with backend await connectWalletAuth({ walletType: wallet, address: account.address, signature, - message, + challengeId, }); } catch (err) { console.error(`[WalletButton] Failed to connect ${wallet}:`, err); setError(err instanceof Error ? err.message : "Connection failed"); diff --git a/apps/noter/package/shared/db/schema.ts b/apps/noter/package/shared/db/schema.ts index 07b5f8761..6655798f8 100644 --- a/apps/noter/package/shared/db/schema.ts +++ b/apps/noter/package/shared/db/schema.ts @@ -214,6 +214,27 @@ export const walletSessions = pgTable( (t) => [index().on(t.userId), index().on(t.expiresAt), index().on(t.walletAddress)] ); +// ════════════════════════════════════════════════════════════════ +// WALLET CHALLENGES (one-time nonces for replay protection) +// ════════════════════════════════════════════════════════════════ + +export const walletChallenges = pgTable( + "wallet_challenges", + { + id: uuid() + .primaryKey() + .$defaultFn(() => uuidv7()), + createdAt: timestamp().defaultNow().notNull(), + + // Server-generated random nonce + nonce: text().notNull(), + + // Challenge expiration (5 minutes) + expiresAt: timestamp().notNull(), + }, + (t) => [index().on(t.expiresAt)] +); + // ════════════════════════════════════════════════════════════════ // NOTES (Apple Notes / Notion-like) // ════════════════════════════════════════════════════════════════ From 23b2a058b440e0c4c99ed29f451f2358c22298cd Mon Sep 17 00:00:00 2001 From: Ashwin-3cS Date: Fri, 27 Mar 2026 13:17:32 +0530 Subject: [PATCH 02/29] fix(noter): address review feedback on wallet challenge flow - Use DELETE...RETURNING in connectWallet to eliminate TOCTOU race - Clean up expired challenges in getChallenge to prevent table bloat - Add Drizzle migration for walletChallenges table --- apps/noter/package.json | 10 +- apps/noter/package/feature/auth/api/route.ts | 20 +- .../migrations/0002_conscious_deathbird.sql | 8 + .../db/migrations/meta/0002_snapshot.json | 1103 +++++++++++++++++ .../shared/db/migrations/meta/_journal.json | 7 + pnpm-lock.yaml | 73 +- 6 files changed, 1181 insertions(+), 40 deletions(-) create mode 100644 apps/noter/package/shared/db/migrations/0002_conscious_deathbird.sql create mode 100644 apps/noter/package/shared/db/migrations/meta/0002_snapshot.json diff --git a/apps/noter/package.json b/apps/noter/package.json index 44e92fc07..cf29afd0e 100644 --- a/apps/noter/package.json +++ b/apps/noter/package.json @@ -18,7 +18,6 @@ "@ai-sdk/openai": "^3.0.41", "@ai-sdk/react": "3.0.39", "@base-ui/react": "^1.2.0", - "@mysten-incubation/memwal": "workspace:*", "@hookform/resolvers": "^5.2.2", "@lexical/code": "^0.41.0", "@lexical/link": "^0.41.0", @@ -30,6 +29,7 @@ "@lexical/table": "^0.41.0", "@lexical/text": "^0.41.0", "@lexical/utils": "^0.41.0", + "@mysten-incubation/memwal": "workspace:*", "@mysten/sui": "^2.5.0", "@mysten/wallet-standard": "^0.15.0", "@mysten/zklogin": "^0.8.1", @@ -45,7 +45,9 @@ "drizzle-orm": "^0.45.1", "drizzle-zod": "^0.8.3", "embla-carousel-react": "^8.6.0", + "esbuild": "~0.27.4", "framer-motion": "^12.34.3", + "get-tsconfig": "^4.7.5", "input-otp": "^1.4.2", "jotai": "^2.18.0", "jwt-decode": "^4.0.0", @@ -68,9 +70,7 @@ "use-debounce": "^10.1.0", "uuidv7": "^1.1.0", "vaul": "^1.1.2", - "zod": "^4.3.6", - "esbuild": "~0.27.4", - "get-tsconfig": "^4.7.5" + "zod": "^4.3.6" }, "devDependencies": { "@tailwindcss/postcss": "^4", @@ -79,7 +79,7 @@ "@types/react": "^19", "@types/react-dom": "^19", "dotenv": "^16.4.5", - "drizzle-kit": "^0.31.9", + "drizzle-kit": "^0.31.10", "eslint": "^9", "eslint-config-next": "16.1.6", "shadcn": "^3.8.5", diff --git a/apps/noter/package/feature/auth/api/route.ts b/apps/noter/package/feature/auth/api/route.ts index 40615fc7e..7114ba88a 100644 --- a/apps/noter/package/feature/auth/api/route.ts +++ b/apps/noter/package/feature/auth/api/route.ts @@ -34,7 +34,7 @@ import { import { OAUTH_PROVIDERS, OAUTH_SCOPES, AUTH_ERRORS } from "../constant"; import { buildOAuthUrl } from "../domain/zklogin"; import { zkLoginSessions, walletSessions, walletChallenges } from "@/shared/db/schema"; -import { eq } from "drizzle-orm"; +import { eq, lt } from "drizzle-orm"; import * as authService from "../domain/service"; export const authRouter = router({ @@ -245,6 +245,11 @@ export const authRouter = router({ */ getChallenge: procedure .mutation(async ({ ctx }) => { + // Clean up expired challenges to prevent table bloat + await ctx.db + .delete(walletChallenges) + .where(lt(walletChallenges.expiresAt, new Date())); + const challengeId = uuidv7(); const nonce = randomBytes(32).toString("hex"); const expiresAt = new Date(Date.now() + 5 * 60 * 1000); // 5 minutes @@ -268,12 +273,11 @@ export const authRouter = router({ const { challengeId, walletType, address, signature } = input; try { - // 1. Fetch and validate challenge + // 1. Atomically consume challenge const [challenge] = await ctx.db - .select() - .from(walletChallenges) + .delete(walletChallenges) .where(eq(walletChallenges.id, challengeId)) - .limit(1); + .returning(); if (!challenge) { throw new TRPCError({ @@ -283,17 +287,13 @@ export const authRouter = router({ } if (challenge.expiresAt < new Date()) { - await ctx.db.delete(walletChallenges).where(eq(walletChallenges.id, challengeId)); throw new TRPCError({ code: "UNAUTHORIZED", message: "Challenge expired", }); } - // 2. Consume challenge (delete before verify to prevent timing-based replay) - await ctx.db.delete(walletChallenges).where(eq(walletChallenges.id, challengeId)); - - // 3. Verify signature against server-issued nonce + // 2. Verify signature against server-issued nonce const signerAddress = await verifyPersonalMessageSignature( new TextEncoder().encode(challenge.nonce), signature, diff --git a/apps/noter/package/shared/db/migrations/0002_conscious_deathbird.sql b/apps/noter/package/shared/db/migrations/0002_conscious_deathbird.sql new file mode 100644 index 000000000..bb82e51fa --- /dev/null +++ b/apps/noter/package/shared/db/migrations/0002_conscious_deathbird.sql @@ -0,0 +1,8 @@ +CREATE TABLE "wallet_challenges" ( + "id" uuid PRIMARY KEY NOT NULL, + "createdAt" timestamp DEFAULT now() NOT NULL, + "nonce" text NOT NULL, + "expiresAt" timestamp NOT NULL +); +--> statement-breakpoint +CREATE INDEX "wallet_challenges_expiresAt_index" ON "wallet_challenges" USING btree ("expiresAt"); \ No newline at end of file diff --git a/apps/noter/package/shared/db/migrations/meta/0002_snapshot.json b/apps/noter/package/shared/db/migrations/meta/0002_snapshot.json new file mode 100644 index 000000000..2c3c80661 --- /dev/null +++ b/apps/noter/package/shared/db/migrations/meta/0002_snapshot.json @@ -0,0 +1,1103 @@ +{ + "id": "c37d77ab-d057-4524-a8df-e5767fc16d94", + "prevId": "b297cb0c-43b8-4e9b-a620-348289847be8", + "version": "7", + "dialect": "postgresql", + "tables": { + "public.chats": { + "name": "chats", + "schema": "", + "columns": { + "id": { + "name": "id", + "type": "uuid", + "primaryKey": true, + "notNull": true + }, + "createdAt": { + "name": "createdAt", + "type": "timestamp", + "primaryKey": false, + "notNull": true, + "default": "now()" + }, + "userId": { + "name": "userId", + "type": "uuid", + "primaryKey": false, + "notNull": true + }, + "title": { + "name": "title", + "type": "text", + "primaryKey": false, + "notNull": false + }, + "model": { + "name": "model", + "type": "text", + "primaryKey": false, + "notNull": false, + "default": "'anthropic/claude-sonnet-4'" + }, + "systemPrompt": { + "name": "systemPrompt", + "type": "text", + "primaryKey": false, + "notNull": false + }, + "temperature": { + "name": "temperature", + "type": "real", + "primaryKey": false, + "notNull": false + } + }, + "indexes": { + "chats_userId_index": { + "name": "chats_userId_index", + "columns": [ + { + "expression": "userId", + "isExpression": false, + "asc": true, + "nulls": "last" + } + ], + "isUnique": false, + "concurrently": false, + "method": "btree", + "with": {} + }, + "chats_userId_createdAt_index": { + "name": "chats_userId_createdAt_index", + "columns": [ + { + "expression": "userId", + "isExpression": false, + "asc": true, + "nulls": "last" + }, + { + "expression": "createdAt", + "isExpression": false, + "asc": true, + "nulls": "last" + } + ], + "isUnique": false, + "concurrently": false, + "method": "btree", + "with": {} + } + }, + "foreignKeys": { + "chats_userId_users_id_fk": { + "name": "chats_userId_users_id_fk", + "tableFrom": "chats", + "tableTo": "users", + "columnsFrom": [ + "userId" + ], + "columnsTo": [ + "id" + ], + "onDelete": "cascade", + "onUpdate": "no action" + } + }, + "compositePrimaryKeys": {}, + "uniqueConstraints": {}, + "policies": {}, + "checkConstraints": {}, + "isRLSEnabled": false + }, + "public.messages": { + "name": "messages", + "schema": "", + "columns": { + "id": { + "name": "id", + "type": "uuid", + "primaryKey": true, + "notNull": true + }, + "createdAt": { + "name": "createdAt", + "type": "timestamp", + "primaryKey": false, + "notNull": true, + "default": "now()" + }, + "chatId": { + "name": "chatId", + "type": "uuid", + "primaryKey": false, + "notNull": true + }, + "role": { + "name": "role", + "type": "message_role", + "typeSchema": "public", + "primaryKey": false, + "notNull": true + }, + "content": { + "name": "content", + "type": "text", + "primaryKey": false, + "notNull": false + }, + "parts": { + "name": "parts", + "type": "jsonb", + "primaryKey": false, + "notNull": false + }, + "model": { + "name": "model", + "type": "text", + "primaryKey": false, + "notNull": false + }, + "status": { + "name": "status", + "type": "ai_message_status", + "typeSchema": "public", + "primaryKey": false, + "notNull": false + }, + "promptTokens": { + "name": "promptTokens", + "type": "integer", + "primaryKey": false, + "notNull": false + }, + "completionTokens": { + "name": "completionTokens", + "type": "integer", + "primaryKey": false, + "notNull": false + }, + "agentRunId": { + "name": "agentRunId", + "type": "uuid", + "primaryKey": false, + "notNull": false + }, + "turnIndex": { + "name": "turnIndex", + "type": "integer", + "primaryKey": false, + "notNull": false + } + }, + "indexes": { + "messages_chatId_index": { + "name": "messages_chatId_index", + "columns": [ + { + "expression": "chatId", + "isExpression": false, + "asc": true, + "nulls": "last" + } + ], + "isUnique": false, + "concurrently": false, + "method": "btree", + "with": {} + }, + "messages_chatId_createdAt_index": { + "name": "messages_chatId_createdAt_index", + "columns": [ + { + "expression": "chatId", + "isExpression": false, + "asc": true, + "nulls": "last" + }, + { + "expression": "createdAt", + "isExpression": false, + "asc": true, + "nulls": "last" + } + ], + "isUnique": false, + "concurrently": false, + "method": "btree", + "with": {} + }, + "messages_agentRunId_index": { + "name": "messages_agentRunId_index", + "columns": [ + { + "expression": "agentRunId", + "isExpression": false, + "asc": true, + "nulls": "last" + } + ], + "isUnique": false, + "concurrently": false, + "method": "btree", + "with": {} + } + }, + "foreignKeys": { + "messages_chatId_chats_id_fk": { + "name": "messages_chatId_chats_id_fk", + "tableFrom": "messages", + "tableTo": "chats", + "columnsFrom": [ + "chatId" + ], + "columnsTo": [ + "id" + ], + "onDelete": "cascade", + "onUpdate": "no action" + } + }, + "compositePrimaryKeys": {}, + "uniqueConstraints": {}, + "policies": {}, + "checkConstraints": {}, + "isRLSEnabled": false + }, + "public.note_memory_highlights": { + "name": "note_memory_highlights", + "schema": "", + "columns": { + "id": { + "name": "id", + "type": "uuid", + "primaryKey": true, + "notNull": true + }, + "createdAt": { + "name": "createdAt", + "type": "timestamp", + "primaryKey": false, + "notNull": true, + "default": "now()" + }, + "noteId": { + "name": "noteId", + "type": "uuid", + "primaryKey": false, + "notNull": true + }, + "userId": { + "name": "userId", + "type": "uuid", + "primaryKey": false, + "notNull": true + }, + "highlightedText": { + "name": "highlightedText", + "type": "text", + "primaryKey": false, + "notNull": true + }, + "highlightedHtml": { + "name": "highlightedHtml", + "type": "text", + "primaryKey": false, + "notNull": false + }, + "startOffset": { + "name": "startOffset", + "type": "integer", + "primaryKey": false, + "notNull": false + }, + "endOffset": { + "name": "endOffset", + "type": "integer", + "primaryKey": false, + "notNull": false + }, + "extractedText": { + "name": "extractedText", + "type": "text", + "primaryKey": false, + "notNull": false + }, + "memoryTitle": { + "name": "memoryTitle", + "type": "text", + "primaryKey": false, + "notNull": false + }, + "memoryContent": { + "name": "memoryContent", + "type": "text", + "primaryKey": false, + "notNull": false + }, + "entities": { + "name": "entities", + "type": "jsonb", + "primaryKey": false, + "notNull": false + }, + "relationships": { + "name": "relationships", + "type": "jsonb", + "primaryKey": false, + "notNull": false + }, + "status": { + "name": "status", + "type": "memory_status", + "typeSchema": "public", + "primaryKey": false, + "notNull": true, + "default": "'preparing'" + }, + "blobId": { + "name": "blobId", + "type": "text", + "primaryKey": false, + "notNull": false + }, + "graphObjectId": { + "name": "graphObjectId", + "type": "text", + "primaryKey": false, + "notNull": false + }, + "transactionId": { + "name": "transactionId", + "type": "text", + "primaryKey": false, + "notNull": false + }, + "approvedAt": { + "name": "approvedAt", + "type": "timestamp", + "primaryKey": false, + "notNull": false + }, + "savedAt": { + "name": "savedAt", + "type": "timestamp", + "primaryKey": false, + "notNull": false + }, + "errorMessage": { + "name": "errorMessage", + "type": "text", + "primaryKey": false, + "notNull": false + } + }, + "indexes": { + "note_memory_highlights_noteId_index": { + "name": "note_memory_highlights_noteId_index", + "columns": [ + { + "expression": "noteId", + "isExpression": false, + "asc": true, + "nulls": "last" + } + ], + "isUnique": false, + "concurrently": false, + "method": "btree", + "with": {} + }, + "note_memory_highlights_userId_index": { + "name": "note_memory_highlights_userId_index", + "columns": [ + { + "expression": "userId", + "isExpression": false, + "asc": true, + "nulls": "last" + } + ], + "isUnique": false, + "concurrently": false, + "method": "btree", + "with": {} + }, + "note_memory_highlights_status_index": { + "name": "note_memory_highlights_status_index", + "columns": [ + { + "expression": "status", + "isExpression": false, + "asc": true, + "nulls": "last" + } + ], + "isUnique": false, + "concurrently": false, + "method": "btree", + "with": {} + }, + "note_memory_highlights_userId_createdAt_index": { + "name": "note_memory_highlights_userId_createdAt_index", + "columns": [ + { + "expression": "userId", + "isExpression": false, + "asc": true, + "nulls": "last" + }, + { + "expression": "createdAt", + "isExpression": false, + "asc": true, + "nulls": "last" + } + ], + "isUnique": false, + "concurrently": false, + "method": "btree", + "with": {} + } + }, + "foreignKeys": { + "note_memory_highlights_noteId_notes_id_fk": { + "name": "note_memory_highlights_noteId_notes_id_fk", + "tableFrom": "note_memory_highlights", + "tableTo": "notes", + "columnsFrom": [ + "noteId" + ], + "columnsTo": [ + "id" + ], + "onDelete": "cascade", + "onUpdate": "no action" + }, + "note_memory_highlights_userId_users_id_fk": { + "name": "note_memory_highlights_userId_users_id_fk", + "tableFrom": "note_memory_highlights", + "tableTo": "users", + "columnsFrom": [ + "userId" + ], + "columnsTo": [ + "id" + ], + "onDelete": "cascade", + "onUpdate": "no action" + } + }, + "compositePrimaryKeys": {}, + "uniqueConstraints": {}, + "policies": {}, + "checkConstraints": {}, + "isRLSEnabled": false + }, + "public.notes": { + "name": "notes", + "schema": "", + "columns": { + "id": { + "name": "id", + "type": "uuid", + "primaryKey": true, + "notNull": true + }, + "createdAt": { + "name": "createdAt", + "type": "timestamp", + "primaryKey": false, + "notNull": true, + "default": "now()" + }, + "userId": { + "name": "userId", + "type": "uuid", + "primaryKey": false, + "notNull": true + }, + "title": { + "name": "title", + "type": "text", + "primaryKey": false, + "notNull": false + }, + "content": { + "name": "content", + "type": "jsonb", + "primaryKey": false, + "notNull": true + }, + "plainText": { + "name": "plainText", + "type": "text", + "primaryKey": false, + "notNull": true + }, + "updatedAt": { + "name": "updatedAt", + "type": "timestamp", + "primaryKey": false, + "notNull": true, + "default": "now()" + } + }, + "indexes": { + "notes_userId_index": { + "name": "notes_userId_index", + "columns": [ + { + "expression": "userId", + "isExpression": false, + "asc": true, + "nulls": "last" + } + ], + "isUnique": false, + "concurrently": false, + "method": "btree", + "with": {} + }, + "notes_userId_updatedAt_index": { + "name": "notes_userId_updatedAt_index", + "columns": [ + { + "expression": "userId", + "isExpression": false, + "asc": true, + "nulls": "last" + }, + { + "expression": "updatedAt", + "isExpression": false, + "asc": true, + "nulls": "last" + } + ], + "isUnique": false, + "concurrently": false, + "method": "btree", + "with": {} + } + }, + "foreignKeys": { + "notes_userId_users_id_fk": { + "name": "notes_userId_users_id_fk", + "tableFrom": "notes", + "tableTo": "users", + "columnsFrom": [ + "userId" + ], + "columnsTo": [ + "id" + ], + "onDelete": "cascade", + "onUpdate": "no action" + } + }, + "compositePrimaryKeys": {}, + "uniqueConstraints": {}, + "policies": {}, + "checkConstraints": {}, + "isRLSEnabled": false + }, + "public.users": { + "name": "users", + "schema": "", + "columns": { + "id": { + "name": "id", + "type": "uuid", + "primaryKey": true, + "notNull": true + }, + "createdAt": { + "name": "createdAt", + "type": "timestamp", + "primaryKey": false, + "notNull": true, + "default": "now()" + }, + "suiAddress": { + "name": "suiAddress", + "type": "text", + "primaryKey": false, + "notNull": true + }, + "authMethod": { + "name": "authMethod", + "type": "auth_method", + "typeSchema": "public", + "primaryKey": false, + "notNull": true + }, + "provider": { + "name": "provider", + "type": "text", + "primaryKey": false, + "notNull": false + }, + "providerSub": { + "name": "providerSub", + "type": "text", + "primaryKey": false, + "notNull": false + }, + "walletType": { + "name": "walletType", + "type": "text", + "primaryKey": false, + "notNull": false + }, + "name": { + "name": "name", + "type": "text", + "primaryKey": false, + "notNull": false + }, + "email": { + "name": "email", + "type": "text", + "primaryKey": false, + "notNull": false + }, + "avatar": { + "name": "avatar", + "type": "text", + "primaryKey": false, + "notNull": false + }, + "lastSeenAt": { + "name": "lastSeenAt", + "type": "timestamp", + "primaryKey": false, + "notNull": false + } + }, + "indexes": { + "users_provider_providerSub_index": { + "name": "users_provider_providerSub_index", + "columns": [ + { + "expression": "provider", + "isExpression": false, + "asc": true, + "nulls": "last" + }, + { + "expression": "providerSub", + "isExpression": false, + "asc": true, + "nulls": "last" + } + ], + "isUnique": false, + "concurrently": false, + "method": "btree", + "with": {} + }, + "users_authMethod_index": { + "name": "users_authMethod_index", + "columns": [ + { + "expression": "authMethod", + "isExpression": false, + "asc": true, + "nulls": "last" + } + ], + "isUnique": false, + "concurrently": false, + "method": "btree", + "with": {} + }, + "users_suiAddress_index": { + "name": "users_suiAddress_index", + "columns": [ + { + "expression": "suiAddress", + "isExpression": false, + "asc": true, + "nulls": "last" + } + ], + "isUnique": false, + "concurrently": false, + "method": "btree", + "with": {} + } + }, + "foreignKeys": {}, + "compositePrimaryKeys": {}, + "uniqueConstraints": { + "users_suiAddress_unique": { + "name": "users_suiAddress_unique", + "nullsNotDistinct": false, + "columns": [ + "suiAddress" + ] + } + }, + "policies": {}, + "checkConstraints": {}, + "isRLSEnabled": false + }, + "public.wallet_challenges": { + "name": "wallet_challenges", + "schema": "", + "columns": { + "id": { + "name": "id", + "type": "uuid", + "primaryKey": true, + "notNull": true + }, + "createdAt": { + "name": "createdAt", + "type": "timestamp", + "primaryKey": false, + "notNull": true, + "default": "now()" + }, + "nonce": { + "name": "nonce", + "type": "text", + "primaryKey": false, + "notNull": true + }, + "expiresAt": { + "name": "expiresAt", + "type": "timestamp", + "primaryKey": false, + "notNull": true + } + }, + "indexes": { + "wallet_challenges_expiresAt_index": { + "name": "wallet_challenges_expiresAt_index", + "columns": [ + { + "expression": "expiresAt", + "isExpression": false, + "asc": true, + "nulls": "last" + } + ], + "isUnique": false, + "concurrently": false, + "method": "btree", + "with": {} + } + }, + "foreignKeys": {}, + "compositePrimaryKeys": {}, + "uniqueConstraints": {}, + "policies": {}, + "checkConstraints": {}, + "isRLSEnabled": false + }, + "public.wallet_sessions": { + "name": "wallet_sessions", + "schema": "", + "columns": { + "id": { + "name": "id", + "type": "uuid", + "primaryKey": true, + "notNull": true + }, + "createdAt": { + "name": "createdAt", + "type": "timestamp", + "primaryKey": false, + "notNull": true, + "default": "now()" + }, + "userId": { + "name": "userId", + "type": "uuid", + "primaryKey": false, + "notNull": false + }, + "walletAddress": { + "name": "walletAddress", + "type": "text", + "primaryKey": false, + "notNull": true + }, + "walletType": { + "name": "walletType", + "type": "text", + "primaryKey": false, + "notNull": true + }, + "signedMessage": { + "name": "signedMessage", + "type": "text", + "primaryKey": false, + "notNull": true + }, + "signature": { + "name": "signature", + "type": "text", + "primaryKey": false, + "notNull": true + }, + "signedAt": { + "name": "signedAt", + "type": "timestamp", + "primaryKey": false, + "notNull": true + }, + "expiresAt": { + "name": "expiresAt", + "type": "timestamp", + "primaryKey": false, + "notNull": true + } + }, + "indexes": { + "wallet_sessions_userId_index": { + "name": "wallet_sessions_userId_index", + "columns": [ + { + "expression": "userId", + "isExpression": false, + "asc": true, + "nulls": "last" + } + ], + "isUnique": false, + "concurrently": false, + "method": "btree", + "with": {} + }, + "wallet_sessions_expiresAt_index": { + "name": "wallet_sessions_expiresAt_index", + "columns": [ + { + "expression": "expiresAt", + "isExpression": false, + "asc": true, + "nulls": "last" + } + ], + "isUnique": false, + "concurrently": false, + "method": "btree", + "with": {} + }, + "wallet_sessions_walletAddress_index": { + "name": "wallet_sessions_walletAddress_index", + "columns": [ + { + "expression": "walletAddress", + "isExpression": false, + "asc": true, + "nulls": "last" + } + ], + "isUnique": false, + "concurrently": false, + "method": "btree", + "with": {} + } + }, + "foreignKeys": { + "wallet_sessions_userId_users_id_fk": { + "name": "wallet_sessions_userId_users_id_fk", + "tableFrom": "wallet_sessions", + "tableTo": "users", + "columnsFrom": [ + "userId" + ], + "columnsTo": [ + "id" + ], + "onDelete": "cascade", + "onUpdate": "no action" + } + }, + "compositePrimaryKeys": {}, + "uniqueConstraints": {}, + "policies": {}, + "checkConstraints": {}, + "isRLSEnabled": false + }, + "public.zklogin_sessions": { + "name": "zklogin_sessions", + "schema": "", + "columns": { + "id": { + "name": "id", + "type": "uuid", + "primaryKey": true, + "notNull": true + }, + "createdAt": { + "name": "createdAt", + "type": "timestamp", + "primaryKey": false, + "notNull": true, + "default": "now()" + }, + "userId": { + "name": "userId", + "type": "uuid", + "primaryKey": false, + "notNull": false + }, + "ephemeralPrivateKey": { + "name": "ephemeralPrivateKey", + "type": "text", + "primaryKey": false, + "notNull": true + }, + "ephemeralPublicKey": { + "name": "ephemeralPublicKey", + "type": "text", + "primaryKey": false, + "notNull": true + }, + "maxEpoch": { + "name": "maxEpoch", + "type": "integer", + "primaryKey": false, + "notNull": true + }, + "randomness": { + "name": "randomness", + "type": "text", + "primaryKey": false, + "notNull": true + }, + "nonce": { + "name": "nonce", + "type": "text", + "primaryKey": false, + "notNull": true + }, + "zkProof": { + "name": "zkProof", + "type": "jsonb", + "primaryKey": false, + "notNull": false + }, + "expiresAt": { + "name": "expiresAt", + "type": "timestamp", + "primaryKey": false, + "notNull": true + } + }, + "indexes": { + "zklogin_sessions_userId_index": { + "name": "zklogin_sessions_userId_index", + "columns": [ + { + "expression": "userId", + "isExpression": false, + "asc": true, + "nulls": "last" + } + ], + "isUnique": false, + "concurrently": false, + "method": "btree", + "with": {} + }, + "zklogin_sessions_expiresAt_index": { + "name": "zklogin_sessions_expiresAt_index", + "columns": [ + { + "expression": "expiresAt", + "isExpression": false, + "asc": true, + "nulls": "last" + } + ], + "isUnique": false, + "concurrently": false, + "method": "btree", + "with": {} + } + }, + "foreignKeys": { + "zklogin_sessions_userId_users_id_fk": { + "name": "zklogin_sessions_userId_users_id_fk", + "tableFrom": "zklogin_sessions", + "tableTo": "users", + "columnsFrom": [ + "userId" + ], + "columnsTo": [ + "id" + ], + "onDelete": "cascade", + "onUpdate": "no action" + } + }, + "compositePrimaryKeys": {}, + "uniqueConstraints": {}, + "policies": {}, + "checkConstraints": {}, + "isRLSEnabled": false + } + }, + "enums": { + "public.ai_message_status": { + "name": "ai_message_status", + "schema": "public", + "values": [ + "streaming", + "awaiting-approval", + "in-progress", + "completed", + "error" + ] + }, + "public.auth_method": { + "name": "auth_method", + "schema": "public", + "values": [ + "zklogin", + "wallet" + ] + }, + "public.memory_status": { + "name": "memory_status", + "schema": "public", + "values": [ + "preparing", + "pending", + "signing", + "uploading", + "indexing", + "saved", + "rejected", + "error" + ] + }, + "public.message_role": { + "name": "message_role", + "schema": "public", + "values": [ + "user", + "assistant" + ] + } + }, + "schemas": {}, + "sequences": {}, + "roles": {}, + "policies": {}, + "views": {}, + "_meta": { + "columns": {}, + "schemas": {}, + "tables": {} + } +} \ No newline at end of file diff --git a/apps/noter/package/shared/db/migrations/meta/_journal.json b/apps/noter/package/shared/db/migrations/meta/_journal.json index f0a9c2978..6b14e3d0a 100644 --- a/apps/noter/package/shared/db/migrations/meta/_journal.json +++ b/apps/noter/package/shared/db/migrations/meta/_journal.json @@ -15,6 +15,13 @@ "when": 1773214559127, "tag": "0001_flat_amazoness", "breakpoints": true + }, + { + "idx": 2, + "version": "7", + "when": 1774594396885, + "tag": "0002_conscious_deathbird", + "breakpoints": true } ] } \ No newline at end of file diff --git a/pnpm-lock.yaml b/pnpm-lock.yaml index 23f0458af..fcb3094c5 100644 --- a/pnpm-lock.yaml +++ b/pnpm-lock.yaml @@ -645,8 +645,8 @@ importers: specifier: ^16.4.5 version: 16.6.1 drizzle-kit: - specifier: ^0.31.9 - version: 0.31.9 + specifier: ^0.31.10 + version: 0.31.10 eslint: specifier: ^9 version: 9.39.4(jiti@2.6.1) @@ -702,8 +702,8 @@ importers: specifier: ^13.7.0 version: 13.12.0(react@19.0.1) '@mysten-incubation/memwal': - specifier: workspace:* - version: link:../../packages/sdk + specifier: ^0.0.1-dev.0 + version: 0.0.1(@mysten/seal@1.1.1(@mysten/sui@2.8.0(typescript@5.9.3)))(@mysten/sui@2.8.0(typescript@5.9.3))(@mysten/walrus@1.0.4(@mysten/sui@2.8.0(typescript@5.9.3)))(ai@6.0.37(zod@3.25.76))(zod@3.25.76) '@noble/ed25519': specifier: ^2.2.3 version: 2.3.0 @@ -3045,6 +3045,26 @@ packages: resolution: {integrity: sha512-cXu86tF4VQVfwz8W1SPbhoRyHJkti6mjH/XJIxp40jhO4j2k1m4KYrEykxqWPkFF3vrK4rgQppBh//AwyGSXPA==} engines: {node: '>=18'} + '@mysten-incubation/memwal@0.0.1': + resolution: {integrity: sha512-kRAFFJBdk3D9XvGHZdPOrnz2x4C7dwCRf0xTaeLFAVTgVwfpk3GmOnJZ1O+pAQyrAhweAzBXNXBWutShnPWgJg==} + peerDependencies: + '@mysten/seal': '>=1.1.0' + '@mysten/sui': '>=2.5.0' + '@mysten/walrus': '>=1.0.3' + ai: '>=4.0.0' + zod: ^3.23.0 + peerDependenciesMeta: + '@mysten/seal': + optional: true + '@mysten/sui': + optional: true + '@mysten/walrus': + optional: true + ai: + optional: true + zod: + optional: true + '@mysten/bcs@1.2.0': resolution: {integrity: sha512-LuKonrGdGW7dq/EM6U2L9/as7dFwnhZnsnINzB/vu08Xfrj0qzWwpLOiXagAa5yZOPLK7anRZydMonczFkUPzA==} @@ -6433,8 +6453,8 @@ packages: resolution: {integrity: sha512-Rcf0nYCAKizwjWQCY+d3zytyuTbDb81NcaPor+8NebESlUz1+9W3uGl0+r9FhU4Qal5Zv9j/7neXCSCe7DHzjA==} hasBin: true - drizzle-kit@0.31.9: - resolution: {integrity: sha512-GViD3IgsXn7trFyBUUHyTFBpH/FsHTxYJ66qdbVggxef4UBPHRYxQaRzYLTuekYnk9i5FIEL9pbBIwMqX/Uwrg==} + drizzle-kit@0.31.10: + resolution: {integrity: sha512-7OZcmQUrdGI+DUNNsKBn1aW8qSoKuTH7d0mYgSP8bAzdFzKoovxEFnoGQp2dVs82EOJeYycqRtciopszwUf8bw==} hasBin: true drizzle-orm@0.34.1: @@ -10269,6 +10289,7 @@ packages: tar@6.1.15: resolution: {integrity: sha512-/zKt9UyngnxIT/EAGYuxaMYgOIJiP81ab9ZfkILq4oNLPFX50qyYmu7jRj9qeXoxmJHjGlbH0+cm2uy1WCs10A==} engines: {node: '>=10'} + deprecated: Old versions of tar are not supported, and contain widely publicized security vulnerabilities, which have been fixed in the current version. Please update. Support for old versions may be purchased (at exorbitant rates) by contacting i@izs.me teeny-request@10.1.0: resolution: {integrity: sha512-3ZnLvgWF29jikg1sAQ1g0o+lr5JX6sVgYvfUJazn7ZjJroDBUTWp44/+cFVX0bULjv4vci+rBD+oGVAkWqhUbw==} @@ -13486,6 +13507,17 @@ snapshots: outvariant: 1.4.3 strict-event-emitter: 0.5.1 + '@mysten-incubation/memwal@0.0.1(@mysten/seal@1.1.1(@mysten/sui@2.8.0(typescript@5.9.3)))(@mysten/sui@2.8.0(typescript@5.9.3))(@mysten/walrus@1.0.4(@mysten/sui@2.8.0(typescript@5.9.3)))(ai@6.0.37(zod@3.25.76))(zod@3.25.76)': + dependencies: + '@noble/ed25519': 2.3.0 + '@noble/hashes': 2.0.1 + optionalDependencies: + '@mysten/seal': 1.1.1(@mysten/sui@2.8.0(typescript@5.9.3)) + '@mysten/sui': 2.8.0(typescript@5.9.3) + '@mysten/walrus': 1.0.4(@mysten/sui@2.8.0(typescript@5.9.3)) + ai: 6.0.37(zod@3.25.76) + zod: 3.25.76 + '@mysten/bcs@1.2.0': dependencies: bs58: 6.0.0 @@ -18543,14 +18575,12 @@ snapshots: transitivePeerDependencies: - supports-color - drizzle-kit@0.31.9: + drizzle-kit@0.31.10: dependencies: '@drizzle-team/brocli': 0.10.2 '@esbuild-kit/esm-loader': 2.6.5 esbuild: 0.25.12 - esbuild-register: 3.6.0(esbuild@0.25.12) - transitivePeerDependencies: - - supports-color + tsx: 4.21.0 drizzle-orm@0.34.1(@opentelemetry/api@1.9.0)(@types/react@18.3.28)(postgres@3.4.8)(react@19.0.1): optionalDependencies: @@ -18819,13 +18849,6 @@ snapshots: transitivePeerDependencies: - supports-color - esbuild-register@3.6.0(esbuild@0.25.12): - dependencies: - debug: 4.4.3 - esbuild: 0.25.12 - transitivePeerDependencies: - - supports-color - esbuild@0.18.20: optionalDependencies: '@esbuild/android-arm': 0.18.20 @@ -18984,8 +19007,8 @@ snapshots: '@next/eslint-plugin-next': 16.1.6 eslint: 9.39.4(jiti@2.6.1) eslint-import-resolver-node: 0.3.9 - eslint-import-resolver-typescript: 3.10.1(eslint-plugin-import@2.32.0(eslint@9.39.4(jiti@2.6.1)))(eslint@9.39.4(jiti@2.6.1)) - eslint-plugin-import: 2.32.0(eslint-import-resolver-typescript@3.10.1(eslint-plugin-import@2.32.0(eslint@9.39.4(jiti@2.6.1)))(eslint@9.39.4(jiti@2.6.1)))(eslint@9.39.4(jiti@2.6.1)) + eslint-import-resolver-typescript: 3.10.1(eslint-plugin-import@2.32.0)(eslint@9.39.4(jiti@2.6.1)) + eslint-plugin-import: 2.32.0(eslint-import-resolver-typescript@3.10.1)(eslint@9.39.4(jiti@2.6.1)) eslint-plugin-jsx-a11y: 6.10.2(eslint@9.39.4(jiti@2.6.1)) eslint-plugin-react: 7.37.5(eslint@9.39.4(jiti@2.6.1)) eslint-plugin-react-hooks: 7.0.1(eslint@9.39.4(jiti@2.6.1)) @@ -19007,7 +19030,7 @@ snapshots: transitivePeerDependencies: - supports-color - eslint-import-resolver-typescript@3.10.1(eslint-plugin-import@2.32.0(eslint@9.39.4(jiti@2.6.1)))(eslint@9.39.4(jiti@2.6.1)): + eslint-import-resolver-typescript@3.10.1(eslint-plugin-import@2.32.0)(eslint@9.39.4(jiti@2.6.1)): dependencies: '@nolyfill/is-core-module': 1.0.39 debug: 4.4.3 @@ -19018,21 +19041,21 @@ snapshots: tinyglobby: 0.2.15 unrs-resolver: 1.11.1 optionalDependencies: - eslint-plugin-import: 2.32.0(eslint-import-resolver-typescript@3.10.1(eslint-plugin-import@2.32.0(eslint@9.39.4(jiti@2.6.1)))(eslint@9.39.4(jiti@2.6.1)))(eslint@9.39.4(jiti@2.6.1)) + eslint-plugin-import: 2.32.0(eslint-import-resolver-typescript@3.10.1)(eslint@9.39.4(jiti@2.6.1)) transitivePeerDependencies: - supports-color - eslint-module-utils@2.12.1(eslint-import-resolver-node@0.3.9)(eslint-import-resolver-typescript@3.10.1(eslint-plugin-import@2.32.0(eslint@9.39.4(jiti@2.6.1)))(eslint@9.39.4(jiti@2.6.1)))(eslint@9.39.4(jiti@2.6.1)): + eslint-module-utils@2.12.1(eslint-import-resolver-node@0.3.9)(eslint-import-resolver-typescript@3.10.1)(eslint@9.39.4(jiti@2.6.1)): dependencies: debug: 3.2.7 optionalDependencies: eslint: 9.39.4(jiti@2.6.1) eslint-import-resolver-node: 0.3.9 - eslint-import-resolver-typescript: 3.10.1(eslint-plugin-import@2.32.0(eslint@9.39.4(jiti@2.6.1)))(eslint@9.39.4(jiti@2.6.1)) + eslint-import-resolver-typescript: 3.10.1(eslint-plugin-import@2.32.0)(eslint@9.39.4(jiti@2.6.1)) transitivePeerDependencies: - supports-color - eslint-plugin-import@2.32.0(eslint-import-resolver-typescript@3.10.1(eslint-plugin-import@2.32.0(eslint@9.39.4(jiti@2.6.1)))(eslint@9.39.4(jiti@2.6.1)))(eslint@9.39.4(jiti@2.6.1)): + eslint-plugin-import@2.32.0(eslint-import-resolver-typescript@3.10.1)(eslint@9.39.4(jiti@2.6.1)): dependencies: '@rtsao/scc': 1.1.0 array-includes: 3.1.9 @@ -19043,7 +19066,7 @@ snapshots: doctrine: 2.1.0 eslint: 9.39.4(jiti@2.6.1) eslint-import-resolver-node: 0.3.9 - eslint-module-utils: 2.12.1(eslint-import-resolver-node@0.3.9)(eslint-import-resolver-typescript@3.10.1(eslint-plugin-import@2.32.0(eslint@9.39.4(jiti@2.6.1)))(eslint@9.39.4(jiti@2.6.1)))(eslint@9.39.4(jiti@2.6.1)) + eslint-module-utils: 2.12.1(eslint-import-resolver-node@0.3.9)(eslint-import-resolver-typescript@3.10.1)(eslint@9.39.4(jiti@2.6.1)) hasown: 2.0.2 is-core-module: 2.16.1 is-glob: 4.0.3 From 9fce7c177e1dd7107042fe227efd1705260d16e5 Mon Sep 17 00:00:00 2001 From: ducnmm <165614309+ducnmm@users.noreply.github.com> Date: Fri, 28 Aug 2026 14:46:33 +0700 Subject: [PATCH 03/29] fix(openclaw): clamp memory_search relevance when cosine distance > 1 (WALM-441) --- .../openclaw-memory-memwal/src/cli/search.ts | 3 ++- packages/openclaw-memory-memwal/src/format.ts | 19 +++++++++++++++++++ .../src/tools/search.ts | 9 +++++---- .../test/plugin.test.mjs | 15 ++++++++++++++- 4 files changed, 40 insertions(+), 6 deletions(-) diff --git a/packages/openclaw-memory-memwal/src/cli/search.ts b/packages/openclaw-memory-memwal/src/cli/search.ts index a3ae3ad82..be4c464d3 100644 --- a/packages/openclaw-memory-memwal/src/cli/search.ts +++ b/packages/openclaw-memory-memwal/src/cli/search.ts @@ -6,6 +6,7 @@ import type { MemWal } from "@mysten-incubation/memwal"; import { resolveAgent } from "../config.js"; +import { relevanceRatio } from "../format.js"; import type { PluginConfig } from "../types.js"; /** Register the `openclaw memwal search` command. */ @@ -25,7 +26,7 @@ export function registerSearchCommand(cmd: any, client: MemWal, config: PluginCo const output = result.results.map((r: any) => ({ text: r.text, blob_id: r.blob_id, - relevance: Math.round((1 - r.distance) * 100) / 100, + relevance: relevanceRatio(r.distance), })); console.log(JSON.stringify(output, null, 2)); } catch (err) { diff --git a/packages/openclaw-memory-memwal/src/format.ts b/packages/openclaw-memory-memwal/src/format.ts index f2960bac9..93a3b0829 100644 --- a/packages/openclaw-memory-memwal/src/format.ts +++ b/packages/openclaw-memory-memwal/src/format.ts @@ -41,6 +41,25 @@ export function escapeForPrompt(text: string): string { return text.replace(/[&<>"']/g, (c) => ESCAPE_MAP[c] ?? c); } +/** + * Map pgvector cosine distance `[0, 2]` to similarity in `[0, 1]`. + * Non-finite values clamp to 0 so a bad hit cannot print negative %. + */ +export function cosineSimilarity(distance: number): number { + if (!Number.isFinite(distance)) return 0; + return Math.min(1, Math.max(0, 1 - distance)); +} + +/** Integer 0–100 relevance for prompt display. */ +export function relevancePercent(distance: number): number { + return Math.round(cosineSimilarity(distance) * 100); +} + +/** Two-decimal 0–1 relevance for tool/CLI details. */ +export function relevanceRatio(distance: number): number { + return Math.round(cosineSimilarity(distance) * 100) / 100; +} + /** * Format recalled memories for prompt injection with security warning. * diff --git a/packages/openclaw-memory-memwal/src/tools/search.ts b/packages/openclaw-memory-memwal/src/tools/search.ts index ca4149664..703a5d72d 100644 --- a/packages/openclaw-memory-memwal/src/tools/search.ts +++ b/packages/openclaw-memory-memwal/src/tools/search.ts @@ -10,7 +10,7 @@ import type { MemWal } from "@mysten-incubation/memwal"; import { Type } from "@sinclair/typebox"; import { looksLikeInjection } from "../capture.js"; -import { escapeForPrompt, toolError, withTimeout } from "../format.js"; +import { escapeForPrompt, relevancePercent, relevanceRatio, toolError, withTimeout } from "../format.js"; import type { PluginConfig } from "../types.js"; import { DEFAULT_SEARCH_LIMIT } from "../constants.js"; @@ -71,10 +71,11 @@ export function registerSearchTool(api: any, client: MemWal, config: PluginConfi }; } - // Walrus Memory returns L2 distance — convert to similarity % for readability + // Cosine distance is [0, 2]; clamp so orthogonal/opposite hits + // cannot print a negative relevance % (WALM-441 / GH #798). const formatted = safe .map((r: any, i: number) => { - const relevance = Math.round((1 - r.distance) * 100); + const relevance = relevancePercent(r.distance); return `${i + 1}. ${escapeForPrompt(r.text)} (${relevance}% relevance)`; }) .join("\n"); @@ -92,7 +93,7 @@ export function registerSearchTool(api: any, client: MemWal, config: PluginConfi memories: safe.map((r: any) => ({ text: r.text, blob_id: r.blob_id, - relevance: Math.round((1 - r.distance) * 100) / 100, + relevance: relevanceRatio(r.distance), })), }, }; diff --git a/packages/openclaw-memory-memwal/test/plugin.test.mjs b/packages/openclaw-memory-memwal/test/plugin.test.mjs index 1c9ef612d..55b911f1f 100644 --- a/packages/openclaw-memory-memwal/test/plugin.test.mjs +++ b/packages/openclaw-memory-memwal/test/plugin.test.mjs @@ -10,7 +10,7 @@ import assert from "node:assert/strict"; import { readFileSync } from "node:fs"; import { parseConfig, resolveAgent, keyPreview } from "../dist/config.js"; -import { withTimeout, withRetry, escapeForPrompt, formatMemoriesForPrompt, stripMemoryTags } from "../dist/format.js"; +import { withTimeout, withRetry, escapeForPrompt, formatMemoriesForPrompt, stripMemoryTags, cosineSimilarity, relevancePercent, relevanceRatio } from "../dist/format.js"; import { looksLikeInjection, shouldCapture } from "../dist/capture.js"; import { registerHooks } from "../dist/hooks/index.js"; @@ -264,3 +264,16 @@ test("trivial and filler turns are not captured", () => { assert.equal(shouldCapture("short"), false); assert.equal(shouldCapture("I prefer TypeScript over Rust for backend services at work"), true); }); + +test("cosine similarity clamps to [0, 1] so relevance cannot go negative", () => { + assert.equal(cosineSimilarity(0), 1); + assert.equal(cosineSimilarity(0.25), 0.75); + assert.equal(cosineSimilarity(1), 0); + assert.equal(cosineSimilarity(1.35), 0); + assert.equal(cosineSimilarity(2), 0); + assert.equal(cosineSimilarity(Number.NaN), 0); + assert.equal(relevancePercent(1.35), 0); + assert.equal(relevanceRatio(1.35), 0); + assert.equal(relevancePercent(0.25), 75); + assert.equal(relevanceRatio(0.25), 0.75); +}); From ba4e336b3af47bb5cdacf5ac17fc01f94202437c Mon Sep 17 00:00:00 2001 From: Le Tien Phat <91601109+Niko1444@users.noreply.github.com> Date: Fri, 28 Aug 2026 17:08:00 +0700 Subject: [PATCH 04/29] feat(sdk): listNamespaces() so an agent can discover namespaces MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit An agent connecting to an unfamiliar account had no way to learn which namespaces hold memories. Recall is similarity-ranked and needs a namespace to search, so the only options were guessing names or falling back to "default" — which undercuts cross-session memory portability. `GET /v1/owners/{owner}/namespaces` already shipped in caff5673. This surfaces it on the TS SDK, metadata-only: no blob fetch, no decryption, and no SEAL session is built or transmitted. Owner resolution needed solving first. The read routes take the address in the path and reject a mismatch against the caller's credentials, but MemWalConfig carries only the delegate key and account id — so the SDK could not build the path at all. `resolveOwner()` learns it from POST /api/stats, which authenticates with the same delegate scheme, needs only a namespace, is rate-limit weight 1, and returns the owner the server resolved. Memoised per client, single-flight guarded like the compatibility probe. Using a stats endpoint as a whoami is indirect. It was chosen over adding a server-side `me` alias because the SDK ships to npm and self-hosters run their own relayer: a path-level change would 403 against every relayer older than it, needing a compatibility branch. The indirection is contained in one private method, no public API mentions owner, and if a self-reference lands later `resolveOwner()` is the only thing that changes. Result types mirror the relayer wire shape (snake_case out), matching the existing RecallResult.dropped_count convention rather than adding a mapping layer. The query string is part of the signed path because the server verifies against path_and_query, not path. MemWalMock gains the same method so the drop-in double keeps parity, aggregating seeded memories by namespace with deterministic timestamps derived from insertion order. Verified live against relayer.dev.memwal.ai, not only against stubs: owner resolution, a plain signed GET, a signed GET carrying a query string, and a cursor round-trip whose base64 cursor contains characters that percent-encode — all 200. Signed-GET-with-query-string had no prior call site in the SDK, so it was the one path that unit stubs could not have validated. That run also showed the relayer returns snapshot_version 2, so the mock now returns 2 rather than disagreeing with the server on a wire-format version. Deferred: Python SDK, the memwal_namespaces MCP tool, and the CLI command — all three need the SDK published first, the same blocker as WALM-428. --- docs/sdk/api-reference.md | 39 +++++++ packages/sdk/src/index.ts | 3 + packages/sdk/src/memwal.ts | 90 +++++++++++++++ packages/sdk/src/mock.ts | 53 +++++++++ packages/sdk/src/types.ts | 41 +++++++ packages/sdk/test/list-namespaces.test.mjs | 128 +++++++++++++++++++++ packages/sdk/test/mock.test.mjs | 42 +++++++ 7 files changed, 396 insertions(+) create mode 100644 packages/sdk/test/list-namespaces.test.mjs diff --git a/docs/sdk/api-reference.md b/docs/sdk/api-reference.md index dfccf3e54..ddf9fb6db 100644 --- a/docs/sdk/api-reference.md +++ b/docs/sdk/api-reference.md @@ -267,6 +267,45 @@ Rebuild missing indexed entries for one namespace from Walrus. Incremental — o } ``` +### `listNamespaces(options?): Promise` + +List the namespaces this account holds memories in. Metadata only — no blob fetch and no decryption. + +Recall is similarity-ranked and needs a namespace to search, so an agent connecting to an unfamiliar account would otherwise have to guess names or fall back to `"default"`. + +- `options.cursor` — the previous page's `next_cursor`, to continue a walk or poll incrementally +- `options.limit` — page size; the relayer defaults to `100` and clamps to `500` + +**Returns:** + +```ts +{ + namespaces: Array<{ + id: string; + name: string; + memory_count: number; + storage_used: number; // bytes + updated_at: string; // MAX(updated_at) across the namespace + }>; + next_cursor: string | null; + has_more: boolean; + snapshot_version: number; +} +``` + +Paginate on `has_more`, not on page length — the relayer clamps `limit`, so a caller asking for more than the cap gets exactly the cap back and would wrongly conclude it was done. + +```ts +let cursor: string | undefined; +let more = true; +while (more) { + const page = await memwal.listNamespaces({ cursor }); + for (const ns of page.namespaces) console.log(ns.name, ns.memory_count); + cursor = page.next_cursor ?? undefined; + more = page.has_more; +} +``` + ### `health(): Promise` Check relayer health. Does not require authentication — a successful response confirms the relayer is reachable, not that your `key`/`accountId` are valid. A signed call (e.g. `remember()`, `recall()`) can still fail with `401` immediately after a passing `health()`. diff --git a/packages/sdk/src/index.ts b/packages/sdk/src/index.ts index c2952622a..612bddaae 100644 --- a/packages/sdk/src/index.ts +++ b/packages/sdk/src/index.ts @@ -76,4 +76,7 @@ export type { RelayerBuildMetadata, RelayerDeprecationNotice, RelayerVersionMetadata, + NamespaceSummary, + NamespacesResult, + ListNamespacesOptions, } from "./types.js"; diff --git a/packages/sdk/src/memwal.ts b/packages/sdk/src/memwal.ts index ad610a8f4..3b2de95e8 100644 --- a/packages/sdk/src/memwal.ts +++ b/packages/sdk/src/memwal.ts @@ -45,6 +45,8 @@ import type { RecallManualOptions, RecallManualResult, RestoreResult, + NamespacesResult, + ListNamespacesOptions, RememberBulkItem, RememberBulkOptions, RememberBulkResult, @@ -199,6 +201,11 @@ export class MemWal { * accepted the write, the caller's next identical attempt reuses the key * and collapses onto the original paid job. */ + /** Resolved owner address for this account. See `resolveOwner()`. */ + private ownerAddress: string | null = null; + /** Single-flight guard so concurrent reads share one owner resolution. */ + private ownerPromise: Promise | null = null; + private pendingRememberKeys = new Map(); private constructor(config: MemWalConfig) { @@ -914,6 +921,89 @@ export class MemWal { /** * Check server health. The endpoint is public and does not require request signing. */ + /** + * List the namespaces this account holds memories in. + * + * Recall is similarity-ranked and needs a namespace to search; without + * this, an agent connecting to an unfamiliar account has to guess names + * or fall back to `"default"`. Returns metadata only — no blob fetch, no + * decryption. + * + * Paginate with `has_more`, NOT page length: the server clamps `limit`, + * so asking for more than the cap returns exactly the cap. + * + * ```ts + * let cursor: string | undefined; + * do { + * const page = await memwal.listNamespaces({ cursor }); + * for (const ns of page.namespaces) console.log(ns.name, ns.memory_count); + * cursor = page.next_cursor ?? undefined; + * var more = page.has_more; + * } while (more); + * ``` + */ + async listNamespaces(options: ListNamespacesOptions = {}): Promise { + const owner = await this.resolveOwner(); + + const params = new URLSearchParams(); + if (options.cursor !== undefined) params.set("updated_after", options.cursor); + if (options.limit !== undefined) params.set("limit", String(options.limit)); + const query = params.toString(); + + // Query string must be part of the signed path: the server verifies + // against `path_and_query`, not `path` (see `auth.rs`). + const path = `/v1/owners/${owner}/namespaces${query ? `?${query}` : ""}`; + + // Metadata-only read — no ciphertext comes back, so no SEAL session + // is built or transmitted. + return this.signedRequest("GET", path, {}, [200], { + includeDelegateKey: false, + }); + } + + /** + * Resolve this account's owner address, memoised for the client's life. + * + * The owner-scoped read routes take the address in the path and reject a + * mismatch against the caller's credentials — but `MemWalConfig` carries + * only the delegate key and account id, so the SDK has to learn its own + * address from the server. + * + * `POST /api/stats` is used because it authenticates with the same + * delegate scheme, needs nothing but a namespace, is rate-limit weight 1, + * and returns the owner the server resolved. Using a stats endpoint as a + * whoami is admittedly indirect; it avoids a server change and keeps this + * working against relayers older than any such change. If a dedicated + * self-reference lands (e.g. accepting `me` as the path owner), this + * method is the only place that needs to change. + */ + private async resolveOwner(): Promise { + if (this.ownerAddress) return this.ownerAddress; + if (this.ownerPromise) return this.ownerPromise; + + this.ownerPromise = (async () => { + const stats = await this.signedRequest<{ owner?: string }>( + "POST", + "/api/stats", + { namespace: this.namespace }, + [200], + { includeDelegateKey: false }, + ); + if (!stats.owner) { + throw new Error( + "Walrus Memory could not resolve this account's owner address " + + "(POST /api/stats returned no owner).", + ); + } + this.ownerAddress = stats.owner; + return stats.owner; + })().finally(() => { + this.ownerPromise = null; + }); + + return this.ownerPromise; + } + async health(): Promise { const res = await fetch(`${this.serverUrl}/health`); if (!res.ok) { diff --git a/packages/sdk/src/mock.ts b/packages/sdk/src/mock.ts index 7f6be213f..e896339ed 100644 --- a/packages/sdk/src/mock.ts +++ b/packages/sdk/src/mock.ts @@ -18,6 +18,8 @@ import type { RememberJobStatus, RememberResult, RestoreResult, + NamespacesResult, + ListNamespacesOptions, } from "./types.js"; import { applyTokenBudget, estimateTokens } from "./tokens.js"; @@ -45,6 +47,14 @@ interface MockMemory { sequence: number; } +/** + * Fixed base for synthesised namespace timestamps. `MockMemory` carries no + * clock, so `updated_at` is derived from insertion order instead — keeping + * mock runs reproducible while preserving the real relayer's property that + * later writes sort later. + */ +const MOCK_NAMESPACE_EPOCH_MS = Date.UTC(2026, 0, 1); + const MOCK_VERSION: RelayerVersionMetadata = { relayerVersion: "memwal-mock", apiVersion: "1.0.0", @@ -374,6 +384,49 @@ export class MemWalMock { }; } + async listNamespaces(options: ListNamespacesOptions = {}): Promise { + const grouped = new Map(); + for (const memory of this.memories) { + const entry = grouped.get(memory.namespace) ?? { count: 0, bytes: 0, sequence: 0 }; + entry.count += 1; + entry.bytes += new TextEncoder().encode(memory.text).length; + entry.sequence = Math.max(entry.sequence, memory.sequence); + grouped.set(memory.namespace, entry); + } + + const all = [...grouped.entries()] + .map(([name, entry]) => ({ + id: `mock-ns-${name}`, + name, + memory_count: entry.count, + storage_used: entry.bytes, + updated_at: new Date( + MOCK_NAMESPACE_EPOCH_MS + entry.sequence * 1000 + ).toISOString(), + })) + .sort((a, b) => + a.updated_at === b.updated_at + ? a.name.localeCompare(b.name) + : a.updated_at.localeCompare(b.updated_at) + ); + + // Mirrors the relayer's keyset walk: `cursor` is an exclusive + // `updated_after` watermark, and `has_more` — not page length — says + // whether to keep going. + const remaining = options.cursor + ? all.filter((ns) => ns.updated_at > options.cursor!) + : all; + const page = remaining.slice(0, options.limit ?? remaining.length); + + return { + namespaces: page, + next_cursor: page.length ? page[page.length - 1].updated_at : null, + has_more: remaining.length > page.length, + // Matches the live relayer's current wire-format version. + snapshot_version: 2, + }; + } + async health(): Promise { return { status: "ok", version: "memwal-mock", ...MOCK_VERSION }; } diff --git a/packages/sdk/src/types.ts b/packages/sdk/src/types.ts index 1ef448755..617f45aef 100644 --- a/packages/sdk/src/types.ts +++ b/packages/sdk/src/types.ts @@ -405,6 +405,47 @@ export interface RecallManualHit { } /** Result from restore() */ +/** One namespace in a `listNamespaces()` page. Mirrors the relayer wire shape. */ +export interface NamespaceSummary { + id: string; + name: string; + memory_count: number; + storage_used: number; + /** + * `MAX(updated_at)` across the namespace's memories — the same value the + * keyset cursor is built from, surfaced so callers can tell *what* + * changed rather than only that their watermark moved. + */ + updated_at: string; +} + +/** Result from listNamespaces() */ +export interface NamespacesResult { + namespaces: NamespaceSummary[]; + /** + * Watermark to hand back as `cursor` on the next call. Populated on every + * page, including the last, so a caller that has finished syncing still + * has a checkpoint to poll from later. + */ + next_cursor: string | null; + /** + * Authoritative "keep paginating" signal. Do NOT infer this from page + * length: the server silently clamps `limit`, so a caller asking for more + * than the cap gets exactly the cap back and would wrongly conclude it + * was done. + */ + has_more: boolean; + snapshot_version: number; +} + +/** Options for listNamespaces() */ +export interface ListNamespacesOptions { + /** Previous page's `next_cursor`, to continue a walk or poll incrementally. */ + cursor?: string; + /** Page size. Server defaults to 100 and clamps to 500. */ + limit?: number; +} + export interface RestoreResult { restored: number; skipped: number; diff --git a/packages/sdk/test/list-namespaces.test.mjs b/packages/sdk/test/list-namespaces.test.mjs new file mode 100644 index 000000000..c6d0b7556 --- /dev/null +++ b/packages/sdk/test/list-namespaces.test.mjs @@ -0,0 +1,128 @@ +import assert from "node:assert/strict"; +import test from "node:test"; + +import { MemWal } from "../dist/memwal.js"; + +const OWNER = "0xowner0000000000000000000000000000000000000000000000000000000001"; +const originalFetch = globalThis.fetch; + +function client() { + return MemWal.create({ + key: new Uint8Array(32).fill(1), + accountId: "0x1", + serverUrl: "https://relayer.example", + }); +} + +/** + * Stub the three calls a listNamespaces() round-trip makes: the compatibility + * preflight, the owner resolution, and the read itself. Records every request + * so tests can assert on paths, headers and call counts. + */ +function stubRelayer(namespacesBody) { + const calls = []; + globalThis.fetch = async (url, init = {}) => { + const u = new URL(url); + calls.push({ path: u.pathname, search: u.search, method: init.method ?? "GET", headers: init.headers ?? {} }); + + if (u.pathname === "/version") { + return Response.json({ + apiVersion: "1.0.0", + relayerVersion: "1.0.0", + minSupportedSdk: { typescript: "0.0.4" }, + }); + } + if (u.pathname === "/api/stats") { + return Response.json({ memory_count: 0, storage_bytes: 0, namespace: "default", owner: OWNER }); + } + if (u.pathname === `/v1/owners/${OWNER}/namespaces`) { + return Response.json(namespacesBody); + } + throw new Error(`unexpected request: ${u.pathname}`); + }; + return calls; +} + +const ONE_PAGE = { + namespaces: [ + { id: "ns-1", name: "work", memory_count: 12, storage_used: 2048, updated_at: "2026-08-20T10:00:00Z" }, + ], + next_cursor: "2026-08-20T10:00:00Z", + has_more: false, + snapshot_version: 1, +}; + +test.afterEach(() => { + globalThis.fetch = originalFetch; +}); + +test("listNamespaces reads the owner-scoped namespaces path", async () => { + const calls = stubRelayer(ONE_PAGE); + + await client().listNamespaces(); + + const read = calls.find((c) => c.path.endsWith("/namespaces")); + assert.ok(read, "expected a request to the namespaces endpoint"); + assert.equal(read.path, `/v1/owners/${OWNER}/namespaces`); + assert.equal(read.method, "GET"); +}); + +test("listNamespaces resolves the owner once and reuses it", async () => { + const calls = stubRelayer(ONE_PAGE); + const memwal = client(); + + await memwal.listNamespaces(); + await memwal.listNamespaces(); + + const statsCalls = calls.filter((c) => c.path === "/api/stats"); + assert.equal(statsCalls.length, 1, "owner resolution must be memoised across calls"); + assert.equal(calls.filter((c) => c.path.endsWith("/namespaces")).length, 2); +}); + +test("listNamespaces forwards cursor as updated_after and passes limit", async () => { + const calls = stubRelayer(ONE_PAGE); + + await client().listNamespaces({ cursor: "2026-08-20T10:00:00Z", limit: 25 }); + + const read = calls.find((c) => c.path.endsWith("/namespaces")); + const params = new URLSearchParams(read.search); + assert.equal(params.get("updated_after"), "2026-08-20T10:00:00Z"); + assert.equal(params.get("limit"), "25"); +}); + +test("listNamespaces omits query params that were not supplied", async () => { + const calls = stubRelayer(ONE_PAGE); + + await client().listNamespaces(); + + const read = calls.find((c) => c.path.endsWith("/namespaces")); + const params = new URLSearchParams(read.search); + assert.equal(params.get("updated_after"), null); + assert.equal(params.get("limit"), null); +}); + +test("listNamespaces returns the relayer's wire shape unchanged", async () => { + stubRelayer(ONE_PAGE); + + const result = await client().listNamespaces(); + + assert.deepEqual(result, ONE_PAGE); + // has_more is the authoritative pagination signal, not page length. + assert.equal(result.has_more, false); + assert.equal(result.namespaces[0].name, "work"); +}); + +test("listNamespaces sends no SEAL session on a metadata-only read", async () => { + const calls = stubRelayer(ONE_PAGE); + + await client().listNamespaces(); + + for (const call of calls.filter((c) => c.path !== "/version")) { + const headers = call.headers; + assert.equal( + headers["x-seal-session"], + undefined, + `${call.path} must not build a decrypt credential for a metadata-only read`, + ); + } +}); diff --git a/packages/sdk/test/mock.test.mjs b/packages/sdk/test/mock.test.mjs index badbc0974..b4c23dace 100644 --- a/packages/sdk/test/mock.test.mjs +++ b/packages/sdk/test/mock.test.mjs @@ -180,3 +180,45 @@ test("MemWalMock provides deterministic embeddings and seed data", async () => { assert.equal((await first.health()).status, "ok"); assert.equal((await first.compatibility()).featureFlags.offlineMock, true); }); + +test("MemWalMock.listNamespaces aggregates seeded memories by namespace", async () => { + const mock = MemWalMock.create({ + initialMemories: [ + { text: "one", namespace: "work" }, + { text: "two", namespace: "work" }, + { text: "three", namespace: "home" }, + ], + }); + + const page = await mock.listNamespaces(); + const byName = Object.fromEntries(page.namespaces.map((n) => [n.name, n])); + + assert.deepEqual(Object.keys(byName).sort(), ["home", "work"]); + assert.equal(byName.work.memory_count, 2); + assert.equal(byName.home.memory_count, 1); + assert.equal(page.has_more, false); +}); + +test("MemWalMock.listNamespaces reports has_more when limit truncates the page", async () => { + const mock = MemWalMock.create({ + initialMemories: [ + { text: "a", namespace: "alpha" }, + { text: "b", namespace: "bravo" }, + { text: "c", namespace: "charlie" }, + ], + }); + + const page = await mock.listNamespaces({ limit: 2 }); + + assert.equal(page.namespaces.length, 2); + assert.equal(page.has_more, true, "has_more is the pagination signal, not page length"); + assert.ok(page.next_cursor, "a truncated page must hand back a cursor"); +}); + +test("MemWalMock.listNamespaces reports the relayer's current snapshot_version", async () => { + // Verified against relayer.dev.memwal.ai on 2026-08-28: the live read API + // returns snapshot_version 2. A double that disagrees with the server on a + // wire-format version is a trap for anyone testing version-gated logic. + const page = await MemWalMock.create().listNamespaces(); + assert.equal(page.snapshot_version, 2); +}); From a9412a1ff4867fbdd30f3dc43abcdaac63946995 Mon Sep 17 00:00:00 2001 From: Le Tien Phat <91601109+Niko1444@users.noreply.github.com> Date: Sat, 29 Aug 2026 09:17:38 +0700 Subject: [PATCH 05/29] chore(changeset): add changeset for listNamespaces() Without this the method would ship in a release whose CHANGELOG never mentions it, riding on an unrelated changeset's version bump. CI does not enforce changesets, so nothing would have caught the omission. --- .changeset/list-namespaces.md | 15 +++++++++++++++ 1 file changed, 15 insertions(+) create mode 100644 .changeset/list-namespaces.md diff --git a/.changeset/list-namespaces.md b/.changeset/list-namespaces.md new file mode 100644 index 000000000..cff6f940b --- /dev/null +++ b/.changeset/list-namespaces.md @@ -0,0 +1,15 @@ +--- +"@mysten-incubation/memwal": patch +--- + +Add `listNamespaces()` so an agent can discover which namespaces an account holds memories in (#634). + +Recall is similarity-ranked and needs a namespace to search. Without this, an agent connecting to an unfamiliar account had to guess names or fall back to `"default"`, which undercuts cross-session memory portability. + +`listNamespaces({ cursor?, limit? })` returns `{ namespaces, next_cursor, has_more, snapshot_version }` over the relayer's existing `GET /v1/owners/{owner}/namespaces`. Each entry carries `id`, `name`, `memory_count`, `storage_used` and `updated_at`. Metadata only — no blob fetch, no decryption, and no SEAL session is built or transmitted. + +Paginate on `has_more`, not on page length: the relayer clamps `limit`, so a caller asking for more than the cap gets exactly the cap back and would wrongly conclude it was done. + +The owner-scoped read routes take the address in the path and reject a mismatch, but `MemWalConfig` carries only the delegate key and account id — so the client resolves its own owner address once, memoised, and callers never supply it. + +`MemWalMock` implements the same method, aggregating seeded memories by namespace with deterministic timestamps derived from insertion order. From 71c2764da0ecd8f6d92016b653cab9e0450cb5bd Mon Sep 17 00:00:00 2001 From: HoangDucBach Date: Sat, 29 Aug 2026 11:28:15 +0700 Subject: [PATCH 06/29] fix(chatbot): return 404 when deleting a nonexistent document (WALM-422) DELETE /api/document destructured the first row from getDocumentsById and read .userId off it without checking that a row came back, so an id matching no document threw TypeError and the request failed with an empty 500 body. The GET handler in the same file already guarded for this, so the two disagreed on the same missing document. Add the same guard to DELETE and cover it with a route-level test that pins both the 404 and the parity with GET. --- .../api/document/document-route.unit.test.ts | 124 ++++++++++++++++++ apps/chatbot/app/(chat)/api/document/route.ts | 6 + 2 files changed, 130 insertions(+) create mode 100644 apps/chatbot/app/(chat)/api/document/document-route.unit.test.ts diff --git a/apps/chatbot/app/(chat)/api/document/document-route.unit.test.ts b/apps/chatbot/app/(chat)/api/document/document-route.unit.test.ts new file mode 100644 index 000000000..10ca3ca65 --- /dev/null +++ b/apps/chatbot/app/(chat)/api/document/document-route.unit.test.ts @@ -0,0 +1,124 @@ +import { beforeEach, describe, expect, it, vi } from "vitest"; + +// Regression test for WALM-422: DELETE /api/document read `.userId` off the +// first row without checking that a row came back, so an id matching no +// document threw TypeError and the request 500s with an empty body. The GET +// handler in the same file already had the guard, so the two disagreed on the +// same missing document. +// +// The DB query layer is mocked so no Postgres connection is needed. + +const OWNER_ID = "11111111-1111-1111-1111-111111111111"; +const OTHER_ID = "22222222-2222-2222-2222-222222222222"; +const DOC_ID = "33333333-3333-3333-3333-333333333333"; +const MISSING_ID = "44444444-4444-4444-4444-444444444444"; +const TIMESTAMP = "2026-01-01T00:00:00.000Z"; + +function makeDoc(userId = OWNER_ID) { + return { + id: DOC_ID, + userId, + title: "A document", + content: "body", + kind: "text" as const, + createdAt: new Date(), + }; +} + +const getDocumentsById = vi.fn(async ({ id }: { id: string }) => + id === DOC_ID ? [makeDoc()] : [] +); +const deleteDocumentsByIdAfterTimestamp = vi.fn(async () => [makeDoc()]); +const saveDocument = vi.fn(async () => makeDoc()); + +vi.mock("@/lib/db/queries", () => ({ + getDocumentsById, + deleteDocumentsByIdAfterTimestamp, + saveDocument, +})); + +const auth = vi.fn(async () => ({ user: { id: OWNER_ID }, expires: "" })); +vi.mock("@/app/(auth)/auth", () => ({ auth })); + +function deleteRequest(id: string, timestamp = TIMESTAMP) { + return new Request( + `http://localhost/api/document?id=${id}×tamp=${timestamp}`, + { method: "DELETE" } + ); +} + +async function callDelete(id: string) { + const { DELETE } = await import("./route"); + return DELETE(deleteRequest(id)); +} + +beforeEach(() => { + vi.clearAllMocks(); + getDocumentsById.mockImplementation(async ({ id }: { id: string }) => + id === DOC_ID ? [makeDoc()] : [] + ); + auth.mockResolvedValue({ user: { id: OWNER_ID }, expires: "" }); +}); + +describe("DELETE /api/document", () => { + it("returns 404 for an id that matches no document", async () => { + const response = await callDelete(MISSING_ID); + + expect(response.status).toBe(404); + await expect(response.json()).resolves.toMatchObject({ + code: "not_found:document", + }); + // Nothing may be deleted on the way to reporting the miss. + expect(deleteDocumentsByIdAfterTimestamp).not.toHaveBeenCalled(); + }); + + it("agrees with GET on the same missing document", async () => { + const { GET } = await import("./route"); + const getResponse = await GET( + new Request(`http://localhost/api/document?id=${MISSING_ID}`) + ); + const deleteResponse = await callDelete(MISSING_ID); + + expect(deleteResponse.status).toBe(getResponse.status); + }); + + it("returns 403 for a document owned by someone else", async () => { + getDocumentsById.mockResolvedValue([makeDoc(OTHER_ID)]); + + const response = await callDelete(DOC_ID); + + expect(response.status).toBe(403); + expect(deleteDocumentsByIdAfterTimestamp).not.toHaveBeenCalled(); + }); + + it("deletes the caller's own document", async () => { + const response = await callDelete(DOC_ID); + + expect(response.status).toBe(200); + expect(deleteDocumentsByIdAfterTimestamp).toHaveBeenCalledWith({ + id: DOC_ID, + timestamp: new Date(TIMESTAMP), + }); + }); + + it("rejects a missing timestamp before reaching the database", async () => { + const { DELETE } = await import("./route"); + const response = await DELETE( + new Request(`http://localhost/api/document?id=${DOC_ID}`, { + method: "DELETE", + }) + ); + + expect(response.status).toBe(400); + expect(getDocumentsById).not.toHaveBeenCalled(); + }); + + it("rejects an unauthenticated caller before reaching the database", async () => { + auth.mockResolvedValue(null as never); + + const response = await callDelete(DOC_ID); + + expect(response.status).toBe(401); + expect(getDocumentsById).not.toHaveBeenCalled(); + }); +}); diff --git a/apps/chatbot/app/(chat)/api/document/route.ts b/apps/chatbot/app/(chat)/api/document/route.ts index fe912a1e6..3bd3c1731 100644 --- a/apps/chatbot/app/(chat)/api/document/route.ts +++ b/apps/chatbot/app/(chat)/api/document/route.ts @@ -113,6 +113,12 @@ export async function DELETE(request: Request) { const [document] = documents; + // Same guard the GET handler above runs. Without it an id that matches no row + // reads `.userId` off undefined and the request 500s with an empty body. + if (!document) { + return new ChatbotError("not_found:document").toResponse(); + } + if (document.userId !== session.user.id) { return new ChatbotError("forbidden:document").toResponse(); } From 4bca853a4e1cdcc34059ad44d75d4ca7f974b208 Mon Sep 17 00:00:00 2001 From: HoangDucBach Date: Sat, 29 Aug 2026 19:57:39 +0700 Subject: [PATCH 07/29] fix(noter): bind getSession and logout to the caller's own session (#779) auth.getSession and auth.logout took a sessionId as procedure input and acted on it, so anyone who learned an id could read that session's user and Sui address, or delete the session, without ever presenting the id as a credential. getSession is a query, so the id also travelled in the request URL on every page load, into access logs and browser history. Both now read the id from ctx.sessionId, which createContext fills from the x-session-id header the way protectedProcedure and the memory REST routes already do. The input is gone, so nothing a caller sends can steer either procedure. createContext now also ignores a header that is not a uuid. The zod input schema used to reject those before they reached a uuid column, where Postgres raises instead of simply not matching. On the client the getSession query key no longer varies per session, so useAuth resets the cached lookup whenever the stored session changes. A cached null left by an expired session would otherwise be read as the answer for the session that had just been established. --- apps/noter/package/feature/auth/api/input.ts | 20 +-- apps/noter/package/feature/auth/api/route.ts | 33 ++-- .../auth/api/session-binding.unit.test.ts | 147 ++++++++++++++++++ .../package/feature/auth/hook/use-auth.ts | 54 +++++-- apps/noter/package/feature/auth/index.ts | 5 +- apps/noter/package/shared/lib/trpc/init.ts | 26 +++- 6 files changed, 234 insertions(+), 51 deletions(-) create mode 100644 apps/noter/package/feature/auth/api/session-binding.unit.test.ts diff --git a/apps/noter/package/feature/auth/api/input.ts b/apps/noter/package/feature/auth/api/input.ts index 9b3230ef1..65c9dcdde 100644 --- a/apps/noter/package/feature/auth/api/input.ts +++ b/apps/noter/package/feature/auth/api/input.ts @@ -8,24 +8,10 @@ */ import { z } from "zod"; -import { - uuidv7Schema, - walletSessionInsertSchema, -} from "@/shared/db/type"; +import { walletSessionInsertSchema } from "@/shared/db/type"; -// ═══════════════════════════════════════════════════════════════ -// Session Management Inputs -// ═══════════════════════════════════════════════════════════════ - -/** - * Input for validating existing session - * Uses common idInputSchema pattern, aliased as sessionId - */ -export const validateSessionInput = z.object({ - sessionId: uuidv7Schema, // Same as idInputSchema.shape.id -}); - -export type ValidateSessionInput = z.infer; +// Session id is deliberately absent here: getSession and logout take it from the +// x-session-id header via the tRPC context, so it can never be supplied as input. // ═══════════════════════════════════════════════════════════════ // Wallet Auth Inputs diff --git a/apps/noter/package/feature/auth/api/route.ts b/apps/noter/package/feature/auth/api/route.ts index 3c5791ef7..f36b607bb 100644 --- a/apps/noter/package/feature/auth/api/route.ts +++ b/apps/noter/package/feature/auth/api/route.ts @@ -9,7 +9,7 @@ import { z } from "zod"; import { verifyPersonalMessageSignature } from "@mysten/sui/verify"; import { normalizeSuiAddress } from "@mysten/sui/utils"; import { uuidv7 } from "uuidv7"; -import { validateSessionInput, connectWalletInput } from "./input"; +import { connectWalletInput } from "./input"; import { AUTH_ERRORS } from "../constant"; import { walletSessions } from "@/shared/db/schema"; import * as authService from "../domain/service"; @@ -38,24 +38,29 @@ const suiAddressSchema = z export const authRouter = router({ /** - * Get current session (for resuming auth state). + * Get the caller's own session (for resuming auth state). * Resolves wallet / enoki sessions only; the legacy zkLogin table is not trusted. + * + * The session id is read from the x-session-id header via ctx, never from the + * input, so a caller cannot read a session it does not already hold. Returns + * null for a missing, unknown, or expired session, which is how the client + * detects a stale stored session and clears it. */ - getSession: procedure - .input(validateSessionInput) - .query(({ ctx, input }) => - authService.getActiveSession(ctx.db, input.sessionId) - ), + getSession: procedure.query(({ ctx }) => + ctx.sessionId ? authService.getActiveSession(ctx.db, ctx.sessionId) : null + ), /** - * Logout - clear session (works for both zkLogin and wallet) + * Logout - clear the caller's own session (works for both zkLogin and wallet). + * Like getSession, the id comes from the header, so knowing another user's + * session id is not enough to end their session. */ - logout: procedure - .input(validateSessionInput) - .mutation(async ({ ctx, input }) => { - await authService.deleteSession(ctx.db, input.sessionId); - return { success: true }; - }), + logout: procedure.mutation(async ({ ctx }) => { + if (ctx.sessionId) { + await authService.deleteSession(ctx.db, ctx.sessionId); + } + return { success: true }; + }), /** * Connect wallet - authenticate with Sui wallet (Slush, Sui Wallet) diff --git a/apps/noter/package/feature/auth/api/session-binding.unit.test.ts b/apps/noter/package/feature/auth/api/session-binding.unit.test.ts new file mode 100644 index 000000000..cc0d974ac --- /dev/null +++ b/apps/noter/package/feature/auth/api/session-binding.unit.test.ts @@ -0,0 +1,147 @@ +import { beforeEach, describe, expect, it, vi } from "vitest"; + +// Route-layer binding tests for getSession / logout (issue #779). Both used to +// take a sessionId as procedure input and act on it, so anyone who learned an id +// could read that session's user and address or delete the session, without ever +// presenting the id as a credential. Both now read the id from ctx.sessionId, +// which createContext fills from the x-session-id header. +// +// These call the REAL authRouter through createCaller and pass a victim id as +// input anyway, cast past the types, to prove the input cannot steer either +// procedure. The service layer is mocked so no DB is needed. + +vi.mock("server-only", () => ({})); + +const CALLER_SESSION = "0192f0a0-0000-7000-8000-000000000001"; +const VICTIM_SESSION = "0192f0a0-0000-7000-8000-000000000002"; + +const VICTIM_SESSION_ROW = { + user: { id: "victim", suiAddress: `0x${"b".repeat(64)}` }, + sessionId: VICTIM_SESSION, + suiAddress: `0x${"b".repeat(64)}`, + expiresAt: new Date(Date.now() + 60_000), +}; + +const CALLER_SESSION_ROW = { + user: { id: "caller", suiAddress: `0x${"a".repeat(64)}` }, + sessionId: CALLER_SESSION, + suiAddress: `0x${"a".repeat(64)}`, + expiresAt: new Date(Date.now() + 60_000), +}; + +const getActiveSession = vi.fn(async (_db: unknown, sessionId: string) => { + if (sessionId === CALLER_SESSION) return CALLER_SESSION_ROW; + if (sessionId === VICTIM_SESSION) return VICTIM_SESSION_ROW; + return null; +}); +const deleteSession = vi.fn(async () => undefined); + +vi.mock("../domain/service", () => ({ + getActiveSession, + deleteSession, + toSafeUser: (u: unknown) => u, + DelegateCredentialConflictError: class extends Error {}, +})); + +vi.mock("../lib/enoki-challenge", () => ({ + issueEnokiChallenge: vi.fn(), + verifyAndConsumeEnokiChallenge: vi.fn(), +})); + +vi.mock("@/shared/lib/shared-redis", () => ({ + SharedRedisUnavailableError: class extends Error {}, +})); + +// sessionId mirrors what createContext read from x-session-id; null means the +// caller presented no usable session header. +async function callerFor(sessionId: string | null) { + const { authRouter } = await import("./route"); + const ctx = { + db: {}, + request: new Request("http://localhost/api/trpc/auth"), + sessionId, + userId: sessionId === CALLER_SESSION ? "caller" : null, + }; + return authRouter.createCaller(ctx as never); +} + +beforeEach(() => { + vi.clearAllMocks(); +}); + +describe("getSession — bound to the caller's own session", () => { + it("returns null when the caller presents no session header", async () => { + const caller = await callerFor(null); + + await expect(caller.getSession()).resolves.toBeNull(); + expect(getActiveSession).not.toHaveBeenCalled(); + }); + + it("ignores a victim session id passed as input", async () => { + const caller = await callerFor(null); + + // The exact request from the report: a valid id, no credential for it. + const result = await ( + caller.getSession as unknown as (input: unknown) => Promise + )({ sessionId: VICTIM_SESSION }); + + expect(result).toBeNull(); + expect(getActiveSession).not.toHaveBeenCalledWith( + expect.anything(), + VICTIM_SESSION + ); + }); + + it("reads the header session even when the input names another one", async () => { + const caller = await callerFor(CALLER_SESSION); + + const result = await ( + caller.getSession as unknown as (input: unknown) => Promise + )({ sessionId: VICTIM_SESSION }); + + expect(result).toEqual(CALLER_SESSION_ROW); + expect(getActiveSession).toHaveBeenCalledWith( + expect.anything(), + CALLER_SESSION + ); + }); + + it("returns null for a header session that no longer resolves", async () => { + const caller = await callerFor("0192f0a0-0000-7000-8000-00000000dead"); + + await expect(caller.getSession()).resolves.toBeNull(); + }); +}); + +describe("logout — ends only the caller's own session", () => { + it("deletes nothing when the caller presents no session header", async () => { + const caller = await callerFor(null); + + await expect(caller.logout()).resolves.toEqual({ success: true }); + expect(deleteSession).not.toHaveBeenCalled(); + }); + + it("ignores a victim session id passed as input", async () => { + const caller = await callerFor(null); + + await ( + caller.logout as unknown as (input: unknown) => Promise + )({ sessionId: VICTIM_SESSION }); + + expect(deleteSession).not.toHaveBeenCalled(); + }); + + it("deletes the header session even when the input names another one", async () => { + const caller = await callerFor(CALLER_SESSION); + + await ( + caller.logout as unknown as (input: unknown) => Promise + )({ sessionId: VICTIM_SESSION }); + + expect(deleteSession).toHaveBeenCalledTimes(1); + expect(deleteSession).toHaveBeenCalledWith( + expect.anything(), + CALLER_SESSION + ); + }); +}); diff --git a/apps/noter/package/feature/auth/hook/use-auth.ts b/apps/noter/package/feature/auth/hook/use-auth.ts index 511faf16c..76639d27d 100644 --- a/apps/noter/package/feature/auth/hook/use-auth.ts +++ b/apps/noter/package/feature/auth/hook/use-auth.ts @@ -27,20 +27,33 @@ export function useAuth() { // Wallet disconnect (clears dapp-kit autoConnect state) const { mutateAsync: disconnectWallet } = useDisconnectWallet(); + const utils = trpc.useUtils(); + // tRPC mutations const connectEnokiMutation = trpc.auth.connectEnoki.useMutation(); const connectDelegateKeyMutation = trpc.auth.connectDelegateKey.useMutation(); const logoutMutation = trpc.auth.logout.useMutation(); - // Session validation query - const sessionQuery = trpc.auth.getSession.useQuery( - { sessionId: session?.sessionId || "" }, - { - enabled: !!session?.sessionId, - retry: false, - refetchOnWindowFocus: false, - refetchOnMount: true, - } + // Session validation query. It takes no input: the server reads the session id + // from the x-session-id header that TRPCProvider attaches, so the cache key no + // longer varies per session and every session change has to reset it — see + // resetSessionQuery below. + const sessionQuery = trpc.auth.getSession.useQuery(undefined, { + enabled: !!session?.sessionId, + retry: false, + refetchOnWindowFocus: false, + refetchOnMount: true, + }); + + /** + * Drop the cached session lookup so a result fetched for one session is never + * read as the answer for the next one. Without this a cached null (a session + * that had expired) would make the effect below clear the session that was + * just established. + */ + const resetSessionQuery = useCallback( + () => utils.auth.getSession.reset(), + [utils] ); /** Initialize authentication from persisted session. */ @@ -79,6 +92,7 @@ export function useAuth() { if (result.sessionData) { setSession(result.sessionData); + await resetSessionQuery(); } if (result.user) { @@ -96,7 +110,7 @@ export function useAuth() { throw error; } }, - [connectEnokiMutation, setSession, setAuthenticated] + [connectEnokiMutation, setSession, setAuthenticated, resetSessionQuery] ); /** Connect with delegate key (manual key + account ID). */ @@ -106,6 +120,7 @@ export function useAuth() { const result = await connectDelegateKeyMutation.mutateAsync(params); setSession(result.sessionData); + await resetSessionQuery(); setAuthenticated({ isAuthenticated: true, @@ -120,14 +135,20 @@ export function useAuth() { throw error; } }, - [connectDelegateKeyMutation, setSession, setAuthenticated] + [ + connectDelegateKeyMutation, + setSession, + setAuthenticated, + resetSessionQuery, + ] ); /** Logout — clear session, auth state, and disconnect wallet (prevents autoConnect). */ const logout = useCallback(async () => { try { if (session?.sessionId) { - await logoutMutation.mutateAsync({ sessionId: session.sessionId }); + // No argument: the server ends the session behind the request header. + await logoutMutation.mutateAsync(); } } catch (error) { console.error("Logout failed:", error); @@ -139,7 +160,14 @@ export function useAuth() { // Wallet may already be disconnected } clearAuth(); - }, [session, logoutMutation, disconnectWallet, clearAuth]); + await resetSessionQuery(); + }, [ + session, + logoutMutation, + disconnectWallet, + clearAuth, + resetSessionQuery, + ]); return { ...auth, diff --git a/apps/noter/package/feature/auth/index.ts b/apps/noter/package/feature/auth/index.ts index c31c168c9..67279edd0 100644 --- a/apps/noter/package/feature/auth/index.ts +++ b/apps/noter/package/feature/auth/index.ts @@ -21,10 +21,7 @@ export { } from "./api/form"; // API Input Schemas (if needed by external features) -export type { - ValidateSessionInput, - ConnectWalletInput, -} from "./api/input"; +export type { ConnectWalletInput } from "./api/input"; // Types export type { diff --git a/apps/noter/package/shared/lib/trpc/init.ts b/apps/noter/package/shared/lib/trpc/init.ts index 1c092db62..b500c7b7f 100644 --- a/apps/noter/package/shared/lib/trpc/init.ts +++ b/apps/noter/package/shared/lib/trpc/init.ts @@ -8,22 +8,42 @@ import { eq } from "drizzle-orm"; export type Context = { db: typeof db; request: Request; + /** + * The session id the caller presented in x-session-id, whether or not it + * resolves to a live session. Procedures that act on a session must read it + * from here rather than from their input, so holding the credential is what + * grants access instead of merely knowing the id. + */ + sessionId: string | null; userId: string | null; }; +const UUID_REGEX = + /^[0-9a-f]{8}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{12}$/i; + function getSessionIdFromRequest(req: Request): string | null { - return req.headers.get("x-session-id"); + const sessionId = req.headers.get("x-session-id")?.trim(); + + // Session ids are compared against uuid columns, where a malformed value makes + // Postgres raise rather than simply miss. Anything that is not a uuid cannot + // name a session, so treat it as no credential at all. + if (!sessionId || !UUID_REGEX.test(sessionId)) { + return null; + } + + return sessionId; } export const createContext = async ( opts: FetchCreateContextFnOptions ): Promise => { + const sessionId = getSessionIdFromRequest(opts.req); const noAuth: Context = { db, request: opts.req, + sessionId, userId: null, }; - const sessionId = getSessionIdFromRequest(opts.req); if (!sessionId) return noAuth; // Sessions are resolved only from wallet/enoki sessions, which require proof @@ -37,7 +57,7 @@ export const createContext = async ( .limit(1); if (walletSession?.userId && walletSession.expiresAt > new Date()) { - return { db, request: opts.req, userId: walletSession.userId }; + return { db, request: opts.req, sessionId, userId: walletSession.userId }; } return noAuth; From 48ef97697e2b34756bfbdaf2c31aa426dba703f1 Mon Sep 17 00:00:00 2001 From: HoangDucBach Date: Sat, 29 Aug 2026 21:24:10 +0700 Subject: [PATCH 08/29] fix(researcher): check the destination before fetching a source file url (#778) processSource downloaded a PDF file part with a bare fetch on the url from the request body, so an authenticated chat request could make the server call anything it can reach: a loopback service, an RFC1918 neighbour, or the metadata endpoint on 169.254.169.254. Route the download through fetchPublicUrl, which parses the url, requires http or https, and rejects loopback, private, link-local, and reserved targets. Hostnames are resolved and every address returned has to be public, so a name pointing into a blocked range is refused like the literal is. Redirects are followed by hand and re-checked at each hop, since fetch's own following would skip the check on the new target. The existing denylist in extractUrlsFromText only prefix-matches a few literals, so it misses userinfo, IPv6, 127.x outside 127.0.0.1, and names that resolve inward. That path reaches its target through Jina Reader rather than from this server, so it is left alone here. --- apps/researcher/lib/rag/ingest/index.ts | 6 +- apps/researcher/lib/rag/ingest/safe-fetch.ts | 277 ++++++++++++++++++ .../lib/rag/ingest/safe-fetch.unit.test.ts | 231 +++++++++++++++ 3 files changed, 512 insertions(+), 2 deletions(-) create mode 100644 apps/researcher/lib/rag/ingest/safe-fetch.ts create mode 100644 apps/researcher/lib/rag/ingest/safe-fetch.unit.test.ts diff --git a/apps/researcher/lib/rag/ingest/index.ts b/apps/researcher/lib/rag/ingest/index.ts index 981e62199..e385b71e0 100644 --- a/apps/researcher/lib/rag/ingest/index.ts +++ b/apps/researcher/lib/rag/ingest/index.ts @@ -4,6 +4,7 @@ import { chunkDocument, estimateTokens } from "./chunking"; import { batchEmbed } from "./embeddings"; import { extractFromUrl, extractFromPdf } from "./extract"; import { generateSourceMetadata } from "./metadata"; +import { fetchPublicUrl } from "./safe-fetch"; import { createSource, createSourceChunks } from "@/lib/db/queries"; import { ChatbotError } from "@/lib/errors"; import type { SourceInput } from "@/lib/ai/source-processing"; @@ -39,8 +40,9 @@ export async function processSource({ rawText = await extractFromPdf(source.file); } else { type = "pdf"; - // Download the PDF from the uploaded file URL - const response = await fetch(source.fileUrl); + // Download the PDF from the uploaded file URL. The URL arrives from the + // request body, so the destination is checked before anything is sent. + const response = await fetchPublicUrl(source.fileUrl); if (!response.ok) { throw new ChatbotError( "bad_request:api", diff --git a/apps/researcher/lib/rag/ingest/safe-fetch.ts b/apps/researcher/lib/rag/ingest/safe-fetch.ts new file mode 100644 index 000000000..223fca4a8 --- /dev/null +++ b/apps/researcher/lib/rag/ingest/safe-fetch.ts @@ -0,0 +1,277 @@ +import { lookup } from "node:dns/promises"; +import { isIP } from "node:net"; + +import { ChatbotError } from "@/lib/errors"; + +// No "server-only" marker here, matching every other unit-tested module in lib: +// the import throws under `node --test`. The node:dns import above keeps this out +// of a client bundle regardless. + +// Outbound fetches for user-supplied source URLs. A file part in a chat request +// names a URL that the server then downloads, so without a destination check the +// request doubles as a probe of whatever the server can reach: loopback services, +// RFC1918 neighbours, and the cloud metadata endpoint on 169.254.169.254. +// +// extractUrlsFromText has a prefix-matching denylist for URLs found in chat text, +// but it only recognises literal 127.0.0.1 and friends at the very start of the +// string. That misses userinfo (http://x@127.0.0.1), IPv6, 127.x outside .0.1, +// hostnames that resolve into a private range, and redirects. This module resolves +// the host and checks every address instead. + +const MAX_REDIRECTS = 5; + +// [network, prefix length]. Everything a request has no business reaching from a +// URL a user typed: loopback, the private ranges, link-local (which carries the +// metadata service), plus the unspecified, multicast, and reserved blocks. +const BLOCKED_V4_RANGES: [string, number][] = [ + ["0.0.0.0", 8], + ["10.0.0.0", 8], + ["100.64.0.0", 10], + ["127.0.0.0", 8], + ["169.254.0.0", 16], + ["172.16.0.0", 12], + ["192.0.0.0", 24], + ["192.168.0.0", 16], + ["198.18.0.0", 15], + ["224.0.0.0", 4], + ["240.0.0.0", 4], +]; + +const BLOCKED_V6_RANGES: [string, number][] = [ + ["::", 128], + ["::1", 128], + ["fc00::", 7], + ["fe80::", 10], + // NAT64 addresses carry an IPv4 destination in their low 32 bits, so they are a + // way back into the ranges above. + ["64:ff9b::", 96], +]; + +function ipv4ToBytes(address: string): number[] | null { + const groups = address.split("."); + + if (groups.length !== 4) { + return null; + } + + const bytes = groups.map((group) => + /^\d{1,3}$/.test(group) ? Number(group) : Number.NaN + ); + + return bytes.every((byte) => byte >= 0 && byte <= 255) ? bytes : null; +} + +function ipv6GroupsToBytes(part: string): number[] | null { + if (part === "") { + return []; + } + + const groups = part.split(":"); + const bytes: number[] = []; + + for (const [index, group] of groups.entries()) { + // A trailing dotted-quad (::ffff:127.0.0.1) stands for the last four bytes. + if (group.includes(".")) { + const embedded = index === groups.length - 1 ? ipv4ToBytes(group) : null; + + if (!embedded) { + return null; + } + bytes.push(...embedded); + continue; + } + + if (!/^[0-9a-f]{1,4}$/i.test(group)) { + return null; + } + const value = Number.parseInt(group, 16); + bytes.push(value >> 8, value & 0xff); + } + + return bytes; +} + +function ipv6ToBytes(address: string): number[] | null { + // Drop any zone id: fe80::1%eth0 addresses the same interface-local target. + const [plain] = address.split("%"); + const halves = plain.split("::"); + + if (halves.length > 2) { + return null; + } + + const head = ipv6GroupsToBytes(halves[0]); + const tail = halves.length === 2 ? ipv6GroupsToBytes(halves[1]) : []; + + if (!head || !tail) { + return null; + } + + if (halves.length === 1) { + return head.length === 16 ? head : null; + } + + const zeroes = 16 - head.length - tail.length; + + return zeroes < 0 + ? null + : [...head, ...new Array(zeroes).fill(0), ...tail]; +} + +function withinRange( + address: number[], + network: number[], + prefixLength: number +): boolean { + let remaining = prefixLength; + + for (let i = 0; i < address.length && remaining > 0; i++) { + const bits = Math.min(8, remaining); + const mask = (0xff << (8 - bits)) & 0xff; + + if ((address[i] & mask) !== (network[i] & mask)) { + return false; + } + remaining -= bits; + } + + return true; +} + +function isMappedIpv4(bytes: number[]): boolean { + return ( + bytes.slice(0, 10).every((byte) => byte === 0) && + bytes[10] === 0xff && + bytes[11] === 0xff + ); +} + +/** + * Whether an IP literal names something outside the public internet. Anything + * unparseable counts as blocked: a value this code cannot reason about must not + * be handed to fetch. + */ +export function isBlockedAddress(address: string): boolean { + const version = isIP(address); + + if (version === 4) { + const bytes = ipv4ToBytes(address); + + return bytes + ? BLOCKED_V4_RANGES.some(([network, prefix]) => + withinRange(bytes, ipv4ToBytes(network) as number[], prefix) + ) + : true; + } + + if (version === 6) { + const bytes = ipv6ToBytes(address); + + if (!bytes) { + return true; + } + if (isMappedIpv4(bytes)) { + return BLOCKED_V4_RANGES.some(([network, prefix]) => + withinRange(bytes.slice(12), ipv4ToBytes(network) as number[], prefix) + ); + } + + return BLOCKED_V6_RANGES.some(([network, prefix]) => + withinRange(bytes, ipv6ToBytes(network) as number[], prefix) + ); + } + + return true; +} + +/** + * Parse a user-supplied URL and confirm it names a public HTTP(S) destination. + * Hostnames are resolved and every returned address has to be public, so a name + * pointing at 127.0.0.1 is rejected as surely as the literal is. + */ +export async function assertPublicUrl(rawUrl: string): Promise { + let url: URL; + + try { + url = new URL(rawUrl); + } catch { + throw new ChatbotError("bad_request:api", "Invalid URL format"); + } + + if (url.protocol !== "http:" && url.protocol !== "https:") { + throw new ChatbotError( + "bad_request:api", + `Unsupported URL scheme: ${url.protocol.replace(":", "")}` + ); + } + + // URL keeps the brackets on an IPv6 host; isIP does not want them. + const host = url.hostname.replace(/^\[|\]$/g, ""); + + if (isIP(host)) { + if (isBlockedAddress(host)) { + throw new ChatbotError( + "bad_request:api", + "URL points at a private or reserved address" + ); + } + + return url; + } + + let addresses: { address: string }[]; + + try { + addresses = await lookup(host, { all: true }); + } catch { + throw new ChatbotError( + "bad_request:api", + `Could not resolve host: ${host}` + ); + } + + if ( + addresses.length === 0 || + addresses.some((entry) => isBlockedAddress(entry.address)) + ) { + throw new ChatbotError( + "bad_request:api", + "URL resolves to a private or reserved address" + ); + } + + return url; +} + +/** + * fetch for user-supplied URLs, with the destination checked before the request + * leaves and again at every redirect. Redirects are followed by hand because + * fetch's own following would skip the check on each new target. + * + * A host that answers with a public address and then a private one on the next + * resolution (DNS rebinding) is not covered; that needs the connection pinned to + * the address that was checked, which fetch does not expose. + */ +export async function fetchPublicUrl( + rawUrl: string, + init?: RequestInit +): Promise { + let target = rawUrl; + + for (let hop = 0; hop <= MAX_REDIRECTS; hop++) { + const url = await assertPublicUrl(target); + const response = await fetch(url, { ...init, redirect: "manual" }); + const location = response.headers.get("location"); + + if (response.status < 300 || response.status >= 400 || !location) { + return response; + } + + target = new URL(location, url).toString(); + } + + throw new ChatbotError( + "bad_request:api", + "Too many redirects while fetching the URL" + ); +} diff --git a/apps/researcher/lib/rag/ingest/safe-fetch.unit.test.ts b/apps/researcher/lib/rag/ingest/safe-fetch.unit.test.ts new file mode 100644 index 000000000..3bc57ed07 --- /dev/null +++ b/apps/researcher/lib/rag/ingest/safe-fetch.unit.test.ts @@ -0,0 +1,231 @@ +import assert from "node:assert/strict"; +import { readFileSync } from "node:fs"; +import { createServer } from "node:http"; +import { resolve } from "node:path"; +import test from "node:test"; +import { ChatbotError } from "@/lib/errors"; +import { assertPublicUrl, fetchPublicUrl, isBlockedAddress } from "./safe-fetch"; + +// Regression tests for issue #778: a PDF file part named a URL that the server +// downloaded with a bare fetch, so a chat request could make the server call +// loopback services, RFC1918 neighbours, or the metadata endpoint. The addresses +// below are the ones an SSRF probe reaches for, plus the encodings that walked +// straight through the old prefix-matching denylist in extractUrlsFromText. + +const BLOCKED = [ + // The address from the report. + "127.0.0.1", + // Loopback outside .0.1, which a prefix match on "127.0.0.1" misses. + "127.1.2.3", + "0.0.0.0", + // Cloud metadata, absent from the old denylist altogether. + "169.254.169.254", + "169.254.0.1", + "10.0.0.1", + "172.16.0.1", + "172.31.255.255", + "192.168.1.1", + "100.64.0.1", + "192.0.0.1", + "198.18.0.1", + "224.0.0.1", + "255.255.255.255", + // IPv6 loopback, in both the compressed and the written-out form. + "::1", + "0:0:0:0:0:0:0:1", + "::", + "fd00::1", + "fc00::1", + "fe80::1", + // Zone ids still name an interface-local target. + "fe80::1%eth0", + // IPv4-mapped and NAT64 forms both carry a blocked IPv4 destination. + "::ffff:127.0.0.1", + "::ffff:169.254.169.254", + "64:ff9b::7f00:1", +]; + +const ALLOWED = [ + "8.8.8.8", + "1.1.1.1", + "93.184.216.34", + "172.32.0.1", + "172.15.255.255", + "128.0.0.1", + "2606:4700:4700::1111", + "::ffff:8.8.8.8", +]; + +// ChatbotError puts the caller-facing detail in `cause` and leaves `message` as +// the generic copy for the error code, so assertions read `cause`. +async function rejectionOf( + call: () => Promise +): Promise { + try { + await call(); + } catch (error) { + assert.ok(error instanceof ChatbotError, `unexpected error type: ${error}`); + return error; + } + + throw new Error("expected the call to reject"); +} + +async function assertRejectedWith( + call: () => Promise, + detail: RegExp +): Promise { + const error = await rejectionOf(call); + + assert.equal(error.statusCode, 400); + assert.match(String(error.cause), detail); +} + +test("isBlockedAddress rejects loopback, private, link-local, and reserved addresses", () => { + for (const address of BLOCKED) { + assert.equal(isBlockedAddress(address), true, `expected blocked: ${address}`); + } +}); + +test("isBlockedAddress allows public addresses", () => { + for (const address of ALLOWED) { + assert.equal(isBlockedAddress(address), false, `expected allowed: ${address}`); + } +}); + +test("isBlockedAddress treats anything unparseable as blocked", () => { + for (const address of ["", "not-an-ip", "127.0.0.256", "1.2.3", "gg::1", "::1::2"]) { + assert.equal(isBlockedAddress(address), true, `expected blocked: ${address}`); + } +}); + +test("assertPublicUrl rejects the URL from the report", async () => { + await assertRejectedWith( + () => assertPublicUrl("http://127.0.0.1:9999/ssrf-proof-token-abc123"), + /private or reserved address/ + ); +}); + +test("assertPublicUrl rejects a loopback host hidden behind userinfo", async () => { + // The old denylist anchored on "http://127.0.0.1", so a username in front of + // the host was enough to slip past it. + await assertRejectedWith( + () => assertPublicUrl("http://user@127.0.0.1/admin"), + /private or reserved address/ + ); +}); + +test("assertPublicUrl rejects bracketed IPv6 loopback", async () => { + await assertRejectedWith( + () => assertPublicUrl("http://[::1]:9999/probe"), + /private or reserved address/ + ); +}); + +test("assertPublicUrl rejects the metadata endpoint", async () => { + await assertRejectedWith( + () => assertPublicUrl("http://169.254.169.254/latest/meta-data/"), + /private or reserved address/ + ); +}); + +test("assertPublicUrl rejects a hostname that resolves to loopback", async () => { + // localhost is a name, not a literal, so only resolution catches it. + await assertRejectedWith( + () => assertPublicUrl("http://localhost:3000/probe"), + /private or reserved address/ + ); +}); + +test("assertPublicUrl rejects non-HTTP schemes", async () => { + for (const url of [ + "file:///etc/passwd", + "ftp://example.com/x", + "gopher://example.com/", + ]) { + await assertRejectedWith(() => assertPublicUrl(url), /Unsupported URL scheme/); + } +}); + +test("assertPublicUrl rejects a malformed URL", async () => { + await assertRejectedWith( + () => assertPublicUrl("not a url"), + /Invalid URL format/ + ); +}); + +test("assertPublicUrl accepts a public literal without touching DNS", async () => { + const url = await assertPublicUrl("https://8.8.8.8/file.pdf"); + + assert.equal(url.hostname, "8.8.8.8"); + assert.equal(url.pathname, "/file.pdf"); +}); + +test("fetchPublicUrl sends nothing to a loopback listener", async () => { + // The report proved the bug by watching a listener log "CAPTURED REQUEST". + // Asserting on the listener, rather than only on the rejection, is what shows + // the request is refused before it leaves rather than after. + const captured: string[] = []; + const server = createServer((request, response) => { + captured.push(request.url ?? ""); + response.end("ok"); + }); + + await new Promise((resolve) => { + server.listen(0, "127.0.0.1", resolve); + }); + + const address = server.address(); + assert.ok(address && typeof address === "object"); + + try { + await assertRejectedWith( + () => fetchPublicUrl(`http://127.0.0.1:${address.port}/ssrf-proof-token`), + /private or reserved address/ + ); + assert.deepEqual(captured, []); + } finally { + await new Promise((resolve) => { + server.close(() => resolve()); + }); + } +}); + +// The guard only helps where it is wired in, and the module that had the bug +// cannot be imported here: it pulls in chunking.ts, whose "server-only" import +// throws under node --test. Reading the source is what pins the call site, so +// restoring the bare fetch fails a test rather than passing quietly. +test("ingest downloads the PDF through the guard rather than bare fetch", () => { + const ingest = readFileSync(resolve("lib/rag/ingest/index.ts"), "utf8"); + + assert.match(ingest, /await fetchPublicUrl\(source\.fileUrl\)/); + assert.doesNotMatch(ingest, /await fetch\(/); +}); + +test("fetchPublicUrl refuses a redirect into a blocked range", async () => { + // fetch would follow a redirect itself, skipping the check on the new target, + // so fetchPublicUrl follows by hand and re-checks each hop. Stubbing fetch is + // the only way to stage a public first hop from a test. + const originalFetch = globalThis.fetch; + const requested: string[] = []; + + globalThis.fetch = (async (input: string | URL | Request) => { + requested.push(String(input)); + + return new Response(null, { + status: 302, + headers: { location: "http://169.254.169.254/latest/meta-data/" }, + }); + }) as typeof fetch; + + try { + await assertRejectedWith( + () => fetchPublicUrl("https://8.8.8.8/file.pdf"), + /private or reserved address/ + ); + // Only the first, allowed hop was ever requested. + assert.deepEqual(requested, ["https://8.8.8.8/file.pdf"]); + } finally { + globalThis.fetch = originalFetch; + } +}); From aaab599cd54ba0352e01a027e3de5dca7dafc9b3 Mon Sep 17 00:00:00 2001 From: HoangDucBach Date: Sat, 29 Aug 2026 22:17:22 +0700 Subject: [PATCH 09/29] fix(researcher): validate AUTH_SECRET before signing or verifying a session (#781) session.ts and proxy.ts encoded process.env.AUTH_SECRET as it came, while enoki-challenge.ts in the same folder already refused anything under 32 characters for that same variable. Two silent failures followed. A short placeholder is brute-forceable offline from one captured cookie. Worse, an unset variable encodes to zero bytes, and jose signs and verifies HS256 with an empty key without complaint, so a deployment missing the variable issued and accepted cookies signed with a key anyone can reproduce; forging a session for any user id needed no secret. .env.example ships AUTH_SECRET= empty and the runtime image does not inherit the builder placeholder, so that is the state a missed variable actually lands in. Add getAuthSecret and getAuthSecretKey, use them in session.ts and proxy.ts, and drop the local copy in enoki-challenge.ts so one rule covers every caller. Validation stays inside the functions: doing it at module scope would run during next build, where the Dockerfile supplies a 22-character placeholder, and would fail the image build rather than a request. proxy.ts reads the secret whether or not a cookie arrived, so a misconfigured deployment fails loudly instead of serving the login page as though the caller were merely signed out. --- apps/researcher/lib/auth/auth-secret.ts | 36 +++++ .../lib/auth/auth-secret.unit.test.ts | 125 ++++++++++++++++++ apps/researcher/lib/auth/enoki-challenge.ts | 13 +- apps/researcher/lib/auth/session.ts | 6 +- apps/researcher/proxy.ts | 7 +- 5 files changed, 172 insertions(+), 15 deletions(-) create mode 100644 apps/researcher/lib/auth/auth-secret.ts create mode 100644 apps/researcher/lib/auth/auth-secret.unit.test.ts diff --git a/apps/researcher/lib/auth/auth-secret.ts b/apps/researcher/lib/auth/auth-secret.ts new file mode 100644 index 000000000..d3fee48aa --- /dev/null +++ b/apps/researcher/lib/auth/auth-secret.ts @@ -0,0 +1,36 @@ +// One rule for AUTH_SECRET across the app. session.ts and proxy.ts used to read +// the variable raw, which had two silent failure modes: +// +// - unset: `new TextEncoder().encode(undefined)` returns zero bytes, and jose +// signs and verifies HS256 with an empty key without complaining. The app +// then hands out and accepts cookies signed with a key every attacker +// already has, so forging a session for any user id takes no secret at all. +// - too short: a placeholder is brute-forceable offline from one captured +// cookie. +// +// enoki-challenge.ts already refused anything under 32 characters for the same +// variable, so the session path was the odd one out. + +const MIN_AUTH_SECRET_LENGTH = 32; + +/** + * The validated AUTH_SECRET. Reading is deliberately lazy: validating at module + * scope would run during `next build`, where the Dockerfile supplies a short + * build-time placeholder, and fail the image build rather than a request. + */ +export function getAuthSecret(): string { + const value = process.env.AUTH_SECRET; + + if (!value || value.length < MIN_AUTH_SECRET_LENGTH) { + throw new Error( + `AUTH_SECRET must contain at least ${MIN_AUTH_SECRET_LENGTH} characters` + ); + } + + return value; +} + +/** The validated secret as a signing key, for jose. */ +export function getAuthSecretKey(): Uint8Array { + return new TextEncoder().encode(getAuthSecret()); +} diff --git a/apps/researcher/lib/auth/auth-secret.unit.test.ts b/apps/researcher/lib/auth/auth-secret.unit.test.ts new file mode 100644 index 000000000..ff90f568e --- /dev/null +++ b/apps/researcher/lib/auth/auth-secret.unit.test.ts @@ -0,0 +1,125 @@ +import assert from "node:assert/strict"; +import { readFileSync } from "node:fs"; +import { resolve } from "node:path"; +import test from "node:test"; +import { SignJWT, jwtVerify } from "jose"; +import { getAuthSecret, getAuthSecretKey } from "./auth-secret"; + +// Regression tests for issue #781: session.ts and proxy.ts signed and verified +// session cookies with the raw AUTH_SECRET value, while enoki-challenge.ts in the +// same folder already required at least 32 characters for that same variable. + +const THIRTY_TWO = "0123456789012345678901234567890a"; + +function withSecret(value: string | undefined, body: () => T): T { + const previous = process.env.AUTH_SECRET; + + if (value === undefined) { + delete process.env.AUTH_SECRET; + } else { + process.env.AUTH_SECRET = value; + } + + try { + return body(); + } finally { + if (previous === undefined) { + delete process.env.AUTH_SECRET; + } else { + process.env.AUTH_SECRET = previous; + } + } +} + +test("getAuthSecret rejects a missing secret", () => { + withSecret(undefined, () => { + assert.throws(getAuthSecret, /at least 32 characters/); + }); +}); + +test("getAuthSecret rejects an empty secret", () => { + // .env.example ships AUTH_SECRET= with no value, so this is the shape a + // copied config actually has. + withSecret("", () => { + assert.throws(getAuthSecret, /at least 32 characters/); + }); +}); + +test("getAuthSecret rejects a secret one character short", () => { + withSecret("0123456789012345678901234567890", () => { + assert.throws(getAuthSecret, /at least 32 characters/); + }); +}); + +test("getAuthSecret rejects the short placeholder from the report", () => { + withSecret("short", () => { + assert.throws(getAuthSecret, /at least 32 characters/); + }); +}); + +test("getAuthSecret accepts a secret at the minimum length", () => { + withSecret(THIRTY_TWO, () => { + assert.equal(getAuthSecret().length, 32); + assert.equal(getAuthSecret(), THIRTY_TWO); + }); +}); + +test("getAuthSecretKey encodes the validated secret", () => { + withSecret(THIRTY_TWO, () => { + assert.deepEqual( + getAuthSecretKey(), + new TextEncoder().encode(THIRTY_TWO) + ); + }); +}); + +test("getAuthSecretKey refuses the empty key that an unset secret used to produce", async () => { + // Why the guard exists at all: encoding an absent value yields zero bytes, and + // jose will sign and verify HS256 with that key. A deployment missing the + // variable therefore issued and accepted cookies signed with a key anyone can + // reproduce, so forging a session for any user id needed no secret. + const emptyKey = new TextEncoder().encode(process.env.AUTH_SECRET_UNSET); + assert.equal(emptyKey.length, 0); + + const forged = await new SignJWT({ userId: "victim" }) + .setProtectedHeader({ alg: "HS256" }) + .setExpirationTime("1h") + .sign(emptyKey); + const { payload } = await jwtVerify(forged, emptyKey); + assert.equal(payload.userId, "victim"); + + withSecret(undefined, () => { + assert.throws(getAuthSecretKey, /at least 32 characters/); + }); +}); + +// The guard only helps where it is wired in. These modules cannot be imported +// here (next/headers, the db client), so their source is what pins the call +// sites: reintroducing a raw read fails a test rather than passing quietly. +for (const path of ["lib/auth/session.ts", "proxy.ts"]) { + test(`${path} reads the secret through the guard`, () => { + const source = readFileSync(resolve(path), "utf8"); + + assert.match(source, /getAuthSecretKey\(\)/); + assert.doesNotMatch(source, /process\.env\.AUTH_SECRET/); + }); +} + +test("enoki-challenge.ts shares the one guard rather than its own copy", () => { + const source = readFileSync(resolve("lib/auth/enoki-challenge.ts"), "utf8"); + + assert.match(source, /getAuthSecretKey\(\)/); + assert.doesNotMatch(source, /process\.env\.AUTH_SECRET/); +}); + +test("the guard stays lazy so a build-time placeholder cannot break the image", () => { + // The Dockerfile builder stage sets a 22-character AUTH_SECRET so `next build` + // can prerender. Validating at module scope would fail the build instead of a + // request, so no module may compute the key while being imported. + for (const path of ["lib/auth/session.ts", "proxy.ts"]) { + const source = readFileSync(resolve(path), "utf8"); + const moduleScopeCall = /^const \w+ = getAuthSecretKey\(\)/m; + + assert.doesNotMatch(source, moduleScopeCall); + } +}); diff --git a/apps/researcher/lib/auth/enoki-challenge.ts b/apps/researcher/lib/auth/enoki-challenge.ts index a9ab4f6d5..6ee16ae95 100644 --- a/apps/researcher/lib/auth/enoki-challenge.ts +++ b/apps/researcher/lib/auth/enoki-challenge.ts @@ -5,6 +5,7 @@ import { normalizeSuiAddress } from "@mysten/sui/utils"; import { verifyPersonalMessageSignature } from "@mysten/sui/verify"; import { jwtVerify, SignJWT } from "jose"; import { cookies } from "next/headers"; +import { getAuthSecretKey } from "@/lib/auth/auth-secret"; import { requireSharedRedisClient, SharedRedisUnavailableError, @@ -16,14 +17,6 @@ const CHALLENGE_REDIS_PREFIX = "enoki-auth:challenge:"; type SuiNetwork = "mainnet" | "testnet"; -function getSecret(): Uint8Array { - const value = process.env.AUTH_SECRET; - if (!value || value.length < 32) { - throw new Error("AUTH_SECRET must contain at least 32 characters"); - } - return new TextEncoder().encode(value); -} - function getNetwork(): SuiNetwork { return process.env.NEXT_PUBLIC_SUI_NETWORK === "mainnet" ? "mainnet" @@ -56,7 +49,7 @@ export async function issueEnokiChallenge(rawAddress: string): Promise { .setJti(nonce) .setIssuedAt() .setExpirationTime(`${CHALLENGE_TTL_SECONDS}s`) - .sign(getSecret()); + .sign(getAuthSecretKey()); try { const redis = await requireSharedRedisClient(); @@ -111,7 +104,7 @@ export async function verifyAndConsumeEnokiChallenge({ try { const address = normalizeSuiAddress(rawAddress); - const { payload } = await jwtVerify(token, getSecret(), { + const { payload } = await jwtVerify(token, getAuthSecretKey(), { algorithms: ["HS256"], audience: "enoki-auth", issuer: "walrus-memory-researcher", diff --git a/apps/researcher/lib/auth/session.ts b/apps/researcher/lib/auth/session.ts index 9e8c7d173..001b389a9 100644 --- a/apps/researcher/lib/auth/session.ts +++ b/apps/researcher/lib/auth/session.ts @@ -3,13 +3,13 @@ import "server-only"; import { jwtVerify } from "jose"; import { cookies } from "next/headers"; import { getUserById } from "@/lib/db/queries"; +import { getAuthSecretKey } from "@/lib/auth/auth-secret"; import { SESSION_MAX_AGE_SECONDS, signSessionIdentity, } from "@/lib/auth/session-token"; const COOKIE_NAME = "session"; -const secret = new TextEncoder().encode(process.env.AUTH_SECRET); type SessionUser = { id: string; @@ -28,7 +28,7 @@ export async function getSession(): Promise<{ user: SessionUser } | null> { if (!token) return null; try { - const { payload } = await jwtVerify(token, secret); + const { payload } = await jwtVerify(token, getAuthSecretKey()); if (typeof payload.userId !== "string") return null; const user = await getUserById(payload.userId); @@ -55,7 +55,7 @@ export async function createSession( ): Promise { const token = await signSessionIdentity( { userId, publicKey, accountId }, - secret + getAuthSecretKey() ); const cookieStore = await cookies(); cookieStore.set(COOKIE_NAME, token, { diff --git a/apps/researcher/proxy.ts b/apps/researcher/proxy.ts index 7f8c1c34d..4dacacf76 100644 --- a/apps/researcher/proxy.ts +++ b/apps/researcher/proxy.ts @@ -1,9 +1,8 @@ import { type NextRequest, NextResponse } from "next/server"; import { jwtVerify } from "jose"; +import { getAuthSecretKey } from "@/lib/auth/auth-secret"; import { isTestEnvironment } from "@/lib/constants"; -const secret = new TextEncoder().encode(process.env.AUTH_SECRET); - export async function proxy(request: NextRequest) { const { pathname } = request.nextUrl; @@ -21,6 +20,10 @@ export async function proxy(request: NextRequest) { return NextResponse.next(); } + // Read the secret whether or not a cookie came with the request. Deferring it + // until a token shows up would let a deployment with no AUTH_SECRET serve the + // login page as if nothing were wrong. + const secret = getAuthSecretKey(); const token = request.cookies.get("session")?.value; let isAuthenticated = false; From ee01813367771dc7221b38c0861005468ef3637f Mon Sep 17 00:00:00 2001 From: HoangDucBach Date: Thu, 3 Sep 2026 14:40:51 +0700 Subject: [PATCH 10/29] fix(chatbot): treat a malformed Bearer header as no session (WALM-416) getToken() URL-decodes Authorization before its decode try/catch, so Bearer %% threw URIError and became HTTP 500 on proxy and guest login. --- .../app/(auth)/api/auth/guest/route.ts | 9 +--- apps/chatbot/lib/session-token.ts | 23 +++++++++ apps/chatbot/lib/session-token.unit.test.ts | 47 +++++++++++++++++++ apps/chatbot/proxy.ts | 10 ++-- 4 files changed, 75 insertions(+), 14 deletions(-) create mode 100644 apps/chatbot/lib/session-token.ts create mode 100644 apps/chatbot/lib/session-token.unit.test.ts diff --git a/apps/chatbot/app/(auth)/api/auth/guest/route.ts b/apps/chatbot/app/(auth)/api/auth/guest/route.ts index 03f4ac083..682f6542b 100644 --- a/apps/chatbot/app/(auth)/api/auth/guest/route.ts +++ b/apps/chatbot/app/(auth)/api/auth/guest/route.ts @@ -1,7 +1,6 @@ import { NextResponse } from "next/server"; -import { getToken } from "next-auth/jwt"; import { signIn } from "@/app/(auth)/auth"; -import { isDevelopmentEnvironment } from "@/lib/constants"; +import { getSessionToken } from "@/lib/session-token"; /** * Validate a redirect target before forwarding to auth. @@ -35,11 +34,7 @@ export async function GET(request: Request) { ? rawRedirectUrl : "/"; - const token = await getToken({ - req: request, - secret: process.env.AUTH_SECRET, - secureCookie: !isDevelopmentEnvironment, - }); + const token = await getSessionToken(request); if (token) { return NextResponse.redirect(new URL("/", request.url)); diff --git a/apps/chatbot/lib/session-token.ts b/apps/chatbot/lib/session-token.ts new file mode 100644 index 000000000..17092e1bf --- /dev/null +++ b/apps/chatbot/lib/session-token.ts @@ -0,0 +1,23 @@ +import { getToken } from "next-auth/jwt"; +import { isDevelopmentEnvironment } from "@/lib/constants"; + +/** + * Read the Auth.js session JWT. A malformed `Authorization: Bearer` value + * (`%%`, lone `%`, …) makes `getToken` throw `URIError` from + * `decodeURIComponent` *before* its inner decode try/catch — that was HTTP 500 + * on proxy and `/api/auth/guest`. Treat it as no session. + */ +export async function getSessionToken(request: Request) { + try { + return await getToken({ + req: request, + secret: process.env.AUTH_SECRET, + secureCookie: !isDevelopmentEnvironment, + }); + } catch (error) { + if (error instanceof URIError) { + return null; + } + throw error; + } +} diff --git a/apps/chatbot/lib/session-token.unit.test.ts b/apps/chatbot/lib/session-token.unit.test.ts new file mode 100644 index 000000000..2823cb7cd --- /dev/null +++ b/apps/chatbot/lib/session-token.unit.test.ts @@ -0,0 +1,47 @@ +import { readFileSync } from "node:fs"; +import { dirname, join } from "node:path"; +import { fileURLToPath } from "node:url"; +import { getToken } from "next-auth/jwt"; +import { describe, expect, it } from "vitest"; +import { getSessionToken } from "@/lib/session-token"; + +function requestWithBearer(value: string): Request { + return new Request("http://127.0.0.1:3001/api/chat", { + headers: { Authorization: `Bearer ${value}` }, + }); +} + +describe("getSessionToken", () => { + it("returns null for a malformed Bearer value instead of throwing", async () => { + const request = requestWithBearer("%%"); + + await expect( + getToken({ req: request, secret: "unit-test-secret-not-for-production" }) + ).rejects.toBeInstanceOf(URIError); + + await expect(getSessionToken(request)).resolves.toBeNull(); + }); + + it("returns null when there is no session cookie or Bearer header", async () => { + await expect( + getSessionToken(new Request("http://127.0.0.1:3001/api/chat")) + ).resolves.toBeNull(); + }); +}); + +describe("malformed Bearer call sites", () => { + const here = dirname(fileURLToPath(import.meta.url)); + + it("proxy and guest route read the session through the wrapper", () => { + const proxy = readFileSync(join(here, "../proxy.ts"), "utf8"); + const guest = readFileSync( + join(here, "../app/(auth)/api/auth/guest/route.ts"), + "utf8" + ); + + expect(proxy).toContain("getSessionToken"); + expect(proxy).not.toContain("getToken("); + expect(guest).toContain("getSessionToken"); + expect(guest).not.toContain("getToken("); + }); +}); diff --git a/apps/chatbot/proxy.ts b/apps/chatbot/proxy.ts index ca5a19dda..ccad67efb 100644 --- a/apps/chatbot/proxy.ts +++ b/apps/chatbot/proxy.ts @@ -1,6 +1,6 @@ import { type NextRequest, NextResponse } from "next/server"; -import { getToken } from "next-auth/jwt"; -import { guestRegex, isDevelopmentEnvironment } from "./lib/constants"; +import { guestRegex } from "./lib/constants"; +import { getSessionToken } from "./lib/session-token"; export async function proxy(request: NextRequest) { const { pathname } = request.nextUrl; @@ -17,11 +17,7 @@ export async function proxy(request: NextRequest) { return NextResponse.next(); } - const token = await getToken({ - req: request, - secret: process.env.AUTH_SECRET, - secureCookie: !isDevelopmentEnvironment, - }); + const token = await getSessionToken(request); if (!token) { const redirectUrl = encodeURIComponent(request.url); From 34d5cd1e47d5d21479e9df909d3560f12b7f762d Mon Sep 17 00:00:00 2001 From: ducnmm <165614309+ducnmm@users.noreply.github.com> Date: Fri, 4 Sep 2026 05:10:43 -0700 Subject: [PATCH 11/29] docs: fix style-guide nits blocking staging promotion (WALM-460) (#853) --- docs/guides/system-prompt-templates.md | 2 +- docs/sdk/api-reference.md | 4 ++-- docs/troubleshooting/overview.md | 2 +- 3 files changed, 4 insertions(+), 4 deletions(-) diff --git a/docs/guides/system-prompt-templates.md b/docs/guides/system-prompt-templates.md index 5ccc4731c..6cebf1645 100644 --- a/docs/guides/system-prompt-templates.md +++ b/docs/guides/system-prompt-templates.md @@ -223,7 +223,7 @@ Four rules carry most of the effect: 1. **Name the trigger, not the goal.** "Call memwal_remember when the user states a preference" works. "Remember important things" does not. 2. **Say "in the same turn, before you finish replying."** Without it, agents acknowledge the fact in prose and never call the tool. -3. **Say what to skip.** An agent told only to write produces noise, and noisy memory makes recall worse. +3. **Say what to skip.** An agent told only to write writes noise, and noisy memory makes recall worse. 4. **Pin the namespace.** Unscoped writes land in the default namespace and mix contexts that should stay apart. See [Memory space](/fundamentals/concepts/memory-space). ## Related diff --git a/docs/sdk/api-reference.md b/docs/sdk/api-reference.md index c4e5ae707..34ec274f9 100644 --- a/docs/sdk/api-reference.md +++ b/docs/sdk/api-reference.md @@ -131,9 +131,9 @@ Search for memories matching a natural language query, scoped to `owner + namesp } ``` -`distance` is cosine distance (lower is more similar). +`distance` is cosine distance. Lower is more similar. -MCP `memwal_recall` displays `score = 1 - distance` (higher = more similar). Do not apply an SDK `maxDistance` threshold to those scores; the polarities are inverted. +MCP `memwal_recall` displays `score = 1 - distance` (higher = more similar). Do not apply an SDK `maxDistance` threshold to those scores. The polarities are inverted. `created_at` is when the fact was **written**, not any date its text describes. diff --git a/docs/troubleshooting/overview.md b/docs/troubleshooting/overview.md index 96526be6a..d49861e04 100644 --- a/docs/troubleshooting/overview.md +++ b/docs/troubleshooting/overview.md @@ -162,7 +162,7 @@ Recall is one search request. A save embeds, encrypts, uploads to Walrus, and in ### Score vs distance -SDK `recall` returns cosine **distance** (lower = more similar), and `maxDistance` drops hits where `distance >= maxDistance`. MCP `memwal_recall` prints **score** as `1 - distance` (higher = more similar). Do not apply an SDK `maxDistance` to MCP scores; that inverts the filter. +SDK `recall` returns cosine **distance** (lower = more similar), and `maxDistance` drops hits where `distance >= maxDistance`. MCP `memwal_recall` prints **score** as `1 - distance` (higher = more similar). Do not apply an SDK `maxDistance` to MCP scores. That inverts the filter. ### How do I check whether the service is reachable? From c7f6261eec23f66339ecbaa523023f88c4d9b122 Mon Sep 17 00:00:00 2001 From: ducnmm <165614309+ducnmm@users.noreply.github.com> Date: Sun, 6 Sep 2026 17:11:42 -0700 Subject: [PATCH 12/29] fix(chatbot): scope getSuggestions to the calling user (WALM-438) (#824) --- .../app/(chat)/api/suggestions/route.ts | 18 ++--- apps/chatbot/artifacts/actions.ts | 21 ++++- .../get-suggestions-ownership.unit.test.ts | 79 +++++++++++++++++++ apps/chatbot/lib/db/queries.ts | 14 +++- .../db/suggestion-query-scope.unit.test.ts | 54 +++++++++++++ 5 files changed, 171 insertions(+), 15 deletions(-) create mode 100644 apps/chatbot/artifacts/get-suggestions-ownership.unit.test.ts create mode 100644 apps/chatbot/lib/db/suggestion-query-scope.unit.test.ts diff --git a/apps/chatbot/app/(chat)/api/suggestions/route.ts b/apps/chatbot/app/(chat)/api/suggestions/route.ts index 303f45ed2..1162ae0bc 100644 --- a/apps/chatbot/app/(chat)/api/suggestions/route.ts +++ b/apps/chatbot/app/(chat)/api/suggestions/route.ts @@ -1,5 +1,5 @@ import { auth } from "@/app/(auth)/auth"; -import { getSuggestionsByDocumentId } from "@/lib/db/queries"; +import { getSuggestionsByDocumentIdForUser } from "@/lib/db/queries"; import { ChatbotError } from "@/lib/errors"; export async function GET(request: Request) { @@ -14,22 +14,20 @@ export async function GET(request: Request) { } const session = await auth(); + const userId = session?.user?.id; - if (!session?.user) { + if (!userId) { return new ChatbotError("unauthorized:suggestions").toResponse(); } - const suggestions = await getSuggestionsByDocumentId({ + const suggestions = await getSuggestionsByDocumentIdForUser({ documentId, + userId, }); - const [suggestion] = suggestions; - - if (!suggestion) { - return Response.json([], { status: 200 }); - } - - if (suggestion.userId !== session.user.id) { + // Owner-scoped lookup already dropped other users' rows. Re-assert so a + // future query refactor cannot leak suggestion text through this route. + if (suggestions.some((suggestion) => suggestion.userId !== userId)) { return new ChatbotError("forbidden:api").toResponse(); } diff --git a/apps/chatbot/artifacts/actions.ts b/apps/chatbot/artifacts/actions.ts index 2000ca111..add108af0 100644 --- a/apps/chatbot/artifacts/actions.ts +++ b/apps/chatbot/artifacts/actions.ts @@ -1,8 +1,23 @@ "use server"; -import { getSuggestionsByDocumentId } from "@/lib/db/queries"; +import { auth } from "@/app/(auth)/auth"; +import { getSuggestionsByDocumentIdForUser } from "@/lib/db/queries"; export async function getSuggestions({ documentId }: { documentId: string }) { - const suggestions = await getSuggestionsByDocumentId({ documentId }); - return suggestions ?? []; + const session = await auth(); + const userId = session?.user?.id; + // No authenticated user id → same empty shape as a missing document, so + // existence of another user's suggestions is not disclosed. + if (!userId) { + return []; + } + + const suggestions = await getSuggestionsByDocumentIdForUser({ + documentId, + userId, + }); + // Defense in depth: even though the lookup is owner-scoped, drop any row + // that somehow isn't the caller's so a future query refactor cannot + // silently reopen the IDOR. + return (suggestions ?? []).filter((row) => row.userId === userId); } diff --git a/apps/chatbot/artifacts/get-suggestions-ownership.unit.test.ts b/apps/chatbot/artifacts/get-suggestions-ownership.unit.test.ts new file mode 100644 index 000000000..ec65d7ce6 --- /dev/null +++ b/apps/chatbot/artifacts/get-suggestions-ownership.unit.test.ts @@ -0,0 +1,79 @@ +import { beforeEach, describe, expect, it, vi } from "vitest"; + +// Regression test for the getSuggestions IDOR (WALM-438 / GH #786). +// The server action used to call an unscoped by-documentId lookup with no +// session check, so guest B could read guest A's suggestion text. The HTTP +// route already rejected that; the action did not. +// +// Two layers, both proven here: +// 1. Auth guard — no session / no user id returns [] and never queries. +// 2. Call-site re-check — even if the DB layer returns a victim row, the +// action strips it so existence/text is not disclosed. + +const OWNER_ID = "11111111-1111-1111-1111-111111111111"; +const ATTACKER_ID = "22222222-2222-2222-2222-222222222222"; +const DOC_ID = "33333333-3333-3333-3333-333333333333"; + +const ownerRow = { + documentId: DOC_ID, + originalText: "secret draft paragraph", + suggestedText: "rewritten secret draft", + userId: OWNER_ID, +}; + +const auth = vi.fn(); +const getSuggestionsByDocumentIdForUser = vi.fn(); + +vi.mock("@/app/(auth)/auth", () => ({ auth })); +vi.mock("@/lib/db/queries", () => ({ + getSuggestionsByDocumentIdForUser, +})); + +beforeEach(() => { + vi.clearAllMocks(); + getSuggestionsByDocumentIdForUser.mockResolvedValue([]); +}); + +describe("getSuggestions ownership", () => { + it("returns [] and does not query when there is no session", async () => { + auth.mockResolvedValue(null); + const { getSuggestions } = await import("./actions"); + + await expect(getSuggestions({ documentId: DOC_ID })).resolves.toEqual([]); + expect(getSuggestionsByDocumentIdForUser).not.toHaveBeenCalled(); + }); + + it("returns [] and does not query when the session has no user id", async () => { + auth.mockResolvedValue({ user: {}, expires: "" }); + const { getSuggestions } = await import("./actions"); + + await expect(getSuggestions({ documentId: DOC_ID })).resolves.toEqual([]); + expect(getSuggestionsByDocumentIdForUser).not.toHaveBeenCalled(); + }); + + it("returns the caller's rows", async () => { + auth.mockResolvedValue({ user: { id: OWNER_ID }, expires: "" }); + getSuggestionsByDocumentIdForUser.mockResolvedValue([ownerRow]); + const { getSuggestions } = await import("./actions"); + + await expect(getSuggestions({ documentId: DOC_ID })).resolves.toEqual([ + ownerRow, + ]); + expect(getSuggestionsByDocumentIdForUser).toHaveBeenCalledWith({ + documentId: DOC_ID, + userId: OWNER_ID, + }); + }); + + it("does not return another user's suggestion text even if the query leaks it", async () => { + auth.mockResolvedValue({ user: { id: ATTACKER_ID }, expires: "" }); + getSuggestionsByDocumentIdForUser.mockResolvedValue([ownerRow]); + const { getSuggestions } = await import("./actions"); + + await expect(getSuggestions({ documentId: DOC_ID })).resolves.toEqual([]); + expect(getSuggestionsByDocumentIdForUser).toHaveBeenCalledWith({ + documentId: DOC_ID, + userId: ATTACKER_ID, + }); + }); +}); diff --git a/apps/chatbot/lib/db/queries.ts b/apps/chatbot/lib/db/queries.ts index 3b86dc4b7..723c593f4 100644 --- a/apps/chatbot/lib/db/queries.ts +++ b/apps/chatbot/lib/db/queries.ts @@ -443,16 +443,26 @@ export async function saveSuggestions({ } } -export async function getSuggestionsByDocumentId({ +// Owner-scoped lookup: filters on both documentId AND userId so a caller can +// never resolve another user's suggestion text. An unscoped by-documentId +// lookup was the footgun behind the getSuggestions IDOR (WALM-438 / GH #786). +export async function getSuggestionsByDocumentIdForUser({ documentId, + userId, }: { documentId: string; + userId: string; }) { try { return await db .select() .from(suggestion) - .where(eq(suggestion.documentId, documentId)); + .where( + and( + eq(suggestion.documentId, documentId), + eq(suggestion.userId, userId) + ) + ); } catch (_error) { throw new ChatbotError( "bad_request:database", diff --git a/apps/chatbot/lib/db/suggestion-query-scope.unit.test.ts b/apps/chatbot/lib/db/suggestion-query-scope.unit.test.ts new file mode 100644 index 000000000..aa2654694 --- /dev/null +++ b/apps/chatbot/lib/db/suggestion-query-scope.unit.test.ts @@ -0,0 +1,54 @@ +import { PgDialect } from "drizzle-orm/pg-core"; +import { beforeEach, describe, expect, it, vi } from "vitest"; + +// Regression test for the DB-layer owner scoping of +// getSuggestionsByDocumentIdForUser (WALM-438 / GH #786). +// +// The action-level suite mocks the query module, so it never executes the +// real SQL. This suite runs the REAL query body against a mocked drizzle +// builder, captures the WHERE clause, and asserts it filters on BOTH +// documentId AND userId. Dropping the userId predicate reopens the IDOR. + +const DOC_VAL = "44444444-4444-4444-4444-444444444444"; +const USER_VAL = "55555555-5555-5555-5555-555555555555"; + +let capturedWhere: unknown; + +const where = vi.fn((clause: unknown) => { + capturedWhere = clause; + return Promise.resolve([]); +}); +const from = vi.fn(() => ({ where })); +const select = vi.fn(() => ({ from })); + +vi.mock("server-only", () => ({})); +vi.mock("postgres", () => ({ default: () => ({}) })); +vi.mock("drizzle-orm/postgres-js", () => ({ + drizzle: () => ({ select }), +})); + +beforeEach(() => { + vi.clearAllMocks(); + capturedWhere = undefined; +}); + +describe("getSuggestionsByDocumentIdForUser DB-layer scoping", () => { + it("filters on both documentId and the owner userId", async () => { + const { getSuggestionsByDocumentIdForUser } = await import("./queries"); + + await getSuggestionsByDocumentIdForUser({ + documentId: DOC_VAL, + userId: USER_VAL, + }); + + expect(where).toHaveBeenCalledTimes(1); + expect(capturedWhere).toBeDefined(); + + const { sql, params } = new PgDialect().sqlToQuery(capturedWhere as never); + + expect(sql).toMatch(/"documentId"/); + expect(sql).toMatch(/"userId"/); + expect(params).toContain(DOC_VAL); + expect(params).toContain(USER_VAL); + }); +}); From 692b5c730d21bf9e609e2db39822ebf4a41af85f Mon Sep 17 00:00:00 2001 From: ducnmm <165614309+ducnmm@users.noreply.github.com> Date: Sun, 6 Sep 2026 18:13:59 -0700 Subject: [PATCH 13/29] feat(mcp): expose maxDistance on memwal_recall (WALM-457) (#850) * feat(mcp): expose maxDistance on memwal_recall (WALM-457) * fix(mcp): do not report maxDistance misses as decrypt failures (WALM-457) * fix(mcp): address review on maxDistance recall cutoff (WALM-457) Undecrypted matches never had a distance computed, so "All matching memories were outside maxDistance." claimed more than the code knew when dropped_count was also non-zero: those hits may well have been inside the cutoff. Report both counts instead of letting the cutoff wording imply the namespace held nothing relevant. Also: - Reject a negative cutoff at both schema boundaries (zod .nonnegative(), JSON Schema minimum: 0). maxDistance: -1 silently dropped every hit and read as an empty namespace. - Document that the cutoff runs on the `limit` nearest matches, so a tight maxDistance returns fewer than `limit` rows. - Refresh the overview line from #849, which predated printing `distance` on the result line alongside `score`. - Fold the two 0.5-boundary tests into one that pins the whole partition, drop an assertion that only restated JS float arithmetic, and cover the cutoff/undecrypted overlap. Co-Authored-By: Claude Opus 5 * docs(mcp): note maxDistance review copy under 0.0.12 (WALM-457) --------- Co-authored-by: Harry Phan Co-authored-by: Claude Opus 5 --- docs/mcp/changelog.mdx | 2 + docs/mcp/overview.md | 2 +- docs/mcp/reference.md | 5 ++ packages/mcp/CHANGELOG.md | 2 + packages/mcp/src/auth-required.ts | 6 ++ packages/mcp/test/tool-definitions.test.mjs | 17 +++++ .../mcp/__tests__/recall-created-at.test.ts | 2 +- .../mcp/__tests__/recall-max-distance.test.ts | 64 +++++++++++++++++++ services/server/scripts/mcp/tools/recall.ts | 64 ++++++++++++++++--- 9 files changed, 153 insertions(+), 11 deletions(-) create mode 100644 services/server/scripts/mcp/__tests__/recall-max-distance.test.ts diff --git a/docs/mcp/changelog.mdx b/docs/mcp/changelog.mdx index 5d7916355..9bc31ad4b 100644 --- a/docs/mcp/changelog.mdx +++ b/docs/mcp/changelog.mdx @@ -38,10 +38,12 @@ This release forwards the MCP client's identity to the relayer so sidecar logs c ### Added - Forward the MCP client's `initialize.clientInfo` to the relayer as `x-memwal-client` / `x-memwal-client-version` so sidecar logs can name the coding agent (Claude Code, Codex, Cursor, …) on each session and tool call. +- Optional `maxDistance` on `memwal_recall`: cosine-distance cutoff (low = similar, must be >= 0). Hits with `distance >= maxDistance` are dropped. Result lines now include both `score` (`1 − cosine distance`) and `distance`. (#373) ### Fixed - Clarify `memwal_restore` `truncated=true` as known-retryable-incomplete: raising `limit` expands the sidecar cap only while `limit < 20`; `truncated=false` is not completeness (WALM-451 `sourceCapped`). +- When every decrypted `memwal_recall` hit misses `maxDistance`, keep the outside-cutoff wording and append any decrypt-drop count instead of replacing the message with a decrypt-failure report. - Resolve the credential directory on every access instead of freezing it at module load, and let `MEMWAL_CREDS_DIR` override it. The login test sandboxed the home directory with `HOME` alone, which `os.homedir()` ignores on Windows, so running the package's test suite there wrote fixture credentials over the developer's real `~/.memwal/credentials.json` and destroyed the delegate key stored in it. (#705) ## 0.0.11 diff --git a/docs/mcp/overview.md b/docs/mcp/overview.md index 06a969328..bf354de34 100644 --- a/docs/mcp/overview.md +++ b/docs/mcp/overview.md @@ -106,7 +106,7 @@ Connectors UI exposes only a single bearer field and cannot supply the required | `memwal_login` | Connect this client to your account through browser wallet sign-in. | | `memwal_logout` | Remove the saved credentials from this machine. | -`memwal_recall` prints `score = 1 - cosine distance` (higher = more similar); the wire and SDK field is `distance`. +`memwal_recall` prints both `score = 1 - cosine distance` (higher = more similar) and the raw `distance` (lower = more similar), which is the wire and SDK field. Its optional `maxDistance` cutoff is expressed in `distance`, not `score`. See [Reference](/mcp/reference) for full parameters, CLI flags, and transports. diff --git a/docs/mcp/reference.md b/docs/mcp/reference.md index 1d7c040dd..d980edc43 100644 --- a/docs/mcp/reference.md +++ b/docs/mcp/reference.md @@ -73,11 +73,16 @@ Save several durable facts in one batched call. Prefer this over repeated `memwa Search the user's Walrus Memory for facts relevant to a query. The agent calls this **proactively** at the start of a task or when the user references past work, decisions, or preferences. Returns matches ranked by relevance. +Each result line includes `score` and `distance`. `distance` is cosine distance (low = similar). Displayed `score` is `1 − cosine distance` (high = similar). Optional `maxDistance` is a cosine-distance cutoff: hits with `distance >= maxDistance` are dropped. Omit `maxDistance` to apply no cutoff. + +The cutoff is applied to the `limit` nearest matches, not before them, so a tight `maxDistance` returns fewer than `limit` results — that is the cutoff biting, not an empty namespace. Raise `limit` to widen the candidate pool the cutoff sees. + | **Parameter** | **Type** | **Required** | **Description** | | --- | --- | --- | --- | | `query` | string | yes | Natural-language query to match against stored memories. | | `limit` | integer (1–100) | no | Max memories to return. Default `10`. | | `namespace` | string | no | Namespace bucket to search. | +| `maxDistance` | number (>= 0) | no | Cosine-distance cutoff (low = similar). Hits with `distance >= maxDistance` are dropped. Omit for no cutoff. | ### memwal_analyze diff --git a/packages/mcp/CHANGELOG.md b/packages/mcp/CHANGELOG.md index ab59b2a8e..14e24f162 100644 --- a/packages/mcp/CHANGELOG.md +++ b/packages/mcp/CHANGELOG.md @@ -5,10 +5,12 @@ ### Added - Forward the MCP client's `initialize.clientInfo` to the relayer as `x-memwal-client` / `x-memwal-client-version` so sidecar logs can name the coding agent (Claude Code, Codex, Cursor, …) on each session and tool call. +- Optional `maxDistance` on `memwal_recall`: cosine-distance cutoff (low = similar, must be >= 0). Hits with `distance >= maxDistance` are dropped. Result lines now include both `score` (`1 − cosine distance`) and `distance`. (#373) ### Fixed - Clarify `memwal_restore` `truncated=true` as known-retryable-incomplete: raising `limit` expands the sidecar cap only while `limit < 20`; `truncated=false` is not completeness (WALM-451 `sourceCapped`). +- When every decrypted `memwal_recall` hit misses `maxDistance`, keep the outside-cutoff wording and append any decrypt-drop count instead of replacing the message with a decrypt-failure report. - Resolve the credential directory on every access instead of freezing it at module load, and let `MEMWAL_CREDS_DIR` override it. The login test sandboxed the home directory with `HOME` alone, which `os.homedir()` ignores on Windows, so running the package's test suite there wrote fixture credentials over the developer's real `~/.memwal/credentials.json` and destroyed the delegate key stored in it. (#705) ## 0.0.11 diff --git a/packages/mcp/src/auth-required.ts b/packages/mcp/src/auth-required.ts index f2ae3558f..beec42d7f 100644 --- a/packages/mcp/src/auth-required.ts +++ b/packages/mcp/src/auth-required.ts @@ -98,6 +98,12 @@ function buildToolDefinitions(proactive: boolean) { query: { type: "string", minLength: 1 }, limit: { type: "integer", minimum: 1, maximum: 100, default: 10 }, namespace: { type: "string" }, + maxDistance: { + type: "number", + minimum: 0, + description: + "Optional cosine-distance cutoff (low = similar; 0 = identical). Hits with distance >= maxDistance are dropped. Omit to apply no cutoff. Displayed score is 1 - distance (high = similar); do not treat score as the cutoff.", + }, }, required: ["query"], additionalProperties: false, diff --git a/packages/mcp/test/tool-definitions.test.mjs b/packages/mcp/test/tool-definitions.test.mjs index 487a50307..6bbdbdb6c 100644 --- a/packages/mcp/test/tool-definitions.test.mjs +++ b/packages/mcp/test/tool-definitions.test.mjs @@ -38,6 +38,23 @@ test("signed-out tools/list keeps conservative remember wording", () => { assert.doesNotMatch(recall, /PROACTIVELY/); }); +test("cold-start memwal_recall schema exposes maxDistance as cosine distance", () => { + const signedIn = TOOL_DEFINITIONS.find((t) => t.name === "memwal_recall"); + const signedOut = SIGNED_OUT_TOOL_DEFINITIONS.find((t) => t.name === "memwal_recall"); + assert.ok(signedIn); + assert.ok(signedOut); + for (const tool of [signedIn, signedOut]) { + const maxDistance = tool.inputSchema.properties.maxDistance; + assert.equal(maxDistance.type, "number"); + // A negative cutoff drops every hit and reads as an empty namespace, + // so the schema rejects it rather than returning a plausible nothing. + assert.equal(maxDistance.minimum, 0); + assert.match(maxDistance.description, /cosine-distance/i); + assert.match(maxDistance.description, /distance >= maxDistance/); + assert.ok(!(tool.inputSchema.required ?? []).includes("maxDistance")); + } +}); + test("memwal_recall is advertised as a read-only search", () => { assert.deepEqual(annotations(TOOL_DEFINITIONS, "memwal_recall"), { readOnlyHint: true, diff --git a/services/server/scripts/mcp/__tests__/recall-created-at.test.ts b/services/server/scripts/mcp/__tests__/recall-created-at.test.ts index 80581f3e8..17d85b0df 100644 --- a/services/server/scripts/mcp/__tests__/recall-created-at.test.ts +++ b/services/server/scripts/mcp/__tests__/recall-created-at.test.ts @@ -54,7 +54,7 @@ test("keeps the existing numbering and score format", () => { const line = formatRecallLine(row({ created_at: "2026-07-06T12:00:00Z" }), 2); assert.ok(line.startsWith("3. "), `expected 1-based numbering, got: ${line}`); - assert.match(line, /\[score=0\.750\]/); + assert.match(line, /\[score=0\.750 distance=0\.250\]/); }); test("a malformed created_at is dropped, not rendered raw", () => { diff --git a/services/server/scripts/mcp/__tests__/recall-max-distance.test.ts b/services/server/scripts/mcp/__tests__/recall-max-distance.test.ts new file mode 100644 index 000000000..801e5767f --- /dev/null +++ b/services/server/scripts/mcp/__tests__/recall-max-distance.test.ts @@ -0,0 +1,64 @@ +/** + * `memwal_recall` can cut weakly related top-K hits with `maxDistance`. + * + * The sidecar is pinned to an SDK that may not accept object-form + * `recall({ maxDistance })`, so the cutoff is applied here on `result.results` + * — same polarity as the SDK: cosine distance, keep `distance < maxDistance`. + */ +import test from "node:test"; +import assert from "node:assert/strict"; + +import { emptyRecallText, filterByMaxDistance, formatRecallLine } from "../tools/recall.js"; + +const row = (text: string, distance: number) => ({ text, distance }); + +test("omitting maxDistance leaves every hit in place", () => { + const results = [ + row("close", 0.1), + row("far", 0.9), + ]; + assert.deepEqual(filterByMaxDistance(results), results); + assert.deepEqual(filterByMaxDistance(results, undefined), results); +}); + +test("the cutoff is exclusive: keeps below maxDistance, drops at or above", () => { + const kept = filterByMaxDistance( + [row("best", 0.1), row("ok", 0.499), row("boundary", 0.5), row("worse", 0.9)], + 0.5, + ); + assert.deepEqual( + kept.map((m) => m.text), + ["best", "ok"], + ); +}); + +test("all hits outside maxDistance is not reported as a decrypt failure", () => { + const results = [row("far", 0.9), row("farther", 1.1)]; + const filtered = filterByMaxDistance(results, 0.5); + assert.equal(filtered.length, 0); + const text = emptyRecallText(results.length, 0); + assert.match(text, /outside maxDistance/); + assert.doesNotMatch(text, /decrypt/); + assert.doesNotMatch(text, /download/); +}); + +test("undecrypted matches are reported alongside the cutoff, not hidden by it", () => { + // Their distance was never computed, so "all outside maxDistance" would + // claim more than we know: they may well have been inside the cutoff. + const text = emptyRecallText(2, 3); + assert.match(text, /outside maxDistance/); + assert.match(text, /3 further matches/); + assert.match(text, /decrypt/); +}); + +test("a single undecrypted match reads as singular", () => { + assert.match(emptyRecallText(2, 1), /1 further match was/); +}); + +test("displayed score is 1 minus cosine distance", () => { + const line = formatRecallLine( + { text: "shipped the composite ranker", distance: 0.25 }, + 0, + ); + assert.match(line, /\[score=0\.750 distance=0\.250\]/); +}); diff --git a/services/server/scripts/mcp/tools/recall.ts b/services/server/scripts/mcp/tools/recall.ts index b0690196d..c48ff4ab5 100644 --- a/services/server/scripts/mcp/tools/recall.ts +++ b/services/server/scripts/mcp/tools/recall.ts @@ -23,6 +23,13 @@ const RECALL_INPUT = { .describe( "Optional namespace bucket to search within. Defaults to the session's namespace." ), + maxDistance: z + .number() + .nonnegative() + .optional() + .describe( + "Optional cosine-distance cutoff (low = similar; 0 = identical). Hits with distance >= maxDistance are dropped. Omit to apply no cutoff. Displayed score is 1 - distance (high = similar); do not treat score as the cutoff." + ), } as const; /** Key deciding whether two results say the same thing: trimmed, internal @@ -60,6 +67,47 @@ export function collapseDuplicates( return { unique, collapsed: results.length - unique.length }; } +/** + * Drop hits whose cosine distance is at or above `maxDistance`. + * + * Same polarity as the SDK: cosine distance is low-is-similar, keep + * `distance < maxDistance`. Omit `maxDistance` (or pass a non-number) to + * leave the list unchanged. + * + * TODO(WALM-428): drop this once the sidecar SDK pin moves off 0.0.3 — 0.1.x + * applies the identical filter inside `recall({ maxDistance })`. + */ +export function filterByMaxDistance( + results: T[], + maxDistance?: number, +): T[] { + if (typeof maxDistance !== "number") return results; + return results.filter((memory) => memory.distance < maxDistance); +} + +/** + * Empty-result copy after the maxDistance filter. + * + * Decrypted hits that all missed the cutoff are not a download/decrypt + * failure, so the decrypt copy must not claim the whole result. But an + * undecrypted match never had a distance computed, so the cutoff copy cannot + * speak for it either: when both counts are non-zero, say both rather than + * letting "outside maxDistance" imply the namespace held nothing relevant. + */ +export function emptyRecallText(resultCount: number, dropped: number): string { + if (resultCount > 0) { + const unchecked = + dropped > 0 + ? ` (${dropped} further ${dropped === 1 ? "match was" : "matches were"} never checked against the cutoff: they failed to download or decrypt.)` + : ""; + return `All matching memories were outside maxDistance.${unchecked}`; + } + if (dropped > 0) { + return `No matching memories could be returned (${dropped} matched but failed to download or decrypt). This is not an empty namespace.`; + } + return "No matching memories found."; +} + /** * Render one recall hit as the line the model sees. * @@ -83,9 +131,10 @@ export function formatRecallLine( index: number, ): string { const score = (1 - memory.distance).toFixed(3); + const distance = memory.distance.toFixed(3); const written = isoDateOrNull(memory.created_at); const stamp = written ? ` [written=${written}]` : ""; - return `${index + 1}. [score=${score}]${stamp} ${memory.text}`; + return `${index + 1}. [score=${score} distance=${distance}]${stamp} ${memory.text}`; } /** `YYYY-MM-DD` for a parseable timestamp, else null. */ @@ -118,25 +167,22 @@ export function registerRecallTool( "Search the user's Walrus Memory for relevant facts before responding. Call this PROACTIVELY at the start of a task, or whenever the user references past work, prior decisions, their preferences, or anything you may have stored earlier — don't wait to be asked. A single focused query is usually enough — recall is a real retrieval over encrypted storage, so do NOT fire multiple redundant searches for the same question. Returns matching memories ranked by semantic relevance, NOT by date: the most recent memory can fall outside `limit` when an older one happens to match your wording more literally, so do not treat the results as a complete or current view of a namespace. Each result carries `written=YYYY-MM-DD`, the date the memory was saved (not any date its text mentions) — check it rather than assuming the top result is the latest.", inputSchema: RECALL_INPUT, }, - wrapTool<{ query: string; limit: number; namespace?: string }>(session, "memwal_recall", async ({ query, limit, namespace }) => { + wrapTool<{ query: string; limit: number; namespace?: string; maxDistance?: number }>(session, "memwal_recall", async ({ query, limit, namespace, maxDistance }) => { const result = await session.memwal.recall(query, limit, namespace); const droppedRaw = (result as { dropped_count?: unknown }).dropped_count; const dropped = typeof droppedRaw === "number" ? droppedRaw : 0; - if (result.results.length === 0) { - const empty = - dropped > 0 - ? `No matching memories could be returned (${dropped} matched but failed to download or decrypt). This is not an empty namespace.` - : "No matching memories found."; + const filtered = filterByMaxDistance(result.results, maxDistance); + if (filtered.length === 0) { return { content: [ { type: "text", - text: empty, + text: emptyRecallText(result.results.length, dropped), }, ], }; } - const { unique, collapsed } = collapseDuplicates(result.results); + const { unique, collapsed } = collapseDuplicates(filtered); const lines = unique.map((m, i) => formatRecallLine(m, i)); // Say what was folded away rather than quietly returning fewer rows // than the caller asked for. It also surfaces that the same fact was From 117d45ea745d42c326efc6dd123f733b8a147f92 Mon Sep 17 00:00:00 2001 From: HoangDucBach Date: Mon, 7 Sep 2026 11:19:00 +0700 Subject: [PATCH 14/29] fix(researcher): unwrap SIIT and IPv4-compatible embeddings in SSRF denylist Mapped IPv4-in-IPv6 already hit the v4 table; deprecated compatible and SIIT forms did not. Also block IPv6 multicast /8 to match v4 multicast. Leave 6to4 and local-use NAT64 out of scope. --- apps/researcher/lib/rag/ingest/safe-fetch.ts | 37 ++++++++++++++----- .../lib/rag/ingest/safe-fetch.unit.test.ts | 9 +++++ 2 files changed, 37 insertions(+), 9 deletions(-) diff --git a/apps/researcher/lib/rag/ingest/safe-fetch.ts b/apps/researcher/lib/rag/ingest/safe-fetch.ts index 223fca4a8..f282590f7 100644 --- a/apps/researcher/lib/rag/ingest/safe-fetch.ts +++ b/apps/researcher/lib/rag/ingest/safe-fetch.ts @@ -45,6 +45,8 @@ const BLOCKED_V6_RANGES: [string, number][] = [ // NAT64 addresses carry an IPv4 destination in their low 32 bits, so they are a // way back into the ranges above. ["64:ff9b::", 96], + // IPv4 already blocks 224.0.0.0/4; the v6 multicast range is the counterpart. + ["ff00::", 8], ]; function ipv4ToBytes(address: string): number[] | null { @@ -138,6 +140,12 @@ function withinRange( return true; } +function isBlockedIpv4Bytes(bytes: number[]): boolean { + return BLOCKED_V4_RANGES.some(([network, prefix]) => + withinRange(bytes, ipv4ToBytes(network) as number[], prefix) + ); +} + function isMappedIpv4(bytes: number[]): boolean { return ( bytes.slice(0, 10).every((byte) => byte === 0) && @@ -146,6 +154,20 @@ function isMappedIpv4(bytes: number[]): boolean { ); } +function isSiitIpv4(bytes: number[]): boolean { + return ( + bytes.slice(0, 8).every((byte) => byte === 0) && + bytes[8] === 0xff && + bytes[9] === 0xff && + bytes[10] === 0 && + bytes[11] === 0 + ); +} + +function isIpv4Compatible(bytes: number[]): boolean { + return bytes.slice(0, 12).every((byte) => byte === 0); +} + /** * Whether an IP literal names something outside the public internet. Anything * unparseable counts as blocked: a value this code cannot reason about must not @@ -157,11 +179,7 @@ export function isBlockedAddress(address: string): boolean { if (version === 4) { const bytes = ipv4ToBytes(address); - return bytes - ? BLOCKED_V4_RANGES.some(([network, prefix]) => - withinRange(bytes, ipv4ToBytes(network) as number[], prefix) - ) - : true; + return bytes ? isBlockedIpv4Bytes(bytes) : true; } if (version === 6) { @@ -170,10 +188,11 @@ export function isBlockedAddress(address: string): boolean { if (!bytes) { return true; } - if (isMappedIpv4(bytes)) { - return BLOCKED_V4_RANGES.some(([network, prefix]) => - withinRange(bytes.slice(12), ipv4ToBytes(network) as number[], prefix) - ); + // Mapped, deprecated IPv4-compatible (::/96), and SIIT stash IPv4 in the + // last 32 bits. Check that payload against the v4 table so ::7f00:1 and + // ::ffff:0:7f00:1 cannot skip a denylist that already unwraps ::ffff:7f00:1. + if (isMappedIpv4(bytes) || isSiitIpv4(bytes) || isIpv4Compatible(bytes)) { + return isBlockedIpv4Bytes(bytes.slice(12)); } return BLOCKED_V6_RANGES.some(([network, prefix]) => diff --git a/apps/researcher/lib/rag/ingest/safe-fetch.unit.test.ts b/apps/researcher/lib/rag/ingest/safe-fetch.unit.test.ts index 3bc57ed07..b2becd456 100644 --- a/apps/researcher/lib/rag/ingest/safe-fetch.unit.test.ts +++ b/apps/researcher/lib/rag/ingest/safe-fetch.unit.test.ts @@ -41,8 +41,14 @@ const BLOCKED = [ "fe80::1%eth0", // IPv4-mapped and NAT64 forms both carry a blocked IPv4 destination. "::ffff:127.0.0.1", + "::ffff:7f00:1", "::ffff:169.254.169.254", "64:ff9b::7f00:1", + // Deprecated IPv4-compatible and SIIT embeddings of loopback. + "::7f00:1", + "::ffff:0:7f00:1", + // IPv6 multicast; v4 multicast 224.0.0.1 is already in the list above. + "ff02::1", ]; const ALLOWED = [ @@ -54,6 +60,9 @@ const ALLOWED = [ "128.0.0.1", "2606:4700:4700::1111", "::ffff:8.8.8.8", + // Unwrap, do not blanket-block: public IPv4 via compatible / SIIT stays public. + "::8.8.8.8", + "::ffff:0:8.8.8.8", ]; // ChatbotError puts the caller-facing detail in `cause` and leaves `message` as From 04257b3247395b47f38818963083c06730c80127 Mon Sep 17 00:00:00 2001 From: HoangDucBach Date: Mon, 7 Sep 2026 11:19:00 +0700 Subject: [PATCH 15/29] fix(researcher): map a malformed redirect Location to ChatbotError A broken Location after 3xx threw TypeError, which the chat route turned into offline:chat. Match assertPublicUrl's 400 instead. --- apps/researcher/lib/rag/ingest/safe-fetch.ts | 6 ++++- .../lib/rag/ingest/safe-fetch.unit.test.ts | 27 +++++++++++++++++++ 2 files changed, 32 insertions(+), 1 deletion(-) diff --git a/apps/researcher/lib/rag/ingest/safe-fetch.ts b/apps/researcher/lib/rag/ingest/safe-fetch.ts index f282590f7..5c8517b5a 100644 --- a/apps/researcher/lib/rag/ingest/safe-fetch.ts +++ b/apps/researcher/lib/rag/ingest/safe-fetch.ts @@ -286,7 +286,11 @@ export async function fetchPublicUrl( return response; } - target = new URL(location, url).toString(); + try { + target = new URL(location, url).toString(); + } catch { + throw new ChatbotError("bad_request:api", "Invalid URL format"); + } } throw new ChatbotError( diff --git a/apps/researcher/lib/rag/ingest/safe-fetch.unit.test.ts b/apps/researcher/lib/rag/ingest/safe-fetch.unit.test.ts index b2becd456..f27420c0b 100644 --- a/apps/researcher/lib/rag/ingest/safe-fetch.unit.test.ts +++ b/apps/researcher/lib/rag/ingest/safe-fetch.unit.test.ts @@ -238,3 +238,30 @@ test("fetchPublicUrl refuses a redirect into a blocked range", async () => { globalThis.fetch = originalFetch; } }); + +test("fetchPublicUrl maps a malformed redirect Location to ChatbotError", async () => { + // new URL(location, url) throws TypeError on a broken Location. Without a + // catch, chat's generic handler turns that into offline:chat (503) instead + // of the same 400 assertPublicUrl uses for a bad user-supplied URL. + const originalFetch = globalThis.fetch; + const requested: string[] = []; + + globalThis.fetch = (async (input: string | URL | Request) => { + requested.push(String(input)); + + return new Response(null, { + status: 302, + headers: { location: "http://[" }, + }); + }) as typeof fetch; + + try { + await assertRejectedWith( + () => fetchPublicUrl("https://8.8.8.8/file.pdf"), + /Invalid URL format/ + ); + assert.deepEqual(requested, ["https://8.8.8.8/file.pdf"]); + } finally { + globalThis.fetch = originalFetch; + } +}); From e0e63a461f858c6c4868ba98d11048448e609e50 Mon Sep 17 00:00:00 2001 From: HoangDucBach Date: Mon, 7 Sep 2026 11:19:00 +0700 Subject: [PATCH 16/29] docs(noter): reword logout JSDoc to the header-only invariant Session ids remain bearer credentials. Logout ignores a body/input id; it does not make knowledge of the id insufficient. --- apps/noter/package/feature/auth/api/route.ts | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/apps/noter/package/feature/auth/api/route.ts b/apps/noter/package/feature/auth/api/route.ts index f36b607bb..91eb89768 100644 --- a/apps/noter/package/feature/auth/api/route.ts +++ b/apps/noter/package/feature/auth/api/route.ts @@ -51,9 +51,9 @@ export const authRouter = router({ ), /** - * Logout - clear the caller's own session (works for both zkLogin and wallet). - * Like getSession, the id comes from the header, so knowing another user's - * session id is not enough to end their session. + * Logout - end only the session presented in x-session-id (zkLogin and wallet). + * A body/input id is ignored. The header value is still a bearer credential: + * presenting a session id there is enough to delete that session. */ logout: procedure.mutation(async ({ ctx }) => { if (ctx.sessionId) { From 85ce261d0a2b5f9b99e30a957cd4ed905331618f Mon Sep 17 00:00:00 2001 From: HoangDucBach Date: Mon, 7 Sep 2026 11:19:00 +0700 Subject: [PATCH 17/29] test(noter): cover createContext UUID guard on x-session-id Route-layer tests inject a parsed sessionId, so they miss a broken malformed-header check. Assert missing/non-uuid headers skip the walletSessions lookup. --- .../lib/trpc/create-context.unit.test.ts | 103 ++++++++++++++++++ 1 file changed, 103 insertions(+) create mode 100644 apps/noter/package/shared/lib/trpc/create-context.unit.test.ts diff --git a/apps/noter/package/shared/lib/trpc/create-context.unit.test.ts b/apps/noter/package/shared/lib/trpc/create-context.unit.test.ts new file mode 100644 index 000000000..39dd1aaab --- /dev/null +++ b/apps/noter/package/shared/lib/trpc/create-context.unit.test.ts @@ -0,0 +1,103 @@ +import { beforeEach, describe, expect, it, vi } from "vitest"; +import type { FetchCreateContextFnOptions } from "@trpc/server/adapters/fetch"; + +// createContext is where the malformed-header guard lives (issue #779). A +// non-uuid x-session-id used to reach eq(walletSessions.id, …) and make +// Postgres raise. The route-layer session-binding tests inject ctx.sessionId +// already parsed, so they cannot catch a regression here. These call the real +// createContext with the db mocked and assert the lookup is skipped unless the +// header is a uuid. + +const VALID_SESSION = "0192f0a0-0000-7000-8000-000000000001"; +const USER_ID = "user-1"; + +const { select, from, where, limit } = vi.hoisted(() => ({ + select: vi.fn(), + from: vi.fn(), + where: vi.fn(), + limit: vi.fn(), +})); + +vi.mock("@/shared/lib/db", () => ({ + db: { + select: (...args: unknown[]) => select(...args), + }, +})); + +function requestOpts(headers?: HeadersInit): FetchCreateContextFnOptions { + return { + req: new Request("http://localhost/api/trpc", { headers }), + resHeaders: new Headers(), + info: {} as FetchCreateContextFnOptions["info"], + }; +} + +async function load() { + return import("./init"); +} + +beforeEach(() => { + vi.clearAllMocks(); + select.mockReturnValue({ from }); + from.mockReturnValue({ where }); + where.mockReturnValue({ limit }); + limit.mockResolvedValue([]); +}); + +describe("createContext — x-session-id UUID guard", () => { + it.each([ + ["missing", undefined], + ["empty", ""], + ["non-uuid", "not-a-uuid"], + ] as const)( + "treats a %s header as no credential and skips the session lookup", + async (_label, header) => { + const { createContext } = await load(); + const ctx = await createContext( + requestOpts( + header === undefined ? undefined : { "x-session-id": header } + ) + ); + + expect(ctx.sessionId).toBeNull(); + expect(ctx.userId).toBeNull(); + expect(select).not.toHaveBeenCalled(); + } + ); + + it("looks up an expired uuid session but does not authenticate it", async () => { + limit.mockResolvedValueOnce([ + { + id: VALID_SESSION, + userId: USER_ID, + expiresAt: new Date(Date.now() - 60_000), + }, + ]); + const { createContext } = await load(); + const ctx = await createContext( + requestOpts({ "x-session-id": VALID_SESSION }) + ); + + expect(select).toHaveBeenCalledOnce(); + expect(ctx.sessionId).toBe(VALID_SESSION); + expect(ctx.userId).toBeNull(); + }); + + it("authenticates a valid uuid session", async () => { + limit.mockResolvedValueOnce([ + { + id: VALID_SESSION, + userId: USER_ID, + expiresAt: new Date(Date.now() + 60_000), + }, + ]); + const { createContext } = await load(); + const ctx = await createContext( + requestOpts({ "x-session-id": VALID_SESSION }) + ); + + expect(select).toHaveBeenCalledOnce(); + expect(ctx.sessionId).toBe(VALID_SESSION); + expect(ctx.userId).toBe(USER_ID); + }); +}); From 60063e8b8d99db5a76ff5f2b7427a4572482b06a Mon Sep 17 00:00:00 2001 From: HoangDucBach Date: Mon, 7 Sep 2026 11:19:00 +0700 Subject: [PATCH 18/29] docs(chatbot): drop incident recap from getSessionToken Keep a one-liner on the URIError branch: Auth.js URL-decodes Bearer outside its JWT try/catch. --- apps/chatbot/lib/session-token.ts | 7 +------ 1 file changed, 1 insertion(+), 6 deletions(-) diff --git a/apps/chatbot/lib/session-token.ts b/apps/chatbot/lib/session-token.ts index 17092e1bf..0472420b4 100644 --- a/apps/chatbot/lib/session-token.ts +++ b/apps/chatbot/lib/session-token.ts @@ -1,12 +1,6 @@ import { getToken } from "next-auth/jwt"; import { isDevelopmentEnvironment } from "@/lib/constants"; -/** - * Read the Auth.js session JWT. A malformed `Authorization: Bearer` value - * (`%%`, lone `%`, …) makes `getToken` throw `URIError` from - * `decodeURIComponent` *before* its inner decode try/catch — that was HTTP 500 - * on proxy and `/api/auth/guest`. Treat it as no session. - */ export async function getSessionToken(request: Request) { try { return await getToken({ @@ -15,6 +9,7 @@ export async function getSessionToken(request: Request) { secureCookie: !isDevelopmentEnvironment, }); } catch (error) { + // Auth.js URL-decodes Bearer outside its JWT try/catch; malformed % sequences are not a session. if (error instanceof URIError) { return null; } From eb3bc18a27f0163c962e82566574953d83ceaed6 Mon Sep 17 00:00:00 2001 From: ducnmm <165614309+ducnmm@users.noreply.github.com> Date: Sun, 6 Sep 2026 23:38:20 -0700 Subject: [PATCH 19/29] docs: document namespace 255-byte cap and restore truncated (WALM-486) (#867) * docs: document namespace 255-byte cap and restore truncated (WALM-486) * docs: note restore limit clamp 1-100 (WALM-486) --- SKILL.md | 19 +++++++++++++++---- 1 file changed, 15 insertions(+), 4 deletions(-) diff --git a/SKILL.md b/SKILL.md index 51495f557..9957b0ffc 100644 --- a/SKILL.md +++ b/SKILL.md @@ -155,7 +155,7 @@ const stored = await memwal.waitForRememberJob(accepted.job_id, { | `recall({ query, limit?, topK?, namespace?, maxDistance? })` *(preferred)* or `recall(query, limit?, namespace?)` | Semantic search for memories | `{ results: [{ blob_id, text, distance }], total }` | | `analyze(text, namespace?)` | Extract facts and accept one memory job per fact | `{ job_ids, facts, fact_count, status, owner }` | | `analyzeAndWait(text, namespace?, opts?)` | Extract facts and wait for all fact jobs to complete | `{ results, facts, total, succeeded, failed, owner }` | -| `restore(namespace, limit?)` | Rebuild missing index entries from Walrus | `{ restored, skipped, total, namespace, owner }` | +| `restore(namespace, limit?)` | Rebuild missing index entries from Walrus | `{ restored, skipped, total, namespace, owner, truncated }` | | `health()` | Check relayer health | `{ status, version }` | | `getPublicKeyHex()` | Get hex-encoded public key | `string` | @@ -274,6 +274,7 @@ interface RestoreResult { total: number; namespace: string; owner: string; + truncated: boolean; } interface HealthResult { @@ -315,9 +316,16 @@ A namespace is an **opaque, flat string label** scoped to a single owner. It is #### Validation -The server accepts any non-empty string as a namespace. There is no length cap, no character whitelist, no normalization (whitespace, case, Unicode). Whatever you send is stored verbatim and matched with exact equality. If you omit the namespace, the server falls back to the literal string `"default"`. +Omit `namespace` and the server uses the literal string `"default"`. An explicit empty string is rejected with HTTP 400 (`namespace cannot be empty`). -> **Implication:** `"my-app"`, `" my-app"` (leading space), `"My-App"`, and `"my-app/"` are four distinct namespaces. Pick a convention and stick to it. +The server then accepts any non-empty UTF-8 string except: + +- more than **255 bytes** (UTF-8 byte length, not character count — Rust `str::len()`) → HTTP 400 `namespace exceeds maximum length of 255 bytes` +- a NUL byte (`\0`) → HTTP 400 `namespace contains a NUL byte` (WALM-439). Tabs, newlines, and other control characters are still allowed so older namespaces stay readable. + +There is no character whitelist, no case folding, no trim, and no Unicode normalization. Whatever passes validation is stored verbatim and matched with exact equality. + +> **Implication:** `"my-app"`, `" my-app"` (leading space), `"My-App"`, and `"my-app/"` are four distinct namespaces. Pick a convention and stick to it. Multi-byte characters (CJK, emoji) consume more than one byte each, so they hit the 255-byte cap sooner than a character count would suggest. #### Flat, not hierarchical @@ -359,6 +367,9 @@ Cross-namespace and cross-owner reads are not just filtered out of results — t | `total` | All on-chain blobs the relayer saw for `(owner, namespace)` | Before the limit was applied | | `namespace` | Echo of the request | | | `owner` | Resolved owner address | | +| `truncated` | Known-retryable-incomplete | `true` is not a hard failure; `false` is not completeness | + +`truncated=true` means this restore is **known-retryable-incomplete**: more missing blobs than `limit` allowed this call to restore, **or** the sidecar's owner-wide candidate fetch hit its cap **and** raising `limit` can still expand that fetch (`limit < 20`). Once the sidecar cap is saturated (`limit >= 20`, cap pinned at 100), truncation follows this call's missing-blob page length, not onchain `total`. A fully restored namespace does not loop. `truncated=false` is **not** proof the sidecar saw every onchain blob; blobs beyond the owner-wide sidecar candidate cap can still be missing. WALM-451 tracks a `sourceCapped` field for that case. Relayers older than WALM-319 omit `truncated`; SDKs default it to `false`. **Silent drops.** A blob that *cannot* be decrypted or embedded (e.g. wrong delegate key, malformed ciphertext, embedding API down) is dropped without counting in `restored` *or* `skipped`. `restored + skipped` is therefore a lower bound on healthy entries, not a strict equality with `total`. @@ -366,7 +377,7 @@ Cross-namespace and cross-owner reads are not just filtered out of results — t * `limit` defaults to `10` in both TypeScript and Python SDKs and matches the server-side default. The Python SDK historically defaulted to `50`; it is now realigned with the server. * `limit` caps the **inspected** blob set, newest-first. It does not cap `restored` independently — if all 10 inspected blobs are already indexed, `restored = 0` and `skipped = 10`. -* There is no enforced server-side maximum, but very large limits will dominate latency (see below). +* The relayer clamps `limit` to 1–100 (values outside that range are clamped, not rejected). #### Pagination From 4142d1a0f7c47930612f974190081375dd8fdecb Mon Sep 17 00:00:00 2001 From: ducnmm <165614309+ducnmm@users.noreply.github.com> Date: Sun, 6 Sep 2026 23:38:24 -0700 Subject: [PATCH 20/29] ci: add production-safe SEAL cross-account synthetic (COMG-715) (#846) * ci: add production-safe SEAL cross-account synthetic (COMG-715) Read-only dry-run of seal_approve so a live identity from account A cannot authorize account B's SEAL key. Missing secrets skip (exit 0). Authorization success pages via SYNTHETIC_SEAL_CROSS_ACCOUNT_FAIL. * docs: use H2 body heading in SEAL synthetic page (COMG-715) * fix(ci): parse JSON-RPC MoveAbort abort code (COMG-715) JSON-RPC returns MoveAbort as a Display string whose MoveLocation contains commas, so [^,]+ never reached abort code 100 and a healthy ENoAccess denial was classified as misconfiguration. * test(ci): drop redundant MoveAbort parser coverage (COMG-715) Keep the nested-comma JSON-RPC fixture and distinct parser/CLI branches. Remove the duplicate hex string, the old-regex assertion, and the classifyNegative helper that restated extractAbortCode results. --- .../synthetic-seal-cross-account.yml | 59 ++ .github/workflows/test.yml | 2 + docs/docs.json | 1 + docs/relayer/synthetic-seal-cross-account.md | 99 +++ scripts/synthetic-seal-cross-account.mjs | 598 ++++++++++++++++++ scripts/synthetic-seal-cross-account.test.mjs | 100 +++ 6 files changed, 859 insertions(+) create mode 100644 .github/workflows/synthetic-seal-cross-account.yml create mode 100644 docs/relayer/synthetic-seal-cross-account.md create mode 100644 scripts/synthetic-seal-cross-account.mjs create mode 100644 scripts/synthetic-seal-cross-account.test.mjs diff --git a/.github/workflows/synthetic-seal-cross-account.yml b/.github/workflows/synthetic-seal-cross-account.yml new file mode 100644 index 000000000..9ed4af724 --- /dev/null +++ b/.github/workflows/synthetic-seal-cross-account.yml @@ -0,0 +1,59 @@ +name: Synthetic SEAL cross-account + +# Production-safe negative synthetic (COMG-715). +# workflow_dispatch only — never on pull_request. Missing secrets skip (exit 0). +# A live cross-account SEAL authorization fails the job with +# SYNTHETIC_SEAL_CROSS_ACCOUNT_FAIL and should page on-call. +# +# Schedule from ops (not enabled here on purpose): +# gh workflow run synthetic-seal-cross-account.yml --ref +# or add `on.schedule` after the two-account secrets exist. + +on: + workflow_dispatch: + inputs: + relayer_url: + description: Relayer base URL for GET /config when SUI_RPC_URL / MEMWAL_PACKAGE_ID are unset + required: false + type: string + default: "" + +permissions: + contents: read + +jobs: + synthetic: + name: Cross-account SEAL deny + runs-on: ubuntu-latest + timeout-minutes: 10 + env: + SEAL_CROSS_ACCOUNT_A_ID: ${{ secrets.SEAL_CROSS_ACCOUNT_A_ID }} + SEAL_CROSS_ACCOUNT_B_ID: ${{ secrets.SEAL_CROSS_ACCOUNT_B_ID }} + SEAL_CROSS_ACCOUNT_A_KEY: ${{ secrets.SEAL_CROSS_ACCOUNT_A_KEY }} + SEAL_CROSS_ACCOUNT_B_KEY: ${{ secrets.SEAL_CROSS_ACCOUNT_B_KEY }} + SUI_RPC_URL: ${{ vars.SUI_RPC_URL }} + SUI_NETWORK: ${{ vars.SUI_NETWORK }} + MEMWAL_PACKAGE_ID: ${{ vars.MEMWAL_PACKAGE_ID }} + MEMWAL_REGISTRY_ID: ${{ vars.MEMWAL_REGISTRY_ID }} + MEMWAL_SEAL_POLICY_PACKAGE_ID: ${{ vars.MEMWAL_SEAL_POLICY_PACKAGE_ID }} + MEMWAL_SERVER_URL: ${{ inputs.relayer_url || vars.MEMWAL_SERVER_URL }} + steps: + - uses: actions/checkout@v4 + + - name: Detect secrets + id: cfg + shell: bash + run: | + if [ -z "${SEAL_CROSS_ACCOUNT_A_ID:-}" ] || [ -z "${SEAL_CROSS_ACCOUNT_B_ID:-}" ] || \ + [ -z "${SEAL_CROSS_ACCOUNT_A_KEY:-}" ] || [ -z "${SEAL_CROSS_ACCOUNT_B_KEY:-}" ]; then + echo "has_secrets=false" >> "$GITHUB_OUTPUT" + else + echo "has_secrets=true" >> "$GITHUB_OUTPUT" + fi + + - name: Setup JS + if: steps.cfg.outputs.has_secrets == 'true' + uses: ./.github/actions/setup-js + + - name: Run cross-account SEAL synthetic + run: node scripts/synthetic-seal-cross-account.mjs diff --git a/.github/workflows/test.yml b/.github/workflows/test.yml index 3e790b7e3..b2e6ab3ef 100644 --- a/.github/workflows/test.yml +++ b/.github/workflows/test.yml @@ -32,6 +32,8 @@ jobs: script: scripts/check-docs-code-sync.mjs - name: Docs / Freshness script: scripts/check-docs-freshness.mjs + - name: SEAL cross-account synthetic parser + script: scripts/synthetic-seal-cross-account.test.mjs steps: - uses: actions/checkout@v4 diff --git a/docs/docs.json b/docs/docs.json index 6b55266c9..6d2de4989 100644 --- a/docs/docs.json +++ b/docs/docs.json @@ -175,6 +175,7 @@ "relayer/self-hosting", "relayer/nautilus-tee", "relayer/observability", + "relayer/synthetic-seal-cross-account", "relayer/versioning-and-compatibility", "relayer/api-reference" ] diff --git a/docs/relayer/synthetic-seal-cross-account.md b/docs/relayer/synthetic-seal-cross-account.md new file mode 100644 index 000000000..49c37858b --- /dev/null +++ b/docs/relayer/synthetic-seal-cross-account.md @@ -0,0 +1,99 @@ +--- +title: "SEAL Cross-Account Synthetic" +description: >- + How operators schedule the production-safe negative synthetic that pages if + an identity from MemWal account A can authorize account B's SEAL key. +keywords: + - Walrus Memory + - MemWal + - SEAL + - synthetic + - cross-account + - CI +goal: + description: Configure two isolated test accounts and run the read-only SEAL cross-account synthetic on a schedule. + requires: + - has_frontmatter: + - title + - description + - keywords + label: Has required frontmatter fields + - min_words: 150 + label: Needs more content depth + - has_questions: true + label: Needs questions for AI search visibility + - has_answer: true + label: Needs answer summary for AI citation +questions: + - "How do I run the SEAL cross-account synthetic?" + - "What secrets does the SEAL cross-account synthetic need?" + - "How do I schedule the SEAL cross-account check in production?" +answer: >- + Run `scripts/synthetic-seal-cross-account.mjs` or dispatch + `.github/workflows/synthetic-seal-cross-account.yml`. The check is read-only: + it dry-runs `seal_approve` and never writes memories or mutates chain. Unset + secrets skip the job (exit 0). If identity A can authorize account B, the job + exits 1 with `SYNTHETIC_SEAL_CROSS_ACCOUNT_FAIL` and should page on-call. + Schedule it with `gh workflow run` or an ops cron that dispatches that workflow. +--- + +## SEAL Cross-Account Synthetic + +COMG-715 adds a production-safe **negative** synthetic on top of the Move unit +test `test_seal_approve_delegate_requires_matching_owner`. The unit test covers +the contract in isolation. This check would **alert** if a live identity from +account A can authorize account B's SEAL key. + +The script is **read-only**. It never calls remember, never fetches SEAL +decryption keys, and never executes a transaction. It `devInspect`s +`account::seal_approve` only. + +## When it runs + +`.github/workflows/synthetic-seal-cross-account.yml` is `workflow_dispatch` +only. It does **not** run on pull requests, so a PR without production secrets +cannot page and cannot touch chain. + +If the two account ids and two delegate keys are unset, the script prints +`skip` and exits 0. + +## Secrets and variables + +Use two **isolated** MemWal accounts (different owners). Each key must be a +delegate or owner of its own account and must **not** be registered on the other. + +| Name | Where | Purpose | +| --- | --- | --- | +| `SEAL_CROSS_ACCOUNT_A_ID` / `SEAL_CROSS_ACCOUNT_B_ID` | GitHub secret | Account object ids | +| `SEAL_CROSS_ACCOUNT_A_KEY` / `SEAL_CROSS_ACCOUNT_B_KEY` | GitHub secret | Delegate private keys (hex or `suiprivkey1…`) | +| `SUI_RPC_URL` | GitHub variable | JSON-RPC fullnode | +| `MEMWAL_PACKAGE_ID` | GitHub variable | Policy package id | +| `MEMWAL_REGISTRY_ID` | GitHub variable | `AccountRegistry` object id | +| `MEMWAL_SERVER_URL` | GitHub variable or workflow input | Optional; `GET /config` fills package id and RPC URL | +| `SUI_NETWORK` | GitHub variable | `mainnet` or `testnet` when the RPC URL does not say | + +Do not commit private keys. Do not reuse the shared benchmark account as both A and B. + +## Schedule (ops) + +Dispatch from a protected ref after the secrets exist: + +```bash +gh workflow run synthetic-seal-cross-account.yml --ref main +``` + +Or Actions → **Synthetic SEAL cross-account** → **Run workflow**. + +To cron it, add `on.schedule` to that workflow once the two-account secrets +are in place (left off by default so an empty repo cannot false-green or page): + +```yaml +on: + schedule: + - cron: "17 6 * * *" # daily 06:17 UTC + workflow_dispatch: +``` + +Point the workflow failure at on-call. A red job whose log contains +`SYNTHETIC_SEAL_CROSS_ACCOUNT_FAIL` means live `seal_approve` allowed a +cross-account identity. That is a SEAL policy regression, not a flake. diff --git a/scripts/synthetic-seal-cross-account.mjs b/scripts/synthetic-seal-cross-account.mjs new file mode 100644 index 000000000..0a5a70fdd --- /dev/null +++ b/scripts/synthetic-seal-cross-account.mjs @@ -0,0 +1,598 @@ +#!/usr/bin/env node +/** + * Production-safe negative synthetic for COMG-715. + * + * Asserts that an identity from MemWal account A cannot authorize account B's + * SEAL key (and the reverse). Read-only: `sui_devInspectTransactionBlock` of + * `account::seal_approve` only. Does not remember, decrypt, fetch SEAL keys, + * or execute a transaction. + * + * Exit 0 — required secrets unset (skip) or deny as expected. + * Exit 1 — SYNTHETIC_SEAL_CROSS_ACCOUNT_FAIL (page: A authorized B). + * Exit 2 — misconfiguration or RPC failure (check is not valid). + * + * node scripts/synthetic-seal-cross-account.mjs + * node scripts/synthetic-seal-cross-account.mjs --help + */ + +import path from "node:path"; +import { fileURLToPath } from "node:url"; + +const E_NO_ACCESS = 100; +const FAIL_TOKEN = "SYNTHETIC_SEAL_CROSS_ACCOUNT_FAIL"; +const OBJECT_ID_RE = /^0x[0-9a-fA-F]{64}$/; +const HEX_32_RE = /^(0x)?[0-9a-fA-F]{64}$/; + +const REQUIRED_SECRETS = [ + ["SEAL_CROSS_ACCOUNT_A_ID", ["SEAL_CROSS_ACCOUNT_A_ID", "MEMWAL_ACCOUNT_A_ID"]], + ["SEAL_CROSS_ACCOUNT_B_ID", ["SEAL_CROSS_ACCOUNT_B_ID", "MEMWAL_ACCOUNT_B_ID"]], + ["SEAL_CROSS_ACCOUNT_A_KEY", ["SEAL_CROSS_ACCOUNT_A_KEY", "MEMWAL_DELEGATE_KEY_A"]], + ["SEAL_CROSS_ACCOUNT_B_KEY", ["SEAL_CROSS_ACCOUNT_B_KEY", "MEMWAL_DELEGATE_KEY_B"]], +]; + +function env(name) { + const v = process.env[name]; + return typeof v === "string" && v.trim() ? v.trim() : ""; +} + +function firstEnv(names) { + for (const name of names) { + const v = env(name); + if (v) return v; + } + return ""; +} + +function usage() { + process.stdout.write(`Production-safe SEAL cross-account synthetic (COMG-715). + +Read-only dry-run of account::seal_approve. Never writes memories or mutates chain. + +Skip (exit 0) when the two account ids and two delegate keys are unset. + +Required to run: + SEAL_CROSS_ACCOUNT_A_ID / SEAL_CROSS_ACCOUNT_B_ID + SEAL_CROSS_ACCOUNT_A_KEY / SEAL_CROSS_ACCOUNT_B_KEY + SUI_RPC_URL and MEMWAL_PACKAGE_ID + (or MEMWAL_SERVER_URL / MEMWAL_RELAYER_URL — GET /config fills both) + +Also used when set: + MEMWAL_REGISTRY_ID AccountRegistry object id + MEMWAL_SEAL_POLICY_PACKAGE_ID seal_approve package (defaults to MEMWAL_PACKAGE_ID) + SUI_NETWORK mainnet|testnet (inferred from the RPC URL when omitted) + +Expected: ENoAccess / deny. If A can authorize B: exit 1 ${FAIL_TOKEN}. +`); +} + +function normalizeHexAddress(value) { + const hex = String(value).replace(/^0x/i, "").toLowerCase(); + if (!/^[0-9a-f]{1,64}$/.test(hex)) { + throw new Error(`not a Sui address: ${value}`); + } + return `0x${hex.padStart(64, "0")}`; +} + +function u64ToLeBytes(value) { + const n = BigInt(value); + if (n < 0n || n > 0xffff_ffff_ffff_ffffn) { + throw new Error(`u64 out of range: ${value}`); + } + const out = new Uint8Array(8); + for (let i = 0; i < 8; i++) out[i] = Number((n >> (8n * BigInt(i))) & 0xffn); + return out; +} + +function sealKeyIdBytes(ownerHex, counter) { + const owner = Uint8Array.from( + normalizeHexAddress(ownerHex) + .slice(2) + .match(/.{2}/g) + .map((b) => parseInt(b, 16)), + ); + const id = new Uint8Array(40); + id.set(owner, 0); + id.set(u64ToLeBytes(counter), 32); + return id; +} + +function shortId(id) { + const n = normalizeHexAddress(id); + return `${n.slice(0, 10)}…${n.slice(-6)}`; +} + +function secretPresence() { + return REQUIRED_SECRETS.map(([label, names]) => ({ + label, + value: firstEnv(names), + })); +} + +function shouldSkip(presence) { + return presence.every((p) => !p.value); +} + +function missingSecrets(presence) { + return presence.filter((p) => !p.value).map((p) => p.label); +} + +async function rpc(url, method, params, timeoutMs = 30_000) { + const controller = new AbortController(); + const timer = setTimeout(() => controller.abort(), timeoutMs); + try { + const res = await fetch(url, { + method: "POST", + headers: { "content-type": "application/json" }, + body: JSON.stringify({ jsonrpc: "2.0", id: 1, method, params }), + signal: controller.signal, + }); + const json = await res.json(); + if (json.error) { + throw new Error(`${method}: ${json.error.message || JSON.stringify(json.error)}`); + } + return json.result; + } finally { + clearTimeout(timer); + } +} + +async function getRelayerConfig(serverUrl) { + const url = `${serverUrl.replace(/\/$/, "")}/config`; + const controller = new AbortController(); + const timer = setTimeout(() => controller.abort(), 15_000); + try { + const res = await fetch(url, { signal: controller.signal }); + if (!res.ok) throw new Error(`GET ${url} → ${res.status}`); + return await res.json(); + } finally { + clearTimeout(timer); + } +} + +function inferNetwork(rpcUrl, explicit) { + if (explicit === "mainnet" || explicit === "testnet") return explicit; + const u = (rpcUrl || "").toLowerCase(); + if (u.includes("testnet")) return "testnet"; + if (u.includes("devnet")) return "devnet"; + if (u.includes("mainnet")) return "mainnet"; + return "mainnet"; +} + +function keypairFromDelegateKey(Ed25519Keypair, decodeSuiPrivateKey, raw) { + try { + if (raw.toLowerCase().startsWith("suiprivkey")) { + const { scheme, secretKey } = decodeSuiPrivateKey(raw); + if (scheme !== "ED25519") { + throw new Error(`delegate key must be Ed25519, got ${scheme}`); + } + return Ed25519Keypair.fromSecretKey(secretKey); + } + if (!HEX_32_RE.test(raw)) { + throw new Error("not hex"); + } + const hex = raw.startsWith("0x") || raw.startsWith("0X") ? raw.slice(2) : raw; + return Ed25519Keypair.fromSecretKey(Uint8Array.from(Buffer.from(hex, "hex"))); + } catch { + throw new Error("invalid delegate key (need 32-byte hex seed or suiprivkey1…)"); + } +} + +function parseAccountObject(objectId, result, packageId) { + const data = result?.data ?? result?.object ?? result; + const content = data?.content ?? {}; + const fields = data?.json ?? content?.fields ?? data?.fields; + const objectType = + data?.type ?? content?.type ?? result?.data?.type ?? result?.data?.content?.type; + if (!fields || typeof objectType !== "string") { + throw new Error(`account ${shortId(objectId)}: missing Move content`); + } + const typeParts = objectType.split("::"); + if (typeParts[1] !== "account" || typeParts[2] !== "MemWalAccount") { + throw new Error(`account ${shortId(objectId)} is ${objectType}, expected MemWalAccount`); + } + const typePkg = normalizeHexAddress(typeParts[0]); + const configuredPkg = normalizeHexAddress(packageId); + // Types keep the first-published package id across upgrades; PACKAGE_ID is + // the current policy package. Equality holds only until the first upgrade. + if (typePkg !== configuredPkg) { + console.log( + `note: ${shortId(objectId)} type package ${shortId(typePkg)} ≠ MEMWAL_PACKAGE_ID ${shortId(configuredPkg)} (expected after an upgrade)`, + ); + } + if (fields.active !== true) { + throw new Error(`account ${shortId(objectId)} is not active`); + } + if (typeof fields.owner !== "string") { + throw new Error(`account ${shortId(objectId)} has no owner`); + } + const rawCounter = fields.access_counter_version; + if (rawCounter === undefined || rawCounter === null) { + throw new Error(`account ${shortId(objectId)} has no access_counter_version`); + } + const delegates = []; + const rawKeys = fields.delegate_keys ?? []; + for (const entry of rawKeys) { + const d = entry?.fields ?? entry; + if (typeof d?.sui_address === "string") { + delegates.push(normalizeHexAddress(d.sui_address)); + } + } + return { + id: normalizeHexAddress(objectId), + owner: normalizeHexAddress(fields.owner), + counter: BigInt(rawCounter), + delegates, + typePackageId: typePkg, + objectType, + }; +} + +function roleOnAccount(address, account) { + const addr = normalizeHexAddress(address); + if (addr === account.owner) return "owner"; + if (account.delegates.includes(addr)) return "delegate"; + return null; +} + +export function extractAbortCode(inspect) { + const effects = inspect?.effects ?? inspect?.transactionEffects ?? inspect; + const status = effects?.status ?? inspect?.status; + if (!status) return { outcome: "unknown", detail: JSON.stringify(inspect).slice(0, 500) }; + + const kind = status.status ?? status; + if (kind === "success" || status.success === true) { + return { outcome: "success", detail: "success" }; + } + + const error = status.error ?? inspect?.error ?? effects?.error; + const text = typeof error === "string" ? error : JSON.stringify(error ?? status); + + // JSON-RPC Display: MoveAbort(, ) in command N. + // Location is MoveLocation { module: ModuleId { ... }, function, ... } and + // contains commas, so the abort code is the decimal after that location. + const moveAbort = text.match(/MoveAbort\([\s\S]*,\s*(\d+)\)/); + if (moveAbort) { + return { outcome: "abort", code: Number(moveAbort[1]), detail: text }; + } + if (error && typeof error === "object") { + const abort = error.MoveAbort ?? error.moveAbort; + if (Array.isArray(abort) && abort.length >= 2) { + return { outcome: "abort", code: Number(abort[1]), detail: text }; + } + if (abort && typeof abort === "object" && abort.abortCode !== undefined) { + return { outcome: "abort", code: Number(abort.abortCode), detail: text }; + } + } + const abortCode = text.match(/abort code:\s*(\d+)/i) ?? text.match(/"abortCode"\s*:\s*"?(\d+)/); + if (abortCode) { + return { outcome: "abort", code: Number(abortCode[1]), detail: text }; + } + return { outcome: "failure", detail: text }; +} + +function pureU8Vector(tx, bytes) { + const arr = Array.from(bytes); + if (tx.pure && typeof tx.pure.vector === "function") { + return tx.pure.vector("u8", arr); + } + return tx.pure("vector", arr); +} + +async function discoverRegistry(network, typePackageId) { + const endpoints = { + mainnet: "https://graphql.mainnet.sui.io/graphql", + testnet: "https://graphql.testnet.sui.io/graphql", + }; + const url = endpoints[network]; + if (!url) return ""; + const type = `${typePackageId}::account::AccountRegistry`; + const controller = new AbortController(); + const timer = setTimeout(() => controller.abort(), 15_000); + try { + const res = await fetch(url, { + method: "POST", + headers: { "content-type": "application/json" }, + body: JSON.stringify({ + query: "query ($type: String!) { objects(first: 5, filter: { type: $type }) { nodes { address } } }", + variables: { type }, + }), + signal: controller.signal, + }); + if (!res.ok) return ""; + const json = await res.json(); + const nodes = json?.data?.objects?.nodes; + if (!Array.isArray(nodes) || nodes.length === 0) return ""; + if (nodes.length > 1) { + throw new Error( + `GraphQL found ${nodes.length} AccountRegistry objects; set MEMWAL_REGISTRY_ID`, + ); + } + return typeof nodes[0]?.address === "string" ? nodes[0].address : ""; + } catch (err) { + if (err instanceof Error && err.message.includes("AccountRegistry")) throw err; + return ""; + } finally { + clearTimeout(timer); + } +} + +async function inspectSealApprove({ + client, + Transaction, + sender, + policyPackageId, + registryId, + accountId, + idBytes, +}) { + const tx = new Transaction(); + tx.setSenderIfNotSet?.(sender); + tx.moveCall({ + target: `${policyPackageId}::account::seal_approve`, + arguments: [ + pureU8Vector(tx, idBytes), + tx.object(registryId), + tx.object(accountId), + ], + }); + + if (typeof client.devInspectTransactionBlock === "function") { + return client.devInspectTransactionBlock({ + sender, + transactionBlock: tx, + }); + } + if (typeof client.core?.simulateTransaction === "function") { + return client.core.simulateTransaction({ sender, transaction: tx }); + } + + const bytes = await tx.build({ client, onlyTransactionKind: true }); + const b64 = Buffer.from(bytes).toString("base64"); + const url = client.transport?.url ?? client.url; + if (!url) { + throw new Error("Sui client has no JSON-RPC URL for sui_devInspectTransactionBlock"); + } + return rpc(url, "sui_devInspectTransactionBlock", [sender, b64, null, null]); +} + +function failSecurity(message) { + const line = `${FAIL_TOKEN}: ${message}`; + console.error(`::error::${line}`); + console.error(line); + console.error("Page on-call. Identity from one MemWal account authorized another account's SEAL key."); + process.exit(1); +} + +function failMisconfig(message) { + console.error(`synthetic-seal-cross-account: misconfig: ${message}`); + process.exit(2); +} + +async function run() { + const presence = secretPresence(); + if (shouldSkip(presence)) { + console.log( + "synthetic-seal-cross-account: skip (required env vars unset; safe in CI without secrets)", + ); + return; + } + const missing = missingSecrets(presence); + if (missing.length) { + failMisconfig(`partial secrets; missing ${missing.join(", ")}`); + } + + const accountAId = firstEnv(["SEAL_CROSS_ACCOUNT_A_ID", "MEMWAL_ACCOUNT_A_ID"]); + const accountBId = firstEnv(["SEAL_CROSS_ACCOUNT_B_ID", "MEMWAL_ACCOUNT_B_ID"]); + const keyA = firstEnv(["SEAL_CROSS_ACCOUNT_A_KEY", "MEMWAL_DELEGATE_KEY_A"]); + const keyB = firstEnv(["SEAL_CROSS_ACCOUNT_B_KEY", "MEMWAL_DELEGATE_KEY_B"]); + + if (!OBJECT_ID_RE.test(accountAId) || !OBJECT_ID_RE.test(accountBId)) { + failMisconfig("account ids must be 0x-prefixed 32-byte object ids"); + } + if (normalizeHexAddress(accountAId) === normalizeHexAddress(accountBId)) { + failMisconfig("account A and account B must be different objects"); + } + + let packageId = firstEnv(["MEMWAL_PACKAGE_ID", "PACKAGE_ID"]); + let rpcUrl = firstEnv(["SUI_RPC_URL"]); + let network = env("SUI_NETWORK"); + const serverUrl = firstEnv(["MEMWAL_SERVER_URL", "MEMWAL_RELAYER_URL"]); + + if ((!packageId || !rpcUrl) && serverUrl) { + const cfg = await getRelayerConfig(serverUrl); + packageId = packageId || cfg.packageId || ""; + rpcUrl = rpcUrl || cfg.suiRpcUrl || ""; + network = network || cfg.network || ""; + console.log(`loaded GET ${serverUrl.replace(/\/$/, "")}/config`); + } + if (!packageId) failMisconfig("MEMWAL_PACKAGE_ID unset (or GET /config)"); + if (!rpcUrl) failMisconfig("SUI_RPC_URL unset (or GET /config)"); + network = inferNetwork(rpcUrl, network); + + let suiMod; + try { + const [jsonRpc, txMod, keysMod, cryptoMod] = await Promise.all([ + import("@mysten/sui/jsonRpc"), + import("@mysten/sui/transactions"), + import("@mysten/sui/keypairs/ed25519"), + import("@mysten/sui/cryptography"), + ]); + suiMod = { jsonRpc, txMod, keysMod, cryptoMod }; + } catch (err) { + failMisconfig( + `cannot import @mysten/sui (${err instanceof Error ? err.message : err}). From repo root: pnpm install --frozen-lockfile`, + ); + } + + const { SuiJsonRpcClient } = suiMod.jsonRpc; + const { Transaction } = suiMod.txMod; + const { Ed25519Keypair } = suiMod.keysMod; + const { decodeSuiPrivateKey } = suiMod.cryptoMod; + if (typeof SuiJsonRpcClient !== "function" || typeof Transaction !== "function") { + failMisconfig("@mysten/sui JSON-RPC client or Transaction missing"); + } + + const client = new SuiJsonRpcClient({ url: rpcUrl, network }); + const keypairA = keypairFromDelegateKey(Ed25519Keypair, decodeSuiPrivateKey, keyA); + const keypairB = keypairFromDelegateKey(Ed25519Keypair, decodeSuiPrivateKey, keyB); + const addrA = normalizeHexAddress(keypairA.getPublicKey().toSuiAddress()); + const addrB = normalizeHexAddress(keypairB.getPublicKey().toSuiAddress()); + if (addrA === addrB) { + failMisconfig("delegate keys A and B derive the same Sui address"); + } + + const readAccount = async (id) => { + let result; + if (typeof client.getObject === "function") { + result = await client.getObject({ + objectId: id, + id, + include: { json: true, type: true }, + options: { showContent: true, showType: true }, + }); + } else { + result = await rpc(rpcUrl, "sui_getObject", [id, { showContent: true, showType: true }]); + } + return parseAccountObject(id, result, packageId); + }; + + const accountA = await readAccount(accountAId); + const accountB = await readAccount(accountBId); + if (accountA.owner === accountB.owner) { + failMisconfig("account A and B share an owner; pick two isolated accounts"); + } + + const roleAonA = roleOnAccount(addrA, accountA); + const roleBonB = roleOnAccount(addrB, accountB); + const roleAonB = roleOnAccount(addrA, accountB); + const roleBonA = roleOnAccount(addrB, accountA); + if (!roleAonA) { + failMisconfig(`key A (${shortId(addrA)}) is not owner/delegate of account A`); + } + if (!roleBonB) { + failMisconfig(`key B (${shortId(addrB)}) is not owner/delegate of account B`); + } + if (roleAonB) { + failMisconfig(`key A is also ${roleAonB} on account B; accounts are not isolated`); + } + if (roleBonA) { + failMisconfig(`key B is also ${roleBonA} on account A; accounts are not isolated`); + } + + const policyPackageId = + firstEnv(["MEMWAL_SEAL_POLICY_PACKAGE_ID", "SEAL_POLICY_PACKAGE_ID"]) || packageId; + let registryId = firstEnv(["MEMWAL_REGISTRY_ID", "REGISTRY_ID"]); + if (!registryId) { + registryId = await discoverRegistry(network, accountA.typePackageId); + } + if (!registryId || !OBJECT_ID_RE.test(normalizeHexAddress(registryId))) { + failMisconfig("MEMWAL_REGISTRY_ID unset and AccountRegistry discovery failed"); + } + registryId = normalizeHexAddress(registryId); + + console.log( + `synthetic-seal-cross-account: ${network} policy=${shortId(policyPackageId)} registry=${shortId(registryId)}`, + ); + console.log( + ` A ${shortId(accountA.id)} owner=${shortId(accountA.owner)} caller=${shortId(addrA)} (${roleAonA}) counter=${accountA.counter}`, + ); + console.log( + ` B ${shortId(accountB.id)} owner=${shortId(accountB.owner)} caller=${shortId(addrB)} (${roleBonB}) counter=${accountB.counter}`, + ); + + const inspect = (sender, accountId, ownerHex, counter) => + inspectSealApprove({ + client, + Transaction, + sender, + policyPackageId, + registryId, + accountId, + idBytes: sealKeyIdBytes(ownerHex, counter), + }); + + const sameAccount = [ + { + name: "A authorizes A (sanity)", + promise: inspect(addrA, accountA.id, accountA.owner, accountA.counter), + expect: "success", + }, + { + name: "B authorizes B (sanity)", + promise: inspect(addrB, accountB.id, accountB.owner, accountB.counter), + expect: "success", + }, + ]; + for (const step of sameAccount) { + const result = extractAbortCode(await step.promise); + if (result.outcome !== "success") { + failMisconfig( + `${step.name} did not succeed (${result.outcome}${result.code !== undefined ? ` ${result.code}` : ""}). Check keys, package, registry.`, + ); + } + console.log(` ok ${step.name}`); + } + + const negatives = [ + { + name: "A on B's account + B's SEAL id", + from: "A", + to: "B", + promise: inspect(addrA, accountB.id, accountB.owner, accountB.counter), + }, + { + name: "A on A's account + B's SEAL id", + from: "A", + to: "B", + promise: inspect(addrA, accountA.id, accountB.owner, accountB.counter), + }, + { + name: "B on A's account + A's SEAL id", + from: "B", + to: "A", + promise: inspect(addrB, accountA.id, accountA.owner, accountA.counter), + }, + { + name: "B on B's account + A's SEAL id", + from: "B", + to: "A", + promise: inspect(addrB, accountB.id, accountA.owner, accountA.counter), + }, + ]; + + for (const step of negatives) { + const result = extractAbortCode(await step.promise); + if (result.outcome === "success") { + failSecurity( + `identity ${step.from} authorized account ${step.to}'s SEAL key (${step.name}); expected ENoAccess`, + ); + } + if (result.outcome === "abort" && result.code === E_NO_ACCESS) { + console.log(` deny ${step.name} (ENoAccess)`); + continue; + } + failMisconfig( + `${step.name}: expected ENoAccess (100), got ${result.outcome}${result.code !== undefined ? ` ${result.code}` : ""}: ${result.detail.slice(0, 300)}`, + ); + } + + console.log("synthetic-seal-cross-account: ok (cross-account seal_approve denied as expected)"); +} + +const isMain = + Boolean(process.argv[1]) && + path.resolve(process.argv[1]) === fileURLToPath(import.meta.url); + +if (isMain) { + const args = process.argv.slice(2); + if (args.includes("--help") || args.includes("-h")) { + usage(); + process.exit(0); + } + + run().catch((err) => { + const msg = err instanceof Error ? err.message : String(err); + console.error(`synthetic-seal-cross-account: ${msg}`); + process.exit(2); + }); +} diff --git a/scripts/synthetic-seal-cross-account.test.mjs b/scripts/synthetic-seal-cross-account.test.mjs new file mode 100644 index 000000000..8c1bc80f6 --- /dev/null +++ b/scripts/synthetic-seal-cross-account.test.mjs @@ -0,0 +1,100 @@ +import assert from "node:assert/strict"; +import { spawnSync } from "node:child_process"; +import test from "node:test"; +import { fileURLToPath } from "node:url"; + +import { extractAbortCode } from "./synthetic-seal-cross-account.mjs"; + +const SCRIPT = fileURLToPath(new URL("./synthetic-seal-cross-account.mjs", import.meta.url)); +const E_NO_ACCESS = 100; + +// Verbatim JSON-RPC Display string from a sui_devInspectTransactionBlock +// effects.status.error (a string, not a structured object). Nested commas +// inside MoveLocation must not hide abort code 100. +const JSON_RPC_SEAL_ENOACCESS = + 'MoveAbort(MoveLocation { module: ModuleId { address: ..., name: Identifier("account") }, function: 25, instruction: 11, function_name: Some("seal_approve") }, 100) in command 0'; + +function inspectFailure(error) { + return { effects: { status: { status: "failure", error } } }; +} + +const SECRET_KEYS = [ + "SEAL_CROSS_ACCOUNT_A_ID", + "SEAL_CROSS_ACCOUNT_B_ID", + "SEAL_CROSS_ACCOUNT_A_KEY", + "SEAL_CROSS_ACCOUNT_B_KEY", + "MEMWAL_ACCOUNT_A_ID", + "MEMWAL_ACCOUNT_B_ID", + "MEMWAL_DELEGATE_KEY_A", + "MEMWAL_DELEGATE_KEY_B", +]; + +function spawnScript(envExtra = {}) { + const env = { ...process.env, ...envExtra }; + for (const key of SECRET_KEYS) { + if (!(key in envExtra)) delete env[key]; + } + return spawnSync(process.execPath, [SCRIPT], { + env, + encoding: "utf8", + }); +} + +test("JSON-RPC MoveAbort Display string with nested location commas is ENoAccess", () => { + const result = extractAbortCode(inspectFailure(JSON_RPC_SEAL_ENOACCESS)); + assert.equal(result.outcome, "abort"); + assert.equal(result.code, E_NO_ACCESS); +}); + +test("simple MoveAbort without nested location commas still parses", () => { + const result = extractAbortCode( + inspectFailure("MoveAbort(MoveLocation { module: 0x2::account }, 100) in command 0"), + ); + assert.equal(result.outcome, "abort"); + assert.equal(result.code, E_NO_ACCESS); +}); + +test("formatMoveAbortMessage abort code phrase is ENoAccess", () => { + const result = extractAbortCode( + inspectFailure( + "MoveAbort in 1st command, abort code: 100, in '0xabc::account::seal_approve' (instruction 11)", + ), + ); + assert.equal(result.outcome, "abort"); + assert.equal(result.code, E_NO_ACCESS); +}); + +test("structured MoveAbort object fallback still works", () => { + const result = extractAbortCode( + inspectFailure({ MoveAbort: [{ module: "account", function_name: "seal_approve" }, 100] }), + ); + assert.equal(result.outcome, "abort"); + assert.equal(result.code, E_NO_ACCESS); +}); + +test("same-account success is not classified as a deny", () => { + const result = extractAbortCode({ effects: { status: { status: "success" } } }); + assert.equal(result.outcome, "success"); +}); + +test("unrelated abort is misconfiguration, not a healthy deny", () => { + const result = extractAbortCode(inspectFailure("MoveAbort(MoveLocation { module: foo }, 1) in command 0")); + assert.equal(result.outcome, "abort"); + assert.equal(result.code, 1); +}); + +test("skips with exit 0 when required secrets are unset", () => { + const result = spawnScript(); + assert.equal(result.status, 0, result.stderr); + assert.match(result.stdout, /skip \(required env vars unset/); +}); + +test("partial secrets exit 2 as misconfiguration", () => { + const result = spawnScript({ + SEAL_CROSS_ACCOUNT_A_ID: `0x${"aa".repeat(32)}`, + SEAL_CROSS_ACCOUNT_B_ID: `0x${"bb".repeat(32)}`, + SEAL_CROSS_ACCOUNT_A_KEY: "aa".repeat(32), + }); + assert.equal(result.status, 2, result.stderr); + assert.match(result.stderr, /partial secrets/); +}); From 93004938117c33598a299fb45ba0a7748da0da80 Mon Sep 17 00:00:00 2001 From: ducnmm <165614309+ducnmm@users.noreply.github.com> Date: Sun, 6 Sep 2026 23:38:29 -0700 Subject: [PATCH 21/29] feat(server): expose writes ok|paused on /health (COMG-717) (#843) * feat(server): expose writes ok|paused on /health (COMG-717) * fix(server): reject writes when WRITES_PAUSED (COMG-717) Make WRITES_PAUSED write-path admission, not a health-only signal. remember, remember_bulk, remember_manual, and analyze return 503 {"error":"writes are paused"} while /health stays HTTP 200 with status ok and writes paused. Reads (recall, restore) stay available. * fix(server): drop stub writes-paused route tests (COMG-717) The throwaway router never mounted production remember/analyze/health handlers, so it could not catch a dropped gate. Keep the helper and 503 mapping unit test; do not grow a handler harness. --- apps/status/server.mjs | 3 +- apps/status/src/App.tsx | 17 +++-- docs/reference/environment-variables.md | 1 + docs/relayer/api-reference.md | 9 ++- services/server/.env.example | 8 +++ services/server/src/routes/admin.rs | 4 +- services/server/src/routes/analyze.rs | 1 + services/server/src/routes/remember.rs | 4 ++ services/server/src/types.rs | 87 ++++++++++++++++++++++--- 9 files changed, 117 insertions(+), 17 deletions(-) diff --git a/apps/status/server.mjs b/apps/status/server.mjs index bbc85e646..47ec71793 100644 --- a/apps/status/server.mjs +++ b/apps/status/server.mjs @@ -354,7 +354,8 @@ async function probeRelayer(name, rawBase, target) { } } - const reportedOk = isRecord(health) && health.status === 'ok' + const writesPaused = isRecord(health) && health.writes === 'paused' + const reportedOk = isRecord(health) && health.status === 'ok' && !writesPaused const status = response.ok ? (reportedOk ? 'operational' : 'degraded') : 'outage' return { diff --git a/apps/status/src/App.tsx b/apps/status/src/App.tsx index 1f993c294..959680876 100644 --- a/apps/status/src/App.tsx +++ b/apps/status/src/App.tsx @@ -11,6 +11,7 @@ interface HealthPayload { relayerVersion?: string apiVersion?: string mode?: string + writes?: string minSupportedSdk?: { typescript?: string python?: string @@ -100,6 +101,7 @@ interface ComponentRow { status: StatusKind uptimeLabel: string history: HistoryBucket[] + writesPaused?: boolean } interface IncidentDay { @@ -164,8 +166,9 @@ function getOverallStatus(snapshot: StatusSnapshot | null, loadState: LoadState) return 'monitoring' } -function getStatusTitle(status: StatusKind) { +function getStatusTitle(status: StatusKind, writesPaused = false) { if (status === 'operational') return 'All Systems Operational' + if (status === 'degraded' && writesPaused) return 'Writes Paused' if (status === 'degraded') return 'Degraded Performance' if (status === 'outage') return 'Service Disruption' return 'Checking System Status' @@ -245,6 +248,7 @@ function buildRows(snapshot: StatusSnapshot | null, loadState: LoadState): Compo status: 'outage', uptimeLabel: 'history unavailable', history: normalizeBuckets(null, 'outage'), + writesPaused: false, }, ] } @@ -256,6 +260,7 @@ function buildRows(snapshot: StatusSnapshot | null, loadState: LoadState): Compo status: component.status, uptimeLabel: formatUptime(history), history: normalizeBuckets(history, component.status), + writesPaused: component.health?.writes === 'paused', } }) } @@ -396,8 +401,9 @@ function calendarRangeLabel(months: CalendarMonth[]) { return `${months[0].label} to ${months[months.length - 1].label}` } -function StatusPill({ status }: { status: StatusKind }) { - return {statusLabel[status]} +function StatusPill({ status, writesPaused }: { status: StatusKind; writesPaused?: boolean }) { + const label = writesPaused ? 'Writes Paused' : statusLabel[status] + return {label} } function Header({ @@ -463,7 +469,7 @@ function ComponentStatusRow({ row }: { row: ComponentRow }) {

{row.name}

- +
@@ -1047,6 +1053,7 @@ export default function App() { }, [refresh]) const overallStatus = getOverallStatus(snapshot, loadState) + const writesPaused = (snapshot?.components ?? []).some((c) => c.health?.writes === 'paused') const rows = useMemo(() => buildRows(snapshot, loadState), [snapshot, loadState]) const uptimeRows = useMemo(() => rows.filter((row) => row.status !== 'monitoring'), [rows]) const productionHistory = snapshot?.histories?.['relayer-production'] @@ -1066,7 +1073,7 @@ export default function App() { {route === 'current' && ( <>
-

{getStatusTitle(overallStatus)}

+

{getStatusTitle(overallStatus, writesPaused)}

{(error || componentError || snapshot?.database?.error) && ( diff --git a/docs/reference/environment-variables.md b/docs/reference/environment-variables.md index 311ffbd45..55b554223 100644 --- a/docs/reference/environment-variables.md +++ b/docs/reference/environment-variables.md @@ -151,6 +151,7 @@ These are not all enforced at boot, but most real deployments need them. | `MCP_MAX_SESSIONS_PER_IP` | `16` | Maximum active MCP sessions from one source IP | | `MCP_MAX_NEW_SESSIONS_PER_IP_PER_MIN` | `30` | Maximum new MCP sessions opened by one source IP per minute | | `TRUSTED_PROXY_HOPS` | `0` | Number of trusted reverse-proxy hops to walk from the right of `X-Forwarded-For`; `0` ignores XFF and uses the TCP peer | +| `WRITES_PAUSED` | `false` | When `1` / `true` / `yes` / `on`, write routes (`POST /api/remember`, `/api/remember/manual`, `/api/remember/bulk`, `/api/analyze`) return HTTP 503 `{"error":"writes are paused"}`. `GET /health` stays HTTP 200 with `status: "ok"` and `writes: "paused"`. Reads (`recall`, `restore`, health) stay available | ### Notes diff --git a/docs/relayer/api-reference.md b/docs/relayer/api-reference.md index c7f3200a3..df431351f 100644 --- a/docs/relayer/api-reference.md +++ b/docs/relayer/api-reference.md @@ -80,7 +80,11 @@ These routes require no authentication. ### `GET /health` -Service liveness check. `status` is `"ok"` when the relayer process is up. `write_ready` is `true` when the encryption sidecar process answered its own `/health` (cached a few seconds). A write outage can still return HTTP 200 with `write_ready: false`. `write_ready: true` is sidecar liveness, not a guarantee that remember or analyze succeed. +Service liveness check. `status` is `"ok"` when the relayer process is up. HTTP 200 means the process is running, not that writes are accepted. + +`writes` is `"ok"` or `"paused"`. `"paused"` when `WRITES_PAUSED` is set (`1` / `true` / `yes`); empty or unset is `"ok"`. That flag is write-path admission, not a health-only signal: `POST /api/remember`, `/api/remember/manual`, `/api/remember/bulk`, and `/api/analyze` then return HTTP 503 with `{"error":"writes are paused"}`. `/health` itself stays HTTP 200 with `status: "ok"` and `writes: "paused"`, so clients can distinguish an intentional pause from an integrator bug. Reads (`recall`, `restore`, remember job status) stay available. + +`write_ready` is `true` when the encryption sidecar process answered its own `/health` (cached a few seconds). That is sidecar liveness only, not a write-pause flag and not a guarantee that remember or analyze succeed. A sidecar outage can still return HTTP 200 with `write_ready: false`. Use `writes`, not `write_ready`, for the pause signal. **Response:** @@ -107,7 +111,8 @@ Service liveness check. `status` is `"ok"` when the relayer process is up. `writ "extract": "extract.v1", "ask": "ask.v1" }, - "write_ready": true + "write_ready": true, + "writes": "ok" } ``` diff --git a/services/server/.env.example b/services/server/.env.example index 012e49c18..742892aae 100644 --- a/services/server/.env.example +++ b/services/server/.env.example @@ -218,6 +218,14 @@ RESTORE_REQUESTS_PER_OWNER_PER_MINUTE=10 # Local dev: ALLOWED_ORIGINS=http://localhost:3000 +# ───────────────────────────────────────────────────────────────────── +# Write pause (maintenance) +# ───────────────────────────────────────────────────────────────────── +# When set, remember / remember_manual / remember_bulk / analyze return +# HTTP 503 {"error":"writes are paused"}. GET /health stays HTTP 200 with +# status "ok" and writes "paused". Reads (recall, restore) stay available. +# WRITES_PAUSED=true + # ───────────────────────────────────────────────────────────────────── # Benchmark mode (RAG-quality benchmarks only — see benchmarks/README.md) # ───────────────────────────────────────────────────────────────────── diff --git a/services/server/src/routes/admin.rs b/services/server/src/routes/admin.rs index 36fdd7b42..82e453974 100644 --- a/services/server/src/routes/admin.rs +++ b/services/server/src/routes/admin.rs @@ -129,7 +129,8 @@ pub async fn stats( /// here means only "the server process is up," not "your delegate /// key/account ID are valid." A caller preflighting credentials before a /// signed call should not treat this as a substitute for that call -/// succeeding. +/// succeeding. `WRITES_PAUSED` does not change this status: `/health` +/// stays HTTP 200 with `writes: "paused"` while write routes return 503. pub async fn health(State(state): State>) -> Json { Json(HealthResponse { status: "ok".to_string(), @@ -149,6 +150,7 @@ pub async fn health(State(state): State>) -> Json ask: ASK_SYSTEM_PROMPT_VERSION.to_string(), }, write_ready: sidecar_write_ready(&state).await, + writes: writes_health_status(state.config.writes_paused), }) } diff --git a/services/server/src/routes/analyze.rs b/services/server/src/routes/analyze.rs index 9836251c8..e69e53837 100644 --- a/services/server/src/routes/analyze.rs +++ b/services/server/src/routes/analyze.rs @@ -121,6 +121,7 @@ pub async fn analyze( Extension(auth): Extension, Json(body): Json, ) -> Result<(StatusCode, Json), AppError> { + reject_if_writes_paused(state.config.writes_paused)?; if body.text.is_empty() { return Err(AppError::BadRequest("Text cannot be empty".into())); } diff --git a/services/server/src/routes/remember.rs b/services/server/src/routes/remember.rs index e0e62b75f..73bf3f2c9 100644 --- a/services/server/src/routes/remember.rs +++ b/services/server/src/routes/remember.rs @@ -754,6 +754,7 @@ pub async fn remember( Extension(auth): Extension, Json(body): Json, ) -> Result<(StatusCode, Json), AppError> { + reject_if_writes_paused(state.config.writes_paused)?; if body.text.is_empty() { return Err(AppError::BadRequest("Text cannot be empty".into())); } @@ -1245,6 +1246,7 @@ pub async fn remember_bulk( Extension(auth): Extension, Json(body): Json, ) -> Result<(StatusCode, Json), AppError> { + reject_if_writes_paused(state.config.writes_paused)?; // ── Validate ────────────────────────────────────────────────────────── if body.items.is_empty() { return Err(AppError::BadRequest("items cannot be empty".into())); @@ -1419,6 +1421,7 @@ pub async fn remember_manual( Extension(auth): Extension, Json(body): Json, ) -> Result, AppError> { + reject_if_writes_paused(state.config.writes_paused)?; if body.encrypted_data.is_empty() { return Err(AppError::BadRequest( "encrypted_data cannot be empty".into(), @@ -2090,6 +2093,7 @@ mod tests { trusted_proxy_hops: 0, allowed_origins: String::new(), benchmark_mode: false, + writes_paused: false, enable_memory_deletion: false, enable_security_delete: false, legacy_db_url: None, diff --git a/services/server/src/types.rs b/services/server/src/types.rs index 1f79c2033..93fc43802 100644 --- a/services/server/src/types.rs +++ b/services/server/src/types.rs @@ -398,6 +398,10 @@ pub struct Config { /// bypassing SEAL + Walrus. **Not for production.** Off by default; /// set `BENCHMARK_MODE=true` to enable. Surfaced via `GET /health`. pub benchmark_mode: bool, + /// Operator write-pause flag from `WRITES_PAUSED`. When true, write + /// routes reject with HTTP 503 and `/health` reports `writes: "paused"` + /// while staying HTTP 200. Distinct from sidecar liveness (`write_ready`). + pub writes_paused: bool, /// Master visibility flag for the memory-deletion feature family. pub enable_memory_deletion: bool, /// Selects the tracked, backend-built security-delete flow. The master @@ -602,6 +606,7 @@ impl Config { benchmark_mode: std::env::var("BENCHMARK_MODE") .map(|v| matches!(v.trim().to_ascii_lowercase().as_str(), "1" | "true" | "yes")) .unwrap_or(false), + writes_paused: env_bool("WRITES_PAUSED"), enable_memory_deletion: env_bool("ENABLE_MEMORY_DELETION"), enable_security_delete: env_bool("ENABLE_SECURITY_DELETE"), legacy_db_url: nonempty_env("LEGACY_DB_URL"), @@ -913,6 +918,30 @@ fn env_bool(name: &str) -> bool { .unwrap_or(false) } +/// `/health` `writes` wire value: `"paused"` when `WRITES_PAUSED` is set. +pub(crate) fn writes_health_status(paused: bool) -> String { + if paused { + "paused".to_string() + } else { + "ok".to_string() + } +} + +/// Stable client-facing body for write-path 503 when `WRITES_PAUSED` is set. +const WRITES_PAUSED_ERROR: &str = "writes are paused"; + +/// Reject write-path admission when `WRITES_PAUSED` is set. +/// +/// Shared by `remember`, `remember_bulk`, `remember_manual`, and `analyze` +/// so `/health` `writes: "paused"` and write rejection stay one flag. +pub(crate) fn reject_if_writes_paused(paused: bool) -> Result<(), AppError> { + if paused { + Err(AppError::WritesPaused(WRITES_PAUSED_ERROR.to_string())) + } else { + Ok(()) + } +} + fn parse_walrus_aggregator_urls(primary: &str, extra_csv: Option<&str>) -> Vec { let mut urls = Vec::new(); let mut push_unique = |raw: &str| { @@ -1331,9 +1360,7 @@ pub fn validate_namespace(namespace: &str) -> Result<(), AppError> { // paths, rejecting them here would strand any namespace already written // with one — unreadable via recall/ask/stats and undeletable via forget. if namespace.contains('\0') { - return Err(AppError::BadRequest( - "namespace contains a NUL byte".into(), - )); + return Err(AppError::BadRequest("namespace contains a NUL byte".into())); } Ok(()) } @@ -1878,6 +1905,10 @@ pub struct HealthResponse { /// This is sidecar liveness, not a guarantee that remember/analyze will /// succeed. `status` stays `"ok"` while the relayer process is up. pub write_ready: bool, + /// Write-path admission: `"ok"` or `"paused"`. `"paused"` when + /// `WRITES_PAUSED` is set; write routes then return HTTP 503. + /// Distinct from `write_ready`. `/health` stays HTTP 200. + pub writes: String, } /// prompt version constants surfaced on `/health`. See the @@ -2057,6 +2088,9 @@ pub enum AppError { /// silently-dropped turn into one retried with exponential backoff — /// closing the bench-completion gap diagnosed during the LME v2 run. UpstreamUnavailable(String), + /// Operator write pause (`WRITES_PAUSED`). HTTP 503 with a stable + /// client-visible message, distinct from transient upstream failures. + WritesPaused(String), } impl std::fmt::Display for AppError { @@ -2071,6 +2105,7 @@ impl std::fmt::Display for AppError { AppError::RateLimited(msg) => write!(f, "Rate Limited: {}", msg), AppError::QuotaExceeded(msg) => write!(f, "Quota Exceeded: {}", msg), AppError::UpstreamUnavailable(msg) => write!(f, "Upstream Unavailable: {}", msg), + AppError::WritesPaused(msg) => write!(f, "Writes Paused: {}", msg), } } } @@ -2102,6 +2137,9 @@ impl axum::response::IntoResponse for AppError { AppError::Conflict(msg) => (axum::http::StatusCode::CONFLICT, msg.clone()), AppError::RateLimited(msg) => (axum::http::StatusCode::TOO_MANY_REQUESTS, msg.clone()), AppError::QuotaExceeded(msg) => (axum::http::StatusCode::PAYMENT_REQUIRED, msg.clone()), + AppError::WritesPaused(msg) => { + (axum::http::StatusCode::SERVICE_UNAVAILABLE, msg.clone()) + } AppError::UpstreamUnavailable(msg) => { // log the upstream details server-side, return // 503 so the SDK / harness will retry per their @@ -2139,6 +2177,7 @@ impl AppError { AppError::RateLimited(_) => "rate_limited", AppError::QuotaExceeded(_) => "quota_exceeded", AppError::UpstreamUnavailable(_) => "upstream_unavailable", + AppError::WritesPaused(_) => "writes_paused", } } } @@ -2267,6 +2306,7 @@ mod tests { trusted_proxy_hops: 0, allowed_origins: String::new(), benchmark_mode: false, + writes_paused: false, enable_memory_deletion: false, enable_security_delete: false, legacy_db_url: None, @@ -2906,7 +2946,10 @@ mod tests { #[test] fn auth_clock_drift_accepts_exact_ceiling() { with_auth_clock_drift_env(Some("900"), || { - assert_eq!(configured_auth_clock_drift_secs(), MAX_AUTH_CLOCK_DRIFT_SECS); + assert_eq!( + configured_auth_clock_drift_secs(), + MAX_AUTH_CLOCK_DRIFT_SECS + ); }); } @@ -2914,24 +2957,36 @@ mod tests { fn auth_clock_drift_rejects_just_over_ceiling() { // Pin the exact inclusive boundary: 900 accepted, 901 falls back. with_auth_clock_drift_env(Some("901"), || { - assert_eq!(configured_auth_clock_drift_secs(), DEFAULT_AUTH_CLOCK_DRIFT_SECS); + assert_eq!( + configured_auth_clock_drift_secs(), + DEFAULT_AUTH_CLOCK_DRIFT_SECS + ); }); } #[test] fn auth_clock_drift_falls_back_when_env_exceeds_cap() { with_auth_clock_drift_env(Some("3600"), || { - assert_eq!(configured_auth_clock_drift_secs(), DEFAULT_AUTH_CLOCK_DRIFT_SECS); + assert_eq!( + configured_auth_clock_drift_secs(), + DEFAULT_AUTH_CLOCK_DRIFT_SECS + ); }); } #[test] fn auth_clock_drift_falls_back_on_negative_or_garbage() { with_auth_clock_drift_env(Some("-5"), || { - assert_eq!(configured_auth_clock_drift_secs(), DEFAULT_AUTH_CLOCK_DRIFT_SECS); + assert_eq!( + configured_auth_clock_drift_secs(), + DEFAULT_AUTH_CLOCK_DRIFT_SECS + ); }); with_auth_clock_drift_env(Some("not-a-number"), || { - assert_eq!(configured_auth_clock_drift_secs(), DEFAULT_AUTH_CLOCK_DRIFT_SECS); + assert_eq!( + configured_auth_clock_drift_secs(), + DEFAULT_AUTH_CLOCK_DRIFT_SECS + ); }); } @@ -3171,11 +3226,13 @@ mod tests { ask: "ask.v1".to_string(), }, write_ready: true, + writes: "ok".to_string(), }; let json = serde_json::to_value(&resp).unwrap(); assert_eq!(json["prompt_versions"]["extract"], "extract.v1"); assert_eq!(json["prompt_versions"]["ask"], "ask.v1"); assert_eq!(json["write_ready"], true); + assert_eq!(json["writes"], "ok"); assert_eq!( json["apiVersion"], crate::compatibility::RELAYER_API_VERSION @@ -3186,4 +3243,18 @@ mod tests { crate::compatibility::MIN_TYPESCRIPT_SDK_VERSION ); } + + #[tokio::test] + async fn writes_paused_maps_to_503_with_stable_message() { + assert_eq!(writes_health_status(false), "ok"); + assert_eq!(writes_health_status(true), "paused"); + assert!(reject_if_writes_paused(false).is_ok()); + let err = reject_if_writes_paused(true).expect_err("paused writes"); + assert_eq!(err.kind(), "writes_paused"); + let resp = axum::response::IntoResponse::into_response(err); + assert_eq!(resp.status(), axum::http::StatusCode::SERVICE_UNAVAILABLE); + let body = axum::body::to_bytes(resp.into_body(), 4096).await.unwrap(); + let json: serde_json::Value = serde_json::from_slice(&body).unwrap(); + assert_eq!(json["error"], WRITES_PAUSED_ERROR); + } } From 3683d6c8c63823334312cf602978a716d93e6550 Mon Sep 17 00:00:00 2001 From: ducnmm <165614309+ducnmm@users.noreply.github.com> Date: Sun, 6 Sep 2026 23:38:52 -0700 Subject: [PATCH 22/29] fix(auth): don't treat Sui RPC 429 as a revoked delegate key (WALM-429) (#811) * fix(auth): don't treat Sui RPC 429 as a revoked delegate key Cache-hit re-verify used to evict on any on-chain error, including gRPC 429. That produced empty 401s that the SDK mapped to memwal_login even when the key was still registered. Keep the cache (or return 503) when Sui is unavailable; only evict on a definitive miss. WALM-429 * fix(auth): fail closed on Sui RPC 429 instead of serving a stale cache Keep the cached mapping so a later verify can succeed, but return 503 with AUTH_UPSTREAM_UNAVAILABLE and Retry-After rather than authenticating a possibly-revoked key for up to 24h. SDKs only use the credential-outage copy when that header is present. * fix(auth): expose Retry-After and stop treating object-not-found as unavailable (WALM-429) Classify GetObject failures by gRPC status code so typo'd x-account-id and NOT_FOUND evict as sign-in failures, while 429/unavailable still 503 with the cache row kept. Expose Retry-After on CORS so browsers can read it. * fix(auth): keep field-parse errors as unavailable RpcError (WALM-429) NotFound is only for absent objects / unparseable ids / NOT_FOUND. A node that returns the object but omits json/fields stays RpcError so auth 503s and keeps the cache row instead of 401+evicting a live key. * chore(auth): slim WALM-429 review follow-up Fold GetObject status tests; drop tautological field-parse cases; trim comments. --- docs/python-sdk/changelog.mdx | 5 +- docs/sdk/changelog.mdx | 8 +- docs/troubleshooting/overview.md | 8 + packages/python-sdk-memwal/CHANGELOG.md | 1 + packages/python-sdk-memwal/memwal/client.py | 19 +- .../tests/test_auth_rejected_message.py | 23 +- packages/sdk/CHANGELOG.md | 4 + packages/sdk/src/manual.ts | 11 +- packages/sdk/src/memwal.ts | 11 +- packages/sdk/src/utils.ts | 13 + .../sdk/test/sanitize-server-error.test.mjs | 30 ++ services/server/src/auth.rs | 293 ++++++++++++++---- services/server/src/main.rs | 30 +- services/server/src/mcp_proxy.rs | 17 +- services/server/src/storage/db.rs | 10 +- services/server/src/storage/sui.rs | 177 +++++++---- 16 files changed, 502 insertions(+), 158 deletions(-) diff --git a/docs/python-sdk/changelog.mdx b/docs/python-sdk/changelog.mdx index 7fdeb6dc3..52bea2a57 100644 --- a/docs/python-sdk/changelog.mdx +++ b/docs/python-sdk/changelog.mdx @@ -29,7 +29,7 @@ questions: - What changes were made in memwal 0.1.4? - Where can I find the release history for the Walrus Memory Python SDK? answer: >- - The latest Python SDK release is 0.1.9. It rejects empty `remember_bulk_async` batches and misaligned relayer `job_ids`, and aligns restore `truncated` docs with WALM-431 retryable semantics. 0.1.8 added `dropped_count` on recall results, `write_ready` on health, `MemWalClockDriftError` for clock-drift 401s, and the `dev` relayer preset. + The latest Python SDK release is 0.1.9. It reports HTTP 503 as a retryable upstream outage instead of a credential failure, rejects empty `remember_bulk_async` batches and misaligned relayer `job_ids`, and aligns restore `truncated` docs with WALM-431 retryable semantics. 0.1.8 added `dropped_count` on recall results, `write_ready` on health, `MemWalClockDriftError` for clock-drift 401s, and the `dev` relayer preset. --- Track what's new, changed, and fixed in `memwal` (Python). @@ -38,10 +38,11 @@ For the latest version, see the [PyPI project page](https://pypi.org/project/mem ## 0.1.9 -This release aligns Python bulk remember with the TypeScript SDK by refusing empty batches and mismatched job ids. +This release reports HTTP 503 as a retryable upstream outage and aligns Python bulk remember with the TypeScript SDK by refusing empty batches and mismatched job ids. ### Fixed +- HTTP 503 from the relayer is reported as a retryable upstream outage, not a credential failure. - `remember_bulk_async` rejects an empty `items` list before the request and raises when the relayer returns a `job_ids` length that does not match the batch. - restore `truncated` docs now match WALM-431 retryable semantics. diff --git a/docs/sdk/changelog.mdx b/docs/sdk/changelog.mdx index a71a384d2..8e66ac7ae 100644 --- a/docs/sdk/changelog.mdx +++ b/docs/sdk/changelog.mdx @@ -28,18 +28,22 @@ questions: - When was bulk remember added to the Walrus Memory SDK? - What security improvements have been made to the MemWal SDK? answer: >- - The latest TypeScript SDK release is 0.1.6. It adds optional `created_at` on `recall()` results, plus `sort` and `scoringWeights` on `RecallOptions`. 0.1.5 added `dropped_count` on recall results and `write_ready` on health, switched `rememberManual` to sending `encryptedData`, and hardened hex decoding and error redaction. + The latest TypeScript SDK release is 0.1.6. It adds optional `created_at` on `recall()` results, plus `sort` and `scoringWeights` on `RecallOptions`. HTTP 503 from the relayer is reported as a retryable upstream outage instead of a sign-in failure. 0.1.5 added `dropped_count` on recall results and `write_ready` on health, switched `rememberManual` to sending `encryptedData`, and hardened hex decoding and error redaction. --- ## 0.1.6 -This release adds write-time on recall results and lets callers sort by recency or pass scoring weights. +This release adds write-time on recall results, lets callers sort by recency or pass scoring weights, and stops mapping relayer 503s to a sign-in failure. ### Added - `recall()` results include optional `created_at` (RFC3339 write-time of the stored fact). - `RecallOptions` accepts `sort: "relevance" | "recent"` and `scoringWeights`. `sort: "recent"` over-fetches semantic candidates (5x `limit`, capped at 50), orders them by write-time descending, then truncates to `limit`. +### Fixed + +- HTTP 503 from the relayer is reported as a retryable upstream outage, not a sign-in failure. Empty-body 401s still point at `memwal_login`. + ## 0.1.5 This release surfaces partial recall results and write-path health, and aligns manual remember with the relayer contract. diff --git a/docs/troubleshooting/overview.md b/docs/troubleshooting/overview.md index d49861e04..f97660bd1 100644 --- a/docs/troubleshooting/overview.md +++ b/docs/troubleshooting/overview.md @@ -56,6 +56,14 @@ To triage quickly, confirm 3 things in order: the key is listed under the correc A version mismatch is rarely the cause here. The relayer reports its minimum supported SDK version, which is TypeScript 0.0.4 at the time of writing, so 0.1.0 is supported. A true version mismatch surfaces as `MemWalCompatibilityError` rather than `AUTH_REJECTED`. +### Intermittent `isn't signed in` / `memwal_login` on recall + +**Symptom:** The same credentials recall successfully one moment and fail the next with `Walrus Memory isn't signed in. Call the memwal_login tool, then retry.` `memwal_login` does not fix it. Other clients or a retry a few seconds later succeed. + +**Cause:** The relayer re-checks the delegate key on Sui on every signed request. When Sui gRPC `GetObject` returns 429 or is otherwise unavailable, older relayers treated that as a revoked key, evicted the cache, and returned an empty HTTP 401. The SDK maps empty 401s to the login hint. The key is usually still registered. + +**Fix:** Retry. This is not a sign-in failure. Current relayers keep the cached mapping when Sui cannot be consulted, or return HTTP 503 `upstream unavailable` on a cache miss. A 503 from the SDK means "retry"; it does not mean call `memwal_login`. If every call 401s, including after a few minutes, then follow the `AUTH_REJECTED` checks above. + ## MCP connection issues This section covers problems that appear before the memory tools work. diff --git a/packages/python-sdk-memwal/CHANGELOG.md b/packages/python-sdk-memwal/CHANGELOG.md index db771be6f..ac409b98d 100644 --- a/packages/python-sdk-memwal/CHANGELOG.md +++ b/packages/python-sdk-memwal/CHANGELOG.md @@ -4,6 +4,7 @@ ### Fixed +- HTTP 503 with `x-auth-error: AUTH_UPSTREAM_UNAVAILABLE` is reported as a retryable credential-verification outage, not a sign-in failure. Other 503s keep the generic sanitized body. - `remember_bulk_async` rejects an empty `items` list before the request and raises when the relayer returns a `job_ids` length that does not match the batch. - restore `truncated` docs now match WALM-431 retryable semantics. diff --git a/packages/python-sdk-memwal/memwal/client.py b/packages/python-sdk-memwal/memwal/client.py index 7164fd719..adef8339e 100644 --- a/packages/python-sdk-memwal/memwal/client.py +++ b/packages/python-sdk-memwal/memwal/client.py @@ -97,6 +97,11 @@ "and dashboard credentials. Full troubleshooting: " "https://docs.wal.app/walrus-memory/troubleshooting/overview#401-auth_rejected-errors" ) +AUTH_UPSTREAM_UNAVAILABLE = "AUTH_UPSTREAM_UNAVAILABLE" +UPSTREAM_UNAVAILABLE_MESSAGE = ( + "Walrus Memory temporarily cannot verify credentials (upstream unavailable). " + "Retry; this is not a sign-in failure." +) logger = logging.getLogger("memwal") @@ -1285,6 +1290,8 @@ async def _signed_request( raise _HttpStatusError( status=response.status_code, body=err_text, + auth_error=response.headers.get("x-auth-error"), + retry_after=response.headers.get("retry-after"), ) return response.json() @@ -1328,15 +1335,25 @@ class _HttpStatusError(MemWalError): explicitly accepted). """ - def __init__(self, status: int, body: str) -> None: + def __init__( + self, + status: int, + body: str, + auth_error: str | None = None, + retry_after: str | None = None, + ) -> None: if status == 401: super().__init__(AUTH_REJECTED_MESSAGE) + elif status == 503 and auth_error == AUTH_UPSTREAM_UNAVAILABLE: + super().__init__(UPSTREAM_UNAVAILABLE_MESSAGE) else: super().__init__( f"Walrus Memory API error ({status}): {_redact_internal_urls(body)}" ) self.status = status self.body = body + self.auth_error = auth_error + self.retry_after = retry_after class MemWalRememberJobNotFound(MemWalError): diff --git a/packages/python-sdk-memwal/tests/test_auth_rejected_message.py b/packages/python-sdk-memwal/tests/test_auth_rejected_message.py index b2dbdfbb3..d514a092d 100644 --- a/packages/python-sdk-memwal/tests/test_auth_rejected_message.py +++ b/packages/python-sdk-memwal/tests/test_auth_rejected_message.py @@ -2,8 +2,29 @@ from __future__ import annotations -from memwal.client import AUTH_REJECTED_MESSAGE +from memwal.client import ( + AUTH_REJECTED_MESSAGE, + AUTH_UPSTREAM_UNAVAILABLE, + UPSTREAM_UNAVAILABLE_MESSAGE, + _HttpStatusError, +) def test_auth_rejected_message_points_to_troubleshooting_guide() -> None: assert "docs.wal.app/walrus-memory/troubleshooting/overview" in AUTH_REJECTED_MESSAGE + + +def test_auth_503_is_retryable_not_a_credential_failure() -> None: + err = _HttpStatusError( + 503, "upstream unavailable", auth_error=AUTH_UPSTREAM_UNAVAILABLE, retry_after="5" + ) + assert str(err) == UPSTREAM_UNAVAILABLE_MESSAGE + assert "sign-in" in str(err) + assert "401" not in str(err) + assert err.retry_after == "5" + + +def test_non_auth_503_keeps_generic_body() -> None: + err = _HttpStatusError(503, "Rate limiter temporarily unavailable") + assert "Rate limiter temporarily unavailable" in str(err) + assert "cannot verify credentials" not in str(err) diff --git a/packages/sdk/CHANGELOG.md b/packages/sdk/CHANGELOG.md index 4aaa09e2e..52d317df9 100644 --- a/packages/sdk/CHANGELOG.md +++ b/packages/sdk/CHANGELOG.md @@ -7,6 +7,10 @@ - `recall()` results include optional `created_at` (RFC3339 write-time of the stored fact). - `RecallOptions` accepts `sort: "relevance" | "recent"` and `scoringWeights`. `sort: "recent"` over-fetches semantic candidates (5x `limit`, capped at 50), orders them by write-time descending, then truncates to `limit`. +### Fixed + +- HTTP 503 with `x-auth-error: AUTH_UPSTREAM_UNAVAILABLE` is reported as a retryable credential-verification outage, not a sign-in failure. Other 503s keep the generic sanitized body. Empty-body 401s still point at `memwal_login`. + ## 0.1.5 ### Added diff --git a/packages/sdk/src/manual.ts b/packages/sdk/src/manual.ts index ba2ff152c..65205bdb5 100644 --- a/packages/sdk/src/manual.ts +++ b/packages/sdk/src/manual.ts @@ -889,14 +889,23 @@ export class MemWalManual { const clockDriftError = clockDriftErrorFromResponse(res); if (clockDriftError) throw clockDriftError; - const { message: sanitized, serverCode } = sanitizeServerError(res.status, raw); + const { message: sanitized, serverCode } = sanitizeServerError( + res.status, + raw, + res.headers.get("x-auth-error"), + ); const err = new Error(sanitized) as Error & { status?: number; serverCode?: string; + retryAfterSeconds?: number; cause?: string; }; err.status = res.status; if (serverCode) err.serverCode = serverCode; + const retryAfter = Number(res.headers.get("retry-after")); + if (Number.isFinite(retryAfter) && retryAfter > 0) { + err.retryAfterSeconds = retryAfter; + } err.cause = raw; throw err; } diff --git a/packages/sdk/src/memwal.ts b/packages/sdk/src/memwal.ts index ad610a8f4..f41f44ee6 100644 --- a/packages/sdk/src/memwal.ts +++ b/packages/sdk/src/memwal.ts @@ -1293,14 +1293,23 @@ export class MemWal { const clockDriftError = clockDriftErrorFromResponse(res); if (clockDriftError) throw clockDriftError; - const { message, serverCode } = sanitizeServerError(res.status, raw); + const { message, serverCode } = sanitizeServerError( + res.status, + raw, + res.headers.get("x-auth-error"), + ); const err = new Error(message) as Error & { status?: number; serverCode?: string; + retryAfterSeconds?: number; cause?: string; }; err.status = res.status; if (serverCode) err.serverCode = serverCode; + const retryAfter = Number(res.headers.get("retry-after")); + if (Number.isFinite(retryAfter) && retryAfter > 0) { + err.retryAfterSeconds = retryAfter; + } // Preserve raw body on `cause` for in-process debugging only. err.cause = raw; throw err; diff --git a/packages/sdk/src/utils.ts b/packages/sdk/src/utils.ts index 2835f8e54..e8190b24a 100644 --- a/packages/sdk/src/utils.ts +++ b/packages/sdk/src/utils.ts @@ -304,6 +304,7 @@ export function redactInternalUrls(text: string): string { export function sanitizeServerError( status: number, rawBody: string, + authError?: string | null, ): { message: string; raw: string; serverCode?: string } { // Number() so a string "401" (some MCP / HTTP paths) still hits this branch. if (Number(status) === 401) { @@ -321,6 +322,18 @@ export function sanitizeServerError( }; } + // Auth-path 503 only: Sui could not be consulted (WALM-429). Other + // relayer 503s (Redis, rate limiter, LLM) keep the generic sanitizer + // so they are not mislabeled as a credential-verification failure. + if (Number(status) === 503 && authError === "AUTH_UPSTREAM_UNAVAILABLE") { + return { + message: + "Walrus Memory temporarily cannot verify credentials (upstream unavailable). Retry; this is not a sign-in failure.", + raw: rawBody, + serverCode: "AUTH_UPSTREAM_UNAVAILABLE", + }; + } + const MAX = 200; let serverCode: string | undefined; let text = rawBody; diff --git a/packages/sdk/test/sanitize-server-error.test.mjs b/packages/sdk/test/sanitize-server-error.test.mjs index 4da755c23..23db9fd14 100644 --- a/packages/sdk/test/sanitize-server-error.test.mjs +++ b/packages/sdk/test/sanitize-server-error.test.mjs @@ -35,6 +35,36 @@ test("non-401 empty bodies still use the placeholder", () => { assert.equal(message, "Walrus Memory server error (500): "); }); +test("auth 503 is retryable credential-verification unavailability, not a login hint", () => { + const { message, serverCode } = sanitizeServerError( + 503, + "upstream unavailable", + "AUTH_UPSTREAM_UNAVAILABLE", + ); + assert.equal(serverCode, "AUTH_UPSTREAM_UNAVAILABLE"); + assert.match(message, /not a sign-in failure/); + assert.doesNotMatch(message, /memwal_login/); +}); + +test("non-auth 503 keeps a generic retryable body, not credential copy", () => { + const { message, serverCode } = sanitizeServerError( + 503, + "Rate limiter temporarily unavailable", + ); + assert.equal(serverCode, undefined); + assert.match(message, /Rate limiter temporarily unavailable/); + assert.doesNotMatch(message, /cannot verify credentials/); + assert.doesNotMatch(message, /memwal_login/); +}); + +test("empty-body 503 without the auth header is generic", () => { + const { message, serverCode } = sanitizeServerError(503, ""); + assert.equal(serverCode, undefined); + assert.equal(message, "Walrus Memory server error (503): "); + assert.doesNotMatch(message, /memwal_login/); + assert.doesNotMatch(message, /isn't signed in/); +}); + test("localhost sidecar URLs are stripped from error text", () => { const { message } = sanitizeServerError( 500, diff --git a/services/server/src/auth.rs b/services/server/src/auth.rs index c4f511b06..4317afbf4 100644 --- a/services/server/src/auth.rs +++ b/services/server/src/auth.rs @@ -50,11 +50,10 @@ async fn constant_time_reject() -> StatusCode { /// Machine-readable reason for a stale/future-dated timestamp, surfaced on the /// `x-auth-error` header so a client can distinguish clock drift from a bad -/// signature. Only emitted for the timestamp check: it depends solely on the -/// client's own clock, not on whether the account or key exists, so exposing it -/// leaks nothing about server-side identity state. Signature, nonce, and -/// account-resolution failures keep the bare uniform 401 (no reason header) so -/// they remain indistinguishable and cannot be used to enumerate accounts. +/// signature. Identity 401s (signature, nonce, account-resolution) keep the +/// bare uniform 401 so they cannot be used to enumerate accounts. 503 auth +/// unavailability uses the same header with `AUTH_UPSTREAM_UNAVAILABLE` — that +/// path does not leak whether the key exists. const ERR_TIMESTAMP_OUT_OF_BOUNDS: &str = "ERR_TIMESTAMP_OUT_OF_BOUNDS"; /// 401 carrying `x-auth-error: `, after the same constant delay as @@ -73,6 +72,72 @@ fn unsupported_legacy_sdk() -> StatusCode { StatusCode::UPGRADE_REQUIRED } +/// Machine-readable 503 from signed HTTP auth / MCP when Sui cannot be +/// consulted. Distinct from other relayer 503s (Redis, rate limiter, LLM) +/// so SDKs only print the credential-verification copy when this header is +/// present (WALM-429). +pub(crate) const AUTH_UPSTREAM_UNAVAILABLE: &str = "AUTH_UPSTREAM_UNAVAILABLE"; + +/// Short Retry-After for auth 503. Callers must backoff rather than +/// immediately re-hitting a 429ing Sui fullnode. Not a substitute for the +/// 100 ms 401 timing pad — that path stays 401-only. +pub(crate) const AUTH_UPSTREAM_RETRY_AFTER_SECS: u64 = 5; + +/// Matches the MCP proxy: Sui could not be consulted, so this is not a login +/// failure. Empty 401 here is what made the SDK print memwal_login (WALM-429). +/// +/// Fail-closed: do not authenticate from a cached mapping while the chain +/// is unreachable. Keep the cache row (do not evict), return 503 with +/// `x-auth-error: AUTH_UPSTREAM_UNAVAILABLE` and `Retry-After`. +pub(crate) fn upstream_unavailable() -> Response { + ( + StatusCode::SERVICE_UNAVAILABLE, + [ + ("retry-after", AUTH_UPSTREAM_RETRY_AFTER_SECS.to_string()), + ("x-auth-error", AUTH_UPSTREAM_UNAVAILABLE.to_string()), + ], + "upstream unavailable", + ) + .into_response() +} + +/// WALM-429: an RPC failure must not evict the cache row *and* must not +/// authenticate from it. +#[derive(Debug)] +enum CacheReverifyAction { + Authenticate { owner: String }, + UnavailableKeepCache { reason: String }, + Evict { reason: String }, +} + +fn cache_reverify_action(result: Result) -> CacheReverifyAction { + match result { + Ok(owner) => CacheReverifyAction::Authenticate { owner }, + Err(e) if e.is_unavailable() => CacheReverifyAction::UnavailableKeepCache { + reason: e.to_string(), + }, + Err(e) => CacheReverifyAction::Evict { + reason: e.to_string(), + }, + } +} + +/// Outcome of resolving a signed delegate key to a MemWal account. +enum AccountResolveError { + /// Identity could not be established. Maps to a timing-normalized 401. + Unauthorized(String), + /// On-chain lookup could not be completed. Maps to 503 — retryable. + Unavailable(String), +} + +impl std::fmt::Display for AccountResolveError { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + match self { + Self::Unauthorized(msg) | Self::Unavailable(msg) => write!(f, "{msg}"), + } + } +} + /// Whether a request whose signed timestamp is `age` seconds old (negative = /// future-dated) is fresh, given the accepted drift window. The window is /// inclusive and symmetric: `|age| <= drift`. `drift == 0` requires an exact @@ -326,12 +391,17 @@ pub async fn verify_signature( } // Step 2: Resolve account — cache → signed header hint/config fallback → registry scan - // Always use constant_time_reject so that timing of the resolution error - // ("account not found" vs "key not in account") cannot be observed by callers. + // Identity failures stay on constant_time_reject (bare 401) so "account not + // found" vs "key not in account" cannot be timed. RPC/scan unavailability + // is 503: a Sui 429 is not a revoke (WALM-429). let (account_id, owner) = match resolve_account(&state, &public_key_hex, &pk_array, account_id_hint).await { Ok(pair) => pair, - Err(e) => { + Err(AccountResolveError::Unavailable(e)) => { + tracing::warn!("Account resolution unavailable: {}", e); + return Ok(upstream_unavailable()); + } + Err(AccountResolveError::Unauthorized(e)) => { tracing::warn!("Account resolution failed: {}", e); return Err(constant_time_reject().await); } @@ -366,34 +436,48 @@ async fn resolve_account( public_key_hex: &str, pk_bytes: &[u8; 32], account_id_hint: Option, -) -> Result<(String, String), String> { +) -> Result<(String, String), AccountResolveError> { // Strategy 1: Check PostgreSQL cache if let Ok(Some((cached_account_id, _cached_owner))) = state.db.get_cached_account(public_key_hex).await { - // Verify the cached mapping is still valid onchain - match verify_delegate_key_onchain( - &state.http_client, - &state.config.sui_rpc_url, - state.sui_grpc_client.as_ref(), - &cached_account_id, - pk_bytes, - &state.config.package_id, - ) - .await - { - Ok(owner) => { + // Re-verify the cached mapping on-chain when Sui is reachable. + // A transient RPC failure is *not* a revoke: keep the row, but + // fail closed with 503 so a revoked key cannot ride a 24h cache + // through a Sui outage. Definitive misses evict. + match cache_reverify_action( + verify_delegate_key_onchain( + &state.http_client, + &state.config.sui_rpc_url, + state.sui_grpc_client.as_ref(), + &cached_account_id, + pk_bytes, + &state.config.package_id, + ) + .await, + ) { + CacheReverifyAction::Authenticate { owner } => { tracing::debug!("account resolved from cache: {}", cached_account_id); return Ok((cached_account_id, owner)); } - Err(_) => { - // Key was revoked on-chain. Delete the stale cache row - // immediately so subsequent requests don't loop: cache-hit → RPC fail → - // fall-through, burning RPC quota and generating log noise on every call. + CacheReverifyAction::UnavailableKeepCache { reason } => { + tracing::warn!( + "on-chain re-verify unavailable for key {} on account {} ({}); keeping cached mapping, not authenticating from it", + public_key_hex, + cached_account_id, + reason + ); + return Err(AccountResolveError::Unavailable(format!( + "on-chain re-verify unavailable for cached account {}: {}", + cached_account_id, reason + ))); + } + CacheReverifyAction::Evict { reason } => { tracing::warn!( - "delegate key {} revoked on-chain for account {}; evicting from cache", + "cached delegate key {} is stale for account {} ({}); evicting from cache", public_key_hex, - cached_account_id + cached_account_id, + reason ); let _ = state.db.delete_cached_key(public_key_hex).await; } @@ -410,7 +494,7 @@ async fn resolve_account( .as_deref() .or(state.config.memwal_account_id.as_deref()) { - let owner = verify_delegate_key_onchain( + match verify_delegate_key_onchain( &state.http_client, &state.config.sui_rpc_url, state.sui_grpc_client.as_ref(), @@ -419,32 +503,41 @@ async fn resolve_account( &state.config.package_id, ) .await - .map_err(|e| { - format!( - "exact account {} verification failed: {}", - exact_account_id, e - ) - })?; - - let _ = state - .db - .cache_delegate_key(public_key_hex, exact_account_id, &owner) - .await; - - tracing::debug!( - "account resolved from exact account id: {}", - exact_account_id - ); - return Ok((exact_account_id.to_string(), owner)); + { + Ok(owner) => { + let _ = state + .db + .cache_delegate_key(public_key_hex, exact_account_id, &owner) + .await; + + tracing::debug!( + "account resolved from exact account id: {}", + exact_account_id + ); + return Ok((exact_account_id.to_string(), owner)); + } + Err(e) if e.is_unavailable() => { + return Err(AccountResolveError::Unavailable(format!( + "exact account {} verification unavailable: {}", + exact_account_id, e + ))); + } + Err(e) => { + return Err(AccountResolveError::Unauthorized(format!( + "exact account {} verification failed: {}", + exact_account_id, e + ))); + } + } } // Strategy 3: The legacy registry scan uses JSON-RPC. Testnet no longer // serves JSON-RPC, so fail closed when a modern signed x-account-id hint // is absent instead of silently contacting a retired endpoint. if state.config.sui_network == "testnet" { - return Err( + return Err(AccountResolveError::Unauthorized( "x-account-id is required for delegate-key authentication on testnet".to_string(), - ); + )); } // Non-testnet compatibility path: scan AccountRegistry only when no exact @@ -453,19 +546,20 @@ async fn resolve_account( // unknown-key floods can't stack unbounded scans, and a per-scan page // cap (MEMWAL_REGISTRY_SCAN_MAX_PAGES) inside the scan itself. Both // rejection messages name the x-account-id remediation, but they surface - // only in server logs: the middleware collapses every auth failure to a - // bare 401 (no oracle). A key past the page cap therefore cannot - // self-resolve — operators must diagnose the lockout from the warn logs - // and either raise the cap or have the client send the header hint, - // which Strategy 2 verifies directly without any scan. + // only in server logs: the middleware collapses identity failures to a + // bare 401 (no oracle) and RPC/scan unavailability to 503. A key past the + // page cap therefore cannot self-resolve — operators must diagnose the + // lockout from the warn logs and either raise the cap or have the client + // send the header hint, which Strategy 2 verifies directly without any + // scan. let _scan_permit = match state.registry_scan_semaphore.try_acquire() { Ok(permit) => permit, Err(_) => { - return Err( + return Err(AccountResolveError::Unavailable( "registry scan concurrency limit reached; retry, or send the x-account-id \ header hint to skip the registry scan" .to_string(), - ); + )); } }; match find_account_by_delegate_key( @@ -486,19 +580,21 @@ async fn resolve_account( .await; return Ok((account_id, owner)); } - Err(e @ OnchainVerifyError::ScanCapExceeded(_)) => { - tracing::warn!("registry scan capped: {}", e); - return Err(format!( + Err(e) if e.is_unavailable() => { + tracing::warn!("registry scan unavailable: {}", e); + return Err(AccountResolveError::Unavailable(format!( "{}; send the x-account-id header hint to authenticate without a scan", e - )); + ))); } Err(e) => { tracing::debug!("registry scan did not find key: {}", e); } } - Err("no account found: not in cache, exact account id, or registry".to_string()) + Err(AccountResolveError::Unauthorized( + "no account found: not in cache, exact account id, or registry".to_string(), + )) } /// Combined auth dispatcher for `read_api_routes`: tries the owner-scoped @@ -767,8 +863,14 @@ mod tests { fn reported_repro_45s_offset_is_within_default_window() { // The issue's repro used a +45s client offset; it is comfortably inside // the default 300s window and is accepted (i.e. does not reproduce). - assert!(is_timestamp_fresh(45, crate::types::DEFAULT_AUTH_CLOCK_DRIFT_SECS)); - assert!(is_timestamp_fresh(-45, crate::types::DEFAULT_AUTH_CLOCK_DRIFT_SECS)); + assert!(is_timestamp_fresh( + 45, + crate::types::DEFAULT_AUTH_CLOCK_DRIFT_SECS + )); + assert!(is_timestamp_fresh( + -45, + crate::types::DEFAULT_AUTH_CLOCK_DRIFT_SECS + )); } #[test] @@ -795,8 +897,14 @@ mod tests { // Pin the exact derivation at default and ceiling so it cannot regress: // default: 2*300 + 300 = 900 // ceiling: 2*900 + 300 = 2100 - assert_eq!(nonce_ttl_secs(crate::types::DEFAULT_AUTH_CLOCK_DRIFT_SECS), 900); - assert_eq!(nonce_ttl_secs(crate::types::MAX_AUTH_CLOCK_DRIFT_SECS), 2100); + assert_eq!( + nonce_ttl_secs(crate::types::DEFAULT_AUTH_CLOCK_DRIFT_SECS), + 900 + ); + assert_eq!( + nonce_ttl_secs(crate::types::MAX_AUTH_CLOCK_DRIFT_SECS), + 2100 + ); } // ── Timestamp-drift reason header (safe to distinguish) ────── @@ -856,6 +964,65 @@ mod tests { assert_eq!(status, StatusCode::UNAUTHORIZED); } + #[tokio::test] + async fn upstream_unavailable_is_503_not_401() { + let resp = upstream_unavailable(); + assert_eq!(resp.status(), StatusCode::SERVICE_UNAVAILABLE); + assert_eq!( + resp.headers() + .get("x-auth-error") + .and_then(|v| v.to_str().ok()), + Some(AUTH_UPSTREAM_UNAVAILABLE), + ); + let retry_after = AUTH_UPSTREAM_RETRY_AFTER_SECS.to_string(); + assert_eq!( + resp.headers() + .get("retry-after") + .and_then(|v| v.to_str().ok()), + Some(retry_after.as_str()), + ); + assert_eq!(AUTH_UPSTREAM_RETRY_AFTER_SECS, 5); + let body = axum::body::to_bytes(resp.into_body(), 64).await.unwrap(); + assert_eq!(&body[..], b"upstream unavailable"); + } + + #[test] + fn resolve_account_cache_hit_rpc_error_keeps_row_without_authenticating() { + // WALM-429: a Sui 429 used to evict the cache and 401 a live key. + // Keep the row (so a later verify can succeed) but do not treat the + // cached mapping as authorization while the chain is unreachable. + let action = cache_reverify_action(Err(OnchainVerifyError::RpcError( + "gRPC GetObject failed: 429 Too Many Requests".into(), + ))); + assert!(matches!( + action, + CacheReverifyAction::UnavailableKeepCache { .. } + )); + } + + #[test] + fn resolve_account_cache_hit_key_not_found_evicts() { + let action = cache_reverify_action(Err(OnchainVerifyError::KeyNotFound("gone".into()))); + assert!(matches!(action, CacheReverifyAction::Evict { .. })); + } + + #[test] + fn resolve_account_cache_hit_object_not_found_evicts() { + // Typo'd x-account-id / gRPC NOT_FOUND is a sign-in failure, not a 503. + let action = cache_reverify_action(Err(OnchainVerifyError::NotFound( + "gRPC GetObject failed: not found".into(), + ))); + assert!(matches!(action, CacheReverifyAction::Evict { .. })); + } + + #[test] + fn resolve_account_cache_hit_ok_authenticates() { + match cache_reverify_action(Ok("0xowner".into())) { + CacheReverifyAction::Authenticate { owner } => assert_eq!(owner, "0xowner"), + other => panic!("expected authenticate, got {other:?}"), + } + } + #[test] fn unsupported_legacy_sdk_returns_upgrade_required() { assert_eq!(unsupported_legacy_sdk(), StatusCode::UPGRADE_REQUIRED); diff --git a/services/server/src/main.rs b/services/server/src/main.rs index c926563f3..5d4614dfc 100644 --- a/services/server/src/main.rs +++ b/services/server/src/main.rs @@ -60,11 +60,9 @@ fn security_delete_cors() -> CorsLayer { } /// CORS layer for the main relayer routes, scoped to the configured origins. -/// `allow_headers` are the request headers a browser may send on a signed -/// request; `expose_headers` lists the response headers a cross-origin client -/// may read — Fetch hides everything else, so `x-auth-error` must be exposed -/// for the browser SDK to read the machine-readable auth-failure reason (e.g. -/// clock-drift vs. bad signature). Only that header is exposed. +/// Fetch hides response headers unless listed in `expose_headers`, so +/// `x-auth-error` and `Retry-After` must be exposed for the browser SDK +/// (clock-drift vs bad signature, and 503 backoff). fn relayer_cors(origins: Vec) -> CorsLayer { CorsLayer::new() .allow_origin(AllowOrigin::list(origins)) @@ -90,7 +88,10 @@ fn relayer_cors(origins: Vec) -> CorsLayer { // so this custom header must be preflight-allowed) "x-admin-api-key".parse::().unwrap(), ]) - .expose_headers(["x-auth-error".parse::().unwrap()]) + .expose_headers([ + "x-auth-error".parse::().unwrap(), + header::RETRY_AFTER, + ]) } #[cfg(test)] @@ -173,11 +174,7 @@ mod cors_tests { } #[tokio::test] - async fn relayer_cors_exposes_only_x_auth_error() { - // Browsers can only read response headers listed in - // Access-Control-Expose-Headers. The clock-drift reason (x-auth-error) - // must be exposed so the browser SDK can distinguish drift from a bad - // signature; nothing else should cross origins. + async fn relayer_cors_exposes_x_auth_error_and_retry_after() { let origin = "https://app.memwal.test"; let app = Router::new() .route("/api/remember", post(|| async {})) @@ -208,10 +205,14 @@ mod cors_tests { names.iter().any(|n| n.eq_ignore_ascii_case("x-auth-error")), "x-auth-error must be exposed, got: {exposed}" ); + assert!( + names.iter().any(|n| n.eq_ignore_ascii_case("retry-after")), + "retry-after must be exposed, got: {exposed}" + ); assert_eq!( names.len(), - 1, - "only x-auth-error should be exposed, got: {exposed}" + 2, + "only x-auth-error and retry-after should be exposed, got: {exposed}" ); } } @@ -1394,8 +1395,7 @@ async fn main() { if let Err(e) = evict_state.db.prune_unconsumed_oauth_clients().await { tracing::error!("MCP OAuth client pruning failed: {}", e); } - if let Err(e) = evict_state.db.sweep_expired_tombstones().await - { + if let Err(e) = evict_state.db.sweep_expired_tombstones().await { tracing::error!("tombstone retention sweep failed: {}", e); } } diff --git a/services/server/src/mcp_proxy.rs b/services/server/src/mcp_proxy.rs index b8c695e30..2d5390820 100644 --- a/services/server/src/mcp_proxy.rs +++ b/services/server/src/mcp_proxy.rs @@ -186,9 +186,8 @@ async fn legacy_delegate_registered( .await { Ok(_) => McpAuthOutcome::Passthrough, - Err(crate::storage::sui::OnchainVerifyError::RpcError(msg)) - | Err(crate::storage::sui::OnchainVerifyError::ScanCapExceeded(msg)) => { - tracing::warn!(error = %msg, "mcp delegate on-chain verify unavailable"); + Err(err) if err.is_unavailable() => { + tracing::warn!(error = %err, "mcp delegate on-chain verify unavailable"); McpAuthOutcome::Unavailable } Err(err) => { @@ -353,9 +352,7 @@ pub async fn sse_proxy( McpAuthOutcome::Unauthorized(err) => { return oauth_unauthorized_response(&state, err.as_ref()) } - McpAuthOutcome::Unavailable => { - return (StatusCode::SERVICE_UNAVAILABLE, "upstream unavailable").into_response() - } + McpAuthOutcome::Unavailable => return crate::auth::upstream_unavailable(), }; if let Err(code) = apply_internal_headers( &mut forwarded, @@ -466,9 +463,7 @@ pub async fn messages_proxy( McpAuthOutcome::Unauthorized(err) => { return oauth_unauthorized_response(&state, err.as_ref()) } - McpAuthOutcome::Unavailable => { - return (StatusCode::SERVICE_UNAVAILABLE, "upstream unavailable").into_response() - } + McpAuthOutcome::Unavailable => return crate::auth::upstream_unavailable(), }; if let Err(code) = apply_internal_headers( &mut forwarded, @@ -586,9 +581,7 @@ pub async fn streamable_proxy( McpAuthOutcome::Unauthorized(err) => { return oauth_unauthorized_response(&state, err.as_ref()) } - McpAuthOutcome::Unavailable => { - return (StatusCode::SERVICE_UNAVAILABLE, "upstream unavailable").into_response() - } + McpAuthOutcome::Unavailable => return crate::auth::upstream_unavailable(), }; if let Err(code) = apply_internal_headers( &mut forwarded, diff --git a/services/server/src/storage/db.rs b/services/server/src/storage/db.rs index 5a223dfc2..09833120b 100644 --- a/services/server/src/storage/db.rs +++ b/services/server/src/storage/db.rs @@ -2199,10 +2199,12 @@ impl VectorDb { /// Immediately remove a single stale/revoked delegate key from the cache. /// - /// Called when `verify_delegate_key_onchain` returns `Err` for a cached entry, - /// meaning the key has been revoked on-chain. Without this, every subsequent - /// request with the revoked key would hit the cache, fail the RPC verify, log - /// noise, and waste an RPC call — in an infinite loop until TTL expiry. + /// Called when a cached entry's on-chain re-verify returns a *definitive* + /// miss (`KeyNotFound` / `AccountDeactivated` / `WrongObjectType`). Do + /// **not** call this for `RpcError` / `ScanCapExceeded` — those mean Sui + /// could not be consulted (WALM-429). Evicting on a 429 made every later + /// request miss the cache, burn a second GetObject, and 401 a still-valid + /// key. pub async fn delete_cached_key(&self, public_key_hex: &str) -> Result { let result = sqlx::query("DELETE FROM delegate_key_cache WHERE public_key = $1") .bind(public_key_hex) diff --git a/services/server/src/storage/sui.rs b/services/server/src/storage/sui.rs index 91698af21..9a4bcc88b 100644 --- a/services/server/src/storage/sui.rs +++ b/services/server/src/storage/sui.rs @@ -80,12 +80,12 @@ pub async fn verify_delegate_key_onchain( let result = rpc_response .result - .ok_or_else(|| OnchainVerifyError::RpcError("No result in RPC response".into()))?; + .ok_or_else(|| OnchainVerifyError::NotFound("No result in RPC response".into()))?; let content = result .data .and_then(|d| d.content) - .ok_or_else(|| OnchainVerifyError::RpcError("Object has no content".into()))?; + .ok_or_else(|| OnchainVerifyError::NotFound("Object has no content".into()))?; // #398: reject foreign/lookalike objects — verify the Move type against the // configured immutable type-origin package id before trusting any field. @@ -360,11 +360,11 @@ pub async fn list_delegate_keys_onchain( let result = rpc_response .result - .ok_or_else(|| OnchainVerifyError::RpcError("No result in RPC response".into()))?; + .ok_or_else(|| OnchainVerifyError::NotFound("No result in RPC response".into()))?; let content = result .data .and_then(|d| d.content) - .ok_or_else(|| OnchainVerifyError::RpcError("Object has no content".into()))?; + .ok_or_else(|| OnchainVerifyError::NotFound("Object has no content".into()))?; ensure_memwal_account_type( content.object_type.as_deref(), @@ -441,23 +441,30 @@ fn grpc_value_as_u64(v: &prost_types::Value) -> Option { } } -/// gRPC counterpart of `verify_delegate_key_onchain` above — same checks -/// (owner, active, delegate_keys membership), fetched via -/// LedgerService.GetObject instead of JSON-RPC's `sui_getObject`. -/// -/// The gRPC `.json` object representation is flatter than JSON-RPC's -/// `.fields` shape and encodes delegate key `public_key` as base64 (not a -/// byte-array) — verified live against real testnet objects while migrating -/// the sidecar and web app to gRPC for this same JSON-RPC sunset. -async fn verify_delegate_key_onchain_grpc( +fn parse_object_id(account_object_id: &str) -> Result { + account_object_id + .parse() + .map_err(|error| OnchainVerifyError::NotFound(format!("invalid object id: {error}"))) +} + +/// Classify LedgerService.GetObject failures by gRPC *code*, never by message +/// text. tonic's Display embeds the code's English name, so matching "not found" +/// in the string would also fire on INTERNAL errors that merely mention a +/// missing object. +fn map_get_object_status(status: tonic::Status) -> OnchainVerifyError { + match status.code() { + tonic::Code::NotFound | tonic::Code::InvalidArgument => { + OnchainVerifyError::NotFound(format!("gRPC GetObject failed: {status}")) + } + _ => OnchainVerifyError::RpcError(format!("gRPC GetObject failed: {status}")), + } +} + +async fn grpc_get_object( mut client: sui_rpc::Client, account_object_id: &str, - public_key_bytes: &[u8], - expected_type_origin_package_id: &str, -) -> Result { - let address: sui_sdk_types::Address = account_object_id - .parse() - .map_err(|e| OnchainVerifyError::RpcError(format!("invalid object id: {}", e)))?; +) -> Result { + let address = parse_object_id(account_object_id)?; let mut request = sui_rpc::proto::sui::rpc::v2::GetObjectRequest::new(&address); request.read_mask = Some(prost_types::FieldMask { paths: vec!["json".to_string(), "object_type".to_string()], @@ -476,11 +483,28 @@ async fn verify_delegate_key_onchain_grpc( started.elapsed(), ); - let object = response - .map_err(|e| OnchainVerifyError::RpcError(format!("gRPC GetObject failed: {}", e)))? + response + .map_err(map_get_object_status)? .into_inner() .object - .ok_or_else(|| OnchainVerifyError::RpcError("gRPC response missing object".into()))?; + .ok_or_else(|| OnchainVerifyError::NotFound("gRPC response missing object".into())) +} + +/// gRPC counterpart of `verify_delegate_key_onchain` above — same checks +/// (owner, active, delegate_keys membership), fetched via +/// LedgerService.GetObject instead of JSON-RPC's `sui_getObject`. +/// +/// The gRPC `.json` object representation is flatter than JSON-RPC's +/// `.fields` shape and encodes delegate key `public_key` as base64 (not a +/// byte-array) — verified live against real testnet objects while migrating +/// the sidecar and web app to gRPC for this same JSON-RPC sunset. +async fn verify_delegate_key_onchain_grpc( + client: sui_rpc::Client, + account_object_id: &str, + public_key_bytes: &[u8], + expected_type_origin_package_id: &str, +) -> Result { + let object = grpc_get_object(client, account_object_id).await?; // #398: verify the Move type before trusting any field (gRPC path). ensure_memwal_account_type( @@ -554,36 +578,11 @@ async fn verify_delegate_key_onchain_grpc( /// representation (unlike JSON-RPC's array-of-numbers), but `/agents` /// doesn't need the key bytes at all — only `sui_address`/`label`/`created_at`. async fn list_delegate_keys_onchain_grpc( - mut client: sui_rpc::Client, + client: sui_rpc::Client, account_object_id: &str, expected_type_origin_package_id: &str, ) -> Result, OnchainVerifyError> { - let address: sui_sdk_types::Address = account_object_id - .parse() - .map_err(|e| OnchainVerifyError::RpcError(format!("invalid object id: {}", e)))?; - let mut request = sui_rpc::proto::sui::rpc::v2::GetObjectRequest::new(&address); - request.read_mask = Some(prost_types::FieldMask { - paths: vec!["json".to_string(), "object_type".to_string()], - }); - - let started = std::time::Instant::now(); - let response = client.ledger_client().get_object(request).await; - let status_label = match &response { - Ok(_) => "200".to_string(), - Err(status) => status.code().to_string(), - }; - crate::observability::observe_external( - "sui_grpc", - "GetObject", - &status_label, - started.elapsed(), - ); - - let object = response - .map_err(|e| OnchainVerifyError::RpcError(format!("gRPC GetObject failed: {}", e)))? - .into_inner() - .object - .ok_or_else(|| OnchainVerifyError::RpcError("gRPC response missing object".into()))?; + let object = grpc_get_object(client, account_object_id).await?; ensure_memwal_account_type( object.object_type.as_deref(), @@ -842,7 +841,7 @@ pub async fn find_account_by_delegate_key( ); return Ok((account_id.to_string(), owner)); } - Err(OnchainVerifyError::KeyNotFound(_)) => { + Err(OnchainVerifyError::KeyNotFound(_) | OnchainVerifyError::NotFound(_)) => { continue; } Err(e) => { @@ -966,6 +965,10 @@ struct ObjectContent { pub enum OnchainVerifyError { RpcError(String), KeyNotFound(String), + /// The named object does not exist, the id is unparseable, or GetObject + /// returned no object. Distinct from `KeyNotFound` (the account exists + /// but this key is not in `delegate_keys`). + NotFound(String), /// Returned when MemWalAccount.active == false. /// Prevents deactivated accounts from authenticating. AccountDeactivated(String), @@ -983,6 +986,7 @@ impl std::fmt::Display for OnchainVerifyError { match self { OnchainVerifyError::RpcError(msg) => write!(f, "Sui RPC error: {}", msg), OnchainVerifyError::KeyNotFound(msg) => write!(f, "Key not found: {}", msg), + OnchainVerifyError::NotFound(msg) => write!(f, "Object not found: {}", msg), OnchainVerifyError::AccountDeactivated(msg) => { write!(f, "Account deactivated: {}", msg) } @@ -998,6 +1002,26 @@ impl std::fmt::Display for OnchainVerifyError { impl std::error::Error for OnchainVerifyError {} +impl OnchainVerifyError { + /// True when the chain could not be consulted, as opposed to a definitive + /// "this key is not registered / this account is dead / this object does + /// not exist" answer. + /// + /// HTTP signed auth and the MCP proxy must not treat these as a revoke: + /// a Sui gRPC 429 is `RpcError`, and logging it as "revoked on-chain" + /// produced intermittent empty 401s that the SDK mapped to memwal_login + /// (WALM-429). + pub fn is_unavailable(&self) -> bool { + match self { + Self::RpcError(_) | Self::ScanCapExceeded(_) => true, + Self::NotFound(_) + | Self::KeyNotFound(_) + | Self::AccountDeactivated(_) + | Self::WrongObjectType(_) => false, + } + } +} + /// Reject any object whose Move type is not /// `{type-origin-package}::account::MemWalAccount`. Sui preserves the original /// publish/type-origin id across package upgrades, so this must never be the @@ -1306,16 +1330,52 @@ mod tests { #[test] fn test_error_variants_are_distinct() { - // Confirm AccountDeactivated is separate from KeyNotFound - // (different auth failure modes → different handling in resolve_account) let deactivated = OnchainVerifyError::AccountDeactivated("msg".into()); let not_found = OnchainVerifyError::KeyNotFound("msg".into()); - // Both are Err variants but must match differently: assert!(matches!( deactivated, OnchainVerifyError::AccountDeactivated(_) )); assert!(matches!(not_found, OnchainVerifyError::KeyNotFound(_))); + assert!(!deactivated.is_unavailable()); + assert!(!not_found.is_unavailable()); + assert!(OnchainVerifyError::RpcError("429".into()).is_unavailable()); + assert!(OnchainVerifyError::ScanCapExceeded("cap".into()).is_unavailable()); + assert!(!OnchainVerifyError::WrongObjectType("type".into()).is_unavailable()); + assert!(!OnchainVerifyError::NotFound("missing object".into()).is_unavailable()); + } + + #[test] + fn get_object_status_classifies_by_grpc_code() { + for (status, unavailable) in [ + (tonic::Status::not_found("no such object"), false), + (tonic::Status::invalid_argument("bad object id"), false), + ( + tonic::Status::resource_exhausted("429 Too Many Requests"), + true, + ), + (tonic::Status::unavailable("fullnode down"), true), + (tonic::Status::deadline_exceeded("timeout"), true), + ( + tonic::Status::internal("internal error: object not found while loading state"), + true, + ), + ] { + let err = map_get_object_status(status); + assert_eq!(err.is_unavailable(), unavailable, "{err}"); + if unavailable { + assert!(matches!(err, OnchainVerifyError::RpcError(_)), "{err}"); + } else { + assert!(matches!(err, OnchainVerifyError::NotFound(_)), "{err}"); + } + } + } + + #[test] + fn invalid_object_id_is_not_unavailable() { + let err = parse_object_id("not-a-sui-object-id").unwrap_err(); + assert!(matches!(err, OnchainVerifyError::NotFound(_))); + assert!(!err.is_unavailable()); } // ── Deactivated account field parsing ──────────────────────── @@ -1326,7 +1386,8 @@ mod tests { assert!(json_account_active(&fields(r#"{"active":true}"#)).unwrap()); assert!(!json_account_active(&fields(r#"{"active":false}"#)).unwrap()); - assert!(json_account_active(&fields(r#"{}"#)).is_err()); + let missing = json_account_active(&fields(r#"{}"#)).unwrap_err(); + assert!(missing.is_unavailable()); assert!(json_account_active(&fields(r#"{"active":"false"}"#)).is_err()); } @@ -1348,7 +1409,8 @@ mod tests { assert!(grpc_account_active(&fields(Some(Kind::BoolValue(true)))).unwrap()); assert!(!grpc_account_active(&fields(Some(Kind::BoolValue(false)))).unwrap()); - assert!(grpc_account_active(&fields(None)).is_err()); + let missing = grpc_account_active(&fields(None)).unwrap_err(); + assert!(missing.is_unavailable()); assert!(grpc_account_active(&fields(Some(Kind::StringValue("false".into())))).is_err()); } @@ -1852,6 +1914,9 @@ mod tests { "0xcf6ad755a1cdff7217865c796778fabe5aa399cb0cf2eba986f4b582047229c6", ) .await; - assert!(result.is_err()); + assert!( + matches!(result, Err(OnchainVerifyError::NotFound(_))), + "missing object must be NotFound, not unavailable RpcError, got: {result:?}" + ); } } From f40885991f8ffea0fbb9ffe5a2036b68144905e7 Mon Sep 17 00:00:00 2001 From: HoangDucBach Date: Mon, 7 Sep 2026 11:19:00 +0700 Subject: [PATCH 23/29] fix(researcher): fail loud when AUTH_SECRET is missing during verify Read the key before jwtVerify's try so a short secret cannot look like an invalid cookie or a failed Enoki ownership check. --- apps/researcher/lib/auth/auth-secret.unit.test.ts | 15 +++++++++++++++ apps/researcher/lib/auth/enoki-challenge.ts | 5 ++++- apps/researcher/lib/auth/session.ts | 7 ++++++- 3 files changed, 25 insertions(+), 2 deletions(-) diff --git a/apps/researcher/lib/auth/auth-secret.unit.test.ts b/apps/researcher/lib/auth/auth-secret.unit.test.ts index ff90f568e..30b25fde4 100644 --- a/apps/researcher/lib/auth/auth-secret.unit.test.ts +++ b/apps/researcher/lib/auth/auth-secret.unit.test.ts @@ -105,6 +105,21 @@ for (const path of ["lib/auth/session.ts", "proxy.ts"]) { }); } +test("session.ts and enoki-challenge.ts read the key before the verify try", () => { + // jwtVerify's try treats every throw as a bad cookie / failed ownership + // check. The key must be read outside that catch so a missing AUTH_SECRET + // fails loud instead of looking like "not authenticated". + for (const path of ["lib/auth/session.ts", "lib/auth/enoki-challenge.ts"]) { + const source = readFileSync(resolve(path), "utf8"); + + assert.match(source, /const secret = getAuthSecretKey\(\);/); + assert.doesNotMatch( + source, + /jwtVerify\(\s*\w+\s*,\s*getAuthSecretKey\(\)/ + ); + } +}); + test("enoki-challenge.ts shares the one guard rather than its own copy", () => { const source = readFileSync(resolve("lib/auth/enoki-challenge.ts"), "utf8"); diff --git a/apps/researcher/lib/auth/enoki-challenge.ts b/apps/researcher/lib/auth/enoki-challenge.ts index 6ee16ae95..f07bfa7d0 100644 --- a/apps/researcher/lib/auth/enoki-challenge.ts +++ b/apps/researcher/lib/auth/enoki-challenge.ts @@ -102,9 +102,12 @@ export async function verifyAndConsumeEnokiChallenge({ return false; } + // A missing/short AUTH_SECRET must not look like a failed ownership check. + const secret = getAuthSecretKey(); + try { const address = normalizeSuiAddress(rawAddress); - const { payload } = await jwtVerify(token, getAuthSecretKey(), { + const { payload } = await jwtVerify(token, secret, { algorithms: ["HS256"], audience: "enoki-auth", issuer: "walrus-memory-researcher", diff --git a/apps/researcher/lib/auth/session.ts b/apps/researcher/lib/auth/session.ts index 001b389a9..cd83aa261 100644 --- a/apps/researcher/lib/auth/session.ts +++ b/apps/researcher/lib/auth/session.ts @@ -27,8 +27,13 @@ export async function getSession(): Promise<{ user: SessionUser } | null> { const token = cookieStore.get(COOKIE_NAME)?.value; if (!token) return null; + // Same order as proxy.ts: a missing/short AUTH_SECRET must throw, not look + // like an invalid cookie. /api/auth/* skips the proxy guard, so this is the + // path GET /api/auth/profile takes. + const secret = getAuthSecretKey(); + try { - const { payload } = await jwtVerify(token, getAuthSecretKey()); + const { payload } = await jwtVerify(token, secret); if (typeof payload.userId !== "string") return null; const user = await getUserById(payload.userId); From d64975f815c3ca165545e3215f1f80ef91fdba16 Mon Sep 17 00:00:00 2001 From: ducnmm <165614309+ducnmm@users.noreply.github.com> Date: Sun, 6 Sep 2026 23:45:55 -0700 Subject: [PATCH 24/29] fix(python-sdk): warn on plaintext remote server_url (WALM-452) (#836) * fix(python-sdk): warn on plaintext remote server_url (WALM-452) Fixes #748 * fix(python-sdk): redact credentials in plaintext server_url warning (WALM-452) Log only scheme/host/port so HTTPX userinfo is not written to memwal logs, while keeping the original URL for transport. --- docs/python-sdk/changelog.mdx | 5 +- packages/python-sdk-memwal/CHANGELOG.md | 1 + packages/python-sdk-memwal/memwal/client.py | 49 +++++++++- .../tests/test_normalize_server_url.py | 96 +++++++++++++++++++ 4 files changed, 148 insertions(+), 3 deletions(-) create mode 100644 packages/python-sdk-memwal/tests/test_normalize_server_url.py diff --git a/docs/python-sdk/changelog.mdx b/docs/python-sdk/changelog.mdx index 52bea2a57..bd4214549 100644 --- a/docs/python-sdk/changelog.mdx +++ b/docs/python-sdk/changelog.mdx @@ -29,7 +29,7 @@ questions: - What changes were made in memwal 0.1.4? - Where can I find the release history for the Walrus Memory Python SDK? answer: >- - The latest Python SDK release is 0.1.9. It reports HTTP 503 as a retryable upstream outage instead of a credential failure, rejects empty `remember_bulk_async` batches and misaligned relayer `job_ids`, and aligns restore `truncated` docs with WALM-431 retryable semantics. 0.1.8 added `dropped_count` on recall results, `write_ready` on health, `MemWalClockDriftError` for clock-drift 401s, and the `dev` relayer preset. + The latest Python SDK release is 0.1.9. It reports HTTP 503 as a retryable upstream outage instead of a credential failure, rejects empty `remember_bulk_async` batches and misaligned relayer `job_ids`, aligns restore `truncated` docs with WALM-431 retryable semantics, and warns when `server_url` uses plaintext HTTP on a non-localhost host without logging URL credentials. 0.1.8 added `dropped_count` on recall results, `write_ready` on health, `MemWalClockDriftError` for clock-drift 401s, and the `dev` relayer preset. --- Track what's new, changed, and fixed in `memwal` (Python). @@ -38,13 +38,14 @@ For the latest version, see the [PyPI project page](https://pypi.org/project/mem ## 0.1.9 -This release reports HTTP 503 as a retryable upstream outage and aligns Python bulk remember with the TypeScript SDK by refusing empty batches and mismatched job ids. +This release reports HTTP 503 as a retryable upstream outage, aligns Python bulk remember with the TypeScript SDK by refusing empty batches and mismatched job ids, and warns when `server_url` uses plaintext HTTP on a non-localhost host. ### Fixed - HTTP 503 from the relayer is reported as a retryable upstream outage, not a credential failure. - `remember_bulk_async` rejects an empty `items` list before the request and raises when the relayer returns a `job_ids` length that does not match the batch. - restore `truncated` docs now match WALM-431 retryable semantics. +- Warn when `server_url` uses plaintext `http://` against a non-localhost host, matching the TypeScript SDK `normalizeServerUrl` guard. Localhost, `127.0.0.1`, `::1`, and `*.localhost` are exempt; invalid URLs are left for the HTTP client to surface. The warning logs only scheme, host, and port so URL userinfo is not written to logs. ## 0.1.8 diff --git a/packages/python-sdk-memwal/CHANGELOG.md b/packages/python-sdk-memwal/CHANGELOG.md index ac409b98d..007dd8aca 100644 --- a/packages/python-sdk-memwal/CHANGELOG.md +++ b/packages/python-sdk-memwal/CHANGELOG.md @@ -7,6 +7,7 @@ - HTTP 503 with `x-auth-error: AUTH_UPSTREAM_UNAVAILABLE` is reported as a retryable credential-verification outage, not a sign-in failure. Other 503s keep the generic sanitized body. - `remember_bulk_async` rejects an empty `items` list before the request and raises when the relayer returns a `job_ids` length that does not match the batch. - restore `truncated` docs now match WALM-431 retryable semantics. +- Warn when `server_url` uses plaintext `http://` against a non-localhost host, matching the TypeScript SDK `normalizeServerUrl` guard. Localhost, `127.0.0.1`, `::1`, and `*.localhost` are exempt; invalid URLs are left for the HTTP client to surface. The warning logs only scheme, host, and port so URL userinfo is not written to logs. ## 0.1.8 diff --git a/packages/python-sdk-memwal/memwal/client.py b/packages/python-sdk-memwal/memwal/client.py index adef8339e..cb964b86a 100644 --- a/packages/python-sdk-memwal/memwal/client.py +++ b/packages/python-sdk-memwal/memwal/client.py @@ -35,6 +35,7 @@ import uuid from datetime import datetime, timezone from typing import Any, Dict, List, Optional, Sequence, Tuple, TypeVar, Union +from urllib.parse import ParseResult, urlparse import httpx import nacl.signing @@ -107,6 +108,52 @@ logger = logging.getLogger("memwal") +def _server_url_for_log(parsed: ParseResult) -> str: + """Scheme/host/port only — never userinfo, path, query, or fragment.""" + + host = parsed.hostname or "" + if ":" in host: + host = f"[{host}]" + if parsed.port is not None: + return f"{parsed.scheme}://{host}:{parsed.port}" + return f"{parsed.scheme}://{host}" + + +def normalize_server_url(url: str) -> str: + """Strip a trailing slash and warn on plaintext HTTP to a remote host. + + Ports the TypeScript ``normalizeServerUrl`` helper: localhost, + ``127.0.0.1``, ``::1``, and ``*.localhost`` are exempt. Invalid URLs + are returned trimmed so the HTTP client can surface the error later. + + The warning logs only scheme/host/port so URL userinfo (HTTPX + credentials) and other sensitive components are not written to logs. + The returned URL is otherwise unchanged and still used for transport. + """ + + trimmed = url.rstrip("/") + try: + parsed = urlparse(trimmed) + host = (parsed.hostname or "").lower() + is_local = ( + host == "localhost" + or host == "127.0.0.1" + or host == "::1" + or host.endswith(".localhost") + ) + if parsed.scheme == "http" and host and not is_local: + logger.warning( + '[memwal] serverUrl "%s" uses plaintext HTTP on a non-localhost host. ' + "Signed requests and any bearer material will be visible to the network. " + "Use https:// in production.", + _server_url_for_log(parsed), + ) + except ValueError: + # invalid URL — let the HTTP call surface the error at request time + pass + return trimmed + + # ============================================================ # Polling helpers (PR #121 parity with TS SDK) # ============================================================ @@ -234,7 +281,7 @@ def __init__(self, config: MemWalConfig) -> None: self._private_key_hex = normalize_private_key(config.key) self._signing_key = build_signing_key(self._private_key_hex) self._account_id = config.account_id - self._server_url = config.server_url.rstrip("/") + self._server_url = normalize_server_url(config.server_url) self._namespace = config.namespace self._client: Optional[httpx.AsyncClient] = None self._server_config: Optional[Dict[str, str]] = None diff --git a/packages/python-sdk-memwal/tests/test_normalize_server_url.py b/packages/python-sdk-memwal/tests/test_normalize_server_url.py new file mode 100644 index 000000000..30189a2e6 --- /dev/null +++ b/packages/python-sdk-memwal/tests/test_normalize_server_url.py @@ -0,0 +1,96 @@ +"""Tests for plaintext HTTP server_url guarding (WALM-452 / #748). + +No network: ``MemWal.create`` only stores config, and +``normalize_server_url`` is a pure parse + log helper. +""" + +from __future__ import annotations + +import logging + +import pytest + +from memwal.client import MemWal, normalize_server_url + +_KEY = "ab" * 32 +_ACCOUNT = "0xdummy" + + +def test_plaintext_remote_warns_and_strips(caplog: pytest.LogCaptureFixture) -> None: + caplog.set_level(logging.WARNING, logger="memwal") + assert ( + normalize_server_url("http://relayer.example.com/") + == "http://relayer.example.com" + ) + assert "plaintext" in caplog.text + + +def test_https_remote_does_not_warn(caplog: pytest.LogCaptureFixture) -> None: + caplog.set_level(logging.WARNING, logger="memwal") + assert ( + normalize_server_url("https://relayer.memory.walrus.xyz") + == "https://relayer.memory.walrus.xyz" + ) + assert caplog.text == "" + + +@pytest.mark.parametrize( + "url", + [ + "http://localhost:8000", + "http://127.0.0.1:8000", + "http://foo.localhost", + "http://[::1]:8000", + ], +) +def test_plaintext_local_does_not_warn( + url: str, caplog: pytest.LogCaptureFixture +) -> None: + caplog.set_level(logging.WARNING, logger="memwal") + assert normalize_server_url(url) == url + assert caplog.text == "" + + +def test_create_warns_on_plaintext_remote(caplog: pytest.LogCaptureFixture) -> None: + caplog.set_level(logging.WARNING, logger="memwal") + client = MemWal.create( + key=_KEY, + account_id=_ACCOUNT, + server_url="http://relayer.example.com/", + ) + assert client._server_url == "http://relayer.example.com" + assert "plaintext" in caplog.text + + +def test_plaintext_remote_warning_omits_url_credentials( + caplog: pytest.LogCaptureFixture, +) -> None: + """HTTPX userinfo must stay on the transport URL and out of the warning.""" + + caplog.set_level(logging.WARNING, logger="memwal") + url = "http://alice:example-secret@relayer.example.com/?token=super-secret" + assert ( + normalize_server_url(url) + == "http://alice:example-secret@relayer.example.com/?token=super-secret" + ) + assert "plaintext" in caplog.text + assert "http://relayer.example.com" in caplog.text + assert "alice" not in caplog.text + assert "example-secret" not in caplog.text + assert "super-secret" not in caplog.text + for rec in caplog.records: + assert "example-secret" not in rec.getMessage() + assert "example-secret" not in str(rec.args) + assert "super-secret" not in rec.getMessage() + assert "super-secret" not in str(rec.args) + + +def test_create_preserves_url_credentials_and_redacts_warning( + caplog: pytest.LogCaptureFixture, +) -> None: + caplog.set_level(logging.WARNING, logger="memwal") + url = "http://alice:example-secret@relayer.example.com/" + client = MemWal.create(key=_KEY, account_id=_ACCOUNT, server_url=url) + assert client._server_url == "http://alice:example-secret@relayer.example.com" + assert "plaintext" in caplog.text + assert "example-secret" not in caplog.text From bb465a9e0d7c8f11c0f906551b4e0f8d1f04e066 Mon Sep 17 00:00:00 2001 From: Le Tien Phat <91601109+Niko1444@users.noreply.github.com> Date: Mon, 7 Sep 2026 13:47:12 +0700 Subject: [PATCH 25/29] fix(sdk): address namespace discovery review feedback --- .changeset/list-namespaces.md | 15 --------- docs/sdk/api-reference.md | 8 ++--- packages/sdk/src/memwal.ts | 23 +++++++------- packages/sdk/src/mock.ts | 36 +++++++++++++++++----- packages/sdk/src/types.ts | 2 +- packages/sdk/test/list-namespaces.test.mjs | 2 +- packages/sdk/test/mock.test.mjs | 35 +++++++++++++++++++++ 7 files changed, 81 insertions(+), 40 deletions(-) delete mode 100644 .changeset/list-namespaces.md diff --git a/.changeset/list-namespaces.md b/.changeset/list-namespaces.md deleted file mode 100644 index cff6f940b..000000000 --- a/.changeset/list-namespaces.md +++ /dev/null @@ -1,15 +0,0 @@ ---- -"@mysten-incubation/memwal": patch ---- - -Add `listNamespaces()` so an agent can discover which namespaces an account holds memories in (#634). - -Recall is similarity-ranked and needs a namespace to search. Without this, an agent connecting to an unfamiliar account had to guess names or fall back to `"default"`, which undercuts cross-session memory portability. - -`listNamespaces({ cursor?, limit? })` returns `{ namespaces, next_cursor, has_more, snapshot_version }` over the relayer's existing `GET /v1/owners/{owner}/namespaces`. Each entry carries `id`, `name`, `memory_count`, `storage_used` and `updated_at`. Metadata only — no blob fetch, no decryption, and no SEAL session is built or transmitted. - -Paginate on `has_more`, not on page length: the relayer clamps `limit`, so a caller asking for more than the cap gets exactly the cap back and would wrongly conclude it was done. - -The owner-scoped read routes take the address in the path and reject a mismatch, but `MemWalConfig` carries only the delegate key and account id — so the client resolves its own owner address once, memoised, and callers never supply it. - -`MemWalMock` implements the same method, aggregating seeded memories by namespace with deterministic timestamps derived from insertion order. diff --git a/docs/sdk/api-reference.md b/docs/sdk/api-reference.md index e9b3fcb34..ffe67e6c2 100644 --- a/docs/sdk/api-reference.md +++ b/docs/sdk/api-reference.md @@ -271,12 +271,12 @@ Rebuild missing indexed entries for one namespace from Walrus. Incremental — o ### `listNamespaces(options?): Promise` -List the namespaces this account holds memories in. Metadata only — no blob fetch and no decryption. +List the namespaces this account holds memories in. Returns metadata only, with no blob fetch or decryption. Recall is similarity-ranked and needs a namespace to search, so an agent connecting to an unfamiliar account would otherwise have to guess names or fall back to `"default"`. -- `options.cursor` — the previous page's `next_cursor`, to continue a walk or poll incrementally -- `options.limit` — page size; the relayer defaults to `100` and clamps to `500` +- `options.cursor`: The previous page's `next_cursor`, to continue a walk or poll incrementally +- `options.limit`: Page size; the relayer defaults to `100` and clamps to `500` **Returns:** @@ -295,7 +295,7 @@ Recall is similarity-ranked and needs a namespace to search, so an agent connect } ``` -Paginate on `has_more`, not on page length — the relayer clamps `limit`, so a caller asking for more than the cap gets exactly the cap back and would wrongly conclude it was done. +Paginate on `has_more`, not on page length. The relayer clamps `limit`, so a caller asking for more than the cap gets exactly the cap back and would wrongly conclude it was done. ```ts let cursor: string | undefined; diff --git a/packages/sdk/src/memwal.ts b/packages/sdk/src/memwal.ts index 5c32f28f7..5217a0122 100644 --- a/packages/sdk/src/memwal.ts +++ b/packages/sdk/src/memwal.ts @@ -195,17 +195,17 @@ export class MemWal { private sessionBuildPromise: Promise | null = null; /** Single-flight guard so concurrent requests share one compatibility probe. */ private compatibilityPromise: Promise | null = null; + /** Resolved owner address for this account. See `resolveOwner()`. */ + private ownerAddress: string | null = null; + /** Single-flight guard so concurrent reads share one owner resolution. */ + private ownerPromise: Promise | null = null; + /** * Keep a generated idempotency key while a remember request has no * acknowledged response. If the transport times out after the server * accepted the write, the caller's next identical attempt reuses the key * and collapses onto the original paid job. */ - /** Resolved owner address for this account. See `resolveOwner()`. */ - private ownerAddress: string | null = null; - /** Single-flight guard so concurrent reads share one owner resolution. */ - private ownerPromise: Promise | null = null; - private pendingRememberKeys = new Map(); private constructor(config: MemWalConfig) { @@ -918,9 +918,6 @@ export class MemWal { return { ...result, truncated: result.truncated ?? false }; } - /** - * Check server health. The endpoint is public and does not require request signing. - */ /** * List the namespaces this account holds memories in. * @@ -934,12 +931,13 @@ export class MemWal { * * ```ts * let cursor: string | undefined; - * do { + * let more = true; + * while (more) { * const page = await memwal.listNamespaces({ cursor }); * for (const ns of page.namespaces) console.log(ns.name, ns.memory_count); * cursor = page.next_cursor ?? undefined; - * var more = page.has_more; - * } while (more); + * more = page.has_more; + * } * ``` */ async listNamespaces(options: ListNamespacesOptions = {}): Promise { @@ -1004,6 +1002,9 @@ export class MemWal { return this.ownerPromise; } + /** + * Check server health. The endpoint is public and does not require request signing. + */ async health(): Promise { const res = await fetch(`${this.serverUrl}/health`); if (!res.ok) { diff --git a/packages/sdk/src/mock.ts b/packages/sdk/src/mock.ts index e896339ed..d3161b781 100644 --- a/packages/sdk/src/mock.ts +++ b/packages/sdk/src/mock.ts @@ -410,18 +410,38 @@ export class MemWalMock { : a.updated_at.localeCompare(b.updated_at) ); - // Mirrors the relayer's keyset walk: `cursor` is an exclusive - // `updated_after` watermark, and `has_more` — not page length — says - // whether to keep going. - const remaining = options.cursor - ? all.filter((ns) => ns.updated_at > options.cursor!) - : all; + // Match the relayer's URL_SAFE_NO_PAD JSON cursor and snapshot walk. + const cursor: { updated_at: string; namespace: string; snapshot_at?: string | null } | null = + options.cursor === undefined ? null : JSON.parse(new TextDecoder().decode( + Uint8Array.from( + atob(options.cursor.replace(/-/g, "+").replace(/_/g, "/")), + (char) => char.charCodeAt(0), + ), + )); + const snapshotAt = cursor?.snapshot_at ?? new Date( + MOCK_NAMESPACE_EPOCH_MS + this.sequence * 1000, + ).toISOString(); + const remaining = all.filter((ns) => + Date.parse(ns.updated_at) <= Date.parse(snapshotAt) && + (!cursor || Date.parse(ns.updated_at) > Date.parse(cursor.updated_at) || + (Date.parse(ns.updated_at) === Date.parse(cursor.updated_at) && + ns.name > cursor.namespace)) + ); const page = remaining.slice(0, options.limit ?? remaining.length); + const hasMore = remaining.length > page.length; + const last = page.at(-1); + const watermark = last ? { updated_at: last.updated_at, namespace: last.name } : cursor; + const nextCursor = watermark ? btoa(Array.from(new TextEncoder().encode(JSON.stringify({ + updated_at: watermark.updated_at, + namespace: watermark.namespace, + snapshot_at: hasMore ? snapshotAt : null, + })), (byte) => String.fromCharCode(byte)).join("")) + .replace(/\+/g, "-").replace(/\//g, "_").replace(/=+$/, "") : null; return { namespaces: page, - next_cursor: page.length ? page[page.length - 1].updated_at : null, - has_more: remaining.length > page.length, + next_cursor: nextCursor, + has_more: hasMore, // Matches the live relayer's current wire-format version. snapshot_version: 2, }; diff --git a/packages/sdk/src/types.ts b/packages/sdk/src/types.ts index de29cb2e1..eb140ee7f 100644 --- a/packages/sdk/src/types.ts +++ b/packages/sdk/src/types.ts @@ -404,7 +404,6 @@ export interface RecallManualHit { distance: number; } -/** Result from restore() */ /** One namespace in a `listNamespaces()` page. Mirrors the relayer wire shape. */ export interface NamespaceSummary { id: string; @@ -446,6 +445,7 @@ export interface ListNamespacesOptions { limit?: number; } +/** Result from restore() */ export interface RestoreResult { restored: number; skipped: number; diff --git a/packages/sdk/test/list-namespaces.test.mjs b/packages/sdk/test/list-namespaces.test.mjs index c6d0b7556..fe7bf6559 100644 --- a/packages/sdk/test/list-namespaces.test.mjs +++ b/packages/sdk/test/list-namespaces.test.mjs @@ -49,7 +49,7 @@ const ONE_PAGE = { ], next_cursor: "2026-08-20T10:00:00Z", has_more: false, - snapshot_version: 1, + snapshot_version: 2, }; test.afterEach(() => { diff --git a/packages/sdk/test/mock.test.mjs b/packages/sdk/test/mock.test.mjs index b4c23dace..45d2d928f 100644 --- a/packages/sdk/test/mock.test.mjs +++ b/packages/sdk/test/mock.test.mjs @@ -222,3 +222,38 @@ test("MemWalMock.listNamespaces reports the relayer's current snapshot_version", const page = await MemWalMock.create().listNamespaces(); assert.equal(page.snapshot_version, 2); }); + +test("MemWalMock namespace cursors use the relayer wire format and reset after a walk", async () => { + const mock = MemWalMock.create({ initialMemories: [ + { text: "a", namespace: "旅行" }, + { text: "b", namespace: "work" }, + ] }); + const first = await mock.listNamespaces({ limit: 1 }); + assert.match(first.next_cursor, /^[A-Za-z0-9_-]+$/); + const cursor = JSON.parse(Buffer.from(first.next_cursor, "base64url").toString("utf8")); + assert.equal(cursor.namespace, "旅行"); + assert.equal(cursor.updated_at, first.namespaces[0].updated_at); + assert.ok(cursor.snapshot_at); + const last = await mock.listNamespaces({ cursor: first.next_cursor }); + assert.deepEqual(last.namespaces.map(ns => ns.name), ["work"]); + assert.equal(last.has_more, false); + assert.equal(JSON.parse(Buffer.from(last.next_cursor, "base64url")).snapshot_at, null); + const empty = await mock.listNamespaces({ cursor: last.next_cursor }); + assert.deepEqual(empty.namespaces, []); + assert.equal(empty.next_cursor, last.next_cursor); +}); + +test("MemWalMock namespace walks defer new writes until the next poll", async () => { + const mock = MemWalMock.create({ initialMemories: [ + { text: "a", namespace: "alpha" }, + { text: "b", namespace: "bravo" }, + ] }); + const first = await mock.listNamespaces({ limit: 1 }); + await mock.remember("new", "bravo"); + const last = await mock.listNamespaces({ cursor: first.next_cursor }); + assert.deepEqual(last.namespaces, []); + assert.equal(last.has_more, false); + const poll = await mock.listNamespaces({ cursor: last.next_cursor }); + assert.deepEqual(poll.namespaces.map(ns => ns.name), ["bravo"]); + assert.equal(poll.namespaces[0].memory_count, 2); +}); From a9a760439347c818b5b7f488d11cb9b65e091327 Mon Sep 17 00:00:00 2001 From: HoangDucBach Date: Mon, 7 Sep 2026 11:19:00 +0700 Subject: [PATCH 26/29] docs(chatbot): drop bug-history comment on DELETE document 404 Match GET's identical guard, which has no comment. --- apps/chatbot/app/(chat)/api/document/route.ts | 2 -- 1 file changed, 2 deletions(-) diff --git a/apps/chatbot/app/(chat)/api/document/route.ts b/apps/chatbot/app/(chat)/api/document/route.ts index 3bd3c1731..5a514090c 100644 --- a/apps/chatbot/app/(chat)/api/document/route.ts +++ b/apps/chatbot/app/(chat)/api/document/route.ts @@ -113,8 +113,6 @@ export async function DELETE(request: Request) { const [document] = documents; - // Same guard the GET handler above runs. Without it an id that matches no row - // reads `.userId` off undefined and the request 500s with an empty body. if (!document) { return new ChatbotError("not_found:document").toResponse(); } From b11f0c07dc16c52e76c954fd71d42a5858336aa3 Mon Sep 17 00:00:00 2001 From: Hoang Duc Bach <124645604+HoangDucBach@users.noreply.github.com> Date: Mon, 7 Sep 2026 13:53:19 +0700 Subject: [PATCH 27/29] fix(chatbot): restore chat, titles and artifacts after upstream model retirements (WALM-354) (#822) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * fix(chatbot): replace the model ids OpenRouter retired Four of the seven ids in chatModels no longer exist upstream, and OpenRouter answers a retired id with a 404 at request time, so picking one only ever produced "Oops, an error occurred!". Two of them were also hardcoded at a call site: getTitleModel used google/gemini-2.0-flash-001, which left every chat named "New chat", and getArtifactModel used anthropic/claude-3.5-haiku, which broke document creation for everyone rather than only whoever selected it. Point the title and artifact models at ids the picker also offers, the way apps/researcher does after the same incident there, and centralise the "-thinking" handling so the marker cannot reach OpenRouter unstripped. * test(chatbot): fail CI when a curated model id is retired upstream The unit tests can only catch an id drifting away from chatModels; they cannot tell that OpenRouter has withdrawn one, which needs a live call. That gap is how the retired title model reached production: nothing in the repo changed, so there was no PR run to go red. Check the ids against the public catalog endpoint, which needs no API key. It gets its own workflow because test.yml has no schedule trigger to hang a weekly run off, and it limits pull-request runs to changes that touch the list so other PRs take no network dependency on OpenRouter. * fix(chatbot): cap the output tokens each model call asks for No call site set maxOutputTokens, so every request was quoted against the model's entire output window. OpenRouter reserves that credit up front for Anthropic and Google models, which rejected the turn outright — "requires more credits, you requested up to 65535 tokens" — however short the reply would have been. OpenAI and DeepSeek do not reserve, which is why only some of the picker appeared broken. Cap the chat turn, the background title call and all six artifact calls. The reasoning cap stays above the thinking budget the chat route requests so the answer is not truncated before it reaches the reply. * fix(chatbot): resolve a stale chat-model cookie to the default Both chat pages handed cookie.value straight to , which sends it as selectedChatModel, and /api/chat rejects an id that is no longer in chatModels. Anyone who had a since-retired model selected therefore had every send fail as a bad request, with nothing to explain it: the picker's fallback is display-only, so it rendered a valid model name while the request body still carried the stale id. Recovery meant reopening the picker, which the error text does not hint at. Coerce the cookie where it is read, matching resolveChatModelId in apps/researcher. * fix(chatbot): make the test mock answer the ids the app actually sends myProvider registers behaviours — chat-model, chat-model-reasoning — but getLanguageModel passed the catalog id through, so every request under PLAYWRIGHT threw NoSuchModelError: No such languageModel: openai/gpt-4o-mini. Every chat turn failed in CI, and the suite stayed green only because no test asserted that a reply arrives. getResponseForPrompt had the matching problem: it searched the whole serialised prompt, and the system prompt contains "hi" inside words like "this", so it always returned the greeting and the weather and default branches were unreachable. Match the newest user turn instead. * test(chatbot): cover the reply, title and reasoning paths end to end The suite exercised the composer and the picker but never asserted that a reply comes back, which is why it stayed green while every chat turn failed under PLAYWRIGHT. These six cover the paths the recent fixes touch: a streamed reply, a reply that survives a reload, intent reaching the model, the background title landing in the sidebar, reasoning streaming for a thinking model, and a retired chat-model cookie no longer rejecting every send. * test(chatbot): make the Playwright warm-up actually compile the routes The warm-up fetched with redirect: "manual", so the proxy's guest-session redirect ended it: every request returned 307 in under 60ms without rendering anything, and the cache stayed cold. Tests then paid the compile themselves — from a cleared .next, 4 to 14 of them failed on first-hit navigations of 40s to 1.3m against a 30s navigationTimeout and a 60s test timeout. CI never reported it because by the second of its two retries the routes were warm. Follow the redirects and keep their cookies, and warm the API and chat routes a message send needs rather than the landing page alone. The same cleared-.next run now warms in 84s and passes 28/28. * chore(chatbot): restore the trailing newline models.ts had Unrelated whitespace churn from the model-id fix. * test(chatbot): pass the warm-up request method explicitly fetchFollowing inferred POST from the path containing /api/chat, which hides the one request that is not a GET behind a string match. * fix(chatbot): cap the suggestions call the earlier sweep missed requestSuggestions streams through getArtifactModel, an Anthropic id, with no maxOutputTokens, so asking for suggestions on a document still hit the credit reservation the other call sites were fixed for. Count model calls against capped calls instead of listing the files, since a named list is what let this one through: it lives under lib/ai/tools rather than beside the other artifact handlers. * test(chatbot): cover createDocument artifact path in Playwright (WALM-354) The mock now emits createDocument for essay/document prompts so CI exercises ARTIFACT_MODEL the same way a retired id used to 404. --- .github/workflows/check-model-ids.yml | 57 ++++++ apps/chatbot/app/(chat)/actions.ts | 2 + apps/chatbot/app/(chat)/api/chat/route.ts | 15 +- apps/chatbot/app/(chat)/chat/[id]/page.tsx | 23 +-- apps/chatbot/app/(chat)/page.tsx | 24 +-- apps/chatbot/artifacts/code/server.ts | 3 + apps/chatbot/artifacts/sheet/server.ts | 3 + apps/chatbot/artifacts/text/server.ts | 3 + apps/chatbot/components/document-preview.tsx | 2 +- apps/chatbot/components/document.tsx | 1 + apps/chatbot/lib/ai/models.mock.ts | 162 ++++++++++++++++-- apps/chatbot/lib/ai/models.ts | 83 ++++++++- apps/chatbot/lib/ai/models.unit.test.ts | 151 ++++++++++++++++ apps/chatbot/lib/ai/providers.ts | 29 ++-- .../lib/ai/tools/request-suggestions.ts | 2 + apps/chatbot/package.json | 3 +- apps/chatbot/playwright.config.ts | 3 +- apps/chatbot/scripts/check-model-ids.ts | 48 ++++++ .../tests/playwright/e2e/conversation.test.ts | 140 +++++++++++++++ apps/chatbot/tests/playwright/global-setup.ts | 116 +++++++++++-- 20 files changed, 769 insertions(+), 101 deletions(-) create mode 100644 .github/workflows/check-model-ids.yml create mode 100644 apps/chatbot/lib/ai/models.unit.test.ts create mode 100644 apps/chatbot/scripts/check-model-ids.ts create mode 100644 apps/chatbot/tests/playwright/e2e/conversation.test.ts diff --git a/.github/workflows/check-model-ids.yml b/.github/workflows/check-model-ids.yml new file mode 100644 index 000000000..aac3a3f93 --- /dev/null +++ b/.github/workflows/check-model-ids.yml @@ -0,0 +1,57 @@ +name: Check model ids + +# Verifies the OpenRouter ids apps/chatbot can send still exist upstream. +# +# Its own workflow rather than a job in test.yml because the failure it catches +# arrives with no commit attached: OpenRouter retires ids on its own timetable, +# which is how a retired title model reached production unnoticed. test.yml has +# no schedule trigger, so a weekly job there would mean guarding every other job +# in the file against a scheduled run. +# +# The catalog endpoint is public, so this needs no key and no environment. + +on: + pull_request: + # Only when the curated list or the check itself moves. Every other PR would + # be taking a network dependency on OpenRouter for nothing. + paths: + - "apps/chatbot/lib/ai/models.ts" + - "apps/chatbot/scripts/check-model-ids.ts" + - ".github/workflows/check-model-ids.yml" + schedule: + # Reads nothing shared, so it does not need the bench-account offsets the + # other weekly suites coordinate around. + - cron: "0 8 * * 1" + workflow_dispatch: + +concurrency: + group: check-model-ids-${{ github.ref }} + cancel-in-progress: ${{ github.event_name == 'pull_request' }} + +permissions: + contents: read + +jobs: + model-ids: + name: Model ids live on OpenRouter + runs-on: ubuntu-latest + timeout-minutes: 5 + + steps: + - name: Checkout + uses: actions/checkout@v4 + + - name: Setup pnpm + uses: pnpm/action-setup@v4 + + - name: Setup Node + uses: actions/setup-node@v4 + with: + node-version: "22" + cache: pnpm + + - name: Install deps + run: pnpm install --frozen-lockfile + + - name: Check curated model ids against the live catalog + run: pnpm --filter @memwal/chatbot check:models diff --git a/apps/chatbot/app/(chat)/actions.ts b/apps/chatbot/app/(chat)/actions.ts index d4d1715b5..76587c2c4 100644 --- a/apps/chatbot/app/(chat)/actions.ts +++ b/apps/chatbot/app/(chat)/actions.ts @@ -3,6 +3,7 @@ import { generateText, type UIMessage } from "ai"; import { cookies } from "next/headers"; import type { VisibilityType } from "@/components/visibility-selector"; +import { MAX_TITLE_OUTPUT_TOKENS } from "@/lib/ai/models"; import { titlePrompt } from "@/lib/ai/prompts"; import { getTitleModel } from "@/lib/ai/providers"; import { @@ -35,6 +36,7 @@ export async function generateTitleFromUserMessage({ model: getTitleModel(), system: titlePrompt, prompt: getTextFromMessage(message), + maxOutputTokens: MAX_TITLE_OUTPUT_TOKENS, }); return text .replace(/^[#*"\s]+/, "") diff --git a/apps/chatbot/app/(chat)/api/chat/route.ts b/apps/chatbot/app/(chat)/api/chat/route.ts index 1c8535643..a956497f6 100644 --- a/apps/chatbot/app/(chat)/api/chat/route.ts +++ b/apps/chatbot/app/(chat)/api/chat/route.ts @@ -16,7 +16,12 @@ import { createResumableStreamContext } from "resumable-stream"; import { auth, type UserType } from "@/app/(auth)/auth"; import { entitlementsByUserType } from "@/lib/ai/entitlements"; import { memoryNamespaceForUser } from "@/lib/ai/memory-namespace"; -import { allowedModelIds } from "@/lib/ai/models"; +import { + allowedModelIds, + isReasoningModelId, + MAX_OUTPUT_TOKENS, + MAX_REASONING_OUTPUT_TOKENS, +} from "@/lib/ai/models"; import { type RequestHints, systemPrompt } from "@/lib/ai/prompts"; import { getLanguageModel, getMemWalModel } from "@/lib/ai/providers"; import { createDocument } from "@/lib/ai/tools/create-document"; @@ -177,10 +182,7 @@ export async function POST(request: Request) { }); } - const isReasoningModel = - selectedChatModel.endsWith("-thinking") || - (selectedChatModel.includes("reasoning") && - !selectedChatModel.includes("non-reasoning")); + const isReasoningModel = isReasoningModelId(selectedChatModel); const modelMessages = await convertToModelMessages(uiMessages); @@ -197,6 +199,9 @@ export async function POST(request: Request) { : getLanguageModel(selectedChatModel), system: systemPrompt({ selectedChatModel, requestHints }), messages: modelMessages, + maxOutputTokens: isReasoningModel + ? MAX_REASONING_OUTPUT_TOKENS + : MAX_OUTPUT_TOKENS, stopWhen: stepCountIs(5), experimental_activeTools: isReasoningModel ? [] diff --git a/apps/chatbot/app/(chat)/chat/[id]/page.tsx b/apps/chatbot/app/(chat)/chat/[id]/page.tsx index 1bd569376..f2c8b5ff8 100644 --- a/apps/chatbot/app/(chat)/chat/[id]/page.tsx +++ b/apps/chatbot/app/(chat)/chat/[id]/page.tsx @@ -5,7 +5,7 @@ import { Suspense } from "react"; import { auth } from "@/app/(auth)/auth"; import { Chat } from "@/components/chat"; import { DataStreamHandler } from "@/components/data-stream-handler"; -import { DEFAULT_CHAT_MODEL } from "@/lib/ai/models"; +import { resolveChatModelId } from "@/lib/ai/models"; import { getChatById, getMessagesByChatId } from "@/lib/db/queries"; import { convertToUIMessages } from "@/lib/utils"; @@ -48,30 +48,15 @@ async function ChatPage({ params }: { params: Promise<{ id: string }> }) { const uiMessages = convertToUIMessages(messagesFromDb); const cookieStore = await cookies(); - const chatModelFromCookie = cookieStore.get("chat-model"); - - if (!chatModelFromCookie) { - return ( - <> - - - - ); - } return ( <> - - - - ); - } - return ( <> ({ const { fullStream } = streamObject({ model: getArtifactModel(), system: codePrompt, + maxOutputTokens: MAX_OUTPUT_TOKENS, prompt: title, schema: z.object({ code: z.string(), @@ -45,6 +47,7 @@ export const codeDocumentHandler = createDocumentHandler<"code">({ const { fullStream } = streamObject({ model: getArtifactModel(), system: updateDocumentPrompt(document.content, "code"), + maxOutputTokens: MAX_OUTPUT_TOKENS, prompt: description, schema: z.object({ code: z.string(), diff --git a/apps/chatbot/artifacts/sheet/server.ts b/apps/chatbot/artifacts/sheet/server.ts index 4d90f2c96..d540e46b0 100644 --- a/apps/chatbot/artifacts/sheet/server.ts +++ b/apps/chatbot/artifacts/sheet/server.ts @@ -1,5 +1,6 @@ import { streamObject } from "ai"; import { z } from "zod"; +import { MAX_OUTPUT_TOKENS } from "@/lib/ai/models"; import { sheetPrompt, updateDocumentPrompt } from "@/lib/ai/prompts"; import { getArtifactModel } from "@/lib/ai/providers"; import { createDocumentHandler } from "@/lib/artifacts/server"; @@ -12,6 +13,7 @@ export const sheetDocumentHandler = createDocumentHandler<"sheet">({ const { fullStream } = streamObject({ model: getArtifactModel(), system: sheetPrompt, + maxOutputTokens: MAX_OUTPUT_TOKENS, prompt: title, schema: z.object({ csv: z.string().describe("CSV data"), @@ -51,6 +53,7 @@ export const sheetDocumentHandler = createDocumentHandler<"sheet">({ const { fullStream } = streamObject({ model: getArtifactModel(), system: updateDocumentPrompt(document.content, "sheet"), + maxOutputTokens: MAX_OUTPUT_TOKENS, prompt: description, schema: z.object({ csv: z.string(), diff --git a/apps/chatbot/artifacts/text/server.ts b/apps/chatbot/artifacts/text/server.ts index 8f7756cc3..75fa53731 100644 --- a/apps/chatbot/artifacts/text/server.ts +++ b/apps/chatbot/artifacts/text/server.ts @@ -1,4 +1,5 @@ import { smoothStream, streamText } from "ai"; +import { MAX_OUTPUT_TOKENS } from "@/lib/ai/models"; import { updateDocumentPrompt } from "@/lib/ai/prompts"; import { getArtifactModel } from "@/lib/ai/providers"; import { createDocumentHandler } from "@/lib/artifacts/server"; @@ -13,6 +14,7 @@ export const textDocumentHandler = createDocumentHandler<"text">({ system: "Write about the given topic. Markdown is supported. Use headings wherever appropriate.", experimental_transform: smoothStream({ chunking: "word" }), + maxOutputTokens: MAX_OUTPUT_TOKENS, prompt: title, }); @@ -41,6 +43,7 @@ export const textDocumentHandler = createDocumentHandler<"text">({ model: getArtifactModel(), system: updateDocumentPrompt(document.content, "text"), experimental_transform: smoothStream({ chunking: "word" }), + maxOutputTokens: MAX_OUTPUT_TOKENS, prompt: description, providerOptions: { openai: { diff --git a/apps/chatbot/components/document-preview.tsx b/apps/chatbot/components/document-preview.tsx index bda6487f3..7d7b3a611 100644 --- a/apps/chatbot/components/document-preview.tsx +++ b/apps/chatbot/components/document-preview.tsx @@ -102,7 +102,7 @@ export function DocumentPreview({ } return ( -
+
{ if (isReadonly) { toast.error( diff --git a/apps/chatbot/lib/ai/models.mock.ts b/apps/chatbot/lib/ai/models.mock.ts index 03a59bd5f..1a4303577 100644 --- a/apps/chatbot/lib/ai/models.mock.ts +++ b/apps/chatbot/lib/ai/models.mock.ts @@ -11,23 +11,128 @@ const mockUsage = { outputTokens: { total: 20, text: 20, reasoning: 0 }, }; +const GREETING_REGEX = /\b(hello|hi|hey)\b/; + +type ModelMessage = { + role?: string; + content?: unknown; +}; + +/** + * Read the newest user turn only. Matching the serialised prompt as a whole + * swept in the system prompt, whose prose contains "hi" inside ordinary words + * like "this", so every request looked like a greeting and the branches below + * could never be told apart from a test. + */ +function lastUserText(prompt: unknown): string { + if (!Array.isArray(prompt)) { + return ""; + } + + const userMessages = (prompt as ModelMessage[]).filter( + (message) => message?.role === "user" + ); + const latest = userMessages.at(-1); + + if (!latest) { + return ""; + } + if (typeof latest.content === "string") { + return latest.content.toLowerCase(); + } + if (!Array.isArray(latest.content)) { + return ""; + } + + return latest.content + .map((part: unknown) => + part && typeof part === "object" && "text" in part + ? String((part as { text: unknown }).text) + : "" + ) + .join(" ") + .toLowerCase(); +} + +const DOCUMENT_PROMPT_REGEX = /\b(essay|create a document|write a document)\b/; +const CREATE_DOCUMENT_CALL_ID = "call_doc"; +const CREATE_DOCUMENT_INPUT = JSON.stringify({ + title: "Test Artifact", + kind: "text", +}); + +function promptAlreadyCalledTools(prompt: unknown): boolean { + if (!Array.isArray(prompt)) { + return false; + } + for (const message of prompt as ModelMessage[]) { + if (message?.role === "tool") { + return true; + } + if (!Array.isArray(message?.content)) { + continue; + } + for (const part of message.content as Array<{ type?: string }>) { + if ( + part?.type === "tool-call" || + part?.type === "tool-result" || + part?.type === "tool-createDocument" + ) { + return true; + } + } + } + return false; +} + +function shouldCreateDocument(prompt: unknown): boolean { + return ( + !promptAlreadyCalledTools(prompt) && + DOCUMENT_PROMPT_REGEX.test(lastUserText(prompt)) + ); +} + function getResponseForPrompt(prompt: unknown): string { - const promptStr = JSON.stringify(prompt).toLowerCase(); + const text = lastUserText(prompt); - if (promptStr.includes("weather") || promptStr.includes("temperature")) { + if (text.includes("weather") || text.includes("temperature")) { return mockResponses.weather; } - if ( - promptStr.includes("hello") || - promptStr.includes("hi") || - promptStr.includes("hey") - ) { + if (GREETING_REGEX.test(text)) { return mockResponses.greeting; } return mockResponses.default; } +function enqueueCreateDocument(controller: ReadableStreamDefaultController) { + controller.enqueue({ + type: "tool-input-start", + id: CREATE_DOCUMENT_CALL_ID, + toolName: "createDocument", + }); + controller.enqueue({ + type: "tool-input-delta", + id: CREATE_DOCUMENT_CALL_ID, + delta: CREATE_DOCUMENT_INPUT, + }); + controller.enqueue({ + type: "tool-input-end", + id: CREATE_DOCUMENT_CALL_ID, + }); + controller.enqueue({ + type: "tool-call", + toolCallId: CREATE_DOCUMENT_CALL_ID, + toolName: "createDocument", + input: CREATE_DOCUMENT_INPUT, + }); + controller.enqueue({ + type: "finish", + finishReason: "tool-calls", + usage: mockUsage, + }); +} + const createMockModel = (): LanguageModel => { return { specificationVersion: "v3", @@ -35,13 +140,44 @@ const createMockModel = (): LanguageModel => { modelId: "mock-model", defaultObjectGenerationMode: "tool", supportedUrls: {}, - doGenerate: async ({ prompt }: { prompt: unknown }) => ({ - finishReason: "stop", - usage: mockUsage, - content: [{ type: "text", text: getResponseForPrompt(prompt) }], - warnings: [], - }), + doGenerate: async ({ prompt }: { prompt: unknown }) => { + if (shouldCreateDocument(prompt)) { + return { + finishReason: "tool-calls", + usage: mockUsage, + content: [ + { + type: "tool-call", + toolCallId: CREATE_DOCUMENT_CALL_ID, + toolName: "createDocument", + input: CREATE_DOCUMENT_INPUT, + }, + ], + warnings: [], + }; + } + return { + finishReason: "stop", + usage: mockUsage, + content: [{ type: "text", text: getResponseForPrompt(prompt) }], + warnings: [], + }; + }, doStream: ({ prompt }: { prompt: unknown }) => { + if (shouldCreateDocument(prompt)) { + return { + stream: new ReadableStream({ + async start(controller) { + await new Promise((resolve) => { + setTimeout(resolve, 500); + }); + enqueueCreateDocument(controller); + controller.close(); + }, + }), + }; + } + const response = getResponseForPrompt(prompt); const words = response.split(" "); diff --git a/apps/chatbot/lib/ai/models.ts b/apps/chatbot/lib/ai/models.ts index 3383dcdba..00f0974ee 100644 --- a/apps/chatbot/lib/ai/models.ts +++ b/apps/chatbot/lib/ai/models.ts @@ -24,21 +24,21 @@ export const chatModels: ChatModel[] = [ }, // Anthropic { - id: "anthropic/claude-3.5-haiku", - name: "Claude 3.5 Haiku", + id: "anthropic/claude-haiku-4.5", + name: "Claude Haiku 4.5", provider: "anthropic", description: "Fast and affordable, great for everyday tasks", }, { - id: "anthropic/claude-3.5-sonnet", - name: "Claude 3.5 Sonnet", + id: "anthropic/claude-sonnet-4.5", + name: "Claude Sonnet 4.5", provider: "anthropic", description: "Best balance of speed and intelligence", }, // Google { - id: "google/gemini-2.0-flash-001", - name: "Gemini 2.0 Flash", + id: "google/gemini-2.5-flash", + name: "Gemini 2.5 Flash", provider: "google", description: "Ultra fast and affordable", }, @@ -49,18 +49,83 @@ export const chatModels: ChatModel[] = [ provider: "deepseek", description: "Strong open-source model", }, - // Reasoning models + // Reasoning models. The -thinking suffix is ours, not OpenRouter's: + // getLanguageModel strips it and wraps the base id (see providers.ts). { - id: "anthropic/claude-3.5-sonnet-thinking", - name: "Claude 3.5 Sonnet (Thinking)", + id: "anthropic/claude-sonnet-4.5-thinking", + name: "Claude Sonnet 4.5 (Thinking)", provider: "reasoning", description: "Extended thinking for complex problems", }, ]; +/** + * Models for the two background calls the picker never covers: chat titles and + * artifact content. + * + * Declared beside the curated list rather than hardcoded at the call site. The + * previous inline ids ("google/gemini-2.0-flash-001" for titles, + * "anthropic/claude-3.5-haiku" for artifacts) were retired upstream and 404'd + * on every request, and nothing tied them back to the models the app maintains. + * Keep both pointing at an id present in `chatModels`. + */ +export const TITLE_MODEL = "google/gemini-2.5-flash"; +export const ARTIFACT_MODEL = "anthropic/claude-haiku-4.5"; + +// An uncapped request is billed as if it will emit the model's whole output +// window, and OpenRouter reserves that much credit up front for Anthropic and +// Google models. Left unset, picking one of those rejects the turn outright +// ("requires more credits ... you requested up to 65535 tokens") however short +// the answer would have been. +export const MAX_OUTPUT_TOKENS = 4096; +// Has to clear the thinking budget the chat route asks for, or the answer is +// truncated before any of it reaches the reply. +export const MAX_REASONING_OUTPUT_TOKENS = 12_000; +export const MAX_TITLE_OUTPUT_TOKENS = 64; + +const THINKING_SUFFIX_REGEX = /-thinking$/; + +// "-thinking" is our own marker, not part of any OpenRouter id. +export function baseOpenRouterId(modelId: string): string { + return modelId.replace(THINKING_SUFFIX_REGEX, ""); +} + +export function isReasoningModelId(modelId: string): boolean { + return ( + THINKING_SUFFIX_REGEX.test(modelId) || + (modelId.includes("reasoning") && !modelId.includes("non-reasoning")) + ); +} + +// Every id the app can send upstream. OpenRouter retires ids on its own +// schedule and answers a retired one with a 404 at request time, so keep the +// full set in one place for check-model-ids.ts to verify against the catalog. +export const openRouterModelIds = [ + ...new Set([ + ...chatModels.map((m) => baseOpenRouterId(m.id)), + TITLE_MODEL, + ARTIFACT_MODEL, + ]), +]; + // Group models by provider for UI export const allowedModelIds = new Set(chatModels.map((m) => m.id)); +/** + * Coerce a persisted `chat-model` cookie to a model the app still supports. + * + * The cookie outlives the curated list: anyone who had a since-retired model + * selected keeps sending that id, and `/api/chat` rejects unknown ids with a + * 400. The picker's own fallback is display-only — it renders the default name + * while `initialChatModel` still carries the stale id into the request body — + * so the coercion has to happen where the cookie is read. + */ +export function resolveChatModelId(cookieValue: string | undefined): string { + return cookieValue && allowedModelIds.has(cookieValue) + ? cookieValue + : DEFAULT_CHAT_MODEL; +} + export const modelsByProvider = chatModels.reduce( (acc, model) => { if (!acc[model.provider]) { diff --git a/apps/chatbot/lib/ai/models.unit.test.ts b/apps/chatbot/lib/ai/models.unit.test.ts new file mode 100644 index 000000000..8252eb976 --- /dev/null +++ b/apps/chatbot/lib/ai/models.unit.test.ts @@ -0,0 +1,151 @@ +import { readdirSync, readFileSync } from "node:fs"; +import { extname, join, resolve } from "node:path"; +import { describe, expect, it } from "vitest"; +import { + allowedModelIds, + ARTIFACT_MODEL, + baseOpenRouterId, + chatModels, + DEFAULT_CHAT_MODEL, + isReasoningModelId, + MAX_OUTPUT_TOKENS, + MAX_REASONING_OUTPUT_TOKENS, + openRouterModelIds, + resolveChatModelId, + TITLE_MODEL, +} from "./models"; + +// apps/researcher hit this in 2026-08: getTitleModel() hardcoded +// "google/gemini-2.0-flash-001", OpenRouter retired it, and every title call +// 404'd. The chatbot carried the same three ids. Nothing here can tell that a +// model was retired upstream — that needs a live call — but these assertions do +// catch the structural cause: an id drifting away from the curated set. +const RETIRED_IDS = [ + "google/gemini-2.0-flash-001", + "anthropic/claude-3.5-haiku", + "anthropic/claude-3.5-sonnet", +]; + +describe("curated model list", () => { + it("points the background models at models the app also offers", () => { + // Titles and artifacts are generated outside the picker, so a bad id here + // breaks those features for everyone rather than only whoever selects it. + expect(allowedModelIds.has(TITLE_MODEL)).toBe(true); + expect(allowedModelIds.has(ARTIFACT_MODEL)).toBe(true); + expect(allowedModelIds.has(DEFAULT_CHAT_MODEL)).toBe(true); + }); + + it("never reintroduces a retired id", () => { + for (const id of RETIRED_IDS) { + expect(allowedModelIds.has(id)).toBe(false); + expect(TITLE_MODEL).not.toBe(id); + expect(ARTIFACT_MODEL).not.toBe(id); + expect(DEFAULT_CHAT_MODEL).not.toBe(id); + expect(openRouterModelIds).not.toContain(id); + } + }); + + it("keeps entries unique and provider-qualified", () => { + expect(allowedModelIds.size).toBe(chatModels.length); + for (const model of chatModels) { + expect(model.id).toContain("/"); + expect(model.name.length).toBeGreaterThan(0); + } + }); + + it("strips the -thinking marker before a request leaves the app", () => { + // OpenRouter has no "-thinking" id; sending one unstripped is a 404. + expect(isReasoningModelId("anthropic/claude-sonnet-4.5-thinking")).toBe(true); + expect(isReasoningModelId("openai/gpt-4o-mini")).toBe(false); + expect(baseOpenRouterId("anthropic/claude-sonnet-4.5-thinking")).toBe( + "anthropic/claude-sonnet-4.5" + ); + for (const id of openRouterModelIds) { + expect(id.endsWith("-thinking")).toBe(false); + } + }); +}); + +describe("chat-model cookie resolution", () => { + it("keeps ids the chat API still accepts", () => { + for (const model of chatModels) { + expect(resolveChatModelId(model.id)).toBe(model.id); + } + }); + + it("falls back to the default for missing, empty or retired ids", () => { + for (const value of [...RETIRED_IDS, "not-a-model", "", " "]) { + expect(resolveChatModelId(value)).toBe(DEFAULT_CHAT_MODEL); + } + expect(resolveChatModelId(undefined)).toBe(DEFAULT_CHAT_MODEL); + }); + + // Both chat pages read the cookie server-side and hand it to , which + // sends it as selectedChatModel, so passing cookie.value straight through is + // what made a retired id reject every send. Guard the call sites too. + it.each([["app/(chat)/page.tsx"], ["app/(chat)/chat/[id]/page.tsx"]])( + "routes the cookie through the resolver in %s", + (page) => { + const source = readFileSync(resolve(page), "utf8"); + expect(source).toMatch( + /resolveChatModelId\(\s*cookieStore\.get\("chat-model"\)\?\.value\s*\)/ + ); + expect(source).not.toMatch(/initialChatModel=\{\w*[Cc]ookie\w*\.value\}/); + } + ); +}); + +const MODEL_CALL_REGEX = /\b(streamText|streamObject|generateText|generateObject)\(/g; +const MAX_OUTPUT_TOKENS_REGEX = /maxOutputTokens:/g; + +function sourceFiles(dir: string): string[] { + return readdirSync(dir, { withFileTypes: true }).flatMap((entry) => { + const path = join(dir, entry.name); + + if (entry.isDirectory()) { + return sourceFiles(path); + } + if (entry.name.includes(".test.") || !/\.tsx?$/.test(extname(path))) { + return []; + } + return [path]; + }); +} + +describe("output token caps", () => { + it("keeps the reasoning cap above the thinking budget the chat route asks for", () => { + const route = readFileSync(resolve("app/(chat)/api/chat/route.ts"), "utf8"); + const budget = route.match(/budgetTokens:\s*([\d_]+)/); + + expect(budget).not.toBeNull(); + expect(MAX_REASONING_OUTPUT_TOKENS).toBeGreaterThan( + Number((budget?.[1] ?? "0").replace(/_/g, "")) + ); + }); + + // An uncapped call is quoted against the model's whole output window, which + // Anthropic and Google reject for credit up front. Counting call sites rather + // than naming them is deliberate: the by-hand sweep that introduced the caps + // missed requestSuggestions, and a named list would have missed it again. + it("caps every model call in the app", () => { + const uncapped = ["app", "artifacts", "lib"] + .flatMap((dir) => sourceFiles(resolve(dir))) + .map((path) => { + const source = readFileSync(path, "utf8"); + return { + path, + calls: source.match(MODEL_CALL_REGEX)?.length ?? 0, + caps: source.match(MAX_OUTPUT_TOKENS_REGEX)?.length ?? 0, + }; + }) + .filter((file) => file.calls > file.caps) + .map((file) => `${file.path} (${file.calls} calls, ${file.caps} capped)`); + + expect(uncapped).toEqual([]); + }); + + it("keeps the caps small enough to be the point of having them", () => { + expect(MAX_OUTPUT_TOKENS).toBeLessThan(32_000); + expect(MAX_REASONING_OUTPUT_TOKENS).toBeLessThan(32_000); + }); +}); diff --git a/apps/chatbot/lib/ai/providers.ts b/apps/chatbot/lib/ai/providers.ts index fe8d66f92..d5eb9128f 100644 --- a/apps/chatbot/lib/ai/providers.ts +++ b/apps/chatbot/lib/ai/providers.ts @@ -6,8 +6,12 @@ import { } from "ai"; import { withMemWal } from "@mysten-incubation/memwal/ai"; import { isTestEnvironment } from "../constants"; - -const THINKING_SUFFIX_REGEX = /-thinking$/; +import { + ARTIFACT_MODEL, + baseOpenRouterId, + isReasoningModelId, + TITLE_MODEL, +} from "./models"; // OpenRouter provider (OpenAI-compatible) const openrouter = createOpenAI({ @@ -35,19 +39,18 @@ export const myProvider = isTestEnvironment : null; export function getLanguageModel(modelId: string) { + // The mock provider registers behaviours, not catalog ids, so every real id + // has to land on one of its aliases — asking it for "openai/gpt-4o-mini" + // throws NoSuchModelError and fails the turn. if (isTestEnvironment && myProvider) { - return myProvider.languageModel(modelId); + return myProvider.languageModel( + isReasoningModelId(modelId) ? "chat-model-reasoning" : "chat-model" + ); } - const isReasoningModel = - modelId.endsWith("-thinking") || - (modelId.includes("reasoning") && !modelId.includes("non-reasoning")); - - if (isReasoningModel) { - const gatewayModelId = modelId.replace(THINKING_SUFFIX_REGEX, ""); - + if (isReasoningModelId(modelId)) { return wrapLanguageModel({ - model: openrouter.chat(gatewayModelId), + model: openrouter.chat(baseOpenRouterId(modelId)), middleware: extractReasoningMiddleware({ tagName: "thinking" }), }); } @@ -59,14 +62,14 @@ export function getTitleModel() { if (isTestEnvironment && myProvider) { return myProvider.languageModel("title-model"); } - return openrouter.chat("google/gemini-2.0-flash-001"); + return openrouter.chat(TITLE_MODEL); } export function getArtifactModel() { if (isTestEnvironment && myProvider) { return myProvider.languageModel("artifact-model"); } - return openrouter.chat("anthropic/claude-3.5-haiku"); + return openrouter.chat(ARTIFACT_MODEL); } /** diff --git a/apps/chatbot/lib/ai/tools/request-suggestions.ts b/apps/chatbot/lib/ai/tools/request-suggestions.ts index 156e1fadf..127db6f40 100644 --- a/apps/chatbot/lib/ai/tools/request-suggestions.ts +++ b/apps/chatbot/lib/ai/tools/request-suggestions.ts @@ -5,6 +5,7 @@ import { getDocumentByIdForUser, saveSuggestions } from "@/lib/db/queries"; import type { Suggestion } from "@/lib/db/schema"; import type { ChatMessage } from "@/lib/types"; import { generateUUID } from "@/lib/utils"; +import { MAX_OUTPUT_TOKENS } from "../models"; import { getArtifactModel } from "../providers"; type RequestSuggestionsProps = { @@ -63,6 +64,7 @@ export const requestSuggestions = ({ model: getArtifactModel(), system: "You are a help writing assistant. Given a piece of writing, please offer suggestions to improve the piece of writing and describe the change. It is very important for the edits to contain full sentences instead of just words. Max 5 suggestions.", + maxOutputTokens: MAX_OUTPUT_TOKENS, prompt: document.content, output: Output.array({ element: z.object({ diff --git a/apps/chatbot/package.json b/apps/chatbot/package.json index 9fbbb5f76..b534e980a 100644 --- a/apps/chatbot/package.json +++ b/apps/chatbot/package.json @@ -20,7 +20,8 @@ "test:e2e": "PLAYWRIGHT=True playwright test", "test:e2e:ui": "PLAYWRIGHT=True playwright test --ui", "test": "pnpm test:e2e", - "test:unit": "vitest run" + "test:unit": "vitest run", + "check:models": "tsx scripts/check-model-ids.ts" }, "dependencies": { "@ai-sdk/gateway": "^3.0.15", diff --git a/apps/chatbot/playwright.config.ts b/apps/chatbot/playwright.config.ts index fdc43dc79..dd7bdcde9 100644 --- a/apps/chatbot/playwright.config.ts +++ b/apps/chatbot/playwright.config.ts @@ -34,7 +34,8 @@ export default defineConfig({ screenshot: "only-on-failure", actionTimeout: 10_000, // 30s to tolerate cold Next.js/Turbopack compile on 2-vCPU CI runners; - // globalSetup also warms `/` to make first-nav fast on the happy path. + // globalSetup also warms every route the suite reaches, so a first-nav + // should already be hitting a compiled route. navigationTimeout: 30_000, }, diff --git a/apps/chatbot/scripts/check-model-ids.ts b/apps/chatbot/scripts/check-model-ids.ts new file mode 100644 index 000000000..42d808948 --- /dev/null +++ b/apps/chatbot/scripts/check-model-ids.ts @@ -0,0 +1,48 @@ +/** + * Verify every OpenRouter id the app can send still exists upstream. + * + * OpenRouter retires ids on its own schedule and answers a retired one with a + * 404 ("No endpoints found for ") only at request time. Nothing in the type + * system or the e2e suite notices, so a dead id surfaces as a generic "Oops, an + * error occurred!" for whoever picks it — and for the title and artifact models, + * which are hardcoded, it silently breaks those features for everyone. + * + * The catalog endpoint is public, so this needs no OPENROUTER_API_KEY. + */ +import { openRouterModelIds } from "../lib/ai/models"; + +const CATALOG_URL = "https://openrouter.ai/api/v1/models"; + +async function main(): Promise { + const response = await fetch(CATALOG_URL, { + signal: AbortSignal.timeout(30_000), + }); + + if (!response.ok) { + throw new Error( + `Could not read the OpenRouter catalog: ${response.status} ${response.statusText}` + ); + } + + const catalog = (await response.json()) as { data: { id: string }[] }; + const live = new Set(catalog.data.map((model) => model.id)); + const retired = openRouterModelIds.filter((id) => !live.has(id)); + + for (const id of openRouterModelIds) { + console.log(`${live.has(id) ? "ok " : "retired"} ${id}`); + } + + if (retired.length > 0) { + throw new Error( + `${retired.length} model id(s) no longer exist on OpenRouter: ${retired.join(", ")}. ` + + "Update lib/ai/models.ts — requests using these ids fail with a 404." + ); + } + + console.log(`\nAll ${openRouterModelIds.length} model ids are live.`); +} + +main().catch((error) => { + console.error(error instanceof Error ? error.message : error); + process.exit(1); +}); diff --git a/apps/chatbot/tests/playwright/e2e/conversation.test.ts b/apps/chatbot/tests/playwright/e2e/conversation.test.ts new file mode 100644 index 000000000..23616cd8c --- /dev/null +++ b/apps/chatbot/tests/playwright/e2e/conversation.test.ts @@ -0,0 +1,140 @@ +import { expect, test } from "@playwright/test"; + +// The mock provider answers a greeting with "Hello! How can I help you today?", +// a weather question with a fixed forecast, and anything else with "This is a +// mock response for testing." (lib/ai/models.mock.ts). +const MOCK_GREETING = /How can I help you today/i; +const MOCK_DEFAULT = /mock response for testing/i; +const MOCK_WEATHER = /sunny and 72/i; +const MODEL_BUTTON_REGEX = /Gemini|Claude|GPT|Grok/i; + +test.describe("Conversation", () => { + test("streams an assistant reply back into the thread", async ({ page }) => { + await page.goto("/"); + await page.getByTestId("multimodal-input").fill("Hello there"); + await page.getByTestId("send-button").click(); + + const assistantMessage = page + .locator('[data-role="assistant"]') + .getByTestId("message-content") + .first(); + + await expect(assistantMessage).toContainText(MOCK_GREETING, { + timeout: 20_000, + }); + }); + + test("keeps the reply after reloading the chat", async ({ page }) => { + await page.goto("/"); + await page.getByTestId("multimodal-input").fill("Tell me something"); + await page.getByTestId("send-button").click(); + + const assistantMessage = page + .locator('[data-role="assistant"]') + .getByTestId("message-content") + .first(); + await expect(assistantMessage).toContainText(MOCK_DEFAULT, { + timeout: 20_000, + }); + + // Sending redirects `/` to /chat/; both the user turn and the assistant + // turn have to come back from the database on a cold load. + await expect(page).toHaveURL(/\/chat\/[0-9a-f-]{36}/, { timeout: 20_000 }); + await page.reload(); + + await expect(page.getByTestId("message-content").first()).toContainText( + "Tell me something" + ); + await expect( + page.locator('[data-role="assistant"]').getByTestId("message-content").first() + ).toContainText(MOCK_DEFAULT); + }); + + test("routes the question to the model instead of a fixed reply", async ({ + page, + }) => { + await page.goto("/"); + await page + .getByTestId("multimodal-input") + .fill("What is the weather in San Francisco?"); + await page.getByTestId("send-button").click(); + + await expect( + page.locator('[data-role="assistant"]').getByTestId("message-content").first() + ).toContainText(MOCK_WEATHER, { timeout: 20_000 }); + }); + + test("titles the chat in the sidebar", async ({ page }) => { + await page.goto("/"); + await page.getByTestId("multimodal-input").fill("Hello there"); + await page.getByTestId("send-button").click(); + + // The title comes from its own model call, separate from the reply. When + // that model id was retired upstream every chat stayed named "New chat". + await expect( + page.getByRole("link", { name: /Test Conversation/i }) + ).toBeVisible({ timeout: 20_000 }); + }); + + test("streams reasoning for a thinking model", async ({ page }) => { + await page.goto("/"); + await page + .locator("button") + .filter({ hasText: MODEL_BUTTON_REGEX }) + .first() + .click(); + await page.getByText(/Claude Sonnet [\d.]+ \(Thinking\)/i).first().click(); + + await page.getByTestId("multimodal-input").fill("Solve this puzzle"); + await page.getByTestId("send-button").click(); + + await expect(page.getByTestId("message-reasoning").first()).toBeVisible({ + timeout: 20_000, + }); + }); + + test("creates a text artifact through the retired-id-safe artifact model", async ({ + page, + }) => { + // ARTIFACT_MODEL used to be a retired OpenRouter id, so createDocument + // 404'd before any panel opened. The mock emits the same tool call the + // chat model would, then the artifact model fills the document. + await page.goto("/"); + await page + .getByTestId("multimodal-input") + .fill("Create a document about Silicon Valley"); + await page.getByTestId("send-button").click(); + + const preview = page.getByTestId("document-preview"); + await expect(preview).toBeVisible({ timeout: 20_000 }); + await expect(preview).toContainText("Test Artifact"); + + await preview.click(); + await expect(page.getByTestId("artifact")).toBeVisible(); + }); + + test("recovers when the chat-model cookie holds a retired id", async ({ + page, + context, + }) => { + // Ids leave chatModels whenever OpenRouter retires a model, but a returning + // reader still carries the old one. Passing it through made /api/chat reject + // every send as a bad request while the picker showed a valid model. + await context.addCookies([ + { + name: "chat-model", + value: "anthropic/claude-3.5-sonnet", + domain: "localhost", + path: "/", + }, + ]); + + await page.goto("/"); + await page.getByTestId("multimodal-input").fill("Hello there"); + await page.getByTestId("send-button").click(); + + await expect( + page.locator('[data-role="assistant"]').getByTestId("message-content").first() + ).toContainText(MOCK_GREETING, { timeout: 20_000 }); + }); +}); diff --git a/apps/chatbot/tests/playwright/global-setup.ts b/apps/chatbot/tests/playwright/global-setup.ts index 98f2a73de..8635cfd81 100644 --- a/apps/chatbot/tests/playwright/global-setup.ts +++ b/apps/chatbot/tests/playwright/global-setup.ts @@ -4,11 +4,11 @@ * Runs once before any test (after the webServer is spawned). Responsible for: * 1. Applying Drizzle migrations so chat/user tables exist. * 2. Failing fast with a clear error if POSTGRES_URL is missing in CI. - * 3. Warming the `/` route so the first `page.goto("/")` in the suite - * doesn't race Turbopack's lazy cold compile against the 15s - * navigationTimeout on CI runners. + * 3. Warming the routes the suite exercises so no test races Turbopack's lazy + * cold compile against the navigationTimeout on CI runners. */ import { spawnSync } from "node:child_process"; +import { randomUUID } from "node:crypto"; export default async function globalSetup(): Promise { const url = process.env.POSTGRES_URL; @@ -33,23 +33,101 @@ export default async function globalSetup(): Promise { throw new Error(`[playwright] Migration failed with exit code ${result.status}`); } - // Prime Next.js/Turbopack's per-route compile cache for `/`. On a cold - // CI runner the first `page.goto("/")` can take 15-30s (the route is - // compiled lazily on first hit), which exceeds the default 15s - // navigationTimeout and flakes tests until retries kick in. Fetching - // here moves the cost to setup time so the first real test navigation - // hits a warm cache. + await warmRoutes(); +} + +const jar = new Map(); + +function storeCookies(response: Response): void { + for (const header of response.headers.getSetCookie()) { + const [pair] = header.split(";"); + const separator = pair.indexOf("="); + + if (separator > 0) { + jar.set(pair.slice(0, separator).trim(), pair.slice(separator + 1).trim()); + } + } +} + +/** + * Fetch a route the way a browser reaches it: following redirects and keeping + * the cookies they set. + * + * Unauthenticated requests are bounced to /api/auth/guest by the proxy, so a + * single non-following request never renders the route it names and compiles + * nothing. Redirects are followed as GET, matching how the proxy sends a + * would-be POST through the guest handshake first. + */ +async function fetchFollowing( + url: string, + init: { method?: string; body?: string } = {} +): Promise { + let target = url; + let method = init.method ?? "GET"; + let body = init.body; + + for (let hop = 0; hop < 5; hop++) { + const response: Response = await fetch(target, { + method, + body, + headers: jar.size > 0 ? { cookie: [...jar].map(([k, v]) => `${k}=${v}`).join("; ") } : {}, + redirect: "manual", + // Matches the webServer budget. A 2-vCPU runner compiles slower than a + // dev laptop, where `/` alone took 39s from a cleared .next. + signal: AbortSignal.timeout(120_000), + }); + + storeCookies(response); + const location = response.headers.get("location"); + + if (response.status < 300 || response.status >= 400 || !location) { + return response; + } + + target = new URL(location, target).toString(); + method = "GET"; + body = undefined; + } + + throw new Error(`Too many redirects while warming ${url}`); +} + +/** + * Prime Turbopack's per-route compile cache before the first test runs. + * + * Routes compile lazily on first hit, and on a cold cache that first hit costs + * seconds to tens of seconds. Warming `/` alone was not enough: sending a + * message still paid for `/api/chat` and `/chat/[id]` mid-test, overrunning both + * the 30s navigationTimeout and the 60s test timeout. Retries hid it, since by + * the second attempt the routes were warm. + * + * Each request is shaped to compile its route without writing anything: an + * unparseable chat body is rejected before it reaches the database, and a random + * chat id belongs to nobody. Failures are not fatal — a slow first test beats a + * suite that cannot start. + */ +async function warmRoutes(): Promise { const port = process.env.PORT ?? "3001"; - const target = `http://localhost:${port}/`; - console.log(`[playwright] Warming ${target} ...`); - try { + const base = `http://localhost:${port}`; + const routes: [string, { method?: string; body?: string }?][] = [ + ["/"], + ["/login"], + ["/register"], + ["/api/history?limit=1"], + ["/api/chat", { method: "POST", body: "{}" }], + [`/api/document?id=${randomUUID()}`], + [`/chat/${randomUUID()}`], + ]; + + for (const [path, init] of routes) { const started = Date.now(); - const res = await fetch(target, { - redirect: "manual", // `/` may 307 to /api/auth/guest; we just want the compile - signal: AbortSignal.timeout(60_000), - }); - console.log(`[playwright] Warm-up done in ${Date.now() - started}ms (status ${res.status})`); - } catch (err) { - console.warn("[playwright] Warm-up failed, continuing:", err); + try { + const response = await fetchFollowing(`${base}${path}`, init); + console.log( + `[playwright] Warmed ${path} in ${Date.now() - started}ms (status ${response.status})` + ); + } catch (err) { + console.warn(`[playwright] Warming ${path} failed, continuing:`, err); + } } } From bf8894ab1e940ddd878a25cdf9639989fbdf6035 Mon Sep 17 00:00:00 2001 From: Le Tien Phat <91601109+Niko1444@users.noreply.github.com> Date: Fri, 28 Aug 2026 14:53:21 +0700 Subject: [PATCH 28/29] fix(analyze): skip the pre-extraction embed on inputs the API will reject /api/analyze accepts 64 KiB but embeds the full input as its dedup query, while text-embedding-3-small caps at 8192 tokens and the embedder does not truncate. Oversized input was therefore always rejected with a 400, ~560ms after the call went out, before falling through to plain extraction. Guard the call with MAX_PRE_EXTRACT_EMBED_BYTES (8 KiB) and record pre_extract_status="skipped_oversized", so the case stops emitting a WARN that reads like an incident. Production evidence (relayer logs, 2026-08-27): a 31,782-byte input was rejected with `Invalid 'input': maximum context length is 8192` after burning 564ms. That size is pinned as a regression test. The threshold equals remember's SUMMARIZE_THRESHOLD_BYTES but is kept separate on purpose: that one decides when to pay for an LLM summary before storing a permanent embedding, this one decides when to abandon a throwaway dedup query vector. Measured bytes-per-token runs ~1.4 (base64) to ~4.5 (prose), so 8 KiB sits below even the densest content's breach point. Large inputs still extract without dedup context. Recovering it needs the pre-extraction fetch_batch timeout addressed first, which dominates this failure mode in production. --- services/server/src/routes/analyze.rs | 87 ++++++++++++++++++++++++++- 1 file changed, 85 insertions(+), 2 deletions(-) diff --git a/services/server/src/routes/analyze.rs b/services/server/src/routes/analyze.rs index e69e53837..083b38266 100644 --- a/services/server/src/routes/analyze.rs +++ b/services/server/src/routes/analyze.rs @@ -106,6 +106,34 @@ const EMBED_TIMEOUT_MS: u64 = 800; const SEARCH_TIMEOUT_MS: u64 = 300; const FETCH_TIMEOUT_MS: u64 = 500; +/// Largest input handed straight to the embedder for the pre-extraction dedup +/// query. `text-embedding-3-small` caps at 8192 tokens and the embedder does +/// not truncate, so oversized input is rejected outright with +/// `Invalid 'input': maximum context length is 8192` — after the request has +/// already cost a round-trip (~560ms observed in production). +/// +/// Measured bytes-per-token over representative corpora runs ~1.4 (base64) +/// to ~4.5 (English prose), so the byte length at which 8192 tokens is +/// reached varies from ~11 KiB to ~36 KiB depending on content. No single +/// byte threshold is exact; this one sits below the densest case so the +/// guard never lets a doomed call through. +/// +/// Deliberately a separate constant from `remember`'s +/// `SUMMARIZE_THRESHOLD_BYTES`, which holds the same value today. That one +/// decides when to pay for an LLM summary before storing a *permanent* +/// embedding; this one decides when to abandon a *throwaway* dedup query +/// vector. Different trade-offs, so they are free to diverge. +const MAX_PRE_EXTRACT_EMBED_BYTES: usize = 8 * 1024; + +/// Whether to skip the pre-extraction dedup embed for an input of this size. +/// +/// Skipping costs the extractor its dedup context, which is why the boundary +/// is inclusive: input of exactly `MAX_PRE_EXTRACT_EMBED_BYTES` is still +/// handed to the embedder. +fn should_skip_pre_extract_embed(text_len: usize) -> bool { + text_len > MAX_PRE_EXTRACT_EMBED_BYTES +} + /// One fact that has finished embed + SEAL encrypt and is ready to enqueue: /// `(plaintext, importance, embedding, ciphertext)`. type PreparedFact = (String, f32, Vec, Vec); @@ -201,6 +229,19 @@ pub async fn analyze( let related_memories: Vec = if !namespace_has_memories { pre_extract_status = "skipped_empty_namespace"; Vec::new() + } else if should_skip_pre_extract_embed(body.text.len()) { + // Oversized input (task WALM-411): the embedding API rejects anything + // past its 8192-token ceiling and the embedder does not truncate, so + // this call cannot succeed. Skip it rather than spend the round-trip + // (~560ms observed in production) to arrive at the same empty context. + // + // The consequence is that large inputs extract without dedup context. + // Recovering it would need summarize-before-embed, the way + // `remember` does it — not affordable here, where pre-extraction is + // inline on the caller's request under a ~1.6s total budget while + // `remember` summarizes inside a spawned job. + pre_extract_status = "skipped_oversized"; + Vec::new() } else { // Embed the input as a query. On embed failure, log + degrade — // do NOT propagate via `?`, even though the embed below uses the @@ -914,8 +955,8 @@ pub async fn analyze( #[cfg(test)] mod tests { use super::{ - analyze_fact_idempotency_key, classify_analyze_job_reuse, AnalyzeJobReuse, - ANALYZE_CONCURRENCY, MAX_ANALYZE_TEXT_BYTES, + analyze_fact_idempotency_key, classify_analyze_job_reuse, should_skip_pre_extract_embed, + AnalyzeJobReuse, ANALYZE_CONCURRENCY, MAX_ANALYZE_TEXT_BYTES, MAX_PRE_EXTRACT_EMBED_BYTES, }; use crate::routes::remember::MAX_REMEMBER_TEXT_BYTES; use crate::services::extractor::MAX_ANALYZE_FACTS; @@ -934,6 +975,48 @@ mod tests { const { assert!(MAX_ANALYZE_TEXT_BYTES < MAX_REMEMBER_TEXT_BYTES) } } + // ── Pre-extraction embed size guard (WALM-411) ─────────────── + + #[test] + fn max_pre_extract_embed_bytes_is_8kb() { + assert_eq!(MAX_PRE_EXTRACT_EMBED_BYTES, 8 * 1024); + } + + #[test] + fn pre_extract_guard_is_reachable_below_the_accepted_input_ceiling() { + // If the guard sat at or above the endpoint's own cap it could never + // fire, and the skip would be dead code. + const { assert!(MAX_PRE_EXTRACT_EMBED_BYTES < MAX_ANALYZE_TEXT_BYTES) } + } + + #[test] + fn embed_is_attempted_at_and_below_the_threshold() { + assert!(!should_skip_pre_extract_embed(0)); + assert!(!should_skip_pre_extract_embed(1)); + assert!(!should_skip_pre_extract_embed( + MAX_PRE_EXTRACT_EMBED_BYTES - 1 + )); + // Boundary: exactly at the limit is still handed to the embedder. + assert!(!should_skip_pre_extract_embed(MAX_PRE_EXTRACT_EMBED_BYTES)); + } + + #[test] + fn embed_is_skipped_above_the_threshold() { + assert!(should_skip_pre_extract_embed( + MAX_PRE_EXTRACT_EMBED_BYTES + 1 + )); + assert!(should_skip_pre_extract_embed(MAX_ANALYZE_TEXT_BYTES)); + } + + #[test] + fn observed_production_rejection_would_now_be_skipped() { + // Regression anchor: a real 31,782-byte /api/analyze input was + // rejected by the embedding API on 2026-08-27 with + // `Invalid 'input': maximum context length is 8192`, after burning + // 564ms. That request must not reach the embedder again. + assert!(should_skip_pre_extract_embed(31_782)); + } + // ── Analyze concurrency + weight ──────────────────── #[test] From 455d4966813a43d45984a1053a2f96d298211e7a Mon Sep 17 00:00:00 2001 From: Le Tien Phat <91601109+Niko1444@users.noreply.github.com> Date: Mon, 7 Sep 2026 14:00:00 +0700 Subject: [PATCH 29/29] fix(analyze): align pre-extraction skip with embedder byte cap --- services/server/src/routes/analyze.rs | 45 +++++++++++------------- services/server/src/services/embedder.rs | 2 +- 2 files changed, 22 insertions(+), 25 deletions(-) diff --git a/services/server/src/routes/analyze.rs b/services/server/src/routes/analyze.rs index 083b38266..d8aa1c8d9 100644 --- a/services/server/src/routes/analyze.rs +++ b/services/server/src/routes/analyze.rs @@ -106,24 +106,15 @@ const EMBED_TIMEOUT_MS: u64 = 800; const SEARCH_TIMEOUT_MS: u64 = 300; const FETCH_TIMEOUT_MS: u64 = 500; -/// Largest input handed straight to the embedder for the pre-extraction dedup -/// query. `text-embedding-3-small` caps at 8192 tokens and the embedder does -/// not truncate, so oversized input is rejected outright with -/// `Invalid 'input': maximum context length is 8192` — after the request has -/// already cost a round-trip (~560ms observed in production). +/// Largest input handed to the embedder for the pre-extraction dedup query. +/// Matches the embedder's local byte cap: larger inputs already fail before +/// any HTTP request. Skipping them here reports an expected skip instead of +/// an embed failure, while preserving dedup attempts for inputs up to 16 KiB. /// -/// Measured bytes-per-token over representative corpora runs ~1.4 (base64) -/// to ~4.5 (English prose), so the byte length at which 8192 tokens is -/// reached varies from ~11 KiB to ~36 KiB depending on content. No single -/// byte threshold is exact; this one sits below the densest case so the -/// guard never lets a doomed call through. -/// -/// Deliberately a separate constant from `remember`'s -/// `SUMMARIZE_THRESHOLD_BYTES`, which holds the same value today. That one -/// decides when to pay for an LLM summary before storing a *permanent* -/// embedding; this one decides when to abandon a *throwaway* dedup query -/// vector. Different trade-offs, so they are free to diverge. -const MAX_PRE_EXTRACT_EMBED_BYTES: usize = 8 * 1024; +/// This is a byte limit, not a token guarantee: dense inputs below it can +/// still exceed the provider's token limit and fall back to plain extraction. +/// It is independent of `remember`'s summarize-before-store cost threshold. +const MAX_PRE_EXTRACT_EMBED_BYTES: usize = 16 * 1024; /// Whether to skip the pre-extraction dedup embed for an input of this size. /// @@ -230,10 +221,9 @@ pub async fn analyze( pre_extract_status = "skipped_empty_namespace"; Vec::new() } else if should_skip_pre_extract_embed(body.text.len()) { - // Oversized input (task WALM-411): the embedding API rejects anything - // past its 8192-token ceiling and the embedder does not truncate, so - // this call cannot succeed. Skip it rather than spend the round-trip - // (~560ms observed in production) to arrive at the same empty context. + // The embedder already rejects inputs above its byte cap locally. + // Classify this expected outcome as a skip instead of logging an + // embed failure; no network round-trip would occur either way. // // The consequence is that large inputs extract without dedup context. // Recovering it would need summarize-before-embed, the way @@ -978,8 +968,8 @@ mod tests { // ── Pre-extraction embed size guard (WALM-411) ─────────────── #[test] - fn max_pre_extract_embed_bytes_is_8kb() { - assert_eq!(MAX_PRE_EXTRACT_EMBED_BYTES, 8 * 1024); + fn max_pre_extract_embed_bytes_is_16kb() { + assert_eq!(MAX_PRE_EXTRACT_EMBED_BYTES, 16 * 1024); } #[test] @@ -987,12 +977,18 @@ mod tests { // If the guard sat at or above the endpoint's own cap it could never // fire, and the skip would be dead code. const { assert!(MAX_PRE_EXTRACT_EMBED_BYTES < MAX_ANALYZE_TEXT_BYTES) } + // Lowering the embedder cap must not leave analyze attempting inputs + // that are guaranteed to fail its local validation. + const { assert!(MAX_PRE_EXTRACT_EMBED_BYTES <= crate::services::embedder::MAX_EMBED_INPUT_BYTES) } } #[test] fn embed_is_attempted_at_and_below_the_threshold() { assert!(!should_skip_pre_extract_embed(0)); assert!(!should_skip_pre_extract_embed(1)); + // Preserve dedup attempts in the band the original 8 KiB guard skipped. + assert!(!should_skip_pre_extract_embed(8 * 1024 + 1)); + assert!(!should_skip_pre_extract_embed(12 * 1024)); assert!(!should_skip_pre_extract_embed( MAX_PRE_EXTRACT_EMBED_BYTES - 1 )); @@ -1013,7 +1009,8 @@ mod tests { // Regression anchor: a real 31,782-byte /api/analyze input was // rejected by the embedding API on 2026-08-27 with // `Invalid 'input': maximum context length is 8192`, after burning - // 564ms. That request must not reach the embedder again. + // 564ms. Since #837, the embedder rejects it locally; analyze should + // classify it as skipped_oversized instead of embed_failed. assert!(should_skip_pre_extract_embed(31_782)); } diff --git a/services/server/src/services/embedder.rs b/services/server/src/services/embedder.rs index c939454a6..90252e6a3 100644 --- a/services/server/src/services/embedder.rs +++ b/services/server/src/services/embedder.rs @@ -27,7 +27,7 @@ pub const EMBEDDING_MODEL: &str = "openai/text-embedding-3-small"; pub const EMBEDDING_DIMS: usize = 1536; /// 16384 = 8192 tokens × 2 chars/token under cl100k; do not use admin 64KiB. -const MAX_EMBED_INPUT_BYTES: usize = 16384; +pub(crate) const MAX_EMBED_INPUT_BYTES: usize = 16384; fn reject_oversized_embed_input(text: &str) -> Result<(), AppError> { if text.len() > MAX_EMBED_INPUT_BYTES {