From e96bef23b8a3b097eb407d24dac021ae78872455 Mon Sep 17 00:00:00 2001 From: liyuanyang Date: Tue, 8 Sep 2026 13:37:18 +0800 Subject: [PATCH 1/2] feat(search): add model-free lexical route for file search (#176) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Files had no retrieval path without an embedding worker. Add a `route=lexical` that uses FTS + trigram over files.name — the same three-tier shape memory Recall already uses — so self-hosted deployments that never configure a worker can still search filenames. - Migration 0024: search_tsv tsvector + pg_trgm GIN on files - search.Service: RouteLexical bypasses worker check, searchLexical with exact/FTS/trigram tiers - API: accept "lexical" in route validation, skip managed embedding path (no cost) - CLI: --route lexical - SPEC.md §6.2, F5.4, docs/mcp.md: document the new lane - Integration test proves lexical works with nil worker while text/auto still fail closed --- CHANGELOG.md | 7 + SPEC.md | 3 +- docs/mcp.md | 5 +- server/cmd/mem/cmds_search.go | 2 +- server/internal/api/api.go | 4 +- server/internal/api/managed_embeddings.go | 5 +- .../migrations/0024_files_lexical_search.sql | 26 +++ server/internal/search/lexical_test.go | 174 ++++++++++++++++++ server/internal/search/search.go | 90 ++++++++- 9 files changed, 302 insertions(+), 14 deletions(-) create mode 100644 server/internal/db/migrations/0024_files_lexical_search.sql create mode 100644 server/internal/search/lexical_test.go diff --git a/CHANGELOG.md b/CHANGELOG.md index 104b02f..95b07d2 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -7,6 +7,13 @@ The project publishes 0.x prerelease versions; a stable release line is not yet ## [Unreleased] +### Added + +- File search gains a model-free lexical route (`route=lexical`). FTS + trigram + over `files.name` — same tier shape as memory Recall — so a deployment with + no embedding worker can still find files by name. CLI: `mem search "query" + --route lexical`. + ### Changed - Migrate GitHub repository, Release, issue, badge, and raw-content coordinates diff --git a/SPEC.md b/SPEC.md index d1e1b55..1dd81ee 100644 --- a/SPEC.md +++ b/SPEC.md @@ -146,7 +146,7 @@ | F5.1 | `mem context "..."` → 返回有大小预算的文件/结构化记忆证据包 | | F5.2 | 每条 evidence 必须含 source kind/id、稳定 citation、内容哈希、片段和 locator | | F5.3 | mem 只走 recall → context pack;回答与行动由调用方 Agent 完成 | -| F5.4 | `source=all|file|memory`;结构化记忆在无 Worker、无模型时也必须可立即召回 | +| F5.4 | `source=all|file|memory`;结构化记忆在无 Worker、无模型时也必须可立即召回;文件词法路由(`route=lexical`)同样无需 Worker | | F5.5 | 联合召回单路失败但仍有证据时返回 `200 + partial=true + warnings[]`;无幸存证据时返回 `502 context_unavailable` | ### F5A · 结构化 Agent 记忆 @@ -534,6 +534,7 @@ embeddings_face ( - `folders (user_id, path)` UNIQUE — 路径唯一性约束 - `memories (workspace_id, idempotency_key_sha256)` UNIQUE — 不落明文幂等键的幂等写入 - `memories` FTS + trigram — 无模型的确定性立即召回 +- `files` FTS + trigram — 文件名无模型词法召回(worker 不可用时降级至此) - `embeddings_* (embedding)` — pgvector HNSW - `file_entities (entity_id)` — 反查"和某人有关的所有文件" diff --git a/docs/mcp.md b/docs/mcp.md index 5fd17a2..6d5601e 100644 --- a/docs/mcp.md +++ b/docs/mcp.md @@ -138,7 +138,7 @@ The canonical product surface is: | `mem_checkpoint_list` | List newest-first bounded checkpoint summaries for one task | | `mem_checkpoint_get` | Get one immutable checkpoint and its full handoff payload | | `mem_resume` | Restore the current task head or a selected historical checkpoint, including resolved and missing evidence | -| `mem_search` | Natural-language search (text / visual / auto fuse); ranked files + snippets | +| `mem_search` | Natural-language search (text / visual / auto fuse); ranked files + snippets. `route=lexical` is model-free (FTS + trigram over file names, no worker needed) | | `mem_context` | Build an evidence-backed context pack for the calling Agent | | `mem_related` | Top-K files related to a `file_id` by embedding similarity | | `mem_face` | Person clusters: `action=list` / `name` / `merge` | @@ -295,7 +295,8 @@ same logical request should supply and retain a stable key so a committed result can replay without another provider invocation or charge. A `504` means the provider outcome is uncertain: do not automatically retry, and do not invent a new key. `mem_context` with `source=memory` stays lexical and -model-independent. +model-independent. `mem_search` with `route=lexical` is likewise model-free: +it uses FTS + trigram over file names and works without a configured worker. Its target output is structured for an Agent to consume: diff --git a/server/cmd/mem/cmds_search.go b/server/cmd/mem/cmds_search.go index ac866df..b608c4b 100644 --- a/server/cmd/mem/cmds_search.go +++ b/server/cmd/mem/cmds_search.go @@ -101,7 +101,7 @@ func newSearchCmd() *cobra.Command { }, } cmd.Flags().StringVar(&typ, "type", "", "mime prefix filter: image|text|application|audio|video") - cmd.Flags().StringVar(&route, "route", "", "search route: text|visual|auto (default auto)") + cmd.Flags().StringVar(&route, "route", "", "search route: text|visual|auto|lexical (default auto)") cmd.Flags().StringVar(&since, "since", "", "YYYY-MM-DD inclusive lower bound on timeline_at") cmd.Flags().StringVar(&until, "until", "", "YYYY-MM-DD inclusive upper bound on timeline_at") cmd.Flags().IntVar(&limit, "limit", 0, "max results (default 10, max 100)") diff --git a/server/internal/api/api.go b/server/internal/api/api.go index e8c6d07..d959ed9 100644 --- a/server/internal/api/api.go +++ b/server/internal/api/api.go @@ -1334,9 +1334,9 @@ func (s *Server) handleSearch(w http.ResponseWriter, r *http.Request) { return } switch req.Route { - case "", search.RouteAuto, search.RouteText, search.RouteVisual: + case "", search.RouteAuto, search.RouteText, search.RouteVisual, search.RouteLexical: default: - writeError(w, http.StatusBadRequest, "bad_route", "route must be auto, text, or visual") + writeError(w, http.StatusBadRequest, "bad_route", "route must be auto, text, visual, or lexical") return } scope, err := pathx.Normalize(req.Scope) diff --git a/server/internal/api/managed_embeddings.go b/server/internal/api/managed_embeddings.go index a4e54f3..d9cd6dc 100644 --- a/server/internal/api/managed_embeddings.go +++ b/server/internal/api/managed_embeddings.go @@ -161,8 +161,9 @@ func (s *Server) managedSearcher( s.Search == nil { return nil, nil, entitlement.ErrEntitlementUnavailable } - // A visual-only query does not invoke the managed text embedding provider. - if query.Route == search.RouteVisual { + // A visual-only or lexical query does not invoke the managed text embedding + // provider. + if query.Route == search.RouteVisual || query.Route == search.RouteLexical { return s.Search, nil, nil } spec, err := s.Search.EmbeddingSpec(r.Context(), query.UserID) diff --git a/server/internal/db/migrations/0024_files_lexical_search.sql b/server/internal/db/migrations/0024_files_lexical_search.sql new file mode 100644 index 0000000..fc3c0c7 --- /dev/null +++ b/server/internal/db/migrations/0024_files_lexical_search.sql @@ -0,0 +1,26 @@ +-- +goose Up +-- Model-free lexical lane for the file corpus. Mirrors the FTS + trigram +-- shape already established for memories (0008) so that filename and path +-- substring search works without an embedding worker. + +-- +goose StatementBegin +ALTER TABLE files + ADD COLUMN IF NOT EXISTS search_tsv tsvector GENERATED ALWAYS AS ( + to_tsvector('simple', coalesce(name, '')) + ) STORED; +-- +goose StatementEnd + +-- +goose StatementBegin +CREATE INDEX IF NOT EXISTS idx_files_search_tsv + ON files USING gin (search_tsv); +-- +goose StatementEnd + +-- +goose StatementBegin +CREATE INDEX IF NOT EXISTS idx_files_name_trgm + ON files USING gin (lower(name) gin_trgm_ops); +-- +goose StatementEnd + +-- +goose Down +DROP INDEX IF EXISTS idx_files_name_trgm; +DROP INDEX IF EXISTS idx_files_search_tsv; +ALTER TABLE files DROP COLUMN IF EXISTS search_tsv; diff --git a/server/internal/search/lexical_test.go b/server/internal/search/lexical_test.go new file mode 100644 index 0000000..2ea2b34 --- /dev/null +++ b/server/internal/search/lexical_test.go @@ -0,0 +1,174 @@ +package search + +import ( + "context" + "os" + "strings" + "testing" + "time" + + "github.com/google/uuid" + "github.com/jackc/pgx/v5/pgxpool" + + memdb "github.com/PeterGuy326/mem/server/internal/db" +) + +func TestLexicalSearchWithoutWorker(t *testing.T) { + dsn := os.Getenv("MEM_TEST_DB") + if dsn == "" { + t.Skip("MEM_TEST_DB not set; skipping lexical search PostgreSQL test") + } + config, err := pgxpool.ParseConfig(dsn) + if err != nil { + t.Fatalf("parse MEM_TEST_DB: %v", err) + } + if !strings.HasSuffix(config.ConnConfig.Database, "_test") { + t.Fatalf( + "refusing to modify non-test database %q; MEM_TEST_DB must end in _test", + config.ConnConfig.Database, + ) + } + ctx, cancel := context.WithTimeout(context.Background(), 120*time.Second) + defer cancel() + database, err := memdb.Open(ctx, dsn) + if err != nil { + t.Fatal(err) + } + t.Cleanup(database.Close) + if err := database.Migrate(ctx); err != nil { + t.Fatal(err) + } + + userID := uuid.New() + if _, err := database.Pool.Exec(ctx, ` + INSERT INTO users (id, email, password_hash) + VALUES ($1, $2, 'test-only') + `, userID, "lexical-"+uuid.NewString()+"@example.test"); err != nil { + t.Fatal(err) + } + t.Cleanup(func() { + cleanupCtx, cleanupCancel := context.WithTimeout(context.Background(), 10*time.Second) + defer cleanupCancel() + database.Pool.Exec(cleanupCtx, `DELETE FROM users WHERE id = $1`, userID) + }) + + now := time.Now().UTC().Truncate(time.Second) + files := []struct { + name string + path string + mime string + }{ + {"quarterly_report.pdf", "/Work/Reports", "application/pdf"}, + {"meeting_notes.md", "/Work/Notes", "text/markdown"}, + {"photo_beach.jpg", "/Photos", "image/jpeg"}, + {"年度总结.docx", "/Work", "application/vnd.openxmlformats-officedocument.wordprocessingml.document"}, + } + for _, f := range files { + if _, err := database.Pool.Exec(ctx, ` + INSERT INTO files ( + user_id, name, path, size, sha256, mime, storage_key, + index_status, created_at, updated_at + ) VALUES ( + $1, $2, $3, 1, $4, $5, $6, + 'ready', $7, $7 + ) + `, userID, f.name, f.path, strings.Repeat("b", 64), f.mime, + "test/"+uuid.NewString(), now); err != nil { + t.Fatal(err) + } + } + + // Service with nil worker — the key precondition for this test. + service := New(database.Pool, nil) + + t.Run("lexical route works without worker", func(t *testing.T) { + hits, err := service.Search(ctx, Query{ + UserID: userID, + Text: "report", + Route: RouteLexical, + Limit: 10, + }) + if err != nil { + t.Fatalf("lexical search failed: %v", err) + } + if len(hits) == 0 { + t.Fatal("lexical search returned no results for 'report'") + } + found := false + for _, h := range hits { + if h.Name == "quarterly_report.pdf" { + found = true + if h.Source != RouteLexical { + t.Errorf("hit source = %q, want %q", h.Source, RouteLexical) + } + break + } + } + if !found { + t.Errorf("expected quarterly_report.pdf in results, got %+v", hits) + } + }) + + t.Run("text route fails without worker", func(t *testing.T) { + _, err := service.Search(ctx, Query{ + UserID: userID, + Text: "report", + Route: RouteText, + Limit: 10, + }) + if err == nil { + t.Fatal("text route should fail without worker") + } + if !strings.Contains(err.Error(), "worker not configured") { + t.Fatalf("unexpected error: %v", err) + } + }) + + t.Run("auto route fails without worker", func(t *testing.T) { + _, err := service.Search(ctx, Query{ + UserID: userID, + Text: "report", + Route: RouteAuto, + Limit: 10, + }) + if err == nil { + t.Fatal("auto route should fail without worker") + } + if !strings.Contains(err.Error(), "worker not configured") { + t.Fatalf("unexpected error: %v", err) + } + }) + + t.Run("trigram match for CJK filename", func(t *testing.T) { + hits, err := service.Search(ctx, Query{ + UserID: userID, + Text: "总结", + Route: RouteLexical, + Limit: 10, + }) + if err != nil { + t.Fatalf("lexical CJK search failed: %v", err) + } + if len(hits) == 0 { + t.Fatal("lexical search returned no results for CJK query '总结'") + } + }) + + t.Run("path filter applies to lexical", func(t *testing.T) { + hits, err := service.Search(ctx, Query{ + UserID: userID, + Text: "notes", + Route: RouteLexical, + PathPrefix: "/Work/Notes", + Limit: 10, + }) + if err != nil { + t.Fatalf("lexical path-filtered search failed: %v", err) + } + for _, h := range hits { + if h.Name == "meeting_notes.md" && h.Path != "/Work/Notes" { + t.Errorf("path filter leaked: got path %q", h.Path) + } + } + }) +} diff --git a/server/internal/search/search.go b/server/internal/search/search.go index c6ff4d0..761d542 100644 --- a/server/internal/search/search.go +++ b/server/internal/search/search.go @@ -149,10 +149,12 @@ var ErrReplayReferenceUnavailable = errors.New("managed embedding replay referen // "text" -> ANN over embeddings_text (Ollama / OpenAI text embedder) // "visual" -> ANN over embeddings_visual via CLIP text encoder // "auto" -> both routes in parallel, merged + deduped by file_id (default) +// "lexical" -> model-free FTS + trigram over files.name (no worker needed) const ( - RouteText = "text" - RouteVisual = "visual" - RouteAuto = "auto" + RouteText = "text" + RouteVisual = "visual" + RouteAuto = "auto" + RouteLexical = "lexical" ) // Hit is one search result. @@ -217,8 +219,10 @@ func (s *Service) Search(ctx context.Context, q Query) ([]Hit, error) { if q.SnippetChars > 16_000 { q.SnippetChars = 16_000 } - if s.worker == nil || !s.worker.Enabled() { - return nil, fmt.Errorf("search disabled: worker not configured") + if q.Route != RouteLexical { + if s.worker == nil || !s.worker.Enabled() { + return nil, fmt.Errorf("search disabled: worker not configured") + } } switch q.Route { @@ -228,8 +232,10 @@ func (s *Service) Search(ctx context.Context, q Query) ([]Hit, error) { return s.searchVisual(ctx, q, text) case "", RouteAuto: return s.searchAuto(ctx, q, text) + case RouteLexical: + return s.searchLexical(ctx, q, text) default: - return nil, fmt.Errorf("unknown route %q (expected text|visual|auto)", q.Route) + return nil, fmt.Errorf("unknown route %q (expected text|visual|auto|lexical)", q.Route) } } @@ -715,6 +721,78 @@ func (s *Service) runVisualANN(ctx context.Context, q Query, vec []float32) ([]H return s.scanHits(ctx, sql, args, RouteVisual, q.SnippetChars) } +// searchLexical is the model-free file recall lane. It uses the same +// three-tier shape as memory Recall (exact phrase → FTS → trigram) so a +// deployment with no worker can still find files by name. +func (s *Service) searchLexical(ctx context.Context, q Query, text string) ([]Hit, error) { + args := []any{q.UserID} + where := []string{"f.user_id = $1"} + args, where = appendPathFilters(args, where, q.PathPrefix, q.AllowedPaths) + args, where = appendMIMEFilter(args, where, q.Type) + if q.Since != nil { + args = append(args, *q.Since) + where = append(where, fmt.Sprintf("COALESCE(f.timeline_at, f.created_at) >= $%d", len(args))) + } + if q.Until != nil { + args = append(args, *q.Until) + where = append(where, fmt.Sprintf("COALESCE(f.timeline_at, f.created_at) <= $%d", len(args))) + } + args = append(args, text) + textArg := len(args) + args = append(args, q.Limit) + limitArg := len(args) + + sql := fmt.Sprintf(` + WITH candidates AS ( + SELECT f.id AS file_id, f.name, f.path, f.mime, f.sha256, + f.summary, f.timeline_at, f.created_at, + strpos(lower(f.name), lower($%d)) > 0 AS exact_phrase, + f.search_tsv @@ plainto_tsquery('simple', $%d) AS fts_match, + ts_rank_cd( + f.search_tsv, + plainto_tsquery('simple', $%d) + )::double precision AS fts_rank, + word_similarity( + lower($%d), + lower(f.name) + )::double precision AS trigram_score + FROM files f + WHERE %s + ), + ranked AS ( + SELECT candidates.*, + CASE + WHEN exact_phrase THEN 1.0::double precision + WHEN fts_match THEN LEAST( + 0.949::double precision, + 0.70::double precision + 0.24::double precision * fts_rank + ) + ELSE LEAST( + 0.699::double precision, + 0.20::double precision + 0.49::double precision * trigram_score + ) + END AS score, + CASE + WHEN exact_phrase THEN 'exact' + WHEN fts_match THEN 'fts' + ELSE 'trigram' + END AS reason + FROM candidates + WHERE exact_phrase + OR fts_match + OR trigram_score >= 0.12 + ) + SELECT 'lexical:' || r.file_id::text, r.file_id, r.name, r.path, r.mime, + r.sha256, -1, r.score::real, r.name, r.summary, + r.timeline_at, r.created_at + FROM ranked r + ORDER BY r.score DESC, r.created_at DESC, r.file_id + LIMIT $%d + `, textArg, textArg, textArg, textArg, strings.Join(where, " AND "), limitArg) + + return s.scanHits(ctx, sql, args, RouteLexical, q.SnippetChars) +} + // scanHits is the common cursor → []Hit loop. Tags every hit with its source route. func (s *Service) scanHits(ctx context.Context, sql string, args []any, route string, snippetChars int) ([]Hit, error) { rows, err := s.pool.Query(ctx, sql, args...) From 13ebe9efa6ba3c733199d374e8db3d107f598ca2 Mon Sep 17 00:00:00 2001 From: liyuanyang Date: Tue, 8 Sep 2026 13:39:38 +0800 Subject: [PATCH 2/2] refactor(search): drop unused reason column from lexical CTE --- server/internal/search/search.go | 7 +------ 1 file changed, 1 insertion(+), 6 deletions(-) diff --git a/server/internal/search/search.go b/server/internal/search/search.go index 761d542..993da16 100644 --- a/server/internal/search/search.go +++ b/server/internal/search/search.go @@ -771,12 +771,7 @@ func (s *Service) searchLexical(ctx context.Context, q Query, text string) ([]Hi 0.699::double precision, 0.20::double precision + 0.49::double precision * trigram_score ) - END AS score, - CASE - WHEN exact_phrase THEN 'exact' - WHEN fts_match THEN 'fts' - ELSE 'trigram' - END AS reason + END AS score FROM candidates WHERE exact_phrase OR fts_match