Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
7 changes: 7 additions & 0 deletions CHANGELOG.md
Original file line number Diff line number Diff line change
Expand Up @@ -7,6 +7,13 @@ The project publishes 0.x prerelease versions; a stable release line is not yet

## [Unreleased]

### Added

- File search gains a model-free lexical route (`route=lexical`). FTS + trigram
over `files.name` — same tier shape as memory Recall — so a deployment with
no embedding worker can still find files by name. CLI: `mem search "query"
--route lexical`.

### Changed

- Migrate GitHub repository, Release, issue, badge, and raw-content coordinates
Expand Down
3 changes: 2 additions & 1 deletion SPEC.md
Original file line number Diff line number Diff line change
Expand Up @@ -146,7 +146,7 @@
| F5.1 | `mem context "..."` → 返回有大小预算的文件/结构化记忆证据包 |
| F5.2 | 每条 evidence 必须含 source kind/id、稳定 citation、内容哈希、片段和 locator |
| F5.3 | mem 只走 recall → context pack;回答与行动由调用方 Agent 完成 |
| F5.4 | `source=all|file|memory`;结构化记忆在无 Worker、无模型时也必须可立即召回 |
| F5.4 | `source=all|file|memory`;结构化记忆在无 Worker、无模型时也必须可立即召回;文件词法路由(`route=lexical`)同样无需 Worker |
| F5.5 | 联合召回单路失败但仍有证据时返回 `200 + partial=true + warnings[]`;无幸存证据时返回 `502 context_unavailable` |

### F5A · 结构化 Agent 记忆
Expand Down Expand Up @@ -534,6 +534,7 @@ embeddings_face (
- `folders (user_id, path)` UNIQUE — 路径唯一性约束
- `memories (workspace_id, idempotency_key_sha256)` UNIQUE — 不落明文幂等键的幂等写入
- `memories` FTS + trigram — 无模型的确定性立即召回
- `files` FTS + trigram — 文件名无模型词法召回(worker 不可用时降级至此)
- `embeddings_* (embedding)` — pgvector HNSW
- `file_entities (entity_id)` — 反查"和某人有关的所有文件"

Expand Down
5 changes: 3 additions & 2 deletions docs/mcp.md
Original file line number Diff line number Diff line change
Expand Up @@ -138,7 +138,7 @@ The canonical product surface is:
| `mem_checkpoint_list` | List newest-first bounded checkpoint summaries for one task |
| `mem_checkpoint_get` | Get one immutable checkpoint and its full handoff payload |
| `mem_resume` | Restore the current task head or a selected historical checkpoint, including resolved and missing evidence |
| `mem_search` | Natural-language search (text / visual / auto fuse); ranked files + snippets |
| `mem_search` | Natural-language search (text / visual / auto fuse); ranked files + snippets. `route=lexical` is model-free (FTS + trigram over file names, no worker needed) |
| `mem_context` | Build an evidence-backed context pack for the calling Agent |
| `mem_related` | Top-K files related to a `file_id` by embedding similarity |
| `mem_face` | Person clusters: `action=list` / `name` / `merge` |
Expand Down Expand Up @@ -295,7 +295,8 @@ same logical request should supply and retain a stable key so a committed
result can replay without another provider invocation or charge. A `504`
means the provider outcome is uncertain: do not automatically retry, and do
not invent a new key. `mem_context` with `source=memory` stays lexical and
model-independent.
model-independent. `mem_search` with `route=lexical` is likewise model-free:
it uses FTS + trigram over file names and works without a configured worker.

Its target output is structured for an Agent to consume:

Expand Down
2 changes: 1 addition & 1 deletion server/cmd/mem/cmds_search.go
Original file line number Diff line number Diff line change
Expand Up @@ -101,7 +101,7 @@ func newSearchCmd() *cobra.Command {
},
}
cmd.Flags().StringVar(&typ, "type", "", "mime prefix filter: image|text|application|audio|video")
cmd.Flags().StringVar(&route, "route", "", "search route: text|visual|auto (default auto)")
cmd.Flags().StringVar(&route, "route", "", "search route: text|visual|auto|lexical (default auto)")
cmd.Flags().StringVar(&since, "since", "", "YYYY-MM-DD inclusive lower bound on timeline_at")
cmd.Flags().StringVar(&until, "until", "", "YYYY-MM-DD inclusive upper bound on timeline_at")
cmd.Flags().IntVar(&limit, "limit", 0, "max results (default 10, max 100)")
Expand Down
4 changes: 2 additions & 2 deletions server/internal/api/api.go
Original file line number Diff line number Diff line change
Expand Up @@ -1334,9 +1334,9 @@ func (s *Server) handleSearch(w http.ResponseWriter, r *http.Request) {
return
}
switch req.Route {
case "", search.RouteAuto, search.RouteText, search.RouteVisual:
case "", search.RouteAuto, search.RouteText, search.RouteVisual, search.RouteLexical:
default:
writeError(w, http.StatusBadRequest, "bad_route", "route must be auto, text, or visual")
writeError(w, http.StatusBadRequest, "bad_route", "route must be auto, text, visual, or lexical")
return
}
scope, err := pathx.Normalize(req.Scope)
Expand Down
5 changes: 3 additions & 2 deletions server/internal/api/managed_embeddings.go
Original file line number Diff line number Diff line change
Expand Up @@ -161,8 +161,9 @@ func (s *Server) managedSearcher(
s.Search == nil {
return nil, nil, entitlement.ErrEntitlementUnavailable
}
// A visual-only query does not invoke the managed text embedding provider.
if query.Route == search.RouteVisual {
// A visual-only or lexical query does not invoke the managed text embedding
// provider.
if query.Route == search.RouteVisual || query.Route == search.RouteLexical {
return s.Search, nil, nil
}
spec, err := s.Search.EmbeddingSpec(r.Context(), query.UserID)
Expand Down
26 changes: 26 additions & 0 deletions server/internal/db/migrations/0024_files_lexical_search.sql
Original file line number Diff line number Diff line change
@@ -0,0 +1,26 @@
-- +goose Up
-- Model-free lexical lane for the file corpus. Mirrors the FTS + trigram
-- shape already established for memories (0008) so that filename and path
-- substring search works without an embedding worker.

-- +goose StatementBegin
ALTER TABLE files
ADD COLUMN IF NOT EXISTS search_tsv tsvector GENERATED ALWAYS AS (
to_tsvector('simple', coalesce(name, ''))
) STORED;
-- +goose StatementEnd

-- +goose StatementBegin
CREATE INDEX IF NOT EXISTS idx_files_search_tsv
ON files USING gin (search_tsv);
-- +goose StatementEnd

-- +goose StatementBegin
CREATE INDEX IF NOT EXISTS idx_files_name_trgm
ON files USING gin (lower(name) gin_trgm_ops);
-- +goose StatementEnd

-- +goose Down
DROP INDEX IF EXISTS idx_files_name_trgm;
DROP INDEX IF EXISTS idx_files_search_tsv;
ALTER TABLE files DROP COLUMN IF EXISTS search_tsv;
174 changes: 174 additions & 0 deletions server/internal/search/lexical_test.go
Original file line number Diff line number Diff line change
@@ -0,0 +1,174 @@
package search

import (
"context"
"os"
"strings"
"testing"
"time"

"github.com/google/uuid"
"github.com/jackc/pgx/v5/pgxpool"

memdb "github.com/PeterGuy326/mem/server/internal/db"
)

func TestLexicalSearchWithoutWorker(t *testing.T) {
dsn := os.Getenv("MEM_TEST_DB")
if dsn == "" {
t.Skip("MEM_TEST_DB not set; skipping lexical search PostgreSQL test")
}
config, err := pgxpool.ParseConfig(dsn)
if err != nil {
t.Fatalf("parse MEM_TEST_DB: %v", err)
}
if !strings.HasSuffix(config.ConnConfig.Database, "_test") {
t.Fatalf(
"refusing to modify non-test database %q; MEM_TEST_DB must end in _test",
config.ConnConfig.Database,
)
}
ctx, cancel := context.WithTimeout(context.Background(), 120*time.Second)
defer cancel()
database, err := memdb.Open(ctx, dsn)
if err != nil {
t.Fatal(err)
}
t.Cleanup(database.Close)
if err := database.Migrate(ctx); err != nil {
t.Fatal(err)
}

userID := uuid.New()
if _, err := database.Pool.Exec(ctx, `
INSERT INTO users (id, email, password_hash)
VALUES ($1, $2, 'test-only')
`, userID, "lexical-"+uuid.NewString()+"@example.test"); err != nil {
t.Fatal(err)
}
t.Cleanup(func() {
cleanupCtx, cleanupCancel := context.WithTimeout(context.Background(), 10*time.Second)
defer cleanupCancel()
database.Pool.Exec(cleanupCtx, `DELETE FROM users WHERE id = $1`, userID)
})

now := time.Now().UTC().Truncate(time.Second)
files := []struct {
name string
path string
mime string
}{
{"quarterly_report.pdf", "/Work/Reports", "application/pdf"},
{"meeting_notes.md", "/Work/Notes", "text/markdown"},
{"photo_beach.jpg", "/Photos", "image/jpeg"},
{"年度总结.docx", "/Work", "application/vnd.openxmlformats-officedocument.wordprocessingml.document"},
}
for _, f := range files {
if _, err := database.Pool.Exec(ctx, `
INSERT INTO files (
user_id, name, path, size, sha256, mime, storage_key,
index_status, created_at, updated_at
) VALUES (
$1, $2, $3, 1, $4, $5, $6,
'ready', $7, $7
)
`, userID, f.name, f.path, strings.Repeat("b", 64), f.mime,
"test/"+uuid.NewString(), now); err != nil {
t.Fatal(err)
}
}

// Service with nil worker — the key precondition for this test.
service := New(database.Pool, nil)

t.Run("lexical route works without worker", func(t *testing.T) {
hits, err := service.Search(ctx, Query{
UserID: userID,
Text: "report",
Route: RouteLexical,
Limit: 10,
})
if err != nil {
t.Fatalf("lexical search failed: %v", err)
}
if len(hits) == 0 {
t.Fatal("lexical search returned no results for 'report'")
}
found := false
for _, h := range hits {
if h.Name == "quarterly_report.pdf" {
found = true
if h.Source != RouteLexical {
t.Errorf("hit source = %q, want %q", h.Source, RouteLexical)
}
break
}
}
if !found {
t.Errorf("expected quarterly_report.pdf in results, got %+v", hits)
}
})

t.Run("text route fails without worker", func(t *testing.T) {
_, err := service.Search(ctx, Query{
UserID: userID,
Text: "report",
Route: RouteText,
Limit: 10,
})
if err == nil {
t.Fatal("text route should fail without worker")
}
if !strings.Contains(err.Error(), "worker not configured") {
t.Fatalf("unexpected error: %v", err)
}
})

t.Run("auto route fails without worker", func(t *testing.T) {
_, err := service.Search(ctx, Query{
UserID: userID,
Text: "report",
Route: RouteAuto,
Limit: 10,
})
if err == nil {
t.Fatal("auto route should fail without worker")
}
if !strings.Contains(err.Error(), "worker not configured") {
t.Fatalf("unexpected error: %v", err)
}
})

t.Run("trigram match for CJK filename", func(t *testing.T) {
hits, err := service.Search(ctx, Query{
UserID: userID,
Text: "总结",
Route: RouteLexical,
Limit: 10,
})
if err != nil {
t.Fatalf("lexical CJK search failed: %v", err)
}
if len(hits) == 0 {
t.Fatal("lexical search returned no results for CJK query '总结'")
}
})

t.Run("path filter applies to lexical", func(t *testing.T) {
hits, err := service.Search(ctx, Query{
UserID: userID,
Text: "notes",
Route: RouteLexical,
PathPrefix: "/Work/Notes",
Limit: 10,
})
if err != nil {
t.Fatalf("lexical path-filtered search failed: %v", err)
}
for _, h := range hits {
if h.Name == "meeting_notes.md" && h.Path != "/Work/Notes" {
t.Errorf("path filter leaked: got path %q", h.Path)
}
}
})
}
85 changes: 79 additions & 6 deletions server/internal/search/search.go
Original file line number Diff line number Diff line change
Expand Up @@ -149,10 +149,12 @@ var ErrReplayReferenceUnavailable = errors.New("managed embedding replay referen
// "text" -> ANN over embeddings_text (Ollama / OpenAI text embedder)
// "visual" -> ANN over embeddings_visual via CLIP text encoder
// "auto" -> both routes in parallel, merged + deduped by file_id (default)
// "lexical" -> model-free FTS + trigram over files.name (no worker needed)
const (
RouteText = "text"
RouteVisual = "visual"
RouteAuto = "auto"
RouteText = "text"
RouteVisual = "visual"
RouteAuto = "auto"
RouteLexical = "lexical"
)

// Hit is one search result.
Expand Down Expand Up @@ -217,8 +219,10 @@ func (s *Service) Search(ctx context.Context, q Query) ([]Hit, error) {
if q.SnippetChars > 16_000 {
q.SnippetChars = 16_000
}
if s.worker == nil || !s.worker.Enabled() {
return nil, fmt.Errorf("search disabled: worker not configured")
if q.Route != RouteLexical {
if s.worker == nil || !s.worker.Enabled() {
return nil, fmt.Errorf("search disabled: worker not configured")
}
}

switch q.Route {
Expand All @@ -228,8 +232,10 @@ func (s *Service) Search(ctx context.Context, q Query) ([]Hit, error) {
return s.searchVisual(ctx, q, text)
case "", RouteAuto:
return s.searchAuto(ctx, q, text)
case RouteLexical:
return s.searchLexical(ctx, q, text)
default:
return nil, fmt.Errorf("unknown route %q (expected text|visual|auto)", q.Route)
return nil, fmt.Errorf("unknown route %q (expected text|visual|auto|lexical)", q.Route)
}
}

Expand Down Expand Up @@ -715,6 +721,73 @@ func (s *Service) runVisualANN(ctx context.Context, q Query, vec []float32) ([]H
return s.scanHits(ctx, sql, args, RouteVisual, q.SnippetChars)
}

// searchLexical is the model-free file recall lane. It uses the same
// three-tier shape as memory Recall (exact phrase → FTS → trigram) so a
// deployment with no worker can still find files by name.
func (s *Service) searchLexical(ctx context.Context, q Query, text string) ([]Hit, error) {
args := []any{q.UserID}
where := []string{"f.user_id = $1"}
args, where = appendPathFilters(args, where, q.PathPrefix, q.AllowedPaths)
args, where = appendMIMEFilter(args, where, q.Type)
if q.Since != nil {
args = append(args, *q.Since)
where = append(where, fmt.Sprintf("COALESCE(f.timeline_at, f.created_at) >= $%d", len(args)))
}
if q.Until != nil {
args = append(args, *q.Until)
where = append(where, fmt.Sprintf("COALESCE(f.timeline_at, f.created_at) <= $%d", len(args)))
}
args = append(args, text)
textArg := len(args)
args = append(args, q.Limit)
limitArg := len(args)

sql := fmt.Sprintf(`
WITH candidates AS (
SELECT f.id AS file_id, f.name, f.path, f.mime, f.sha256,
f.summary, f.timeline_at, f.created_at,
strpos(lower(f.name), lower($%d)) > 0 AS exact_phrase,
f.search_tsv @@ plainto_tsquery('simple', $%d) AS fts_match,
ts_rank_cd(
f.search_tsv,
plainto_tsquery('simple', $%d)
)::double precision AS fts_rank,
word_similarity(
lower($%d),
lower(f.name)
)::double precision AS trigram_score
FROM files f
WHERE %s
),
ranked AS (
SELECT candidates.*,
CASE
WHEN exact_phrase THEN 1.0::double precision
WHEN fts_match THEN LEAST(
0.949::double precision,
0.70::double precision + 0.24::double precision * fts_rank
)
ELSE LEAST(
0.699::double precision,
0.20::double precision + 0.49::double precision * trigram_score
)
END AS score
FROM candidates
WHERE exact_phrase
OR fts_match
OR trigram_score >= 0.12
)
SELECT 'lexical:' || r.file_id::text, r.file_id, r.name, r.path, r.mime,
r.sha256, -1, r.score::real, r.name, r.summary,
r.timeline_at, r.created_at
FROM ranked r
ORDER BY r.score DESC, r.created_at DESC, r.file_id
LIMIT $%d
`, textArg, textArg, textArg, textArg, strings.Join(where, " AND "), limitArg)

return s.scanHits(ctx, sql, args, RouteLexical, q.SnippetChars)
}

// scanHits is the common cursor → []Hit loop. Tags every hit with its source route.
func (s *Service) scanHits(ctx context.Context, sql string, args []any, route string, snippetChars int) ([]Hit, error) {
rows, err := s.pool.Query(ctx, sql, args...)
Expand Down
Loading