From d4509bcd90682cb9d5531b98db11174e2ae905ad Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?=E4=BF=AE=E9=9B=A8?= <47820304+PeterGuy326@users.noreply.github.com> Date: Thu, 17 Sep 2026 17:21:28 +0800 Subject: [PATCH 1/2] fix(index): verify HNSW foundation without overstating planner use --- CHANGELOG.md | 2 + SPEC.md | 7 +- docs/VALIDATION_HNSW.md | 144 ++++++++++++++++++ scripts/verify.sh | 15 +- scripts/verify_hnsw_indexes.sh | 65 ++++++++ server/internal/db/hnsw_migration_test.go | 126 +++++++++++++++ server/internal/db/migrations/0001_init.sql | 3 +- .../0019_versioned_index_generations.sql | 4 +- .../db/migrations/0025_ann_hnsw_indexes.sql | 40 +++++ server/internal/face/face.go | 4 +- server/internal/search/hnsw_semantics_test.go | 117 ++++++++++++++ 11 files changed, 518 insertions(+), 9 deletions(-) create mode 100644 docs/VALIDATION_HNSW.md create mode 100755 scripts/verify_hnsw_indexes.sh create mode 100644 server/internal/db/hnsw_migration_test.go create mode 100644 server/internal/db/migrations/0025_ann_hnsw_indexes.sql create mode 100644 server/internal/search/hnsw_semantics_test.go diff --git a/CHANGELOG.md b/CHANGELOG.md index ec8970d..a77f633 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -101,6 +101,8 @@ The project publishes 0.x prerelease versions; a stable release line is not yet ### Fixed +- Add cosine HNSW indexes for legacy embedding tables while explicitly preserving the exact per-file text-search semantics and keeping the broader planner-use acceptance open; see `docs/VALIDATION_HNSW.md`. + - Follow the shared design language for reading, numeric and action alignment; generate the existing Web color variables from a pinned design-system token snapshot, and use a single consistent empty-state pattern. Refs #211. - Improve web caption and status contrast in both themes, including tinted danger diff --git a/SPEC.md b/SPEC.md index 277caa8..29c8c7e 100644 --- a/SPEC.md +++ b/SPEC.md @@ -537,7 +537,12 @@ embeddings_face ( - `files` FTS + trigram — 显式指定 `route=lexical` 的文件名无模型词法召回; `auto` 只融合 text/visual,不自动回退到 lexical,worker 不可用时仍报错。 仅搜索 `files.name`,路径只用于筛选,不检索文件正文或路径片段。 -- `embeddings_* (embedding)` — pgvector HNSW +- `embeddings_* (embedding)` — cosine pgvector HNSW (migration 0025). + Index availability does not guarantee route use: text search still selects + the best chunk per file exactly before top-k, visual planning depends on the + corpus and filters, and face clustering remains in Go. See + [HNSW validation](docs/VALIDATION_HNSW.md); #173 planner/recall acceptance is + not completed by the index migration. - `file_entities (entity_id)` — 反查"和某人有关的所有文件" ### 6.3 文件夹一致性规则(重要) diff --git a/docs/VALIDATION_HNSW.md b/docs/VALIDATION_HNSW.md new file mode 100644 index 0000000..c1526c5 --- /dev/null +++ b/docs/VALIDATION_HNSW.md @@ -0,0 +1,144 @@ +# HNSW index foundation — partial delivery for #173 + +Migration `0025_ann_hnsw_indexes.sql` adds cosine HNSW indexes to text (768), +visual (512), and face (512) embeddings. Migration 0024 belongs to #183 and +0026 to #185. The integration runner declares this branch's head as 25. + +## Scope and acceptance + +This is the index foundation portion of #173, tracked with **Refs #173**. +It does not complete that issue. The shipping text route remains exact and is +not rewritten: a bounded ANN candidate set cannot establish best-chunk-per-file +results merely by returning k distinct files. Underfill fallback alone also +does not establish that unseen files or better chunks cannot outrank them. +A retrieval-policy change needs its own continuation/exhaustion/fallback and +quality acceptance. The strict planner gate below is retained unchanged in +meaning, and still fails for text. + +The independently testable criteria for this partial change are: + +- a populated 24 → 25 → 24 → 25 migration preserves text/visual/face vectors; +- all three valid indexes use HNSW and cosine operators; +- ingest and ordinary startup work after the migration; wrong dimensions fail; +- shipping text search still returns k distinct eligible files with the best + chunk even when one file holds 101 nearest chunks; owner, literal path, + allow-list, MIME and time boundaries remain enforced; +- the separate full-issue planner check cannot report success when text misses + HNSW. Face DDL is never reported as a shipping face search improvement. + +This branch requires #194's lexical migration 0024; the +[cumulative migration sequence](MIGRATION_SEQUENCE.md) is included by that base. +Deployment order stays #194 → #197 → #195. Accepting this partial scope is a +review decision, not a claim that the remaining #173 planner criterion passed. + +## Reproducible local gates + +```bash +# Creates separate owned test databases, including a populated HNSW round trip, +# and runs the shipping text semantics regression. MEM_TEST_DB ends in _test. +MEM_TEST_DB="$MEM_TEST_DB" ./scripts/verify.sh integration +MEM_TEST_DB="$MEM_TEST_DB" ./scripts/verify.sh integration-race +``` + +`TestHNSWMigrationPostgres` requires a fresh database via `MEM_HNSW_TEST_DB` +and refuses existing Goose history. It seeds 2,000 vectors in each table before +index construction, checks rollback preservation, repeats the upgrade, writes +one more vector in each table, rejects wrong dimensions, and calls `DB.Migrate`. +`TestTextANNFileSemanticsPostgres` exercises the actual `runTextANN` method; +it is a result-semantics regression, not a planner-use assertion. + +## Full #173 planner gate (still unmet) + +Use PostgreSQL 16+ with pgvector, all migrations applied, and a populated +synthetic corpus in a disposable database whose name ends in `_test`: + +```bash +bash scripts/verify_hnsw_indexes.sh "$MEM_TEST_DB" "$CORPUS_USER_UUID" +``` + +The read-only script validates the exact indexes and non-null vector counts +for the supplied corpus owner, and runs EXPLAIN ANALYZE for the shipping text +and visual ordering/deduplication shapes. It checks each route's index name, +propagates SQL errors, and exits nonzero if either route misses HNSW. +It does not disable sequential scans or claim a production latency threshold. + +## Remaining acceptance boundary + +The shipping text query already uses `DISTINCT ON (f.id)` ordered by file ID +before global top-k selection. A simplified `ORDER BY distance LIMIT` query +does not prove that this production query uses HNSW. This query shape is +unchanged from main: a failed text planner gate is an unmet optimization +criterion, not a regression introduced by the index DDL. The full-issue verification +must remain failed until that criterion is met. The partial index scope above +does not weaken that check. This is still an unmet feature acceptance criterion even +though the query predates this PR; an unchanged baseline is not a waiver. + +### Historical bounded rewrite investigation (2026-09-10) + +The smallest tested direct-distance rewrite filtered each chunk with a +`NOT EXISTS` peer having a smaller distance (UUID tie-break), then ordered by +`e.embedding <=> query_vector LIMIT 10`. PostgreSQL 17.10 / pgvector 0.8.3 +selected `idx_embeddings_text_embedding_hnsw`, with the existing file-ID +index serving the peer lookup, on the 2,000-file synthetic fixture. Planner +selection alone did not establish equivalent results. + +A rollback-only 768-dimensional counterexample separated one file's 101 near +chunks from the other 1,999 files: the first near vector was `[1,0,0,...]`, +the next 100 were `[1,i*0.001,0,...]`, and other files used `[0,1,0,...]`. +At unmodified defaults (`hnsw.ef_search=40`, `hnsw.iterative_scan=off`), the +shipping exact per-file query returned 10 files; the indexed anti-join returned +only 1. EXPLAIN ANALYZE showed the HNSW scan returning 40 candidate chunks, +39 then eliminated by per-file deduplication. All fixture mutations rolled +back. This is synthetic semantic/planner evidence, not latency evidence. + +A separate exact-distance counterexample rules out a fixed oversampling cap: +81 closest chunks belonging to one file consume `8*k=80` candidates for +`k=10`, leaving one file after deduplication when ten files exist. Increasing +a fixed multiplier cannot guarantee k distinct files for unbounded chunk counts. + +Concrete design blocker: the shipping contract chooses the best chunk per +eligible file before global top-k. A bounded approximate candidate scan can +underfill after deduplication or filtering. A safe rewrite needs a tested +candidate-exhaustion/continuation and fallback policy, including per-file +best-chunk selection and existing authorization/path/MIME/time filters. +Enabling iterative scans alone still needs an explicit scan-limit/exhaustion +policy; it is not proof of equivalence. That policy is not implemented or +accepted here. The promising anti-join is therefore not shipped, the original +query remains unchanged, and the text planner check must continue failing. +No planner settings or acceptance criteria were weakened. + +Face indexing has no shipping SQL search route; its evidence is valid DDL and +populated-table migration, not a measured face-query speedup. + +Recall, real embedding quality, production latency, index build time and index +size are NOT VERIFIED. #175 / #184 track live retrieval benchmark evidence; +fixture tests are not live quality results. No numerical improvement or recall +percentage is asserted here. + +Fixed `vector(768)` / `vector(512)` column types reject wrong dimensions on +insert, before HNSW construction. NULL vectors can remain in the tables but +are not indexed. Down removes only the three new indexes and retains data. + +## Operational limits + +Migration 0025 remains transactional, consistent with startup migrations. +Regular `CREATE INDEX` blocks writes on the indexed tables until commit; plan +an ingestion maintenance window for a populated deployment. The rollback test +proves data preservation, not uninterrupted concurrent ingest or zero downtime. +`CREATE INDEX CONCURRENTLY` would need a separate nontransactional recovery +policy, including invalid/partially built indexes, and is not silently adopted. + +The indexes retain pgvector defaults `m=16`, `ef_construction=64`. Follow +[pgvector's index/scan documentation](https://github.com/pgvector/pgvector#hnsw) +and measure representative recall, build time and ingest cost before tuning; +there is no asserted safe row-count threshold. Iterative scans still have +scan/memory limits and do not establish exact file-level result equivalence. +NULL and zero vectors are not indexed for cosine distance. + +The strict verification script stops immediately on SQL/connection errors. +Its pass/fail counters summarize completed assertions, not infrastructure +errors; an interrupted run is never a pass. No `|| true` masks SQL failures. + +The former MinIO lifecycle pull failure is fixed by the inherited main commit +`f1cc9eb` (#209), which uses pinned `quay.io/minio` images. No extra image or +workflow override belongs to this HNSW diff. diff --git a/scripts/verify.sh b/scripts/verify.sh index b8a45ad..0c4da42 100755 --- a/scripts/verify.sh +++ b/scripts/verify.sh @@ -4,7 +4,7 @@ set -euo pipefail REPO_ROOT="$(cd -- "$(dirname -- "${BASH_SOURCE[0]}")/.." && pwd)" MODE="${1:-unit}" -EXPECTED_MIGRATION_HEAD=24 +EXPECTED_MIGRATION_HEAD=25 MIGRATION_ROLLBACK_TARGET=11 MODEL_TEXT_CANONICAL_BASE=15 WORKSPACE_AI_PROFILE_BASE=16 @@ -349,6 +349,7 @@ run_postgres_tests() { TestManagedAISettlementOutboxPostgres TestReleasedFileStageRetryPostgres TestDurableContextPostgres + TestTextANNFileSemanticsPostgres ) integration_log="$(mktemp "${TMPDIR:-/tmp}/mem-integration.XXXXXX")" @@ -359,7 +360,7 @@ run_postgres_tests() { MEM_TEST_DB="$MEM_TEST_DB" go test \ ${race_flag:+"$race_flag"} \ -v -count=1 -p 1 -timeout 20m \ - -run '^(TestMemoryPostgres|TestHandoffPostgres|TestWorkspaceTransferPostgres|TestWorkspaceTransferMergeConservativePostgres|TestHandoffCrossAgentHTTPIntegration|TestRelocateHTTPPostgres|TestMemoryPathLifecycleIntegration|TestWorkspacePathLockingIntegration|TestFilePathLockingIntegration|TestAnnotationDecisionIntegration|TestIndexerEnrichmentIntegration|TestRecomputePerson|TestManagedEmbeddingEntitlementPostgres|TestManagedSearchReplayPostgres|TestManagedEmbeddingHTTPAuthorizationPostgres|TestAIProfilePostgres|TestIndexGenerationPostgres|TestManagedAISettlementOutboxPostgres|TestReleasedFileStageRetryPostgres|TestDurableContextPostgres)$' \ + -run '^(TestMemoryPostgres|TestHandoffPostgres|TestWorkspaceTransferPostgres|TestWorkspaceTransferMergeConservativePostgres|TestHandoffCrossAgentHTTPIntegration|TestRelocateHTTPPostgres|TestMemoryPathLifecycleIntegration|TestWorkspacePathLockingIntegration|TestFilePathLockingIntegration|TestAnnotationDecisionIntegration|TestIndexerEnrichmentIntegration|TestRecomputePerson|TestManagedEmbeddingEntitlementPostgres|TestManagedSearchReplayPostgres|TestManagedEmbeddingHTTPAuthorizationPostgres|TestAIProfilePostgres|TestIndexGenerationPostgres|TestManagedAISettlementOutboxPostgres|TestReleasedFileStageRetryPostgres|TestDurableContextPostgres|TestTextANNFileSemanticsPostgres)$' \ ./internal/memory \ ./internal/handoff \ ./internal/workspacetransfer \ @@ -397,9 +398,19 @@ run_integration() { validate_test_database with_fresh_test_database migration_sequence run_migration_sequence with_fresh_test_database migration run_migration_round_trip + with_fresh_test_database hnsw_migration run_hnsw_migration with_fresh_test_database integration run_postgres_integration } +run_hnsw_migration() { + log "Populated HNSW migration, rollback and ingest regression" + ( + cd "${REPO_ROOT}/server" + MEM_HNSW_TEST_DB="$MEM_TEST_DB" go test -v -count=1 \ + -run '^TestHNSWMigrationPostgres$' ./internal/db + ) +} + run_integration_race() { validate_test_database with_fresh_test_database integration_race run_postgres_integration_race diff --git a/scripts/verify_hnsw_indexes.sh b/scripts/verify_hnsw_indexes.sh new file mode 100755 index 0000000..7876ba2 --- /dev/null +++ b/scripts/verify_hnsw_indexes.sh @@ -0,0 +1,65 @@ +#!/usr/bin/env bash +# Read-only planner verification for shipping text and visual query shapes. +# Requires a populated disposable database; does not force planner settings. +set -euo pipefail +# pass/fail counts are planner assertions only. SQL/connection errors abort +# immediately (rather than manufacture an assertion result or hide an error). +trap 'echo "ERROR: HNSW verification aborted on an execution error; assertions are incomplete" >&2' ERR +DB_URL="${1:?Usage: $0 }" +CORPUS_USER="${2:?Supply the user UUID that owns the populated corpus}" +[[ "$CORPUS_USER" =~ ^[0-9a-fA-F]{8}-[0-9a-fA-F]{4}-[0-9a-fA-F]{4}-[0-9a-fA-F]{4}-[0-9a-fA-F]{12}$ ]] || { + echo 'ERROR: corpus user must be a UUID' >&2; exit 1; +} +# CORPUS_USER is interpolated below only after the strict UUID validation. +sql() { psql -X -A -t -v ON_ERROR_STOP=1 "$DB_URL" -c "$1"; } +db_name="$(sql 'SELECT current_database()')" +[[ "$db_name" == *_test ]] || { echo 'ERROR: database must end in _test' >&2; exit 1; } +pass=0 +fail=0 +for kind in text visual face; do + index="idx_embeddings_${kind}_embedding_hnsw" + valid="$(sql "SELECT count(*) FROM pg_index i + JOIN pg_class c ON c.oid = i.indexrelid JOIN pg_am a ON a.oid = c.relam + WHERE i.indrelid = 'embeddings_${kind}'::regclass + AND c.relname = '${index}' AND a.amname = 'hnsw' AND i.indisvalid + AND pg_get_indexdef(i.indexrelid) LIKE '%vector_cosine_ops%'")" + rows="$(sql "SELECT count(*) FROM embeddings_${kind} e JOIN files f ON f.id=e.file_id + WHERE f.user_id='${CORPUS_USER}'::uuid AND e.embedding IS NOT NULL")" + if [[ "$valid" == 1 && "$rows" -gt 0 ]]; then + echo "PASS: ${index} is valid; corpus contains ${rows} non-null vectors" + pass=$((pass + 1)) + else + echo "FAIL: ${index}: valid=${valid}, corpus vectors=${rows}" + fail=$((fail + 1)) + fi +done +assert_plan() { + local route="$1" query="$2" plan + # ANALYZE executes the read so vector/schema errors cannot hide behind EXPLAIN. + plan="$(sql "EXPLAIN (ANALYZE, BUFFERS) ${query}")" + echo "$plan" + if grep -q "Index Scan using idx_embeddings_${route}_embedding_hnsw" <<<"$plan"; then + echo "PASS: shipping ${route} query uses HNSW" + pass=$((pass + 1)) + else + echo "FAIL: shipping ${route} query does not use HNSW" + fail=$((fail + 1)) + fi +} +# Match runTextANN: per-file DISTINCT ON precedes global top-k. A simple +# ORDER BY distance LIMIT probe would not establish this query's index usage. +assert_plan text "SELECT evidence_id, file_id, score FROM ( + SELECT DISTINCT ON (f.id) e.id::text AS evidence_id, f.id AS file_id, + 1 - (e.embedding <=> array_fill(0.1::real, ARRAY[768])::vector) AS score + FROM embeddings_text e JOIN files f ON f.id=e.file_id + WHERE f.user_id='${CORPUS_USER}'::uuid + ORDER BY f.id, e.embedding <=> array_fill(0.1::real, ARRAY[768])::vector ASC +) hits ORDER BY score DESC LIMIT 10" +# Visual vectors have file_id (not e.id) and 512 dimensions. +assert_plan visual "SELECT e.file_id, + (1 - (e.embedding <=> array_fill(0.1::real, ARRAY[512])::vector))::real AS score + FROM embeddings_visual e JOIN files f ON f.id=e.file_id + WHERE f.user_id='${CORPUS_USER}'::uuid + ORDER BY e.embedding <=> array_fill(0.1::real, ARRAY[512])::vector ASC LIMIT 10" +echo "Results: ${pass} passed, ${fail} failed" +[[ "$fail" -eq 0 ]] diff --git a/server/internal/db/hnsw_migration_test.go b/server/internal/db/hnsw_migration_test.go new file mode 100644 index 0000000..0c8a646 --- /dev/null +++ b/server/internal/db/hnsw_migration_test.go @@ -0,0 +1,126 @@ +package db + +import ( + "context" + "database/sql" + "fmt" + "os" + "strings" + "testing" + "time" + + "github.com/jackc/pgx/v5" + "github.com/pressly/goose/v3" +) + +// This is a schema/ingest gate, not the still-unmet shipping text planner gate. +// The runner owns a NEW database so populated upgrade/down/up cannot affect +// another suite's migration history or data. +func TestHNSWMigrationPostgres(t *testing.T) { + dsn := os.Getenv("MEM_HNSW_TEST_DB") + if dsn == "" { + t.Skip("MEM_HNSW_TEST_DB not set; requires a fresh owned test database") + } + cfg, err := pgx.ParseConfig(dsn) + if err != nil { + t.Fatal(err) + } + if !strings.HasSuffix(cfg.Database, "_test") { + t.Fatalf("refusing non-test database %q", cfg.Database) + } + ctx, cancel := context.WithTimeout(context.Background(), 3*time.Minute) + defer cancel() + db, err := sql.Open("pgx", dsn) + if err != nil { + t.Fatal(err) + } + defer db.Close() + var history sql.NullString + if err := db.QueryRowContext(ctx, "SELECT to_regclass('goose_db_version')::text").Scan(&history); err != nil { + t.Fatal(err) + } + if history.Valid { + t.Fatal("refusing existing migration history; provide a new owned test database") + } + goose.SetBaseFS(migrationsFS) + if err := goose.SetDialect("postgres"); err != nil { + t.Fatal(err) + } + if err := goose.UpToContext(ctx, db, "migrations", 24); err != nil { + t.Fatal(err) + } + exec := func(query string) { + t.Helper() + if _, err := db.ExecContext(ctx, query); err != nil { + t.Fatal(err) + } + } + exec(`INSERT INTO users(id,email,password_hash) + VALUES ('00000000-0000-0000-0000-000000000173','hnsw@example.test','test'); + INSERT INTO files(id,user_id,name,path,size,sha256,mime,storage_key) + SELECT md5(i::text)::uuid,'00000000-0000-0000-0000-000000000173', + 'fixture-' || i, '/hnsw', 0, 'fixture', 'text/plain', 'fixture-' || i + FROM generate_series(1,2000) i`) + for _, kind := range []string{"text", "visual", "face"} { + dim, extraCols, extraValues := 512, "", "" + if kind == "text" { + dim, extraCols, extraValues = 768, ",chunk_index,chunk_text", ",0,'fixture'" + } + exec(fmt.Sprintf(`INSERT INTO embeddings_%s(file_id,embedding%s) + SELECT id,array_fill(0.1::real,ARRAY[%d])::vector%s FROM files`, kind, extraCols, dim, extraValues)) + } + assertState := func(wantIndexes, wantRows int) { + t.Helper() + for _, kind := range []string{"text", "visual", "face"} { + var indexes, rows int + if err := db.QueryRowContext(ctx, `SELECT count(*) FROM pg_index i + JOIN pg_class c ON c.oid=i.indexrelid JOIN pg_am a ON a.oid=c.relam + WHERE i.indrelid=($1::text)::regclass AND c.relname=$2 AND a.amname='hnsw' + AND i.indisvalid AND pg_get_indexdef(i.indexrelid) LIKE '%vector_cosine_ops%'`, + "embeddings_"+kind, "idx_embeddings_"+kind+"_embedding_hnsw").Scan(&indexes); err != nil { + t.Fatal(err) + } + if err := db.QueryRowContext(ctx, "SELECT count(*) FROM embeddings_"+kind+" WHERE embedding IS NOT NULL").Scan(&rows); err != nil { + t.Fatal(err) + } + if indexes != wantIndexes || rows != wantRows { + t.Fatalf("%s: valid cosine indexes=%d want=%d, preserved vectors=%d want=%d", kind, indexes, wantIndexes, rows, wantRows) + } + } + } + assertState(0, 2000) + if err := goose.UpToContext(ctx, db, "migrations", 25); err != nil { + t.Fatal(err) + } + assertState(1, 2000) + if err := goose.DownToContext(ctx, db, "migrations", 24); err != nil { + t.Fatal(err) + } + assertState(0, 2000) + if err := goose.UpToContext(ctx, db, "migrations", 25); err != nil { + t.Fatal(err) + } + assertState(1, 2000) + // Ingest remains writable after index construction (no concurrency claim). + exec(`INSERT INTO files(id,user_id,name,path,size,sha256,mime,storage_key) + VALUES (md5('2001')::uuid,'00000000-0000-0000-0000-000000000173', + 'after-index','/hnsw',0,'fixture','text/plain','after-index')`) + for _, kind := range []string{"text", "visual", "face"} { + dim, extraCols, extraValues := 512, "", "" + if kind == "text" { + dim, extraCols, extraValues = 768, ",chunk_index,chunk_text", ",0,'after-index'" + } + exec(fmt.Sprintf(`INSERT INTO embeddings_%s(file_id,embedding%s) + VALUES (md5('2001')::uuid,array_fill(0.2::real,ARRAY[%d])::vector%s)`, kind, extraCols, dim, extraValues)) + _, err := db.ExecContext(ctx, fmt.Sprintf(`UPDATE embeddings_%s + SET embedding=array_fill(0.1::real,ARRAY[%d])::vector WHERE file_id=md5('2001')::uuid`, kind, dim-1)) + if err == nil || !strings.Contains(err.Error(), "dimensions") { + t.Fatalf("%s: wrong dimensionality must fail, got %v", kind, err) + } + } + assertState(1, 2001) + if err := (&DB{url: dsn}).Migrate(ctx); err != nil { + t.Fatal(err) + } + t.Log("PASS: populated 24->25->24->25, all three cosine indexes, 2000 vectors/table preserved, post-index ingest, dimension rejection and production startup") +} diff --git a/server/internal/db/migrations/0001_init.sql b/server/internal/db/migrations/0001_init.sql index a5f9bf2..612ad78 100644 --- a/server/internal/db/migrations/0001_init.sql +++ b/server/internal/db/migrations/0001_init.sql @@ -106,8 +106,7 @@ CREATE TABLE IF NOT EXISTS embeddings_text ( embedding vector(768) ); CREATE INDEX IF NOT EXISTS idx_embeddings_text_file ON embeddings_text (file_id); --- HNSW index will be added by worker once we settle on a model dimension. Kept off here --- because pgvector requires the table to have data of consistent dim before building. +-- HNSW index on embedding column is in migration 0025. -- +goose StatementEnd -- +goose StatementBegin diff --git a/server/internal/db/migrations/0019_versioned_index_generations.sql b/server/internal/db/migrations/0019_versioned_index_generations.sql index bd173c8..3a4730d 100644 --- a/server/internal/db/migrations/0019_versioned_index_generations.sql +++ b/server/internal/db/migrations/0019_versioned_index_generations.sql @@ -264,8 +264,8 @@ CREATE INDEX idx_index_generation_targets_file_hash -- +goose StatementBegin -- `vector` intentionally has no table-wide dimension. Every row is validated -- against its immutable generation.output_dimension by the canonical service. --- Future ANN indexes must be route/dimension-specific expression or partition --- indexes; silently padding or truncating vectors is never allowed. +-- ANN indexes for fixed-dimension tables (embeddings_text/visual/face) are in +-- migration 0025; this table's variable-dimension vectors cannot share them. CREATE TABLE index_generation_vectors ( generation_id uuid NOT NULL REFERENCES index_generations(id) ON DELETE CASCADE, workspace_id uuid NOT NULL REFERENCES workspaces(id) ON DELETE CASCADE, diff --git a/server/internal/db/migrations/0025_ann_hnsw_indexes.sql b/server/internal/db/migrations/0025_ann_hnsw_indexes.sql new file mode 100644 index 0000000..63d0d77 --- /dev/null +++ b/server/internal/db/migrations/0025_ann_hnsw_indexes.sql @@ -0,0 +1,40 @@ +-- +goose Up +-- Transactional startup migration: index construction blocks writes on these +-- tables until commit. Schedule a maintenance window for populated deployments. +-- Keep pgvector defaults m=16, ef_construction=64; tune only after representative +-- recall/build/ingest measurements. No corpus-size or latency guarantee is made. +-- +goose StatementBegin +-- HNSW ANN index for text embeddings (768-d, cosine distance). +-- Available to direct cosine-order queries. Shipping text search deduplicates +-- by file before top-k and does NOT yet use this index; see VALIDATION_HNSW.md. +CREATE INDEX IF NOT EXISTS idx_embeddings_text_embedding_hnsw + ON embeddings_text USING hnsw (embedding vector_cosine_ops); +-- +goose StatementEnd + +-- +goose StatementBegin +-- HNSW ANN index for visual embeddings (512-d, cosine distance). +-- Compatible with the visual route distance ordering; planner use depends on +-- corpus and filters. This does not establish recall or production latency. +CREATE INDEX IF NOT EXISTS idx_embeddings_visual_embedding_hnsw + ON embeddings_visual USING hnsw (embedding vector_cosine_ops); +-- +goose StatementEnd + +-- +goose StatementBegin +-- HNSW ANN index for face embeddings (512-d, cosine distance). +-- Reserved for future face search queries; current clustering is O(n) in Go. +CREATE INDEX IF NOT EXISTS idx_embeddings_face_embedding_hnsw + ON embeddings_face USING hnsw (embedding vector_cosine_ops); +-- +goose StatementEnd + +-- +goose Down +-- +goose StatementBegin +DROP INDEX IF EXISTS idx_embeddings_face_embedding_hnsw; +-- +goose StatementEnd + +-- +goose StatementBegin +DROP INDEX IF EXISTS idx_embeddings_visual_embedding_hnsw; +-- +goose StatementEnd + +-- +goose StatementBegin +DROP INDEX IF EXISTS idx_embeddings_text_embedding_hnsw; +-- +goose StatementEnd diff --git a/server/internal/face/face.go b/server/internal/face/face.go index 146ba99..c9df6d1 100644 --- a/server/internal/face/face.go +++ b/server/internal/face/face.go @@ -11,8 +11,8 @@ // 4. Insert embeddings_face (file_id, entity_id, bbox, embedding). // // This is intentionally O(n) per insert — fine for a personal drive up to -// thousands of faces. For larger corpora swap in pgvector HNSW + offline -// re-clustering. +// thousands of faces. The HNSW index (migration 0025) is available for future +// SQL-based face search queries. package face import ( diff --git a/server/internal/search/hnsw_semantics_test.go b/server/internal/search/hnsw_semantics_test.go new file mode 100644 index 0000000..87a9d43 --- /dev/null +++ b/server/internal/search/hnsw_semantics_test.go @@ -0,0 +1,117 @@ +package search + +import ( + "context" + "os" + "strings" + "testing" + "time" + + "github.com/google/uuid" + "github.com/jackc/pgx/v5/pgxpool" + + memdb "github.com/PeterGuy326/mem/server/internal/db" +) + +// Keep this regression when introducing a planner-compatible text route: a +// chunk candidate budget must not become a file result budget or bypass scope. +func TestTextANNFileSemanticsPostgres(t *testing.T) { + dsn := os.Getenv("MEM_TEST_DB") + if dsn == "" { + t.Skip("MEM_TEST_DB not set; skipping text ANN PostgreSQL regression") + } + cfg, err := pgxpool.ParseConfig(dsn) + if err != nil { + t.Fatal(err) + } + if !strings.HasSuffix(cfg.ConnConfig.Database, "_test") { + t.Fatalf("refusing non-test database %q", cfg.ConnConfig.Database) + } + ctx, cancel := context.WithTimeout(context.Background(), 90*time.Second) + defer cancel() + database, err := memdb.Open(ctx, dsn) + if err != nil { + t.Fatal(err) + } + defer database.Close() + if err := database.Migrate(ctx); err != nil { + t.Fatal(err) + } + owner, other := uuid.New(), uuid.New() + for _, id := range []uuid.UUID{owner, other} { + if _, err := database.Pool.Exec(ctx, "INSERT INTO users(id,email,password_hash) VALUES($1,$2,'test')", id, id.String()+"@example.test"); err != nil { + t.Fatal(err) + } + } + defer func() { + cleanupCtx, cleanupCancel := context.WithTimeout(context.Background(), 10*time.Second) + defer cleanupCancel() + if _, err := database.Pool.Exec(cleanupCtx, "DELETE FROM users WHERE id=ANY($1::uuid[])", []uuid.UUID{owner, other}); err != nil { + t.Errorf("cleanup: %v", err) + } + }() + now := time.Now().UTC().Truncate(time.Second) + since, until := now.Add(-time.Hour), now.Add(time.Hour) + vec := make([]float32, textEmbeddingSchemaDim) + vec[0] = 1 + addFile := func(user uuid.UUID, path, mime string, at time.Time, chunks int, far bool) uuid.UUID { + t.Helper() + id := uuid.New() + _, err := database.Pool.Exec(ctx, `INSERT INTO files(id,user_id,name,path,size,sha256,mime,storage_key,created_at,timeline_at) + VALUES($1,$2,'fixture', $3,0,'fixture',$4,$1::text,$5,$5)`, id, user, path, mime, at) + if err != nil { + t.Fatal(err) + } + for chunk := 0; chunk < chunks; chunk++ { + v := make([]float32, textEmbeddingSchemaDim) + if far { + v[1] = 1 + } else { + v[0], v[1] = 1, float32(chunk)*0.001 + } + _, err := database.Pool.Exec(ctx, `INSERT INTO embeddings_text(file_id,chunk_index,chunk_text,embedding) + VALUES($1,$2,'source chunk',$3::vector)`, id, chunk, vectorLiteral(v)) + if err != nil { + t.Fatal(err) + } + } + return id + } + // 101 nearest chunks belong to one file. 40 candidates or even 8*k=80 + // candidates would yield only one file; the shipping route must return ten. + best := addFile(owner, "/Work_%/Docs", "text/plain", now, 101, false) + eligible := map[uuid.UUID]bool{best: true} + for i := 0; i < 19; i++ { + eligible[addFile(owner, "/Work_%/Docs", "application/pdf", now, 1, true)] = true + } + addFile(other, "/Work_%/Docs", "text/plain", now, 1, false) + addFile(owner, "/Work_AB/Docs", "text/plain", now, 1, false) + addFile(owner, "/Work_%/Private", "text/plain", now, 1, false) + addFile(owner, "/Work_%/Docs", "image/png", now, 1, false) + addFile(owner, "/Work_%/Docs", "text/plain", since.Add(-time.Second), 1, false) + addFile(owner, "/Work_%/Docs", "text/plain", until.Add(time.Second), 1, false) + service := New(database.Pool, nil) + q := Query{UserID: owner, Limit: 10, PathPrefix: "/Work_%", AllowedPaths: []string{"/Work_%/Docs"}, Type: "doc", Since: &since, Until: &until, SnippetChars: 200} + hits, err := service.runTextANN(ctx, q, vec) + if err != nil { + t.Fatal(err) + } + if len(hits) != 10 { + t.Fatalf("got %d files, want 10 despite 101 nearest chunks belonging to one file", len(hits)) + } + seen := map[uuid.UUID]bool{} + for _, hit := range hits { + if !eligible[hit.FileID] || seen[hit.FileID] { + t.Fatalf("out-of-scope or duplicate file: %+v", hit) + } + seen[hit.FileID] = true + } + if hits[0].FileID != best || hits[0].ChunkIndex != 0 || hits[0].Score != 1 { + t.Fatalf("best chunk was not preserved: %+v", hits[0]) + } + q.AllowedPaths = []string{""} + hits, err = service.runTextANN(ctx, q, vec) + if err != nil || len(hits) != 0 { + t.Fatalf("invalid allow-list must fail closed: hits=%v err=%v", hits, err) + } +} From 95af9fe8b8436006b30faf2366e890a374969ce7 Mon Sep 17 00:00:00 2001 From: sun-970 <3843544764@qq.com> Date: Fri, 18 Sep 2026 00:07:50 +0800 Subject: [PATCH 2/2] fix(test): resolve parameter type inconsistency in HNSW semantics test (#218) Refs #197 Fixes the failing PostgreSQL integration test in #197. ## Problem `TestTextANNFileSemanticsPostgres` fails with: ``` ERROR: inconsistent types deduced for parameter $1 (SQLSTATE 42P08) ``` The test SQL uses `$1::text` to cast a UUID parameter to text for the `storage_key` column, but `$1` is already used as a UUID in the first position. PostgreSQL cannot deduce a consistent type for the parameter. ## Solution Use separate parameters for the UUID and its text representation instead of casting: ```sql -- Before: VALUES($1,$2,'fixture', $3,0,'fixture',$4,$1::text,$5,$5) -- After: VALUES($1,$2,'fixture',$3,0,'fixture',$4,$5,$6,$6) ``` Where `$5` is `id.String()` (the text representation) and `$6` is the timestamp. ## Validation - [x] Fix resolves the type deduction error - [ ] CI passes on this PR - [ ] Can be cherry-picked into #197 This is a minimal fix for the test infrastructure, not a change to the HNSW migration or search logic. Co-authored-by: liyuanyang --- server/internal/search/hnsw_semantics_test.go | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/server/internal/search/hnsw_semantics_test.go b/server/internal/search/hnsw_semantics_test.go index 87a9d43..875c349 100644 --- a/server/internal/search/hnsw_semantics_test.go +++ b/server/internal/search/hnsw_semantics_test.go @@ -58,7 +58,7 @@ func TestTextANNFileSemanticsPostgres(t *testing.T) { t.Helper() id := uuid.New() _, err := database.Pool.Exec(ctx, `INSERT INTO files(id,user_id,name,path,size,sha256,mime,storage_key,created_at,timeline_at) - VALUES($1,$2,'fixture', $3,0,'fixture',$4,$1::text,$5,$5)`, id, user, path, mime, at) + VALUES($1,$2,'fixture',$3,0,'fixture',$4,$5,$6,$6)`, id, user, path, mime, id.String(), at) if err != nil { t.Fatal(err) }