Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
6 changes: 6 additions & 0 deletions CHANGELOG.md
Original file line number Diff line number Diff line change
Expand Up @@ -46,6 +46,12 @@ All notable changes to `doc` are documented here.

### Added

- PostgreSQL full-text search with tsvector/GIN index and `matchField` provenance (#71): both
`GET /api/doc?keyword=` and `GET /api/v1/documents?query=` now report `matchField`
(`title` | `content` | `both`) so host UIs can highlight where a hit was found. A plain-text
extraction of TipTap `content` is persisted in `contentSearch` and indexed via a trigger-maintained
`search_vector` tsvector column with `websearch_to_tsquery()`. The existing `contains` fallback
remains active for SQLite and unmigrated environments.
- Extend document search to content body and add time-range and sort filters (#68): both
`GET /api/doc?keyword=` and `GET /api/v1/documents?query=` now match title OR content
(case-insensitive). New `after` / `before` (ISO 8601) filter by `updatedAt`; `sort` accepts
Expand Down
3 changes: 2 additions & 1 deletion package.json
Original file line number Diff line number Diff line change
Expand Up @@ -26,7 +26,8 @@
"format:fix": "prettier --write --list-different .",
"prepare": "husky",
"db:preflight": "prisma db execute --file prisma/preflight/ensure-share-relation-unique.sql --schema prisma/schema.prisma",
"db:push": "npm run db:preflight && prisma db push",
"db:search-index": "prisma db execute --file prisma/preflight/ensure-search-index.sql --schema prisma/schema.prisma",
"db:push": "npm run db:preflight && prisma db push && npm run db:search-index",
"test": "vitest",
"test:e2e": "playwright test",
"test:e2e:full": "DOC_E2E_FULL_LOOP=1 playwright test e2e/workspace.spec.ts",
Expand Down
54 changes: 54 additions & 0 deletions prisma/preflight/ensure-search-index.sql
Original file line number Diff line number Diff line change
@@ -0,0 +1,54 @@
DO $$
BEGIN
IF to_regclass('"Doc"') IS NULL THEN
RETURN;
END IF;

IF NOT EXISTS (
SELECT 1 FROM information_schema.columns
WHERE table_name = 'Doc' AND column_name = 'contentSearch'
) THEN
RETURN;
END IF;

IF NOT EXISTS (
SELECT 1 FROM information_schema.columns
WHERE table_name = 'Doc' AND column_name = 'search_vector'
) THEN
ALTER TABLE "Doc" ADD COLUMN "search_vector" tsvector;
END IF;

CREATE OR REPLACE FUNCTION "doc_search_vector_update"()
RETURNS TRIGGER AS $trigger$
BEGIN
NEW."search_vector" :=
setweight(to_tsvector('english', coalesce(NEW.title, '')), 'A') ||
setweight(to_tsvector('english', coalesce(NEW."contentSearch", '')), 'B');
RETURN NEW;
END;
$trigger$ LANGUAGE plpgsql;

IF NOT EXISTS (
SELECT 1 FROM pg_trigger
WHERE tgname = 'doc_search_vector_trigger'
) THEN
CREATE TRIGGER doc_search_vector_trigger
BEFORE INSERT OR UPDATE ON "Doc"
FOR EACH ROW
EXECUTE FUNCTION "doc_search_vector_update"();
END IF;

UPDATE "Doc"
SET "search_vector" =
setweight(to_tsvector('english', coalesce(title, '')), 'A') ||
setweight(to_tsvector('english', coalesce("contentSearch", '')), 'B')
WHERE "search_vector" IS NULL;

IF NOT EXISTS (
SELECT 1 FROM pg_indexes
WHERE indexname = 'Doc_search_vector_idx'
) THEN
CREATE INDEX "Doc_search_vector_idx" ON "Doc" USING GIN ("search_vector");
END IF;
END
$$;
5 changes: 5 additions & 0 deletions prisma/schema.prisma
Original file line number Diff line number Diff line change
Expand Up @@ -22,6 +22,11 @@ model Doc {
title String
content String
contentBinary Bytes?
contentSearch String? // plain-text extraction of TipTap JSON for search
/// Trigger-maintained tsvector. Prisma has no first-class tsvector type; the
/// column and GIN index are created by prisma/preflight/ensure-search-index.sql
/// and are not read or written through the Prisma client.
search_vector Unsupported("tsvector")?
isDeleted Boolean @default(false)
isStar Boolean @default(false)
sortOrder Int
Expand Down
11 changes: 7 additions & 4 deletions services/collaboration/src/db/doc.ts
Original file line number Diff line number Diff line change
Expand Up @@ -3,6 +3,7 @@ import type { QueryResultRow } from 'pg'
import { pgClient, reconnect } from './client.js'
import { errorMessage } from '../lib/error.js'
import { sendEmail } from '../lib/mailer.js'
import { extractPlainTextFromJson } from '../lib/tiptap-text-extractor.js'

export interface StoredDocumentRow extends QueryResultRow {
content: string | null
Expand All @@ -21,8 +22,9 @@ export interface MonitorDocumentRow extends QueryResultRow {
*/
export async function updateDocJsonStr(id: string, jsonStr: string): Promise<number> {
try {
const sql = `update "Doc" set content = $1, "updatedAt" = $2 where id = $3`
const values = [jsonStr, new Date(), id]
const contentSearch = extractPlainTextFromJson(jsonStr)
const sql = `update "Doc" set content = $1, "contentSearch" = $2, "updatedAt" = $3 where id = $4`
const values = [jsonStr, contentSearch || null, new Date(), id]
const result = await pgClient.query(sql, values)
return result.rowCount ?? 0
} catch (error) {
Expand Down Expand Up @@ -69,8 +71,9 @@ export async function updateDocBinary(id: string, binary: Uint8Array): Promise<n
// 同时更新正文二进制和 JSON 镜像,保证恢复后的状态一致。
export async function updateDocBinaryAndJson(id: string, binary: Uint8Array, jsonStr: string): Promise<number> {
try {
const sql = `update "Doc" set "contentBinary" = $1, content = $2, "updatedAt" = $3 where id = $4`
const values = [binary, jsonStr, new Date(), id]
const contentSearch = extractPlainTextFromJson(jsonStr)
const sql = `update "Doc" set "contentBinary" = $1, content = $2, "contentSearch" = $3, "updatedAt" = $4 where id = $5`
const values = [binary, jsonStr, contentSearch || null, new Date(), id]
const result = await pgClient.query(sql, values)
return result.rowCount ?? 0
} catch (error) {
Expand Down
34 changes: 34 additions & 0 deletions services/collaboration/src/lib/tiptap-text-extractor.ts
Original file line number Diff line number Diff line change
@@ -0,0 +1,34 @@
interface TipTapNode {
type: string
text?: string
content?: TipTapNode[]
}

export function extractPlainText(value: unknown): string {
if (value == null || typeof value !== 'object' || Array.isArray(value)) return ''
const parts: string[] = []
collectText(value as TipTapNode, parts)
return parts.join(' ').replace(/\s+/g, ' ').trim()
}

export function extractPlainTextFromJson(jsonStr: string): string {
if (!jsonStr?.trim()) return ''
try {
const parsed = JSON.parse(jsonStr)
return extractPlainText(parsed)
} catch {
return ''
}
}

function collectText(node: TipTapNode, parts: string[]) {
if (node.type === 'text' && typeof node.text === 'string') {
parts.push(node.text)
return
}
if (node.content && Array.isArray(node.content)) {
for (const child of node.content) {
collectText(child, parts)
}
}
}
1 change: 1 addition & 0 deletions src/__tests__/api/doc-create-permissions.test.ts
Original file line number Diff line number Diff line change
Expand Up @@ -131,6 +131,7 @@ describe('POST /api/doc permissions', () => {
title: 'Source copy',
content: '{"type":"doc"}',
contentBinary: Buffer.from('binary'),
contentSearch: null,
parentId: null,
sortOrder: 1024,
userId: 'owner',
Expand Down
34 changes: 34 additions & 0 deletions src/__tests__/lib/api-v1-documents.test.ts
Original file line number Diff line number Diff line change
Expand Up @@ -9,6 +9,7 @@ const mocks = vi.hoisted(() => ({
findMany: vi.fn(),
updateMany: vi.fn(),
nextSortOrder: vi.fn(),
fullTextSearch: vi.fn(),
}))

vi.mock('server-only', () => ({}))
Expand All @@ -26,6 +27,10 @@ vi.mock('@/db/db', () => ({
vi.mock('@/lib/doc-sort-order', () => ({
getNextSortOrderForParent: mocks.nextSortOrder,
}))
vi.mock('@/lib/doc-search', async (importOriginal) => {
const actual = await importOriginal<typeof import('@/lib/doc-search')>()
return { ...actual, fullTextSearch: mocks.fullTextSearch }
})

import { ApiV1Error } from '@/lib/api-v1'
import {
Expand Down Expand Up @@ -159,6 +164,10 @@ describe('TipTap API codec', () => {
})

describe('v1 document service', () => {
beforeEach(() => {
mocks.fullTextSearch.mockResolvedValue(null)
})

test('lists only the principal documents with stable pagination', async () => {
mocks.findMany.mockResolvedValue([
metadata,
Expand Down Expand Up @@ -212,6 +221,31 @@ describe('v1 document service', () => {
)
})

test('falls back to contains when full-text search returns no hits', async () => {
mocks.fullTextSearch.mockResolvedValue([])
mocks.findMany.mockResolvedValue([])

await listApiDocuments('user-1', new URLSearchParams({ query: '中文关键词' }))

const where = mocks.findMany.mock.calls[0][0].where
expect(where.id).toBeUndefined()
expect(where.OR).toEqual([
{ title: { contains: '中文关键词', mode: 'insensitive' } },
{ content: { contains: '中文关键词', mode: 'insensitive' } },
])
})

test('restricts ids when full-text search returns hits', async () => {
mocks.fullTextSearch.mockResolvedValue([{ id: 'doc-1', matchField: 'title' }])
mocks.findMany.mockResolvedValue([])

await listApiDocuments('user-1', new URLSearchParams({ query: 'Example' }))

const where = mocks.findMany.mock.calls[0][0].where
expect(where.id).toEqual({ in: ['doc-1'] })
expect(where.OR).toBeUndefined()
})

test('filters by time range with after and before params', async () => {
mocks.findMany.mockResolvedValue([])

Expand Down
108 changes: 108 additions & 0 deletions src/__tests__/lib/doc-search.test.ts
Original file line number Diff line number Diff line change
@@ -0,0 +1,108 @@
// @vitest-environment node

import { beforeEach, describe, expect, test, vi } from 'vitest'

const mocks = vi.hoisted(() => ({
queryRaw: vi.fn(),
queryRawUnsafe: vi.fn(),
}))

vi.mock('server-only', () => ({}))
vi.mock('@/db/db', () => ({
db: {
$queryRaw: mocks.queryRaw,
$queryRawUnsafe: mocks.queryRawUnsafe,
},
}))

import { computeMatchField, fullTextSearch, hasSearchVector } from '@/lib/doc-search'

beforeEach(() => {
vi.clearAllMocks()
vi.resetModules()
})

describe('computeMatchField', () => {
test('returns "title" when query matches only title', () => {
expect(computeMatchField('My Document', 'some content', 'document')).toBe('title')
})

test('returns "content" when query matches only content', () => {
expect(computeMatchField('My Document', 'some content about testing', 'testing')).toBe('content')
})

test('returns "both" when query matches title and content', () => {
expect(computeMatchField('My Document', 'document content', 'document')).toBe('both')
})

test('returns "content" when contentSearch is null', () => {
expect(computeMatchField('My Document', null, 'content')).toBe('content')
})

test('is case-insensitive', () => {
expect(computeMatchField('MY DOCUMENT', 'SOME CONTENT', 'document')).toBe('title')
expect(computeMatchField('my document', 'some content', 'document')).toBe('title')
})
})

describe('hasSearchVector', () => {
test('returns true when search_vector column exists', async () => {
const { hasSearchVector: freshHasSearchVector } = await import('@/lib/doc-search')
mocks.queryRaw.mockResolvedValueOnce([{ exists: true }])
expect(await freshHasSearchVector()).toBe(true)
})

test('returns false when search_vector column does not exist', async () => {
const { hasSearchVector: freshHasSearchVector } = await import('@/lib/doc-search')
mocks.queryRaw.mockResolvedValueOnce([{ exists: false }])
expect(await freshHasSearchVector()).toBe(false)
})

test('returns false on database error', async () => {
const { hasSearchVector: freshHasSearchVector } = await import('@/lib/doc-search')
mocks.queryRaw.mockRejectedValueOnce(new Error('connection failed'))
expect(await freshHasSearchVector()).toBe(false)
})
})

describe('fullTextSearch', () => {
test('returns null when search_vector is not available', async () => {
const { fullTextSearch: freshFullTextSearch } = await import('@/lib/doc-search')
mocks.queryRaw.mockResolvedValueOnce([{ exists: false }])
const result = await freshFullTextSearch('user-1', 'test', { isDeleted: false })
expect(result).toBeNull()
})

test('returns search hits with matchField when search_vector is available', async () => {
const { fullTextSearch: freshFullTextSearch } = await import('@/lib/doc-search')
mocks.queryRaw.mockResolvedValueOnce([{ exists: true }])
mocks.queryRawUnsafe.mockResolvedValueOnce([
{ id: 'doc-1', title: 'Test Document', contentSearch: 'some content' },
{ id: 'doc-2', title: 'Another Doc', contentSearch: 'test content here' },
])

const result = await freshFullTextSearch('user-1', 'test', { isDeleted: false })
expect(result).toEqual([
{ id: 'doc-1', matchField: 'title' },
{ id: 'doc-2', matchField: 'content' },
])
})

test('returns null on query error', async () => {
const { fullTextSearch: freshFullTextSearch } = await import('@/lib/doc-search')
mocks.queryRaw.mockResolvedValueOnce([{ exists: true }])
mocks.queryRawUnsafe.mockRejectedValueOnce(new Error('query failed'))

const result = await freshFullTextSearch('user-1', 'test', { isDeleted: false })
expect(result).toBeNull()
})

test('returns an empty array when tsquery matches nothing (caller must fall back)', async () => {
const { fullTextSearch: freshFullTextSearch } = await import('@/lib/doc-search')
mocks.queryRaw.mockResolvedValueOnce([{ exists: true }])
mocks.queryRawUnsafe.mockResolvedValueOnce([])

const result = await freshFullTextSearch('user-1', '中文关键词', { isDeleted: false })
expect(result).toEqual([])
})
})
Loading
Loading