diff --git a/.ncurc.json b/.ncurc.json new file mode 100644 index 0000000..3750254 --- /dev/null +++ b/.ncurc.json @@ -0,0 +1,3 @@ +{ + "reject": ["flexsearch"] +} diff --git a/flexsearch.config.ts b/flexsearch.config.ts index e0d44be..4ea1c86 100644 --- a/flexsearch.config.ts +++ b/flexsearch.config.ts @@ -1,4 +1,10 @@ import type { FlexSearchConfig } from 'docusaurus-plugin-mcp-server'; +import { stemmer } from '@orama/stemmers/russian'; +import { stopwords } from '@orama/stopwords/russian'; + +const STOPWORDS = new Set(stopwords); + +export const WORD_SEPARATOR = /[\s\-_.,;:!?'"()[\]{}«»—–]+/; // FlexSearch tuned for Russian content. The plugin's defaults are tuned for // English (forward tokenize + bidirectional context + English stemmer) and @@ -7,12 +13,17 @@ import type { FlexSearchConfig } from 'docusaurus-plugin-mcp-server'; // - tokenize: 'strict' indexes whole words only (no prefix explosion). // - context: false drops the bidirectional context that bloats the index. // - resolution: 3 is enough for relevance ranking on ~100 docs. -// - encode: lowercase split on Russian punctuation, no English stemmer. +// - encode: lowercase split on Russian punctuation, drop Russian stopwords, +// then the Snowball Russian stemmer: «подписка», «подписки» and «подписку» +// become one stem. A stem is never longer than its word, so the index +// doesn't grow. Stopwords matter because a query matches on any of its +// words (worker/search-provider.ts): «как», «где», «мне» would match +// every article and drown the ranking. // // This config MUST be identical at build time (docusaurus.config.ts) and at // runtime (worker/index.ts) — otherwise the runtime provider deserializes the // index with the wrong shape and returns no results. -const flexsearchConfig: FlexSearchConfig = { +const flexsearchConfig = { tokenize: 'strict', resolution: 3, context: false, @@ -20,8 +31,9 @@ const flexsearchConfig: FlexSearchConfig = { encode: (str: string) => String(str) .toLowerCase() - .split(/[\s\-_.,;:!?'"()[\]{}«»—–]+/) - .filter(Boolean), -}; + .split(WORD_SEPARATOR) + .filter((word) => word && !STOPWORDS.has(word)) + .map(stemmer), +} satisfies FlexSearchConfig; export default flexsearchConfig; diff --git a/package.json b/package.json index ed3ad0e..8f28e65 100644 --- a/package.json +++ b/package.json @@ -23,9 +23,11 @@ "@docusaurus/preset-classic": "3.10.1", "@mdx-js/react": "^3.1.1", "@orama/plugin-docusaurus-v3": "^3.1.18", + "@orama/stemmers": "^3.1.18", + "@orama/stopwords": "^3.1.18", "clsx": "^2.1.1", "docusaurus-plugin-mcp-server": "^1.0.0", - "flexsearch": "0.8.212", + "flexsearch": "0.7.43", "hono": "^4.12.27", "prism-react-renderer": "^2.4.1", "react": "^19.2.7", diff --git a/pnpm-lock.yaml b/pnpm-lock.yaml index 0571413..20a4cc4 100644 --- a/pnpm-lock.yaml +++ b/pnpm-lock.yaml @@ -32,6 +32,12 @@ importers: '@orama/plugin-docusaurus-v3': specifier: ^3.1.18 version: 3.1.18(1186ddf6ac51dfc8a88c39eec90528fc) + '@orama/stemmers': + specifier: ^3.1.18 + version: 3.1.18 + '@orama/stopwords': + specifier: ^3.1.18 + version: 3.1.18 clsx: specifier: ^2.1.1 version: 2.1.1 @@ -39,8 +45,8 @@ importers: specifier: ^1.0.0 version: 1.0.0(@docusaurus/core@3.10.1(@docusaurus/faster@3.10.1(@docusaurus/types@3.10.1(@swc/core@1.15.32)(postcss@8.5.6)(react-dom@19.2.7(react@19.2.7))(react@19.2.7))(postcss@8.5.6))(@mdx-js/react@3.1.1(@types/react@19.2.17)(react@19.2.7))(@rspack/core@1.7.11)(@swc/core@1.15.32)(postcss@8.5.6)(react-dom@19.2.7(react@19.2.7))(react@19.2.7)(typescript@6.0.3))(@docusaurus/types@3.10.1(@swc/core@1.15.32)(postcss@8.5.6)(react-dom@19.2.7(react@19.2.7))(react@19.2.7))(react-dom@19.2.7(react@19.2.7))(react@19.2.7)(zod@3.25.76) flexsearch: - specifier: 0.8.212 - version: 0.8.212 + specifier: 0.7.43 + version: 0.7.43 hono: specifier: ^4.12.27 version: 4.12.27 @@ -2110,6 +2116,14 @@ packages: react: '>=17.0.0 <20.0.0' react-dom: '>=17.0.0 <20.0.0' + '@orama/stemmers@3.1.18': + resolution: {integrity: sha512-xXs5fpiUvBs4jJZtnayJmUSomdxxiwwwvTbAzPHSyI6geLM7bxXHgpVlYz4JeMBycHX8bHF7j9m7F35BxhuZ8g==} + engines: {node: '>= 20.0.0'} + + '@orama/stopwords@3.1.18': + resolution: {integrity: sha512-W8V7m7RnCme+99OmKl/xs5rf6OUhFpr0aPGVmPrXzTLSg4ZqSbRY2euS2S/lgjjYi/0NhEWqwoq8nDY6Ihx4EA==} + engines: {node: '>= 20.0.0'} + '@orama/switch@3.1.18': resolution: {integrity: sha512-KjAHr/7qiteLWLE284EEA2ZPX9dI3kWi01Ws1paB/oYP1hGijDizqxLuXEC1TwTYvVE07eCMeqho1Ql/rFCasQ==} peerDependencies: @@ -4009,9 +4023,6 @@ packages: flexsearch@0.7.43: resolution: {integrity: sha512-c5o/+Um8aqCSOXGcZoqZOm+NqtVwNsvVpWv6lfmSclU954O3wvQKxxK8zj74fPaSJbXpSLTs4PRhh+wnoCXnKg==} - flexsearch@0.8.212: - resolution: {integrity: sha512-wSyJr1GUWoOOIISRu+X2IXiOcVfg9qqBRyCPRUdLMIGJqPzMo+jMRlvE83t14v1j0dRMEaBbER/adQjp6Du2pw==} - follow-redirects@1.15.11: resolution: {integrity: sha512-deG2P0JfjrTxl50XGCDyfI97ZGVCxIpfKYmfyrQ54n5FO/0gfIES8C/Psl6kWVDolizcaaxZJnTS0QSMxvnsBQ==} engines: {node: '>=4.0'} @@ -10552,6 +10563,10 @@ snapshots: - '@types/react' - babel-plugin-macros + '@orama/stemmers@3.1.18': {} + + '@orama/stopwords@3.1.18': {} + '@orama/switch@3.1.18(@orama/core@0.1.11)(@orama/orama@3.1.18)(@oramacloud/client@2.1.4)': dependencies: '@orama/core': 0.1.11 @@ -12676,8 +12691,6 @@ snapshots: flexsearch@0.7.43: {} - flexsearch@0.8.212: {} - follow-redirects@1.15.11: {} form-data-encoder@2.1.4: {} diff --git a/worker/index.ts b/worker/index.ts index 8cbb222..9874fe5 100644 --- a/worker/index.ts +++ b/worker/index.ts @@ -1,9 +1,9 @@ import { Hono } from 'hono'; import { cors } from 'hono/cors'; -import { McpDocsServer } from 'docusaurus-plugin-mcp-server'; +import { McpDocsServer, type ProcessedDoc } from 'docusaurus-plugin-mcp-server'; import docs from '../build/mcp/docs.json'; import searchIndex from '../build/mcp/search-index.json'; -import flexsearchConfig from '../flexsearch.config.ts'; +import { HelpSearchProvider } from './search-provider.ts'; const NAME = 'hexlet-help'; const VERSION = '1.0.0'; @@ -12,17 +12,15 @@ const BASE_URL = 'https://help.hexlet.io'; let server: McpDocsServer | null = null; const getServer = (): McpDocsServer => (server ??= new McpDocsServer({ - docs: docs as Record, + docs: docs as Record, searchIndexData: searchIndex, name: NAME, version: VERSION, baseUrl: BASE_URL, - // The built-in 'flexsearch' provider is bundled statically (no dynamic - // import), so it works in the Worker. This config must match the one used - // at build time in docusaurus.config.ts, or the index deserializes wrong. - flexsearch: flexsearchConfig, - // eslint-disable-next-line @typescript-eslint/no-explicit-any - } as any)); + // Reads the index built with flexsearch.config.ts, but ranks articles + // that match only some of the query words instead of dropping them. + search: new HelpSearchProvider(), + })); const app = new Hono(); diff --git a/worker/search-provider.ts b/worker/search-provider.ts new file mode 100644 index 0000000..b319d02 --- /dev/null +++ b/worker/search-provider.ts @@ -0,0 +1,162 @@ +import FlexSearch from 'flexsearch'; +import type { + ProcessedDoc, + SearchOptions, + SearchProvider, + SearchProviderInitData, + SearchResult, +} from 'docusaurus-plugin-mcp-server'; +import flexsearchConfig, { WORD_SEPARATOR } from '../flexsearch.config.ts'; + +// In the order the plugin indexes the fields. +const FIELD_WEIGHTS: Record = { + title: 3, + content: 1, + headings: 2, + description: 1.5, +}; + +const SNIPPET_LENGTH = 200; + +const encode = flexsearchConfig.encode; + +// The built-in FlexSearch provider hands the whole query to FlexSearch, which +// intersects its words: a question asked in full finds nothing unless one +// article contains every word of it. This provider reads the same index but +// ranks articles by the words they do contain. The document shape below +// mirrors the plugin's createSearchIndex, which built the index at build time, +// and the flexsearch version in package.json must be the one the plugin +// depends on (0.7): 0.8 reads a 0.7 export without an error and finds nothing, +// so initialize() checks that the index finds an article by its own title. +export class HelpSearchProvider implements SearchProvider { + readonly name = 'help-flexsearch'; + + private docs: Record | null = null; + + private index = new FlexSearch.Document({ + ...flexsearchConfig, + document: { + id: 'id', + index: Object.keys(FIELD_WEIGHTS), + store: ['title', 'description'], + }, + }); + + async initialize(_context: unknown, initData?: SearchProviderInitData): Promise { + if (!initData?.docs || !initData.indexData) { + throw new Error('[HelpSearch] docs and indexData are required'); + } + for (const [key, value] of Object.entries(initData.indexData)) { + await this.index.import(key, value as string); + } + const [first] = Object.values(initData.docs); + if (first && this.index.search(first.title).length === 0) { + throw new Error('[HelpSearch] The index finds nothing: does flexsearch match the plugin?'); + } + this.docs = initData.docs; + } + + isReady(): boolean { + return this.docs !== null; + } + + async search(query: string, options?: SearchOptions): Promise { + const docs = this.getDocs(); + const limit = options?.limit ?? 16; + const total = this.getDocCount(); + const words = wordsByStem(query); + const terms = new Set(words.keys()); + + // Every word is looked up on its own, so an article matching only part of + // the question still counts. A word found in few articles says more about + // the question than one found everywhere (IDF), and a word in the title + // says more than one in the body (field weight). + const scores = new Map(); + for (const word of words.values()) { + const weights = new Map(); + for (const { field, result } of this.index.search(word, { limit: total })) { + const weight = FIELD_WEIGHTS[field]; + for (const id of result) { + const docId = String(id); + weights.set(docId, Math.max(weights.get(docId) ?? 0, weight)); + } + } + const idf = Math.log(1 + total / weights.size); + for (const [docId, weight] of weights) { + scores.set(docId, (scores.get(docId) ?? 0) + idf * weight); + } + } + + return [...scores] + .sort(([, a], [, b]) => b - a) + .slice(0, limit) + .flatMap(([url, score]) => { + const doc = docs[url]; + if (!doc) return []; + return { + url, + route: doc.route, + title: doc.title, + score, + snippet: snippet(doc.markdown, terms), + matchingHeadings: doc.headings + .map((heading) => heading.text) + .filter((text) => encode(text).some((stem) => terms.has(stem))) + .slice(0, 3), + }; + }); + } + + async getDocument(url: string): Promise { + return this.getDocs()[url] ?? null; + } + + getDocCount(): number { + return Object.keys(this.docs ?? {}).length; + } + + async healthCheck(): Promise<{ healthy: boolean; message?: string }> { + if (!this.isReady()) { + return { healthy: false, message: 'Help search provider not initialized' }; + } + return { healthy: true, message: `Help search ready with ${this.getDocCount()} documents` }; + } + + private getDocs(): Record { + if (!this.docs) { + throw new Error('[HelpSearch] Provider not initialized'); + } + return this.docs; + } +} + +// One word per stem: the index encodes the query itself, so it gets the word +// rather than the stem — a stem encoded again may lose another suffix. +function wordsByStem(query: string): Map { + const words = new Map(); + for (const word of query.split(WORD_SEPARATOR)) { + for (const stem of encode(word)) { + if (!words.has(stem)) words.set(stem, word); + } + } + return words; +} + +// A stem is the lowercased start of its word (with «ё» read as «е»), so the +// earliest stem found in the text marks where the article answers the query. +function snippet(markdown: string, terms: Set): string { + const text = markdown.toLowerCase().replaceAll('ё', 'е'); + const positions = [...terms].map((term) => text.indexOf(term)).filter((index) => index !== -1); + const start = positions.length > 0 ? Math.max(0, Math.min(...positions) - 50) : 0; + const end = Math.min(markdown.length, start + SNIPPET_LENGTH); + const body = markdown + .slice(start, end) + .replace(/^#{1,6}\s+/gm, '') + .replace(/!\[([^\]]*)\]\([^)]+\)/g, '') + .replace(/\[([^\]]+)\]\([^)]+\)/g, '$1') + .replace(/```[a-z]*\n?/g, '') + .replace(/`([^`]+)`/g, '$1') + .replace(/\s+/g, ' ') + .trim(); + return `${start > 0 ? '...' : ''}${body}${end < markdown.length ? '...' : ''}`; +}