Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
3 changes: 3 additions & 0 deletions .ncurc.json
Original file line number Diff line number Diff line change
@@ -0,0 +1,3 @@
{
"reject": ["flexsearch"]
}
22 changes: 17 additions & 5 deletions flexsearch.config.ts
Original file line number Diff line number Diff line change
@@ -1,4 +1,10 @@
import type { FlexSearchConfig } from 'docusaurus-plugin-mcp-server';
import { stemmer } from '@orama/stemmers/russian';
import { stopwords } from '@orama/stopwords/russian';

const STOPWORDS = new Set(stopwords);

export const WORD_SEPARATOR = /[\s\-_.,;:!?'"()[\]{}«»—–]+/;

// FlexSearch tuned for Russian content. The plugin's defaults are tuned for
// English (forward tokenize + bidirectional context + English stemmer) and
Expand All @@ -7,21 +13,27 @@ import type { FlexSearchConfig } from 'docusaurus-plugin-mcp-server';
// - tokenize: 'strict' indexes whole words only (no prefix explosion).
// - context: false drops the bidirectional context that bloats the index.
// - resolution: 3 is enough for relevance ranking on ~100 docs.
// - encode: lowercase split on Russian punctuation, no English stemmer.
// - encode: lowercase split on Russian punctuation, drop Russian stopwords,
// then the Snowball Russian stemmer: «подписка», «подписки» and «подписку»
// become one stem. A stem is never longer than its word, so the index
// doesn't grow. Stopwords matter because a query matches on any of its
// words (worker/search-provider.ts): «как», «где», «мне» would match
// every article and drown the ranking.
//
// This config MUST be identical at build time (docusaurus.config.ts) and at
// runtime (worker/index.ts) — otherwise the runtime provider deserializes the
// index with the wrong shape and returns no results.
const flexsearchConfig: FlexSearchConfig = {
const flexsearchConfig = {
tokenize: 'strict',
resolution: 3,
context: false,
cache: 100,
encode: (str: string) =>
String(str)
.toLowerCase()
.split(/[\s\-_.,;:!?'"()[\]{}«»—–]+/)
.filter(Boolean),
};
.split(WORD_SEPARATOR)
.filter((word) => word && !STOPWORDS.has(word))
.map(stemmer),
} satisfies FlexSearchConfig;

export default flexsearchConfig;
4 changes: 3 additions & 1 deletion package.json
Original file line number Diff line number Diff line change
Expand Up @@ -23,9 +23,11 @@
"@docusaurus/preset-classic": "3.10.1",
"@mdx-js/react": "^3.1.1",
"@orama/plugin-docusaurus-v3": "^3.1.18",
"@orama/stemmers": "^3.1.18",
"@orama/stopwords": "^3.1.18",
"clsx": "^2.1.1",
"docusaurus-plugin-mcp-server": "^1.0.0",
"flexsearch": "0.8.212",
"flexsearch": "0.7.43",
"hono": "^4.12.27",
"prism-react-renderer": "^2.4.1",
"react": "^19.2.7",
Expand Down
27 changes: 20 additions & 7 deletions pnpm-lock.yaml

Some generated files are not rendered by default. Learn more about how customized files appear on GitHub.

16 changes: 7 additions & 9 deletions worker/index.ts
Original file line number Diff line number Diff line change
@@ -1,9 +1,9 @@
import { Hono } from 'hono';
import { cors } from 'hono/cors';
import { McpDocsServer } from 'docusaurus-plugin-mcp-server';
import { McpDocsServer, type ProcessedDoc } from 'docusaurus-plugin-mcp-server';
import docs from '../build/mcp/docs.json';
import searchIndex from '../build/mcp/search-index.json';
import flexsearchConfig from '../flexsearch.config.ts';
import { HelpSearchProvider } from './search-provider.ts';

const NAME = 'hexlet-help';
const VERSION = '1.0.0';
Expand All @@ -12,17 +12,15 @@ const BASE_URL = 'https://help.hexlet.io';
let server: McpDocsServer | null = null;
const getServer = (): McpDocsServer =>
(server ??= new McpDocsServer({
docs: docs as Record<string, unknown>,
docs: docs as Record<string, ProcessedDoc>,
searchIndexData: searchIndex,
name: NAME,
version: VERSION,
baseUrl: BASE_URL,
// The built-in 'flexsearch' provider is bundled statically (no dynamic
// import), so it works in the Worker. This config must match the one used
// at build time in docusaurus.config.ts, or the index deserializes wrong.
flexsearch: flexsearchConfig,
// eslint-disable-next-line @typescript-eslint/no-explicit-any
} as any));
// Reads the index built with flexsearch.config.ts, but ranks articles
// that match only some of the query words instead of dropping them.
search: new HelpSearchProvider(),
}));

const app = new Hono();

Expand Down
162 changes: 162 additions & 0 deletions worker/search-provider.ts
Original file line number Diff line number Diff line change
@@ -0,0 +1,162 @@
import FlexSearch from 'flexsearch';
import type {
ProcessedDoc,
SearchOptions,
SearchProvider,
SearchProviderInitData,
SearchResult,
} from 'docusaurus-plugin-mcp-server';
import flexsearchConfig, { WORD_SEPARATOR } from '../flexsearch.config.ts';

// In the order the plugin indexes the fields.
const FIELD_WEIGHTS: Record<string, number> = {
title: 3,
content: 1,
headings: 2,
description: 1.5,
};

const SNIPPET_LENGTH = 200;

const encode = flexsearchConfig.encode;

// The built-in FlexSearch provider hands the whole query to FlexSearch, which
// intersects its words: a question asked in full finds nothing unless one
// article contains every word of it. This provider reads the same index but
// ranks articles by the words they do contain. The document shape below
// mirrors the plugin's createSearchIndex, which built the index at build time,
// and the flexsearch version in package.json must be the one the plugin
// depends on (0.7): 0.8 reads a 0.7 export without an error and finds nothing,
// so initialize() checks that the index finds an article by its own title.
export class HelpSearchProvider implements SearchProvider {
readonly name = 'help-flexsearch';

private docs: Record<string, ProcessedDoc> | null = null;

private index = new FlexSearch.Document({
...flexsearchConfig,
document: {
id: 'id',
index: Object.keys(FIELD_WEIGHTS),
store: ['title', 'description'],
},
});

async initialize(_context: unknown, initData?: SearchProviderInitData): Promise<void> {
if (!initData?.docs || !initData.indexData) {
throw new Error('[HelpSearch] docs and indexData are required');
}
for (const [key, value] of Object.entries(initData.indexData)) {
await this.index.import(key, value as string);
}
const [first] = Object.values(initData.docs);
if (first && this.index.search(first.title).length === 0) {
throw new Error('[HelpSearch] The index finds nothing: does flexsearch match the plugin?');
}
this.docs = initData.docs;
}

isReady(): boolean {
return this.docs !== null;
}

async search(query: string, options?: SearchOptions): Promise<SearchResult[]> {
const docs = this.getDocs();
const limit = options?.limit ?? 16;
const total = this.getDocCount();
const words = wordsByStem(query);
const terms = new Set(words.keys());

// Every word is looked up on its own, so an article matching only part of
// the question still counts. A word found in few articles says more about
// the question than one found everywhere (IDF), and a word in the title
// says more than one in the body (field weight).
const scores = new Map<string, number>();
for (const word of words.values()) {
const weights = new Map<string, number>();
for (const { field, result } of this.index.search(word, { limit: total })) {
const weight = FIELD_WEIGHTS[field];
for (const id of result) {
const docId = String(id);
weights.set(docId, Math.max(weights.get(docId) ?? 0, weight));
}
}
const idf = Math.log(1 + total / weights.size);
for (const [docId, weight] of weights) {
scores.set(docId, (scores.get(docId) ?? 0) + idf * weight);
}
}

return [...scores]
.sort(([, a], [, b]) => b - a)
.slice(0, limit)
.flatMap(([url, score]) => {
const doc = docs[url];
if (!doc) return [];
return {
url,
route: doc.route,
title: doc.title,
score,
snippet: snippet(doc.markdown, terms),
matchingHeadings: doc.headings
.map((heading) => heading.text)
.filter((text) => encode(text).some((stem) => terms.has(stem)))
.slice(0, 3),
};
});
}

async getDocument(url: string): Promise<ProcessedDoc | null> {
return this.getDocs()[url] ?? null;
}

getDocCount(): number {
return Object.keys(this.docs ?? {}).length;
}

async healthCheck(): Promise<{ healthy: boolean; message?: string }> {
if (!this.isReady()) {
return { healthy: false, message: 'Help search provider not initialized' };
}
return { healthy: true, message: `Help search ready with ${this.getDocCount()} documents` };
}

private getDocs(): Record<string, ProcessedDoc> {
if (!this.docs) {
throw new Error('[HelpSearch] Provider not initialized');
}
return this.docs;
}
}

// One word per stem: the index encodes the query itself, so it gets the word
// rather than the stem — a stem encoded again may lose another suffix.
function wordsByStem(query: string): Map<string, string> {
const words = new Map<string, string>();
for (const word of query.split(WORD_SEPARATOR)) {
for (const stem of encode(word)) {
if (!words.has(stem)) words.set(stem, word);
}
}
return words;
}

// A stem is the lowercased start of its word (with «ё» read as «е»), so the
// earliest stem found in the text marks where the article answers the query.
function snippet(markdown: string, terms: Set<string>): string {
const text = markdown.toLowerCase().replaceAll('ё', 'е');
const positions = [...terms].map((term) => text.indexOf(term)).filter((index) => index !== -1);
const start = positions.length > 0 ? Math.max(0, Math.min(...positions) - 50) : 0;
const end = Math.min(markdown.length, start + SNIPPET_LENGTH);
const body = markdown
.slice(start, end)
.replace(/^#{1,6}\s+/gm, '')
.replace(/!\[([^\]]*)\]\([^)]+\)/g, '')
.replace(/\[([^\]]+)\]\([^)]+\)/g, '$1')
.replace(/```[a-z]*\n?/g, '')
.replace(/`([^`]+)`/g, '$1')
.replace(/\s+/g, ' ')
.trim();
return `${start > 0 ? '...' : ''}${body}${end < markdown.length ? '...' : ''}`;
}
Loading