diff --git a/backend/indexing/embedder.py b/backend/indexing/embedder.py index 806e620..21c529f 100644 --- a/backend/indexing/embedder.py +++ b/backend/indexing/embedder.py @@ -34,6 +34,17 @@ def model(self): logger.info("Embedding model loaded.") return self._model + def load(self) -> None: + """ + Force the model to load/download immediately (eager loading). + + Intended to be called during application startup so that the + (potentially slow) model download/initialization happens before + the server begins accepting HTTP requests, rather than blocking + the first incoming request. + """ + _ = self.model + @property def dimension(self) -> int: """Embedding vector dimension.""" diff --git a/backend/main.py b/backend/main.py index dfc0ecf..79466d2 100644 --- a/backend/main.py +++ b/backend/main.py @@ -43,6 +43,14 @@ async def lifespan(app: FastAPI): app.state.orchestrator = Orchestrator() app.state.database = Database() + # Eagerly load the embedding model now so the (slow) download/init + # happens during startup instead of blocking the first HTTP request. + # Without this, the embedder's lazy `@property` defers loading until + # first use, causing 502s while the model downloads on first request. + logger.info("Eagerly loading embedding model before accepting requests...") + app.state.orchestrator.embedder.load() + logger.info("Embedding model ready.") + logger.info( f"Research Agent API ready. " f"Vector store: {app.state.orchestrator.vector_store.count()} passages"