From d3b002627c0610b7bfd7b1f873a5ec205b6bec12 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?=D0=9A=D0=B8=D1=80=D0=B8=D0=BB=D0=BB=20=D0=A0=D0=BE=D0=BC?= =?UTF-8?q?=D0=B0=D0=BD=D0=BE=D0=B2?= Date: Fri, 5 Jun 2026 11:29:32 +0000 Subject: [PATCH] Update src/vector_store.py --- src/vector_store.py | 144 ++++++++++++-------------------------------- 1 file changed, 37 insertions(+), 107 deletions(-) diff --git a/src/vector_store.py b/src/vector_store.py index 969cc7e..f6b13f4 100644 --- a/src/vector_store.py +++ b/src/vector_store.py @@ -1,125 +1,55 @@ -"""Vector store and knowledge base implementation using Qdrant and Ollama embeddings. - -This module defines a `KnowledgeBase` class that manages a Qdrant collection, provides methods to add documents (with chunking) and perform semantic search. -""" - -from __future__ import annotations - -import os from pathlib import Path -from typing import List, Dict, Any - +from uuid import uuid4 from langchain_ollama import OllamaEmbeddings from langchain_qdrant import QdrantVectorStore -from langchain_core.documents import Document -from langchain_text_splitters import RecursiveCharacterTextSplitter from qdrant_client import QdrantClient from qdrant_client.http.models import Distance, VectorParams - -# Default configuration constants -DEFAULT_COLLECTION_NAME = "knowledge_base" -DEFAULT_VECTOR_SIZE = 3072 # size of nomic-embed-text embeddings -DEFAULT_DISTANCE = Distance.COSINE -DEFAULT_QDRANT_PATH = Path("./qdrant_data") +from langchain_text_splitters import RecursiveCharacterTextSplitter +from langchain_core.documents import Document class KnowledgeBase: - """A thin wrapper around QdrantVectorStore. - - The class ensures that the collection is created only once and provides - convenient methods for adding documents and performing semantic search. - """ - - def __init__( - self, - collection_name: str = DEFAULT_COLLECTION_NAME, - host: str | None = None, - port: int | None = None, - path: str | None = None, - api_key: str | None = None, - ) -> None: - """Create or connect to a Qdrant collection. - - Parameters - ---------- - collection_name: str - Name of the Qdrant collection. - host, port: str/int - Optional host and port for a remote Qdrant instance. - path: str - Path for an on‑disk Qdrant instance (used in local mode). - api_key: str - API key for Qdrant Cloud. - """ - - # Determine client connection. - if host and port: - self.client = QdrantClient(url=f"{host}:{port}") - elif path: - self.client = QdrantClient(path=path) - else: - # Default to a persistent on‑disk client. - self.client = QdrantClient(path=str(DEFAULT_QDRANT_PATH)) - - self.collection_name = collection_name - - # Create collection if it does not exist. - if collection_name not in self.client.get_collections().collections: - self.client.create_collection( - collection_name=collection_name, - vectors_config=VectorParams(size=DEFAULT_VECTOR_SIZE, distance=DEFAULT_DISTANCE), - ) - - # Embedding model from Ollama. - self.embeddings = OllamaEmbeddings(model="nomic-embed-text") - - # Vector store wrapper. - self.store = QdrantVectorStore( + """A simple wrapper around Qdrant for storing and searching documents.""" + def __init__(self, collection_name: str = "rag_kb", persist_path: str = "qdrant_data"): + # Initialize Qdrant client (in‑memory or on‑disk) + self.client = QdrantClient(path=persist_path) + # Create collection with cosine distance and 3072‑dim vectors (nomic‑embed‑text output) + self.client.create_collection( + collection_name=collection_name, + vectors_config=VectorParams(size=3072, distance=Distance.COSINE), + ) + # Create vector store wrapper + self.vector_store = QdrantVectorStore( client=self.client, collection_name=collection_name, - embedding=self.embeddings, + embedding=OllamaEmbeddings(model="nomic-embed-text"), ) + # Text splitter for chunking documents + self.splitter = RecursiveCharacterTextSplitter(chunk_size=1000, chunk_overlap=200) - # Text splitter for chunking. - self.splitter = RecursiveCharacterTextSplitter( - chunk_size=500, chunk_overlap=50, length_function=len - ) - - # --------------------------------------------------------------------- - # Public API - # --------------------------------------------------------------------- - - def add_document(self, title: str, content: str) -> None: + def add_document(self, title: str, content: str): """Add a document to the knowledge base. - The content is split into chunks, embedded, and stored. + The content is split into chunks, each chunk is embedded and stored. """ - # Split into Document objects with metadata. - docs = self.splitter.create_documents([content]) - for i, doc in enumerate(docs): - # Attach metadata: title and chunk index. - doc.metadata.update({"title": title, "chunk_index": i}) - # Add to store. - self.store.add_documents(docs) + # Create Document objects with metadata + docs = self.splitter.create_documents([content], metadata={"title": title}) + # Add to vector store + self.vector_store.add_documents(docs) - def search(self, query: str, limit: int = 5) -> List[Dict[str, Any]]: - """Perform a semantic search and return results. + def search(self, query: str, limit: int = 10): + """Search the knowledge base for the most relevant documents. - Returns a list of dictionaries containing the chunk content and metadata. + Returns a list of dictionaries containing page_content, metadata and score. """ - results = self.store.similarity_search(query, k=limit) - output = [] - for doc in results: - output.append( - { - "content": doc.page_content, - "title": doc.metadata.get("title"), - "chunk_index": doc.metadata.get("chunk_index"), - } - ) - return output + results = self.vector_store.similarity_search_with_score(query, k=limit) + return [ + { + "page_content": doc.page_content, + "metadata": doc.metadata, + "score": score, + } + for doc, score in results + ] - def get_all_documents(self) -> List[Document]: - """Return all documents stored in the collection.""" - return self.store.get_all_documents() - -# End of src/vector_store.py \ No newline at end of file +# Global singleton instance used by tools and the agent +kb = KnowledgeBase() \ No newline at end of file