diff --git a/vector_store.py b/vector_store.py index 01f1fef..da69a63 100644 --- a/vector_store.py +++ b/vector_store.py @@ -1,93 +1,72 @@ -"""Module for interacting with Qdrant vector store using Ollama embeddings. +"""Vector store implementation using Qdrant and Ollama embeddings. -This module provides a simple wrapper around LangChain's QdrantVectorStore. -It handles initialization, adding documents (with chunking), and semantic search. +This module provides a simple wrapper around QdrantVectorStore that handles +- Initialization of the Qdrant client and collection. +- Chunking of documents using RecursiveCharacterTextSplitter. +- Adding documents with embeddings from Ollama. +- Semantic search. """ from pathlib import Path -from typing import List +from typing import List, Dict, Any from langchain_ollama import OllamaEmbeddings from langchain_qdrant import QdrantVectorStore from langchain_text_splitters import RecursiveCharacterTextSplitter +from langchain.schema import Document +# Global configuration +QDRANT_HOST = "localhost" +QDRANT_PORT = 6333 +COLLECTION_NAME = "knowledge_base" +EMBEDDING_MODEL = "nomic-embed-text" -class QdrantStore: - """A wrapper around QdrantVectorStore. +# Initialize embeddings and splitter +embeddings = OllamaEmbeddings(model=EMBEDDING_MODEL) +text_splitter = RecursiveCharacterTextSplitter(chunk_size=500, chunk_overlap=50) - Parameters - ---------- - collection_name: str - Name of the collection in Qdrant. Defaults to "knowledge_base". - url: str - URL of the Qdrant instance. Defaults to "http://localhost:6333". +# Create or connect to Qdrant collection +vector_store = QdrantVectorStore( + client_kwargs={"host": QDRANT_HOST, "port": QDRANT_PORT}, + collection_name=COLLECTION_NAME, + embeddings=embeddings, +) + +# Ensure collection exists +if not vector_store.client.has_collection(COLLECTION_NAME): + vector_store.client.create_collection(COLLECTION_NAME) + +def add_documents(docs: List[Dict[str, str]]) -> None: + """Add a list of documents to the vector store. + + Each document dict must contain ``title`` and ``content`` keys. + The content is split into chunks before being stored. """ + documents: List[Document] = [] + for doc in docs: + title = doc.get("title", "") + content = doc.get("content", "") + # Split content into chunks + chunks = text_splitter.split_text(content) + for i, chunk in enumerate(chunks): + meta = {"title": title, "chunk_index": i} + documents.append(Document(page_content=chunk, metadata=meta)) + vector_store.add_documents(documents) - def __init__(self, collection_name: str = "knowledge_base", url: str = "http://localhost:6333"): - self.collection_name = collection_name - self.url = url - # Use Ollama embeddings model - self.embeddings = OllamaEmbeddings(model="nomic-embed-text") - # Initialize an empty vector store (will create collection if not exists) - self.vector_store = QdrantVectorStore.from_texts( - [], self.embeddings, url=self.url, collection_name=self.collection_name - ) +def search(query: str, k: int = 5) -> List[Document]: + """Semantic search in the vector store. - def _split_text(self, text: str) -> List[str]: - """Split a long text into manageable chunks. + Returns a list of Documents ordered by relevance. + """ + return vector_store.similarity_search(query, k=k) - Uses RecursiveCharacterTextSplitter with a chunk size of 1000 characters and - an overlap of 200 characters to preserve context. - """ - splitter = RecursiveCharacterTextSplitter(chunk_size=1000, chunk_overlap=200) - return splitter.split_text(text) +# Convenience: add a single document - def add_document(self, content: str, title: str) -> None: - """Add a document to the vector store. +def add_document(title: str, content: str) -> None: + add_documents([{"title": title, "content": content}]) - Parameters - ---------- - content: str - Full text of the document. - title: str - Title or identifier for the document. - """ - chunks = self._split_text(content) - # Prepare metadata for each chunk - metadatas = [{"title": title, "chunk_index": i} for i in range(len(chunks))] - self.vector_store.add_texts(chunks, metadatas) +# Convenience: search and return plain strings - def search(self, query: str, max_results: int = 5): - """Perform a semantic search. - - Parameters - ---------- - query: str - The search query. - max_results: int - Number of top results to return. - - Returns - ------- - List[Document] - List of LangChain Document objects. - """ - return self.vector_store.similarity_search(query, k=max_results) - - def load_documents_from_dir(self, directory: str) -> None: - """Load all .txt files from a directory into the vector store. - - Parameters - ---------- - directory: str - Path to the directory containing text files. - """ - path = Path(directory) - for file_path in path.rglob("*.txt"): - with file_path.open("r", encoding="utf-8") as f: - content = f.read() - title = file_path.stem - self.add_document(content, title) - -# Singleton instance used by tools -store = QdrantStore() +def search_text(query: str, k: int = 5) -> List[str]: + docs = search(query, k) + return [f"{doc.metadata.get('title', 'Untitled')} (chunk {doc.metadata.get('chunk_index', 0)}): {doc.page_content[:200]}..." for doc in docs] \ No newline at end of file