Files
task-6a02e23da6fe2e4ac16acf65/src/vector_store.py
T
2026-06-04 22:57:42 +00:00

92 lines
3.4 KiB
Python

"""Vector store implementation using Qdrant and Ollama embeddings.
This module provides a KnowledgeBase class that wraps a QdrantVectorStore and exposes
methods for adding documents and searching the knowledge base.
"""
from pathlib import Path
from typing import List, Dict, Any
from langchain_ollama import OllamaEmbeddings
from langchain_qdrant import QdrantVectorStore
from langchain_text_splitters import RecursiveCharacterTextSplitter
from langchain_core.documents import Document
from qdrant_client import QdrantClient
class KnowledgeBase:
"""A simple wrapper around QdrantVectorStore.
Parameters
----------
collection_name: str
Name of the Qdrant collection to use.
host: str, optional
Qdrant host address. Defaults to ``localhost``.
port: int, optional
Qdrant port. Defaults to ``6333``.
"""
def __init__(self, collection_name: str = "knowledge_base", host: str = "localhost", port: int = 6333):
self.collection_name = collection_name
self.client = QdrantClient(host=host, port=port)
# Ensure the collection exists
self.client.recreate_collection(
collection_name=self.collection_name,
vectors_config={"size": 1024, "distance": "Cosine"},
)
self.embeddings = OllamaEmbeddings(model="nomic-embed-text")
self.vector_store = QdrantVectorStore(
client=self.client,
collection_name=self.collection_name,
embeddings=self.embeddings,
)
self.splitter = RecursiveCharacterTextSplitter(chunk_size=1000, chunk_overlap=200)
def add_document(self, content: str, title: str) -> None:
"""Add a document to the knowledge base.
The document is split into chunks, embedded, and stored in Qdrant.
"""
chunks = self.splitter.split_text(content)
documents: List[Document] = []
for idx, chunk in enumerate(chunks):
meta = {"title": title, "chunk_index": idx}
documents.append(Document(page_content=chunk, metadata=meta))
self.vector_store.add_documents(documents)
def search(self, query: str, max_results: int = 5) -> List[Dict[str, Any]]:
"""Search the knowledge base for the most relevant chunks.
Returns a list of dictionaries containing the chunk content, title, and score.
"""
results = self.vector_store.similarity_search_with_score(query, k=max_results)
output = []
for doc, score in results:
output.append({
"content": doc.page_content,
"title": doc.metadata.get("title"),
"chunk_index": doc.metadata.get("chunk_index"),
"score": score,
})
return output
def load_from_directory(self, directory: str) -> None:
"""Load all .txt files from a directory into the knowledge base.
Parameters
----------
directory: str
Path to the directory containing text files.
"""
path = Path(directory)
for file_path in path.rglob("*.txt"):
title = file_path.stem
content = file_path.read_text(encoding="utf-8")
self.add_document(content, title)
# Example usage (uncomment for quick test)
# if __name__ == "__main__":
# kb = KnowledgeBase()
# kb.load_from_directory("data")
# print(kb.search("What is Python?", 3))