125 lines
4.4 KiB
Python
125 lines
4.4 KiB
Python
"""Vector store and knowledge base implementation using Qdrant and Ollama embeddings.
|
||
|
||
This module defines a `KnowledgeBase` class that manages a Qdrant collection, provides methods to add documents (with chunking) and perform semantic search.
|
||
"""
|
||
|
||
from __future__ import annotations
|
||
|
||
import os
|
||
from pathlib import Path
|
||
from typing import List, Dict, Any
|
||
|
||
from langchain_ollama import OllamaEmbeddings
|
||
from langchain_qdrant import QdrantVectorStore
|
||
from langchain_core.documents import Document
|
||
from langchain_text_splitters import RecursiveCharacterTextSplitter
|
||
from qdrant_client import QdrantClient
|
||
from qdrant_client.http.models import Distance, VectorParams
|
||
|
||
# Default configuration constants
|
||
DEFAULT_COLLECTION_NAME = "knowledge_base"
|
||
DEFAULT_VECTOR_SIZE = 3072 # size of nomic-embed-text embeddings
|
||
DEFAULT_DISTANCE = Distance.COSINE
|
||
DEFAULT_QDRANT_PATH = Path("./qdrant_data")
|
||
|
||
class KnowledgeBase:
|
||
"""A thin wrapper around QdrantVectorStore.
|
||
|
||
The class ensures that the collection is created only once and provides
|
||
convenient methods for adding documents and performing semantic search.
|
||
"""
|
||
|
||
def __init__(
|
||
self,
|
||
collection_name: str = DEFAULT_COLLECTION_NAME,
|
||
host: str | None = None,
|
||
port: int | None = None,
|
||
path: str | None = None,
|
||
api_key: str | None = None,
|
||
) -> None:
|
||
"""Create or connect to a Qdrant collection.
|
||
|
||
Parameters
|
||
----------
|
||
collection_name: str
|
||
Name of the Qdrant collection.
|
||
host, port: str/int
|
||
Optional host and port for a remote Qdrant instance.
|
||
path: str
|
||
Path for an on‑disk Qdrant instance (used in local mode).
|
||
api_key: str
|
||
API key for Qdrant Cloud.
|
||
"""
|
||
|
||
# Determine client connection.
|
||
if host and port:
|
||
self.client = QdrantClient(url=f"{host}:{port}")
|
||
elif path:
|
||
self.client = QdrantClient(path=path)
|
||
else:
|
||
# Default to a persistent on‑disk client.
|
||
self.client = QdrantClient(path=str(DEFAULT_QDRANT_PATH))
|
||
|
||
self.collection_name = collection_name
|
||
|
||
# Create collection if it does not exist.
|
||
if collection_name not in self.client.get_collections().collections:
|
||
self.client.create_collection(
|
||
collection_name=collection_name,
|
||
vectors_config=VectorParams(size=DEFAULT_VECTOR_SIZE, distance=DEFAULT_DISTANCE),
|
||
)
|
||
|
||
# Embedding model from Ollama.
|
||
self.embeddings = OllamaEmbeddings(model="nomic-embed-text")
|
||
|
||
# Vector store wrapper.
|
||
self.store = QdrantVectorStore(
|
||
client=self.client,
|
||
collection_name=collection_name,
|
||
embedding=self.embeddings,
|
||
)
|
||
|
||
# Text splitter for chunking.
|
||
self.splitter = RecursiveCharacterTextSplitter(
|
||
chunk_size=500, chunk_overlap=50, length_function=len
|
||
)
|
||
|
||
# ---------------------------------------------------------------------
|
||
# Public API
|
||
# ---------------------------------------------------------------------
|
||
|
||
def add_document(self, title: str, content: str) -> None:
|
||
"""Add a document to the knowledge base.
|
||
|
||
The content is split into chunks, embedded, and stored.
|
||
"""
|
||
# Split into Document objects with metadata.
|
||
docs = self.splitter.create_documents([content])
|
||
for i, doc in enumerate(docs):
|
||
# Attach metadata: title and chunk index.
|
||
doc.metadata.update({"title": title, "chunk_index": i})
|
||
# Add to store.
|
||
self.store.add_documents(docs)
|
||
|
||
def search(self, query: str, limit: int = 5) -> List[Dict[str, Any]]:
|
||
"""Perform a semantic search and return results.
|
||
|
||
Returns a list of dictionaries containing the chunk content and metadata.
|
||
"""
|
||
results = self.store.similarity_search(query, k=limit)
|
||
output = []
|
||
for doc in results:
|
||
output.append(
|
||
{
|
||
"content": doc.page_content,
|
||
"title": doc.metadata.get("title"),
|
||
"chunk_index": doc.metadata.get("chunk_index"),
|
||
}
|
||
)
|
||
return output
|
||
|
||
def get_all_documents(self) -> List[Document]:
|
||
"""Return all documents stored in the collection."""
|
||
return self.store.get_all_documents()
|
||
|
||
# End of src/vector_store.py |