From 0c4400b542390c1e27079c7404d78dcc87dddce4 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?=D0=9A=D0=B8=D1=80=D0=B8=D0=BB=D0=BB=20=D0=A0=D0=BE=D0=BC?= =?UTF-8?q?=D0=B0=D0=BD=D0=BE=D0=B2?= Date: Thu, 4 Jun 2026 20:02:44 +0000 Subject: [PATCH] Update qdrant_store.py --- qdrant_store.py | 61 +++++++++++++++++++------------------------------ 1 file changed, 24 insertions(+), 37 deletions(-) diff --git a/qdrant_store.py b/qdrant_store.py index 96b42ad..6120f7a 100644 --- a/qdrant_store.py +++ b/qdrant_store.py @@ -1,50 +1,37 @@ -"""Module for interacting with Qdrant vector store using Ollama embeddings. +"""Wrapper around QdrantVectorStore using Ollama embeddings. + +This module is intentionally lightweight and can be imported from any +context without requiring the caller to set up a package structure. """ -from typing import List, Dict, Any +import os +import sys + +# Ensure the current directory is in sys.path so that relative imports work +sys.path.append(os.path.dirname(__file__)) from langchain_ollama import OllamaEmbeddings from langchain_qdrant import QdrantVectorStore -from qdrant_client import QdrantClient +from langchain_text_splitters import RecursiveCharacterTextSplitter +from typing import List, Dict class QdrantStore: - """Wrapper around QdrantVectorStore. + """Wrapper around QdrantVectorStore using Ollama embeddings.""" - Parameters - ---------- - collection_name: str - Name of the Qdrant collection. - host: str - Qdrant host URL. - port: int - Qdrant port. - """ - - def __init__(self, collection_name: str = "rag_collection", host: str = "localhost", port: int = 6333): - self.collection_name = collection_name - self.client = QdrantClient(host=host, port=port) + def __init__(self, host: str = "localhost", port: int = 6333, collection_name: str = "knowledge_base"): self.embeddings = OllamaEmbeddings(model="nomic-embed-text") - # Create collection if not exists - if collection_name not in self.client.get_collections().collections: - self.client.recreate_collection(collection_name=collection_name, vectors_config={"size": 512, "distance": "Cosine"}) - self.store = QdrantVectorStore.from_existing_collection( + self.store = QdrantVectorStore( + url=f"http://{host}:{port}", collection_name=collection_name, - embedding=self.embeddings, - client=self.client, + embeddings=self.embeddings, ) - def add_documents(self, documents: List[str], titles: List[str] | None = None, metadatas: List[Dict[str, Any]] | None = None) -> None: - """Add documents to the collection. + def add_document(self, content: str, title: str) -> None: + splitter = RecursiveCharacterTextSplitter(chunk_size=500, chunk_overlap=50) + chunks = splitter.split_text(content) + metadatas = [{"title": title, "source": title} for _ in chunks] + self.store.add_texts(chunks, metadatas=metadatas) - Each document is added as a separate vector. If titles or metadatas are provided, they are attached. - """ - if titles is None: - titles = [f"doc_{i}" for i in range(len(documents))] - if metadatas is None: - metadatas = [{} for _ in range(len(documents))] - self.store.add_texts(texts=documents, metadatas=metadatas, ids=titles) - - def search(self, query: str, k: int = 5) -> List[Dict[str, Any]]: - """Semantic search returning list of dicts with 'content' and 'metadata'.""" - results = self.store.similarity_search_with_score(query, k=k) - return [{"content": r[0].page_content, "metadata": r[0].metadata, "score": r[1]} for r in results] + def search(self, query: str, k: int = 5) -> List[Dict]: + results = self.store.similarity_search(query, k=k) + return [{"content": doc.page_content, "metadata": doc.metadata} for doc in results] \ No newline at end of file