52 lines
1.8 KiB
Python
52 lines
1.8 KiB
Python
"""
|
|
RAG vector store using Qdrant and Ollama embeddings.
|
|
"""
|
|
|
|
from pathlib import Path
|
|
from typing import List
|
|
|
|
import chromadb # kept for compatibility if needed
|
|
from langchain.embeddings.ollama import OllamaEmbeddings
|
|
from langchain.text_splitter import RecursiveCharacterTextSplitter
|
|
from langchain.schema.document import Document
|
|
from langchain.vectorstores import Qdrant
|
|
|
|
CHROMA_DIR = "./chroma_db"
|
|
EMBED_MODEL = "nomic-embed-text"
|
|
|
|
|
|
def create_vectorstore(persist_directory: str = CHROMA_DIR):
|
|
"""Create or load a Qdrant vector store.
|
|
|
|
Parameters
|
|
----------
|
|
persist_directory:
|
|
Directory where the Qdrant database is stored. If it does not exist, it will be created.
|
|
"""
|
|
embeddings = OllamaEmbeddings(model=EMBED_MODEL)
|
|
# Qdrant can use a local file store via `path` argument
|
|
client = Qdrant(persist_directory=persist_directory, embedding_function=embeddings)
|
|
return client
|
|
|
|
|
|
def load_documents(directory: str, vectorstore) -> None:
|
|
"""Load all .txt and .md files from *directory*, chunk them and add to the vector store.
|
|
|
|
The function does not return anything; it mutates the provided collection.
|
|
"""
|
|
text_splitter = RecursiveCharacterTextSplitter(chunk_size=1000, chunk_overlap=200)
|
|
docs: List[Document] = []
|
|
for path in Path(directory).rglob("*.txt"):
|
|
content = path.read_text(encoding="utf-8")
|
|
docs.extend(text_splitter.create_documents([content], metadata={"source": str(path)}))
|
|
for path in Path(directory).rglob("*.md"):
|
|
content = path.read_text(encoding="utf-8")
|
|
docs.extend(text_splitter.create_documents([content], metadata={"source": str(path)}))
|
|
|
|
if docs:
|
|
# Qdrant expects texts and metadatas lists
|
|
vectorstore.add_texts(
|
|
texts=[doc.page_content for doc in docs],
|
|
metadatas=[doc.metadata for doc in docs],
|
|
)
|