Files
brojs-task-6a1864f78a94f887…/vectorstore.py
T

47 lines
1.7 KiB
Python

"""
RAG vector store using ChromaDB and Ollama embeddings.
"""
from pathlib import Path
from typing import List, Iterable
import chromadb
from langchain.embeddings.ollama import OllamaEmbeddings
from langchain.text_splitter import RecursiveCharacterTextSplitter
from langchain.schema.document import Document
CHROMA_DIR = "./chroma_db"
EMBED_MODEL = "nomic-embed-text"
def create_vectorstore(persist_directory: str = CHROMA_DIR):
"""Create or load a Chroma vector store.
Parameters
----------
persist_directory:
Directory where the Chroma database is stored. If it does not exist, it will be created.
"""
embeddings = OllamaEmbeddings(model=EMBED_MODEL)
client = chromadb.PersistentClient(path=persist_directory)
# Use a single collection named "documents"
return client.get_or_create_collection(name="documents", embedding_function=embeddings)
def load_documents(directory: str, vectorstore) -> None:
"""Load all .txt and .md files from *directory*, chunk them and add to the vector store.
The function does not return anything; it mutates the provided collection.
"""
text_splitter = RecursiveCharacterTextSplitter(chunk_size=1000, chunk_overlap=200)
docs: List[Document] = []
for path in Path(directory).rglob("*.txt"):
content = path.read_text(encoding="utf-8")
docs.extend(text_splitter.create_documents([content], metadata={"source": str(path)}))
for path in Path(directory).rglob("*.md"):
content = path.read_text(encoding="utf-8")
docs.extend(text_splitter.create_documents([content], metadata={"source": str(path)}))
if docs:
vectorstore.add(documents=docs)