diff --git a/vectorstore.py b/vectorstore.py index 74308ad..0a701f0 100644 --- a/vectorstore.py +++ b/vectorstore.py @@ -1,56 +1,53 @@ -""" -Vector store utilities for ChromaDB with Ollama embeddings. -""" +"""Module for creating and loading a Chroma vector store with Ollama embeddings.""" -import os from pathlib import Path +from typing import List -from langchain_chroma import Chroma from langchain_ollama import OllamaEmbeddings +from langchain_chroma import Chroma from langchain_text_splitters import RecursiveCharacterTextSplitter from langchain.docstore.document import Document + def create_vectorstore(persist_directory: str = "./chroma_db") -> Chroma: - """Create or load a Chroma vector store. + """Create a Chroma vector store with Ollama embeddings. Parameters ---------- persist_directory: str - Directory where the Chroma database is persisted. + Directory where the vector store will be persisted. Returns ------- Chroma - A Chroma vector store instance. + The created Chroma vector store. """ embeddings = OllamaEmbeddings(model="nomic-embed-text") return Chroma(persist_directory=persist_directory, embedding_function=embeddings) -def load_documents(directory: str, vectorstore: Chroma) -> None: - """Load text and markdown files from *directory*, chunk them, and add to *vectorstore*. + +def load_documents(directory: str, vectorstore: Chroma, chunk_size: int = 1000, chunk_overlap: int = 200) -> None: + """Load .txt and .md files from a directory, split them into chunks, and add to the vector store. Parameters ---------- directory: str - Path to the folder containing .txt and .md files. + Path to the directory containing documents. vectorstore: Chroma - The Chroma vector store to which documents will be added. + The vector store to add documents to. + chunk_size: int + Maximum size of each chunk. + chunk_overlap: int + Number of characters to overlap between chunks. """ - docs = [] - for file_path in Path(directory).rglob("*.*"): - if file_path.suffix.lower() in ".txt .md".split(): - try: - text = file_path.read_text(encoding="utf-8") - except Exception as e: - print(f"Failed to read {file_path}: {e}") - continue - docs.append(Document(page_content=text, metadata={"source": str(file_path)})) - - if not docs: - print("No documents found to load.") - return - - splitter = RecursiveCharacterTextSplitter(chunk_size=1000, chunk_overlap=200) - chunks = splitter.split_documents(docs) - vectorstore.add_documents(chunks) - print(f"Loaded {len(chunks)} chunks into the vector store.") \ No newline at end of file + splitter = RecursiveCharacterTextSplitter(chunk_size=chunk_size, chunk_overlap=chunk_overlap) + documents: List[Document] = [] + for file_path in Path(directory).glob("**/*"): + if file_path.suffix.lower() in {".txt", ".md"}: + text = file_path.read_text(encoding="utf-8") + chunks = splitter.split_text(text) + for chunk in chunks: + documents.append(Document(page_content=chunk, metadata={"source": str(file_path)})) + if documents: + vectorstore.add_documents(documents) + vectorstore.persist()