From 144be41b2a619ef346e403c54edaa978b3663ad1 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?=D0=9A=D0=B8=D1=80=D0=B8=D0=BB=D0=BB=20=D0=A0=D0=BE=D0=BC?= =?UTF-8?q?=D0=B0=D0=BD=D0=BE=D0=B2?= Date: Tue, 2 Jun 2026 07:13:00 +0000 Subject: [PATCH] Update vectorstore.py --- vectorstore.py | 46 +++++++++++++++++++++++----------------------- 1 file changed, 23 insertions(+), 23 deletions(-) diff --git a/vectorstore.py b/vectorstore.py index 0a701f0..2c6a67e 100644 --- a/vectorstore.py +++ b/vectorstore.py @@ -1,53 +1,53 @@ -"""Module for creating and loading a Chroma vector store with Ollama embeddings.""" +""" +Vector store utilities for ChromaDB. +""" from pathlib import Path -from typing import List from langchain_ollama import OllamaEmbeddings from langchain_chroma import Chroma from langchain_text_splitters import RecursiveCharacterTextSplitter -from langchain.docstore.document import Document -def create_vectorstore(persist_directory: str = "./chroma_db") -> Chroma: - """Create a Chroma vector store with Ollama embeddings. +def create_vectorstore(persist_directory: str = "./chroma_db"): + """Create or load a Chroma vector store. Parameters ---------- persist_directory: str - Directory where the vector store will be persisted. + Directory where the Chroma database is persisted. Returns ------- Chroma - The created Chroma vector store. + The Chroma vector store instance. """ embeddings = OllamaEmbeddings(model="nomic-embed-text") return Chroma(persist_directory=persist_directory, embedding_function=embeddings) -def load_documents(directory: str, vectorstore: Chroma, chunk_size: int = 1000, chunk_overlap: int = 200) -> None: - """Load .txt and .md files from a directory, split them into chunks, and add to the vector store. +def load_documents(directory: str, vectorstore: Chroma, chunk_size: int = 1000, chunk_overlap: int = 200): + """Read all .txt and .md files from a directory, split them into chunks and add to the vectorstore. Parameters ---------- directory: str - Path to the directory containing documents. + Path to the folder containing documents. vectorstore: Chroma - The vector store to add documents to. + The vector store to which documents will be added. chunk_size: int - Maximum size of each chunk. + Maximum number of characters per chunk. chunk_overlap: int - Number of characters to overlap between chunks. + Number of overlapping characters between consecutive chunks. """ splitter = RecursiveCharacterTextSplitter(chunk_size=chunk_size, chunk_overlap=chunk_overlap) - documents: List[Document] = [] - for file_path in Path(directory).glob("**/*"): - if file_path.suffix.lower() in {".txt", ".md"}: - text = file_path.read_text(encoding="utf-8") - chunks = splitter.split_text(text) - for chunk in chunks: - documents.append(Document(page_content=chunk, metadata={"source": str(file_path)})) - if documents: - vectorstore.add_documents(documents) - vectorstore.persist() + docs = [] + for file_path in Path(directory).rglob("*.txt"): + docs.append(file_path.read_text(encoding="utf-8")) + for file_path in Path(directory).rglob("*.md"): + docs.append(file_path.read_text(encoding="utf-8")) + if not docs: + return + texts = splitter.split_text("\n\n".join(docs)) + vectorstore.add_texts(texts) + vectorstore.persist()