Files
task-6a1864f78a94f887e50d46da/vectorstore.py
T
2026-06-02 07:24:32 +00:00

94 lines
3.2 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
"""Utilities for creating and populating a ChromaDB vector store.
This module contains two helper functions:
* :func:`create_vectorstore` returns a :class:`langchain_chroma.Chroma` instance backed by
an ``OllamaEmbeddings`` model.
* :func:`load_documents` reads ``.txt``/``.md`` files from a directory, splits them into
chunks using :class:`langchain_text_splitters.RecursiveCharacterTextSplitter`, and adds
the chunks to the vector store.
The vector store is persisted in ``./chroma_db`` by default, so it survives program
restarts.
"""
from pathlib import Path
from typing import Iterable
from langchain_chroma import Chroma
from langchain_ollama import OllamaEmbeddings
from langchain_text_splitters import RecursiveCharacterTextSplitter
# ---------------------------------------------------------------------------
# Vector store creation
# ---------------------------------------------------------------------------
def create_vectorstore(persist_directory: str = "./chroma_db") -> Chroma:
"""Create a Chroma vector store backed by Ollama embeddings.
Parameters
----------
persist_directory: str
Path to the directory where the Chroma DB will be stored.
Returns
-------
Chroma
A Chroma vector store instance.
"""
embeddings = OllamaEmbeddings(model="nomic-embed-text")
return Chroma(
persist_directory=persist_directory,
embedding_function=embeddings,
)
# ---------------------------------------------------------------------------
# Document ingestion
# ---------------------------------------------------------------------------
def load_documents(directory: str | Path, vectorstore: Chroma) -> None:
"""Load all ``.txt`` and ``.md`` files from *directory* into *vectorstore*.
The files are split into chunks using
:class:`langchain_text_splitters.RecursiveCharacterTextSplitter` before being
added to the vector store.
Parameters
----------
directory: str | Path
Directory containing the documents.
vectorstore: Chroma
The vector store to populate.
"""
path = Path(directory)
if not path.is_dir():
raise ValueError(f"{directory!r} is not a directory")
splitter = RecursiveCharacterTextSplitter(chunk_size=1000, chunk_overlap=200)
docs = []
for file in path.rglob("*.txt"):
docs.append(file.read_text(encoding="utf-8"))
for file in path.rglob("*.md"):
docs.append(file.read_text(encoding="utf-8"))
if not docs:
print("No documents found in", directory)
return
# Split all documents into chunks
chunks = splitter.split_text("\n\n".join(docs))
# Create LangChain Document objects
from langchain.docstore.document import Document
documents = [Document(page_content=chunk) for chunk in chunks]
vectorstore.add_documents(documents)
vectorstore.persist()
print(f"Added {len(documents)} chunks to the vector store.")
# ---------------------------------------------------------------------------
# Example usage (uncomment to run manually)
# ---------------------------------------------------------------------------
# if __name__ == "__main__":
# store = create_vectorstore()
# load_documents("documents", store)