Решение готово к публикации: add vectorstore.py

This commit is contained in:
2026-06-11 09:16:31 +00:00
parent e34bbb8aa1
commit 30620c0201
+46
View File
@@ -0,0 +1,46 @@
"""
RAG vector store using ChromaDB and Ollama embeddings.
"""
from pathlib import Path
from typing import List, Iterable
import chromadb
from langchain.embeddings.ollama import OllamaEmbeddings
from langchain.text_splitter import RecursiveCharacterTextSplitter
from langchain.schema.document import Document
CHROMA_DIR = "./chroma_db"
EMBED_MODEL = "nomic-embed-text"
def create_vectorstore(persist_directory: str = CHROMA_DIR):
"""Create or load a Chroma vector store.
Parameters
----------
persist_directory:
Directory where the Chroma database is stored. If it does not exist, it will be created.
"""
embeddings = OllamaEmbeddings(model=EMBED_MODEL)
client = chromadb.PersistentClient(path=persist_directory)
# Use a single collection named "documents"
return client.get_or_create_collection(name="documents", embedding_function=embeddings)
def load_documents(directory: str, vectorstore) -> None:
"""Load all .txt and .md files from *directory*, chunk them and add to the vector store.
The function does not return anything; it mutates the provided collection.
"""
text_splitter = RecursiveCharacterTextSplitter(chunk_size=1000, chunk_overlap=200)
docs: List[Document] = []
for path in Path(directory).rglob("*.txt"):
content = path.read_text(encoding="utf-8")
docs.extend(text_splitter.create_documents([content], metadata={"source": str(path)}))
for path in Path(directory).rglob("*.md"):
content = path.read_text(encoding="utf-8")
docs.extend(text_splitter.create_documents([content], metadata={"source": str(path)}))
if docs:
vectorstore.add(documents=docs)