From 0dff9baa7b89b0c3dec40b97c8c73cbc9071a533 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?=D0=A0=D0=B8=D0=BD=D0=B0=D1=80=20=D0=9C=D0=B8=D1=80=D0=B7?= =?UTF-8?q?=D0=B0=D0=B3=D0=B8=D1=82=D0=BE=D0=B2?= Date: Thu, 18 Jun 2026 09:15:00 +0000 Subject: [PATCH] Publish solution for task 6a02e23da6fe2e4ac16acf65: update load_documents.py --- load_documents.py | 25 +++++++++++++------------ 1 file changed, 13 insertions(+), 12 deletions(-) diff --git a/load_documents.py b/load_documents.py index f5ba73a..1fe884a 100644 --- a/load_documents.py +++ b/load_documents.py @@ -1,14 +1,13 @@ -"""Utility script to load documents from a directory into the Qdrant vector store. - -The script walks through the specified directory, reads all .txt files, splits them into chunks using -`RecursiveCharacterTextSplitter`, embeds the chunks with `OllamaEmbeddings`, and stores them in the -local Qdrant collection via the helper functions defined in :mod:`main`. - -Usage: - python load_documents.py /path/to/docs - -The script prints the number of documents added. -""" +# Utility script to load documents from a directory into the Chroma vector store. +# +# The script walks through the specified directory, reads all .txt files, splits them into chunks using +# `RecursiveCharacterTextSplitter`, embeds the chunks with `OllamaEmbeddings`, and stores them in the +# local Chroma collection via the helper functions defined in :mod:`main`. +# +# Usage: +# python load_documents.py /path/to/docs +# +# The script prints the number of documents added. import os import sys @@ -48,7 +47,9 @@ def load_documents_from_dir(directory: str) -> int: title = file_path.stem documents = chunk_document(content, title) ids = [str(uuid4()) for _ in documents] - vector_store.add_documents(documents, ids=ids) + embeddings = embedding.embed_documents([doc.page_content for doc in documents]) + collection = vector_store.get_collection(name="rag_memory") + collection.add(ids=ids, documents=[doc.page_content for doc in documents], embeddings=embeddings, metadatas=[doc.metadata for doc in documents]) total_chunks += len(documents) return total_chunks