Publish solution for task 6a02e23da6fe2e4ac16acf65: update load_documents.py

This commit is contained in:
2026-06-18 09:15:00 +00:00
parent 7d6f099248
commit 0dff9baa7b
+13 -12
View File
@@ -1,14 +1,13 @@
"""Utility script to load documents from a directory into the Qdrant vector store.
The script walks through the specified directory, reads all .txt files, splits them into chunks using
`RecursiveCharacterTextSplitter`, embeds the chunks with `OllamaEmbeddings`, and stores them in the
local Qdrant collection via the helper functions defined in :mod:`main`.
Usage:
python load_documents.py /path/to/docs
The script prints the number of documents added.
"""
# Utility script to load documents from a directory into the Chroma vector store.
#
# The script walks through the specified directory, reads all .txt files, splits them into chunks using
# `RecursiveCharacterTextSplitter`, embeds the chunks with `OllamaEmbeddings`, and stores them in the
# local Chroma collection via the helper functions defined in :mod:`main`.
#
# Usage:
# python load_documents.py /path/to/docs
#
# The script prints the number of documents added.
import os
import sys
@@ -48,7 +47,9 @@ def load_documents_from_dir(directory: str) -> int:
title = file_path.stem
documents = chunk_document(content, title)
ids = [str(uuid4()) for _ in documents]
vector_store.add_documents(documents, ids=ids)
embeddings = embedding.embed_documents([doc.page_content for doc in documents])
collection = vector_store.get_collection(name="rag_memory")
collection.add(ids=ids, documents=[doc.page_content for doc in documents], embeddings=embeddings, metadatas=[doc.metadata for doc in documents])
total_chunks += len(documents)
return total_chunks