Files
2026-05-27 14:37:31 +00:00

72 lines
2.7 KiB
Python
Raw Permalink Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
"""
LangChain tools for adding and searching a Qdrant knowledge base.
The module exposes two @tooldecorated functions:
* add_to_knowledge_base(content: str, title: str = "document")
* search_knowledge_base(query: str, max_results: int = 5)
Both tools use an inmemory Qdrant client and Ollama embeddings.
"""
import os
from typing import List
from langchain_core.documents import Document
from langchain_text_splitters import RecursiveCharacterTextSplitter
from langchain_ollama import ChatOllama, OllamaEmbeddings
from langchain_qdrant import QdrantVectorStore
from langchain.tools import tool
# --- Configuration -------------------------------------------------------
QDRANT_COLLECTION = "rag_agent"
EMBEDDING_MODEL = "nomic-embed-text"
LLM_MODEL = "llama3"
# Initialize embeddings and vector store (inmemory Qdrant)
embeddings = OllamaEmbeddings(model=EMBEDDING_MODEL, base_url="http://localhost:11434")
vector_store = QdrantVectorStore(client=None, collection_name=QDRANT_COLLECTION, embedding=embeddings)
# Ensure the collection exists (creates if not present)
try:
vector_store.client.create_collection(
name=QDRANT_COLLECTION,
vectors_config=dict(size=embeddings.embed_query("test").shape[0], distance="cosine"),
)
except Exception:
# Collection already exists ignore
pass
# Text splitter for chunking documents
splitter = RecursiveCharacterTextSplitter(chunk_size=500, chunk_overlap=50)
@tool("Add a document to the knowledge base")
def add_to_knowledge_base(content: str, title: str = "document") -> str:
"""
Splits *content* into chunks, embeds them via Ollama and stores in Qdrant.
Returns a confirmation string with number of chunks added.
"""
# Split content
docs: List[Document] = splitter.split_text(content)
for i, chunk in enumerate(docs):
doc = Document(page_content=chunk, metadata={"title": title, "chunk_index": i})
vector_store.add_documents([doc])
return f"Added {len(docs)} chunks from '{title}'."
@tool("Search the knowledge base")
def search_knowledge_base(query: str, max_results: int = 5) -> str:
"""
Performs a semantic similarity search in Qdrant.
Returns a formatted string with top results and their metadata.
"""
docs = vector_store.similarity_search(query, k=max_results)
if not docs:
return "No relevant documents found."
lines: List[str] = []
for i, doc in enumerate(docs, 1):
title = doc.metadata.get("title", "unknown")
idx = doc.metadata.get("chunk_index", "?")
lines.append(f"{i}. [{title} - chunk {idx}]\n{doc.page_content[:200]}...")
return "\n\n".join(lines)
# Expose tool names for agent creation
TOOLS = [add_to_knowledge_base, search_knowledge_base]