import os import asyncio from langchain_openai import ChatOpenAI from langchain.tools import tool from deepagents import create_deep_agent from deepagents.backends import FilesystemBackend, LocalShellBackend, CompositeBackend from langchain_core.messages import HumanMessage from langchain_core.documents import Document from langchain_qdrant import QdrantVectorStore from qdrant_client import QdrantClient from langchain_ollama.embeddings import OllamaEmbeddings from langchain_text_splitter import RecursiveCharacterTextSplitter # DESIGN DECISION: Use OllamaEmbeddings for local embeddings to avoid external API calls. # NECESSITY: Assignment requires local LLM and embeddings via Ollama. # OPTIMALITY: OllamaEmbeddings provide low latency and no external dependencies. # ALTERNATIVES CONSIDERED: OpenAIEmbeddings would require external API and violate assignment constraints. embeddings = OllamaEmbeddings(model="nomic-embed-text") # DESIGN DECISION: Initialize QdrantVectorStore with local Qdrant client and OllamaEmbeddings. # NECESSITY: RAG requires a vector store; Qdrant is specified in the stack. # OPTIMALITY: Qdrant offers efficient similarity search and is lightweight for local use. # ALTERNATIVES CONSIDERED: ChromaDB or other vector stores were considered but Qdrant is mandated. qdrant_client = QdrantClient(host="localhost", port=6333) vector_store = QdrantVectorStore( client=qdrant_client, collection_name="knowledge", embedding_function=embeddings ) # DESIGN DECISION: Use RecursiveCharacterTextSplitter with chunk_size=500 and chunk_overlap=100. # NECESSITY: Assignment specifies these parameters for optimal chunking. # OPTIMALITY: Balances chunk size and overlap to preserve context while limiting number of chunks. # ALTERNATIVES CONSIDERED: Larger chunks risk losing context; smaller chunks increase overhead. splitter = RecursiveCharacterTextSplitter(chunk_size=500, chunk_overlap=100) @tool def search_knowledge_base(query: str, max_results: int = 3) -> str: """Search the knowledge base for relevant information.""" docs = vector_store.similarity_search(query, k=max_results) return "\n".join(d.page_content for d in docs) if docs else "No results." @tool def add_to_knowledge_base(content: str, title: str = "doc") -> str: """Add content to the knowledge base.""" chunks = splitter.split_text(content) docs = [Document(page_content=chunk, metadata={"title": title}) for chunk in chunks] vector_store.add_documents(docs) return f"Added {len(docs)} chunks for {title}." # DESIGN DECISION: Use OpenRouter via langchain_openai for LLM. # NECESSITY: Assignment mandates OpenRouter usage. # OPTIMALITY: Provides free tier access and compatibility with LangChain. # ALTERNATIVES CONSIDERED: Local LLMs would require GPU resources. llm = ChatOpenAI( model="openai/gpt-oss-20b:free", base_url="https://openrouter.ai/api/v1", api_key=os.getenv("OPENAI_API_KEY"), temperature=0.0, ) backend = CompositeBackend([ LocalShellBackend(workspace_dir="./workspace"), FilesystemBackend(), ]) agent = create_deep_agent( model=llm, tools=[search_knowledge_base, add_to_knowledge_base], backend=backend, system_prompt="You are a helpful agent with a knowledge base. Use the tools search_knowledge_base and add_to_knowledge_base as needed.", ) async def main(): print("RAG Agent CLI. Commands: /add