Files
task-6a1864f78a94f887e50d46da/main.py
T
2026-06-04 16:00:00 +00:00

100 lines
3.7 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
import os
import asyncio
from pathlib import Path
from typing import List
from langchain_openai import ChatOpenAI
from langchain_core.messages import HumanMessage
from langchain.tools import tool
from langchain_ollama import OllamaEmbeddings
from langchain_chroma import Chroma
from langchain_text_splitters import RecursiveCharacterTextSplitter
from langchain_tavily import TavilySearchResults
from deepagents import create_deep_agent
from deepagents.backends import FilesystemBackend, LocalShellBackend, CompositeBackend
# --------------------------- LLM ---------------------------
llm = ChatOpenAI(
model="openai/gpt-oss-20b:free",
base_url="https://openrouter.ai/api/v1",
api_key=os.getenv("OPENAI_API_KEY"),
temperature=0.0,
)
# --------------------------- Backend ---------------------------
backend = CompositeBackend([
LocalShellBackend(workspace_dir="./workspace"),
FilesystemBackend(),
])
# --------------------------- Vectorstore ---------------------------
PERSIST_DIR = Path("./chroma_db")
PERSIST_DIR.mkdir(parents=True, exist_ok=True)
embeddings = OllamaEmbeddings(model="nomic-embed-text")
vectorstore = Chroma(persist_directory=str(PERSIST_DIR), embedding_function=embeddings)
# Load documents from ./documents if not already loaded
DOCS_DIR = Path("./documents")
if DOCS_DIR.exists():
for file in DOCS_DIR.glob("**/*.*"):
if file.suffix.lower() in {".txt", ".md"}:
text = file.read_text(encoding="utf-8")
splitter = RecursiveCharacterTextSplitter(chunk_size=1000, chunk_overlap=200)
docs = splitter.split_text(text)
vectorstore.add_texts(docs, metadatas=[{"source": str(file)} for _ in docs])
vectorstore.persist()
# --------------------------- Tools ---------------------------
@tool
def search_local_kb(query: str, top_k: int = 3) -> str:
"""Semantic search in the local ChromaDB knowledge base."""
retriever = vectorstore.as_retriever(search_kwargs={"k": top_k})
docs = retriever.invoke(query)
if not docs:
return "No local knowledge found."
return "\n---\n".join([f"{d.page_content[:500]}..." for d in docs])
@tool
def web_search(query: str) -> str:
"""Web search using Tavily."""
tavily = TavilySearchResults(api_key=os.getenv("TAVILY_API_KEY"))
results = tavily.invoke(query)
if not results:
return "No web results found."
return "\n---\n".join([f"{r['title']}: {r['content'][:500]}..." for r in results])
# --------------------------- Agent ---------------------------
SYSTEM_PROMPT = (
"You are an AI assistant that can answer questions using either a local knowledge base or the web. "
"If the answer can be found in the local documents, use the `search_local_kb` tool and prefix the response with `[Local KB]`. "
"If the answer requires uptodate information, use the `web_search` tool and prefix the response with `[Web Search]`. "
"Always indicate the source in the response."
)
agent = create_deep_agent(
model=llm,
tools=[search_local_kb, web_search],
backend=backend,
system_prompt=SYSTEM_PROMPT,
)
# --------------------------- CLI ---------------------------
async def main():
print("RAG Agent ready. Type 'exit' to quit.")
while True:
user_input = input("\nЗапрос: ")
if user_input.lower() in {"exit", "quit"}:
print("Goodbye!")
break
result = await agent.ainvoke(
{"messages": [HumanMessage(content=user_input)]},
{"configurable": {"thread_id": "session-1"}},
)
# The last message is the assistant's reply
reply = result["messages"][-1].content
print(f"\n{reply}")
if __name__ == "__main__":
asyncio.run(main())