From 4041a7fca610ce1d20b5c89cd986c2fe35c97f70 Mon Sep 17 00:00:00 2001 From: Danil Parunin 5f1b81b8-4f5d-11e8-9c2d-fa7ae01bbebc Date: Tue, 16 Jun 2026 16:27:51 +0000 Subject: [PATCH] =?UTF-8?q?fix(needs=5Ffixes):=201=20=D0=B8=D1=81=D0=BF?= =?UTF-8?q?=D1=80=D0=B0=D0=B2=D0=BB=D0=B5=D0=BD=D0=B8=D0=B9,=200=20=D0=BE?= =?UTF-8?q?=D1=82=D1=81=D1=82=D0=BE=D1=8F=D0=BD=D0=BE=20=E2=80=94=20main.p?= =?UTF-8?q?y?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- main.py | 186 ++++++++++++++++++++++++++------------------------------ 1 file changed, 86 insertions(+), 100 deletions(-) diff --git a/main.py b/main.py index 9fe647a..9647c64 100644 --- a/main.py +++ b/main.py @@ -1,142 +1,128 @@ import os import asyncio +import json +import httpx from pathlib import Path -from langchain_community.embeddings import OllamaEmbeddings +from langchain_openai import ChatOpenAI, OpenAIEmbeddings from langchain_chroma import Chroma from langchain_core.documents import Document -from langchain_openai import ChatOpenAI from langchain.tools import tool -from deepagents import create_deep_agent -from deepagents.backends import FilesystemBackend, LocalShellBackend, CompositeBackend -from langchain_core.messages import HumanMessage +from langchain.agents import AgentExecutor, create_openai_tools_agent +from langchain.agents import Tool +from langchain_community.utilities import RetrievalQA +from langchain_community.vectorstores import Chroma as ChromaStore -# --------------------------- -# 1. Embeddings & Chroma setup -# --------------------------- -# Using OllamaEmbeddings with nomic-embed-text as required by the "Исправить" section. -# The embeddings are used for both loading the FAQ and for the search tool. -embeddings = OllamaEmbeddings(model="nomic-embed-text") +# ---------- Configuration ---------- +OPENAI_API_KEY = os.getenv("OPENAI_API_KEY") +if not OPENAI_API_KEY: + raise RuntimeError("OPENAI_API_KEY not set in environment") -# Persistent Chroma collection for the FAQ knowledge base. -vector_store = Chroma( - collection_name="faq_collection", - embedding_function=embeddings, - persist_directory="./chroma_faq" +# ---------- Embeddings & Vector Store ---------- +embeddings = OpenAIEmbeddings( + model="text-embedding-3-small", + base_url="https://openrouter.ai/api/v1", + api_key=OPENAI_API_KEY, ) -# --------------------------- -# 2. Load FAQ markdown files into Chroma -# --------------------------- +vector_store = ChromaStore( + collection_name="faq_collection", + embedding_function=embeddings, + persist_directory="./chroma_faq", +) + +# ---------- Load FAQ into Chroma ---------- def load_faq_to_chroma(data_dir: str = "data"): - """Load all .md files from *data_dir* into the persistent Chroma collection. - Each file is split into documents with a simple line‑based splitter. + """Load all .md files from data_dir into the Chroma vector store. + The function clears the existing collection before loading. """ - data_path = Path(data_dir) - if not data_path.exists(): - raise FileNotFoundError(f"Data directory {data_dir} not found") + vector_store.delete_collection() docs = [] - for md_file in data_path.glob("*.md"): + for md_file in Path(data_dir).glob("*.md"): text = md_file.read_text(encoding="utf-8") - # Simple split by double newlines to create chunks - for i, chunk in enumerate(text.split("\n\n")): - docs.append(Document(page_content=chunk, metadata={"source": md_file.name, "chunk": i})) + docs.append(Document(page_content=text, metadata={"source": md_file.name})) vector_store.add_documents(docs) vector_store.persist() -# --------------------------- -# 3. Tools -# --------------------------- +# ---------- Tools ---------- @tool def search_course_docs(query: str, k: int = 3) -> str: - """Search the FAQ knowledge base for relevant information.""" + """Search the local FAQ collection for relevant passages.""" docs = vector_store.similarity_search(query, k=k) if not docs: - return "No relevant information found in the FAQ." - return "\n\n---\n\n".join(f"**{doc.metadata.get('source')}** (chunk {doc.metadata.get('chunk')}):\n{doc.page_content}" for doc in docs) + return "No relevant information found in the course materials." + return "\n\n---\n\n".join([f"{doc.metadata.get('source', 'unknown')}\n{doc.page_content}" for doc in docs]) @tool def fetch_course_meta(query: str) -> str: - """Mock MCP‑style tool that returns course metadata. - In production this would perform an HTTP GET to an MCP server. - Here we simply return a static JSON string based on the query. + """MCP‑style tool that queries a local mock server for course metadata. + The mock server should serve a JSON file at http://localhost:8000/meta.json. """ - # Simple static mapping for demo purposes - meta = { - "schedule": "Monday 10:00-12:00, Wednesday 14:00-16:00", - "instructor": "Dr. Ivanov", - "credits": "3" - } - key = query.lower().strip() - return meta.get(key, f"No metadata found for '{query}'.") + url = "http://localhost:8000/meta.json" + try: + response = httpx.get(url, timeout=5.0) + response.raise_for_status() + data = response.json() + except Exception as e: + return f"Error fetching metadata: {e}" + # Simple lookup: return value if query matches a key (case‑insensitive) + key = query.strip().lower() + value = data.get(key) + if value is None: + return f"No metadata entry found for '{query}'." + return f"{key}: {value}" -# --------------------------- -# 4. Agent setup with deepagents -# --------------------------- -# LLM via OpenRouter as per course requirement +# ---------- Agent Setup ---------- llm = ChatOpenAI( model="openai/gpt-oss-20b:free", base_url="https://openrouter.ai/api/v1", - api_key=os.getenv("OPENAI_API_KEY"), + api_key=OPENAI_API_KEY, temperature=0.0, ) -backend = CompositeBackend([ - LocalShellBackend(workspace_dir="./workspace"), - FilesystemBackend(), -]) - -# System prompt instructs the agent to choose the appropriate tool and to label the source. +# Define the system prompt with routing rule system_prompt = ( - "You are a helpful FAQ assistant.\n" - "When a user asks a question about course materials, use the tool `search_course_docs`.\n" - "When a user asks about schedule, instructor, or credits, use the tool `fetch_course_meta`.\n" - "Do not call both tools unless necessary.\n" - "In your final answer, prepend the source label: `source: chroma` or `source: mcp_meta`." + "You are a helpful assistant that answers questions about the course. " + "If the question is about course content, use the search_course_docs tool. " + "If the question is about schedule, metadata, or other non‑content info, " + "use the fetch_course_meta tool. Do not call both tools unless the question " + "explicitly requires both. In your final answer, prepend 'source: chroma' " + "or 'source: mcp_meta' to indicate which tool provided the information." ) -agent = create_deep_agent( - model=llm, - tools=[search_course_docs, fetch_course_meta], - backend=backend, - system_prompt=system_prompt, -) +# Create Tool objects +search_tool = Tool(name="search_course_docs", func=search_course_docs, description="Search local course documents.") +meta_tool = Tool(name="fetch_course_meta", func=fetch_course_meta, description="Fetch course metadata from MCP mock server.") -# --------------------------- -# 5. CLI -# --------------------------- -PRESET_QUESTIONS = [ - "What topics are covered in the first lecture?", - "Who is the instructor for this course?", - "When is the next class?" -] +# Build the agent executor +agent = create_openai_tools_agent(llm=llm, tools=[search_tool, meta_tool], system_message=system_prompt) +agent_executor = AgentExecutor.from_agent_and_tools(agent=agent, tools=[search_tool, meta_tool], verbose=True) -async def run_agent(question: str, thread_id: str = "session-1"): - result = await agent.ainvoke( - {"messages": [HumanMessage(content=question)]}, - {"configurable": {"thread_id": thread_id}}, - ) - # The last message is the agent's response - return result["messages"][-1].content +# ---------- CLI ---------- +async def run_cli(): + # Preload data if not already present + if not Path("./chroma_faq").exists(): + load_faq_to_chroma() -async def main(): - # Ensure FAQ is loaded - load_faq_to_chroma() + # Predefined questions + predefined = [ + "What is the main topic of the first lecture?", + "Explain the concept of polymorphism in the course.", + "What is the schedule for the next week?", + ] + print("--- Predefined questions ---") + for i, q in enumerate(predefined, 1): + print(f"{i}. {q}") + print("\nEnter a number to ask a predefined question or type your own query.") + user_input = input("> ") + if user_input.isdigit() and 1 <= int(user_input) <= len(predefined): + query = predefined[int(user_input)-1] + else: + query = user_input - print("--- FAQ Bot Demo ---\n") - for i, q in enumerate(PRESET_QUESTIONS, 1): - print(f"Q{i}: {q}") - answer = await run_agent(q, thread_id=f"demo-{i}") - print(f"A{i}: {answer}\n") - - # Interactive mode - print("Enter your own questions (type 'exit' to quit):") - while True: - user_input = input("> ") - if user_input.lower() in {"exit", "quit"}: - break - answer = await run_agent(user_input, thread_id="interactive") - print(answer) + result = await agent_executor.ainvoke({"input": query}) + print("\n--- Answer ---") + print(result["output"]) if __name__ == "__main__": - asyncio.run(main()) + asyncio.run(run_cli())