From df181b410ae38bfd0dc018823ff9530432fbc1a9 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?=D0=90=D1=80=D1=82=D0=B5=D0=BC=20=D0=92=D0=BB=D0=B0=D0=B4?= =?UTF-8?q?=D0=B8=D0=BC=D0=B8=D1=80=D0=BE=D0=B2=D0=B8=D1=87=20=D0=91=D0=B0?= =?UTF-8?q?=D0=B1=D0=B0=D0=B9=D0=BA=D0=B8=D0=BD?= Date: Thu, 28 May 2026 16:00:44 +0000 Subject: [PATCH] feat: solution for 6a02e23da6fe2e4ac16acf65 --- .../6a02e23da6fe2e4ac16acf65/solution.py | 162 +++++++++--------- 1 file changed, 79 insertions(+), 83 deletions(-) diff --git a/solutions/6a02e23da6fe2e4ac16acf65/solution.py b/solutions/6a02e23da6fe2e4ac16acf65/solution.py index f291fe7..988b5ec 100644 --- a/solutions/6a02e23da6fe2e4ac16acf65/solution.py +++ b/solutions/6a02e23da6fe2e4ac16acf65/solution.py @@ -1,61 +1,62 @@ -from pathlib import Path +from langchain_openai import ChatOpenAI +from pydantic import SecretStr +from langchain.tools import tool +from langchain_qdrant import QdrantVectorStore +from qdrant_client import QdrantClient +from qdrant_client.http.models import Distance, VectorParams +from langchain_ollama import OllamaEmbeddings +from langchain_core.documents import Document +from langchain_text_splitters import RecursiveCharacterTextSplitter +from langchain.agents import create_agent +import os -# ---------- LLM & Embeddings ---------- -from langchain_ollama import Ollama, OllamaEmbeddings - -llm = Ollama( - model="openai/gpt-oss-20b", # e.g. "llama3" - base_url="http://localhost:11434", +# ---------- LLM ---------- +llm = ChatOpenAI( + model="openai/gpt-oss-20b", + base_url='https://platform.brojs.ru/jrnl-bh/api/inference/v1', + api_key=SecretStr("jrnl_30283ab953615cbb6846ff9940a1eedce0b76d7b2f59a2394f29e74643e6a90d"), temperature=0.7, ) +# ---------- Embeddings ---------- embeddings = OllamaEmbeddings(model="nomic-embed-text") -# ---------- Qdrant Vector Store ---------- -from qdrant_client import QdrantClient -from qdrant_client.http.models import Distance, VectorParams -from langchain_qdrant import QdrantVectorStore +# ---------- Qdrant ---------- +client = QdrantClient(":memory:") +collection_name = "knowledge_base" -client = QdrantClient(":memory:") # in‑memory for demo; replace with path or URL as needed -# Determine vector size from the embedding model -sample_vector = embeddings.embed_query("test")[0] -vector_size = len(sample_vector) - -client.create_collection( - collection_name="knowledge_base", - vectors_config=VectorParams(size=vector_size, distance=Distance.COSINE), -) +try: + client.get_collection(collection_name) +except Exception: + # Use a typical embedding size for nomic-embed-text (768) + client.create_collection( + collection_name=collection_name, + vectors_config=VectorParams(size=768, distance=Distance.COSINE), + ) vector_store = QdrantVectorStore( client=client, - collection_name="knowledge_base", + collection_name=collection_name, embedding=embeddings, ) -# ---------- Text Splitter ---------- -from langchain_text_splitters import RecursiveCharacterTextSplitter - +# ---------- Text splitter ---------- splitter = RecursiveCharacterTextSplitter(chunk_size=500, chunk_overlap=50) # ---------- Tools ---------- -from langchain.tools import tool -from langchain_core.documents import Document - -@tool("search_knowledge_base") +@tool def search_knowledge_base(query: str, max_results: int = 5) -> str: """Search the knowledge base for relevant documents.""" - docs = vector_store.similarity_search_with_score(query, k=max_results) - if not docs: + docs_with_score = vector_store.similarity_search_with_score(query, k=max_results) + if not docs_with_score: return "No results found." - response_lines = [] - for i, (doc, score) in enumerate(docs, start=1): - title = doc.metadata.get("title", "Untitled") - snippet = doc.page_content[:200] + ("..." if len(doc.page_content) > 200 else "") - response_lines.append(f"{i}. [{score:.2f}] {title}\n{snippet}") - return "\n\n".join(response_lines) + return "\n".join( + f"{i+1}. {doc.page_content[:200]}..." + for i, (doc, _) in enumerate(docs_with_score) + ) -@tool("add_to_knowledge_base") -def add_to_knowledge_base(content: str, title: str = "Untitled") -> str: +@tool +def add_to_knowledge_base(content: str, title: str = "") -> str: """Add a new document to the knowledge base.""" chunks = splitter.split_text(content) docs = [Document(page_content=c, metadata={"title": title}) for c in chunks] @@ -63,64 +64,59 @@ def add_to_knowledge_base(content: str, title: str = "Untitled") -> str: return f"Added {len(chunks)} chunks under title '{title}'." # ---------- Agent ---------- -from langchain.agents import create_agent -from langchain_core.messages import HumanMessage +system_prompt = """ +You are an assistant that can search and add information to a knowledge base. +Use the tools `search_knowledge_base` and `add_to_knowledge_base` as needed. +""" agent = create_agent( model=llm, tools=[search_knowledge_base, add_to_knowledge_base], - system_message="You are a helpful assistant that can search and update the knowledge base.", + system_prompt=system_prompt, ) -# ---------- Document Loader ---------- -def load_documents_from_dir(directory: str): - """Load all .txt files from directory into vector store.""" - for file_path in Path(directory).glob("*.txt"): - text = file_path.read_text(encoding="utf-8") - add_to_knowledge_base(text, title=file_path.stem) - # ---------- CLI ---------- +def load_directory(path: str): + """Load all text files from a directory into the knowledge base.""" + for root, _, files in os.walk(path): + for file in files: + if file.lower().endswith(".txt"): + with open(os.path.join(root, file), encoding="utf-8") as f: + content = f.read() + add_to_knowledge_base(content=content, title=file) + def main(): - print( - "Welcome to the RAG Agent. Commands:\n" - "/add <file>\n" - "/search <query>\n" - "/quit\n" - ) + print("Welcome to the RAG agent. Commands: /add <file>, /search <query>, /load <dir>, /quit") while True: - user_input = input("> ").strip() - if not user_input: - continue - if user_input.lower() in ("/quit", "exit"): + try: + inp = input("> ").strip() + except EOFError: break - - if user_input.startswith("/add"): + if not inp: + continue + if inp.lower() in ("/quit", "exit"): + print("Goodbye!") + break + if inp.startswith("/add "): + _, file_path = inp.split(maxsplit=1) try: - _, title, file_path = user_input.split(maxsplit=2) - content = Path(file_path).read_text(encoding="utf-8") - print(add_to_knowledge_base(content, title)) + with open(file_path, encoding="utf-8") as f: + content = f.read() + print(add_to_knowledge_base(content=content, title=os.path.basename(file_path))) except Exception as e: - print(f"Error adding document: {e}") - - elif user_input.startswith("/search"): - query = user_input[len("/search") :].strip() - if not query: - print("Please provide a search query.") - continue - result = agent.invoke({"messages": [HumanMessage(content=query)]}) - for msg in result["messages"]: - if hasattr(msg, "content"): - print(msg.content) - + print(f"Error adding file: {e}") + elif inp.startswith("/search "): + _, query = inp.split(maxsplit=1) + print(search_knowledge_base(query=query)) + elif inp.startswith("/load "): + _, dir_path = inp.split(maxsplit=1) + load_directory(dir_path) + print(f"Loaded documents from {dir_path}") else: - # Treat as normal chat message - result = agent.invoke({"messages": [HumanMessage(content=user_input)]}) - for msg in result["messages"]: - if hasattr(msg, "content"): - print(msg.content) + # Regular conversation + response = agent.invoke({"messages": [{"role": "human", "content": inp}]}) + msg = response["messages"][-1] + print(msg.content) if __name__ == "__main__": - # Optional: load initial docs from a folder named 'docs' - if Path("docs").exists(): - load_documents_from_dir("docs") main() \ No newline at end of file