From 98119268ed1df3599ec8778e1c6f644fdbff42bb Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?=D0=90=D1=80=D1=82=D1=83=D1=80=20=D0=9A=D1=83=D0=B7=D0=B0?= =?UTF-8?q?=D1=85=D0=BC=D0=B5=D1=82=D0=BE=D0=B2?= Date: Wed, 1 Jul 2026 14:06:25 +0000 Subject: [PATCH] Delete directory 'src' --- src/__init__.py | 1 - src/agent.js | 63 ------------ src/agent.py | 46 --------- src/bot.js | 23 ----- src/bot.py | 103 ------------------- src/chromadb_client.py | 115 --------------------- src/cli.py | 55 ---------- src/config.py | 22 ---- src/database.py | 100 ------------------ src/embedding.py | 28 ------ src/embeddings.py | 14 --- src/faq_bot.py | 82 --------------- src/index.js | 43 -------- src/index.py | 223 ----------------------------------------- src/ingest.js | 55 ---------- src/ingest.py | 72 ------------- src/main.py | 153 ---------------------------- src/mcp_tool.py | 24 ----- src/mcp_tools.py | 33 ------ src/middleware.js | 23 ----- src/retriever.py | 42 -------- src/tools.py | 85 ---------------- src/vector_store.py | 96 ------------------ src/vectorstore.js | 71 ------------- 24 files changed, 1572 deletions(-) delete mode 100644 src/__init__.py delete mode 100644 src/agent.js delete mode 100644 src/agent.py delete mode 100644 src/bot.js delete mode 100644 src/bot.py delete mode 100644 src/chromadb_client.py delete mode 100644 src/cli.py delete mode 100644 src/config.py delete mode 100644 src/database.py delete mode 100644 src/embedding.py delete mode 100644 src/embeddings.py delete mode 100644 src/faq_bot.py delete mode 100644 src/index.js delete mode 100644 src/index.py delete mode 100644 src/ingest.js delete mode 100644 src/ingest.py delete mode 100644 src/main.py delete mode 100644 src/mcp_tool.py delete mode 100644 src/mcp_tools.py delete mode 100644 src/middleware.js delete mode 100644 src/retriever.py delete mode 100644 src/tools.py delete mode 100644 src/vector_store.py delete mode 100644 src/vectorstore.js diff --git a/src/__init__.py b/src/__init__.py deleted file mode 100644 index dc494e4..0000000 --- a/src/__init__.py +++ /dev/null @@ -1 +0,0 @@ -# Package initialization \ No newline at end of file diff --git a/src/agent.js b/src/agent.js deleted file mode 100644 index 64dd268..0000000 --- a/src/agent.js +++ /dev/null @@ -1,63 +0,0 @@ -import { initializeAgentExecutorWithOptions } from "langchain/agents"; -import { OpenAI } from "langchain/llms/openai"; -import { RetrievalQAChain } from "langchain/chains"; -import { RetrievalQA } from "langchain/chains/retrieval_qa"; -import { OpenAIEmbeddings } from "langchain/embeddings/openai"; -import { ChromaClient } from "chromadb"; - -const llm = new OpenAI({ - temperature: 0, - openAIApiKey: process.env.OPENAI_API_KEY, -}); - -const client = new ChromaClient({ - path: process.env.CHROMA_DB_PATH || "./chromadb", -}); - -export async function createAgent(collectionName) { - const collection = await client.getOrCreateCollection({ - name: collectionName, - }); - - const retriever = { - async getRelevantDocuments(query) { - const embedding = await new OpenAIEmbeddings({ - openAIApiKey: process.env.OPENAI_API_KEY, - }).embedQuery(query); - const results = await collection.query({ - queryEmbeddings: [embedding], - nResults: 5, - }); - return results.documents[0].map((doc, idx) => ({ - pageContent: doc, - metadata: results.metadatas[0][idx], - })); - }, - }; - - const qaChain = RetrievalQAChain.fromLLM(llm, retriever, { - returnSourceDocuments: true, - }); - - const agent = await initializeAgentExecutorWithOptions( - [], - llm, - { - agentType: "chat-conversational-react-description", - memory: undefined, - verbose: true, - tools: [ - { - name: "retrieval", - func: async (input) => { - const docs = await qaChain.call({ input }); - return docs.output; - }, - description: "Use this tool to retrieve answers from the knowledge base", - }, - ], - } - ); - - return agent; -} \ No newline at end of file diff --git a/src/agent.py b/src/agent.py deleted file mode 100644 index 4376e79..0000000 --- a/src/agent.py +++ /dev/null @@ -1,46 +0,0 @@ -import os -from typing import List, Dict, Any - -from langchain_community.llms import Ollama -from langchain_core.prompts import ChatPromptTemplate, MessagesPlaceholder -from langchain_core.runnables import RunnablePassthrough -from langchain_core.tools import BaseTool -from langchain.agents import AgentExecutor, create_openai_tools_agent -from langchain.schema import HumanMessage, SystemMessage - -from .tools import search_course_docs, fetch_course_meta - -# Load tools -TOOLS: List[BaseTool] = [search_course_docs, fetch_course_meta] - -# System prompt guiding the agent -SYSTEM_PROMPT = """ -You are a helpful assistant for a machine learning course. Your job is to answer user questions. - -- If the question is about course materials, lecture slides, assignments, or any content that can be found in the FAQ documents, use the tool `search_course_docs`. -- If the question is about course schedule, instructor information, or other metadata, use the tool `fetch_course_meta`. -- Do not use both tools unless absolutely necessary. -- In your answer, always include a source tag: `source: chroma` if you used the FAQ tool, or `source: mcp_meta` if you used the metadata tool. -""" - -def build_agent() -> AgentExecutor: - """ - Build and return a LangChain AgentExecutor with the defined tools and system prompt. - """ - llm = Ollama(model="llama3", temperature=0.0) - - # Prompt template - prompt = ChatPromptTemplate.from_messages( - [ - SystemMessage(content=SYSTEM_PROMPT), - MessagesPlaceholder(variable_name="history"), - HumanMessage(content="{input}"), - ] - ) - - # Create the agent - agent = create_openai_tools_agent(llm=llm, tools=TOOLS, prompt=prompt) - - # Wrap with AgentExecutor - agent_executor = AgentExecutor(agent=agent, tools=TOOLS, verbose=True, handle_parsing_errors=True) - return agent_executor \ No newline at end of file diff --git a/src/bot.js b/src/bot.js deleted file mode 100644 index ef6e02d..0000000 --- a/src/bot.js +++ /dev/null @@ -1,23 +0,0 @@ -/** - * Minimal Context‑Aware Prompt (MCP) tool. - * Generates a prompt that can be used for vector search. - */ -export function generatePrompt(question) { - return `Answer the following question based on the knowledge base: "${question}"`; -} - -/** - * Handles a user query by generating a prompt, searching the vector store, - * and returning the best answer. - * @param {string} question - * @param {ChromaVectorStore} vectorStore - * @returns {Promise} - */ -export async function answerQuestion(question, vectorStore) { - const prompt = generatePrompt(question); - const results = await vectorStore.similaritySearch(prompt, 1); - if (results.length === 0) { - return "I couldn't find an answer to that question."; - } - return results[0]; -} \ No newline at end of file diff --git a/src/bot.py b/src/bot.py deleted file mode 100644 index d655252..0000000 --- a/src/bot.py +++ /dev/null @@ -1,103 +0,0 @@ -""" -FAQBot implementation. - -The bot uses LangChain's RetrievalQA chain with a ChromaDB vector store -and an LLM (OpenAI or a dummy fallback). It exposes a single method -`ask(question: str) -> str` that returns the best answer from the FAQ. -""" - -import os -from typing import Optional - -import chromadb -from langchain.chains import RetrievalQA -from langchain.llms import OpenAI -from langchain.vectorstores import Chroma - -# Import the vector store helper -from .vector_store import get_vector_store, FAQ_DATA - - -class DummyLLM: - """ - A minimal LLM that simply echoes the prompt. - Used when no OpenAI API key is available. - """ - - def __call__(self, prompt: str) -> str: - return prompt - - -class DummyEmbedding: - """ - Dummy embedding function that returns a fixed vector of zeros. - This avoids the need for an external embedding service during runtime. - """ - - def __call__(self, texts): - return [[0.0] * 768 for _ in texts] - - -class FAQBot: - """ - FAQ Bot that answers user questions based on a predefined FAQ dataset. - """ - - def __init__(self, persist_dir: str, openai_api_key: Optional[str] = None): - """ - Initialize the bot. - - Parameters - ---------- - persist_dir : str - Directory where the ChromaDB data is persisted. - openai_api_key : Optional[str] - OpenAI API key. If None, a DummyLLM is used. - """ - self.persist_dir = persist_dir - self.openai_api_key = openai_api_key - - # Load or create the vector store - collection = get_vector_store(persist_dir) - - # Wrap the collection with LangChain's Chroma wrapper - self.vectorstore = Chroma( - collection=collection, - embedding_function=DummyEmbedding(), - ) - - # Choose LLM - if openai_api_key: - self.llm = OpenAI(temperature=0, openai_api_key=openai_api_key) - else: - self.llm = DummyLLM() - - # Build RetrievalQA chain - self.chain = RetrievalQA.from_chain_type( - llm=self.llm, - chain_type="stuff", - retriever=self.vectorstore.as_retriever(), - return_source_documents=False, - ) - - def ask(self, question: str) -> str: - """ - Ask the bot a question. - - Parameters - ---------- - question : str - The user question. - - Returns - ------- - str - The bot's answer. - """ - try: - response = self.chain.run(question) - if not response: - return "I don't have an answer for that." - return response.strip() - except Exception as exc: - return f"Error processing your question: {exc}" \ No newline at end of file diff --git a/src/chromadb_client.py b/src/chromadb_client.py deleted file mode 100644 index 2dab15a..0000000 --- a/src/chromadb_client.py +++ /dev/null @@ -1,115 +0,0 @@ -""" -Chromadb client wrapper for storing and querying FAQ documents. -""" - -import os -from typing import List, Dict, Any - -import chromadb -from chromadb.config import Settings -from chromadb.utils import embedding_functions - -import openai -from dotenv import load_dotenv - -load_dotenv() - -# Ensure OpenAI API key is set -OPENAI_API_KEY = os.getenv("OPENAI_API_KEY") -if not OPENAI_API_KEY: - raise RuntimeError("OPENAI_API_KEY not set in environment") - -openai.api_key = OPENAI_API_KEY - - -class ChromadbClient: - """ - A simple wrapper around ChromaDB for storing FAQ documents and performing similarity searches. - """ - - def __init__(self, collection_name: str = "faq_collection", persist_directory: str = "chromadb"): - """ - Initialize the ChromaDB client and collection. - - :param collection_name: Name of the collection to use. - :param persist_directory: Directory to persist the database. - """ - self.client = chromadb.Client(Settings( - chroma_db_impl="duckdb+parquet", - persist_directory=persist_directory, - )) - self.collection_name = collection_name - self.collection = self.client.get_or_create_collection(name=collection_name) - - def _embed_text(self, text: str) -> List[float]: - """ - Generate embeddings for a given text using OpenAI embeddings. - - :param text: Text to embed. - :return: List of floats representing the embedding. - """ - response = openai.Embedding.create( - input=text, - model="text-embedding-ada-002", - ) - return response["data"][0]["embedding"] - - def add_documents(self, documents: List[Dict[str, Any]]) -> None: - """ - Add a list of documents to the collection. - - Each document should be a dict with keys: - - id: unique identifier - - text: the content of the document - - metadata: optional dict of metadata - - :param documents: List of document dicts. - """ - ids = [] - embeddings = [] - metadatas = [] - texts = [] - - for doc in documents: - doc_id = str(doc["id"]) - text = doc["text"] - metadata = doc.get("metadata", {}) - - ids.append(doc_id) - embeddings.append(self._embed_text(text)) - metadatas.append(metadata) - texts.append(text) - - self.collection.add( - ids=ids, - embeddings=embeddings, - metadatas=metadatas, - documents=texts, - ) - - def query(self, query_text: str, top_k: int = 3) -> List[Dict[str, Any]]: - """ - Query the collection for the most similar documents to the query_text. - - :param query_text: The query string. - :param top_k: Number of top results to return. - :return: List of dicts containing id, score, metadata, and document text. - """ - query_embedding = self._embed_text(query_text) - results = self.collection.query( - query_embeddings=[query_embedding], - n_results=top_k, - include=["documents", "metadatas", "distances"], - ) - - # ChromaDB returns lists; we flatten them - hits = [] - for i in range(len(results["ids"][0])): - hit = { - "id": results["ids"][0][i], - "score": 1 - results["distances"][0][i], # convert distance to similarity - "metadata": results["metadatas"][0][i], - "document": results["documents"][0][i], - } - hits.append(hit) - return hits \ No newline at end of file diff --git a/src/cli.py b/src/cli.py deleted file mode 100644 index 4536c6e..0000000 --- a/src/cli.py +++ /dev/null @@ -1,55 +0,0 @@ -import argparse -import sys - -from .agent import build_agent - -PRESET_QUESTIONS = [ - { - "question": "What is the deadline for Assignment 1?", - "description": "Should use FAQ tool", - }, - { - "question": "How many lectures are there in the course?", - "description": "Should use FAQ tool", - }, - { - "question": "What is the course schedule for next week?", - "description": "Should use metadata tool", - }, -] - -def run_preset_questions(agent): - print("\nRunning preset questions:\n") - for idx, item in enumerate(PRESET_QUESTIONS, 1): - print(f"Q{idx}: {item['question']}") - response = agent.invoke({"input": item["question"]}) - print(f"A{idx}: {response['output']}\n") - -def interactive_mode(agent): - print("\nEnter your questions (type 'exit' to quit):") - while True: - try: - user_input = input("\n> ") - except (KeyboardInterrupt, EOFError): - print("\nExiting.") - break - if user_input.lower() in {"exit", "quit"}: - print("Goodbye!") - break - response = agent.invoke({"input": user_input}) - print(f"\n{response['output']}") - -def main(): - parser = argparse.ArgumentParser(description="FAQ Bot CLI") - parser.add_argument("--interactive", action="store_true", help="Start interactive mode") - args = parser.parse_args() - - agent = build_agent() - - if args.interactive: - interactive_mode(agent) - else: - run_preset_questions(agent) - -if __name__ == "__main__": - main() \ No newline at end of file diff --git a/src/config.py b/src/config.py deleted file mode 100644 index 9ec1ea5..0000000 --- a/src/config.py +++ /dev/null @@ -1,22 +0,0 @@ -import os -from pydantic import BaseSettings - -class Settings(BaseSettings): - # ChromaDB configuration - chroma_db_path: str = "./chroma_db" - chroma_collection_name: str = "faq_collection" - - # Ollama embedding configuration - ollama_embed_model: str = "all-MiniLM-L6-v2" - ollama_host: str = "http://localhost" - ollama_port: int = 11434 - - # OpenAI LLM configuration - openai_api_key: str = "" - openai_model: str = "gpt-3.5-turbo" - - class Config: - env_file = ".env" - env_file_encoding = "utf-8" - -settings = Settings() \ No newline at end of file diff --git a/src/database.py b/src/database.py deleted file mode 100644 index 5d1e7e8..0000000 --- a/src/database.py +++ /dev/null @@ -1,100 +0,0 @@ -""" -ChromaDB wrapper for storing and querying FAQ documents. -""" - -import chromadb -from chromadb.config import Settings -from typing import List, Dict, Any - -class ChromaDB: - """ - Wrapper around ChromaDB to handle FAQ documents. - """ - - def __init__(self, persist_path: str = "chromadb"): - """ - Initialize the ChromaDB client and collection. - - Parameters - ---------- - persist_path : str, optional - Directory to persist the database. Defaults to "chromadb". - """ - self.client = chromadb.Client(Settings(persist_directory=persist_path)) - self.collection_name = "faq" - self.collection = self.client.get_or_create_collection(name=self.collection_name) - - def add_document(self, text: str, embedding: List[float], doc_id: str = None): - """ - Add a single document to the collection. - - Parameters - ---------- - text : str - The document text. - embedding : List[float] - The embedding vector for the document. - doc_id : str, optional - Optional document ID. If None, an auto-generated ID is used. - """ - if doc_id is None: - # Generate a simple incremental ID - existing_ids = self.collection.get()["ids"] - doc_id = str(len(existing_ids)) - self.collection.add( - documents=[text], - embeddings=[embedding], - ids=[doc_id] - ) - - def query(self, embedding: List[float], k: int = 5) -> List[Dict[str, Any]]: - """ - Query the collection for the top-k most similar documents. - - Parameters - ---------- - embedding : List[float] - The query embedding. - k : int, optional - Number of results to return. Defaults to 5. - - Returns - ------- - List[Dict[str, Any]] - List of dictionaries containing 'id', 'document', and 'distance'. - """ - results = self.collection.query( - query_embeddings=[embedding], - n_results=k, - include=["documents", "distances", "ids"] - ) - docs = [] - for doc, dist, doc_id in zip(results["documents"][0], results["distances"][0], results["ids"][0]): - docs.append({"id": doc_id, "document": doc, "distance": dist}) - return docs - - def is_empty(self) -> bool: - """ - Check if the collection has any documents. - - Returns - ------- - bool - True if empty, False otherwise. - """ - return len(self.collection.get()["ids"]) == 0 - - def load_sample_data(self, sample_data: List[Dict[str, str]]): - """ - Load a list of sample documents into the collection. - - Parameters - ---------- - sample_data : List[Dict[str, str]] - List of dictionaries with keys 'text' and optional 'id'. - """ - for item in sample_data: - text = item["text"] - doc_id = item.get("id") - embedding = embed_text(text) - self.add_document(text, embedding, doc_id=doc_id) \ No newline at end of file diff --git a/src/embedding.py b/src/embedding.py deleted file mode 100644 index dc23f5c..0000000 --- a/src/embedding.py +++ /dev/null @@ -1,28 +0,0 @@ -""" -Embedding utilities using Ollama's embed-text model. -""" - -import ollama -from typing import List - -def embed_text(text: str, model: str = "embed-text") -> List[float]: - """ - Generate an embedding for the given text using Ollama's embed-text model. - - Parameters - ---------- - text : str - The input text to embed. - model : str, optional - The Ollama model name. Defaults to "embed-text". - - Returns - ------- - List[float] - The embedding vector. - """ - try: - result = ollama.embeddings(model=model, prompt=text) - return result["embedding"] - except Exception as e: - raise RuntimeError(f"Failed to embed text: {e}") from e \ No newline at end of file diff --git a/src/embeddings.py b/src/embeddings.py deleted file mode 100644 index b148b8a..0000000 --- a/src/embeddings.py +++ /dev/null @@ -1,14 +0,0 @@ -from langchain_ollama import OllamaEmbeddings -from src.config import settings - -# Instantiate the Ollama embeddings once for reuse -ollama_embeddings = OllamaEmbeddings( - model=settings.ollama_embed_model, - base_url=f"{settings.ollama_host}:{settings.ollama_port}" -) - -def get_embedding(text: str): - """ - Return the embedding vector for a single text string. - """ - return ollama_embeddings.embed_query(text) \ No newline at end of file diff --git a/src/faq_bot.py b/src/faq_bot.py deleted file mode 100644 index 8bc5f13..0000000 --- a/src/faq_bot.py +++ /dev/null @@ -1,82 +0,0 @@ -""" -FAQ Bot entry point. - -The bot loads FAQ documents from the `data/` directory, stores them in -ChromaDB, and then enters an interactive loop where the user can ask -questions. The bot returns the top 3 most relevant answers. -""" - -import os -import sys -from pathlib import Path - -from src.vector_store import VectorStore - -# --------------------------------------------------------------------------- # -# Helper functions -# --------------------------------------------------------------------------- # -def load_documents(folder: Path) -> list: - """ - Load all .txt files from the given folder as documents. - - Each file becomes a single document with its content as text. - """ - docs = [] - for file in folder.glob("*.txt"): - text = file.read_text(encoding="utf-8") - docs.append({"text": text, "metadata": {"source": file.name}}) - return docs - - -# --------------------------------------------------------------------------- # -# Main logic -# --------------------------------------------------------------------------- # -def main() -> None: - # Load environment variables (e.g. OPENAI_API_KEY) - from dotenv import load_dotenv - - load_dotenv() - - # Resolve data directory relative to the project root - project_root = Path(__file__).resolve().parent.parent - data_dir = project_root / "data" - - # Initialize vector store - store = VectorStore() - - # If the collection is empty, load documents - if store.collection.count() == 0: - print("Loading documents into ChromaDB...") - docs = load_documents(data_dir) - if not docs: - print(f"No .txt files found in {data_dir}. Exiting.") - sys.exit(1) - store.add_documents(docs) - print(f"Added {len(docs)} documents.") - - print("\nFAQ Bot is ready. Type your question (or 'exit' to quit).") - - while True: - try: - query = input("\nQ: ") - except EOFError: - break - - if query.lower() in ("exit", "quit"): - break - - results = store.query(query, top_k=3) - if not results: - print("No answer found.") - continue - - print("\nTop answers:") - for i, res in enumerate(results, 1): - snippet = res["text"][:200].replace("\n", " ") - print(f"{i}. {snippet}... (distance: {res['distance']:.4f})") - - print("\nGoodbye!") - - -if __name__ == "__main__": - main() \ No newline at end of file diff --git a/src/index.js b/src/index.js deleted file mode 100644 index db8e47e..0000000 --- a/src/index.js +++ /dev/null @@ -1,43 +0,0 @@ -import readlineSync from 'readline-sync'; -import ChromaVectorStore from './vectorStore.js'; -import { answerQuestion } from './bot.js'; - -/** - * Sample FAQ dataset. - * In a real application this would be loaded from a file or database. - */ -const faqData = [ - { id: '1', text: 'What is ChromaDB?', metadata: { category: 'database' } }, - { id: '2', text: 'How do I install ChromaDB?', metadata: { category: 'installation' } }, - { id: '3', text: 'What is an MCP-tool?', metadata: { category: 'concept' } }, - { id: '4', text: 'How to use the FAQ bot?', metadata: { category: 'usage' } }, -]; - -/** - * Main entry point. - */ -async function main() { - const vectorStore = new ChromaVectorStore(); - await vectorStore.init('faq'); - - // Load data into the collection if it is empty. - // For simplicity we always add the data; in production you would check existence. - await vectorStore.addDocuments(faqData); - - console.log('FAQ bot is ready. Type your question (or "exit" to quit).'); - - while (true) { - const question = readlineSync.question('> '); - if (question.trim().toLowerCase() === 'exit') { - console.log('Goodbye!'); - break; - } - const answer = await answerQuestion(question, vectorStore); - console.log(`Answer: ${answer}`); - } -} - -main().catch(err => { - console.error('Error:', err); - process.exit(1); -}); \ No newline at end of file diff --git a/src/index.py b/src/index.py deleted file mode 100644 index 68b9133..0000000 --- a/src/index.py +++ /dev/null @@ -1,223 +0,0 @@ -#!/usr/bin/env python3 -""" -FAQ Bot using QDrant as the vector store. - -This script provides: -- Data ingestion from a text file into QDrant. -- Querying the vector store to retrieve relevant FAQ answers. -- A simple CLI interface for ingestion and querying. - -Author: Artur Kuzakhmetov -""" - -import os -import sys -import json -import argparse -from pathlib import Path -from typing import List, Tuple - -import openai -from qdrant_client import QdrantClient -from qdrant_client.http import models as qdrant_models -from qdrant_client.http.models import PointStruct - -# --------------------------------------------------------------------------- # -# Configuration -# --------------------------------------------------------------------------- # - -# Environment variables -OPENAI_API_KEY = os.getenv("OPENAI_API_KEY") -QDRANT_URL = os.getenv("QDRANT_URL", "http://localhost:6333") -QDRANT_API_KEY = os.getenv("QDRANT_API_KEY") # Optional, if QDrant requires auth -QDRANT_COLLECTION = os.getenv("QDRANT_COLLECTION", "faq_collection") - -# OpenAI embedding model -EMBEDDING_MODEL = "text-embedding-ada-002" -EMBEDDING_DIM = 1536 # Dimension of Ada-002 embeddings - -# --------------------------------------------------------------------------- # -# Helper functions -# --------------------------------------------------------------------------- # - -def split_text_into_chunks(text: str, max_chunk_size: int = 500) -> List[str]: - """ - Split a large text into smaller chunks suitable for embedding. - Splits on paragraph boundaries and ensures each chunk is <= max_chunk_size. - """ - paragraphs = [p.strip() for p in text.split("\n\n") if p.strip()] - chunks = [] - current_chunk = "" - for para in paragraphs: - if len(current_chunk) + len(para) + 1 <= max_chunk_size: - current_chunk += (" " if current_chunk else "") + para - else: - if current_chunk: - chunks.append(current_chunk.strip()) - current_chunk = para - if current_chunk: - chunks.append(current_chunk.strip()) - return chunks - -def embed_texts(texts: List[str]) -> List[List[float]]: - """ - Generate embeddings for a list of texts using OpenAI's embedding API. - """ - if not OPENAI_API_KEY: - raise RuntimeError("OPENAI_API_KEY environment variable is not set.") - openai.api_key = OPENAI_API_KEY - embeddings = [] - for text in texts: - response = openai.Embedding.create( - input=text, - model=EMBEDDING_MODEL - ) - embeddings.append(response["data"][0]["embedding"]) - return embeddings - -# --------------------------------------------------------------------------- # -# QDrant Vector Store Wrapper -# --------------------------------------------------------------------------- # - -class QdrantVectorStore: - def __init__(self, url: str = QDRANT_URL, api_key: str = QDRANT_API_KEY, collection_name: str = QDRANT_COLLECTION): - self.client = QdrantClient(url=url, api_key=api_key) - self.collection_name = collection_name - self._ensure_collection() - - def _ensure_collection(self): - """ - Create the collection if it does not exist. - """ - collections = self.client.get_collections() - if self.collection_name not in [c.name for c in collections.collections]: - self.client.create_collection( - collection_name=self.collection_name, - vectors_config=qdrant_models.VectorParams( - size=EMBEDDING_DIM, - distance="Cosine" - ) - ) - - def upsert(self, texts: List[str], embeddings: List[List[float]]): - """ - Upsert a batch of texts and their embeddings into QDrant. - """ - points = [] - for idx, (text, embedding) in enumerate(zip(texts, embeddings)): - point_id = f"{self.collection_name}_{idx}_{hash(text) % 1000000}" - points.append( - PointStruct( - id=point_id, - vector=embedding, - payload={"text": text} - ) - ) - self.client.upsert( - collection_name=self.collection_name, - points=points - ) - - def search(self, query_embedding: List[float], top_k: int = 5) -> List[Tuple[str, float]]: - """ - Search the collection for the most similar vectors to the query embedding. - Returns a list of (text, score) tuples. - """ - search_result = self.client.search( - collection_name=self.collection_name, - query_vector=query_embedding, - limit=top_k, - with_payload=True, - score=True - ) - results = [] - for hit in search_result: - text = hit.payload.get("text", "") - score = hit.score - results.append((text, score)) - return results - - def delete_collection(self): - """ - Delete the entire collection. Use with caution. - """ - self.client.delete_collection(self.collection_name) - -# --------------------------------------------------------------------------- # -# Bot Logic -# --------------------------------------------------------------------------- # - -def ingest_data(file_path: str, vector_store: QdrantVectorStore): - """ - Read a text file, split into chunks, embed, and store in QDrant. - """ - if not Path(file_path).is_file(): - raise FileNotFoundError(f"File not found: {file_path}") - - with open(file_path, "r", encoding="utf-8") as f: - raw_text = f.read() - - chunks = split_text_into_chunks(raw_text) - embeddings = embed_texts(chunks) - vector_store.upsert(chunks, embeddings) - print(f"Ingested {len(chunks)} chunks into collection '{vector_store.collection_name}'.") - -def query_faq(question: str, vector_store: QdrantVectorStore, top_k: int = 5) -> str: - """ - Query the FAQ bot with a question and return a formatted answer. - """ - query_embedding = embed_texts([question])[0] - results = vector_store.search(query_embedding, top_k=top_k) - if not results: - return "Sorry, I couldn't find an answer to your question." - - answer_parts = [] - for idx, (text, score) in enumerate(results, start=1): - answer_parts.append(f"{idx}. (Score: {score:.4f})\n{text}\n") - return "\n".join(answer_parts) - -def get_response(question: str, top_k: int = 5) -> str: - """ - Public API for external tools (e.g., MCP-tool) to get a bot response. - """ - vector_store = QdrantVectorStore() - return query_faq(question, vector_store, top_k=top_k) - -# --------------------------------------------------------------------------- # -# CLI Interface -# --------------------------------------------------------------------------- # - -def main(): - parser = argparse.ArgumentParser(description="FAQ Bot CLI") - subparsers = parser.add_subparsers(dest="command", required=True) - - ingest_parser = subparsers.add_parser("ingest", help="Ingest a text file into QDrant") - ingest_parser.add_argument("file", help="Path to the text file to ingest") - - query_parser = subparsers.add_parser("query", help="Query the FAQ bot") - query_parser.add_argument("question", help="Your question") - query_parser.add_argument("--top_k", type=int, default=5, help="Number of top results to return") - - delete_parser = subparsers.add_parser("delete", help="Delete the QDrant collection (use with caution)") - - args = parser.parse_args() - - vector_store = QdrantVectorStore() - - if args.command == "ingest": - ingest_data(args.file, vector_store) - elif args.command == "query": - answer = query_faq(args.question, vector_store, top_k=args.top_k) - print(answer) - elif args.command == "delete": - confirm = input(f"Are you sure you want to delete collection '{vector_store.collection_name}'? (yes/no): ") - if confirm.lower() == "yes": - vector_store.delete_collection() - print("Collection deleted.") - else: - print("Deletion aborted.") - else: - parser.print_help() - -if __name__ == "__main__": - main() \ No newline at end of file diff --git a/src/ingest.js b/src/ingest.js deleted file mode 100644 index eeea324..0000000 --- a/src/ingest.js +++ /dev/null @@ -1,55 +0,0 @@ -require('dotenv').config(); -const fs = require('fs'); -const path = require('path'); -const { OpenAI } = require('openai'); -const { ChromaClient } = require('chromadb'); - -const openai = new OpenAI({ apiKey: process.env.OPENAI_API_KEY }); -const chroma = new ChromaClient({ path: 'chromadb' }); - -const COLLECTION_NAME = 'faq_collection'; -const FAQ_FILE = path.join(__dirname, '..', 'faq.json'); - -async function ingest() { - try { - const rawData = fs.readFileSync(FAQ_FILE, 'utf-8'); - const faqEntries = JSON.parse(rawData); - - const collection = await chroma.getOrCreateCollection({ - name: COLLECTION_NAME, - metadata: { description: 'FAQ embeddings' } - }); - - const documents = []; - const embeddings = []; - const ids = []; - const metadatas = []; - - for (let i = 0; i < faqEntries.length; i++) { - const { question, answer } = faqEntries[i]; - const embeddingResponse = await openai.embeddings.create({ - model: 'text-embedding-ada-002', - input: question - }); - const embedding = embeddingResponse.data[0].embedding; - - documents.push(question); - embeddings.push(embedding); - ids.push(`faq-${i}`); - metadatas.push({ answer }); - } - - await collection.add({ - documents, - embeddings, - ids, - metadatas - }); - - console.log(`Ingested ${faqEntries.length} FAQ entries into collection '${COLLECTION_NAME}'.`); - } catch (err) { - console.error('Error during ingestion:', err); - } -} - -ingest(); \ No newline at end of file diff --git a/src/ingest.py b/src/ingest.py deleted file mode 100644 index 6ba811a..0000000 --- a/src/ingest.py +++ /dev/null @@ -1,72 +0,0 @@ -""" -Ingestion logic for FAQ documents into ChromaDB. -""" - -import os -from pathlib import Path -from typing import List - -from chromadb import Client -from chromadb.api.types import Documents, EmbeddingFunction -from chromadb.config import Settings - -from langchain.embeddings.openai import OpenAIEmbeddings -from langchain.text_splitter import RecursiveCharacterTextSplitter - -def _load_faq_pairs(file_path: Path) -> List[tuple]: - """ - Load FAQ pairs from a text file. - Expected format: - Q: - A: - Each pair separated by a blank line. - """ - pairs = [] - with file_path.open("r", encoding="utf-8") as f: - content = f.read() - raw_pairs = content.strip().split("\n\n") - for raw in raw_pairs: - lines = raw.strip().splitlines() - if len(lines) < 2: - continue - q_line = lines[0].strip() - a_line = lines[1].strip() - if q_line.lower().startswith("q:") and a_line.lower().startswith("a:"): - question = q_line[2:].strip() - answer = a_line[2:].strip() - pairs.append((question, answer)) - return pairs - -def ingest_faq(file_path: Path, client: Client, collection_name: str): - """ - Ingest FAQ pairs into the specified ChromaDB collection. - """ - pairs = _load_faq_pairs(file_path) - if not pairs: - raise ValueError("No valid FAQ pairs found in the file.") - - # Prepare documents and metadata - documents = [] - metadatas = [] - ids = [] - - for idx, (q, a) in enumerate(pairs): - # Combine question and answer for embedding - doc = f"Q: {q}\nA: {a}" - documents.append(doc) - metadatas.append({"question": q, "answer": a}) - ids.append(str(idx)) - - # Use OpenAI embeddings - embedding = OpenAIEmbeddings() - - # Create or get collection - collection = client.get_or_create_collection(name=collection_name) - - # Add documents to collection - collection.add( - documents=documents, - metadatas=metadatas, - ids=ids, - embedding_function=embedding - ) \ No newline at end of file diff --git a/src/main.py b/src/main.py deleted file mode 100644 index b6ce9a9..0000000 --- a/src/main.py +++ /dev/null @@ -1,153 +0,0 @@ -""" -Main entry point for the FAQ bot using ChromaDB and a single MCP-tool. -""" - -import os -import json -import sys -from typing import List, Dict, Any - -import openai -from dotenv import load_dotenv - -from chromadb_client import ChromadbClient -from mcp_tool import MCPTool - -load_dotenv() - -# Ensure OpenAI API key is set -OPENAI_API_KEY = os.getenv("OPENAI_API_KEY") -if not OPENAI_API_KEY: - print("Error: OPENAI_API_KEY not set in environment.", file=sys.stderr) - sys.exit(1) - -openai.api_key = OPENAI_API_KEY - -# Initialize the ChromaDB client -db_client = ChromadbClient() - -# Load FAQ documents from a local file (JSON lines format) -FAQ_FILE = os.getenv("FAQ_FILE", "data/faq.jsonl") - -def load_faq_documents(file_path: str) -> List[Dict[str, Any]]: - """ - Load FAQ documents from a JSON lines file. - - Each line should be a JSON object with keys: - - id: unique identifier - - text: the content of the FAQ - - metadata: optional dict - """ - docs = [] - if not os.path.exists(file_path): - print(f"FAQ file {file_path} not found. Skipping load.", file=sys.stderr) - return docs - - with open(file_path, "r", encoding="utf-8") as f: - for line in f: - try: - doc = json.loads(line.strip()) - docs.append(doc) - except json.JSONDecodeError: - continue - return docs - -# Load and add documents to the collection if not already present -if not db_client.collection.count(): - print("Loading FAQ documents into ChromaDB...") - faq_docs = load_faq_documents(FAQ_FILE) - if faq_docs: - db_client.add_documents(faq_docs) - print(f"Added {len(faq_docs)} documents.") - else: - print("No FAQ documents loaded.", file=sys.stderr) - -# Instantiate the MCP-tool -mcp_tool = MCPTool() - -# Define the function schema for OpenAI function calling -function_schema = { - "name": mcp_tool.name, - "description": mcp_tool.description, - "parameters": { - "type": "object", - "properties": {}, - "required": [], - }, -} - -def ask_question(question: str) -> str: - """ - Ask a question to the bot. The bot will: - 1. Retrieve relevant FAQ documents from ChromaDB. - 2. Use OpenAI LLM to generate an answer, possibly invoking the MCP-tool. - """ - # Retrieve top 3 relevant documents - hits = db_client.query(question, top_k=3) - - # Build context from hits - context = "\n\n".join([f"Document {hit['id']}:\n{hit['document']}" for hit in hits]) - - # Construct the prompt for the LLM - messages = [ - {"role": "system", "content": "You are an FAQ assistant. Use the provided documents to answer questions."}, - {"role": "user", "content": f"Question: {question}\n\nContext:\n{context}"}, - ] - - # Call OpenAI with function calling enabled - response = openai.ChatCompletion.create( - model="gpt-4o-mini", - messages=messages, - functions=[function_schema], - function_call="auto", - ) - - # Parse the response - reply = response["choices"][0]["message"] - if reply.get("function_call"): - # The model wants to call the MCP-tool - func_name = reply["function_call"]["name"] - if func_name == mcp_tool.name: - # Execute the tool - tool_response = mcp_tool({}) - # Send the tool response back to the model - tool_message = { - "role": "tool", - "name": func_name, - "content": json.dumps(tool_response), - } - # Re-send the conversation with the tool response - messages.append(reply) - messages.append(tool_message) - # Get the final answer - final_response = openai.ChatCompletion.create( - model="gpt-4o-mini", - messages=messages, - ) - return final_response["choices"][0]["message"]["content"] - else: - return f"Unknown function call: {func_name}" - else: - return reply["content"] - -def main(): - print("FAQ Bot (ChromaDB + MCP-tool). Type 'exit' to quit.") - while True: - try: - user_input = input("\nYou: ").strip() - except (EOFError, KeyboardInterrupt): - print("\nGoodbye!") - break - - if user_input.lower() in {"exit", "quit"}: - print("Goodbye!") - break - - if not user_input: - continue - - answer = ask_question(user_input) - print(f"\nBot: {answer}") - -if __name__ == "__main__": - main() \ No newline at end of file diff --git a/src/mcp_tool.py b/src/mcp_tool.py deleted file mode 100644 index 589778a..0000000 --- a/src/mcp_tool.py +++ /dev/null @@ -1,24 +0,0 @@ -""" -A single MCP-tool implementation for the FAQ bot. -""" - -import datetime -from typing import Dict, Any - -class MCPTool: - """ - Example MCP-tool that returns the current UTC datetime. - """ - - name = "get_current_utc_time" - description = "Returns the current UTC datetime in ISO 8601 format." - - def __call__(self, arguments: Dict[str, Any]) -> Dict[str, Any]: - """ - Execute the tool. - - :param arguments: Dictionary of arguments (unused in this simple tool). - :return: Dictionary with the result. - """ - now = datetime.datetime.utcnow().isoformat() + "Z" - return {"current_time": now} \ No newline at end of file diff --git a/src/mcp_tools.py b/src/mcp_tools.py deleted file mode 100644 index 8533b4a..0000000 --- a/src/mcp_tools.py +++ /dev/null @@ -1,33 +0,0 @@ -""" -Integration with MCP-tools for generating answers. -""" - -import mcp_tools -from typing import List - -def generate_answer(context: List[str], question: str) -> str: - """ - Generate an answer using MCP-tools given context and a question. - - Parameters - ---------- - context : List[str] - List of context strings retrieved from the database. - question : str - The user's question. - - Returns - ------- - str - The generated answer. - """ - # Combine context into a single string - context_text = "\n".join(context) - # Construct a prompt for MCP-tools - prompt = f"Question: {question}\nContext:\n{context_text}\nAnswer:" - # Use MCP-tools to generate the answer - try: - response = mcp_tools.generate(prompt=prompt) - return response - except Exception as e: - raise RuntimeError(f"Failed to generate answer with MCP-tools: {e}") from e \ No newline at end of file diff --git a/src/middleware.js b/src/middleware.js deleted file mode 100644 index 3f3372f..0000000 --- a/src/middleware.js +++ /dev/null @@ -1,23 +0,0 @@ -const moderate = require('moderate-censor'); - -/** - * Moderates user input using moderate-censor. - * @param {string} text - * @returns {Promise<{allowed: boolean, reasons: string[]}>} - */ -async function moderateInput(text) { - try { - const result = await moderate.moderate(text); - if (result.isAllowed) { - return { allowed: true, reasons: [] }; - } else { - return { allowed: false, reasons: result.reasons || [] }; - } - } catch (err) { - console.error('Moderation error:', err); - // If moderation fails, default to allowing to avoid blocking legitimate queries - return { allowed: true, reasons: [] }; - } -} - -module.exports = { moderateInput }; \ No newline at end of file diff --git a/src/retriever.py b/src/retriever.py deleted file mode 100644 index b98d393..0000000 --- a/src/retriever.py +++ /dev/null @@ -1,42 +0,0 @@ -""" -Retrieval and answer generation logic using LangChain. -""" - -import os -from typing import Any - -from chromadb import Client -from chromadb.config import Settings - -from langchain.embeddings.openai import OpenAIEmbeddings -from langchain.llms.openai import OpenAIChat -from langchain.chains import RetrievalQA -from langchain.vectorstores import Chroma - -def get_answer(question: str, client: Client, collection_name: str, k: int = 3) -> str: - """ - Retrieve relevant FAQ chunks and generate an answer using OpenAIChat. - """ - # Set up embeddings and LLM - embedding = OpenAIEmbeddings() - llm = OpenAIChat(temperature=0) - - # Load vector store - vectorstore = Chroma( - client=client, - collection_name=collection_name, - embedding_function=embedding - ) - - # Build RetrievalQA chain - qa_chain = RetrievalQA.from_chain_type( - llm=llm, - chain_type="stuff", - retriever=vectorstore.as_retriever(search_kwargs={"k": k}), - return_source_documents=True - ) - - # Run chain - result = qa_chain({"question": question}) - answer = result.get("answer", "") - return answer.strip() \ No newline at end of file diff --git a/src/tools.py b/src/tools.py deleted file mode 100644 index 18c61b0..0000000 --- a/src/tools.py +++ /dev/null @@ -1,85 +0,0 @@ -import json -import os -from pathlib import Path -from typing import List, Dict, Any - -import httpx -from langchain_community.document_loaders import TextLoader -from langchain_community.embeddings import OllamaEmbeddings -from langchain_community.vectorstores import Chroma -from langchain_core.documents import Document -from langchain_core.tools import tool - -# Path to the data directory -DATA_DIR = Path(__file__).parent.parent / "data" -CHROMA_DIR = Path(__file__).parent.parent / "chroma_faq" - -def load_faq_to_chroma() -> Chroma: - """ - Load all .md files from the data directory, chunk them, embed with Ollama, - and persist into a Chroma vector store. - """ - # Check if the Chroma collection already exists - if CHROMA_DIR.exists(): - # Load existing collection - return Chroma(persist_directory=str(CHROMA_DIR), embedding_function=OllamaEmbeddings(model="nomic-embed-text")) - - # Gather all markdown files - md_files = list(DATA_DIR.glob("*.md")) - documents: List[Document] = [] - - for md_file in md_files: - loader = TextLoader(str(md_file), encoding="utf-8") - docs = loader.load() - documents.extend(docs) - - # Create embeddings - embeddings = OllamaEmbeddings(model="nomic-embed-text") - - # Create Chroma vector store - chroma = Chroma.from_documents( - documents=documents, - embedding=embeddings, - persist_directory=str(CHROMA_DIR), - ) - return chroma - -@tool -def search_course_docs(query: str, k: int = 3) -> List[Dict[str, Any]]: - """ - Search the local FAQ Chroma vector store for relevant documents. - - Returns a list of dictionaries containing the content and metadata. - """ - chroma = load_faq_to_chroma() - results = chroma.similarity_search(query, k=k) - output = [] - for doc in results: - output.append( - { - "content": doc.page_content, - "metadata": doc.metadata, - } - ) - return output - -@tool -def fetch_course_meta(query: str) -> Dict[str, Any]: - """ - Simulate an MCP-style HTTP tool that returns course metadata - matching the query. The metadata is read from a local JSON file. - """ - meta_path = DATA_DIR / "course_meta.json" - with open(meta_path, "r", encoding="utf-8") as f: - data = json.load(f) - - # Simple keyword matching in schedule and instructor fields - results = {} - if "schedule" in query.lower(): - results["schedule"] = data.get("schedule", []) - if "instructor" in query.lower() or "professor" in query.lower(): - results["instructor"] = data.get("instructor", {}) - if not results: - # Default to returning the whole metadata if no keyword matched - results = data - return results \ No newline at end of file diff --git a/src/vector_store.py b/src/vector_store.py deleted file mode 100644 index b06205f..0000000 --- a/src/vector_store.py +++ /dev/null @@ -1,96 +0,0 @@ -""" -Vector store abstraction over ChromaDB. - -The `VectorStore` class encapsulates all interactions with the ChromaDB -collection. It uses the MCP-tool to generate embeddings for documents -and queries. -""" - -import chromadb -from chromadb.config import Settings -from typing import List, Dict - -from .mcp_tool import get_embedding - - -class VectorStore: - """ - Wrapper around a ChromaDB collection. - - Parameters - ---------- - collection_name : str, optional - Name of the collection to use. Defaults to "faq". - """ - - def __init__(self, collection_name: str = "faq"): - self.client = chromadb.Client(Settings()) - self.collection = self.client.get_or_create_collection(name=collection_name) - - def add_documents(self, documents: List[Dict[str, str]]) -> None: - """ - Add a list of documents to the collection. - - Each document must contain a 'text' key and may optionally contain - a 'metadata' dictionary. - - Parameters - ---------- - documents : List[Dict[str, str]] - List of documents to add. - """ - ids = [] - texts = [] - embeddings = [] - metadatas = [] - - for i, doc in enumerate(documents): - ids.append(str(i)) - texts.append(doc["text"]) - embeddings.append(get_embedding(doc["text"])) - metadatas.append(doc.get("metadata", {})) - - self.collection.add( - ids=ids, - documents=texts, - embeddings=embeddings, - metadatas=metadatas, - ) - - def query(self, query_text: str, top_k: int = 5) -> List[Dict[str, str]]: - """ - Retrieve the most relevant documents for a query. - - Parameters - ---------- - query_text : str - The query string. - top_k : int, optional - Number of results to return. Defaults to 5. - - Returns - ------- - List[Dict[str, str]] - List of result dictionaries containing 'text', 'distance', - and 'metadata'. - """ - embedding = get_embedding(query_text) - results = self.collection.query( - query_embeddings=[embedding], - n_results=top_k, - ) - - output = [] - for doc, dist, meta in zip( - results["documents"][0], - results["distances"][0], - results["metadatas"][0], - ): - output.append( - { - "text": doc, - "distance": dist, - "metadata": meta, - } - ) - return output \ No newline at end of file diff --git a/src/vectorstore.js b/src/vectorstore.js deleted file mode 100644 index 0b51cf5..0000000 --- a/src/vectorstore.js +++ /dev/null @@ -1,71 +0,0 @@ -import { ChromaClient } from 'chromadb'; - -/** - * Simple embedding utility. - * Produces a 768‑dimensional vector where each dimension is a count of - * the number of words that hash to that index. - */ -function embed(text) { - const vector = new Array(768).fill(0); - const words = text.toLowerCase().split(/\s+/); - for (const word of words) { - const hash = [...word].reduce((acc, ch) => acc + ch.charCodeAt(0), 0); - const idx = hash % 768; - vector[idx] += 1; - } - return vector; -} - -/** - * Wrapper around ChromaDB providing a minimal API for the bot. - */ -class ChromaVectorStore { - constructor() { - this.client = new ChromaClient(); - this.collection = null; - } - - /** - * Initializes the collection. Creates it if it does not exist. - * @param {string} name - Collection name. - */ - async init(name = 'faq') { - this.collection = await this.client.getOrCreateCollection({ - name, - }); - } - - /** - * Adds documents to the collection. - * @param {Array<{id?: string, text: string, metadata?: object}>} docs - */ - async addDocuments(docs) { - const ids = docs.map((d, idx) => d.id ?? `doc-${idx}`); - const metadatas = docs.map(d => d.metadata ?? {}); - const embeddings = docs.map(d => embed(d.text)); - await this.collection.add({ - ids, - documents: docs.map(d => d.text), - metadatas, - embeddings, - }); - } - - /** - * Performs a similarity search. - * @param {string} queryText - * @param {number} k - * @returns {Promise>} Top k documents. - */ - async similaritySearch(queryText, k = 3) { - const queryEmbedding = embed(queryText); - const results = await this.collection.query({ - queryEmbeddings: [queryEmbedding], - nResults: k, - }); - return results[0].documents; - } -} - -export default ChromaVectorStore; -export { embed }; \ No newline at end of file