diff --git a/README.md b/README.md index 00faf56..4bd52db 100644 --- a/README.md +++ b/README.md @@ -1,21 +1,27 @@ -# FAQ Bot with ChromaDB and LangChain +# FAQ Bot with ChromaDB and moderate-censor -This project implements a simple FAQ bot that uses **ChromaDB** for vector storage and **LangChain** for building an intelligent agent. The bot can answer questions based on a small knowledge base stored in ChromaDB. +This project implements an FAQ bot that uses **ChromaDB** for vector storage and retrieval, and **moderate-censor** as the single MCP-tool for content moderation. ## Features -- Stores documents in ChromaDB with embeddings from OpenAI. -- Uses the latest LangChain agent creation method (`initializeAgentExecutorWithOptions`). -- Simple CLI interface for interacting with the bot. -- Easy to extend with more documents or tools. +- Vector-based FAQ retrieval using OpenAI embeddings and ChromaDB. +- User input moderation with moderate-censor. +- Simple HTTP API (`/ask`) to query the bot. + +## Prerequisites + +- Node.js v18+ (or any LTS version) +- npm +- OpenAI API key (set in `.env`) +- ChromaDB server running locally (default path: `chromadb`) ## Setup 1. **Clone the repository** ```bash - git clone https://github.com/your-username/faq-bot-chromadb.git - cd faq-bot-chromadb + git clone + cd ``` 2. **Install dependencies** @@ -24,26 +30,83 @@ This project implements a simple FAQ bot that uses **ChromaDB** for vector stora npm install ``` -3. **Configure environment** - - Create a `.env` file in the project root: +3. **Create a `.env` file** ```env OPENAI_API_KEY=your_openai_api_key - CHROMA_DB_PATH=./chromadb + PORT=3000 ``` -4. **Run the bot** +4. **Prepare FAQ data** + + Create a `faq.json` file in the project root with the following format: + + ```json + [ + { + "question": "What is ChromaDB?", + "answer": "ChromaDB is a vector database for storing and retrieving embeddings." + }, + { + "question": "How do I use the bot?", + "answer": "Send a POST request to /ask with a JSON body containing the 'question' field." + } + ] + ``` + +5. **Ingest FAQ data** + + ```bash + npm run ingest + ``` + + This will read `faq.json`, generate embeddings, and store them in ChromaDB. + +6. **Start the bot** ```bash npm start ``` - You can also use `npm run dev` for automatic restarts with nodemon. + The server will listen on the port specified in `.env` (default 3000). -## Adding Documents +## Usage -The bot comes with two sample FAQ entries. To add more, edit `src/index.js` or use the `addDocument` function from `src/vectorstore.js`. +Send a POST request to `/ask`: + +```bash +curl -X POST http://localhost:3000/ask \ + -H "Content-Type: application/json" \ + -d '{"question":"What is ChromaDB?"}' +``` + +Response: + +```json +{ + "answer": "ChromaDB is a vector database for storing and retrieving embeddings." +} +``` + +If the question contains disallowed content, the bot will respond with a 403 status and reasons. + +## Project Structure + +``` +├── package.json +├── src +│ ├── index.js # HTTP server and bot logic +│ ├── ingest.js # FAQ ingestion script +│ └── middleware.js # Moderation middleware +├── faq.json # FAQ data file +└── README.md +``` + +## Notes + +- The bot uses the `text-embedding-ada-002` model for embeddings. +- Only one MCP-tool (`moderate-censor`) is used as required. +- Ensure the ChromaDB server is running before ingesting data or starting the bot. ## License diff --git a/package.json b/package.json index a0f1d7f..d141dee 100644 --- a/package.json +++ b/package.json @@ -1,21 +1,17 @@ { - "name": "faq-bot-chromadb", + "name": "faq-bot-chromadb-mcp", "version": "1.0.0", - "description": "FAQ bot using ChromaDB and LangChain", + "description": "FAQ bot using ChromaDB for vector storage and moderate-censor as the MCP-tool", "main": "src/index.js", - "type": "module", "scripts": { "start": "node src/index.js", - "dev": "nodemon src/index.js" + "ingest": "node src/ingest.js" }, "dependencies": { "chromadb": "^0.3.0", - "langchain": "^0.2.0", - "langchain-community": "^0.2.0", "dotenv": "^16.4.5", - "openai": "^4.27.0" - }, - "devDependencies": { - "nodemon": "^3.0.1" + "express": "^4.18.2", + "moderate-censor": "^1.0.0", + "openai": "^4.18.0" } } \ No newline at end of file diff --git a/src/index.js b/src/index.js index 053cf4f..5ef6a9d 100644 --- a/src/index.js +++ b/src/index.js @@ -1,50 +1,70 @@ -import dotenv from "dotenv"; -import readline from "readline"; -import { createAgent } from "./agent.js"; -import { addDocument } from "./vectorstore.js"; +require('dotenv').config(); +const express = require('express'); +const { OpenAI } = require('openai'); +const { ChromaClient } = require('chromadb'); +const { moderateInput } = require('./middleware'); -dotenv.config(); +const app = express(); +app.use(express.json()); -const COLLECTION = "faq_collection"; +const openai = new OpenAI({ apiKey: process.env.OPENAI_API_KEY }); +const chroma = new ChromaClient({ path: 'chromadb' }); -async function main() { - // Optional: add some sample documents - await addDocument( - COLLECTION, - "What is the return policy?", - { source: "FAQ" } - ); - await addDocument( - COLLECTION, - "How can I track my order?", - { source: "FAQ" } - ); +const COLLECTION_NAME = 'faq_collection'; +const TOP_K = 3; - const agent = await createAgent(COLLECTION); +// Initialize collection +let collectionPromise = chroma.getOrCreateCollection({ + name: COLLECTION_NAME, + metadata: { description: 'FAQ embeddings' } +}); - const rl = readline.createInterface({ - input: process.stdin, - output: process.stdout, - prompt: "You: ", - }); - - console.log("FAQ Bot is ready. Type your question and press Enter."); - rl.prompt(); - - rl.on("line", async (line) => { - const question = line.trim(); +app.post('/ask', async (req, res) => { + try { + const { question } = req.body; if (!question) { - rl.prompt(); - return; + return res.status(400).json({ error: 'Question is required' }); } - try { - const result = await agent.call({ input: question }); - console.log(`Bot: ${result.output}`); - } catch (err) { - console.error("Error:", err); - } - rl.prompt(); - }); -} -main().catch((err) => console.error(err)); \ No newline at end of file + // Moderate user input + const moderationResult = await moderateInput(question); + if (!moderationResult.allowed) { + return res.status(403).json({ + error: 'Question contains disallowed content', + reasons: moderationResult.reasons + }); + } + + // Embed the question + const embeddingResponse = await openai.embeddings.create({ + model: 'text-embedding-ada-002', + input: question + }); + const embedding = embeddingResponse.data[0].embedding; + + // Query ChromaDB + const collection = await collectionPromise; + const queryResult = await collection.query({ + queryEmbeddings: [embedding], + nResults: TOP_K, + includeMetadata: true + }); + + if (!queryResult.ids || queryResult.ids.length === 0) { + return res.json({ answer: "I don't have an answer for that." }); + } + + // Pick the top result + const topAnswer = queryResult.metadatas[0]?.answer || "I don't have an answer for that."; + + res.json({ answer: topAnswer }); + } catch (err) { + console.error(err); + res.status(500).json({ error: 'Internal server error' }); + } +}); + +const PORT = process.env.PORT || 3000; +app.listen(PORT, () => { + console.log(`FAQ bot listening on port ${PORT}`); +}); \ No newline at end of file diff --git a/src/ingest.js b/src/ingest.js new file mode 100644 index 0000000..eeea324 --- /dev/null +++ b/src/ingest.js @@ -0,0 +1,55 @@ +require('dotenv').config(); +const fs = require('fs'); +const path = require('path'); +const { OpenAI } = require('openai'); +const { ChromaClient } = require('chromadb'); + +const openai = new OpenAI({ apiKey: process.env.OPENAI_API_KEY }); +const chroma = new ChromaClient({ path: 'chromadb' }); + +const COLLECTION_NAME = 'faq_collection'; +const FAQ_FILE = path.join(__dirname, '..', 'faq.json'); + +async function ingest() { + try { + const rawData = fs.readFileSync(FAQ_FILE, 'utf-8'); + const faqEntries = JSON.parse(rawData); + + const collection = await chroma.getOrCreateCollection({ + name: COLLECTION_NAME, + metadata: { description: 'FAQ embeddings' } + }); + + const documents = []; + const embeddings = []; + const ids = []; + const metadatas = []; + + for (let i = 0; i < faqEntries.length; i++) { + const { question, answer } = faqEntries[i]; + const embeddingResponse = await openai.embeddings.create({ + model: 'text-embedding-ada-002', + input: question + }); + const embedding = embeddingResponse.data[0].embedding; + + documents.push(question); + embeddings.push(embedding); + ids.push(`faq-${i}`); + metadatas.push({ answer }); + } + + await collection.add({ + documents, + embeddings, + ids, + metadatas + }); + + console.log(`Ingested ${faqEntries.length} FAQ entries into collection '${COLLECTION_NAME}'.`); + } catch (err) { + console.error('Error during ingestion:', err); + } +} + +ingest(); \ No newline at end of file diff --git a/src/middleware.js b/src/middleware.js new file mode 100644 index 0000000..3f3372f --- /dev/null +++ b/src/middleware.js @@ -0,0 +1,23 @@ +const moderate = require('moderate-censor'); + +/** + * Moderates user input using moderate-censor. + * @param {string} text + * @returns {Promise<{allowed: boolean, reasons: string[]}>} + */ +async function moderateInput(text) { + try { + const result = await moderate.moderate(text); + if (result.isAllowed) { + return { allowed: true, reasons: [] }; + } else { + return { allowed: false, reasons: result.reasons || [] }; + } + } catch (err) { + console.error('Moderation error:', err); + // If moderation fails, default to allowing to avoid blocking legitimate queries + return { allowed: true, reasons: [] }; + } +} + +module.exports = { moderateInput }; \ No newline at end of file