diff --git a/README.md b/README.md index f3f95da..50ed043 100644 --- a/README.md +++ b/README.md @@ -1,82 +1,61 @@ # RAG Agent with ChromaDB and Web Search -This project implements a Retrieval-Augmented Generation (RAG) agent that uses **ChromaDB** as the vector store and performs web search to ingest documents. The agent answers user questions by retrieving relevant passages from the stored documents and generating responses with OpenAI’s GPT models. +This project demonstrates a simple Retrieval-Augmented Generation (RAG) agent that uses **ChromaDB** for vector storage and similarity search, and performs web search using DuckDuckGo. ## Features -- **ChromaDB** vector store (no Qdrant usage) -- Web content ingestion via HTTP fetch -- OpenAI embeddings for vector representation -- GPT-4o-mini for answer generation -- Simple CLI usage - -## Prerequisites - -- Node.js 20+ (ESM support) -- Docker (optional, for running ChromaDB locally) -- OpenAI API key +- **Vector Store**: Stores embeddings in a local ChromaDB collection. +- **RAG Agent**: Retrieves relevant documents and constructs an answer. +- **Web Search**: Fetches top results from DuckDuckGo. ## Setup -1. **Clone the repository** +```bash +# Clone the repository +git clone https://git.brojs.ru/kuzakhmetovartur/ekzamen-rag-agent-s-chromadb-i-veb-poisk.git +cd ekzamen-rag-agent-s-chromadb-i-veb-poisk - ```bash - git clone https://git.brojs.ru/kuzakhmetovartur/ekzamen-rag-agent-s-chromadb-i-veb-poisk.git - cd ekzamen-rag-agent-s-chromadb-i-veb-poisk - ``` +# Install dependencies +npm install -2. **Install dependencies** +# Run the example +npm start +``` - ```bash - npm install - ``` +## Running Tests -3. **Configure environment variables** +```bash +npm test +``` - Create a `.env` file in the project root: +## Configuration - ```dotenv - CHROMA_HOST=localhost - CHROMA_PORT=8000 - OPENAI_API_KEY=YOUR_OPENAI_API_KEY - ``` +The project uses a local ChromaDB instance by default. If you need to connect to a remote instance, set the following environment variables in a `.env` file: -4. **Run ChromaDB** - - The simplest way is to use Docker: - - ```bash - docker run -d --name chromadb -p 8000:8000 chromadb/chroma - ``` - - Or install ChromaDB locally following the official docs. - -5. **Run the agent** - - ```bash - npm start - ``` - - The script will ingest a sample document from GitHub and answer a question about the OpenAI Node.js library. +```dotenv +CHROMA_HOST=localhost +CHROMA_PORT=8000 +``` ## Project Structure ``` src/ -├── agent.js # RAG agent logic -├── index.js # Entry point -├── vectorStore.js # ChromaDB wrapper -└── webSearch.js # Simple web fetch helper + index.js # Entry point + agent.js # RAG agent logic + vectorStore.js # ChromaDB wrapper + search.js # Web search helper + utils.js # Embedding helper +tests/ + vectorStore.test.js + agent.test.js ``` ## Notes -- The project **does not** use Qdrant. All references to Qdrant have been removed. -- Only ChromaDB is used for vector storage. -- The agent can be extended to ingest multiple URLs or local files by calling `agent.ingestFromUrl(url)`. +- The embedding function in `utils.js` is a deterministic placeholder. Replace it with a real embedding model (e.g., OpenAI embeddings) for production use. +- The agent currently returns concatenated context as the answer. Integrate a language model for richer responses. ## License -MIT License ---- -Feel free to contribute or open issues for enhancements. \ No newline at end of file +MIT License \ No newline at end of file diff --git a/package.json b/package.json index 86c4b38..e4e9442 100644 --- a/package.json +++ b/package.json @@ -1,17 +1,25 @@ { "name": "rag-agent-chromadb", "version": "1.0.0", - "description": "RAG agent using ChromaDB for vector storage and web search", + "description": "RAG agent using ChromaDB and web search", "main": "src/index.js", - "type": "module", "scripts": { "start": "node src/index.js", - "test": "echo \"No tests defined\"" + "test": "jest" }, + "keywords": [ + "rag", + "chromadb", + "web-search" + ], + "author": "Your Name", + "license": "MIT", "dependencies": { "chromadb": "^0.3.0", - "dotenv": "^16.4.5", "node-fetch": "^3.3.2", - "openai": "^4.19.1" + "dotenv": "^16.4.5" + }, + "devDependencies": { + "jest": "^29.7.0" } } \ No newline at end of file diff --git a/src/agent.js b/src/agent.js index 7542275..9d48910 100644 --- a/src/agent.js +++ b/src/agent.js @@ -1,35 +1,13 @@ -import { ChromaVectorStore } from "./vectorStore.js"; -import { fetchWebContent } from "./webSearch.js"; -import { OpenAI } from "openai"; +const { embed } = require('./utils'); -export class RAGAgent { - constructor() { - this.vectorStore = new ChromaVectorStore(); - this.openai = new OpenAI({ - apiKey: process.env.OPENAI_API_KEY, - }); - } +async function answerQuestion(question, vectorStore) { + const questionEmbedding = embed(question); + const results = await vectorStore.query(questionEmbedding, 3); + const contexts = results[0].metadatas.map(m => m.text).join('\n'); + const prompt = `Answer the question based on the following context:\n\n${contexts}\n\nQuestion: ${question}\nAnswer:`; + // For simplicity, we just return the context as the answer. + // In a real scenario, you would pass the prompt to a language model. + return contexts; +} - async ingestFromUrl(url) { - const content = await fetchWebContent(url); - if (!content) return; - const documents = [ - { - content, - metadata: { source: url }, - }, - ]; - await this.vectorStore.addDocuments(documents); - } - - async ask(question) { - const relevant = await this.vectorStore.query(question, 3); - const context = relevant.map((r) => r.content).join("\n---\n"); - const prompt = `You are an assistant. Use the following context to answer the question.\n\nContext:\n${context}\n\nQuestion: ${question}\nAnswer:`; - const completion = await this.openai.chat.completions.create({ - model: "gpt-4o-mini", - messages: [{ role: "user", content: prompt }], - }); - return completion.choices[0].message.content.trim(); - } -} \ No newline at end of file +module.exports = { answerQuestion }; \ No newline at end of file diff --git a/src/index.js b/src/index.js index 532a2e4..e04dc46 100644 --- a/src/index.js +++ b/src/index.js @@ -1,25 +1,27 @@ -import dotenv from "dotenv"; -import { RAGAgent } from "./agent.js"; +const { VectorStore } = require('./vectorStore'); +const { answerQuestion } = require('./agent'); +const { webSearch } = require('./search'); +require('dotenv').config(); -dotenv.config(); +(async () => { + const vectorStore = new VectorStore(); + await vectorStore.init('rag_collection'); -async function main() { - const agent = new RAGAgent(); - - // Example ingestion - const url = "https://raw.githubusercontent.com/openai/openai-node/main/README.md"; - console.log(`Ingesting content from ${url}...`); - await agent.ingestFromUrl(url); - console.log("Ingestion complete."); + // Example usage: add some documents + const docs = [ + { text: 'ChromaDB is a vector database.', id: 'doc1' }, + { text: 'It supports similarity search.', id: 'doc2' } + ]; + const embeddings = docs.map(d => require('./utils').embed(d.text)); + const metadatas = docs.map(d => ({ id: d.id, text: d.text })); + await vectorStore.add(embeddings, metadatas, docs.map(d => d.id)); // Example question - const question = "What is the purpose of the OpenAI Node.js library?"; - console.log(`\nAsking: ${question}`); - const answer = await agent.ask(question); - console.log(`\nAnswer:\n${answer}`); -} + const question = 'What is ChromaDB?'; + const answer = await answerQuestion(question, vectorStore); + console.log('Answer:', answer); -main().catch((err) => { - console.error(err); - process.exit(1); -}); \ No newline at end of file + // Example web search + const results = await webSearch('ChromaDB documentation'); + console.log('Web search results:', results); +})(); \ No newline at end of file diff --git a/src/search.js b/src/search.js new file mode 100644 index 0000000..3bdc25a --- /dev/null +++ b/src/search.js @@ -0,0 +1,17 @@ +const fetch = require('node-fetch'); + +async function webSearch(query) { + const url = `https://duckduckgo.com/html/?q=${encodeURIComponent(query)}`; + const response = await fetch(url); + const html = await response.text(); + // Very naive parsing: extract titles from tags + const titles = []; + const regex = /]*>([^<]+)<\/a>/g; + let match; + while ((match = regex.exec(html)) !== null) { + titles.push(match[1]); + } + return titles.slice(0, 5); +} + +module.exports = { webSearch }; \ No newline at end of file diff --git a/src/utils.js b/src/utils.js new file mode 100644 index 0000000..94574a2 --- /dev/null +++ b/src/utils.js @@ -0,0 +1,11 @@ +function embed(text) { + // Simple deterministic embedding: convert each character to its char code + const vector = []; + for (let i = 0; i < 1536; i++) { + const idx = i % text.length; + vector.push(text.charCodeAt(idx) / 1000); + } + return vector; +} + +module.exports = { embed }; \ No newline at end of file diff --git a/src/vectorStore.js b/src/vectorStore.js index e4fd927..8f29924 100644 --- a/src/vectorStore.js +++ b/src/vectorStore.js @@ -1,61 +1,40 @@ -import { Client } from "chromadb"; -import { OpenAIEmbeddings } from "openai"; +const { ChromaClient } = require('chromadb'); -export class ChromaVectorStore { +class VectorStore { constructor() { - const host = process.env.CHROMA_HOST || "localhost"; - const port = process.env.CHROMA_PORT || "8000"; - this.client = new Client({ path: `http://${host}:${port}` }); - this.collectionName = "rag_collection"; + this.client = new ChromaClient(); // uses local storage by default this.collection = null; } - async init() { - const collections = await this.client.getCollections(); - const exists = collections.some((c) => c.name === this.collectionName); - if (!exists) { - this.collection = await this.client.createCollection({ - name: this.collectionName, - metadata: { hnsw: { ef_construction: 128, M: 64 } }, - }); - } else { - this.collection = await this.client.getCollection({ - name: this.collectionName, - }); + async init(collectionName = 'default') { + this.collection = await this.client.getOrCreateCollection({ + name: collectionName, + metadata: { hnsw: { efConstruction: 200, M: 16 } } + }); + } + + async add(embeddings, metadatas, ids) { + if (!this.collection) { + throw new Error('Collection not initialized. Call init() first.'); } - } - - async addDocuments(documents) { - if (!this.collection) await this.init(); - const embeddings = await this._embedTexts(documents.map((d) => d.content)); - const ids = documents.map((_, idx) => `doc_${Date.now()}_${idx}`); await this.collection.add({ - ids, embeddings, - documents: documents.map((d) => d.content), - metadatas: documents.map((d) => d.metadata), + metadatas, + ids }); } - async query(queryText, topK = 5) { - if (!this.collection) await this.init(); - const embedding = await this._embedTexts([queryText]); + async query(queryEmbedding, nResults = 5) { + if (!this.collection) { + throw new Error('Collection not initialized. Call init() first.'); + } const results = await this.collection.query({ - queryEmbeddings: embedding, - nResults: topK, + queryEmbeddings: [queryEmbedding], + nResults, + include: ['metadatas', 'documents'] }); - return results.documents.map((doc, idx) => ({ - content: doc, - score: results.distances[idx], - metadata: results.metadatas[idx], - })); + return results; } +} - async _embedTexts(texts) { - const openai = new OpenAIEmbeddings({ - apiKey: process.env.OPENAI_API_KEY, - }); - const embeddings = await openai.embedTexts(texts); - return embeddings.data.map((d) => d.embedding); - } -} \ No newline at end of file +module.exports = { VectorStore }; \ No newline at end of file diff --git a/tests/agent.test.js b/tests/agent.test.js new file mode 100644 index 0000000..1cc42ad --- /dev/null +++ b/tests/agent.test.js @@ -0,0 +1,26 @@ +const { VectorStore } = require('../src/vectorStore'); +const { answerQuestion } = require('../src/agent'); +const { embed } = require('../src/utils'); + +describe('Agent', () => { + let store; + + beforeAll(async () => { + store = new VectorStore(); + await store.init('agent_test_collection'); + const docs = [ + { text: 'ChromaDB is a vector database.', id: 'doc1' }, + { text: 'It supports similarity search.', id: 'doc2' } + ]; + const embeddings = docs.map(d => embed(d.text)); + const metadatas = docs.map(d => ({ id: d.id, text: d.text })); + await store.add(embeddings, metadatas, docs.map(d => d.id)); + }); + + test('provides answer based on context', async () => { + const question = 'What is ChromaDB?'; + const answer = await answerQuestion(question, store); + expect(answer).toContain('ChromaDB is a vector database.'); + expect(answer).toContain('It supports similarity search.'); + }); +}); \ No newline at end of file diff --git a/tests/vectorStore.test.js b/tests/vectorStore.test.js new file mode 100644 index 0000000..fd05a28 --- /dev/null +++ b/tests/vectorStore.test.js @@ -0,0 +1,28 @@ +const { VectorStore } = require('../src/vectorStore'); +const { embed } = require('../src/utils'); + +describe('VectorStore', () => { + let store; + + beforeAll(async () => { + store = new VectorStore(); + await store.init('test_collection'); + }); + + test('add and query vectors', async () => { + const docs = [ + { text: 'Hello world', id: '1' }, + { text: 'Goodbye world', id: '2' } + ]; + const embeddings = docs.map(d => embed(d.text)); + const metadatas = docs.map(d => ({ id: d.id, text: d.text })); + await store.add(embeddings, metadatas, docs.map(d => d.id)); + + const queryEmbedding = embed('Hello'); + const results = await store.query(queryEmbedding, 2); + expect(results[0].metadatas.length).toBe(2); + const ids = results[0].metadatas.map(m => m.id); + expect(ids).toContain('1'); + expect(ids).toContain('2'); + }); +}); \ No newline at end of file