feat: solution for 'Повторный экзамен: FAQ-бот — ChromaDB + один MCP-tool'
This commit is contained in:
@@ -1,46 +1,102 @@
|
|||||||
# FAQ Bot – ChromaDB + LangChain
|
# FAQ Bot – ChromaDB + LangChain
|
||||||
|
|
||||||
This project implements a simple FAQ bot that answers user questions based on a predefined FAQ dataset.
|
This project implements a simple FAQ bot that uses **ChromaDB** for vector storage and **LangChain** as the single MCP‑tool to retrieve and generate answers.
|
||||||
The bot uses **ChromaDB** for vector storage and **LangChain** as the single MCP‑tool to process queries.
|
The bot can ingest FAQ documents, store embeddings in ChromaDB, and answer user questions via a command‑line interface.
|
||||||
|
|
||||||
## Features
|
## Features
|
||||||
|
|
||||||
- Persistent vector store (ChromaDB) – data is saved to disk and reused across runs.
|
- **Vector storage** – ChromaDB (DuckDB + Parquet backend)
|
||||||
- Retrieval‑based QA using LangChain’s `RetrievalQA` chain.
|
- **Embedding model** – OpenAI embeddings (`text-embedding-3-small`)
|
||||||
- Simple command‑line interface.
|
- **LLM** – OpenAI Chat (`gpt-4o-mini` by default)
|
||||||
- Unit tests covering vector store creation, bot answering, and unknown‑question handling.
|
- **MCP‑tool** – LangChain (only one MCP‑tool used)
|
||||||
|
- **CLI** – `python -m src.main ingest|ask`
|
||||||
|
- **Unit tests** – `pytest`
|
||||||
|
|
||||||
## Setup
|
## Setup
|
||||||
|
|
||||||
```bash
|
1. **Clone the repository**
|
||||||
# Create a virtual environment (optional but recommended)
|
|
||||||
python -m venv .venv
|
|
||||||
source .venv/bin/activate # On Windows use `.venv\Scripts\activate`
|
|
||||||
|
|
||||||
# Install dependencies
|
```bash
|
||||||
|
git clone https://git.brojs.ru/kuzakhmetovartur/povtornyy-ekzamen-faq-bot-chromadb-odin-.git
|
||||||
|
cd povtornyy-ekzamen-faq-bot-chromadb-odin-
|
||||||
|
```
|
||||||
|
|
||||||
|
2. **Create a virtual environment**
|
||||||
|
|
||||||
|
```bash
|
||||||
|
python3 -m venv .venv
|
||||||
|
source .venv/bin/activate
|
||||||
|
```
|
||||||
|
|
||||||
|
3. **Install dependencies**
|
||||||
|
|
||||||
|
```bash
|
||||||
pip install -r requirements.txt
|
pip install -r requirements.txt
|
||||||
```
|
```
|
||||||
|
|
||||||
|
4. **Set OpenAI API key**
|
||||||
|
|
||||||
|
```bash
|
||||||
|
export OPENAI_API_KEY="sk-..."
|
||||||
|
```
|
||||||
|
|
||||||
## Usage
|
## Usage
|
||||||
|
|
||||||
```bash
|
### Ingest FAQ file
|
||||||
python src/main.py
|
|
||||||
|
Prepare a text file with FAQ pairs in the following format:
|
||||||
|
|
||||||
|
```
|
||||||
|
Q: What is Python?
|
||||||
|
A: Python is a programming language.
|
||||||
|
|
||||||
|
Q: What is ChromaDB?
|
||||||
|
A: ChromaDB is a vector database.
|
||||||
```
|
```
|
||||||
|
|
||||||
You will be prompted to type a question. The bot will reply with the best answer from the FAQ dataset.
|
Run:
|
||||||
|
|
||||||
## Running Tests
|
```bash
|
||||||
|
python -m src.main ingest path/to/faq.txt --collection faq_collection
|
||||||
|
```
|
||||||
|
|
||||||
|
### Ask a question
|
||||||
|
|
||||||
|
```bash
|
||||||
|
python -m src.main ask "What is Python?" --collection faq_collection
|
||||||
|
```
|
||||||
|
|
||||||
|
The bot will print the generated answer.
|
||||||
|
|
||||||
|
## Testing
|
||||||
|
|
||||||
|
Run the test suite:
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
pytest
|
pytest
|
||||||
```
|
```
|
||||||
|
|
||||||
All tests should pass.
|
## Project Structure
|
||||||
|
|
||||||
## FAQ Dataset
|
```
|
||||||
|
src/
|
||||||
|
├── main.py # CLI entry point
|
||||||
|
├── ingest.py # Ingestion logic
|
||||||
|
└── retriever.py # Retrieval & answer generation
|
||||||
|
tests/
|
||||||
|
├── test_ingest.py
|
||||||
|
└── test_retrieval.py
|
||||||
|
requirements.txt
|
||||||
|
README.md
|
||||||
|
```
|
||||||
|
|
||||||
The dataset is embedded in the code (3 entries). Feel free to extend it in `src/vector_store.py`.
|
## Notes
|
||||||
|
|
||||||
## License
|
- The bot uses the default OpenAI embeddings and LLM.
|
||||||
|
If you want to change the model, edit the `OpenAIEmbeddings()` and `OpenAIChat()` calls in `src/ingest.py` and `src/retriever.py`.
|
||||||
|
- ChromaDB data is persisted in the `chromadb/` directory relative to the project root.
|
||||||
|
- The deadline for the assignment is **31.08.2026**. All code is committed to the specified Git repository.
|
||||||
|
|
||||||
MIT License
|
---
|
||||||
|
|
||||||
|
Happy coding!
|
||||||
+3
-3
@@ -1,4 +1,4 @@
|
|||||||
chromadb==0.4.22
|
chromadb==0.4.22
|
||||||
langchain==0.0.346
|
langchain==0.1.12
|
||||||
openai==0.27.8
|
openai==1.12.0
|
||||||
pytest==7.4.3
|
pytest==8.2.2
|
||||||
@@ -0,0 +1,72 @@
|
|||||||
|
"""
|
||||||
|
Ingestion logic for FAQ documents into ChromaDB.
|
||||||
|
"""
|
||||||
|
|
||||||
|
import os
|
||||||
|
from pathlib import Path
|
||||||
|
from typing import List
|
||||||
|
|
||||||
|
from chromadb import Client
|
||||||
|
from chromadb.api.types import Documents, EmbeddingFunction
|
||||||
|
from chromadb.config import Settings
|
||||||
|
|
||||||
|
from langchain.embeddings.openai import OpenAIEmbeddings
|
||||||
|
from langchain.text_splitter import RecursiveCharacterTextSplitter
|
||||||
|
|
||||||
|
def _load_faq_pairs(file_path: Path) -> List[tuple]:
|
||||||
|
"""
|
||||||
|
Load FAQ pairs from a text file.
|
||||||
|
Expected format:
|
||||||
|
Q: <question>
|
||||||
|
A: <answer>
|
||||||
|
Each pair separated by a blank line.
|
||||||
|
"""
|
||||||
|
pairs = []
|
||||||
|
with file_path.open("r", encoding="utf-8") as f:
|
||||||
|
content = f.read()
|
||||||
|
raw_pairs = content.strip().split("\n\n")
|
||||||
|
for raw in raw_pairs:
|
||||||
|
lines = raw.strip().splitlines()
|
||||||
|
if len(lines) < 2:
|
||||||
|
continue
|
||||||
|
q_line = lines[0].strip()
|
||||||
|
a_line = lines[1].strip()
|
||||||
|
if q_line.lower().startswith("q:") and a_line.lower().startswith("a:"):
|
||||||
|
question = q_line[2:].strip()
|
||||||
|
answer = a_line[2:].strip()
|
||||||
|
pairs.append((question, answer))
|
||||||
|
return pairs
|
||||||
|
|
||||||
|
def ingest_faq(file_path: Path, client: Client, collection_name: str):
|
||||||
|
"""
|
||||||
|
Ingest FAQ pairs into the specified ChromaDB collection.
|
||||||
|
"""
|
||||||
|
pairs = _load_faq_pairs(file_path)
|
||||||
|
if not pairs:
|
||||||
|
raise ValueError("No valid FAQ pairs found in the file.")
|
||||||
|
|
||||||
|
# Prepare documents and metadata
|
||||||
|
documents = []
|
||||||
|
metadatas = []
|
||||||
|
ids = []
|
||||||
|
|
||||||
|
for idx, (q, a) in enumerate(pairs):
|
||||||
|
# Combine question and answer for embedding
|
||||||
|
doc = f"Q: {q}\nA: {a}"
|
||||||
|
documents.append(doc)
|
||||||
|
metadatas.append({"question": q, "answer": a})
|
||||||
|
ids.append(str(idx))
|
||||||
|
|
||||||
|
# Use OpenAI embeddings
|
||||||
|
embedding = OpenAIEmbeddings()
|
||||||
|
|
||||||
|
# Create or get collection
|
||||||
|
collection = client.get_or_create_collection(name=collection_name)
|
||||||
|
|
||||||
|
# Add documents to collection
|
||||||
|
collection.add(
|
||||||
|
documents=documents,
|
||||||
|
metadatas=metadatas,
|
||||||
|
ids=ids,
|
||||||
|
embedding_function=embedding
|
||||||
|
)
|
||||||
+53
-27
@@ -1,41 +1,67 @@
|
|||||||
|
#!/usr/bin/env python3
|
||||||
"""
|
"""
|
||||||
Command‑line interface for the FAQ bot.
|
FAQ Bot using ChromaDB and LangChain
|
||||||
"""
|
"""
|
||||||
|
|
||||||
import argparse
|
|
||||||
import os
|
import os
|
||||||
|
import sys
|
||||||
|
import argparse
|
||||||
|
from pathlib import Path
|
||||||
|
|
||||||
from .bot import FAQBot
|
from chromadb import Client
|
||||||
|
from chromadb.config import Settings
|
||||||
|
|
||||||
|
from langchain.embeddings.openai import OpenAIEmbeddings
|
||||||
|
from langchain.llms.openai import OpenAIChat
|
||||||
|
from langchain.chains import RetrievalQA
|
||||||
|
from langchain.vectorstores import Chroma
|
||||||
|
|
||||||
|
from ingest import ingest_faq
|
||||||
|
from retriever import get_answer
|
||||||
|
|
||||||
|
def init_chroma(collection_name: str) -> Client:
|
||||||
|
"""
|
||||||
|
Initialize a ChromaDB client and create a collection if it does not exist.
|
||||||
|
"""
|
||||||
|
client = Client(Settings(
|
||||||
|
chroma_db_impl="duckdb+parquet",
|
||||||
|
persist_directory="chromadb",
|
||||||
|
))
|
||||||
|
# Ensure collection exists
|
||||||
|
if collection_name not in client.list_collections():
|
||||||
|
client.create_collection(name=collection_name)
|
||||||
|
return client
|
||||||
|
|
||||||
def main():
|
def main():
|
||||||
parser = argparse.ArgumentParser(description="FAQ Bot CLI")
|
parser = argparse.ArgumentParser(description="FAQ Bot CLI")
|
||||||
parser.add_argument(
|
subparsers = parser.add_subparsers(dest="command", required=True)
|
||||||
"--persist-dir",
|
|
||||||
type=str,
|
ingest_parser = subparsers.add_parser("ingest", help="Ingest FAQ file into ChromaDB")
|
||||||
default="chromadb_persist",
|
ingest_parser.add_argument("faq_file", type=Path, help="Path to FAQ text file")
|
||||||
help="Directory to persist ChromaDB data",
|
ingest_parser.add_argument("--collection", type=str, default="faq_collection", help="Chroma collection name")
|
||||||
)
|
|
||||||
parser.add_argument(
|
query_parser = subparsers.add_parser("ask", help="Ask a question to the FAQ bot")
|
||||||
"--openai-key",
|
query_parser.add_argument("question", type=str, help="Your question")
|
||||||
type=str,
|
query_parser.add_argument("--collection", type=str, default="faq_collection", help="Chroma collection name")
|
||||||
default=os.getenv("OPENAI_API_KEY"),
|
|
||||||
help="OpenAI API key (optional)",
|
|
||||||
)
|
|
||||||
args = parser.parse_args()
|
args = parser.parse_args()
|
||||||
|
|
||||||
bot = FAQBot(persist_dir=args.persist_dir, openai_api_key=args.openai_key)
|
# Ensure OpenAI API key is set
|
||||||
|
if "OPENAI_API_KEY" not in os.environ:
|
||||||
|
print("Error: OPENAI_API_KEY environment variable not set.", file=sys.stderr)
|
||||||
|
sys.exit(1)
|
||||||
|
|
||||||
print("FAQ Bot is ready. Type your question (Ctrl+C to exit).")
|
client = init_chroma(args.collection)
|
||||||
while True:
|
|
||||||
try:
|
if args.command == "ingest":
|
||||||
question = input("\n> ")
|
ingest_faq(args.faq_file, client, args.collection)
|
||||||
if not question.strip():
|
print(f"Ingestion completed. Collection '{args.collection}' updated.")
|
||||||
continue
|
elif args.command == "ask":
|
||||||
answer = bot.ask(question)
|
answer = get_answer(args.question, client, args.collection)
|
||||||
print(f"\nAnswer: {answer}")
|
print("\nAnswer:\n")
|
||||||
except (KeyboardInterrupt, EOFError):
|
print(answer)
|
||||||
print("\nGoodbye!")
|
else:
|
||||||
break
|
parser.print_help()
|
||||||
|
|
||||||
if __name__ == "__main__":
|
if __name__ == "__main__":
|
||||||
main()
|
main()
|
||||||
@@ -0,0 +1,42 @@
|
|||||||
|
"""
|
||||||
|
Retrieval and answer generation logic using LangChain.
|
||||||
|
"""
|
||||||
|
|
||||||
|
import os
|
||||||
|
from typing import Any
|
||||||
|
|
||||||
|
from chromadb import Client
|
||||||
|
from chromadb.config import Settings
|
||||||
|
|
||||||
|
from langchain.embeddings.openai import OpenAIEmbeddings
|
||||||
|
from langchain.llms.openai import OpenAIChat
|
||||||
|
from langchain.chains import RetrievalQA
|
||||||
|
from langchain.vectorstores import Chroma
|
||||||
|
|
||||||
|
def get_answer(question: str, client: Client, collection_name: str, k: int = 3) -> str:
|
||||||
|
"""
|
||||||
|
Retrieve relevant FAQ chunks and generate an answer using OpenAIChat.
|
||||||
|
"""
|
||||||
|
# Set up embeddings and LLM
|
||||||
|
embedding = OpenAIEmbeddings()
|
||||||
|
llm = OpenAIChat(temperature=0)
|
||||||
|
|
||||||
|
# Load vector store
|
||||||
|
vectorstore = Chroma(
|
||||||
|
client=client,
|
||||||
|
collection_name=collection_name,
|
||||||
|
embedding_function=embedding
|
||||||
|
)
|
||||||
|
|
||||||
|
# Build RetrievalQA chain
|
||||||
|
qa_chain = RetrievalQA.from_chain_type(
|
||||||
|
llm=llm,
|
||||||
|
chain_type="stuff",
|
||||||
|
retriever=vectorstore.as_retriever(search_kwargs={"k": k}),
|
||||||
|
return_source_documents=True
|
||||||
|
)
|
||||||
|
|
||||||
|
# Run chain
|
||||||
|
result = qa_chain({"question": question})
|
||||||
|
answer = result.get("answer", "")
|
||||||
|
return answer.strip()
|
||||||
@@ -0,0 +1,44 @@
|
|||||||
|
import os
|
||||||
|
import tempfile
|
||||||
|
import shutil
|
||||||
|
import json
|
||||||
|
from pathlib import Path
|
||||||
|
|
||||||
|
import chromadb
|
||||||
|
from chromadb.config import Settings
|
||||||
|
|
||||||
|
from src.ingest import ingest_faq
|
||||||
|
|
||||||
|
def test_ingest_faq(tmp_path):
|
||||||
|
# Create a temporary FAQ file
|
||||||
|
faq_content = """Q: What is Python?
|
||||||
|
A: Python is a programming language.
|
||||||
|
|
||||||
|
Q: What is ChromaDB?
|
||||||
|
A: ChromaDB is a vector database."""
|
||||||
|
faq_file = tmp_path / "faq.txt"
|
||||||
|
faq_file.write_text(faq_content, encoding="utf-8")
|
||||||
|
|
||||||
|
# Initialize a temporary ChromaDB client
|
||||||
|
db_dir = tmp_path / "chromadb"
|
||||||
|
client = chromadb.Client(Settings(
|
||||||
|
chroma_db_impl="duckdb+parquet",
|
||||||
|
persist_directory=str(db_dir)
|
||||||
|
))
|
||||||
|
|
||||||
|
collection_name = "test_collection"
|
||||||
|
|
||||||
|
# Ingest
|
||||||
|
ingest_faq(faq_file, client, collection_name)
|
||||||
|
|
||||||
|
# Verify collection exists and has documents
|
||||||
|
collection = client.get_collection(name=collection_name)
|
||||||
|
assert collection.count() == 2
|
||||||
|
|
||||||
|
# Verify metadata
|
||||||
|
docs = collection.get(ids=["0", "1"])
|
||||||
|
assert docs["metadatas"][0]["question"] == "What is Python?"
|
||||||
|
assert docs["metadatas"][1]["answer"] == "ChromaDB is a vector database."
|
||||||
|
|
||||||
|
# Clean up
|
||||||
|
shutil.rmtree(db_dir)
|
||||||
@@ -0,0 +1,51 @@
|
|||||||
|
import os
|
||||||
|
import tempfile
|
||||||
|
import shutil
|
||||||
|
import json
|
||||||
|
from pathlib import Path
|
||||||
|
|
||||||
|
import chromadb
|
||||||
|
from chromadb.config import Settings
|
||||||
|
|
||||||
|
from src.ingest import ingest_faq
|
||||||
|
from src.retriever import get_answer
|
||||||
|
|
||||||
|
def test_retrieval(tmp_path, monkeypatch):
|
||||||
|
# Mock OpenAI API key
|
||||||
|
monkeypatch.setenv("OPENAI_API_KEY", "test_key")
|
||||||
|
|
||||||
|
# Create a temporary FAQ file
|
||||||
|
faq_content = """Q: What is Python?
|
||||||
|
A: Python is a programming language.
|
||||||
|
|
||||||
|
Q: What is ChromaDB?
|
||||||
|
A: ChromaDB is a vector database."""
|
||||||
|
faq_file = tmp_path / "faq.txt"
|
||||||
|
faq_file.write_text(faq_content, encoding="utf-8")
|
||||||
|
|
||||||
|
# Initialize a temporary ChromaDB client
|
||||||
|
db_dir = tmp_path / "chromadb"
|
||||||
|
client = chromadb.Client(Settings(
|
||||||
|
chroma_db_impl="duckdb+parquet",
|
||||||
|
persist_directory=str(db_dir)
|
||||||
|
))
|
||||||
|
|
||||||
|
collection_name = "test_collection"
|
||||||
|
|
||||||
|
# Ingest
|
||||||
|
ingest_faq(faq_file, client, collection_name)
|
||||||
|
|
||||||
|
# Mock OpenAIChat to avoid real API calls
|
||||||
|
class DummyLLM:
|
||||||
|
def __call__(self, *args, **kwargs):
|
||||||
|
return "Dummy answer"
|
||||||
|
|
||||||
|
# Patch the LLM in retriever
|
||||||
|
monkeypatch.setattr("src.retriever.OpenAIChat", DummyLLM)
|
||||||
|
|
||||||
|
# Retrieve answer
|
||||||
|
answer = get_answer("What is Python?", client, collection_name)
|
||||||
|
assert answer == "Dummy answer"
|
||||||
|
|
||||||
|
# Clean up
|
||||||
|
shutil.rmtree(db_dir)
|
||||||
Reference in New Issue
Block a user