Build a RAG Chatbot from Scratch — Part 5: Deployment and Production Considerations
Final part — we containerize the chatbot, add rate limiting, and deploy.
Step 1: Dockerfile
FROM python:3.12-slim
WORKDIR /app
COPY requirements.txt .
RUN pip install --no-cache-dir -r requirements.txt
COPY . .
EXPOSE 8000
CMD ["uvicorn", "app.main:app", "--host", "0.0.0.0", "--port", "8000", "--workers", "2"]
Step 2: Rate Limiting
Prevent abuse with token bucket rate limiting:
# app/middleware/rate_limit.py
from fastapi import Request, HTTPException
from app.core.cache import redis_client
import time
RATE_LIMIT = 20 # requests per minute
async def rate_limit_middleware(request: Request, call_next):
if request.url.path == "/chat/stream":
client_ip = request.client.host
key = f"rate:{client_ip}"
current = redis_client.get(key)
if current and int(current) >= RATE_LIMIT:
raise HTTPException(429, "Too many requests. Please wait.")
pipe = redis_client.pipeline()
pipe.incr(key)
pipe.expire(key, 60)
pipe.execute()
return await call_next(request)
Step 3: docker-compose.yml
version: '3.8'
services:
api:
build: .
ports:
- "8000:8000"
environment:
OPENAI_API_KEY: ${OPENAI_API_KEY}
CHROMA_PERSIST_DIR: /data/chroma
volumes:
- chroma_data:/data/chroma
healthcheck:
test: ["CMD", "curl", "-f", "http://localhost:8000/health"]
interval: 30s
retries: 3
redis:
image: redis:7-alpine
volumes:
- redis_data:/data
volumes:
chroma_data:
redis_data:
Step 4: Cost Optimization
# Track and limit costs
COST_PER_1K_TOKENS = {
"gpt-4o-mini": {"input": 0.00015, "output": 0.0006},
"text-embedding-3-small": 0.00002
}
def estimate_cost(model: str, input_tokens: int, output_tokens: int) -> float:
costs = COST_PER_1K_TOKENS.get(model, {})
input_cost = input_tokens / 1000 * costs.get("input", 0)
output_cost = output_tokens / 1000 * costs.get("output", 0)
return round(input_cost + output_cost, 6)
Step 5: Health Check
@app.get("/health")
def health():
try:
chroma_client.heartbeat()
redis_client.ping()
return {"status": "healthy", "vector_store": "ok", "cache": "ok"}
except Exception as e:
return {"status": "unhealthy", "error": str(e)}
Deployment
docker-compose up -d --build
curl http://localhost:8000/health
# {"status":"healthy","vector_store":"ok","cache":"ok"}
Series Summary
We built a production RAG chatbot with document ingestion, semantic retrieval, streaming responses, hallucination detection, and deployment — all for pennies per query.
Advertisement