Build a RAG Chatbot from Scratch — Part 4: Evaluation and Quality Metrics
Why Evaluate RAG Systems?
RAG systems fail in two ways: (1) retrieving wrong documents, and (2) generating answers that aren’t grounded in the retrieved context. We need metrics for both.
Step 1: Retrieval Evaluation
Measure how well our retriever finds relevant chunks:
# app/evaluation/retrieval_metrics.py
def recall_at_k(relevant_ids: set[str], retrieved_ids: list[str], k: int = 5) -> float:
"""What fraction of relevant docs were retrieved in top-k?"""
if not relevant_ids:
return 0.0
retrieved_at_k = set(retrieved_ids[:k])
return len(relevant_ids & retrieved_at_k) / len(relevant_ids)
def mrr(relevant_ids: set[str], retrieved_ids: list[str]) -> float:
"""Mean Reciprocal Rank — position of the first relevant doc."""
for i, doc_id in enumerate(retrieved_ids, start=1):
if doc_id in relevant_ids:
return 1.0 / i
return 0.0
def precision_at_k(relevant_ids: set[str], retrieved_ids: list[str], k: int = 5) -> float:
"""What fraction of top-k retrieved docs are relevant?"""
retrieved_at_k = set(retrieved_ids[:k])
if not retrieved_at_k:
return 0.0
return len(relevant_ids & retrieved_at_k) / len(retrieved_at_k)
Step 2: Answer Faithfulness (Hallucination Detection)
Use an LLM to check if the generated answer is fully supported by the context:
# app/evaluation/faithfulness.py
from openai import OpenAI
from app.config import settings
client = OpenAI(api_key=settings.openai_api_key)
def check_faithfulness(answer: str, context: str) -> dict:
"""Use GPT-4 to evaluate if the answer is grounded in context."""
response = client.chat.completions.create(
model="gpt-4o",
messages=[{
"role": "system",
"content": """Evaluate whether the ANSWER is fully supported by the CONTEXT.
Return JSON:
{
"faithful": true/false,
"score": 0-10,
"unsupported_claims": ["claim not in context"],
"explanation": "reason"
}"""
}, {
"role": "user",
"content": f"CONTEXT:\n{context}\n\nANSWER:\n{answer}"
}],
response_format={"type": "json_object"}
)
return json.loads(response.choices[0].message.content)
Step 3: End-to-End Test Suite
# tests/test_rag_quality.py
import pytest
from app.generation import answer_question
from app.evaluation.faithfulness import check_faithfulness
TEST_CASES = [
{
"query": "What is the vacation policy?",
"expected_keywords": ["vacation", "days", "policy"],
"min_sources": 1,
"min_faithfulness": 7
},
{
"query": "How do I reset my password?",
"expected_keywords": ["reset", "password"],
"min_sources": 1,
"min_faithfulness": 7
}
]
@pytest.mark.parametrize("case", TEST_CASES)
def test_answer_quality(case):
result = answer_question(case["query"])
# Check answer contains expected keywords
answer_lower = result["answer"].lower()
for keyword in case["expected_keywords"]:
assert keyword in answer_lower, f"Missing keyword: {keyword}"
# Check sources exist
assert len(result["sources"]) >= case["min_sources"], "No sources returned"
@pytest.mark.parametrize("case", TEST_CASES)
def test_no_hallucination(case):
result = answer_question(case["query"])
# Build context from sources
context = "\n".join([s["content"] for s in result["sources"]])
faithfulness = check_faithfulness(result["answer"], context)
assert faithfulness["score"] >= case["min_faithfulness"], \
f"Faithfulness score too low: {faithfulness['score']}\nUnsupported: {faithfulness.get('unsupported_claims')}"
Step 4: Run the Evaluation Suite
pytest tests/test_rag_quality.py -v
# Output:
# test_answer_quality[vacation_policy] PASSED
# test_answer_quality[password_reset] PASSED
# test_no_hallucination[vacation_policy] PASSED
# test_no_hallucination[password_reset] PASSED
Summary
- Recall@k measures whether relevant docs appear in top-k results
- MRR tracks the position of the first relevant result
- Faithfulness scoring uses GPT-4 to detect unsupported claims
- Keyword assertions in tests catch totally wrong answers
- Automated eval runs as part of CI/CD pipeline
Advertisement