Build a RAG Chatbot from Scratch — Part 4: Evaluation and Quality Metrics

Why Evaluate RAG Systems?

RAG systems fail in two ways: (1) retrieving wrong documents, and (2) generating answers that aren’t grounded in the retrieved context. We need metrics for both.

Step 1: Retrieval Evaluation

Measure how well our retriever finds relevant chunks:

# app/evaluation/retrieval_metrics.py
def recall_at_k(relevant_ids: set[str], retrieved_ids: list[str], k: int = 5) -> float:
    """What fraction of relevant docs were retrieved in top-k?"""
    if not relevant_ids:
        return 0.0
    retrieved_at_k = set(retrieved_ids[:k])
    return len(relevant_ids & retrieved_at_k) / len(relevant_ids)

def mrr(relevant_ids: set[str], retrieved_ids: list[str]) -> float:
    """Mean Reciprocal Rank — position of the first relevant doc."""
    for i, doc_id in enumerate(retrieved_ids, start=1):
        if doc_id in relevant_ids:
            return 1.0 / i
    return 0.0

def precision_at_k(relevant_ids: set[str], retrieved_ids: list[str], k: int = 5) -> float:
    """What fraction of top-k retrieved docs are relevant?"""
    retrieved_at_k = set(retrieved_ids[:k])
    if not retrieved_at_k:
        return 0.0
    return len(relevant_ids & retrieved_at_k) / len(retrieved_at_k)

Step 2: Answer Faithfulness (Hallucination Detection)

Use an LLM to check if the generated answer is fully supported by the context:

# app/evaluation/faithfulness.py
from openai import OpenAI
from app.config import settings

client = OpenAI(api_key=settings.openai_api_key)

def check_faithfulness(answer: str, context: str) -> dict:
    """Use GPT-4 to evaluate if the answer is grounded in context."""
    response = client.chat.completions.create(
        model="gpt-4o",
        messages=[{
            "role": "system",
            "content": """Evaluate whether the ANSWER is fully supported by the CONTEXT.
Return JSON:
{
  "faithful": true/false,
  "score": 0-10,
  "unsupported_claims": ["claim not in context"],
  "explanation": "reason"
}"""
        }, {
            "role": "user",
            "content": f"CONTEXT:\n{context}\n\nANSWER:\n{answer}"
        }],
        response_format={"type": "json_object"}
    )
    return json.loads(response.choices[0].message.content)

Step 3: End-to-End Test Suite

# tests/test_rag_quality.py
import pytest
from app.generation import answer_question
from app.evaluation.faithfulness import check_faithfulness

TEST_CASES = [
    {
        "query": "What is the vacation policy?",
        "expected_keywords": ["vacation", "days", "policy"],
        "min_sources": 1,
        "min_faithfulness": 7
    },
    {
        "query": "How do I reset my password?",
        "expected_keywords": ["reset", "password"],
        "min_sources": 1,
        "min_faithfulness": 7
    }
]

@pytest.mark.parametrize("case", TEST_CASES)
def test_answer_quality(case):
    result = answer_question(case["query"])

    # Check answer contains expected keywords
    answer_lower = result["answer"].lower()
    for keyword in case["expected_keywords"]:
        assert keyword in answer_lower, f"Missing keyword: {keyword}"

    # Check sources exist
    assert len(result["sources"]) >= case["min_sources"], "No sources returned"

@pytest.mark.parametrize("case", TEST_CASES)
def test_no_hallucination(case):
    result = answer_question(case["query"])

    # Build context from sources
    context = "\n".join([s["content"] for s in result["sources"]])

    faithfulness = check_faithfulness(result["answer"], context)
    assert faithfulness["score"] >= case["min_faithfulness"], \
        f"Faithfulness score too low: {faithfulness['score']}\nUnsupported: {faithfulness.get('unsupported_claims')}"

Step 4: Run the Evaluation Suite

pytest tests/test_rag_quality.py -v

# Output:
# test_answer_quality[vacation_policy] PASSED
# test_answer_quality[password_reset] PASSED
# test_no_hallucination[vacation_policy] PASSED
# test_no_hallucination[password_reset] PASSED

Summary

  • Recall@k measures whether relevant docs appear in top-k results
  • MRR tracks the position of the first relevant result
  • Faithfulness scoring uses GPT-4 to detect unsupported claims
  • Keyword assertions in tests catch totally wrong answers
  • Automated eval runs as part of CI/CD pipeline

← Part 3 | Part 5 →


Advertisement