Build a RAG Chatbot from Scratch — Part 2: Retrieval and Prompt Assembly

In this part, we query our vector store and construct prompts that ground the LLM in retrieved documents.

Step 1: Semantic Retrieval

# app/retrieval.py
from app.embeddings import get_embedding
from app.vector_store import get_or_create_collection

def retrieve(query: str, k: int = 5, threshold: float = 0.7) -> list[dict]:
    """Retrieve the k most relevant chunks for a query."""
    query_embedding = get_embedding(query)
    collection = get_or_create_collection()

    results = collection.query(
        query_embeddings=[query_embedding],
        n_results=k
    )

    chunks = []
    for i, doc in enumerate(results["documents"][0]):
        distance = results["distances"][0][i]
        similarity = 1 - distance  # Cosine distance → similarity

        if similarity < threshold:
            continue

        chunks.append({
            "content": doc,
            "similarity": round(similarity, 3),
            "metadata": results["metadatas"][0][i]
        })

    return chunks

Threshold filtering prevents irrelevant chunks from being included. A similarity below 0.7 means the chunk is too different from the question to be useful.

Step 2: Prompt Assembly

# app/prompts.py
SYSTEM_PROMPT = """You are a helpful assistant that answers questions based ONLY on the provided context.

Rules:
- If the answer is in the context, cite the source
- If the context doesn't contain the answer, say "I don't have enough information to answer this"
- Do not use outside knowledge
- Keep answers concise and factual
- Include relevant code snippets or examples when present in the context
"""

def build_rag_prompt(query: str, chunks: list[dict]) -> str:
    """Assemble retrieved chunks into a single prompt."""
    context_parts = []
    for i, chunk in enumerate(chunks):
        source = chunk["metadata"].get("source", "unknown")
        context_parts.append(
            f"[Source {i+1}: {source}]\n{chunk['content']}"
        )

    context = "\n\n---\n\n".join(context_parts)

    return context, SYSTEM_PROMPT

Why a system prompt? It constrains the LLM to only use provided context, preventing hallucinations.

Step 3: Generate Answer

# app/generation.py
from openai import OpenAI
from app.config import settings
from app.retrieval import retrieve
from app.prompts import build_rag_prompt

client = OpenAI(api_key=settings.openai_api_key)

def answer_question(query: str) -> dict:
    """Full RAG pipeline: retrieve → prompt → generate."""
    # Retrieve relevant chunks
    chunks = retrieve(query)

    if not chunks:
        return {
            "answer": "No relevant documents found to answer this question.",
            "sources": []
        }

    # Build prompt
    context, system_prompt = build_rag_prompt(query, chunks)

    # Generate response
    response = client.chat.completions.create(
        model=settings.model,
        messages=[
            {"role": "system", "content": system_prompt},
            {"role": "user", "content": f"Context:\n{context}\n\nQuestion: {query}"}
        ],
        temperature=0.3,  # Low temp = more factual
        max_tokens=1000
    )

    return {
        "answer": response.choices[0].message.content,
        "sources": [
            {"source": c["metadata"]["source"], "similarity": c["similarity"]}
            for c in chunks
        ],
        "tokens_used": response.usage.total_tokens
    }

Step 4: Q&A API Endpoint

# app/api/chat.py
from fastapi import APIRouter
from pydantic import BaseModel
from app.generation import answer_question

router = APIRouter(prefix="/chat", tags=["chat"])

class Question(BaseModel):
    query: str
    max_sources: int = 5

@router.post("/")
def ask(question: Question):
    return answer_question(question.query)

Step 5: Improving Retrieval with Reranking

For better accuracy, add a lightweight reranking step:

def rerank_chunks(query: str, chunks: list[dict]) -> list[dict]:
    """Use keyword overlap as a cheap reranking signal."""
    query_words = set(query.lower().split())

    for chunk in chunks:
        chunk_words = set(chunk["content"].lower().split())
        overlap = len(query_words & chunk_words) / len(query_words)
        chunk["keyword_score"] = round(overlap, 3)
        chunk["combined_score"] = chunk["similarity"] * 0.7 + overlap * 0.3

    return sorted(chunks, key=lambda c: c["combined_score"], reverse=True)

This combines semantic similarity (70%) with keyword overlap (30%) for a hybrid score.

Verification

# Ingest sample documents first
curl -X POST http://localhost:8000/ingest/pdf -F "file=@docs/company_handbook.pdf"

# Ask a question
curl -X POST http://localhost:8000/chat/ \
  -H "Content-Type: application/json" \
  -d '{"query": "What is the vacation policy?"}'

# Response:
# {
#   "answer": "Employees receive 20 days of paid vacation per year...",
#   "sources": [
#     {"source": "company_handbook.pdf", "similarity": 0.92}
#   ],
#   "tokens_used": 450
# }

Summary

  • Semantic retrieval finds relevant chunks using cosine similarity
  • Threshold filtering (0.7+) prevents irrelevant context from polluting prompts
  • System prompt constrains LLM to use only provided context
  • Hybrid reranking combines semantic + keyword scores for better precision
  • Source citations included in every response for transparency

← Part 1 | Part 3 →


Advertisement