Build a RAG Chatbot from Scratch — Part 2: Retrieval and Prompt Assembly
In this part, we query our vector store and construct prompts that ground the LLM in retrieved documents.
Step 1: Semantic Retrieval
# app/retrieval.py
from app.embeddings import get_embedding
from app.vector_store import get_or_create_collection
def retrieve(query: str, k: int = 5, threshold: float = 0.7) -> list[dict]:
"""Retrieve the k most relevant chunks for a query."""
query_embedding = get_embedding(query)
collection = get_or_create_collection()
results = collection.query(
query_embeddings=[query_embedding],
n_results=k
)
chunks = []
for i, doc in enumerate(results["documents"][0]):
distance = results["distances"][0][i]
similarity = 1 - distance # Cosine distance → similarity
if similarity < threshold:
continue
chunks.append({
"content": doc,
"similarity": round(similarity, 3),
"metadata": results["metadatas"][0][i]
})
return chunks
Threshold filtering prevents irrelevant chunks from being included. A similarity below 0.7 means the chunk is too different from the question to be useful.
Step 2: Prompt Assembly
# app/prompts.py
SYSTEM_PROMPT = """You are a helpful assistant that answers questions based ONLY on the provided context.
Rules:
- If the answer is in the context, cite the source
- If the context doesn't contain the answer, say "I don't have enough information to answer this"
- Do not use outside knowledge
- Keep answers concise and factual
- Include relevant code snippets or examples when present in the context
"""
def build_rag_prompt(query: str, chunks: list[dict]) -> str:
"""Assemble retrieved chunks into a single prompt."""
context_parts = []
for i, chunk in enumerate(chunks):
source = chunk["metadata"].get("source", "unknown")
context_parts.append(
f"[Source {i+1}: {source}]\n{chunk['content']}"
)
context = "\n\n---\n\n".join(context_parts)
return context, SYSTEM_PROMPT
Why a system prompt? It constrains the LLM to only use provided context, preventing hallucinations.
Step 3: Generate Answer
# app/generation.py
from openai import OpenAI
from app.config import settings
from app.retrieval import retrieve
from app.prompts import build_rag_prompt
client = OpenAI(api_key=settings.openai_api_key)
def answer_question(query: str) -> dict:
"""Full RAG pipeline: retrieve → prompt → generate."""
# Retrieve relevant chunks
chunks = retrieve(query)
if not chunks:
return {
"answer": "No relevant documents found to answer this question.",
"sources": []
}
# Build prompt
context, system_prompt = build_rag_prompt(query, chunks)
# Generate response
response = client.chat.completions.create(
model=settings.model,
messages=[
{"role": "system", "content": system_prompt},
{"role": "user", "content": f"Context:\n{context}\n\nQuestion: {query}"}
],
temperature=0.3, # Low temp = more factual
max_tokens=1000
)
return {
"answer": response.choices[0].message.content,
"sources": [
{"source": c["metadata"]["source"], "similarity": c["similarity"]}
for c in chunks
],
"tokens_used": response.usage.total_tokens
}
Step 4: Q&A API Endpoint
# app/api/chat.py
from fastapi import APIRouter
from pydantic import BaseModel
from app.generation import answer_question
router = APIRouter(prefix="/chat", tags=["chat"])
class Question(BaseModel):
query: str
max_sources: int = 5
@router.post("/")
def ask(question: Question):
return answer_question(question.query)
Step 5: Improving Retrieval with Reranking
For better accuracy, add a lightweight reranking step:
def rerank_chunks(query: str, chunks: list[dict]) -> list[dict]:
"""Use keyword overlap as a cheap reranking signal."""
query_words = set(query.lower().split())
for chunk in chunks:
chunk_words = set(chunk["content"].lower().split())
overlap = len(query_words & chunk_words) / len(query_words)
chunk["keyword_score"] = round(overlap, 3)
chunk["combined_score"] = chunk["similarity"] * 0.7 + overlap * 0.3
return sorted(chunks, key=lambda c: c["combined_score"], reverse=True)
This combines semantic similarity (70%) with keyword overlap (30%) for a hybrid score.
Verification
# Ingest sample documents first
curl -X POST http://localhost:8000/ingest/pdf -F "file=@docs/company_handbook.pdf"
# Ask a question
curl -X POST http://localhost:8000/chat/ \
-H "Content-Type: application/json" \
-d '{"query": "What is the vacation policy?"}'
# Response:
# {
# "answer": "Employees receive 20 days of paid vacation per year...",
# "sources": [
# {"source": "company_handbook.pdf", "similarity": 0.92}
# ],
# "tokens_used": 450
# }
Summary
- Semantic retrieval finds relevant chunks using cosine similarity
- Threshold filtering (0.7+) prevents irrelevant context from polluting prompts
- System prompt constrains LLM to use only provided context
- Hybrid reranking combines semantic + keyword scores for better precision
- Source citations included in every response for transparency
Advertisement