Add lesson: RAG (Retrieval-Augmented Generation)

This commit is contained in:
Rohit Ghumare
2026-03-31 17:23:04 +01:00
parent 40278f2158
commit 99190f2d85
4 changed files with 891 additions and 0 deletions
@@ -0,0 +1,343 @@
import math
from collections import Counter
def chunk_text(text, chunk_size=200, overlap=50):
words = text.split()
chunks = []
start = 0
while start < len(words):
end = start + chunk_size
chunk = " ".join(words[start:end])
chunks.append(chunk)
start += chunk_size - overlap
return chunks
def build_vocabulary(documents):
vocab = set()
for doc in documents:
vocab.update(doc.lower().split())
return sorted(vocab)
def compute_tf(text, vocab):
words = text.lower().split()
count = Counter(words)
total = len(words)
if total == 0:
return [0.0] * len(vocab)
return [count.get(word, 0) / total for word in vocab]
def compute_idf(documents, vocab):
n = len(documents)
idf = []
for word in vocab:
doc_count = sum(1 for doc in documents if word in doc.lower().split())
idf.append(math.log((n + 1) / (doc_count + 1)) + 1)
return idf
def tfidf_embed(text, vocab, idf):
tf = compute_tf(text, vocab)
return [t * i for t, i in zip(tf, idf)]
def cosine_similarity(a, b):
dot_product = sum(x * y for x, y in zip(a, b))
norm_a = math.sqrt(sum(x * x for x in a))
norm_b = math.sqrt(sum(x * x for x in b))
if norm_a == 0 or norm_b == 0:
return 0.0
return dot_product / (norm_a * norm_b)
def search(query_embedding, stored_embeddings, top_k=5):
scores = []
for i, emb in enumerate(stored_embeddings):
sim = cosine_similarity(query_embedding, emb)
scores.append((i, sim))
scores.sort(key=lambda x: x[1], reverse=True)
return scores[:top_k]
def build_rag_prompt(query, retrieved_chunks):
context = "\n\n---\n\n".join(
f"[Source {i+1}]\n{chunk}"
for i, chunk in enumerate(retrieved_chunks)
)
return (
"Answer the question based ONLY on the following context.\n"
"If the context doesn't contain enough information, "
"say \"I don't have enough information to answer that.\"\n\n"
f"Context:\n{context}\n\n"
f"Question: {query}\n\n"
"Answer:"
)
def simple_generate(prompt, retrieved_chunks):
query_section = prompt.lower().split("question:")[-1]
query_words = set(query_section.split())
stop_words = {"the", "a", "an", "is", "are", "was", "were", "what", "how",
"why", "when", "where", "do", "does", "for", "of", "in", "to",
"and", "or", "on", "at", "by", "it", "its", "this", "that"}
query_words = query_words - stop_words
best_sentence = ""
best_score = 0
for chunk in retrieved_chunks:
for sentence in chunk.split("."):
sentence = sentence.strip()
if len(sentence) < 10:
continue
words = set(sentence.lower().split())
overlap = len(query_words & words)
if overlap > best_score:
best_score = overlap
best_sentence = sentence
return best_sentence if best_sentence else "I don't have enough information."
class RAGPipeline:
def __init__(self, chunk_size=200, overlap=50, top_k=5):
self.chunk_size = chunk_size
self.overlap = overlap
self.top_k = top_k
self.chunks = []
self.embeddings = []
self.vocab = []
self.idf = []
self.sources = []
def index(self, documents, source_names=None):
all_chunks = []
all_sources = []
for i, doc in enumerate(documents):
doc_chunks = chunk_text(doc, self.chunk_size, self.overlap)
all_chunks.extend(doc_chunks)
name = source_names[i] if source_names else f"doc_{i}"
all_sources.extend([name] * len(doc_chunks))
self.chunks = all_chunks
self.sources = all_sources
self.vocab = build_vocabulary(all_chunks)
self.idf = compute_idf(all_chunks, self.vocab)
self.embeddings = [
tfidf_embed(chunk, self.vocab, self.idf)
for chunk in all_chunks
]
return len(all_chunks)
def query(self, question, top_k=None):
k = top_k or self.top_k
query_emb = tfidf_embed(question, self.vocab, self.idf)
results = search(query_emb, self.embeddings, k)
retrieved = []
for idx, score in results:
retrieved.append({
"chunk": self.chunks[idx],
"source": self.sources[idx],
"score": score,
"index": idx
})
chunk_texts = [r["chunk"] for r in retrieved]
prompt = build_rag_prompt(question, chunk_texts)
answer = simple_generate(prompt, chunk_texts)
return {
"question": question,
"answer": answer,
"prompt": prompt,
"retrieved": retrieved
}
SAMPLE_DOCUMENTS = [
"""Acme Corp Refund Policy.
All standard plan customers are eligible for a full refund within 30 days of purchase.
Enterprise plan customers receive an extended 60-day refund window with pro-rated refunds
calculated from the date of cancellation. Refunds are processed within 5-7 business days
and returned to the original payment method. No refunds are available after the refund
window closes. Customers must submit refund requests through the support portal or by
contacting their account manager directly. Annual subscriptions that are cancelled mid-term
will receive a pro-rated credit for the remaining months.""",
"""Acme Corp Product Overview.
Acme Corp offers three product tiers: Starter, Professional, and Enterprise.
The Starter plan includes basic features for individual users at $29 per month.
The Professional plan adds team collaboration, advanced analytics, and priority
support for $99 per month per user. The Enterprise plan includes everything in
Professional plus custom integrations, dedicated account management, SSO,
audit logs, and a 99.99% uptime SLA. Enterprise pricing is custom and starts
at $500 per month for up to 50 users. All plans include a 14-day free trial
with no credit card required.""",
"""Acme Corp Security Practices.
Acme Corp maintains SOC 2 Type II compliance and undergoes annual third-party
security audits. All data is encrypted at rest using AES-256 and in transit
using TLS 1.3. Customer data is stored in isolated tenants within AWS
us-east-1 and eu-west-1 regions. Data residency can be configured per
organization for Enterprise customers. Backups are performed every 6 hours
with 30-day retention. Acme Corp does not sell or share customer data with
third parties. Enterprise customers can request data deletion within 24 hours.
Bug bounty program available through HackerOne.""",
"""Acme Corp API Documentation.
The Acme API uses REST with JSON request and response bodies. Authentication
is via Bearer tokens issued through OAuth 2.0. Rate limits are 100 requests
per minute for Starter, 1000 for Professional, and 10000 for Enterprise.
Rate limit headers are included in every response: X-RateLimit-Limit,
X-RateLimit-Remaining, and X-RateLimit-Reset. Exceeding the rate limit
returns HTTP 429 with a Retry-After header. The API supports pagination
via cursor-based pagination using the next_cursor field. Webhooks are
available for real-time event notifications on Professional and Enterprise
plans. API versioning uses date-based versions in the URL path.""",
"""Acme Corp Uptime and Reliability.
Acme Corp guarantees 99.9% uptime for Professional plans and 99.99% uptime
for Enterprise plans. Uptime is calculated monthly excluding scheduled
maintenance windows which are announced 72 hours in advance. If uptime
falls below the guaranteed level, customers receive service credits:
10% credit for each 0.1% below the SLA threshold, up to a maximum of
30% of the monthly fee. Service credits must be requested within 30 days
of the incident. Status page updates are posted at status.acme.com
within 5 minutes of any detected incident. Post-incident reports are
published within 48 hours for any outage exceeding 15 minutes."""
]
if __name__ == "__main__":
print("=" * 60)
print("STEP 1: Document Chunking")
print("=" * 60)
sample = SAMPLE_DOCUMENTS[0]
chunks = chunk_text(sample, chunk_size=30, overlap=10)
print(f" Document length: {len(sample.split())} words")
print(f" Chunk size: 30 words, overlap: 10 words")
print(f" Number of chunks: {len(chunks)}")
for i, chunk in enumerate(chunks):
print(f"\n Chunk {i}: ({len(chunk.split())} words)")
print(f" {chunk[:100]}...")
print("\n" + "=" * 60)
print("STEP 2: TF-IDF Embedding")
print("=" * 60)
mini_docs = [
"The cat sat on the mat",
"The dog sat on the rug",
"Machine learning is a branch of artificial intelligence"
]
vocab = build_vocabulary(mini_docs)
idf = compute_idf(mini_docs, vocab)
print(f" Vocabulary size: {len(vocab)}")
print(f" Sample words and IDF scores:")
for word, score in sorted(zip(vocab, idf), key=lambda x: x[1], reverse=True)[:8]:
print(f" {word:20s} IDF={score:.3f}")
emb1 = tfidf_embed(mini_docs[0], vocab, idf)
emb2 = tfidf_embed(mini_docs[1], vocab, idf)
emb3 = tfidf_embed(mini_docs[2], vocab, idf)
print(f"\n Embedding dimensions: {len(emb1)}")
print(f" Non-zero entries in 'cat sat on mat': {sum(1 for v in emb1 if v > 0)}")
print(f" Non-zero entries in 'dog sat on rug': {sum(1 for v in emb2 if v > 0)}")
print(f" Non-zero entries in 'machine learning': {sum(1 for v in emb3 if v > 0)}")
print("\n" + "=" * 60)
print("STEP 3: Cosine Similarity")
print("=" * 60)
sim_12 = cosine_similarity(emb1, emb2)
sim_13 = cosine_similarity(emb1, emb3)
sim_23 = cosine_similarity(emb2, emb3)
print(f" 'cat on mat' vs 'dog on rug': {sim_12:.4f} (similar structure)")
print(f" 'cat on mat' vs 'machine learning': {sim_13:.4f} (unrelated)")
print(f" 'dog on rug' vs 'machine learning': {sim_23:.4f} (unrelated)")
print(f"\n As expected: similar sentences score higher.")
print("\n" + "=" * 60)
print("STEP 4: Full RAG Pipeline")
print("=" * 60)
rag = RAGPipeline(chunk_size=50, overlap=10, top_k=3)
source_names = [
"refund-policy.md",
"product-overview.md",
"security.md",
"api-docs.md",
"uptime-sla.md"
]
num_chunks = rag.index(SAMPLE_DOCUMENTS, source_names)
print(f" Indexed {len(SAMPLE_DOCUMENTS)} documents into {num_chunks} chunks")
print(f" Vocabulary size: {len(rag.vocab)} terms")
queries = [
"What is the refund policy for enterprise customers?",
"What are the API rate limits?",
"How is customer data encrypted?",
"What happens if uptime falls below the SLA?",
"How much does the Professional plan cost?"
]
for query in queries:
print(f"\n Query: {query}")
result = rag.query(query, top_k=3)
print(f" Answer: {result['answer']}")
print(f" Retrieved {len(result['retrieved'])} chunks:")
for r in result["retrieved"]:
preview = r["chunk"][:80].replace("\n", " ")
print(f" [{r['source']}] score={r['score']:.4f} | {preview}...")
print("\n" + "=" * 60)
print("STEP 5: Chunk Size Comparison")
print("=" * 60)
test_query = "What is the refund policy for enterprise customers?"
for chunk_size in [20, 50, 100, 200]:
rag_test = RAGPipeline(chunk_size=chunk_size, overlap=max(5, chunk_size // 5))
n = rag_test.index(SAMPLE_DOCUMENTS)
result = rag_test.query(test_query, top_k=3)
top_score = result["retrieved"][0]["score"] if result["retrieved"] else 0
print(f" chunk_size={chunk_size:>3d}: {n:>3d} chunks, "
f"top_score={top_score:.4f}, "
f"answer_len={len(result['answer'])}")
print("\n" + "=" * 60)
print("STEP 6: Prompt Inspection")
print("=" * 60)
result = rag.query("What encryption does Acme use?", top_k=2)
prompt_lines = result["prompt"].split("\n")
print(f" Prompt length: {len(result['prompt'])} chars")
print(f" Prompt lines: {len(prompt_lines)}")
print(f"\n First 5 lines of generated prompt:")
for line in prompt_lines[:5]:
print(f" {line}")
print(f" ...")
print(f" Last 3 lines of generated prompt:")
for line in prompt_lines[-3:]:
print(f" {line}")
print("\n" + "=" * 60)
print("SUMMARY")
print("=" * 60)
print(" RAG pipeline: Query -> Embed -> Search -> Augment -> Generate")
print(f" Documents indexed: {len(SAMPLE_DOCUMENTS)}")
print(f" Total chunks: {num_chunks}")
print(f" Vocabulary size: {len(rag.vocab)}")
print(f" Embedding dimensions: {len(rag.vocab)}")
print(" Similarity metric: cosine similarity")
print(" Embedding method: TF-IDF")
print("\n In production, replace TF-IDF with neural embeddings")
print(" (text-embedding-3-small) and the simple generator with")
print(" an actual LLM API call. The pipeline stays the same.")
+419
View File
@@ -0,0 +1,419 @@
# RAG (Retrieval-Augmented Generation)
> Your LLM knows everything up to its training cutoff. It knows nothing about your company's docs, your codebase, or last week's meeting notes. RAG solves this by retrieving relevant documents and stuffing them into the prompt. It's the most deployed pattern in production AI. If you build one thing from this course, build a RAG pipeline.
**Type:** Build
**Languages:** Python
**Prerequisites:** Phase 10 (LLMs from Scratch), Phase 11 Lessons 01-05
**Time:** ~90 minutes
## The Problem
You build a chatbot for your company. A customer asks "What's the refund policy for enterprise plans?" The LLM responds with a generic answer about typical SaaS refund policies. The actual policy, buried in a 200-page internal wiki, says enterprise customers get a 60-day window with pro-rated refunds. The LLM has never seen this document. It cannot know what it was not trained on.
Fine-tuning is one solution. Take the LLM, train it on your internal docs, and deploy the updated model. This works but has serious problems. Fine-tuning costs thousands of dollars in compute. The model becomes stale the moment a document changes. You have no way to know which source the model drew from. And if the company acquires another product line next month, you fine-tune again.
RAG is the other solution. Leave the model untouched. When a question comes in, search your document store for relevant passages, paste them into the prompt before the question, and let the model answer using those passages as context. The document store can be updated in minutes. You can see exactly which documents were retrieved. The model itself never changes. This is why RAG is the dominant pattern in production: it's cheaper, fresher, more auditable, and works with any LLM.
## The Concept
### The RAG Pattern
The entire pattern fits in four steps:
```mermaid
graph LR
Q["User Query"] --> R["Retrieve"]
R --> A["Augment Prompt"]
A --> G["Generate"]
G --> Ans["Answer"]
subgraph "Retrieve"
R --> Embed["Embed query"]
Embed --> Search["Search vector store"]
Search --> TopK["Return top-k chunks"]
end
subgraph "Augment"
TopK --> Format["Format chunks into prompt"]
Format --> Combine["Combine with user question"]
end
subgraph "Generate"
Combine --> LLM["LLM generates answer"]
LLM --> Cite["Answer grounded in retrieved docs"]
end
```
Query -> Retrieve -> Augment prompt -> Generate. Every RAG system follows this pattern. The differences between production RAG systems are in the details of each step: how you chunk, how you embed, how you search, and how you construct the prompt.
### Why RAG Beats Fine-Tuning
| Concern | Fine-tuning | RAG |
|---------|------------|-----|
| Cost | $1,000-$100,000+ per training run | $0.01-$0.10 per query (embedding + LLM) |
| Freshness | Stale until retrained | Updated in minutes by re-indexing docs |
| Auditability | Cannot trace answer to source | Can show exact retrieved passages |
| Hallucination | Still hallucinates freely | Grounded in retrieved documents |
| Data privacy | Training data baked into weights | Documents stay in your vector store |
Fine-tuning changes the model's weights permanently. RAG changes the model's context temporarily. For most applications, temporary context is what you want.
The one case where fine-tuning wins: when you need the model to adopt a specific style, tone, or reasoning pattern that cannot be achieved through prompting alone. For factual knowledge retrieval, RAG wins every time.
### Embedding Models
An embedding model converts text into a dense vector. Similar texts produce vectors that are close together in this high-dimensional space. "How do I reset my password?" and "I need to change my password" produce nearly identical vectors despite sharing few words. "The cat sat on the mat" produces a very different vector.
Common embedding models:
| Model | Dimensions | Provider | Notes |
|-------|-----------|----------|-------|
| text-embedding-3-small | 1536 | OpenAI | Best price/performance for most use cases |
| text-embedding-3-large | 3072 | OpenAI | Higher accuracy, 2x the dimensions |
| text-embedding-ada-002 | 1536 | OpenAI | Legacy, replaced by 3-small |
| all-MiniLM-L6-v2 | 384 | Open source (Sentence Transformers) | Runs locally, fast, good for prototyping |
| voyage-3 | 1024 | Voyage AI | Strong on code and technical content |
| Cohere embed-v3 | 1024 | Cohere | Good multilingual support |
For this lesson, we build our own simple embedding using TF-IDF. Not because TF-IDF is what production systems use, but because it makes the concept concrete: text goes in, a vector comes out, similar texts produce similar vectors.
### Vector Similarity
Given two vectors, how do you measure similarity? Three options:
**Cosine similarity**: the cosine of the angle between two vectors. Ranges from -1 (opposite) to 1 (identical). Ignores magnitude, only cares about direction. This is the default for RAG.
```
cosine_sim(a, b) = dot(a, b) / (||a|| * ||b||)
```
**Dot product**: the raw inner product. Larger vectors get higher scores. Useful when magnitude carries information (longer documents might be more relevant).
```
dot(a, b) = sum(a_i * b_i)
```
**L2 (Euclidean) distance**: straight-line distance in the vector space. Smaller distance = more similar. Sensitive to magnitude differences.
```
L2(a, b) = sqrt(sum((a_i - b_i)^2))
```
Cosine similarity is the standard. It handles documents of different lengths gracefully because it normalizes by magnitude. When someone says "vector search," they almost always mean cosine similarity.
### Chunking Strategies
Documents are too long to embed as single vectors. A 50-page PDF might produce a terrible embedding because it contains dozens of topics. Instead, you split documents into chunks and embed each chunk separately.
**Fixed-size chunking**: split every N tokens. Simple and predictable. A 512-token chunk with 50-token overlap means chunk 1 is tokens 0-511, chunk 2 is tokens 462-973, and so on. The overlap ensures you do not split a sentence at an unlucky boundary.
**Semantic chunking**: split at natural boundaries. Paragraphs, sections, or markdown headers. Each chunk is a coherent unit of meaning. More complex to implement but produces better retrieval.
**Recursive chunking**: try to split at the largest boundary first (section headers). If a section is still too large, split at paragraph boundaries. If a paragraph is still too large, split at sentence boundaries. This is the LangChain RecursiveCharacterTextSplitter approach and it works well in practice.
Chunk size matters more than people think:
- Too small (64-128 tokens): each chunk lacks context. "It increased 15% last quarter" means nothing without knowing what "it" refers to.
- Too large (2048+ tokens): each chunk covers multiple topics, diluting relevance. When you search for revenue data, you get a chunk that's 10% about revenue and 90% about headcount.
- Sweet spot (256-512 tokens): enough context to be self-contained, focused enough to be relevant.
Most production RAG systems use 256-512 token chunks with 50-token overlap. Anthropic's RAG guidelines recommend this range.
### Vector Databases
Once you have embeddings, you need somewhere to store and search them. Options:
| Database | Type | Best for |
|----------|------|----------|
| FAISS | Library (in-process) | Prototyping, small to medium datasets |
| Chroma | Lightweight DB | Local development, small deployments |
| Pinecone | Managed service | Production without ops overhead |
| Weaviate | Open source DB | Self-hosted production |
| pgvector | Postgres extension | Already using Postgres |
| Qdrant | Open source DB | High-performance self-hosted |
For this lesson, we build a simple in-memory vector store. It stores vectors in a list and does brute-force cosine similarity search. This is equivalent to FAISS with a flat index. It scales to maybe 100,000 vectors before getting slow. Production systems use approximate nearest neighbor (ANN) algorithms like HNSW to search millions of vectors in milliseconds.
### The Full Pipeline
```mermaid
graph TD
subgraph "Indexing (offline)"
D["Documents"] --> C["Chunk"]
C --> E["Embed each chunk"]
E --> S["Store vectors + text"]
end
subgraph "Querying (online)"
Q["User query"] --> QE["Embed query"]
QE --> VS["Vector search (top-k)"]
VS --> P["Build prompt with chunks"]
P --> LLM["LLM generates answer"]
end
S -.->|"same vector space"| VS
```
The indexing phase runs once per document (or when documents update). The querying phase runs on every user request. In production, indexing might process millions of documents over hours. Querying must respond in under a second.
### Real Numbers
Most production RAG systems use these parameters:
- **k = 5 to 10** retrieved chunks per query
- **Chunk size = 256 to 512 tokens** with 50-token overlap
- **Context budget**: 2,500-5,000 tokens of retrieved content per query
- **Total prompt**: ~8,000-16,000 tokens (system prompt + retrieved chunks + conversation history + user query)
- **Embedding dimension**: 384-3072 depending on model
- **Indexing throughput**: 100-1,000 documents per second with API embeddings
- **Query latency**: 50-200ms for retrieval, 500-3000ms for generation
## Build It
### Step 1: Document Chunking
```python
def chunk_text(text, chunk_size=200, overlap=50):
words = text.split()
chunks = []
start = 0
while start < len(words):
end = start + chunk_size
chunk = " ".join(words[start:end])
chunks.append(chunk)
start += chunk_size - overlap
return chunks
```
### Step 2: TF-IDF Embeddings
We build a simple embedding function. TF-IDF (Term Frequency-Inverse Document Frequency) is not a neural embedding, but it converts text to vectors in a way that captures word importance. Frequent words in a document get higher TF. Rare words across the corpus get higher IDF. The product gives a vector where important, distinctive words have high values.
```python
import math
from collections import Counter
def build_vocabulary(documents):
vocab = set()
for doc in documents:
vocab.update(doc.lower().split())
return sorted(vocab)
def compute_tf(text, vocab):
words = text.lower().split()
count = Counter(words)
total = len(words)
return [count.get(word, 0) / total for word in vocab]
def compute_idf(documents, vocab):
n = len(documents)
idf = []
for word in vocab:
doc_count = sum(1 for doc in documents if word in doc.lower().split())
idf.append(math.log((n + 1) / (doc_count + 1)) + 1)
return idf
def tfidf_embed(text, vocab, idf):
tf = compute_tf(text, vocab)
return [t * i for t, i in zip(tf, idf)]
```
### Step 3: Cosine Similarity Search
```python
def cosine_similarity(a, b):
dot = sum(x * y for x, y in zip(a, b))
norm_a = math.sqrt(sum(x * x for x in a))
norm_b = math.sqrt(sum(x * x for x in b))
if norm_a == 0 or norm_b == 0:
return 0.0
return dot / (norm_a * norm_b)
def search(query_embedding, stored_embeddings, top_k=5):
scores = []
for i, emb in enumerate(stored_embeddings):
sim = cosine_similarity(query_embedding, emb)
scores.append((i, sim))
scores.sort(key=lambda x: x[1], reverse=True)
return scores[:top_k]
```
### Step 4: Prompt Construction
This is where the "augmented" in RAG happens. Take the retrieved chunks, format them into a prompt, and ask the LLM to answer based on the provided context.
```python
def build_rag_prompt(query, retrieved_chunks):
context = "\n\n---\n\n".join(
f"[Source {i+1}]\n{chunk}"
for i, chunk in enumerate(retrieved_chunks)
)
return f"""Answer the question based ONLY on the following context.
If the context doesn't contain enough information, say "I don't have enough information to answer that."
Context:
{context}
Question: {query}
Answer:"""
```
### Step 5: The Complete RAG Pipeline
```python
class RAGPipeline:
def __init__(self):
self.chunks = []
self.embeddings = []
self.vocab = []
self.idf = []
def index(self, documents):
all_chunks = []
for doc in documents:
all_chunks.extend(chunk_text(doc))
self.chunks = all_chunks
self.vocab = build_vocabulary(all_chunks)
self.idf = compute_idf(all_chunks, self.vocab)
self.embeddings = [
tfidf_embed(chunk, self.vocab, self.idf)
for chunk in all_chunks
]
def query(self, question, top_k=5):
query_emb = tfidf_embed(question, self.vocab, self.idf)
results = search(query_emb, self.embeddings, top_k)
retrieved = [(self.chunks[i], score) for i, score in results]
prompt = build_rag_prompt(
question, [chunk for chunk, _ in retrieved]
)
return prompt, retrieved
```
### Step 6: Generation (simulated)
In production, this is where you call the LLM API. For this lesson, we simulate generation by extracting the most relevant sentence from the retrieved context.
```python
def simple_generate(prompt, retrieved_chunks):
query_words = set(prompt.lower().split("question:")[-1].split())
best_sentence = ""
best_score = 0
for chunk in retrieved_chunks:
for sentence in chunk.split("."):
sentence = sentence.strip()
if not sentence:
continue
words = set(sentence.lower().split())
overlap = len(query_words & words)
if overlap > best_score:
best_score = overlap
best_sentence = sentence
return best_sentence if best_sentence else "I don't have enough information."
```
## Use It
With a real embedding model and LLM, the code barely changes:
```python
from openai import OpenAI
client = OpenAI()
def embed(text):
response = client.embeddings.create(
model="text-embedding-3-small",
input=text
)
return response.data[0].embedding
def generate(prompt):
response = client.chat.completions.create(
model="gpt-4o-mini",
messages=[{"role": "user", "content": prompt}],
temperature=0
)
return response.choices[0].message.content
```
Or with Anthropic:
```python
import anthropic
client = anthropic.Anthropic()
def generate(prompt):
response = client.messages.create(
model="claude-sonnet-4-20250514",
max_tokens=1024,
messages=[{"role": "user", "content": prompt}]
)
return response.content[0].text
```
The pipeline is the same. Swap the embedding function. Swap the generation function. The retrieval logic, chunking, prompt construction -- all identical regardless of which models you use.
For vector storage at scale, replace the brute-force search with a proper vector database:
```python
import chromadb
client = chromadb.Client()
collection = client.create_collection("my_docs")
collection.add(
documents=chunks,
ids=[f"chunk_{i}" for i in range(len(chunks))]
)
results = collection.query(
query_texts=["What is the refund policy?"],
n_results=5
)
```
Chroma handles the embedding internally (it uses all-MiniLM-L6-v2 by default) and stores the vectors in a local database. Same pattern, different plumbing.
## Ship It
This lesson produces:
- `outputs/prompt-rag-architect.md` -- a prompt for designing RAG systems for specific use cases
- `outputs/skill-rag-pipeline.md` -- a skill that teaches agents how to build and debug RAG pipelines
## Exercises
1. Replace the TF-IDF embeddings with a simple bag-of-words approach (binary: 1 if word present, 0 if not). Compare retrieval quality on the sample documents. TF-IDF should outperform because it weights rare words higher.
2. Experiment with chunk sizes: try 50, 100, 200, and 500 words on the same document set. For each size, run the same 5 queries and count how many return a relevant chunk in the top-3. Find the sweet spot where retrieval quality peaks.
3. Add metadata to each chunk (source document name, chunk position). Modify the prompt template to include source attribution so the LLM cites its sources.
4. Implement a simple evaluation: given 10 question-answer pairs, run each question through the RAG pipeline, and measure what percentage of retrieved chunks contain the answer. This is retrieval recall at k.
5. Build a conversation-aware RAG pipeline: maintain a history of the last 3 exchanges and include them in the prompt alongside the retrieved chunks. Test with follow-up questions like "What about enterprise?" after asking about pricing.
## Key Terms
| Term | What people say | What it actually means |
|------|----------------|----------------------|
| RAG | "AI that reads your docs" | Retrieve relevant documents, paste them into the prompt, and generate an answer grounded in those documents |
| Embedding | "Convert text to numbers" | A dense vector representation of text where similar meanings produce similar vectors |
| Vector database | "Search engine for AI" | A data store optimized for storing vectors and finding the nearest neighbors by similarity |
| Chunking | "Split docs into pieces" | Breaking documents into smaller segments (typically 256-512 tokens) so each can be embedded and retrieved independently |
| Cosine similarity | "How similar are two vectors" | The cosine of the angle between two vectors; 1 = identical direction, 0 = orthogonal, -1 = opposite |
| Top-k retrieval | "Get the k best matches" | Return the k most similar chunks to the query from the vector store |
| Context window | "How much text the LLM can see" | The maximum number of tokens the LLM can process in a single request; retrieved chunks must fit within this |
| Augmented generation | "Answer using given context" | Generating a response using retrieved documents as context rather than relying solely on trained knowledge |
| TF-IDF | "Word importance scoring" | Term Frequency times Inverse Document Frequency; weights words by how distinctive they are within a corpus |
| Indexing | "Preparing docs for search" | The offline process of chunking, embedding, and storing documents so they can be searched at query time |
## Further Reading
- Lewis et al., "Retrieval-Augmented Generation for Knowledge-Intensive NLP Tasks" (2020) -- the original RAG paper from Facebook AI Research that formalized the retrieve-then-generate pattern
- Anthropic's RAG documentation (docs.anthropic.com) -- practical guidelines for chunk sizes, prompt construction, and evaluation
- Pinecone Learning Center, "What is RAG?" -- clear visual explanations of the RAG pipeline with production considerations
- Sentence-BERT: Reimers & Gurevych (2019) -- the paper behind the all-MiniLM embedding models, showing how to train bi-encoders for semantic similarity
@@ -0,0 +1,60 @@
---
name: prompt-rag-architect
description: Design RAG systems for specific use cases with concrete architecture decisions
phase: 11
lesson: 6
---
You are a RAG system architect. Given a use case description, design a complete RAG pipeline with specific, justified decisions for every component.
Gather these inputs before designing:
1. **Document corpus**: What are the documents? (PDFs, wiki pages, code, chat logs, emails)
2. **Corpus size**: How many documents? Total token count?
3. **Update frequency**: How often do documents change?
4. **Query patterns**: What kinds of questions will users ask?
5. **Latency requirements**: How fast must the response be?
6. **Accuracy requirements**: Is a wrong answer worse than no answer?
For each component, choose and justify:
**Chunking strategy:**
- Fixed 256 tokens + 50 overlap: default for most use cases
- Semantic (paragraph/section boundaries): for well-structured docs like wikis
- Recursive (headers -> paragraphs -> sentences): for mixed-format corpora
- Code-aware (function/class boundaries): for codebases
**Embedding model:**
- text-embedding-3-small (1536d): best value for general text
- text-embedding-3-large (3072d): when retrieval accuracy is critical
- all-MiniLM-L6-v2 (384d): when data cannot leave the network
- voyage-code-2: for code-heavy corpora
**Vector store:**
- In-memory (FAISS flat): prototyping, < 100K vectors
- FAISS HNSW: single-machine, < 10M vectors, low latency
- pgvector: already using Postgres, < 5M vectors
- Pinecone/Weaviate/Qdrant: production scale, > 1M vectors
**Retrieval parameters:**
- top_k = 3-5: for focused, single-topic questions
- top_k = 5-10: for broad questions or multi-hop reasoning
- top_k = 10-20: when using a reranker to filter down
**Prompt template:**
- Direct context injection: for simple Q&A
- Citation-aware template: when users need to verify sources
- Conversational template: when maintaining chat history
**Common failure modes to warn about:**
- Chunk boundary splits: important info spread across two chunks, neither retrieved
- Vocabulary mismatch: user says "cancel" but docs say "terminate subscription"
- Stale index: documents updated but embeddings not re-generated
- Context overflow: too many retrieved chunks exceed the model's context window
- Hallucination despite context: model ignores retrieved docs and generates from training data
For each design, provide:
- Architecture diagram (as ASCII or description)
- Estimated cost per 1000 queries
- Expected latency breakdown (embed query + vector search + LLM generation)
- Top 3 risks and mitigations
@@ -0,0 +1,69 @@
---
name: skill-rag-pipeline
description: Build and debug RAG pipelines from first principles
version: 1.0.0
phase: 11
lesson: 6
tags: [rag, retrieval, embeddings, vector-search, llm-engineering]
---
# RAG Pipeline Pattern
Every RAG system follows this pattern:
```
documents -> chunk -> embed -> store
query -> embed -> search(top_k) -> build_prompt -> generate
```
Indexing happens once per document. Querying happens on every user request.
## When to use RAG
- The LLM needs access to private or recent documents
- Fine-tuning is too expensive or too slow to update
- You need to cite sources for answers
- The knowledge base changes frequently
## When NOT to use RAG
- The answer is general knowledge the LLM already has
- The task is creative (writing, brainstorming) not factual
- You need the model to adopt a specific reasoning style (use fine-tuning)
## Implementation checklist
1. Chunk documents into 256-512 token segments with 50-token overlap
2. Embed each chunk using a consistent embedding model
3. Store embeddings in a vector database with the original text
4. At query time, embed the user's question with the same model
5. Retrieve top-k (5-10) most similar chunks via cosine similarity
6. Build a prompt: system instruction + retrieved context + user question
7. Generate the answer, grounding it in the retrieved context
8. Return the answer with source references
## Common mistakes
- Using different embedding models for indexing and querying (vectors are incompatible)
- Chunks too small (lose context) or too large (dilute relevance)
- Not including overlap between chunks (splits sentences at boundaries)
- Forgetting to re-index when documents change
- Returning retrieved chunks to the user without generating a coherent answer
- Not setting temperature=0 for factual RAG queries (higher temperature = more hallucination)
## Debugging retrieval
If the right chunks are not being retrieved:
1. Print the query embedding and verify it's non-zero
2. Check cosine similarities manually for a known-relevant chunk
3. Try rephrasing the query to match document vocabulary
4. Verify the embedding model matches between index and query time
5. Check if the relevant content was lost during chunking
## Production parameters
- Chunk size: 256-512 tokens
- Overlap: 50 tokens (10-20% of chunk size)
- Top-k: 5-10 for most use cases
- Temperature: 0 for factual answers
- Embedding model: text-embedding-3-small (cost effective) or text-embedding-3-large (higher accuracy)