RAG Advanced - Building Production-Grade Retrieval-Augmented Generation
Table of Contents
Introduction
Retrieval-Augmented Generation (RAG) has become the standard approach for grounding Large Language Models (LLMs) in domain-specific knowledge. While basic RAG systems are straightforward to build, production-grade RAG requires sophisticated techniques to handle the nuances of real-world information retrieval.
Basic RAG limitations that advanced techniques address:
- Semantic gap: Dense embeddings miss lexical matches; sparse retrieval misses semantic similarity
- Retrieval noise: Top-k results often include irrelevant documents
- Query ambiguity: User queries may be vague, multi-faceted, or poorly formed
- Context fragmentation: Fixed-size chunks lose document structure and context
- Quality measurement: No systematic way to evaluate RAG pipeline performance
In this tutorial, you will learn advanced RAG techniques that significantly improve retrieval quality, answer accuracy, and overall system reliability. Each technique is production-tested and can be combined for maximum effectiveness.
Prerequisites
- Python 3.10 or higher
- Basic understanding of RAG architecture (embeddings, vector stores, LLMs)
- Familiarity with LangChain or similar frameworks
- An OpenAI API key (or other LLM provider)
- 8GB+ RAM recommended for local embedding models
Installation and Setup
# Core dependencies
pip install langchain langchain-openai langchain-community
pip install chromadb faiss-cpu
pip install sentence-transformers
pip install rank-bm25
For evaluation
pip install ragas
For cross-encoder reranking
pip install transformers torch
Additional utilities
pip install tiktoken numpy pandas
Setup environment:
import os
os.environ["OPENAIAPIKEY"] = "your-api-key"
from langchainopenai import ChatOpenAI, OpenAIEmbeddings
from langchain.textsplitter import RecursiveCharacterTextSplitter
from langchaincommunity.vectorstores import Chroma, FAISS
Verify setup
llm = ChatOpenAI(model="gpt-4o-mini", temperature=0)
embeddings = OpenAIEmbeddings(model="text-embedding-3-small")
print("Setup complete.")
Hybrid Search: Dense + Sparse Retrieval
Hybrid search combines the strengths of dense vector retrieval (semantic understanding) with sparse retrieval (exact keyword matching) for more robust document retrieval.
import numpy as np
from rankbm25 import BM25Okapi
from langchainopenai import OpenAIEmbeddings
from langchaincommunity.vectorstores import FAISS
from langchain.schema import Document
from typing import List, Tuple
class HybridSearchRetriever:
"""
Combines dense (embedding) and sparse (BM25) retrieval
with configurable fusion weights.
"""
def init(
self,
documents: List[Document],
embeddings,
denseweight: float = 0.6,
sparseweight: float = 0.4,
):
self.documents = documents
self.denseweight = denseweight
self.sparseweight = sparseweight
# Build dense index
self.vectorstore = FAISS.fromdocuments(documents, embeddings)
# Build sparse index (BM25)
tokenizeddocs = [doc.pagecontent.lower().split() for doc in documents]
self.bm25 = BM25Okapi(tokenizeddocs)
def normalizescores(self, scores: List[float]) -> List[float]:
"""Min-max normalize scores to [0, 1]."""
if not scores:
return scores
mins, maxs = min(scores), max(scores)
if maxs == mins:
return [0.5] len(scores)
return [(s - mins) / (maxs - mins) for s in scores]
def retrieve(self, query: str, topk: int = 10) -> List[Tuple[Document, float]]:
"""
Retrieve documents using reciprocal rank fusion of
dense and sparse results.
"""
# Dense retrieval
denseresults = self.vectorstore.similaritysearchwithscore(
query, k=topk 2
)
densedocscores = {
doc.pagecontent: 1.0 / (rank + 1)
for rank, (doc, score) in enumerate(denseresults)
}
# Sparse retrieval (BM25)
tokenizedquery = query.lower().split()
bm25scores = self.bm25.getscores(tokenizedquery)
sparseranked = sorted(
enumerate(bm25scores), key=lambda x: -x[1]
)[:topk 2]
sparsedocscores = {
self.documents[idx].pagecontent: 1.0 / (rank + 1)
for rank, (idx, score) in enumerate(sparseranked)
}
# Reciprocal Rank Fusion
allcontents = set(densedocscores.keys()) | set(sparsedocscores.keys())
fusedscores = {}
for content in allcontents:
densescore = densedocscores.get(content, 0.0)
sparsescore = sparsedocscores.get(content, 0.0)
fusedscores[content] = (
self.denseweight densescore +
self.sparseweight sparsescore
)
# Sort by fused score and return topk
sortedresults = sorted(fusedscores.items(), key=lambda x: -x[1])[:topk]
results = []
contenttodoc = {doc.pagecontent: doc for doc in self.documents}
for content, score in sortedresults:
if content in contenttodoc:
results.append((contenttodoc[content], score))
return results
Usage
documents = [
Document(pagecontent="Machine learning uses statistical methods to learn from data."),
Document(pagecontent="Python is the most popular programming language for ML."),
Document(pagecontent="Neural networks are inspired by biological brain structures."),
Document(pagecontent="The BM25 algorithm is widely used in information retrieval."),
Document(pagecontent="Transformers revolutionized natural language processing in 2017."),
]
embeddings = OpenAIEmbeddings(model="text-embedding-3-small")
hybridretriever = HybridSearchRetriever(
documents=documents,
embeddings=embeddings,
denseweight=0.6,
sparseweight=0.4,
)
results = hybridretriever.retrieve("BM25 information retrieval algorithm", topk=3)
for doc, score in results:
print(f" [{score:.3f}] {doc.pagecontent[:80]}...")
Reranking with Cross-Encoders
Cross-encoder reranking is one of the most impactful improvements you can make to a RAG system. Unlike bi-encoders (used for initial retrieval), cross-encoders jointly encode the query and document, producing much more accurate relevance scores.
from transformers import AutoModelForSequenceClassification, AutoTokenizer
import torch
from langchain.schema import Document
from typing import List, Tuple
class CrossEncoderReranker:
"""
Reranks retrieved documents using a cross-encoder model.
This provides much more accurate relevance scoring than
embedding similarity alone.
"""
def init(
self,
modelname: str = "cross-encoder/ms-marco-MiniLM-L-12-v2",
device: str = None,
batchsize: int = 32,
):
self.device = device or ("cuda" if torch.cuda.isavailable() else "cpu")
self.batchsize = batchsize
self.tokenizer = AutoTokenizer.frompretrained(modelname)
self.model = AutoModelForSequenceClassification.frompretrained(modelname)
self.model.to(self.device)
self.model.eval()
@torch.nograd()
def scorepairs(self, query: str, documents: List[str]) -> List[float]:
"""Score query-document pairs using the cross-encoder."""
scores = []
for i in range(0, len(documents), self.batchsize):
batchdocs = documents[i:i + self.batchsize]
pairs = [[query, doc] for doc in batchdocs]
inputs = self.tokenizer(
pairs,
padding=True,
truncation=True,
maxlength=512,
returntensors="pt",
).to(self.device)
outputs = self.model(inputs)
batchscores = outputs.logits.squeeze(-1).cpu().tolist()
if isinstance(batchscores, float):
batchscores = [batchscores]
scores.extend(batchscores)
return scores
def rerank(
self,
query: str,
documents: List[Document],
topk: int = 5,
scorethreshold: float = None,
) -> List[Tuple[Document, float]]:
"""Rerank documents and return topk most relevant."""
doctexts = [doc.pagecontent for doc in documents]
scores = self.scorepairs(query, doctexts)
# Pair documents with scores and sort
scoreddocs = list(zip(documents, scores))
scoreddocs.sort(key=lambda x: -x[1])
# Apply threshold filter if specified
if scorethreshold is not None:
scoreddocs = [
(doc, score) for doc, score in scoreddocs
if score >= scorethreshold
]
return scoreddocs[:topk]
Usage: Two-stage retrieval + reranking
reranker = CrossEncoderReranker()
Stage 1: Retrieve candidates (over-fetch)
initialresults = hybridretriever.retrieve(
"How do transformers work in NLP?",
topk=20 # Retrieve more than needed
)
candidatedocs = [doc for doc, in initialresults]
Stage 2: Rerank with cross-encoder
reranked = reranker.rerank(
query="How do transformers work in NLP?",
documents=candidatedocs,
topk=5,
scorethreshold=0.0,
)
print("Reranked results:")
for doc, score in reranked:
print(f" [{score:.4f}] {doc.pagecontent[:80]}...")
Query Transformation Techniques
Raw user queries are often suboptimal for retrieval. Query transformation techniques reformulate queries to improve retrieval quality.
from langchainopenai import ChatOpenAI
from langchain.prompts import ChatPromptTemplate
from typing import List
class QueryTransformer:
"""
Transforms user queries to improve retrieval quality
using multiple strategies.
"""
def init(self, llm=None):
self.llm = llm or ChatOpenAI(model="gpt-4o-mini", temperature=0.3)
def multi
query(self, query: str, numqueries: int = 3) -> List[str]:
"""
Generate multiple reformulations of the query to capture
different aspects and phrasings.
"""
prompt = ChatPromptTemplate.from
template(
"You are an expert at reformulating search queries. "
"Generate {numqueries} different versions of the following "
"question that would help retrieve relevant documents. "
"Each version should approach the question from a different angle.\n\n"
"Original question: {query}\n\n"
"Provide each query on a new line, numbered 1-{numqueries}."
)
response = self.llm.invoke(
prompt.formatmessages(query=query, numqueries=numqueries)
)
# Parse numbered queries
queries = []
for line in response.content.strip().split("\n"):
line = line.strip()
if line and line[0].isdigit():
# Remove numbering
cleaned = line.split(".", 1)[-1].strip() if "." in line else line
queries.append(cleaned)
return queries[:numqueries]
def stepback(self, query: str) -> str:
"""
Generate a more abstract 'step-back' query that captures
the broader context needed to answer the specific question.
"""
prompt = ChatPromptTemplate.fromtemplate(
"You are an expert at understanding the deeper context behind "
"questions. Given the following specific question, generate a "
"single broader, more abstract question that would help gather "
"background knowledge needed to fully answer the original.\n\n"
"Original question: {query}\n\n"
"Step-back question:"
)
response = self.llm.invoke(prompt.formatmessages(query=query))
return response.content.strip()
def hyde(self, query: str) -> str:
"""
Hypothetical Document Embeddings (HyDE): Generate a hypothetical
answer that can be used as a query for better semantic matching.
"""
prompt = ChatPromptTemplate.fromtemplate(
"Write a short, factual passage (100-150 words) that would be "
"a perfect answer to the following question. Write it as if it "
"were from a reference document, not as a direct response.\n\n"
"Question: {query}\n\n"
"Passage:"
)
response = self.llm.invoke(prompt.formatmessages(query=query))
return response.content.strip()
def decompose(self, query: str) -> List[str]:
"""
Decompose a complex query into simpler sub-queries
that can be answered independently.
"""
prompt = ChatPromptTemplate.fromtemplate(
"Break down the following complex question into 2-4 simpler "
"sub-questions that together would provide a complete answer "
"to the original question.\n\n"
"Complex question: {query}\n\n"
"Sub-questions (one per line, numbered):"
)
response = self.llm.invoke(prompt.formatmessages(query=query))
subqueries = []
for line in response.content.strip().split("\n"):
line = line.strip()
if line and line[0].isdigit():
cleaned = line.split(".", 1)[-1].strip() if "." in line else line
subqueries.append(cleaned)
return subqueries
Usage examples
transformer = QueryTransformer()
originalquery = "How does RAG handle conflicting information from multiple sources?"
Multi-query: retrieve with multiple reformulations
multiqueries = transformer.multiquery(originalquery)
print("Multi-query reformulations:")
for i, q in enumerate(multiqueries, 1):
print(f" {i}. {q}")
Step-back: get broader context
stepbackquery = transformer.stepback(originalquery)
print(f"\nStep-back query: {stepbackquery}")
HyDE: generate hypothetical document
hydedoc = transformer.hyde(originalquery)
print(f"\nHyDE document: {hydedoc[:200]}...")
Decompose: break into sub-questions
subqueries = transformer.decompose(originalquery)
print("\nDecomposed sub-queries:")
for i, q in enumerate(subqueries, 1):
print(f" {i}. {q}")
Retrieval with Multi-Query Fusion
def retrievewithmultiquery(
retriever: HybridSearchRetriever,
reranker: CrossEncoderReranker,
transformer: QueryTransformer,
query: str,
topk: int = 5,
) -> List[Tuple[Document, float]]:
"""Full retrieval pipeline with multi-query fusion and reranking."""
# Generate multiple query variants
queries = [query] + transformer.multiquery(query, numqueries=3)
# Retrieve candidates from all queries
allcandidates = {}
for q in queries:
results = retriever.retrieve(q, topk=10)
for doc, score in results:
content = doc.pagecontent
if content not in allcandidates or score > allcandidates[content][1]:
allcandidates[content] = (doc, score)
candidatedocs = [doc for doc, in allcandidates.values()]
print(f"Unique candidates from {len(queries)} queries: {len(candidatedocs)}")
# Rerank all candidates
reranked = reranker.rerank(query, candidatedocs, topk=topk)
return reranked
Parent-Child Chunking Strategy
Parent-child chunking solves the context fragmentation problem by indexing small chunks for precise retrieval while returning larger parent chunks for comprehensive context.
from langchain.schema import Document
from langchain.textsplitter import RecursiveCharacterTextSplitter
from typing import Dict, List, Optional
import uuid
import hashlib
class ParentChildChunker:
"""
Implements parent-child chunking: small chunks for retrieval,
large chunks for context in the LLM prompt.
"""
def init(
self,
parentchunksize: int = 2000,
parentchunkoverlap: int = 200,
childchunksize: int = 400,
childchunkoverlap: int = 50,
):
self.parentsplitter = RecursiveCharacterTextSplitter(
chunksize=parentchunksize,
chunkoverlap=parentchunkoverlap,
separators=["\n\n", "\n", ". ", " ", ""],
)
self.childsplitter = RecursiveCharacterTextSplitter(
chunksize=childchunksize,
chunkoverlap=childchunkoverlap,
separators=["\n\n", "\n", ". ", " ", ""],
)
self.parentstore: Dict[str, Document] = {}
self.childtoparent: Dict[str, str] = {}
def generateid(self, text: str) -> str:
"""Generate a deterministic ID from content."""
return hashlib.md5(text.encode()).hexdigest()[:12]
def processdocuments(self, documents: List[Document]) -> List[Document]:
"""
Split documents into parent and child chunks.
Returns child chunks (for indexing) with parent references.
"""
childdocuments = []
for doc in documents:
# Create parent chunks
parentchunks = self.parentsplitter.splittext(doc.pagecontent)
for parenttext in parentchunks:
parentid = self.generateid(parenttext)
parentdoc = Document(
pagecontent=parenttext,
metadata={
doc.metadata,
"chunktype": "parent",
"parentid": parentid,
}
)
self.parentstore[parentid] = parentdoc
# Create child chunks from this parent
childchunks = self.childsplitter.splittext(parenttext)
for childtext in childchunks:
childid = self.generateid(childtext)
childdoc = Document(
pagecontent=childtext,
metadata={
doc.metadata,
"chunktype": "child",
"childid": childid,
"parentid": parentid,
}
)
childdocuments.append(childdoc)
self.childtoparent[childid] = parentid
print(f"Created {len(self.parentstore)} parent chunks")
print(f"Created {len(childdocuments)} child chunks")
return childdocuments
def getparent(self, childdoc: Document) -> Optional[Document]:
"""Retrieve the parent chunk for a given child chunk."""
parentid = childdoc.metadata.get("parentid")
if parentid and parentid in self.parentstore:
return self.parentstore[parentid]
return None
def retrievewithparents(
self,
childresults: List[Document],
deduplicate: bool = True,
) -> List[Document]:
"""
Given retrieved child chunks, return their parent chunks
for richer context.
"""
seenparentids = set()
parentresults = []
for childdoc in childresults:
parentdoc = self.getparent(childdoc)
if parentdoc:
parentid = parentdoc.metadata["parentid"]
if not deduplicate or parentid not in seenparentids:
seenparentids.add(parentid)
parentresults.append(parentdoc)
return parentresults
Usage
chunker = ParentChildChunker(
parentchunksize=2000,
childchunksize=400,
)
Process documents
rawdocuments = [
Document(
pagecontent="Long document text here..." 100,
metadata={"source": "technicalmanual.pdf", "page": 1}
),
]
childdocs = chunker.processdocuments(rawdocuments)
Index child chunks in vector store
vectorstore = FAISS.fromdocuments(childdocs, embeddings)
At retrieval time: search children, return parents
query = "How to configure the authentication system?"
childresults = vectorstore.similaritysearch(query, k=5)
parentresults = chunker.retrievewithparents(childresults)
print(f"Retrieved {len(childresults)} child chunks")
print(f"Mapped to {len(parentresults)} unique parent chunks")
for parent in parentresults:
print(f" Parent [{parent.metadata['parentid']}]: "
f"{len(parent.pagecontent)} chars")
Recursive Retrieval
Recursive retrieval iteratively refines results by retrieving, analyzing, and re-querying based on gaps in the retrieved information.
from langchainopenai import ChatOpenAI
from langchain.prompts import ChatPromptTemplate
from langchain.schema import Document
from typing import List, Tuple, Dict
import json
class RecursiveRetriever:
"""
Implements recursive retrieval that iteratively refines
the search based on analysis of retrieved documents.
"""
def init(self, retriever, reranker, llm=None, maxiterations: int = 3):
self.retriever = retriever
self.reranker = reranker
self.llm = llm or ChatOpenAI(model="gpt-4o-mini", temperature=0)
self.maxiterations = maxiterations
def analyzecoverage(
self,
query: str,
retrieveddocs: List[Document],
) -> Dict:
"""Analyze whether retrieved documents sufficiently answer the query."""
context = "\n\n---\n\n".join([doc.pagecontent for doc in retrieveddocs])
prompt = ChatPromptTemplate.fromtemplate(
"Analyze whether the following retrieved documents sufficiently "
"answer the query. Identify any information gaps.\n\n"
"Query: {query}\n\n"
"Retrieved Documents:\n{context}\n\n"
"Respond in JSON format:\n"
'{{\n'
' "coveragescore": <0.0-1.0>,\n'
' "issufficient": ,\n'
' "gaps": ["gap1", "gap2"],\n'
' "followupqueries": ["query1", "query2"]\n'
'}}'
)
response = self.llm.invoke(
prompt.formatmessages(query=query, context=context[:4000])
)
try:
return json.loads(response.content)
except json.JSONDecodeError:
return {
"coveragescore": 0.5,
"issufficient": True,
"gaps": [],
"followupqueries": [],
}
def retrieve(
self,
query: str,
topk: int = 5,
) -> Tuple[List[Document], Dict]:
"""
Recursively retrieve documents until sufficient coverage
is achieved or max iterations reached.
"""
alldocuments = []
seencontents = set()
retrievallog = {"iterations": []}
currentqueries = [query]
for iteration in range(self.maxiterations):
print(f"\nIteration {iteration + 1}/{self.maxiterations}")
iterationinfo = {
"iteration": iteration + 1,
"queries": currentqueries,
"newdocs": 0,
}
# Retrieve for all current queries
for q in currentqueries:
results = self.retriever.retrieve(q, topk=topk)
for doc, score in results:
if doc.pagecontent not in seencontents:
seencontents.add(doc.pagecontent)
alldocuments.append(doc)
iterationinfo["newdocs"] += 1
print(f" Queries: {len(currentqueries)}, "
f"New docs: {iterationinfo['newdocs']}, "
f"Total docs: {len(alldocuments)}")
# Rerank all accumulated documents
reranked = self.reranker.rerank(
query=query,
documents=alldocuments,
topk=min(topk 2, len(alldocuments)),
)
topdocs = [doc for doc, in reranked]
# Analyze coverage
analysis = self.analyzecoverage(query, topdocs[:topk])
iterationinfo["coveragescore"] = analysis["coveragescore"]
iterationinfo["issufficient"] = analysis["issufficient"]
retrievallog["iterations"].append(iterationinfo)
print(f" Coverage: {analysis['coveragescore']:.2f}, "
f"Sufficient: {analysis['issufficient']}")
if analysis["issufficient"]:
break
# Generate follow-up queries for the next iteration
currentqueries = analysis.get("followupqueries", [])
if not currentqueries:
break
# Final reranking
finalresults = self.reranker.rerank(
query=query,
documents=alldocuments,
topk=topk,
)
finaldocs = [doc for doc, in finalresults]
retrievallog["totaldocumentsretrieved"] = len(alldocuments)
retrievallog["finaldocuments"] = len(finaldocs)
return finaldocs, retrievallog
Usage
recursiveretriever = RecursiveRetriever(
retriever=hybridretriever,
reranker=reranker,
maxiterations=3,
)
docs, log = recursiveretriever.retrieve(
"What are the best practices for securing a RAG pipeline in production?",
topk=5,
)
print(f"\nFinal results: {len(docs)} documents")
print(f"Iterations used: {len(log['iterations'])}")
for iteration in log["iterations"]:
print(f" Iteration {iteration['iteration']}: "
f"coverage={iteration['coveragescore']:.2f}")
Evaluation with RAGAS
RAGAS (Retrieval-Augmented Generation Assessment) provides automated evaluation metrics for RAG systems.
from ragas import evaluate
from ragas.metrics import (
faithfulness,
answerrelevancy,
contextprecision,
contextrecall,
contextrelevancy,
answercorrectness,
)
from datasets import Dataset
from typing import List, Dict
def createevaluationdataset(
questions: List[str],
groundtruths: List[str],
ragpipeline,
) -> Dataset:
"""
Create an evaluation dataset by running the RAG pipeline
on a set of questions with known answers.
"""
evaldata = {
"question": [],
"answer": [],
"contexts": [],
"groundtruth": [],
}
for question, groundtruth in zip(questions, groundtruths):
# Run the RAG pipeline
result = ragpipeline.query(question)
evaldata["question"].append(question)
evaldata["answer"].append(result["answer"])
evaldata["contexts"].append(
[doc.pagecontent for doc in result["sourcedocuments"]]
)
evaldata["groundtruth"].append(groundtruth)
return Dataset.fromdict(evaldata)
def runragasevaluation(evaldataset: Dataset) -> Dict:
"""
Run comprehensive RAGAS evaluation.
Metrics:
- faithfulness: Is the answer grounded in the retrieved context?
- answerrelevancy: Is the answer relevant to the question?
- contextprecision: Are the top-ranked contexts relevant?
- contextrecall: Does the context contain the ground truth info?
"""
results = evaluate(
evaldataset,
metrics=[
faithfulness,
answerrelevancy,
contextprecision,
contextrecall,
],
)
return results
Define evaluation questions and ground truths
evalquestions = [
"What is hybrid search in RAG?",
"How does cross-encoder reranking improve retrieval?",
"What is the parent-child chunking strategy?",
"How do you evaluate a RAG pipeline?",
]
evalgroundtruths = [
"Hybrid search combines dense vector retrieval with sparse BM25 retrieval "
"to leverage both semantic understanding and exact keyword matching.",
"Cross-encoder reranking jointly encodes the query and document to produce "
"more accurate relevance scores than bi-encoder similarity alone.",
"Parent-child chunking indexes small chunks for precise retrieval while "
"returning larger parent chunks to provide comprehensive context.",
"RAG pipelines can be evaluated using metrics like faithfulness, answer "
"relevancy, context precision, and context recall from RAGAS.",
]
Run evaluation
evaldataset = createevaluationdataset(
evalquestions, evalgroundtruths, ragpipeline
)
results = runragasevaluation(evaldataset)
print("\nRAGAS Evaluation Results:")
print(f" Faithfulness: {results['faithfulness']:.4f}")
print(f" Answer Relevancy: {results['answerrelevancy']:.4f}")
print(f" Context Precision: {results['contextprecision']:.4f}")
print(f" Context Recall: {results['contextrecall']:.4f}")
Custom evaluation function for ongoing monitoring
def evaluatesinglequery(
question: str,
answer: str,
contexts: List[str],
groundtruth: str,
) -> Dict:
"""Evaluate a single RAG query for monitoring."""
dataset = Dataset.fromdict({
"question": [question],
"answer": [answer],
"contexts": [contexts],
"groundtruth": [groundtruth],
})
results = evaluate(
dataset,
metrics=[faithfulness, answerrelevancy],
)
return {
"faithfulness": results["faithfulness"],
"answerrelevancy": results["answerrelevancy"],
}
Production RAG Pipeline
Bringing all techniques together into a production-ready pipeline.
from langchainopenai import ChatOpenAI, OpenAIEmbeddings
from langchain.prompts import ChatPromptTemplate
from langchain.schema import Document
from typing import List, Dict, Optional
import time
import logging
logging.basicConfig(level=logging.INFO)
logger = logging.getLogger(name)
class ProductionRAGPipeline:
"""
Production-grade RAG pipeline combining:
- Hybrid search (dense + sparse)
- Cross-encoder reranking
- Query transformation
- Parent-child chunking
- Comprehensive logging and metrics
"""
def init(
self,
documents: List[Document],
embedding
model: str = "text-embedding-3-small",
llmmodel: str = "gpt-4o",
rerankermodel: str = "cross-encoder/ms-marco-MiniLM-L-12-v2",
parentchunksize: int = 2000,
childchunksize: int = 400,
denseweight: float = 0.6,
sparseweight: float = 0.4,
):
self.llm = ChatOpenAI(model=llmmodel, temperature=0)
self.embeddings = OpenAIEmbeddings(model=embeddingmodel)
# Initialize chunker
self.chunker = ParentChildChunker(
parentchunksize=parentchunksize,
childchunksize=childchunksize,
)
childdocs = self.chunker.processdocuments(documents)
# Initialize hybrid retriever with child docs
self.retriever = HybridSearchRetriever(
documents=childdocs,
embeddings=self.embeddings,
denseweight=denseweight,
sparseweight=sparseweight,
)
# Initialize reranker