Building a Python RAG System from Scratch
- How does retrieval decide what matches?
- Cosine similarity. Two embeddings that point the same way score near 1.
- What does Part 1 depend on?
- NumPy and an OpenAI-compatible embeddings and chat API. No LangChain.
- What does Part 2 add?
- PDF parsing, chunking, a FAISS index on disk, a tool-calling agent, and a Streamlit page to upload files and ask questions.
- Where is the architecture?
- The companion article is RAG Architecture and Overview.
RAG implementation does not have to be a black box. This guide is split into two parts. First, a pure Python implementation of the core retrieval-augmented pipeline without external frameworks. Second, a production-ready visual PDF Q&A app built using LangChain, FAISS, and Streamlit. The stages these files implement are the ingestion and query pipeline in RAG Architecture and Overview.
Part 1: Pure Python RAG from scratch
1. Mathematical foundation: cosine similarity
How does a computer determine if two sentences mean the same thing? It calculates the cosine of the angle between their vector embeddings:
Scores range from −1 to 1. Values closer to 1 indicate a smaller angle in vector space and higher semantic similarity.
2. Building a generic embedding class
First, create a base class and use OpenAI’s API to transform raw text into numerical vectors:
import os
import numpy as np
from typing import List
from openai import OpenAI
class BaseEmbeddings:
"""Base class for text embeddings"""
def __init__(self, path: str = '', is_api: bool = True):
self.path = path
self.is_api = is_api
def get_embedding(self, text: str, model: str) -> List[float]:
raise NotImplementedError
@classmethod
def cosine_similarity(cls, vector1: List[float], vector2: List[float]) -> float:
"""Calculate cosine similarity between two vectors"""
dot_product = np.dot(vector1, vector2)
magnitude = np.linalg.norm(vector1) * np.linalg.norm(vector2)
if not magnitude:
return 0.0
return float(dot_product / magnitude)
class OpenAIEmbedding(BaseEmbeddings):
"""OpenAI Embedding Implementation"""
def __init__(self, path: str = '', is_api: bool = True):
super().__init__(path, is_api)
if self.is_api:
self.client = OpenAI(
api_key=os.getenv("OPENAI_API_KEY"),
base_url=os.getenv("OPENAI_BASE_URL", "https://api.openai.com/v1")
)
def get_embedding(self, text: str, model: str = "text-embedding-3-small") -> List[float]:
text = text.replace("\n", " ")
return self.client.embeddings.create(input=[text], model=model).data[0].embedding
3. Implementing a lightweight in-memory vector store
Next, build an in-memory vector store to hold document embeddings and search for top matches:
class VectorStore:
"""Simple vector storage and retrieval class"""
def __init__(self, document: List[str] = None):
self.document = document if document else []
self.vectors = []
def get_vector(self, embedding_model: BaseEmbeddings):
"""Batch-generate embeddings for stored documents"""
self.vectors = [embedding_model.get_embedding(doc) for doc in self.document]
return self.vectors
def query(self, query: str, embedding_model: BaseEmbeddings, k: int = 1) -> List[str]:
"""Retrieve top-k document chunks closest to the query"""
query_vector = embedding_model.get_embedding(query)
similarities = [
BaseEmbeddings.cosine_similarity(query_vector, vec)
for vec in self.vectors
]
top_k_indices = np.argsort(similarities)[-k:][::-1]
return [self.document[idx] for idx in top_k_indices]
4. Tying it all together
Finally, connect retrieval with model generation:
PROMPT_TEMPLATE = """Answer the question based on the reference paragraphs below. If the reference is irrelevant, use your general knowledge.
Reference:
{context}
Question: {question}
Helpful Answer:"""
def run_mini_rag(question: str, documents: List[str]) -> str:
# 1. Initialize model and vector store
emb_model = OpenAIEmbedding()
vector_store = VectorStore(documents)
vector_store.get_vector(emb_model)
# 2. Retrieve top matching context
matched_context = vector_store.query(question, emb_model, k=1)[0]
# 3. Construct prompt & call LLM
full_prompt = PROMPT_TEMPLATE.format(context=matched_context, question=question)
client = OpenAI(api_key=os.getenv("OPENAI_API_KEY"))
response = client.chat.completions.create(
model="gpt-4o-mini",
messages=[{"role": "user", "content": full_prompt}]
)
return response.choices[0].message.content
# Test Run
if __name__ == "__main__":
docs = [
"Machine learning is a branch of artificial intelligence focused on learning from data.",
"To switch Kernels in JupyterLab: Open the notebook, click the Kernel dropdown menu in the top right corner, and select your target virtual environment."
]
ans = run_mini_rag("How do I change the Kernel in Jupyter?", docs)
print("AI Response:", ans)
Part 2: LangChain, FAISS, and Streamlit
In production, you do not need to rebuild these primitives. Using LangChain and Streamlit, you can put together a complete multi-PDF interactive system in a few dozen lines of code:
import os
import streamlit as st
from dotenv import load_dotenv
from PyPDF2 import PdfReader
from langchain.text_splitter import RecursiveCharacterTextSplitter
from langchain_community.vectorstores import FAISS
from langchain_community.embeddings import DashScopeEmbeddings
from langchain.tools.retriever import create_retriever_tool
from langchain.agents import AgentExecutor, create_tool_calling_agent
from langchain.chat_models import init_chat_model
from langchain_core.prompts import ChatPromptTemplate
load_dotenv(override=True)
# 1. PDF Extraction and Text Chunking
def read_pdfs(pdf_docs):
text = ""
for pdf in pdf_docs:
pdf_reader = PdfReader(pdf)
for page in pdf_reader.pages:
text += page.extract_text() or ""
return text
def get_text_chunks(text):
# Recursive splitter preserves paragraph context
text_splitter = RecursiveCharacterTextSplitter(chunk_size=1000, chunk_overlap=200)
return text_splitter.split_text(text)
# 2. Vector DB Generation using FAISS
def create_vector_store(text_chunks):
embeddings = DashScopeEmbeddings(model="text-embedding-v1")
vectorstore = FAISS.from_texts(text_chunks, embedding=embeddings)
vectorstore.save_local("faiss_db")
# 3. LangChain Agent Pipeline
def execute_rag_agent(user_question):
embeddings = DashScopeEmbeddings(model="text-embedding-v1")
new_db = FAISS.load_local("faiss_db", embeddings, allow_dangerous_deserialization=True)
retriever = new_db.as_retriever()
# Wrap retriever as an Agent tool
retriever_tool = create_retriever_tool(
retriever,
"pdf_extractor",
"Searches and extracts context from uploaded PDF documents."
)
llm = init_chat_model("deepseek-chat", model_provider="deepseek")
prompt = ChatPromptTemplate.from_messages([
("system", "You are an AI assistant. Answer questions thoroughly based on the provided context. State if the answer is not in the context."),
("placeholder", "{chat_history}"),
("human", "{input}"),
("placeholder", "{agent_scratchpad}"),
])
agent = create_tool_calling_agent(llm, [retriever_tool], prompt)
agent_executor = AgentExecutor(agent=agent, tools=[retriever_tool], verbose=True)
response = agent_executor.invoke({"input": user_question})
return response['output']
# 4. Streamlit Frontend UI
def main():
st.set_page_config(page_title="PDF Smart Assistant", layout="wide")
st.header("📄 RAG-Powered PDF Q&A System")
user_question = st.text_input("Ask a question about your documents:")
if user_question:
if os.path.exists("faiss_db"):
with st.spinner("Searching context & generating response..."):
answer = execute_rag_agent(user_question)
st.write("**Answer:**", answer)
else:
st.error("Please upload and process a PDF document first!")
with st.sidebar:
st.title("Document Upload")
pdf_docs = st.file_uploader("Upload PDF Files", accept_multiple_files=True, type=['pdf'])
if st.button("Process Documents"):
with st.spinner("Parsing & Vectorizing..."):
raw_text = read_pdfs(pdf_docs)
chunks = get_text_chunks(raw_text)
create_vector_store(chunks)
st.success("Indexing complete! Ask away.")
if __name__ == "__main__":
main()
Key takeaway
Part 1 strips away abstractions to demonstrate how RAG works under the hood via cosine similarity and prompt engineering. Part 2 shows how open-source libraries can be combined into a working frontend application for real-world deployment.
The model that reads the retrieved chunks can be a local one. Check whether it fits the machine with the RunLocalModel checker before you point the agent at it.