-
Notifications
You must be signed in to change notification settings - Fork 1
Expand file tree
/
Copy pathContextual_Chunking.py
More file actions
138 lines (103 loc) · 4.38 KB
/
Copy pathContextual_Chunking.py
File metadata and controls
138 lines (103 loc) · 4.38 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
"""
Contextual Retrieval (2025/2026 Standard)
Solves the "Lost in the Middle" problem by prepending context to each chunk
"""
import os
from pypdf import PdfReader
from pathlib import Path
from dotenv import load_dotenv
from langchain_text_splitters import RecursiveCharacterTextSplitter
from langchain_openai import ChatOpenAI
from langchain_core.prompts import ChatPromptTemplate
def extract_text_from_pdf(pdf_path):
"""Extract all text from a PDF file."""
reader = PdfReader(pdf_path)
text = ""
for page in reader.pages:
text += page.extract_text() + "\n"
return text
def create_base_chunks(text, chunk_size=800, chunk_overlap=100):
"""Create initial chunks using recursive splitter."""
text_splitter = RecursiveCharacterTextSplitter(
chunk_size=chunk_size,
chunk_overlap=chunk_overlap,
separators=["\n\n", "\n", ".", " ", ""]
)
return text_splitter.split_text(text)
def generate_context(chunk, document_name, llm):
"""
Use LLM to generate contextual information for a chunk.
This prepends context like:
[Context: From ITC Q1 2025 Financial Report, Revenue Section]
"""
prompt = ChatPromptTemplate.from_messages([
("system", "You are a document analyzer. Generate a concise 1-sentence context description for the given text chunk."),
("user", """Document: {doc_name}
Chunk:
{chunk}
Generate a brief context sentence (max 20 words) that describes what this chunk is about and where it's from. Format: "Context: [your description]"
Just return the context line, nothing else.""")
])
chain = prompt | llm
response = chain.invoke({
"doc_name": document_name,
"chunk": chunk[:500] # Only send first 500 chars to save tokens
})
return response.content.strip()
def main():
# Load environment variables
load_dotenv()
# PDF files
pdfs = [
"/Volumes/vibecoding/RAG-Complete Cook Book/ITC-August-Q1-2526.pdf",
"/Volumes/vibecoding/RAG-Complete Cook Book/ITC-October-Q2-2526.pdf"
]
print("📄 Contextual Retrieval - The 2025/2026 Standard")
print(" Solves: 'Lost in the Middle' problem")
print(" Method: LLM prepends context to each chunk")
print("=" * 70)
# Initialize LLM (using fast, cheap model)
print("\n🤖 Initializing LLM (gpt-5-nano for context generation)...")
llm = ChatOpenAI(model="gpt-5-nano", temperature=0)
print("✅ LLM ready. Processing PDFs...\n")
# Process only first PDF for demo (to save time/tokens)
pdf_path = pdfs[0]
pdf_file = Path(pdf_path)
if not pdf_file.exists():
print(f"❌ File not found: {pdf_file.name}")
return
print(f"📂 Processing: {pdf_file.name}")
# Extract text
text = extract_text_from_pdf(pdf_path)
# Create base chunks
print(" Creating base chunks...")
base_chunks = create_base_chunks(text, chunk_size=800, chunk_overlap=100)
print(f" Total chunks: {len(base_chunks)}")
# Generate contextualized versions (only for first 3 chunks to save time)
print("\n Generating contextualized versions (first 3 chunks)...")
for i in range(min(3, len(base_chunks))):
chunk = base_chunks[i]
print(f"\n{'='*70}")
print(f"CHUNK {i+1} COMPARISON")
print(f"{'='*70}")
# Show original chunk
print("\n📋 ORIGINAL CHUNK:")
preview = chunk[:300].replace('\n', ' ')
print(f" {preview}...")
print(f" Length: {len(chunk)} chars")
# Generate and show contextualized version
print("\n🎯 CONTEXTUALIZED VERSION:")
context = generate_context(chunk, pdf_file.stem, llm)
contextualized_chunk = f"{context}\n\n{chunk}"
context_preview = contextualized_chunk[:400].replace('\n', ' ')
print(f" {context_preview}...")
print(f" Length: {len(contextualized_chunk)} chars")
print(f"\n💡 Why this helps:")
print(f" Even if search query doesn't match exact words in chunk,")
print(f" the context provides searchable keywords and metadata.")
print("\n" + "=" * 70)
print("\n✨ Key Benefit:")
print(" Chunks now carry their own context, making them findable")
print(" even when key terms are in different paragraphs.")
if __name__ == "__main__":
main()