khalil commited on
Commit
317cf00
·
verified ·
1 Parent(s): a7f0ffa

Create app.py

Browse files
Files changed (1) hide show
  1. app.py +74 -0
app.py ADDED
@@ -0,0 +1,74 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ import os
2
+ import fitz # PyMuPDF for PDF processing
3
+ import numpy as np
4
+ from sklearn.feature_extraction.text import TfidfVectorizer
5
+ from sklearn.metrics.pairwise import cosine_similarity
6
+ import streamlit as st
7
+ from groq import Groq
8
+ from tempfile import NamedTemporaryFile
9
+
10
+ # Set up the Groq client with your API key
11
+ client = Groq(api_key="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")
12
+
13
+ # Step 1: Function to extract text from PDF
14
+ def extract_text_from_pdf(pdf_path):
15
+ doc = fitz.open(pdf_path)
16
+ text = ""
17
+ for page in doc:
18
+ text += page.get_text()
19
+ doc.close()
20
+ return text
21
+
22
+ # Step 2: Function to split extracted text into chunks for retrieval
23
+ def chunk_text(text, chunk_size=1000):
24
+ words = text.split()
25
+ chunks = []
26
+ for i in range(0, len(words), chunk_size):
27
+ chunk = " ".join(words[i:i+chunk_size])
28
+ chunks.append(chunk)
29
+ return chunks
30
+
31
+ # Step 3: Retrieve the most relevant chunk using TF-IDF and cosine similarity
32
+ def retrieve_chunk(question, chunks):
33
+ vectorizer = TfidfVectorizer().fit_transform([question] + chunks)
34
+ question_vector = vectorizer[0]
35
+ chunk_vectors = vectorizer[1:]
36
+ similarities = cosine_similarity(question_vector, chunk_vectors).flatten()
37
+ best_chunk_index = np.argmax(similarities)
38
+ return chunks[best_chunk_index]
39
+
40
+ # Step 4: Generate an answer using the Groq API's language model
41
+ def generate_answer(retrieved_text, question):
42
+ prompt = f"Based on the following text, answer the question:\n\nText: {retrieved_text}\n\nQuestion: {question}"
43
+ chat_completion = client.chat.completions.create(
44
+ messages=[{"role": "user", "content": prompt}],
45
+ model="llama3-8b-8192"
46
+ )
47
+ return chat_completion.choices[0].message.content
48
+
49
+ # Step 5: Streamlit UI for PDF upload and Q&A
50
+ def main():
51
+ st.title("PDF Question-Answer Chatbot")
52
+
53
+ uploaded_file = st.file_uploader("Upload a PDF", type="pdf")
54
+
55
+ if uploaded_file:
56
+ with NamedTemporaryFile(delete=False) as tmp_file:
57
+ tmp_file.write(uploaded_file.getvalue())
58
+ pdf_path = tmp_file.name
59
+
60
+ # Extract text from the uploaded PDF and chunk it
61
+ text = extract_text_from_pdf(pdf_path)
62
+ chunks = chunk_text(text)
63
+
64
+ question = st.text_input("Ask a question:")
65
+ if st.button("Get Answer"):
66
+ if question:
67
+ retrieved_text = retrieve_chunk(question, chunks)
68
+ answer = generate_answer(retrieved_text, question)
69
+ st.write("Answer:", answer)
70
+ else:
71
+ st.write("Please enter a question.")
72
+
73
+ if __name__ == "__main__":
74
+ main()