zainab-zaman-chadhar commited on
Commit
d9fe47d
Β·
verified Β·
1 Parent(s): 0b3fc85

Update app.py

Browse files
Files changed (1) hide show
  1. app.py +135 -78
app.py CHANGED
@@ -1,91 +1,148 @@
 
1
  import streamlit as st
2
- import os
3
  from PyPDF2 import PdfReader
4
- import docx
5
- from pptx import Presentation
6
- import openpyxl
7
- from sentence_transformers import SentenceTransformer
8
- import faiss
9
- from transformers import AutoTokenizer, AutoModelForCausalLM
10
- import torch
11
 
12
- # -----------------------------
13
- # Streamlit UI
14
- # -----------------------------
15
- st.title("πŸ“˜ Free Document QA App (Hugging Face)")
16
 
17
- uploaded_file = st.file_uploader("Upload a document (PDF, DOCX, PPTX, XLSX)")
 
 
 
 
18
 
19
- if uploaded_file:
20
- st.write(f"Uploaded: {uploaded_file.name}")
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
21
 
22
- # -----------------------------
23
- # Extract text
24
- # -----------------------------
25
- context = ""
26
- if uploaded_file.type == "application/pdf":
27
- reader = PdfReader(uploaded_file)
28
- for page in reader.pages:
29
- context += page.extract_text() + "\n"
30
- elif uploaded_file.type in ["application/vnd.openxmlformats-officedocument.wordprocessingml.document"]:
31
- doc = docx.Document(uploaded_file)
32
- for para in doc.paragraphs:
33
- context += para.text + "\n"
34
- elif uploaded_file.type in ["application/vnd.openxmlformats-officedocument.presentationml.presentation"]:
35
- prs = Presentation(uploaded_file)
36
- for slide in prs.slides:
37
- for shape in slide.shapes:
38
- if hasattr(shape, "text"):
39
- context += shape.text + "\n"
40
- elif uploaded_file.type in ["application/vnd.openxmlformats-officedocument.spreadsheetml.sheet"]:
41
- wb = openpyxl.load_workbook(uploaded_file)
42
- for sheet in wb.worksheets:
43
- for row in sheet.iter_rows(values_only=True):
44
- context += " ".join([str(cell) for cell in row if cell]) + "\n"
45
- else:
46
- st.warning("Unsupported file type.")
47
- context = ""
48
 
49
- # -----------------------------
50
- # Split into chunks
51
- # -----------------------------
52
- CHUNK_SIZE = 500 # characters
53
- chunks = [context[i:i+CHUNK_SIZE] for i in range(0, len(context), CHUNK_SIZE)]
54
 
55
- # -----------------------------
56
- # Embeddings
57
- # -----------------------------
58
- embed_model = SentenceTransformer('all-MiniLM-L6-v2')
59
- embeddings = embed_model.encode(chunks)
60
 
61
- # FAISS index
62
- index = faiss.IndexFlatL2(embeddings.shape[1])
63
- index.add(embeddings)
 
 
 
 
 
64
 
65
- # -----------------------------
66
- # Load HF model for answering
67
- # -----------------------------
68
- st.info("Loading model... (this may take a minute)")
69
- hf_model_name = "TheBloke/guanaco-7B-GPTQ" # works on CPU or GPU
70
- tokenizer = AutoTokenizer.from_pretrained(hf_model_name)
71
- model = AutoModelForCausalLM.from_pretrained(hf_model_name, device_map="auto")
72
 
73
- # -----------------------------
74
- # Ask question
75
- # -----------------------------
76
- question = st.text_input("Ask a question based on the document:")
77
 
78
- if question:
79
- # Search top relevant chunk
80
- q_emb = embed_model.encode([question])
81
- D, I = index.search(q_emb, k=2) # top 2 chunks
82
- relevant_text = "\n".join([chunks[i] for i in I[0]])
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
83
 
84
- # Generate answer
85
- prompt = f"Answer the question using ONLY the context below.\n\nContext:\n{relevant_text}\n\nQuestion: {question}\nAnswer:"
86
- inputs = tokenizer(prompt, return_tensors="pt")
87
- with st.spinner("πŸ€– Generating answer..."):
88
- outputs = model.generate(**inputs, max_new_tokens=300)
89
- answer = tokenizer.decode(outputs[0], skip_special_tokens=True)
90
- st.success("βœ… Answer:")
91
- st.write(answer)
 
 
1
+ # app.py
2
  import streamlit as st
 
3
  from PyPDF2 import PdfReader
4
+ from langchain.text_splitter import RecursiveCharacterTextSplitter
5
+ from langchain.vectorstores import FAISS
6
+ from langchain.embeddings import HuggingFaceEmbeddings
7
+ from langchain.chains.question_answering import load_qa_chain
8
+ from langchain.llms import HuggingFacePipeline
9
+ from transformers import AutoModelForCausalLM, AutoTokenizer, pipeline
10
+ import tempfile
11
 
12
+ # --- Streamlit page config ---
13
+ st.set_page_config(page_title="πŸ“š Multi PDF Chatbot", layout="wide", page_icon="πŸ€–")
 
 
14
 
15
+ # --- Header ---
16
+ st.markdown(
17
+ "<h1 style='text-align: center; color: #2F4F4F;'>πŸ“š Multi-PDF Chat Agent πŸ€–</h1>",
18
+ unsafe_allow_html=True
19
+ )
20
 
21
+ # --- Sidebar Styling ---
22
+ st.markdown(
23
+ """
24
+ <style>
25
+ section[data-testid="stSidebar"] {
26
+ background-color: #111827;
27
+ padding: 30px 20px;
28
+ }
29
+ .profile-img-container {
30
+ display: flex;
31
+ justify-content: center;
32
+ margin-bottom: 15px;
33
+ }
34
+ .profile-img-container img {
35
+ border-radius: 12px;
36
+ height: 100px;
37
+ width: 100px;
38
+ object-fit: cover;
39
+ border: 2px solid #4ade80;
40
+ box-shadow: 0 0 10px rgba(0,0,0,0.5);
41
+ }
42
+ .upload-section {
43
+ background-color: #1f2937;
44
+ padding: 20px;
45
+ border-radius: 12px;
46
+ margin-top: 20px;
47
+ margin-bottom: 30px;
48
+ }
49
+ .upload-section h4 {
50
+ color: #facc15;
51
+ margin-bottom: 15px;
52
+ }
53
+ .footer {
54
+ text-align: center;
55
+ font-size: 13px;
56
+ color: #9ca3af;
57
+ }
58
+ .footer a {
59
+ color: #facc15;
60
+ text-decoration: none;
61
+ }
62
+ .footer a:hover {
63
+ color: #ffffff;
64
+ }
65
+ </style>
66
+ """,
67
+ unsafe_allow_html=True
68
+ )
69
 
70
+ # --- Sidebar Layout ---
71
+ with st.sidebar:
72
+ # Profile image
73
+ col1, col2, col3 = st.columns([1, 2, 1])
74
+ with col2:
75
+ st.image("assets/img/main.png", width=100)
76
+
77
+ # Upload PDFs
78
+ st.markdown('<div class="upload-section">', unsafe_allow_html=True)
79
+ st.markdown("#### πŸ“ Upload PDF Files")
80
+ pdf_docs = st.file_uploader("Drag and drop your PDFs here", accept_multiple_files=True, label_visibility="collapsed")
81
+ st.markdown('</div>', unsafe_allow_html=True)
82
+
83
+ # Footer
84
+ st.markdown(
85
+ """
86
+ <div class="footer">
87
+ Built with ❀️ by <a href="https://github.com/Danish7861" target="_blank">Danish Shahzad</a>
88
+ </div>
89
+ """,
90
+ unsafe_allow_html=True
91
+ )
 
 
 
 
92
 
93
+ # --- Main Logic ---
 
 
 
 
94
 
95
+ # Store FAISS vector store in session state
96
+ if "vectorstore" not in st.session_state:
97
+ st.session_state.vectorstore = None
 
 
98
 
99
+ if pdf_docs:
100
+ if st.button("πŸ“€ Submit & Process"):
101
+ with st.spinner("Processing PDFs..."):
102
+ all_text = ""
103
+ for pdf_file in pdf_docs:
104
+ pdf_reader = PdfReader(pdf_file)
105
+ for page in pdf_reader.pages:
106
+ all_text += page.extract_text() + "\n"
107
 
108
+ # Split text into chunks
109
+ splitter = RecursiveCharacterTextSplitter(chunk_size=1000, chunk_overlap=100)
110
+ chunks = splitter.split_text(all_text)
 
 
 
 
111
 
112
+ # Create embeddings
113
+ embeddings = HuggingFaceEmbeddings(model_name="sentence-transformers/all-MiniLM-L6-v2")
114
+ st.session_state.vectorstore = FAISS.from_texts(chunks, embeddings)
115
+ st.success("βœ… PDF content indexed successfully!")
116
 
117
+ # --- Load local LLM ---
118
+ @st.cache_resource(show_spinner=False)
119
+ def load_local_llm():
120
+ tokenizer = AutoTokenizer.from_pretrained("TheBloke/guanaco-7B-GGML")
121
+ model = AutoModelForCausalLM.from_pretrained("TheBloke/guanaco-7B-GGML")
122
+ pipe = pipeline("text-generation", model=model, tokenizer=tokenizer, max_length=512)
123
+ return HuggingFacePipeline(pipeline=pipe)
124
+
125
+ llm = load_local_llm()
126
+
127
+ # --- Question input ---
128
+ user_question = st.text_input("πŸ” Ask something from your uploaded PDFs:")
129
+ if user_question:
130
+ if st.session_state.vectorstore is None:
131
+ st.warning("⚠️ Please upload and process PDFs first.")
132
+ else:
133
+ with st.spinner("Thinking... πŸ’­"):
134
+ relevant_docs = st.session_state.vectorstore.similarity_search(user_question)
135
+ chain = load_qa_chain(llm, chain_type="stuff")
136
+ answer = chain.run(input_documents=relevant_docs, question=user_question)
137
+ st.success("Answer:")
138
+ st.write(answer)
139
 
140
+ # --- Fixed footer ---
141
+ st.markdown(
142
+ """
143
+ <div style="position: fixed; bottom: 0; width: 100%; background-color: #2F4F4F; padding: 10px; color: white; text-align: center; font-size: 13px;">
144
+ πŸ“„ Multi-PDF Chatbot | Powered by LangChain, FAISS & Local AI
145
+ </div>
146
+ """,
147
+ unsafe_allow_html=True
148
+ )