decodingdatascience commited on
Commit
c32433e
·
verified ·
1 Parent(s): 28c2229

Update app.py

Browse files
Files changed (1) hide show
  1. app.py +316 -95
app.py CHANGED
@@ -1,141 +1,362 @@
1
- # for hugingface to create app.py
2
- # !pip install -U gradio pinecone llama-index llama-index-vector-stores-pinecone llama-index-readers-file pypdf
3
- from llama_index.core import VectorStoreIndex, SimpleDirectoryReader, StorageContext, Settings
4
- # --- Imports ---
5
  import logging
 
6
  import sys
7
- import gradio as gr
 
8
 
 
 
9
  from pinecone import Pinecone, ServerlessSpec
10
- from llama_index.core import VectorStoreIndex, SimpleDirectoryReader, StorageContext , Settings
11
- from llama_index.vector_stores.pinecone import PineconeVectorStore
12
- from llama_index.readers.file import PDFReader
13
- from llama_index.llms.openai import OpenAI
 
 
 
14
  from llama_index.embeddings.openai import OpenAIEmbedding
15
- # --- Logging ---
 
 
 
 
 
 
 
16
  logging.basicConfig(stream=sys.stdout, level=logging.INFO)
 
17
 
18
- #load keys for huggingface
19
- import os
 
 
20
  OPENAI_API_KEY = os.getenv("OPENAI_API_KEY")
21
  PINECONE_API_KEY = os.getenv("PINECONE_API_KEY")
 
22
 
 
 
23
 
24
- Settings.llm = OpenAI(model="gpt-4o-mini", temperature=0.2)
25
- Settings.embed_model = OpenAIEmbedding(model="text-embedding-ada-002")
26
- Settings.chunk_size = 600
27
- Settings.chunk_overlap = 200
28
 
29
- # Define a system prompt
30
- system_prompt = '''
31
- You are AYesha, the Decoding Data Science (DDS) Enterprise HR Chatbot. Answer questions exclusively using the attached DDS HR Handbook. Base all responses on the most up-to-date information available in the handbook. Only respond to queries directly related to DDS HR policies as outlined in the handbook.
 
 
32
 
33
- - If a question pertains to topics outside DDS HR policies, respond politely, clarifying that you are a human resources bot and only answer DDS HR questions.
34
- - For questions you cannot answer (e.g., requests for old policies, salary details, or confidential information), politely decline and direct the user to email connect@decodingdatascience.com.
35
- - Never answer questions about anything outside of your scope.
36
- - Persist in following these constraints for any follow-up questions.
37
- - Before answering, carefully check that the information and query are within the allowed scope. Follow chain-of-thought reasoning:
38
- 1. First, reason step-by-step whether the question is covered in the current handbook and is within HR.
39
- 2. Only after confirming, produce a final answer.
40
 
41
- Format answers as concise, professional responses. Do not wrap answers in code blocks or any special formatting.
 
 
 
 
 
 
 
42
 
43
- Output requirements:
44
- - For allowed HR questions, answer concisely based only on the latest DDS HR handbook information.
45
- - For forbidden topics, output: “I’m sorry, I can only answer questions about the latest DDS HR policies. For confidential or other queries, please email connect@decodingdatascience.com.”
 
46
 
 
 
47
 
48
- **Example 1**
49
- User: What is the leave encashment policy at DDS?
50
- Reasoning: This is an HR policy question found in the latest handbook.
51
- Final Answer: [Provide answer summarized from the latest handbook’s section on leave encashment]
52
 
53
- **Example 2**
54
- User: Can you tell me the salary range for Data Scientists?
55
- Reasoning: Salary details are confidential and not shared by this bot.
56
- Final Answer: I’m sorry, I can only answer questions about the latest DDS HR policies. For confidential or other queries, please email connect@decodingdatascience.com.
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
57
 
58
- **Example 3**
59
- User: Can you explain what DDS does as a company overall?
60
- Reasoning: This is not an HR question, so it cannot be answered.
61
- Final Answer: I’m sorry, I only answer DDS HR policy questions as outlined in the handbook.
62
 
63
- (Real-world examples should be longer and use precise wording from the handbook where appropriate.)
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
64
 
65
- **Important instructions:**
66
- - Only answer questions directly supported by the latest DDS HR handbook.
67
- - Decline politely and redirect to the provided email address for any questions outside scope or for confidential information.
68
- - Always reason before concluding. Only present the answer after checking scope and source.
69
 
70
- Remember: As AYesha, the DDS HR Enterprise Chatbot, you must never provide information outside authorized HR handbook content and always respond respectfully according to these constraints.
 
 
 
 
71
 
72
- '''
 
 
 
 
73
 
 
 
74
 
 
 
 
75
 
 
 
 
 
76
 
77
- # --- Initialize Pinecone ---
78
- pc = Pinecone(api_key=PINECONE_API_KEY)
79
- index_name = "quickstart"
80
- dimension = 1536
81
 
82
- # --- Delete index if it already exists (optional) ---
83
- existing_indexes = [idx["name"] for idx in pc.list_indexes()]
 
 
 
84
 
85
- if index_name in existing_indexes:
86
- pc.delete_index(index_name)
87
 
88
- # --- Create Pinecone index ---
89
- pc.create_index(
90
- name=index_name,
91
- dimension=dimension,
92
- metric="euclidean",
93
- spec=ServerlessSpec(cloud="aws", region="us-east-1"),
94
  )
95
 
96
- pinecone_index = pc.Index(index_name)
97
 
98
- # --- Load PDF documents from folder ---
99
- documents = SimpleDirectoryReader(
100
- input_dir="Data",
101
- required_exts=[".pdf"],
102
- file_extractor={".pdf": PDFReader()}
103
- ).load_data()
104
 
105
- if not documents:
106
- raise ValueError("No PDF documents were loaded from the 'data' folder.")
 
107
 
108
- # --- Create Vector Index ---
109
- vector_store = PineconeVectorStore(pinecone_index=pinecone_index)
110
- storage_context = StorageContext.from_defaults(vector_store=vector_store)
111
 
112
- index = VectorStoreIndex.from_documents(
113
- documents,
114
- storage_context=storage_context
115
- )
116
 
117
- # --- Query Engine ---
118
- query_engine = index.as_query_engine(system_prompt=system_prompt)
 
 
 
119
 
120
- # --- Gradio App ---
121
- def query_doc(prompt):
122
- try:
123
- response = query_engine.query(prompt)
124
- return str(response)
125
- except Exception as e:
126
- return f"Error: {str(e)}"
127
 
128
- gr.Interface(
129
- fn=query_doc,
130
- inputs=gr.Textbox(label="Ask a question about the document"),
131
- outputs=gr.Textbox(label="Answer"),
132
- title="DDS Enterprise HR Chatbot",
133
- description="Ask questions related to HR for latest Information."
134
- ).launch(share=True)
135
 
 
136
 
 
 
 
137
 
 
 
 
 
 
 
 
 
 
138
 
139
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
140
 
141
 
 
 
 
 
 
 
1
+ # DDS Enterprise HR Chatbot + Telegram Bot
2
+ # Hugging Face Spaces app.py
3
+
 
4
  import logging
5
+ import os
6
  import sys
7
+ import threading
8
+ import time
9
 
10
+ import gradio as gr
11
+ import requests
12
  from pinecone import Pinecone, ServerlessSpec
13
+
14
+ from llama_index.core import (
15
+ Settings,
16
+ SimpleDirectoryReader,
17
+ StorageContext,
18
+ VectorStoreIndex,
19
+ )
20
  from llama_index.embeddings.openai import OpenAIEmbedding
21
+ from llama_index.llms.openai import OpenAI
22
+ from llama_index.readers.file import PDFReader
23
+ from llama_index.vector_stores.pinecone import PineconeVectorStore
24
+
25
+
26
+ # ---------------------------------------------------------
27
+ # 1. Logging
28
+ # ---------------------------------------------------------
29
  logging.basicConfig(stream=sys.stdout, level=logging.INFO)
30
+ logger = logging.getLogger("dds-hr-enterprise")
31
 
32
+
33
+ # ---------------------------------------------------------
34
+ # 2. Secrets from Hugging Face
35
+ # ---------------------------------------------------------
36
  OPENAI_API_KEY = os.getenv("OPENAI_API_KEY")
37
  PINECONE_API_KEY = os.getenv("PINECONE_API_KEY")
38
+ TELEGRAM_BOT_TOKEN = os.getenv("TELEGRAM_BOT_TOKEN")
39
 
40
+ if not OPENAI_API_KEY:
41
+ raise ValueError("OPENAI_API_KEY is missing from Hugging Face Secrets.")
42
 
43
+ if not PINECONE_API_KEY:
44
+ raise ValueError("PINECONE_API_KEY is missing from Hugging Face Secrets.")
 
 
45
 
46
+ if not TELEGRAM_BOT_TOKEN:
47
+ logger.warning(
48
+ "TELEGRAM_BOT_TOKEN is missing. Web chatbot will work, "
49
+ "but Telegram integration will stay disabled."
50
+ )
51
 
 
 
 
 
 
 
 
52
 
53
+ # ---------------------------------------------------------
54
+ # 3. LlamaIndex / OpenAI settings
55
+ # ---------------------------------------------------------
56
+ Settings.llm = OpenAI(
57
+ model="gpt-4o-mini",
58
+ temperature=0.2,
59
+ api_key=OPENAI_API_KEY,
60
+ )
61
 
62
+ Settings.embed_model = OpenAIEmbedding(
63
+ model="text-embedding-ada-002",
64
+ api_key=OPENAI_API_KEY,
65
+ )
66
 
67
+ Settings.chunk_size = 600
68
+ Settings.chunk_overlap = 200
69
 
 
 
 
 
70
 
71
+ # ---------------------------------------------------------
72
+ # 4. System prompt
73
+ # ---------------------------------------------------------
74
+ SYSTEM_PROMPT = """
75
+ You are AYesha, the Decoding Data Science (DDS) Enterprise HR Chatbot.
76
+
77
+ Answer questions exclusively using the latest DDS HR Handbook content supplied
78
+ to the retrieval system.
79
+
80
+ Rules:
81
+ - Only answer questions directly related to DDS HR policies in the handbook.
82
+ - Do not answer general questions unrelated to DDS HR.
83
+ - Do not provide confidential information such as salary details.
84
+ - If the answer is not supported by the handbook, do not guess.
85
+ - For confidential, unsupported, or out-of-scope requests, respond:
86
+ "I’m sorry, I can only answer questions about the latest DDS HR policies.
87
+ For confidential or other queries, please email
88
+ connect@decodingdatascience.com."
89
+ - Keep responses concise, professional, and easy to understand.
90
+ - Base the final answer only on retrieved handbook information.
91
+ """
92
+
93
+
94
+ # ---------------------------------------------------------
95
+ # 5. Pinecone
96
+ # ---------------------------------------------------------
97
+ pc = Pinecone(api_key=PINECONE_API_KEY)
98
 
99
+ INDEX_NAME = "quickstart"
100
+ DIMENSION = 1536
 
 
101
 
102
+ existing_indexes = [idx["name"] for idx in pc.list_indexes()]
103
+ index_was_created = INDEX_NAME not in existing_indexes
104
+
105
+ if index_was_created:
106
+ logger.info("Creating Pinecone index '%s'...", INDEX_NAME)
107
+ pc.create_index(
108
+ name=INDEX_NAME,
109
+ dimension=DIMENSION,
110
+ metric="cosine",
111
+ spec=ServerlessSpec(
112
+ cloud="aws",
113
+ region="us-east-1",
114
+ ),
115
+ )
116
+
117
+ # Give Pinecone a moment to make the new index ready.
118
+ time.sleep(5)
119
+ else:
120
+ logger.info(
121
+ "Using existing Pinecone index '%s' — NOT deleting/rebuilding it.",
122
+ INDEX_NAME,
123
+ )
124
+
125
+ pinecone_index = pc.Index(INDEX_NAME)
126
+ vector_store = PineconeVectorStore(pinecone_index=pinecone_index)
127
 
 
 
 
 
128
 
129
+ # ---------------------------------------------------------
130
+ # 6. Build vectors once, or reconnect to existing vectors
131
+ # ---------------------------------------------------------
132
+ if index_was_created:
133
+ logger.info("Loading PDFs from Data/ and creating embeddings...")
134
 
135
+ documents = SimpleDirectoryReader(
136
+ input_dir="Data",
137
+ required_exts=[".pdf"],
138
+ file_extractor={".pdf": PDFReader()},
139
+ ).load_data()
140
 
141
+ if not documents:
142
+ raise ValueError("No PDF documents were loaded from the 'Data' folder.")
143
 
144
+ storage_context = StorageContext.from_defaults(
145
+ vector_store=vector_store
146
+ )
147
 
148
+ index = VectorStoreIndex.from_documents(
149
+ documents,
150
+ storage_context=storage_context,
151
+ )
152
 
153
+ logger.info("DDS HR handbook indexed successfully.")
 
 
 
154
 
155
+ else:
156
+ # Reuse the vectors already stored in Pinecone.
157
+ index = VectorStoreIndex.from_vector_store(
158
+ vector_store=vector_store
159
+ )
160
 
 
 
161
 
162
+ # ---------------------------------------------------------
163
+ # 7. Query engine shared by BOTH Gradio and Telegram
164
+ # ---------------------------------------------------------
165
+ query_engine = index.as_query_engine(
166
+ system_prompt=SYSTEM_PROMPT,
167
+ similarity_top_k=4,
168
  )
169
 
 
170
 
171
+ def query_doc(prompt: str) -> str:
172
+ """Single HR answering function used by web app and Telegram."""
173
+ if not prompt or not prompt.strip():
174
+ return "Please enter an HR policy question."
 
 
175
 
176
+ try:
177
+ response = query_engine.query(prompt.strip())
178
+ return str(response)
179
 
180
+ except Exception as exc:
181
+ logger.exception("Query error")
182
+ return "Sorry, I couldn't process your question right now. Please try again."
183
 
 
 
 
 
184
 
185
+ # ---------------------------------------------------------
186
+ # 8. Telegram integration
187
+ # ---------------------------------------------------------
188
+ def telegram_api(method: str) -> str:
189
+ return f"https://api.telegram.org/bot{TELEGRAM_BOT_TOKEN}/{method}"
190
 
 
 
 
 
 
 
 
191
 
192
+ def telegram_send_message(chat_id: int, text: str) -> None:
193
+ """Send answer back to Telegram. Split long answers if required."""
194
+ if not TELEGRAM_BOT_TOKEN:
195
+ return
 
 
 
196
 
197
+ text = str(text)
198
 
199
+ # Telegram messages have a size limit; 4000 gives us a safe margin.
200
+ for start in range(0, len(text), 4000):
201
+ chunk = text[start:start + 4000]
202
 
203
+ response = requests.post(
204
+ telegram_api("sendMessage"),
205
+ json={
206
+ "chat_id": chat_id,
207
+ "text": chunk,
208
+ },
209
+ timeout=20,
210
+ )
211
+ response.raise_for_status()
212
 
213
 
214
+ def telegram_typing(chat_id: int) -> None:
215
+ try:
216
+ requests.post(
217
+ telegram_api("sendChatAction"),
218
+ json={
219
+ "chat_id": chat_id,
220
+ "action": "typing",
221
+ },
222
+ timeout=10,
223
+ )
224
+ except Exception:
225
+ # Typing indicator is optional, so ignore failures.
226
+ pass
227
+
228
+
229
+ def telegram_polling_loop() -> None:
230
+ """Receive Telegram messages using long polling."""
231
+ if not TELEGRAM_BOT_TOKEN:
232
+ return
233
+
234
+ logger.info("Starting DDS HR Telegram bot...")
235
+
236
+ # getUpdates and webhooks cannot be used at the same time.
237
+ # Remove an old webhook if one exists.
238
+ try:
239
+ requests.post(
240
+ telegram_api("deleteWebhook"),
241
+ json={"drop_pending_updates": True},
242
+ timeout=20,
243
+ )
244
+ except Exception as exc:
245
+ logger.warning("Could not clear Telegram webhook: %s", exc)
246
+
247
+ offset = None
248
+
249
+ while True:
250
+ try:
251
+ params = {
252
+ "timeout": 45,
253
+ "allowed_updates": ["message"],
254
+ }
255
+
256
+ if offset is not None:
257
+ params["offset"] = offset
258
+
259
+ response = requests.get(
260
+ telegram_api("getUpdates"),
261
+ params=params,
262
+ timeout=55,
263
+ )
264
+ response.raise_for_status()
265
+
266
+ payload = response.json()
267
+
268
+ if not payload.get("ok"):
269
+ logger.warning("Telegram API response: %s", payload)
270
+ time.sleep(3)
271
+ continue
272
+
273
+ for update in payload.get("result", []):
274
+ offset = update["update_id"] + 1
275
+
276
+ message = update.get("message")
277
+ if not message:
278
+ continue
279
+
280
+ chat_id = message.get("chat", {}).get("id")
281
+ text = (message.get("text") or "").strip()
282
+
283
+ if not chat_id or not text:
284
+ continue
285
+
286
+ logger.info("Telegram question received from chat_id=%s", chat_id)
287
+
288
+ # Telegram commands
289
+ if text.startswith("/start"):
290
+ telegram_send_message(
291
+ chat_id,
292
+ "Welcome to AYesha — DDS Enterprise HR Chatbot.\n\n"
293
+ "Ask me a question about the latest DDS HR policies.",
294
+ )
295
+ continue
296
+
297
+ if text.startswith("/help"):
298
+ telegram_send_message(
299
+ chat_id,
300
+ "Ask a normal HR question, for example:\n"
301
+ "• What is the annual leave policy?\n"
302
+ "• What is the probation policy?\n"
303
+ "• How do I request leave?",
304
+ )
305
+ continue
306
+
307
+ if text.startswith("/"):
308
+ telegram_send_message(
309
+ chat_id,
310
+ "Please type your HR policy question as a normal message.",
311
+ )
312
+ continue
313
+
314
+ telegram_typing(chat_id)
315
+
316
+ answer = query_doc(text)
317
+ telegram_send_message(chat_id, answer)
318
+
319
+ except Exception as exc:
320
+ logger.exception("Telegram polling error: %s", exc)
321
+ time.sleep(5)
322
+
323
+
324
+ def start_telegram_bot() -> None:
325
+ """Run Telegram polling without blocking the Gradio web application."""
326
+ if not TELEGRAM_BOT_TOKEN:
327
+ logger.warning("Telegram bot not started: TELEGRAM_BOT_TOKEN missing.")
328
+ return
329
+
330
+ bot_thread = threading.Thread(
331
+ target=telegram_polling_loop,
332
+ daemon=True,
333
+ name="telegram-bot",
334
+ )
335
+ bot_thread.start()
336
+ logger.info("Telegram bot background thread started.")
337
+
338
+
339
+ # Start Telegram before Gradio blocks the main thread.
340
+ start_telegram_bot()
341
+
342
+
343
+ # ---------------------------------------------------------
344
+ # 9. Gradio web app
345
+ # ---------------------------------------------------------
346
+ demo = gr.Interface(
347
+ fn=query_doc,
348
+ inputs=gr.Textbox(
349
+ label="Ask a question about DDS HR policies",
350
+ placeholder="Example: What is the annual leave policy?",
351
+ ),
352
+ outputs=gr.Textbox(label="Answer"),
353
+ title="DDS Enterprise HR Chatbot",
354
+ description="Ask questions based on the latest DDS HR Handbook.",
355
+ )
356
 
357
 
358
+ if __name__ == "__main__":
359
+ demo.launch(
360
+ server_name="0.0.0.0",
361
+ server_port=7860,
362
+ )