Learnix-AI-Lab commited on
Commit
c3d3fc9
·
verified ·
1 Parent(s): 8ea5325

Upload 2 files

Browse files
Files changed (2) hide show
  1. app.py +575 -0
  2. requirements.txt +6 -0
app.py ADDED
@@ -0,0 +1,575 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ import os
2
+ import re
3
+ import math
4
+ import tempfile
5
+ from typing import Tuple
6
+
7
+ import gradio as gr
8
+ import torch
9
+ from transformers import AutoTokenizer, AutoModelForCausalLM
10
+
11
+ from pypdf import PdfReader
12
+ from docx import Document
13
+
14
+
15
+ # =========================
16
+ # Model Configuration
17
+ # =========================
18
+
19
+ MODEL_ID = os.getenv("MODEL_ID", "Qwen/Qwen2.5-0.5B-Instruct")
20
+
21
+ MAX_INPUT_CHARS = 12000
22
+ MAX_NEW_TOKENS = 700
23
+
24
+
25
+ # =========================
26
+ # Load Local Small LLM
27
+ # =========================
28
+
29
+ print(f"Loading model: {MODEL_ID}")
30
+
31
+ tokenizer = AutoTokenizer.from_pretrained(MODEL_ID)
32
+
33
+ model = AutoModelForCausalLM.from_pretrained(
34
+ MODEL_ID,
35
+ torch_dtype=torch.float32,
36
+ device_map="cpu",
37
+ low_cpu_mem_usage=True,
38
+ )
39
+
40
+ model.eval()
41
+
42
+ print("Model loaded successfully.")
43
+
44
+
45
+ # =========================
46
+ # File Reading Functions
47
+ # =========================
48
+
49
+ def read_pdf(file_path: str) -> str:
50
+ text = ""
51
+ reader = PdfReader(file_path)
52
+
53
+ for i, page in enumerate(reader.pages):
54
+ page_text = page.extract_text() or ""
55
+ text += f"\n\n--- Page {i + 1} ---\n{page_text}"
56
+
57
+ return text.strip()
58
+
59
+
60
+ def read_docx(file_path: str) -> str:
61
+ doc = Document(file_path)
62
+ paragraphs = []
63
+
64
+ for para in doc.paragraphs:
65
+ if para.text.strip():
66
+ paragraphs.append(para.text.strip())
67
+
68
+ return "\n".join(paragraphs).strip()
69
+
70
+
71
+ def read_txt(file_path: str) -> str:
72
+ with open(file_path, "r", encoding="utf-8", errors="ignore") as f:
73
+ return f.read().strip()
74
+
75
+
76
+ def extract_text_from_file(file) -> str:
77
+ if file is None:
78
+ return ""
79
+
80
+ file_path = file.name
81
+ ext = os.path.splitext(file_path)[1].lower()
82
+
83
+ try:
84
+ if ext == ".pdf":
85
+ return read_pdf(file_path)
86
+
87
+ elif ext == ".docx":
88
+ return read_docx(file_path)
89
+
90
+ elif ext in [".txt", ".md"]:
91
+ return read_txt(file_path)
92
+
93
+ else:
94
+ return "Unsupported file format. Please upload PDF, DOCX, TXT, or MD."
95
+ except Exception as e:
96
+ return f"Error reading file: {str(e)}"
97
+
98
+
99
+ # =========================
100
+ # Text Utility
101
+ # =========================
102
+
103
+ def clean_text(text: str) -> str:
104
+ text = text.replace("\x00", " ")
105
+ text = re.sub(r"\n{3,}", "\n\n", text)
106
+ text = re.sub(r"[ \t]{2,}", " ", text)
107
+ return text.strip()
108
+
109
+
110
+ def limit_text(text: str, max_chars: int = MAX_INPUT_CHARS) -> str:
111
+ text = clean_text(text)
112
+
113
+ if len(text) <= max_chars:
114
+ return text
115
+
116
+ beginning = text[: int(max_chars * 0.65)]
117
+ ending = text[-int(max_chars * 0.35):]
118
+
119
+ return (
120
+ beginning
121
+ + "\n\n[... middle part shortened because report is long ...]\n\n"
122
+ + ending
123
+ )
124
+
125
+
126
+ def estimate_confidence(report_text: str) -> Tuple[int, str]:
127
+ words = report_text.split()
128
+ word_count = len(words)
129
+
130
+ score = 50
131
+
132
+ if word_count > 500:
133
+ score += 10
134
+ if word_count > 1200:
135
+ score += 10
136
+ if re.search(r"\bmethodology\b|\bmethod\b", report_text, re.I):
137
+ score += 8
138
+ if re.search(r"\bresult\b|\bresults\b|\banalysis\b", report_text, re.I):
139
+ score += 8
140
+ if re.search(r"\bconclusion\b", report_text, re.I):
141
+ score += 6
142
+ if re.search(r"\breference\b|\breferences\b|\bcitation\b", report_text, re.I):
143
+ score += 5
144
+ if re.search(r"\bobjective\b|\bgoal\b|\baim\b", report_text, re.I):
145
+ score += 5
146
+
147
+ if word_count < 250:
148
+ score -= 15
149
+
150
+ score = max(20, min(95, score))
151
+
152
+ if score >= 80:
153
+ level = "Strong defense readiness"
154
+ elif score >= 65:
155
+ level = "Moderate defense readiness"
156
+ elif score >= 50:
157
+ level = "Basic defense readiness"
158
+ else:
159
+ level = "Needs more preparation"
160
+
161
+ return score, level
162
+
163
+
164
+ # =========================
165
+ # Local LLM Function
166
+ # =========================
167
+
168
+ def ask_local_llm(prompt: str) -> str:
169
+ messages = [
170
+ {
171
+ "role": "system",
172
+ "content": (
173
+ "You are Assignment Defense AI. "
174
+ "Your job is to help students defend their assignments in front of teachers. "
175
+ "Use simple English. If useful, explain in Bangla-English style. "
176
+ "Be practical, clear, and exam-focused. "
177
+ "Do not invent facts that are not in the report."
178
+ ),
179
+ },
180
+ {
181
+ "role": "user",
182
+ "content": prompt,
183
+ },
184
+ ]
185
+
186
+ try:
187
+ text = tokenizer.apply_chat_template(
188
+ messages,
189
+ tokenize=False,
190
+ add_generation_prompt=True,
191
+ )
192
+ except Exception:
193
+ text = (
194
+ "System: You are Assignment Defense AI.\n"
195
+ f"User: {prompt}\n"
196
+ "Assistant:"
197
+ )
198
+
199
+ inputs = tokenizer(
200
+ text,
201
+ return_tensors="pt",
202
+ truncation=True,
203
+ max_length=4096,
204
+ )
205
+
206
+ with torch.no_grad():
207
+ outputs = model.generate(
208
+ **inputs,
209
+ max_new_tokens=MAX_NEW_TOKENS,
210
+ temperature=0.4,
211
+ do_sample=True,
212
+ top_p=0.9,
213
+ repetition_penalty=1.12,
214
+ pad_token_id=tokenizer.eos_token_id,
215
+ )
216
+
217
+ generated_tokens = outputs[0][inputs["input_ids"].shape[-1]:]
218
+
219
+ response = tokenizer.decode(
220
+ generated_tokens,
221
+ skip_special_tokens=True,
222
+ )
223
+
224
+ return response.strip()
225
+
226
+
227
+ # =========================
228
+ # Main AI Features
229
+ # =========================
230
+
231
+ def generate_defense_pack(file, manual_text: str, difficulty: str, language_style: str):
232
+ uploaded_text = extract_text_from_file(file)
233
+ manual_text = manual_text.strip() if manual_text else ""
234
+
235
+ if uploaded_text and manual_text:
236
+ report_text = uploaded_text + "\n\nAdditional Notes:\n" + manual_text
237
+ elif uploaded_text:
238
+ report_text = uploaded_text
239
+ elif manual_text:
240
+ report_text = manual_text
241
+ else:
242
+ return (
243
+ "Please upload a report or paste your assignment text.",
244
+ "",
245
+ "",
246
+ "",
247
+ "",
248
+ "",
249
+ )
250
+
251
+ report_text = limit_text(report_text)
252
+
253
+ confidence_score, confidence_level = estimate_confidence(report_text)
254
+
255
+ prompt = f"""
256
+ You are preparing a student for assignment defense/viva.
257
+
258
+ Language style: {language_style}
259
+ Difficulty level: {difficulty}
260
+
261
+ Here is the student's report:
262
+
263
+ {report_text}
264
+
265
+ Create a complete defense preparation pack with these sections:
266
+
267
+ 1. Report Summary
268
+ - Explain the assignment in simple words.
269
+ - Mention the main topic, objective, method, and result if available.
270
+
271
+ 2. 12 Viva Questions
272
+ - Questions should be realistic and teacher-like.
273
+ - Include easy, medium, and challenging questions.
274
+
275
+ 3. Easy Answers
276
+ - Give short, clear answers for each viva question.
277
+ - Answers should sound natural for a student.
278
+
279
+ 4. Defense Speech
280
+ - Write a 1 to 2 minute formal speech for presenting this assignment.
281
+ - Make it confident but not overdramatic.
282
+
283
+ 5. Weak Points Teacher May Ask
284
+ - Identify possible weak areas in the report.
285
+ - Give safe answer strategy for each weak point.
286
+
287
+ 6. Final Preparation Tips
288
+ - Give practical tips before viva.
289
+ """
290
+
291
+ ai_output = ask_local_llm(prompt)
292
+
293
+ confidence_report = f"""
294
+ # Confidence Score
295
+
296
+ **Score:** {confidence_score}/100
297
+ **Level:** {confidence_level}
298
+
299
+ ## Meaning
300
+ This score is estimated from your report structure, length, and presence of important academic sections.
301
+
302
+ ## How to improve
303
+ - Understand the objective clearly.
304
+ - Memorize the methodology, not the full report.
305
+ - Prepare 5–10 key terms from your assignment.
306
+ - Practice explaining the report in 60 seconds.
307
+ - Be honest if you do not know an answer.
308
+ """
309
+
310
+ summary = ai_output
311
+ viva_questions = extract_section(ai_output, "12 Viva Questions", "Easy Answers")
312
+ easy_answers = extract_section(ai_output, "Easy Answers", "Defense Speech")
313
+ speech = extract_section(ai_output, "Defense Speech", "Weak Points")
314
+ weak_points = extract_section(ai_output, "Weak Points", "Final Preparation")
315
+
316
+ return (
317
+ summary,
318
+ viva_questions,
319
+ easy_answers,
320
+ speech,
321
+ weak_points,
322
+ confidence_report,
323
+ )
324
+
325
+
326
+ def extract_section(text: str, start_keyword: str, end_keyword: str) -> str:
327
+ try:
328
+ pattern = rf"(?is){re.escape(start_keyword)}(.*?){re.escape(end_keyword)}"
329
+ match = re.search(pattern, text)
330
+
331
+ if match:
332
+ return match.group(1).strip()
333
+
334
+ return "Section generated inside the full defense pack. Please check the Full Defense Pack tab."
335
+ except Exception:
336
+ return "Section extraction failed. Please check the Full Defense Pack tab."
337
+
338
+
339
+ def evaluate_practice_answer(question: str, student_answer: str, report_context: str):
340
+ if not question.strip() or not student_answer.strip():
341
+ return "Please provide both the viva question and your answer."
342
+
343
+ report_context = limit_text(report_context or "", 4000)
344
+
345
+ prompt = f"""
346
+ You are a strict but helpful viva teacher.
347
+
348
+ Report context:
349
+ {report_context}
350
+
351
+ Viva question:
352
+ {question}
353
+
354
+ Student answer:
355
+ {student_answer}
356
+
357
+ Evaluate the answer.
358
+
359
+ Give:
360
+ 1. Score out of 10
361
+ 2. What was good
362
+ 3. What was weak
363
+ 4. Better answer
364
+ 5. One short tip for the student
365
+
366
+ Use simple English.
367
+ """
368
+
369
+ return ask_local_llm(prompt)
370
+
371
+
372
+ def create_custom_questions(file, manual_text: str, topic_focus: str, number_of_questions: int):
373
+ uploaded_text = extract_text_from_file(file)
374
+ manual_text = manual_text.strip() if manual_text else ""
375
+
376
+ report_text = uploaded_text + "\n\n" + manual_text
377
+ report_text = limit_text(report_text)
378
+
379
+ if not report_text.strip():
380
+ return "Please upload or paste your report first."
381
+
382
+ prompt = f"""
383
+ Create {number_of_questions} viva questions from this report.
384
+
385
+ Focus area: {topic_focus if topic_focus else "overall assignment"}
386
+
387
+ For each question, include:
388
+ - Question
389
+ - Short easy answer
390
+ - Difficulty: Easy/Medium/Hard
391
+
392
+ Report:
393
+ {report_text}
394
+ """
395
+
396
+ return ask_local_llm(prompt)
397
+
398
+
399
+ # =========================
400
+ # Gradio UI
401
+ # =========================
402
+
403
+ custom_css = """
404
+ .gradio-container {
405
+ max-width: 1100px !important;
406
+ margin: auto !important;
407
+ }
408
+ .main-title {
409
+ text-align: center;
410
+ font-size: 34px;
411
+ font-weight: 800;
412
+ margin-bottom: 8px;
413
+ }
414
+ .sub-title {
415
+ text-align: center;
416
+ font-size: 16px;
417
+ opacity: 0.85;
418
+ margin-bottom: 25px;
419
+ }
420
+ """
421
+
422
+ with gr.Blocks(css=custom_css, theme=gr.themes.Soft()) as demo:
423
+ gr.HTML(
424
+ """
425
+ <div class="main-title">Assignment Defense AI</div>
426
+ <div class="sub-title">
427
+ Upload your report, get viva questions, easy answers, defense speech, and confidence score.
428
+ </div>
429
+ """
430
+ )
431
+
432
+ with gr.Row():
433
+ with gr.Column(scale=1):
434
+ file_input = gr.File(
435
+ label="Upload Report",
436
+ file_types=[".pdf", ".docx", ".txt", ".md"],
437
+ )
438
+
439
+ manual_text = gr.Textbox(
440
+ label="Or paste assignment text / notes",
441
+ placeholder="Paste your assignment text here...",
442
+ lines=10,
443
+ )
444
+
445
+ difficulty = gr.Radio(
446
+ choices=["Easy", "Medium", "Hard"],
447
+ value="Medium",
448
+ label="Viva Difficulty",
449
+ )
450
+
451
+ language_style = gr.Radio(
452
+ choices=[
453
+ "Simple English",
454
+ "Bangla-English Mixed",
455
+ "Formal Academic English",
456
+ ],
457
+ value="Simple English",
458
+ label="Answer Style",
459
+ )
460
+
461
+ generate_btn = gr.Button(
462
+ "Generate Defense Pack",
463
+ variant="primary",
464
+ )
465
+
466
+ with gr.Column(scale=2):
467
+ with gr.Tab("Full Defense Pack"):
468
+ full_output = gr.Markdown()
469
+
470
+ with gr.Tab("Viva Questions"):
471
+ questions_output = gr.Markdown()
472
+
473
+ with gr.Tab("Easy Answers"):
474
+ answers_output = gr.Markdown()
475
+
476
+ with gr.Tab("Defense Speech"):
477
+ speech_output = gr.Markdown()
478
+
479
+ with gr.Tab("Weak Points"):
480
+ weak_output = gr.Markdown()
481
+
482
+ with gr.Tab("Confidence Score"):
483
+ confidence_output = gr.Markdown()
484
+
485
+ generate_btn.click(
486
+ fn=generate_defense_pack,
487
+ inputs=[file_input, manual_text, difficulty, language_style],
488
+ outputs=[
489
+ full_output,
490
+ questions_output,
491
+ answers_output,
492
+ speech_output,
493
+ weak_output,
494
+ confidence_output,
495
+ ],
496
+ )
497
+
498
+ gr.Markdown("---")
499
+
500
+ gr.Markdown("## Practice Viva Evaluator")
501
+
502
+ with gr.Row():
503
+ with gr.Column():
504
+ practice_question = gr.Textbox(
505
+ label="Viva Question",
506
+ placeholder="Example: Why did you choose this methodology?",
507
+ lines=3,
508
+ )
509
+
510
+ practice_answer = gr.Textbox(
511
+ label="Your Answer",
512
+ placeholder="Type your answer here...",
513
+ lines=6,
514
+ )
515
+
516
+ report_context = gr.Textbox(
517
+ label="Optional Report Context",
518
+ placeholder="Paste a small part of your report if needed...",
519
+ lines=6,
520
+ )
521
+
522
+ evaluate_btn = gr.Button("Evaluate My Answer")
523
+
524
+ with gr.Column():
525
+ evaluation_output = gr.Markdown()
526
+
527
+ evaluate_btn.click(
528
+ fn=evaluate_practice_answer,
529
+ inputs=[practice_question, practice_answer, report_context],
530
+ outputs=evaluation_output,
531
+ )
532
+
533
+ gr.Markdown("---")
534
+
535
+ gr.Markdown("## Custom Viva Question Generator")
536
+
537
+ with gr.Row():
538
+ with gr.Column():
539
+ topic_focus = gr.Textbox(
540
+ label="Focus Topic",
541
+ placeholder="Example: methodology, result, networking, algorithm, water cycle...",
542
+ )
543
+
544
+ number_of_questions = gr.Slider(
545
+ minimum=5,
546
+ maximum=25,
547
+ value=10,
548
+ step=1,
549
+ label="Number of Questions",
550
+ )
551
+
552
+ custom_btn = gr.Button("Generate Custom Questions")
553
+
554
+ with gr.Column():
555
+ custom_output = gr.Markdown()
556
+
557
+ custom_btn.click(
558
+ fn=create_custom_questions,
559
+ inputs=[file_input, manual_text, topic_focus, number_of_questions],
560
+ outputs=custom_output,
561
+ )
562
+
563
+ gr.Markdown(
564
+ """
565
+ ### Notes
566
+ - This app uses a small local Hugging Face model.
567
+ - No API key is required.
568
+ - First run may take time because the model downloads automatically.
569
+ - For better quality, later you can add Gemini/Groq API as an optional mode.
570
+ """
571
+ )
572
+
573
+
574
+ if __name__ == "__main__":
575
+ demo.launch()
requirements.txt ADDED
@@ -0,0 +1,6 @@
 
 
 
 
 
 
 
1
+ gradio
2
+ torch
3
+ transformers
4
+ pypdf
5
+ python-docx
6
+ accelerate