import os import json from flask import Flask, request, render_template, jsonify from werkzeug.utils import secure_filename import google.generativeai as genai from dotenv import load_dotenv import traceback import re import fitz import docx import werkzeug load_dotenv() app = Flask(__name__) app.config['UPLOAD_FOLDER'] = 'uploads' app.config['MAX_CONTENT_LENGTH'] = 200 * 1024 * 1024 os.makedirs(app.config['UPLOAD_FOLDER'], exist_ok=True) try: genai.configure(api_key=os.getenv("GEMINI_API_KEY")) safety_settings = [ {"category": "HARM_CATEGORY_HARASSMENT", "threshold": "BLOCK_NONE"}, {"category": "HARM_CATEGORY_HATE_SPEECH", "threshold": "BLOCK_NONE"}, {"category": "HARM_CATEGORY_SEXUALLY_EXPLICIT", "threshold": "BLOCK_NONE"}, {"category": "HARM_CATEGORY_DANGEROUS_CONTENT", "threshold": "BLOCK_NONE"}, ] model = genai.GenerativeModel('gemini-1.5-flash', safety_settings=safety_settings) except Exception as e: print(f"Lỗi khi cấu hình Gemini API: {e}") model = None def robust_json_parser(dirty_string): match = re.search(r'```json\s*([\s\S]*?)\s*```', dirty_string) if match: clean_string = match.group(1) try: return json.loads(clean_string) except json.JSONDecodeError: pass try: start = dirty_string.find('[') end = dirty_string.rfind(']') if start != -1 and end != -1: potential_json = dirty_string[start:end+1] return json.loads(potential_json) except (json.JSONDecodeError, ValueError): return None return None def extract_text_from_file(filepath): text = "" if filepath.endswith('.pdf'): with fitz.open(filepath) as doc: for page in doc: text += page.get_text() elif filepath.endswith('.docx'): doc = docx.Document(filepath) for para in doc.paragraphs: text += para.text + "\n" return text @app.route('/') def index(): return render_template('index.html') @app.route('/process', methods=['POST']) def process_file(): try: if not model: return jsonify({"error": "Lỗi: Gemini API chưa được cấu hình."}), 500 if 'file' not in request.files: return jsonify({"error": "Không có file nào."}), 400 file = request.files['file'] action = request.form.get('action', 'quiz') if file.filename == '': return jsonify({"error": "Chưa chọn file."}), 400 if file: filename = secure_filename(file.filename) filepath = os.path.join(app.config['UPLOAD_FOLDER'], filename) try: file.save(filepath) extracted_text = extract_text_from_file(filepath) if not extracted_text.strip(): return jsonify({"error": "Không thể đọc nội dung file."}), 400 truncated_text = extracted_text[:100000] if action == 'quiz': num_questions = request.form.get('num_questions', 10, type=int) difficulty = request.form.get('difficulty', 'Trung bình', type=str) prompt = f""" VAI TRÒ: Anh là một chuyên gia khảo thí, nhiệm vụ của anh là tạo ra một bài kiểm tra chất lượng cao. NHIỆM VỤ: Dựa vào nội dung của văn bản được cung cấp, hãy tạo ra chính xác {num_questions} câu hỏi trắc nghiệm ở mức độ "{difficulty}". QUY TẮC BẤT DI BẤT DỊCH (CỰC KỲ QUAN TRỌNG): 1. CHỈ DÙNG THÔNG TIN CÓ TRONG VĂN BẢN: Mọi câu hỏi và đáp án phải có thể được suy ra hoặc tìm thấy trực tiếp từ nội dung văn bản. NGHIÊM CẤM tuyệt đối việc bịa đặt thông tin hoặc hỏi những câu không liên quan. 2. KHÔNG HỎI VỀ ĐỊNH DẠNG: Không được hỏi về những thứ như "tần suất xuất hiện của ký tự", "số lượng đoạn văn", hay phong cách viết của tác giả. Chỉ tập trung vào NỘI DUNG và Ý NGHĨA của văn bản. 3. CHẤT LƯỢNG CÂU HỎI: Câu hỏi phải rõ ràng, dễ hiểu. Các lựa chọn nhiễu (sai) phải hợp lý nhưng không chính xác. 4. ĐỊNH DẠNG ĐẦU RA: Bắt buộc chỉ trả về MỘT chuỗi JSON duy nhất, không có bất kỳ lời nói đầu, giải thích hay markdown nào khác. Chuỗi JSON là một danh sách các đối tượng, mỗi đối tượng có key "question", "options" (danh sách 4 chuỗi), và "answer". VĂN BẢN NGUỒN ĐỂ PHÂN TÍCH: --- {truncated_text} --- """ response = model.generate_content(prompt) if not response.parts: block_reason = str(response.prompt_feedback) return jsonify({"error": f"API đã từ chối tạo nội dung. Lý do: {block_reason}."}), 500 parsed_json = robust_json_parser(response.text) if parsed_json is None: return jsonify({"error": "Lỗi: AI trả về định dạng không thể xử lý. Vui lòng thử lại với số câu hỏi ít hơn."}), 500 return jsonify(parsed_json) finally: if os.path.exists(filepath): os.remove(filepath) return jsonify({"error": "Yêu cầu không hợp lệ."}), 400 except werkzeug.exceptions.RequestEntityTooLarge: return jsonify({"error": f"Lỗi: File quá lớn (Tối đa 200MB)."}), 413 except Exception: print(traceback.format_exc()) return jsonify({"error": "Một lỗi không xác định đã xảy ra phía máy chủ."}), 500 @app.route('/explain', methods=['POST']) def explain_answer(): try: data = request.json question = data.get('question') correct_answer = data.get('correct_answer') if not question or not correct_answer: return jsonify({"error": "Dữ liệu không hợp lệ."}), 400 prompt = f""" VAI TRÒ: Một giáo viên giỏi. NHIỆM VỤ: Giải thích ngắn gọn (khoảng 2-3 câu) tại sao đáp án của câu hỏi sau lại là lựa chọn đó. Tập trung vào kiến thức cốt lõi. Câu hỏi: "{question}" Đáp án đúng: "{correct_answer}" Bắt đầu giải thích của bạn bằng "Đáp án đúng là '{correct_answer}' vì..." """ response = model.generate_content(prompt) if not response.parts: return jsonify({"explanation": "Không thể tạo giải thích vào lúc này."}) return jsonify({"explanation": response.text}) except Exception: print(traceback.format_exc()) return jsonify({"explanation": "Lỗi khi tạo giải thích."}) if __name__ == '__main__': app.run(debug=True)