Spaces:
Sleeping
Sleeping
| #!/usr/bin/env python | |
| # coding=utf-8 | |
| import datetime | |
| import re | |
| from fpdf import FPDF | |
| from langdetect import detect | |
| import gradio as gr | |
| from tools.common import prepend_metadata_questions # ✅ Import helper | |
| # === PDF Export Function === | |
| def export_text_to_pdf(text, metadata=None, output_path=None, language="en"): | |
| if output_path is None: | |
| timestamp = datetime.datetime.now().strftime("%Y%m%d_%H%M%S") | |
| output_path = f"data_governance_record_{timestamp}.pdf" | |
| pdf = FPDF() | |
| pdf.add_page() | |
| pdf.set_auto_page_break(auto=True, margin=15) | |
| # Title | |
| pdf.set_font("Arial", 'B', 16) | |
| pdf.set_text_color(0, 51, 102) | |
| title = "Data Governance and Quality Record" if language == "en" else "Dossier de Gouvernance et Qualité des Données" | |
| pdf.cell(0, 15, title, ln=True, align='C') | |
| pdf.ln(10) | |
| # Metadata block | |
| if metadata: | |
| pdf.set_font("Arial", '', 12) | |
| pdf.set_text_color(90, 90, 90) | |
| pdf.multi_cell(0, 10, f"Organization: {metadata.get('organization', 'N/A')}") | |
| pdf.multi_cell(0, 10, f"Completed by: {metadata.get('completed_by', 'N/A')} ({metadata.get('role', 'N/A')})") | |
| pdf.multi_cell(0, 10, f"Timestamp: {metadata.get('timestamp', 'N/A')}") | |
| pdf.ln(5) | |
| # Body | |
| pdf.set_font("Arial", '', 12) | |
| pdf.set_text_color(0, 0, 0) | |
| for line in text.strip().split('\n'): | |
| line = line.strip() | |
| if line.startswith("## "): | |
| section_title = line.replace("## ", "").strip() | |
| pdf.set_font("Arial", 'B', 13) | |
| pdf.set_text_color(30, 30, 120) | |
| pdf.ln(8) | |
| pdf.cell(0, 10, section_title, ln=True) | |
| pdf.set_font("Arial", '', 12) | |
| pdf.set_text_color(0, 0, 0) | |
| elif line.startswith("- **"): | |
| match = re.match(r"- \*\*(.+?)\*\*: (.+)", line) | |
| if match: | |
| label, answer = match.groups() | |
| pdf.set_font("Arial", 'B', 12) | |
| pdf.cell(0, 10, f"{label}:", ln=True) | |
| pdf.set_font("Arial", '', 12) | |
| pdf.multi_cell(0, 10, answer) | |
| elif line == "---": | |
| pdf.line(10, pdf.get_y(), 200, pdf.get_y()) | |
| pdf.ln(5) | |
| else: | |
| pdf.multi_cell(0, 10, line) | |
| pdf.output(output_path) | |
| return output_path | |
| # === Core Questions === | |
| BASE_QUESTIONS = [ | |
| ("dataset_description", "Please describe the dataset(s) used."), | |
| ("data_sources", "What are the sources of the data?"), | |
| ("data_collection_method", "How was the data collected?"), | |
| ("preprocessing", "What preprocessing steps were applied?"), | |
| ("representativeness", "Is the data representative of the use case?"), | |
| ("bias_handling", "How are biases identified and mitigated?"), | |
| ("data_split", "How is the data split (training/testing/validation)?"), | |
| ("missing_data", "How is missing or incomplete data handled?"), | |
| ("updates", "How is data kept up-to-date or refreshed?"), | |
| ("access_control", "Who has access to the data and under what conditions?") | |
| ] | |
| QUESTIONS = prepend_metadata_questions(BASE_QUESTIONS) # ✅ Prepend metadata | |
| def get_questions(): | |
| return QUESTIONS | |
| # === Run Tool === | |
| def run_tool(): | |
| state = {"step": 0, "answers": {}} | |
| def step_by_step_agent(user_input, state): | |
| step = state["step"] | |
| answers = state["answers"] | |
| if step > 0: | |
| key, _ = QUESTIONS[step - 1] | |
| answers[key] = user_input | |
| if step < len(QUESTIONS): | |
| next_q = QUESTIONS[step][1] | |
| state["step"] += 1 | |
| return next_q, state, None | |
| content = "\n".join([f"- **{label}**: {answers.get(key, '')}" for key, label in QUESTIONS]) | |
| detected_lang = detect(content) | |
| # ✅ Extract metadata from prepended fields | |
| metadata = { | |
| "organization": answers.get("organization_name", "N/A"), | |
| "completed_by": answers.get("user_name", "N/A"), | |
| "role": answers.get("user_role", "N/A"), | |
| "timestamp": datetime.datetime.now().strftime("%Y-%m-%d %H:%M:%S") | |
| } | |
| pdf_path = export_text_to_pdf(content, metadata=metadata, language=detected_lang) | |
| return "✅ Documentation complete. Download below.", {"done": True}, pdf_path | |
| with gr.Blocks(title="Data Governance Tool") as demo: | |
| chatbot = gr.Chatbot( | |
| label="📊 Data Governance Assistant", | |
| value=[{"role": "assistant", "content": QUESTIONS[0][1]}], | |
| type="messages" | |
| ) | |
| msg = gr.Textbox(label="Your answer") | |
| state_var = gr.State(state) | |
| file_output = gr.File(label="Download PDF") | |
| reset_btn = gr.Button("🔁 Restart") | |
| def chat_logic(msg_in, state_in): | |
| reply, updated_state, file = step_by_step_agent(msg_in, state_in) | |
| messages = [{"role": "user", "content": msg_in}] | |
| if reply: | |
| messages.append({"role": "assistant", "content": reply}) | |
| return messages, updated_state, file | |
| def reset(): | |
| return [{"role": "assistant", "content": QUESTIONS[0][1]}], {"step": 0, "answers": {}}, None | |
| msg.submit(chat_logic, [msg, state_var], [chatbot, state_var, file_output]) | |
| reset_btn.click(reset, outputs=[chatbot, state_var, file_output]) | |
| demo.launch(show_api=False) | |