File size: 4,974 Bytes
ebf1e57
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
# dora_data_loss_prevention.py
import datetime
import re
from fpdf import FPDF
from langdetect import detect
import gradio as gr
from tools.common import prepend_metadata_questions

# === PDF Export ===
def export_text_to_pdf(text, metadata=None, output_path=None, language="en"):
    if output_path is None:
        timestamp = datetime.datetime.now().strftime("%Y%m%d_%H%M%S")
        output_path = f"data_loss_prevention_{timestamp}.pdf"

    pdf = FPDF()
    pdf.add_page()
    pdf.set_auto_page_break(auto=True, margin=15)

    pdf.set_font("Arial", 'B', 16)
    pdf.set_text_color(0, 51, 102)
    title = "Data Loss Prevention Strategy - DORA (Optional)" if language == "en" else "Stratégie de Prévention des Pertes de Données - DORA"
    pdf.cell(0, 15, title, ln=True, align='C')
    pdf.ln(10)

    if metadata:
        pdf.set_font("Arial", '', 12)
        pdf.set_text_color(90, 90, 90)
        pdf.multi_cell(0, 10, f"Organization: {metadata.get('organization_name', 'N/A')}")
        pdf.multi_cell(0, 10, f"Completed by: {metadata.get('user_name', 'N/A')} ({metadata.get('user_role', 'N/A')})")
        pdf.multi_cell(0, 10, f"Timestamp: {metadata.get('timestamp', 'N/A')}")
        pdf.ln(5)

    pdf.set_font("Arial", '', 12)
    pdf.set_text_color(0, 0, 0)
    for line in text.strip().split('\n'):
        line = line.strip()
        if line.startswith("## "):
            section = line.replace("## ", "").strip()
            pdf.set_font("Arial", 'B', 13)
            pdf.set_text_color(30, 30, 120)
            pdf.ln(8)
            pdf.cell(0, 10, section, ln=True)
            pdf.set_font("Arial", '', 12)
            pdf.set_text_color(0, 0, 0)
        elif line.startswith("- **"):
            match = re.match(r"- \*\*(.+?)\*\*: (.+)", line)
            if match:
                label, value = match.groups()
                pdf.set_font("Arial", 'B', 12)
                pdf.cell(0, 10, f"{label}:", ln=True)
                pdf.set_font("Arial", '', 12)
                pdf.multi_cell(0, 10, value)
        elif line == "---":
            pdf.line(10, pdf.get_y(), 200, pdf.get_y())
            pdf.ln(5)
        else:
            pdf.multi_cell(0, 10, line)
    pdf.output(output_path)
    return output_path

# === Questions ===
QUESTIONS = prepend_metadata_questions([
    ("dlp_policies", "Describe the DLP policies currently in place."),
    ("tools_technologies", "What tools and technologies are used for DLP?"),
    ("sensitive_data_types", "Which types of sensitive data are protected?"),
    ("data_exfiltration_prevention", "How do you prevent data exfiltration or leaks?"),
    ("training_awareness", "Is DLP part of employee training or awareness programs?"),
    ("incident_history", "Any previous incidents of data loss or leakage? What lessons were learned?")
])


def get_questions():
    return QUESTIONS


def run_tool():
    state = {"step": 0, "answers": {}}

    def step_by_step_agent(user_input, state):
        step = state["step"]
        answers = state["answers"]

        if step > 0:
            key, _ = QUESTIONS[step - 1]
            answers[key] = user_input

        if step < len(QUESTIONS):
            next_question = QUESTIONS[step][1]
            state["step"] += 1
            return next_question, state, None

        metadata = {
            "user_name": answers.get("user_name", "N/A"),
            "user_role": answers.get("user_role", "N/A"),
            "organization_name": answers.get("organization_name", "N/A"),
            "timestamp": datetime.datetime.now().strftime("%Y-%m-%d %H:%M:%S")
        }

        content = "\n".join([f"- **{label}**: {answers.get(key, '')}" for key, label in QUESTIONS])
        lang = detect(content if len(content.strip()) > 3 else "Placeholder content")
        pdf_path = export_text_to_pdf(content, metadata=metadata, language=lang)
        return "✅ Report generated. Download below.", {"done": True}, pdf_path

    with gr.Blocks(title="DORA Data Loss Prevention Tool") as demo:
        chatbot = gr.Chatbot(label="🛡️ DLP Strategy Assistant", value=[{"role": "assistant", "content": QUESTIONS[0][1]}], type="messages")
        msg = gr.Textbox(label="Your answer")
        state_var = gr.State(state)
        file_output = gr.File(label="Download PDF")
        reset_btn = gr.Button("🔁 Restart")

        def chat_logic(msg_in, state_in):
            reply, updated_state, file = step_by_step_agent(msg_in, state_in)
            messages = [{"role": "user", "content": msg_in}]
            if reply:
                messages.append({"role": "assistant", "content": reply})
            return messages, updated_state, file

        def reset():
            return [{"role": "assistant", "content": QUESTIONS[0][1]}], {"step": 0, "answers": {}}, None

        msg.submit(chat_logic, [msg, state_var], [chatbot, state_var, file_output])
        reset_btn.click(reset, outputs=[chatbot, state_var, file_output])

    demo.launch(show_api=False)