File size: 6,713 Bytes
2c95ce1
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
ο»Ώimport gradio as gr
from codon_table import translate_dna_to_text, get_example_sequences, CODON_TABLE

def process_dna_sequence(sequence, reading_frame, detailed_mode, example_dropdown):
    """Process DNA sequence and return explanation"""
    
    # If user selected an example, use that
    if example_dropdown and example_dropdown != "Choose an example...":
        examples = get_example_sequences()
        if example_dropdown in examples:
            sequence = examples[example_dropdown]
    
    if not sequence or sequence.strip() == "":
        return "Please enter a DNA sequence or select an example."
    
    try:
        result = translate_dna_to_text(sequence, reading_frame, detailed_mode)
        return result
    except Exception as e:
        return f"❌ Error processing sequence: {str(e)}\n\nPlease check your input and try again."

def get_genetic_code_table():
    """Generate a formatted genetic code reference table"""
    output = ["# 🧬 Genetic Code Reference\n"]
    output.append("| Codon | Amino Acid | Type | Description |")
    output.append("|-------|------------|------|-------------|")
    
    # Group by amino acid for better organization
    amino_acid_groups = {}
    for codon, info in CODON_TABLE.items():
        aa = info['amino_acid']
        if aa not in amino_acid_groups:
            amino_acid_groups[aa] = []
        amino_acid_groups[aa].append((codon, info))
    
    # Sort amino acids, with special codons first
    special_order = ['Methionine', 'STOP']
    regular_amino_acids = sorted([aa for aa in amino_acid_groups.keys() if aa not in special_order])
    
    for aa in special_order + regular_amino_acids:
        if aa in amino_acid_groups:
            for codon, info in sorted(amino_acid_groups[aa]):
                icon = "πŸš€" if info['type'] == 'start' else "πŸ›‘" if info['type'] == 'stop' else "πŸ”€"
                output.append(f"| {codon} | {aa} | {icon} | {info['description'][:50]}{'...' if len(info['description']) > 50 else ''} |")
    
    return "\n".join(output)

# Create the Gradio interface
with gr.Blocks(
    title="Gene2Text: DNA Codon Explainer",
    theme=gr.themes.Soft(),
    css="""
    .gradio-container {
        max-width: 1200px !important;
    }
    .output-markdown {
        font-family: 'Segoe UI', Tahoma, Geneva, Verdana, sans-serif;
    }
    """
) as app:
    
    gr.Markdown("""
    # 🧬 Gene2Text: Interpretable Codon-by-Codon Describer
    
    **Transform DNA sequences into readable explanations!** This tool takes raw DNA sequences and explains what each three-letter codon codes for, making molecular biology accessible to everyone.
    
    Perfect for:
    - πŸŽ“ **Students** learning molecular biology
    - πŸ‘©β€πŸ« **Teachers** explaining genetic concepts  
    - πŸ”¬ **Researchers** quickly interpreting sequences
    - πŸ€” **Anyone curious** about how DNA codes for proteins
    """)
    
    with gr.Row():
        with gr.Column(scale=2):
            gr.Markdown("## πŸ“ Input Your DNA Sequence")
            
            # Example dropdown
            example_dropdown = gr.Dropdown(
                choices=["Choose an example..."] + list(get_example_sequences().keys()),
                value="Choose an example...",
                label="πŸ“š Or select an example sequence:",
                info="Choose a pre-loaded example to see how the tool works"
            )
            
            # Main input
            sequence_input = gr.Textbox(
                label="🧬 DNA Sequence",
                placeholder="Enter your DNA sequence here (e.g., ATG GCT TAA)\nSpaces and line breaks will be automatically removed.",
                lines=4,
                info="Enter nucleotides: A, T, G, C only. Other characters will be filtered out."
            )
            
            with gr.Row():
                reading_frame = gr.Radio(
                    choices=[0, 1, 2],
                    value=0,
                    label="πŸ“ Reading Frame",
                    info="Choose which nucleotide to start reading from (0=first, 1=second, 2=third)"
                )
                
                detailed_mode = gr.Checkbox(
                    value=True,
                    label="πŸ” Detailed Descriptions",
                    info="Include biological context and amino acid properties"
                )
            
            submit_btn = gr.Button("πŸ”¬ Analyze Sequence", variant="primary", size="lg")
        
        with gr.Column(scale=3):
            gr.Markdown("## πŸ“‹ Results")
            output_text = gr.Markdown(
                value="Enter a DNA sequence to see the codon-by-codon breakdown here...",
                elem_classes=["output-markdown"]
            )
    
    # Genetic Code Reference (collapsible)
    with gr.Accordion("πŸ“– Genetic Code Reference Table", open=False):
        genetic_code_display = gr.Markdown(get_genetic_code_table())
    
    # Educational content
    with gr.Accordion("πŸ’‘ How It Works", open=False):
        gr.Markdown("""
        ### The Genetic Code Explained
        
        **DNA β†’ RNA β†’ Protein** is the central dogma of molecular biology:
        
        1. **Codons**: DNA is read in groups of 3 nucleotides called codons
        2. **Translation**: Each codon codes for a specific amino acid (or stop signal)
        3. **Proteins**: Amino acids chain together to form proteins
        4. **Reading Frames**: DNA can be read in 3 different frames, giving different results
        
        **Special Codons:**
        - πŸš€ **ATG**: Start codon (Methionine) - where protein synthesis begins
        - πŸ›‘ **TAA, TAG, TGA**: Stop codons - where protein synthesis ends
        
        **Why This Matters:**
        Understanding how DNA codes for proteins helps us comprehend genetics, evolution, 
        disease mechanisms, and biotechnology applications.
        """)
    
    # Event handlers
    submit_btn.click(
        fn=process_dna_sequence,
        inputs=[sequence_input, reading_frame, detailed_mode, example_dropdown],
        outputs=output_text
    )
    
    # Auto-update when example is selected
    example_dropdown.change(
        fn=process_dna_sequence,
        inputs=[sequence_input, reading_frame, detailed_mode, example_dropdown],
        outputs=output_text
    )
    
    # Footer
    gr.Markdown("""
    ---
    **Built with ❀️ for biology education** | Made with [Gradio](https://gradio.app) | 
    Perfect for classrooms, labs, and curious minds everywhere!
    """)

# Launch the app
if __name__ == "__main__":
    app.launch(
        share=True,
        server_name="0.0.0.0",
        server_port=7860,
        show_error=True
    )