File size: 49,640 Bytes
b89cfc5
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
cc25d65
 
b89cfc5
 
 
 
 
 
cc25d65
 
 
 
 
 
 
 
b89cfc5
cc25d65
 
 
 
 
 
 
 
 
b89cfc5
cc25d65
 
 
 
b89cfc5
cc25d65
 
b89cfc5
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
417
418
419
420
421
422
423
424
425
426
427
428
429
430
431
432
433
434
435
436
437
438
439
440
441
442
443
444
445
446
447
448
449
450
451
452
453
454
455
456
457
458
459
460
461
462
463
464
465
466
467
468
469
470
471
472
473
474
475
476
477
478
479
480
481
482
483
484
485
486
487
488
489
490
491
492
493
494
495
496
497
498
499
500
501
502
503
504
505
506
507
508
509
510
511
512
513
514
515
516
517
518
519
520
521
522
523
524
525
526
527
528
529
530
531
532
533
534
535
536
537
538
539
540
541
542
543
544
545
546
547
548
549
550
551
552
553
554
555
556
557
558
559
560
561
562
563
564
565
566
567
568
569
570
571
572
573
574
575
576
577
578
579
580
581
582
583
584
585
586
587
588
589
590
591
592
593
594
595
596
597
598
599
600
601
602
603
604
605
606
607
608
609
610
611
612
613
614
615
616
617
618
619
620
621
622
623
624
625
626
627
628
629
630
631
632
633
634
635
636
637
638
639
640
641
642
643
644
645
646
647
648
649
650
651
652
653
654
655
656
657
658
659
660
661
662
663
664
665
666
667
668
669
670
671
672
673
674
675
676
677
678
679
680
681
682
683
684
685
686
687
688
689
690
691
692
693
694
695
696
697
698
699
700
701
702
703
704
705
706
707
708
709
710
711
712
713
714
715
716
717
718
719
720
721
722
723
724
725
726
727
728
729
730
731
732
733
734
735
736
737
738
739
740
741
742
743
744
745
746
747
748
749
750
751
752
753
754
755
756
757
758
759
760
761
762
763
764
765
766
767
768
769
770
771
772
773
774
775
776
777
778
779
780
781
782
783
784
785
786
787
788
789
790
791
792
793
794
795
796
797
798
799
800
801
802
803
804
805
806
807
808
809
810
811
812
813
814
815
816
817
818
819
820
821
822
823
824
825
826
827
828
829
830
831
832
833
834
835
836
837
838
839
840
841
842
843
844
845
846
847
848
849
850
851
852
853
854
855
856
857
858
859
860
861
862
863
864
865
866
867
868
869
870
871
872
873
874
875
876
877
878
879
880
881
882
883
884
885
886
887
888
889
890
891
892
893
894
895
896
897
898
899
900
901
902
903
904
905
906
907
908
909
910
911
912
913
914
915
916
917
918
919
920
921
922
923
924
925
926
927
928
929
930
931
932
933
934
935
936
937
938
939
940
941
942
943
944
945
946
947
import dspy
import src.agents.memory_agents as m
import asyncio
from concurrent.futures import ThreadPoolExecutor
import os
from dotenv import load_dotenv

load_dotenv()

AGENTS_WITH_DESCRIPTION = {
    "preprocessing_agent": "Cleans and prepares a DataFrame using Pandas and NumPy—handles missing values, detects column types, and converts date strings to datetime.",
    "statistical_analytics_agent": "Performs statistical analysis (e.g., regression, seasonal decomposition) using statsmodels, with proper handling of categorical data and missing values.",
    "sk_learn_agent": "Trains and evaluates machine learning models using scikit-learn, including classification, regression, and clustering with feature importance insights.",
    "data_viz_agent": "Generates interactive visualizations with Plotly, selecting the best chart type to reveal trends, comparisons, and insights based on the analysis goal."
}

PLANNER_AGENTS_WITH_DESCRIPTION = {
    "planner_preprocessing_agent": (
        "Cleans and prepares a DataFrame using Pandas and NumPy—"
        "handles missing values, detects column types, and converts date strings to datetime. "
        "Outputs a cleaned DataFrame for the planner_statistical_analytics_agent."
    ),
    "planner_statistical_analytics_agent": (
        "Takes the cleaned DataFrame from preprocessing, performs statistical analysis "
        "(e.g., regression, seasonal decomposition) using statsmodels with proper handling "
        "of categorical data and remaining missing values. "
        "Produces summary statistics and model diagnostics for the planner_sk_learn_agent."
    ),
    "planner_sk_learn_agent": (
        "Receives summary statistics and the cleaned data, trains and evaluates machine "
        "learning models using scikit-learn (classification, regression, clustering), "
        "and generates performance metrics and feature importance. "
        "Passes the trained models and evaluation results to the planner_data_viz_agent."
    ),
    "planner_data_viz_agent": (
        "Consumes trained models and evaluation results to create interactive visualizations "
        "with Plotly—selects the best chart type, applies styling, and annotates insights. "
        "Delivers ready-to-share figures that communicate model performance and key findings."
    ),
}

def get_agent_description(agent_name, is_planner=False):
    if is_planner:
        return PLANNER_AGENTS_WITH_DESCRIPTION[agent_name.lower()] if agent_name.lower() in PLANNER_AGENTS_WITH_DESCRIPTION else "No description available for this agent"
    else:
        return AGENTS_WITH_DESCRIPTION[agent_name.lower()] if agent_name.lower() in AGENTS_WITH_DESCRIPTION else "No description available for this agent"


# Agent to make a Chat history name from a query
class chat_history_name_agent(dspy.Signature):
    """You are an agent that takes a query and returns a name for the chat history"""
    query = dspy.InputField(desc="The query to make a name for")
    name = dspy.OutputField(desc="A name for the chat history (max 3 words)")

class dataset_description_agent(dspy.Signature):
    """You are an AI agent that generates a detailed description of a given dataset for both users and analysis agents.
Your description should serve two key purposes:
1. Provide users with context about the dataset's purpose, structure, and key attributes.
2. Give analysis agents critical data handling instructions to prevent common errors.

For data handling instructions, you must always include Python data types and address the following:
- Data type warnings (e.g., numeric columns stored as strings that need conversion).
- Null value handling recommendations.
- Format inconsistencies that require preprocessing.
- Explicit warnings about columns that appear numeric but are stored as strings (e.g., '10' vs 10).
- Explicit Python data types for each major column (e.g., int, float, str, bool, datetime).
- Columns with numeric values that should be treated as categorical (e.g., zip codes, IDs).
- Any date parsing or standardization required (e.g., MM/DD/YYYY to datetime).
- Any other technical considerations that would affect downstream analysis or modeling.
- List all columns and their data types with exact case sensitive spelling

If an existing description is provided, enhance it with both business context and technical guidance for analysis agents, preserving accurate information from the existing description or what the user has written.

Ensure the description is comprehensive and provides actionable insights for both users and analysis agents.


Example:
This housing dataset contains property details including price, square footage, bedrooms, and location data.
It provides insights into real estate market trends across different neighborhoods and property types.

TECHNICAL CONSIDERATIONS FOR ANALYSIS:
- price (str): Appears numeric but is stored as strings with a '$' prefix and commas (e.g., "$350,000"). Requires cleaning with str.replace('$','').replace(',','') and conversion to float.
- square_footage (str): Contains unit suffix like 'sq ft' (e.g., "1,200 sq ft"). Remove suffix and commas before converting to int.
- bedrooms (int): Correctly typed but may contain null values (~5% missing) – consider imputation or filtering.
- zip_code (int): Numeric column but should be treated as str or category to preserve leading zeros and prevent unintended numerical analysis.
- year_built (float): May contain missing values (~15%) – consider mean/median imputation or exclusion depending on use case.
- listing_date (str): Dates stored in "MM/DD/YYYY" format – convert to datetime using pd.to_datetime().
- property_type (str): Categorical column with inconsistent capitalization (e.g., "Condo", "condo", "CONDO") – normalize to lowercase for consistent grouping.
    """
    dataset = dspy.InputField(desc="The dataset to describe, including headers, sample data, null counts, and data types.")
    existing_description = dspy.InputField(desc="An existing description to improve upon (if provided).", default="")
    description = dspy.OutputField(desc="A comprehensive dataset description with business context and technical guidance for analysis agents.")



class analytical_planner(dspy.Signature):
    """
    You are a data analytics planner agent. You have access to three inputs:
    1. Datasets
    2. Data Agent descriptions
    3. User-defined Goal

    You take these three inputs to develop a comprehensive plan to achieve the user-defined goal from the data & Agents available.

    In case you think the user-defined goal is infeasible, you can ask the user to redefine or add more description to the goal.

    Your output includes:
    - Plan: The ordered sequence of agents (e.g., Agent1->Agent2->Agent3)
    - Plan Description: Explanation of why each agent is used in that sequence
    - Plan Instructions: A dictionary with keys as agent names and values as dictionaries specifying:
        - 'create': variables this agent should generate
        - 'receive': variables this agent should get from previous agents

    Format:
    plan: Agent1->Agent2->Agent3
   
    plan_instructions: {
        "Agent1": {"create": ["var1"], "receive": []},
        "Agent2": {"create": ["var2"], "receive": ["var1"]},
        ...
    }
    plan_desc: Use Agent1 for this reason, then Agent2 for this reason, and finally Agent3 for this reason.
    """
    dataset = dspy.InputField(desc="Available datasets loaded in the system, use this df, columns set df as copy of df")
    Agent_desc = dspy.InputField(desc="The agents available in the system")
    goal = dspy.InputField(desc="The user defined goal")

    plan = dspy.OutputField(desc="The plan that would achieve the user defined goal", prefix='Plan:')
    plan_instructions = dspy.OutputField(desc="Detailed variable-level instructions per agent for the plan")
    plan_desc = dspy.OutputField(desc="The reasoning behind the chosen plan")

class planner_data_viz_agent(dspy.Signature):
    """
    You are the data visualization agent in a multi-agent analytics system.

    You are given:
    - A user-defined goal describing what visualization the user wants
    - A dataset and styling index
    - Agent-specific plan instructions that tell you:
        - What variables you should **receive** (e.g., df_cleaned, regression_model)
        - What visualization components you are expected to **create** (e.g., scatter_plot, bar_chart)

    Your job is to:
    - Use Plotly to generate a visualization that fulfills the user goal and adheres to plan instructions
    - Display the chart using Plotly's `fig.show()` method
    - Respect performance thresholds:
        - If `len(df) > 50000`, sample the dataset before visualizing using:
        ```python
        if len(df) > 50000:
            df = df.sample(5000, random_state=42)
        ```
    - Only this agent handles visualization — no other agent creates visual plots
    - When using `update_layout()`, use only one format for x/y-axis labels (e.g., 'K', 'M', or 1,000 — not both)
    - Use trendlines **only if explicitly requested** in the user goal
    - NEVER include `goal`, `dataset`, or `styling_index` in your final output
    - NEVER output the raw dataset
    - Produce:
        - Valid, runnable Plotly code
        - A clear and concise summary (no code in the summary)

    Input Fields:
    - goal: The user's visualization intent (e.g., "plot sales over time with trendline")
    - dataset: Context on the dataframe (`df`) including its name and available columns
    - styling_index: Plotly-specific formatting or design hints
    - plan_instructions: A dictionary with:
        - 'create': list of visual assets to generate (e.g., 'scatter_plot')
        - 'receive': list of input variables needed (e.g., 'regression_model', 'df_cleaned')

    Output Fields:
    - code: Plotly Python code that implements and displays the chart
    - summary: A plain-language explanation of what the chart shows (without any code)
    """
    goal = dspy.InputField(desc="User-defined chart goal (e.g. trendlines, scatter plots)")
    dataset = dspy.InputField(desc="Details of the dataframe (`df`) and its columns")
    styling_index = dspy.InputField(desc="Instructions for plot styling and layout formatting")
    plan_instructions = dspy.InputField(desc="Variables to create and receive for visualization purposes")

    code = dspy.OutputField(desc="Plotly Python code for the visualization")
    summary = dspy.OutputField(desc="Plain-language summary of what is being visualized")

class planner_preprocessing_agent(dspy.Signature):
    """
    You are a preprocessing agent in a multi-agent data analytics system.

    You are given:
    - A dataset (already loaded as `df`)
    - A user-defined analysis goal
    - Agent-specific plan instructions telling you what variables you are expected to create and what variables you are receiving from previous agents

    Your job is to:
    - Generate Python code using NumPy and Pandas to preprocess the data and produce any intermediate variables listed in 'create'
    - Follow best practices in preprocessing, including:
        - Identifying and separating numeric and categorical columns into: numeric_columns and categorical_columns
        - Handling null values appropriately
        - Converting string-based date columns to datetime using safe conversions
        - Creating a correlation matrix for numeric columns
    - Ensure variable names follow the instructions exactly
    - Do not read data from CSV; `df` is already loaded
    - If you need to convert to datetime, use:
        ```python
        def safe_to_datetime(date):
            try:
                return pd.to_datetime(date, errors='coerce', cache=False)
            except (ValueError, TypeError):
                return pd.NaT
        df['datetime_column'] = df['datetime_column'].apply(safe_to_datetime)
        ```
    - If visualizing (e.g., heatmaps), use Plotly (not matplotlib)
    - Write output to the console using `print()`, as this is standard Python code

    Your output should include:
    - Python code that satisfies the instructions
    - A summary explaining what was done and why

    Input Fields:
    - dataset: The original dataset (`df`) and column info
    - goal: The user's goal (e.g., predictive modeling, exploration, cleaning)
    - plan_instructions: A dictionary with keys:
        - 'create': list of variable names you are required to generate
        - 'receive': list of variable names you are receiving

    Output Fields:
    - code: Python code that does the requested preprocessing
    - summary: Comments on what analysis/preprocessing was performed and why
    """
    dataset = dspy.InputField(desc="The dataset, preloaded as df")
    goal = dspy.InputField(desc="User-defined goal for the analysis")
    plan_instructions = dspy.InputField(desc="Agent-level instructions about what to create and receive")
    
    code = dspy.OutputField(desc="Generated Python code for preprocessing")
    summary = dspy.OutputField(desc="Explanation of what was done and why")

class planner_statistical_analytics_agent(dspy.Signature):
    """
    You are a statistical analytics agent in a multi-agent data analytics system.

    You are given:
    - A dataset (usually a cleaned or transformed version like `df_cleaned`)
    - A user-defined goal (e.g., regression, seasonal decomposition)
    - Agent-specific plan instructions that define:
        - Which variables you are expected to **create** (e.g., regression_model)
        - Which variables you will **receive** (e.g., df_cleaned, target_variable)

    Your job is to:
    - Generate executable Python code using the StatsModels library to perform the required statistical analysis.
    - Use best practices for model building and error handling.
    - Follow these constraints:
        - Strings should be handled as categorical variables via `C(col)` in model formulas
        - Add a constant using `sm.add_constant()`
        - Do not modify the DataFrame's index
        - Convert X and y to float before fitting
        - Handle missing values before modeling
        - Avoid data visualization here (that is handled by other agents)
        - Write output to the console using `print()`

    If the goal is regression:
        - Use statsmodels OLS with proper handling of categorical variables

    If the goal is seasonal decomposition:
        - Use `statsmodels.tsa.seasonal_decompose`
        - Ensure the time series and period are correctly provided

    Use this function structure as a safe template:
    ```python
    import statsmodels.api as sm

    def statistical_model(X, y, goal, period=None):
        try:
            X = X.dropna()
            y = y.loc[X.index].dropna()
            X = X.loc[y.index]
            for col in X.select_dtypes(include=['object', 'category']).columns:
                X[col] = X[col].astype('category')
            X = sm.add_constant(X)
            if goal == 'regression':
                formula = 'y ~ ' + ' + '.join([f'C({col})' if X[col].dtype.name == 'category' else col for col in X.columns])
                model = sm.OLS(y.astype(float), X.astype(float)).fit()
                return model.summary()
            elif goal == 'seasonal_decompose':
                if period is None:
                    raise ValueError("Period must be specified for seasonal decomposition")
                decomposition = sm.tsa.seasonal_decompose(y, period=period)
                return decomposition
            else:
                raise ValueError("Unknown goal specified.")
        except Exception as e:
            return f"An error occurred: {e}"
    ```

    Input Fields:
    - dataset: Typically a cleaned dataset like `df_cleaned`
    - goal: The user's intended statistical analysis
    - plan_instructions: A dictionary with keys:
        - 'create': list of variable names to generate (e.g., 'regression_model')
        - 'receive': list of variables provided to this agent (e.g., df_cleaned, X, y, period)

    Output Fields:
    - code: The Python code implementing the statistical analysis
    - summary: Explanation of the analysis logic and rationale
    """
    dataset = dspy.InputField(desc="Preprocessed dataset, often named df_cleaned")
    goal = dspy.InputField(desc="The user's statistical analysis goal, e.g., regression or seasonal_decompose")
    plan_instructions = dspy.InputField(desc="Instructions on variables to create and receive for statistical modeling")
    
    code = dspy.OutputField(desc="Python code for statistical modeling using statsmodels")
    summary = dspy.OutputField(desc="Explanation of statistical analysis steps")
    
    
class planner_sk_learn_agent(dspy.Signature):
    """
    You are a machine learning agent in a multi-agent data analytics pipeline.

    You are given:
    - A dataset (often cleaned and feature-engineered)
    - A user-defined goal (e.g., classification, regression, clustering)
    - Agent-specific plan instructions specifying:
        - Which variables you are expected to **create** (e.g., trained_model, predictions)
        - Which variables you will **receive** (e.g., df_cleaned, target_variable, feature_columns)

    Your responsibilities:
    - Use the scikit-learn library to implement the appropriate ML pipeline
    - Always split data into training and testing sets where applicable
    - Use `print()` for all outputs.
    - Ensure your code is:
        - Reproducible (set `random_state=42`)
        - Modular (avoid deeply nested code)
        - Focused on model building, not visualization (leave plotting to the `data_viz_agent`)
    - Your task may include:
        - Preprocessing inputs (e.g., encoding)
        - Model selection and training
        - Evaluation (e.g., accuracy, RMSE, classification report)

    You must not:
    - Visualize anything (that's another agent's job)
    - Rely on hardcoded column names — use those passed via `plan_instructions`

    Example safe structure:
    ```python
    from sklearn.model_selection import train_test_split
    from sklearn.linear_model import LogisticRegression
    from sklearn.metrics import classification_report

    X = df[feature_columns]
    y = df[target_column]
    X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)

    model = LogisticRegression()
    model.fit(X_train, y_train)

    predictions = model.predict(X_test)
    st.write(classification_report(y_test, predictions))
    ```

    Input Fields:
    - dataset: Typically a cleaned dataset (like df_cleaned)
    - goal: User-defined machine learning goal (e.g., "predict churn", "cluster customers")
    - plan_instructions: A dictionary with:
        - 'create': list of output variables to generate (e.g., 'trained_model', 'metrics')
        - 'receive': list of inputs (e.g., df_cleaned, feature_columns, target_column)

    Output Fields:
    - code: Python code implementing the machine learning task
    - summary: A description of what the model does, how it's evaluated, and why it fits the goal
    """
    dataset = dspy.InputField(desc="Input dataset, often cleaned and feature-selected (e.g., df_cleaned)")
    goal = dspy.InputField(desc="The user's machine learning goal (e.g., classification or regression)")
    plan_instructions = dspy.InputField(desc="Instructions indicating what to create and what variables to receive")

    code = dspy.OutputField(desc="Scikit-learn based machine learning code")
    summary = dspy.OutputField(desc="Explanation of the ML approach and evaluation")

    

class goal_refiner_agent(dspy.Signature):
    # Called to refine the query incase user query not elaborate
    """You take a user-defined goal given to a AI data analyst planner agent, 
    you make the goal more elaborate using the datasets available and agent_desc"""
    dataset = dspy.InputField(desc="Available datasets loaded in the system, use this df,columns  set df as copy of df")
    Agent_desc = dspy.InputField(desc= "The agents available in the system")
    goal = dspy.InputField(desc="The user defined goal ")
    refined_goal = dspy.OutputField(desc='Refined goal that helps the planner agent plan better')

class preprocessing_agent(dspy.Signature):
    """You are a AI data-preprocessing agent. Generate clean and efficient Python code using NumPy and Pandas to perform introductory data preprocessing on a pre-loaded DataFrame df, based on the user's analysis goals.

    Preprocessing Requirements:

    1. Identify Column Types
    - Separate columns into numeric and categorical using:
        categorical_columns = df.select_dtypes(include=[object, 'category']).columns.tolist()
        numeric_columns = df.select_dtypes(include=[np.number]).columns.tolist()

    2. Handle Missing Values
    - Numeric columns: Impute missing values using the mean of each column
    - Categorical columns: Impute missing values using the mode of each column

    3. Convert Date Strings to Datetime
    - For any column suspected to represent dates (in string format), convert it to datetime using:
        def safe_to_datetime(date):
            try:
                return pd.to_datetime(date, errors='coerce', cache=False)
            except (ValueError, TypeError):
                return pd.NaT
        df['datetime_column'] = df['datetime_column'].apply(safe_to_datetime)
    - Replace 'datetime_column' with the actual column names containing date-like strings

    Important Notes:
    - Do NOT create a correlation matrix — correlation analysis is outside the scope of preprocessing
    - Do NOT generate any plots or visualizations

    Output Instructions:
    1. Include the full preprocessing Python code
    2. Provide a brief bullet-point summary of the steps performed. Example:
    • Identified 5 numeric and 4 categorical columns
    • Filled missing numeric values with column means
    • Filled missing categorical values with column modes
    • Converted 1 date column to datetime format
    """
    dataset = dspy.InputField(desc="Available datasets loaded in the system, use this df, column_names  set df as copy of df")
    goal = dspy.InputField(desc="The user defined goal could ")
    code = dspy.OutputField(desc ="The code that does the data preprocessing and introductory analysis")
    summary = dspy.OutputField(desc="A concise bullet-point summary of the preprocessing operations performed")
    


class statistical_analytics_agent(dspy.Signature):
    # Statistical Analysis Agent, builds statistical models using StatsModel Package
    """ 
    You are a statistical analytics agent. Your task is to take a dataset and a user-defined goal and output Python code that performs the appropriate statistical analysis to achieve that goal. Follow these guidelines:

    IMPORTANT: You may be provided with previous interaction history. The section marked "### Current Query:" contains the user's current request. Any text in "### Previous Interaction History:" is for context only and is NOT part of the current request.

    Data Handling:

    Always handle strings as categorical variables in a regression using statsmodels C(string_column).
    Do not change the index of the DataFrame.
    Convert X and y into float when fitting a model.
    Error Handling:

    Always check for missing values and handle them appropriately.
    Ensure that categorical variables are correctly processed.
    Provide clear error messages if the model fitting fails.
    Regression:

    For regression, use statsmodels and ensure that a constant term is added to the predictor using sm.add_constant(X).
    Handle categorical variables using C(column_name) in the model formula.
    Fit the model with model = sm.OLS(y.astype(float), X.astype(float)).fit().
    Seasonal Decomposition:

    Ensure the period is set correctly when performing seasonal decomposition.
    Verify the number of observations works for the decomposition.
    Output:

    Ensure the code is executable and as intended.
    Also choose the correct type of model for the problem
    Avoid adding data visualization code.

    Use code like this to prevent failing:
    import pandas as pd
    import numpy as np
    import statsmodels.api as sm

    def statistical_model(X, y, goal, period=None):
        try:
            # Check for missing values and handle them
            X = X.dropna()
            y = y.loc[X.index].dropna()

            # Ensure X and y are aligned
            X = X.loc[y.index]

            # Convert categorical variables
            for col in X.select_dtypes(include=['object', 'category']).columns:
                X[col] = X[col].astype('category')

            # Add a constant term to the predictor
            X = sm.add_constant(X)

            # Fit the model
            if goal == 'regression':
                # Handle categorical variables in the model formula
                formula = 'y ~ ' + ' + '.join([f'C({col})' if X[col].dtype.name == 'category' else col for col in X.columns])
                model = sm.OLS(y.astype(float), X.astype(float)).fit()
                return model.summary()

            elif goal == 'seasonal_decompose':
                if period is None:
                    raise ValueError("Period must be specified for seasonal decomposition")
                decomposition = sm.tsa.seasonal_decompose(y, period=period)
                return decomposition

            else:
                raise ValueError("Unknown goal specified. Please provide a valid goal.")

        except Exception as e:
            return f"An error occurred: {e}"

    # Example usage:
    result = statistical_analysis(X, y, goal='regression')
    print(result)

    If visualizing use plotly

    Provide a concise bullet-point summary of the statistical analysis performed.
    
    Example Summary:
    • Applied linear regression with OLS to predict house prices based on 5 features
    • Model achieved R-squared of 0.78
    • Significant predictors include square footage (p<0.001) and number of bathrooms (p<0.01)
    • Detected strong seasonal pattern with 12-month periodicity
    • Forecast shows 15% growth trend over next quarter

    """
    dataset = dspy.InputField(desc="Available datasets loaded in the system, use this df,columns  set df as copy of df")
    goal = dspy.InputField(desc="The user defined goal for the analysis to be performed")
    code = dspy.OutputField(desc ="The code that does the statistical analysis using statsmodel")
    summary = dspy.OutputField(desc="A concise bullet-point summary of the statistical analysis performed and key findings")
    

class sk_learn_agent(dspy.Signature):
    # Machine Learning Agent, performs task using sci-kit learn
    """You are a machine learning agent. 
    Your task is to take a dataset and a user-defined goal, and output Python code that performs the appropriate machine learning analysis to achieve that goal. 
    You should use the scikit-learn library.

    IMPORTANT: You may be provided with previous interaction history. The section marked "### Current Query:" contains the user's current request. Any text in "### Previous Interaction History:" is for context only and is NOT part of the current request.

    Make sure your output is as intended!

    Provide a concise bullet-point summary of the machine learning operations performed.
    
    Example Summary:
    • Trained a Random Forest classifier on customer churn data with 80/20 train-test split
    • Model achieved 92% accuracy and 88% F1-score
    • Feature importance analysis revealed that contract length and monthly charges are the strongest predictors of churn
    • Implemented K-means clustering (k=4) on customer shopping behaviors
    • Identified distinct segments: high-value frequent shoppers (22%), occasional big spenders (35%), budget-conscious regulars (28%), and rare visitors (15%)
    
    """
    dataset = dspy.InputField(desc="Available datasets loaded in the system, use this df,columns. set df as copy of df")
    goal = dspy.InputField(desc="The user defined goal ")
    code = dspy.OutputField(desc ="The code that does the Exploratory data analysis")
    summary = dspy.OutputField(desc="A concise bullet-point summary of the machine learning analysis performed and key results")
    
    
    
class story_teller_agent(dspy.Signature):
    # Optional helper agent, which can be called to build a analytics story 
    # For all of the analysis performed
    """ You are a story teller agent, taking output from different data analytics agents, you compose a compelling story for what was done """
    agent_analysis_list =dspy.InputField(desc="A list of analysis descriptions from every agent")
    story = dspy.OutputField(desc="A coherent story combining the whole analysis")

class code_combiner_agent(dspy.Signature):
    # Combines code from different agents into one script
    """ You are a code combine agent, taking Python code output from many agents and combining the operations into 1 output
    You also fix any errors in the code. 

    IMPORTANT: You may be provided with previous interaction history. The section marked "### Current Query:" contains the user's current request. Any text in "### Previous Interaction History:" is for context only and is NOT part of the current request.

    Double check column_names/dtypes using dataset, also check if applied logic works for the datatype
    df = df.copy()
    Also add this to display Plotly chart
    fig.show()

    Make sure your output is as intended!

    Provide a concise bullet-point summary of the code integration performed.
    
    Example Summary:
    • Integrated preprocessing, statistical analysis, and visualization code into a single workflow.
    • Fixed variable scope issues, standardized DataFrame handling (e.g., using `df.copy()`), and corrected errors.
    • Validated column names and data types against the dataset definition to prevent runtime issues.
    • Ensured visualizations are displayed correctly (e.g., added `fig.show()`).

    """
    dataset = dspy.InputField(desc="Use this double check column_names, data types")
    agent_code_list =dspy.InputField(desc="A list of code given by each agent")
    refined_complete_code = dspy.OutputField(desc="Refined complete code base")
    summary = dspy.OutputField(desc="A concise 4 bullet-point summary of the code integration performed and improvements made")
    
    
class data_viz_agent(dspy.Signature):
    # Visualizes data using Plotly
    """    
    You are an AI agent responsible for generating interactive data visualizations using Plotly.

    IMPORTANT Instructions:

    - The section marked "### Current Query:" contains the user's request. Any text in "### Previous Interaction History:" is for context only and should NOT be treated as part of the current request.
    - You must only use the tools provided to you. This agent handles visualization only.
    - If len(df) > 50000, always sample the dataset before visualization using:  
    if len(df) > 50000:  
        df = df.sample(50000, random_state=1)

    - Each visualization must be generated as a **separate figure** using go.Figure().  
    Do NOT use subplots under any circumstances.

    - Each figure must be returned individually using:  
    fig.to_html(full_html=False)

    - Use update_layout with xaxis and yaxis **only once per figure**.

    - Enhance readability and clarity by:  
    • Using low opacity (0.4-0.7) where appropriate  
    • Applying visually distinct colors for different elements or categories  

    - Make sure the visual **answers the user's specific goal**:  
    • Identify what insight or comparison the user is trying to achieve  
    • Choose the visualization type and features (e.g., color, size, grouping) to emphasize that goal  
    • For example, if the user asks for "trends in revenue," use a time series line chart; if they ask for "top-performing categories," use a bar chart sorted by value  
    • Prioritize highlighting patterns, outliers, or comparisons relevant to the question

    - Never include the dataset or styling index in the output.

    - If there are no relevant columns for the requested visualization, respond with:  
    "No relevant columns found to generate this visualization."

    - Use only one number format consistently: either 'K', 'M', or comma-separated values like 1,000/1,000,000. Do not mix formats.

    - Only include trendlines in scatter plots if the user explicitly asks for them.

    - Output only the code and a concise bullet-point summary of what the visualization reveals.

    - Always end each visualization with:  
    fig.to_html(full_html=False)

    Example Summary:  
    • Created an interactive scatter plot of sales vs. marketing spend with color-coded product categories  
    • Included a trend line showing positive correlation (r=0.72)  
    • Highlighted outliers where high marketing spend resulted in low sales  
    • Generated a time series chart of monthly revenue from 2020-2023  
    • Added annotations for key business events  
    • Visualization reveals 35% YoY growth with seasonal peaks in Q4
    """
    goal = dspy.InputField(desc="user defined goal which includes information about data and chart they want to plot")
    dataset = dspy.InputField(desc=" Provides information about the data in the data frame. Only use column names and dataframe_name as in this context")
    styling_index = dspy.InputField(desc='Provides instructions on how to style your Plotly plots')
    code= dspy.OutputField(desc="Plotly code that visualizes what the user needs according to the query & dataframe_index & styling_context")
    summary = dspy.OutputField(desc="A concise bullet-point summary of the visualization created and key insights revealed")
    
    

class code_fix(dspy.Signature):
    """
You are an expert AI developer and data analyst assistant, skilled at identifying and resolving issues in Python code related to data analytics. Another agent has attempted to generate Python code for a data analytics task but produced code that is broken or throws an error.

Your task is to:
1. Carefully examine the provided **faulty_code** and the corresponding **error** message.
2. Identify the **exact cause** of the failure based on the error and surrounding context.
3. Modify only the necessary portion(s) of the code to fix the issue, utilizing the **dataset_context** to inform your corrections.
4. Ensure the **intended behavior** of the original code is preserved (e.g., if the code is meant to filter, group, or visualize data, that functionality must be preserved).
5. Ensure the final output is **runnable**, **error-free**, and **logically consistent**.

Strict instructions:
- Do **not** modify any working parts of the code unnecessarily.
- Do **not** change variable names, structure, or logic unless it directly contributes to resolving the issue.
- Do **not** output anything besides the corrected, full version of the code (i.e., no explanations, comments, or logs).
- Avoid introducing new dependencies or libraries unless absolutely required to fix the problem.
- The output must be complete and executable as-is.

Be precise, minimal, and reliable. Prioritize functional correctness.
    """
    dataset_context = dspy.InputField(desc="The dataset context to be used for the code fix")
    faulty_code = dspy.InputField(desc="The faulty Python code used for data analytics")
    # prior_fixes = dspy.InputField(desc="If a fix for this code exists in our error retriever", default="use the error message")
    error = dspy.InputField(desc="The error message thrown when running the code")
    fixed_code = dspy.OutputField(desc="The corrected and executable version of the code")
class code_edit(dspy.Signature):
    """
You are an expert AI code editor that specializes in modifying existing data analytics code based on user requests. The user provides a working or partially working code snippet, a natural language prompt describing the desired change, and dataset context information.

Your job is to:
1. Analyze the provided original_code, user_prompt, and dataset_context.
2. Modify only the part(s) of the code that are relevant to the user's request, using the dataset context to inform your edits.
3. Leave all unrelated parts of the code unchanged, unless the user explicitly requests a full rewrite or broader changes.
4. Ensure that your changes maintain or improve the functionality and correctness of the code.

Your edits may include:
- Bug fixes or logic corrections (if requested)
- Plot and visualization styling changes
- Optimization or simplification
- Code reformatting or restructuring (if asked for)
- Adjusting data processing or analysis steps
- Any other edits specifically described in the user prompt

Strict requirements:
- Do not change variable names, function structures, or logic outside the scope of the user's request.
- Do not refactor, optimize, or rewrite unless explicitly instructed.
- Ensure the edited code remains complete and executable.
- Output only the modified code, without any additional explanation, comments, or extra formatting.

Make your edits precise, minimal, and faithful to the user's instructions, using the dataset context to guide your modifications.
    """
    dataset_context = dspy.InputField(desc="The dataset context to be used for the code edit, including information about the dataset's shape, columns, types, and null values")
    original_code = dspy.InputField(desc="The original code the user wants modified")
    user_prompt = dspy.InputField(desc="The user instruction describing how the code should be changed")
    edited_code = dspy.OutputField(desc="The updated version of the code reflecting the user's request, incorporating changes informed by the dataset context")

# The ind module is called when agent_name is 
# explicitly mentioned in the query
class auto_analyst_ind(dspy.Module):
    """Handles individual agent execution when explicitly specified in query"""
    
    def __init__(self, agents, retrievers):
        # Initialize agent modules and retrievers
        self.agents = {}
        self.agent_inputs = {}
        self.agent_desc = []
        
        # Create modules from agent signatures
        for i, a in enumerate(agents):
            name = a.__pydantic_core_schema__['schema']['model_name']
            self.agents[name] = dspy.ChainOfThoughtWithHint(a)
            self.agent_inputs[name] = {x.strip() for x in str(agents[i].__pydantic_core_schema__['cls']).split('->')[0].split('(')[1].split(',')}
            self.agent_desc.append(get_agent_description(name))
            
        # Initialize components
        self.memory_summarize_agent = dspy.ChainOfThought(m.memory_summarize_agent)
        self.dataset = retrievers['dataframe_index'].as_retriever(k=1)
        self.styling_index = retrievers['style_index'].as_retriever(similarity_top_k=1)
        self.code_combiner_agent = dspy.ChainOfThought(code_combiner_agent)
        
        # Initialize thread pool
        self.executor = ThreadPoolExecutor(max_workers=min(4, os.cpu_count() * 2))
    
    def execute_agent(self, specified_agent, inputs):
        """Execute agent and generate memory summary in parallel"""
        try:
            # Execute main agent
            agent_result = self.agents[specified_agent.strip()](**inputs)
            return specified_agent.strip(), dict(agent_result)
            
        except Exception as e:
            return specified_agent.strip(), {"error": str(e)}

    def execute_agent_with_memory(self, specified_agent, inputs, query):
        """Execute agent and generate memory summary in parallel"""
        try:
            # Execute main agent
            agent_result = self.agents[specified_agent.strip()](**inputs)
            agent_dict = dict(agent_result)
            
            # Generate memory summary
            memory_result = self.memory_summarize_agent(
                agent_response=specified_agent+' '+agent_dict['code']+'\n'+agent_dict['summary'],
                user_goal=query
            )
            
            return {
                specified_agent.strip(): agent_dict,
                'memory_'+specified_agent.strip(): str(memory_result.summary)
            }
        except Exception as e:
            return {"error": str(e)}

    def forward(self, query, specified_agent):
        try:
            # If specified_agent contains multiple agents separated by commas
            # This is for handling multiple @agent mentions in one query
            if "," in specified_agent:
                agent_list = [agent.strip() for agent in specified_agent.split(",")]
                return self.execute_multiple_agents(query, agent_list)
            
            # Process query with specified agent (single agent case)
            dict_ = {}
            dict_['dataset'] = self.dataset.retrieve(query)[0].text
            dict_['styling_index'] = self.styling_index.retrieve(query)[0].text
            dict_['hint'] = []
            dict_['goal'] = query
            dict_['Agent_desc'] = str(self.agent_desc)

            # Prepare inputs
            inputs = {x:dict_[x] for x in self.agent_inputs[specified_agent.strip()]}
            inputs['hint'] = str(dict_['hint']).replace('[','').replace(']','')
            
            # Execute agent
            result = self.agents[specified_agent.strip()](**inputs)
            output_dict = {specified_agent.strip(): dict(result)}

            if "error" in output_dict:
                return {"response": f"Error executing agent: {output_dict['error']}"}

            return output_dict

        except Exception as e:
            return {"response": f"This is the error from the system: {str(e)}"}
    
    def execute_multiple_agents(self, query, agent_list):
        """Execute multiple agents sequentially on the same query"""
        try:
            # Initialize resources
            dict_ = {}
            dict_['dataset'] = self.dataset.retrieve(query)[0].text
            dict_['styling_index'] = self.styling_index.retrieve(query)[0].text
            dict_['hint'] = []
            dict_['goal'] = query
            dict_['Agent_desc'] = str(self.agent_desc)
            
            results = {}
            code_list = []
            
            # Execute each agent sequentially
            for agent_name in agent_list:
                if agent_name not in self.agents:
                    results[agent_name] = {"error": f"Agent '{agent_name}' not found"}
                    continue
                
                # Prepare inputs for this agent
                inputs = {x:dict_[x] for x in self.agent_inputs[agent_name] if x in dict_}
                inputs['hint'] = str(dict_['hint']).replace('[','').replace(']','')
                
                # Execute agent
                agent_result = self.agents[agent_name](**inputs)
                agent_dict = dict(agent_result)
                results[agent_name] = agent_dict
                
                # Collect code for later combination
                if 'code' in agent_dict:
                    code_list.append(agent_dict['code'])
            
            return results
            
        except Exception as e:
            return {"response": f"Error executing multiple agents: {str(e)}"}

# This is the auto_analyst with planner
class auto_analyst(dspy.Module):
    """Main analyst module that coordinates multiple agents using a planner"""
    
    def __init__(self, agents, retrievers):
        # Initialize agent modules and retrievers
        self.agents = {}
        self.agent_inputs = {}
        self.agent_desc = []
        
        for i, a in enumerate(agents):
            name = a.__pydantic_core_schema__['schema']['model_name']
            self.agents[name] = dspy.ChainOfThought(a)
            self.agent_inputs[name] = {x.strip() for x in str(agents[i].__pydantic_core_schema__['cls']).split('->')[0].split('(')[1].split(',')}
            self.agent_desc.append({name: get_agent_description(name, is_planner=True)})
        
        # Initialize coordination agents
        self.planner = dspy.ChainOfThought(analytical_planner)
        self.refine_goal = dspy.ChainOfThought(goal_refiner_agent)
        # self.code_combiner_agent = dspy.ChainOfThought(code_combiner_agent)
        self.story_teller = dspy.ChainOfThought(story_teller_agent)
        self.memory_summarize_agent = dspy.ChainOfThought(m.memory_summarize_agent)
        self.parallelizer = dspy.Parallel(num_threads=10)
                
        # Initialize retrievers
        self.dataset = retrievers['dataframe_index'].as_retriever(k=1)
        self.styling_index = retrievers['style_index'].as_retriever(similarity_top_k=1)

    def forward(self, query):
        # Create shared dictionary for inputs
        dict_ = {}
        dict_['dataset'] = self.dataset.retrieve(query)[0].text
        dict_['styling_index'] = self.styling_index.retrieve(query)[0].text
        dict_['hint'] = []
        dict_['goal'] = query
        dict_['Agent_desc'] = str(self.agent_desc)

        # Get plan from planner agent
        plan = self.planner(goal=dict_['goal'], dataset=dict_['dataset'], Agent_desc=dict_['Agent_desc'])
        
        # Parse the plan
        output_dict = {'analytical_planner': plan}
        plan_text = plan.plan.replace('Plan', '').replace(':', '').strip()
        
        # If no plan is created, refine the goal and try again
        if not plan_text or not plan_text.split('->'):
            refined_goal = self.refine_goal(dataset=dict_['dataset'], goal=dict_['goal'], Agent_desc=dict_['Agent_desc'])
            return self.forward(query=refined_goal.refined_goal)
        
        plan_list = [p.strip() for p in plan_text.split('->') if p.strip()]
        
        # Get plan instructions if available, or create default empty ones
        try:
            plan_instructions = eval(plan.plan_instructions)
        except:
            plan_instructions = {p.strip(): {'create': [], 'receive': []} for p in plan_list}
        
        # Prepare parallel execution list
        parallel_list = []
        for p in plan_list:
            inputs = {x: dict_[x] for x in self.agent_inputs[p.strip()] if x != 'plan_instructions'}
            if 'plan_instructions' in self.agent_inputs[p.strip()]:
                if p.strip() in plan_instructions:
                    inputs['plan_instructions'] = plan_instructions[p.strip()]
                else:
                    inputs['plan_instructions'] = {'create': [], 'receive': []}
            parallel_list.append((self.agents[p.strip()], inputs))
        
        # Execute agents in parallel
        results = self.parallelizer(parallel_list)
        
        # Process results
        output_dict = {'analytical_planner': dict(plan)}
        code_list = []
        
        for i, result in enumerate(results):
            agent_name = plan_list[i].strip()
            output_dict[agent_name] = dict(result)
            if hasattr(result, 'code'):
                code_list.append(result.code)
                
        code_combiner_info = {}
        # try:
        #     with dspy.context(lm=dspy.LM(model="gemini/gemini-2.5-pro-preview-03-25", api_key = os.environ['GEMINI_API_KEY'], max_tokens=10000)):
        #         combiner_result = self.code_combiner_agent(agent_code_list=str(code_list), dataset=dict_['dataset'])
        #         code_combiner_info = {
        #             'code_combiner_agent_name': 'gemini-2.5-pro-preview-03-25',
        #             'code_list': str(code_list),
        #             'combiner_results': dict(combiner_result)
        #         }
        # except:
        #     try: 
        #         # Set max_tokens to a fixed value of 10000 for all models
        #         with dspy.context(lm=dspy.LM(model="o3-mini", max_tokens=10000, temperature=1.0, api_key=os.getenv("OPENAI_API_KEY"))):
        #             combiner_result = self.code_combiner_agent(agent_code_list=str(code_list), dataset=dict_['dataset'])
        #             code_combiner_info = {
        #                 'code_combiner_agent_name': 'o3-mini',
        #                 'code_list': str(code_list),
        #                 'combiner_results': dict(combiner_result)
        #             }
        #     except:
        #         try: 
        #             with dspy.context(lm=dspy.LM(model="claude-3-7-sonnet-latest", max_tokens=100000, temperature=1.0, api_key=os.getenv("ANTHROPIC_API_KEY"))):
        #                 combiner_result = self.code_combiner_agent(agent_code_list=str(code_list), dataset=dict_['dataset'])
        #                 code_combiner_info = {
        #                     'code_combiner_agent_name': 'claude-3-7-sonnet-latest',
        #                     'code_list': str(code_list),
        #                     'combiner_results': dict(combiner_result)
        #                 }
        #         except Exception as e:
        #             code_combiner_info = {
        #                 'code_combiner_agent_name': 'none',
        #                 'code_list': str(code_list),
        #                 'combiner_results': {"error": f"Error in code combiner: {str(e)}"}
        #             }
        
        # # Add the code combiner info to the output dictionary
        output_dict['code_combiner_agent'] = code_combiner_info
            
        return output_dict