File size: 4,985 Bytes
e317359
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
import matplotlib.pyplot as plt
import numpy as np
import os


def normalize_data(mean, std):
    """
    Normalize the data using Min-Max Scaling to bring the mean and standard deviation between 0 and 1.

    Parameters:
    - mean: The mean values of the data.
    - std: The standard deviation values of the data.

    Returns:
    - Tuple of normalized mean and standard deviation.
    """
    # Apply Min-Max Scaling to normalize the data between 0 and 1
    mean_normalized = (mean - mean.min()) / (mean.max() - mean.min())
    # Normalize std by the same range for consistency
    std_normalized = std / (mean.max() - mean.min())
    return mean_normalized, std_normalized


def plot_radar_chart(df, output_dir):
    """
    Plot a radar chart for the given DataFrame and save it to the specified output directory.

    Parameters:
    - df: DataFrame containing the data to plot.
    - output_dir: Directory where the radar chart will be saved.
    """
    # Calculate mean and std across rows for each column
    mean = df.mean()
    std = df.std()

    # Normalize data
    mean_normalized, std_normalized = normalize_data(mean, std)

    # Number of variables we're plotting.
    num_vars = len(df.columns)

    # Split the circle into even parts and save angles so we know where to put each axis.
    angles = np.linspace(0, 2 * np.pi, num_vars, endpoint=False).tolist()
    # Complete the loop for the plot
    angles += angles[:1]
    mean_normalized = mean_normalized.tolist() + mean_normalized.tolist()[:1]
    std_normalized = std_normalized.tolist() + std_normalized.tolist()[:1]

    # Draw the plot
    fig, ax = plt.subplots(figsize=(6, 6), subplot_kw=dict(polar=True))
    ax.fill(angles, mean_normalized, color='red', alpha=0.25)
    ax.plot(angles, mean_normalized, color='red', label='Mean')

    # Draw error bars
    for angle, mean, error in zip(angles, mean_normalized, std_normalized):
        ax.errorbar(angle, mean, yerr=error, color='black', capsize=3)

    ax.set_ylim(0, 1)

    # Labels for each feature
    labels = df.columns.tolist()
    ax.set_xticks(angles[:-1])
    ax.set_xticklabels(labels, size=12)

    # Title and legend
    plt.title('Time Series Benchmark Features (Normalized)',
              size=15, color='red', y=1.1)
    ax.legend(loc='upper right', bbox_to_anchor=(1.1, 1.1))

    # Save the figure
    plt.savefig(os.path.join(output_dir, 'character_radar.png'))
    plt.close()


def persist_analysis(features_df, output_dir):
    """
    Persist the analysis results by saving the features DataFrame and its description to CSV files.

    Parameters:
    - features_df: DataFrame containing the features to be saved.
    - output_dir: Directory where the CSV files will be saved.
    """
    features_df.to_csv(f"{output_dir}/features.csv")
    features_df.describe().to_csv(f"{output_dir}/features_description.csv")

    # Create histograms of each characteristic
    for column in features_df.columns:
        plot_feature_histogram(features_df, column, output_dir)
    # Create a ring plot showing the mean and std of some key features, trend and seasonality should be there for sure.
    plot_radar_chart(features_df, output_dir)


def plot_histogram(freq_distribution_dict, name, output_dir):
    """
    Plot the histogram for the given frequency distribution dictionary and save it to the output directory.

    Parameters:
    - freq_distribution_dict: Dictionary with frequency strings as keys and counts as values.
    - name: Name of the frequency distribution.
    - output_dir: Directory where the histogram will be saved.
    """
    fig, ax = plt.subplots()
    ax.bar(freq_distribution_dict.keys(), freq_distribution_dict.values())
    ax.set_xlabel(f"Frequency of {name}")
    ax.set_ylabel("Count")
    ax.set_title(f"Distribution of {name} frequencies")
    plt.savefig(f"{output_dir}/{name}_frequency_distribution.png")
    plt.close()


def plot_feature_histogram(dataframe, column_name, output_directory):
    """
    Plot a histogram for the given column in the DataFrame and save it to the specified output directory.

    Parameters:
    - dataframe: The DataFrame containing the data.
    - column_name: The column name for which to plot the histogram.
    - output_directory: The directory where the plot will be saved.
    """
    # Check if the column exists in the dataframe
    if column_name not in dataframe.columns:
        raise ValueError(
            f"Column '{column_name}' does not exist in the dataframe.")

    # Drop rows with NA values in the specified column
    clean_data = dataframe[column_name].dropna()

    # Plot the histogram
    plt.figure(figsize=(10, 6))
    plt.hist(clean_data, bins=10, edgecolor='black')
    plt.title(f'Histogram of {column_name}')
    plt.xlabel(column_name)
    plt.ylabel('Frequency')

    # Save the plot
    plot_filename = os.path.join(
        output_directory, f"{column_name}_histogram.png")
    plt.savefig(plot_filename)
    plt.close()