Spaces:
Running
Running
File size: 4,985 Bytes
e317359 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 | import matplotlib.pyplot as plt
import numpy as np
import os
def normalize_data(mean, std):
"""
Normalize the data using Min-Max Scaling to bring the mean and standard deviation between 0 and 1.
Parameters:
- mean: The mean values of the data.
- std: The standard deviation values of the data.
Returns:
- Tuple of normalized mean and standard deviation.
"""
# Apply Min-Max Scaling to normalize the data between 0 and 1
mean_normalized = (mean - mean.min()) / (mean.max() - mean.min())
# Normalize std by the same range for consistency
std_normalized = std / (mean.max() - mean.min())
return mean_normalized, std_normalized
def plot_radar_chart(df, output_dir):
"""
Plot a radar chart for the given DataFrame and save it to the specified output directory.
Parameters:
- df: DataFrame containing the data to plot.
- output_dir: Directory where the radar chart will be saved.
"""
# Calculate mean and std across rows for each column
mean = df.mean()
std = df.std()
# Normalize data
mean_normalized, std_normalized = normalize_data(mean, std)
# Number of variables we're plotting.
num_vars = len(df.columns)
# Split the circle into even parts and save angles so we know where to put each axis.
angles = np.linspace(0, 2 * np.pi, num_vars, endpoint=False).tolist()
# Complete the loop for the plot
angles += angles[:1]
mean_normalized = mean_normalized.tolist() + mean_normalized.tolist()[:1]
std_normalized = std_normalized.tolist() + std_normalized.tolist()[:1]
# Draw the plot
fig, ax = plt.subplots(figsize=(6, 6), subplot_kw=dict(polar=True))
ax.fill(angles, mean_normalized, color='red', alpha=0.25)
ax.plot(angles, mean_normalized, color='red', label='Mean')
# Draw error bars
for angle, mean, error in zip(angles, mean_normalized, std_normalized):
ax.errorbar(angle, mean, yerr=error, color='black', capsize=3)
ax.set_ylim(0, 1)
# Labels for each feature
labels = df.columns.tolist()
ax.set_xticks(angles[:-1])
ax.set_xticklabels(labels, size=12)
# Title and legend
plt.title('Time Series Benchmark Features (Normalized)',
size=15, color='red', y=1.1)
ax.legend(loc='upper right', bbox_to_anchor=(1.1, 1.1))
# Save the figure
plt.savefig(os.path.join(output_dir, 'character_radar.png'))
plt.close()
def persist_analysis(features_df, output_dir):
"""
Persist the analysis results by saving the features DataFrame and its description to CSV files.
Parameters:
- features_df: DataFrame containing the features to be saved.
- output_dir: Directory where the CSV files will be saved.
"""
features_df.to_csv(f"{output_dir}/features.csv")
features_df.describe().to_csv(f"{output_dir}/features_description.csv")
# Create histograms of each characteristic
for column in features_df.columns:
plot_feature_histogram(features_df, column, output_dir)
# Create a ring plot showing the mean and std of some key features, trend and seasonality should be there for sure.
plot_radar_chart(features_df, output_dir)
def plot_histogram(freq_distribution_dict, name, output_dir):
"""
Plot the histogram for the given frequency distribution dictionary and save it to the output directory.
Parameters:
- freq_distribution_dict: Dictionary with frequency strings as keys and counts as values.
- name: Name of the frequency distribution.
- output_dir: Directory where the histogram will be saved.
"""
fig, ax = plt.subplots()
ax.bar(freq_distribution_dict.keys(), freq_distribution_dict.values())
ax.set_xlabel(f"Frequency of {name}")
ax.set_ylabel("Count")
ax.set_title(f"Distribution of {name} frequencies")
plt.savefig(f"{output_dir}/{name}_frequency_distribution.png")
plt.close()
def plot_feature_histogram(dataframe, column_name, output_directory):
"""
Plot a histogram for the given column in the DataFrame and save it to the specified output directory.
Parameters:
- dataframe: The DataFrame containing the data.
- column_name: The column name for which to plot the histogram.
- output_directory: The directory where the plot will be saved.
"""
# Check if the column exists in the dataframe
if column_name not in dataframe.columns:
raise ValueError(
f"Column '{column_name}' does not exist in the dataframe.")
# Drop rows with NA values in the specified column
clean_data = dataframe[column_name].dropna()
# Plot the histogram
plt.figure(figsize=(10, 6))
plt.hist(clean_data, bins=10, edgecolor='black')
plt.title(f'Histogram of {column_name}')
plt.xlabel(column_name)
plt.ylabel('Frequency')
# Save the plot
plot_filename = os.path.join(
output_directory, f"{column_name}_histogram.png")
plt.savefig(plot_filename)
plt.close()
|