self-supervised-prostate-cancer / codes /classification /mlp_classifier_weighted.py
buseyaren
Translate remaining pipeline scripts/comments to English and update README
dc14622
Raw
History Blame Contribute Delete
18.1 kB
import numpy as np
import pandas as pd
from sklearn.neural_network import MLPClassifier
from sklearn.model_selection import train_test_split
from sklearn.metrics import classification_report, confusion_matrix
import matplotlib.pyplot as plt
import seaborn as sns
import joblib
from sklearn.preprocessing import StandardScaler, LabelEncoder
from collections import Counter
from imblearn.over_sampling import SMOTE
def load_features_from_extraction(output_dir):
"""
Loads feature files produced by `extract_features_hierarchical.py`.
No need to read an extra CSV - labels are already stored in `fused_features.csv`.
"""
from pathlib import Path
output_dir = Path(output_dir)
features_path = output_dir / 'fused_features.npy'
if not features_path.exists():
raise FileNotFoundError(f"Feature file not found: {features_path}")
print(f"Loading features from {features_path}")
features = np.load(features_path)
print(f"Loaded {len(features)} features with shape {features.shape}")
csv_path = output_dir / 'fused_features.csv'
if not csv_path.exists():
raise FileNotFoundError(f"CSV file not found: {csv_path}")
print(f"Loading labels from {csv_path}")
df = pd.read_csv(csv_path)
if 'label' in df.columns:
labels = df['label'].values
valid_mask = labels != -1
elif 'gleason_grade' in df.columns:
from sklearn.preprocessing import LabelEncoder
label_encoder = LabelEncoder()
grades = df['gleason_grade'].values
valid_mask = (grades != 'unknown') & (pd.notna(grades))
valid_grades = grades[valid_mask]
labels = np.full(len(grades), -1)
if len(valid_grades) > 0:
labels[valid_mask] = label_encoder.fit_transform(valid_grades)
else:
raise ValueError("Could not find the 'label' or 'gleason_grade' column in the CSV!")
grades = df['gleason_grade'].values if 'gleason_grade' in df.columns else None
patient_ids = df['patient_id'].values if 'patient_id' in df.columns else None
features = features[valid_mask]
# Use grades as labels (string format)
if grades is not None:
labels = grades[valid_mask]
else:
labels = labels[valid_mask]
grades = grades[valid_mask] if grades is not None else None
patient_ids = patient_ids[valid_mask] if patient_ids is not None else None
print(f"Total matched samples: {len(features)}")
if grades is not None:
print(f"Unique Gleason grades: {np.unique(grades)}")
return features, labels, grades, patient_ids
def create_weighted_mlp_classifier(y_train_encoded, label_encoder):
"""
Create a weighted MLP classifier based on class performance.
"""
# Compute class distribution
class_counts = Counter(y_train_encoded)
total_samples = len(y_train_encoded)
# Compute class weights (inverse proportional)
class_weights = {}
for class_id in range(len(label_encoder.classes_)):
if class_id in class_counts:
# Give higher weight to classes with fewer examples
weight = total_samples / (len(class_counts) * class_counts[class_id])
class_weights[class_id] = weight
else:
class_weights[class_id] = 1.0
print("Class weights:")
for class_id, weight in class_weights.items():
class_name = label_encoder.inverse_transform([class_id])[0]
print(f" {class_name}: {weight:.3f}")
# Create the weighted MLP classifier
weighted_mlp = MLPClassifier(
# Larger, deeper network architecture
hidden_layer_sizes=(2048, 1024, 512, 256),
# Activation and solver settings
activation='relu',
solver='adam',
# Regularizasyon
alpha=0.00001,
# Learning parameters
batch_size=128,
learning_rate='adaptive',
learning_rate_init=0.001,
max_iter=500,
# Early stopping settings
early_stopping=True,
validation_fraction=0.1,
n_iter_no_change=20,
# Other settings
random_state=42,
verbose=True,
# Momentum
momentum=0.9,
nesterovs_momentum=True,
# Learning rate schedule
power_t=0.5
)
return weighted_mlp, class_weights
def create_sample_weights(y_train_encoded, class_weights):
"""
Creates sample weights for each example.
"""
sample_weights = np.ones(len(y_train_encoded))
for i, class_id in enumerate(y_train_encoded):
sample_weights[i] = class_weights.get(class_id, 1.0)
print(f"Sample weights created - min: {sample_weights.min():.3f}, max: {sample_weights.max():.3f}")
return sample_weights
def create_weighted_dataset(X_train, y_train_encoded, class_weights, label_encoder):
"""
Creates a weighted dataset by duplicating samples according to weights.
"""
print("Creating weighted dataset by duplicating samples...")
# Compute how many times each class should be duplicated
duplication_factors = {}
for class_id, weight in class_weights.items():
# Round the weight to determine duplication count
duplication_factors[class_id] = max(1, int(round(weight)))
print("Duplication factors:")
for class_id, factor in duplication_factors.items():
class_name = label_encoder.inverse_transform([class_id])[0]
print(f" {class_name}: {factor}x")
# Duplicate the data
X_weighted = []
y_weighted = []
for i, class_id in enumerate(y_train_encoded):
# Add the original sample
X_weighted.append(X_train[i])
y_weighted.append(class_id)
# Add extra copies according to weight
extra_copies = duplication_factors.get(class_id, 1) - 1
for _ in range(extra_copies):
X_weighted.append(X_train[i])
y_weighted.append(class_id)
X_weighted = np.array(X_weighted)
y_weighted = np.array(y_weighted)
print(f"Weighted dataset created: {len(X_weighted)} samples (original: {len(X_train)})")
# Show the new class distribution
new_counts = Counter(y_weighted)
print("\nNew class distribution after weighting:")
for class_id, count in new_counts.items():
class_name = label_encoder.inverse_transform([class_id])[0]
print(f" {class_name}: {count}")
return X_weighted, y_weighted
def apply_weighted_smote(X_train, y_train_encoded, label_encoder, sampling_strategy='auto'):
"""
Applies weighted SMOTE - it does not fully balance classes; it only slightly increases underrepresented classes.
"""
print("Applying weighted SMOTE...")
# Save the original class distribution
original_counts = Counter(y_train_encoded)
print("Original class distribution:")
for class_id, count in original_counts.items():
class_name = label_encoder.inverse_transform([class_id])[0]
print(f" {class_name}: {count}")
# Create up to 80% of the largest class (not full balancing)
max_samples = max(original_counts.values())
target_samples = int(max_samples * 0.8)
# Sampling strategy'yi ayarla
if sampling_strategy == 'auto':
# Determine target number of samples for each class
sampling_strategy = {}
for class_id in original_counts:
current_count = original_counts[class_id]
if current_count < target_samples:
sampling_strategy[class_id] = target_samples
else:
sampling_strategy[class_id] = current_count
print(f"Target samples per class: {target_samples}")
print("Sampling strategy:", sampling_strategy)
# SMOTE uygula
smote = SMOTE(sampling_strategy=sampling_strategy, random_state=42)
X_resampled, y_resampled = smote.fit_resample(X_train, y_train_encoded)
# Show the new class distribution
new_counts = Counter(y_resampled)
print("\nNew class distribution after weighted SMOTE:")
for class_id, count in new_counts.items():
class_name = label_encoder.inverse_transform([class_id])[0]
print(f" {class_name}: {count}")
return X_resampled, y_resampled
def main():
# Load train features (CSV okumaya gerek yok)
print("=== LOADING TRAIN FEATURES ===")
X_train, y_train, train_grades, train_cases = load_features_from_extraction(
'feature_extraction/extractedfusedfeatures_train' # Train feature extraction output folder
)
# Load test features
print("\n=== LOADING TEST FEATURES ===")
X_test, y_test, test_grades, test_cases = load_features_from_extraction(
'feature_extraction/extractedfusedfeatures_test' # Test feature extraction output folder
)
# Remove 2+4 class if present
print("\n=== FILTERING DATA ===")
if train_grades is not None:
train_mask = train_grades != '2+4'
X_train = X_train[train_mask]
y_train = y_train[train_mask]
train_grades = train_grades[train_mask]
train_cases = train_cases[train_mask] if train_cases is not None else None
if test_grades is not None:
test_mask = test_grades != '2+4'
X_test = X_test[test_mask]
y_test = y_test[test_mask]
test_grades = test_grades[test_mask]
test_cases = test_cases[test_mask] if test_cases is not None else None
print(f"Final training set: {X_train.shape[0]} samples")
print(f"Final test set: {X_test.shape[0]} samples")
print(f"Training classes: {np.unique(y_train)}")
print(f"Test classes: {np.unique(y_test)}")
# Encode string labels to numbers
print("\n=== ENCODING LABELS ===")
label_encoder = LabelEncoder()
y_train_encoded = label_encoder.fit_transform(y_train)
y_test_encoded = label_encoder.transform(y_test)
print(f"Unique classes after encoding: {np.unique(y_train_encoded)}")
print("Label mapping:", dict(zip(label_encoder.classes_, range(len(label_encoder.classes_)))))
# Print class distribution before SMOTE
print("\nClass distribution before SMOTE:")
print(Counter(y_train_encoded))
# Visualize class distribution
plt.figure(figsize=(12, 6))
class_counts = Counter(y_train_encoded)
class_names = [label_encoder.inverse_transform([i])[0] for i in range(len(label_encoder.classes_))]
class_values = [class_counts.get(i, 0) for i in range(len(label_encoder.classes_))]
plt.bar(class_names, class_values, color='skyblue', edgecolor='black')
plt.title('Class Distribution - Training Data')
plt.xlabel('Gleason Grade')
plt.ylabel('Number of Samples')
plt.xticks(rotation=45)
plt.tight_layout()
plt.savefig('class_distribution_training.png')
print("Saved class distribution plot: class_distribution_training.png")
plt.show()
# Apply weighted SMOTE (not full balancing)
print("\n=== APPLYING WEIGHTED SMOTE ===")
X_train_resampled, y_train_resampled = apply_weighted_smote(X_train, y_train_encoded, label_encoder)
# Create weighted MLP model
print("\n=== CREATING WEIGHTED MLP MODEL ===")
weighted_mlp, class_weights = create_weighted_mlp_classifier(y_train_resampled, label_encoder)
# Create weighted dataset by duplicating samples
print("\n=== CREATING WEIGHTED DATASET ===")
X_train_weighted, y_train_weighted = create_weighted_dataset(X_train_resampled, y_train_resampled, class_weights, label_encoder)
# Scale the weighted features
print("\n=== SCALING WEIGHTED FEATURES ===")
scaler = StandardScaler()
X_train_scaled = scaler.fit_transform(X_train_weighted)
X_test_scaled = scaler.transform(X_test)
print(f"\nFinal training set shape after weighting: {X_train_scaled.shape}")
# Model training
print("\n=== TRAINING WEIGHTED MLP MODEL ===")
print("Starting model training...")
weighted_mlp.fit(X_train_scaled, y_train_weighted)
# Evaluate on test set
print("\n=== EVALUATING WEIGHTED MODEL ON TEST SET ===")
y_pred = weighted_mlp.predict(X_test_scaled)
# Convert numeric predictions back to original labels for the report
y_test_original = label_encoder.inverse_transform(y_test_encoded)
y_pred_original = label_encoder.inverse_transform(y_pred)
print("\nClassification Report:")
print(classification_report(y_test_original, y_pred_original))
# Create confusion matrix
cm = confusion_matrix(y_test_original, y_pred_original)
plt.figure(figsize=(12, 10))
sns.heatmap(cm, annot=True, fmt='d', cmap='Blues',
xticklabels=label_encoder.classes_,
yticklabels=label_encoder.classes_)
plt.xlabel('Predicted')
plt.ylabel('Actual')
plt.title('Weighted MLP Confusion Matrix - 90 Epoch Features')
plt.xticks(rotation=45)
plt.yticks(rotation=45)
plt.tight_layout()
plt.savefig('weighted_mlp_confusion_matrix_90ep.png')
print("Saved confusion matrix to weighted_mlp_confusion_matrix_90ep.png")
# Save the trained model, scaler and label encoder
joblib.dump(weighted_mlp, 'tcga_gleason_weighted_mlp_model_90ep.joblib')
joblib.dump(scaler, 'tcga_gleason_weighted_mlp_scaler_90ep.joblib')
joblib.dump(label_encoder, 'tcga_gleason_weighted_label_encoder_90ep.joblib')
joblib.dump(class_weights, 'tcga_gleason_class_weights_90ep.joblib')
print("Weighted model, scaler, label encoder and class weights saved")
# Plot learning curve
plt.figure(figsize=(12, 8))
plt.plot(weighted_mlp.loss_curve_)
plt.title('Weighted MLP Learning Curve - 90 Epoch Features')
plt.xlabel('Iterations')
plt.ylabel('Loss')
plt.grid(True)
plt.savefig('weighted_mlp_learning_curve_90ep.png')
print("Saved learning curve to weighted_mlp_learning_curve_90ep.png")
# Validation score curve
if hasattr(weighted_mlp, 'validation_scores_'):
plt.figure(figsize=(12, 8))
plt.plot(weighted_mlp.validation_scores_)
plt.title('Weighted MLP Validation Score Curve - 90 Epoch Features')
plt.xlabel('Iterations')
plt.ylabel('Score')
plt.grid(True)
plt.savefig('weighted_mlp_validation_curve_90ep.png')
print("Saved validation curve to weighted_mlp_validation_curve_90ep.png")
print("\nDone! You can now use the trained weighted model for predictions.")
# Function to predict on new samples
def predict_gleason_grade_weighted(embedding_vector,
model_path='tcga_gleason_weighted_mlp_model_90ep.joblib',
scaler_path='tcga_gleason_weighted_mlp_scaler_90ep.joblib',
encoder_path='tcga_gleason_weighted_label_encoder_90ep.joblib',
weights_path='tcga_gleason_class_weights_90ep.joblib'):
"""Predict Gleason grade using weighted model with class weights information"""
model = joblib.load(model_path)
scaler = joblib.load(scaler_path)
label_encoder = joblib.load(encoder_path)
class_weights = joblib.load(weights_path)
# Reshape and scale the input
embedding_vector = np.array(embedding_vector).reshape(1, -1)
embedding_vector_scaled = scaler.transform(embedding_vector)
# Get prediction and probabilities
prediction = model.predict(embedding_vector_scaled)
probabilities = model.predict_proba(embedding_vector_scaled)
# Convert numeric prediction back to original label
prediction_original = label_encoder.inverse_transform(prediction)
# Get the weight of the predicted class
predicted_class_id = prediction[0]
predicted_class_weight = class_weights.get(predicted_class_id, 1.0)
return {
'predicted_grade': prediction_original[0],
'probabilities': dict(zip(label_encoder.classes_, probabilities[0])),
'class_weight': predicted_class_weight,
'confidence_score': np.max(probabilities[0])
}
def compare_models_performance():
"""Compare regular and weighted model performance"""
print("=== MODEL PERFORMANCE COMPARISON ===")
# Load both models
try:
regular_model = joblib.load('tcga_gleason_mlp_model_90ep.joblib')
weighted_model = joblib.load('tcga_gleason_weighted_mlp_model_90ep.joblib')
print("Both models loaded successfully!")
print(f"Regular model type: {type(regular_model)}")
print(f"Weighted model type: {type(weighted_model)}")
# Compare model architectures
print(f"\nRegular model hidden layers: {regular_model.hidden_layer_sizes}")
print(f"Weighted model hidden layers: {weighted_model.hidden_layer_sizes}")
# Compare final loss
if hasattr(regular_model, 'loss_curve_') and hasattr(weighted_model, 'loss_curve_'):
print(f"\nRegular model final loss: {regular_model.loss_curve_[-1]:.6f}")
print(f"Weighted model final loss: {weighted_model.loss_curve_[-1]:.6f}")
# Compare validation scores
if hasattr(regular_model, 'validation_scores_') and hasattr(weighted_model, 'validation_scores_'):
print(f"Regular model final validation score: {regular_model.validation_scores_[-1]:.6f}")
print(f"Weighted model final validation score: {weighted_model.validation_scores_[-1]:.6f}")
except FileNotFoundError as e:
print(f"Model file not found: {e}")
print("Please run the training first to generate both models.")
if __name__ == "__main__":
main()