Spaces:
Sleeping
Sleeping
| from joblib import Parallel, delayed, dump, load | |
| import numpy as np | |
| from scipy.stats import mode | |
| from sklearn.datasets import load_iris | |
| from sklearn.model_selection import KFold | |
| from sklearn.metrics import accuracy_score | |
| from src.decisiontreeclassifier import DecisionTreeClassifier | |
| from sklearn.preprocessing import LabelEncoder | |
| import pandas as pd | |
| from sklearn.model_selection import ParameterGrid | |
| from sklearn.ensemble import ExtraTreesClassifier | |
| from shapely.geometry import Point, Polygon | |
| import multiprocessing | |
| class RandomForest: | |
| def __init__(self, n_estimators=400, max_features=1.0, | |
| min_samples_leaf=2, random_state=42): | |
| self.n_estimators = n_estimators | |
| self.max_features = max_features | |
| self.min_samples_leaf = min_samples_leaf | |
| self.random_state = random_state | |
| self.model = None | |
| self.minimum_collision_count = 2 | |
| self.decision_threshold = 50.0 | |
| def _engineer_features(X): | |
| X = np.asarray(X, dtype=float) | |
| longitude = X[:, 0] | |
| latitude = X[:, 1] | |
| lon_rad = np.radians(longitude) | |
| lat_rad = np.radians(latitude) | |
| return np.column_stack([ | |
| longitude, | |
| latitude, | |
| np.sin(lon_rad), | |
| np.cos(lon_rad), | |
| np.sin(lat_rad), | |
| np.cos(lat_rad), | |
| longitude * latitude, | |
| longitude ** 2, | |
| latitude ** 2, | |
| ]) | |
| def fit(self, X, y): | |
| # A single proximity record is nearly uniform across the country and is | |
| # not learnable from location. Model elevated/repeated activity instead. | |
| binary_y = (np.asarray(y, dtype=float) >= self.minimum_collision_count).astype(int) | |
| engineered_X = self._engineer_features(X) | |
| self.model = ExtraTreesClassifier( | |
| n_estimators=self.n_estimators, | |
| max_features=self.max_features, | |
| min_samples_leaf=self.min_samples_leaf, | |
| class_weight="balanced", | |
| n_jobs=-1, | |
| random_state=self.random_state, | |
| ) | |
| self.model.fit(engineered_X, binary_y) | |
| dump(self, 'random_forest_model.joblib') | |
| print("Model saved to random_forest_model.joblib") | |
| return self | |
| def predict(self, X): | |
| if isinstance(X, pd.DataFrame): | |
| X = X.to_numpy() | |
| if self.model is None: | |
| raise RuntimeError("The collision-risk model has not been trained.") | |
| return self.model.predict_proba(self._engineer_features(X))[:, 1] * 100.0 | |
| def predict_with_location(self, X_df): | |
| points_df = X_df[X_df['Type'] == 'Point'] | |
| # Assuming 'polygons_df' has been correctly filtered to include only polygon rows | |
| polygons_df = X_df[(X_df['Type'] == 'Polygon') | (X_df['group'].isin(X_df[X_df['Type'] == 'Polygon']['group'])) & (X_df['Type'] != 'Point')] | |
| for index, row in points_df.iterrows(): | |
| latitude, longitude = row['Latitude'], row['Longitude'] | |
| point = np.array([[longitude, latitude]]) | |
| X_df.at[index, 'Prediction'] = self.predict(point)[0] | |
| grouped_polygons = polygons_df.groupby('group') | |
| for group_number, group_df in grouped_polygons: | |
| polygon_coords = [(x, y) for x, y in zip(group_df['Longitude'], group_df['Latitude'])] | |
| polygon = Polygon(polygon_coords) | |
| minx, miny, maxx, maxy = polygon.bounds | |
| sumOfPredictions = 0 | |
| count = 0 | |
| for lat in np.arange(miny, maxy, 1/69): | |
| for lon in np.arange(minx, maxx, 1/(np.cos(np.radians(lat)) * 69)): | |
| if polygon.contains(Point(lon, lat)): | |
| point = np.array([[lon, lat]]) | |
| sumOfPredictions += self.predict(point)[0] | |
| count += 1 | |
| # Update the prediction for this group in X_df | |
| if count > 0: | |
| average_prediction = sumOfPredictions / count | |
| # Find rows belonging to this group and update | |
| X_df.loc[X_df['group'] == group_number, 'Prediction'] = average_prediction | |
| return X_df | |
| def load_model(self, filename): | |
| try: | |
| return load(filename) | |
| except Exception as e: | |
| print("Model not found. Exception:", str(e)) | |
| return None | |
| def train_tree(self, indices, X, y, max_features, n_features, random_state): | |
| np.random.seed(random_state) # Ensure reproducibility for each tree | |
| X_subset, y_subset = X[indices], y[indices] | |
| features_indices = np.random.choice(n_features, size=max_features, replace=False) | |
| tree = DecisionTreeClassifier(min_samples_split=25, max_depth=25, feature_selection_strategy='sqrt') | |
| #tree = DecisionTreeClassifier(min_samples_split=1, max_depth=1, feature_selection_strategy='sqrt') | |
| tree.fit(X_subset[:, features_indices], y_subset) | |
| return tree, features_indices # Return a tuple of the tree and its feature indices | |
| def evaluate_model_with_kfold(self, X, y, n_splits=5): | |
| kf = KFold(n_splits=n_splits, shuffle=True, random_state=42) | |
| accuracies = [] | |
| for train_index, test_index in kf.split(X): | |
| X_train, X_test = X[train_index], X[test_index] | |
| y_train, y_test = y[train_index], y[test_index] | |
| model = RandomForest(n_estimators=10, max_features='sqrt', random_state=42) | |
| model.fit(X_train, y_train) | |
| predictions = (model.predict(X_test) >= 50).astype(int) | |
| accuracy = accuracy_score((y_test >= self.minimum_collision_count).astype(int), predictions) | |
| accuracies.append(accuracy) | |
| return np.mean(accuracies), np.std(accuracies) | |
| def model_predict(self, X_set: np.array) -> np.array: | |
| """Returns the predicted labels for a given data set""" | |
| pred_probs = self.predict_proba(X_set) | |
| preds = np.argmax(pred_probs, axis=1) | |
| return preds | |
| #To make a prediction with a list of trained base learners, we will average the predicted probabilities for each class of every base learner. | |
| #The average will be the predicted probability of the random forest model. | |
| def _predict_proba_w_base_learners(self, X_set: np.array) -> list: | |
| """ | |
| Creates list of predictions for all base learners | |
| """ | |
| pred_prob_list = [] | |
| for base_learner in self.base_learner_list: | |
| pred_prob_list.append(base_learner.predict_proba(X_set)) | |
| return pred_prob_list | |
| def predict_proba(self, X_set: np.array) -> list: | |
| """Returns the predicted probs for a given data set""" | |
| pred_probs = [] | |
| base_learners_pred_probs = RandomForest._predict_proba_w_base_learners(X_set) | |
| # Average the predicted probabilities of base learners | |
| for obs in range(X_set.shape[0]): | |
| base_learner_probs_for_obs = [a[obs] for a in base_learners_pred_probs] | |
| # Calculate the average for each index | |
| obs_average_pred_probs = np.mean(base_learner_probs_for_obs, axis=0) | |
| pred_probs.append(obs_average_pred_probs) | |
| return pred_probs | |
| def grid_search_cv(self, X, y, param_grid, n_splits=5): | |
| best_score = 0 | |
| best_params = None | |
| for params in ParameterGrid(param_grid): | |
| print("Evaluating parameters:", params) | |
| model = DecisionTreeClassifier( | |
| max_depth=params['max_depth'], | |
| min_samples_split=params['min_samples_split'], | |
| feature_selection_strategy=params['feature_selection_strategy'] | |
| ) | |
| # Use your evaluate_model_with_kfold or similar function | |
| mean_accuracy, _ = RandomForest.evaluate_model_with_kfold_test(X, y, model, n_splits=n_splits) | |
| print(f"Mean Accuracy: {mean_accuracy}") | |
| if mean_accuracy > best_score: | |
| best_score = mean_accuracy | |
| best_params = params | |
| return best_score, best_params | |
| def evaluate_model_with_kfold_test(self, X, y, model, n_splits=2): | |
| kf = KFold(n_splits=n_splits, shuffle=True, random_state=42) | |
| accuracies = [] | |
| for train_index, test_index in kf.split(X): | |
| X_train, X_test = X[train_index], X[test_index] | |
| y_train, y_test = y[train_index], y[test_index] | |
| model.fit(X_train, y_train) | |
| predictions = model.predict(X_test) | |
| accuracy = accuracy_score(y_test, predictions) | |
| accuracies.append(accuracy) | |
| return np.mean(accuracies), np.std(accuracies) | |