Spaces:
Sleeping
Sleeping
| # import argparse | |
| import os | |
| import pandas as pd | |
| import time | |
| import mlflow | |
| from mlflow.models.signature import infer_signature | |
| from sklearn.model_selection import train_test_split | |
| from sklearn.pipeline import Pipeline | |
| from sklearn.preprocessing import StandardScaler, OneHotEncoder | |
| from sklearn.compose import ColumnTransformer | |
| from sklearn.linear_model import LinearRegression | |
| from sklearn.metrics import mean_squared_error | |
| # Définir l'URI de suivi MLflow | |
| mlflow.set_tracking_uri("http://0.0.0.0:5000") | |
| if __name__ == "__main__": | |
| os.environ["APP_URI"] = "https://zacbl-getaround.hf.space" | |
| # Set your variables for your environment | |
| EXPERIMENT_NAME="Car_Price_Prediction" | |
| print("training model...") | |
| # Time execution | |
| start_time = time.time() | |
| # Call mlflow autolog | |
| mlflow.sklearn.autolog() | |
| with mlflow.start_run() as run: | |
| # Import dataset | |
| df = pd.read_csv("https://full-stack-assets.s3.eu-west-3.amazonaws.com/Deployment/get_around_pricing_project.csv", sep=";") | |
| # Split the data into train and test : | |
| X = df.drop(columns = 'rental_price_per_day', axis = 1) | |
| y = df['rental_price_per_day'] | |
| X_train, X_test, y_train, y_test = train_test_split(X, y, test_size = 0.2, random_state = 42) | |
| # Define the numerical and categorical columns : | |
| num_cols = X.select_dtypes(include = ['int64', 'float64']).columns | |
| cat_cols = X.select_dtypes(include = ['object', 'bool']).columns | |
| # Train / test split | |
| X_train, X_test, y_train, y_test = train_test_split(X, y, test_size = 0.2) | |
| # Define the pipeline : | |
| num_pipeline = Pipeline([ | |
| ('scaler', StandardScaler()) | |
| ]) | |
| full_pipeline = ColumnTransformer([ | |
| ('num', num_pipeline, num_cols), | |
| ('cat', OneHotEncoder(), cat_cols) | |
| ]) | |
| # Make a pipeline containing the full pipeline and the model : | |
| model = Pipeline([ | |
| ('preprocessing', full_pipeline), | |
| ('model', LinearRegression()) | |
| ]) | |
| # Fit the model : | |
| model.fit(X_train, y_train) | |
| # Predict the price : | |
| predictions = model.predict(X_test) | |
| # Calculate the mean squared error : | |
| mean_squared_error(y_test, predictions) | |
| # Print r2 score : | |
| model.score(X_test, y_test) | |
| # Enregistrer le run_id dans un fichier | |
| run_id = run.info.run_id | |
| with open("run_id.txt", "w") as f: | |
| f.write(run_id) | |
| print("...Done!") | |
| print("Saving model...") | |
| mlflow.sklearn.log_model(model, "model", signature=infer_signature(X_train, predictions)) | |
| print("...Model saved!") | |
| print(f"---Total training time: {time.time()-start_time}") |