MLModels / src /modeling.py
pcsekhar's picture
Deploy house price predictor
9e03a51 verified
Raw
History Blame Contribute Delete
3.93 kB
from __future__ import annotations
from dataclasses import dataclass
from math import sqrt
from pathlib import Path
from typing import Any
import joblib
import pandas as pd
from sklearn.compose import ColumnTransformer
from sklearn.ensemble import RandomForestRegressor
from sklearn.impute import SimpleImputer
from sklearn.metrics import mean_absolute_error, mean_squared_error, r2_score
from sklearn.model_selection import train_test_split
from sklearn.pipeline import Pipeline
from sklearn.preprocessing import OneHotEncoder
from src.data import DEFAULT_FEATURES, TARGET_COLUMN, select_model_frame
NUMERIC_FEATURES = [
"OverallQual",
"GrLivArea",
"GarageCars",
"TotalBsmtSF",
"FullBath",
"YearBuilt",
]
CATEGORICAL_FEATURES = ["Neighborhood", "HouseStyle"]
DEFAULT_MODEL_PATH = Path("models") / "house_price_model.joblib"
@dataclass(frozen=True)
class TrainedModel:
pipeline: Pipeline
feature_names: list[str]
metrics: dict[str, float]
class HousePriceModel:
def __init__(self, trained: TrainedModel):
self._trained = trained
@property
def metrics(self) -> dict[str, float]:
return self._trained.metrics
@property
def feature_names(self) -> list[str]:
return self._trained.feature_names
@classmethod
def load(cls, path: str | Path = DEFAULT_MODEL_PATH) -> "HousePriceModel":
payload = joblib.load(path)
return cls(
TrainedModel(
pipeline=payload["pipeline"],
feature_names=list(payload["feature_names"]),
metrics=dict(payload["metrics"]),
)
)
def predict(self, features: dict[str, Any]) -> float:
frame = pd.DataFrame([{name: features.get(name) for name in self.feature_names}])
prediction = self._trained.pipeline.predict(frame)[0]
return round(float(max(prediction, 0.0)), 2)
def build_pipeline() -> Pipeline:
numeric_pipeline = Pipeline(
steps=[
("imputer", SimpleImputer(strategy="median")),
]
)
categorical_pipeline = Pipeline(
steps=[
("imputer", SimpleImputer(strategy="most_frequent")),
("encoder", OneHotEncoder(handle_unknown="ignore", sparse_output=False)),
]
)
preprocessor = ColumnTransformer(
transformers=[
("num", numeric_pipeline, NUMERIC_FEATURES),
("cat", categorical_pipeline, CATEGORICAL_FEATURES),
]
)
regressor = RandomForestRegressor(n_estimators=250, random_state=42, min_samples_leaf=2)
return Pipeline(steps=[("preprocessor", preprocessor), ("regressor", regressor)])
def train_model(frame: pd.DataFrame, artifact_path: str | Path = DEFAULT_MODEL_PATH) -> TrainedModel:
model_frame = select_model_frame(frame).dropna(subset=[TARGET_COLUMN])
features = model_frame[DEFAULT_FEATURES]
target = model_frame[TARGET_COLUMN]
if len(model_frame) >= 10:
x_train, x_test, y_train, y_test = train_test_split(
features, target, test_size=0.2, random_state=42
)
else:
x_train, x_test, y_train, y_test = features, features, target, target
pipeline = build_pipeline()
pipeline.fit(x_train, y_train)
predictions = pipeline.predict(x_test)
metrics = {
"rmse": round(sqrt(mean_squared_error(y_test, predictions)), 2),
"mae": round(mean_absolute_error(y_test, predictions), 2),
"r2": round(r2_score(y_test, predictions), 4),
}
trained = TrainedModel(pipeline=pipeline, feature_names=list(DEFAULT_FEATURES), metrics=metrics)
artifact_path = Path(artifact_path)
artifact_path.parent.mkdir(parents=True, exist_ok=True)
joblib.dump(
{
"pipeline": trained.pipeline,
"feature_names": trained.feature_names,
"metrics": trained.metrics,
},
artifact_path,
)
return trained