Last active
December 16, 2020 13:15
-
-
Save deepak-karkala/5ca9cc9f97d8f5aecae91edea2060f01 to your computer and use it in GitHub Desktop.
Preprocessing Pipeline
This file contains hidden or bidirectional Unicode text that may be interpreted or compiled differently than what appears below. To review, open the file in an editor that reveals hidden Unicode characters.
Learn more about bidirectional Unicode characters
| # Preprocessing pipeline for Numerical and Categorical features | |
| from sklearn.pipeline import Pipeline | |
| from sklearn.preprocessing import StandardScaler | |
| from sklearn.preprocessing import OrdinalEncoder | |
| from sklearn.compose import ColumnTransformer | |
| ############ Pipeline for numerical features ############# | |
| # Numerical features | |
| numerical_attribs = [ 'Accommodates', 'Bedrooms', 'Beds', 'Minimum Nights', | |
| 'Availability 30', 'Availability 60', 'Availability 90', | |
| 'Availability 365', 'Number of Reviews', 'Reviews per Month', | |
| 'Review Scores Rating', 'Review Scores Accuracy', 'Review Scores Cleanliness', | |
| 'Review Scores Checkin', 'Review Scores Communication', 'Review Scores Location', | |
| 'Review Scores Value', 'Host Response Rate'] | |
| # Pipeline for numerical features | |
| # 1. SimpleImputer: Replace NULL values with median | |
| # 2. FunctionTransformer: Add extra features | |
| # 3. StandardScaler: Normalise values | |
| numerical_pipeline = Pipeline([ | |
| ('imputer', SimpleImputer(strategy="median")), | |
| ('attribs_adder', FunctionTransformer(add_extra_features, validate=False)), | |
| ('std_scaler', StandardScaler()), | |
| ]) | |
| ########## Pipeline for categorical features ############# | |
| # Categorical features | |
| categorical_attribs = ["Country", "City", "Neighbourhood Cleansed", | |
| "Property Type", "Room Type", "Bed Type", "Cancellation Policy", | |
| "Host Response Time"] | |
| # OrdinalEncoder(): Encode categorical features as integers | |
| categorical_pipeline = Pipeline([ | |
| ('ordinal_encoder', OrdinalEncoder()), | |
| ]) | |
| ########## Combined Pipeline for all features ############ | |
| preprocessing_pipeline = ColumnTransformer([ | |
| ("categorical", categorical_pipeline, categorical_attribs), | |
| ("numerical", numerical_pipeline, numerical_attribs), | |
| ]) | |
| # Labels | |
| label = ["Price"] | |
| # Combine numerical and categorical features | |
| df_attribs = df[categorical_attribs + numerical_attribs + label].copy() | |
| # Fit Preprocessing pipeline | |
| df_prepared = preprocessing_pipeline.fit_transform(df[categorical_attribs + numerical_attribs]) | |
| # Save preprocessing pipeline | |
| save_model(model=preprocessing_pipeline, save_path="preprocessing_pipeline.pkl") |
Sign up for free
to join this conversation on GitHub.
Already have an account?
Sign in to comment