Skip to content

Instantly share code, notes, and snippets.

@deepak-karkala
Last active December 16, 2020 13:15
Show Gist options
  • Select an option

  • Save deepak-karkala/5ca9cc9f97d8f5aecae91edea2060f01 to your computer and use it in GitHub Desktop.

Select an option

Save deepak-karkala/5ca9cc9f97d8f5aecae91edea2060f01 to your computer and use it in GitHub Desktop.
Preprocessing Pipeline
# Preprocessing pipeline for Numerical and Categorical features
from sklearn.pipeline import Pipeline
from sklearn.preprocessing import StandardScaler
from sklearn.preprocessing import OrdinalEncoder
from sklearn.compose import ColumnTransformer
############ Pipeline for numerical features #############
# Numerical features
numerical_attribs = [ 'Accommodates', 'Bedrooms', 'Beds', 'Minimum Nights',
'Availability 30', 'Availability 60', 'Availability 90',
'Availability 365', 'Number of Reviews', 'Reviews per Month',
'Review Scores Rating', 'Review Scores Accuracy', 'Review Scores Cleanliness',
'Review Scores Checkin', 'Review Scores Communication', 'Review Scores Location',
'Review Scores Value', 'Host Response Rate']
# Pipeline for numerical features
# 1. SimpleImputer: Replace NULL values with median
# 2. FunctionTransformer: Add extra features
# 3. StandardScaler: Normalise values
numerical_pipeline = Pipeline([
('imputer', SimpleImputer(strategy="median")),
('attribs_adder', FunctionTransformer(add_extra_features, validate=False)),
('std_scaler', StandardScaler()),
])
########## Pipeline for categorical features #############
# Categorical features
categorical_attribs = ["Country", "City", "Neighbourhood Cleansed",
"Property Type", "Room Type", "Bed Type", "Cancellation Policy",
"Host Response Time"]
# OrdinalEncoder(): Encode categorical features as integers
categorical_pipeline = Pipeline([
('ordinal_encoder', OrdinalEncoder()),
])
########## Combined Pipeline for all features ############
preprocessing_pipeline = ColumnTransformer([
("categorical", categorical_pipeline, categorical_attribs),
("numerical", numerical_pipeline, numerical_attribs),
])
# Labels
label = ["Price"]
# Combine numerical and categorical features
df_attribs = df[categorical_attribs + numerical_attribs + label].copy()
# Fit Preprocessing pipeline
df_prepared = preprocessing_pipeline.fit_transform(df[categorical_attribs + numerical_attribs])
# Save preprocessing pipeline
save_model(model=preprocessing_pipeline, save_path="preprocessing_pipeline.pkl")
Sign up for free to join this conversation on GitHub. Already have an account? Sign in to comment