Last active
June 30, 2019 22:10
-
-
Save ikatsov/d13da5c568692be9c9e4747447d3119a to your computer and use it in GitHub Desktop.
This file contains hidden or bidirectional Unicode text that may be interpreted or compiled differently than what appears below. To review, open the file in an editor that reveals hidden Unicode characters.
Learn more about bidirectional Unicode characters
| def add_derived_columns(df): # step 1: add JID and standartize timestamps | |
| df_ext = df.copy() | |
| df_ext['jid'] = df_ext['uid'].map(str) + '_' + df_ext['conversion_id'].map(str) | |
| min_max_scaler = MinMaxScaler() | |
| for cname in ('timestamp', 'time_since_last_click'): | |
| x = df_ext[cname].values.reshape(-1, 1) | |
| df_ext[cname + '_norm'] = min_max_scaler.fit_transform(x) | |
| return df_ext | |
| def sample_campaigns(df, n_campaigns): # step 2.1: reduce the dataset by sampling campaigns | |
| campaigns = np.random.choice( df['campaign'].unique(), n_campaigns, replace = False ) | |
| return df[ df['campaign'].isin(campaigns) ] | |
| def filter_journeys_by_length(df, min_touchpoints): # step 2.2: remove short (trivial) journeys | |
| grouped = df.groupby(['jid'])['uid'].count().reset_index(name="count") | |
| return df[df['jid'].isin( grouped[grouped['count'] >= min_touchpoints]['jid'].values )] | |
| def balance_conversions(df): # step 3: balance the dataset: | |
| df_minority = df[df.conversion == 1] # The number of converted and non-converted events should be equal. | |
| df_majority = df[df.conversion == 0] # We take all converted journeys and iteratively add non-converted | |
| # samples until the datset is balanced. We do it this way becasue | |
| df_majority_jids = np.array_split( # we are trying to balance the number of events, but can add only | |
| df_majority['jid'].unique(), # the whole journeys. | |
| 100 * df_majority.shape[0]/df_minority.shape[0] ) | |
| df_majority_sampled = pd.DataFrame(data=None, columns=df.columns) | |
| for jid_chunk in df_majority_jids: | |
| df_majority_sampled = pd.concat([ | |
| df_majority_sampled, | |
| df_majority[df_majority.jid.isin(jid_chunk)] | |
| ]) | |
| if df_majority_sampled.shape[0] > df_minority.shape[0]: | |
| break | |
| return pd.concat([df_majority_sampled, df_minority]).sample(frac=1).reset_index(drop=True) | |
| def map_one_hot(df, column_names, result_column_name): # step 4: one-hot encoding for categorical variables | |
| mapper = {} # We use custom mapping becasue IDs in the orginal dataset | |
| for i, col_name in enumerate(column_names): # are not sequential, and standard one-not encoding | |
| for val in df[col_name].unique(): # provided by Keras does not handle this properly. | |
| mapper[val*10 + i] = len(mapper) | |
| def one_hot(values): | |
| v = np.zeros( len(mapper) ) | |
| for i, val in enumerate(values): | |
| mapped_val_id = mapper[val*10 + i] | |
| v[mapped_val_id] = 1 | |
| return v | |
| df_ext = df.copy() | |
| df_ext[result_column_name] = df_ext[column_names].values.tolist() | |
| df_ext[result_column_name] = df_ext[result_column_name].map(one_hot) | |
| return df_ext | |
| n_campaigns = 400 | |
| data_file = 'data/criteo_attribution_dataset.tsv.gz' | |
| df0 = pd.read_csv(data_file, sep='\t', compression='gzip') | |
| df1 = add_derived_columns(df0) | |
| df2 = sample_campaigns(df1, n_campaigns) | |
| df3 = filter_journeys_by_length(df2, 2) | |
| df4 = balance_conversions(df3) | |
| # all categories are mapped to one vector | |
| df5 = map_one_hot(df4, ['cat1', 'cat2', 'cat3', 'cat4', 'cat5', 'cat6', 'cat8'], 'cats') | |
| # the final dataframe used for modeling | |
| df6 = map_one_hot(df5, ['campaign'], 'campaigns').sort_values(by=['timestamp_norm']) |
Sign up for free
to join this conversation on GitHub.
Already have an account?
Sign in to comment