Skip to content

Instantly share code, notes, and snippets.

@ikatsov
Last active June 30, 2019 22:10
Show Gist options
  • Select an option

  • Save ikatsov/d13da5c568692be9c9e4747447d3119a to your computer and use it in GitHub Desktop.

Select an option

Save ikatsov/d13da5c568692be9c9e4747447d3119a to your computer and use it in GitHub Desktop.
def add_derived_columns(df): # step 1: add JID and standartize timestamps
df_ext = df.copy()
df_ext['jid'] = df_ext['uid'].map(str) + '_' + df_ext['conversion_id'].map(str)
min_max_scaler = MinMaxScaler()
for cname in ('timestamp', 'time_since_last_click'):
x = df_ext[cname].values.reshape(-1, 1)
df_ext[cname + '_norm'] = min_max_scaler.fit_transform(x)
return df_ext
def sample_campaigns(df, n_campaigns): # step 2.1: reduce the dataset by sampling campaigns
campaigns = np.random.choice( df['campaign'].unique(), n_campaigns, replace = False )
return df[ df['campaign'].isin(campaigns) ]
def filter_journeys_by_length(df, min_touchpoints): # step 2.2: remove short (trivial) journeys
grouped = df.groupby(['jid'])['uid'].count().reset_index(name="count")
return df[df['jid'].isin( grouped[grouped['count'] >= min_touchpoints]['jid'].values )]
def balance_conversions(df): # step 3: balance the dataset:
df_minority = df[df.conversion == 1] # The number of converted and non-converted events should be equal.
df_majority = df[df.conversion == 0] # We take all converted journeys and iteratively add non-converted
# samples until the datset is balanced. We do it this way becasue
df_majority_jids = np.array_split( # we are trying to balance the number of events, but can add only
df_majority['jid'].unique(), # the whole journeys.
100 * df_majority.shape[0]/df_minority.shape[0] )
df_majority_sampled = pd.DataFrame(data=None, columns=df.columns)
for jid_chunk in df_majority_jids:
df_majority_sampled = pd.concat([
df_majority_sampled,
df_majority[df_majority.jid.isin(jid_chunk)]
])
if df_majority_sampled.shape[0] > df_minority.shape[0]:
break
return pd.concat([df_majority_sampled, df_minority]).sample(frac=1).reset_index(drop=True)
def map_one_hot(df, column_names, result_column_name): # step 4: one-hot encoding for categorical variables
mapper = {} # We use custom mapping becasue IDs in the orginal dataset
for i, col_name in enumerate(column_names): # are not sequential, and standard one-not encoding
for val in df[col_name].unique(): # provided by Keras does not handle this properly.
mapper[val*10 + i] = len(mapper)
def one_hot(values):
v = np.zeros( len(mapper) )
for i, val in enumerate(values):
mapped_val_id = mapper[val*10 + i]
v[mapped_val_id] = 1
return v
df_ext = df.copy()
df_ext[result_column_name] = df_ext[column_names].values.tolist()
df_ext[result_column_name] = df_ext[result_column_name].map(one_hot)
return df_ext
n_campaigns = 400
data_file = 'data/criteo_attribution_dataset.tsv.gz'
df0 = pd.read_csv(data_file, sep='\t', compression='gzip')
df1 = add_derived_columns(df0)
df2 = sample_campaigns(df1, n_campaigns)
df3 = filter_journeys_by_length(df2, 2)
df4 = balance_conversions(df3)
# all categories are mapped to one vector
df5 = map_one_hot(df4, ['cat1', 'cat2', 'cat3', 'cat4', 'cat5', 'cat6', 'cat8'], 'cats')
# the final dataframe used for modeling
df6 = map_one_hot(df5, ['campaign'], 'campaigns').sort_values(by=['timestamp_norm'])
Sign up for free to join this conversation on GitHub. Already have an account? Sign in to comment