This file contains hidden or bidirectional Unicode text that may be interpreted or compiled differently than what appears below. To review, open the file in an editor that reveals hidden Unicode characters.
Learn more about bidirectional Unicode characters
| events = [ | |
| 0, # no action | |
| 1, # visit | |
| 2 # purchase | |
| ] | |
| offers = [ | |
| 1, # advertisement | |
| 2, # small discount | |
| 3 # large discount | |
| ] |
This file contains hidden or bidirectional Unicode text that may be interpreted or compiled differently than what appears below. To review, open the file in an editor that reveals hidden Unicode characters.
Learn more about bidirectional Unicode characters
| k = 100 # number of time steps | |
| m = 3 # number of offers (campaigns) | |
| n = 1000 # number of customers | |
| def generate_profiles(n, k, m): | |
| p_offers = [1 / m] * m # offer probabilities | |
| t_offers = np.linspace(0, k, m + 2).tolist()[1 : -1] # offer campaign times | |
| t_offer_jit = 5 # offer time jitter, std dev | |
| P = np.zeros((n, k)) # matrix of events |
This file contains hidden or bidirectional Unicode text that may be interpreted or compiled differently than what appears below. To review, open the file in an editor that reveals hidden Unicode characters.
Learn more about bidirectional Unicode characters
| def get_event_pr(d, f): | |
| f_ids = offer_seq(f) # sequence of offer IDs received by the customer | |
| f_ids = np.concatenate((offer_seq(f), np.zeros(3 - len(f_ids)))) | |
| if((f_ids[0] == 1 and f_ids[1] == 3) or | |
| (f_ids[1] == 1 and f_ids[2] == 3) or | |
| (f_ids[0] == 1 and f_ids[2] == 3)): | |
| p_events = [0.70, 0.08, 0.22] # higher probability of purchase | |
| else: | |
| p_events = [0.90, 0.08, 0.02] # default behavior |
This file contains hidden or bidirectional Unicode text that may be interpreted or compiled differently than what appears below. To review, open the file in an editor that reveals hidden Unicode characters.
Learn more about bidirectional Unicode characters
| P, F, D = generate_profiles(n, k, m) # training set | |
| Pt, Ft, Dt = generate_profiles(n, k, m) # test set | |
| visualize_profiles(P) | |
| visualize_profiles(F) |
This file contains hidden or bidirectional Unicode text that may be interpreted or compiled differently than what appears below. To review, open the file in an editor that reveals hidden Unicode characters.
Learn more about bidirectional Unicode characters
| # p, f, d - rows of matrices P, F, and D that correspond to a given customer | |
| # t_start, t_end - time interval | |
| def state_features(p, f, d, t_start, t_end): | |
| p_frame = p[0 : t_end] | |
| f_frame = f[0 : t_end] | |
| return np.array([ | |
| d[0], # demographic features | |
| count(p_frame, 1), # visits | |
| index(f_frame, 1, k), # first time offer #1 was issued | |
| index(f_frame, 2, k), # first time offer #2 was issued |
This file contains hidden or bidirectional Unicode text that may be interpreted or compiled differently than what appears below. To review, open the file in an editor that reveals hidden Unicode characters.
Learn more about bidirectional Unicode characters
| def prepare_trajectories(P, F, D): | |
| T = [] | |
| for u in range(0, n): | |
| offer_times = find_offer_times(F[u]).tolist() | |
| ranges = offer_time_ranges(offer_times) | |
| T_u = [] | |
| for r in range(0, len(ranges)): | |
| (t_start, t_end) = ranges[r] | |
| state = state_features(P[u], F[u], D[u], 0, t_start) |
This file contains hidden or bidirectional Unicode text that may be interpreted or compiled differently than what appears below. To review, open the file in an editor that reveals hidden Unicode characters.
Learn more about bidirectional Unicode characters
| T = prepare_trajectories(P, F, D) | |
| Tt = prepare_trajectories(Pt, Ft, Dt) |
This file contains hidden or bidirectional Unicode text that may be interpreted or compiled differently than what appears below. To review, open the file in an editor that reveals hidden Unicode characters.
Learn more about bidirectional Unicode characters
| def Q_0(sa): | |
| return [1] | |
| Q = Q_0 | |
| iterations = 6 | |
| for i in range(iterations): # FQI iterations | |
| X = [] | |
| Y = [] | |
| for sample in T.reshape((n * (m + 1), m + 1)): | |
| x = np.append(sample[0], sample[1]) # feature vector consists of state-action pairs |
This file contains hidden or bidirectional Unicode text that may be interpreted or compiled differently than what appears below. To review, open the file in an editor that reveals hidden Unicode characters.
Learn more about bidirectional Unicode characters
| # use the test set to evaluate the policy | |
| states = Tt[:, :, 0].flatten().tolist() | |
| values = [] | |
| best_actions = [] | |
| for s in states: | |
| a_best, v_best = best_action(Q, s, offers) | |
| values.append(v_best) | |
| best_actions.append(a_best) |
This file contains hidden or bidirectional Unicode text that may be interpreted or compiled differently than what appears below. To review, open the file in an editor that reveals hidden Unicode characters.
Learn more about bidirectional Unicode characters
| def evaluate_policy_return(T, behavioral_policy, target_policy): | |
| returns = [] | |
| for trajectory in T: | |
| importance_weight = 1 | |
| trajectory_return = 0 | |
| for transition in trajectory: | |
| state, action, reward = transition[0 : 3] | |
| action_prob_b = behavioral_policy(state, action) | |
| action_prob_t = target_policy(state, action) | |