Created
April 4, 2017 15:32
-
-
Save Awuor87/f0b496f04dd200e05fa6c646009ac9ab to your computer and use it in GitHub Desktop.
Building a Recommendation Engine in Python
This file contains hidden or bidirectional Unicode text that may be interpreted or compiled differently than what appears below. To review, open the file in an editor that reveals hidden Unicode characters.
Learn more about bidirectional Unicode characters
| import networkx | |
| from operator import itemgetter | |
| import matplotlib.pyplot | |
| # read the data from the amazon-books.txt; | |
| # populate amazonProducts nested dicitonary; | |
| # key = ASIN; value = MetaData associated with ASIN | |
| fhr = open('./amazon-books.txt', 'r', encoding='utf-8', errors='ignore') | |
| amazonBooks = {} | |
| fhr.readline() | |
| for line in fhr: | |
| cell = line.split('\t') | |
| MetaData = {} | |
| MetaData['Id'] = cell[0].strip() | |
| ASIN = cell[1].strip() | |
| MetaData['Title'] = cell[2].strip() | |
| MetaData['Categories'] = cell[3].strip() | |
| MetaData['Group'] = cell[4].strip() | |
| MetaData['Copurchased'] = cell[5].strip() | |
| MetaData['SalesRank'] = int(cell[6].strip()) | |
| MetaData['TotalReviews'] = int(cell[7].strip()) | |
| MetaData['AvgRating'] = float(cell[8].strip()) | |
| MetaData['DegreeCentrality'] = int(cell[9].strip()) | |
| MetaData['ClusteringCoeff'] = float(cell[10].strip()) | |
| amazonBooks[ASIN] = MetaData | |
| fhr.close() | |
| # read the data from amazon-books-copurchase.adjlist; | |
| # assign it to copurchaseGraph weighted Graph; | |
| # node = ASIN, edge= copurchase, edge weight = category similarity | |
| fhr=open("amazon-books-copurchase.edgelist", 'rb') | |
| copurchaseGraph=networkx.read_weighted_edgelist(fhr) | |
| fhr.close() | |
| # now let's assume a person is considering buying the following book; | |
| # what else can we recommend to them based on copurchase behavior | |
| # we've seen from other users? | |
| print ("Looking for Recommendations for Customer Purchasing this Book:") | |
| print ("--------------------------------------------------------------") | |
| asin = '0805047905' | |
| # Understand the metadata associated with this book by printing out the | |
| # features associates with the book | |
| print ("ASIN = ", asin) | |
| print ("Title = ", amazonBooks[asin]['Title']) | |
| print ("SalesRank = ", amazonBooks[asin]['SalesRank']) | |
| print ("TotalReviews = ", amazonBooks[asin]['TotalReviews']) | |
| print ("AvgRating = ", amazonBooks[asin]['AvgRating']) | |
| print ("DegreeCentrality = ", amazonBooks[asin]['DegreeCentrality']) | |
| print ("ClusteringCoeff = ", amazonBooks[asin]['ClusteringCoeff']) | |
| print() | |
| # Create variable dcl from the copurchaseGraph data using the networkx.degree package | |
| # Create new variable dc is equal to dcl of given asin | |
| # print dc | |
| dcl = networkx.degree(copurchaseGraph) | |
| dc = dcl[asin] | |
| print ("Degree Centrality:", dc) | |
| print() | |
| # Get ego network of given asin at depth 1 using networkx.ego_graph package | |
| # and assign to variable ego | |
| # print number of nodes in ego | |
| # print number of edges in ego | |
| ego = networkx.ego_graph(copurchaseGraph, asin, radius=1) | |
| print ("Ego Network:", | |
| "Nodes =", ego.number_of_nodes(), | |
| "Edges =", ego.number_of_edges()) | |
| print() | |
| # Get clustering coefficient of given asin | |
| # Get clustering coefficient of ego using networkx.average_clustering and | |
| # assign to variable cc | |
| # print clustering coefficient, round to two decimal places | |
| cc = networkx.average_clustering(ego) | |
| print ("Clustering Coefficient:", round(cc,2)) | |
| print() | |
| # Use island method on ego network with a threshold of 0.65 to trim down the | |
| # ego network | |
| # Set threshold to 0.68 | |
| # Create empty tuple called egotrim using the networkx.Graph() to represent | |
| # the trimmed network | |
| # loop node 1, node 2, edge in the ego network edges data: | |
| # if edge weight is greater than or equal to the threshold: | |
| # add node 1, node 2, edge weight to the egotrim tuple | |
| #print threshold of the trimmed network | |
| #print number of nodes in the egotrim tuple | |
| #print number of edges in the egotrim tuple | |
| # print list of egotrim network to obtain the asin of the books in the | |
| # trimmed network | |
| threshold = 0.68 | |
| egotrim = networkx.Graph() | |
| for n1, n2, e in ego.edges(data=True): | |
| if e['weight'] >= threshold: | |
| egotrim.add_edge(n1,n2,e) | |
| print ("Trimmed Ego Network:", | |
| "Threshold=", threshold, | |
| "Nodes =", egotrim.number_of_nodes(), | |
| "Edges =", egotrim.number_of_edges()) | |
| print() | |
| print("Asin in the trimmed network: ", list(egotrim)) | |
| # Print out the neighbors of the asin in the trimmed network | |
| # Create variable called neighbors which contains the neighbors of the | |
| # given asin in the trimmed network | |
| # print list of neighbors | |
| print() | |
| neighbors = egotrim.neighbors(asin) | |
| print() | |
| print("Neighbours in the trimmed network: ", list(neighbors)) | |
| # Write a for loop statement to reiterate in the neighbors tuple and | |
| # print out the recommendations for customers | |
| # purchasing this book | |
| # Loop with neighbor asin as nb_asin in neighbors tuple: | |
| # Print nb_asin | |
| # Print Title from the amazonBooks dataset | |
| # Print AvgRating from the amazonBooks dataset | |
| # Print TotalReviews from the amazonBooks dataset | |
| print() | |
| for nb_asin in neighbors: | |
| print("Asin: ", nb_asin) | |
| print("Book Title: ", amazonBooks[nb_asin]["Title"]) | |
| print("Average Rating:", amazonBooks[nb_asin]["AvgRating"]) | |
| print("Number of Reviews: ", amazonBooks[nb_asin]["TotalReviews"]) |
Sign up for free
to join this conversation on GitHub.
Already have an account?
Sign in to comment