{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# Importation\nimport numpy as np\nimport pandas as pd\nimport seaborn as sns\nfrom matplotlib import pyplot as plt\nfrom tqdm.notebook import tqdm","metadata":{"execution":{"iopub.status.busy":"2022-07-25T15:02:01.603257Z","iopub.execute_input":"2022-07-25T15:02:01.603596Z","iopub.status.idle":"2022-07-25T15:02:02.569282Z","shell.execute_reply.started":"2022-07-25T15:02:01.603495Z","shell.execute_reply":"2022-07-25T15:02:02.568520Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Data Importation**","metadata":{}},{"cell_type":"code","source":"articles = pd.read_csv(\"../input/h-and-m-personalized-fashion-recommendations/articles.csv\")\ncustomers = pd.read_csv(\"../input/h-and-m-personalized-fashion-recommendations/customers.csv\")\ntransactions = pd.read_csv(\"../input/h-and-m-personalized-fashion-recommendations/transactions_train.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-07-25T15:02:02.570891Z","iopub.execute_input":"2022-07-25T15:02:02.571218Z","iopub.status.idle":"2022-07-25T15:03:08.126269Z","shell.execute_reply.started":"2022-07-25T15:02:02.571180Z","shell.execute_reply":"2022-07-25T15:03:08.125520Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We keep only the last week of the transactions :","metadata":{}},{"cell_type":"code","source":"transactions_1w = transactions[transactions['t_dat'] >= '2020-09-15'].copy()","metadata":{"execution":{"iopub.status.busy":"2022-07-25T15:03:08.127739Z","iopub.execute_input":"2022-07-25T15:03:08.128006Z","iopub.status.idle":"2022-07-25T15:03:11.967010Z","shell.execute_reply.started":"2022-07-25T15:03:08.127972Z","shell.execute_reply":"2022-07-25T15:03:11.966222Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"In order to predict some articles to other articles we need to merge our dataset with the articles dataframe :","metadata":{}},{"cell_type":"code","source":"df1 = transactions_1w.merge(articles, on='article_id')","metadata":{"execution":{"iopub.status.busy":"2022-07-25T15:03:11.971889Z","iopub.execute_input":"2022-07-25T15:03:11.973934Z","iopub.status.idle":"2022-07-25T15:03:13.125072Z","shell.execute_reply.started":"2022-07-25T15:03:11.973893Z","shell.execute_reply":"2022-07-25T15:03:13.124336Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df1.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-25T15:03:13.126319Z","iopub.execute_input":"2022-07-25T15:03:13.126589Z","iopub.status.idle":"2022-07-25T15:03:13.135283Z","shell.execute_reply.started":"2022-07-25T15:03:13.126554Z","shell.execute_reply":"2022-07-25T15:03:13.134453Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We can see that there is a total of 266 364 transactions the last week","metadata":{}},{"cell_type":"markdown","source":"# Find Items Purchased Together","metadata":{}},{"cell_type":"markdown","source":"The most purchased items are displayed :","metadata":{}},{"cell_type":"code","source":"vc = df1.article_id.value_counts()\nprint(vc)","metadata":{"execution":{"iopub.status.busy":"2022-07-25T15:03:13.136823Z","iopub.execute_input":"2022-07-25T15:03:13.137309Z","iopub.status.idle":"2022-07-25T15:03:13.152108Z","shell.execute_reply.started":"2022-07-25T15:03:13.137272Z","shell.execute_reply":"2022-07-25T15:03:13.151374Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"vc = df1.article_id.value_counts()\nvc1 = df1.product_type_no.value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-07-25T15:04:22.170513Z","iopub.execute_input":"2022-07-25T15:04:22.171122Z","iopub.status.idle":"2022-07-25T15:04:22.182499Z","shell.execute_reply.started":"2022-07-25T15:04:22.171080Z","shell.execute_reply":"2022-07-25T15:04:22.181779Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"vc.describe()","metadata":{"execution":{"iopub.status.busy":"2022-07-25T15:04:25.501577Z","iopub.execute_input":"2022-07-25T15:04:25.501954Z","iopub.status.idle":"2022-07-25T15:04:25.515165Z","shell.execute_reply.started":"2022-07-25T15:04:25.501923Z","shell.execute_reply":"2022-07-25T15:04:25.514344Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We can see that of all the items purchased in the last week, the average number of purchases on these items is 14.","metadata":{}},{"cell_type":"code","source":"best_articles = []\nfor i in range(len(vc)):\n    if vc.values[i] >= 14:\n        best_articles.append(vc.values[i])\n\nprint(len(best_articles))","metadata":{"execution":{"iopub.status.busy":"2022-07-25T15:04:37.937534Z","iopub.execute_input":"2022-07-25T15:04:37.938113Z","iopub.status.idle":"2022-07-25T15:04:37.982606Z","shell.execute_reply.started":"2022-07-25T15:04:37.938075Z","shell.execute_reply":"2022-07-25T15:04:37.981716Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"3911 is the index of the last article which has been bought more than 14 times. So we decided to cut this vc list in order to cut near the average and to make the execution easier :","metadata":{}},{"cell_type":"code","source":"# FIND ITEMS PURCHASED TOGETHER\npairs = {}\nfor j,i in enumerate(vc.index.values[:3911]): # We select only the first 3911 articles\n    USERS = df1.loc[df1.article_id==i.item(),'customer_id'].unique() # We create a list of the customer who bought one of those articles\n    vc2 = df1.loc[(df1.customer_id.isin(USERS))&(df1.article_id!=i.item()),'article_id'].value_counts() # We create a list of the articles bought by these users which is not the article in quetsion and it is sorted by values\n    pairs[i.item()] = [vc2.index[0], vc2.index[1], vc2.index[2]] # We save only the 3 articles that appears most of the time","metadata":{"execution":{"iopub.status.busy":"2022-07-25T15:04:40.955625Z","iopub.execute_input":"2022-07-25T15:04:40.955892Z","iopub.status.idle":"2022-07-25T15:06:16.933170Z","shell.execute_reply.started":"2022-07-25T15:04:40.955861Z","shell.execute_reply":"2022-07-25T15:06:16.931149Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We do the same as previously but with the other part of the vc list :","metadata":{}},{"cell_type":"code","source":"pairs2 = {}\nfor j,i in enumerate(vc.index.values[3911:18683]):\n    USERS2 = df1.loc[df1.article_id==i.item(),'customer_id'].unique()\n    vc22 = df1.loc[(df1.customer_id.isin(USERS2))&(df1.article_id!=i.item()),'article_id'].value_counts()\n    # Sometimes there are no other articles bought with the article in question or few\n    if len(vc22) == 0:\n        pairs2[i.item()] = []\n    elif len(vc22) == 1:\n        pairs2[i.item()] = [vc22.index[0]]\n    elif len(vc22) == 2:\n        pairs2[i.item()] = [vc22.index[0], vc22.index[1]]\n    else:\n        pairs2[i.item()] = [vc22.index[0], vc22.index[1], vc22.index[2]]                   ","metadata":{"execution":{"iopub.status.busy":"2022-07-25T15:06:16.935757Z","iopub.status.idle":"2022-07-25T15:06:16.936208Z","shell.execute_reply.started":"2022-07-25T15:06:16.935959Z","shell.execute_reply":"2022-07-25T15:06:16.935983Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"For an article :","metadata":{}},{"cell_type":"code","source":"list(pairs.keys())[0]","metadata":{"execution":{"iopub.status.busy":"2022-07-25T15:06:16.937266Z","iopub.status.idle":"2022-07-25T15:06:16.937657Z","shell.execute_reply.started":"2022-07-25T15:06:16.937424Z","shell.execute_reply":"2022-07-25T15:06:16.937445Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We have 3 articles the most frequently purchased :","metadata":{}},{"cell_type":"code","source":"list(pairs.values())[0]","metadata":{"execution":{"iopub.status.busy":"2022-07-25T15:06:16.938719Z","iopub.status.idle":"2022-07-25T15:06:16.939110Z","shell.execute_reply.started":"2022-07-25T15:06:16.938893Z","shell.execute_reply":"2022-07-25T15:06:16.938915Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"For an article :","metadata":{}},{"cell_type":"code","source":"list(pairs2.keys())[5]","metadata":{"execution":{"iopub.status.busy":"2022-07-25T15:06:16.940364Z","iopub.status.idle":"2022-07-25T15:06:16.940930Z","shell.execute_reply.started":"2022-07-25T15:06:16.940683Z","shell.execute_reply":"2022-07-25T15:06:16.940708Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We have 3 articles the most frequently purchased :","metadata":{}},{"cell_type":"code","source":"list(pairs2.values())[0]","metadata":{"execution":{"iopub.status.busy":"2022-07-25T15:06:16.942202Z","iopub.status.idle":"2022-07-25T15:06:16.942660Z","shell.execute_reply.started":"2022-07-25T15:06:16.942417Z","shell.execute_reply":"2022-07-25T15:06:16.942442Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(list(pairs.keys()))","metadata":{"execution":{"iopub.status.busy":"2022-07-25T15:06:16.943952Z","iopub.status.idle":"2022-07-25T15:06:16.944405Z","shell.execute_reply.started":"2022-07-25T15:06:16.944166Z","shell.execute_reply":"2022-07-25T15:06:16.944189Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Now, we will combine the two dictionaries :","metadata":{}},{"cell_type":"code","source":"pairs.update(pairs2)","metadata":{"execution":{"iopub.status.busy":"2022-07-25T15:06:16.945866Z","iopub.status.idle":"2022-07-25T15:06:16.946329Z","shell.execute_reply.started":"2022-07-25T15:06:16.946080Z","shell.execute_reply":"2022-07-25T15:06:16.946104Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Now that we have 3 articles predicted for 1 article we need to have access to the articles purchased by 1 article :","metadata":{}},{"cell_type":"code","source":"def specification (df):\n    groupby_customer = df.groupby('customer_id') # group by customer\n    \n    customers_articles = {}\n\n    for key in groupby_customer.groups.keys():\n        temp = groupby_customer.get_group(key) # unique customer\n        customers_articles[key]=temp.article_id.values.tolist() # list of all the articles purchased by an unique customer\n    return customers_articles ","metadata":{"execution":{"iopub.status.busy":"2022-07-25T15:06:16.948283Z","iopub.status.idle":"2022-07-25T15:06:16.948858Z","shell.execute_reply.started":"2022-07-25T15:06:16.948616Z","shell.execute_reply":"2022-07-25T15:06:16.948642Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cc = specification(df1)","metadata":{"execution":{"iopub.status.busy":"2022-07-25T15:06:16.950276Z","iopub.status.idle":"2022-07-25T15:06:16.951199Z","shell.execute_reply.started":"2022-07-25T15:06:16.950974Z","shell.execute_reply":"2022-07-25T15:06:16.950998Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"For a customer :","metadata":{}},{"cell_type":"code","source":"list(cc.keys())[6]","metadata":{"execution":{"iopub.status.busy":"2022-07-25T15:06:16.952231Z","iopub.status.idle":"2022-07-25T15:06:16.952927Z","shell.execute_reply.started":"2022-07-25T15:06:16.952680Z","shell.execute_reply":"2022-07-25T15:06:16.952705Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We have all the articles he purchased :","metadata":{}},{"cell_type":"code","source":"list(cc.values())[6]","metadata":{"execution":{"iopub.status.busy":"2022-07-25T15:06:16.954261Z","iopub.status.idle":"2022-07-25T15:06:16.954679Z","shell.execute_reply.started":"2022-07-25T15:06:16.954446Z","shell.execute_reply":"2022-07-25T15:06:16.954469Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"But we have a problem here ! Because if a customer buys the same item twice, we have the article_id two times. So, we need to have, for each customer, the article_id and the number of times it has purchased :","metadata":{}},{"cell_type":"code","source":"purchase_dict_1w = {}\narticles_id = []\n\nfor i in range(len(list(cc.keys()))):\n    \n    cust_id = list(cc.keys())[i]\n    articles_id = list(cc.values())[i] # articles purchased buy a specific customer\n    \n    if cust_id not in purchase_dict_1w:\n        purchase_dict_1w[cust_id] = {}\n    \n    for i in range(len(articles_id)):\n        art_id = articles_id[i]\n        if articles_id[i] not in purchase_dict_1w[cust_id]:\n            purchase_dict_1w[cust_id][art_id] = 0 # if the customer purchased this article we will put a 0\n        else:\n            purchase_dict_1w[cust_id][art_id] += 1 # if he purchased it again we add 1\n    \nprint(len(purchase_dict_1w))","metadata":{"execution":{"iopub.status.busy":"2022-07-25T15:06:16.955831Z","iopub.status.idle":"2022-07-25T15:06:16.956377Z","shell.execute_reply.started":"2022-07-25T15:06:16.956133Z","shell.execute_reply":"2022-07-25T15:06:16.956157Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"So now for a customer :","metadata":{}},{"cell_type":"code","source":"list(purchase_dict_1w.keys())[9]","metadata":{"execution":{"iopub.status.busy":"2022-07-25T15:06:16.957869Z","iopub.status.idle":"2022-07-25T15:06:16.958285Z","shell.execute_reply.started":"2022-07-25T15:06:16.958057Z","shell.execute_reply":"2022-07-25T15:06:16.958080Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We have the article_id with the number of time that the customer has purchased it :","metadata":{}},{"cell_type":"code","source":"list(purchase_dict_1w.values())[9] # We have little issue, we need to +1 all the number print","metadata":{"execution":{"iopub.status.busy":"2022-07-25T15:06:16.959781Z","iopub.status.idle":"2022-07-25T15:06:16.960257Z","shell.execute_reply.started":"2022-07-25T15:06:16.960017Z","shell.execute_reply":"2022-07-25T15:06:16.960041Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"But we also want to sorted this dictionnary :","metadata":{}},{"cell_type":"code","source":"purchase_dict_1w_sorted = {}\nfor i in range(len(list(purchase_dict_1w.values()))):\n    dico = list(purchase_dict_1w.values())[i] # We stock in a dico all the value for each customer\n    sortedDict = sorted(dico.items(), key=lambda x: x[1], reverse=True) # We sorted it \n    purchase_dict_1w_sorted[list(purchase_dict_1w.keys())[i]] = sortedDict # We made it correspond with the good key","metadata":{"execution":{"iopub.status.busy":"2022-07-25T15:06:16.961617Z","iopub.status.idle":"2022-07-25T15:06:16.962016Z","shell.execute_reply.started":"2022-07-25T15:06:16.961798Z","shell.execute_reply":"2022-07-25T15:06:16.961821Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"So for the same customer :","metadata":{}},{"cell_type":"code","source":"list(purchase_dict_1w_sorted.keys())[9]","metadata":{"execution":{"iopub.status.busy":"2022-07-25T15:06:16.963564Z","iopub.status.idle":"2022-07-25T15:06:16.964215Z","shell.execute_reply.started":"2022-07-25T15:06:16.963966Z","shell.execute_reply":"2022-07-25T15:06:16.963994Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We have the articles he purchased ordered :","metadata":{}},{"cell_type":"code","source":"list(purchase_dict_1w_sorted.values())[9]","metadata":{"execution":{"iopub.status.busy":"2022-07-25T15:03:13.296279Z","iopub.status.idle":"2022-07-25T15:03:13.297168Z","shell.execute_reply.started":"2022-07-25T15:03:13.296935Z","shell.execute_reply":"2022-07-25T15:03:13.296959Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"So now, we all we have created we can make some predictions :","metadata":{}},{"cell_type":"code","source":"def prediction(pairs):\n    prediction = {}\n    USERS = df1['customer_id'].unique() # for an unique customer\n    for i in range(len(USERS)):\n        values = list(purchase_dict_1w_sorted.values())[i] # to access the customer_id\n        pred = []\n        for j in range(len(values)):\n            article = values[j][0] # to go through the articles id\n            if article in list(pairs.keys()): # the specific article has to belong to the pairs dictionary\n                for k in range(len(pairs[article])):\n                    pred.append(pairs[article][k]) # we load the 3 predictions of this article\n        prediction[list(purchase_dict_1w_sorted.keys())[i]] = pred\n    return prediction","metadata":{"execution":{"iopub.status.busy":"2022-07-25T15:06:17.567273Z","iopub.execute_input":"2022-07-25T15:06:17.567525Z","iopub.status.idle":"2022-07-25T15:06:17.574175Z","shell.execute_reply.started":"2022-07-25T15:06:17.567497Z","shell.execute_reply":"2022-07-25T15:06:17.573386Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"PREDICTION = prediction(pairs)","metadata":{"execution":{"iopub.status.busy":"2022-07-25T15:03:13.300199Z","iopub.status.idle":"2022-07-25T15:03:13.300892Z","shell.execute_reply.started":"2022-07-25T15:03:13.300650Z","shell.execute_reply":"2022-07-25T15:03:13.300676Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Finally, for a customer :","metadata":{}},{"cell_type":"code","source":"list(PREDICTION.keys())[9]","metadata":{"execution":{"iopub.status.busy":"2022-07-25T15:03:13.302203Z","iopub.status.idle":"2022-07-25T15:03:13.302701Z","shell.execute_reply.started":"2022-07-25T15:03:13.302425Z","shell.execute_reply":"2022-07-25T15:03:13.302452Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We have a list of predictions !","metadata":{}},{"cell_type":"code","source":"list(PREDICTION.values())[9]","metadata":{"execution":{"iopub.status.busy":"2022-07-25T15:03:13.304042Z","iopub.status.idle":"2022-07-25T15:03:13.304606Z","shell.execute_reply.started":"2022-07-25T15:03:13.304356Z","shell.execute_reply":"2022-07-25T15:03:13.304381Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We create a list of the top-12 article purchased the last week :","metadata":{}},{"cell_type":"code","source":"dummy_list_1w = list((df1['article_id'].value_counts()).index)[:12]\ndummy_list_1w","metadata":{"execution":{"iopub.status.busy":"2022-07-25T15:03:13.305644Z","iopub.status.idle":"2022-07-25T15:03:13.306300Z","shell.execute_reply.started":"2022-07-25T15:03:13.306047Z","shell.execute_reply":"2022-07-25T15:03:13.306073Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# We create the same list but we change the type of the data\ndummy_pred_1w = []\nfor i in range(len(dummy_list_1w)):\n    dummy_pred_1w.append(str(dummy_list_1w[i]))\ndummy_pred_1w","metadata":{"execution":{"iopub.status.busy":"2022-07-25T15:03:13.308744Z","iopub.status.idle":"2022-07-25T15:03:13.309739Z","shell.execute_reply.started":"2022-07-25T15:03:13.309486Z","shell.execute_reply":"2022-07-25T15:03:13.309512Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The problem that we have right now is that we want to have only the 12 best predictions for each customer who made a transaction the last week. So sometimes we need to cut the list of predictions beacause we have too many of them but sometimes the list is too long so we will add the right amount of the top-12 article purchased in order to have the right number of 12 predictions for each customer :","metadata":{}},{"cell_type":"code","source":"def submission(prediction):\n    pred = []\n    for i in range(len(list(prediction.keys()))):\n        prediction_str = []\n        for j in range(len(list(prediction.values())[i])):\n            prediction_str.append(str(list(prediction.values())[i][j])) # we save in a list all the prediction values as str\n        \n        if len(prediction_str)>12:\n            s = ' '.join(prediction_str[:12]) # if the prediction list is longuer than 12, we keep only the first 12\n        else:\n            add_s = dummy_pred_1w[:(12-len(prediction_str))]\n            s = ' '.join(prediction_str+add_s) # otherwise we add articles from the dummy_list\n        pred.append([list(prediction.keys())[i],s])    \n    return pred","metadata":{"execution":{"iopub.status.busy":"2022-07-25T15:03:13.310842Z","iopub.status.idle":"2022-07-25T15:03:13.311613Z","shell.execute_reply.started":"2022-07-25T15:03:13.311357Z","shell.execute_reply":"2022-07-25T15:03:13.311383Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Now we have a dictionary with the predictions :","metadata":{}},{"cell_type":"code","source":"SUB = submission(PREDICTION)","metadata":{"execution":{"iopub.status.busy":"2022-07-25T15:03:13.312863Z","iopub.status.idle":"2022-07-25T15:03:13.313710Z","shell.execute_reply.started":"2022-07-25T15:03:13.313463Z","shell.execute_reply":"2022-07-25T15:03:13.313487Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We create a DataFrame from it :","metadata":{}},{"cell_type":"code","source":"SUBMISSION = pd.DataFrame(data=SUB, columns = ['customer_id', 'prediction'])","metadata":{"execution":{"iopub.status.busy":"2022-07-25T15:03:13.314770Z","iopub.status.idle":"2022-07-25T15:03:13.315654Z","shell.execute_reply.started":"2022-07-25T15:03:13.315402Z","shell.execute_reply":"2022-07-25T15:03:13.315427Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Just to have a look :","metadata":{}},{"cell_type":"code","source":"SUBMISSION","metadata":{"execution":{"iopub.status.busy":"2022-07-25T15:03:13.316704Z","iopub.status.idle":"2022-07-25T15:03:13.317406Z","shell.execute_reply.started":"2022-07-25T15:03:13.317165Z","shell.execute_reply":"2022-07-25T15:03:13.317190Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# We save a csv in order to submit it\nSUBMISSION.to_csv('submission_content_based.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-25T15:03:13.318949Z","iopub.status.idle":"2022-07-25T15:03:13.319987Z","shell.execute_reply.started":"2022-07-25T15:03:13.319696Z","shell.execute_reply":"2022-07-25T15:03:13.319723Z"},"trusted":true},"execution_count":null,"outputs":[]}]}