{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Exploring H&M fashion comp part 1","metadata":{}},{"cell_type":"markdown","source":"### Setting up","metadata":{}},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        if filename.endswith('csv'):\n            print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-04-18T13:05:05.575789Z","iopub.execute_input":"2023-04-18T13:05:05.576152Z","iopub.status.idle":"2023-04-18T13:06:57.041355Z","shell.execute_reply.started":"2023-04-18T13:05:05.576109Z","shell.execute_reply":"2023-04-18T13:06:57.040273Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Struggled a bit to use fastkaggle, so switched to using kaggles import data from the competition","metadata":{"execution":{"iopub.status.busy":"2023-04-16T11:23:45.584904Z","iopub.execute_input":"2023-04-16T11:23:45.585581Z","iopub.status.idle":"2023-04-16T11:23:45.590121Z","shell.execute_reply.started":"2023-04-16T11:23:45.585546Z","shell.execute_reply":"2023-04-16T11:23:45.589009Z"}}},{"cell_type":"markdown","source":"### Data analysis","metadata":{}},{"cell_type":"code","source":"sample_submission = pd.read_csv('/kaggle/input/h-and-m-personalized-fashion-recommendations/sample_submission.csv')\narticles = pd.read_csv('/kaggle/input/h-and-m-personalized-fashion-recommendations/articles.csv')\ntransactions_train = pd.read_csv('/kaggle/input/h-and-m-personalized-fashion-recommendations/transactions_train.csv')\ncustomers = pd.read_csv('/kaggle/input/h-and-m-personalized-fashion-recommendations/customers.csv')","metadata":{"execution":{"iopub.status.busy":"2023-04-18T13:06:57.043277Z","iopub.execute_input":"2023-04-18T13:06:57.043659Z","iopub.status.idle":"2023-04-18T13:08:25.759544Z","shell.execute_reply.started":"2023-04-18T13:06:57.043621Z","shell.execute_reply":"2023-04-18T13:08:25.758485Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Sample Submissions\nDescription:\n\nsample_submission.csv - a sample submission file in the correct format","metadata":{}},{"cell_type":"code","source":"print(sample_submission.head(10))","metadata":{"execution":{"iopub.status.busy":"2023-04-18T13:08:25.775557Z","iopub.execute_input":"2023-04-18T13:08:25.776196Z","iopub.status.idle":"2023-04-18T13:08:25.787017Z","shell.execute_reply.started":"2023-04-18T13:08:25.776149Z","shell.execute_reply":"2023-04-18T13:08:25.785855Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(sample_submission.info())","metadata":{"execution":{"iopub.status.busy":"2023-04-18T13:08:25.789913Z","iopub.execute_input":"2023-04-18T13:08:25.790377Z","iopub.status.idle":"2023-04-18T13:08:25.910668Z","shell.execute_reply.started":"2023-04-18T13:08:25.790327Z","shell.execute_reply":"2023-04-18T13:08:25.909548Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Articles\nDescription:\n\narticles.csv - detailed metadata for each article_id available for purchase","metadata":{}},{"cell_type":"code","source":"print(articles.head(10))","metadata":{"execution":{"iopub.status.busy":"2023-04-18T13:08:25.912381Z","iopub.execute_input":"2023-04-18T13:08:25.912758Z","iopub.status.idle":"2023-04-18T13:08:25.929030Z","shell.execute_reply.started":"2023-04-18T13:08:25.912717Z","shell.execute_reply":"2023-04-18T13:08:25.927986Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(articles.info())","metadata":{"execution":{"iopub.status.busy":"2023-04-18T13:08:25.930605Z","iopub.execute_input":"2023-04-18T13:08:25.931304Z","iopub.status.idle":"2023-04-18T13:08:25.995924Z","shell.execute_reply.started":"2023-04-18T13:08:25.931265Z","shell.execute_reply":"2023-04-18T13:08:25.994770Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Transactions\nDescription:\n\ntransactions_train.csv - the training data, consisting of the purchases each customer for each date, as well as additional information. Duplicate rows correspond to multiple purchases of the same item. Your task is to predict the article_ids each customer will purchase during the 7-day period immediately after the training data period.","metadata":{}},{"cell_type":"code","source":"print(transactions_train.head(10))","metadata":{"execution":{"iopub.status.busy":"2023-04-18T13:08:25.997580Z","iopub.execute_input":"2023-04-18T13:08:25.997943Z","iopub.status.idle":"2023-04-18T13:08:26.006249Z","shell.execute_reply.started":"2023-04-18T13:08:25.997904Z","shell.execute_reply":"2023-04-18T13:08:26.005013Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(transactions_train.info())","metadata":{"execution":{"iopub.status.busy":"2023-04-18T13:08:26.007888Z","iopub.execute_input":"2023-04-18T13:08:26.008687Z","iopub.status.idle":"2023-04-18T13:08:26.022986Z","shell.execute_reply.started":"2023-04-18T13:08:26.008649Z","shell.execute_reply":"2023-04-18T13:08:26.021872Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Customers\nDescription:\n\ncustomers.csv - metadata for each customer_id in dataset","metadata":{}},{"cell_type":"code","source":"print(customers.head(10))","metadata":{"execution":{"iopub.status.busy":"2023-04-18T13:08:26.025232Z","iopub.execute_input":"2023-04-18T13:08:26.025853Z","iopub.status.idle":"2023-04-18T13:08:26.037770Z","shell.execute_reply.started":"2023-04-18T13:08:26.025818Z","shell.execute_reply":"2023-04-18T13:08:26.036746Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(customers.info())","metadata":{"execution":{"iopub.status.busy":"2023-04-18T13:08:26.043108Z","iopub.execute_input":"2023-04-18T13:08:26.043409Z","iopub.status.idle":"2023-04-18T13:08:26.282823Z","shell.execute_reply.started":"2023-04-18T13:08:26.043383Z","shell.execute_reply":"2023-04-18T13:08:26.281698Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Starting","metadata":{}},{"cell_type":"markdown","source":"So, I have a lot of data on different things, but I want to start by making a simple model to see what kind of results I can expect.\n\nI have decided to start by looking at the data from the transaction_train file, because it includes the purchase history of the customers, and by only looking at the id for the items purchased, I should be able to create a model for some kind of predictions.","metadata":{}},{"cell_type":"markdown","source":"First I just use the to_string method on the data.","metadata":{}},{"cell_type":"code","source":"print(transactions_train.head(10))","metadata":{"execution":{"iopub.status.busy":"2023-04-18T13:08:26.284433Z","iopub.execute_input":"2023-04-18T13:08:26.285433Z","iopub.status.idle":"2023-04-18T13:08:26.294407Z","shell.execute_reply.started":"2023-04-18T13:08:26.285389Z","shell.execute_reply":"2023-04-18T13:08:26.293155Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(transactions_train.info())","metadata":{"execution":{"iopub.status.busy":"2023-04-18T13:08:26.296388Z","iopub.execute_input":"2023-04-18T13:08:26.297032Z","iopub.status.idle":"2023-04-18T13:08:26.308692Z","shell.execute_reply.started":"2023-04-18T13:08:26.296995Z","shell.execute_reply":"2023-04-18T13:08:26.307544Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"This is way too much data to use for experimentations, because I want to try to use some smaller different models.","metadata":{}},{"cell_type":"code","source":"### Define how big I want the smaller dataset to be compared to the other one\ndef take_percentage(percent=10):\n    row_count = len(transactions_train)\n    return (row_count // 100) * percent\n    ","metadata":{"execution":{"iopub.status.busy":"2023-04-18T13:08:26.310447Z","iopub.execute_input":"2023-04-18T13:08:26.311119Z","iopub.status.idle":"2023-04-18T13:08:26.318055Z","shell.execute_reply.started":"2023-04-18T13:08:26.311081Z","shell.execute_reply":"2023-04-18T13:08:26.316780Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"### Size of smaller dataset\nsize = take_percentage(1)\n\nsmaller_transactions = transactions_train.head(size)","metadata":{"execution":{"iopub.status.busy":"2023-04-18T13:08:26.319385Z","iopub.execute_input":"2023-04-18T13:08:26.320334Z","iopub.status.idle":"2023-04-18T13:08:26.328304Z","shell.execute_reply.started":"2023-04-18T13:08:26.320278Z","shell.execute_reply":"2023-04-18T13:08:26.327358Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(smaller_transactions.to_string(max_rows=10))","metadata":{"execution":{"iopub.status.busy":"2023-04-18T13:08:26.329727Z","iopub.execute_input":"2023-04-18T13:08:26.330179Z","iopub.status.idle":"2023-04-18T13:08:26.348788Z","shell.execute_reply.started":"2023-04-18T13:08:26.330142Z","shell.execute_reply":"2023-04-18T13:08:26.347621Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"smaller_transactions.info()","metadata":{"execution":{"iopub.status.busy":"2023-04-18T13:08:26.350476Z","iopub.execute_input":"2023-04-18T13:08:26.350912Z","iopub.status.idle":"2023-04-18T13:08:26.391409Z","shell.execute_reply.started":"2023-04-18T13:08:26.350872Z","shell.execute_reply":"2023-04-18T13:08:26.390305Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Now this is a more managable size to play with some models.\n\nThere is a problem here where there might be customers without any data, but I will ignore this for now, and see how it works with this smaller subset.","metadata":{}},{"cell_type":"markdown","source":"From here on I am taking inspiration from the approach from the notebook on collaborative filtering from the fastbook github at: https://github.com/fastai/fastbook/blob/master/08_collab.ipynb","metadata":{}},{"cell_type":"code","source":"### Import the relevant fastai packages\nfrom fastai.collab import *\nfrom fastai.tabular.all import *","metadata":{"execution":{"iopub.status.busy":"2023-04-18T13:08:26.393087Z","iopub.execute_input":"2023-04-18T13:08:26.393464Z","iopub.status.idle":"2023-04-18T13:08:29.210506Z","shell.execute_reply.started":"2023-04-18T13:08:26.393428Z","shell.execute_reply":"2023-04-18T13:08:29.209467Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Since the transactions set now has no actual information about the products that are bought, I just quickly merge it with the articles set, that includes all the info about the different articles.","metadata":{}},{"cell_type":"code","source":"articles.head()","metadata":{"execution":{"iopub.status.busy":"2023-04-18T13:08:29.211839Z","iopub.execute_input":"2023-04-18T13:08:29.212888Z","iopub.status.idle":"2023-04-18T13:08:29.241386Z","shell.execute_reply.started":"2023-04-18T13:08:29.212847Z","shell.execute_reply":"2023-04-18T13:08:29.240460Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"transactions = smaller_transactions.merge(articles)","metadata":{"execution":{"iopub.status.busy":"2023-04-18T13:08:29.242540Z","iopub.execute_input":"2023-04-18T13:08:29.243360Z","iopub.status.idle":"2023-04-18T13:08:30.229321Z","shell.execute_reply.started":"2023-04-18T13:08:29.243284Z","shell.execute_reply":"2023-04-18T13:08:30.228230Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"transactions.head()","metadata":{"execution":{"iopub.status.busy":"2023-04-18T13:08:30.230923Z","iopub.execute_input":"2023-04-18T13:08:30.231407Z","iopub.status.idle":"2023-04-18T13:08:30.258285Z","shell.execute_reply.started":"2023-04-18T13:08:30.231366Z","shell.execute_reply":"2023-04-18T13:08:30.257422Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Now as a result I have a dataset with both the purchases as well as the information about the articles purchased.","metadata":{}},{"cell_type":"markdown","source":"Now I'll just try to make a dataloader and see what happens.","metadata":{}},{"cell_type":"code","source":"dls = CollabDataLoaders.from_df(transactions,bs=64)\ndls.show_batch()","metadata":{"execution":{"iopub.status.busy":"2023-04-18T13:08:30.259588Z","iopub.execute_input":"2023-04-18T13:08:30.260026Z","iopub.status.idle":"2023-04-18T13:08:31.554431Z","shell.execute_reply.started":"2023-04-18T13:08:30.259987Z","shell.execute_reply":"2023-04-18T13:08:31.553306Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"### This prints too much\n#dls.classes","metadata":{"execution":{"iopub.status.busy":"2023-04-18T13:08:31.556152Z","iopub.execute_input":"2023-04-18T13:08:31.556540Z","iopub.status.idle":"2023-04-18T13:08:31.561347Z","shell.execute_reply.started":"2023-04-18T13:08:31.556501Z","shell.execute_reply":"2023-04-18T13:08:31.560131Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#n_customers = len(dls.classes['customer_id'])\n#n_articles = len(dls.classes['article_id'])","metadata":{"execution":{"iopub.status.busy":"2023-04-18T13:11:25.610825Z","iopub.execute_input":"2023-04-18T13:11:25.611399Z","iopub.status.idle":"2023-04-18T13:11:25.616721Z","shell.execute_reply.started":"2023-04-18T13:11:25.611303Z","shell.execute_reply":"2023-04-18T13:11:25.615672Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"It seems like I set up the dataloaders wrong here, ","metadata":{}},{"cell_type":"markdown","source":"Lets look a bit more at the data.","metadata":{}},{"cell_type":"code","source":"transactions['article_id'].value_counts().head(10)","metadata":{"execution":{"iopub.status.busy":"2023-04-18T13:11:26.840305Z","iopub.execute_input":"2023-04-18T13:11:26.840799Z","iopub.status.idle":"2023-04-18T13:11:26.859180Z","shell.execute_reply.started":"2023-04-18T13:11:26.840756Z","shell.execute_reply":"2023-04-18T13:11:26.858113Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"After thinking a bit, I realized that the dataloader for collaborative filtering are supposed to have some \"set\" values. User, Item and Rating. The problem here was that I had not defined these, and they were just automatically designated to the three first columns of the table.\n\nBut there is also the problem that there is no rating system here. So how do I fix that? Well... I'll just add a column for it and set it to 1, since all the items here are bought.","metadata":{}},{"cell_type":"code","source":"transactions['rating'] = 1","metadata":{"execution":{"iopub.status.busy":"2023-04-18T13:11:27.430015Z","iopub.execute_input":"2023-04-18T13:11:27.430493Z","iopub.status.idle":"2023-04-18T13:11:27.440680Z","shell.execute_reply.started":"2023-04-18T13:11:27.430450Z","shell.execute_reply":"2023-04-18T13:11:27.439366Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"This should do the trick, lets look at the data.","metadata":{}},{"cell_type":"code","source":"transactions.head(5)","metadata":{"execution":{"iopub.status.busy":"2023-04-18T13:11:28.778237Z","iopub.execute_input":"2023-04-18T13:11:28.778749Z","iopub.status.idle":"2023-04-18T13:11:28.811624Z","shell.execute_reply.started":"2023-04-18T13:11:28.778698Z","shell.execute_reply":"2023-04-18T13:11:28.810431Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"A column of ones has magically appeared, so lets try to make a new dataloader.","metadata":{}},{"cell_type":"code","source":"dls = CollabDataLoaders.from_df(transactions,user_name='customer_id',item_name='article_id',rating_name='rating',bs=64)\ndls.show_batch()","metadata":{"execution":{"iopub.status.busy":"2023-04-18T13:11:29.920273Z","iopub.execute_input":"2023-04-18T13:11:29.920777Z","iopub.status.idle":"2023-04-18T13:11:31.801592Z","shell.execute_reply.started":"2023-04-18T13:11:29.920731Z","shell.execute_reply":"2023-04-18T13:11:31.800415Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Hmm, but there are no ratings of zero, which might have to be fixed to before I use it, but why not just try.","metadata":{}},{"cell_type":"code","source":"### This still prints too much\n#dls.classes","metadata":{"execution":{"iopub.status.busy":"2023-04-18T13:11:31.803580Z","iopub.execute_input":"2023-04-18T13:11:31.804264Z","iopub.status.idle":"2023-04-18T13:11:31.809142Z","shell.execute_reply.started":"2023-04-18T13:11:31.804221Z","shell.execute_reply":"2023-04-18T13:11:31.808011Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"n_customers = len(dls.classes['customer_id'])\nn_articles = len(dls.classes['article_id'])","metadata":{"execution":{"iopub.status.busy":"2023-04-18T13:11:35.068555Z","iopub.execute_input":"2023-04-18T13:11:35.069015Z","iopub.status.idle":"2023-04-18T13:11:35.077986Z","shell.execute_reply.started":"2023-04-18T13:11:35.068973Z","shell.execute_reply":"2023-04-18T13:11:35.076785Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Now lets jus throw it into a neural network and see what comes out.","metadata":{}},{"cell_type":"code","source":"learn = collab_learner(dls, use_nn=True, y_range=(0, 1.5), layers=[100,50])\nlearn.fit_one_cycle(5, 5e-3, wd=0.1)","metadata":{"execution":{"iopub.status.busy":"2023-04-18T13:08:31.685116Z","iopub.status.idle":"2023-04-18T13:08:31.685959Z","shell.execute_reply.started":"2023-04-18T13:08:31.685700Z","shell.execute_reply":"2023-04-18T13:08:31.685726Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"This model must be amazing, 0 validation loss and 0 training loss.\n\nSo... something is obviously wrong with this one.","metadata":{}},{"cell_type":"markdown","source":"All the ratings I have used for the model are the same, 1. I'm have some ideas but I am not sure what to do next, so I will move a bit away from collaborative filtering, and instead try to modify the dataframe in another way.","metadata":{}},{"cell_type":"code","source":"### I go back to the transactions set i made before\n\ntransactions.head(5)","metadata":{"execution":{"iopub.status.busy":"2023-04-18T13:11:48.092821Z","iopub.execute_input":"2023-04-18T13:11:48.093277Z","iopub.status.idle":"2023-04-18T13:11:48.142362Z","shell.execute_reply.started":"2023-04-18T13:11:48.093234Z","shell.execute_reply":"2023-04-18T13:11:48.141351Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"transactions.tail(5)","metadata":{"execution":{"iopub.status.busy":"2023-04-18T13:11:53.160444Z","iopub.execute_input":"2023-04-18T13:11:53.160908Z","iopub.status.idle":"2023-04-18T13:11:53.199355Z","shell.execute_reply.started":"2023-04-18T13:11:53.160866Z","shell.execute_reply":"2023-04-18T13:11:53.198369Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"That is interesting, so with my 300000+ rows in this smaller dataset, I only managed to cover a single week of purchases.\n\nBut here I want to predict the purchases of specific customers a week after a period, so I need the data to describe just that.","metadata":{}},{"cell_type":"code","source":"### I will start by picking some customers\n\nchosen_customers = customers.head(100)","metadata":{"execution":{"iopub.status.busy":"2023-04-18T13:11:56.376156Z","iopub.execute_input":"2023-04-18T13:11:56.376647Z","iopub.status.idle":"2023-04-18T13:11:56.386976Z","shell.execute_reply.started":"2023-04-18T13:11:56.376603Z","shell.execute_reply":"2023-04-18T13:11:56.385951Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"chosen_customers.head(3)","metadata":{"execution":{"iopub.status.busy":"2023-04-18T13:12:04.912471Z","iopub.execute_input":"2023-04-18T13:12:04.912939Z","iopub.status.idle":"2023-04-18T13:12:04.936821Z","shell.execute_reply.started":"2023-04-18T13:12:04.912894Z","shell.execute_reply":"2023-04-18T13:12:04.935829Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"I will ignore all the information other than the customer ID for now, and I want to get an impression of how many purchases they made over in some timeframe.","metadata":{}},{"cell_type":"code","source":"transactions_train.head(2)","metadata":{"execution":{"iopub.status.busy":"2023-04-18T13:12:10.114785Z","iopub.execute_input":"2023-04-18T13:12:10.115250Z","iopub.status.idle":"2023-04-18T13:12:10.136383Z","shell.execute_reply.started":"2023-04-18T13:12:10.115198Z","shell.execute_reply":"2023-04-18T13:12:10.135300Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"transactions_train.tail(2)","metadata":{"execution":{"iopub.status.busy":"2023-04-18T13:12:11.300401Z","iopub.execute_input":"2023-04-18T13:12:11.300875Z","iopub.status.idle":"2023-04-18T13:12:11.322029Z","shell.execute_reply.started":"2023-04-18T13:12:11.300831Z","shell.execute_reply":"2023-04-18T13:12:11.321036Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Lets create a new dataframe which only has the transaction data for the selected customers.","metadata":{}},{"cell_type":"code","source":"chosen_customers['customer_id'].head(2)","metadata":{"execution":{"iopub.status.busy":"2023-04-18T13:22:08.142377Z","iopub.execute_input":"2023-04-18T13:22:08.142728Z","iopub.status.idle":"2023-04-18T13:22:08.161616Z","shell.execute_reply.started":"2023-04-18T13:22:08.142686Z","shell.execute_reply":"2023-04-18T13:22:08.159854Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"chosen_customers_id = chosen_customers['customer_id'].values.tolist()\nchosen_customers_id","metadata":{"execution":{"iopub.status.busy":"2023-04-18T13:23:13.867288Z","iopub.execute_input":"2023-04-18T13:23:13.867787Z","iopub.status.idle":"2023-04-18T13:23:13.879891Z","shell.execute_reply.started":"2023-04-18T13:23:13.867741Z","shell.execute_reply":"2023-04-18T13:23:13.878592Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Now I have a list of the chosen customer id's.\n\nSo I start extracting them from the transactions set.","metadata":{}},{"cell_type":"code","source":"test_customer1 = chosen_customers_id[0]\ntest_customer1","metadata":{"execution":{"iopub.status.busy":"2023-04-18T13:26:38.768133Z","iopub.execute_input":"2023-04-18T13:26:38.768636Z","iopub.status.idle":"2023-04-18T13:26:38.777052Z","shell.execute_reply.started":"2023-04-18T13:26:38.768591Z","shell.execute_reply":"2023-04-18T13:26:38.776022Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_from_c1 = transactions_train.loc[transactions_train['customer_id'] == test_customer1]\ndata_from_c1","metadata":{"execution":{"iopub.status.busy":"2023-04-18T13:30:18.868860Z","iopub.execute_input":"2023-04-18T13:30:18.869548Z","iopub.status.idle":"2023-04-18T13:30:18.947070Z","shell.execute_reply.started":"2023-04-18T13:30:18.869499Z","shell.execute_reply":"2023-04-18T13:30:18.945561Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Lets continue with this approach in the next notebook, since this one is getting a bit long.","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}