{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Just another EDA notebook\nI'm just sharing the basic EDA work that I did before looking at other folks' notebooks. Give it an upvote if you found it useful...\n\n# Contest overview\n## Objective\n* Show customers products they'll be more likely to buy. \n* Make 12 predictions, ranked 1-12 for each customer, higher ranked that customer buys = better score.\n\n## Data\nBasic observations on looking at the data in the explorer\n* Images\n    * Folders are not organized to anything useful, simply by three digits of item id\n    * file id= article id\n* Articles.csv = 25 columns x 109m rows, describing the items for sale\n* customers.csv = 7 columns 1.4m rows\n    * Age, member status, fashion news, \n    * postal code - hash value one value with 9%, seems strange\n* transcactions_train.csv\n    * One line per customer/item bought... if mutliple items bought then mutliple lines\n    * Price column... min = 0, max = 0.59, mean = 0.03... is this a % of price paid?\n    * sale_channel_id = either 1 or 2... not sure what this means\n\n\n\n\n","metadata":{}},{"cell_type":"code","source":"import os\nimport pandas as pd\npd.options.plotting.backend = \"plotly\"\nfrom glob import glob\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom pathlib import Path","metadata":{"execution":{"iopub.status.busy":"2022-03-01T20:51:55.920515Z","iopub.execute_input":"2022-03-01T20:51:55.921089Z","iopub.status.idle":"2022-03-01T20:51:56.941881Z","shell.execute_reply.started":"2022-03-01T20:51:55.92099Z","shell.execute_reply":"2022-03-01T20:51:56.941259Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n#takes ~90 seconds without accelerator\npath = '../input/h-and-m-personalized-fashion-recommendations/'\n\narticles_df = pd.read_csv(path + 'articles.csv')\ncustomers_df = pd.read_csv(path + 'customers.csv')\nsample_submission_df =  pd.read_csv(path + 'sample_submission.csv')\ntransactions_train_df =  pd.read_csv(path + 'transactions_train.csv')\n\nimages_jpg = glob(path + \"images/*/*.jpg\")\nf'df shapes: articles{articles_df.shape}, customers{customers_df.shape}, transactions{transactions_train_df.shape}, no of images: {len(images_jpg)}, submissions{sample_submission_df.shape}'","metadata":{"execution":{"iopub.status.busy":"2022-03-01T20:51:56.943148Z","iopub.execute_input":"2022-03-01T20:51:56.943667Z","iopub.status.idle":"2022-03-01T20:53:19.275398Z","shell.execute_reply.started":"2022-03-01T20:51:56.943637Z","shell.execute_reply":"2022-03-01T20:53:19.27418Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"articles_df.describe()","metadata":{"execution":{"iopub.status.busy":"2022-03-01T20:53:19.278089Z","iopub.execute_input":"2022-03-01T20:53:19.278763Z","iopub.status.idle":"2022-03-01T20:53:19.404104Z","shell.execute_reply.started":"2022-03-01T20:53:19.278714Z","shell.execute_reply":"2022-03-01T20:53:19.402735Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"articles_df.tail()","metadata":{"execution":{"iopub.status.busy":"2022-03-01T20:53:19.405709Z","iopub.execute_input":"2022-03-01T20:53:19.40608Z","iopub.status.idle":"2022-03-01T20:53:19.434217Z","shell.execute_reply.started":"2022-03-01T20:53:19.406035Z","shell.execute_reply":"2022-03-01T20:53:19.433333Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"customers_df.tail()","metadata":{"execution":{"iopub.status.busy":"2022-03-01T20:53:19.436524Z","iopub.execute_input":"2022-03-01T20:53:19.437055Z","iopub.status.idle":"2022-03-01T20:53:19.451712Z","shell.execute_reply.started":"2022-03-01T20:53:19.437015Z","shell.execute_reply":"2022-03-01T20:53:19.450783Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"transactions_train_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-03-01T20:53:19.452944Z","iopub.execute_input":"2022-03-01T20:53:19.453971Z","iopub.status.idle":"2022-03-01T20:53:19.471971Z","shell.execute_reply.started":"2022-03-01T20:53:19.453926Z","shell.execute_reply":"2022-03-01T20:53:19.471324Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Distributions\nI'm curious to see:\n* Total items:\n * What are the most popular items?\n * What are the most popular items by customers in postal codes/ages/etc?\n* Items over time:\n * Total items sold in a trend line over time, broken down by various attributes of the items\n * Should we be predicting certian items at certian times of the year?","metadata":{}},{"cell_type":"code","source":"transactions_count_df = pd.DataFrame.from_dict(transactions_train_df['article_id'].value_counts()) #creates a dictionary of counts of items as a df\ntransactions_count_df.reset_index(level=0, inplace=True) #want index to be standard pd row numbers, not the item id...\ntransactions_count_df.rename(columns={\"index\":'article_id','article_id':'count'}, inplace=True)\ntransactions_count_df.tail()","metadata":{"execution":{"iopub.status.busy":"2022-03-01T20:53:19.47357Z","iopub.execute_input":"2022-03-01T20:53:19.473847Z","iopub.status.idle":"2022-03-01T20:53:21.225559Z","shell.execute_reply.started":"2022-03-01T20:53:19.473816Z","shell.execute_reply":"2022-03-01T20:53:21.224509Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"transactions_count_df['index1'] = transactions_count_df.index #don't want to index on count\ntransactions_count_df.drop(['index1'], axis=1, inplace= True) \ntransactions_count_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-03-01T20:53:21.22683Z","iopub.execute_input":"2022-03-01T20:53:21.227146Z","iopub.status.idle":"2022-03-01T20:53:21.241205Z","shell.execute_reply.started":"2022-03-01T20:53:21.227116Z","shell.execute_reply":"2022-03-01T20:53:21.240138Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Lots of zeros or ones...\ntransactions_count_df['count'].plot()","metadata":{"execution":{"iopub.status.busy":"2022-03-01T20:53:21.242291Z","iopub.execute_input":"2022-03-01T20:53:21.24254Z","iopub.status.idle":"2022-03-01T20:53:24.319652Z","shell.execute_reply.started":"2022-03-01T20:53:21.242507Z","shell.execute_reply":"2022-03-01T20:53:24.318803Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#df_temp.plot()\ntransactions_count_df[transactions_count_df < 1000]['count'].plot()","metadata":{"execution":{"iopub.status.busy":"2022-03-01T20:53:24.32098Z","iopub.execute_input":"2022-03-01T20:53:24.321798Z","iopub.status.idle":"2022-03-01T20:53:24.897821Z","shell.execute_reply.started":"2022-03-01T20:53:24.32176Z","shell.execute_reply":"2022-03-01T20:53:24.896699Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# all sales by date\ntransactions_train_df.tail()","metadata":{"execution":{"iopub.status.busy":"2022-03-01T20:53:24.899443Z","iopub.execute_input":"2022-03-01T20:53:24.899697Z","iopub.status.idle":"2022-03-01T20:53:24.910741Z","shell.execute_reply.started":"2022-03-01T20:53:24.899667Z","shell.execute_reply":"2022-03-01T20:53:24.90986Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"articles_df.columns","metadata":{"execution":{"iopub.status.busy":"2022-03-01T20:53:24.912506Z","iopub.execute_input":"2022-03-01T20:53:24.913136Z","iopub.status.idle":"2022-03-01T20:53:24.924447Z","shell.execute_reply.started":"2022-03-01T20:53:24.913096Z","shell.execute_reply":"2022-03-01T20:53:24.923522Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# eyeballing some columns to merge with transactions\narticle_cols = ['article_id', 'product_type_name', 'product_group_name', 'department_no', 'department_name', 'index_code', 'index_group_no', \n                'index_group_name', 'section_no', 'section_name', 'garment_group_no', 'garment_group_name']\n\narticles_df[article_cols].tail()","metadata":{"execution":{"iopub.status.busy":"2022-03-01T20:53:24.925467Z","iopub.execute_input":"2022-03-01T20:53:24.926022Z","iopub.status.idle":"2022-03-01T20:53:24.957739Z","shell.execute_reply.started":"2022-03-01T20:53:24.925989Z","shell.execute_reply":"2022-03-01T20:53:24.957112Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"transactions_with_ref_df = pd.merge(transactions_train_df, articles_df[article_cols], on= 'article_id', how='left')\ntransactions_with_ref_df.tail()","metadata":{"execution":{"iopub.status.busy":"2022-03-01T20:53:24.960129Z","iopub.execute_input":"2022-03-01T20:53:24.960374Z","iopub.status.idle":"2022-03-01T20:53:50.822319Z","shell.execute_reply.started":"2022-03-01T20:53:24.960344Z","shell.execute_reply":"2022-03-01T20:53:50.821326Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# group new df by weeks... see seasonal trends... y = number of transactions, x = week\n# stacked bar chart by garment group name (seems reasonable to try first)\n# ","metadata":{},"execution_count":null,"outputs":[]}]}