{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\n#import os\n#for dirname, _, filenames in os.walk('/kaggle/input'):\n#    for filename in filenames:\n#        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-03-04T18:54:09.416056Z","iopub.execute_input":"2022-03-04T18:54:09.416967Z","iopub.status.idle":"2022-03-04T18:54:09.424325Z","shell.execute_reply.started":"2022-03-04T18:54:09.416930Z","shell.execute_reply":"2022-03-04T18:54:09.422604Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_tt = pd.read_csv('/kaggle/input/h-and-m-personalized-fashion-recommendations/transactions_train.csv')\ndf_tt","metadata":{"execution":{"iopub.status.busy":"2022-03-04T18:54:12.284668Z","iopub.execute_input":"2022-03-04T18:54:12.285004Z","iopub.status.idle":"2022-03-04T18:55:29.639334Z","shell.execute_reply.started":"2022-03-04T18:54:12.284973Z","shell.execute_reply":"2022-03-04T18:55:29.638825Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"num_cols = [col for col in df_tt.columns if df_tt[col].dtype in ['int64', 'float64']]\ndf_tt[num_cols].describe()","metadata":{"execution":{"iopub.status.busy":"2022-03-04T18:55:32.213411Z","iopub.execute_input":"2022-03-04T18:55:32.213881Z","iopub.status.idle":"2022-03-04T18:55:34.772007Z","shell.execute_reply.started":"2022-03-04T18:55:32.213846Z","shell.execute_reply":"2022-03-04T18:55:34.770944Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cat_cols = [col for col in df_tt.columns if df_tt[col].dtype in ['O']]\ndf_tt[cat_cols].describe()","metadata":{"execution":{"iopub.status.busy":"2022-03-04T18:55:53.051460Z","iopub.execute_input":"2022-03-04T18:55:53.051777Z","iopub.status.idle":"2022-03-04T18:56:12.844277Z","shell.execute_reply.started":"2022-03-04T18:55:53.051745Z","shell.execute_reply":"2022-03-04T18:56:12.843023Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_ss = pd.read_csv('/kaggle/input/h-and-m-personalized-fashion-recommendations/sample_submission.csv')\ndf_ss","metadata":{"execution":{"iopub.status.busy":"2022-03-04T18:56:46.942614Z","iopub.execute_input":"2022-03-04T18:56:46.943276Z","iopub.status.idle":"2022-03-04T18:56:53.058440Z","shell.execute_reply.started":"2022-03-04T18:56:46.943243Z","shell.execute_reply":"2022-03-04T18:56:53.056864Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_ss = pd.read_csv('/kaggle/input/h-and-m-personalized-fashion-recommendations/sample_submission.csv')\ndf_ss","metadata":{"execution":{"iopub.status.busy":"2022-03-04T18:58:16.449403Z","iopub.execute_input":"2022-03-04T18:58:16.450311Z","iopub.status.idle":"2022-03-04T18:58:21.303285Z","shell.execute_reply.started":"2022-03-04T18:58:16.450264Z","shell.execute_reply":"2022-03-04T18:58:21.302567Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_cus = pd.read_csv('/kaggle/input/h-and-m-personalized-fashion-recommendations/customers.csv')\ndf_cus","metadata":{"execution":{"iopub.status.busy":"2022-03-04T19:00:17.228256Z","iopub.execute_input":"2022-03-04T19:00:17.228554Z","iopub.status.idle":"2022-03-04T19:00:23.699118Z","shell.execute_reply.started":"2022-03-04T19:00:17.228519Z","shell.execute_reply":"2022-03-04T19:00:23.698167Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"num_cols = [col for col in df_cus.columns if df_cus[col].dtype in ['int64', 'float64']]\ndf_cus[num_cols].describe()","metadata":{"execution":{"iopub.status.busy":"2022-03-04T19:00:36.407533Z","iopub.execute_input":"2022-03-04T19:00:36.407868Z","iopub.status.idle":"2022-03-04T19:00:36.596408Z","shell.execute_reply.started":"2022-03-04T19:00:36.407835Z","shell.execute_reply":"2022-03-04T19:00:36.595218Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cat_cols = [col for col in df_cus.columns if df_cus[col].dtype in ['O']]\ndf_cus[cat_cols].describe()","metadata":{"execution":{"iopub.status.busy":"2022-03-04T19:02:11.419853Z","iopub.execute_input":"2022-03-04T19:02:11.420248Z","iopub.status.idle":"2022-03-04T19:02:14.504500Z","shell.execute_reply.started":"2022-03-04T19:02:11.420222Z","shell.execute_reply":"2022-03-04T19:02:14.503430Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_art = pd.read_csv('/kaggle/input/h-and-m-personalized-fashion-recommendations/articles.csv')\ndf_art","metadata":{"execution":{"iopub.status.busy":"2022-03-04T19:03:06.126173Z","iopub.execute_input":"2022-03-04T19:03:06.126445Z","iopub.status.idle":"2022-03-04T19:03:07.414959Z","shell.execute_reply.started":"2022-03-04T19:03:06.126419Z","shell.execute_reply":"2022-03-04T19:03:07.414087Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"num_cols = [col for col in df_art.columns if df_art[col].dtype in ['int64', 'float64']]\ndf_art[num_cols].describe()","metadata":{"execution":{"iopub.status.busy":"2022-03-04T19:03:21.282233Z","iopub.execute_input":"2022-03-04T19:03:21.282686Z","iopub.status.idle":"2022-03-04T19:03:21.349431Z","shell.execute_reply.started":"2022-03-04T19:03:21.282658Z","shell.execute_reply":"2022-03-04T19:03:21.348550Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cat_cols = [col for col in df_art.columns if df_art[col].dtype in ['O']]\ndf_art[cat_cols].describe()","metadata":{"execution":{"iopub.status.busy":"2022-03-04T19:03:28.148466Z","iopub.execute_input":"2022-03-04T19:03:28.148931Z","iopub.status.idle":"2022-03-04T19:03:28.456277Z","shell.execute_reply.started":"2022-03-04T19:03:28.148897Z","shell.execute_reply":"2022-03-04T19:03:28.455080Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import glob\nimport os\n\n# number of image files\nimg = [os.path.basename(p) for p in glob.glob('/kaggle/input/h-and-m-personalized-fashion-recommendations/images/*/*.jpg', recursive=True)\n       if os.path.isfile(p)]\nlen(img)","metadata":{"execution":{"iopub.status.busy":"2022-03-04T19:11:26.020410Z","iopub.execute_input":"2022-03-04T19:11:26.020911Z","iopub.status.idle":"2022-03-04T19:12:30.451842Z","shell.execute_reply.started":"2022-03-04T19:11:26.020875Z","shell.execute_reply":"2022-03-04T19:12:30.450455Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import glob\nimport os\n\n# number of image folders\nfolders = [os.path.basename(p) for p in glob.glob('/kaggle/input/h-and-m-personalized-fashion-recommendations/images/*')]\nlen(folders)","metadata":{"execution":{"iopub.status.busy":"2022-03-04T19:09:16.191016Z","iopub.execute_input":"2022-03-04T19:09:16.191288Z","iopub.status.idle":"2022-03-04T19:09:16.212388Z","shell.execute_reply.started":"2022-03-04T19:09:16.191262Z","shell.execute_reply":"2022-03-04T19:09:16.211270Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\n\ndef apk(actual, predicted, k=12, default=0.0):\n\n    if len(predicted) > k:\n        predicted = predicted[:k]\n        \n    score = 0.0\n    num_hits = 0.0\n    \n    for i, p in enumerate(predicted):\n        if p in actual and p not in predicted[:i]:\n            num_hits += 1.0\n            score += num_hits / (i+1.0)\n            \n    if not actual:\n        return default\n    \n    return score / min(len(actual), k)\n\ndef mapk(actual, predicted, k=12, default=0.0):\n    return np.mean([apk(a, p, k, default) for a, p in zip(actual,predicted)])\n\n# check\n#mapk([[1,2,3,4,5,6,7,8,9,10,11,12]], [[1,2,3,4,5,6,7,8,9,10,11,12]], 12, 0.0)","metadata":{"execution":{"iopub.status.busy":"2022-02-27T22:08:20.381723Z","iopub.execute_input":"2022-02-27T22:08:20.381977Z","iopub.status.idle":"2022-02-27T22:08:20.391973Z","shell.execute_reply.started":"2022-02-27T22:08:20.381945Z","shell.execute_reply":"2022-02-27T22:08:20.391226Z"},"trusted":true},"execution_count":null,"outputs":[]}]}