{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-04-03T11:44:17.326042Z","iopub.execute_input":"2022-04-03T11:44:17.326431Z","iopub.status.idle":"2022-04-03T11:44:19.493824Z","shell.execute_reply.started":"2022-04-03T11:44:17.326394Z","shell.execute_reply":"2022-04-03T11:44:19.492653Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**First, read the data files**","metadata":{}},{"cell_type":"code","source":"# import useful libraries\nimport numpy as np\nimport pandas as pd\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n%matplotlib inline\nimport seaborn as sns\nsns.set (color_codes=True)\n\n# read the data set of articles\ndataA = pd.read_csv ('../input/h-and-m-personalized-fashion-recommendations/articles.csv')\n\n# printing the data\ndataA.head (10)","metadata":{"execution":{"iopub.status.busy":"2022-04-21T09:11:33.518601Z","iopub.execute_input":"2022-04-21T09:11:33.518953Z","iopub.status.idle":"2022-04-21T09:11:35.660997Z","shell.execute_reply.started":"2022-04-21T09:11:33.518916Z","shell.execute_reply":"2022-04-21T09:11:35.660076Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Checking the missing values**","metadata":{}},{"cell_type":"code","source":"# checking the missing values\ndataA.isnull ().sum ()","metadata":{"execution":{"iopub.status.busy":"2022-04-21T09:11:38.034722Z","iopub.execute_input":"2022-04-21T09:11:38.035257Z","iopub.status.idle":"2022-04-21T09:11:38.202191Z","shell.execute_reply.started":"2022-04-21T09:11:38.035222Z","shell.execute_reply":"2022-04-21T09:11:38.201453Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Dropping the missing values\ndataA = dataA.dropna ()\ndataA.count ()","metadata":{"execution":{"iopub.status.busy":"2022-04-21T09:11:42.450250Z","iopub.execute_input":"2022-04-21T09:11:42.451192Z","iopub.status.idle":"2022-04-21T09:11:42.812008Z","shell.execute_reply.started":"2022-04-21T09:11:42.451151Z","shell.execute_reply":"2022-04-21T09:11:42.811073Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# after dropping the values\nprint (dataA.isnull ().sum ())","metadata":{"execution":{"iopub.status.busy":"2022-04-21T09:11:44.307255Z","iopub.execute_input":"2022-04-21T09:11:44.307558Z","iopub.status.idle":"2022-04-21T09:11:44.473266Z","shell.execute_reply.started":"2022-04-21T09:11:44.307525Z","shell.execute_reply":"2022-04-21T09:11:44.472539Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Select the necessary variables**","metadata":{}},{"cell_type":"code","source":"dataA = dataA [['product_code', 'colour_group_code', 'garment_group_no']]\ndataA.head (10)","metadata":{"execution":{"iopub.status.busy":"2022-04-21T09:12:04.769107Z","iopub.execute_input":"2022-04-21T09:12:04.770049Z","iopub.status.idle":"2022-04-21T09:12:04.782465Z","shell.execute_reply.started":"2022-04-21T09:12:04.769987Z","shell.execute_reply":"2022-04-21T09:12:04.781393Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Processing the dataset of customers**","metadata":{}},{"cell_type":"code","source":"# read the dataset of customers\ndataC = pd.read_csv ('../input/h-and-m-personalized-fashion-recommendations/customers.csv')\n\n# printing the data\ndataC.head (10)","metadata":{"execution":{"iopub.status.busy":"2022-04-21T09:12:06.839829Z","iopub.execute_input":"2022-04-21T09:12:06.840181Z","iopub.status.idle":"2022-04-21T09:12:12.170228Z","shell.execute_reply.started":"2022-04-21T09:12:06.840146Z","shell.execute_reply":"2022-04-21T09:12:12.169219Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# checking the missing values\ndataC.isnull ().sum ()","metadata":{"execution":{"iopub.status.busy":"2022-04-21T09:12:14.330384Z","iopub.execute_input":"2022-04-21T09:12:14.330690Z","iopub.status.idle":"2022-04-21T09:12:14.943445Z","shell.execute_reply.started":"2022-04-21T09:12:14.330658Z","shell.execute_reply":"2022-04-21T09:12:14.942428Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# dropping the missing values\ndataC = dataC.dropna ()\ndataC.count ()","metadata":{"execution":{"iopub.status.busy":"2022-04-21T09:12:16.331276Z","iopub.execute_input":"2022-04-21T09:12:16.331585Z","iopub.status.idle":"2022-04-21T09:12:17.240488Z","shell.execute_reply.started":"2022-04-21T09:12:16.331554Z","shell.execute_reply":"2022-04-21T09:12:17.239601Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# after dropping the values\nprint (dataC.isnull().sum ())","metadata":{"execution":{"iopub.status.busy":"2022-04-21T09:12:18.714863Z","iopub.execute_input":"2022-04-21T09:12:18.715195Z","iopub.status.idle":"2022-04-21T09:12:18.955295Z","shell.execute_reply.started":"2022-04-21T09:12:18.715161Z","shell.execute_reply":"2022-04-21T09:12:18.953905Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Select the necessary variables**","metadata":{}},{"cell_type":"code","source":"dataC = dataC [['age', 'club_member_status', 'fashion_news_frequency']]\ndataC.head (10)","metadata":{"execution":{"iopub.status.busy":"2022-04-21T09:12:22.747970Z","iopub.execute_input":"2022-04-21T09:12:22.748834Z","iopub.status.idle":"2022-04-21T09:12:22.787947Z","shell.execute_reply.started":"2022-04-21T09:12:22.748777Z","shell.execute_reply":"2022-04-21T09:12:22.787164Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Processing the dataset transactions**","metadata":{}},{"cell_type":"code","source":"# read the dataset of transactions\ndataT = pd.read_csv ('../input/h-and-m-personalized-fashion-recommendations/transactions_train.csv')\n\n# printing the data \ndataT.head (10)","metadata":{"execution":{"iopub.status.busy":"2022-04-21T09:12:30.848053Z","iopub.execute_input":"2022-04-21T09:12:30.848377Z","iopub.status.idle":"2022-04-21T09:13:34.330816Z","shell.execute_reply.started":"2022-04-21T09:12:30.848345Z","shell.execute_reply":"2022-04-21T09:13:34.329923Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Checking the missing values**","metadata":{}},{"cell_type":"code","source":"# checking the missing values\ndataT.isnull ().sum ()","metadata":{"execution":{"iopub.status.busy":"2022-04-21T09:13:37.204913Z","iopub.execute_input":"2022-04-21T09:13:37.205391Z","iopub.status.idle":"2022-04-21T09:13:44.071717Z","shell.execute_reply.started":"2022-04-21T09:13:37.205350Z","shell.execute_reply":"2022-04-21T09:13:44.070765Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Select the necessary variables**","metadata":{}},{"cell_type":"code","source":"dataT = dataT [['t_dat', 'article_id', 'price']]\ndataT.head (10)","metadata":{"execution":{"iopub.status.busy":"2022-04-21T09:13:46.361133Z","iopub.execute_input":"2022-04-21T09:13:46.361476Z","iopub.status.idle":"2022-04-21T09:13:46.801477Z","shell.execute_reply.started":"2022-04-21T09:13:46.361439Z","shell.execute_reply":"2022-04-21T09:13:46.800604Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Create a sample from dataset transactions and sample from customers dataset and sample from articles**","metadata":{}},{"cell_type":"code","source":"# create a sample from articles dataset\ndataA = dataA.sample (n=1000)\n\n# check the shape of sample dataset\ndataA.shape","metadata":{"execution":{"iopub.status.busy":"2022-04-21T09:13:55.131874Z","iopub.execute_input":"2022-04-21T09:13:55.132578Z","iopub.status.idle":"2022-04-21T09:13:55.142361Z","shell.execute_reply.started":"2022-04-21T09:13:55.132519Z","shell.execute_reply":"2022-04-21T09:13:55.141587Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# create a sample from dataset transactions\ndataT = dataT.sample (n=1000)\n\n# check the shape of sample dataset\ndataT.shape","metadata":{"execution":{"iopub.status.busy":"2022-04-21T09:13:57.633655Z","iopub.execute_input":"2022-04-21T09:13:57.637023Z","iopub.status.idle":"2022-04-21T09:13:59.449840Z","shell.execute_reply.started":"2022-04-21T09:13:57.636956Z","shell.execute_reply":"2022-04-21T09:13:59.449226Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# create a sample from customers dataset \ndataC = dataC.sample (n=1000)\n\n# check the shape of sample dataset\ndataC.shape","metadata":{"execution":{"iopub.status.busy":"2022-04-21T09:14:00.834473Z","iopub.execute_input":"2022-04-21T09:14:00.835334Z","iopub.status.idle":"2022-04-21T09:14:00.854734Z","shell.execute_reply.started":"2022-04-21T09:14:00.835287Z","shell.execute_reply":"2022-04-21T09:14:00.854151Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Create features**","metadata":{}},{"cell_type":"code","source":"pd.get_dummies (dataA)\ndataA.columns","metadata":{"execution":{"iopub.status.busy":"2022-04-21T09:14:40.212306Z","iopub.execute_input":"2022-04-21T09:14:40.212679Z","iopub.status.idle":"2022-04-21T09:14:40.223831Z","shell.execute_reply.started":"2022-04-21T09:14:40.212640Z","shell.execute_reply":"2022-04-21T09:14:40.223179Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.get_dummies (dataC)\ndataC.columns","metadata":{"execution":{"iopub.status.busy":"2022-04-21T09:15:26.433654Z","iopub.execute_input":"2022-04-21T09:15:26.433943Z","iopub.status.idle":"2022-04-21T09:15:26.449917Z","shell.execute_reply.started":"2022-04-21T09:15:26.433913Z","shell.execute_reply":"2022-04-21T09:15:26.449117Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.get_dummies (dataT)\ndataT.columns","metadata":{"execution":{"iopub.status.busy":"2022-04-21T09:15:27.998971Z","iopub.execute_input":"2022-04-21T09:15:27.999859Z","iopub.status.idle":"2022-04-21T09:15:28.017641Z","shell.execute_reply.started":"2022-04-21T09:15:27.999818Z","shell.execute_reply":"2022-04-21T09:15:28.016682Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Sample data for creating model**","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\ny = dataT ['price']\nX = dataC ['age']\n\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size = 0.3, random_state = 42)","metadata":{"execution":{"iopub.status.busy":"2022-04-21T09:17:19.636277Z","iopub.execute_input":"2022-04-21T09:17:19.636774Z","iopub.status.idle":"2022-04-21T09:17:19.644735Z","shell.execute_reply.started":"2022-04-21T09:17:19.636729Z","shell.execute_reply":"2022-04-21T09:17:19.643942Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Create K-fold cross-validation model**","metadata":{}},{"cell_type":"code","source":"# create a sample\nX = pd.concat ([X_train, X_test])\ny = pd.concat ([y_train, y_test])\n\n\n# import required libraries\n\nfrom sklearn import model_selection\nfrom sklearn.dummy import DummyClassifier\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.tree import DecisionTreeClassifier\nfrom sklearn.neighbors import KNeighborsClassifier\nfrom sklearn.naive_bayes import GaussianNB\nfrom sklearn.svm import SVC\nfrom sklearn.ensemble import RandomForestClassifier\nimport xgboost\n\nfor model in [\n    DummyClassifier,\n    LogisticRegression,\n    DecisionTreeClassifier,\n    KNeighborsClassifier,\n    GaussianNB,\n    SVC,\n    RandomForestClassifier,\n    xgboost.XGBClassifier,\n]:\n    cls = model ()\n    kfold = model_selection.KFold (n_splits = 10, random_state = None)\n    s = model_selection.cross_val_score (cls, X, y, scoring = \"roc_auc\", cv = kfold)\n    print (f\"{model.__name__:22} AUC: \"\n          f\"{s.mean():.3f} STD: {s.std():.2f}\")","metadata":{"execution":{"iopub.status.busy":"2022-04-21T09:17:21.358892Z","iopub.execute_input":"2022-04-21T09:17:21.361291Z"},"trusted":true},"execution_count":null,"outputs":[]}]}