{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-16T18:39:52.205332Z","iopub.execute_input":"2022-08-16T18:39:52.206465Z","iopub.status.idle":"2022-08-16T18:39:52.217706Z","shell.execute_reply.started":"2022-08-16T18:39:52.206413Z","shell.execute_reply":"2022-08-16T18:39:52.216502Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Baseline\n\n1. Based a lot of work here off the top solutions. Also thank you https://www.kaggle.com/datasets/munumbutt/amexfeather for the feather dataset\n2. Approach - loaded data, converted to numerics, train-test split, xgb","metadata":{}},{"cell_type":"code","source":"# Load Data\nimport pandas as pd, numpy as np # CPU libraries\nimport gc\n\n\ntrain = pd.read_feather('../input/amexfeather/train_data.ftr')\nprint(train.shape)\n#Reduce Train\npartTrain = train.iloc[1:5531451, :]\nprint(partTrain.shape)\n\ndel train\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-08-16T18:39:52.267825Z","iopub.execute_input":"2022-08-16T18:39:52.268232Z","iopub.status.idle":"2022-08-16T18:39:56.834918Z","shell.execute_reply.started":"2022-08-16T18:39:52.268198Z","shell.execute_reply":"2022-08-16T18:39:56.833635Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#List non-numeric columns\nallCols = partTrain.columns.tolist()\nnumericCols = partTrain.select_dtypes(include=np.number).columns.tolist()\n\nprint(\"all\", len(allCols))\nprint(\"numeric\", len(numericCols))\n\nnonNumeric = list(set(allCols) - set(numericCols))\nprint(\"non numeric\", len(nonNumeric))\n# converting pandas \"categorical\" dtype to numeric\npartTrain[nonNumeric] = partTrain[nonNumeric].apply(pd.to_numeric, errors='coerce')\n\n\ndel allCols, numericCols\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-08-16T18:39:56.836682Z","iopub.execute_input":"2022-08-16T18:39:56.837027Z","iopub.status.idle":"2022-08-16T18:40:26.963820Z","shell.execute_reply.started":"2022-08-16T18:39:56.836996Z","shell.execute_reply":"2022-08-16T18:40:26.962535Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#train_test split\nX = partTrain.drop(columns=[\"target\"],axis=1)\ny = partTrain[\"target\"]\n\ndel partTrain\n \nfrom sklearn.model_selection import train_test_split\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size = 0.33,random_state=100)\n\ndel X\ndel y\n\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-08-16T18:40:26.964961Z","iopub.execute_input":"2022-08-16T18:40:26.965304Z","iopub.status.idle":"2022-08-16T18:41:02.249855Z","shell.execute_reply.started":"2022-08-16T18:40:26.965275Z","shell.execute_reply":"2022-08-16T18:41:02.248582Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#implement XGB baseline\n\nfrom xgboost import XGBClassifier\n\n# Create as many classifiers as there are clusters\nxgb_classifier0 = XGBClassifier(objective='binary:logistic', \n                      n_estimators=200,\n                      eta=0.2,\n                      seed=12,\n                      learning_rate=0.02,\n                      use_label_encoder=False,\n                      eval_metric='aucpr',                      \n                    )\n\n\nprint(\"Print Xtrain\", X_train.shape)\nprint(\"Print y_train\", y_train.shape)\nxgb_classifier0.fit(X_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2022-08-16T18:41:02.251986Z","iopub.execute_input":"2022-08-16T18:41:02.252349Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#generate output\n\ngc.collect()\n\nimport pandas as pd # Remove when running from the beginning\npartTest = pd.read_feather('../input/amexfeather/test_data.ftr')\npartTest = partTest.drop_duplicates(['customer_ID'])\n\npartTest[nonNumeric] = partTest[nonNumeric].apply(pd.to_numeric, errors='coerce')\n\ny_pred_prob = xgb_classifier0.predict_proba(partTest)[:,1]\nprint(y_pred_prob.shape)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Create Submission\nsubmission = pd.read_feather('../input/amexfeather/test_data.ftr')\nsubmission = submission.drop_duplicates(['customer_ID'])\nsubmission[\"prediction\"] = y_pred_prob\nsubmission = submission[['customer_ID', \"prediction\"]]\nprint(submission.head, submission.shape)\nsubmission.to_csv('submission.csv',index=False)\nprint('Submission file shape is', submission.shape )","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}