{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# Main pycaret\n!pip install --ignore-installed pycaret","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","scrolled":true,"_kg_hide-output":true,"execution":{"iopub.status.busy":"2022-06-02T06:52:34.629741Z","iopub.execute_input":"2022-06-02T06:52:34.630705Z","iopub.status.idle":"2022-06-02T06:56:18.588600Z","shell.execute_reply.started":"2022-06-02T06:52:34.630602Z","shell.execute_reply":"2022-06-02T06:56:18.586879Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport gc\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"execution":{"iopub.status.busy":"2022-06-02T06:56:18.591702Z","iopub.execute_input":"2022-06-02T06:56:18.592823Z","iopub.status.idle":"2022-06-02T06:56:18.605759Z","shell.execute_reply.started":"2022-06-02T06:56:18.592772Z","shell.execute_reply":"2022-06-02T06:56:18.604769Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## The train and test aggregation files are created from these two notebooks.\n### [Notebook 1](https://www.kaggle.com/code/sravanneeli/train-file-pckl-creation-from-parquet-files)\n### [Notebook 2](https://www.kaggle.com/code/sravanneeli/test-file-pckl-creation-from-parquet-files)","metadata":{}},{"cell_type":"code","source":"np.random.seed(42)","metadata":{"execution":{"iopub.status.busy":"2022-06-02T06:56:18.607760Z","iopub.execute_input":"2022-06-02T06:56:18.608511Z","iopub.status.idle":"2022-06-02T06:56:18.627659Z","shell.execute_reply.started":"2022-06-02T06:56:18.608465Z","shell.execute_reply":"2022-06-02T06:56:18.626848Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.read_pickle('../input/amex-data-pckl-files/train_agg.pkl', compression='gzip')\ntrain_labels = pd.read_csv('../input/amex-default-prediction/train_labels.csv')","metadata":{"execution":{"iopub.status.busy":"2022-06-02T06:56:18.629802Z","iopub.execute_input":"2022-06-02T06:56:18.630295Z","iopub.status.idle":"2022-06-02T06:56:32.172604Z","shell.execute_reply.started":"2022-06-02T06:56:18.630258Z","shell.execute_reply":"2022-06-02T06:56:32.171620Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-06-02T06:56:32.173770Z","iopub.execute_input":"2022-06-02T06:56:32.174163Z","iopub.status.idle":"2022-06-02T06:56:32.382582Z","shell.execute_reply.started":"2022-06-02T06:56:32.174129Z","shell.execute_reply":"2022-06-02T06:56:32.381484Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cat_cols = []\nfor col in train_df:\n    if \"last\" in col:\n        cat_cols.append(col)","metadata":{"execution":{"iopub.status.busy":"2022-06-02T06:56:32.383974Z","iopub.execute_input":"2022-06-02T06:56:32.384353Z","iopub.status.idle":"2022-06-02T06:56:32.395009Z","shell.execute_reply.started":"2022-06-02T06:56:32.384318Z","shell.execute_reply":"2022-06-02T06:56:32.393979Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"while True:\n    try:\n        from pycaret.classification import *\n        break\n    except:\n        pass","metadata":{"execution":{"iopub.status.busy":"2022-06-02T06:56:32.396550Z","iopub.execute_input":"2022-06-02T06:56:32.397241Z","iopub.status.idle":"2022-06-02T06:56:35.953220Z","shell.execute_reply.started":"2022-06-02T06:56:32.397191Z","shell.execute_reply":"2022-06-02T06:56:35.951904Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_labels = train_labels.sort_values(by=['customer_ID']).drop('customer_ID', axis=1)\ntrain_df = pd.concat([train_df, train_labels], axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-06-02T06:56:35.955114Z","iopub.execute_input":"2022-06-02T06:56:35.955633Z","iopub.status.idle":"2022-06-02T06:56:36.756628Z","shell.execute_reply.started":"2022-06-02T06:56:35.955587Z","shell.execute_reply":"2022-06-02T06:56:36.755584Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-06-02T06:56:36.757776Z","iopub.execute_input":"2022-06-02T06:56:36.758111Z","iopub.status.idle":"2022-06-02T06:56:37.089281Z","shell.execute_reply.started":"2022-06-02T06:56:36.758081Z","shell.execute_reply":"2022-06-02T06:56:37.088273Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# COMPETITION METRIC FROM Konstantin Yakovlev\n# https://www.kaggle.com/kyakovlev\n# https://www.kaggle.com/competitions/amex-default-prediction/discussion/327534\ndef amex_metric_mod(y_true, y_pred):\n\n    labels     = np.transpose(np.array([y_true, y_pred]))\n    labels     = labels[labels[:, 1].argsort()[::-1]]\n    weights    = np.where(labels[:,0]==0, 20, 1)\n    cut_vals   = labels[np.cumsum(weights) <= int(0.04 * np.sum(weights))]\n    top_four   = np.sum(cut_vals[:,0]) / np.sum(labels[:,0])\n\n    gini = [0,0]\n    for i in [1,0]:\n        labels         = np.transpose(np.array([y_true, y_pred]))\n        labels         = labels[labels[:, i].argsort()[::-1]]\n        weight         = np.where(labels[:,0]==0, 20, 1)\n        weight_random  = np.cumsum(weight / np.sum(weight))\n        total_pos      = np.sum(labels[:, 0] *  weight)\n        cum_pos_found  = np.cumsum(labels[:, 0] * weight)\n        lorentz        = cum_pos_found / total_pos\n        gini[i]        = np.sum((lorentz - weight_random) * weight)\n\n    return 0.5 * (gini[1]/gini[0] + top_four)","metadata":{"execution":{"iopub.status.busy":"2022-06-02T06:56:37.091736Z","iopub.execute_input":"2022-06-02T06:56:37.092199Z","iopub.status.idle":"2022-06-02T06:56:37.101852Z","shell.execute_reply.started":"2022-06-02T06:56:37.092163Z","shell.execute_reply":"2022-06-02T06:56:37.100728Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"clf1 = setup(data = train_df.drop('customer_ID', axis=1),\n             session_id=2286,\n             fold=5,\n             categorical_features=cat_cols,\n             use_gpu=True,\n             target = 'target',\n             silent = True)","metadata":{"execution":{"iopub.status.busy":"2022-06-02T06:56:37.103427Z","iopub.execute_input":"2022-06-02T06:56:37.104158Z","iopub.status.idle":"2022-06-02T06:58:23.537243Z","shell.execute_reply.started":"2022-06-02T06:56:37.104111Z","shell.execute_reply":"2022-06-02T06:58:23.536173Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"add_metric(\"amex_customer_metric\", name=\"Amex Metric\", score_func=amex_metric_mod)","metadata":{"execution":{"iopub.status.busy":"2022-06-02T06:58:23.538731Z","iopub.execute_input":"2022-06-02T06:58:23.539349Z","iopub.status.idle":"2022-06-02T06:58:23.548263Z","shell.execute_reply.started":"2022-06-02T06:58:23.539310Z","shell.execute_reply":"2022-06-02T06:58:23.547276Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del(train_df)\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-06-02T06:58:23.549912Z","iopub.execute_input":"2022-06-02T06:58:23.550738Z","iopub.status.idle":"2022-06-02T06:58:23.746115Z","shell.execute_reply.started":"2022-06-02T06:58:23.550669Z","shell.execute_reply":"2022-06-02T06:58:23.744985Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# model training\nmodel = compare_models(include = ['lightgbm'])","metadata":{"execution":{"iopub.status.busy":"2022-06-02T06:58:23.747569Z","iopub.execute_input":"2022-06-02T06:58:23.747917Z","iopub.status.idle":"2022-06-02T07:06:15.972717Z","shell.execute_reply.started":"2022-06-02T06:58:23.747873Z","shell.execute_reply":"2022-06-02T07:06:15.971605Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-06-02T07:06:15.980319Z","iopub.execute_input":"2022-06-02T07:06:15.980716Z","iopub.status.idle":"2022-06-02T07:06:16.156268Z","shell.execute_reply.started":"2022-06-02T07:06:15.980683Z","shell.execute_reply":"2022-06-02T07:06:16.155037Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predict_model(model, raw_score=True)","metadata":{"execution":{"iopub.status.busy":"2022-06-02T07:06:16.157523Z","iopub.execute_input":"2022-06-02T07:06:16.158162Z","iopub.status.idle":"2022-06-02T07:06:23.906979Z","shell.execute_reply.started":"2022-06-02T07:06:16.158116Z","shell.execute_reply":"2022-06-02T07:06:23.905979Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df = pd.read_pickle('../input/amex-data-pckl-files/test_agg.pkl', compression='gzip')","metadata":{"execution":{"iopub.status.busy":"2022-06-02T07:06:23.908564Z","iopub.execute_input":"2022-06-02T07:06:23.909674Z","iopub.status.idle":"2022-06-02T07:06:33.984205Z","shell.execute_reply.started":"2022-06-02T07:06:23.909627Z","shell.execute_reply":"2022-06-02T07:06:33.982621Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred = []\nfor i in range(0, test_df.shape[0], 10000):\n    y_pred.extend(predict_model(model, data=test_df.iloc[i:i+10000, :],raw_score=True)['Score_1'].to_list())","metadata":{"execution":{"iopub.status.busy":"2022-06-02T07:06:33.985298Z","iopub.status.idle":"2022-06-02T07:06:33.985718Z","shell.execute_reply.started":"2022-06-02T07:06:33.985517Z","shell.execute_reply":"2022-06-02T07:06:33.985535Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub_df = pd.read_csv('../input/amex-default-prediction/sample_submission.csv')","metadata":{"execution":{"iopub.status.busy":"2022-06-02T07:06:33.987536Z","iopub.status.idle":"2022-06-02T07:06:33.988309Z","shell.execute_reply.started":"2022-06-02T07:06:33.988082Z","shell.execute_reply":"2022-06-02T07:06:33.988106Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub_df['prediction'] = y_pred","metadata":{"execution":{"iopub.status.busy":"2022-06-02T07:06:33.989494Z","iopub.status.idle":"2022-06-02T07:06:33.990011Z","shell.execute_reply.started":"2022-06-02T07:06:33.989797Z","shell.execute_reply":"2022-06-02T07:06:33.989817Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub_df.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-06-02T07:06:33.991317Z","iopub.status.idle":"2022-06-02T07:06:33.991750Z","shell.execute_reply.started":"2022-06-02T07:06:33.991556Z","shell.execute_reply":"2022-06-02T07:06:33.991577Z"},"trusted":true},"execution_count":null,"outputs":[]}]}