{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-12-31T14:29:12.638764Z","iopub.execute_input":"2022-12-31T14:29:12.639640Z","iopub.status.idle":"2022-12-31T14:29:12.670297Z","shell.execute_reply.started":"2022-12-31T14:29:12.639552Z","shell.execute_reply":"2022-12-31T14:29:12.669425Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install pyforest\nfrom pyforest import*","metadata":{"execution":{"iopub.status.busy":"2022-12-31T14:29:12.672210Z","iopub.execute_input":"2022-12-31T14:29:12.672554Z","iopub.status.idle":"2022-12-31T14:29:28.744270Z","shell.execute_reply.started":"2022-12-31T14:29:12.672517Z","shell.execute_reply":"2022-12-31T14:29:28.743211Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Importing datasets\n","metadata":{}},{"cell_type":"code","source":"test_data=pd.read_csv(\"/kaggle/input/amex-default-prediction/train_data.csv\",nrows=100000)\ntrain_data= pd.read_csv(\"/kaggle/input/amex-default-prediction/train_data.csv\",nrows=100000)\ntrain_labels= pd.read_csv(\"/kaggle/input/amex-default-prediction/train_labels.csv\",nrows=100000)","metadata":{"execution":{"iopub.status.busy":"2022-12-31T14:29:28.745958Z","iopub.execute_input":"2022-12-31T14:29:28.746302Z","iopub.status.idle":"2022-12-31T14:29:41.831572Z","shell.execute_reply.started":"2022-12-31T14:29:28.746272Z","shell.execute_reply":"2022-12-31T14:29:41.830591Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_labels[\"target\"].value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-12-31T14:29:41.834229Z","iopub.execute_input":"2022-12-31T14:29:41.834600Z","iopub.status.idle":"2022-12-31T14:29:41.852046Z","shell.execute_reply.started":"2022-12-31T14:29:41.834549Z","shell.execute_reply":"2022-12-31T14:29:41.850871Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.head()","metadata":{"execution":{"iopub.status.busy":"2022-12-31T14:29:41.853988Z","iopub.execute_input":"2022-12-31T14:29:41.854559Z","iopub.status.idle":"2022-12-31T14:29:41.888025Z","shell.execute_reply.started":"2022-12-31T14:29:41.854518Z","shell.execute_reply":"2022-12-31T14:29:41.886933Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.shape","metadata":{"execution":{"iopub.status.busy":"2022-12-31T14:29:41.889688Z","iopub.execute_input":"2022-12-31T14:29:41.890084Z","iopub.status.idle":"2022-12-31T14:29:41.897637Z","shell.execute_reply.started":"2022-12-31T14:29:41.890034Z","shell.execute_reply":"2022-12-31T14:29:41.896630Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"count=0\nfor col in train_data.columns:\n    if train_data[col].dtypes==\"object\":\n        count+=1\n        print(col,count)\n        \n        ","metadata":{"execution":{"iopub.status.busy":"2022-12-31T14:29:41.899151Z","iopub.execute_input":"2022-12-31T14:29:41.900116Z","iopub.status.idle":"2022-12-31T14:29:41.918450Z","shell.execute_reply.started":"2022-12-31T14:29:41.900049Z","shell.execute_reply":"2022-12-31T14:29:41.917366Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nobject_data= train_data[[\"customer_ID\",\"S_2\",\"D_63\",\"D_64\"]]\nobject_data.head()","metadata":{"execution":{"iopub.status.busy":"2022-12-31T14:29:41.920769Z","iopub.execute_input":"2022-12-31T14:29:41.921374Z","iopub.status.idle":"2022-12-31T14:29:41.937785Z","shell.execute_reply.started":"2022-12-31T14:29:41.921341Z","shell.execute_reply":"2022-12-31T14:29:41.936915Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.info()","metadata":{"execution":{"iopub.status.busy":"2022-12-31T14:29:41.941304Z","iopub.execute_input":"2022-12-31T14:29:41.941739Z","iopub.status.idle":"2022-12-31T14:29:41.959976Z","shell.execute_reply.started":"2022-12-31T14:29:41.941687Z","shell.execute_reply":"2022-12-31T14:29:41.958879Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df= pd.merge(train_data, train_labels, how=\"inner\", on=[\"customer_ID\"])\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2022-12-31T14:29:41.961659Z","iopub.execute_input":"2022-12-31T14:29:41.962922Z","iopub.status.idle":"2022-12-31T14:29:42.355822Z","shell.execute_reply.started":"2022-12-31T14:29:41.962878Z","shell.execute_reply":"2022-12-31T14:29:42.354657Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.tail()","metadata":{"execution":{"iopub.status.busy":"2022-12-31T14:29:42.357167Z","iopub.execute_input":"2022-12-31T14:29:42.357892Z","iopub.status.idle":"2022-12-31T14:29:42.385774Z","shell.execute_reply.started":"2022-12-31T14:29:42.357849Z","shell.execute_reply":"2022-12-31T14:29:42.384839Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.isna().sum()","metadata":{"execution":{"iopub.status.busy":"2022-12-31T14:29:42.387378Z","iopub.execute_input":"2022-12-31T14:29:42.387807Z","iopub.status.idle":"2022-12-31T14:29:42.444115Z","shell.execute_reply.started":"2022-12-31T14:29:42.387769Z","shell.execute_reply":"2022-12-31T14:29:42.443122Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"droping feautures with missing values >=80% in the train dataframe","metadata":{}},{"cell_type":"code","source":"i =0\nfor col in df.columns:\n    if (df[col].isna().sum()/len(df[col])*100) >= 80:\n        df.drop(col, axis=1,inplace=True)\n        i+=1\n    \n    #print(\"Total features dropped:\"i)","metadata":{"execution":{"iopub.status.busy":"2022-12-31T14:29:42.445483Z","iopub.execute_input":"2022-12-31T14:29:42.445906Z","iopub.status.idle":"2022-12-31T14:29:43.717366Z","shell.execute_reply.started":"2022-12-31T14:29:42.445863Z","shell.execute_reply":"2022-12-31T14:29:43.716276Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.isna().sum()","metadata":{"execution":{"iopub.status.busy":"2022-12-31T14:29:43.719277Z","iopub.execute_input":"2022-12-31T14:29:43.719664Z","iopub.status.idle":"2022-12-31T14:29:43.774263Z","shell.execute_reply.started":"2022-12-31T14:29:43.719626Z","shell.execute_reply":"2022-12-31T14:29:43.773254Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.describe().T","metadata":{"execution":{"iopub.status.busy":"2022-12-31T14:29:43.776127Z","iopub.execute_input":"2022-12-31T14:29:43.776509Z","iopub.status.idle":"2022-12-31T14:29:44.881725Z","shell.execute_reply.started":"2022-12-31T14:29:43.776473Z","shell.execute_reply":"2022-12-31T14:29:44.880771Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"replacing NAN with the mean and processing data","metadata":{}},{"cell_type":"code","source":"def pre_process(df):\n   \n    data=df\n        \n    data['Year']=data['S_2'].apply(lambda x:np.int(x[0:4]))\n    data['Month']=data['S_2'].apply(lambda x:np.int(x[5:7]))\n    data['Day']=data['S_2'].apply(lambda x:np.int(x[8:10]))\n    data=data.drop('S_2', axis=1)\n    \n    \n    data_na=data\n    data_na= data_na.fillna(0)\n    data2=data_na.drop('customer_ID', axis=1)\n    \n\n    \n   \n  \n    categorical_cols=['D_63','D_64','D_68','B_30','B_38','D_114','D_116','D_117','D_120','D_126']\n    dummies=pd.get_dummies(data2, columns=categorical_cols)\n\n    #the above transformed data is an array so convert it to a dataframe \n    dummy_data=pd.DataFrame(dummies, index= data2.index)\n\n    #now concatenate the original data and the dummified data using pandas\n    concatenated_data= pd.concat([data2, dummy_data], axis=1)\n    \n    data3= concatenated_data.drop(['D_63','D_64','D_68','B_30','B_38','D_114','D_116','D_117','D_120','D_126'], axis=1)\n    \n    data3= data3.astype(float)\n    data3=data3.astype(int)\n    \n    data3=data3.loc[:, ~data3.T.duplicated(keep='first')]\n    \n    return data3","metadata":{"execution":{"iopub.status.busy":"2022-12-31T14:29:44.883353Z","iopub.execute_input":"2022-12-31T14:29:44.883749Z","iopub.status.idle":"2022-12-31T14:29:44.896099Z","shell.execute_reply.started":"2022-12-31T14:29:44.883696Z","shell.execute_reply":"2022-12-31T14:29:44.894693Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"##currently taking a bit more time to run on the kaggle servers \n\n#pros_train= pre_process(train_data)","metadata":{"execution":{"iopub.status.busy":"2022-12-31T14:29:44.898183Z","iopub.execute_input":"2022-12-31T14:29:44.898637Z","iopub.status.idle":"2022-12-31T14:29:44.908553Z","shell.execute_reply.started":"2022-12-31T14:29:44.898603Z","shell.execute_reply":"2022-12-31T14:29:44.907523Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data=train_data\n        \ndata['Year']=data['S_2'].apply(lambda x:np.int(x[0:4]))\ndata['Month']=data['S_2'].apply(lambda x:np.int(x[5:7]))\ndata['Day']=data['S_2'].apply(lambda x:np.int(x[8:10]))\ndata=data.drop('S_2', axis=1)\n    \n    \ndata_na=data\ndata_na= data_na.fillna(0)\ndata2=data_na.drop('customer_ID', axis=1)\n    \n\n    \n   \n  \ncategorical_cols=['D_63','D_64','D_68','B_30','B_38','D_114','D_116','D_117','D_120','D_126']\ndummies=pd.get_dummies(data2, columns=categorical_cols)\n\n#the above transformed data is an array so convert it to a dataframe \ndummy_data=pd.DataFrame(dummies, index= data2.index)\n\n#now concatenate the original data and the dummified data using pandas\nconcatenated_data= pd.concat([data2, dummy_data], axis=1)\n    \ndata3= concatenated_data.drop(['D_63','D_64','D_68','B_30','B_38','D_114','D_116','D_117','D_120','D_126'], axis=1)\n    \ndata3= data3.astype(float)\ndata3=data3.astype(int)\n    \ndata3=data3.loc[:, ~data3.T.duplicated(keep='first')]","metadata":{"execution":{"iopub.status.busy":"2022-12-31T14:29:44.910574Z","iopub.execute_input":"2022-12-31T14:29:44.911033Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}