{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport matplotlib.pyplot as plt, gc, os\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-01-12T00:33:00.461929Z","iopub.execute_input":"2023-01-12T00:33:00.462498Z","iopub.status.idle":"2023-01-12T00:33:00.474980Z","shell.execute_reply.started":"2023-01-12T00:33:00.462458Z","shell.execute_reply":"2023-01-12T00:33:00.474371Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport os\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport numpy as np\nimport pandas as pd\nimport warnings\nwarnings.filterwarnings('ignore')","metadata":{"execution":{"iopub.status.busy":"2023-01-12T00:33:00.476343Z","iopub.execute_input":"2023-01-12T00:33:00.476561Z","iopub.status.idle":"2023-01-12T00:33:00.614832Z","shell.execute_reply.started":"2023-01-12T00:33:00.476531Z","shell.execute_reply":"2023-01-12T00:33:00.614215Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_parquet(\"/kaggle/input/amex-data-integer-dtypes-parquet-format/train.parquet\")\ntest = pd.read_parquet(\"/kaggle/input/amex-data-integer-dtypes-parquet-format/test.parquet\")\ntrain_labels = pd.read_csv(\"../input/amex-default-prediction/train_labels.csv\")\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2023-01-12T00:33:00.615731Z","iopub.execute_input":"2023-01-12T00:33:00.615926Z","iopub.status.idle":"2023-01-12T00:33:47.467597Z","shell.execute_reply.started":"2023-01-12T00:33:00.615902Z","shell.execute_reply":"2023-01-12T00:33:47.466380Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#shapes\ntrain.shape, test.shape, train_labels.shape","metadata":{"execution":{"iopub.status.busy":"2023-01-12T00:33:47.469371Z","iopub.execute_input":"2023-01-12T00:33:47.469770Z","iopub.status.idle":"2023-01-12T00:33:47.477838Z","shell.execute_reply.started":"2023-01-12T00:33:47.469722Z","shell.execute_reply":"2023-01-12T00:33:47.476640Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#count NaN or missing values in the DataFrame\nprint('/nCount total NaN at each column in a DataFrame: /n/n',train.isnull().sum())","metadata":{"execution":{"iopub.status.busy":"2023-01-12T00:33:47.481154Z","iopub.execute_input":"2023-01-12T00:33:47.482073Z","iopub.status.idle":"2023-01-12T00:33:50.268323Z","shell.execute_reply.started":"2023-01-12T00:33:47.482008Z","shell.execute_reply":"2023-01-12T00:33:50.267493Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Columns that have more than 50% of missing values\ncolumns=train.columns[train.isna().sum()/len(train)*100>50]\ncolumns","metadata":{"execution":{"iopub.status.busy":"2023-01-12T00:33:50.269404Z","iopub.execute_input":"2023-01-12T00:33:50.269633Z","iopub.status.idle":"2023-01-12T00:33:52.985345Z","shell.execute_reply.started":"2023-01-12T00:33:50.269598Z","shell.execute_reply":"2023-01-12T00:33:52.984645Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Drop the columns which have more than 50% of missing values.\ntrain = train.drop(columns, axis=1)\ntest = test.drop(columns, axis=1)","metadata":{"execution":{"iopub.status.busy":"2023-01-12T00:33:52.986400Z","iopub.execute_input":"2023-01-12T00:33:52.986618Z","iopub.status.idle":"2023-01-12T00:33:56.241699Z","shell.execute_reply.started":"2023-01-12T00:33:52.986591Z","shell.execute_reply":"2023-01-12T00:33:56.240816Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#lower than 50% but >0\nna_rate_s = train.isna().sum()/len(train)*100\nna_columns= list(na_rate_s [na_rate_s > 0].index)\nna_columns","metadata":{"execution":{"iopub.status.busy":"2023-01-12T00:33:56.242959Z","iopub.execute_input":"2023-01-12T00:33:56.243218Z","iopub.status.idle":"2023-01-12T00:33:58.759460Z","shell.execute_reply.started":"2023-01-12T00:33:56.243186Z","shell.execute_reply":"2023-01-12T00:33:58.758727Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#for the lower than 50% but >0 of missing values, we fill with median values.\nfor column in na_columns:\n    fill_value = train[column].mean()\n    train[column] = train[column].fillna(fill_value)\ntrain.isna().sum()","metadata":{"execution":{"iopub.status.busy":"2023-01-12T00:33:58.760576Z","iopub.execute_input":"2023-01-12T00:33:58.760811Z","iopub.status.idle":"2023-01-12T00:34:02.717142Z","shell.execute_reply.started":"2023-01-12T00:33:58.760784Z","shell.execute_reply":"2023-01-12T00:34:02.716409Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"na_rate_s = test.isna().sum()/len(test)*100\nna_columns= list(na_rate_s [na_rate_s > 0].index)\nna_columns","metadata":{"execution":{"iopub.status.busy":"2023-01-12T00:34:02.718240Z","iopub.execute_input":"2023-01-12T00:34:02.718469Z","iopub.status.idle":"2023-01-12T00:34:07.876927Z","shell.execute_reply.started":"2023-01-12T00:34:02.718442Z","shell.execute_reply":"2023-01-12T00:34:07.876017Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for column in na_columns:\n    fill_value = test[column].mean()\n    test[column] = test[column].fillna(fill_value)\ntest.isna().sum()","metadata":{"execution":{"iopub.status.busy":"2023-01-12T00:34:07.878294Z","iopub.execute_input":"2023-01-12T00:34:07.878595Z","iopub.status.idle":"2023-01-12T00:34:16.465196Z","shell.execute_reply.started":"2023-01-12T00:34:07.878558Z","shell.execute_reply":"2023-01-12T00:34:16.464369Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.reset_index(inplace=True)\ntest.reset_index(inplace=True)","metadata":{"execution":{"iopub.status.busy":"2023-01-12T00:34:16.466339Z","iopub.execute_input":"2023-01-12T00:34:16.466558Z","iopub.status.idle":"2023-01-12T00:34:16.553393Z","shell.execute_reply.started":"2023-01-12T00:34:16.466533Z","shell.execute_reply":"2023-01-12T00:34:16.552569Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.reset_index(drop=True, inplace=True)\ntest.reset_index(drop=True, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2023-01-12T00:34:16.554615Z","iopub.execute_input":"2023-01-12T00:34:16.554875Z","iopub.status.idle":"2023-01-12T00:34:16.559227Z","shell.execute_reply.started":"2023-01-12T00:34:16.554843Z","shell.execute_reply":"2023-01-12T00:34:16.558514Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#shape\ntrain.shape, train_labels.shape, test.shape","metadata":{"execution":{"iopub.status.busy":"2023-01-12T00:34:16.562440Z","iopub.execute_input":"2023-01-12T00:34:16.562652Z","iopub.status.idle":"2023-01-12T00:34:16.570329Z","shell.execute_reply.started":"2023-01-12T00:34:16.562621Z","shell.execute_reply":"2023-01-12T00:34:16.569674Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import calendar ","metadata":{"execution":{"iopub.status.busy":"2023-01-12T00:34:16.571206Z","iopub.execute_input":"2023-01-12T00:34:16.571401Z","iopub.status.idle":"2023-01-12T00:34:16.578793Z","shell.execute_reply.started":"2023-01-12T00:34:16.571376Z","shell.execute_reply":"2023-01-12T00:34:16.578090Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import datetime \ndef get_last_day(cur_date):\n    year = cur_date.year\n    month = cur_date.month\n    last_day = calendar.monthrange(year,month)[1]\n    return datetime.date(year,month,last_day)","metadata":{"execution":{"iopub.status.busy":"2023-01-12T00:34:16.579720Z","iopub.execute_input":"2023-01-12T00:34:16.579927Z","iopub.status.idle":"2023-01-12T00:34:16.587857Z","shell.execute_reply.started":"2023-01-12T00:34:16.579901Z","shell.execute_reply":"2023-01-12T00:34:16.587210Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#change the column to datetime data type. \n#the original S_2 looks like datetime but actually numbers.\n\ntrain['S_2'] =pd.to_datetime(train['S_2'])","metadata":{"execution":{"iopub.status.busy":"2023-01-12T00:34:16.588734Z","iopub.execute_input":"2023-01-12T00:34:16.588922Z","iopub.status.idle":"2023-01-12T00:34:17.805515Z","shell.execute_reply.started":"2023-01-12T00:34:16.588899Z","shell.execute_reply":"2023-01-12T00:34:17.804548Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#to show the datetime of each month_last_date\ntrain['month_last_date'] = train['S_2'].apply(get_last_day)\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2023-01-12T00:34:17.806762Z","iopub.execute_input":"2023-01-12T00:34:17.806998Z","iopub.status.idle":"2023-01-12T00:35:03.639605Z","shell.execute_reply.started":"2023-01-12T00:34:17.806966Z","shell.execute_reply":"2023-01-12T00:35:03.638808Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#calculate the number of customer id in each datetime.\ncustomer_last_date_series=train.groupby(by=['customer_ID','month_last_date'])['customer_ID'].count()\n\n#to check if the number of the same customer id exceeds 1. 1 indicates no duplicate customer id each datetime.\ncustomer_last_date_series.sort_values(ascending=False)","metadata":{"execution":{"iopub.status.busy":"2023-01-12T00:35:03.640786Z","iopub.execute_input":"2023-01-12T00:35:03.640990Z","iopub.status.idle":"2023-01-12T00:35:08.369247Z","shell.execute_reply.started":"2023-01-12T00:35:03.640965Z","shell.execute_reply":"2023-01-12T00:35:08.368340Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#change the data type from series to dateframe using reset_index()\nmonth_customer_df = train.groupby(by='month_last_date')['customer_ID'].count().reset_index()\n\n#change the column name from customer_ID to customer_number\nmonth_customer_df.columns = ['month_last_date','customer_number']\nmonth_customer_df","metadata":{"execution":{"iopub.status.busy":"2023-01-12T00:35:08.370506Z","iopub.execute_input":"2023-01-12T00:35:08.370736Z","iopub.status.idle":"2023-01-12T00:35:10.108314Z","shell.execute_reply.started":"2023-01-12T00:35:08.370708Z","shell.execute_reply":"2023-01-12T00:35:10.107574Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#merge train and train_labels which shows the target(1 is default,0 is not default)\ntrain = pd.merge(left = train,right=train_labels,how='inner',on='customer_ID')\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2023-01-12T00:35:10.109426Z","iopub.execute_input":"2023-01-12T00:35:10.109674Z","iopub.status.idle":"2023-01-12T00:36:59.058117Z","shell.execute_reply.started":"2023-01-12T00:35:10.109636Z","shell.execute_reply":"2023-01-12T00:36:59.057461Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_labels['target'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-01-12T00:36:59.059277Z","iopub.execute_input":"2023-01-12T00:36:59.059811Z","iopub.status.idle":"2023-01-12T00:36:59.069280Z","shell.execute_reply.started":"2023-01-12T00:36:59.059774Z","shell.execute_reply":"2023-01-12T00:36:59.068656Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#calculate the total number of default in each datetime. \nmonth_default_df = train[train['target']==1].groupby(by='month_last_date')['customer_ID'].count().reset_index()\nmonth_default_df.columns = ['month_last_date','default_number']\nmonth_default_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-01-12T00:36:59.070189Z","iopub.execute_input":"2023-01-12T00:36:59.070659Z","iopub.status.idle":"2023-01-12T00:37:02.242127Z","shell.execute_reply.started":"2023-01-12T00:36:59.070628Z","shell.execute_reply":"2023-01-12T00:37:02.241552Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#dateframe: number of customer and number of default in each datetime. \ncounter_df=pd.merge(left=month_customer_df,right=month_default_df,how ='outer',on='month_last_date')\ncounter_df","metadata":{"execution":{"iopub.status.busy":"2023-01-12T00:37:02.243038Z","iopub.execute_input":"2023-01-12T00:37:02.243241Z","iopub.status.idle":"2023-01-12T00:37:02.254958Z","shell.execute_reply.started":"2023-01-12T00:37:02.243216Z","shell.execute_reply":"2023-01-12T00:37:02.254442Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#calculate the default rate\ncounter_df['pd_rate'] = counter_df['default_number'] / counter_df['customer_number']\ncounter_df","metadata":{"execution":{"iopub.status.busy":"2023-01-12T00:37:02.255838Z","iopub.execute_input":"2023-01-12T00:37:02.256036Z","iopub.status.idle":"2023-01-12T00:37:02.272864Z","shell.execute_reply.started":"2023-01-12T00:37:02.256011Z","shell.execute_reply":"2023-01-12T00:37:02.272202Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize = (10,8))\nplt.plot(counter_df['month_last_date'],counter_df['pd_rate'])\nplt.xlabel('month last date')\nplt.ylabel('default rate')\nplt.title('Amex default prediction')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-01-12T00:37:02.273712Z","iopub.execute_input":"2023-01-12T00:37:02.273912Z","iopub.status.idle":"2023-01-12T00:37:02.491164Z","shell.execute_reply.started":"2023-01-12T00:37:02.273887Z","shell.execute_reply":"2023-01-12T00:37:02.490540Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head()","metadata":{"execution":{"iopub.status.busy":"2023-01-12T00:37:02.491989Z","iopub.execute_input":"2023-01-12T00:37:02.492176Z","iopub.status.idle":"2023-01-12T00:37:02.517566Z","shell.execute_reply.started":"2023-01-12T00:37:02.492152Z","shell.execute_reply":"2023-01-12T00:37:02.517001Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# select features\n# x is the selected features after dropping unrelated features. y is the predicted target.\nX = train.drop(['index','customer_ID','S_2','target','month_last_date'],axis=1)\ny = train['target']\nX.head()","metadata":{"execution":{"iopub.status.busy":"2023-01-12T00:37:02.518407Z","iopub.execute_input":"2023-01-12T00:37:02.518606Z","iopub.status.idle":"2023-01-12T00:37:03.423634Z","shell.execute_reply.started":"2023-01-12T00:37:02.518583Z","shell.execute_reply":"2023-01-12T00:37:03.422951Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#build up the logistic regression model with data x and y. \nfrom sklearn.linear_model import LogisticRegression\nclf = LogisticRegression(random_state=0,solver='sag').fit(X,y)","metadata":{"execution":{"iopub.status.busy":"2023-01-12T00:37:03.424650Z","iopub.execute_input":"2023-01-12T00:37:03.425069Z","iopub.status.idle":"2023-01-12T00:42:13.955226Z","shell.execute_reply.started":"2023-01-12T00:37:03.425035Z","shell.execute_reply":"2023-01-12T00:42:13.954242Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#accuracy of the model in train data，actual value y(target). \nclf.score(X,y)","metadata":{"execution":{"iopub.status.busy":"2023-01-12T00:42:13.956603Z","iopub.execute_input":"2023-01-12T00:42:13.956895Z","iopub.status.idle":"2023-01-12T00:42:16.372363Z","shell.execute_reply.started":"2023-01-12T00:42:13.956863Z","shell.execute_reply":"2023-01-12T00:42:16.371583Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#predict value y(target_predict) of train data\ny_pred_train=clf.predict(X)\ny_pred_train","metadata":{"execution":{"iopub.status.busy":"2023-01-12T00:42:16.373696Z","iopub.execute_input":"2023-01-12T00:42:16.373936Z","iopub.status.idle":"2023-01-12T00:42:18.338631Z","shell.execute_reply.started":"2023-01-12T00:42:16.373906Z","shell.execute_reply":"2023-01-12T00:42:18.337637Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#add the column of target_predict to train data. \ntrain_copy=train.copy()\ntrain_copy['target_predict']=y_pred_train\ntrain_copy.head()","metadata":{"execution":{"iopub.status.busy":"2023-01-12T00:42:18.340677Z","iopub.execute_input":"2023-01-12T00:42:18.341319Z","iopub.status.idle":"2023-01-12T00:42:19.467014Z","shell.execute_reply.started":"2023-01-12T00:42:18.341268Z","shell.execute_reply":"2023-01-12T00:42:19.466164Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#calculate the total number of predicted default in each datetime. \nmonth_default_df_copy = train_copy[train_copy['target_predict']==1].groupby(by='month_last_date')['customer_ID'].count().reset_index()\nmonth_default_df_copy.columns = ['month_last_date','predicted_default_number']\nmonth_default_df_copy.head()","metadata":{"execution":{"iopub.status.busy":"2023-01-12T00:42:19.468200Z","iopub.execute_input":"2023-01-12T00:42:19.468459Z","iopub.status.idle":"2023-01-12T00:42:20.702518Z","shell.execute_reply.started":"2023-01-12T00:42:19.468411Z","shell.execute_reply":"2023-01-12T00:42:20.701748Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"counter_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-01-12T00:42:20.703625Z","iopub.execute_input":"2023-01-12T00:42:20.703840Z","iopub.status.idle":"2023-01-12T00:42:20.713718Z","shell.execute_reply.started":"2023-01-12T00:42:20.703812Z","shell.execute_reply":"2023-01-12T00:42:20.712830Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"counter_df=pd.merge(left=counter_df,right=month_default_df_copy,how='outer',on='month_last_date')\ncounter_df","metadata":{"execution":{"iopub.status.busy":"2023-01-12T00:42:20.714886Z","iopub.execute_input":"2023-01-12T00:42:20.715308Z","iopub.status.idle":"2023-01-12T00:42:20.736158Z","shell.execute_reply.started":"2023-01-12T00:42:20.715272Z","shell.execute_reply":"2023-01-12T00:42:20.735469Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"counter_df['predicted_default_rate']=counter_df['predicted_default_number']/counter_df['customer_number']\ncounter_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-01-12T00:42:20.737304Z","iopub.execute_input":"2023-01-12T00:42:20.737543Z","iopub.status.idle":"2023-01-12T00:42:20.755521Z","shell.execute_reply.started":"2023-01-12T00:42:20.737514Z","shell.execute_reply":"2023-01-12T00:42:20.754661Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#y is actual value, y_pred_train is predcited value\nfrom sklearn.metrics import classification_report\nprint(classification_report(y,y_pred_train))","metadata":{"execution":{"iopub.status.busy":"2023-01-12T00:42:20.756549Z","iopub.execute_input":"2023-01-12T00:42:20.756762Z","iopub.status.idle":"2023-01-12T00:42:29.437395Z","shell.execute_reply.started":"2023-01-12T00:42:20.756734Z","shell.execute_reply":"2023-01-12T00:42:29.436686Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"counter_df","metadata":{"execution":{"iopub.status.busy":"2023-01-12T00:42:29.438538Z","iopub.execute_input":"2023-01-12T00:42:29.439078Z","iopub.status.idle":"2023-01-12T00:42:29.451431Z","shell.execute_reply.started":"2023-01-12T00:42:29.439042Z","shell.execute_reply":"2023-01-12T00:42:29.450841Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize = (10,8))\nplt.plot(counter_df['month_last_date'],counter_df['pd_rate'],label='actual default rate')\nplt.plot(counter_df['month_last_date'],counter_df['predicted_default_rate'],label='predicted default rate')\nplt.legend()#show the labels\nplt.xlabel('month last date')\nplt.ylabel('default rate')\nplt.title('Amex default prediction')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-01-12T00:42:29.452352Z","iopub.execute_input":"2023-01-12T00:42:29.452664Z","iopub.status.idle":"2023-01-12T00:42:29.677378Z","shell.execute_reply.started":"2023-01-12T00:42:29.452627Z","shell.execute_reply":"2023-01-12T00:42:29.676657Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#the importances of features in train data. The coefficient is the importances related to each feature.\n#abs represnets the absolute value. \nfeature_importances = abs(clf.coef_[0])\nfeature_importances","metadata":{"execution":{"iopub.status.busy":"2023-01-12T00:42:29.678431Z","iopub.execute_input":"2023-01-12T00:42:29.678687Z","iopub.status.idle":"2023-01-12T00:42:29.686031Z","shell.execute_reply.started":"2023-01-12T00:42:29.678656Z","shell.execute_reply":"2023-01-12T00:42:29.685515Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#dataframe built up with features and importances. \nfeature_importance_df= pd.DataFrame(\n    {\n        \"feature\":X.columns,\"importance\":feature_importances\n    }\n)\nfeature_importance_df","metadata":{"execution":{"iopub.status.busy":"2023-01-12T00:42:29.690198Z","iopub.execute_input":"2023-01-12T00:42:29.690392Z","iopub.status.idle":"2023-01-12T00:42:29.703799Z","shell.execute_reply.started":"2023-01-12T00:42:29.690368Z","shell.execute_reply":"2023-01-12T00:42:29.703176Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#sort the dataframe by descending of importances. \nfeature_importance_df = feature_importance_df.sort_values(by='importance',ascending = False)\ntop10_feature_importance_df=feature_importance_df.head(10)\ntop10_feature_importance_df","metadata":{"execution":{"iopub.status.busy":"2023-01-12T00:42:29.704701Z","iopub.execute_input":"2023-01-12T00:42:29.704914Z","iopub.status.idle":"2023-01-12T00:42:29.718907Z","shell.execute_reply.started":"2023-01-12T00:42:29.704889Z","shell.execute_reply":"2023-01-12T00:42:29.718287Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#plot the dataframe \nimport matplotlib.pyplot as plt\n\nplt.bar(top10_feature_importance_df['feature'],top10_feature_importance_df['importance'])\nplt.xlabel('feature name')\nplt.ylabel('feature importance')\nplt.title(\"feature importance bar\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-01-12T00:42:29.719804Z","iopub.execute_input":"2023-01-12T00:42:29.720013Z","iopub.status.idle":"2023-01-12T00:42:29.883884Z","shell.execute_reply.started":"2023-01-12T00:42:29.719986Z","shell.execute_reply":"2023-01-12T00:42:29.883310Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#test data\ntest.head()","metadata":{"execution":{"iopub.status.busy":"2023-01-12T00:42:29.884766Z","iopub.execute_input":"2023-01-12T00:42:29.884962Z","iopub.status.idle":"2023-01-12T00:42:29.909136Z","shell.execute_reply.started":"2023-01-12T00:42:29.884938Z","shell.execute_reply":"2023-01-12T00:42:29.908469Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#drop unrelated features of the train data \ntest_X =test.drop(['index','customer_ID','S_2'],axis = 1)\ntest_X","metadata":{"execution":{"iopub.status.busy":"2023-01-12T00:42:29.910137Z","iopub.execute_input":"2023-01-12T00:42:29.910438Z","iopub.status.idle":"2023-01-12T00:42:32.868015Z","shell.execute_reply.started":"2023-01-12T00:42:29.910400Z","shell.execute_reply":"2023-01-12T00:42:32.867273Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#predict the target of test data using the model\ny_predict =clf.predict(test_X)\ny_predict ","metadata":{"execution":{"iopub.status.busy":"2023-01-12T00:42:32.869212Z","iopub.execute_input":"2023-01-12T00:42:32.869489Z","iopub.status.idle":"2023-01-12T00:42:36.711049Z","shell.execute_reply.started":"2023-01-12T00:42:32.869457Z","shell.execute_reply":"2023-01-12T00:42:36.710325Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test['prediction']=y_predict\ntest[['customer_ID','prediction']].to_csv('out.csv',index=False)","metadata":{"execution":{"iopub.status.busy":"2023-01-12T00:42:36.712058Z","iopub.execute_input":"2023-01-12T00:42:36.712266Z","iopub.status.idle":"2023-01-12T00:43:20.861798Z","shell.execute_reply.started":"2023-01-12T00:42:36.712240Z","shell.execute_reply":"2023-01-12T00:43:20.861014Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import cross_val_score\nscores=cross_val_score(clf,X,y,cv=5)","metadata":{"execution":{"iopub.status.busy":"2023-01-12T00:43:20.863019Z","iopub.execute_input":"2023-01-12T00:43:20.863265Z","iopub.status.idle":"2023-01-12T01:04:16.515483Z","shell.execute_reply.started":"2023-01-12T00:43:20.863235Z","shell.execute_reply":"2023-01-12T01:04:16.514479Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"%0.2f accuracy with a standard deviation of %0.2f\" % (scores.mean(),scores.std()))","metadata":{"execution":{"iopub.status.busy":"2023-01-12T01:04:16.516826Z","iopub.execute_input":"2023-01-12T01:04:16.517040Z","iopub.status.idle":"2023-01-12T01:04:16.522322Z","shell.execute_reply.started":"2023-01-12T01:04:16.517014Z","shell.execute_reply":"2023-01-12T01:04:16.521613Z"},"trusted":true},"execution_count":null,"outputs":[]}]}