{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.model_selection import train_test_split\nimport matplotlib.pyplot as plt\n\nimport time\nimport tracemalloc\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-06-30T09:30:59.152006Z","iopub.execute_input":"2022-06-30T09:30:59.152438Z","iopub.status.idle":"2022-06-30T09:30:59.157592Z","shell.execute_reply.started":"2022-06-30T09:30:59.152403Z","shell.execute_reply":"2022-06-30T09:30:59.156538Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"start_time = time.time()\ntracemalloc.start()\ntrain = pd.read_feather('../input/amexfeather/train_data.ftr')\ntrain.head(5)\nprint(\"--- %s seconds ---\" % (time.time() - start_time))\nprint(tracemalloc.get_traced_memory())","metadata":{"execution":{"iopub.status.busy":"2022-06-30T09:30:59.162079Z","iopub.execute_input":"2022-06-30T09:30:59.162467Z","iopub.status.idle":"2022-06-30T09:31:17.557435Z","shell.execute_reply.started":"2022-06-30T09:30:59.162434Z","shell.execute_reply":"2022-06-30T09:31:17.555989Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.shape","metadata":{"execution":{"iopub.status.busy":"2022-06-30T09:31:17.559863Z","iopub.execute_input":"2022-06-30T09:31:17.560325Z","iopub.status.idle":"2022-06-30T09:31:17.572654Z","shell.execute_reply.started":"2022-06-30T09:31:17.560285Z","shell.execute_reply":"2022-06-30T09:31:17.571075Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Since there are too many variables, some of them will be deleted.","metadata":{}},{"cell_type":"markdown","source":"### **Removal of Columns that Contains too many N/As**","metadata":{}},{"cell_type":"code","source":"# Calculate percentage of N/As in columns and Rename the result dataframe\nstart_time = time.time()\ntracemalloc.start()\n\nnulls = pd.DataFrame(train.isnull().sum(axis = 0)* 100 / len(train)).reset_index()\nnulls = nulls.rename(columns={'index':'var',0: \"percent\"})\n\n# Use the columns with <30% N/As\nmany_nulls = nulls[nulls['percent']>70][\"var\"]\ntrain.drop(columns = many_nulls,axis=1, inplace=True)\n\nprint(\"--- %s seconds ---\" % (time.time() - start_time))\nprint(tracemalloc.get_traced_memory())","metadata":{"execution":{"iopub.status.busy":"2022-06-30T09:31:17.574706Z","iopub.execute_input":"2022-06-30T09:31:17.575359Z","iopub.status.idle":"2022-06-30T09:31:26.019860Z","shell.execute_reply.started":"2022-06-30T09:31:17.575313Z","shell.execute_reply":"2022-06-30T09:31:26.018250Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### **Delete Customer ID, Date & Categorical Column**","metadata":{}},{"cell_type":"code","source":"#train['year'] = pd.to_datetime(train['S_2']).dt.year\n#train['month'] = pd.to_datetime(train['S_2']).dt.month\n#train['day'] = pd.to_datetime(train['S_2']).dt.day\ntrain.drop(['customer_ID','S_2'],inplace=True,axis=1)\n\n","metadata":{"execution":{"iopub.status.busy":"2022-06-30T09:31:26.023868Z","iopub.execute_input":"2022-06-30T09:31:26.024931Z","iopub.status.idle":"2022-06-30T09:31:29.220169Z","shell.execute_reply.started":"2022-06-30T09:31:26.024882Z","shell.execute_reply":"2022-06-30T09:31:29.217650Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cats = train.select_dtypes(include=['category'])\ncats.columns\ntrain.drop(['D_63', 'D_64', 'D_68', 'B_30', 'B_38', 'D_114', 'D_116', 'D_117',\n       'D_120', 'D_126'],inplace=True,axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-06-30T09:31:29.225349Z","iopub.execute_input":"2022-06-30T09:31:29.228767Z","iopub.status.idle":"2022-06-30T09:31:38.139656Z","shell.execute_reply.started":"2022-06-30T09:31:29.228680Z","shell.execute_reply":"2022-06-30T09:31:38.134659Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = train.replace(np.nan, 0)\n#train.head(1000).to_csv('./out1.csv')","metadata":{"execution":{"iopub.status.busy":"2022-06-30T09:31:38.172372Z","iopub.execute_input":"2022-06-30T09:31:38.182088Z","iopub.status.idle":"2022-06-30T09:31:47.111534Z","shell.execute_reply.started":"2022-06-30T09:31:38.181476Z","shell.execute_reply":"2022-06-30T09:31:47.110232Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## **Machine Learning Starts**","metadata":{}},{"cell_type":"code","source":"#train_sample = train.head(10000)\n#del train","metadata":{"execution":{"iopub.status.busy":"2022-06-30T09:31:47.114372Z","iopub.execute_input":"2022-06-30T09:31:47.115064Z","iopub.status.idle":"2022-06-30T09:31:47.121660Z","shell.execute_reply.started":"2022-06-30T09:31:47.115000Z","shell.execute_reply":"2022-06-30T09:31:47.119991Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#train.head(100).to_csv('./out.csv')\n#X = train_sample.loc[:, train_sample.columns!=['customer_ID','target']]\nX = train.drop(['target'],inplace=False,axis=1)\ny = train['target']\ndel train","metadata":{"execution":{"iopub.status.busy":"2022-06-30T09:31:47.124347Z","iopub.execute_input":"2022-06-30T09:31:47.125367Z","iopub.status.idle":"2022-06-30T09:31:52.955903Z","shell.execute_reply.started":"2022-06-30T09:31:47.125318Z","shell.execute_reply":"2022-06-30T09:31:52.954391Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import StandardScaler\n\nsc = StandardScaler()\nX = sc.fit_transform(X)\n","metadata":{"execution":{"iopub.status.busy":"2022-06-30T09:31:52.958033Z","iopub.execute_input":"2022-06-30T09:31:52.958489Z","iopub.status.idle":"2022-06-30T09:32:30.320048Z","shell.execute_reply.started":"2022-06-30T09:31:52.958450Z","shell.execute_reply":"2022-06-30T09:32:30.318572Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train,X_test,y_train,y_test=train_test_split(X,y,test_size=0.3)\n","metadata":{"execution":{"iopub.status.busy":"2022-06-30T09:32:30.324231Z","iopub.execute_input":"2022-06-30T09:32:30.324984Z","iopub.status.idle":"2022-06-30T09:32:49.271558Z","shell.execute_reply.started":"2022-06-30T09:32:30.324932Z","shell.execute_reply":"2022-06-30T09:32:49.269974Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.linear_model import LogisticRegression\n\n\nlogit = LogisticRegression()\n\nlogit.fit( X_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2022-06-30T09:33:10.705285Z","iopub.execute_input":"2022-06-30T09:33:10.706701Z","iopub.status.idle":"2022-06-30T09:35:00.599803Z","shell.execute_reply.started":"2022-06-30T09:33:10.706642Z","shell.execute_reply":"2022-06-30T09:35:00.598032Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import confusion_matrix, accuracy_score\n\nprint(\"Score the X-train with Y-train is : \", round(logit.score(X_train,y_train),2))\nprint(\"Score the X-test  with Y-test  is : \", round(logit.score(X_test,y_test),2))\n\ny_pred=logit.predict(X_test)\n\n\ncm = confusion_matrix(y_test,y_pred)\nprint(cm)\n#accuracy_score(y_test, y_test)","metadata":{"execution":{"iopub.status.busy":"2022-06-30T09:37:22.636634Z","iopub.execute_input":"2022-06-30T09:37:22.637526Z","iopub.status.idle":"2022-06-30T09:37:32.979426Z","shell.execute_reply.started":"2022-06-30T09:37:22.637473Z","shell.execute_reply":"2022-06-30T09:37:32.977696Z"},"trusted":true},"execution_count":null,"outputs":[]}]}