{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport os\nimport matplotlib.pyplot as plt\nimport plotly.express as px\nimport seaborn as sns\nfrom sklearn import preprocessing\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.metrics import roc_auc_score\nplt.style.use('fivethirtyeight')\npd.set_option('max_columns', 500) #for displaying all columns\ncolor_pal = plt.rcParams[\"axes.prop_cycle\"].by_key()[\"color\"]\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-09T13:03:28.715228Z","iopub.execute_input":"2022-07-09T13:03:28.715790Z","iopub.status.idle":"2022-07-09T13:03:32.236573Z","shell.execute_reply.started":"2022-07-09T13:03:28.715680Z","shell.execute_reply":"2022-07-09T13:03:32.235297Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv(\"../input/railofy-challenge/Railofy_training_data_for_model.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-07-09T13:03:32.239429Z","iopub.execute_input":"2022-07-09T13:03:32.239768Z","iopub.status.idle":"2022-07-09T13:03:32.537892Z","shell.execute_reply.started":"2022-07-09T13:03:32.239740Z","shell.execute_reply":"2022-07-09T13:03:32.536590Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.columns","metadata":{"execution":{"iopub.status.busy":"2022-07-09T13:03:32.539566Z","iopub.execute_input":"2022-07-09T13:03:32.540027Z","iopub.status.idle":"2022-07-09T13:03:32.552517Z","shell.execute_reply.started":"2022-07-09T13:03:32.539984Z","shell.execute_reply":"2022-07-09T13:03:32.551332Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.isna().sum()","metadata":{"execution":{"iopub.status.busy":"2022-07-09T13:03:32.555591Z","iopub.execute_input":"2022-07-09T13:03:32.556349Z","iopub.status.idle":"2022-07-09T13:03:32.579836Z","shell.execute_reply.started":"2022-07-09T13:03:32.556300Z","shell.execute_reply":"2022-07-09T13:03:32.578522Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* We have 24 feature columns in the train dataset.\n\n* No missing data\n","metadata":{}},{"cell_type":"code","source":"train.describe()","metadata":{"execution":{"iopub.status.busy":"2022-07-09T13:03:32.582751Z","iopub.execute_input":"2022-07-09T13:03:32.583612Z","iopub.status.idle":"2022-07-09T13:03:32.723683Z","shell.execute_reply.started":"2022-07-09T13:03:32.583561Z","shell.execute_reply":"2022-07-09T13:03:32.721993Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-09T13:03:32.725567Z","iopub.execute_input":"2022-07-09T13:03:32.725879Z","iopub.status.idle":"2022-07-09T13:03:32.753254Z","shell.execute_reply.started":"2022-07-09T13:03:32.725852Z","shell.execute_reply":"2022-07-09T13:03:32.751982Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* QT, SL, BKT_2, CL_1, CL_2, CL_3 are categorical columns\n\n* pk is a primary key, identifier column","metadata":{}},{"cell_type":"markdown","source":"# Target Distribution","metadata":{}},{"cell_type":"code","source":"sns.countplot(data=train,x='target')","metadata":{"execution":{"iopub.status.busy":"2022-07-09T13:03:32.755025Z","iopub.execute_input":"2022-07-09T13:03:32.755886Z","iopub.status.idle":"2022-07-09T13:03:32.979700Z","shell.execute_reply.started":"2022-07-09T13:03:32.755848Z","shell.execute_reply":"2022-07-09T13:03:32.978380Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* We have almost 25000 entries with target 0 and 12000 entries with target 1.\n\n* Target is imbalanced, this would need to be accounted for while setting up the model. ","metadata":{}},{"cell_type":"markdown","source":"# Feature visualization and preprocessing","metadata":{}},{"cell_type":"code","source":"categorical_columns = ['QT','SL','BKT_2','CL_1','CL_2','CL_3']\ni = 1\nplt.figure(figsize=(15,15))\nfor columns in categorical_columns:\n    plt.subplot(3,2,i)\n    sns.countplot(x=columns,hue=\"target\",data=train)\n    i+=1","metadata":{"execution":{"iopub.status.busy":"2022-07-09T13:03:32.981925Z","iopub.execute_input":"2022-07-09T13:03:32.982788Z","iopub.status.idle":"2022-07-09T13:03:34.031492Z","shell.execute_reply.started":"2022-07-09T13:03:32.982737Z","shell.execute_reply":"2022-07-09T13:03:34.030479Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"numeric = [ 'GRCA', 'CCA', 'JD', 'ODD', 'JS', 'ODS', 'NDTD',\n       'CURP', 'GROP', 'CANP', 'SBRA', 'SCRA', 'GRA', 'CURA', 'RPW', 'CUCA',\n       'CAR']\ni = 1\nplt.figure(figsize=(20,20))\nfor columns in numeric:\n    plt.subplot(9,3,i)\n    sns.kdeplot(x=columns,data=train)\n    #plt.title(columns)\n    i+=1","metadata":{"execution":{"iopub.status.busy":"2022-07-09T13:03:34.032844Z","iopub.execute_input":"2022-07-09T13:03:34.033235Z","iopub.status.idle":"2022-07-09T13:03:39.098675Z","shell.execute_reply.started":"2022-07-09T13:03:34.033204Z","shell.execute_reply":"2022-07-09T13:03:39.097441Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* As the scale of all variables seems to be different we would need to normalise before inputting to the model.","metadata":{}},{"cell_type":"code","source":"le = preprocessing.LabelEncoder()\ntrain['QT'] = le.fit_transform(train['QT'])\ny = train['target']\nX = train.drop(['pk','target'],axis=1)\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.33, random_state=42)\nscaler = preprocessing.StandardScaler().fit(X_train)\nscaled = scaler.transform(X_train)\nscaled_test = scaler.transform(X_test)","metadata":{"execution":{"iopub.status.busy":"2022-07-09T13:03:39.102920Z","iopub.execute_input":"2022-07-09T13:03:39.104076Z","iopub.status.idle":"2022-07-09T13:03:39.166861Z","shell.execute_reply.started":"2022-07-09T13:03:39.104024Z","shell.execute_reply":"2022-07-09T13:03:39.165995Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Baseline Model","metadata":{}},{"cell_type":"code","source":"clf = RandomForestClassifier(class_weight=\"balanced\")\nclf.fit(scaled,y_train)\npred = clf.predict(scaled_test)\nroc_auc_score(y_test,pred)","metadata":{"execution":{"iopub.status.busy":"2022-07-09T13:03:39.168483Z","iopub.execute_input":"2022-07-09T13:03:39.169202Z","iopub.status.idle":"2022-07-09T13:03:46.275999Z","shell.execute_reply.started":"2022-07-09T13:03:39.169158Z","shell.execute_reply":"2022-07-09T13:03:46.274831Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Creating a submission file","metadata":{}},{"cell_type":"code","source":"TEST = pd.read_csv(\"../input/railofy-challenge/Railofy_testing_data_for_model.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-07-09T13:03:46.277388Z","iopub.execute_input":"2022-07-09T13:03:46.277734Z","iopub.status.idle":"2022-07-09T13:03:46.332450Z","shell.execute_reply.started":"2022-07-09T13:03:46.277704Z","shell.execute_reply":"2022-07-09T13:03:46.331150Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = TEST.drop(['pk'],axis=1)\nX['QT'] = le.fit_transform(X['QT'])\nscaled_te = scaler.transform(X)\nsubmission_pred = clf.predict_proba(scaled_te)","metadata":{"execution":{"iopub.status.busy":"2022-07-09T13:03:46.334048Z","iopub.execute_input":"2022-07-09T13:03:46.334635Z","iopub.status.idle":"2022-07-09T13:03:46.458612Z","shell.execute_reply.started":"2022-07-09T13:03:46.334601Z","shell.execute_reply":"2022-07-09T13:03:46.457466Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"prediction = pd.DataFrame()\nprediction['pk'] = TEST['pk']\nprediction['target'] = submission_pred[:,1]","metadata":{"execution":{"iopub.status.busy":"2022-07-09T13:03:46.460040Z","iopub.execute_input":"2022-07-09T13:03:46.460372Z","iopub.status.idle":"2022-07-09T13:03:46.469081Z","shell.execute_reply.started":"2022-07-09T13:03:46.460344Z","shell.execute_reply":"2022-07-09T13:03:46.467791Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"prediction.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-09T13:03:46.470450Z","iopub.execute_input":"2022-07-09T13:03:46.470756Z","iopub.status.idle":"2022-07-09T13:03:46.494865Z","shell.execute_reply.started":"2022-07-09T13:03:46.470728Z","shell.execute_reply":"2022-07-09T13:03:46.493982Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"prediction.to_csv(\"submission.csv\",index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-09T13:03:46.496132Z","iopub.execute_input":"2022-07-09T13:03:46.496969Z","iopub.status.idle":"2022-07-09T13:03:46.519268Z","shell.execute_reply.started":"2022-07-09T13:03:46.496918Z","shell.execute_reply":"2022-07-09T13:03:46.518045Z"},"trusted":true},"execution_count":null,"outputs":[]}]}