{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np \nimport pandas as pd \n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n#Viz\nimport matplotlib.pyplot as plt\nimport seaborn as sns\ncolor = sns.color_palette()\nimport matplotlib as mpl\n%matplotlib inline","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-08T18:54:43.628531Z","iopub.execute_input":"2022-08-08T18:54:43.629200Z","iopub.status.idle":"2022-08-08T18:54:44.916054Z","shell.execute_reply.started":"2022-08-08T18:54:43.629110Z","shell.execute_reply":"2022-08-08T18:54:44.914860Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Import and Explore","metadata":{}},{"cell_type":"code","source":"train=pd.read_csv('/kaggle/input/tabular-playground-series-aug-2022/train.csv')\ntest=pd.read_csv('/kaggle/input/tabular-playground-series-aug-2022/test.csv')\nsample=pd.read_csv('/kaggle/input/tabular-playground-series-aug-2022/sample_submission.csv')","metadata":{"execution":{"iopub.status.busy":"2022-08-08T18:54:44.918047Z","iopub.execute_input":"2022-08-08T18:54:44.918355Z","iopub.status.idle":"2022-08-08T18:54:45.224172Z","shell.execute_reply.started":"2022-08-08T18:54:44.918326Z","shell.execute_reply":"2022-08-08T18:54:45.222907Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.shape","metadata":{"execution":{"iopub.status.busy":"2022-08-08T18:54:45.225835Z","iopub.execute_input":"2022-08-08T18:54:45.226299Z","iopub.status.idle":"2022-08-08T18:54:45.238759Z","shell.execute_reply.started":"2022-08-08T18:54:45.226256Z","shell.execute_reply":"2022-08-08T18:54:45.237397Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.shape","metadata":{"execution":{"iopub.status.busy":"2022-08-08T18:54:45.242751Z","iopub.execute_input":"2022-08-08T18:54:45.243303Z","iopub.status.idle":"2022-08-08T18:54:45.252530Z","shell.execute_reply.started":"2022-08-08T18:54:45.243250Z","shell.execute_reply":"2022-08-08T18:54:45.251295Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-08T18:54:45.254647Z","iopub.execute_input":"2022-08-08T18:54:45.255053Z","iopub.status.idle":"2022-08-08T18:54:45.277580Z","shell.execute_reply.started":"2022-08-08T18:54:45.255011Z","shell.execute_reply":"2022-08-08T18:54:45.276208Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#lets drop the id column first:\ntrain.drop(columns=['id'],inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-08T18:54:45.278836Z","iopub.execute_input":"2022-08-08T18:54:45.279152Z","iopub.status.idle":"2022-08-08T18:54:45.295864Z","shell.execute_reply.started":"2022-08-08T18:54:45.279124Z","shell.execute_reply":"2022-08-08T18:54:45.294900Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.hist(figsize=(12,12))\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-08T18:54:45.297323Z","iopub.execute_input":"2022-08-08T18:54:45.297933Z","iopub.status.idle":"2022-08-08T18:54:48.003517Z","shell.execute_reply.started":"2022-08-08T18:54:45.297897Z","shell.execute_reply":"2022-08-08T18:54:48.002047Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"1. target column is highly skewed\n1. distribution is rather normal in float type features\n","metadata":{}},{"cell_type":"code","source":"train.info()","metadata":{"execution":{"iopub.status.busy":"2022-08-08T18:54:48.004934Z","iopub.execute_input":"2022-08-08T18:54:48.005906Z","iopub.status.idle":"2022-08-08T18:54:48.033858Z","shell.execute_reply.started":"2022-08-08T18:54:48.005833Z","shell.execute_reply":"2022-08-08T18:54:48.032554Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"3 object type (nominal categories), 7 int type (ordinal categories) so 10 categorical type data\n\n16 numerical type data","metadata":{}},{"cell_type":"code","source":"train.describe().T.style.background_gradient(cmap='afmhot',axis=0) ","metadata":{"execution":{"iopub.status.busy":"2022-08-08T18:54:48.035290Z","iopub.execute_input":"2022-08-08T18:54:48.036097Z","iopub.status.idle":"2022-08-08T18:54:48.227865Z","shell.execute_reply.started":"2022-08-08T18:54:48.036050Z","shell.execute_reply":"2022-08-08T18:54:48.226690Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The further we go, the more missing data we have. let's see what percentage of data is missing from each column.","metadata":{}},{"cell_type":"code","source":"missing_prcnt=[train[col].isna().sum()/train.shape[0] *100 for col in train.columns]\nmiss_tbl=pd.DataFrame(missing_prcnt,columns=['%missing'],index=train.columns)","metadata":{"execution":{"iopub.status.busy":"2022-08-08T18:54:48.233980Z","iopub.execute_input":"2022-08-08T18:54:48.234391Z","iopub.status.idle":"2022-08-08T18:54:48.255915Z","shell.execute_reply.started":"2022-08-08T18:54:48.234354Z","shell.execute_reply":"2022-08-08T18:54:48.254781Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"miss_tbl.style.background_gradient(cmap='gist_stern',axis=0) ","metadata":{"execution":{"iopub.status.busy":"2022-08-08T18:54:48.257188Z","iopub.execute_input":"2022-08-08T18:54:48.257576Z","iopub.status.idle":"2022-08-08T18:54:48.277072Z","shell.execute_reply.started":"2022-08-08T18:54:48.257530Z","shell.execute_reply":"2022-08-08T18:54:48.275770Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Now let's explore the different types of categorical data:**","metadata":{}},{"cell_type":"code","source":"train.attribute_0.unique() # 2 nominal categories","metadata":{"execution":{"iopub.status.busy":"2022-08-08T18:54:48.278614Z","iopub.execute_input":"2022-08-08T18:54:48.279746Z","iopub.status.idle":"2022-08-08T18:54:48.289122Z","shell.execute_reply.started":"2022-08-08T18:54:48.279701Z","shell.execute_reply":"2022-08-08T18:54:48.287943Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.attribute_1.unique() # 3 nominal categories","metadata":{"execution":{"iopub.status.busy":"2022-08-08T18:54:48.290500Z","iopub.execute_input":"2022-08-08T18:54:48.291312Z","iopub.status.idle":"2022-08-08T18:54:48.303901Z","shell.execute_reply.started":"2022-08-08T18:54:48.291267Z","shell.execute_reply":"2022-08-08T18:54:48.302858Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.product_code.unique() #5 nominal categories","metadata":{"execution":{"iopub.status.busy":"2022-08-08T18:54:48.305946Z","iopub.execute_input":"2022-08-08T18:54:48.307093Z","iopub.status.idle":"2022-08-08T18:54:48.315668Z","shell.execute_reply.started":"2022-08-08T18:54:48.307017Z","shell.execute_reply":"2022-08-08T18:54:48.314816Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.attribute_2.unique() # 4 ordinal categories","metadata":{"execution":{"iopub.status.busy":"2022-08-08T18:54:48.317294Z","iopub.execute_input":"2022-08-08T18:54:48.318266Z","iopub.status.idle":"2022-08-08T18:54:48.330347Z","shell.execute_reply.started":"2022-08-08T18:54:48.318223Z","shell.execute_reply":"2022-08-08T18:54:48.328697Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.attribute_3.unique() # 4 ordinal categories","metadata":{"execution":{"iopub.status.busy":"2022-08-08T18:54:48.332801Z","iopub.execute_input":"2022-08-08T18:54:48.333236Z","iopub.status.idle":"2022-08-08T18:54:48.341774Z","shell.execute_reply.started":"2022-08-08T18:54:48.333194Z","shell.execute_reply":"2022-08-08T18:54:48.340514Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.measurement_0.nunique() # 29 ordinal categories","metadata":{"execution":{"iopub.status.busy":"2022-08-08T18:54:48.343292Z","iopub.execute_input":"2022-08-08T18:54:48.343929Z","iopub.status.idle":"2022-08-08T18:54:48.354000Z","shell.execute_reply.started":"2022-08-08T18:54:48.343886Z","shell.execute_reply":"2022-08-08T18:54:48.352799Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.measurement_0.value_counts().sort_values()","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2022-08-08T18:54:48.356662Z","iopub.execute_input":"2022-08-08T18:54:48.357166Z","iopub.status.idle":"2022-08-08T18:54:48.366812Z","shell.execute_reply.started":"2022-08-08T18:54:48.357136Z","shell.execute_reply":"2022-08-08T18:54:48.365917Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.measurement_1.nunique() # 30 ordinal categories","metadata":{"execution":{"iopub.status.busy":"2022-08-08T18:54:48.368270Z","iopub.execute_input":"2022-08-08T18:54:48.368833Z","iopub.status.idle":"2022-08-08T18:54:48.377093Z","shell.execute_reply.started":"2022-08-08T18:54:48.368800Z","shell.execute_reply":"2022-08-08T18:54:48.376253Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.measurement_1.value_counts().sort_values()","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2022-08-08T18:54:48.378509Z","iopub.execute_input":"2022-08-08T18:54:48.379060Z","iopub.status.idle":"2022-08-08T18:54:48.388920Z","shell.execute_reply.started":"2022-08-08T18:54:48.379028Z","shell.execute_reply":"2022-08-08T18:54:48.387704Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.measurement_2.nunique() # 35 ordinal categories","metadata":{"execution":{"iopub.status.busy":"2022-08-08T18:54:48.390341Z","iopub.execute_input":"2022-08-08T18:54:48.390951Z","iopub.status.idle":"2022-08-08T18:54:48.398950Z","shell.execute_reply.started":"2022-08-08T18:54:48.390898Z","shell.execute_reply":"2022-08-08T18:54:48.397540Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.measurement_2.value_counts().sort_values()","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2022-08-08T18:54:48.401859Z","iopub.execute_input":"2022-08-08T18:54:48.402907Z","iopub.status.idle":"2022-08-08T18:54:48.414139Z","shell.execute_reply.started":"2022-08-08T18:54:48.402856Z","shell.execute_reply":"2022-08-08T18:54:48.412981Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Visualization","metadata":{}},{"cell_type":"code","source":"#we are going to find correlated features, just note that correlation (spearman) is used for numerical and ordinal data \nmask_con_corr = train.corr(method='spearman')[(train.corr(method='spearman') >= 0.3) | (train.corr(method='spearman') <= -0.3)]\nmask = np.triu(np.ones_like(train.corr()))\ng=sns.set(rc = {'figure.figsize':(12,12)})\ng=sns.heatmap(data=mask_con_corr,cmap='summer',mask=mask, annot=True, fmt='0.2f',linewidths=.5)\ng.set_title(\"Heatmap of Correlation\")","metadata":{"execution":{"iopub.status.busy":"2022-08-08T18:54:48.415796Z","iopub.execute_input":"2022-08-08T18:54:48.416646Z","iopub.status.idle":"2022-08-08T18:54:53.431629Z","shell.execute_reply.started":"2022-08-08T18:54:48.416600Z","shell.execute_reply":"2022-08-08T18:54:53.430478Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Some features are rather correlated those features are:\n\n> [measurement_5, measurement_6, measurement_7, measurement_8] & measurement_17\n\n> attribute_2 & attribute_3\n\n> attribute_2 & measurement_1\n\n> attribute_3 & measurement_1\n","metadata":{}},{"cell_type":"markdown","source":"complete pair plot takes some time to complete, but we can see that every pair of numerical features create a circular cluster and no specific patterns can be seen, below is an example of these clusters:","metadata":{}},{"cell_type":"code","source":"pairplot_cols=list(train.select_dtypes(['float']).columns)\npairplot_cols.append('failure')\nsns.pairplot(data=train.loc[:,pairplot_cols[-3:]],hue='failure',palette='CMRmap',kind='scatter')","metadata":{"execution":{"iopub.status.busy":"2022-08-08T18:54:53.432845Z","iopub.execute_input":"2022-08-08T18:54:53.433157Z","iopub.status.idle":"2022-08-08T18:54:57.512117Z","shell.execute_reply.started":"2022-08-08T18:54:53.433128Z","shell.execute_reply":"2022-08-08T18:54:57.510795Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Looking at the countplot we can see, two groups of failure (0,1) have almost the same count ratios in each feature\nfor example: attribute_0, material 7 count is much higher than material 5 count, in both groups of failure or in attribute_1, there's a descending pattern in both groups of failure.","metadata":{}},{"cell_type":"code","source":"fig, axes = plt.subplots(2,4,figsize=(20,20))\nbarplot_cols=list(train.select_dtypes(exclude=['float']).columns)\nfor name, ax in zip(barplot_cols, axes.flatten()):\n    sns.countplot(x=name,hue='failure',data=train,ax=ax)","metadata":{"execution":{"iopub.status.busy":"2022-08-08T18:54:57.513723Z","iopub.execute_input":"2022-08-08T18:54:57.514561Z","iopub.status.idle":"2022-08-08T18:54:59.967957Z","shell.execute_reply.started":"2022-08-08T18:54:57.514517Z","shell.execute_reply":"2022-08-08T18:54:59.966807Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from scipy.stats import chi2_contingency\ndef chi2_calc(df,target):\n    scores=[]\n    for col in df.columns:\n        ct=pd.crosstab(df[col],target)\n        stat,p,dof,expected=chi2_contingency(ct)\n        scores.append(p)\n    return pd.DataFrame(scores, index=df.columns, columns=['P value']).sort_values(by='P value')","metadata":{"execution":{"iopub.status.busy":"2022-08-08T18:54:59.969609Z","iopub.execute_input":"2022-08-08T18:54:59.970215Z","iopub.status.idle":"2022-08-08T18:54:59.976921Z","shell.execute_reply.started":"2022-08-08T18:54:59.970164Z","shell.execute_reply":"2022-08-08T18:54:59.975728Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"chi2_calc(train.select_dtypes(['object','int']),train.failure)","metadata":{"execution":{"iopub.status.busy":"2022-08-08T18:54:59.978625Z","iopub.execute_input":"2022-08-08T18:54:59.978943Z","iopub.status.idle":"2022-08-08T18:55:00.119613Z","shell.execute_reply.started":"2022-08-08T18:54:59.978910Z","shell.execute_reply":"2022-08-08T18:55:00.118504Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"As we can see, there are only a few features with p-value under 0.05, meaning that a lot of categorical features don't have a statistically significant relation with the target, we may be better off dropping them. in this note book I won't drop them because I want to establish a baseline.\n\nFeatures with P-values over 0.05:\n\n\n* measurement_2\n* attribute_1\n* measurement_0\n* measurement_1\n","metadata":{}},{"cell_type":"code","source":"#let's separate the target column from the rest of the dataset\ny=train.pop('failure')","metadata":{"execution":{"iopub.status.busy":"2022-08-08T18:55:00.124337Z","iopub.execute_input":"2022-08-08T18:55:00.125217Z","iopub.status.idle":"2022-08-08T18:55:00.130742Z","shell.execute_reply.started":"2022-08-08T18:55:00.125177Z","shell.execute_reply":"2022-08-08T18:55:00.129512Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Creating the Pipeline","metadata":{}},{"cell_type":"code","source":"from sklearn.pipeline import Pipeline\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.preprocessing import OrdinalEncoder,OneHotEncoder,StandardScaler\nfrom sklearn.compose import ColumnTransformer\n\n############################################## Creating Preprocessing Pipelines#################################################\n# there are 3 types of data, nominal,ordinal and numeric. I'll create three pipelines for preprocessing\n\npipeline_ord=Pipeline(steps=[('SimpleImputer',SimpleImputer(strategy='most_frequent'))])\npipeline_nom=Pipeline(steps=[('SimpleImputer',SimpleImputer(strategy='most_frequent')),('OneHotEncoder',OneHotEncoder(handle_unknown='error',sparse=False))])\npipeline_numeric=Pipeline(steps=[('SimpleImputer',SimpleImputer(strategy='median')),('StandardScaler',StandardScaler())])\n\n########################################### Creating A ColumnTransformer############################################\n\nfrom sklearn.compose import ColumnTransformer\n\nordinals=train.select_dtypes(include='int').columns\nnominals=train.select_dtypes(include='object').columns\nnumericals=train.select_dtypes(include='float').columns\nclmn_trns=ColumnTransformer(transformers=[('ordinals',pipeline_ord,ordinals),\n                                          ('nominals',pipeline_nom,nominals),\n                                          ('numericals',pipeline_numeric,numericals)])\n\n","metadata":{"execution":{"iopub.status.busy":"2022-08-08T18:55:00.132448Z","iopub.execute_input":"2022-08-08T18:55:00.132893Z","iopub.status.idle":"2022-08-08T18:55:00.420344Z","shell.execute_reply.started":"2022-08-08T18:55:00.132848Z","shell.execute_reply":"2022-08-08T18:55:00.419129Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Scoring","metadata":{}},{"cell_type":"code","source":"from sklearn.ensemble import RandomForestClassifier,GradientBoostingClassifier\nfrom xgboost import XGBClassifier\nfrom lightgbm import LGBMClassifier\nfrom catboost import CatBoostClassifier\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.model_selection import cross_val_score\n\nclassifiers = []\nrandom_state=0\nclassifiers.append(RandomForestClassifier(random_state=random_state))\nclassifiers.append(GradientBoostingClassifier(random_state=random_state)) \nclassifiers.append(XGBClassifier(random_state=random_state))\nclassifiers.append(LGBMClassifier(random_state=random_state))\nclassifiers.append(CatBoostClassifier(random_state=random_state,verbose=0))\n\n# CREATING A FOR LOOP FOR SCORING EACH MODEL\ncv_results = []\ncv=StratifiedKFold(n_splits=2, random_state=random_state, shuffle=True)\nfor classifier in classifiers :\n    classif = Pipeline(steps=[\n        ('clmn_trns',clmn_trns), #COLUMN TRANSFORMER\n        ('classifier', classifier)])\n    cvs=cross_val_score(classif, train, y, scoring = \"roc_auc\", cv = cv, n_jobs=-1)\n    cv_results.append(cvs)\n\ncv_means = []\ncv_std = []\nfor cv_result in cv_results:\n    cv_means.append(cv_result.mean())\n    cv_std.append(cv_result.std())\n\n#CREATING A DATAFRAME OF MODEL SCORES\ncv_res = pd.DataFrame({\"CrossValMeans\":cv_means,\"CrossValSDs\": cv_std,\"Algorithm\":[\"RandomForestClassifier\",\n                                                                                   \"GradientBoostingClassifier\",\n                                                                                   \"XGBClassifier\",\n                                                                                   \"LGBMClassifier\",\n                                                                                   \"CatBoostClassifier\"]})\n#PLOTTING\ng=sns.set(rc={'figure.figsize':(5,5)})\ng = sns.barplot(x=\"CrossValMeans\",y=\"Algorithm\",data = cv_res.sort_values(by='CrossValMeans'), palette=\"summer\",**{'xerr':cv_std})\ng.set_xlabel(\"Mean Roc-Auc\")\ng = g.set_title(\"Cross validation scores\")\n\n\ncv_res.sort_values(by='CrossValMeans')","metadata":{"execution":{"iopub.status.busy":"2022-08-08T18:55:00.421823Z","iopub.execute_input":"2022-08-08T18:55:00.422152Z","iopub.status.idle":"2022-08-08T18:55:43.820461Z","shell.execute_reply.started":"2022-08-08T18:55:00.422121Z","shell.execute_reply":"2022-08-08T18:55:43.819015Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"As we can see, on average the score is around 50 (not very different from a no skill model), so we need to make changes to improve this score!","metadata":{}}]}