{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-01T10:47:14.444374Z","iopub.execute_input":"2022-08-01T10:47:14.444806Z","iopub.status.idle":"2022-08-01T10:47:14.455335Z","shell.execute_reply.started":"2022-08-01T10:47:14.444773Z","shell.execute_reply":"2022-08-01T10:47:14.454208Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import seaborn as sns\nimport matplotlib.pyplot as plt\nfrom sklearn import tree\nfrom sklearn.tree import DecisionTreeClassifier\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.svm import SVC\nfrom sklearn.neighbors import KNeighborsClassifier\nfrom sklearn.naive_bayes import GaussianNB\nfrom sklearn.metrics import classification_report, confusion_matrix\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import OneHotEncoder\nfrom sklearn.metrics import classification_report, confusion_matrix, roc_curve, f1_score\n# import warnings filter\nfrom warnings import simplefilter\n# ignore all future warnings\nsimplefilter(action='ignore', category=FutureWarning)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T10:47:14.463528Z","iopub.execute_input":"2022-08-01T10:47:14.464315Z","iopub.status.idle":"2022-08-01T10:47:14.474048Z","shell.execute_reply.started":"2022-08-01T10:47:14.464278Z","shell.execute_reply":"2022-08-01T10:47:14.472720Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv('/kaggle/input/tabular-playground-series-aug-2022/train.csv')\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-01T10:47:14.477186Z","iopub.execute_input":"2022-08-01T10:47:14.478071Z","iopub.status.idle":"2022-08-01T10:47:14.677625Z","shell.execute_reply.started":"2022-08-01T10:47:14.478016Z","shell.execute_reply":"2022-08-01T10:47:14.676434Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(len(df))\ndf.nunique()","metadata":{"execution":{"iopub.status.busy":"2022-08-01T10:47:14.679192Z","iopub.execute_input":"2022-08-01T10:47:14.679557Z","iopub.status.idle":"2022-08-01T10:47:14.717253Z","shell.execute_reply.started":"2022-08-01T10:47:14.679525Z","shell.execute_reply":"2022-08-01T10:47:14.715565Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.dtypes","metadata":{"execution":{"iopub.status.busy":"2022-08-01T10:47:14.721134Z","iopub.execute_input":"2022-08-01T10:47:14.721662Z","iopub.status.idle":"2022-08-01T10:47:14.733825Z","shell.execute_reply.started":"2022-08-01T10:47:14.721614Z","shell.execute_reply":"2022-08-01T10:47:14.732561Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.set(rc={'figure.figsize':(34,30)})\nfor i,col in enumerate(df.columns,1):\n    #if df.dtypes[col]!='object':\n    plt.subplot(5,6,i)\n    sns.histplot(df[col], label = col)\n    plt.legend()","metadata":{"execution":{"iopub.status.busy":"2022-08-01T10:47:14.735141Z","iopub.execute_input":"2022-08-01T10:47:14.735583Z","iopub.status.idle":"2022-08-01T10:47:26.346648Z","shell.execute_reply.started":"2022-08-01T10:47:14.735528Z","shell.execute_reply":"2022-08-01T10:47:26.345435Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"It looks like attribute columns are categorical features ","metadata":{}},{"cell_type":"code","source":"sns.set(rc={'figure.figsize':(24,20)})\nsns.heatmap(df.corr(), annot=True,fmt='.2f')","metadata":{"execution":{"iopub.status.busy":"2022-08-01T10:47:26.348476Z","iopub.execute_input":"2022-08-01T10:47:26.348903Z","iopub.status.idle":"2022-08-01T10:47:30.096331Z","shell.execute_reply.started":"2022-08-01T10:47:26.348865Z","shell.execute_reply":"2022-08-01T10:47:30.094718Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.dropna()","metadata":{"execution":{"iopub.status.busy":"2022-08-01T10:47:30.098516Z","iopub.execute_input":"2022-08-01T10:47:30.099023Z","iopub.status.idle":"2022-08-01T10:47:30.161339Z","shell.execute_reply.started":"2022-08-01T10:47:30.098978Z","shell.execute_reply":"2022-08-01T10:47:30.160183Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Too many rows with missing values so can't just remove these rows","metadata":{}},{"cell_type":"code","source":"for col in df.columns:\n    ratio_null = len(df[df[col].isna()])/len(df)\n    if ratio_null!=0:\n        print(col,'%.2f percent missing values'%(100*ratio_null))","metadata":{"execution":{"iopub.status.busy":"2022-08-01T10:47:30.163128Z","iopub.execute_input":"2022-08-01T10:47:30.163545Z","iopub.status.idle":"2022-08-01T10:47:30.202131Z","shell.execute_reply.started":"2022-08-01T10:47:30.163509Z","shell.execute_reply":"2022-08-01T10:47:30.201166Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"All columns have less than 10 % missing values so no need to drop any columns\nAll columns with missing values have continuous data so it is safe to replace missing values with mean.","metadata":{}},{"cell_type":"code","source":"df = df.fillna(df.mean())","metadata":{"execution":{"iopub.status.busy":"2022-08-01T10:47:30.206356Z","iopub.execute_input":"2022-08-01T10:47:30.206811Z","iopub.status.idle":"2022-08-01T10:47:30.561376Z","shell.execute_reply.started":"2022-08-01T10:47:30.206773Z","shell.execute_reply":"2022-08-01T10:47:30.560170Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = df.drop(['id', 'failure'], axis = 1)\ny = df['failure']","metadata":{"execution":{"iopub.status.busy":"2022-08-01T10:47:30.586783Z","iopub.execute_input":"2022-08-01T10:47:30.587226Z","iopub.status.idle":"2022-08-01T10:47:30.599366Z","shell.execute_reply.started":"2022-08-01T10:47:30.587179Z","shell.execute_reply":"2022-08-01T10:47:30.597252Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"categorical_cols = ['product_code', 'attribute_0', 'attribute_1', 'attribute_2',\n       'attribute_3']\nenc = OneHotEncoder(handle_unknown='ignore')\nencoded_data  = pd.DataFrame(enc.fit_transform(X[categorical_cols]).toarray())\nencoded_data","metadata":{"execution":{"iopub.status.busy":"2022-08-01T10:47:30.605588Z","iopub.execute_input":"2022-08-01T10:47:30.607027Z","iopub.status.idle":"2022-08-01T10:47:30.707255Z","shell.execute_reply.started":"2022-08-01T10:47:30.606961Z","shell.execute_reply":"2022-08-01T10:47:30.705842Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = X[ ['loading', 'measurement_0', 'measurement_1', 'measurement_2',\n       'measurement_3', 'measurement_4', 'measurement_5', 'measurement_6',\n       'measurement_7', 'measurement_8', 'measurement_9', 'measurement_10',\n       'measurement_11', 'measurement_12', 'measurement_13', 'measurement_14',\n       'measurement_15', 'measurement_16', 'measurement_17']]\nX = X.join(encoded_data)\nX.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-01T10:47:30.709066Z","iopub.execute_input":"2022-08-01T10:47:30.709449Z","iopub.status.idle":"2022-08-01T10:47:30.754968Z","shell.execute_reply.started":"2022-08-01T10:47:30.709416Z","shell.execute_reply":"2022-08-01T10:47:30.753670Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.3, random_state=0)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T10:47:30.756859Z","iopub.execute_input":"2022-08-01T10:47:30.758048Z","iopub.status.idle":"2022-08-01T10:47:30.790052Z","shell.execute_reply.started":"2022-08-01T10:47:30.757996Z","shell.execute_reply":"2022-08-01T10:47:30.788856Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"clf = tree.DecisionTreeClassifier()\nclf = clf.fit(X_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T10:47:30.792373Z","iopub.execute_input":"2022-08-01T10:47:30.792859Z","iopub.status.idle":"2022-08-01T10:47:31.753045Z","shell.execute_reply.started":"2022-08-01T10:47:30.792815Z","shell.execute_reply":"2022-08-01T10:47:31.751687Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred = clf.predict(X_test)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T10:47:31.754789Z","iopub.execute_input":"2022-08-01T10:47:31.755171Z","iopub.status.idle":"2022-08-01T10:47:31.768376Z","shell.execute_reply.started":"2022-08-01T10:47:31.755133Z","shell.execute_reply":"2022-08-01T10:47:31.766757Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(classification_report(y_test, y_pred))","metadata":{"execution":{"iopub.status.busy":"2022-08-01T10:47:31.770256Z","iopub.execute_input":"2022-08-01T10:47:31.770648Z","iopub.status.idle":"2022-08-01T10:47:31.799799Z","shell.execute_reply.started":"2022-08-01T10:47:31.770605Z","shell.execute_reply":"2022-08-01T10:47:31.798303Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Trying various models with default parameters","metadata":{}},{"cell_type":"code","source":"# Import required libraries for machine learning classifiers\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.svm import LinearSVC\nfrom sklearn.tree import DecisionTreeClassifier\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.naive_bayes import GaussianNB\nfrom xgboost import XGBClassifier","metadata":{"execution":{"iopub.status.busy":"2022-08-01T10:47:31.801835Z","iopub.execute_input":"2022-08-01T10:47:31.802217Z","iopub.status.idle":"2022-08-01T10:47:31.809385Z","shell.execute_reply.started":"2022-08-01T10:47:31.802182Z","shell.execute_reply":"2022-08-01T10:47:31.807973Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"log_model = LogisticRegression(max_iter=10000)\nsvc_model = LinearSVC(dual=False)\ndtr_model = DecisionTreeClassifier()\ngnb_model = GaussianNB()\nrf = RandomForestClassifier(n_estimators = 10) \nxg = XGBClassifier()\nmodels = [log_model, svc_model, dtr_model, gnb_model, rf, xg]","metadata":{"execution":{"iopub.status.busy":"2022-08-01T10:47:31.817778Z","iopub.execute_input":"2022-08-01T10:47:31.818263Z","iopub.status.idle":"2022-08-01T10:47:31.827698Z","shell.execute_reply.started":"2022-08-01T10:47:31.818217Z","shell.execute_reply":"2022-08-01T10:47:31.826077Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"names = ['logistic regression', 'SVC', 'decision tree', 'naive bayes', 'random forest','xg_boost']\ni=0\nfor model in models:\n    model.fit(X_train, y_train)\n \n    # performing predictions on the test dataset\n    y_pred = model.predict(X_test)\n  \n    # using metrics module for f1 calculation\n    print(\"F1 SCORE OF THE \"+names[i], f1_score(y_test, y_pred))\n    i+=1","metadata":{"execution":{"iopub.status.busy":"2022-08-01T10:47:31.830069Z","iopub.execute_input":"2022-08-01T10:47:31.830568Z","iopub.status.idle":"2022-08-01T10:47:38.542816Z","shell.execute_reply.started":"2022-08-01T10:47:31.830527Z","shell.execute_reply":"2022-08-01T10:47:38.541263Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Decision tree performed the best, which is unusual, will further explore the results","metadata":{}},{"cell_type":"code","source":"dtr_model = tree.DecisionTreeClassifier()\ndtr_model.fit(X,y)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T10:47:38.548438Z","iopub.execute_input":"2022-08-01T10:47:38.548873Z","iopub.status.idle":"2022-08-01T10:47:39.859687Z","shell.execute_reply.started":"2022-08-01T10:47:38.548839Z","shell.execute_reply":"2022-08-01T10:47:39.858142Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test = pd.read_csv('/kaggle/input/tabular-playground-series-aug-2022/test.csv')\ndf_test.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-01T10:47:39.861554Z","iopub.execute_input":"2022-08-01T10:47:39.862684Z","iopub.status.idle":"2022-08-01T10:47:39.988138Z","shell.execute_reply.started":"2022-08-01T10:47:39.862639Z","shell.execute_reply":"2022-08-01T10:47:39.986721Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test = df_test.fillna(df_test.mean())\nX_test = df_test.drop(['id'], axis = 1)\ncategorical_cols = ['product_code', 'attribute_0', 'attribute_1', 'attribute_2',\n       'attribute_3']\nencoded_data  = pd.DataFrame(enc.transform(X_test[categorical_cols]).toarray())\nencoded_data\nX_test = X_test[ ['loading', 'measurement_0', 'measurement_1', 'measurement_2',\n       'measurement_3', 'measurement_4', 'measurement_5', 'measurement_6',\n       'measurement_7', 'measurement_8', 'measurement_9', 'measurement_10',\n       'measurement_11', 'measurement_12', 'measurement_13', 'measurement_14',\n       'measurement_15', 'measurement_16', 'measurement_17']]\nX_test = X_test.join(encoded_data)\nX_test.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-01T10:47:39.989783Z","iopub.execute_input":"2022-08-01T10:47:39.990341Z","iopub.status.idle":"2022-08-01T10:47:40.304867Z","shell.execute_reply.started":"2022-08-01T10:47:39.990305Z","shell.execute_reply":"2022-08-01T10:47:40.303511Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred = dtr_model.predict_proba(X_test)\npred","metadata":{"execution":{"iopub.status.busy":"2022-08-01T10:47:40.306328Z","iopub.execute_input":"2022-08-01T10:47:40.306689Z","iopub.status.idle":"2022-08-01T10:47:40.328762Z","shell.execute_reply.started":"2022-08-01T10:47:40.306658Z","shell.execute_reply":"2022-08-01T10:47:40.327445Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.DataFrame(pred).nunique()","metadata":{"execution":{"iopub.status.busy":"2022-08-01T10:47:40.330245Z","iopub.execute_input":"2022-08-01T10:47:40.330600Z","iopub.status.idle":"2022-08-01T10:47:40.342096Z","shell.execute_reply.started":"2022-08-01T10:47:40.330568Z","shell.execute_reply":"2022-08-01T10:47:40.340669Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Clearly overfitting as the probability values are either 0 or 1","metadata":{}},{"cell_type":"code","source":"df_test['failure'] = pred[:,1]\ndf_test[['id', 'failure']].to_csv('Submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T10:47:40.344490Z","iopub.execute_input":"2022-08-01T10:47:40.344964Z","iopub.status.idle":"2022-08-01T10:47:40.407939Z","shell.execute_reply.started":"2022-08-01T10:47:40.344917Z","shell.execute_reply":"2022-08-01T10:47:40.406746Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"[i for i in dir(dtr_model) if i[0]!='_']","metadata":{"execution":{"iopub.status.busy":"2022-08-01T11:03:40.115000Z","iopub.execute_input":"2022-08-01T11:03:40.115514Z","iopub.status.idle":"2022-08-01T11:03:40.127831Z","shell.execute_reply.started":"2022-08-01T11:03:40.115476Z","shell.execute_reply":"2022-08-01T11:03:40.126286Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dtr_model.feature_importances_","metadata":{"execution":{"iopub.status.busy":"2022-08-01T11:04:09.481427Z","iopub.execute_input":"2022-08-01T11:04:09.481821Z","iopub.status.idle":"2022-08-01T11:04:09.490229Z","shell.execute_reply.started":"2022-08-01T11:04:09.481789Z","shell.execute_reply":"2022-08-01T11:04:09.489151Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.barh(list(map(str,X.columns)),dtr_model.feature_importances_)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T11:07:57.176535Z","iopub.execute_input":"2022-08-01T11:07:57.177021Z","iopub.status.idle":"2022-08-01T11:07:57.786487Z","shell.execute_reply.started":"2022-08-01T11:07:57.176984Z","shell.execute_reply":"2022-08-01T11:07:57.784950Z"},"trusted":true},"execution_count":null,"outputs":[]}]}