{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"from IPython.core.display import HTML\nHTML(\"\"\"\n<style>\n.output_png {\n    display: table-cell;\n    text-align: center;\n    vertical-align: middle;\n    horizontal-align: middle;\n}\nh1,h2 {\n    text-align: center;\n    background-color: blue;\n    padding: 20px;\n    margin: 0;\n    color: white;\n    font-family: ariel;\n    border-radius: 80px\n}\n\nh3{\n    text-align: center;\n    border-style: solid;\n    border-width: 3px;\n    padding: 12px;\n    margin: 0;\n    color: black;\n    font-family: ariel;\n    border-radius: 80px;\n    border-color: gold;\n}\n\nbody, p {\n    font-family: ariel;\n    font-size: 15px;\n    color: charcoal;\n}\ndiv {\n    font-size: 14px;\n    margin: 0;\n\n}\n\nh4 {\n    padding: 0px;\n    margin: 0;\n    font-family: ariel;\n    color: purple;\n}\n</style>\n\"\"\")","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-08-03T15:36:00.259302Z","iopub.execute_input":"2022-08-03T15:36:00.259819Z","iopub.status.idle":"2022-08-03T15:36:00.269248Z","shell.execute_reply.started":"2022-08-03T15:36:00.259785Z","shell.execute_reply":"2022-08-03T15:36:00.268051Z"},"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nfrom IPython.display import display,HTML\nc1,c2,f1,f2,fs1,fs2=\\\n'#8F003C','#eb3446','Tourney','Smokum',45,10\ndef dhtml(string,fontcolor=c1,font=f1,fontsize=fs1):\n    display(HTML(\"\"\"<style>\n    @import 'https://fonts.googleapis.com/css?family=\"\"\"\\\n    +font+\"\"\";</style>\n    <h4 class='font-effect-3d-float' style='font-family:\"\"\"+\\\n    font+\"\"\"; color:\"\"\"+fontcolor+\"\"\"; font-size:\"\"\"+\\\n    str(fontsize)+\"\"\"px;'>%s</h4>\"\"\"%string))\n    \n    \ndhtml('🔥💥 Just Another TPS EDA + Baseline 💥🔥' )","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-08-03T15:36:00.338355Z","iopub.execute_input":"2022-08-03T15:36:00.338799Z","iopub.status.idle":"2022-08-03T15:36:00.347673Z","shell.execute_reply.started":"2022-08-03T15:36:00.338763Z","shell.execute_reply":"2022-08-03T15:36:00.346461Z"},"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Important Note regarding Public Leaderboard\n\nThis leaderboard is calculated with approximately 26% of the test data. The final results will be based on the other 74%, so we should **trust our cv scores and not try to overfit to the public leaderboard**","metadata":{}},{"cell_type":"markdown","source":"## Importing Libraries","metadata":{}},{"cell_type":"code","source":"# Basic Imports\nimport pandas as pd\nimport numpy as np\nfrom matplotlib import pyplot as plt\nimport seaborn as sns\nimport warnings\nwarnings.filterwarnings('ignore')\n\n# GroupKFold, perfect for this dataset\n# For more information on this, check out this discussion by @AmbrosM https://www.kaggle.com/competitions/tabular-playground-series-aug-2022/discussion/341070\nfrom sklearn.model_selection import GroupKFold\n\n# Imputation Imports\nfrom sklearn.experimental import enable_iterative_imputer\nfrom sklearn.impute import SimpleImputer, IterativeImputer, KNNImputer\nfrom sklearn.linear_model import LinearRegression\nfrom xgboost import XGBRegressor\n\n# Pipelining & preprocessing\nfrom sklearn.pipeline import make_pipeline,Pipeline\nfrom sklearn.compose import ColumnTransformer\nfrom sklearn.preprocessing import OneHotEncoder, StandardScaler, PowerTransformer\n\n# Model used\nfrom sklearn.linear_model import LogisticRegression\n\n# Metrics\nfrom sklearn.metrics import roc_auc_score\n\n","metadata":{"_kg_hide-input":false,"execution":{"iopub.status.busy":"2022-08-03T15:36:00.502584Z","iopub.execute_input":"2022-08-03T15:36:00.502999Z","iopub.status.idle":"2022-08-03T15:36:00.511682Z","shell.execute_reply.started":"2022-08-03T15:36:00.502965Z","shell.execute_reply":"2022-08-03T15:36:00.510056Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Loading in and looking at the data","metadata":{}},{"cell_type":"code","source":"train = pd.read_csv('../input/tabular-playground-series-aug-2022/train.csv',\n                    index_col='id')\ntest = pd.read_csv('../input/tabular-playground-series-aug-2022/test.csv',\n                    index_col='id')","metadata":{"execution":{"iopub.status.busy":"2022-08-03T15:36:00.647462Z","iopub.execute_input":"2022-08-03T15:36:00.647886Z","iopub.status.idle":"2022-08-03T15:36:00.901783Z","shell.execute_reply.started":"2022-08-03T15:36:00.647854Z","shell.execute_reply":"2022-08-03T15:36:00.900644Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-03T15:36:00.903959Z","iopub.execute_input":"2022-08-03T15:36:00.904301Z","iopub.status.idle":"2022-08-03T15:36:00.934319Z","shell.execute_reply.started":"2022-08-03T15:36:00.904272Z","shell.execute_reply":"2022-08-03T15:36:00.933154Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Basic EDA","metadata":{}},{"cell_type":"code","source":"print(f'Train data shape is {train.shape}')\nprint(f'Test data shape is {test.shape}')","metadata":{"execution":{"iopub.status.busy":"2022-08-03T15:36:00.936461Z","iopub.execute_input":"2022-08-03T15:36:00.937657Z","iopub.status.idle":"2022-08-03T15:36:00.943903Z","shell.execute_reply.started":"2022-08-03T15:36:00.937611Z","shell.execute_reply":"2022-08-03T15:36:00.942801Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Missing Values","metadata":{}},{"cell_type":"code","source":"#Missing values\nnull = pd.concat([train.isna().sum(), test.isna().sum()], axis=0).rename(\"null_count\").reset_index()\nnull[\"tt\"] = [\"train\"]*len(train.columns) + [\"test\"]*len(test.columns)\nnull = null.sort_values('null_count',ascending = True)\nnull.drop(null.loc[null['null_count']==0].index,inplace = True)\nplt.figure(figsize=(10,10))\nax = sns.barplot(data = null, y=\"index\", x=\"null_count\", hue=\"tt\", orient=\"h\",palette = ['cyan','red'])\nax.bar_label(ax.containers[0])\nax.bar_label(ax.containers[1])\nax.set_ylabel('Columns with Null values', backgroundcolor = 'yellow',labelpad = 20)\nax.set_xlabel('Number of Null Values', size = 'large',backgroundcolor = 'yellow',labelpad = 20)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-03T15:36:00.972874Z","iopub.execute_input":"2022-08-03T15:36:00.973263Z","iopub.status.idle":"2022-08-03T15:36:01.833383Z","shell.execute_reply.started":"2022-08-03T15:36:00.973231Z","shell.execute_reply":"2022-08-03T15:36:01.832500Z"},"_kg_hide-input":true,"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Insights:** \n* Null values are only there for float value columns\n* Similar percentages for train and test","metadata":{}},{"cell_type":"markdown","source":"### Looking at the Target Column - Failure","metadata":{}},{"cell_type":"code","source":"data = [100*train['failure'].value_counts()[0]/len(train['failure']),100*train['failure'].value_counts()[1]/len(train['failure'])]\nkeys = [f'Target 0',f'Target 1']\n\nplt.figure(figsize = (7,7))\nplt.title('Target Column (Failure) Distribution')\nax = plt.pie(data,labels = keys,colors = ['red','green'],explode = [0,0.1],autopct='%.0f%%')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-03T15:36:01.835046Z","iopub.execute_input":"2022-08-03T15:36:01.835561Z","iopub.status.idle":"2022-08-03T15:36:01.952548Z","shell.execute_reply.started":"2022-08-03T15:36:01.835515Z","shell.execute_reply":"2022-08-03T15:36:01.951285Z"},"_kg_hide-input":true,"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Insight:**\n* Moderately Imbalanced Data","metadata":{}},{"cell_type":"markdown","source":"### The Continuous Float Columns","metadata":{}},{"cell_type":"code","source":"other_cols = ['product_code',\"attribute_0\", \"attribute_1\", \"attribute_2\", \"attribute_3\",'measurement_1','measurement_2','measurement_0']\nnumerical_cols   = [col for col in test.columns if col not in other_cols]\n\nplt.figure(figsize=(20,20))\nplt.title('Train and Test Distributions for the Continuous Float Columns')\ndf = train.copy()\ndf_test = test.copy()\ndf['Legend'] = 'Train'\ndf_test['Legend'] = 'Test'\n\ni=1\nfor col in numerical_cols:\n    plt.subplot(4,4,i)\n    sns.histplot(data=pd.concat([df_test, df]), x=col, hue='Legend',palette = ['red','green'])\n    i=i+1","metadata":{"execution":{"iopub.status.busy":"2022-08-03T15:36:01.954423Z","iopub.execute_input":"2022-08-03T15:36:01.955181Z","iopub.status.idle":"2022-08-03T15:36:14.396349Z","shell.execute_reply.started":"2022-08-03T15:36:01.955134Z","shell.execute_reply":"2022-08-03T15:36:14.395129Z"},"_kg_hide-input":true,"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize = (15,15))\nplt.title('Feature Correlation Plot')\nsns.heatmap(df[numerical_cols].corr(), annot=True,cmap='viridis')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-03T15:36:14.399493Z","iopub.execute_input":"2022-08-03T15:36:14.400274Z","iopub.status.idle":"2022-08-03T15:36:16.336763Z","shell.execute_reply.started":"2022-08-03T15:36:14.400228Z","shell.execute_reply":"2022-08-03T15:36:16.335582Z"},"_kg_hide-input":true,"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Insights:**\n* Loading is slightly skewed so I try to apply log transformation to make the distribution more symmetric.\n* Measurement_17 is relatively strongly correlated with features measurement_4 to 9\n","metadata":{}},{"cell_type":"code","source":"plt.figure()\nplt.title('Log Transformed Loading Feature')\nax = sns.histplot(np.log(train['loading']),color = ['green'])\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-03T15:36:16.338430Z","iopub.execute_input":"2022-08-03T15:36:16.339242Z","iopub.status.idle":"2022-08-03T15:36:16.704793Z","shell.execute_reply.started":"2022-08-03T15:36:16.339206Z","shell.execute_reply":"2022-08-03T15:36:16.703585Z"},"_kg_hide-input":true,"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Categorical Columns","metadata":{}},{"cell_type":"markdown","source":"**Interesting Insight:**\n* Each Product code has a unique value for each of the 4 attributes as shown below. This is true for all product codes in both the train and test set.","metadata":{}},{"cell_type":"code","source":"both = pd.concat([train,test])\nboth[['product_code','attribute_0','attribute_1','attribute_2', 'attribute_3']].drop_duplicates().set_index('product_code')","metadata":{"execution":{"iopub.status.busy":"2022-08-03T15:36:16.706397Z","iopub.execute_input":"2022-08-03T15:36:16.707092Z","iopub.status.idle":"2022-08-03T15:36:16.753567Z","shell.execute_reply.started":"2022-08-03T15:36:16.707053Z","shell.execute_reply":"2022-08-03T15:36:16.752346Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def explore_cat(col,col1,col2,explode1,explode2):\n    data = [100*train[col].value_counts()[train[col].unique()[i]]/sum(train[col].value_counts()) for i in range(len(train[col].unique()))]\n    keys = train[col].unique().tolist()\n\n    data_test = [100*test[col].value_counts()[test[col].unique()[i]]/sum(test[col].value_counts()) for i in range(len(test[col].unique()))]\n    keys_test = test[col].unique().tolist()\n\n    plt.figure(figsize = (14,7))\n    plt.subplot(1,2,1)\n    plt.title(f'Train {col.capitalize()} Distribution')\n    ax = plt.pie(data,labels = keys,colors = col1,explode =explode1,autopct='%.0f%%')\n    plt.subplot(1,2,2)\n    plt.title(f'Test {col.capitalize()} Distribution')\n    ax = plt.pie(data_test,labels = keys_test,colors =col2,explode = explode2,autopct='%.0f%%')\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-03T15:59:40.412553Z","iopub.execute_input":"2022-08-03T15:59:40.413339Z","iopub.status.idle":"2022-08-03T15:59:40.424192Z","shell.execute_reply.started":"2022-08-03T15:59:40.413295Z","shell.execute_reply":"2022-08-03T15:59:40.422829Z"},"_kg_hide-input":true,"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"explore_cat('attribute_0',['red','green'],['green','red'],[0.1,0],[0.1,0])","metadata":{"execution":{"iopub.status.busy":"2022-08-03T16:03:02.324973Z","iopub.execute_input":"2022-08-03T16:03:02.325503Z","iopub.status.idle":"2022-08-03T16:03:02.554403Z","shell.execute_reply.started":"2022-08-03T16:03:02.325466Z","shell.execute_reply":"2022-08-03T16:03:02.552720Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"explore_cat('attribute_1',['blue','red','green'],['green','blue','red'],[0,0,0],[0,0,0])","metadata":{"execution":{"iopub.status.busy":"2022-08-03T16:04:59.089430Z","iopub.execute_input":"2022-08-03T16:04:59.090734Z","iopub.status.idle":"2022-08-03T16:04:59.335198Z","shell.execute_reply.started":"2022-08-03T16:04:59.090678Z","shell.execute_reply":"2022-08-03T16:04:59.334006Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"explore_cat('attribute_2',['green','yellow','orange','red'],['red','green','blue'],[0,0,0,0],[0,0,0])","metadata":{"execution":{"iopub.status.busy":"2022-08-03T16:07:27.063809Z","iopub.execute_input":"2022-08-03T16:07:27.064925Z","iopub.status.idle":"2022-08-03T16:07:27.281592Z","shell.execute_reply.started":"2022-08-03T16:07:27.064876Z","shell.execute_reply":"2022-08-03T16:07:27.280144Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"explore_cat('attribute_3',['green','yellow','cyan','red'],['blue','orange','red','green'],[0.1,0,0,0.1],[0,0,0.1,0.1])","metadata":{"execution":{"iopub.status.busy":"2022-08-03T16:09:23.469206Z","iopub.execute_input":"2022-08-03T16:09:23.470603Z","iopub.status.idle":"2022-08-03T16:09:23.700919Z","shell.execute_reply.started":"2022-08-03T16:09:23.470551Z","shell.execute_reply":"2022-08-03T16:09:23.698955Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Train and Test Distributions for the High Cardinality Categorical Columns","metadata":{}},{"cell_type":"code","source":"other_cols = ['measurement_0','measurement_1','measurement_2']\n\nplt.figure(figsize=(20,5))\ndf = train.copy()\ndf_test = test.copy()\ndf['Legend'] = 'Train'\ndf_test['Legend'] = 'Test'\n\ni=1\nfor col in other_cols:\n    plt.subplot(1,3,i)\n    sns.histplot(data=pd.concat([df_test, df]), x=col, hue='Legend',palette = ['red','green'])\n    i=i+1","metadata":{"execution":{"iopub.status.busy":"2022-08-03T15:41:57.149855Z","iopub.execute_input":"2022-08-03T15:41:57.150348Z","iopub.status.idle":"2022-08-03T15:41:59.164876Z","shell.execute_reply.started":"2022-08-03T15:41:57.150315Z","shell.execute_reply":"2022-08-03T15:41:59.163615Z"},"_kg_hide-input":true,"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Dropping some Categorical Variables for now","metadata":{}},{"cell_type":"code","source":"train.drop([ \"attribute_2\", \"attribute_3\"],axis = 'columns',inplace=True)\ntest.drop([  \"attribute_2\", \"attribute_3\"],axis = 'columns',inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T15:36:18.045550Z","iopub.status.idle":"2022-08-03T15:36:18.045971Z","shell.execute_reply.started":"2022-08-03T15:36:18.045766Z","shell.execute_reply":"2022-08-03T15:36:18.045785Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Partial Imputation of Measurement_17 using Highly correlated features \n(just experimenting)","metadata":{}},{"cell_type":"code","source":"impute_df = train.loc[(train['measurement_5'].isna() == False)&(train['measurement_6'].isna()==False)\n          &(train['measurement_7'].isna() == False)&(train['measurement_8'].isna() == False)\n          &(train['measurement_17'].isna() == False),\n        ['measurement_5','measurement_6','measurement_7','measurement_8','measurement_17']]\nimpute_df_target = train.loc[(train['measurement_5'].isna() == False)&(train['measurement_6'].isna()==False)\n          &(train['measurement_7'].isna() == False)&(train['measurement_8'].isna() == False)\n          &(train['measurement_17'].isna() == True),\n        ['measurement_5','measurement_6','measurement_7','measurement_8','measurement_17']]\nimpute_x = impute_df.drop('measurement_17',axis = 'columns')\nimpute_y = impute_df['measurement_17']\nimpute_model = LinearRegression()\nimpute_model.fit(impute_x,impute_y)\ntrain.loc[impute_df_target.index,'measurement_17'] = impute_model.predict(impute_df_target.drop('measurement_17',axis = 'columns'))","metadata":{"execution":{"iopub.status.busy":"2022-08-03T15:36:18.048496Z","iopub.status.idle":"2022-08-03T15:36:18.048922Z","shell.execute_reply.started":"2022-08-03T15:36:18.048725Z","shell.execute_reply":"2022-08-03T15:36:18.048745Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Preprocessing Pipelines","metadata":{}},{"cell_type":"code","source":"# inspired by THE DEVESTATOR's notebook https://www.kaggle.com/code/thedevastator/tps-aug-simple-baseline\ncategorical_cols = ['product_code',\"attribute_0\", \"attribute_1\", \"attribute_2\", \"attribute_3\"]\nnumerical_cols   = [col for col in test.columns if col not in categorical_cols]\ncategorical_cols = ['attribute_0','attribute_1']\n\n\n# Preprocessing for categorical data\nnumerical_transformer = Pipeline(steps=[\n    ('imputer', KNNImputer(n_neighbors=3)),\n    ('scaler', StandardScaler())\n])\n\n# Preprocessing for categorical data\ncategorical_transformer = Pipeline(steps=[\n    ('onehot', OneHotEncoder(handle_unknown='ignore'))\n])\n\n# Bundle preprocessing for numerical and categorical data\npreprocessor = ColumnTransformer(\n    transformers=[\n        ('num', numerical_transformer, numerical_cols),\n        ('cat', categorical_transformer, categorical_cols)\n    ])","metadata":{"execution":{"iopub.status.busy":"2022-08-03T15:36:18.050325Z","iopub.status.idle":"2022-08-03T15:36:18.050996Z","shell.execute_reply.started":"2022-08-03T15:36:18.050782Z","shell.execute_reply":"2022-08-03T15:36:18.050804Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## GroupKFold + Model","metadata":{}},{"cell_type":"code","source":"auc = []\ntest_pred = []\n# GroupKFold\nkf = GroupKFold(n_splits=5)\nfor fold, (idx_train, idx_valid) in enumerate(kf.split(train, train['failure'], train['product_code'])):\n    xtrain = train.iloc[idx_train][test.columns]\n    xvalid = train.iloc[idx_valid][test.columns]\n    ytrain = train.iloc[idx_train]['failure']\n    yvalid = train.iloc[idx_valid]['failure']\n    xtest = test.copy()\n    \n    # Log Transforming the slightly skewed feature and dropping roduct_code\n    for df in [xtrain,xvalid,xtest]:\n        df['loading'] = np.log(df['loading'])\n        df.drop('product_code',axis = 'columns',inplace = True)\n\n    features = [f for f in xtrain.columns if f != 'product_code']\n    # l1 penalty used because first simple model run revealed several low coeff features\n    model = make_pipeline(preprocessor,LogisticRegression(penalty='l1', random_state=1,solver = 'liblinear',C= 0.01,class_weight = 'balanced'))\n    model.fit(xtrain[features], ytrain)\n\n    preds_valid = model.predict_proba(xvalid[features])[:,1]\n    score = roc_auc_score(yvalid, preds_valid)\n    print(f\"Fold {fold}: auc = {score:0.5f}\")\n    auc.append(score)\n\n    test_pred.append(model.predict_proba(xtest[features])[:,1])\n\nprint(f\"Average auc = {sum(auc) / 5:.5f}\")","metadata":{"execution":{"iopub.status.busy":"2022-08-03T15:36:18.052730Z","iopub.status.idle":"2022-08-03T15:36:18.053924Z","shell.execute_reply.started":"2022-08-03T15:36:18.053714Z","shell.execute_reply":"2022-08-03T15:36:18.053736Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Submission","metadata":{}},{"cell_type":"code","source":"submission = pd.DataFrame({'id': test.index,\n                           'failure': sum(test_pred)/5})\nsubmission.to_csv('submission.csv', index=False)\nsubmission","metadata":{"execution":{"iopub.status.busy":"2022-08-03T15:36:18.054946Z","iopub.status.idle":"2022-08-03T15:36:18.055512Z","shell.execute_reply.started":"2022-08-03T15:36:18.055321Z","shell.execute_reply":"2022-08-03T15:36:18.055341Z"},"trusted":true},"execution_count":null,"outputs":[]}]}