{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-01T06:07:43.018839Z","iopub.execute_input":"2022-08-01T06:07:43.019267Z","iopub.status.idle":"2022-08-01T06:07:43.036306Z","shell.execute_reply.started":"2022-08-01T06:07:43.019166Z","shell.execute_reply":"2022-08-01T06:07:43.034579Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns\nimport plotly.io as pio\nimport plotly.express as px\nimport plotly.graph_objects as go\nimport plotly.figure_factory as ff\nfrom plotly.subplots import make_subplots\nfrom plotly.offline import init_notebook_mode, iplot","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-08-01T06:07:43.038959Z","iopub.execute_input":"2022-08-01T06:07:43.039593Z","iopub.status.idle":"2022-08-01T06:07:44.387566Z","shell.execute_reply.started":"2022-08-01T06:07:43.039534Z","shell.execute_reply":"2022-08-01T06:07:44.386551Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train= pd.read_csv('/kaggle/input/tabular-playground-series-aug-2022/train.csv', sep=',', index_col='id')\ntest= pd.read_csv('/kaggle/input/tabular-playground-series-aug-2022/test.csv', sep=',', index_col='id')\ntrain.head()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-08-01T06:07:44.388793Z","iopub.execute_input":"2022-08-01T06:07:44.389439Z","iopub.status.idle":"2022-08-01T06:07:44.616672Z","shell.execute_reply.started":"2022-08-01T06:07:44.389401Z","shell.execute_reply":"2022-08-01T06:07:44.615435Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(5,5))\nplt.pie(train['failure'].value_counts().values, colors=['darkred','blue'], labels=['0','1'])\nmy_circle=plt.Circle( (0,0), 0.7, color='white')\np=plt.gcf()\np.gca().add_artist(my_circle)\nplt.title('target_distribution', fontsize=30)\nprint(train['failure'].value_counts(normalize=True))","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-08-01T06:07:44.619560Z","iopub.execute_input":"2022-08-01T06:07:44.619939Z","iopub.status.idle":"2022-08-01T06:07:44.779216Z","shell.execute_reply.started":"2022-08-01T06:07:44.619907Z","shell.execute_reply":"2022-08-01T06:07:44.777810Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"desc = train.describe().reset_index().iloc[1:,:-1]\ndesc_t = test.describe().reset_index().iloc[1:,:]\n\nfig, ax = plt.subplots(5,5, figsize=(25,25))\n\nfig.suptitle('Train features Describe', fontsize=30)\n\nfor i,x in enumerate(desc.columns[1:]):\n    \n    \n    if i<5:\n        a=i\n        b=0\n    if i>=5 and i<10:\n        a=i-5\n        b=1\n    if i>=10 and i<15:\n        a=i-10\n        b=2\n    if i>=15 and i<20:\n        a=i-15\n        b=3\n    if i>=20 and i<25:\n        a=i-20\n        b=4\n    \n    \n    ax[b,a].plot(desc['index'],desc[x], marker='.', color='purple', label='train')\n    ax[b,a].plot(desc['index'],desc_t[x], marker='.', color='lime', label='test')\n    ax[b,a].legend(['train','test'])\n    ax[b,a].set_title(x)","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-08-01T06:07:44.781562Z","iopub.execute_input":"2022-08-01T06:07:44.782741Z","iopub.status.idle":"2022-08-01T06:07:48.158275Z","shell.execute_reply.started":"2022-08-01T06:07:44.782643Z","shell.execute_reply":"2022-08-01T06:07:48.156961Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots(5,5, figsize=(25,25))\n\nfig.suptitle('Feature Distribution, train/test df', fontsize=30)\n\nfor i,x in enumerate(desc.columns[1:]):\n    \n    \n    if i<5:\n        a=i\n        b=0\n    if i>=5 and i<10:\n        a=i-5\n        b=1\n    if i>=10 and i<15:\n        a=i-10\n        b=2\n    if i>=15 and i<20:\n        a=i-15\n        b=3\n    if i>=20 and i<25:\n        a=i-20\n        b=4\n    \n    ax[b,a].hist(train[x], color='purple', label='train', alpha=0.8)\n    ax[b,a].hist(test[x], color='lime', label='test', alpha=0.8)\n    ax[b,a].legend(['train','test'])\n    ax[b,a].set_title(x)","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-08-01T06:07:48.159945Z","iopub.execute_input":"2022-08-01T06:07:48.160635Z","iopub.status.idle":"2022-08-01T06:07:52.514512Z","shell.execute_reply.started":"2022-08-01T06:07:48.160593Z","shell.execute_reply":"2022-08-01T06:07:52.513121Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cols=train.columns[:-1]\nunique_df = pd.DataFrame(train[train.columns[:-1]]).nunique().reset_index()\nunique_df.columns=['features','count']\n\nfig1 = px.bar(unique_df, y='count', x=cols)\n\nfig1.update_layout(title='Feature cardinality in train set',\n                  xaxis_title='features',\n                  yaxis_title='# unique values',\n                  titlefont={'size': 28, 'family':'Serif'},\n                  template='simple_white',\n                  showlegend=True,\n                  width=1000, height=500)\nfig1.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-08-01T06:07:52.516484Z","iopub.execute_input":"2022-08-01T06:07:52.516993Z","iopub.status.idle":"2022-08-01T06:07:53.107561Z","shell.execute_reply.started":"2022-08-01T06:07:52.516955Z","shell.execute_reply":"2022-08-01T06:07:53.106160Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"unique_df = pd.DataFrame(test[test.columns]).nunique().reset_index()\nunique_df.columns=['features','count']\n\nfig1 = px.bar(unique_df, y='count', x=cols)\n\nfig1.update_layout(title='Feature cardinality in test set',\n                  xaxis_title='features',\n                  yaxis_title='# unique values',\n                  titlefont={'size': 28, 'family':'Serif'},\n                  template='simple_white',\n                  showlegend=True,\n                  width=1000, height=500)\nfig1.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-08-01T06:07:53.109745Z","iopub.execute_input":"2022-08-01T06:07:53.110242Z","iopub.status.idle":"2022-08-01T06:07:53.219667Z","shell.execute_reply.started":"2022-08-01T06:07:53.110196Z","shell.execute_reply":"2022-08-01T06:07:53.218369Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"desc = train[train['failure']==0].describe().reset_index().iloc[1:,:]\ndesc_t = train[train['failure']==1].describe().reset_index().iloc[1:,:]\n\nfig, ax = plt.subplots(5,5, figsize=(22,22))\n\nfig.suptitle('Train class label describe', fontsize=30)\n\nfor i,x in enumerate(desc.columns[1:-1]):\n    \n    if i<5:\n        a=i\n        b=0\n    if i>=5 and i<10:\n        a=i-5\n        b=1\n    if i>=10 and i<15:\n        a=i-10\n        b=2\n    if i>=15 and i<20:\n        a=i-15\n        b=3\n    if i>=20 and i<25:\n        a=i-20\n        b=4\n    \n    ax[b,a].plot(desc['index'],desc[x], marker='.', color='blue', label='train_0')\n    ax[b,a].plot(desc['index'],desc_t[x], marker='.', color='darkred', label='train_1')\n    ax[b,a].legend(['train_0','train_1'])\n    ax[b,a].set_title(x)","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-08-01T06:07:53.221394Z","iopub.execute_input":"2022-08-01T06:07:53.221802Z","iopub.status.idle":"2022-08-01T06:07:56.546755Z","shell.execute_reply.started":"2022-08-01T06:07:53.221766Z","shell.execute_reply":"2022-08-01T06:07:56.545486Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"t0 = train[train['failure']==0]\nt1 = train[train['failure']==1]\n\nfig, ax = plt.subplots(5,5, figsize=(25,25))\n\nfig.suptitle('Feature Distribution, train/test df', fontsize=30)\n\nfor i,x in enumerate(desc.columns[1:]):\n    \n    \n    if i<5:\n        a=i\n        b=0\n    if i>=5 and i<10:\n        a=i-5\n        b=1\n    if i>=10 and i<15:\n        a=i-10\n        b=2\n    if i>=15 and i<20:\n        a=i-15\n        b=3\n    if i>=20 and i<25:\n        a=i-20\n        b=4\n    \n    ax[b,a].hist(t0[x], color='blue', label='train', alpha=0.8)\n    ax[b,a].hist(t1[x], color='darkred', label='test', alpha=0.8)\n    ax[b,a].legend(['train0','train1'])\n    ax[b,a].set_title(x)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T06:09:49.684544Z","iopub.execute_input":"2022-08-01T06:09:49.685050Z","iopub.status.idle":"2022-08-01T06:09:54.199082Z","shell.execute_reply.started":"2022-08-01T06:09:49.685014Z","shell.execute_reply":"2022-08-01T06:09:54.198113Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(20,5))\n_=plt.plot(t0.isna().sum()[0:-1], color='blue')\n_=plt.plot(t1.isna().sum()[0:-1], color='darkred')\n_=plt.title('Train: NA by feature/class')\n_=plt.legend(['train_0','train_1'])\n_=plt.xticks(rotation=90)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T06:25:27.181529Z","iopub.execute_input":"2022-08-01T06:25:27.182126Z","iopub.status.idle":"2022-08-01T06:25:27.537038Z","shell.execute_reply.started":"2022-08-01T06:25:27.182079Z","shell.execute_reply":"2022-08-01T06:25:27.535785Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(20,5))\n_=plt.plot(t0.mean(numeric_only=True)[0:-1], color='blue')\n_=plt.plot(t1.mean(numeric_only=True)[0:-1], color='darkred')\n_=plt.title('Train: mean by feature/class')\n_=plt.legend(['train_0','train_1'])\n_=plt.xticks(rotation=90)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T06:18:16.718907Z","iopub.execute_input":"2022-08-01T06:18:16.719327Z","iopub.status.idle":"2022-08-01T06:18:17.001423Z","shell.execute_reply.started":"2022-08-01T06:18:16.719295Z","shell.execute_reply":"2022-08-01T06:18:16.999937Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(20,5))\n_=plt.plot(t0.median(numeric_only=True)[0:-1], color='blue')\n_=plt.plot(t1.median(numeric_only=True)[0:-1], color='darkred')\n_=plt.title('Train: median by feature/class')\n_=plt.legend(['train_0','train_1'])\n_=plt.xticks(rotation=90)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T06:20:00.048125Z","iopub.execute_input":"2022-08-01T06:20:00.048648Z","iopub.status.idle":"2022-08-01T06:20:00.345680Z","shell.execute_reply.started":"2022-08-01T06:20:00.048611Z","shell.execute_reply":"2022-08-01T06:20:00.343997Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(20,5))\n_=plt.plot(t0.std(numeric_only=True)[0:-1], color='blue')\n_=plt.plot(t1.std(numeric_only=True)[0:-1], color='darkred')\n_=plt.title('Train: std by feature/class')\n_=plt.legend(['train_0','train_1'])\n_=plt.xticks(rotation=90)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T06:20:58.433546Z","iopub.execute_input":"2022-08-01T06:20:58.434061Z","iopub.status.idle":"2022-08-01T06:20:58.714998Z","shell.execute_reply.started":"2022-08-01T06:20:58.434024Z","shell.execute_reply":"2022-08-01T06:20:58.713290Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(20,5))\n_=plt.plot(t0.skew(numeric_only=True)[0:-1], color='blue')\n_=plt.plot(t1.skew(numeric_only=True)[0:-1], color='darkred')\n_=plt.title('Train: skewness by feature/class')\n_=plt.legend(['train_0','train_1'])\n_=plt.xticks(rotation=90)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T06:21:37.246851Z","iopub.execute_input":"2022-08-01T06:21:37.247285Z","iopub.status.idle":"2022-08-01T06:21:37.532598Z","shell.execute_reply.started":"2022-08-01T06:21:37.247250Z","shell.execute_reply":"2022-08-01T06:21:37.531464Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(20,5))\n_=plt.plot(t0.kurt(numeric_only=True)[0:-1], color='blue')\n_=plt.plot(t1.kurt(numeric_only=True)[0:-1], color='darkred')\n_=plt.title('Train: kurtosis by feature/class')\n_=plt.legend(['train_0','train_1'])\n_=plt.xticks(rotation=90)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T06:22:49.083653Z","iopub.execute_input":"2022-08-01T06:22:49.084083Z","iopub.status.idle":"2022-08-01T06:22:49.374064Z","shell.execute_reply.started":"2022-08-01T06:22:49.084051Z","shell.execute_reply":"2022-08-01T06:22:49.372860Z"},"trusted":true},"execution_count":null,"outputs":[]}]}