{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-27T12:17:09.642727Z","iopub.execute_input":"2022-07-27T12:17:09.643551Z","iopub.status.idle":"2022-07-27T12:17:09.654600Z","shell.execute_reply.started":"2022-07-27T12:17:09.643501Z","shell.execute_reply":"2022-07-27T12:17:09.653406Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv(\"../input/tabular-playground-series-feb-2022/train.csv\")\ntest = pd.read_csv(\"../input/tabular-playground-series-feb-2022/test.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-07-27T12:16:24.954197Z","iopub.execute_input":"2022-07-27T12:16:24.954613Z","iopub.status.idle":"2022-07-27T12:17:09.640598Z","shell.execute_reply.started":"2022-07-27T12:16:24.954581Z","shell.execute_reply":"2022-07-27T12:17:09.639449Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = train.sample(10000)","metadata":{"execution":{"iopub.status.busy":"2022-07-27T12:17:20.225862Z","iopub.execute_input":"2022-07-27T12:17:20.226260Z","iopub.status.idle":"2022-07-27T12:17:20.276293Z","shell.execute_reply.started":"2022-07-27T12:17:20.226227Z","shell.execute_reply":"2022-07-27T12:17:20.275155Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Variable Description\ndef description(df):\n    print(f\"Dataset Shape: {df.shape}\")\n    summary = pd.DataFrame(df.dtypes,columns=['dtypes'])\n    summary = summary.reset_index()\n    summary['Name'] = summary['index']\n    summary = summary[['Name','dtypes']]\n    summary['Missing'] = df.isnull().sum().values    \n    summary['Uniques'] = df.nunique().values\n    summary['First Value'] = df.iloc[0].values\n    summary['Second Value'] = df.iloc[1].values\n    summary['Third Value'] = df.iloc[2].values\n    return summary\n\ndescription(df)","metadata":{"execution":{"iopub.status.busy":"2022-07-27T12:17:25.697843Z","iopub.execute_input":"2022-07-27T12:17:25.698870Z","iopub.status.idle":"2022-07-27T12:17:25.820027Z","shell.execute_reply.started":"2022-07-27T12:17:25.698830Z","shell.execute_reply":"2022-07-27T12:17:25.818888Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 1. Statistical analysis\n","metadata":{}},{"cell_type":"code","source":"train.drop([\"row_id\", \"target\"], axis = 1 , inplace = True)\ntest.drop([\"row_id\"] , axis = 1 , inplace = True)","metadata":{"execution":{"iopub.status.busy":"2022-07-27T12:18:20.415655Z","iopub.execute_input":"2022-07-27T12:18:20.416079Z","iopub.status.idle":"2022-07-27T12:18:20.669391Z","shell.execute_reply.started":"2022-07-27T12:18:20.416046Z","shell.execute_reply":"2022-07-27T12:18:20.668596Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"FEATURES = [col for col in train.columns]","metadata":{"execution":{"iopub.status.busy":"2022-07-27T12:18:35.323404Z","iopub.execute_input":"2022-07-27T12:18:35.323826Z","iopub.status.idle":"2022-07-27T12:18:35.328913Z","shell.execute_reply.started":"2022-07-27T12:18:35.323779Z","shell.execute_reply":"2022-07-27T12:18:35.327961Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.describe().T.sort_values(by='std' , ascending = False)\\\n                     .style.background_gradient(cmap='Pastel1')\\\n                     .bar(subset=[\"max\"], color='#7CAE00')\\\n                     .bar(subset=[\"mean\",], color='#F8766D')\\\n                     .bar(subset=[\"std\",], color='#00BFC4')","metadata":{"execution":{"iopub.status.busy":"2022-07-27T12:18:52.181933Z","iopub.execute_input":"2022-07-27T12:18:52.183865Z","iopub.status.idle":"2022-07-27T12:18:56.559744Z","shell.execute_reply.started":"2022-07-27T12:18:52.183776Z","shell.execute_reply":"2022-07-27T12:18:56.558875Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 2. Numerical feature distributions","metadata":{}},{"cell_type":"code","source":"df = pd.concat([train[FEATURES], test[FEATURES]], axis=0)\n\ncat_features = [col for col in FEATURES if df[col].nunique() < 25]\ncont_features = [col for col in FEATURES if df[col].nunique() >= 25]","metadata":{"execution":{"iopub.status.busy":"2022-07-27T12:20:06.421281Z","iopub.execute_input":"2022-07-27T12:20:06.421730Z","iopub.status.idle":"2022-07-27T12:20:09.308026Z","shell.execute_reply.started":"2022-07-27T12:20:06.421691Z","shell.execute_reply":"2022-07-27T12:20:09.306895Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns\n\nncols = 5\nnrows = 10\nn_features = cont_features[:10]\nfig, axes = plt.subplots(nrows, ncols, figsize=(25, 50))\n\nfor r in range(nrows):\n    for c in range(ncols):\n        col = n_features[ncols]\n        sns.kdeplot(x=train[col], ax=axes[r, c], color='#F8766D', label='Train data' , fill =True)\n        sns.kdeplot(x=test[col], ax=axes[r, c], color='#7CAE00', label='Test data', fill =True)\n        axes[r,c].legend()\n        axes[r, c].set_ylabel('')\n        axes[r, c].set_xlabel(col, fontsize=10)\n        axes[r, c].tick_params(labelsize=7, width=2)\n        axes[r, c].xaxis.offsetText.set_fontsize(10)\n        axes[r, c].yaxis.offsetText.set_fontsize(10)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-27T12:21:47.820660Z","iopub.execute_input":"2022-07-27T12:21:47.821095Z","iopub.status.idle":"2022-07-27T12:22:55.856396Z","shell.execute_reply.started":"2022-07-27T12:21:47.821060Z","shell.execute_reply":"2022-07-27T12:22:55.855497Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 3. Target Distribution","metadata":{}},{"cell_type":"code","source":"train = pd.read_csv(\"../input/tabular-playground-series-feb-2022/train.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-07-27T12:23:12.455842Z","iopub.execute_input":"2022-07-27T12:23:12.456931Z","iopub.status.idle":"2022-07-27T12:23:31.576746Z","shell.execute_reply.started":"2022-07-27T12:23:12.456885Z","shell.execute_reply":"2022-07-27T12:23:31.575540Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pie, ax = plt.subplots(figsize=[18,12])\ntrain.groupby('target').size().plot(kind='pie',autopct='%0.9f',title='Target', ylabel='', startangle=90, labeldistance=1.15, wedgeprops = { 'linewidth' : 3, 'edgecolor' : 'white' })","metadata":{"execution":{"iopub.status.busy":"2022-07-27T12:23:31.578524Z","iopub.execute_input":"2022-07-27T12:23:31.578963Z","iopub.status.idle":"2022-07-27T12:23:31.896861Z","shell.execute_reply.started":"2022-07-27T12:23:31.578933Z","shell.execute_reply":"2022-07-27T12:23:31.895665Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 4. Clusters analysis using tsne","metadata":{}},{"cell_type":"code","source":"cols = train.columns.to_list()\ncols.remove('target')","metadata":{"execution":{"iopub.status.busy":"2022-07-27T12:24:35.752780Z","iopub.execute_input":"2022-07-27T12:24:35.753184Z","iopub.status.idle":"2022-07-27T12:24:35.758505Z","shell.execute_reply.started":"2022-07-27T12:24:35.753153Z","shell.execute_reply":"2022-07-27T12:24:35.757571Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.manifold import TSNE\nfrom sklearn.preprocessing import StandardScaler","metadata":{"execution":{"iopub.status.busy":"2022-07-27T12:27:20.154031Z","iopub.execute_input":"2022-07-27T12:27:20.154465Z","iopub.status.idle":"2022-07-27T12:27:20.160205Z","shell.execute_reply.started":"2022-07-27T12:27:20.154431Z","shell.execute_reply":"2022-07-27T12:27:20.159129Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_subset = train.sample(10000, random_state= 42)\n\ntsne = TSNE(n_components=2, random_state=0, perplexity= 25, n_iter=3000)\ntransformed_data = tsne.fit_transform(StandardScaler().fit_transform(train_subset[cols].values))","metadata":{"execution":{"iopub.status.busy":"2022-07-27T12:27:22.020204Z","iopub.execute_input":"2022-07-27T12:27:22.020604Z","iopub.status.idle":"2022-07-27T12:30:19.787725Z","shell.execute_reply.started":"2022-07-27T12:27:22.020569Z","shell.execute_reply":"2022-07-27T12:30:19.786764Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tsne_data = np.vstack((transformed_data.T, train_subset.target)).T\n\ntsne_df = pd.DataFrame(data=tsne_data, columns=(\"X\", \"Y\", \"target\"))\n\nfig = sns.FacetGrid(tsne_df, hue=\"target\", height=10).map(plt.scatter, 'X', 'Y').add_legend(fontsize = 12)\nplt.title('Perplexity= 25, n_iter=3000, n_sample = 10000', fontsize = 17)\nleg = ax.legend(fontsize = 18)\n\n\n#plt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-27T12:30:20.679006Z","iopub.execute_input":"2022-07-27T12:30:20.679416Z","iopub.status.idle":"2022-07-27T12:30:21.547682Z","shell.execute_reply.started":"2022-07-27T12:30:20.679379Z","shell.execute_reply":"2022-07-27T12:30:21.546547Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig.savefig('tsne.png', dpi=200) ","metadata":{"execution":{"iopub.status.busy":"2022-07-27T12:30:21.549494Z","iopub.execute_input":"2022-07-27T12:30:21.549782Z","iopub.status.idle":"2022-07-27T12:30:22.383032Z","shell.execute_reply.started":"2022-07-27T12:30:21.549755Z","shell.execute_reply":"2022-07-27T12:30:22.382060Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 5. Clusters analysis using LDA","metadata":{}},{"cell_type":"code","source":"train = pd.read_csv(\"../input/tabular-playground-series-feb-2022/train.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-07-27T12:34:06.411230Z","iopub.execute_input":"2022-07-27T12:34:06.411697Z","iopub.status.idle":"2022-07-27T12:34:23.078182Z","shell.execute_reply.started":"2022-07-27T12:34:06.411661Z","shell.execute_reply":"2022-07-27T12:34:23.076980Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.discriminant_analysis import LinearDiscriminantAnalysis as LDA\n\ntrain_sub = train.sample(50000, random_state= 42)\nlda_data = LDA(n_components=3).fit_transform(train_sub.drop(columns='target'),train_sub.target)\nplt.figure(figsize=(15,12))\nsns.scatterplot(x = lda_data[:, 0], y = lda_data[:, 1], hue = 'target', data=train_sub)\nplt.savefig('LDA4.png', dpi=200) \nplt.title('n= 3, n_sample = 50000')","metadata":{"execution":{"iopub.status.busy":"2022-07-27T12:34:23.080316Z","iopub.execute_input":"2022-07-27T12:34:23.081563Z","iopub.status.idle":"2022-07-27T12:34:32.208436Z","shell.execute_reply.started":"2022-07-27T12:34:23.081514Z","shell.execute_reply":"2022-07-27T12:34:32.207625Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install jovian\nimport jovian","metadata":{"execution":{"iopub.status.busy":"2022-07-27T12:46:24.072954Z","iopub.execute_input":"2022-07-27T12:46:24.073402Z","iopub.status.idle":"2022-07-27T12:46:40.011738Z","shell.execute_reply.started":"2022-07-27T12:46:24.073365Z","shell.execute_reply":"2022-07-27T12:46:40.010166Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"jovian.commit(project='Data science exploratory analysis code used by data engineers')","metadata":{"execution":{"iopub.status.busy":"2022-07-27T12:49:34.718638Z","iopub.execute_input":"2022-07-27T12:49:34.719077Z","iopub.status.idle":"2022-07-27T12:49:53.857111Z","shell.execute_reply.started":"2022-07-27T12:49:34.719040Z","shell.execute_reply":"2022-07-27T12:49:53.855341Z"},"trusted":true},"execution_count":null,"outputs":[]}]}