{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"##Import Libraries","metadata":{"id":"r-6rqO5Ec4C_"}},{"cell_type":"markdown","source":"### Table of Contents : \n\n   * [Import Libraries](#sec1)\n   * [Import CSV Files](#sec2)\n   * [Basic Data Information & EDA](#sec3)\n   * [Data Preprocessing](#sec4)\n   * [PCA](#sec5)\n   * [Train-Test Split](#sec6)","metadata":{}},{"cell_type":"markdown","source":"# Import Libraries <a class=\"anchor\" id=\"sec1\"></a>","metadata":{}},{"cell_type":"code","source":"#importing libraries\nimport warnings\nwarnings.filterwarnings('ignore')\nimport pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns","metadata":{"id":"4OKHsGyrc7lv","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.set_option('display.max_columns', None)\npd.set_option('display.max_rows', None)\npd.set_option('display.expand_frame_repr', False)\npd.set_option('max_colwidth', -1)","metadata":{"id":"_eRibkpAdKoO","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Import CSV Files <a class=\"anchor\" id=\"sec2\"></a>","metadata":{"id":"wvZd3Zu5dctw"}},{"cell_type":"code","source":"df=pd.read_csv(\"../input/tabular-playground-series-aug-2021/train.csv\")","metadata":{"id":"brUVL06Zdixq","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head()","metadata":{"id":"U7XYJpIodtjN","outputId":"3020c6a4-c52e-4453-8b7a-8a78c3a62ad7","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.select_dtypes(include=['int64']).nunique().sort_values(ascending=True)","metadata":{"id":"IQZVmGcNOwA0","outputId":"1715b687-9d8e-4b57-aae1-ca4c4596d45a","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.select_dtypes(include=['float']).nunique().sort_values(ascending=True)","metadata":{"id":"ELksXwRPftBx","outputId":"68aa655a-1270-4e54-ed1b-2b48c611770b","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Basic Data Information & EDA <a class=\"anchor\" id=\"sec3\"></a>","metadata":{"id":"y5HpA_bhLYPS"}},{"cell_type":"code","source":"df.columns","metadata":{"id":"cqX6Jub-T4Sb","outputId":"58f06506-6d08-4df9-eb3a-85edb1e3d5a8","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.shape","metadata":{"id":"eu4TZbCQfLJV","outputId":"1d776004-456e-4707-d865-ba6aae479625","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.describe()","metadata":{"id":"tmQ9BuDodvoh","outputId":"2b9e2b7c-59ec-4434-e9a2-f8a82b7ad3c3","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.isnull().sum()","metadata":{"id":"iTkpBFiif5Ww","outputId":"538f4d8a-f1c5-4d09-c7a0-02df45538b74","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Hence there is no missing values in the data.","metadata":{"id":"UtmQMXQMLj_k"}},{"cell_type":"code","source":"for i in range(1,99):\n    sns.distplot(df.iloc[:,i])\n    plt.show()","metadata":{"id":"gGMEPVUggH1r","outputId":"ef36fc20-aff5-4154-a80f-87b5afe759bf","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.distplot(df.iloc[:,3])","metadata":{"id":"-4LbYHVrD2Bc","outputId":"3d1fabe2-153e-4d3d-ff37-d30b318d02db","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in range(1,99):\n    sns.boxplot(df.iloc[:,i])\n    plt.show()","metadata":{"id":"yRlJt6KSCExr","outputId":"64759992-527d-4f33-9b6b-de7a7befe6ab","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The data has outliers and also it is not normally distributed","metadata":{"id":"hltDpo0LgZ0l"}},{"cell_type":"markdown","source":"## Data Preprocessing <a class=\"anchor\" id=\"sec4\"></a>","metadata":{"id":"gOFdHWYemQRs"}},{"cell_type":"code","source":"from sklearn.preprocessing import RobustScaler\nr=RobustScaler()\ndf_r=r.fit_transform(df)","metadata":{"id":"ZxpwlSW-mTSo","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_r","metadata":{"id":"GpUq_Srw6ERK","outputId":"18cc4440-cefc-4463-813b-122b465a577e","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_r=pd.DataFrame(df_r)\ndf_r.head()","metadata":{"id":"wKniAZco6fMZ","outputId":"6c63df4c-bdfc-4d20-cd03-fdaa2c6719f8","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in range(1,99):\n    sns.distplot(df_r.iloc[:,i])\n    plt.show()","metadata":{"id":"mr-PwvM4rDjS","outputId":"91adb558-4975-4447-cdde-09c1a7142300","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in range(1,99):\n    sns.boxplot(df_r.iloc[:,i])\n    plt.show()","metadata":{"id":"yMxQGG6lrEgX","outputId":"03914925-c22d-4be9-c2d4-fc51c06258f6","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import StandardScaler\nobject = StandardScaler()\ndf_s=object.fit_transform(df)","metadata":{"id":"TzZQE80-mLiw","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_s=pd.DataFrame(df_s)","metadata":{"id":"aWNDJP0D805l","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_s.head()","metadata":{"id":"Y0r9D5XPgtoq","outputId":"19c7822f-d511-45b9-c990-20f86255d074","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in range(1,99):\n    sns.distplot(df_s.iloc[:,i])\n    plt.show()","metadata":{"id":"T31-QDiK8nhV","outputId":"b8f0db98-a01a-46fb-ffd3-328c63cc0827","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in range(1,99):\n    sns.boxplot(df_r.iloc[:,i])\n    plt.show()","metadata":{"id":"BFsT-N8R8u0A","outputId":"bf8052aa-2b75-4f90-8b33-f30ca95a2187","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Hence, neither robust scaler nor standard scaler was useful in this dataset.","metadata":{"id":"-0SJfdH2--dt"}},{"cell_type":"code","source":"#PCA","metadata":{"id":"neNSj0vKgOId","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head()","metadata":{"id":"gENBoDPrg7K_","outputId":"fb22719c-023b-4e15-d31b-b2603bf1a1a3","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#\n# Scale the dataset; \n#\nfrom sklearn.preprocessing import StandardScaler\nsc = StandardScaler()","metadata":{"id":"j8NLL-A2hEfX","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"a=df.drop('loss',axis=1).values\nb=df['loss'].values","metadata":{"id":"YfRfUkn1gPrS","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sc.fit(a)","metadata":{"id":"hkE-JLEphsy1","outputId":"f6414abc-cf6b-4fae-ced5-57a377660677","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"a_std = sc.transform(a)","metadata":{"id":"u1ngPRDyibwW","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.decomposition import PCA\npca=PCA()","metadata":{"id":"sXNVFazjjFmq","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"a_pca = pca.fit_transform(a_std)","metadata":{"id":"qLRq-lauaD5M","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Determine explained variance using explained_variance_ration_ attribute\n#\nexp_var_pca = pca.explained_variance_ratio_","metadata":{"id":"jQ9rITs3aV1K","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"exp_var_pca","metadata":{"id":"vEFQ8s6waXW6","outputId":"00204732-1d77-4129-9b6a-3381cdcb4e6d","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Cumulative sum of eigenvalues; This will be used to create step plot\n# for visualizing the variance explained by each principal component.\n#\ncum_sum_eigenvalues = np.cumsum(exp_var_pca)","metadata":{"id":"9Dau4SntaeCg","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cum_sum_eigenvalues","metadata":{"id":"J0_Ie-28amXc","outputId":"51c8c268-bd81-4b50-c3b5-a05f359b4445","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.bar(range(0,len(exp_var_pca)), exp_var_pca, alpha=0.5, align='center', label='Individual explained variance')\nplt.step(range(0,len(cum_sum_eigenvalues)), cum_sum_eigenvalues, where='mid',label='Cumulative explained variance')\nplt.ylabel('Explained variance ratio')\nplt.xlabel('Principal component index')\nplt.legend(loc='best')\nplt.tight_layout()\nplt.show()","metadata":{"id":"QyZ0kmK6apCi","outputId":"b4d48989-35df-462a-d6c0-1f8a49b5c02d","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## PCA <a class=\"anchor\" id=\"sec5\"></a>","metadata":{"id":"0TnGf_LaOFe3"}},{"cell_type":"code","source":"from sklearn.decomposition import PCA\npca=PCA(n_components=80,random_state=1000)","metadata":{"id":"G4LqrXyYOIqW","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"a_pca_80=pca.fit_transform(a_std)","metadata":{"id":"NirGFU4fQ5KZ","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Determine explained variance using explained_variance_ration_ attribute\n#\nexp_var_pca = pca.explained_variance_ratio_","metadata":{"id":"6QwxIy8ARR2O","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"exp_var_pca ","metadata":{"id":"zHhdRtV8RUP4","outputId":"30b4a917-757c-4c12-a98f-1ed6bb5f12ff","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Cumulative sum of eigenvalues; This will be used to create step plot\n# for visualizing the variance explained by each principal component.\n#\ncum_sum_eigenvalues = np.cumsum(exp_var_pca)","metadata":{"id":"x17uIH_jRco_","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cum_sum_eigenvalues","metadata":{"id":"4w3lLumhRf7F","outputId":"ea4458e9-a5aa-431c-d662-21a7af2fc18f","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.bar(range(0,len(exp_var_pca)), exp_var_pca, alpha=0.5, align='center', label='Individual explained variance')\nplt.step(range(0,len(cum_sum_eigenvalues)), cum_sum_eigenvalues, where='mid',label='Cumulative explained variance')\nplt.ylabel('Explained variance ratio')\nplt.xlabel('Principal component index')\nplt.legend(loc='best')\nplt.tight_layout()\nplt.show()","metadata":{"id":"xTLil0SJRqNd","outputId":"b0b1ae83-b870-45a2-bec2-2761c4c44c9f","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_new=pd.DataFrame(a_pca_80,columns=['f1','f2','f3','f4','f5','f6','f7','f8','f9','f10','f11','f12','f13','f14','f15','f16','f17','f18','f19','f20','f21','f22','f23','f24','f25','f26','f27','f28','f29','f30','f31','f32','f33','f34','f35','f36','f37','f38','f39','f40','f41','f42','f43','f44','f45','f46','f47','f48','f49','f50','f51','f52','f53','f54','f55','f56','f57','f58','f59','f60','f61','f62',\n                                      'f63','f64','f65','f66','f67','f68','f69','f70','f71','f72','f73','f74','f75','f76','f77','f78','f79','f80'])","metadata":{"id":"juPj2JvEUL3w","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_new['loss']=b","metadata":{"id":"324iNXqIXj8R","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_new.head()","metadata":{"id":"gfb4N9VnYHXD","outputId":"dd766ddf-e285-45c8-982d-38c932bbd16e","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in range(0,79):\n    sns.distplot(df_new.iloc[:,i])\n    plt.show()","metadata":{"id":"zO_gvmH1YPa5","outputId":"45e538ca-b0c9-4efc-f483-16b5e222c510","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"##PCA","metadata":{"id":"ap4uXPg7sz2H"}},{"cell_type":"code","source":"df_new1=pd.DataFrame(a_pca,columns=['f0','f1','f2','f3','f4','f5','f6','f7','f8','f9','f10','f11','f12','f13','f14','f15','f16','f17','f18','f19','f20','f21','f22','f23','f24','f25','f26','f27','f28','f29','f30','f31','f32','f33','f34','f35','f36','f37','f38','f39','f40','f41','f42','f43','f44','f45','f46','f47','f48','f49','f50','f51','f52','f53','f54','f55','f56','f57','f58','f59','f60','f61','f62',\n                                      'f63','f64','f65','f66','f67','f68','f69','f70','f71','f72','f73','f74','f75','f76','f77','f78','f79','f80','f81','f82','f83','f84','f85','f86','f87','f88','f89','f90','f91','f92','f93','f94','f95','f96','f97','f98','f99','loss'])","metadata":{"id":"_Iac1jNRs1KE","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_new1.head()","metadata":{"id":"3knGHzXWtoIT","outputId":"b5c1a5f5-0d5b-4a51-8de7-6ec8027a942c","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in range(0,100):\n    sns.distplot(df_new1.iloc[:,i])\n    plt.show()","metadata":{"id":"4L_CQWqcuLEu","outputId":"c652ffcb-8405-4607-8df7-487c6821ff9c","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Train-Test Split <a class=\"anchor\" id=\"sec6\"></a>","metadata":{"id":"Bx5ngnYIMxxQ"}},{"cell_type":"code","source":"x = df_new.drop(\"loss\",axis=1)\ny = df_new[[\"loss\"]]\nfrom sklearn.model_selection import train_test_split\nx_train,x_test,y_train,y_test = train_test_split(x,y,test_size=0.20,random_state=49)\nprint(\"x_train :\",x_train.shape)\nprint(\"x_test :\",x_test.shape)\nprint(\"y_train :\",y_train.shape)\nprint(\"y_test :\",y_test.shape)","metadata":{"id":"1hKKKBcXM15B","outputId":"d8e282de-49ae-4c74-f201-e4af2711e356","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Linear Regression","metadata":{"id":"VTu4q0gANooz","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.linear_model import LinearRegression\nmodel = LinearRegression()\nmodel.fit(x_train,y_train)","metadata":{"id":"hQ8dYPsAOiqo","outputId":"79ffa5f9-9b9b-485d-f1bf-384f791da73a","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds = model.predict(x_test)","metadata":{"id":"tSGcdZMFOlng","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import mean_squared_error, r2_score\nmean_squared_error(y_test,preds)\nr2_score(y_test,preds)","metadata":{"id":"DXpEPVwoPEF-","outputId":"20c42b35-9a6d-4fb6-8d98-985fdbeb3428","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"##Random Forest","metadata":{"id":"6pQGVo9-PJE7","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.ensemble import RandomForestRegressor\nR_model = RandomForestRegressor()\nR_model.fit(x_train,y_train)\npreds = R_model.predict(x_test)","metadata":{"id":"TVYJawilPb86","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import r2_score\nr = r2_score(y_test,preds)\nprint(\"R2score when we predict using Randomn forest is \",r)","metadata":{"id":"a-5RoJqaPo2M","outputId":"afac0fc4-aecd-4f92-93ac-0471eaa3b374","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#Train-Test Split2","metadata":{"id":"JUCFaiOoudIA"}},{"cell_type":"code","source":"x = df_new1.drop(\"loss\",axis=1)\ny = df_new1[[\"loss\"]]\nfrom sklearn.model_selection import train_test_split\nx_train,x_test,y_train,y_test = train_test_split(x,y,test_size=0.20,random_state=49)\nprint(\"x_train :\",x_train.shape)\nprint(\"x_test :\",x_test.shape)\nprint(\"y_train :\",y_train.shape)\nprint(\"y_test :\",y_test.shape)","metadata":{"id":"McOmaa8-ueCt","outputId":"8f42e03b-78b8-4298-e2ff-2163ba3ba36a","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Linear Regression","metadata":{"id":"MHqF17YgunPj","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.linear_model import LinearRegression\nmodel = LinearRegression()\nmodel.fit(x_train,y_train)","metadata":{"id":"b_1nxX3vuqRF","outputId":"652a8f19-4c94-49eb-8059-b9e94ae73e17","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds = model.predict(x_test)","metadata":{"id":"PI0l6H4kutjo","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import mean_squared_error, r2_score\nmean_squared_error(y_test,preds)\nr2_score(y_test,preds)","metadata":{"id":"5DnGM22Su3JM","outputId":"6869dd21-410c-4c23-ec0a-d2258301b149","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"##Random Forest","metadata":{"id":"eNBAheCLu3ug","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.ensemble import RandomForestRegressor\nR_model = RandomForestRegressor()\nR_model.fit(x_train,y_train)\npreds = R_model.predict(x_test)","metadata":{"id":"2-yUlfG-u7L1","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import r2_score\nr = r2_score(y_test,preds)\nprint(\"R2score when we predict using Randomn forest is \",r)","metadata":{"id":"niaC5Mmyu-Pc","outputId":"e09edb25-2277-4c73-dca5-204d63dda0b0","trusted":true},"execution_count":null,"outputs":[]}]}