{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport matplotlib.pyplot as plt \nimport seaborn as sns\n%matplotlib inline\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\nimport warnings \nwarnings.filterwarnings('ignore')\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-22T17:57:52.630354Z","iopub.execute_input":"2022-07-22T17:57:52.630824Z","iopub.status.idle":"2022-07-22T17:57:52.649129Z","shell.execute_reply.started":"2022-07-22T17:57:52.630788Z","shell.execute_reply":"2022-07-22T17:57:52.647860Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = pd.read_csv('/kaggle/input/forest-cover-type-prediction/train.csv')\ndf_test = pd.read_csv('/kaggle/input/forest-cover-type-prediction/test.csv')","metadata":{"execution":{"iopub.status.busy":"2022-07-22T17:57:52.652139Z","iopub.execute_input":"2022-07-22T17:57:52.653022Z","iopub.status.idle":"2022-07-22T17:57:54.545422Z","shell.execute_reply.started":"2022-07-22T17:57:52.652971Z","shell.execute_reply":"2022-07-22T17:57:54.544058Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-22T17:57:54.547508Z","iopub.execute_input":"2022-07-22T17:57:54.547960Z","iopub.status.idle":"2022-07-22T17:57:54.580141Z","shell.execute_reply.started":"2022-07-22T17:57:54.547921Z","shell.execute_reply":"2022-07-22T17:57:54.579213Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.dtypes","metadata":{"execution":{"iopub.status.busy":"2022-07-22T17:57:54.581406Z","iopub.execute_input":"2022-07-22T17:57:54.581940Z","iopub.status.idle":"2022-07-22T17:57:54.591140Z","shell.execute_reply.started":"2022-07-22T17:57:54.581909Z","shell.execute_reply":"2022-07-22T17:57:54.590091Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"There is no categorical data, just numerical.","metadata":{}},{"cell_type":"code","source":"pd.set_option('display.max_columns',None)\ndf_train.describe()","metadata":{"execution":{"iopub.status.busy":"2022-07-22T17:57:54.593757Z","iopub.execute_input":"2022-07-22T17:57:54.594192Z","iopub.status.idle":"2022-07-22T17:57:54.788521Z","shell.execute_reply.started":"2022-07-22T17:57:54.594154Z","shell.execute_reply":"2022-07-22T17:57:54.787377Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Inferences**\n1. The count is 15120 for each column, which means that no data point is missing.\n2. Soil type 7 and 15 are all zeroes so we can remove them.\n3. Wilderness_Area and Soil_Type are one hot encoded. Hence, they could be converted back for analysis.\n4. We also need to standardize the data as it is not scaled.","metadata":{}},{"cell_type":"code","source":"# Removing Soil_type 7 & 15\ndf_train = df_train.drop(['Soil_Type7','Soil_Type15'],axis=1)\ndf_test = df_test.drop(['Soil_Type7','Soil_Type15'],axis=1)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-22T17:57:54.790002Z","iopub.execute_input":"2022-07-22T17:57:54.790441Z","iopub.status.idle":"2022-07-22T17:57:54.876519Z","shell.execute_reply.started":"2022-07-22T17:57:54.790402Z","shell.execute_reply":"2022-07-22T17:57:54.875156Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Correlation matrix**","metadata":{}},{"cell_type":"code","source":"size = 10 \ncorrmatrix = df_train.iloc[:,:size].corr()\nf,ax = plt.subplots(figsize=(10,8))\nsns.heatmap(corrmatrix,vmax=0.8,square = True)","metadata":{"execution":{"iopub.status.busy":"2022-07-22T17:57:54.878294Z","iopub.execute_input":"2022-07-22T17:57:54.879400Z","iopub.status.idle":"2022-07-22T17:57:55.300221Z","shell.execute_reply.started":"2022-07-22T17:57:54.879348Z","shell.execute_reply":"2022-07-22T17:57:55.299069Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n# X = df_train.iloc[:, :-1]\n# y = df_train.iloc[:,-1]\n\nX = df_train.drop(['Id'],axis=1)\ny = df_train['Cover_Type']\nX_train, X_test, y_train, y_test = train_test_split(X,y,test_size=0.25,random_state=42)\n\nfrom sklearn.preprocessing import StandardScaler\n\nfeatures_to_scale = ['Elevation', 'Aspect','Slope','Horizontal_Distance_To_Hydrology','Vertical_Distance_To_Hydrology',\n'Horizontal_Distance_To_Roadways','Hillshade_9am','Hillshade_Noon','Hillshade_3pm','Horizontal_Distance_To_Fire_Points']\n\nsc_X = StandardScaler()\nX_train[features_to_scale] = sc_X.fit_transform(X_train[features_to_scale])","metadata":{"execution":{"iopub.status.busy":"2022-07-22T18:06:13.481200Z","iopub.execute_input":"2022-07-22T18:06:13.481621Z","iopub.status.idle":"2022-07-22T18:06:13.515458Z","shell.execute_reply.started":"2022-07-22T18:06:13.481589Z","shell.execute_reply":"2022-07-22T18:06:13.514499Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-22T18:06:18.571742Z","iopub.execute_input":"2022-07-22T18:06:18.572563Z","iopub.status.idle":"2022-07-22T18:06:18.609071Z","shell.execute_reply.started":"2022-07-22T18:06:18.572523Z","shell.execute_reply":"2022-07-22T18:06:18.608104Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Lets try a RandomForestClassifier \nfrom sklearn.ensemble import RandomForestClassifier \nRFClassifier = RandomForestClassifier(n_estimators=100, random_state=42)\nRFClassifier.fit(X_train, y_train)\n\ny_pred = RFClassifier.predict(X_test)\n\nfrom sklearn.metrics import accuracy_score, confusion_matrix\n\nprint(accuracy_score(y_test,y_pred))\n# rfclassifier_prob = RFClassifier.predict_proba(X_test)\n# confusion_matrix(y_test,y_pred)","metadata":{"execution":{"iopub.status.busy":"2022-07-22T18:06:19.790662Z","iopub.execute_input":"2022-07-22T18:06:19.791081Z","iopub.status.idle":"2022-07-22T18:06:21.297387Z","shell.execute_reply.started":"2022-07-22T18:06:19.791048Z","shell.execute_reply":"2022-07-22T18:06:21.296228Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# from xgboost import XGBClassifier \n\n# xgb = XGBClassifier(learning_rate=0.09,n_estimator=500,max_depth = 30,nthread = 4,objective = 'multi:softprob',subsample=0.75)\n\n# xgb.fit(X_train, y_train)\n\n# y_pred = xgb.predict(X_test)","metadata":{"execution":{"iopub.status.busy":"2022-07-22T18:06:23.149065Z","iopub.execute_input":"2022-07-22T18:06:23.149484Z","iopub.status.idle":"2022-07-22T18:06:23.154443Z","shell.execute_reply.started":"2022-07-22T18:06:23.149447Z","shell.execute_reply":"2022-07-22T18:06:23.153235Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from lightgbm import LGBMClassifier \nlgb = LGBMClassifier(learning_rate=0.09,\n                    num_leaves = 500,\n                     boosting_type='gbdt',\n                     objective = 'multiclass',\n                     metric = 'multi_logloss',\n                     max_depth = 30,\n                     subsample=0.75\n                    )\n\nlgb.fit(X_train, y_train)\n\ny_pred = lgb.predict(X_test)\nprint(accuracy_score(y_test,y_pred))","metadata":{"execution":{"iopub.status.busy":"2022-07-22T18:06:23.441832Z","iopub.execute_input":"2022-07-22T18:06:23.442639Z","iopub.status.idle":"2022-07-22T18:06:28.057570Z","shell.execute_reply.started":"2022-07-22T18:06:23.442589Z","shell.execute_reply":"2022-07-22T18:06:28.056727Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.ensemble import ExtraTreesClassifier \n\netc = ExtraTreesClassifier(n_estimators=500,n_jobs=-1,random_state=0)\netc.fit(X_train, y_train)\n\ny_pred = etc.predict(X_test)\nprint(accuracy_score(y_test, y_pred))","metadata":{"execution":{"iopub.status.busy":"2022-07-22T18:06:28.059128Z","iopub.execute_input":"2022-07-22T18:06:28.059924Z","iopub.status.idle":"2022-07-22T18:06:30.911317Z","shell.execute_reply.started":"2022-07-22T18:06:28.059886Z","shell.execute_reply":"2022-07-22T18:06:30.909825Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ids = df_test['Id']\n# df_test.drop(['Id'],axis=1,inplace=True)\ndf_test[features_to_scale] = sc_X.fit_transform(df_test[features_to_scale])\ny_result = etc.predict(df_test)\n\ny_result = pd.Series(y_result,name='Cover_Type')\nids = pd.Series(ids, name='Id')\nsubmission = pd.concat([ids,y_result],axis=1)\nsubmission.to_csv('/kaggle/working/submission.csv',index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-22T18:09:17.393377Z","iopub.execute_input":"2022-07-22T18:09:17.393847Z","iopub.status.idle":"2022-07-22T18:09:30.523222Z","shell.execute_reply.started":"2022-07-22T18:09:17.393810Z","shell.execute_reply":"2022-07-22T18:09:30.521873Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}