{"cells":[{"metadata":{"id":"AV2i0SA99uhT","trusted":true},"cell_type":"code","source":"import pandas as pd \nimport numpy as np \nimport matplotlib.pyplot as plt \nfrom sklearn.ensemble import ExtraTreesClassifier \nfrom sklearn.ensemble import BaggingClassifier\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.tree import DecisionTreeClassifier\n\n# To measure performance\nfrom sklearn import metrics\n","execution_count":null,"outputs":[]},{"metadata":{"id":"PA9f5E4m89n2","outputId":"7b354df6-2acf-454a-a5a5-da53a28e9bb6","trusted":true},"cell_type":"code","source":"import pandas as pd\n\ndf = pd.read_csv('../input/womendata2019/WEvents2019.csv')\ndf.head()\n","execution_count":null,"outputs":[]},{"metadata":{"id":"Vm4NwpzB8kzu","trusted":true},"cell_type":"code","source":"df=df[:50000]","execution_count":null,"outputs":[]},{"metadata":{"id":"fZwWa0m69VqQ","trusted":true},"cell_type":"code","source":"# Seperating the dependent and independent variables \ny = df['WFinalScore']  \nX=df.drop(['Season','LTeamID','LCurrentScore','WCurrentScore','WFinalScore','EventType','EventSubType','X','Y','Area'],axis=1)\n","execution_count":null,"outputs":[]},{"metadata":{"id":"zdLLZA6A9ii1","trusted":true},"cell_type":"code","source":"# Splitting Dataset\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size = 0.3)\n","execution_count":null,"outputs":[]},{"metadata":{"id":"GjnYANF49Ojw","outputId":"2b33d0e8-b318-4310-9d08-0dba762bb6ba","trusted":true},"cell_type":"code","source":"from sklearn.model_selection import GridSearchCV\nfrom sklearn.ensemble import RandomForestRegressor\nfrom sklearn.linear_model import LinearRegression\nfrom sklearn.linear_model import Lasso\nfrom sklearn.tree import DecisionTreeRegressor\nfrom sklearn.svm.libsvm import cross_validation\nfrom sklearn.model_selection import ShuffleSplit\ndef find_best_model_using_gridsearchcv(X,y):\n    algos = {\n        'linear_regression' : {\n            'model': LinearRegression(),\n            'params': {\n                'normalize': [True, False]\n            }\n        },\n        'lasso': {\n            'model': Lasso(),\n            'params': {\n                'alpha': [1,2],\n                'selection': ['random', 'cyclic']\n            }\n        },\n\n        'Random_Forest' : {\n            'model': RandomForestRegressor(n_estimators=20),\n            'params': {\n                'criterion' : ['mse','friedman_mse'],\n                \n            }\n        },\n\n        'decision_tree': {\n            'model': DecisionTreeRegressor(),\n            'params': {\n                'criterion' : ['mse','friedman_mse'],\n                'splitter': ['best','random']\n            }\n        }\n    }\n    scores = []\n    cv = ShuffleSplit(n_splits=5, test_size=0.2, random_state=0)\n    for algo_name, config in algos.items():\n        gs =  GridSearchCV(config['model'], config['params'], cv=cv, return_train_score=False)\n        gs.fit(X,y)\n        scores.append({\n            'model': algo_name,\n            'best_score': gs.best_score_,\n            'best_params': gs.best_params_\n        })\n\n    return pd.DataFrame(scores,columns=['model','best_score','best_params'])\n\nfind_best_model_using_gridsearchcv(X,y)\n","execution_count":null,"outputs":[]},{"metadata":{"id":"AE9sr15zhHk-","trusted":true},"cell_type":"code","source":"# Building the model \nextra_tree_forest = ExtraTreesClassifier(n_estimators = 5, \n                                        criterion ='entropy', max_features = 2) \n  \n# Training the model \nextra_tree_forest.fit(X_train, y_train) \n  \n# Computing the importance of each feature \nfeature_importance = extra_tree_forest.feature_importances_ \n  \n# Normalizing the individual importances \nfeature_importance_normalized = np.std([tree.feature_importances_ for tree in \n                                        extra_tree_forest.estimators_], \n                                        axis = 0) ","execution_count":null,"outputs":[]},{"metadata":{"id":"tqzqDq-whm9h","trusted":true},"cell_type":"code","source":"y_pred=extra_tree_forest.predict(X_test)","execution_count":null,"outputs":[]},{"metadata":{"id":"gApnKv-LhsES","outputId":"a6f0982e-4da8-4310-8842-e213b5552fbf","trusted":true},"cell_type":"code","source":"df = pd.DataFrame({'Actual': y_test.ravel(), 'Predicted': y_pred.ravel()})\ndf\n","execution_count":null,"outputs":[]},{"metadata":{"id":"n8CFyQNXhvQZ","outputId":"31016966-a917-472d-d9da-0a82d6144bd1","trusted":true},"cell_type":"code","source":"df1 = df.head(25)\ndf1.plot(kind='bar',figsize=(16,10))\nplt.grid(which='major', linestyle='-', linewidth='0.5', color='green')\nplt.grid(which='minor', linestyle=':', linewidth='0.5', color='black')\nplt.show()\n","execution_count":null,"outputs":[]},{"metadata":{"id":"6GXD6_yphz5J","outputId":"73967f6f-2860-4958-93b1-1ce3e9dbde44","trusted":true},"cell_type":"code","source":"extra_tree_forest.score(X_test,y_test)","execution_count":null,"outputs":[]},{"metadata":{"id":"j9SI2uzZh46W","outputId":"c9e10557-a73f-4ac9-c290-dac3447eccfc","trusted":true},"cell_type":"code","source":"from sklearn.ensemble import RandomForestClassifier\nmodel = RandomForestClassifier(n_estimators=20)\nmodel.fit(X_train, y_train)\n","execution_count":null,"outputs":[]},{"metadata":{"id":"17WnuwX5h9Sb","outputId":"74f10752-57db-4292-8d6e-8511f4f31c8a","trusted":true},"cell_type":"code","source":"model.score(X_test, y_test)\n","execution_count":null,"outputs":[]}],"metadata":{"colab":{"name":"March Madness.ipynb","provenance":[]},"kernelspec":{"name":"python3","display_name":"Python 3"}},"nbformat":4,"nbformat_minor":4}