{"cells":[{"metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","trusted":true,"collapsed":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nimport os\nprint(os.listdir(\"../input\"))\n\n# Any results you write to the current directory are saved as output.","execution_count":1,"outputs":[]},{"metadata":{"collapsed":true,"trusted":true,"_uuid":"13d4598beae9ba5dbc5f5a45c13a9bb491844e3f"},"cell_type":"code","source":"test = pd.read_csv('../input/test.csv')\ntrain = pd.read_csv('../input/train.csv')","execution_count":2,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c58528bc4cb0515c7c32f291be531b9417013792","collapsed":true},"cell_type":"code","source":"test.head(10)","execution_count":3,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"9e291f918d2bd3e91d62057e81269b47aae28a95","collapsed":true},"cell_type":"code","source":"train.describe()","execution_count":4,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"bfc9836e1bec0f1ebf44d8351539be14e8d0b15f","collapsed":true},"cell_type":"code","source":"train.describe(include='all')","execution_count":5,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"10df1cb0bf2f1df8c3d90242f5cb252c97df28aa","collapsed":true},"cell_type":"code","source":"train.shape","execution_count":6,"outputs":[]},{"metadata":{"collapsed":true,"trusted":true,"_uuid":"ee0e828471ae8724b06b1d1c7bb55740c54bea9a"},"cell_type":"code","source":"test['is_train'] = False\ntrain['is_train'] = True\n\nall_df = pd.concat([test, train], axis = 0)","execution_count":7,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c254598765236f3793ee0e1ae6e2d93470cf2d8a","collapsed":true},"cell_type":"code","source":"test.head()","execution_count":8,"outputs":[]},{"metadata":{"collapsed":true,"trusted":true,"_uuid":"bd59831ee3b2345c8f3a6e74be877ed537dfcbd8"},"cell_type":"code","source":"from sklearn import  preprocessing\nle = preprocessing.LabelEncoder()","execution_count":9,"outputs":[]},{"metadata":{"collapsed":true,"trusted":true,"_uuid":"cf66c5e89f64e9a8b05fc8fc05cf293cffe102da"},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"collapsed":true,"trusted":true,"_uuid":"2e265bb3278e132aaf701a1daf672f50edac5a2a"},"cell_type":"code","source":"cat_vars = [\"user_id\", \"region\", \"city\", \"parent_category_name\", \"category_name\", \"param_1\", \"param_2\", \"param_3\", \"user_type\"]\nfor col in cat_vars:\n    all_df[col] = all_df[col].astype('str')\n    le.fit(all_df[col])\n    all_df[col] = le.transform(all_df[col])","execution_count":10,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"16b071dd588d2ae206243bf85495a0621f4045a6","collapsed":true},"cell_type":"code","source":"all_df.head()","execution_count":11,"outputs":[]},{"metadata":{"collapsed":true,"trusted":true,"_uuid":"167f5a06b107bf49562cd5bfe295d0dc05e72193"},"cell_type":"code","source":"cols_to_drop = [\"item_id\", \"title\", \"description\", \"activation_date\", \"image\"]\nall_df = all_df.drop(cols_to_drop, axis = 1)","execution_count":12,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b7bbac8082c5cf16da7074198c56587f4da130ad","collapsed":true},"cell_type":"code","source":"all_df.head()","execution_count":13,"outputs":[]},{"metadata":{"trusted":true,"collapsed":true,"_uuid":"9f2b1a843e57b56c088d273ec06c8de2aafb4332"},"cell_type":"code","source":"all_df = all_df.fillna(0)","execution_count":14,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"0eb7f0edb1d14c09461125ff0fb6540930fb5a67","collapsed":true},"cell_type":"code","source":"all_df.head()","execution_count":15,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e36449eabc0e6a28d169ccbcec847b2d9a453505","collapsed":true},"cell_type":"code","source":"all_df.loc[all_df['param_1']==110].head()","execution_count":16,"outputs":[]},{"metadata":{"trusted":true,"collapsed":true,"_uuid":"ee243ec650a0db2ac137d04be55f87f4be1cfaef"},"cell_type":"code","source":"train_df= all_df.loc[all_df['is_train']==True].drop(['is_train'], axis = 1)\ntest_df = all_df.loc[all_df['is_train'] == False].drop(['is_train', 'deal_probability'], axis = 1)","execution_count":17,"outputs":[]},{"metadata":{"trusted":true,"collapsed":true,"_uuid":"67e898e22ef24d5daafa367d2c370f648ac51532"},"cell_type":"code","source":"from sklearn.model_selection import train_test_split","execution_count":18,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"699b2a054c27f1a7b0248491cc7de20bef6abdb2","collapsed":true},"cell_type":"code","source":"train_X, valid_X, train_y, valid_y = train_test_split(train_df.drop('deal_probability', axis=1), \n                                                      train_df['deal_probability'], test_size=0.2)","execution_count":19,"outputs":[]},{"metadata":{"trusted":true,"collapsed":true,"_uuid":"0c07516046e7e299839ce2761f826b775d77c33c"},"cell_type":"code","source":"rf_params={\n    'n_estimators' : [100, 200, 300],\n    'n_jobs' : [-1], \n}","execution_count":32,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"cc0522fca8be93baae138654d5c58ff9310fd47d","collapsed":true},"cell_type":"code","source":"from sklearn.ensemble import RandomForestRegressor\nfrom sklearn.model_selection import GridSearchCV\nrf = RandomForestRegressor()\ngrid_search = GridSearchCV(rf, param_grid=rf_params)\ngrid_search.fit(train_X,train_y)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ddeec47900668c0bac93f55287dd8f03e01379d5","collapsed":true},"cell_type":"code","source":"from sklearn import linear_model\nreg = linear_model.Ridge (alpha = .5)\nreg.fit(train_X,train_y)","execution_count":22,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f9b47f1d2e8f769bab09b815b134800d9808672b","collapsed":true},"cell_type":"code","source":"rf","execution_count":23,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f5f464ba94b0529fea0d43d8464492f02ce38a95","collapsed":true},"cell_type":"code","source":"rf.score(valid_X, valid_y)","execution_count":28,"outputs":[]},{"metadata":{"trusted":true,"collapsed":true,"_uuid":"d786893b6f789a521140a2d394702e7743e7070f"},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"collapsed":true,"_uuid":"04a3df419b77a367f8f8611282c3018a889c7fbe"},"cell_type":"code","source":"import xgboost as xgb \nxgb_model = xgb.XGBRegressor()\nxgb_model.fit(train_X,train_y)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"collapsed":true,"_uuid":"23530ce2a160557b43447bb01acf7a5aad686581"},"cell_type":"code","source":"from sklearn.ensemble import RandomForestRegressor\nrf = RandomForestRegressor(n_jobs = -1, n_estimators = 100)\nrf.fit(train_X,train_y)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"67bdd4b83b61a53cbc1ca739ae860782c55a7032","collapsed":true},"cell_type":"code","source":"from sklearn.ensemble import RandomForestRegressor\nrf = RandomForestRegressor(verbose = 1, n_estimators=20)\nrf.fit(train_X,train_y)","execution_count":37,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"314144cfd088c369d756aa094eb2a1e75290bc00","collapsed":true},"cell_type":"code","source":"from sklearn.metrics import mean_squared_error as mse\nprint(np.sqrt(mse(rf.predict(valid_X), valid_y)))","execution_count":38,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"3c6ce2664f45c9246b50508f60adb0efdcae338b","collapsed":true},"cell_type":"code","source":"pred_test_y = rf.predict(test_df)","execution_count":39,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e3c72e26c5cfaed93adc1e3c27558b9524a9e504","collapsed":true},"cell_type":"code","source":"pred_test_y[:5]","execution_count":40,"outputs":[]},{"metadata":{"trusted":true,"collapsed":true,"_uuid":"668d53630646075aa4bedb5dbcd05ca48d42b357"},"cell_type":"code","source":"pred_test_y[pred_test_y>1] = 1\npred_test_y[pred_test_y<0] = 0\ntest_id = test['item_id'].values\nsub = pd.DataFrame({'item_id':test_id})\nsub['deal_probability'] = pred_test_y\nsub.to_csv('rf_test.csv', index=False)","execution_count":41,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"049af4ec53525d0907b5d922189847c8d4d302a8","collapsed":true},"cell_type":"code","source":"sub.head()","execution_count":42,"outputs":[]},{"metadata":{"trusted":true,"collapsed":true,"_uuid":"47a6adaf5f1af84c05dcb9043bbd120cc54a5650"},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.5","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}