{"cells":[{"metadata":{"_uuid":"bf4803011e37e2337ede6f439474cfc73248b9b4"},"cell_type":"markdown","source":"The approach is very simple. We will drop any column which can not be converted to numerical feature in astraight forward way and consider only the following columns for prediction\n1. region\n2. city\n3. parent_category_name\n4. category_name\n5. param_1\n6. param_2\n7. param_3\n8. price\n9. item_seq_number\n10. user_type\n\nWe will just ignore the rest of the columns for now\nA lot of things can be and should be improved over this e.g. hyperparameter optimization and most importantly **we should definitely use title and description to improve**.\nYou may consider my code a bit untidy. But I will improve this as I move forward"},{"metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","trusted":true},"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport os","execution_count":1,"outputs":[]},{"metadata":{"_uuid":"4e590392ff4d69daf09a9d8911aebf1de8d0dc44"},"cell_type":"markdown","source":"First, lets load the training data"},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","collapsed":true,"trusted":true},"cell_type":"code","source":"data_raw = pd.read_csv('../input/train.csv')","execution_count":3,"outputs":[]},{"metadata":{"_uuid":"d0ab0df8029834df5c6a259fa923f6122017127e"},"cell_type":"markdown","source":"Drop anything that we'll not be needing for now"},{"metadata":{"collapsed":true,"trusted":true,"_uuid":"1938fca1542ab45efb4c9d01552a86bf9febec58"},"cell_type":"code","source":"data_raw = data_raw.drop('item_id', axis=1).drop('user_id', axis=1).drop('title', axis=1).drop('description', axis=1)\\\n        .drop('activation_date', axis=1).drop('user_type', axis=1).drop('image', axis=1).drop('image_top_1', axis=1)","execution_count":4,"outputs":[]},{"metadata":{"_uuid":"11104778ebcb3acc2b9ffbdd417e37adeae87017"},"cell_type":"markdown","source":"**Make all those feature mentioned categorical now. Because there are managable number of unique values in those columns**"},{"metadata":{"collapsed":true,"trusted":true,"_uuid":"d037c144381993a17e2bde4e861718c9657b19a9"},"cell_type":"code","source":"region_set = data_raw['region'].unique()\ndef convert_region(x):\n    tp = np.where(region_set == x['region'])\n    return tp[0][0]\ndata_raw['region'] = data_raw.apply(convert_region, axis=1)","execution_count":6,"outputs":[]},{"metadata":{"collapsed":true,"trusted":true,"_uuid":"2de00e591a515a943f62979c09f87ba6247c8034"},"cell_type":"code","source":"city_set = data_raw['city'].unique()\ndef convert_city(x):\n    tp = np.where(city_set == x['city'])\n    return tp[0][0]\ndata_raw['city'] = data_raw.apply(convert_city, axis=1)","execution_count":7,"outputs":[]},{"metadata":{"collapsed":true,"trusted":true,"_uuid":"98b0cc64cb3429fb1d39361879303d24dc2075bf"},"cell_type":"code","source":"cat_set = data_raw['category_name'].unique()\ndef convert_cat(x):\n    tp = np.where(cat_set == x['category_name'])\n    return tp[0][0]\ndata_raw['category_name'] = data_raw.apply(convert_cat, axis=1)","execution_count":9,"outputs":[]},{"metadata":{"collapsed":true,"trusted":true,"_uuid":"3ee9a1a267560edc64e38b0ca8422f81b2865b82"},"cell_type":"code","source":"p1_set = data_raw['param_1'].unique()\ndef convert_p1(x):\n    tp = np.where(p1_set == x['param_1'])\n    try:\n        return tp[0][0]\n    except Exception:\n        return 0\ndata_raw['param_1'] = data_raw.apply(convert_p1, axis=1)","execution_count":8,"outputs":[]},{"metadata":{"collapsed":true,"trusted":true,"_uuid":"8c44467e1dbedf124c4c76f199e8f2817aef87d3"},"cell_type":"code","source":"p2_set = data_raw['param_2'].unique()\ndef convert_p2(x):\n    tp = np.where(p2_set == x['param_2'])\n    try:\n        return tp[0][0]\n    except Exception:\n        return 0\ndata_raw['param_2'] = data_raw.apply(convert_p2, axis=1)","execution_count":10,"outputs":[]},{"metadata":{"collapsed":true,"trusted":true,"_uuid":"4f3847015f23149d93198b43a586b56da907211c"},"cell_type":"code","source":"p3_set = data_raw['param_3'].unique()\ndef convert_p3(x):\n    tp = np.where(p3_set == x['param_3'])\n    try:\n        return tp[0][0]\n    except Exception:\n        return 0\ndata_raw['param_3'] = data_raw.apply(convert_p3, axis=1)","execution_count":11,"outputs":[]},{"metadata":{"collapsed":true,"trusted":true,"_uuid":"a36bd835b2806ab77761a375c2d08d949621a081"},"cell_type":"code","source":"parent_category_name_set = data_raw['parent_category_name'].unique()\ndef convert_parent_category_name(x):\n    tp = np.where(parent_category_name_set == x['parent_category_name'])\n    return tp[0][0]\ndata_raw['parent_category_name'] = data_raw.apply(convert_parent_category_name, axis=1)","execution_count":12,"outputs":[]},{"metadata":{"_uuid":"71748b0d47886ae80d98b55ba8e0d00c1a469932"},"cell_type":"markdown","source":"Now our data is ready. We will prepare our `x_train` and `y_train`"},{"metadata":{"collapsed":true,"trusted":true,"_uuid":"e51ad80251cf647d5f3215cbe527962d0be8b26e"},"cell_type":"code","source":"x_train = data_raw.drop('deal_probability', axis=1).as_matrix()\ny_train = data_raw['deal_probability'].as_matrix()","execution_count":14,"outputs":[]},{"metadata":{"collapsed":true,"trusted":true,"_uuid":"c329e99bd7c2896d3635de24e5b6b69350b026ec"},"cell_type":"code","source":"import xgboost as xgb","execution_count":15,"outputs":[]},{"metadata":{"_uuid":"d07ca4ced3590461d7266b3d2a86797871c958e9"},"cell_type":"markdown","source":"**Lets train our simple xgb model**"},{"metadata":{"collapsed":true,"trusted":true,"_uuid":"daebd7e964b0421ee3608438bac1c522d893fb31"},"cell_type":"code","source":"xgdmat=xgb.DMatrix(x_train, y_train)\nparams={'seed':0,'colsample_bytree':0.8,'objective':'reg:linear','max_depth':24,'min_child_weight':24}\nfinal_gb=xgb.train(params,xgdmat)","execution_count":16,"outputs":[]},{"metadata":{"_uuid":"3f35c4ded3bb54133b17759607e069ae713e2ec1"},"cell_type":"markdown","source":"**Reading test data**"},{"metadata":{"collapsed":true,"trusted":true,"_uuid":"7b558b45229590c1e59a7d8ca449a6334c554aba"},"cell_type":"code","source":"test_df = pd.read_csv('../input/test.csv')","execution_count":17,"outputs":[]},{"metadata":{"_uuid":"46c1fa9d06264ac809dc4a6910e0a9b1e7b8bca1"},"cell_type":"markdown","source":"Dropping the columns we don't need"},{"metadata":{"collapsed":true,"trusted":true,"_uuid":"bf6c11e5af73ff74283f21696f8904e7daaaaa17"},"cell_type":"code","source":"test_df = test_df.drop('user_id', axis=1).drop('title', axis=1).drop('description', axis=1)\\\n        .drop('activation_date', axis=1).drop('user_type', axis=1).drop('image', axis=1).drop('image_top_1', axis=1)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d1567862cddfcfefe2b66324c8b7f0154e7b51c3"},"cell_type":"markdown","source":"**Make all columns categorical instead of string, as done in training set**"},{"metadata":{"collapsed":true,"trusted":true,"_uuid":"05320527b49bc20df45ac91213755e6196b75e8e"},"cell_type":"code","source":"def convert_region(x):\n    tp = np.where(region_set == x['region'])\n    return tp[0][0]\n# creating region catagory as numerical feature\ntest_df['region'] = test_df.apply(convert_region, axis=1)","execution_count":18,"outputs":[]},{"metadata":{"collapsed":true,"trusted":true,"_uuid":"84cc9624eed07e7b76916b91f6bdc1f415f2b5cd"},"cell_type":"code","source":"def convert_city(x):\n    tp = np.where(city_set == x['city'])\n    try:\n        return tp[0][0]\n    except Exception:\n        return 0\ntest_df['city'] = test_df.apply(convert_city, axis=1)","execution_count":19,"outputs":[]},{"metadata":{"collapsed":true,"trusted":true,"_uuid":"9e520e8bb0b46e8a56babc035d8be93b80d6b3d3"},"cell_type":"code","source":"def convert_cat(x):\n    tp = np.where(cat_set == x['category_name'])\n    try:\n        return tp[0][0]\n    except Exception:\n        return 0\ntest_df['category_name'] = test_df.apply(convert_cat, axis=1)","execution_count":20,"outputs":[]},{"metadata":{"collapsed":true,"trusted":true,"_uuid":"0097b3c39726f5bca4b4bf17c233811e7fc378e3"},"cell_type":"code","source":"def convert_p1(x):\n    tp = np.where(p1_set == x['param_1'])\n    try:\n        return tp[0][0]\n    except Exception:\n        return 0\ntest_df['param_1'] = test_df.apply(convert_p1, axis=1)","execution_count":21,"outputs":[]},{"metadata":{"collapsed":true,"trusted":true,"_uuid":"4e822a32ed31afe4b09b2d9e1c0f7a8820c556ea"},"cell_type":"code","source":"def convert_p2(x):\n    tp = np.where(p2_set == x['param_2'])\n    try:\n        return tp[0][0]\n    except Exception:\n        return 0\ntest_df['param_2'] = test_df.apply(convert_p2, axis=1)","execution_count":22,"outputs":[]},{"metadata":{"collapsed":true,"trusted":true,"_uuid":"2b255bd75fc3b7de2cdb10aad3ae020f2298fcb3"},"cell_type":"code","source":"def convert_p3(x):\n    tp = np.where(p3_set == x['param_3'])\n    try:\n        return tp[0][0]\n    except Exception:\n        return 0\ntest_df['param_3'] = test_df.apply(convert_p3, axis=1)","execution_count":23,"outputs":[]},{"metadata":{"collapsed":true,"trusted":true,"_uuid":"29b140c0f30d9e7d1b1a7c2a6a6ebb071f6b221f"},"cell_type":"code","source":"def convert_parent_category_name(x):\n    tp = np.where(parent_category_name_set == x['parent_category_name'])\n    try:\n        return tp[0][0]\n    except Exception:\n        return 0\ntest_df['parent_category_name'] = test_df.apply(convert_parent_category_name, axis=1)","execution_count":24,"outputs":[]},{"metadata":{"_uuid":"2a386177dc2ab8576d59914ae7a90de3d2fe7925"},"cell_type":"markdown","source":"**prepare `x_test`**"},{"metadata":{"collapsed":true,"trusted":true,"_uuid":"7406ab7e248d575eac6725be5c59e3acd7c732d6"},"cell_type":"code","source":"x_test = test_df.drop('item_id', axis=1).as_matrix()","execution_count":28,"outputs":[]},{"metadata":{"collapsed":true,"trusted":true,"_uuid":"275e27e62c643fbc6769779cea0b38e1ad27245e"},"cell_type":"code","source":"tesdmat=xgb.DMatrix(x_test)\ny_pred=final_gb.predict(tesdmat)","execution_count":29,"outputs":[]},{"metadata":{"collapsed":true,"trusted":true,"_uuid":"8f1909f1bc6b7db0c64e129aff3916996b57f147"},"cell_type":"code","source":"y_pred[y_pred<0] = 0","execution_count":31,"outputs":[]},{"metadata":{"_uuid":"3a3bf56f24ca0726e39b3ecf3ee2f4ae1c1c9cb1"},"cell_type":"markdown","source":"At this point our prediction is ready. Lets prepare the output dataframe"},{"metadata":{"collapsed":true,"trusted":true,"_uuid":"f5a5f690fc1eea028593c321a622f38caaa9ed0a"},"cell_type":"code","source":"result = pd.DataFrame({ 'deal_probability': y_pred, 'item_id': test_df['item_id']})","execution_count":45,"outputs":[]},{"metadata":{"_uuid":"b07f3826f33417b997f8504f78f0af6be34b8951"},"cell_type":"markdown","source":"**All done. Lets create our submission file**"},{"metadata":{"collapsed":true,"trusted":true,"_uuid":"16cc5cad443706b9ff0785a1c716ebfa71872487"},"cell_type":"code","source":"result.to_csv('submission.csv', index=False)","execution_count":46,"outputs":[]},{"metadata":{"trusted":true,"collapsed":true,"_uuid":"31b36cd701a4d755cc35a3f6f864a519e1e5b86f"},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.5","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}