{"cells":[{"metadata":{"_uuid":"f693d883762977620e72f52f823d6de216fcf531","_cell_guid":"ff2ff2d9-b657-4e65-82a8-3682c7d78b09"},"cell_type":"markdown","source":"Nothing much going on here. Just dumping some of the structured features into a GBDT model."},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"collapsed":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nimport os\nprint(os.listdir(\"../input\"))\n\n# Any results you write to the current directory are saved as output.","execution_count":1,"outputs":[]},{"metadata":{"collapsed":true,"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport catboost as cb","execution_count":3,"outputs":[]},{"metadata":{"_uuid":"708bf90c8fc5879983e1d9ad9e67c6720cfddbb7","_cell_guid":"0a8bb466-706c-479f-9c23-0ec0f605f467","trusted":true,"collapsed":true},"cell_type":"code","source":"!head ../input/sample_submission.csv","execution_count":4,"outputs":[]},{"metadata":{"collapsed":true,"_uuid":"84a60d14783e0a76b24c59a59ae15767ec2f9b51","_cell_guid":"edf21e11-9f6e-44f9-8dbc-96b481780dbd","trusted":true},"cell_type":"code","source":"train = pd.read_csv('../input/train.csv')\ntest = pd.read_csv('../input/test.csv')","execution_count":5,"outputs":[]},{"metadata":{"_uuid":"3465cfbb0c137794c6e807b06415c41975f2de66","_cell_guid":"dbc84137-0478-4672-a6da-fbf88e941c95","trusted":true,"collapsed":true},"cell_type":"code","source":"train.columns.values","execution_count":6,"outputs":[]},{"metadata":{"collapsed":true,"_uuid":"b3ec58df3d25b9f5086507db2472d0ce5e2ac9bd","_cell_guid":"4fcb7f06-797e-4254-8f0f-41919c87ece2","trusted":true},"cell_type":"code","source":"# restrict to numerical features and categorical features that aren't overly specific (i.e. title)\nfeatures = [f for f in train.columns.values if not f in ['item_id','user_id','title','description','activation_date','image','deal_probability']]\n# Treat everything but 'price' and 'item_seq_number' as categorical\nnumerical = ['price','item_seq_number','image_top_1']\ncat_ix = [i for i,f in enumerate(features) if not f in numerical]","execution_count":7,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"3d193ca58f39a4fee4d08761d2545043f535b275","collapsed":true},"cell_type":"code","source":"train.loc[:,'item_seq_number'] = train.loc[:,'item_seq_number'].astype(float)\ntest.loc[:,'item_seq_number'] = test.loc[:,'item_seq_number'].astype(float)","execution_count":10,"outputs":[]},{"metadata":{"collapsed":true,"_uuid":"b293c44e25fb8d4b0cce21d389f4dd053b90525f","_cell_guid":"37c59242-6818-4b56-b88d-5bef5256b0ca","trusted":true},"cell_type":"code","source":"# Fill missing features (use mean for numerical features)\nfor f in features:\n    if f in numerical:\n        mean = train[f].mean()\n        train.loc[:,f] = train[f].fillna(mean)\n        test.loc[:,f] = test[f].fillna(mean)\n    else:\n        train.loc[:,f] = train[f].fillna('NULL')\n        test.loc[:,f] = test[f].fillna('NULL')","execution_count":null,"outputs":[]},{"metadata":{"collapsed":true,"_uuid":"5102871915a96e8de8fc8b3b65100372495e644d","_cell_guid":"98a34fe6-a11e-40b8-a51a-534b7f96df67","trusted":false},"cell_type":"code","source":"cbr = cb.CatBoostRegressor()\ncbr.fit(train[features],train['deal_probability'],cat_features=cat_ix)","execution_count":null,"outputs":[]},{"metadata":{"collapsed":true,"_uuid":"49d0d3eaffb9e461ebd6b84e6a063d2dcbf0e038","_cell_guid":"82cadf2a-518d-47e9-bf30-f5cd2f7dfff6","trusted":false},"cell_type":"code","source":"cbr.score(train[features],train['deal_probability'])","execution_count":null,"outputs":[]},{"metadata":{"collapsed":true,"_uuid":"9c810f7ca388ff8efd0a87a27e179534df7b1d02","_cell_guid":"9451b439-ccd8-49cb-a676-e4af75172de3","trusted":false},"cell_type":"code","source":"test.loc[:,'deal_probability'] = cbr.predict(test[features])\ntest.loc[:,'deal_probability'] = test['deal_probability'].clip(lower=0,upper=1)","execution_count":null,"outputs":[]},{"metadata":{"collapsed":true,"_uuid":"53dd9060e64737b6dcf6a25c14a4da440c40379a","_cell_guid":"26657d20-cb58-48aa-b9da-6a86817f37f4","trusted":false},"cell_type":"code","source":"test[['item_id','deal_probability']].to_csv('submission1.csv',index=False)","execution_count":null,"outputs":[]},{"metadata":{"collapsed":true,"_uuid":"3ebbf8663c1f87bc4829fff799523383996d1068","_cell_guid":"807f3108-3c04-494b-8467-b96f436a91ab","trusted":false},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"language_info":{"name":"python","version":"3.6.5","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"}},"nbformat":4,"nbformat_minor":1}