{"cells":[{"metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nfrom sklearn.ensemble import RandomForestRegressor\nimport gc\nimport lightgbm\nfrom sklearn.model_selection import train_test_split\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nimport os\nprint(os.listdir(\"../input\"))\n\n# Any results you write to the current directory are saved as output.","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"41a5e0be-8997-4139-ae69-503793f5fa1d","_uuid":"d09a670384de1c53cbdd26d6f637634f784829ad","collapsed":true,"trusted":true},"cell_type":"code","source":"train = pd.read_csv(\"../input/train.csv\")\ntest = pd.read_csv(\"../input/test.csv\")","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"ef810c5a-9cfc-41fb-90cc-6fad71b297df","_uuid":"a98e4af942af77f16ac9b596a15a29f510742e49","trusted":true},"cell_type":"code","source":"train.head()","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"f3880370-8e42-4dfe-9451-64576a6c23de","_uuid":"5beb379e12d228c3166effd3b8acac780f92ceee","trusted":true},"cell_type":"code","source":"train.describe()","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"036d4aa0-a74f-43e7-a2b4-ece613669a81","_uuid":"578a49cf56a30e950569339a4b6d0a8c323003a9","trusted":true},"cell_type":"code","source":"cols = ['parent_category_name', 'category_name', 'price', 'user_type', 'item_seq_number', 'image_top_1']\ndummy_cols = ['parent_category_name', 'category_name','user_type']\ny = train['deal_probability'].copy()\nx_train = train[cols].copy().fillna(0)\nx_test  = test[cols].copy().fillna(0)\ndel train, test; gc.collect()\n\nn = len(x_train)\nx = pd.concat([x_train, x_test])\nx = pd.get_dummies(x, columns=dummy_cols)\nx.head()","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"f71f92a8-83d8-4964-9663-b04c586d87ae","_uuid":"9a1ea4e286d17b104258406fcd556833bfdb24b9","trusted":true},"cell_type":"code","source":"x_train = x.iloc[:n, :]\nx_test = x.iloc[n:, :]\ndel x; gc.collect()","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"45b19acd-3a60-40ae-8760-60f970fec2ae","_uuid":"9d118956daaecf9008a5b68b578f76f05717e6f5","trusted":true},"cell_type":"code","source":"# https://www.kaggle.com/ezietsman/simple-python-lightgbm-example\n\n# Create training and validation sets\nx, x_val, y, y_val = train_test_split(x_train, y, test_size=0.2, random_state=42)\n\n# Create the LightGBM data containers\ntrain_data = lightgbm.Dataset(x, label=y)\nval_data = lightgbm.Dataset(x_val, label=y_val)\n\n# Train the model\nparameters = {\n    'task': 'train',\n    'boosting_type': 'gbdt',\n    'objective': 'regression',\n    'metric': 'rmse',\n    'num_leaves': 31,\n    'learning_rate': 0.05,\n    'feature_fraction': 0.9,\n    'bagging_fraction': 0.8,\n    'bagging_freq': 5,\n    'verbose': 100\n}\n\n\nmodel = lightgbm.train(parameters,\n                       train_data,\n                       valid_sets=val_data,\n                       num_boost_round=2000,\n                       early_stopping_rounds=100)\n\n# Create a submission\ny_pred = model.predict(x_test)","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"853c3df6-4907-4e56-b779-433337dddc26","_uuid":"2bb22face049fb84e90b432bc27600d405fbb664","collapsed":true,"trusted":true},"cell_type":"code","source":"sub = pd.read_csv('../input/sample_submission.csv')","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"7b1382dc-ce46-4ed8-81ef-a1caa1d00122","_uuid":"860cc580c103eff45b9372485bb468aadb5bd014","collapsed":true,"trusted":true},"cell_type":"code","source":"sub['deal_probability'] = y_pred\nsub['deal_probability'].clip(0.0, 1.0, inplace=True)","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"c825dc95-d9d9-48cc-951c-e6a5c1248963","_uuid":"e8e8f950896c9c057e2bb23c40e44f14744ae573","trusted":true},"cell_type":"code","source":"sub.to_csv('simple_mean_benchmark.csv', index=False)\nsub.head()","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.4","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}