{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"collapsed":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nimport os\nprint(os.listdir(\"../input\"))\n\n# Any results you write to the current directory are saved as output.","execution_count":2,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true,"collapsed":true},"cell_type":"code","source":"#Reading the training data\ntrain_df = pd.read_csv('../input/train.csv')","execution_count":18,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"4b0c3ba6f45d557af73cc275aa9254d5864489b4","collapsed":true},"cell_type":"code","source":"train_df.head(15)","execution_count":4,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"62d8f34174adcf43c8554291022d395fcc411219","collapsed":true},"cell_type":"code","source":"#Columns in the training dataframe\nprint(train_df.columns)","execution_count":5,"outputs":[]},{"metadata":{"trusted":true,"collapsed":true,"_uuid":"d3db7d57dc2144212444c35e08ac9a58ab2835ec"},"cell_type":"code","source":"#Gives a categorical mapping.\n#Given a column assigns integers to the unique data of that column\ndef get_map(col_name):\n    keys = list(train_df[col_name].unique())\n    values = list(range(len(keys)))\n    map_dict = dict(zip(keys, values))\n    \n    return map_dict","execution_count":8,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"9b8c928974f07f785fee38f8385210b66e9679f0","collapsed":true},"cell_type":"code","source":"region_map = get_map('region')\nregion_map","execution_count":9,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d9d7b39a45c7d693bdb71a30d12316828ff33999","collapsed":true},"cell_type":"code","source":"city_map = get_map('city')\nlen(city_map)","execution_count":11,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"59f3db019ac1f0c9909e495b3493e01cf4194a7f","collapsed":true},"cell_type":"code","source":"parent_category_map = get_map('parent_category_name')\nparent_category_map","execution_count":12,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"3ee3a66e183aa20b7e61aca67e74a93afb244f66","collapsed":true},"cell_type":"code","source":"category_map = get_map('category_name')\ncategory_map","execution_count":13,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1b843cd98df9b207d3c3676013a70d8f5507cd8c","collapsed":true},"cell_type":"code","source":"user_type_map = get_map('user_type')\nuser_type_map","execution_count":14,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d334eebc705e7155f32f2851c85231925e36bf16","collapsed":true},"cell_type":"code","source":"#Adds the categorical mappings to the dataframe\ndef add_codes (col_name, mapping):\n    train_df[col_name + '_code'] = train_df[col_name].apply(lambda x : mapping[x])\n    return None","execution_count":19,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"5fa5d0bb6e02e9a1f59a094c2a4ebee0d8c7394e","collapsed":true},"cell_type":"code","source":"add_codes('region', region_map)\ntrain_df[['region', 'region_code']][:10]","execution_count":20,"outputs":[]},{"metadata":{"trusted":true,"collapsed":true,"_uuid":"7778a731e906c1d5ff32cca854f148e29390ddb9"},"cell_type":"code","source":"add_codes('city', city_map)\nadd_codes('parent_category_name', parent_category_map)\nadd_codes('category_name', category_map)\nadd_codes('user_type', user_type_map)","execution_count":21,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b5dc558cb287d25c6ce7c9860559d0a8f82c85fd","collapsed":true},"cell_type":"code","source":"train_df[['region_code', 'city_code', 'parent_category_name_code', 'category_name_code', 'user_type_code']][:10]","execution_count":22,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"6bc51d51448f183b6bf089f01f052603b5573a7a","collapsed":true},"cell_type":"code","source":"#Combining all the parameters in param_1, param_2, param_3.\ntrain_df['param_combined'] = train_df.apply(lambda row: ' '.join([str(row['param_1']), str(row['param_2']),  str(row['param_3'])]), axis=1)\ntrain_df['param_combined'][:10]","execution_count":23,"outputs":[]},{"metadata":{"trusted":true,"collapsed":true,"_uuid":"ab2bfee71db2145862acc9bc3f5d5c51549a1639"},"cell_type":"code","source":"#Finding the lenght(number of words) in a columns and adding them to the dataframe\ndef add_len (col_name):\n    train_df[col_name] = train_df[col_name].fillna(' ')\n    train_df[col_name + '_len'] = train_df[col_name].apply(lambda x : len(x.split()))\n    return None","execution_count":24,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"419f23bd7040feb1152e4f9a0b3b243584161053","collapsed":true},"cell_type":"code","source":"add_len('title')\ntrain_df[['title', 'title_len']][:10]","execution_count":25,"outputs":[]},{"metadata":{"trusted":true,"collapsed":true,"_uuid":"ae204e3fdb712be1a2205fea1af31e58f9e56218"},"cell_type":"code","source":"add_len('description')\nadd_len('param_combined')","execution_count":26,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d368b348c9832f7d4475818c5b452cbdcc8b3479","collapsed":true},"cell_type":"code","source":"train_df[['title_len', 'description_len', 'param_combined_len']][:10]","execution_count":27,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"cf489c722f6508ac8889083c260acb234ac2c9eb","collapsed":true},"cell_type":"code","source":"#importing the periods_train.csv\npr_train_df = pd.read_csv('../input/periods_train.csv', parse_dates = ['date_from', 'date_to', 'activation_date'])","execution_count":30,"outputs":[]},{"metadata":{"trusted":true,"collapsed":true,"_uuid":"56aa804d35a0b76a9967c988affd9794f137391c"},"cell_type":"code","source":"#Finding the time period for which the advertisement was there\ntrain_df['period'] = pr_train_df['date_to'] - pr_train_df['date_from']","execution_count":31,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"9ef6850f26ac17a5866b4dc6d09a9f6f0ea82e4a","collapsed":true},"cell_type":"code","source":"#Converting the time periods to number of days in integer.\ntrain_df['period'] = train_df['period'].astype('int64')/(864 * 10e10)","execution_count":35,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"cee30699a04728b75775a62bb6e1bec3950bdc1a","collapsed":true},"cell_type":"code","source":"train_df['period'][:10]","execution_count":36,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ac931a05b95070044891df6cb6ff5bc82999493b","collapsed":true},"cell_type":"code","source":"#Showing all the newly created columns\ntrain_df[['region_code', \n          'city_code', \n          'parent_category_name_code', \n          'category_name_code', \n          'param_combined_len', \n          'title_len', \n          'description_len', \n          'price', \n          'user_type_code',\n          'period']][:10]","execution_count":38,"outputs":[]},{"metadata":{"trusted":true,"collapsed":true,"_uuid":"a1e9ce1563f6e7abaebf5a3a656a3ef2755bdbd5"},"cell_type":"code","source":"#Making the training data NumPy array.\ntrain_data = train_df[['region_code', \n                      'city_code', \n                      'parent_category_name_code', \n                      'category_name_code', \n                      'param_combined_len', \n                      'title_len', \n                      'description_len', \n                      'price', \n                      'user_type_code',\n                      'period']].values","execution_count":46,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b4eda1d5d7e7beab8409888c5a7d4f83563965a6","collapsed":true},"cell_type":"code","source":"train_data.shape","execution_count":47,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a66ff3666c7b0cb11d73ce5f51d614adc6e6ebcd","collapsed":true},"cell_type":"code","source":"train_data[:3]","execution_count":48,"outputs":[]},{"metadata":{"trusted":true,"collapsed":true,"_uuid":"3839676ea91d4d115296015a438b1a7f952dbdc2"},"cell_type":"code","source":"#Createing the labels\nlabels = train_df['deal_probability'].values","execution_count":49,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"44d9cd4081df0151c4e72b8ea543819715a35c93","collapsed":true},"cell_type":"code","source":"labels = labels.reshape(len(labels), 1)","execution_count":52,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"17e9e070597c8729ca65fc53d1b78ed5b18a0cbd","collapsed":true},"cell_type":"code","source":"labels.shape","execution_count":53,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"2d90952caa4b71555b16e4c32497e3d0aae0f6f1","collapsed":true},"cell_type":"code","source":"labels[:5]","execution_count":54,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"97c3eaf2ea7c70e5020d1cd5c28dbdd0bc292291","collapsed":true},"cell_type":"code","source":"from keras.layers import Dense, Activation\nfrom keras.models import Sequential\nfrom keras.initializers import glorot_uniform\nfrom keras.optimizers import Adam\nfrom keras.regularizers import l2","execution_count":63,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"72b0b0c1200bd79508bc998a554825ef6bfa6150","collapsed":true},"cell_type":"code","source":"#Implementing a simple 2-Layer Neural Net.\nmodel = Sequential()\nmodel.add(Dense(units = 10, \n                activation = 'relu', \n                kernel_initializer = 'glorot_uniform',\n                kernel_regularizer = l2(0.01),\n                input_dim = 10))\nmodel.add(Dense(units = 1, \n                activation = 'sigmoid', \n                kernel_initializer = 'glorot_uniform', \n                kernel_regularizer = l2(0.01)))\nmodel.compile(optimizer = Adam(lr = 0.0001,\n                               beta_1 = 0.9,\n                               beta_2 = 0.999,\n                               epsilon = 10e-8), \n              loss = 'mean_squared_error', \n              metrics = ['mse'])","execution_count":71,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"7bd9308611766b559bb4697d04ae22541989ba23","collapsed":true},"cell_type":"code","source":"model.fit(train_data, labels, batch_size = 128, epochs = 3, validation_split = 0.01)","execution_count":72,"outputs":[]},{"metadata":{"trusted":true,"collapsed":true,"_uuid":"8746025b89650f3716c43b745c9de301eff0329a"},"cell_type":"code","source":"#Why is the acc: 0.0 and mse:inf always. Please Help!","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.5","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}