{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"scrolled":true},"cell_type":"code","source":"%matplotlib inline\nimport warnings\nwarnings.filterwarnings(\"ignore\")\nimport os\nimport gc\nimport numpy as np\nimport pandas as pd\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nfrom tqdm import tqdm\n\nDATA_DIR = '../input/avito-demand-prediction/'\ntextdata_path = '../input/adp-prepare-kfold-text/textdata.csv'\ntarget_col = 'deal_probability'\nos.listdir(DATA_DIR)","execution_count":1,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"594f249b0e4d245879d1423659dc5dbf21a52e3d"},"cell_type":"code","source":"geo_detail_path = '../input/region-and-city-details-with-lat-lon-and-clusters/avito_region_city_features.csv'\ngeo_detail = pd.read_csv(geo_detail_path)\nfor c in ['city_region', 'region_id', 'city_region_id']:\n    del geo_detail[c]\ngeo_detail.head(2).T","execution_count":2,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a65eb50098e0a6a22c6da0fd3da76429d6325be3"},"cell_type":"code","source":"active_period_feats = pd.read_csv('../input/adp-active-user-feats/active_period_feats.csv')\nactive_period_feats.head()","execution_count":3,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"6b224736e0e371421fec9591a4b89d7fb8d4f3aa"},"cell_type":"code","source":"active_period_feats.columns.tolist()","execution_count":4,"outputs":[]},{"metadata":{"trusted":true,"collapsed":true,"_uuid":"2ad744c6027426f389b38dfefe4577c87f7e162f"},"cell_type":"code","source":"act_feat_cols = ['avg_days_from_act_user',\n                 'avg_days_up_user',\n                 'avg_days_up_sum_user',\n                 'avg_times_up_user',\n                 'n_user_items']","execution_count":5,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"usecols = ['user_id', #'item_id',\n           'region', 'city', 'parent_category_name', 'category_name', \n           'param_1', 'param_2', 'param_3', \n           'activation_date',\n           'title', 'description', \n           'price', 'item_seq_number', \n           'user_type', \n           'image_top_1', 'image']\neval_sets = pd.read_csv(textdata_path, usecols=['eval_set'])['eval_set'].values\ntrain_num = (eval_sets!=10).sum()\neval_sets = eval_sets[:train_num]\ntrain = pd.read_csv(DATA_DIR+'train.csv', usecols=usecols+[target_col])\ntest = pd.read_csv(DATA_DIR+'test.csv', usecols=usecols)\ntrain = train.merge(active_period_feats, on='user_id', how='left')\ntest = test.merge(active_period_feats, on='user_id', how='left')\ndel active_period_feats; gc.collect()","execution_count":6,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"0ec2b0b119b870d084226cb6dfd3d8b61d6181e1"},"cell_type":"code","source":"def get_dow(df):\n    f = lambda x:pd.to_datetime(x).dayofweek\n    unq = df['activation_date'].unique().tolist()\n    d = dict([u, f(u)] for u in unq)\n    df['dow'] = df['activation_date'].map(d.get)\n    return df\ntrain = get_dow(train)\ntest = get_dow(test)\ndel train['activation_date'], test['activation_date']; gc.collect()","execution_count":7,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"284150e83adf5d084363cef5f0304e37df068bd8"},"cell_type":"code","source":"len(set(train['user_id'].values.tolist()) & set(test['user_id'].values.tolist())), len(set(test['user_id'].values.tolist()))","execution_count":8,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"70a837911b84fc6734325f590e85a8347cface4b"},"cell_type":"code","source":"common_indexes = set(train['user_id'].values.tolist()) & set(test['user_id'].values.tolist())\ncommon_indexes = list(common_indexes)\ntrain['user_common'] = 0\ntest['user_common'] = 0\ntrain = train.set_index('user_id')\ntest = test.set_index('user_id')\ntrain.loc[common_indexes, 'user_common'] = 1\ntest.loc[common_indexes, 'user_common'] = 1\ntrain = train.reset_index()\ntest = test.reset_index()\ndel common_indexes; gc.collect()","execution_count":9,"outputs":[]},{"metadata":{"trusted":true,"collapsed":true,"_uuid":"6966eb97051715b69a3f25579c550f6cf30b1a9f"},"cell_type":"code","source":"train['user_id_common'] = train['user_id'].values\ntrain.loc[train['user_common']==0, 'user_id_common'] = 'unknown'\ntest['user_id_common'] = test['user_id'].values\ntest.loc[test['user_common']==0, 'user_id_common'] = 'unknown'","execution_count":10,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ef2f848be55b191fde5d4c6004817cd531dbd1c2"},"cell_type":"code","source":"train.head(2).T","execution_count":11,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d439f9f9500a5831806dd6435823dc9d68e7a0ef"},"cell_type":"code","source":"train_num == len(train)","execution_count":12,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b8fca9e67c9729ed74a37480cce0afd5e7eb4434"},"cell_type":"code","source":"y = train[target_col].values\ndel train[target_col]; gc.collect()\ntrain_num = len(train)\ndf = pd.concat([train, test], ignore_index=True)\ndel train, test; gc.collect()","execution_count":13,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"80fa6a443a99d7746f8b048eafee99e35358fb91"},"cell_type":"code","source":"geo = df[['city', 'region']].merge(geo_detail, on=['city', 'region'], how='left')\ngeo.head(1).T","execution_count":14,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"3a382d0fc0a94cf5b211a82edf4c0276c3c72146"},"cell_type":"code","source":"geo.columns.tolist()","execution_count":15,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"54cc9bfa1def6e9dcadef426f8e9bcde8633c090","collapsed":true},"cell_type":"code","source":"geo_cols = ['latitude',\n            'longitude',\n            'lat_lon_hdbscan_cluster_05_03',\n            'lat_lon_hdbscan_cluster_10_03',\n            'lat_lon_hdbscan_cluster_20_03']\ngeo = geo[geo_cols]\ngeo.to_csv('geo_detail.csv', index=False)","execution_count":16,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"901801c179b83a37bb22e51f47baee46ed03c795"},"cell_type":"code","source":"df['image'].isnull().sum()","execution_count":17,"outputs":[]},{"metadata":{"trusted":true,"collapsed":true,"_uuid":"a2605cc08df180d21fd606a6d7211afd76757794"},"cell_type":"code","source":"df['image'] = (~df['image'].isnull()).astype('int8')","execution_count":18,"outputs":[]},{"metadata":{"trusted":true,"collapsed":true,"_uuid":"3d98ecfba2194930034d370642a84c45864cae76"},"cell_type":"code","source":"del df['user_common']; gc.collect();","execution_count":19,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"2da64aeaf0949a2c92855f7ed75a649d41cbbd08"},"cell_type":"code","source":"df['image_top_1'].isnull().sum()","execution_count":20,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c1ef71596234dc70c87e5647b0adb97e62f286d9"},"cell_type":"code","source":"df['image_top_1'].min(), df['image_top_1'].max()","execution_count":21,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a2382c044a334689bfae5b171f097cd13ec2d8e1"},"cell_type":"code","source":"df.head(3).T","execution_count":22,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"fc88ce257f8e5d7a0ddb2f6552fe1fc4ad558e84"},"cell_type":"code","source":"enc_cols = ['user_id', 'user_id_common',\n            'region', 'city', 'parent_category_name', 'category_name', \n            'param_1', 'param_2', 'param_3', \n            'user_type', \n            'image_top_1']\n\n# pair_cols = [('image_top_1', 'city'), \n#              ('image_top_1', 'region'), \n#              ('image_top_1', 'param_1'), \n#              ('city', 'region'), \n#              ('city', 'param_1'), \n#              ('region', 'param_1')]\n\n# for i, pair in enumerate(pair_cols):\n#     print('column pairing', i, pair)\n#     p_colname = 'P_'+'X'.join(pair)\n#     df[p_colname] = ''\n#     for p in pair:\n#         df[p_colname] += df[p].fillna('unknown').astype(str)\n#     enc_cols.append(p_colname)\n\nenc_dict = {}\nfor i, c in enumerate(enc_cols):\n    print('label encoding', i, c)\n    values, names = pd.factorize(df[c].fillna('unknown'))\n    df[c] = values\n    #enc_dict[c] = pd.DataFrame(names.values, columns=['lbe'])\n    #enc_dict[c].to_csv(c+'_enc.csv', index=False)","execution_count":23,"outputs":[]},{"metadata":{"trusted":true,"collapsed":true,"_uuid":"b1a0c41f9fffe2d098161160fb3a6979a65943f6"},"cell_type":"code","source":"df['price_bin'] = pd.cut(np.log1p(df['price']), 256, labels=np.arange(256))\ndf['price_bin'] = df['price_bin'].astype('float').fillna(-1)\ndf['price_bin'] = df['price_bin'].astype('int')\n\ndf['item_seq_bin'] = pd.cut(np.log1p(df['item_seq_number']), 512, labels=np.arange(512))\ndf['item_seq_bin'] = df['item_seq_bin'].astype('float').fillna(-1)\ndf['item_seq_bin'] = df['item_seq_bin'].astype('int')","execution_count":24,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"546e1baca38a44f3ee4a226a052f0c4b9a9029eb"},"cell_type":"code","source":"df.head(3).T","execution_count":25,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"7f059d0fb804f8fdfd9a26c5eae5b8cec77c80d3"},"cell_type":"code","source":"del df['title'], df['description']; gc.collect();\ndf.info()","execution_count":26,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"13f60f75142dd558fa815acf69f1b9c4395de4f3","scrolled":false},"cell_type":"code","source":"def reduce_memory(df):\n    for c in df.columns:\n        if df[c].dtype=='int':\n            if df[c].min()<0:\n                if df[c].abs().max()<2**7:\n                    df[c] = df[c].astype('int8')\n                elif df[c].abs().max()<2**15:\n                    df[c] = df[c].astype('int16')\n                elif df[c].abs().max()<2**31:\n                    df[c] = df[c].astype('int32')\n                else:\n                    continue\n            else:\n                if df[c].max()<2**8:\n                    df[c] = df[c].astype('uint8')\n                elif df[c].max()<2**16:\n                    df[c] = df[c].astype('uint16')\n                elif df[c].max()<2**32:\n                    df[c] = df[c].astype('uint32')\n                else:\n                    continue\n    return df\ndf = reduce_memory(df)\nprint(df.info())","execution_count":27,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"2728b3dd3aed1f27f3eda0e4add85987e2e995b1"},"cell_type":"code","source":"act_feat_cols","execution_count":29,"outputs":[]},{"metadata":{"trusted":true,"collapsed":true,"_uuid":"f44e7b81dd8234a3d313e8e7656bb2fc9d7c01b6"},"cell_type":"code","source":"cols = ['user_id',\n        'user_id_common',\n        'region',\n        'city',\n        'parent_category_name',\n        'category_name',\n        'param_1',\n        'param_2',\n        'param_3',\n        'price', 'price_bin', \n        'item_seq_number', 'item_seq_bin', \n        'user_type',\n        'image',\n        'image_top_1',\n        'dow'] + \\\n        act_feat_cols\ndf = df[cols]","execution_count":30,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"0305e9afcdc4526a30a829bc1b3a1155e3148a27"},"cell_type":"code","source":"df.shape","execution_count":31,"outputs":[]},{"metadata":{"trusted":true,"collapsed":true,"_uuid":"9451146317a9382ee96baf0719cf9266f23622fe"},"cell_type":"code","source":"df.to_csv('data_lbe.csv', index=False)","execution_count":32,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.5","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}