{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport os\n\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.metrics import roc_auc_score\nimport lightgbm as lgb\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport gc\n\n%matplotlib inline\n\nprint(os.listdir(\"../input\"))\n","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"974a88751e19a806e76ebb2334463e02ff7a9753"},"cell_type":"markdown","source":"### Read meta data"},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"train_meta = pd.read_csv('../input/training_set_metadata.csv')\ntest_meta = pd.read_csv('../input/test_set_metadata.csv')","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"80356635be130e92208849afba2689e3a9f62fa6"},"cell_type":"markdown","source":"### Drop object_id, target and create is_train feature"},{"metadata":{"trusted":true,"_uuid":"2ba81acad3c25b1d39db8aad1d38a2702f48715e"},"cell_type":"code","source":"# Make sure I can rerun that any time\nif 'object_id' in train_meta:\n    del train_meta['object_id'], train_meta['target']\n    del test_meta['object_id']\n    gc.collect()\n    \ntrain_meta['is_train'] = 1\ntest_meta['is_train'] = 0","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"0e4b30c8f602b1b34c323757365b64c850a0d4d1"},"cell_type":"markdown","source":"### Create full dataset"},{"metadata":{"trusted":true,"_uuid":"800d9c319bce32ca2427bcd32678fa9bb317b60c"},"cell_type":"code","source":"np.random.seed(10)\nfull_meta = pd.concat([train_meta, test_meta], axis=0).sample(frac=1.0)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"7a2b3abcaf3688b24903ac361ca2e1e7a8f2d139"},"cell_type":"markdown","source":"### Run a first Adversarial validation"},{"metadata":{"trusted":true,"_uuid":"821cc493658da3345ca6533ba6af5d598d07ece4"},"cell_type":"code","source":"folds = StratifiedKFold(n_splits=5, shuffle=True, random_state=1)\n\nfeatures = [f for f in full_meta if f not in ['is_train']]\nimportances = pd.DataFrame()\n\nlgb_params = {\n    'n_estimators': 200,\n    'boosting_type': 'rf',\n    'learning_rate': .05,\n    'num_leaves': 127,\n    'subsample': 0.621,\n    'colsample_bytree': 0.7,\n    'max_depth': 7,\n    'bagging_freq': 1,\n}\noof_preds = np.zeros(full_meta.shape[0])\nfor fold_, (trn_, val_) in enumerate(folds.split(full_meta, full_meta['is_train'])):\n    trn_x, trn_y = full_meta[features].iloc[trn_], full_meta['is_train'].iloc[trn_]\n    val_x, val_y = full_meta[features].iloc[val_], full_meta['is_train'].iloc[val_]\n    \n    clf = lgb.LGBMClassifier(**lgb_params)\n    \n    clf.fit(trn_x, trn_y)\n    \n    oof_preds[val_] = clf.predict_proba(val_x)[:, 1]\n    \n    imp_df = pd.DataFrame()\n    imp_df['feature'] = features\n    imp_df['gain'] = clf.booster_.feature_importance(importance_type='gain')\n    imp_df['split'] = clf.booster_.feature_importance(importance_type='split')\n    imp_df['fold'] = fold_ + 1\n    importances = pd.concat([importances, imp_df], axis=0, sort=False)\n\n    del trn_x, trn_y, val_x, val_y\n    gc.collect()\n\nprint('Train/Test separation score : %.6f' % roc_auc_score(full_meta['is_train'], oof_preds))","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"a327a0dc9f4ddcc1cf5677adf912bf364e8e4f6e"},"cell_type":"markdown","source":"### Display big contributors"},{"metadata":{"trusted":true,"_uuid":"a1aba9df9ec7688537c94a293a28b8adb4758b3f"},"cell_type":"code","source":"mean_gain = importances[['gain', 'feature']].groupby('feature').mean()\nimportances['mean_gain'] = importances['feature'].map(mean_gain['gain'])\n\nplt.figure(figsize=(8, 12))\nsns.barplot(x='gain', y='feature', data=importances.sort_values('mean_gain', ascending=False))\nplt.tight_layout()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"afc756f785d7d797911c25f4de425a391589aa47"},"cell_type":"markdown","source":"This shows that **hostgal_specz** is the biggest contributor by far and that is due to :"},{"metadata":{"trusted":true,"_uuid":"1319770f4b4dfcc4806adedd4353fac7fadf0e80"},"cell_type":"code","source":"nulls = pd.concat([train_meta.isnull().mean(), test_meta.isnull().mean()], axis=1).sort_index()\nnulls.columns = ['train', 'test']\nnulls\n","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"2b61c3bf5e75b7c2de2aaf33fbfd594054a96f12"},"cell_type":"markdown","source":"Most of the samples in test don't have a **hostgal_specz** measure as it is stated in the data section of the challenge.\n\nTherefore any model trained on it may have trouble generalyzing ..."},{"metadata":{"trusted":true,"_uuid":"67e379429c378c6f636f05898ab403584a163f25"},"cell_type":"code","source":"del train_meta, test_meta, full_meta\ngc.collect()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"6c2f7acadbfa7b25224c181b4140b83ff8e2a663"},"cell_type":"markdown","source":"### Can we run adversarial validation on train and test datasets"},{"metadata":{"trusted":true,"_uuid":"0d14b956175297710a1f07a8bf75a93af301dbfd"},"cell_type":"code","source":"train = pd.read_csv('../input/training_set.csv')\ntrain.shape\n# Drop object_id\ndel train['object_id']\ntrain['is_train'] = 1","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"a30aeebee6503d3a7bacb54378066dad18db6c33"},"cell_type":"markdown","source":"Test contains a lot more samples so will go through them by chuncks of 5e6"},{"metadata":{"trusted":true,"_uuid":"127456ccbe87a9b57b7b737b2c6680de8af9dc31"},"cell_type":"code","source":"import time\nstart = time.time()\nchunks = 5000000\nimportances = pd.DataFrame()\nfor i_c, df in enumerate(pd.read_csv('../input/test_set.csv', chunksize=chunks, iterator=True)):\n    # Drop object_id\n    del df['object_id']\n    # Add is_train feature\n    df['is_train'] = 0\n    # Concat\n    full_data = pd.concat([df, train], axis=0).sample(frac=1.0)\n    y = full_data['is_train']\n    del full_data['is_train'], df\n    gc.collect()\n    \n    # Run a lightgbm\n    folds = StratifiedKFold(n_splits=4, shuffle=True, random_state=1)\n    \n    oof_preds = np.zeros(full_data.shape[0])\n    for fold_, (trn_, val_) in enumerate(folds.split(full_data, y)):\n        trn_x, trn_y = full_data.iloc[trn_], y.iloc[trn_]\n        val_x, val_y = full_data.iloc[val_], y.iloc[val_]\n\n        clf = lgb.LGBMClassifier(**lgb_params)\n\n        clf.fit(trn_x, trn_y)\n\n        oof_preds[val_] = clf.predict_proba(val_x)[:, 1]\n\n        imp_df = pd.DataFrame()\n        imp_df['feature'] = full_data.columns\n        imp_df['gain'] = clf.booster_.feature_importance(importance_type='gain')\n        imp_df['split'] = clf.booster_.feature_importance(importance_type='split')\n        imp_df['fold'] = i_c * folds.n_splits + fold_ + 1\n        importances = pd.concat([importances, imp_df], axis=0, sort=False)\n\n        del trn_x, trn_y, val_x, val_y\n        gc.collect()\n\n    print('Train/Test separation score : %.6f' % roc_auc_score(y, oof_preds))\n    \n    print('Chunk %3d Adversarial validation done [%5.1fmin spent]' % (i_c+1, (time.time() - start)/60))\n    \n    del full_data\n    gc.collect()\n    \n    if i_c > 4 :\n        break","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"ad2fd69151e55e0c9fdd46151d81b420633f9c2a"},"cell_type":"markdown","source":"### Display contributors"},{"metadata":{"trusted":true,"_uuid":"0f581c4484336dbd8909859f649bdbbf622220d7"},"cell_type":"code","source":"mean_gain = importances[['gain', 'feature']].groupby('feature').mean()\nimportances['mean_gain'] = importances['feature'].map(mean_gain['gain'])\n\nplt.figure(figsize=(8, 12))\nsns.barplot(x='gain', y='feature', data=importances.sort_values('mean_gain', ascending=False))\nplt.tight_layout()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"34ec4303b5b5b79665c3d7f2161e6da702bb7f96"},"cell_type":"markdown","source":"It looks like only meta data features may be a problem.\nWe now need to find a way to predict **class 99**"},{"metadata":{"trusted":true,"_uuid":"91ab45ba2a3c6b04ce3872ece9fabf220bb2194f"},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}