{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"%matplotlib inline\nimport warnings\nwarnings.filterwarnings('ignore')\nimport os\nimport gc\nimport time\nimport pickle\nimport feather\nimport numpy as np\nimport pandas as pd\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nfrom tqdm._tqdm_notebook import tqdm_notebook as tqdm\ntqdm.pandas()","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"DATA_DIR = '../input/'\n# train = pd.read_csv(DATA_DIR+'training_set.csv')\n# test_set_id = pd.read_csv(DATA_DIR+'test_set.csv', usecols=['object_id'])\n# test_set_id.shape # (453653104, 1)\ntrain = pd.read_csv(DATA_DIR+'training_set_metadata.csv')\ntest = pd.read_csv(DATA_DIR+'test_set_metadata.csv')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"4d052647b8dc28a9451fc145fed5a1c5db7b9521"},"cell_type":"code","source":"train.shape, test.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"929067ff9140ad6bf0cbc9e8d53b8e0410bd913f"},"cell_type":"code","source":"display(train.head())\ndisplay(test.head())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1b4e7408cb92a9a3c3e220f444fc41a52eaa01f6"},"cell_type":"code","source":"train.columns","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"07925a1e4c71eade315fe50afd423cb22dafc268","scrolled":true},"cell_type":"code","source":"feat_cols = [\n    'ra', 'decl', 'gal_l', 'gal_b', 'ddf', \n    'hostgal_specz', 'hostgal_photoz', 'hostgal_photoz_err', \n    'distmod', 'mwebv'\n]\nlen(feat_cols)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"6d6777a2686d7838271f8768fe6ccb8a396ea84e"},"cell_type":"code","source":"for c in feat_cols:\n    print(\n        'nan-ratio of {:>18} [train: {:.4f}; test: {:.4f}]'\n        .format(\n            c, \n            train[c].isnull().sum() / train.shape[0], \n            test[c].isnull().sum() / test.shape[0])\n    )","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"72e8535440a5b10f91bfe13160389d07a955443c","scrolled":false},"cell_type":"code","source":"for c in feat_cols:\n    plt.figure(figsize=[20, 4])\n    plt.subplot(1, 2, 1)\n    sns.violinplot(x='target', y=c, data=train)\n    plt.grid()\n    plt.subplot(1, 2, 2)\n    sns.distplot(train[c].dropna())\n    sns.distplot(test[c].dropna())\n    plt.legend(['train', 'test'])\n    plt.grid()\n    plt.show();","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ae132148fd307cb9b5c026091e31ecd01f9ac24d"},"cell_type":"code","source":"target = train['target'].values.copy()\ndel train['target']","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e20a0693da51c59c5251843c64a0599da5c8c3e5"},"cell_type":"code","source":"train_ids = train['object_id'].copy()\ntest_ids = test['object_id'].copy()\ndel train['object_id'], test['object_id'];","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"scrolled":false,"_uuid":"1c6e607bdebe98b25e209b25c9503e55f24a8323"},"cell_type":"code","source":"train['target'] = target.copy()\nsns.pairplot(train[feat_cols+['target']].dropna(), hue='target', vars=feat_cols)\ndel train['target']","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b0d5640017e419fdb79d191d8b72f23064543fb1"},"cell_type":"code","source":"data = pd.concat([\n    train[feat_cols], \n    test[feat_cols].sample(frac=5 * train.shape[0]/test.shape[0])\n], ignore_index=True)\ndata['is_test'] = 1\ndata['is_test'][:train.shape[0]] = 0\nsns.pairplot(data, hue='is_test', vars=feat_cols, plot_kws={'alpha': 0.25})\ndel data; gc.collect();","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"ca387e2a854f9c4b15934a6d63f8fcc31a746e7d"},"cell_type":"markdown","source":"## Inspired by the [great naitive benchmark kernel](https://www.kaggle.com/kyleboone/naive-benchmark-galactic-vs-extragalactic)  \n- We separate the train/test metadata to galactic&extragalactic parts\n- And train two lgb classifiers\n"},{"metadata":{"trusted":true,"_uuid":"de908966edb3ac5a36b310383602a720e2ff861b"},"cell_type":"code","source":"tmp = False\nfor i in [6, 16, 53, 65, 92]:\n    tmp|=(target==i)\nprint(\n    tmp.sum(), \n    (train[['hostgal_specz', 'hostgal_photoz', 'hostgal_photoz_err']].sum(1)==0).sum(), \n    (train['distmod'].isnull()).sum()\n)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c738c95da9014261f061c94974fa505f473022f5"},"cell_type":"code","source":"print(\n    (test[[\n        'hostgal_specz', 'hostgal_photoz', 'hostgal_photoz_err'\n    ]].sum(1)==0).sum(), (test['distmod'].isnull()).sum()\n)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"c24a7f044dbd4cc01a091b554dad9ac2b2803084"},"cell_type":"markdown","source":"### Several conditions seem to be same"},{"metadata":{"trusted":true,"_uuid":"a4f835d523fdcc2fa619c4fb08aec56766417bd7"},"cell_type":"code","source":"train_mask = train['distmod'].isnull().values\ntest_mask = test['distmod'].isnull().values","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"4e33845c4c026223dc52dadcf80bb48bca688410"},"cell_type":"markdown","source":"## Class Weight\n- by https://www.kaggle.com/c/PLAsTiCC-2018/discussion/67194#397146"},{"metadata":{"trusted":true,"_uuid":"aa6316850e99b4b04e203863bec05e66c14d4802"},"cell_type":"code","source":"labels2weight = {x:1 for x in np.unique(target)}\nlabels2weight[64] = 2\nlabels2weight[15] = 2","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"3f0db679e58fb0a7c9df7c7f948ab48fa97a2c23"},"cell_type":"code","source":"import lightgbm as lgb\n\nround_params = dict(num_boost_round = 20000,\n                    early_stopping_rounds = 100,\n                    verbose_eval = 50)\nparams = {\n    \"objective\": \"multiclass\",\n    \"metric\": \"multi_logloss\",\n    #\"num_class\": len(np.unique(y)),\n    #\"two_round\": True,\n    \"num_leaves\" : 30,\n    \"min_child_samples\" : 30,\n    \"learning_rate\" : 0.03,\n    \"feature_fraction\" : 0.75,\n    \"bagging_fraction\" : 0.75,\n    \"bagging_freq\" : 1,\n    \"seed\" : 42,\n    \"lambda_l2\": 1e-2,\n    \"verbosity\" : -1\n}\n\ndef lgb_cv_train(X, labels, X_test, \n                 params=params, round_params=round_params):\n    print('X', X.shape, 'labels', labels.shape, 'X_test', X_test.shape)\n    print('unique labels', np.unique(labels))\n    \n    labels2y = dict(map(reversed, enumerate(np.unique(labels))))\n    y2labels = dict(enumerate(np.unique(labels)))\n    y = np.array(list(map(labels2y.get, labels)))\n    weight = np.array(list(map(labels2weight.get, labels)))\n    \n    params['num_class'] = len(np.unique(y))\n    cv_raw = lgb.cv(\n        params, \n        lgb.Dataset(X, label=y, weight=weight), \n        nfold=5, \n        **round_params\n    )\n    best_round = np.argmin(cv_raw['multi_logloss-mean'])\n    best_score = cv_raw['multi_logloss-mean'][best_round]\n    print(f'best_round: {best_round}', f'best_score: {best_score}')\n    model = lgb.train(\n        params, \n        lgb.Dataset(X, label=y, weight=weight), \n        num_boost_round=best_round, \n    )\n    pred = model.predict(X_test)\n    pred_labels = pd.DataFrame(\n        {f'class_{c}': pred[:, i] for i,c in enumerate(np.unique(labels))}\n    )\n    res = dict(\n        model=model,\n        best_round=best_round,\n        best_score=best_score,\n        pred_labels=pred_labels\n    )\n    return res","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"2b2208f3c91391b2bce09c529542ed2aaf63075d"},"cell_type":"code","source":"feat_extra_li = ['hostgal_specz', 'hostgal_photoz', 'hostgal_photoz_err', 'distmod']\nfeat_gal_cols = ['ra', 'decl', 'gal_l', 'gal_b', 'ddf', 'mwebv']\nfeat_extra_cols = feat_gal_cols + feat_extra_li\nprint(feat_gal_cols)\nprint(feat_extra_cols)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"628962dce6bce3a1b0858157d2dac0cfa27541a9"},"cell_type":"code","source":"np.unique(target[train_mask]), np.unique(target[~train_mask])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"00468ec72f04b8f80570d7baea2bc2810283d828"},"cell_type":"code","source":"%%time\nres_gal = lgb_cv_train(\n    train.loc[train_mask, feat_gal_cols], \n    target[train_mask], \n    test.loc[test_mask, feat_gal_cols]\n)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"8be0bd16826537efc52bf74d89dff3143f557ab1"},"cell_type":"markdown","source":"### Assign the unknown class with average probability"},{"metadata":{"trusted":true,"_uuid":"9656ef4fa34557a079044c2028b3e9c1ec57c9c2"},"cell_type":"code","source":"res_gal['pred_labels'].head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"98f43c6ae324aa48a53fef92de75ed412232ac75"},"cell_type":"code","source":"n_gal = res_gal['pred_labels'].shape[1]\nres_gal['pred_labels'] = res_gal['pred_labels'] * n_gal/(n_gal+1)\nres_gal['pred_labels']['class_99'] = 1/(n_gal+1)\nres_gal['pred_labels'].head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"9b99003553a1384289576e7b7b0db412ed6614a7"},"cell_type":"code","source":"%%time\nres_extra = lgb_cv_train(\n    train.loc[~train_mask, feat_extra_cols], \n    target[~train_mask], \n    test.loc[~test_mask, feat_extra_cols]\n)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d0351a5c8f8e0b68135fa29b5cc1db78a6d9962a"},"cell_type":"code","source":"res_extra['pred_labels'].head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"701a6f985de2ffe680f0897af143795780c5c451"},"cell_type":"code","source":"n_extra = res_extra['pred_labels'].shape[1]\nres_extra['pred_labels'] = res_extra['pred_labels'] * n_extra/(n_extra+1)\nres_extra['pred_labels']['class_99'] = 1/(n_extra+1)\nres_extra['pred_labels'].head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"8ebd453c15daac9bdb60668dfff274b3f38ec09e"},"cell_type":"code","source":"sub = pd.read_csv(DATA_DIR+'sample_submission.csv')\nsub = sub.set_index('object_id')\nsub[:] = 0\nsub.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d5a98ccc7c273d4796bcd7854b8ca64415847f48"},"cell_type":"code","source":"classnames = sub.columns.tolist()\nprint(sub.shape, classnames)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"78beaf7cf45790e45eafe4bc2e8785a07a4c6eca"},"cell_type":"code","source":"for c in res_gal['pred_labels'].columns:\n    sub.loc[test_mask, c] = res_gal['pred_labels'][c].values\nfor c in res_extra['pred_labels'].columns:\n    sub.loc[~test_mask, c] = res_extra['pred_labels'][c].values","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d30708ae2ab81eb701d8294040cab473cd564856"},"cell_type":"code","source":"sub.tail(10)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"bddb275961c1e076dfa770bf54553d164d94709f"},"cell_type":"code","source":"%%time\nscore = res_gal['best_score'] * (train_mask).sum()/train.shape[0]\nscore+= res_extra['best_score'] * (~train_mask).sum()/train.shape[0]\nsub.reset_index().to_csv(f'meta_lgb_{score}.csv', index=False, float_format='%.6f')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ca14a17718c9b9b1c3695913012c3870d62e4a91"},"cell_type":"code","source":"os.listdir()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"2ff633cc947f832f17eb38cd69e20e428b46be5d"},"cell_type":"markdown","source":"## Simple Adversarial Validation\n- run on sampled test set to save time"},{"metadata":{"trusted":true,"_uuid":"365f0a5f2dd5945b0e65750a2a7310cc9b3252dd"},"cell_type":"code","source":"train['is_train'] = 1\ntest['is_train'] = 0\nX = pd.concat(\n    [train, \n     test\n     .sample(frac=5*train.shape[0]/test.shape[0])],\n    ignore_index=True\n)\ny = X['is_train'].values.copy()\ndel X['is_train'], train['is_train'], test['is_train']\ndel X['hostgal_specz'] # this is obvious different disttributed in train/test and will make auc=0.99\nX.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1b74a9cca03032a3964dfac8fdce30995d9e770c"},"cell_type":"code","source":"from sklearn.model_selection import KFold\nfrom sklearn.metrics import roc_auc_score\nparams['objective'] = 'binary'\nparams['metric'] = 'auc'\nparams['num_class'] = 1\n\nprint('X', X.shape, 'y', y.shape)\npred_adv = np.zeros(X.shape[0])\nkf = KFold(n_splits=5, shuffle=True, random_state=42)\ndtrain = lgb.Dataset(X, label=y)\ndtrain.construct()\nmodels = []\nfor trn_idx, val_idx in kf.split(X):\n    model = lgb.train(\n        params, \n        dtrain.subset(trn_idx), \n        valid_sets=[dtrain.subset(trn_idx), dtrain.subset(val_idx)], \n        valid_names=['train', 'valid'],\n        **round_params\n    )\n    models.append(model)\n    pred_adv[val_idx] = model.predict(X.iloc[val_idx])\nadv_score = roc_auc_score(y, pred_adv)\nprint('oof score:', adv_score)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"cda8b25ecb162f4ee861db01cd2d934afc80fa59"},"cell_type":"markdown","source":"### Maybe this explains why native benchmark outperforms LGB classifier trained on train(meta) dataset\n### Check the feature importance"},{"metadata":{"trusted":true,"_uuid":"d7dd671f261c4b5722d4216d14157145761a2455"},"cell_type":"code","source":"def get_imp_plot(name, feature_name, lgb_feat_imps, nfolds=5, savefig=False):\n    lgb_imps = pd.DataFrame(\n        np.vstack(lgb_feat_imps).T, \n        columns=['fold_{}'.format(i) for i in range(nfolds)],\n        index=feature_name,\n    )\n    lgb_imps['fold_mean'] = lgb_imps.mean(1)\n    lgb_imps = lgb_imps.loc[\n        lgb_imps['fold_mean'].sort_values(ascending=False).index\n    ]\n    lgb_imps.reset_index().to_csv(f'{name}_lgb_imps.csv', index=False)\n    del lgb_imps['fold_mean']; gc.collect();\n\n    max_num_features = min(len(feature_name), 300)\n    f, ax = plt.subplots(figsize=[8, max_num_features//2])\n    data = lgb_imps.iloc[:max_num_features].copy()\n    data_mean = data.mean(1).sort_values(ascending=False)\n    data = data.loc[data_mean.index]\n    data_index = data.index.copy()\n    data = [data[c].values for c in data.columns]\n    data = np.hstack(data)\n    data = pd.DataFrame(data, index=data_index.tolist()*nfolds, columns=['igb_imp'])\n    data = data.reset_index()\n    data.columns = ['feature_name', 'igb_imp']\n    sns.barplot(x='igb_imp', y='feature_name', data=data, orient='h', ax=ax)\n    plt.grid()\n    if savefig:\n        plt.savefig(f'{name}_lgb_imp.png')\n\nget_imp_plot(\n    'adv', \n    X.columns.tolist(), \n    [model.feature_importance()/model.best_iteration for model in models]\n)","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}