{"cells":[{"metadata":{"_cell_guid":"54027fc5-278b-43be-942f-1b41c307ed0a","_uuid":"173fd44a8641082ecc38fd851c49ff8494c83928"},"cell_type":"markdown","source":"This kernel is based on my earlier \"[Simple Linear Stacking](https://www.kaggle.com/aharless/simple-linear-stacking-lb-9730)\" kernel. (Click on the link for more information.) This one represents the base model results as ranks instead of logits. (On the same base model data, the public LB performance was almost identical, but this one was .0001 higher.)  Also, this now uses the updated version of [anttip's FM_FTRL](https://www.kaggle.com/anttip/talkingdata-wordbatch-fm-ftrl-lb-0-9752).\n\nThe latest version uses a new LightGBM model that was fit seprately to multiple periods and recombined, and it otpimizes the stacker's regularization parameter using a time-based split of the validation data."},{"metadata":{"_cell_guid":"f571fc89-4de5-4f43-b9bf-c7c7390292a4","collapsed":true,"_uuid":"844d76db2685e46d9872df1269a873a2919bd8a1","trusted":true},"cell_type":"code","source":"# File containing validation data\n# (These are selected from the last day of the original training set\n#  to correspond to the times of day used in the test set.)\nVAL_FILE = '../input/training-and-validation-data-pickle/validation.pkl.gz'","execution_count":15,"outputs":[]},{"metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","trusted":true},"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport os\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.metrics import roc_auc_score\nfrom sklearn.model_selection import train_test_split\n\nprint(os.listdir(\"../input\"))","execution_count":16,"outputs":[]},{"metadata":{"_cell_guid":"18eb2734-ebcc-49ec-9194-4ddaad317c5f","collapsed":true,"_uuid":"cbc85f52d166fd3e4cb07f75f1d7c24e0fd1f526","trusted":true},"cell_type":"code","source":"almost_zero = 1e-10\nalmost_one = 1 - almost_zero","execution_count":17,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","collapsed":true,"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"# Just names to identify the models\nbase_models = {\n    'lgb1 ': \"assemblage of krishna's LGBM with time deltas\",\n    'wbftl': \"anttip's Wordbatch FM-FTRL\",\n    'nngpu': \"Downampled Neural Network run on GPU\"\n    }","execution_count":18,"outputs":[]},{"metadata":{"_cell_guid":"2d708642-8f1b-4fd3-b33e-500def92647a","collapsed":true,"_uuid":"87839ab18b15bfc16f71016685a5540ca56bd6dc","trusted":true},"cell_type":"code","source":"# Files with validation set predictions from each of the base models\n# (These were fit on a subset of the training data that ends a day before\n#  the end of the full training set.)\ncvfiles = {\n    'lgb1 ': '../input/validation-of-pas-un-m-lange/val_krishnas_r_lgb_bag3.csv',\n    'wbftl': '../input/validate-anttip-s-wordbatch-fm-ftrl-9752-version/wordbatch_fm_ftrl_val.csv',\n    'nngpu': '../input/gpu-nn-validation-more-features/gpu_val3.csv'\n    }","execution_count":19,"outputs":[]},{"metadata":{"_cell_guid":"5b5459d8-0ac9-4c8e-9c22-3e19517d5eb6","collapsed":true,"_uuid":"8e81ec28924b1440ab3604d71a45504f7998e3a4","trusted":true},"cell_type":"code","source":"# Files with test set predictions\n# (These were fit on the full training set\n#  or on a subset at the end, to accommodate memory limits.)\nsubfiles = {\n    'lgb1 ': '../input/ceci-n-est-pas-un-m-lange/sub_krishnas_r_lgb_bag4.csv',\n    'wbftl': '../input/anttip-s-wordbatch-fm-ftrl-9752-version/wordbatch_fm_ftrl.csv',\n    'nngpu': '../input/new-talkingdata-gpu-example-with-multiple-runs/gpu_test3.csv'\n    }","execution_count":20,"outputs":[]},{"metadata":{"_cell_guid":"2fb78497-fea9-442d-b090-228d0979af6a","collapsed":true,"_uuid":"8d6a2eef353372f65772b45eb91486554b249ad4","trusted":true},"cell_type":"code","source":"# Public leaderbaord scores, for comparison\nlbscores = {\n    'lgb1 ': .9759,\n    'wbftl': .9752,\n    'nngpu': .9695\n}","execution_count":21,"outputs":[]},{"metadata":{"_cell_guid":"914b72c1-5cfe-4175-9de4-06af6aa38a45","_uuid":"1f7b5e8d323c5582b8a1d87f3da8152dd1fa83e1"},"cell_type":"markdown","source":"You can click on the \"Data\" tab and follow the links to the kernels that generated each of the outputs above. Usually there are \"forked from\" links that you can follow to see where the models originated. (In the case of Pranav's LGBM, though, the link is misleading, because the original was [a separate kernel in R](https://www.kaggle.com/pranav84/single-lightgbm-in-r-with-75-mln-rows-lb-0-9690), and the fork was from a different Python kernel.)"},{"metadata":{"_cell_guid":"61a43e55-9ba9-41d3-8448-595270d4681e","collapsed":true,"_uuid":"5f2b6a6f7e495c51e75ed6b2f41ed9d9b147acac","trusted":true},"cell_type":"code","source":"model_order = [m for m in base_models] # To keep order consistent when converting to array","execution_count":22,"outputs":[]},{"metadata":{"_cell_guid":"81a2a11d-fa38-4aa2-8b28-8c97fd72a1b0","collapsed":true,"_uuid":"8ba754b121183e83d1542c558150ae1a3bb8313a","trusted":true},"cell_type":"code","source":"cvdata = pd.DataFrame( { \n    m:pd.read_csv(cvfiles[m])['is_attributed'].rank()\n    for m in base_models\n    } )\nX_train = np.array(cvdata[model_order])\ny_train = pd.read_pickle(VAL_FILE)['is_attributed']\nn = len(y_train)\nX_train /= n","execution_count":23,"outputs":[]},{"metadata":{"_cell_guid":"0c713caf-2ee7-432a-8dbd-d10590ce0fd9","_uuid":"ec8a93d77e9063ab6aad20a6c910b54b9798a425","trusted":true},"cell_type":"code","source":"cvdata.head()","execution_count":24,"outputs":[]},{"metadata":{"_cell_guid":"9765374d-f5df-414b-89fd-8ca186453de9","_uuid":"b071581ff35f4900a110c342b275390d1cd4d225","trusted":true},"cell_type":"code","source":"cvdata.corr()","execution_count":25,"outputs":[]},{"metadata":{"_uuid":"1d88a9f12e4009f4a663e5c54138289e4d3977c7"},"cell_type":"markdown","source":"Before fitting the stacking model, let's optimize the regularization parameter."},{"metadata":{"trusted":true,"_uuid":"55845ca437e9275cb5bca7cbd6d3c4220fa691c6"},"cell_type":"code","source":"X, X_val, y, y_val = train_test_split(X_train, y_train, test_size=.33, shuffle=False )\nfor c in [1e-6, 1e-5, 1e-4, 1e-3, 1e-2, 1e-1, 1, 1e1, 1e2, 1e3, 1e4, 1e5, 1e6]:\n    mod = LogisticRegression(C=c)\n    mod.fit(X, y)\n    val_pred = mod.predict_proba(X_val)[:,1]\n    print( c, roc_auc_score(y_val, mod.predict_proba(X_val)[:,1]) )","execution_count":26,"outputs":[]},{"metadata":{"_uuid":"6a019fffe2a20dbd5842997cfc7a1b6f91dd4134"},"cell_type":"markdown","source":"Looks like it wants no regularization, which isn't too surprising, given the simplicity of the model.."},{"metadata":{"_cell_guid":"b3ce0323-0e5f-409e-b404-7c14b616d769","_uuid":"7b8e096669df5ae976feaef5a3165b60de08105f","trusted":true},"cell_type":"code","source":"stack_model = LogisticRegression(C=1e6)\nstack_model.fit(X_train, y_train)\nstack_model.coef_","execution_count":27,"outputs":[]},{"metadata":{"_cell_guid":"f1981258-f50c-47cb-9417-7a52f97255be","_uuid":"63faf7a8a6af88edc671ddffa34a20b131dfc946"},"cell_type":"markdown","source":"Note that the evaluation criterion for this competition (AUC) depends only on rank. Therefore:\n1. Any linear transformation applied to the coefficients won't affect the score of the result.\n2. We don't care much about the value of the intercept term (not shown above).\n3. It doesn't matter whether the coefficients sum to 1.\n4. If we normalize the coefficients to sum to 1, the result has the same interpretation as blending weights."},{"metadata":{"_cell_guid":"c4531d65-0675-446c-80bb-2d21c8dd96f2","_uuid":"17bd987ce19a889c637bd27d48e3bafc60582ee2","trusted":true},"cell_type":"code","source":"weights = stack_model.coef_/stack_model.coef_.sum()\ncolumns = cvdata[model_order].columns\nscores = [ roc_auc_score( y_train, cvdata[c] )  for c in columns ]\nnames = [ base_models[c] for c in columns ]\nlb = [ lbscores[c] for c in columns ]\npd.DataFrame( data={'LB score': lb, 'CV score':scores, 'weight':weights.reshape(-1)}, index=names )","execution_count":28,"outputs":[]},{"metadata":{"_cell_guid":"2c08fb71-6977-4683-a83b-00be98a9afa5","_uuid":"6551c01befb1d8ce2426f6114576fa34f8d2f9e0","trusted":true},"cell_type":"code","source":"print(  'Stacker score: ', roc_auc_score( y_train, stack_model.predict_proba(X_train)[:,1] )  )","execution_count":29,"outputs":[]},{"metadata":{"_cell_guid":"638c438d-f89a-4f16-862e-1963cab66982","collapsed":true,"_uuid":"6695317f3b2022328774893364f04263a8236329","trusted":false},"cell_type":"code","source":"final_sub = pd.DataFrame()\nsubs = {m:pd.read_csv(subfiles[m]).rename({'is_attributed':m},axis=1) for m in base_models}\nfirst_model = list(base_models.keys())[0]\nfinal_sub['click_id'] = subs[first_model]['click_id']","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"2aa5ec3e-8a7d-47a9-8e42-15daf7d2f970","_uuid":"d89f2ac0df48446ddd04b62edaa77f8a2a1acef4","trusted":false,"collapsed":true},"cell_type":"code","source":"df = subs[first_model]\nfor m in subs:\n    if m != first_model:\n        df = df.merge(subs[m], on='click_id')  # being careful in case clicks are in different order\ndf.head()","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"a03efa79-dea6-495d-a071-27543000e1a5","_uuid":"3fa0b547b3435bb5f650a03eda1ff2e49238cef3","trusted":false,"collapsed":true},"cell_type":"code","source":"X_test = np.array( df.drop(['click_id'],axis=1)[model_order].rank()/df.shape[0] )\nfinal_sub['is_attributed'] = stack_model.predict_proba(X_test)[:,1]\nfinal_sub.head(10)","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"263bb278-223a-46a8-a266-2a7d6740ea63","_uuid":"691704fdd769e3a8b81ba0a425e264d253e42bae","trusted":false,"collapsed":true},"cell_type":"code","source":"final_sub['is_attributed'] = final_sub['is_attributed'].rank(method='dense') / 1e8\npd.options.display.float_format = ('{:,.8f}').format\nfinal_sub.head(10)","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"c29b10eb-1897-4b68-af20-2f600652d140","collapsed":true,"_uuid":"bd2cdb0cce3eaf2f6cfee9fe2e3a25d1aacda47c","trusted":false},"cell_type":"code","source":"final_sub.to_csv(\"sub_stacked.csv\", index=False, float_format='%.8f')","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.4","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}