{"cells":[{"metadata":{},"cell_type":"markdown","source":"## If you enjoy this approach or it helped you somehow, please UPvote this kernel :)"},{"metadata":{},"cell_type":"markdown","source":"## There are just *so many features*. Besides, their meaning are *unknown* to us. Therefore, why don't try to somehow reduce this high dimensionality?\n## SVD to rescue!"},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport pickle\nfrom collections import Counter\n\nfrom sklearn.decomposition import TruncatedSVD\nfrom sklearn.manifold import TSNE\nimport os\nfrom sklearn.multioutput import MultiOutputRegressor\nfrom xgboost import XGBRegressor","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"# self exlanatory function\ndef load_year(y):\n    dtypes = pickle.load(open(\n        f'../input/kdd2020-cpr/{y}_dtypes.pkl', 'rb'\n    ))\n    del dtypes['date']\n    df = pd.read_csv(\n        f'../input/kdd2020-cpr/{y}.csv',\n        dtype=dtypes, parse_dates=['date']\n    )\n    return df","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"YEAR = 2018\nCOMPONENTS = 100\nXGB_ESTIMATORS = 50\nMAGICIAN = 'TruncatedSVD'","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"ls ../input","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"%%time\ndf = load_year(YEAR)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"%%time\ndef drop_fullnull(df, inplace=False):\n    mask = df.isnull().all()\n    labels = df.columns[mask]\n    \n    shape = df.shape\n    print(labels)\n    if inplace:\n        df.drop(labels=labels, axis=1, inplace=True)\n    else:\n        df = df.drop(labels=labels, axis=1)\n    if labels.any():\n        if shape == df.shape:\n            print('lables:', labels)\n        else:\n            print(shape, df.shape)\n    return df\n\n# I tried this but thinks end up more complicated\n# since the shape may change\n# clean = drop_fullnull(df)\n# clean.shape","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Let's \"train\" the SVD instance"},{"metadata":{"trusted":true},"cell_type":"code","source":"magician = TruncatedSVD(\n    n_components=COMPONENTS,\n    random_state=0\n)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"cp = df.copy()\n\n# Simplest imputation of all\ncp.fillna(0, inplace=True)\n\nout_mask = df.columns.str.contains('output')\nout_cols = df.columns[df.columns.str.contains('output')]\ncp_out = cp.loc[:, out_mask]\ncp.drop(['id'] + out_cols.tolist(),\n        axis=1, inplace=True\n)\ncp['day'] = cp.date.dt.day\ncp['month'] = cp.date.dt.month\ncp['year'] = cp.date.dt.year\ncp.drop('date', inplace=True, axis=1)\n\nmagician.fit(cp)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"xgb_params = {\n    'n_estimators': XGB_ESTIMATORS,\n    'random_state': 0,\n    'n_jobs': -1,\n    'learning_rate': .1,\n    'max_depth': 10,\n    'tree_method': 'gpu_hist',\n    'verbosity': 2,\n    'objective': 'reg:squarederror',\n    \n}\nmodel = XGBRegressor(**xgb_params)\nclf = MultiOutputRegressor(model)\n\nclf.fit(pd.DataFrame(magician.transform(cp)), cp_out);","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"And now let's predict and submit"},{"metadata":{"trusted":true},"cell_type":"code","source":"d9 = load_year(2019)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def svd_preprocess(df):\n    cp = df.copy()\n    ids = df.id.copy()\n    \n    cp.fillna(0, inplace=True)\n    cp.drop(['id'], axis=1, inplace=True, errors='ignore')\n    cp['day'] = cp.date.dt.day\n    cp['month'] = cp.date.dt.month\n    cp['year'] = cp.date.dt.year\n    cp.drop('date', inplace=True, axis=1)\n    return cp, ids\n\nX, ids = svd_preprocess(d9)\n\ndf_in = pd.DataFrame(magician.transform(X))\ndf_in.index = ids","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"y_pred = clf.predict(df_in)\n\nout_cols_flat = out_cols.ravel()\nid_col = []\nfor i in ids:\n    id_col.extend([f'{i}_{sufix}' for sufix in out_cols_flat])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df_sub = pd.DataFrame(\n    {'id':id_col , 'value':y_pred.ravel()}\n)\ndf_sub.set_index('id', inplace=True)\ndf_sub.to_csv(\n    f'submission-{YEAR}-{COMPONENTS}{MAGICIAN}-{XGB_ESTIMATORS}xgb.csv',\n    index='id'\n)\n# If you want to use the model later, just uncomment\n# line below\n# pickle.dump(clf, open('clf.pkl', 'wb'))","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}