{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Jane Street Market Prediction (#2.2)\n## Imputation, downsizing dataset.\n\nLoaded by all training notebooks.<br>\nEvaluated in https://www.kaggle.com/wendellavila/janestreet-preprocessing-selection/\n\nNotebook Navigation<br>\n[All](https://www.kaggle.com/wendellavila/janestreet-index/) | [#1](https://www.kaggle.com/wendellavila/janestreet-model-selection/) | [#2.1](https://www.kaggle.com/wendellavila/janestreet-preprocessing-selection) | [#2.2](https://www.kaggle.com/wendellavila/janestreet-data-preprocessing) | [#3](https://www.kaggle.com/wendellavila/janestreet-regularization-selection) | [#4.1](https://www.kaggle.com/wendellavila/janestreet-hyperparameter-tuning) | [#4.2](https://www.kaggle.com/wendellavila/janestreet-hyperparameter-evaluation) | [#5.1](https://www.kaggle.com/wendellavila/janestreet-pca) | [#5.2](https://www.kaggle.com/wendellavila/janestreet-autoencoder) | [#5.3](https://www.kaggle.com/wendellavila/janestreet-dimensionality-reduction-evaluation) |[#6](https://www.kaggle.com/wendellavila/janestreet-ensemble)","metadata":{}},{"cell_type":"markdown","source":"## Imports","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nimport sklearn\n#from sklearn.experimental import enable_iterative_imputer\nfrom sklearn.impute import SimpleImputer#, IterativeImputer\npd.set_option('display.max_columns', 300)","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-14T03:11:24.585481Z","iopub.execute_input":"2022-07-14T03:11:24.586127Z","iopub.status.idle":"2022-07-14T03:11:25.879785Z","shell.execute_reply.started":"2022-07-14T03:11:24.586089Z","shell.execute_reply":"2022-07-14T03:11:25.878874Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Misc","metadata":{}},{"cell_type":"code","source":"#downsizing dataframe for faster loading\ndef reduce_dtypes(df):\n    \n    for col in df.columns:\n        col_type = df[col].dtype\n        \n        if col_type != object:\n            c_min = df[col].min()\n            c_max = df[col].max()\n            if str(col_type)[:3] == 'int':\n                if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                    df[col] = df[col].astype(np.int8)\n                elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                    df[col] = df[col].astype(np.int16)\n                elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                    df[col] = df[col].astype(np.int32)\n                else:\n                    df[col] = df[col].astype(np.int64)  \n            else:\n                if c_min > np.finfo(np.float16).min and c_max < np.finfo(np.float16).max:\n                    df[col] = df[col].astype(np.float16)\n                elif c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                    df[col] = df[col].astype(np.float32)\n                else:\n                    df[col] = df[col].astype(np.float64)\n        else:\n            df[col] = df[col].astype('category')\n            \n    return df","metadata":{"execution":{"iopub.status.busy":"2022-07-14T03:11:59.590637Z","iopub.execute_input":"2022-07-14T03:11:59.591018Z","iopub.status.idle":"2022-07-14T03:11:59.608642Z","shell.execute_reply.started":"2022-07-14T03:11:59.590987Z","shell.execute_reply":"2022-07-14T03:11:59.607505Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def missing_info(df):\n    print(\"Total N° of NaN: \", df.isnull().sum().sum())\n    col_nan = df.columns[df.isnull().any()]\n    print(\"N° of columns with NaN: \", len(col_nan))\n    df[col_nan]","metadata":{"execution":{"iopub.status.busy":"2022-07-06T15:05:54.51407Z","iopub.execute_input":"2022-07-06T15:05:54.514387Z","iopub.status.idle":"2022-07-06T15:05:54.522777Z","shell.execute_reply.started":"2022-07-06T15:05:54.514355Z","shell.execute_reply":"2022-07-06T15:05:54.521869Z"},"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def add_missing_indicator(df):\n    col_nan = df.columns[df.isnull().any()]\n    missing_i = df[col_nan].isnull().astype('float64').add_suffix('_missing')\n    return pd.concat([df, missing_i], axis=\"columns\")","metadata":{"execution":{"iopub.status.busy":"2022-07-06T15:05:54.524116Z","iopub.execute_input":"2022-07-06T15:05:54.526017Z","iopub.status.idle":"2022-07-06T15:05:54.534151Z","shell.execute_reply.started":"2022-07-06T15:05:54.525953Z","shell.execute_reply":"2022-07-06T15:05:54.532922Z"},"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Visualizing data","metadata":{}},{"cell_type":"markdown","source":"# data = reduce_dtypes(pd.read_hdf('../input/jane-street-market-train-data-best-formats/jane_street_train.h5'))\n# data = pd.read_hdf('../input/jane-street-market-train-data-best-formats/jane_street_train.h5')\n# features = [c for c in data.columns if 'feature' in c]\n# data","metadata":{}},{"cell_type":"code","source":"# missing_info(data)\n# missing = pd.DataFrame(df[col_nan].isnull().sum().sort_values(ascending=False)*100/df.shape[0],columns=['missing %']).T\n# missing.style.background_gradient(cmap='Blues', axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-07-06T15:05:54.546588Z","iopub.execute_input":"2022-07-06T15:05:54.547102Z","iopub.status.idle":"2022-07-06T15:05:54.558864Z","shell.execute_reply.started":"2022-07-06T15:05:54.547054Z","shell.execute_reply":"2022-07-06T15:05:54.557982Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # Display the histogram \n# fig,axes = plt.subplots(nrows=45,ncols=3,figsize=(25,250))\n\n# for i in range(2,137):\n#     sns.distplot(data.iloc[:,i],ax=axes[(i-2)//3,(i-2)%3])","metadata":{"execution":{"iopub.status.busy":"2022-07-06T15:05:54.630553Z","iopub.execute_input":"2022-07-06T15:05:54.63095Z","iopub.status.idle":"2022-07-06T15:05:54.634661Z","shell.execute_reply.started":"2022-07-06T15:05:54.630918Z","shell.execute_reply":"2022-07-06T15:05:54.633887Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# del data","metadata":{"execution":{"iopub.status.busy":"2022-07-06T15:05:55.183884Z","iopub.execute_input":"2022-07-06T15:05:55.184438Z","iopub.status.idle":"2022-07-06T15:05:55.187637Z","shell.execute_reply.started":"2022-07-06T15:05:55.184402Z","shell.execute_reply":"2022-07-06T15:05:55.18672Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Preprocessing Pipeline","metadata":{}},{"cell_type":"code","source":"def preprocessing_pipeline(imputation='none', name='none', addIndicator=False, removeExtraResp=False, remove0=False):\n    print(\"Loading data...\")\n    data = pd.read_hdf('../input/jane-street-market-train-data-best-formats/jane_street_train.h5')\n    print(\"Data loaded. Working...\")\n    features = [c for c in data.columns if 'feature' in c]\n    resps = [c for c in data.columns if 'resp' in c]\n    \n    #filtering out rows with weight == 0\n    if(remove0 == True):\n        data = data.query('weight > 0').reset_index(drop=True)\n    \n    if(removeExtraResp == True):\n        data = data[['date'] + ['weight'] + ['resp'] + features]\n    else:\n        data = data[['date'] + ['weight'] + resps + features]\n        \n    if(addIndicator == True):\n        data = add_missing_indicator(data)\n    \n    data['action'] = (data['resp'] > 0.000000001)*1\n    \n    train_data = data[data['date']<450]\n    val_data = data[data['date']>=450]\n    del data\n    \n    print(\"Imputting...\")\n    #imputation\n    if(imputation == 'mean'):\n        mean = train_data.mean()\n        train_data.fillna(value=mean, inplace=True)\n        val_data.fillna(value=mean, inplace=True)\n    elif(imputation == 'ffil'):\n        train_data.fillna(method='ffill', inplace=True)\n        val_data.fillna(method='ffill', inplace=True)\n        mean = train_data.mean()\n        train_data.fillna(value=mean, inplace=True)\n        val_data.fillna(value=mean, inplace=True)\n#     elif(imputation == 'iterative'):\n#         imp = IterativeImputer(max_iter=10)\n#         temp_data = pd.DataFrame(imp.fit_transform(train_data))\n#         temp_data.columns=train_data.columns\n#         temp_data.index=train_data.index\n#         train_data = temp_data\n        \n#         temp_data = pd.DataFrame(imp.transform(val_data))\n#         temp_data.columns=val_data.columns\n#         temp_data.index=val_data.index\n#         val_data = temp_data\n#         del temp_data\n    \n    #reducing dtypes of dataframe for faster loading\n    train_data = reduce_dtypes(train_data)\n    val_data = reduce_dtypes(val_data)\n    train_data.to_pickle(f'train-{name}.pkl')\n    val_data.to_pickle(f'val-{name}.pkl')\n    print(\"\\nTrain missing values:\")\n    missing_info(train_data)\n    print(\"\\nVal missing values:\")\n    missing_info(val_data)\n    del train_data, val_data\n    print(\"Finished.\")\n    ","metadata":{"execution":{"iopub.status.busy":"2022-07-06T15:05:56.68733Z","iopub.execute_input":"2022-07-06T15:05:56.687944Z","iopub.status.idle":"2022-07-06T15:05:56.705399Z","shell.execute_reply.started":"2022-07-06T15:05:56.687908Z","shell.execute_reply":"2022-07-06T15:05:56.704362Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Execution","metadata":{}},{"cell_type":"code","source":"preprocessing_pipeline('original', 'original')","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preprocessing_pipeline('mean', 'mean')","metadata":{"execution":{"iopub.status.busy":"2022-07-06T15:05:57.667642Z","iopub.execute_input":"2022-07-06T15:05:57.668055Z","iopub.status.idle":"2022-07-06T15:05:57.671999Z","shell.execute_reply.started":"2022-07-06T15:05:57.668017Z","shell.execute_reply":"2022-07-06T15:05:57.671138Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preprocessing_pipeline('ffil', 'ffil')","metadata":{"execution":{"iopub.status.busy":"2022-07-06T15:05:58.176609Z","iopub.execute_input":"2022-07-06T15:05:58.177018Z","iopub.status.idle":"2022-07-06T15:05:58.181414Z","shell.execute_reply.started":"2022-07-06T15:05:58.176982Z","shell.execute_reply":"2022-07-06T15:05:58.180014Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preprocessing_pipeline('mean', 'mean-indicator', addIndicator=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-06T15:05:59.920915Z","iopub.execute_input":"2022-07-06T15:05:59.92148Z","iopub.status.idle":"2022-07-06T15:09:15.883075Z","shell.execute_reply.started":"2022-07-06T15:05:59.921426Z","shell.execute_reply":"2022-07-06T15:09:15.881931Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preprocessing_pipeline('ffil', 'ffil-indicator', addIndicator=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-06T15:09:15.885621Z","iopub.execute_input":"2022-07-06T15:09:15.885961Z","iopub.status.idle":"2022-07-06T15:09:37.919318Z","shell.execute_reply.started":"2022-07-06T15:09:15.885923Z","shell.execute_reply":"2022-07-06T15:09:37.917361Z"},"trusted":true},"execution_count":null,"outputs":[]}]}